Merge remote-tracking branch 'net-next/master' into mac80211-next
Resolve the merge conflict between Felix's/my and Toke's patches coming into the tree through net and mac80211-next respectively. Most of Felix's changes go away due to Toke's new infrastructure work, my patch changes to "goto begin" (the label wasn't there before) instead of returning NULL so flow control towards drivers is preserved better. Signed-off-by: Johannes Berg <johannes.berg@intel.com>
This commit is contained in:
@@ -101,8 +101,6 @@ static void lowpan_ndisc_802154_update(struct neighbour *n, u32 flags,
|
||||
ieee802154_be16_to_le16(&neigh->short_addr, lladdr_short);
|
||||
if (!lowpan_802154_is_valid_src_short_addr(neigh->short_addr))
|
||||
neigh->short_addr = cpu_to_le16(IEEE802154_ADDR_SHORT_UNSPEC);
|
||||
} else {
|
||||
neigh->short_addr = cpu_to_le16(IEEE802154_ADDR_SHORT_UNSPEC);
|
||||
}
|
||||
write_unlock_bh(&n->lock);
|
||||
}
|
||||
|
||||
@@ -335,7 +335,7 @@ int batadv_v_elp_iface_enable(struct batadv_hard_iface *hard_iface)
|
||||
goto out;
|
||||
|
||||
skb_reserve(hard_iface->bat_v.elp_skb, ETH_HLEN + NET_IP_ALIGN);
|
||||
elp_buff = skb_push(hard_iface->bat_v.elp_skb, BATADV_ELP_HLEN);
|
||||
elp_buff = skb_put(hard_iface->bat_v.elp_skb, BATADV_ELP_HLEN);
|
||||
elp_packet = (struct batadv_elp_packet *)elp_buff;
|
||||
memset(elp_packet, 0, BATADV_ELP_HLEN);
|
||||
|
||||
|
||||
@@ -460,6 +460,29 @@ static int batadv_check_unicast_packet(struct batadv_priv *bat_priv,
|
||||
return 0;
|
||||
}
|
||||
|
||||
/**
|
||||
* batadv_last_bonding_get - Get last_bonding_candidate of orig_node
|
||||
* @orig_node: originator node whose last bonding candidate should be retrieved
|
||||
*
|
||||
* Return: last bonding candidate of router or NULL if not found
|
||||
*
|
||||
* The object is returned with refcounter increased by 1.
|
||||
*/
|
||||
static struct batadv_orig_ifinfo *
|
||||
batadv_last_bonding_get(struct batadv_orig_node *orig_node)
|
||||
{
|
||||
struct batadv_orig_ifinfo *last_bonding_candidate;
|
||||
|
||||
spin_lock_bh(&orig_node->neigh_list_lock);
|
||||
last_bonding_candidate = orig_node->last_bonding_candidate;
|
||||
|
||||
if (last_bonding_candidate)
|
||||
kref_get(&last_bonding_candidate->refcount);
|
||||
spin_unlock_bh(&orig_node->neigh_list_lock);
|
||||
|
||||
return last_bonding_candidate;
|
||||
}
|
||||
|
||||
/**
|
||||
* batadv_last_bonding_replace - Replace last_bonding_candidate of orig_node
|
||||
* @orig_node: originator node whose bonding candidates should be replaced
|
||||
@@ -530,7 +553,7 @@ batadv_find_router(struct batadv_priv *bat_priv,
|
||||
* router - obviously there are no other candidates.
|
||||
*/
|
||||
rcu_read_lock();
|
||||
last_candidate = orig_node->last_bonding_candidate;
|
||||
last_candidate = batadv_last_bonding_get(orig_node);
|
||||
if (last_candidate)
|
||||
last_cand_router = rcu_dereference(last_candidate->router);
|
||||
|
||||
@@ -622,6 +645,9 @@ next:
|
||||
batadv_orig_ifinfo_put(next_candidate);
|
||||
}
|
||||
|
||||
if (last_candidate)
|
||||
batadv_orig_ifinfo_put(last_candidate);
|
||||
|
||||
return router;
|
||||
}
|
||||
|
||||
|
||||
@@ -26,11 +26,13 @@
|
||||
|
||||
#include <linux/module.h>
|
||||
#include <linux/debugfs.h>
|
||||
#include <linux/stringify.h>
|
||||
#include <asm/ioctls.h>
|
||||
|
||||
#include <net/bluetooth/bluetooth.h>
|
||||
#include <linux/proc_fs.h>
|
||||
|
||||
#include "leds.h"
|
||||
#include "selftest.h"
|
||||
|
||||
/* Bluetooth sockets */
|
||||
@@ -712,13 +714,16 @@ static struct net_proto_family bt_sock_family_ops = {
|
||||
struct dentry *bt_debugfs;
|
||||
EXPORT_SYMBOL_GPL(bt_debugfs);
|
||||
|
||||
#define VERSION __stringify(BT_SUBSYS_VERSION) "." \
|
||||
__stringify(BT_SUBSYS_REVISION)
|
||||
|
||||
static int __init bt_init(void)
|
||||
{
|
||||
int err;
|
||||
|
||||
sock_skb_cb_check_size(sizeof(struct bt_skb_cb));
|
||||
|
||||
BT_INFO("Core ver %s", BT_SUBSYS_VERSION);
|
||||
BT_INFO("Core ver %s", VERSION);
|
||||
|
||||
err = bt_selftest();
|
||||
if (err < 0)
|
||||
@@ -726,6 +731,8 @@ static int __init bt_init(void)
|
||||
|
||||
bt_debugfs = debugfs_create_dir("bluetooth", NULL);
|
||||
|
||||
bt_leds_init();
|
||||
|
||||
err = bt_sysfs_init();
|
||||
if (err < 0)
|
||||
return err;
|
||||
@@ -785,6 +792,8 @@ static void __exit bt_exit(void)
|
||||
|
||||
bt_sysfs_cleanup();
|
||||
|
||||
bt_leds_cleanup();
|
||||
|
||||
debugfs_remove_recursive(bt_debugfs);
|
||||
}
|
||||
|
||||
@@ -792,7 +801,7 @@ subsys_initcall(bt_init);
|
||||
module_exit(bt_exit);
|
||||
|
||||
MODULE_AUTHOR("Marcel Holtmann <marcel@holtmann.org>");
|
||||
MODULE_DESCRIPTION("Bluetooth Core ver " BT_SUBSYS_VERSION);
|
||||
MODULE_VERSION(BT_SUBSYS_VERSION);
|
||||
MODULE_DESCRIPTION("Bluetooth Core ver " VERSION);
|
||||
MODULE_VERSION(VERSION);
|
||||
MODULE_LICENSE("GPL");
|
||||
MODULE_ALIAS_NETPROTO(PF_BLUETOOTH);
|
||||
|
||||
@@ -1562,6 +1562,7 @@ int hci_dev_do_close(struct hci_dev *hdev)
|
||||
auto_off = hci_dev_test_and_clear_flag(hdev, HCI_AUTO_OFF);
|
||||
|
||||
if (!auto_off && hdev->dev_type == HCI_PRIMARY &&
|
||||
!hci_dev_test_flag(hdev, HCI_USER_CHANNEL) &&
|
||||
hci_dev_test_flag(hdev, HCI_MGMT))
|
||||
__mgmt_power_off(hdev);
|
||||
|
||||
|
||||
+35
-14
@@ -971,14 +971,14 @@ void __hci_req_enable_advertising(struct hci_request *req)
|
||||
hci_req_add(req, HCI_OP_LE_SET_ADV_ENABLE, sizeof(enable), &enable);
|
||||
}
|
||||
|
||||
static u8 create_default_scan_rsp_data(struct hci_dev *hdev, u8 *ptr)
|
||||
static u8 append_local_name(struct hci_dev *hdev, u8 *ptr, u8 ad_len)
|
||||
{
|
||||
u8 ad_len = 0;
|
||||
size_t name_len;
|
||||
int max_len;
|
||||
|
||||
max_len = HCI_MAX_AD_LENGTH - ad_len - 2;
|
||||
name_len = strlen(hdev->dev_name);
|
||||
if (name_len > 0) {
|
||||
size_t max_len = HCI_MAX_AD_LENGTH - ad_len - 2;
|
||||
if (name_len > 0 && max_len > 0) {
|
||||
|
||||
if (name_len > max_len) {
|
||||
name_len = max_len;
|
||||
@@ -997,22 +997,42 @@ static u8 create_default_scan_rsp_data(struct hci_dev *hdev, u8 *ptr)
|
||||
return ad_len;
|
||||
}
|
||||
|
||||
static u8 create_default_scan_rsp_data(struct hci_dev *hdev, u8 *ptr)
|
||||
{
|
||||
return append_local_name(hdev, ptr, 0);
|
||||
}
|
||||
|
||||
static u8 create_instance_scan_rsp_data(struct hci_dev *hdev, u8 instance,
|
||||
u8 *ptr)
|
||||
{
|
||||
struct adv_info *adv_instance;
|
||||
u32 instance_flags;
|
||||
u8 scan_rsp_len = 0;
|
||||
|
||||
adv_instance = hci_find_adv_instance(hdev, instance);
|
||||
if (!adv_instance)
|
||||
return 0;
|
||||
|
||||
/* TODO: Set the appropriate entries based on advertising instance flags
|
||||
* here once flags other than 0 are supported.
|
||||
*/
|
||||
instance_flags = adv_instance->flags;
|
||||
|
||||
if ((instance_flags & MGMT_ADV_FLAG_APPEARANCE) && hdev->appearance) {
|
||||
ptr[0] = 3;
|
||||
ptr[1] = EIR_APPEARANCE;
|
||||
put_unaligned_le16(hdev->appearance, ptr + 2);
|
||||
scan_rsp_len += 4;
|
||||
ptr += 4;
|
||||
}
|
||||
|
||||
memcpy(ptr, adv_instance->scan_rsp_data,
|
||||
adv_instance->scan_rsp_len);
|
||||
|
||||
return adv_instance->scan_rsp_len;
|
||||
scan_rsp_len += adv_instance->scan_rsp_len;
|
||||
ptr += adv_instance->scan_rsp_len;
|
||||
|
||||
if (instance_flags & MGMT_ADV_FLAG_LOCAL_NAME)
|
||||
scan_rsp_len = append_local_name(hdev, ptr, scan_rsp_len);
|
||||
|
||||
return scan_rsp_len;
|
||||
}
|
||||
|
||||
void __hci_req_update_scan_rsp_data(struct hci_request *req, u8 instance)
|
||||
@@ -1194,7 +1214,7 @@ static void adv_timeout_expire(struct work_struct *work)
|
||||
|
||||
hci_req_init(&req, hdev);
|
||||
|
||||
hci_req_clear_adv_instance(hdev, &req, instance, false);
|
||||
hci_req_clear_adv_instance(hdev, NULL, &req, instance, false);
|
||||
|
||||
if (list_empty(&hdev->adv_instances))
|
||||
__hci_req_disable_advertising(&req);
|
||||
@@ -1284,8 +1304,9 @@ static void cancel_adv_timeout(struct hci_dev *hdev)
|
||||
* setting.
|
||||
* - force == false: Only instances that have a timeout will be removed.
|
||||
*/
|
||||
void hci_req_clear_adv_instance(struct hci_dev *hdev, struct hci_request *req,
|
||||
u8 instance, bool force)
|
||||
void hci_req_clear_adv_instance(struct hci_dev *hdev, struct sock *sk,
|
||||
struct hci_request *req, u8 instance,
|
||||
bool force)
|
||||
{
|
||||
struct adv_info *adv_instance, *n, *next_instance = NULL;
|
||||
int err;
|
||||
@@ -1311,7 +1332,7 @@ void hci_req_clear_adv_instance(struct hci_dev *hdev, struct hci_request *req,
|
||||
rem_inst = adv_instance->instance;
|
||||
err = hci_remove_adv_instance(hdev, rem_inst);
|
||||
if (!err)
|
||||
mgmt_advertising_removed(NULL, hdev, rem_inst);
|
||||
mgmt_advertising_removed(sk, hdev, rem_inst);
|
||||
}
|
||||
} else {
|
||||
adv_instance = hci_find_adv_instance(hdev, instance);
|
||||
@@ -1325,7 +1346,7 @@ void hci_req_clear_adv_instance(struct hci_dev *hdev, struct hci_request *req,
|
||||
|
||||
err = hci_remove_adv_instance(hdev, instance);
|
||||
if (!err)
|
||||
mgmt_advertising_removed(NULL, hdev, instance);
|
||||
mgmt_advertising_removed(sk, hdev, instance);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1716,7 +1737,7 @@ void __hci_abort_conn(struct hci_request *req, struct hci_conn *conn,
|
||||
* function. To be safe hard-code one of the
|
||||
* values that's suitable for SCO.
|
||||
*/
|
||||
rej.reason = HCI_ERROR_REMOTE_LOW_RESOURCES;
|
||||
rej.reason = HCI_ERROR_REJ_LIMITED_RESOURCES;
|
||||
|
||||
hci_req_add(req, HCI_OP_REJECT_SYNC_CONN_REQ,
|
||||
sizeof(rej), &rej);
|
||||
|
||||
@@ -73,8 +73,9 @@ void __hci_req_update_scan_rsp_data(struct hci_request *req, u8 instance);
|
||||
|
||||
int __hci_req_schedule_adv_instance(struct hci_request *req, u8 instance,
|
||||
bool force);
|
||||
void hci_req_clear_adv_instance(struct hci_dev *hdev, struct hci_request *req,
|
||||
u8 instance, bool force);
|
||||
void hci_req_clear_adv_instance(struct hci_dev *hdev, struct sock *sk,
|
||||
struct hci_request *req, u8 instance,
|
||||
bool force);
|
||||
|
||||
void __hci_req_update_class(struct hci_request *req);
|
||||
|
||||
|
||||
+388
-8
@@ -26,6 +26,7 @@
|
||||
|
||||
#include <linux/export.h>
|
||||
#include <linux/utsname.h>
|
||||
#include <linux/sched.h>
|
||||
#include <asm/unaligned.h>
|
||||
|
||||
#include <net/bluetooth/bluetooth.h>
|
||||
@@ -38,6 +39,8 @@
|
||||
static LIST_HEAD(mgmt_chan_list);
|
||||
static DEFINE_MUTEX(mgmt_chan_list_lock);
|
||||
|
||||
static DEFINE_IDA(sock_cookie_ida);
|
||||
|
||||
static atomic_t monitor_promisc = ATOMIC_INIT(0);
|
||||
|
||||
/* ----- HCI socket interface ----- */
|
||||
@@ -52,6 +55,8 @@ struct hci_pinfo {
|
||||
__u32 cmsg_mask;
|
||||
unsigned short channel;
|
||||
unsigned long flags;
|
||||
__u32 cookie;
|
||||
char comm[TASK_COMM_LEN];
|
||||
};
|
||||
|
||||
void hci_sock_set_flag(struct sock *sk, int nr)
|
||||
@@ -74,6 +79,38 @@ unsigned short hci_sock_get_channel(struct sock *sk)
|
||||
return hci_pi(sk)->channel;
|
||||
}
|
||||
|
||||
u32 hci_sock_get_cookie(struct sock *sk)
|
||||
{
|
||||
return hci_pi(sk)->cookie;
|
||||
}
|
||||
|
||||
static bool hci_sock_gen_cookie(struct sock *sk)
|
||||
{
|
||||
int id = hci_pi(sk)->cookie;
|
||||
|
||||
if (!id) {
|
||||
id = ida_simple_get(&sock_cookie_ida, 1, 0, GFP_KERNEL);
|
||||
if (id < 0)
|
||||
id = 0xffffffff;
|
||||
|
||||
hci_pi(sk)->cookie = id;
|
||||
get_task_comm(hci_pi(sk)->comm, current);
|
||||
return true;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
static void hci_sock_free_cookie(struct sock *sk)
|
||||
{
|
||||
int id = hci_pi(sk)->cookie;
|
||||
|
||||
if (id) {
|
||||
hci_pi(sk)->cookie = 0xffffffff;
|
||||
ida_simple_remove(&sock_cookie_ida, id);
|
||||
}
|
||||
}
|
||||
|
||||
static inline int hci_test_bit(int nr, const void *addr)
|
||||
{
|
||||
return *((const __u32 *) addr + (nr >> 5)) & ((__u32) 1 << (nr & 31));
|
||||
@@ -305,6 +342,60 @@ void hci_send_to_monitor(struct hci_dev *hdev, struct sk_buff *skb)
|
||||
kfree_skb(skb_copy);
|
||||
}
|
||||
|
||||
void hci_send_monitor_ctrl_event(struct hci_dev *hdev, u16 event,
|
||||
void *data, u16 data_len, ktime_t tstamp,
|
||||
int flag, struct sock *skip_sk)
|
||||
{
|
||||
struct sock *sk;
|
||||
__le16 index;
|
||||
|
||||
if (hdev)
|
||||
index = cpu_to_le16(hdev->id);
|
||||
else
|
||||
index = cpu_to_le16(MGMT_INDEX_NONE);
|
||||
|
||||
read_lock(&hci_sk_list.lock);
|
||||
|
||||
sk_for_each(sk, &hci_sk_list.head) {
|
||||
struct hci_mon_hdr *hdr;
|
||||
struct sk_buff *skb;
|
||||
|
||||
if (hci_pi(sk)->channel != HCI_CHANNEL_CONTROL)
|
||||
continue;
|
||||
|
||||
/* Ignore socket without the flag set */
|
||||
if (!hci_sock_test_flag(sk, flag))
|
||||
continue;
|
||||
|
||||
/* Skip the original socket */
|
||||
if (sk == skip_sk)
|
||||
continue;
|
||||
|
||||
skb = bt_skb_alloc(6 + data_len, GFP_ATOMIC);
|
||||
if (!skb)
|
||||
continue;
|
||||
|
||||
put_unaligned_le32(hci_pi(sk)->cookie, skb_put(skb, 4));
|
||||
put_unaligned_le16(event, skb_put(skb, 2));
|
||||
|
||||
if (data)
|
||||
memcpy(skb_put(skb, data_len), data, data_len);
|
||||
|
||||
skb->tstamp = tstamp;
|
||||
|
||||
hdr = (void *)skb_push(skb, HCI_MON_HDR_SIZE);
|
||||
hdr->opcode = cpu_to_le16(HCI_MON_CTRL_EVENT);
|
||||
hdr->index = index;
|
||||
hdr->len = cpu_to_le16(skb->len - HCI_MON_HDR_SIZE);
|
||||
|
||||
hci_send_to_channel(HCI_CHANNEL_MONITOR, skb,
|
||||
HCI_SOCK_TRUSTED, NULL);
|
||||
kfree_skb(skb);
|
||||
}
|
||||
|
||||
read_unlock(&hci_sk_list.lock);
|
||||
}
|
||||
|
||||
static struct sk_buff *create_monitor_event(struct hci_dev *hdev, int event)
|
||||
{
|
||||
struct hci_mon_hdr *hdr;
|
||||
@@ -384,6 +475,129 @@ static struct sk_buff *create_monitor_event(struct hci_dev *hdev, int event)
|
||||
return skb;
|
||||
}
|
||||
|
||||
static struct sk_buff *create_monitor_ctrl_open(struct sock *sk)
|
||||
{
|
||||
struct hci_mon_hdr *hdr;
|
||||
struct sk_buff *skb;
|
||||
u16 format;
|
||||
u8 ver[3];
|
||||
u32 flags;
|
||||
|
||||
/* No message needed when cookie is not present */
|
||||
if (!hci_pi(sk)->cookie)
|
||||
return NULL;
|
||||
|
||||
switch (hci_pi(sk)->channel) {
|
||||
case HCI_CHANNEL_RAW:
|
||||
format = 0x0000;
|
||||
ver[0] = BT_SUBSYS_VERSION;
|
||||
put_unaligned_le16(BT_SUBSYS_REVISION, ver + 1);
|
||||
break;
|
||||
case HCI_CHANNEL_USER:
|
||||
format = 0x0001;
|
||||
ver[0] = BT_SUBSYS_VERSION;
|
||||
put_unaligned_le16(BT_SUBSYS_REVISION, ver + 1);
|
||||
break;
|
||||
case HCI_CHANNEL_CONTROL:
|
||||
format = 0x0002;
|
||||
mgmt_fill_version_info(ver);
|
||||
break;
|
||||
default:
|
||||
/* No message for unsupported format */
|
||||
return NULL;
|
||||
}
|
||||
|
||||
skb = bt_skb_alloc(14 + TASK_COMM_LEN , GFP_ATOMIC);
|
||||
if (!skb)
|
||||
return NULL;
|
||||
|
||||
flags = hci_sock_test_flag(sk, HCI_SOCK_TRUSTED) ? 0x1 : 0x0;
|
||||
|
||||
put_unaligned_le32(hci_pi(sk)->cookie, skb_put(skb, 4));
|
||||
put_unaligned_le16(format, skb_put(skb, 2));
|
||||
memcpy(skb_put(skb, sizeof(ver)), ver, sizeof(ver));
|
||||
put_unaligned_le32(flags, skb_put(skb, 4));
|
||||
*skb_put(skb, 1) = TASK_COMM_LEN;
|
||||
memcpy(skb_put(skb, TASK_COMM_LEN), hci_pi(sk)->comm, TASK_COMM_LEN);
|
||||
|
||||
__net_timestamp(skb);
|
||||
|
||||
hdr = (void *)skb_push(skb, HCI_MON_HDR_SIZE);
|
||||
hdr->opcode = cpu_to_le16(HCI_MON_CTRL_OPEN);
|
||||
if (hci_pi(sk)->hdev)
|
||||
hdr->index = cpu_to_le16(hci_pi(sk)->hdev->id);
|
||||
else
|
||||
hdr->index = cpu_to_le16(HCI_DEV_NONE);
|
||||
hdr->len = cpu_to_le16(skb->len - HCI_MON_HDR_SIZE);
|
||||
|
||||
return skb;
|
||||
}
|
||||
|
||||
static struct sk_buff *create_monitor_ctrl_close(struct sock *sk)
|
||||
{
|
||||
struct hci_mon_hdr *hdr;
|
||||
struct sk_buff *skb;
|
||||
|
||||
/* No message needed when cookie is not present */
|
||||
if (!hci_pi(sk)->cookie)
|
||||
return NULL;
|
||||
|
||||
switch (hci_pi(sk)->channel) {
|
||||
case HCI_CHANNEL_RAW:
|
||||
case HCI_CHANNEL_USER:
|
||||
case HCI_CHANNEL_CONTROL:
|
||||
break;
|
||||
default:
|
||||
/* No message for unsupported format */
|
||||
return NULL;
|
||||
}
|
||||
|
||||
skb = bt_skb_alloc(4, GFP_ATOMIC);
|
||||
if (!skb)
|
||||
return NULL;
|
||||
|
||||
put_unaligned_le32(hci_pi(sk)->cookie, skb_put(skb, 4));
|
||||
|
||||
__net_timestamp(skb);
|
||||
|
||||
hdr = (void *)skb_push(skb, HCI_MON_HDR_SIZE);
|
||||
hdr->opcode = cpu_to_le16(HCI_MON_CTRL_CLOSE);
|
||||
if (hci_pi(sk)->hdev)
|
||||
hdr->index = cpu_to_le16(hci_pi(sk)->hdev->id);
|
||||
else
|
||||
hdr->index = cpu_to_le16(HCI_DEV_NONE);
|
||||
hdr->len = cpu_to_le16(skb->len - HCI_MON_HDR_SIZE);
|
||||
|
||||
return skb;
|
||||
}
|
||||
|
||||
static struct sk_buff *create_monitor_ctrl_command(struct sock *sk, u16 index,
|
||||
u16 opcode, u16 len,
|
||||
const void *buf)
|
||||
{
|
||||
struct hci_mon_hdr *hdr;
|
||||
struct sk_buff *skb;
|
||||
|
||||
skb = bt_skb_alloc(6 + len, GFP_ATOMIC);
|
||||
if (!skb)
|
||||
return NULL;
|
||||
|
||||
put_unaligned_le32(hci_pi(sk)->cookie, skb_put(skb, 4));
|
||||
put_unaligned_le16(opcode, skb_put(skb, 2));
|
||||
|
||||
if (buf)
|
||||
memcpy(skb_put(skb, len), buf, len);
|
||||
|
||||
__net_timestamp(skb);
|
||||
|
||||
hdr = (void *)skb_push(skb, HCI_MON_HDR_SIZE);
|
||||
hdr->opcode = cpu_to_le16(HCI_MON_CTRL_COMMAND);
|
||||
hdr->index = cpu_to_le16(index);
|
||||
hdr->len = cpu_to_le16(skb->len - HCI_MON_HDR_SIZE);
|
||||
|
||||
return skb;
|
||||
}
|
||||
|
||||
static void __printf(2, 3)
|
||||
send_monitor_note(struct sock *sk, const char *fmt, ...)
|
||||
{
|
||||
@@ -458,6 +672,26 @@ static void send_monitor_replay(struct sock *sk)
|
||||
read_unlock(&hci_dev_list_lock);
|
||||
}
|
||||
|
||||
static void send_monitor_control_replay(struct sock *mon_sk)
|
||||
{
|
||||
struct sock *sk;
|
||||
|
||||
read_lock(&hci_sk_list.lock);
|
||||
|
||||
sk_for_each(sk, &hci_sk_list.head) {
|
||||
struct sk_buff *skb;
|
||||
|
||||
skb = create_monitor_ctrl_open(sk);
|
||||
if (!skb)
|
||||
continue;
|
||||
|
||||
if (sock_queue_rcv_skb(mon_sk, skb))
|
||||
kfree_skb(skb);
|
||||
}
|
||||
|
||||
read_unlock(&hci_sk_list.lock);
|
||||
}
|
||||
|
||||
/* Generate internal stack event */
|
||||
static void hci_si_event(struct hci_dev *hdev, int type, int dlen, void *data)
|
||||
{
|
||||
@@ -585,6 +819,7 @@ static int hci_sock_release(struct socket *sock)
|
||||
{
|
||||
struct sock *sk = sock->sk;
|
||||
struct hci_dev *hdev;
|
||||
struct sk_buff *skb;
|
||||
|
||||
BT_DBG("sock %p sk %p", sock, sk);
|
||||
|
||||
@@ -593,8 +828,24 @@ static int hci_sock_release(struct socket *sock)
|
||||
|
||||
hdev = hci_pi(sk)->hdev;
|
||||
|
||||
if (hci_pi(sk)->channel == HCI_CHANNEL_MONITOR)
|
||||
switch (hci_pi(sk)->channel) {
|
||||
case HCI_CHANNEL_MONITOR:
|
||||
atomic_dec(&monitor_promisc);
|
||||
break;
|
||||
case HCI_CHANNEL_RAW:
|
||||
case HCI_CHANNEL_USER:
|
||||
case HCI_CHANNEL_CONTROL:
|
||||
/* Send event to monitor */
|
||||
skb = create_monitor_ctrl_close(sk);
|
||||
if (skb) {
|
||||
hci_send_to_channel(HCI_CHANNEL_MONITOR, skb,
|
||||
HCI_SOCK_TRUSTED, NULL);
|
||||
kfree_skb(skb);
|
||||
}
|
||||
|
||||
hci_sock_free_cookie(sk);
|
||||
break;
|
||||
}
|
||||
|
||||
bt_sock_unlink(&hci_sk_list, sk);
|
||||
|
||||
@@ -721,6 +972,27 @@ static int hci_sock_ioctl(struct socket *sock, unsigned int cmd,
|
||||
goto done;
|
||||
}
|
||||
|
||||
/* When calling an ioctl on an unbound raw socket, then ensure
|
||||
* that the monitor gets informed. Ensure that the resulting event
|
||||
* is only send once by checking if the cookie exists or not. The
|
||||
* socket cookie will be only ever generated once for the lifetime
|
||||
* of a given socket.
|
||||
*/
|
||||
if (hci_sock_gen_cookie(sk)) {
|
||||
struct sk_buff *skb;
|
||||
|
||||
if (capable(CAP_NET_ADMIN))
|
||||
hci_sock_set_flag(sk, HCI_SOCK_TRUSTED);
|
||||
|
||||
/* Send event to monitor */
|
||||
skb = create_monitor_ctrl_open(sk);
|
||||
if (skb) {
|
||||
hci_send_to_channel(HCI_CHANNEL_MONITOR, skb,
|
||||
HCI_SOCK_TRUSTED, NULL);
|
||||
kfree_skb(skb);
|
||||
}
|
||||
}
|
||||
|
||||
release_sock(sk);
|
||||
|
||||
switch (cmd) {
|
||||
@@ -784,6 +1056,7 @@ static int hci_sock_bind(struct socket *sock, struct sockaddr *addr,
|
||||
struct sockaddr_hci haddr;
|
||||
struct sock *sk = sock->sk;
|
||||
struct hci_dev *hdev = NULL;
|
||||
struct sk_buff *skb;
|
||||
int len, err = 0;
|
||||
|
||||
BT_DBG("sock %p sk %p", sock, sk);
|
||||
@@ -822,7 +1095,35 @@ static int hci_sock_bind(struct socket *sock, struct sockaddr *addr,
|
||||
atomic_inc(&hdev->promisc);
|
||||
}
|
||||
|
||||
hci_pi(sk)->channel = haddr.hci_channel;
|
||||
|
||||
if (!hci_sock_gen_cookie(sk)) {
|
||||
/* In the case when a cookie has already been assigned,
|
||||
* then there has been already an ioctl issued against
|
||||
* an unbound socket and with that triggerd an open
|
||||
* notification. Send a close notification first to
|
||||
* allow the state transition to bounded.
|
||||
*/
|
||||
skb = create_monitor_ctrl_close(sk);
|
||||
if (skb) {
|
||||
hci_send_to_channel(HCI_CHANNEL_MONITOR, skb,
|
||||
HCI_SOCK_TRUSTED, NULL);
|
||||
kfree_skb(skb);
|
||||
}
|
||||
}
|
||||
|
||||
if (capable(CAP_NET_ADMIN))
|
||||
hci_sock_set_flag(sk, HCI_SOCK_TRUSTED);
|
||||
|
||||
hci_pi(sk)->hdev = hdev;
|
||||
|
||||
/* Send event to monitor */
|
||||
skb = create_monitor_ctrl_open(sk);
|
||||
if (skb) {
|
||||
hci_send_to_channel(HCI_CHANNEL_MONITOR, skb,
|
||||
HCI_SOCK_TRUSTED, NULL);
|
||||
kfree_skb(skb);
|
||||
}
|
||||
break;
|
||||
|
||||
case HCI_CHANNEL_USER:
|
||||
@@ -884,9 +1185,38 @@ static int hci_sock_bind(struct socket *sock, struct sockaddr *addr,
|
||||
}
|
||||
}
|
||||
|
||||
atomic_inc(&hdev->promisc);
|
||||
hci_pi(sk)->channel = haddr.hci_channel;
|
||||
|
||||
if (!hci_sock_gen_cookie(sk)) {
|
||||
/* In the case when a cookie has already been assigned,
|
||||
* this socket will transition from a raw socket into
|
||||
* an user channel socket. For a clean transition, send
|
||||
* the close notification first.
|
||||
*/
|
||||
skb = create_monitor_ctrl_close(sk);
|
||||
if (skb) {
|
||||
hci_send_to_channel(HCI_CHANNEL_MONITOR, skb,
|
||||
HCI_SOCK_TRUSTED, NULL);
|
||||
kfree_skb(skb);
|
||||
}
|
||||
}
|
||||
|
||||
/* The user channel is restricted to CAP_NET_ADMIN
|
||||
* capabilities and with that implicitly trusted.
|
||||
*/
|
||||
hci_sock_set_flag(sk, HCI_SOCK_TRUSTED);
|
||||
|
||||
hci_pi(sk)->hdev = hdev;
|
||||
|
||||
/* Send event to monitor */
|
||||
skb = create_monitor_ctrl_open(sk);
|
||||
if (skb) {
|
||||
hci_send_to_channel(HCI_CHANNEL_MONITOR, skb,
|
||||
HCI_SOCK_TRUSTED, NULL);
|
||||
kfree_skb(skb);
|
||||
}
|
||||
|
||||
atomic_inc(&hdev->promisc);
|
||||
break;
|
||||
|
||||
case HCI_CHANNEL_MONITOR:
|
||||
@@ -900,6 +1230,8 @@ static int hci_sock_bind(struct socket *sock, struct sockaddr *addr,
|
||||
goto done;
|
||||
}
|
||||
|
||||
hci_pi(sk)->channel = haddr.hci_channel;
|
||||
|
||||
/* The monitor interface is restricted to CAP_NET_RAW
|
||||
* capabilities and with that implicitly trusted.
|
||||
*/
|
||||
@@ -908,9 +1240,10 @@ static int hci_sock_bind(struct socket *sock, struct sockaddr *addr,
|
||||
send_monitor_note(sk, "Linux version %s (%s)",
|
||||
init_utsname()->release,
|
||||
init_utsname()->machine);
|
||||
send_monitor_note(sk, "Bluetooth subsystem version %s",
|
||||
BT_SUBSYS_VERSION);
|
||||
send_monitor_note(sk, "Bluetooth subsystem version %u.%u",
|
||||
BT_SUBSYS_VERSION, BT_SUBSYS_REVISION);
|
||||
send_monitor_replay(sk);
|
||||
send_monitor_control_replay(sk);
|
||||
|
||||
atomic_inc(&monitor_promisc);
|
||||
break;
|
||||
@@ -925,6 +1258,8 @@ static int hci_sock_bind(struct socket *sock, struct sockaddr *addr,
|
||||
err = -EPERM;
|
||||
goto done;
|
||||
}
|
||||
|
||||
hci_pi(sk)->channel = haddr.hci_channel;
|
||||
break;
|
||||
|
||||
default:
|
||||
@@ -946,6 +1281,8 @@ static int hci_sock_bind(struct socket *sock, struct sockaddr *addr,
|
||||
if (capable(CAP_NET_ADMIN))
|
||||
hci_sock_set_flag(sk, HCI_SOCK_TRUSTED);
|
||||
|
||||
hci_pi(sk)->channel = haddr.hci_channel;
|
||||
|
||||
/* At the moment the index and unconfigured index events
|
||||
* are enabled unconditionally. Setting them on each
|
||||
* socket when binding keeps this functionality. They
|
||||
@@ -956,16 +1293,40 @@ static int hci_sock_bind(struct socket *sock, struct sockaddr *addr,
|
||||
* received by untrusted users. Example for such events
|
||||
* are changes to settings, class of device, name etc.
|
||||
*/
|
||||
if (haddr.hci_channel == HCI_CHANNEL_CONTROL) {
|
||||
if (hci_pi(sk)->channel == HCI_CHANNEL_CONTROL) {
|
||||
if (!hci_sock_gen_cookie(sk)) {
|
||||
/* In the case when a cookie has already been
|
||||
* assigned, this socket will transtion from
|
||||
* a raw socket into a control socket. To
|
||||
* allow for a clean transtion, send the
|
||||
* close notification first.
|
||||
*/
|
||||
skb = create_monitor_ctrl_close(sk);
|
||||
if (skb) {
|
||||
hci_send_to_channel(HCI_CHANNEL_MONITOR, skb,
|
||||
HCI_SOCK_TRUSTED, NULL);
|
||||
kfree_skb(skb);
|
||||
}
|
||||
}
|
||||
|
||||
/* Send event to monitor */
|
||||
skb = create_monitor_ctrl_open(sk);
|
||||
if (skb) {
|
||||
hci_send_to_channel(HCI_CHANNEL_MONITOR, skb,
|
||||
HCI_SOCK_TRUSTED, NULL);
|
||||
kfree_skb(skb);
|
||||
}
|
||||
|
||||
hci_sock_set_flag(sk, HCI_MGMT_INDEX_EVENTS);
|
||||
hci_sock_set_flag(sk, HCI_MGMT_UNCONF_INDEX_EVENTS);
|
||||
hci_sock_set_flag(sk, HCI_MGMT_GENERIC_EVENTS);
|
||||
hci_sock_set_flag(sk, HCI_MGMT_OPTION_EVENTS);
|
||||
hci_sock_set_flag(sk, HCI_MGMT_SETTING_EVENTS);
|
||||
hci_sock_set_flag(sk, HCI_MGMT_DEV_CLASS_EVENTS);
|
||||
hci_sock_set_flag(sk, HCI_MGMT_LOCAL_NAME_EVENTS);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
|
||||
hci_pi(sk)->channel = haddr.hci_channel;
|
||||
sk->sk_state = BT_BOUND;
|
||||
|
||||
done:
|
||||
@@ -1133,6 +1494,19 @@ static int hci_mgmt_cmd(struct hci_mgmt_chan *chan, struct sock *sk,
|
||||
goto done;
|
||||
}
|
||||
|
||||
if (chan->channel == HCI_CHANNEL_CONTROL) {
|
||||
struct sk_buff *skb;
|
||||
|
||||
/* Send event to monitor */
|
||||
skb = create_monitor_ctrl_command(sk, index, opcode, len,
|
||||
buf + sizeof(*hdr));
|
||||
if (skb) {
|
||||
hci_send_to_channel(HCI_CHANNEL_MONITOR, skb,
|
||||
HCI_SOCK_TRUSTED, NULL);
|
||||
kfree_skb(skb);
|
||||
}
|
||||
}
|
||||
|
||||
if (opcode >= chan->handler_count ||
|
||||
chan->handlers[opcode].func == NULL) {
|
||||
BT_DBG("Unknown op %u", opcode);
|
||||
@@ -1440,6 +1814,9 @@ static int hci_sock_setsockopt(struct socket *sock, int level, int optname,
|
||||
|
||||
BT_DBG("sk %p, opt %d", sk, optname);
|
||||
|
||||
if (level != SOL_HCI)
|
||||
return -ENOPROTOOPT;
|
||||
|
||||
lock_sock(sk);
|
||||
|
||||
if (hci_pi(sk)->channel != HCI_CHANNEL_RAW) {
|
||||
@@ -1523,6 +1900,9 @@ static int hci_sock_getsockopt(struct socket *sock, int level, int optname,
|
||||
|
||||
BT_DBG("sk %p, opt %d", sk, optname);
|
||||
|
||||
if (level != SOL_HCI)
|
||||
return -ENOPROTOOPT;
|
||||
|
||||
if (get_user(len, optlen))
|
||||
return -EFAULT;
|
||||
|
||||
|
||||
@@ -11,6 +11,8 @@
|
||||
|
||||
#include "leds.h"
|
||||
|
||||
DEFINE_LED_TRIGGER(bt_power_led_trigger);
|
||||
|
||||
struct hci_basic_led_trigger {
|
||||
struct led_trigger led_trigger;
|
||||
struct hci_dev *hdev;
|
||||
@@ -24,6 +26,21 @@ void hci_leds_update_powered(struct hci_dev *hdev, bool enabled)
|
||||
if (hdev->power_led)
|
||||
led_trigger_event(hdev->power_led,
|
||||
enabled ? LED_FULL : LED_OFF);
|
||||
|
||||
if (!enabled) {
|
||||
struct hci_dev *d;
|
||||
|
||||
read_lock(&hci_dev_list_lock);
|
||||
|
||||
list_for_each_entry(d, &hci_dev_list, list) {
|
||||
if (test_bit(HCI_UP, &d->flags))
|
||||
enabled = true;
|
||||
}
|
||||
|
||||
read_unlock(&hci_dev_list_lock);
|
||||
}
|
||||
|
||||
led_trigger_event(bt_power_led_trigger, enabled ? LED_FULL : LED_OFF);
|
||||
}
|
||||
|
||||
static void power_activate(struct led_classdev *led_cdev)
|
||||
@@ -72,3 +89,13 @@ void hci_leds_init(struct hci_dev *hdev)
|
||||
/* initialize power_led */
|
||||
hdev->power_led = led_allocate_basic(hdev, power_activate, "power");
|
||||
}
|
||||
|
||||
void bt_leds_init(void)
|
||||
{
|
||||
led_trigger_register_simple("bluetooth-power", &bt_power_led_trigger);
|
||||
}
|
||||
|
||||
void bt_leds_cleanup(void)
|
||||
{
|
||||
led_trigger_unregister_simple(bt_power_led_trigger);
|
||||
}
|
||||
|
||||
@@ -7,10 +7,20 @@
|
||||
*/
|
||||
|
||||
#if IS_ENABLED(CONFIG_BT_LEDS)
|
||||
|
||||
void hci_leds_update_powered(struct hci_dev *hdev, bool enabled);
|
||||
void hci_leds_init(struct hci_dev *hdev);
|
||||
|
||||
void bt_leds_init(void);
|
||||
void bt_leds_cleanup(void);
|
||||
|
||||
#else
|
||||
|
||||
static inline void hci_leds_update_powered(struct hci_dev *hdev,
|
||||
bool enabled) {}
|
||||
static inline void hci_leds_init(struct hci_dev *hdev) {}
|
||||
|
||||
static inline void bt_leds_init(void) {}
|
||||
static inline void bt_leds_cleanup(void) {}
|
||||
|
||||
#endif
|
||||
|
||||
+277
-76
@@ -38,7 +38,7 @@
|
||||
#include "mgmt_util.h"
|
||||
|
||||
#define MGMT_VERSION 1
|
||||
#define MGMT_REVISION 13
|
||||
#define MGMT_REVISION 14
|
||||
|
||||
static const u16 mgmt_commands[] = {
|
||||
MGMT_OP_READ_INDEX_LIST,
|
||||
@@ -104,6 +104,8 @@ static const u16 mgmt_commands[] = {
|
||||
MGMT_OP_REMOVE_ADVERTISING,
|
||||
MGMT_OP_GET_ADV_SIZE_INFO,
|
||||
MGMT_OP_START_LIMITED_DISCOVERY,
|
||||
MGMT_OP_READ_EXT_INFO,
|
||||
MGMT_OP_SET_APPEARANCE,
|
||||
};
|
||||
|
||||
static const u16 mgmt_events[] = {
|
||||
@@ -141,6 +143,7 @@ static const u16 mgmt_events[] = {
|
||||
MGMT_EV_LOCAL_OOB_DATA_UPDATED,
|
||||
MGMT_EV_ADVERTISING_ADDED,
|
||||
MGMT_EV_ADVERTISING_REMOVED,
|
||||
MGMT_EV_EXT_INFO_CHANGED,
|
||||
};
|
||||
|
||||
static const u16 mgmt_untrusted_commands[] = {
|
||||
@@ -149,6 +152,7 @@ static const u16 mgmt_untrusted_commands[] = {
|
||||
MGMT_OP_READ_UNCONF_INDEX_LIST,
|
||||
MGMT_OP_READ_CONFIG_INFO,
|
||||
MGMT_OP_READ_EXT_INDEX_LIST,
|
||||
MGMT_OP_READ_EXT_INFO,
|
||||
};
|
||||
|
||||
static const u16 mgmt_untrusted_events[] = {
|
||||
@@ -162,6 +166,7 @@ static const u16 mgmt_untrusted_events[] = {
|
||||
MGMT_EV_NEW_CONFIG_OPTIONS,
|
||||
MGMT_EV_EXT_INDEX_ADDED,
|
||||
MGMT_EV_EXT_INDEX_REMOVED,
|
||||
MGMT_EV_EXT_INFO_CHANGED,
|
||||
};
|
||||
|
||||
#define CACHE_TIMEOUT msecs_to_jiffies(2 * 1000)
|
||||
@@ -256,13 +261,6 @@ static int mgmt_limited_event(u16 event, struct hci_dev *hdev, void *data,
|
||||
flag, skip_sk);
|
||||
}
|
||||
|
||||
static int mgmt_generic_event(u16 event, struct hci_dev *hdev, void *data,
|
||||
u16 len, struct sock *skip_sk)
|
||||
{
|
||||
return mgmt_send_event(event, hdev, HCI_CHANNEL_CONTROL, data, len,
|
||||
HCI_MGMT_GENERIC_EVENTS, skip_sk);
|
||||
}
|
||||
|
||||
static int mgmt_event(u16 event, struct hci_dev *hdev, void *data, u16 len,
|
||||
struct sock *skip_sk)
|
||||
{
|
||||
@@ -278,6 +276,14 @@ static u8 le_addr_type(u8 mgmt_addr_type)
|
||||
return ADDR_LE_DEV_RANDOM;
|
||||
}
|
||||
|
||||
void mgmt_fill_version_info(void *ver)
|
||||
{
|
||||
struct mgmt_rp_read_version *rp = ver;
|
||||
|
||||
rp->version = MGMT_VERSION;
|
||||
rp->revision = cpu_to_le16(MGMT_REVISION);
|
||||
}
|
||||
|
||||
static int read_version(struct sock *sk, struct hci_dev *hdev, void *data,
|
||||
u16 data_len)
|
||||
{
|
||||
@@ -285,8 +291,7 @@ static int read_version(struct sock *sk, struct hci_dev *hdev, void *data,
|
||||
|
||||
BT_DBG("sock %p", sk);
|
||||
|
||||
rp.version = MGMT_VERSION;
|
||||
rp.revision = cpu_to_le16(MGMT_REVISION);
|
||||
mgmt_fill_version_info(&rp);
|
||||
|
||||
return mgmt_cmd_complete(sk, MGMT_INDEX_NONE, MGMT_OP_READ_VERSION, 0,
|
||||
&rp, sizeof(rp));
|
||||
@@ -572,8 +577,8 @@ static int new_options(struct hci_dev *hdev, struct sock *skip)
|
||||
{
|
||||
__le32 options = get_missing_options(hdev);
|
||||
|
||||
return mgmt_generic_event(MGMT_EV_NEW_CONFIG_OPTIONS, hdev, &options,
|
||||
sizeof(options), skip);
|
||||
return mgmt_limited_event(MGMT_EV_NEW_CONFIG_OPTIONS, hdev, &options,
|
||||
sizeof(options), HCI_MGMT_OPTION_EVENTS, skip);
|
||||
}
|
||||
|
||||
static int send_options_rsp(struct sock *sk, u16 opcode, struct hci_dev *hdev)
|
||||
@@ -862,6 +867,107 @@ static int read_controller_info(struct sock *sk, struct hci_dev *hdev,
|
||||
sizeof(rp));
|
||||
}
|
||||
|
||||
static inline u16 eir_append_data(u8 *eir, u16 eir_len, u8 type, u8 *data,
|
||||
u8 data_len)
|
||||
{
|
||||
eir[eir_len++] = sizeof(type) + data_len;
|
||||
eir[eir_len++] = type;
|
||||
memcpy(&eir[eir_len], data, data_len);
|
||||
eir_len += data_len;
|
||||
|
||||
return eir_len;
|
||||
}
|
||||
|
||||
static inline u16 eir_append_le16(u8 *eir, u16 eir_len, u8 type, u16 data)
|
||||
{
|
||||
eir[eir_len++] = sizeof(type) + sizeof(data);
|
||||
eir[eir_len++] = type;
|
||||
put_unaligned_le16(data, &eir[eir_len]);
|
||||
eir_len += sizeof(data);
|
||||
|
||||
return eir_len;
|
||||
}
|
||||
|
||||
static u16 append_eir_data_to_buf(struct hci_dev *hdev, u8 *eir)
|
||||
{
|
||||
u16 eir_len = 0;
|
||||
size_t name_len;
|
||||
|
||||
if (hci_dev_test_flag(hdev, HCI_BREDR_ENABLED))
|
||||
eir_len = eir_append_data(eir, eir_len, EIR_CLASS_OF_DEV,
|
||||
hdev->dev_class, 3);
|
||||
|
||||
if (hci_dev_test_flag(hdev, HCI_LE_ENABLED))
|
||||
eir_len = eir_append_le16(eir, eir_len, EIR_APPEARANCE,
|
||||
hdev->appearance);
|
||||
|
||||
name_len = strlen(hdev->dev_name);
|
||||
eir_len = eir_append_data(eir, eir_len, EIR_NAME_COMPLETE,
|
||||
hdev->dev_name, name_len);
|
||||
|
||||
name_len = strlen(hdev->short_name);
|
||||
eir_len = eir_append_data(eir, eir_len, EIR_NAME_SHORT,
|
||||
hdev->short_name, name_len);
|
||||
|
||||
return eir_len;
|
||||
}
|
||||
|
||||
static int read_ext_controller_info(struct sock *sk, struct hci_dev *hdev,
|
||||
void *data, u16 data_len)
|
||||
{
|
||||
char buf[512];
|
||||
struct mgmt_rp_read_ext_info *rp = (void *)buf;
|
||||
u16 eir_len;
|
||||
|
||||
BT_DBG("sock %p %s", sk, hdev->name);
|
||||
|
||||
memset(&buf, 0, sizeof(buf));
|
||||
|
||||
hci_dev_lock(hdev);
|
||||
|
||||
bacpy(&rp->bdaddr, &hdev->bdaddr);
|
||||
|
||||
rp->version = hdev->hci_ver;
|
||||
rp->manufacturer = cpu_to_le16(hdev->manufacturer);
|
||||
|
||||
rp->supported_settings = cpu_to_le32(get_supported_settings(hdev));
|
||||
rp->current_settings = cpu_to_le32(get_current_settings(hdev));
|
||||
|
||||
|
||||
eir_len = append_eir_data_to_buf(hdev, rp->eir);
|
||||
rp->eir_len = cpu_to_le16(eir_len);
|
||||
|
||||
hci_dev_unlock(hdev);
|
||||
|
||||
/* If this command is called at least once, then the events
|
||||
* for class of device and local name changes are disabled
|
||||
* and only the new extended controller information event
|
||||
* is used.
|
||||
*/
|
||||
hci_sock_set_flag(sk, HCI_MGMT_EXT_INFO_EVENTS);
|
||||
hci_sock_clear_flag(sk, HCI_MGMT_DEV_CLASS_EVENTS);
|
||||
hci_sock_clear_flag(sk, HCI_MGMT_LOCAL_NAME_EVENTS);
|
||||
|
||||
return mgmt_cmd_complete(sk, hdev->id, MGMT_OP_READ_EXT_INFO, 0, rp,
|
||||
sizeof(*rp) + eir_len);
|
||||
}
|
||||
|
||||
static int ext_info_changed(struct hci_dev *hdev, struct sock *skip)
|
||||
{
|
||||
char buf[512];
|
||||
struct mgmt_ev_ext_info_changed *ev = (void *)buf;
|
||||
u16 eir_len;
|
||||
|
||||
memset(buf, 0, sizeof(buf));
|
||||
|
||||
eir_len = append_eir_data_to_buf(hdev, ev->eir);
|
||||
ev->eir_len = cpu_to_le16(eir_len);
|
||||
|
||||
return mgmt_limited_event(MGMT_EV_EXT_INFO_CHANGED, hdev, ev,
|
||||
sizeof(*ev) + eir_len,
|
||||
HCI_MGMT_EXT_INFO_EVENTS, skip);
|
||||
}
|
||||
|
||||
static int send_settings_rsp(struct sock *sk, u16 opcode, struct hci_dev *hdev)
|
||||
{
|
||||
__le32 settings = cpu_to_le32(get_current_settings(hdev));
|
||||
@@ -922,7 +1028,7 @@ static int clean_up_hci_state(struct hci_dev *hdev)
|
||||
hci_req_add(&req, HCI_OP_WRITE_SCAN_ENABLE, 1, &scan);
|
||||
}
|
||||
|
||||
hci_req_clear_adv_instance(hdev, NULL, 0x00, false);
|
||||
hci_req_clear_adv_instance(hdev, NULL, NULL, 0x00, false);
|
||||
|
||||
if (hci_dev_test_flag(hdev, HCI_LE_ADV))
|
||||
__hci_req_disable_advertising(&req);
|
||||
@@ -1000,8 +1106,8 @@ static int new_settings(struct hci_dev *hdev, struct sock *skip)
|
||||
{
|
||||
__le32 ev = cpu_to_le32(get_current_settings(hdev));
|
||||
|
||||
return mgmt_generic_event(MGMT_EV_NEW_SETTINGS, hdev, &ev,
|
||||
sizeof(ev), skip);
|
||||
return mgmt_limited_event(MGMT_EV_NEW_SETTINGS, hdev, &ev,
|
||||
sizeof(ev), HCI_MGMT_SETTING_EVENTS, skip);
|
||||
}
|
||||
|
||||
int mgmt_new_settings(struct hci_dev *hdev)
|
||||
@@ -1690,7 +1796,7 @@ static int set_le(struct sock *sk, struct hci_dev *hdev, void *data, u16 len)
|
||||
enabled = lmp_host_le_capable(hdev);
|
||||
|
||||
if (!val)
|
||||
hci_req_clear_adv_instance(hdev, NULL, 0x00, true);
|
||||
hci_req_clear_adv_instance(hdev, NULL, NULL, 0x00, true);
|
||||
|
||||
if (!hdev_is_powered(hdev) || val == enabled) {
|
||||
bool changed = false;
|
||||
@@ -2435,6 +2541,8 @@ static int send_pin_code_neg_reply(struct sock *sk, struct hci_dev *hdev,
|
||||
if (!cmd)
|
||||
return -ENOMEM;
|
||||
|
||||
cmd->cmd_complete = addr_cmd_complete;
|
||||
|
||||
err = hci_send_cmd(hdev, HCI_OP_PIN_CODE_NEG_REPLY,
|
||||
sizeof(cp->addr.bdaddr), &cp->addr.bdaddr);
|
||||
if (err < 0)
|
||||
@@ -2513,8 +2621,8 @@ static int set_io_capability(struct sock *sk, struct hci_dev *hdev, void *data,
|
||||
BT_DBG("");
|
||||
|
||||
if (cp->io_capability > SMP_IO_KEYBOARD_DISPLAY)
|
||||
return mgmt_cmd_complete(sk, hdev->id, MGMT_OP_SET_IO_CAPABILITY,
|
||||
MGMT_STATUS_INVALID_PARAMS, NULL, 0);
|
||||
return mgmt_cmd_status(sk, hdev->id, MGMT_OP_SET_IO_CAPABILITY,
|
||||
MGMT_STATUS_INVALID_PARAMS);
|
||||
|
||||
hci_dev_lock(hdev);
|
||||
|
||||
@@ -2932,6 +3040,35 @@ static int user_passkey_neg_reply(struct sock *sk, struct hci_dev *hdev,
|
||||
HCI_OP_USER_PASSKEY_NEG_REPLY, 0);
|
||||
}
|
||||
|
||||
static void adv_expire(struct hci_dev *hdev, u32 flags)
|
||||
{
|
||||
struct adv_info *adv_instance;
|
||||
struct hci_request req;
|
||||
int err;
|
||||
|
||||
adv_instance = hci_find_adv_instance(hdev, hdev->cur_adv_instance);
|
||||
if (!adv_instance)
|
||||
return;
|
||||
|
||||
/* stop if current instance doesn't need to be changed */
|
||||
if (!(adv_instance->flags & flags))
|
||||
return;
|
||||
|
||||
cancel_adv_timeout(hdev);
|
||||
|
||||
adv_instance = hci_get_next_instance(hdev, adv_instance->instance);
|
||||
if (!adv_instance)
|
||||
return;
|
||||
|
||||
hci_req_init(&req, hdev);
|
||||
err = __hci_req_schedule_adv_instance(&req, adv_instance->instance,
|
||||
true);
|
||||
if (err)
|
||||
return;
|
||||
|
||||
hci_req_run(&req, NULL);
|
||||
}
|
||||
|
||||
static void set_name_complete(struct hci_dev *hdev, u8 status, u16 opcode)
|
||||
{
|
||||
struct mgmt_cp_set_local_name *cp;
|
||||
@@ -2947,13 +3084,17 @@ static void set_name_complete(struct hci_dev *hdev, u8 status, u16 opcode)
|
||||
|
||||
cp = cmd->param;
|
||||
|
||||
if (status)
|
||||
if (status) {
|
||||
mgmt_cmd_status(cmd->sk, hdev->id, MGMT_OP_SET_LOCAL_NAME,
|
||||
mgmt_status(status));
|
||||
else
|
||||
} else {
|
||||
mgmt_cmd_complete(cmd->sk, hdev->id, MGMT_OP_SET_LOCAL_NAME, 0,
|
||||
cp, sizeof(*cp));
|
||||
|
||||
if (hci_dev_test_flag(hdev, HCI_LE_ADV))
|
||||
adv_expire(hdev, MGMT_ADV_FLAG_LOCAL_NAME);
|
||||
}
|
||||
|
||||
mgmt_pending_remove(cmd);
|
||||
|
||||
unlock:
|
||||
@@ -2993,8 +3134,9 @@ static int set_local_name(struct sock *sk, struct hci_dev *hdev, void *data,
|
||||
if (err < 0)
|
||||
goto failed;
|
||||
|
||||
err = mgmt_generic_event(MGMT_EV_LOCAL_NAME_CHANGED, hdev,
|
||||
data, len, sk);
|
||||
err = mgmt_limited_event(MGMT_EV_LOCAL_NAME_CHANGED, hdev, data,
|
||||
len, HCI_MGMT_LOCAL_NAME_EVENTS, sk);
|
||||
ext_info_changed(hdev, sk);
|
||||
|
||||
goto failed;
|
||||
}
|
||||
@@ -3017,7 +3159,7 @@ static int set_local_name(struct sock *sk, struct hci_dev *hdev, void *data,
|
||||
/* The name is stored in the scan response data and so
|
||||
* no need to udpate the advertising data here.
|
||||
*/
|
||||
if (lmp_le_capable(hdev))
|
||||
if (lmp_le_capable(hdev) && hci_dev_test_flag(hdev, HCI_ADVERTISING))
|
||||
__hci_req_update_scan_rsp_data(&req, hdev->cur_adv_instance);
|
||||
|
||||
err = hci_req_run(&req, set_name_complete);
|
||||
@@ -3029,6 +3171,40 @@ failed:
|
||||
return err;
|
||||
}
|
||||
|
||||
static int set_appearance(struct sock *sk, struct hci_dev *hdev, void *data,
|
||||
u16 len)
|
||||
{
|
||||
struct mgmt_cp_set_appearance *cp = data;
|
||||
u16 apperance;
|
||||
int err;
|
||||
|
||||
BT_DBG("");
|
||||
|
||||
if (!lmp_le_capable(hdev))
|
||||
return mgmt_cmd_status(sk, hdev->id, MGMT_OP_SET_APPEARANCE,
|
||||
MGMT_STATUS_NOT_SUPPORTED);
|
||||
|
||||
apperance = le16_to_cpu(cp->appearance);
|
||||
|
||||
hci_dev_lock(hdev);
|
||||
|
||||
if (hdev->appearance != apperance) {
|
||||
hdev->appearance = apperance;
|
||||
|
||||
if (hci_dev_test_flag(hdev, HCI_LE_ADV))
|
||||
adv_expire(hdev, MGMT_ADV_FLAG_APPEARANCE);
|
||||
|
||||
ext_info_changed(hdev, sk);
|
||||
}
|
||||
|
||||
err = mgmt_cmd_complete(sk, hdev->id, MGMT_OP_SET_APPEARANCE, 0, NULL,
|
||||
0);
|
||||
|
||||
hci_dev_unlock(hdev);
|
||||
|
||||
return err;
|
||||
}
|
||||
|
||||
static void read_local_oob_data_complete(struct hci_dev *hdev, u8 status,
|
||||
u16 opcode, struct sk_buff *skb)
|
||||
{
|
||||
@@ -4869,7 +5045,7 @@ static int clock_info_cmd_complete(struct mgmt_pending_cmd *cmd, u8 status)
|
||||
int err;
|
||||
|
||||
memset(&rp, 0, sizeof(rp));
|
||||
memcpy(&rp.addr, &cmd->param, sizeof(rp.addr));
|
||||
memcpy(&rp.addr, cmd->param, sizeof(rp.addr));
|
||||
|
||||
if (status)
|
||||
goto complete;
|
||||
@@ -5501,17 +5677,6 @@ unlock:
|
||||
return err;
|
||||
}
|
||||
|
||||
static inline u16 eir_append_data(u8 *eir, u16 eir_len, u8 type, u8 *data,
|
||||
u8 data_len)
|
||||
{
|
||||
eir[eir_len++] = sizeof(type) + data_len;
|
||||
eir[eir_len++] = type;
|
||||
memcpy(&eir[eir_len], data, data_len);
|
||||
eir_len += data_len;
|
||||
|
||||
return eir_len;
|
||||
}
|
||||
|
||||
static void read_local_oob_ext_data_complete(struct hci_dev *hdev, u8 status,
|
||||
u16 opcode, struct sk_buff *skb)
|
||||
{
|
||||
@@ -5815,6 +5980,8 @@ static u32 get_supported_adv_flags(struct hci_dev *hdev)
|
||||
flags |= MGMT_ADV_FLAG_DISCOV;
|
||||
flags |= MGMT_ADV_FLAG_LIMITED_DISCOV;
|
||||
flags |= MGMT_ADV_FLAG_MANAGED_FLAGS;
|
||||
flags |= MGMT_ADV_FLAG_APPEARANCE;
|
||||
flags |= MGMT_ADV_FLAG_LOCAL_NAME;
|
||||
|
||||
if (hdev->adv_tx_power != HCI_TX_POWER_INVALID)
|
||||
flags |= MGMT_ADV_FLAG_TX_POWER;
|
||||
@@ -5871,28 +6038,59 @@ static int read_adv_features(struct sock *sk, struct hci_dev *hdev,
|
||||
return err;
|
||||
}
|
||||
|
||||
static bool tlv_data_is_valid(struct hci_dev *hdev, u32 adv_flags, u8 *data,
|
||||
u8 len, bool is_adv_data)
|
||||
static u8 tlv_data_max_len(u32 adv_flags, bool is_adv_data)
|
||||
{
|
||||
u8 max_len = HCI_MAX_AD_LENGTH;
|
||||
int i, cur_len;
|
||||
bool flags_managed = false;
|
||||
bool tx_power_managed = false;
|
||||
|
||||
if (is_adv_data) {
|
||||
if (adv_flags & (MGMT_ADV_FLAG_DISCOV |
|
||||
MGMT_ADV_FLAG_LIMITED_DISCOV |
|
||||
MGMT_ADV_FLAG_MANAGED_FLAGS)) {
|
||||
flags_managed = true;
|
||||
MGMT_ADV_FLAG_MANAGED_FLAGS))
|
||||
max_len -= 3;
|
||||
}
|
||||
|
||||
if (adv_flags & MGMT_ADV_FLAG_TX_POWER) {
|
||||
tx_power_managed = true;
|
||||
if (adv_flags & MGMT_ADV_FLAG_TX_POWER)
|
||||
max_len -= 3;
|
||||
}
|
||||
} else {
|
||||
/* at least 1 byte of name should fit in */
|
||||
if (adv_flags & MGMT_ADV_FLAG_LOCAL_NAME)
|
||||
max_len -= 3;
|
||||
|
||||
if (adv_flags & (MGMT_ADV_FLAG_APPEARANCE))
|
||||
max_len -= 4;
|
||||
}
|
||||
|
||||
return max_len;
|
||||
}
|
||||
|
||||
static bool flags_managed(u32 adv_flags)
|
||||
{
|
||||
return adv_flags & (MGMT_ADV_FLAG_DISCOV |
|
||||
MGMT_ADV_FLAG_LIMITED_DISCOV |
|
||||
MGMT_ADV_FLAG_MANAGED_FLAGS);
|
||||
}
|
||||
|
||||
static bool tx_power_managed(u32 adv_flags)
|
||||
{
|
||||
return adv_flags & MGMT_ADV_FLAG_TX_POWER;
|
||||
}
|
||||
|
||||
static bool name_managed(u32 adv_flags)
|
||||
{
|
||||
return adv_flags & MGMT_ADV_FLAG_LOCAL_NAME;
|
||||
}
|
||||
|
||||
static bool appearance_managed(u32 adv_flags)
|
||||
{
|
||||
return adv_flags & MGMT_ADV_FLAG_APPEARANCE;
|
||||
}
|
||||
|
||||
static bool tlv_data_is_valid(u32 adv_flags, u8 *data, u8 len, bool is_adv_data)
|
||||
{
|
||||
int i, cur_len;
|
||||
u8 max_len;
|
||||
|
||||
max_len = tlv_data_max_len(adv_flags, is_adv_data);
|
||||
|
||||
if (len > max_len)
|
||||
return false;
|
||||
|
||||
@@ -5900,10 +6098,21 @@ static bool tlv_data_is_valid(struct hci_dev *hdev, u32 adv_flags, u8 *data,
|
||||
for (i = 0, cur_len = 0; i < len; i += (cur_len + 1)) {
|
||||
cur_len = data[i];
|
||||
|
||||
if (flags_managed && data[i + 1] == EIR_FLAGS)
|
||||
if (data[i + 1] == EIR_FLAGS &&
|
||||
(!is_adv_data || flags_managed(adv_flags)))
|
||||
return false;
|
||||
|
||||
if (tx_power_managed && data[i + 1] == EIR_TX_POWER)
|
||||
if (data[i + 1] == EIR_TX_POWER && tx_power_managed(adv_flags))
|
||||
return false;
|
||||
|
||||
if (data[i + 1] == EIR_NAME_COMPLETE && name_managed(adv_flags))
|
||||
return false;
|
||||
|
||||
if (data[i + 1] == EIR_NAME_SHORT && name_managed(adv_flags))
|
||||
return false;
|
||||
|
||||
if (data[i + 1] == EIR_APPEARANCE &&
|
||||
appearance_managed(adv_flags))
|
||||
return false;
|
||||
|
||||
/* If the current field length would exceed the total data
|
||||
@@ -6027,8 +6236,8 @@ static int add_advertising(struct sock *sk, struct hci_dev *hdev,
|
||||
goto unlock;
|
||||
}
|
||||
|
||||
if (!tlv_data_is_valid(hdev, flags, cp->data, cp->adv_data_len, true) ||
|
||||
!tlv_data_is_valid(hdev, flags, cp->data + cp->adv_data_len,
|
||||
if (!tlv_data_is_valid(flags, cp->data, cp->adv_data_len, true) ||
|
||||
!tlv_data_is_valid(flags, cp->data + cp->adv_data_len,
|
||||
cp->scan_rsp_len, false)) {
|
||||
err = mgmt_cmd_status(sk, hdev->id, MGMT_OP_ADD_ADVERTISING,
|
||||
MGMT_STATUS_INVALID_PARAMS);
|
||||
@@ -6175,7 +6384,7 @@ static int remove_advertising(struct sock *sk, struct hci_dev *hdev,
|
||||
|
||||
hci_req_init(&req, hdev);
|
||||
|
||||
hci_req_clear_adv_instance(hdev, &req, cp->instance, true);
|
||||
hci_req_clear_adv_instance(hdev, sk, &req, cp->instance, true);
|
||||
|
||||
if (list_empty(&hdev->adv_instances))
|
||||
__hci_req_disable_advertising(&req);
|
||||
@@ -6211,23 +6420,6 @@ unlock:
|
||||
return err;
|
||||
}
|
||||
|
||||
static u8 tlv_data_max_len(u32 adv_flags, bool is_adv_data)
|
||||
{
|
||||
u8 max_len = HCI_MAX_AD_LENGTH;
|
||||
|
||||
if (is_adv_data) {
|
||||
if (adv_flags & (MGMT_ADV_FLAG_DISCOV |
|
||||
MGMT_ADV_FLAG_LIMITED_DISCOV |
|
||||
MGMT_ADV_FLAG_MANAGED_FLAGS))
|
||||
max_len -= 3;
|
||||
|
||||
if (adv_flags & MGMT_ADV_FLAG_TX_POWER)
|
||||
max_len -= 3;
|
||||
}
|
||||
|
||||
return max_len;
|
||||
}
|
||||
|
||||
static int get_adv_size_info(struct sock *sk, struct hci_dev *hdev,
|
||||
void *data, u16 data_len)
|
||||
{
|
||||
@@ -6356,6 +6548,9 @@ static const struct hci_mgmt_handler mgmt_handlers[] = {
|
||||
{ remove_advertising, MGMT_REMOVE_ADVERTISING_SIZE },
|
||||
{ get_adv_size_info, MGMT_GET_ADV_SIZE_INFO_SIZE },
|
||||
{ start_limited_discovery, MGMT_START_DISCOVERY_SIZE },
|
||||
{ read_ext_controller_info,MGMT_READ_EXT_INFO_SIZE,
|
||||
HCI_MGMT_UNTRUSTED },
|
||||
{ set_appearance, MGMT_SET_APPEARANCE_SIZE },
|
||||
};
|
||||
|
||||
void mgmt_index_added(struct hci_dev *hdev)
|
||||
@@ -6494,9 +6689,12 @@ void __mgmt_power_off(struct hci_dev *hdev)
|
||||
|
||||
mgmt_pending_foreach(0, hdev, cmd_complete_rsp, &status);
|
||||
|
||||
if (memcmp(hdev->dev_class, zero_cod, sizeof(zero_cod)) != 0)
|
||||
mgmt_generic_event(MGMT_EV_CLASS_OF_DEV_CHANGED, hdev,
|
||||
zero_cod, sizeof(zero_cod), NULL);
|
||||
if (memcmp(hdev->dev_class, zero_cod, sizeof(zero_cod)) != 0) {
|
||||
mgmt_limited_event(MGMT_EV_CLASS_OF_DEV_CHANGED, hdev,
|
||||
zero_cod, sizeof(zero_cod),
|
||||
HCI_MGMT_DEV_CLASS_EVENTS, NULL);
|
||||
ext_info_changed(hdev, NULL);
|
||||
}
|
||||
|
||||
new_settings(hdev, match.sk);
|
||||
|
||||
@@ -7092,9 +7290,11 @@ void mgmt_set_class_of_dev_complete(struct hci_dev *hdev, u8 *dev_class,
|
||||
mgmt_pending_foreach(MGMT_OP_ADD_UUID, hdev, sk_lookup, &match);
|
||||
mgmt_pending_foreach(MGMT_OP_REMOVE_UUID, hdev, sk_lookup, &match);
|
||||
|
||||
if (!status)
|
||||
mgmt_generic_event(MGMT_EV_CLASS_OF_DEV_CHANGED, hdev,
|
||||
dev_class, 3, NULL);
|
||||
if (!status) {
|
||||
mgmt_limited_event(MGMT_EV_CLASS_OF_DEV_CHANGED, hdev, dev_class,
|
||||
3, HCI_MGMT_DEV_CLASS_EVENTS, NULL);
|
||||
ext_info_changed(hdev, NULL);
|
||||
}
|
||||
|
||||
if (match.sk)
|
||||
sock_put(match.sk);
|
||||
@@ -7123,8 +7323,9 @@ void mgmt_set_local_name_complete(struct hci_dev *hdev, u8 *name, u8 status)
|
||||
return;
|
||||
}
|
||||
|
||||
mgmt_generic_event(MGMT_EV_LOCAL_NAME_CHANGED, hdev, &ev, sizeof(ev),
|
||||
cmd ? cmd->sk : NULL);
|
||||
mgmt_limited_event(MGMT_EV_LOCAL_NAME_CHANGED, hdev, &ev, sizeof(ev),
|
||||
HCI_MGMT_LOCAL_NAME_EVENTS, cmd ? cmd->sk : NULL);
|
||||
ext_info_changed(hdev, cmd ? cmd->sk : NULL);
|
||||
}
|
||||
|
||||
static inline bool has_uuid(u8 *uuid, u16 uuid_count, u8 (*uuids)[16])
|
||||
|
||||
@@ -21,12 +21,41 @@
|
||||
SOFTWARE IS DISCLAIMED.
|
||||
*/
|
||||
|
||||
#include <asm/unaligned.h>
|
||||
|
||||
#include <net/bluetooth/bluetooth.h>
|
||||
#include <net/bluetooth/hci_core.h>
|
||||
#include <net/bluetooth/hci_mon.h>
|
||||
#include <net/bluetooth/mgmt.h>
|
||||
|
||||
#include "mgmt_util.h"
|
||||
|
||||
static struct sk_buff *create_monitor_ctrl_event(__le16 index, u32 cookie,
|
||||
u16 opcode, u16 len, void *buf)
|
||||
{
|
||||
struct hci_mon_hdr *hdr;
|
||||
struct sk_buff *skb;
|
||||
|
||||
skb = bt_skb_alloc(6 + len, GFP_ATOMIC);
|
||||
if (!skb)
|
||||
return NULL;
|
||||
|
||||
put_unaligned_le32(cookie, skb_put(skb, 4));
|
||||
put_unaligned_le16(opcode, skb_put(skb, 2));
|
||||
|
||||
if (buf)
|
||||
memcpy(skb_put(skb, len), buf, len);
|
||||
|
||||
__net_timestamp(skb);
|
||||
|
||||
hdr = (void *)skb_push(skb, HCI_MON_HDR_SIZE);
|
||||
hdr->opcode = cpu_to_le16(HCI_MON_CTRL_EVENT);
|
||||
hdr->index = index;
|
||||
hdr->len = cpu_to_le16(skb->len - HCI_MON_HDR_SIZE);
|
||||
|
||||
return skb;
|
||||
}
|
||||
|
||||
int mgmt_send_event(u16 event, struct hci_dev *hdev, unsigned short channel,
|
||||
void *data, u16 data_len, int flag, struct sock *skip_sk)
|
||||
{
|
||||
@@ -52,14 +81,18 @@ int mgmt_send_event(u16 event, struct hci_dev *hdev, unsigned short channel,
|
||||
__net_timestamp(skb);
|
||||
|
||||
hci_send_to_channel(channel, skb, flag, skip_sk);
|
||||
kfree_skb(skb);
|
||||
|
||||
if (channel == HCI_CHANNEL_CONTROL)
|
||||
hci_send_monitor_ctrl_event(hdev, event, data, data_len,
|
||||
skb_get_ktime(skb), flag, skip_sk);
|
||||
|
||||
kfree_skb(skb);
|
||||
return 0;
|
||||
}
|
||||
|
||||
int mgmt_cmd_status(struct sock *sk, u16 index, u16 cmd, u8 status)
|
||||
{
|
||||
struct sk_buff *skb;
|
||||
struct sk_buff *skb, *mskb;
|
||||
struct mgmt_hdr *hdr;
|
||||
struct mgmt_ev_cmd_status *ev;
|
||||
int err;
|
||||
@@ -80,17 +113,30 @@ int mgmt_cmd_status(struct sock *sk, u16 index, u16 cmd, u8 status)
|
||||
ev->status = status;
|
||||
ev->opcode = cpu_to_le16(cmd);
|
||||
|
||||
mskb = create_monitor_ctrl_event(hdr->index, hci_sock_get_cookie(sk),
|
||||
MGMT_EV_CMD_STATUS, sizeof(*ev), ev);
|
||||
if (mskb)
|
||||
skb->tstamp = mskb->tstamp;
|
||||
else
|
||||
__net_timestamp(skb);
|
||||
|
||||
err = sock_queue_rcv_skb(sk, skb);
|
||||
if (err < 0)
|
||||
kfree_skb(skb);
|
||||
|
||||
if (mskb) {
|
||||
hci_send_to_channel(HCI_CHANNEL_MONITOR, mskb,
|
||||
HCI_SOCK_TRUSTED, NULL);
|
||||
kfree_skb(mskb);
|
||||
}
|
||||
|
||||
return err;
|
||||
}
|
||||
|
||||
int mgmt_cmd_complete(struct sock *sk, u16 index, u16 cmd, u8 status,
|
||||
void *rp, size_t rp_len)
|
||||
{
|
||||
struct sk_buff *skb;
|
||||
struct sk_buff *skb, *mskb;
|
||||
struct mgmt_hdr *hdr;
|
||||
struct mgmt_ev_cmd_complete *ev;
|
||||
int err;
|
||||
@@ -114,10 +160,24 @@ int mgmt_cmd_complete(struct sock *sk, u16 index, u16 cmd, u8 status,
|
||||
if (rp)
|
||||
memcpy(ev->data, rp, rp_len);
|
||||
|
||||
mskb = create_monitor_ctrl_event(hdr->index, hci_sock_get_cookie(sk),
|
||||
MGMT_EV_CMD_COMPLETE,
|
||||
sizeof(*ev) + rp_len, ev);
|
||||
if (mskb)
|
||||
skb->tstamp = mskb->tstamp;
|
||||
else
|
||||
__net_timestamp(skb);
|
||||
|
||||
err = sock_queue_rcv_skb(sk, skb);
|
||||
if (err < 0)
|
||||
kfree_skb(skb);
|
||||
|
||||
if (mskb) {
|
||||
hci_send_to_channel(HCI_CHANNEL_MONITOR, mskb,
|
||||
HCI_SOCK_TRUSTED, NULL);
|
||||
kfree_skb(mskb);
|
||||
}
|
||||
|
||||
return err;
|
||||
}
|
||||
|
||||
|
||||
+4
-1
@@ -3387,7 +3387,10 @@ int smp_register(struct hci_dev *hdev)
|
||||
if (!lmp_sc_capable(hdev)) {
|
||||
debugfs_create_file("force_bredr_smp", 0644, hdev->debugfs,
|
||||
hdev, &force_bredr_smp_fops);
|
||||
return 0;
|
||||
|
||||
/* Flag can be already set here (due to power toggle) */
|
||||
if (!hci_dev_test_flag(hdev, HCI_FORCE_BREDR_SMP))
|
||||
return 0;
|
||||
}
|
||||
|
||||
if (WARN_ON(hdev->smp_bredr_data)) {
|
||||
|
||||
+4
-2
@@ -227,9 +227,11 @@ static int __init br_init(void)
|
||||
br_fdb_test_addr_hook = br_fdb_test_addr;
|
||||
#endif
|
||||
|
||||
pr_info("bridge: automatic filtering via arp/ip/ip6tables has been "
|
||||
"deprecated. Update your scripts to load br_netfilter if you "
|
||||
#if IS_MODULE(CONFIG_BRIDGE_NETFILTER)
|
||||
pr_info("bridge: filtering via arp/ip/ip6tables is no longer available "
|
||||
"by default. Update your scripts to load br_netfilter if you "
|
||||
"need this.\n");
|
||||
#endif
|
||||
|
||||
return 0;
|
||||
|
||||
|
||||
@@ -30,6 +30,7 @@
|
||||
#include <linux/netfilter_ipv6.h>
|
||||
#include <linux/netfilter_arp.h>
|
||||
#include <linux/in_route.h>
|
||||
#include <linux/rculist.h>
|
||||
#include <linux/inetdevice.h>
|
||||
|
||||
#include <net/ip.h>
|
||||
@@ -395,11 +396,10 @@ bridged_dnat:
|
||||
skb->dev = nf_bridge->physindev;
|
||||
nf_bridge_update_protocol(skb);
|
||||
nf_bridge_push_encap_header(skb);
|
||||
NF_HOOK_THRESH(NFPROTO_BRIDGE,
|
||||
NF_BR_PRE_ROUTING,
|
||||
net, sk, skb, skb->dev, NULL,
|
||||
br_nf_pre_routing_finish_bridge,
|
||||
1);
|
||||
br_nf_hook_thresh(NF_BR_PRE_ROUTING,
|
||||
net, sk, skb, skb->dev,
|
||||
NULL,
|
||||
br_nf_pre_routing_finish);
|
||||
return 0;
|
||||
}
|
||||
ether_addr_copy(eth_hdr(skb)->h_dest, dev->dev_addr);
|
||||
@@ -417,10 +417,8 @@ bridged_dnat:
|
||||
skb->dev = nf_bridge->physindev;
|
||||
nf_bridge_update_protocol(skb);
|
||||
nf_bridge_push_encap_header(skb);
|
||||
NF_HOOK_THRESH(NFPROTO_BRIDGE, NF_BR_PRE_ROUTING, net, sk, skb,
|
||||
skb->dev, NULL,
|
||||
br_handle_frame_finish, 1);
|
||||
|
||||
br_nf_hook_thresh(NF_BR_PRE_ROUTING, net, sk, skb, skb->dev, NULL,
|
||||
br_handle_frame_finish);
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -992,6 +990,43 @@ static struct notifier_block brnf_notifier __read_mostly = {
|
||||
.notifier_call = brnf_device_event,
|
||||
};
|
||||
|
||||
/* recursively invokes nf_hook_slow (again), skipping already-called
|
||||
* hooks (< NF_BR_PRI_BRNF).
|
||||
*
|
||||
* Called with rcu read lock held.
|
||||
*/
|
||||
int br_nf_hook_thresh(unsigned int hook, struct net *net,
|
||||
struct sock *sk, struct sk_buff *skb,
|
||||
struct net_device *indev,
|
||||
struct net_device *outdev,
|
||||
int (*okfn)(struct net *, struct sock *,
|
||||
struct sk_buff *))
|
||||
{
|
||||
struct nf_hook_entry *elem;
|
||||
struct nf_hook_state state;
|
||||
int ret;
|
||||
|
||||
elem = rcu_dereference(net->nf.hooks[NFPROTO_BRIDGE][hook]);
|
||||
|
||||
while (elem && (elem->ops.priority <= NF_BR_PRI_BRNF))
|
||||
elem = rcu_dereference(elem->next);
|
||||
|
||||
if (!elem)
|
||||
return okfn(net, sk, skb);
|
||||
|
||||
/* We may already have this, but read-locks nest anyway */
|
||||
rcu_read_lock();
|
||||
nf_hook_state_init(&state, elem, hook, NF_BR_PRI_BRNF + 1,
|
||||
NFPROTO_BRIDGE, indev, outdev, sk, net, okfn);
|
||||
|
||||
ret = nf_hook_slow(skb, &state);
|
||||
rcu_read_unlock();
|
||||
if (ret == 1)
|
||||
ret = okfn(net, sk, skb);
|
||||
|
||||
return ret;
|
||||
}
|
||||
|
||||
#ifdef CONFIG_SYSCTL
|
||||
static
|
||||
int brnf_sysctl_call_tables(struct ctl_table *ctl, int write,
|
||||
|
||||
@@ -187,10 +187,9 @@ static int br_nf_pre_routing_finish_ipv6(struct net *net, struct sock *sk, struc
|
||||
skb->dev = nf_bridge->physindev;
|
||||
nf_bridge_update_protocol(skb);
|
||||
nf_bridge_push_encap_header(skb);
|
||||
NF_HOOK_THRESH(NFPROTO_BRIDGE, NF_BR_PRE_ROUTING,
|
||||
net, sk, skb, skb->dev, NULL,
|
||||
br_nf_pre_routing_finish_bridge,
|
||||
1);
|
||||
br_nf_hook_thresh(NF_BR_PRE_ROUTING,
|
||||
net, sk, skb, skb->dev, NULL,
|
||||
br_nf_pre_routing_finish_bridge);
|
||||
return 0;
|
||||
}
|
||||
ether_addr_copy(eth_hdr(skb)->h_dest, dev->dev_addr);
|
||||
@@ -207,9 +206,8 @@ static int br_nf_pre_routing_finish_ipv6(struct net *net, struct sock *sk, struc
|
||||
skb->dev = nf_bridge->physindev;
|
||||
nf_bridge_update_protocol(skb);
|
||||
nf_bridge_push_encap_header(skb);
|
||||
NF_HOOK_THRESH(NFPROTO_BRIDGE, NF_BR_PRE_ROUTING, net, sk, skb,
|
||||
skb->dev, NULL,
|
||||
br_handle_frame_finish, 1);
|
||||
br_nf_hook_thresh(NF_BR_PRE_ROUTING, net, sk, skb,
|
||||
skb->dev, NULL, br_handle_frame_finish);
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -91,7 +91,7 @@ ebt_log_packet(struct net *net, u_int8_t pf, unsigned int hooknum,
|
||||
if (loginfo->type == NF_LOG_TYPE_LOG)
|
||||
bitmask = loginfo->u.log.logflags;
|
||||
else
|
||||
bitmask = NF_LOG_MASK;
|
||||
bitmask = NF_LOG_DEFAULT_MASK;
|
||||
|
||||
if ((bitmask & EBT_LOG_IP) && eth_hdr(skb)->h_proto ==
|
||||
htons(ETH_P_IP)) {
|
||||
|
||||
@@ -24,7 +24,7 @@ ebt_redirect_tg(struct sk_buff *skb, const struct xt_action_param *par)
|
||||
return EBT_DROP;
|
||||
|
||||
if (par->hooknum != NF_BR_BROUTING)
|
||||
/* rcu_read_lock()ed by nf_hook_slow */
|
||||
/* rcu_read_lock()ed by nf_hook_thresh */
|
||||
ether_addr_copy(eth_hdr(skb)->h_dest,
|
||||
br_port_get_rcu(par->in)->br->dev->dev_addr);
|
||||
else
|
||||
|
||||
@@ -146,7 +146,7 @@ ebt_basic_match(const struct ebt_entry *e, const struct sk_buff *skb,
|
||||
return 1;
|
||||
if (NF_INVF(e, EBT_IOUT, ebt_dev_check(e->out, out)))
|
||||
return 1;
|
||||
/* rcu_read_lock()ed by nf_hook_slow */
|
||||
/* rcu_read_lock()ed by nf_hook_thresh */
|
||||
if (in && (p = br_port_get_rcu(in)) != NULL &&
|
||||
NF_INVF(e, EBT_ILOGICALIN,
|
||||
ebt_dev_check(e->logical_in, p->br->dev)))
|
||||
|
||||
@@ -13,79 +13,11 @@
|
||||
#include <linux/module.h>
|
||||
#include <linux/netfilter_bridge.h>
|
||||
#include <net/netfilter/nf_tables.h>
|
||||
#include <net/netfilter/nf_tables_bridge.h>
|
||||
#include <linux/ip.h>
|
||||
#include <linux/ipv6.h>
|
||||
#include <net/netfilter/nf_tables_ipv4.h>
|
||||
#include <net/netfilter/nf_tables_ipv6.h>
|
||||
|
||||
int nft_bridge_iphdr_validate(struct sk_buff *skb)
|
||||
{
|
||||
struct iphdr *iph;
|
||||
u32 len;
|
||||
|
||||
if (!pskb_may_pull(skb, sizeof(struct iphdr)))
|
||||
return 0;
|
||||
|
||||
iph = ip_hdr(skb);
|
||||
if (iph->ihl < 5 || iph->version != 4)
|
||||
return 0;
|
||||
|
||||
len = ntohs(iph->tot_len);
|
||||
if (skb->len < len)
|
||||
return 0;
|
||||
else if (len < (iph->ihl*4))
|
||||
return 0;
|
||||
|
||||
if (!pskb_may_pull(skb, iph->ihl*4))
|
||||
return 0;
|
||||
|
||||
return 1;
|
||||
}
|
||||
EXPORT_SYMBOL_GPL(nft_bridge_iphdr_validate);
|
||||
|
||||
int nft_bridge_ip6hdr_validate(struct sk_buff *skb)
|
||||
{
|
||||
struct ipv6hdr *hdr;
|
||||
u32 pkt_len;
|
||||
|
||||
if (!pskb_may_pull(skb, sizeof(struct ipv6hdr)))
|
||||
return 0;
|
||||
|
||||
hdr = ipv6_hdr(skb);
|
||||
if (hdr->version != 6)
|
||||
return 0;
|
||||
|
||||
pkt_len = ntohs(hdr->payload_len);
|
||||
if (pkt_len + sizeof(struct ipv6hdr) > skb->len)
|
||||
return 0;
|
||||
|
||||
return 1;
|
||||
}
|
||||
EXPORT_SYMBOL_GPL(nft_bridge_ip6hdr_validate);
|
||||
|
||||
static inline void nft_bridge_set_pktinfo_ipv4(struct nft_pktinfo *pkt,
|
||||
struct sk_buff *skb,
|
||||
const struct nf_hook_state *state)
|
||||
{
|
||||
if (nft_bridge_iphdr_validate(skb))
|
||||
nft_set_pktinfo_ipv4(pkt, skb, state);
|
||||
else
|
||||
nft_set_pktinfo(pkt, skb, state);
|
||||
}
|
||||
|
||||
static inline void nft_bridge_set_pktinfo_ipv6(struct nft_pktinfo *pkt,
|
||||
struct sk_buff *skb,
|
||||
const struct nf_hook_state *state)
|
||||
{
|
||||
#if IS_ENABLED(CONFIG_IPV6)
|
||||
if (nft_bridge_ip6hdr_validate(skb) &&
|
||||
nft_set_pktinfo_ipv6(pkt, skb, state) == 0)
|
||||
return;
|
||||
#endif
|
||||
nft_set_pktinfo(pkt, skb, state);
|
||||
}
|
||||
|
||||
static unsigned int
|
||||
nft_do_chain_bridge(void *priv,
|
||||
struct sk_buff *skb,
|
||||
@@ -95,13 +27,13 @@ nft_do_chain_bridge(void *priv,
|
||||
|
||||
switch (eth_hdr(skb)->h_proto) {
|
||||
case htons(ETH_P_IP):
|
||||
nft_bridge_set_pktinfo_ipv4(&pkt, skb, state);
|
||||
nft_set_pktinfo_ipv4_validate(&pkt, skb, state);
|
||||
break;
|
||||
case htons(ETH_P_IPV6):
|
||||
nft_bridge_set_pktinfo_ipv6(&pkt, skb, state);
|
||||
nft_set_pktinfo_ipv6_validate(&pkt, skb, state);
|
||||
break;
|
||||
default:
|
||||
nft_set_pktinfo(&pkt, skb, state);
|
||||
nft_set_pktinfo_unspec(&pkt, skb, state);
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -207,12 +139,20 @@ static int __init nf_tables_bridge_init(void)
|
||||
int ret;
|
||||
|
||||
nf_register_afinfo(&nf_br_afinfo);
|
||||
nft_register_chain_type(&filter_bridge);
|
||||
ret = nft_register_chain_type(&filter_bridge);
|
||||
if (ret < 0)
|
||||
goto err1;
|
||||
|
||||
ret = register_pernet_subsys(&nf_tables_bridge_net_ops);
|
||||
if (ret < 0) {
|
||||
nft_unregister_chain_type(&filter_bridge);
|
||||
nf_unregister_afinfo(&nf_br_afinfo);
|
||||
}
|
||||
if (ret < 0)
|
||||
goto err2;
|
||||
|
||||
return ret;
|
||||
|
||||
err2:
|
||||
nft_unregister_chain_type(&filter_bridge);
|
||||
err1:
|
||||
nf_unregister_afinfo(&nf_br_afinfo);
|
||||
return ret;
|
||||
}
|
||||
|
||||
|
||||
@@ -14,7 +14,6 @@
|
||||
#include <linux/netfilter/nf_tables.h>
|
||||
#include <net/netfilter/nf_tables.h>
|
||||
#include <net/netfilter/nft_reject.h>
|
||||
#include <net/netfilter/nf_tables_bridge.h>
|
||||
#include <net/netfilter/ipv4/nf_reject.h>
|
||||
#include <net/netfilter/ipv6/nf_reject.h>
|
||||
#include <linux/ip.h>
|
||||
@@ -37,6 +36,30 @@ static void nft_reject_br_push_etherhdr(struct sk_buff *oldskb,
|
||||
skb_pull(nskb, ETH_HLEN);
|
||||
}
|
||||
|
||||
static int nft_bridge_iphdr_validate(struct sk_buff *skb)
|
||||
{
|
||||
struct iphdr *iph;
|
||||
u32 len;
|
||||
|
||||
if (!pskb_may_pull(skb, sizeof(struct iphdr)))
|
||||
return 0;
|
||||
|
||||
iph = ip_hdr(skb);
|
||||
if (iph->ihl < 5 || iph->version != 4)
|
||||
return 0;
|
||||
|
||||
len = ntohs(iph->tot_len);
|
||||
if (skb->len < len)
|
||||
return 0;
|
||||
else if (len < (iph->ihl*4))
|
||||
return 0;
|
||||
|
||||
if (!pskb_may_pull(skb, iph->ihl*4))
|
||||
return 0;
|
||||
|
||||
return 1;
|
||||
}
|
||||
|
||||
/* We cannot use oldskb->dev, it can be either bridge device (NF_BRIDGE INPUT)
|
||||
* or the bridge port (NF_BRIDGE PREROUTING).
|
||||
*/
|
||||
@@ -143,6 +166,25 @@ static void nft_reject_br_send_v4_unreach(struct net *net,
|
||||
br_forward(br_port_get_rcu(dev), nskb, false, true);
|
||||
}
|
||||
|
||||
static int nft_bridge_ip6hdr_validate(struct sk_buff *skb)
|
||||
{
|
||||
struct ipv6hdr *hdr;
|
||||
u32 pkt_len;
|
||||
|
||||
if (!pskb_may_pull(skb, sizeof(struct ipv6hdr)))
|
||||
return 0;
|
||||
|
||||
hdr = ipv6_hdr(skb);
|
||||
if (hdr->version != 6)
|
||||
return 0;
|
||||
|
||||
pkt_len = ntohs(hdr->payload_len);
|
||||
if (pkt_len + sizeof(struct ipv6hdr) > skb->len)
|
||||
return 0;
|
||||
|
||||
return 1;
|
||||
}
|
||||
|
||||
static void nft_reject_br_send_v6_tcp_reset(struct net *net,
|
||||
struct sk_buff *oldskb,
|
||||
const struct net_device *dev,
|
||||
|
||||
+43
-32
@@ -4055,12 +4055,17 @@ static inline int nf_ingress(struct sk_buff *skb, struct packet_type **pt_prev,
|
||||
{
|
||||
#ifdef CONFIG_NETFILTER_INGRESS
|
||||
if (nf_hook_ingress_active(skb)) {
|
||||
int ingress_retval;
|
||||
|
||||
if (*pt_prev) {
|
||||
*ret = deliver_skb(skb, *pt_prev, orig_dev);
|
||||
*pt_prev = NULL;
|
||||
}
|
||||
|
||||
return nf_hook_ingress(skb);
|
||||
rcu_read_lock();
|
||||
ingress_retval = nf_hook_ingress(skb);
|
||||
rcu_read_unlock();
|
||||
return ingress_retval;
|
||||
}
|
||||
#endif /* CONFIG_NETFILTER_INGRESS */
|
||||
return 0;
|
||||
@@ -5584,6 +5589,7 @@ static inline bool netdev_adjacent_is_neigh_list(struct net_device *dev,
|
||||
|
||||
static int __netdev_adjacent_dev_insert(struct net_device *dev,
|
||||
struct net_device *adj_dev,
|
||||
u16 ref_nr,
|
||||
struct list_head *dev_list,
|
||||
void *private, bool master)
|
||||
{
|
||||
@@ -5593,7 +5599,7 @@ static int __netdev_adjacent_dev_insert(struct net_device *dev,
|
||||
adj = __netdev_find_adj(adj_dev, dev_list);
|
||||
|
||||
if (adj) {
|
||||
adj->ref_nr++;
|
||||
adj->ref_nr += ref_nr;
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -5603,7 +5609,7 @@ static int __netdev_adjacent_dev_insert(struct net_device *dev,
|
||||
|
||||
adj->dev = adj_dev;
|
||||
adj->master = master;
|
||||
adj->ref_nr = 1;
|
||||
adj->ref_nr = ref_nr;
|
||||
adj->private = private;
|
||||
dev_hold(adj_dev);
|
||||
|
||||
@@ -5642,6 +5648,7 @@ free_adj:
|
||||
|
||||
static void __netdev_adjacent_dev_remove(struct net_device *dev,
|
||||
struct net_device *adj_dev,
|
||||
u16 ref_nr,
|
||||
struct list_head *dev_list)
|
||||
{
|
||||
struct netdev_adjacent *adj;
|
||||
@@ -5654,10 +5661,10 @@ static void __netdev_adjacent_dev_remove(struct net_device *dev,
|
||||
BUG();
|
||||
}
|
||||
|
||||
if (adj->ref_nr > 1) {
|
||||
pr_debug("%s to %s ref_nr-- = %d\n", dev->name, adj_dev->name,
|
||||
adj->ref_nr-1);
|
||||
adj->ref_nr--;
|
||||
if (adj->ref_nr > ref_nr) {
|
||||
pr_debug("%s to %s ref_nr-%d = %d\n", dev->name, adj_dev->name,
|
||||
ref_nr, adj->ref_nr-ref_nr);
|
||||
adj->ref_nr -= ref_nr;
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -5676,21 +5683,22 @@ static void __netdev_adjacent_dev_remove(struct net_device *dev,
|
||||
|
||||
static int __netdev_adjacent_dev_link_lists(struct net_device *dev,
|
||||
struct net_device *upper_dev,
|
||||
u16 ref_nr,
|
||||
struct list_head *up_list,
|
||||
struct list_head *down_list,
|
||||
void *private, bool master)
|
||||
{
|
||||
int ret;
|
||||
|
||||
ret = __netdev_adjacent_dev_insert(dev, upper_dev, up_list, private,
|
||||
master);
|
||||
ret = __netdev_adjacent_dev_insert(dev, upper_dev, ref_nr, up_list,
|
||||
private, master);
|
||||
if (ret)
|
||||
return ret;
|
||||
|
||||
ret = __netdev_adjacent_dev_insert(upper_dev, dev, down_list, private,
|
||||
false);
|
||||
ret = __netdev_adjacent_dev_insert(upper_dev, dev, ref_nr, down_list,
|
||||
private, false);
|
||||
if (ret) {
|
||||
__netdev_adjacent_dev_remove(dev, upper_dev, up_list);
|
||||
__netdev_adjacent_dev_remove(dev, upper_dev, ref_nr, up_list);
|
||||
return ret;
|
||||
}
|
||||
|
||||
@@ -5698,9 +5706,10 @@ static int __netdev_adjacent_dev_link_lists(struct net_device *dev,
|
||||
}
|
||||
|
||||
static int __netdev_adjacent_dev_link(struct net_device *dev,
|
||||
struct net_device *upper_dev)
|
||||
struct net_device *upper_dev,
|
||||
u16 ref_nr)
|
||||
{
|
||||
return __netdev_adjacent_dev_link_lists(dev, upper_dev,
|
||||
return __netdev_adjacent_dev_link_lists(dev, upper_dev, ref_nr,
|
||||
&dev->all_adj_list.upper,
|
||||
&upper_dev->all_adj_list.lower,
|
||||
NULL, false);
|
||||
@@ -5708,17 +5717,19 @@ static int __netdev_adjacent_dev_link(struct net_device *dev,
|
||||
|
||||
static void __netdev_adjacent_dev_unlink_lists(struct net_device *dev,
|
||||
struct net_device *upper_dev,
|
||||
u16 ref_nr,
|
||||
struct list_head *up_list,
|
||||
struct list_head *down_list)
|
||||
{
|
||||
__netdev_adjacent_dev_remove(dev, upper_dev, up_list);
|
||||
__netdev_adjacent_dev_remove(upper_dev, dev, down_list);
|
||||
__netdev_adjacent_dev_remove(dev, upper_dev, ref_nr, up_list);
|
||||
__netdev_adjacent_dev_remove(upper_dev, dev, ref_nr, down_list);
|
||||
}
|
||||
|
||||
static void __netdev_adjacent_dev_unlink(struct net_device *dev,
|
||||
struct net_device *upper_dev)
|
||||
struct net_device *upper_dev,
|
||||
u16 ref_nr)
|
||||
{
|
||||
__netdev_adjacent_dev_unlink_lists(dev, upper_dev,
|
||||
__netdev_adjacent_dev_unlink_lists(dev, upper_dev, ref_nr,
|
||||
&dev->all_adj_list.upper,
|
||||
&upper_dev->all_adj_list.lower);
|
||||
}
|
||||
@@ -5727,17 +5738,17 @@ static int __netdev_adjacent_dev_link_neighbour(struct net_device *dev,
|
||||
struct net_device *upper_dev,
|
||||
void *private, bool master)
|
||||
{
|
||||
int ret = __netdev_adjacent_dev_link(dev, upper_dev);
|
||||
int ret = __netdev_adjacent_dev_link(dev, upper_dev, 1);
|
||||
|
||||
if (ret)
|
||||
return ret;
|
||||
|
||||
ret = __netdev_adjacent_dev_link_lists(dev, upper_dev,
|
||||
ret = __netdev_adjacent_dev_link_lists(dev, upper_dev, 1,
|
||||
&dev->adj_list.upper,
|
||||
&upper_dev->adj_list.lower,
|
||||
private, master);
|
||||
if (ret) {
|
||||
__netdev_adjacent_dev_unlink(dev, upper_dev);
|
||||
__netdev_adjacent_dev_unlink(dev, upper_dev, 1);
|
||||
return ret;
|
||||
}
|
||||
|
||||
@@ -5747,8 +5758,8 @@ static int __netdev_adjacent_dev_link_neighbour(struct net_device *dev,
|
||||
static void __netdev_adjacent_dev_unlink_neighbour(struct net_device *dev,
|
||||
struct net_device *upper_dev)
|
||||
{
|
||||
__netdev_adjacent_dev_unlink(dev, upper_dev);
|
||||
__netdev_adjacent_dev_unlink_lists(dev, upper_dev,
|
||||
__netdev_adjacent_dev_unlink(dev, upper_dev, 1);
|
||||
__netdev_adjacent_dev_unlink_lists(dev, upper_dev, 1,
|
||||
&dev->adj_list.upper,
|
||||
&upper_dev->adj_list.lower);
|
||||
}
|
||||
@@ -5801,7 +5812,7 @@ static int __netdev_upper_dev_link(struct net_device *dev,
|
||||
list_for_each_entry(j, &upper_dev->all_adj_list.upper, list) {
|
||||
pr_debug("Interlinking %s with %s, non-neighbour\n",
|
||||
i->dev->name, j->dev->name);
|
||||
ret = __netdev_adjacent_dev_link(i->dev, j->dev);
|
||||
ret = __netdev_adjacent_dev_link(i->dev, j->dev, i->ref_nr);
|
||||
if (ret)
|
||||
goto rollback_mesh;
|
||||
}
|
||||
@@ -5811,7 +5822,7 @@ static int __netdev_upper_dev_link(struct net_device *dev,
|
||||
list_for_each_entry(i, &upper_dev->all_adj_list.upper, list) {
|
||||
pr_debug("linking %s's upper device %s with %s\n",
|
||||
upper_dev->name, i->dev->name, dev->name);
|
||||
ret = __netdev_adjacent_dev_link(dev, i->dev);
|
||||
ret = __netdev_adjacent_dev_link(dev, i->dev, i->ref_nr);
|
||||
if (ret)
|
||||
goto rollback_upper_mesh;
|
||||
}
|
||||
@@ -5820,7 +5831,7 @@ static int __netdev_upper_dev_link(struct net_device *dev,
|
||||
list_for_each_entry(i, &dev->all_adj_list.lower, list) {
|
||||
pr_debug("linking %s's lower device %s with %s\n", dev->name,
|
||||
i->dev->name, upper_dev->name);
|
||||
ret = __netdev_adjacent_dev_link(i->dev, upper_dev);
|
||||
ret = __netdev_adjacent_dev_link(i->dev, upper_dev, i->ref_nr);
|
||||
if (ret)
|
||||
goto rollback_lower_mesh;
|
||||
}
|
||||
@@ -5838,7 +5849,7 @@ rollback_lower_mesh:
|
||||
list_for_each_entry(i, &dev->all_adj_list.lower, list) {
|
||||
if (i == to_i)
|
||||
break;
|
||||
__netdev_adjacent_dev_unlink(i->dev, upper_dev);
|
||||
__netdev_adjacent_dev_unlink(i->dev, upper_dev, i->ref_nr);
|
||||
}
|
||||
|
||||
i = NULL;
|
||||
@@ -5848,7 +5859,7 @@ rollback_upper_mesh:
|
||||
list_for_each_entry(i, &upper_dev->all_adj_list.upper, list) {
|
||||
if (i == to_i)
|
||||
break;
|
||||
__netdev_adjacent_dev_unlink(dev, i->dev);
|
||||
__netdev_adjacent_dev_unlink(dev, i->dev, i->ref_nr);
|
||||
}
|
||||
|
||||
i = j = NULL;
|
||||
@@ -5860,7 +5871,7 @@ rollback_mesh:
|
||||
list_for_each_entry(j, &upper_dev->all_adj_list.upper, list) {
|
||||
if (i == to_i && j == to_j)
|
||||
break;
|
||||
__netdev_adjacent_dev_unlink(i->dev, j->dev);
|
||||
__netdev_adjacent_dev_unlink(i->dev, j->dev, i->ref_nr);
|
||||
}
|
||||
if (i == to_i)
|
||||
break;
|
||||
@@ -5940,16 +5951,16 @@ void netdev_upper_dev_unlink(struct net_device *dev,
|
||||
*/
|
||||
list_for_each_entry(i, &dev->all_adj_list.lower, list)
|
||||
list_for_each_entry(j, &upper_dev->all_adj_list.upper, list)
|
||||
__netdev_adjacent_dev_unlink(i->dev, j->dev);
|
||||
__netdev_adjacent_dev_unlink(i->dev, j->dev, i->ref_nr);
|
||||
|
||||
/* remove also the devices itself from lower/upper device
|
||||
* list
|
||||
*/
|
||||
list_for_each_entry(i, &dev->all_adj_list.lower, list)
|
||||
__netdev_adjacent_dev_unlink(i->dev, upper_dev);
|
||||
__netdev_adjacent_dev_unlink(i->dev, upper_dev, i->ref_nr);
|
||||
|
||||
list_for_each_entry(i, &upper_dev->all_adj_list.upper, list)
|
||||
__netdev_adjacent_dev_unlink(dev, i->dev);
|
||||
__netdev_adjacent_dev_unlink(dev, i->dev, i->ref_nr);
|
||||
|
||||
call_netdevice_notifiers_info(NETDEV_CHANGEUPPER, dev,
|
||||
&changeupper_info.info);
|
||||
|
||||
+138
-18
@@ -1362,6 +1362,11 @@ static inline int bpf_try_make_writable(struct sk_buff *skb,
|
||||
return err;
|
||||
}
|
||||
|
||||
static int bpf_try_make_head_writable(struct sk_buff *skb)
|
||||
{
|
||||
return bpf_try_make_writable(skb, skb_headlen(skb));
|
||||
}
|
||||
|
||||
static inline void bpf_push_mac_rcsum(struct sk_buff *skb)
|
||||
{
|
||||
if (skb_at_tc_ingress(skb))
|
||||
@@ -1441,6 +1446,28 @@ static const struct bpf_func_proto bpf_skb_load_bytes_proto = {
|
||||
.arg4_type = ARG_CONST_STACK_SIZE,
|
||||
};
|
||||
|
||||
BPF_CALL_2(bpf_skb_pull_data, struct sk_buff *, skb, u32, len)
|
||||
{
|
||||
/* Idea is the following: should the needed direct read/write
|
||||
* test fail during runtime, we can pull in more data and redo
|
||||
* again, since implicitly, we invalidate previous checks here.
|
||||
*
|
||||
* Or, since we know how much we need to make read/writeable,
|
||||
* this can be done once at the program beginning for direct
|
||||
* access case. By this we overcome limitations of only current
|
||||
* headroom being accessible.
|
||||
*/
|
||||
return bpf_try_make_writable(skb, len ? : skb_headlen(skb));
|
||||
}
|
||||
|
||||
static const struct bpf_func_proto bpf_skb_pull_data_proto = {
|
||||
.func = bpf_skb_pull_data,
|
||||
.gpl_only = false,
|
||||
.ret_type = RET_INTEGER,
|
||||
.arg1_type = ARG_PTR_TO_CTX,
|
||||
.arg2_type = ARG_ANYTHING,
|
||||
};
|
||||
|
||||
BPF_CALL_5(bpf_l3_csum_replace, struct sk_buff *, skb, u32, offset,
|
||||
u64, from, u64, to, u64, flags)
|
||||
{
|
||||
@@ -1567,6 +1594,7 @@ BPF_CALL_5(bpf_csum_diff, __be32 *, from, u32, from_size,
|
||||
static const struct bpf_func_proto bpf_csum_diff_proto = {
|
||||
.func = bpf_csum_diff,
|
||||
.gpl_only = false,
|
||||
.pkt_access = true,
|
||||
.ret_type = RET_INTEGER,
|
||||
.arg1_type = ARG_PTR_TO_STACK,
|
||||
.arg2_type = ARG_CONST_STACK_SIZE_OR_ZERO,
|
||||
@@ -1575,6 +1603,26 @@ static const struct bpf_func_proto bpf_csum_diff_proto = {
|
||||
.arg5_type = ARG_ANYTHING,
|
||||
};
|
||||
|
||||
BPF_CALL_2(bpf_csum_update, struct sk_buff *, skb, __wsum, csum)
|
||||
{
|
||||
/* The interface is to be used in combination with bpf_csum_diff()
|
||||
* for direct packet writes. csum rotation for alignment as well
|
||||
* as emulating csum_sub() can be done from the eBPF program.
|
||||
*/
|
||||
if (skb->ip_summed == CHECKSUM_COMPLETE)
|
||||
return (skb->csum = csum_add(skb->csum, csum));
|
||||
|
||||
return -ENOTSUPP;
|
||||
}
|
||||
|
||||
static const struct bpf_func_proto bpf_csum_update_proto = {
|
||||
.func = bpf_csum_update,
|
||||
.gpl_only = false,
|
||||
.ret_type = RET_INTEGER,
|
||||
.arg1_type = ARG_PTR_TO_CTX,
|
||||
.arg2_type = ARG_ANYTHING,
|
||||
};
|
||||
|
||||
static inline int __bpf_rx_skb(struct net_device *dev, struct sk_buff *skb)
|
||||
{
|
||||
return dev_forward_skb(dev, skb);
|
||||
@@ -1602,6 +1650,8 @@ static inline int __bpf_tx_skb(struct net_device *dev, struct sk_buff *skb)
|
||||
BPF_CALL_3(bpf_clone_redirect, struct sk_buff *, skb, u32, ifindex, u64, flags)
|
||||
{
|
||||
struct net_device *dev;
|
||||
struct sk_buff *clone;
|
||||
int ret;
|
||||
|
||||
if (unlikely(flags & ~(BPF_F_INGRESS)))
|
||||
return -EINVAL;
|
||||
@@ -1610,14 +1660,25 @@ BPF_CALL_3(bpf_clone_redirect, struct sk_buff *, skb, u32, ifindex, u64, flags)
|
||||
if (unlikely(!dev))
|
||||
return -EINVAL;
|
||||
|
||||
skb = skb_clone(skb, GFP_ATOMIC);
|
||||
if (unlikely(!skb))
|
||||
clone = skb_clone(skb, GFP_ATOMIC);
|
||||
if (unlikely(!clone))
|
||||
return -ENOMEM;
|
||||
|
||||
bpf_push_mac_rcsum(skb);
|
||||
/* For direct write, we need to keep the invariant that the skbs
|
||||
* we're dealing with need to be uncloned. Should uncloning fail
|
||||
* here, we need to free the just generated clone to unclone once
|
||||
* again.
|
||||
*/
|
||||
ret = bpf_try_make_head_writable(skb);
|
||||
if (unlikely(ret)) {
|
||||
kfree_skb(clone);
|
||||
return -ENOMEM;
|
||||
}
|
||||
|
||||
bpf_push_mac_rcsum(clone);
|
||||
|
||||
return flags & BPF_F_INGRESS ?
|
||||
__bpf_rx_skb(dev, skb) : __bpf_tx_skb(dev, skb);
|
||||
__bpf_rx_skb(dev, clone) : __bpf_tx_skb(dev, clone);
|
||||
}
|
||||
|
||||
static const struct bpf_func_proto bpf_clone_redirect_proto = {
|
||||
@@ -1716,6 +1777,22 @@ static const struct bpf_func_proto bpf_get_hash_recalc_proto = {
|
||||
.arg1_type = ARG_PTR_TO_CTX,
|
||||
};
|
||||
|
||||
BPF_CALL_1(bpf_set_hash_invalid, struct sk_buff *, skb)
|
||||
{
|
||||
/* After all direct packet write, this can be used once for
|
||||
* triggering a lazy recalc on next skb_get_hash() invocation.
|
||||
*/
|
||||
skb_clear_hash(skb);
|
||||
return 0;
|
||||
}
|
||||
|
||||
static const struct bpf_func_proto bpf_set_hash_invalid_proto = {
|
||||
.func = bpf_set_hash_invalid,
|
||||
.gpl_only = false,
|
||||
.ret_type = RET_INTEGER,
|
||||
.arg1_type = ARG_PTR_TO_CTX,
|
||||
};
|
||||
|
||||
BPF_CALL_3(bpf_skb_vlan_push, struct sk_buff *, skb, __be16, vlan_proto,
|
||||
u16, vlan_tci)
|
||||
{
|
||||
@@ -2063,19 +2140,14 @@ static const struct bpf_func_proto bpf_skb_change_tail_proto = {
|
||||
|
||||
bool bpf_helper_changes_skb_data(void *func)
|
||||
{
|
||||
if (func == bpf_skb_vlan_push)
|
||||
return true;
|
||||
if (func == bpf_skb_vlan_pop)
|
||||
return true;
|
||||
if (func == bpf_skb_store_bytes)
|
||||
return true;
|
||||
if (func == bpf_skb_change_proto)
|
||||
return true;
|
||||
if (func == bpf_skb_change_tail)
|
||||
return true;
|
||||
if (func == bpf_l3_csum_replace)
|
||||
return true;
|
||||
if (func == bpf_l4_csum_replace)
|
||||
if (func == bpf_skb_vlan_push ||
|
||||
func == bpf_skb_vlan_pop ||
|
||||
func == bpf_skb_store_bytes ||
|
||||
func == bpf_skb_change_proto ||
|
||||
func == bpf_skb_change_tail ||
|
||||
func == bpf_skb_pull_data ||
|
||||
func == bpf_l3_csum_replace ||
|
||||
func == bpf_l4_csum_replace)
|
||||
return true;
|
||||
|
||||
return false;
|
||||
@@ -2352,7 +2424,7 @@ BPF_CALL_3(bpf_skb_under_cgroup, struct sk_buff *, skb, struct bpf_map *, map,
|
||||
struct cgroup *cgrp;
|
||||
struct sock *sk;
|
||||
|
||||
sk = skb->sk;
|
||||
sk = skb_to_full_sk(skb);
|
||||
if (!sk || !sk_fullsock(sk))
|
||||
return -ENOENT;
|
||||
if (unlikely(idx >= array->map.max_entries))
|
||||
@@ -2440,8 +2512,12 @@ tc_cls_act_func_proto(enum bpf_func_id func_id)
|
||||
return &bpf_skb_store_bytes_proto;
|
||||
case BPF_FUNC_skb_load_bytes:
|
||||
return &bpf_skb_load_bytes_proto;
|
||||
case BPF_FUNC_skb_pull_data:
|
||||
return &bpf_skb_pull_data_proto;
|
||||
case BPF_FUNC_csum_diff:
|
||||
return &bpf_csum_diff_proto;
|
||||
case BPF_FUNC_csum_update:
|
||||
return &bpf_csum_update_proto;
|
||||
case BPF_FUNC_l3_csum_replace:
|
||||
return &bpf_l3_csum_replace_proto;
|
||||
case BPF_FUNC_l4_csum_replace:
|
||||
@@ -2474,6 +2550,8 @@ tc_cls_act_func_proto(enum bpf_func_id func_id)
|
||||
return &bpf_get_route_realm_proto;
|
||||
case BPF_FUNC_get_hash_recalc:
|
||||
return &bpf_get_hash_recalc_proto;
|
||||
case BPF_FUNC_set_hash_invalid:
|
||||
return &bpf_set_hash_invalid_proto;
|
||||
case BPF_FUNC_perf_event_output:
|
||||
return &bpf_skb_event_output_proto;
|
||||
case BPF_FUNC_get_smp_processor_id:
|
||||
@@ -2491,6 +2569,8 @@ xdp_func_proto(enum bpf_func_id func_id)
|
||||
switch (func_id) {
|
||||
case BPF_FUNC_perf_event_output:
|
||||
return &bpf_xdp_event_output_proto;
|
||||
case BPF_FUNC_get_smp_processor_id:
|
||||
return &bpf_get_smp_processor_id_proto;
|
||||
default:
|
||||
return sk_filter_func_proto(func_id);
|
||||
}
|
||||
@@ -2533,6 +2613,45 @@ static bool sk_filter_is_valid_access(int off, int size,
|
||||
return __is_valid_access(off, size, type);
|
||||
}
|
||||
|
||||
static int tc_cls_act_prologue(struct bpf_insn *insn_buf, bool direct_write,
|
||||
const struct bpf_prog *prog)
|
||||
{
|
||||
struct bpf_insn *insn = insn_buf;
|
||||
|
||||
if (!direct_write)
|
||||
return 0;
|
||||
|
||||
/* if (!skb->cloned)
|
||||
* goto start;
|
||||
*
|
||||
* (Fast-path, otherwise approximation that we might be
|
||||
* a clone, do the rest in helper.)
|
||||
*/
|
||||
*insn++ = BPF_LDX_MEM(BPF_B, BPF_REG_6, BPF_REG_1, CLONED_OFFSET());
|
||||
*insn++ = BPF_ALU32_IMM(BPF_AND, BPF_REG_6, CLONED_MASK);
|
||||
*insn++ = BPF_JMP_IMM(BPF_JEQ, BPF_REG_6, 0, 7);
|
||||
|
||||
/* ret = bpf_skb_pull_data(skb, 0); */
|
||||
*insn++ = BPF_MOV64_REG(BPF_REG_6, BPF_REG_1);
|
||||
*insn++ = BPF_ALU64_REG(BPF_XOR, BPF_REG_2, BPF_REG_2);
|
||||
*insn++ = BPF_RAW_INSN(BPF_JMP | BPF_CALL, 0, 0, 0,
|
||||
BPF_FUNC_skb_pull_data);
|
||||
/* if (!ret)
|
||||
* goto restore;
|
||||
* return TC_ACT_SHOT;
|
||||
*/
|
||||
*insn++ = BPF_JMP_IMM(BPF_JEQ, BPF_REG_0, 0, 2);
|
||||
*insn++ = BPF_ALU32_IMM(BPF_MOV, BPF_REG_0, TC_ACT_SHOT);
|
||||
*insn++ = BPF_EXIT_INSN();
|
||||
|
||||
/* restore: */
|
||||
*insn++ = BPF_MOV64_REG(BPF_REG_1, BPF_REG_6);
|
||||
/* start: */
|
||||
*insn++ = prog->insnsi[0];
|
||||
|
||||
return insn - insn_buf;
|
||||
}
|
||||
|
||||
static bool tc_cls_act_is_valid_access(int off, int size,
|
||||
enum bpf_access_type type,
|
||||
enum bpf_reg_type *reg_type)
|
||||
@@ -2810,6 +2929,7 @@ static const struct bpf_verifier_ops tc_cls_act_ops = {
|
||||
.get_func_proto = tc_cls_act_func_proto,
|
||||
.is_valid_access = tc_cls_act_is_valid_access,
|
||||
.convert_ctx_access = tc_cls_act_convert_ctx_access,
|
||||
.gen_prologue = tc_cls_act_prologue,
|
||||
};
|
||||
|
||||
static const struct bpf_verifier_ops xdp_ops = {
|
||||
|
||||
+10
-11
@@ -2286,7 +2286,7 @@ out:
|
||||
|
||||
static inline void set_pkt_overhead(struct pktgen_dev *pkt_dev)
|
||||
{
|
||||
pkt_dev->pkt_overhead = LL_RESERVED_SPACE(pkt_dev->odev);
|
||||
pkt_dev->pkt_overhead = 0;
|
||||
pkt_dev->pkt_overhead += pkt_dev->nr_labels*sizeof(u32);
|
||||
pkt_dev->pkt_overhead += VLAN_TAG_SIZE(pkt_dev);
|
||||
pkt_dev->pkt_overhead += SVLAN_TAG_SIZE(pkt_dev);
|
||||
@@ -2777,13 +2777,13 @@ static void pktgen_finalize_skb(struct pktgen_dev *pkt_dev, struct sk_buff *skb,
|
||||
}
|
||||
|
||||
static struct sk_buff *pktgen_alloc_skb(struct net_device *dev,
|
||||
struct pktgen_dev *pkt_dev,
|
||||
unsigned int extralen)
|
||||
struct pktgen_dev *pkt_dev)
|
||||
{
|
||||
unsigned int extralen = LL_RESERVED_SPACE(dev);
|
||||
struct sk_buff *skb = NULL;
|
||||
unsigned int size = pkt_dev->cur_pkt_size + 64 + extralen +
|
||||
pkt_dev->pkt_overhead;
|
||||
unsigned int size;
|
||||
|
||||
size = pkt_dev->cur_pkt_size + 64 + extralen + pkt_dev->pkt_overhead;
|
||||
if (pkt_dev->flags & F_NODE) {
|
||||
int node = pkt_dev->node >= 0 ? pkt_dev->node : numa_node_id();
|
||||
|
||||
@@ -2796,8 +2796,9 @@ static struct sk_buff *pktgen_alloc_skb(struct net_device *dev,
|
||||
skb = __netdev_alloc_skb(dev, size, GFP_NOWAIT);
|
||||
}
|
||||
|
||||
/* the caller pre-fetches from skb->data and reserves for the mac hdr */
|
||||
if (likely(skb))
|
||||
skb_reserve(skb, LL_RESERVED_SPACE(dev));
|
||||
skb_reserve(skb, extralen - 16);
|
||||
|
||||
return skb;
|
||||
}
|
||||
@@ -2830,16 +2831,14 @@ static struct sk_buff *fill_packet_ipv4(struct net_device *odev,
|
||||
mod_cur_headers(pkt_dev);
|
||||
queue_map = pkt_dev->cur_queue_map;
|
||||
|
||||
datalen = (odev->hard_header_len + 16) & ~0xf;
|
||||
|
||||
skb = pktgen_alloc_skb(odev, pkt_dev, datalen);
|
||||
skb = pktgen_alloc_skb(odev, pkt_dev);
|
||||
if (!skb) {
|
||||
sprintf(pkt_dev->result, "No memory");
|
||||
return NULL;
|
||||
}
|
||||
|
||||
prefetchw(skb->data);
|
||||
skb_reserve(skb, datalen);
|
||||
skb_reserve(skb, 16);
|
||||
|
||||
/* Reserve for ethernet and IP header */
|
||||
eth = (__u8 *) skb_push(skb, 14);
|
||||
@@ -2959,7 +2958,7 @@ static struct sk_buff *fill_packet_ipv6(struct net_device *odev,
|
||||
mod_cur_headers(pkt_dev);
|
||||
queue_map = pkt_dev->cur_queue_map;
|
||||
|
||||
skb = pktgen_alloc_skb(odev, pkt_dev, 16);
|
||||
skb = pktgen_alloc_skb(odev, pkt_dev);
|
||||
if (!skb) {
|
||||
sprintf(pkt_dev->result, "No memory");
|
||||
return NULL;
|
||||
|
||||
+175
-19
@@ -843,7 +843,10 @@ static inline int rtnl_vfinfo_size(const struct net_device *dev,
|
||||
size += nla_total_size(num_vfs * sizeof(struct nlattr));
|
||||
size += num_vfs *
|
||||
(nla_total_size(sizeof(struct ifla_vf_mac)) +
|
||||
nla_total_size(sizeof(struct ifla_vf_vlan)) +
|
||||
nla_total_size(MAX_VLAN_LIST_LEN *
|
||||
sizeof(struct nlattr)) +
|
||||
nla_total_size(MAX_VLAN_LIST_LEN *
|
||||
sizeof(struct ifla_vf_vlan_info)) +
|
||||
nla_total_size(sizeof(struct ifla_vf_spoofchk)) +
|
||||
nla_total_size(sizeof(struct ifla_vf_rate)) +
|
||||
nla_total_size(sizeof(struct ifla_vf_link_state)) +
|
||||
@@ -1111,14 +1114,15 @@ static noinline_for_stack int rtnl_fill_vfinfo(struct sk_buff *skb,
|
||||
struct nlattr *vfinfo)
|
||||
{
|
||||
struct ifla_vf_rss_query_en vf_rss_query_en;
|
||||
struct nlattr *vf, *vfstats, *vfvlanlist;
|
||||
struct ifla_vf_link_state vf_linkstate;
|
||||
struct ifla_vf_vlan_info vf_vlan_info;
|
||||
struct ifla_vf_spoofchk vf_spoofchk;
|
||||
struct ifla_vf_tx_rate vf_tx_rate;
|
||||
struct ifla_vf_stats vf_stats;
|
||||
struct ifla_vf_trust vf_trust;
|
||||
struct ifla_vf_vlan vf_vlan;
|
||||
struct ifla_vf_rate vf_rate;
|
||||
struct nlattr *vf, *vfstats;
|
||||
struct ifla_vf_mac vf_mac;
|
||||
struct ifla_vf_info ivi;
|
||||
|
||||
@@ -1135,11 +1139,14 @@ static noinline_for_stack int rtnl_fill_vfinfo(struct sk_buff *skb,
|
||||
* IFLA_VF_LINK_STATE_AUTO which equals zero
|
||||
*/
|
||||
ivi.linkstate = 0;
|
||||
/* VLAN Protocol by default is 802.1Q */
|
||||
ivi.vlan_proto = htons(ETH_P_8021Q);
|
||||
if (dev->netdev_ops->ndo_get_vf_config(dev, vfs_num, &ivi))
|
||||
return 0;
|
||||
|
||||
vf_mac.vf =
|
||||
vf_vlan.vf =
|
||||
vf_vlan_info.vf =
|
||||
vf_rate.vf =
|
||||
vf_tx_rate.vf =
|
||||
vf_spoofchk.vf =
|
||||
@@ -1150,6 +1157,9 @@ static noinline_for_stack int rtnl_fill_vfinfo(struct sk_buff *skb,
|
||||
memcpy(vf_mac.mac, ivi.mac, sizeof(ivi.mac));
|
||||
vf_vlan.vlan = ivi.vlan;
|
||||
vf_vlan.qos = ivi.qos;
|
||||
vf_vlan_info.vlan = ivi.vlan;
|
||||
vf_vlan_info.qos = ivi.qos;
|
||||
vf_vlan_info.vlan_proto = ivi.vlan_proto;
|
||||
vf_tx_rate.rate = ivi.max_tx_rate;
|
||||
vf_rate.min_tx_rate = ivi.min_tx_rate;
|
||||
vf_rate.max_tx_rate = ivi.max_tx_rate;
|
||||
@@ -1158,10 +1168,8 @@ static noinline_for_stack int rtnl_fill_vfinfo(struct sk_buff *skb,
|
||||
vf_rss_query_en.setting = ivi.rss_query_en;
|
||||
vf_trust.setting = ivi.trusted;
|
||||
vf = nla_nest_start(skb, IFLA_VF_INFO);
|
||||
if (!vf) {
|
||||
nla_nest_cancel(skb, vfinfo);
|
||||
return -EMSGSIZE;
|
||||
}
|
||||
if (!vf)
|
||||
goto nla_put_vfinfo_failure;
|
||||
if (nla_put(skb, IFLA_VF_MAC, sizeof(vf_mac), &vf_mac) ||
|
||||
nla_put(skb, IFLA_VF_VLAN, sizeof(vf_vlan), &vf_vlan) ||
|
||||
nla_put(skb, IFLA_VF_RATE, sizeof(vf_rate),
|
||||
@@ -1177,17 +1185,23 @@ static noinline_for_stack int rtnl_fill_vfinfo(struct sk_buff *skb,
|
||||
&vf_rss_query_en) ||
|
||||
nla_put(skb, IFLA_VF_TRUST,
|
||||
sizeof(vf_trust), &vf_trust))
|
||||
return -EMSGSIZE;
|
||||
goto nla_put_vf_failure;
|
||||
vfvlanlist = nla_nest_start(skb, IFLA_VF_VLAN_LIST);
|
||||
if (!vfvlanlist)
|
||||
goto nla_put_vf_failure;
|
||||
if (nla_put(skb, IFLA_VF_VLAN_INFO, sizeof(vf_vlan_info),
|
||||
&vf_vlan_info)) {
|
||||
nla_nest_cancel(skb, vfvlanlist);
|
||||
goto nla_put_vf_failure;
|
||||
}
|
||||
nla_nest_end(skb, vfvlanlist);
|
||||
memset(&vf_stats, 0, sizeof(vf_stats));
|
||||
if (dev->netdev_ops->ndo_get_vf_stats)
|
||||
dev->netdev_ops->ndo_get_vf_stats(dev, vfs_num,
|
||||
&vf_stats);
|
||||
vfstats = nla_nest_start(skb, IFLA_VF_STATS);
|
||||
if (!vfstats) {
|
||||
nla_nest_cancel(skb, vf);
|
||||
nla_nest_cancel(skb, vfinfo);
|
||||
return -EMSGSIZE;
|
||||
}
|
||||
if (!vfstats)
|
||||
goto nla_put_vf_failure;
|
||||
if (nla_put_u64_64bit(skb, IFLA_VF_STATS_RX_PACKETS,
|
||||
vf_stats.rx_packets, IFLA_VF_STATS_PAD) ||
|
||||
nla_put_u64_64bit(skb, IFLA_VF_STATS_TX_PACKETS,
|
||||
@@ -1199,11 +1213,19 @@ static noinline_for_stack int rtnl_fill_vfinfo(struct sk_buff *skb,
|
||||
nla_put_u64_64bit(skb, IFLA_VF_STATS_BROADCAST,
|
||||
vf_stats.broadcast, IFLA_VF_STATS_PAD) ||
|
||||
nla_put_u64_64bit(skb, IFLA_VF_STATS_MULTICAST,
|
||||
vf_stats.multicast, IFLA_VF_STATS_PAD))
|
||||
return -EMSGSIZE;
|
||||
vf_stats.multicast, IFLA_VF_STATS_PAD)) {
|
||||
nla_nest_cancel(skb, vfstats);
|
||||
goto nla_put_vf_failure;
|
||||
}
|
||||
nla_nest_end(skb, vfstats);
|
||||
nla_nest_end(skb, vf);
|
||||
return 0;
|
||||
|
||||
nla_put_vf_failure:
|
||||
nla_nest_cancel(skb, vf);
|
||||
nla_put_vfinfo_failure:
|
||||
nla_nest_cancel(skb, vfinfo);
|
||||
return -EMSGSIZE;
|
||||
}
|
||||
|
||||
static int rtnl_fill_link_ifmap(struct sk_buff *skb, struct net_device *dev)
|
||||
@@ -1448,6 +1470,7 @@ static const struct nla_policy ifla_info_policy[IFLA_INFO_MAX+1] = {
|
||||
static const struct nla_policy ifla_vf_policy[IFLA_VF_MAX+1] = {
|
||||
[IFLA_VF_MAC] = { .len = sizeof(struct ifla_vf_mac) },
|
||||
[IFLA_VF_VLAN] = { .len = sizeof(struct ifla_vf_vlan) },
|
||||
[IFLA_VF_VLAN_LIST] = { .type = NLA_NESTED },
|
||||
[IFLA_VF_TX_RATE] = { .len = sizeof(struct ifla_vf_tx_rate) },
|
||||
[IFLA_VF_SPOOFCHK] = { .len = sizeof(struct ifla_vf_spoofchk) },
|
||||
[IFLA_VF_RATE] = { .len = sizeof(struct ifla_vf_rate) },
|
||||
@@ -1704,7 +1727,37 @@ static int do_setvfinfo(struct net_device *dev, struct nlattr **tb)
|
||||
err = -EOPNOTSUPP;
|
||||
if (ops->ndo_set_vf_vlan)
|
||||
err = ops->ndo_set_vf_vlan(dev, ivv->vf, ivv->vlan,
|
||||
ivv->qos);
|
||||
ivv->qos,
|
||||
htons(ETH_P_8021Q));
|
||||
if (err < 0)
|
||||
return err;
|
||||
}
|
||||
|
||||
if (tb[IFLA_VF_VLAN_LIST]) {
|
||||
struct ifla_vf_vlan_info *ivvl[MAX_VLAN_LIST_LEN];
|
||||
struct nlattr *attr;
|
||||
int rem, len = 0;
|
||||
|
||||
err = -EOPNOTSUPP;
|
||||
if (!ops->ndo_set_vf_vlan)
|
||||
return err;
|
||||
|
||||
nla_for_each_nested(attr, tb[IFLA_VF_VLAN_LIST], rem) {
|
||||
if (nla_type(attr) != IFLA_VF_VLAN_INFO ||
|
||||
nla_len(attr) < NLA_HDRLEN) {
|
||||
return -EINVAL;
|
||||
}
|
||||
if (len >= MAX_VLAN_LIST_LEN)
|
||||
return -EOPNOTSUPP;
|
||||
ivvl[len] = nla_data(attr);
|
||||
|
||||
len++;
|
||||
}
|
||||
if (len == 0)
|
||||
return -EINVAL;
|
||||
|
||||
err = ops->ndo_set_vf_vlan(dev, ivvl[0]->vf, ivvl[0]->vlan,
|
||||
ivvl[0]->qos, ivvl[0]->vlan_proto);
|
||||
if (err < 0)
|
||||
return err;
|
||||
}
|
||||
@@ -3577,6 +3630,91 @@ static bool stats_attr_valid(unsigned int mask, int attrid, int idxattr)
|
||||
(!idxattr || idxattr == attrid);
|
||||
}
|
||||
|
||||
#define IFLA_OFFLOAD_XSTATS_FIRST (IFLA_OFFLOAD_XSTATS_UNSPEC + 1)
|
||||
static int rtnl_get_offload_stats_attr_size(int attr_id)
|
||||
{
|
||||
switch (attr_id) {
|
||||
case IFLA_OFFLOAD_XSTATS_CPU_HIT:
|
||||
return sizeof(struct rtnl_link_stats64);
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int rtnl_get_offload_stats(struct sk_buff *skb, struct net_device *dev,
|
||||
int *prividx)
|
||||
{
|
||||
struct nlattr *attr = NULL;
|
||||
int attr_id, size;
|
||||
void *attr_data;
|
||||
int err;
|
||||
|
||||
if (!(dev->netdev_ops && dev->netdev_ops->ndo_has_offload_stats &&
|
||||
dev->netdev_ops->ndo_get_offload_stats))
|
||||
return -ENODATA;
|
||||
|
||||
for (attr_id = IFLA_OFFLOAD_XSTATS_FIRST;
|
||||
attr_id <= IFLA_OFFLOAD_XSTATS_MAX; attr_id++) {
|
||||
if (attr_id < *prividx)
|
||||
continue;
|
||||
|
||||
size = rtnl_get_offload_stats_attr_size(attr_id);
|
||||
if (!size)
|
||||
continue;
|
||||
|
||||
if (!dev->netdev_ops->ndo_has_offload_stats(attr_id))
|
||||
continue;
|
||||
|
||||
attr = nla_reserve_64bit(skb, attr_id, size,
|
||||
IFLA_OFFLOAD_XSTATS_UNSPEC);
|
||||
if (!attr)
|
||||
goto nla_put_failure;
|
||||
|
||||
attr_data = nla_data(attr);
|
||||
memset(attr_data, 0, size);
|
||||
err = dev->netdev_ops->ndo_get_offload_stats(attr_id, dev,
|
||||
attr_data);
|
||||
if (err)
|
||||
goto get_offload_stats_failure;
|
||||
}
|
||||
|
||||
if (!attr)
|
||||
return -ENODATA;
|
||||
|
||||
*prividx = 0;
|
||||
return 0;
|
||||
|
||||
nla_put_failure:
|
||||
err = -EMSGSIZE;
|
||||
get_offload_stats_failure:
|
||||
*prividx = attr_id;
|
||||
return err;
|
||||
}
|
||||
|
||||
static int rtnl_get_offload_stats_size(const struct net_device *dev)
|
||||
{
|
||||
int nla_size = 0;
|
||||
int attr_id;
|
||||
int size;
|
||||
|
||||
if (!(dev->netdev_ops && dev->netdev_ops->ndo_has_offload_stats &&
|
||||
dev->netdev_ops->ndo_get_offload_stats))
|
||||
return 0;
|
||||
|
||||
for (attr_id = IFLA_OFFLOAD_XSTATS_FIRST;
|
||||
attr_id <= IFLA_OFFLOAD_XSTATS_MAX; attr_id++) {
|
||||
if (!dev->netdev_ops->ndo_has_offload_stats(attr_id))
|
||||
continue;
|
||||
size = rtnl_get_offload_stats_attr_size(attr_id);
|
||||
nla_size += nla_total_size_64bit(size);
|
||||
}
|
||||
|
||||
if (nla_size != 0)
|
||||
nla_size += nla_total_size(0);
|
||||
|
||||
return nla_size;
|
||||
}
|
||||
|
||||
static int rtnl_fill_statsinfo(struct sk_buff *skb, struct net_device *dev,
|
||||
int type, u32 pid, u32 seq, u32 change,
|
||||
unsigned int flags, unsigned int filter_mask,
|
||||
@@ -3586,6 +3724,7 @@ static int rtnl_fill_statsinfo(struct sk_buff *skb, struct net_device *dev,
|
||||
struct nlmsghdr *nlh;
|
||||
struct nlattr *attr;
|
||||
int s_prividx = *prividx;
|
||||
int err;
|
||||
|
||||
ASSERT_RTNL();
|
||||
|
||||
@@ -3614,8 +3753,6 @@ static int rtnl_fill_statsinfo(struct sk_buff *skb, struct net_device *dev,
|
||||
const struct rtnl_link_ops *ops = dev->rtnl_link_ops;
|
||||
|
||||
if (ops && ops->fill_linkxstats) {
|
||||
int err;
|
||||
|
||||
*idxattr = IFLA_STATS_LINK_XSTATS;
|
||||
attr = nla_nest_start(skb,
|
||||
IFLA_STATS_LINK_XSTATS);
|
||||
@@ -3639,8 +3776,6 @@ static int rtnl_fill_statsinfo(struct sk_buff *skb, struct net_device *dev,
|
||||
if (master)
|
||||
ops = master->rtnl_link_ops;
|
||||
if (ops && ops->fill_linkxstats) {
|
||||
int err;
|
||||
|
||||
*idxattr = IFLA_STATS_LINK_XSTATS_SLAVE;
|
||||
attr = nla_nest_start(skb,
|
||||
IFLA_STATS_LINK_XSTATS_SLAVE);
|
||||
@@ -3655,6 +3790,24 @@ static int rtnl_fill_statsinfo(struct sk_buff *skb, struct net_device *dev,
|
||||
}
|
||||
}
|
||||
|
||||
if (stats_attr_valid(filter_mask, IFLA_STATS_LINK_OFFLOAD_XSTATS,
|
||||
*idxattr)) {
|
||||
*idxattr = IFLA_STATS_LINK_OFFLOAD_XSTATS;
|
||||
attr = nla_nest_start(skb, IFLA_STATS_LINK_OFFLOAD_XSTATS);
|
||||
if (!attr)
|
||||
goto nla_put_failure;
|
||||
|
||||
err = rtnl_get_offload_stats(skb, dev, prividx);
|
||||
if (err == -ENODATA)
|
||||
nla_nest_cancel(skb, attr);
|
||||
else
|
||||
nla_nest_end(skb, attr);
|
||||
|
||||
if (err && err != -ENODATA)
|
||||
goto nla_put_failure;
|
||||
*idxattr = 0;
|
||||
}
|
||||
|
||||
nlmsg_end(skb, nlh);
|
||||
|
||||
return 0;
|
||||
@@ -3708,6 +3861,9 @@ static size_t if_nlmsg_stats_size(const struct net_device *dev,
|
||||
}
|
||||
}
|
||||
|
||||
if (stats_attr_valid(filter_mask, IFLA_STATS_LINK_OFFLOAD_XSTATS, 0))
|
||||
size += rtnl_get_offload_stats_size(dev);
|
||||
|
||||
return size;
|
||||
}
|
||||
|
||||
|
||||
+69
-34
@@ -3097,11 +3097,31 @@ struct sk_buff *skb_segment(struct sk_buff *head_skb,
|
||||
sg = !!(features & NETIF_F_SG);
|
||||
csum = !!can_checksum_protocol(features, proto);
|
||||
|
||||
/* GSO partial only requires that we trim off any excess that
|
||||
* doesn't fit into an MSS sized block, so take care of that
|
||||
* now.
|
||||
*/
|
||||
if (sg && csum && (features & NETIF_F_GSO_PARTIAL)) {
|
||||
if (sg && csum && (mss != GSO_BY_FRAGS)) {
|
||||
if (!(features & NETIF_F_GSO_PARTIAL)) {
|
||||
struct sk_buff *iter;
|
||||
|
||||
if (!list_skb ||
|
||||
!net_gso_ok(features, skb_shinfo(head_skb)->gso_type))
|
||||
goto normal;
|
||||
|
||||
/* Split the buffer at the frag_list pointer.
|
||||
* This is based on the assumption that all
|
||||
* buffers in the chain excluding the last
|
||||
* containing the same amount of data.
|
||||
*/
|
||||
skb_walk_frags(head_skb, iter) {
|
||||
if (skb_headlen(iter))
|
||||
goto normal;
|
||||
|
||||
len -= iter->len;
|
||||
}
|
||||
}
|
||||
|
||||
/* GSO partial only requires that we trim off any excess that
|
||||
* doesn't fit into an MSS sized block, so take care of that
|
||||
* now.
|
||||
*/
|
||||
partial_segs = len / mss;
|
||||
if (partial_segs > 1)
|
||||
mss *= partial_segs;
|
||||
@@ -3109,6 +3129,7 @@ struct sk_buff *skb_segment(struct sk_buff *head_skb,
|
||||
partial_segs = 0;
|
||||
}
|
||||
|
||||
normal:
|
||||
headroom = skb_headroom(head_skb);
|
||||
pos = skb_headlen(head_skb);
|
||||
|
||||
@@ -3300,21 +3321,29 @@ perform_csum_check:
|
||||
*/
|
||||
segs->prev = tail;
|
||||
|
||||
/* Update GSO info on first skb in partial sequence. */
|
||||
if (partial_segs) {
|
||||
struct sk_buff *iter;
|
||||
int type = skb_shinfo(head_skb)->gso_type;
|
||||
unsigned short gso_size = skb_shinfo(head_skb)->gso_size;
|
||||
|
||||
/* Update type to add partial and then remove dodgy if set */
|
||||
type |= SKB_GSO_PARTIAL;
|
||||
type |= (features & NETIF_F_GSO_PARTIAL) / NETIF_F_GSO_PARTIAL * SKB_GSO_PARTIAL;
|
||||
type &= ~SKB_GSO_DODGY;
|
||||
|
||||
/* Update GSO info and prepare to start updating headers on
|
||||
* our way back down the stack of protocols.
|
||||
*/
|
||||
skb_shinfo(segs)->gso_size = skb_shinfo(head_skb)->gso_size;
|
||||
skb_shinfo(segs)->gso_segs = partial_segs;
|
||||
skb_shinfo(segs)->gso_type = type;
|
||||
SKB_GSO_CB(segs)->data_offset = skb_headroom(segs) + doffset;
|
||||
for (iter = segs; iter; iter = iter->next) {
|
||||
skb_shinfo(iter)->gso_size = gso_size;
|
||||
skb_shinfo(iter)->gso_segs = partial_segs;
|
||||
skb_shinfo(iter)->gso_type = type;
|
||||
SKB_GSO_CB(iter)->data_offset = skb_headroom(iter) + doffset;
|
||||
}
|
||||
|
||||
if (tail->len - doffset <= gso_size)
|
||||
skb_shinfo(tail)->gso_size = 0;
|
||||
else if (tail != segs)
|
||||
skb_shinfo(tail)->gso_segs = DIV_ROUND_UP(tail->len - doffset, gso_size);
|
||||
}
|
||||
|
||||
/* Following permits correct backpressure, for protocols
|
||||
@@ -4493,17 +4522,24 @@ int skb_ensure_writable(struct sk_buff *skb, int write_len)
|
||||
}
|
||||
EXPORT_SYMBOL(skb_ensure_writable);
|
||||
|
||||
/* remove VLAN header from packet and update csum accordingly. */
|
||||
static int __skb_vlan_pop(struct sk_buff *skb, u16 *vlan_tci)
|
||||
/* remove VLAN header from packet and update csum accordingly.
|
||||
* expects a non skb_vlan_tag_present skb with a vlan tag payload
|
||||
*/
|
||||
int __skb_vlan_pop(struct sk_buff *skb, u16 *vlan_tci)
|
||||
{
|
||||
struct vlan_hdr *vhdr;
|
||||
unsigned int offset = skb->data - skb_mac_header(skb);
|
||||
int offset = skb->data - skb_mac_header(skb);
|
||||
int err;
|
||||
|
||||
__skb_push(skb, offset);
|
||||
if (WARN_ONCE(offset,
|
||||
"__skb_vlan_pop got skb with skb->data not at mac header (offset %d)\n",
|
||||
offset)) {
|
||||
return -EINVAL;
|
||||
}
|
||||
|
||||
err = skb_ensure_writable(skb, VLAN_ETH_HLEN);
|
||||
if (unlikely(err))
|
||||
goto pull;
|
||||
return err;
|
||||
|
||||
skb_postpull_rcsum(skb, skb->data + (2 * ETH_ALEN), VLAN_HLEN);
|
||||
|
||||
@@ -4520,12 +4556,14 @@ static int __skb_vlan_pop(struct sk_buff *skb, u16 *vlan_tci)
|
||||
skb_set_network_header(skb, ETH_HLEN);
|
||||
|
||||
skb_reset_mac_len(skb);
|
||||
pull:
|
||||
__skb_pull(skb, offset);
|
||||
|
||||
return err;
|
||||
}
|
||||
EXPORT_SYMBOL(__skb_vlan_pop);
|
||||
|
||||
/* Pop a vlan tag either from hwaccel or from payload.
|
||||
* Expects skb->data at mac header.
|
||||
*/
|
||||
int skb_vlan_pop(struct sk_buff *skb)
|
||||
{
|
||||
u16 vlan_tci;
|
||||
@@ -4535,9 +4573,7 @@ int skb_vlan_pop(struct sk_buff *skb)
|
||||
if (likely(skb_vlan_tag_present(skb))) {
|
||||
skb->vlan_tci = 0;
|
||||
} else {
|
||||
if (unlikely((skb->protocol != htons(ETH_P_8021Q) &&
|
||||
skb->protocol != htons(ETH_P_8021AD)) ||
|
||||
skb->len < VLAN_ETH_HLEN))
|
||||
if (unlikely(!eth_type_vlan(skb->protocol)))
|
||||
return 0;
|
||||
|
||||
err = __skb_vlan_pop(skb, &vlan_tci);
|
||||
@@ -4545,9 +4581,7 @@ int skb_vlan_pop(struct sk_buff *skb)
|
||||
return err;
|
||||
}
|
||||
/* move next vlan tag to hw accel tag */
|
||||
if (likely((skb->protocol != htons(ETH_P_8021Q) &&
|
||||
skb->protocol != htons(ETH_P_8021AD)) ||
|
||||
skb->len < VLAN_ETH_HLEN))
|
||||
if (likely(!eth_type_vlan(skb->protocol)))
|
||||
return 0;
|
||||
|
||||
vlan_proto = skb->protocol;
|
||||
@@ -4560,29 +4594,30 @@ int skb_vlan_pop(struct sk_buff *skb)
|
||||
}
|
||||
EXPORT_SYMBOL(skb_vlan_pop);
|
||||
|
||||
/* Push a vlan tag either into hwaccel or into payload (if hwaccel tag present).
|
||||
* Expects skb->data at mac header.
|
||||
*/
|
||||
int skb_vlan_push(struct sk_buff *skb, __be16 vlan_proto, u16 vlan_tci)
|
||||
{
|
||||
if (skb_vlan_tag_present(skb)) {
|
||||
unsigned int offset = skb->data - skb_mac_header(skb);
|
||||
int offset = skb->data - skb_mac_header(skb);
|
||||
int err;
|
||||
|
||||
/* __vlan_insert_tag expect skb->data pointing to mac header.
|
||||
* So change skb->data before calling it and change back to
|
||||
* original position later
|
||||
*/
|
||||
__skb_push(skb, offset);
|
||||
if (WARN_ONCE(offset,
|
||||
"skb_vlan_push got skb with skb->data not at mac header (offset %d)\n",
|
||||
offset)) {
|
||||
return -EINVAL;
|
||||
}
|
||||
|
||||
err = __vlan_insert_tag(skb, skb->vlan_proto,
|
||||
skb_vlan_tag_get(skb));
|
||||
if (err) {
|
||||
__skb_pull(skb, offset);
|
||||
if (err)
|
||||
return err;
|
||||
}
|
||||
|
||||
skb->protocol = skb->vlan_proto;
|
||||
skb->mac_len += VLAN_HLEN;
|
||||
|
||||
skb_postpush_rcsum(skb, skb->data + (2 * ETH_ALEN), VLAN_HLEN);
|
||||
__skb_pull(skb, offset);
|
||||
}
|
||||
__vlan_hwaccel_put_tag(skb, vlan_proto, vlan_tci);
|
||||
return 0;
|
||||
|
||||
+4
-1
@@ -1340,7 +1340,6 @@ static struct sock *sk_prot_alloc(struct proto *prot, gfp_t priority,
|
||||
if (!try_module_get(prot->owner))
|
||||
goto out_free_sec;
|
||||
sk_tx_queue_clear(sk);
|
||||
cgroup_sk_alloc(&sk->sk_cgrp_data);
|
||||
}
|
||||
|
||||
return sk;
|
||||
@@ -1400,6 +1399,7 @@ struct sock *sk_alloc(struct net *net, int family, gfp_t priority,
|
||||
sock_net_set(sk, net);
|
||||
atomic_set(&sk->sk_wmem_alloc, 1);
|
||||
|
||||
cgroup_sk_alloc(&sk->sk_cgrp_data);
|
||||
sock_update_classid(&sk->sk_cgrp_data);
|
||||
sock_update_netprioidx(&sk->sk_cgrp_data);
|
||||
}
|
||||
@@ -1544,6 +1544,9 @@ struct sock *sk_clone_lock(const struct sock *sk, const gfp_t priority)
|
||||
newsk->sk_priority = 0;
|
||||
newsk->sk_incoming_cpu = raw_smp_processor_id();
|
||||
atomic64_set(&newsk->sk_cookie, 0);
|
||||
|
||||
cgroup_sk_alloc(&newsk->sk_cgrp_data);
|
||||
|
||||
/*
|
||||
* Before updating sk_refcnt, we must commit prior changes to memory
|
||||
* (Documentation/RCU/rculist_nulls.txt for details)
|
||||
|
||||
@@ -43,7 +43,6 @@ void sk_stream_write_space(struct sock *sk)
|
||||
rcu_read_unlock();
|
||||
}
|
||||
}
|
||||
EXPORT_SYMBOL(sk_stream_write_space);
|
||||
|
||||
/**
|
||||
* sk_stream_wait_connect - Wait for a socket to get into the connected state
|
||||
|
||||
+5
-3
@@ -378,9 +378,11 @@ static int dsa_switch_setup_one(struct dsa_switch *ds, struct device *parent)
|
||||
if (ret < 0)
|
||||
goto out;
|
||||
|
||||
ret = ops->set_addr(ds, dst->master_netdev->dev_addr);
|
||||
if (ret < 0)
|
||||
goto out;
|
||||
if (ops->set_addr) {
|
||||
ret = ops->set_addr(ds, dst->master_netdev->dev_addr);
|
||||
if (ret < 0)
|
||||
goto out;
|
||||
}
|
||||
|
||||
if (!ds->slave_mii_bus && ops->phy_read) {
|
||||
ds->slave_mii_bus = devm_mdiobus_alloc(parent);
|
||||
|
||||
+5
-7
@@ -304,13 +304,11 @@ static int dsa_ds_apply(struct dsa_switch_tree *dst, struct dsa_switch *ds)
|
||||
if (err < 0)
|
||||
return err;
|
||||
|
||||
err = ds->ops->set_addr(ds, dst->master_netdev->dev_addr);
|
||||
if (err < 0)
|
||||
return err;
|
||||
|
||||
err = ds->ops->set_addr(ds, dst->master_netdev->dev_addr);
|
||||
if (err < 0)
|
||||
return err;
|
||||
if (ds->ops->set_addr) {
|
||||
err = ds->ops->set_addr(ds, dst->master_netdev->dev_addr);
|
||||
if (err < 0)
|
||||
return err;
|
||||
}
|
||||
|
||||
if (!ds->slave_mii_bus && ds->ops->phy_read) {
|
||||
ds->slave_mii_bus = devm_mdiobus_alloc(ds->dev);
|
||||
|
||||
+28
-7
@@ -69,6 +69,30 @@ static inline bool dsa_port_is_bridged(struct dsa_slave_priv *p)
|
||||
return !!p->bridge_dev;
|
||||
}
|
||||
|
||||
static void dsa_port_set_stp_state(struct dsa_switch *ds, int port, u8 state)
|
||||
{
|
||||
struct dsa_port *dp = &ds->ports[port];
|
||||
|
||||
if (ds->ops->port_stp_state_set)
|
||||
ds->ops->port_stp_state_set(ds, port, state);
|
||||
|
||||
if (ds->ops->port_fast_age) {
|
||||
/* Fast age FDB entries or flush appropriate forwarding database
|
||||
* for the given port, if we are moving it from Learning or
|
||||
* Forwarding state, to Disabled or Blocking or Listening state.
|
||||
*/
|
||||
|
||||
if ((dp->stp_state == BR_STATE_LEARNING ||
|
||||
dp->stp_state == BR_STATE_FORWARDING) &&
|
||||
(state == BR_STATE_DISABLED ||
|
||||
state == BR_STATE_BLOCKING ||
|
||||
state == BR_STATE_LISTENING))
|
||||
ds->ops->port_fast_age(ds, port);
|
||||
}
|
||||
|
||||
dp->stp_state = state;
|
||||
}
|
||||
|
||||
static int dsa_slave_open(struct net_device *dev)
|
||||
{
|
||||
struct dsa_slave_priv *p = netdev_priv(dev);
|
||||
@@ -104,8 +128,7 @@ static int dsa_slave_open(struct net_device *dev)
|
||||
goto clear_promisc;
|
||||
}
|
||||
|
||||
if (ds->ops->port_stp_state_set)
|
||||
ds->ops->port_stp_state_set(ds, p->port, stp_state);
|
||||
dsa_port_set_stp_state(ds, p->port, stp_state);
|
||||
|
||||
if (p->phy)
|
||||
phy_start(p->phy);
|
||||
@@ -147,8 +170,7 @@ static int dsa_slave_close(struct net_device *dev)
|
||||
if (ds->ops->port_disable)
|
||||
ds->ops->port_disable(ds, p->port, p->phy);
|
||||
|
||||
if (ds->ops->port_stp_state_set)
|
||||
ds->ops->port_stp_state_set(ds, p->port, BR_STATE_DISABLED);
|
||||
dsa_port_set_stp_state(ds, p->port, BR_STATE_DISABLED);
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -354,7 +376,7 @@ static int dsa_slave_stp_state_set(struct net_device *dev,
|
||||
if (switchdev_trans_ph_prepare(trans))
|
||||
return ds->ops->port_stp_state_set ? 0 : -EOPNOTSUPP;
|
||||
|
||||
ds->ops->port_stp_state_set(ds, p->port, attr->u.stp_state);
|
||||
dsa_port_set_stp_state(ds, p->port, attr->u.stp_state);
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -556,8 +578,7 @@ static void dsa_slave_bridge_port_leave(struct net_device *dev)
|
||||
/* Port left the bridge, put in BR_STATE_DISABLED by the bridge layer,
|
||||
* so allow it to be in BR_STATE_FORWARDING to be kept functional
|
||||
*/
|
||||
if (ds->ops->port_stp_state_set)
|
||||
ds->ops->port_stp_state_set(ds, p->port, BR_STATE_FORWARDING);
|
||||
dsa_port_set_stp_state(ds, p->port, BR_STATE_FORWARDING);
|
||||
}
|
||||
|
||||
static int dsa_slave_port_attr_get(struct net_device *dev,
|
||||
|
||||
@@ -640,6 +640,21 @@ config TCP_CONG_CDG
|
||||
D.A. Hayes and G. Armitage. "Revisiting TCP congestion control using
|
||||
delay gradients." In Networking 2011. Preprint: http://goo.gl/No3vdg
|
||||
|
||||
config TCP_CONG_BBR
|
||||
tristate "BBR TCP"
|
||||
default n
|
||||
---help---
|
||||
|
||||
BBR (Bottleneck Bandwidth and RTT) TCP congestion control aims to
|
||||
maximize network utilization and minimize queues. It builds an explicit
|
||||
model of the the bottleneck delivery rate and path round-trip
|
||||
propagation delay. It tolerates packet loss and delay unrelated to
|
||||
congestion. It can operate over LAN, WAN, cellular, wifi, or cable
|
||||
modem links. It can coexist with flows that use loss-based congestion
|
||||
control, and can operate with shallow buffers, deep buffers,
|
||||
bufferbloat, policers, or AQM schemes that do not provide a delay
|
||||
signal. It requires the fq ("Fair Queue") pacing packet scheduler.
|
||||
|
||||
choice
|
||||
prompt "Default TCP congestion control"
|
||||
default DEFAULT_CUBIC
|
||||
@@ -674,6 +689,9 @@ choice
|
||||
config DEFAULT_CDG
|
||||
bool "CDG" if TCP_CONG_CDG=y
|
||||
|
||||
config DEFAULT_BBR
|
||||
bool "BBR" if TCP_CONG_BBR=y
|
||||
|
||||
config DEFAULT_RENO
|
||||
bool "Reno"
|
||||
endchoice
|
||||
|
||||
+2
-1
@@ -8,7 +8,7 @@ obj-y := route.o inetpeer.o protocol.o \
|
||||
inet_timewait_sock.o inet_connection_sock.o \
|
||||
tcp.o tcp_input.o tcp_output.o tcp_timer.o tcp_ipv4.o \
|
||||
tcp_minisocks.o tcp_cong.o tcp_metrics.o tcp_fastopen.o \
|
||||
tcp_recovery.o \
|
||||
tcp_rate.o tcp_recovery.o \
|
||||
tcp_offload.o datagram.o raw.o udp.o udplite.o \
|
||||
udp_offload.o arp.o icmp.o devinet.o af_inet.o igmp.o \
|
||||
fib_frontend.o fib_semantics.o fib_trie.o \
|
||||
@@ -41,6 +41,7 @@ obj-$(CONFIG_INET_DIAG) += inet_diag.o
|
||||
obj-$(CONFIG_INET_TCP_DIAG) += tcp_diag.o
|
||||
obj-$(CONFIG_INET_UDP_DIAG) += udp_diag.o
|
||||
obj-$(CONFIG_NET_TCPPROBE) += tcp_probe.o
|
||||
obj-$(CONFIG_TCP_CONG_BBR) += tcp_bbr.o
|
||||
obj-$(CONFIG_TCP_CONG_BIC) += tcp_bic.o
|
||||
obj-$(CONFIG_TCP_CONG_CDG) += tcp_cdg.o
|
||||
obj-$(CONFIG_TCP_CONG_CUBIC) += tcp_cubic.o
|
||||
|
||||
+10
-4
@@ -1192,7 +1192,7 @@ EXPORT_SYMBOL(inet_sk_rebuild_header);
|
||||
struct sk_buff *inet_gso_segment(struct sk_buff *skb,
|
||||
netdev_features_t features)
|
||||
{
|
||||
bool udpfrag = false, fixedid = false, encap;
|
||||
bool udpfrag = false, fixedid = false, gso_partial, encap;
|
||||
struct sk_buff *segs = ERR_PTR(-EINVAL);
|
||||
const struct net_offload *ops;
|
||||
unsigned int offset = 0;
|
||||
@@ -1245,6 +1245,8 @@ struct sk_buff *inet_gso_segment(struct sk_buff *skb,
|
||||
if (IS_ERR_OR_NULL(segs))
|
||||
goto out;
|
||||
|
||||
gso_partial = !!(skb_shinfo(segs)->gso_type & SKB_GSO_PARTIAL);
|
||||
|
||||
skb = segs;
|
||||
do {
|
||||
iph = (struct iphdr *)(skb_mac_header(skb) + nhoff);
|
||||
@@ -1259,9 +1261,13 @@ struct sk_buff *inet_gso_segment(struct sk_buff *skb,
|
||||
iph->id = htons(id);
|
||||
id += skb_shinfo(skb)->gso_segs;
|
||||
}
|
||||
tot_len = skb_shinfo(skb)->gso_size +
|
||||
SKB_GSO_CB(skb)->data_offset +
|
||||
skb->head - (unsigned char *)iph;
|
||||
|
||||
if (gso_partial)
|
||||
tot_len = skb_shinfo(skb)->gso_size +
|
||||
SKB_GSO_CB(skb)->data_offset +
|
||||
skb->head - (unsigned char *)iph;
|
||||
else
|
||||
tot_len = skb->len - nhoff;
|
||||
} else {
|
||||
if (!fixedid)
|
||||
iph->id = htons(id++);
|
||||
|
||||
+8
-21
@@ -182,26 +182,13 @@ static void fib_flush(struct net *net)
|
||||
struct fib_table *tb;
|
||||
|
||||
hlist_for_each_entry_safe(tb, tmp, head, tb_hlist)
|
||||
flushed += fib_table_flush(tb);
|
||||
flushed += fib_table_flush(net, tb);
|
||||
}
|
||||
|
||||
if (flushed)
|
||||
rt_cache_flush(net);
|
||||
}
|
||||
|
||||
void fib_flush_external(struct net *net)
|
||||
{
|
||||
struct fib_table *tb;
|
||||
struct hlist_head *head;
|
||||
unsigned int h;
|
||||
|
||||
for (h = 0; h < FIB_TABLE_HASHSZ; h++) {
|
||||
head = &net->ipv4.fib_table_hash[h];
|
||||
hlist_for_each_entry(tb, head, tb_hlist)
|
||||
fib_table_flush_external(tb);
|
||||
}
|
||||
}
|
||||
|
||||
/*
|
||||
* Find address type as if only "dev" was present in the system. If
|
||||
* on_dev is NULL then all interfaces are taken into consideration.
|
||||
@@ -590,13 +577,13 @@ int ip_rt_ioctl(struct net *net, unsigned int cmd, void __user *arg)
|
||||
if (cmd == SIOCDELRT) {
|
||||
tb = fib_get_table(net, cfg.fc_table);
|
||||
if (tb)
|
||||
err = fib_table_delete(tb, &cfg);
|
||||
err = fib_table_delete(net, tb, &cfg);
|
||||
else
|
||||
err = -ESRCH;
|
||||
} else {
|
||||
tb = fib_new_table(net, cfg.fc_table);
|
||||
if (tb)
|
||||
err = fib_table_insert(tb, &cfg);
|
||||
err = fib_table_insert(net, tb, &cfg);
|
||||
else
|
||||
err = -ENOBUFS;
|
||||
}
|
||||
@@ -719,7 +706,7 @@ static int inet_rtm_delroute(struct sk_buff *skb, struct nlmsghdr *nlh)
|
||||
goto errout;
|
||||
}
|
||||
|
||||
err = fib_table_delete(tb, &cfg);
|
||||
err = fib_table_delete(net, tb, &cfg);
|
||||
errout:
|
||||
return err;
|
||||
}
|
||||
@@ -741,7 +728,7 @@ static int inet_rtm_newroute(struct sk_buff *skb, struct nlmsghdr *nlh)
|
||||
goto errout;
|
||||
}
|
||||
|
||||
err = fib_table_insert(tb, &cfg);
|
||||
err = fib_table_insert(net, tb, &cfg);
|
||||
errout:
|
||||
return err;
|
||||
}
|
||||
@@ -828,9 +815,9 @@ static void fib_magic(int cmd, int type, __be32 dst, int dst_len, struct in_ifad
|
||||
cfg.fc_scope = RT_SCOPE_HOST;
|
||||
|
||||
if (cmd == RTM_NEWROUTE)
|
||||
fib_table_insert(tb, &cfg);
|
||||
fib_table_insert(net, tb, &cfg);
|
||||
else
|
||||
fib_table_delete(tb, &cfg);
|
||||
fib_table_delete(net, tb, &cfg);
|
||||
}
|
||||
|
||||
void fib_add_ifaddr(struct in_ifaddr *ifa)
|
||||
@@ -1254,7 +1241,7 @@ static void ip_fib_net_exit(struct net *net)
|
||||
|
||||
hlist_for_each_entry_safe(tb, tmp, head, tb_hlist) {
|
||||
hlist_del(&tb->tb_hlist);
|
||||
fib_table_flush(tb);
|
||||
fib_table_flush(net, tb);
|
||||
fib_free_table(tb);
|
||||
}
|
||||
}
|
||||
|
||||
+10
-2
@@ -164,6 +164,14 @@ static struct fib_table *fib_empty_table(struct net *net)
|
||||
return NULL;
|
||||
}
|
||||
|
||||
static int call_fib_rule_notifiers(struct net *net,
|
||||
enum fib_event_type event_type)
|
||||
{
|
||||
struct fib_notifier_info info;
|
||||
|
||||
return call_fib_notifiers(net, event_type, &info);
|
||||
}
|
||||
|
||||
static const struct nla_policy fib4_rule_policy[FRA_MAX+1] = {
|
||||
FRA_GENERIC_POLICY,
|
||||
[FRA_FLOW] = { .type = NLA_U32 },
|
||||
@@ -220,7 +228,7 @@ static int fib4_rule_configure(struct fib_rule *rule, struct sk_buff *skb,
|
||||
rule4->tos = frh->tos;
|
||||
|
||||
net->ipv4.fib_has_custom_rules = true;
|
||||
fib_flush_external(rule->fr_net);
|
||||
call_fib_rule_notifiers(net, FIB_EVENT_RULE_ADD);
|
||||
|
||||
err = 0;
|
||||
errout:
|
||||
@@ -242,7 +250,7 @@ static int fib4_rule_delete(struct fib_rule *rule)
|
||||
net->ipv4.fib_num_tclassid_users--;
|
||||
#endif
|
||||
net->ipv4.fib_has_custom_rules = true;
|
||||
fib_flush_external(rule->fr_net);
|
||||
call_fib_rule_notifiers(net, FIB_EVENT_RULE_DEL);
|
||||
errout:
|
||||
return err;
|
||||
}
|
||||
|
||||
+60
-106
@@ -73,6 +73,7 @@
|
||||
#include <linux/slab.h>
|
||||
#include <linux/export.h>
|
||||
#include <linux/vmalloc.h>
|
||||
#include <linux/notifier.h>
|
||||
#include <net/net_namespace.h>
|
||||
#include <net/ip.h>
|
||||
#include <net/protocol.h>
|
||||
@@ -80,10 +81,47 @@
|
||||
#include <net/tcp.h>
|
||||
#include <net/sock.h>
|
||||
#include <net/ip_fib.h>
|
||||
#include <net/switchdev.h>
|
||||
#include <trace/events/fib.h>
|
||||
#include "fib_lookup.h"
|
||||
|
||||
static BLOCKING_NOTIFIER_HEAD(fib_chain);
|
||||
|
||||
int register_fib_notifier(struct notifier_block *nb)
|
||||
{
|
||||
return blocking_notifier_chain_register(&fib_chain, nb);
|
||||
}
|
||||
EXPORT_SYMBOL(register_fib_notifier);
|
||||
|
||||
int unregister_fib_notifier(struct notifier_block *nb)
|
||||
{
|
||||
return blocking_notifier_chain_unregister(&fib_chain, nb);
|
||||
}
|
||||
EXPORT_SYMBOL(unregister_fib_notifier);
|
||||
|
||||
int call_fib_notifiers(struct net *net, enum fib_event_type event_type,
|
||||
struct fib_notifier_info *info)
|
||||
{
|
||||
info->net = net;
|
||||
return blocking_notifier_call_chain(&fib_chain, event_type, info);
|
||||
}
|
||||
|
||||
static int call_fib_entry_notifiers(struct net *net,
|
||||
enum fib_event_type event_type, u32 dst,
|
||||
int dst_len, struct fib_info *fi,
|
||||
u8 tos, u8 type, u32 tb_id, u32 nlflags)
|
||||
{
|
||||
struct fib_entry_notifier_info info = {
|
||||
.dst = dst,
|
||||
.dst_len = dst_len,
|
||||
.fi = fi,
|
||||
.tos = tos,
|
||||
.type = type,
|
||||
.tb_id = tb_id,
|
||||
.nlflags = nlflags,
|
||||
};
|
||||
return call_fib_notifiers(net, event_type, &info.info);
|
||||
}
|
||||
|
||||
#define MAX_STAT_DEPTH 32
|
||||
|
||||
#define KEYLENGTH (8*sizeof(t_key))
|
||||
@@ -1076,7 +1114,8 @@ static int fib_insert_alias(struct trie *t, struct key_vector *tp,
|
||||
}
|
||||
|
||||
/* Caller must hold RTNL. */
|
||||
int fib_table_insert(struct fib_table *tb, struct fib_config *cfg)
|
||||
int fib_table_insert(struct net *net, struct fib_table *tb,
|
||||
struct fib_config *cfg)
|
||||
{
|
||||
struct trie *t = (struct trie *)tb->tb_data;
|
||||
struct fib_alias *fa, *new_fa;
|
||||
@@ -1175,17 +1214,6 @@ int fib_table_insert(struct fib_table *tb, struct fib_config *cfg)
|
||||
new_fa->tb_id = tb->tb_id;
|
||||
new_fa->fa_default = -1;
|
||||
|
||||
err = switchdev_fib_ipv4_add(key, plen, fi,
|
||||
new_fa->fa_tos,
|
||||
cfg->fc_type,
|
||||
cfg->fc_nlflags,
|
||||
tb->tb_id);
|
||||
if (err) {
|
||||
switchdev_fib_ipv4_abort(fi);
|
||||
kmem_cache_free(fn_alias_kmem, new_fa);
|
||||
goto out;
|
||||
}
|
||||
|
||||
hlist_replace_rcu(&fa->fa_list, &new_fa->fa_list);
|
||||
|
||||
alias_free_mem_rcu(fa);
|
||||
@@ -1193,6 +1221,11 @@ int fib_table_insert(struct fib_table *tb, struct fib_config *cfg)
|
||||
fib_release_info(fi_drop);
|
||||
if (state & FA_S_ACCESSED)
|
||||
rt_cache_flush(cfg->fc_nlinfo.nl_net);
|
||||
|
||||
call_fib_entry_notifiers(net, FIB_EVENT_ENTRY_ADD,
|
||||
key, plen, fi,
|
||||
new_fa->fa_tos, cfg->fc_type,
|
||||
tb->tb_id, cfg->fc_nlflags);
|
||||
rtmsg_fib(RTM_NEWROUTE, htonl(key), new_fa, plen,
|
||||
tb->tb_id, &cfg->fc_nlinfo, nlflags);
|
||||
|
||||
@@ -1228,30 +1261,22 @@ int fib_table_insert(struct fib_table *tb, struct fib_config *cfg)
|
||||
new_fa->tb_id = tb->tb_id;
|
||||
new_fa->fa_default = -1;
|
||||
|
||||
/* (Optionally) offload fib entry to switch hardware. */
|
||||
err = switchdev_fib_ipv4_add(key, plen, fi, tos, cfg->fc_type,
|
||||
cfg->fc_nlflags, tb->tb_id);
|
||||
if (err) {
|
||||
switchdev_fib_ipv4_abort(fi);
|
||||
goto out_free_new_fa;
|
||||
}
|
||||
|
||||
/* Insert new entry to the list. */
|
||||
err = fib_insert_alias(t, tp, l, new_fa, fa, key);
|
||||
if (err)
|
||||
goto out_sw_fib_del;
|
||||
goto out_free_new_fa;
|
||||
|
||||
if (!plen)
|
||||
tb->tb_num_default++;
|
||||
|
||||
rt_cache_flush(cfg->fc_nlinfo.nl_net);
|
||||
call_fib_entry_notifiers(net, FIB_EVENT_ENTRY_ADD, key, plen, fi, tos,
|
||||
cfg->fc_type, tb->tb_id, cfg->fc_nlflags);
|
||||
rtmsg_fib(RTM_NEWROUTE, htonl(key), new_fa, plen, new_fa->tb_id,
|
||||
&cfg->fc_nlinfo, nlflags);
|
||||
succeeded:
|
||||
return 0;
|
||||
|
||||
out_sw_fib_del:
|
||||
switchdev_fib_ipv4_del(key, plen, fi, tos, cfg->fc_type, tb->tb_id);
|
||||
out_free_new_fa:
|
||||
kmem_cache_free(fn_alias_kmem, new_fa);
|
||||
out:
|
||||
@@ -1490,7 +1515,8 @@ static void fib_remove_alias(struct trie *t, struct key_vector *tp,
|
||||
}
|
||||
|
||||
/* Caller must hold RTNL. */
|
||||
int fib_table_delete(struct fib_table *tb, struct fib_config *cfg)
|
||||
int fib_table_delete(struct net *net, struct fib_table *tb,
|
||||
struct fib_config *cfg)
|
||||
{
|
||||
struct trie *t = (struct trie *) tb->tb_data;
|
||||
struct fib_alias *fa, *fa_to_delete;
|
||||
@@ -1543,9 +1569,9 @@ int fib_table_delete(struct fib_table *tb, struct fib_config *cfg)
|
||||
if (!fa_to_delete)
|
||||
return -ESRCH;
|
||||
|
||||
switchdev_fib_ipv4_del(key, plen, fa_to_delete->fa_info, tos,
|
||||
cfg->fc_type, tb->tb_id);
|
||||
|
||||
call_fib_entry_notifiers(net, FIB_EVENT_ENTRY_DEL, key, plen,
|
||||
fa_to_delete->fa_info, tos, cfg->fc_type,
|
||||
tb->tb_id, 0);
|
||||
rtmsg_fib(RTM_DELROUTE, htonl(key), fa_to_delete, plen, tb->tb_id,
|
||||
&cfg->fc_nlinfo, 0);
|
||||
|
||||
@@ -1734,82 +1760,8 @@ out:
|
||||
return NULL;
|
||||
}
|
||||
|
||||
/* Caller must hold RTNL */
|
||||
void fib_table_flush_external(struct fib_table *tb)
|
||||
{
|
||||
struct trie *t = (struct trie *)tb->tb_data;
|
||||
struct key_vector *pn = t->kv;
|
||||
unsigned long cindex = 1;
|
||||
struct hlist_node *tmp;
|
||||
struct fib_alias *fa;
|
||||
|
||||
/* walk trie in reverse order */
|
||||
for (;;) {
|
||||
unsigned char slen = 0;
|
||||
struct key_vector *n;
|
||||
|
||||
if (!(cindex--)) {
|
||||
t_key pkey = pn->key;
|
||||
|
||||
/* cannot resize the trie vector */
|
||||
if (IS_TRIE(pn))
|
||||
break;
|
||||
|
||||
/* resize completed node */
|
||||
pn = resize(t, pn);
|
||||
cindex = get_index(pkey, pn);
|
||||
|
||||
continue;
|
||||
}
|
||||
|
||||
/* grab the next available node */
|
||||
n = get_child(pn, cindex);
|
||||
if (!n)
|
||||
continue;
|
||||
|
||||
if (IS_TNODE(n)) {
|
||||
/* record pn and cindex for leaf walking */
|
||||
pn = n;
|
||||
cindex = 1ul << n->bits;
|
||||
|
||||
continue;
|
||||
}
|
||||
|
||||
hlist_for_each_entry_safe(fa, tmp, &n->leaf, fa_list) {
|
||||
struct fib_info *fi = fa->fa_info;
|
||||
|
||||
/* if alias was cloned to local then we just
|
||||
* need to remove the local copy from main
|
||||
*/
|
||||
if (tb->tb_id != fa->tb_id) {
|
||||
hlist_del_rcu(&fa->fa_list);
|
||||
alias_free_mem_rcu(fa);
|
||||
continue;
|
||||
}
|
||||
|
||||
/* record local slen */
|
||||
slen = fa->fa_slen;
|
||||
|
||||
if (!fi || !(fi->fib_flags & RTNH_F_OFFLOAD))
|
||||
continue;
|
||||
|
||||
switchdev_fib_ipv4_del(n->key, KEYLENGTH - fa->fa_slen,
|
||||
fi, fa->fa_tos, fa->fa_type,
|
||||
tb->tb_id);
|
||||
}
|
||||
|
||||
/* update leaf slen */
|
||||
n->slen = slen;
|
||||
|
||||
if (hlist_empty(&n->leaf)) {
|
||||
put_child_root(pn, n->key, NULL);
|
||||
node_free(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/* Caller must hold RTNL. */
|
||||
int fib_table_flush(struct fib_table *tb)
|
||||
int fib_table_flush(struct net *net, struct fib_table *tb)
|
||||
{
|
||||
struct trie *t = (struct trie *)tb->tb_data;
|
||||
struct key_vector *pn = t->kv;
|
||||
@@ -1858,9 +1810,11 @@ int fib_table_flush(struct fib_table *tb)
|
||||
continue;
|
||||
}
|
||||
|
||||
switchdev_fib_ipv4_del(n->key, KEYLENGTH - fa->fa_slen,
|
||||
fi, fa->fa_tos, fa->fa_type,
|
||||
tb->tb_id);
|
||||
call_fib_entry_notifiers(net, FIB_EVENT_ENTRY_DEL,
|
||||
n->key,
|
||||
KEYLENGTH - fa->fa_slen,
|
||||
fi, fa->fa_tos, fa->fa_type,
|
||||
tb->tb_id, 0);
|
||||
hlist_del_rcu(&fa->fa_list);
|
||||
fib_release_info(fa->fa_info);
|
||||
alias_free_mem_rcu(fa);
|
||||
|
||||
@@ -24,7 +24,7 @@ static struct sk_buff *gre_gso_segment(struct sk_buff *skb,
|
||||
__be16 protocol = skb->protocol;
|
||||
u16 mac_len = skb->mac_len;
|
||||
int gre_offset, outer_hlen;
|
||||
bool need_csum, ufo;
|
||||
bool need_csum, ufo, gso_partial;
|
||||
|
||||
if (!skb->encapsulation)
|
||||
goto out;
|
||||
@@ -69,6 +69,8 @@ static struct sk_buff *gre_gso_segment(struct sk_buff *skb,
|
||||
goto out;
|
||||
}
|
||||
|
||||
gso_partial = !!(skb_shinfo(segs)->gso_type & SKB_GSO_PARTIAL);
|
||||
|
||||
outer_hlen = skb_tnl_header_len(skb);
|
||||
gre_offset = outer_hlen - tnl_hlen;
|
||||
skb = segs;
|
||||
@@ -96,7 +98,7 @@ static struct sk_buff *gre_gso_segment(struct sk_buff *skb,
|
||||
greh = (struct gre_base_hdr *)skb_transport_header(skb);
|
||||
pcsum = (__sum16 *)(greh + 1);
|
||||
|
||||
if (skb_is_gso(skb)) {
|
||||
if (gso_partial) {
|
||||
unsigned int partial_adj;
|
||||
|
||||
/* Adjust checksum to account for the fact that
|
||||
|
||||
+3
-2
@@ -312,6 +312,7 @@ static int ip_rcv_finish(struct net *net, struct sock *sk, struct sk_buff *skb)
|
||||
{
|
||||
const struct iphdr *iph = ip_hdr(skb);
|
||||
struct rtable *rt;
|
||||
struct net_device *dev = skb->dev;
|
||||
|
||||
/* if ingress device is enslaved to an L3 master device pass the
|
||||
* skb to its handler for processing
|
||||
@@ -341,7 +342,7 @@ static int ip_rcv_finish(struct net *net, struct sock *sk, struct sk_buff *skb)
|
||||
*/
|
||||
if (!skb_valid_dst(skb)) {
|
||||
int err = ip_route_input_noref(skb, iph->daddr, iph->saddr,
|
||||
iph->tos, skb->dev);
|
||||
iph->tos, dev);
|
||||
if (unlikely(err)) {
|
||||
if (err == -EXDEV)
|
||||
__NET_INC_STATS(net, LINUX_MIB_IPRPFILTER);
|
||||
@@ -370,7 +371,7 @@ static int ip_rcv_finish(struct net *net, struct sock *sk, struct sk_buff *skb)
|
||||
__IP_UPD_PO_STATS(net, IPSTATS_MIB_INBCAST, skb->len);
|
||||
} else if (skb->pkt_type == PACKET_BROADCAST ||
|
||||
skb->pkt_type == PACKET_MULTICAST) {
|
||||
struct in_device *in_dev = __in_dev_get_rcu(skb->dev);
|
||||
struct in_device *in_dev = __in_dev_get_rcu(dev);
|
||||
|
||||
/* RFC 1122 3.3.6:
|
||||
*
|
||||
|
||||
+14
-1
@@ -88,6 +88,7 @@ static int vti_rcv_cb(struct sk_buff *skb, int err)
|
||||
struct net_device *dev;
|
||||
struct pcpu_sw_netstats *tstats;
|
||||
struct xfrm_state *x;
|
||||
struct xfrm_mode *inner_mode;
|
||||
struct ip_tunnel *tunnel = XFRM_TUNNEL_SKB_CB(skb)->tunnel.ip4;
|
||||
u32 orig_mark = skb->mark;
|
||||
int ret;
|
||||
@@ -105,7 +106,19 @@ static int vti_rcv_cb(struct sk_buff *skb, int err)
|
||||
}
|
||||
|
||||
x = xfrm_input_state(skb);
|
||||
family = x->inner_mode->afinfo->family;
|
||||
|
||||
inner_mode = x->inner_mode;
|
||||
|
||||
if (x->sel.family == AF_UNSPEC) {
|
||||
inner_mode = xfrm_ip2inner_mode(x, XFRM_MODE_SKB_CB(skb)->protocol);
|
||||
if (inner_mode == NULL) {
|
||||
XFRM_INC_STATS(dev_net(skb->dev),
|
||||
LINUX_MIB_XFRMINSTATEMODEERROR);
|
||||
return -EINVAL;
|
||||
}
|
||||
}
|
||||
|
||||
family = inner_mode->afinfo->family;
|
||||
|
||||
skb->mark = be32_to_cpu(tunnel->parms.i_key);
|
||||
ret = xfrm_policy_check(NULL, XFRM_POLICY_IN, skb, family);
|
||||
|
||||
+7
-3
@@ -2076,6 +2076,7 @@ static int __ipmr_fill_mroute(struct mr_table *mrt, struct sk_buff *skb,
|
||||
struct rta_mfc_stats mfcs;
|
||||
struct nlattr *mp_attr;
|
||||
struct rtnexthop *nhp;
|
||||
unsigned long lastuse;
|
||||
int ct;
|
||||
|
||||
/* If cache is unresolved, don't try to parse IIF and OIF */
|
||||
@@ -2105,12 +2106,14 @@ static int __ipmr_fill_mroute(struct mr_table *mrt, struct sk_buff *skb,
|
||||
|
||||
nla_nest_end(skb, mp_attr);
|
||||
|
||||
lastuse = READ_ONCE(c->mfc_un.res.lastuse);
|
||||
lastuse = time_after_eq(jiffies, lastuse) ? jiffies - lastuse : 0;
|
||||
|
||||
mfcs.mfcs_packets = c->mfc_un.res.pkt;
|
||||
mfcs.mfcs_bytes = c->mfc_un.res.bytes;
|
||||
mfcs.mfcs_wrong_if = c->mfc_un.res.wrong_if;
|
||||
if (nla_put_64bit(skb, RTA_MFC_STATS, sizeof(mfcs), &mfcs, RTA_PAD) ||
|
||||
nla_put_u64_64bit(skb, RTA_EXPIRES,
|
||||
jiffies_to_clock_t(c->mfc_un.res.lastuse),
|
||||
nla_put_u64_64bit(skb, RTA_EXPIRES, jiffies_to_clock_t(lastuse),
|
||||
RTA_PAD))
|
||||
return -EMSGSIZE;
|
||||
|
||||
@@ -2120,7 +2123,7 @@ static int __ipmr_fill_mroute(struct mr_table *mrt, struct sk_buff *skb,
|
||||
|
||||
int ipmr_get_route(struct net *net, struct sk_buff *skb,
|
||||
__be32 saddr, __be32 daddr,
|
||||
struct rtmsg *rtm, int nowait)
|
||||
struct rtmsg *rtm, int nowait, u32 portid)
|
||||
{
|
||||
struct mfc_cache *cache;
|
||||
struct mr_table *mrt;
|
||||
@@ -2165,6 +2168,7 @@ int ipmr_get_route(struct net *net, struct sk_buff *skb,
|
||||
return -ENOMEM;
|
||||
}
|
||||
|
||||
NETLINK_CB(skb2).portid = portid;
|
||||
skb_push(skb2, sizeof(struct iphdr));
|
||||
skb_reset_network_header(skb2);
|
||||
iph = ip_hdr(skb2);
|
||||
|
||||
@@ -156,7 +156,7 @@ static struct nf_loginfo trace_loginfo = {
|
||||
.u = {
|
||||
.log = {
|
||||
.level = 4,
|
||||
.logflags = NF_LOG_MASK,
|
||||
.logflags = NF_LOG_DEFAULT_MASK,
|
||||
},
|
||||
},
|
||||
};
|
||||
|
||||
@@ -110,7 +110,7 @@ static unsigned int ipv4_helper(void *priv,
|
||||
if (!help)
|
||||
return NF_ACCEPT;
|
||||
|
||||
/* rcu_read_lock()ed by nf_hook_slow */
|
||||
/* rcu_read_lock()ed by nf_hook_thresh */
|
||||
helper = rcu_dereference(help->helper);
|
||||
if (!helper)
|
||||
return NF_ACCEPT;
|
||||
|
||||
@@ -149,7 +149,7 @@ icmp_error_message(struct net *net, struct nf_conn *tmpl, struct sk_buff *skb,
|
||||
return -NF_ACCEPT;
|
||||
}
|
||||
|
||||
/* rcu_read_lock()ed by nf_hook_slow */
|
||||
/* rcu_read_lock()ed by nf_hook_thresh */
|
||||
innerproto = __nf_ct_l4proto_find(PF_INET, origtuple.dst.protonum);
|
||||
|
||||
/* Ordinarily, we'd expect the inverted tupleproto, but it's
|
||||
|
||||
@@ -30,7 +30,7 @@ static struct nf_loginfo default_loginfo = {
|
||||
.u = {
|
||||
.log = {
|
||||
.level = LOGLEVEL_NOTICE,
|
||||
.logflags = NF_LOG_MASK,
|
||||
.logflags = NF_LOG_DEFAULT_MASK,
|
||||
},
|
||||
},
|
||||
};
|
||||
|
||||
@@ -29,7 +29,7 @@ static struct nf_loginfo default_loginfo = {
|
||||
.u = {
|
||||
.log = {
|
||||
.level = LOGLEVEL_NOTICE,
|
||||
.logflags = NF_LOG_MASK,
|
||||
.logflags = NF_LOG_DEFAULT_MASK,
|
||||
},
|
||||
},
|
||||
};
|
||||
@@ -46,7 +46,7 @@ static void dump_ipv4_packet(struct nf_log_buf *m,
|
||||
if (info->type == NF_LOG_TYPE_LOG)
|
||||
logflags = info->u.log.logflags;
|
||||
else
|
||||
logflags = NF_LOG_MASK;
|
||||
logflags = NF_LOG_DEFAULT_MASK;
|
||||
|
||||
ih = skb_header_pointer(skb, iphoff, sizeof(_iph), &_iph);
|
||||
if (ih == NULL) {
|
||||
@@ -76,7 +76,7 @@ static void dump_ipv4_packet(struct nf_log_buf *m,
|
||||
if (ntohs(ih->frag_off) & IP_OFFSET)
|
||||
nf_log_buf_add(m, "FRAG:%u ", ntohs(ih->frag_off) & IP_OFFSET);
|
||||
|
||||
if ((logflags & XT_LOG_IPOPT) &&
|
||||
if ((logflags & NF_LOG_IPOPT) &&
|
||||
ih->ihl * 4 > sizeof(struct iphdr)) {
|
||||
const unsigned char *op;
|
||||
unsigned char _opt[4 * 15 - sizeof(struct iphdr)];
|
||||
@@ -250,7 +250,7 @@ static void dump_ipv4_packet(struct nf_log_buf *m,
|
||||
}
|
||||
|
||||
/* Max length: 15 "UID=4294967295 " */
|
||||
if ((logflags & XT_LOG_UID) && !iphoff)
|
||||
if ((logflags & NF_LOG_UID) && !iphoff)
|
||||
nf_log_dump_sk_uid_gid(m, skb->sk);
|
||||
|
||||
/* Max length: 16 "MARK=0xFFFFFFFF " */
|
||||
@@ -282,7 +282,7 @@ static void dump_ipv4_mac_header(struct nf_log_buf *m,
|
||||
if (info->type == NF_LOG_TYPE_LOG)
|
||||
logflags = info->u.log.logflags;
|
||||
|
||||
if (!(logflags & XT_LOG_MACDECODE))
|
||||
if (!(logflags & NF_LOG_MACDECODE))
|
||||
goto fallback;
|
||||
|
||||
switch (dev->type) {
|
||||
|
||||
@@ -88,8 +88,8 @@ gre_manip_pkt(struct sk_buff *skb,
|
||||
const struct nf_conntrack_tuple *tuple,
|
||||
enum nf_nat_manip_type maniptype)
|
||||
{
|
||||
const struct gre_hdr *greh;
|
||||
struct gre_hdr_pptp *pgreh;
|
||||
const struct gre_base_hdr *greh;
|
||||
struct pptp_gre_header *pgreh;
|
||||
|
||||
/* pgreh includes two optional 32bit fields which are not required
|
||||
* to be there. That's where the magic '8' comes from */
|
||||
@@ -97,18 +97,19 @@ gre_manip_pkt(struct sk_buff *skb,
|
||||
return false;
|
||||
|
||||
greh = (void *)skb->data + hdroff;
|
||||
pgreh = (struct gre_hdr_pptp *)greh;
|
||||
pgreh = (struct pptp_gre_header *)greh;
|
||||
|
||||
/* we only have destination manip of a packet, since 'source key'
|
||||
* is not present in the packet itself */
|
||||
if (maniptype != NF_NAT_MANIP_DST)
|
||||
return true;
|
||||
switch (greh->version) {
|
||||
case GRE_VERSION_1701:
|
||||
|
||||
switch (greh->flags & GRE_VERSION) {
|
||||
case GRE_VERSION_0:
|
||||
/* We do not currently NAT any GREv0 packets.
|
||||
* Try to behave like "nf_nat_proto_unknown" */
|
||||
break;
|
||||
case GRE_VERSION_PPTP:
|
||||
case GRE_VERSION_1:
|
||||
pr_debug("call_id -> 0x%04x\n", ntohs(tuple->dst.u.gre.key));
|
||||
pgreh->call_id = tuple->dst.u.gre.key;
|
||||
break;
|
||||
|
||||
@@ -21,7 +21,7 @@ nft_do_chain_arp(void *priv,
|
||||
{
|
||||
struct nft_pktinfo pkt;
|
||||
|
||||
nft_set_pktinfo(&pkt, skb, state);
|
||||
nft_set_pktinfo_unspec(&pkt, skb, state);
|
||||
|
||||
return nft_do_chain(&pkt, priv);
|
||||
}
|
||||
@@ -80,7 +80,10 @@ static int __init nf_tables_arp_init(void)
|
||||
{
|
||||
int ret;
|
||||
|
||||
nft_register_chain_type(&filter_arp);
|
||||
ret = nft_register_chain_type(&filter_arp);
|
||||
if (ret < 0)
|
||||
return ret;
|
||||
|
||||
ret = register_pernet_subsys(&nf_tables_arp_net_ops);
|
||||
if (ret < 0)
|
||||
nft_unregister_chain_type(&filter_arp);
|
||||
|
||||
@@ -103,7 +103,10 @@ static int __init nf_tables_ipv4_init(void)
|
||||
{
|
||||
int ret;
|
||||
|
||||
nft_register_chain_type(&filter_ipv4);
|
||||
ret = nft_register_chain_type(&filter_ipv4);
|
||||
if (ret < 0)
|
||||
return ret;
|
||||
|
||||
ret = register_pernet_subsys(&nf_tables_ipv4_net_ops);
|
||||
if (ret < 0)
|
||||
nft_unregister_chain_type(&filter_ipv4);
|
||||
|
||||
@@ -31,6 +31,7 @@ static unsigned int nf_route_table_hook(void *priv,
|
||||
__be32 saddr, daddr;
|
||||
u_int8_t tos;
|
||||
const struct iphdr *iph;
|
||||
int err;
|
||||
|
||||
/* root is playing with raw sockets. */
|
||||
if (skb->len < sizeof(struct iphdr) ||
|
||||
@@ -46,15 +47,17 @@ static unsigned int nf_route_table_hook(void *priv,
|
||||
tos = iph->tos;
|
||||
|
||||
ret = nft_do_chain(&pkt, priv);
|
||||
if (ret != NF_DROP && ret != NF_QUEUE) {
|
||||
if (ret != NF_DROP && ret != NF_STOLEN) {
|
||||
iph = ip_hdr(skb);
|
||||
|
||||
if (iph->saddr != saddr ||
|
||||
iph->daddr != daddr ||
|
||||
skb->mark != mark ||
|
||||
iph->tos != tos)
|
||||
if (ip_route_me_harder(state->net, skb, RTN_UNSPEC))
|
||||
ret = NF_DROP;
|
||||
iph->tos != tos) {
|
||||
err = ip_route_me_harder(state->net, skb, RTN_UNSPEC);
|
||||
if (err < 0)
|
||||
ret = NF_DROP_ERR(err);
|
||||
}
|
||||
}
|
||||
return ret;
|
||||
}
|
||||
|
||||
+79
-55
@@ -46,6 +46,8 @@
|
||||
#include <net/sock.h>
|
||||
#include <net/raw.h>
|
||||
|
||||
#define TCPUDP_MIB_MAX max_t(u32, UDP_MIB_MAX, TCP_MIB_MAX)
|
||||
|
||||
/*
|
||||
* Report socket allocation statistics [mea@utu.fi]
|
||||
*/
|
||||
@@ -356,22 +358,22 @@ static void icmp_put(struct seq_file *seq)
|
||||
atomic_long_t *ptr = net->mib.icmpmsg_statistics->mibs;
|
||||
|
||||
seq_puts(seq, "\nIcmp: InMsgs InErrors InCsumErrors");
|
||||
for (i = 0; icmpmibmap[i].name != NULL; i++)
|
||||
for (i = 0; icmpmibmap[i].name; i++)
|
||||
seq_printf(seq, " In%s", icmpmibmap[i].name);
|
||||
seq_puts(seq, " OutMsgs OutErrors");
|
||||
for (i = 0; icmpmibmap[i].name != NULL; i++)
|
||||
for (i = 0; icmpmibmap[i].name; i++)
|
||||
seq_printf(seq, " Out%s", icmpmibmap[i].name);
|
||||
seq_printf(seq, "\nIcmp: %lu %lu %lu",
|
||||
snmp_fold_field(net->mib.icmp_statistics, ICMP_MIB_INMSGS),
|
||||
snmp_fold_field(net->mib.icmp_statistics, ICMP_MIB_INERRORS),
|
||||
snmp_fold_field(net->mib.icmp_statistics, ICMP_MIB_CSUMERRORS));
|
||||
for (i = 0; icmpmibmap[i].name != NULL; i++)
|
||||
for (i = 0; icmpmibmap[i].name; i++)
|
||||
seq_printf(seq, " %lu",
|
||||
atomic_long_read(ptr + icmpmibmap[i].index));
|
||||
seq_printf(seq, " %lu %lu",
|
||||
snmp_fold_field(net->mib.icmp_statistics, ICMP_MIB_OUTMSGS),
|
||||
snmp_fold_field(net->mib.icmp_statistics, ICMP_MIB_OUTERRORS));
|
||||
for (i = 0; icmpmibmap[i].name != NULL; i++)
|
||||
for (i = 0; icmpmibmap[i].name; i++)
|
||||
seq_printf(seq, " %lu",
|
||||
atomic_long_read(ptr + (icmpmibmap[i].index | 0x100)));
|
||||
}
|
||||
@@ -379,14 +381,16 @@ static void icmp_put(struct seq_file *seq)
|
||||
/*
|
||||
* Called from the PROCfs module. This outputs /proc/net/snmp.
|
||||
*/
|
||||
static int snmp_seq_show(struct seq_file *seq, void *v)
|
||||
static int snmp_seq_show_ipstats(struct seq_file *seq, void *v)
|
||||
{
|
||||
int i;
|
||||
struct net *net = seq->private;
|
||||
u64 buff64[IPSTATS_MIB_MAX];
|
||||
int i;
|
||||
|
||||
memset(buff64, 0, IPSTATS_MIB_MAX * sizeof(u64));
|
||||
|
||||
seq_puts(seq, "Ip: Forwarding DefaultTTL");
|
||||
|
||||
for (i = 0; snmp4_ipstats_list[i].name != NULL; i++)
|
||||
for (i = 0; snmp4_ipstats_list[i].name; i++)
|
||||
seq_printf(seq, " %s", snmp4_ipstats_list[i].name);
|
||||
|
||||
seq_printf(seq, "\nIp: %d %d",
|
||||
@@ -394,54 +398,74 @@ static int snmp_seq_show(struct seq_file *seq, void *v)
|
||||
net->ipv4.sysctl_ip_default_ttl);
|
||||
|
||||
BUILD_BUG_ON(offsetof(struct ipstats_mib, mibs) != 0);
|
||||
for (i = 0; snmp4_ipstats_list[i].name != NULL; i++)
|
||||
seq_printf(seq, " %llu",
|
||||
snmp_fold_field64(net->mib.ip_statistics,
|
||||
snmp4_ipstats_list[i].entry,
|
||||
offsetof(struct ipstats_mib, syncp)));
|
||||
snmp_get_cpu_field64_batch(buff64, snmp4_ipstats_list,
|
||||
net->mib.ip_statistics,
|
||||
offsetof(struct ipstats_mib, syncp));
|
||||
for (i = 0; snmp4_ipstats_list[i].name; i++)
|
||||
seq_printf(seq, " %llu", buff64[i]);
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int snmp_seq_show_tcp_udp(struct seq_file *seq, void *v)
|
||||
{
|
||||
unsigned long buff[TCPUDP_MIB_MAX];
|
||||
struct net *net = seq->private;
|
||||
int i;
|
||||
|
||||
memset(buff, 0, TCPUDP_MIB_MAX * sizeof(unsigned long));
|
||||
|
||||
seq_puts(seq, "\nTcp:");
|
||||
for (i = 0; snmp4_tcp_list[i].name; i++)
|
||||
seq_printf(seq, " %s", snmp4_tcp_list[i].name);
|
||||
|
||||
seq_puts(seq, "\nTcp:");
|
||||
snmp_get_cpu_field_batch(buff, snmp4_tcp_list,
|
||||
net->mib.tcp_statistics);
|
||||
for (i = 0; snmp4_tcp_list[i].name; i++) {
|
||||
/* MaxConn field is signed, RFC 2012 */
|
||||
if (snmp4_tcp_list[i].entry == TCP_MIB_MAXCONN)
|
||||
seq_printf(seq, " %ld", buff[i]);
|
||||
else
|
||||
seq_printf(seq, " %lu", buff[i]);
|
||||
}
|
||||
|
||||
memset(buff, 0, TCPUDP_MIB_MAX * sizeof(unsigned long));
|
||||
|
||||
snmp_get_cpu_field_batch(buff, snmp4_udp_list,
|
||||
net->mib.udp_statistics);
|
||||
seq_puts(seq, "\nUdp:");
|
||||
for (i = 0; snmp4_udp_list[i].name; i++)
|
||||
seq_printf(seq, " %s", snmp4_udp_list[i].name);
|
||||
seq_puts(seq, "\nUdp:");
|
||||
for (i = 0; snmp4_udp_list[i].name; i++)
|
||||
seq_printf(seq, " %lu", buff[i]);
|
||||
|
||||
memset(buff, 0, TCPUDP_MIB_MAX * sizeof(unsigned long));
|
||||
|
||||
/* the UDP and UDP-Lite MIBs are the same */
|
||||
seq_puts(seq, "\nUdpLite:");
|
||||
snmp_get_cpu_field_batch(buff, snmp4_udp_list,
|
||||
net->mib.udplite_statistics);
|
||||
for (i = 0; snmp4_udp_list[i].name; i++)
|
||||
seq_printf(seq, " %s", snmp4_udp_list[i].name);
|
||||
seq_puts(seq, "\nUdpLite:");
|
||||
for (i = 0; snmp4_udp_list[i].name; i++)
|
||||
seq_printf(seq, " %lu", buff[i]);
|
||||
|
||||
seq_putc(seq, '\n');
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int snmp_seq_show(struct seq_file *seq, void *v)
|
||||
{
|
||||
snmp_seq_show_ipstats(seq, v);
|
||||
|
||||
icmp_put(seq); /* RFC 2011 compatibility */
|
||||
icmpmsg_put(seq);
|
||||
|
||||
seq_puts(seq, "\nTcp:");
|
||||
for (i = 0; snmp4_tcp_list[i].name != NULL; i++)
|
||||
seq_printf(seq, " %s", snmp4_tcp_list[i].name);
|
||||
snmp_seq_show_tcp_udp(seq, v);
|
||||
|
||||
seq_puts(seq, "\nTcp:");
|
||||
for (i = 0; snmp4_tcp_list[i].name != NULL; i++) {
|
||||
/* MaxConn field is signed, RFC 2012 */
|
||||
if (snmp4_tcp_list[i].entry == TCP_MIB_MAXCONN)
|
||||
seq_printf(seq, " %ld",
|
||||
snmp_fold_field(net->mib.tcp_statistics,
|
||||
snmp4_tcp_list[i].entry));
|
||||
else
|
||||
seq_printf(seq, " %lu",
|
||||
snmp_fold_field(net->mib.tcp_statistics,
|
||||
snmp4_tcp_list[i].entry));
|
||||
}
|
||||
|
||||
seq_puts(seq, "\nUdp:");
|
||||
for (i = 0; snmp4_udp_list[i].name != NULL; i++)
|
||||
seq_printf(seq, " %s", snmp4_udp_list[i].name);
|
||||
|
||||
seq_puts(seq, "\nUdp:");
|
||||
for (i = 0; snmp4_udp_list[i].name != NULL; i++)
|
||||
seq_printf(seq, " %lu",
|
||||
snmp_fold_field(net->mib.udp_statistics,
|
||||
snmp4_udp_list[i].entry));
|
||||
|
||||
/* the UDP and UDP-Lite MIBs are the same */
|
||||
seq_puts(seq, "\nUdpLite:");
|
||||
for (i = 0; snmp4_udp_list[i].name != NULL; i++)
|
||||
seq_printf(seq, " %s", snmp4_udp_list[i].name);
|
||||
|
||||
seq_puts(seq, "\nUdpLite:");
|
||||
for (i = 0; snmp4_udp_list[i].name != NULL; i++)
|
||||
seq_printf(seq, " %lu",
|
||||
snmp_fold_field(net->mib.udplite_statistics,
|
||||
snmp4_udp_list[i].entry));
|
||||
|
||||
seq_putc(seq, '\n');
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -469,21 +493,21 @@ static int netstat_seq_show(struct seq_file *seq, void *v)
|
||||
struct net *net = seq->private;
|
||||
|
||||
seq_puts(seq, "TcpExt:");
|
||||
for (i = 0; snmp4_net_list[i].name != NULL; i++)
|
||||
for (i = 0; snmp4_net_list[i].name; i++)
|
||||
seq_printf(seq, " %s", snmp4_net_list[i].name);
|
||||
|
||||
seq_puts(seq, "\nTcpExt:");
|
||||
for (i = 0; snmp4_net_list[i].name != NULL; i++)
|
||||
for (i = 0; snmp4_net_list[i].name; i++)
|
||||
seq_printf(seq, " %lu",
|
||||
snmp_fold_field(net->mib.net_statistics,
|
||||
snmp4_net_list[i].entry));
|
||||
|
||||
seq_puts(seq, "\nIpExt:");
|
||||
for (i = 0; snmp4_ipextstats_list[i].name != NULL; i++)
|
||||
for (i = 0; snmp4_ipextstats_list[i].name; i++)
|
||||
seq_printf(seq, " %s", snmp4_ipextstats_list[i].name);
|
||||
|
||||
seq_puts(seq, "\nIpExt:");
|
||||
for (i = 0; snmp4_ipextstats_list[i].name != NULL; i++)
|
||||
for (i = 0; snmp4_ipextstats_list[i].name; i++)
|
||||
seq_printf(seq, " %llu",
|
||||
snmp_fold_field64(net->mib.ip_statistics,
|
||||
snmp4_ipextstats_list[i].entry,
|
||||
|
||||
+10
-3
@@ -476,12 +476,18 @@ u32 ip_idents_reserve(u32 hash, int segs)
|
||||
atomic_t *p_id = ip_idents + hash % IP_IDENTS_SZ;
|
||||
u32 old = ACCESS_ONCE(*p_tstamp);
|
||||
u32 now = (u32)jiffies;
|
||||
u32 delta = 0;
|
||||
u32 new, delta = 0;
|
||||
|
||||
if (old != now && cmpxchg(p_tstamp, old, now) == old)
|
||||
delta = prandom_u32_max(now - old);
|
||||
|
||||
return atomic_add_return(segs + delta, p_id) - segs;
|
||||
/* Do not use atomic_add_return() as it makes UBSAN unhappy */
|
||||
do {
|
||||
old = (u32)atomic_read(p_id);
|
||||
new = old + delta + segs;
|
||||
} while (atomic_cmpxchg(p_id, old, new) != old);
|
||||
|
||||
return new - segs;
|
||||
}
|
||||
EXPORT_SYMBOL(ip_idents_reserve);
|
||||
|
||||
@@ -2494,7 +2500,8 @@ static int rt_fill_info(struct net *net, __be32 dst, __be32 src, u32 table_id,
|
||||
IPV4_DEVCONF_ALL(net, MC_FORWARDING)) {
|
||||
int err = ipmr_get_route(net, skb,
|
||||
fl4->saddr, fl4->daddr,
|
||||
r, nowait);
|
||||
r, nowait, portid);
|
||||
|
||||
if (err <= 0) {
|
||||
if (!nowait) {
|
||||
if (err == 0)
|
||||
|
||||
+22
-4
@@ -387,7 +387,7 @@ void tcp_init_sock(struct sock *sk)
|
||||
|
||||
icsk->icsk_rto = TCP_TIMEOUT_INIT;
|
||||
tp->mdev_us = jiffies_to_usecs(TCP_TIMEOUT_INIT);
|
||||
tp->rtt_min[0].rtt = ~0U;
|
||||
minmax_reset(&tp->rtt_min, tcp_time_stamp, ~0U);
|
||||
|
||||
/* So many TCP implementations out there (incorrectly) count the
|
||||
* initial SYN frame in their delayed-ACK and congestion control
|
||||
@@ -396,6 +396,9 @@ void tcp_init_sock(struct sock *sk)
|
||||
*/
|
||||
tp->snd_cwnd = TCP_INIT_CWND;
|
||||
|
||||
/* There's a bubble in the pipe until at least the first ACK. */
|
||||
tp->app_limited = ~0U;
|
||||
|
||||
/* See draft-stevens-tcpca-spec-01 for discussion of the
|
||||
* initialization of these values.
|
||||
*/
|
||||
@@ -1014,6 +1017,9 @@ int tcp_sendpage(struct sock *sk, struct page *page, int offset,
|
||||
flags);
|
||||
|
||||
lock_sock(sk);
|
||||
|
||||
tcp_rate_check_app_limited(sk); /* is sending application-limited? */
|
||||
|
||||
res = do_tcp_sendpages(sk, page, offset, size, flags);
|
||||
release_sock(sk);
|
||||
return res;
|
||||
@@ -1115,6 +1121,8 @@ int tcp_sendmsg(struct sock *sk, struct msghdr *msg, size_t size)
|
||||
|
||||
timeo = sock_sndtimeo(sk, flags & MSG_DONTWAIT);
|
||||
|
||||
tcp_rate_check_app_limited(sk); /* is sending application-limited? */
|
||||
|
||||
/* Wait for a connection to finish. One exception is TCP Fast Open
|
||||
* (passive side) where data is allowed to be sent before a connection
|
||||
* is fully established.
|
||||
@@ -2704,7 +2712,7 @@ void tcp_get_info(struct sock *sk, struct tcp_info *info)
|
||||
{
|
||||
const struct tcp_sock *tp = tcp_sk(sk); /* iff sk_type == SOCK_STREAM */
|
||||
const struct inet_connection_sock *icsk = inet_csk(sk);
|
||||
u32 now = tcp_time_stamp;
|
||||
u32 now = tcp_time_stamp, intv;
|
||||
unsigned int start;
|
||||
int notsent_bytes;
|
||||
u64 rate64;
|
||||
@@ -2794,6 +2802,15 @@ void tcp_get_info(struct sock *sk, struct tcp_info *info)
|
||||
info->tcpi_min_rtt = tcp_min_rtt(tp);
|
||||
info->tcpi_data_segs_in = tp->data_segs_in;
|
||||
info->tcpi_data_segs_out = tp->data_segs_out;
|
||||
|
||||
info->tcpi_delivery_rate_app_limited = tp->rate_app_limited ? 1 : 0;
|
||||
rate = READ_ONCE(tp->rate_delivered);
|
||||
intv = READ_ONCE(tp->rate_interval_us);
|
||||
if (rate && intv) {
|
||||
rate64 = (u64)rate * tp->mss_cache * USEC_PER_SEC;
|
||||
do_div(rate64, intv);
|
||||
put_unaligned(rate64, &info->tcpi_delivery_rate);
|
||||
}
|
||||
}
|
||||
EXPORT_SYMBOL_GPL(tcp_get_info);
|
||||
|
||||
@@ -3261,11 +3278,12 @@ static void __init tcp_init_mem(void)
|
||||
|
||||
void __init tcp_init(void)
|
||||
{
|
||||
unsigned long limit;
|
||||
int max_rshare, max_wshare, cnt;
|
||||
unsigned long limit;
|
||||
unsigned int i;
|
||||
|
||||
sock_skb_cb_check_size(sizeof(struct tcp_skb_cb));
|
||||
BUILD_BUG_ON(sizeof(struct tcp_skb_cb) >
|
||||
FIELD_SIZEOF(struct sk_buff, cb));
|
||||
|
||||
percpu_counter_init(&tcp_sockets_allocated, 0, GFP_KERNEL);
|
||||
percpu_counter_init(&tcp_orphan_count, 0, GFP_KERNEL);
|
||||
|
||||
@@ -0,0 +1,896 @@
|
||||
/* Bottleneck Bandwidth and RTT (BBR) congestion control
|
||||
*
|
||||
* BBR congestion control computes the sending rate based on the delivery
|
||||
* rate (throughput) estimated from ACKs. In a nutshell:
|
||||
*
|
||||
* On each ACK, update our model of the network path:
|
||||
* bottleneck_bandwidth = windowed_max(delivered / elapsed, 10 round trips)
|
||||
* min_rtt = windowed_min(rtt, 10 seconds)
|
||||
* pacing_rate = pacing_gain * bottleneck_bandwidth
|
||||
* cwnd = max(cwnd_gain * bottleneck_bandwidth * min_rtt, 4)
|
||||
*
|
||||
* The core algorithm does not react directly to packet losses or delays,
|
||||
* although BBR may adjust the size of next send per ACK when loss is
|
||||
* observed, or adjust the sending rate if it estimates there is a
|
||||
* traffic policer, in order to keep the drop rate reasonable.
|
||||
*
|
||||
* BBR is described in detail in:
|
||||
* "BBR: Congestion-Based Congestion Control",
|
||||
* Neal Cardwell, Yuchung Cheng, C. Stephen Gunn, Soheil Hassas Yeganeh,
|
||||
* Van Jacobson. ACM Queue, Vol. 14 No. 5, September-October 2016.
|
||||
*
|
||||
* There is a public e-mail list for discussing BBR development and testing:
|
||||
* https://groups.google.com/forum/#!forum/bbr-dev
|
||||
*
|
||||
* NOTE: BBR *must* be used with the fq qdisc ("man tc-fq") with pacing enabled,
|
||||
* since pacing is integral to the BBR design and implementation.
|
||||
* BBR without pacing would not function properly, and may incur unnecessary
|
||||
* high packet loss rates.
|
||||
*/
|
||||
#include <linux/module.h>
|
||||
#include <net/tcp.h>
|
||||
#include <linux/inet_diag.h>
|
||||
#include <linux/inet.h>
|
||||
#include <linux/random.h>
|
||||
#include <linux/win_minmax.h>
|
||||
|
||||
/* Scale factor for rate in pkt/uSec unit to avoid truncation in bandwidth
|
||||
* estimation. The rate unit ~= (1500 bytes / 1 usec / 2^24) ~= 715 bps.
|
||||
* This handles bandwidths from 0.06pps (715bps) to 256Mpps (3Tbps) in a u32.
|
||||
* Since the minimum window is >=4 packets, the lower bound isn't
|
||||
* an issue. The upper bound isn't an issue with existing technologies.
|
||||
*/
|
||||
#define BW_SCALE 24
|
||||
#define BW_UNIT (1 << BW_SCALE)
|
||||
|
||||
#define BBR_SCALE 8 /* scaling factor for fractions in BBR (e.g. gains) */
|
||||
#define BBR_UNIT (1 << BBR_SCALE)
|
||||
|
||||
/* BBR has the following modes for deciding how fast to send: */
|
||||
enum bbr_mode {
|
||||
BBR_STARTUP, /* ramp up sending rate rapidly to fill pipe */
|
||||
BBR_DRAIN, /* drain any queue created during startup */
|
||||
BBR_PROBE_BW, /* discover, share bw: pace around estimated bw */
|
||||
BBR_PROBE_RTT, /* cut cwnd to min to probe min_rtt */
|
||||
};
|
||||
|
||||
/* BBR congestion control block */
|
||||
struct bbr {
|
||||
u32 min_rtt_us; /* min RTT in min_rtt_win_sec window */
|
||||
u32 min_rtt_stamp; /* timestamp of min_rtt_us */
|
||||
u32 probe_rtt_done_stamp; /* end time for BBR_PROBE_RTT mode */
|
||||
struct minmax bw; /* Max recent delivery rate in pkts/uS << 24 */
|
||||
u32 rtt_cnt; /* count of packet-timed rounds elapsed */
|
||||
u32 next_rtt_delivered; /* scb->tx.delivered at end of round */
|
||||
struct skb_mstamp cycle_mstamp; /* time of this cycle phase start */
|
||||
u32 mode:3, /* current bbr_mode in state machine */
|
||||
prev_ca_state:3, /* CA state on previous ACK */
|
||||
packet_conservation:1, /* use packet conservation? */
|
||||
restore_cwnd:1, /* decided to revert cwnd to old value */
|
||||
round_start:1, /* start of packet-timed tx->ack round? */
|
||||
tso_segs_goal:7, /* segments we want in each skb we send */
|
||||
idle_restart:1, /* restarting after idle? */
|
||||
probe_rtt_round_done:1, /* a BBR_PROBE_RTT round at 4 pkts? */
|
||||
unused:5,
|
||||
lt_is_sampling:1, /* taking long-term ("LT") samples now? */
|
||||
lt_rtt_cnt:7, /* round trips in long-term interval */
|
||||
lt_use_bw:1; /* use lt_bw as our bw estimate? */
|
||||
u32 lt_bw; /* LT est delivery rate in pkts/uS << 24 */
|
||||
u32 lt_last_delivered; /* LT intvl start: tp->delivered */
|
||||
u32 lt_last_stamp; /* LT intvl start: tp->delivered_mstamp */
|
||||
u32 lt_last_lost; /* LT intvl start: tp->lost */
|
||||
u32 pacing_gain:10, /* current gain for setting pacing rate */
|
||||
cwnd_gain:10, /* current gain for setting cwnd */
|
||||
full_bw_cnt:3, /* number of rounds without large bw gains */
|
||||
cycle_idx:3, /* current index in pacing_gain cycle array */
|
||||
unused_b:6;
|
||||
u32 prior_cwnd; /* prior cwnd upon entering loss recovery */
|
||||
u32 full_bw; /* recent bw, to estimate if pipe is full */
|
||||
};
|
||||
|
||||
#define CYCLE_LEN 8 /* number of phases in a pacing gain cycle */
|
||||
|
||||
/* Window length of bw filter (in rounds): */
|
||||
static const int bbr_bw_rtts = CYCLE_LEN + 2;
|
||||
/* Window length of min_rtt filter (in sec): */
|
||||
static const u32 bbr_min_rtt_win_sec = 10;
|
||||
/* Minimum time (in ms) spent at bbr_cwnd_min_target in BBR_PROBE_RTT mode: */
|
||||
static const u32 bbr_probe_rtt_mode_ms = 200;
|
||||
/* Skip TSO below the following bandwidth (bits/sec): */
|
||||
static const int bbr_min_tso_rate = 1200000;
|
||||
|
||||
/* We use a high_gain value of 2/ln(2) because it's the smallest pacing gain
|
||||
* that will allow a smoothly increasing pacing rate that will double each RTT
|
||||
* and send the same number of packets per RTT that an un-paced, slow-starting
|
||||
* Reno or CUBIC flow would:
|
||||
*/
|
||||
static const int bbr_high_gain = BBR_UNIT * 2885 / 1000 + 1;
|
||||
/* The pacing gain of 1/high_gain in BBR_DRAIN is calculated to typically drain
|
||||
* the queue created in BBR_STARTUP in a single round:
|
||||
*/
|
||||
static const int bbr_drain_gain = BBR_UNIT * 1000 / 2885;
|
||||
/* The gain for deriving steady-state cwnd tolerates delayed/stretched ACKs: */
|
||||
static const int bbr_cwnd_gain = BBR_UNIT * 2;
|
||||
/* The pacing_gain values for the PROBE_BW gain cycle, to discover/share bw: */
|
||||
static const int bbr_pacing_gain[] = {
|
||||
BBR_UNIT * 5 / 4, /* probe for more available bw */
|
||||
BBR_UNIT * 3 / 4, /* drain queue and/or yield bw to other flows */
|
||||
BBR_UNIT, BBR_UNIT, BBR_UNIT, /* cruise at 1.0*bw to utilize pipe, */
|
||||
BBR_UNIT, BBR_UNIT, BBR_UNIT /* without creating excess queue... */
|
||||
};
|
||||
/* Randomize the starting gain cycling phase over N phases: */
|
||||
static const u32 bbr_cycle_rand = 7;
|
||||
|
||||
/* Try to keep at least this many packets in flight, if things go smoothly. For
|
||||
* smooth functioning, a sliding window protocol ACKing every other packet
|
||||
* needs at least 4 packets in flight:
|
||||
*/
|
||||
static const u32 bbr_cwnd_min_target = 4;
|
||||
|
||||
/* To estimate if BBR_STARTUP mode (i.e. high_gain) has filled pipe... */
|
||||
/* If bw has increased significantly (1.25x), there may be more bw available: */
|
||||
static const u32 bbr_full_bw_thresh = BBR_UNIT * 5 / 4;
|
||||
/* But after 3 rounds w/o significant bw growth, estimate pipe is full: */
|
||||
static const u32 bbr_full_bw_cnt = 3;
|
||||
|
||||
/* "long-term" ("LT") bandwidth estimator parameters... */
|
||||
/* The minimum number of rounds in an LT bw sampling interval: */
|
||||
static const u32 bbr_lt_intvl_min_rtts = 4;
|
||||
/* If lost/delivered ratio > 20%, interval is "lossy" and we may be policed: */
|
||||
static const u32 bbr_lt_loss_thresh = 50;
|
||||
/* If 2 intervals have a bw ratio <= 1/8, their bw is "consistent": */
|
||||
static const u32 bbr_lt_bw_ratio = BBR_UNIT / 8;
|
||||
/* If 2 intervals have a bw diff <= 4 Kbit/sec their bw is "consistent": */
|
||||
static const u32 bbr_lt_bw_diff = 4000 / 8;
|
||||
/* If we estimate we're policed, use lt_bw for this many round trips: */
|
||||
static const u32 bbr_lt_bw_max_rtts = 48;
|
||||
|
||||
/* Do we estimate that STARTUP filled the pipe? */
|
||||
static bool bbr_full_bw_reached(const struct sock *sk)
|
||||
{
|
||||
const struct bbr *bbr = inet_csk_ca(sk);
|
||||
|
||||
return bbr->full_bw_cnt >= bbr_full_bw_cnt;
|
||||
}
|
||||
|
||||
/* Return the windowed max recent bandwidth sample, in pkts/uS << BW_SCALE. */
|
||||
static u32 bbr_max_bw(const struct sock *sk)
|
||||
{
|
||||
struct bbr *bbr = inet_csk_ca(sk);
|
||||
|
||||
return minmax_get(&bbr->bw);
|
||||
}
|
||||
|
||||
/* Return the estimated bandwidth of the path, in pkts/uS << BW_SCALE. */
|
||||
static u32 bbr_bw(const struct sock *sk)
|
||||
{
|
||||
struct bbr *bbr = inet_csk_ca(sk);
|
||||
|
||||
return bbr->lt_use_bw ? bbr->lt_bw : bbr_max_bw(sk);
|
||||
}
|
||||
|
||||
/* Return rate in bytes per second, optionally with a gain.
|
||||
* The order here is chosen carefully to avoid overflow of u64. This should
|
||||
* work for input rates of up to 2.9Tbit/sec and gain of 2.89x.
|
||||
*/
|
||||
static u64 bbr_rate_bytes_per_sec(struct sock *sk, u64 rate, int gain)
|
||||
{
|
||||
rate *= tcp_mss_to_mtu(sk, tcp_sk(sk)->mss_cache);
|
||||
rate *= gain;
|
||||
rate >>= BBR_SCALE;
|
||||
rate *= USEC_PER_SEC;
|
||||
return rate >> BW_SCALE;
|
||||
}
|
||||
|
||||
/* Pace using current bw estimate and a gain factor. In order to help drive the
|
||||
* network toward lower queues while maintaining high utilization and low
|
||||
* latency, the average pacing rate aims to be slightly (~1%) lower than the
|
||||
* estimated bandwidth. This is an important aspect of the design. In this
|
||||
* implementation this slightly lower pacing rate is achieved implicitly by not
|
||||
* including link-layer headers in the packet size used for the pacing rate.
|
||||
*/
|
||||
static void bbr_set_pacing_rate(struct sock *sk, u32 bw, int gain)
|
||||
{
|
||||
struct bbr *bbr = inet_csk_ca(sk);
|
||||
u64 rate = bw;
|
||||
|
||||
rate = bbr_rate_bytes_per_sec(sk, rate, gain);
|
||||
rate = min_t(u64, rate, sk->sk_max_pacing_rate);
|
||||
if (bbr->mode != BBR_STARTUP || rate > sk->sk_pacing_rate)
|
||||
sk->sk_pacing_rate = rate;
|
||||
}
|
||||
|
||||
/* Return count of segments we want in the skbs we send, or 0 for default. */
|
||||
static u32 bbr_tso_segs_goal(struct sock *sk)
|
||||
{
|
||||
struct bbr *bbr = inet_csk_ca(sk);
|
||||
|
||||
return bbr->tso_segs_goal;
|
||||
}
|
||||
|
||||
static void bbr_set_tso_segs_goal(struct sock *sk)
|
||||
{
|
||||
struct tcp_sock *tp = tcp_sk(sk);
|
||||
struct bbr *bbr = inet_csk_ca(sk);
|
||||
u32 min_segs;
|
||||
|
||||
min_segs = sk->sk_pacing_rate < (bbr_min_tso_rate >> 3) ? 1 : 2;
|
||||
bbr->tso_segs_goal = min(tcp_tso_autosize(sk, tp->mss_cache, min_segs),
|
||||
0x7FU);
|
||||
}
|
||||
|
||||
/* Save "last known good" cwnd so we can restore it after losses or PROBE_RTT */
|
||||
static void bbr_save_cwnd(struct sock *sk)
|
||||
{
|
||||
struct tcp_sock *tp = tcp_sk(sk);
|
||||
struct bbr *bbr = inet_csk_ca(sk);
|
||||
|
||||
if (bbr->prev_ca_state < TCP_CA_Recovery && bbr->mode != BBR_PROBE_RTT)
|
||||
bbr->prior_cwnd = tp->snd_cwnd; /* this cwnd is good enough */
|
||||
else /* loss recovery or BBR_PROBE_RTT have temporarily cut cwnd */
|
||||
bbr->prior_cwnd = max(bbr->prior_cwnd, tp->snd_cwnd);
|
||||
}
|
||||
|
||||
static void bbr_cwnd_event(struct sock *sk, enum tcp_ca_event event)
|
||||
{
|
||||
struct tcp_sock *tp = tcp_sk(sk);
|
||||
struct bbr *bbr = inet_csk_ca(sk);
|
||||
|
||||
if (event == CA_EVENT_TX_START && tp->app_limited) {
|
||||
bbr->idle_restart = 1;
|
||||
/* Avoid pointless buffer overflows: pace at est. bw if we don't
|
||||
* need more speed (we're restarting from idle and app-limited).
|
||||
*/
|
||||
if (bbr->mode == BBR_PROBE_BW)
|
||||
bbr_set_pacing_rate(sk, bbr_bw(sk), BBR_UNIT);
|
||||
}
|
||||
}
|
||||
|
||||
/* Find target cwnd. Right-size the cwnd based on min RTT and the
|
||||
* estimated bottleneck bandwidth:
|
||||
*
|
||||
* cwnd = bw * min_rtt * gain = BDP * gain
|
||||
*
|
||||
* The key factor, gain, controls the amount of queue. While a small gain
|
||||
* builds a smaller queue, it becomes more vulnerable to noise in RTT
|
||||
* measurements (e.g., delayed ACKs or other ACK compression effects). This
|
||||
* noise may cause BBR to under-estimate the rate.
|
||||
*
|
||||
* To achieve full performance in high-speed paths, we budget enough cwnd to
|
||||
* fit full-sized skbs in-flight on both end hosts to fully utilize the path:
|
||||
* - one skb in sending host Qdisc,
|
||||
* - one skb in sending host TSO/GSO engine
|
||||
* - one skb being received by receiver host LRO/GRO/delayed-ACK engine
|
||||
* Don't worry, at low rates (bbr_min_tso_rate) this won't bloat cwnd because
|
||||
* in such cases tso_segs_goal is 1. The minimum cwnd is 4 packets,
|
||||
* which allows 2 outstanding 2-packet sequences, to try to keep pipe
|
||||
* full even with ACK-every-other-packet delayed ACKs.
|
||||
*/
|
||||
static u32 bbr_target_cwnd(struct sock *sk, u32 bw, int gain)
|
||||
{
|
||||
struct bbr *bbr = inet_csk_ca(sk);
|
||||
u32 cwnd;
|
||||
u64 w;
|
||||
|
||||
/* If we've never had a valid RTT sample, cap cwnd at the initial
|
||||
* default. This should only happen when the connection is not using TCP
|
||||
* timestamps and has retransmitted all of the SYN/SYNACK/data packets
|
||||
* ACKed so far. In this case, an RTO can cut cwnd to 1, in which
|
||||
* case we need to slow-start up toward something safe: TCP_INIT_CWND.
|
||||
*/
|
||||
if (unlikely(bbr->min_rtt_us == ~0U)) /* no valid RTT samples yet? */
|
||||
return TCP_INIT_CWND; /* be safe: cap at default initial cwnd*/
|
||||
|
||||
w = (u64)bw * bbr->min_rtt_us;
|
||||
|
||||
/* Apply a gain to the given value, then remove the BW_SCALE shift. */
|
||||
cwnd = (((w * gain) >> BBR_SCALE) + BW_UNIT - 1) / BW_UNIT;
|
||||
|
||||
/* Allow enough full-sized skbs in flight to utilize end systems. */
|
||||
cwnd += 3 * bbr->tso_segs_goal;
|
||||
|
||||
/* Reduce delayed ACKs by rounding up cwnd to the next even number. */
|
||||
cwnd = (cwnd + 1) & ~1U;
|
||||
|
||||
return cwnd;
|
||||
}
|
||||
|
||||
/* An optimization in BBR to reduce losses: On the first round of recovery, we
|
||||
* follow the packet conservation principle: send P packets per P packets acked.
|
||||
* After that, we slow-start and send at most 2*P packets per P packets acked.
|
||||
* After recovery finishes, or upon undo, we restore the cwnd we had when
|
||||
* recovery started (capped by the target cwnd based on estimated BDP).
|
||||
*
|
||||
* TODO(ycheng/ncardwell): implement a rate-based approach.
|
||||
*/
|
||||
static bool bbr_set_cwnd_to_recover_or_restore(
|
||||
struct sock *sk, const struct rate_sample *rs, u32 acked, u32 *new_cwnd)
|
||||
{
|
||||
struct tcp_sock *tp = tcp_sk(sk);
|
||||
struct bbr *bbr = inet_csk_ca(sk);
|
||||
u8 prev_state = bbr->prev_ca_state, state = inet_csk(sk)->icsk_ca_state;
|
||||
u32 cwnd = tp->snd_cwnd;
|
||||
|
||||
/* An ACK for P pkts should release at most 2*P packets. We do this
|
||||
* in two steps. First, here we deduct the number of lost packets.
|
||||
* Then, in bbr_set_cwnd() we slow start up toward the target cwnd.
|
||||
*/
|
||||
if (rs->losses > 0)
|
||||
cwnd = max_t(s32, cwnd - rs->losses, 1);
|
||||
|
||||
if (state == TCP_CA_Recovery && prev_state != TCP_CA_Recovery) {
|
||||
/* Starting 1st round of Recovery, so do packet conservation. */
|
||||
bbr->packet_conservation = 1;
|
||||
bbr->next_rtt_delivered = tp->delivered; /* start round now */
|
||||
/* Cut unused cwnd from app behavior, TSQ, or TSO deferral: */
|
||||
cwnd = tcp_packets_in_flight(tp) + acked;
|
||||
} else if (prev_state >= TCP_CA_Recovery && state < TCP_CA_Recovery) {
|
||||
/* Exiting loss recovery; restore cwnd saved before recovery. */
|
||||
bbr->restore_cwnd = 1;
|
||||
bbr->packet_conservation = 0;
|
||||
}
|
||||
bbr->prev_ca_state = state;
|
||||
|
||||
if (bbr->restore_cwnd) {
|
||||
/* Restore cwnd after exiting loss recovery or PROBE_RTT. */
|
||||
cwnd = max(cwnd, bbr->prior_cwnd);
|
||||
bbr->restore_cwnd = 0;
|
||||
}
|
||||
|
||||
if (bbr->packet_conservation) {
|
||||
*new_cwnd = max(cwnd, tcp_packets_in_flight(tp) + acked);
|
||||
return true; /* yes, using packet conservation */
|
||||
}
|
||||
*new_cwnd = cwnd;
|
||||
return false;
|
||||
}
|
||||
|
||||
/* Slow-start up toward target cwnd (if bw estimate is growing, or packet loss
|
||||
* has drawn us down below target), or snap down to target if we're above it.
|
||||
*/
|
||||
static void bbr_set_cwnd(struct sock *sk, const struct rate_sample *rs,
|
||||
u32 acked, u32 bw, int gain)
|
||||
{
|
||||
struct tcp_sock *tp = tcp_sk(sk);
|
||||
struct bbr *bbr = inet_csk_ca(sk);
|
||||
u32 cwnd = 0, target_cwnd = 0;
|
||||
|
||||
if (!acked)
|
||||
return;
|
||||
|
||||
if (bbr_set_cwnd_to_recover_or_restore(sk, rs, acked, &cwnd))
|
||||
goto done;
|
||||
|
||||
/* If we're below target cwnd, slow start cwnd toward target cwnd. */
|
||||
target_cwnd = bbr_target_cwnd(sk, bw, gain);
|
||||
if (bbr_full_bw_reached(sk)) /* only cut cwnd if we filled the pipe */
|
||||
cwnd = min(cwnd + acked, target_cwnd);
|
||||
else if (cwnd < target_cwnd || tp->delivered < TCP_INIT_CWND)
|
||||
cwnd = cwnd + acked;
|
||||
cwnd = max(cwnd, bbr_cwnd_min_target);
|
||||
|
||||
done:
|
||||
tp->snd_cwnd = min(cwnd, tp->snd_cwnd_clamp); /* apply global cap */
|
||||
if (bbr->mode == BBR_PROBE_RTT) /* drain queue, refresh min_rtt */
|
||||
tp->snd_cwnd = min(tp->snd_cwnd, bbr_cwnd_min_target);
|
||||
}
|
||||
|
||||
/* End cycle phase if it's time and/or we hit the phase's in-flight target. */
|
||||
static bool bbr_is_next_cycle_phase(struct sock *sk,
|
||||
const struct rate_sample *rs)
|
||||
{
|
||||
struct tcp_sock *tp = tcp_sk(sk);
|
||||
struct bbr *bbr = inet_csk_ca(sk);
|
||||
bool is_full_length =
|
||||
skb_mstamp_us_delta(&tp->delivered_mstamp, &bbr->cycle_mstamp) >
|
||||
bbr->min_rtt_us;
|
||||
u32 inflight, bw;
|
||||
|
||||
/* The pacing_gain of 1.0 paces at the estimated bw to try to fully
|
||||
* use the pipe without increasing the queue.
|
||||
*/
|
||||
if (bbr->pacing_gain == BBR_UNIT)
|
||||
return is_full_length; /* just use wall clock time */
|
||||
|
||||
inflight = rs->prior_in_flight; /* what was in-flight before ACK? */
|
||||
bw = bbr_max_bw(sk);
|
||||
|
||||
/* A pacing_gain > 1.0 probes for bw by trying to raise inflight to at
|
||||
* least pacing_gain*BDP; this may take more than min_rtt if min_rtt is
|
||||
* small (e.g. on a LAN). We do not persist if packets are lost, since
|
||||
* a path with small buffers may not hold that much.
|
||||
*/
|
||||
if (bbr->pacing_gain > BBR_UNIT)
|
||||
return is_full_length &&
|
||||
(rs->losses || /* perhaps pacing_gain*BDP won't fit */
|
||||
inflight >= bbr_target_cwnd(sk, bw, bbr->pacing_gain));
|
||||
|
||||
/* A pacing_gain < 1.0 tries to drain extra queue we added if bw
|
||||
* probing didn't find more bw. If inflight falls to match BDP then we
|
||||
* estimate queue is drained; persisting would underutilize the pipe.
|
||||
*/
|
||||
return is_full_length ||
|
||||
inflight <= bbr_target_cwnd(sk, bw, BBR_UNIT);
|
||||
}
|
||||
|
||||
static void bbr_advance_cycle_phase(struct sock *sk)
|
||||
{
|
||||
struct tcp_sock *tp = tcp_sk(sk);
|
||||
struct bbr *bbr = inet_csk_ca(sk);
|
||||
|
||||
bbr->cycle_idx = (bbr->cycle_idx + 1) & (CYCLE_LEN - 1);
|
||||
bbr->cycle_mstamp = tp->delivered_mstamp;
|
||||
bbr->pacing_gain = bbr_pacing_gain[bbr->cycle_idx];
|
||||
}
|
||||
|
||||
/* Gain cycling: cycle pacing gain to converge to fair share of available bw. */
|
||||
static void bbr_update_cycle_phase(struct sock *sk,
|
||||
const struct rate_sample *rs)
|
||||
{
|
||||
struct bbr *bbr = inet_csk_ca(sk);
|
||||
|
||||
if ((bbr->mode == BBR_PROBE_BW) && !bbr->lt_use_bw &&
|
||||
bbr_is_next_cycle_phase(sk, rs))
|
||||
bbr_advance_cycle_phase(sk);
|
||||
}
|
||||
|
||||
static void bbr_reset_startup_mode(struct sock *sk)
|
||||
{
|
||||
struct bbr *bbr = inet_csk_ca(sk);
|
||||
|
||||
bbr->mode = BBR_STARTUP;
|
||||
bbr->pacing_gain = bbr_high_gain;
|
||||
bbr->cwnd_gain = bbr_high_gain;
|
||||
}
|
||||
|
||||
static void bbr_reset_probe_bw_mode(struct sock *sk)
|
||||
{
|
||||
struct bbr *bbr = inet_csk_ca(sk);
|
||||
|
||||
bbr->mode = BBR_PROBE_BW;
|
||||
bbr->pacing_gain = BBR_UNIT;
|
||||
bbr->cwnd_gain = bbr_cwnd_gain;
|
||||
bbr->cycle_idx = CYCLE_LEN - 1 - prandom_u32_max(bbr_cycle_rand);
|
||||
bbr_advance_cycle_phase(sk); /* flip to next phase of gain cycle */
|
||||
}
|
||||
|
||||
static void bbr_reset_mode(struct sock *sk)
|
||||
{
|
||||
if (!bbr_full_bw_reached(sk))
|
||||
bbr_reset_startup_mode(sk);
|
||||
else
|
||||
bbr_reset_probe_bw_mode(sk);
|
||||
}
|
||||
|
||||
/* Start a new long-term sampling interval. */
|
||||
static void bbr_reset_lt_bw_sampling_interval(struct sock *sk)
|
||||
{
|
||||
struct tcp_sock *tp = tcp_sk(sk);
|
||||
struct bbr *bbr = inet_csk_ca(sk);
|
||||
|
||||
bbr->lt_last_stamp = tp->delivered_mstamp.stamp_jiffies;
|
||||
bbr->lt_last_delivered = tp->delivered;
|
||||
bbr->lt_last_lost = tp->lost;
|
||||
bbr->lt_rtt_cnt = 0;
|
||||
}
|
||||
|
||||
/* Completely reset long-term bandwidth sampling. */
|
||||
static void bbr_reset_lt_bw_sampling(struct sock *sk)
|
||||
{
|
||||
struct bbr *bbr = inet_csk_ca(sk);
|
||||
|
||||
bbr->lt_bw = 0;
|
||||
bbr->lt_use_bw = 0;
|
||||
bbr->lt_is_sampling = false;
|
||||
bbr_reset_lt_bw_sampling_interval(sk);
|
||||
}
|
||||
|
||||
/* Long-term bw sampling interval is done. Estimate whether we're policed. */
|
||||
static void bbr_lt_bw_interval_done(struct sock *sk, u32 bw)
|
||||
{
|
||||
struct bbr *bbr = inet_csk_ca(sk);
|
||||
u32 diff;
|
||||
|
||||
if (bbr->lt_bw) { /* do we have bw from a previous interval? */
|
||||
/* Is new bw close to the lt_bw from the previous interval? */
|
||||
diff = abs(bw - bbr->lt_bw);
|
||||
if ((diff * BBR_UNIT <= bbr_lt_bw_ratio * bbr->lt_bw) ||
|
||||
(bbr_rate_bytes_per_sec(sk, diff, BBR_UNIT) <=
|
||||
bbr_lt_bw_diff)) {
|
||||
/* All criteria are met; estimate we're policed. */
|
||||
bbr->lt_bw = (bw + bbr->lt_bw) >> 1; /* avg 2 intvls */
|
||||
bbr->lt_use_bw = 1;
|
||||
bbr->pacing_gain = BBR_UNIT; /* try to avoid drops */
|
||||
bbr->lt_rtt_cnt = 0;
|
||||
return;
|
||||
}
|
||||
}
|
||||
bbr->lt_bw = bw;
|
||||
bbr_reset_lt_bw_sampling_interval(sk);
|
||||
}
|
||||
|
||||
/* Token-bucket traffic policers are common (see "An Internet-Wide Analysis of
|
||||
* Traffic Policing", SIGCOMM 2016). BBR detects token-bucket policers and
|
||||
* explicitly models their policed rate, to reduce unnecessary losses. We
|
||||
* estimate that we're policed if we see 2 consecutive sampling intervals with
|
||||
* consistent throughput and high packet loss. If we think we're being policed,
|
||||
* set lt_bw to the "long-term" average delivery rate from those 2 intervals.
|
||||
*/
|
||||
static void bbr_lt_bw_sampling(struct sock *sk, const struct rate_sample *rs)
|
||||
{
|
||||
struct tcp_sock *tp = tcp_sk(sk);
|
||||
struct bbr *bbr = inet_csk_ca(sk);
|
||||
u32 lost, delivered;
|
||||
u64 bw;
|
||||
s32 t;
|
||||
|
||||
if (bbr->lt_use_bw) { /* already using long-term rate, lt_bw? */
|
||||
if (bbr->mode == BBR_PROBE_BW && bbr->round_start &&
|
||||
++bbr->lt_rtt_cnt >= bbr_lt_bw_max_rtts) {
|
||||
bbr_reset_lt_bw_sampling(sk); /* stop using lt_bw */
|
||||
bbr_reset_probe_bw_mode(sk); /* restart gain cycling */
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
/* Wait for the first loss before sampling, to let the policer exhaust
|
||||
* its tokens and estimate the steady-state rate allowed by the policer.
|
||||
* Starting samples earlier includes bursts that over-estimate the bw.
|
||||
*/
|
||||
if (!bbr->lt_is_sampling) {
|
||||
if (!rs->losses)
|
||||
return;
|
||||
bbr_reset_lt_bw_sampling_interval(sk);
|
||||
bbr->lt_is_sampling = true;
|
||||
}
|
||||
|
||||
/* To avoid underestimates, reset sampling if we run out of data. */
|
||||
if (rs->is_app_limited) {
|
||||
bbr_reset_lt_bw_sampling(sk);
|
||||
return;
|
||||
}
|
||||
|
||||
if (bbr->round_start)
|
||||
bbr->lt_rtt_cnt++; /* count round trips in this interval */
|
||||
if (bbr->lt_rtt_cnt < bbr_lt_intvl_min_rtts)
|
||||
return; /* sampling interval needs to be longer */
|
||||
if (bbr->lt_rtt_cnt > 4 * bbr_lt_intvl_min_rtts) {
|
||||
bbr_reset_lt_bw_sampling(sk); /* interval is too long */
|
||||
return;
|
||||
}
|
||||
|
||||
/* End sampling interval when a packet is lost, so we estimate the
|
||||
* policer tokens were exhausted. Stopping the sampling before the
|
||||
* tokens are exhausted under-estimates the policed rate.
|
||||
*/
|
||||
if (!rs->losses)
|
||||
return;
|
||||
|
||||
/* Calculate packets lost and delivered in sampling interval. */
|
||||
lost = tp->lost - bbr->lt_last_lost;
|
||||
delivered = tp->delivered - bbr->lt_last_delivered;
|
||||
/* Is loss rate (lost/delivered) >= lt_loss_thresh? If not, wait. */
|
||||
if (!delivered || (lost << BBR_SCALE) < bbr_lt_loss_thresh * delivered)
|
||||
return;
|
||||
|
||||
/* Find average delivery rate in this sampling interval. */
|
||||
t = (s32)(tp->delivered_mstamp.stamp_jiffies - bbr->lt_last_stamp);
|
||||
if (t < 1)
|
||||
return; /* interval is less than one jiffy, so wait */
|
||||
t = jiffies_to_usecs(t);
|
||||
/* Interval long enough for jiffies_to_usecs() to return a bogus 0? */
|
||||
if (t < 1) {
|
||||
bbr_reset_lt_bw_sampling(sk); /* interval too long; reset */
|
||||
return;
|
||||
}
|
||||
bw = (u64)delivered * BW_UNIT;
|
||||
do_div(bw, t);
|
||||
bbr_lt_bw_interval_done(sk, bw);
|
||||
}
|
||||
|
||||
/* Estimate the bandwidth based on how fast packets are delivered */
|
||||
static void bbr_update_bw(struct sock *sk, const struct rate_sample *rs)
|
||||
{
|
||||
struct tcp_sock *tp = tcp_sk(sk);
|
||||
struct bbr *bbr = inet_csk_ca(sk);
|
||||
u64 bw;
|
||||
|
||||
bbr->round_start = 0;
|
||||
if (rs->delivered < 0 || rs->interval_us <= 0)
|
||||
return; /* Not a valid observation */
|
||||
|
||||
/* See if we've reached the next RTT */
|
||||
if (!before(rs->prior_delivered, bbr->next_rtt_delivered)) {
|
||||
bbr->next_rtt_delivered = tp->delivered;
|
||||
bbr->rtt_cnt++;
|
||||
bbr->round_start = 1;
|
||||
bbr->packet_conservation = 0;
|
||||
}
|
||||
|
||||
bbr_lt_bw_sampling(sk, rs);
|
||||
|
||||
/* Divide delivered by the interval to find a (lower bound) bottleneck
|
||||
* bandwidth sample. Delivered is in packets and interval_us in uS and
|
||||
* ratio will be <<1 for most connections. So delivered is first scaled.
|
||||
*/
|
||||
bw = (u64)rs->delivered * BW_UNIT;
|
||||
do_div(bw, rs->interval_us);
|
||||
|
||||
/* If this sample is application-limited, it is likely to have a very
|
||||
* low delivered count that represents application behavior rather than
|
||||
* the available network rate. Such a sample could drag down estimated
|
||||
* bw, causing needless slow-down. Thus, to continue to send at the
|
||||
* last measured network rate, we filter out app-limited samples unless
|
||||
* they describe the path bw at least as well as our bw model.
|
||||
*
|
||||
* So the goal during app-limited phase is to proceed with the best
|
||||
* network rate no matter how long. We automatically leave this
|
||||
* phase when app writes faster than the network can deliver :)
|
||||
*/
|
||||
if (!rs->is_app_limited || bw >= bbr_max_bw(sk)) {
|
||||
/* Incorporate new sample into our max bw filter. */
|
||||
minmax_running_max(&bbr->bw, bbr_bw_rtts, bbr->rtt_cnt, bw);
|
||||
}
|
||||
}
|
||||
|
||||
/* Estimate when the pipe is full, using the change in delivery rate: BBR
|
||||
* estimates that STARTUP filled the pipe if the estimated bw hasn't changed by
|
||||
* at least bbr_full_bw_thresh (25%) after bbr_full_bw_cnt (3) non-app-limited
|
||||
* rounds. Why 3 rounds: 1: rwin autotuning grows the rwin, 2: we fill the
|
||||
* higher rwin, 3: we get higher delivery rate samples. Or transient
|
||||
* cross-traffic or radio noise can go away. CUBIC Hystart shares a similar
|
||||
* design goal, but uses delay and inter-ACK spacing instead of bandwidth.
|
||||
*/
|
||||
static void bbr_check_full_bw_reached(struct sock *sk,
|
||||
const struct rate_sample *rs)
|
||||
{
|
||||
struct bbr *bbr = inet_csk_ca(sk);
|
||||
u32 bw_thresh;
|
||||
|
||||
if (bbr_full_bw_reached(sk) || !bbr->round_start || rs->is_app_limited)
|
||||
return;
|
||||
|
||||
bw_thresh = (u64)bbr->full_bw * bbr_full_bw_thresh >> BBR_SCALE;
|
||||
if (bbr_max_bw(sk) >= bw_thresh) {
|
||||
bbr->full_bw = bbr_max_bw(sk);
|
||||
bbr->full_bw_cnt = 0;
|
||||
return;
|
||||
}
|
||||
++bbr->full_bw_cnt;
|
||||
}
|
||||
|
||||
/* If pipe is probably full, drain the queue and then enter steady-state. */
|
||||
static void bbr_check_drain(struct sock *sk, const struct rate_sample *rs)
|
||||
{
|
||||
struct bbr *bbr = inet_csk_ca(sk);
|
||||
|
||||
if (bbr->mode == BBR_STARTUP && bbr_full_bw_reached(sk)) {
|
||||
bbr->mode = BBR_DRAIN; /* drain queue we created */
|
||||
bbr->pacing_gain = bbr_drain_gain; /* pace slow to drain */
|
||||
bbr->cwnd_gain = bbr_high_gain; /* maintain cwnd */
|
||||
} /* fall through to check if in-flight is already small: */
|
||||
if (bbr->mode == BBR_DRAIN &&
|
||||
tcp_packets_in_flight(tcp_sk(sk)) <=
|
||||
bbr_target_cwnd(sk, bbr_max_bw(sk), BBR_UNIT))
|
||||
bbr_reset_probe_bw_mode(sk); /* we estimate queue is drained */
|
||||
}
|
||||
|
||||
/* The goal of PROBE_RTT mode is to have BBR flows cooperatively and
|
||||
* periodically drain the bottleneck queue, to converge to measure the true
|
||||
* min_rtt (unloaded propagation delay). This allows the flows to keep queues
|
||||
* small (reducing queuing delay and packet loss) and achieve fairness among
|
||||
* BBR flows.
|
||||
*
|
||||
* The min_rtt filter window is 10 seconds. When the min_rtt estimate expires,
|
||||
* we enter PROBE_RTT mode and cap the cwnd at bbr_cwnd_min_target=4 packets.
|
||||
* After at least bbr_probe_rtt_mode_ms=200ms and at least one packet-timed
|
||||
* round trip elapsed with that flight size <= 4, we leave PROBE_RTT mode and
|
||||
* re-enter the previous mode. BBR uses 200ms to approximately bound the
|
||||
* performance penalty of PROBE_RTT's cwnd capping to roughly 2% (200ms/10s).
|
||||
*
|
||||
* Note that flows need only pay 2% if they are busy sending over the last 10
|
||||
* seconds. Interactive applications (e.g., Web, RPCs, video chunks) often have
|
||||
* natural silences or low-rate periods within 10 seconds where the rate is low
|
||||
* enough for long enough to drain its queue in the bottleneck. We pick up
|
||||
* these min RTT measurements opportunistically with our min_rtt filter. :-)
|
||||
*/
|
||||
static void bbr_update_min_rtt(struct sock *sk, const struct rate_sample *rs)
|
||||
{
|
||||
struct tcp_sock *tp = tcp_sk(sk);
|
||||
struct bbr *bbr = inet_csk_ca(sk);
|
||||
bool filter_expired;
|
||||
|
||||
/* Track min RTT seen in the min_rtt_win_sec filter window: */
|
||||
filter_expired = after(tcp_time_stamp,
|
||||
bbr->min_rtt_stamp + bbr_min_rtt_win_sec * HZ);
|
||||
if (rs->rtt_us >= 0 &&
|
||||
(rs->rtt_us <= bbr->min_rtt_us || filter_expired)) {
|
||||
bbr->min_rtt_us = rs->rtt_us;
|
||||
bbr->min_rtt_stamp = tcp_time_stamp;
|
||||
}
|
||||
|
||||
if (bbr_probe_rtt_mode_ms > 0 && filter_expired &&
|
||||
!bbr->idle_restart && bbr->mode != BBR_PROBE_RTT) {
|
||||
bbr->mode = BBR_PROBE_RTT; /* dip, drain queue */
|
||||
bbr->pacing_gain = BBR_UNIT;
|
||||
bbr->cwnd_gain = BBR_UNIT;
|
||||
bbr_save_cwnd(sk); /* note cwnd so we can restore it */
|
||||
bbr->probe_rtt_done_stamp = 0;
|
||||
}
|
||||
|
||||
if (bbr->mode == BBR_PROBE_RTT) {
|
||||
/* Ignore low rate samples during this mode. */
|
||||
tp->app_limited =
|
||||
(tp->delivered + tcp_packets_in_flight(tp)) ? : 1;
|
||||
/* Maintain min packets in flight for max(200 ms, 1 round). */
|
||||
if (!bbr->probe_rtt_done_stamp &&
|
||||
tcp_packets_in_flight(tp) <= bbr_cwnd_min_target) {
|
||||
bbr->probe_rtt_done_stamp = tcp_time_stamp +
|
||||
msecs_to_jiffies(bbr_probe_rtt_mode_ms);
|
||||
bbr->probe_rtt_round_done = 0;
|
||||
bbr->next_rtt_delivered = tp->delivered;
|
||||
} else if (bbr->probe_rtt_done_stamp) {
|
||||
if (bbr->round_start)
|
||||
bbr->probe_rtt_round_done = 1;
|
||||
if (bbr->probe_rtt_round_done &&
|
||||
after(tcp_time_stamp, bbr->probe_rtt_done_stamp)) {
|
||||
bbr->min_rtt_stamp = tcp_time_stamp;
|
||||
bbr->restore_cwnd = 1; /* snap to prior_cwnd */
|
||||
bbr_reset_mode(sk);
|
||||
}
|
||||
}
|
||||
}
|
||||
bbr->idle_restart = 0;
|
||||
}
|
||||
|
||||
static void bbr_update_model(struct sock *sk, const struct rate_sample *rs)
|
||||
{
|
||||
bbr_update_bw(sk, rs);
|
||||
bbr_update_cycle_phase(sk, rs);
|
||||
bbr_check_full_bw_reached(sk, rs);
|
||||
bbr_check_drain(sk, rs);
|
||||
bbr_update_min_rtt(sk, rs);
|
||||
}
|
||||
|
||||
static void bbr_main(struct sock *sk, const struct rate_sample *rs)
|
||||
{
|
||||
struct bbr *bbr = inet_csk_ca(sk);
|
||||
u32 bw;
|
||||
|
||||
bbr_update_model(sk, rs);
|
||||
|
||||
bw = bbr_bw(sk);
|
||||
bbr_set_pacing_rate(sk, bw, bbr->pacing_gain);
|
||||
bbr_set_tso_segs_goal(sk);
|
||||
bbr_set_cwnd(sk, rs, rs->acked_sacked, bw, bbr->cwnd_gain);
|
||||
}
|
||||
|
||||
static void bbr_init(struct sock *sk)
|
||||
{
|
||||
struct tcp_sock *tp = tcp_sk(sk);
|
||||
struct bbr *bbr = inet_csk_ca(sk);
|
||||
u64 bw;
|
||||
|
||||
bbr->prior_cwnd = 0;
|
||||
bbr->tso_segs_goal = 0; /* default segs per skb until first ACK */
|
||||
bbr->rtt_cnt = 0;
|
||||
bbr->next_rtt_delivered = 0;
|
||||
bbr->prev_ca_state = TCP_CA_Open;
|
||||
bbr->packet_conservation = 0;
|
||||
|
||||
bbr->probe_rtt_done_stamp = 0;
|
||||
bbr->probe_rtt_round_done = 0;
|
||||
bbr->min_rtt_us = tcp_min_rtt(tp);
|
||||
bbr->min_rtt_stamp = tcp_time_stamp;
|
||||
|
||||
minmax_reset(&bbr->bw, bbr->rtt_cnt, 0); /* init max bw to 0 */
|
||||
|
||||
/* Initialize pacing rate to: high_gain * init_cwnd / RTT. */
|
||||
bw = (u64)tp->snd_cwnd * BW_UNIT;
|
||||
do_div(bw, (tp->srtt_us >> 3) ? : USEC_PER_MSEC);
|
||||
sk->sk_pacing_rate = 0; /* force an update of sk_pacing_rate */
|
||||
bbr_set_pacing_rate(sk, bw, bbr_high_gain);
|
||||
|
||||
bbr->restore_cwnd = 0;
|
||||
bbr->round_start = 0;
|
||||
bbr->idle_restart = 0;
|
||||
bbr->full_bw = 0;
|
||||
bbr->full_bw_cnt = 0;
|
||||
bbr->cycle_mstamp.v64 = 0;
|
||||
bbr->cycle_idx = 0;
|
||||
bbr_reset_lt_bw_sampling(sk);
|
||||
bbr_reset_startup_mode(sk);
|
||||
}
|
||||
|
||||
static u32 bbr_sndbuf_expand(struct sock *sk)
|
||||
{
|
||||
/* Provision 3 * cwnd since BBR may slow-start even during recovery. */
|
||||
return 3;
|
||||
}
|
||||
|
||||
/* In theory BBR does not need to undo the cwnd since it does not
|
||||
* always reduce cwnd on losses (see bbr_main()). Keep it for now.
|
||||
*/
|
||||
static u32 bbr_undo_cwnd(struct sock *sk)
|
||||
{
|
||||
return tcp_sk(sk)->snd_cwnd;
|
||||
}
|
||||
|
||||
/* Entering loss recovery, so save cwnd for when we exit or undo recovery. */
|
||||
static u32 bbr_ssthresh(struct sock *sk)
|
||||
{
|
||||
bbr_save_cwnd(sk);
|
||||
return TCP_INFINITE_SSTHRESH; /* BBR does not use ssthresh */
|
||||
}
|
||||
|
||||
static size_t bbr_get_info(struct sock *sk, u32 ext, int *attr,
|
||||
union tcp_cc_info *info)
|
||||
{
|
||||
if (ext & (1 << (INET_DIAG_BBRINFO - 1)) ||
|
||||
ext & (1 << (INET_DIAG_VEGASINFO - 1))) {
|
||||
struct tcp_sock *tp = tcp_sk(sk);
|
||||
struct bbr *bbr = inet_csk_ca(sk);
|
||||
u64 bw = bbr_bw(sk);
|
||||
|
||||
bw = bw * tp->mss_cache * USEC_PER_SEC >> BW_SCALE;
|
||||
memset(&info->bbr, 0, sizeof(info->bbr));
|
||||
info->bbr.bbr_bw_lo = (u32)bw;
|
||||
info->bbr.bbr_bw_hi = (u32)(bw >> 32);
|
||||
info->bbr.bbr_min_rtt = bbr->min_rtt_us;
|
||||
info->bbr.bbr_pacing_gain = bbr->pacing_gain;
|
||||
info->bbr.bbr_cwnd_gain = bbr->cwnd_gain;
|
||||
*attr = INET_DIAG_BBRINFO;
|
||||
return sizeof(info->bbr);
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
static void bbr_set_state(struct sock *sk, u8 new_state)
|
||||
{
|
||||
struct bbr *bbr = inet_csk_ca(sk);
|
||||
|
||||
if (new_state == TCP_CA_Loss) {
|
||||
struct rate_sample rs = { .losses = 1 };
|
||||
|
||||
bbr->prev_ca_state = TCP_CA_Loss;
|
||||
bbr->full_bw = 0;
|
||||
bbr->round_start = 1; /* treat RTO like end of a round */
|
||||
bbr_lt_bw_sampling(sk, &rs);
|
||||
}
|
||||
}
|
||||
|
||||
static struct tcp_congestion_ops tcp_bbr_cong_ops __read_mostly = {
|
||||
.flags = TCP_CONG_NON_RESTRICTED,
|
||||
.name = "bbr",
|
||||
.owner = THIS_MODULE,
|
||||
.init = bbr_init,
|
||||
.cong_control = bbr_main,
|
||||
.sndbuf_expand = bbr_sndbuf_expand,
|
||||
.undo_cwnd = bbr_undo_cwnd,
|
||||
.cwnd_event = bbr_cwnd_event,
|
||||
.ssthresh = bbr_ssthresh,
|
||||
.tso_segs_goal = bbr_tso_segs_goal,
|
||||
.get_info = bbr_get_info,
|
||||
.set_state = bbr_set_state,
|
||||
};
|
||||
|
||||
static int __init bbr_register(void)
|
||||
{
|
||||
BUILD_BUG_ON(sizeof(struct bbr) > ICSK_CA_PRIV_SIZE);
|
||||
return tcp_register_congestion_control(&tcp_bbr_cong_ops);
|
||||
}
|
||||
|
||||
static void __exit bbr_unregister(void)
|
||||
{
|
||||
tcp_unregister_congestion_control(&tcp_bbr_cong_ops);
|
||||
}
|
||||
|
||||
module_init(bbr_register);
|
||||
module_exit(bbr_unregister);
|
||||
|
||||
MODULE_AUTHOR("Van Jacobson <vanj@google.com>");
|
||||
MODULE_AUTHOR("Neal Cardwell <ncardwell@google.com>");
|
||||
MODULE_AUTHOR("Yuchung Cheng <ycheng@google.com>");
|
||||
MODULE_AUTHOR("Soheil Hassas Yeganeh <soheil@google.com>");
|
||||
MODULE_LICENSE("Dual BSD/GPL");
|
||||
MODULE_DESCRIPTION("TCP BBR (Bottleneck Bandwidth and RTT)");
|
||||
+6
-6
@@ -56,7 +56,7 @@ MODULE_PARM_DESC(use_shadow, "use shadow window heuristic");
|
||||
module_param(use_tolerance, bool, 0644);
|
||||
MODULE_PARM_DESC(use_tolerance, "use loss tolerance heuristic");
|
||||
|
||||
struct minmax {
|
||||
struct cdg_minmax {
|
||||
union {
|
||||
struct {
|
||||
s32 min;
|
||||
@@ -74,10 +74,10 @@ enum cdg_state {
|
||||
};
|
||||
|
||||
struct cdg {
|
||||
struct minmax rtt;
|
||||
struct minmax rtt_prev;
|
||||
struct minmax *gradients;
|
||||
struct minmax gsum;
|
||||
struct cdg_minmax rtt;
|
||||
struct cdg_minmax rtt_prev;
|
||||
struct cdg_minmax *gradients;
|
||||
struct cdg_minmax gsum;
|
||||
bool gfilled;
|
||||
u8 tail;
|
||||
u8 state;
|
||||
@@ -353,7 +353,7 @@ static void tcp_cdg_cwnd_event(struct sock *sk, const enum tcp_ca_event ev)
|
||||
{
|
||||
struct cdg *ca = inet_csk_ca(sk);
|
||||
struct tcp_sock *tp = tcp_sk(sk);
|
||||
struct minmax *gradients;
|
||||
struct cdg_minmax *gradients;
|
||||
|
||||
switch (ev) {
|
||||
case CA_EVENT_CWND_RESTART:
|
||||
|
||||
+1
-1
@@ -69,7 +69,7 @@ int tcp_register_congestion_control(struct tcp_congestion_ops *ca)
|
||||
int ret = 0;
|
||||
|
||||
/* all algorithms must implement ssthresh and cong_avoid ops */
|
||||
if (!ca->ssthresh || !ca->cong_avoid) {
|
||||
if (!ca->ssthresh || !(ca->cong_avoid || ca->cong_control)) {
|
||||
pr_err("%s does not implement required ops\n", ca->name);
|
||||
return -EINVAL;
|
||||
}
|
||||
|
||||
+79
-79
@@ -289,6 +289,7 @@ static bool tcp_ecn_rcv_ecn_echo(const struct tcp_sock *tp, const struct tcphdr
|
||||
static void tcp_sndbuf_expand(struct sock *sk)
|
||||
{
|
||||
const struct tcp_sock *tp = tcp_sk(sk);
|
||||
const struct tcp_congestion_ops *ca_ops = inet_csk(sk)->icsk_ca_ops;
|
||||
int sndmem, per_mss;
|
||||
u32 nr_segs;
|
||||
|
||||
@@ -309,7 +310,8 @@ static void tcp_sndbuf_expand(struct sock *sk)
|
||||
* Cubic needs 1.7 factor, rounded to 2 to include
|
||||
* extra cushion (application might react slowly to POLLOUT)
|
||||
*/
|
||||
sndmem = 2 * nr_segs * per_mss;
|
||||
sndmem = ca_ops->sndbuf_expand ? ca_ops->sndbuf_expand(sk) : 2;
|
||||
sndmem *= nr_segs * per_mss;
|
||||
|
||||
if (sk->sk_sndbuf < sndmem)
|
||||
sk->sk_sndbuf = min(sndmem, sysctl_tcp_wmem[2]);
|
||||
@@ -899,12 +901,29 @@ static void tcp_verify_retransmit_hint(struct tcp_sock *tp, struct sk_buff *skb)
|
||||
tp->retransmit_high = TCP_SKB_CB(skb)->end_seq;
|
||||
}
|
||||
|
||||
/* Sum the number of packets on the wire we have marked as lost.
|
||||
* There are two cases we care about here:
|
||||
* a) Packet hasn't been marked lost (nor retransmitted),
|
||||
* and this is the first loss.
|
||||
* b) Packet has been marked both lost and retransmitted,
|
||||
* and this means we think it was lost again.
|
||||
*/
|
||||
static void tcp_sum_lost(struct tcp_sock *tp, struct sk_buff *skb)
|
||||
{
|
||||
__u8 sacked = TCP_SKB_CB(skb)->sacked;
|
||||
|
||||
if (!(sacked & TCPCB_LOST) ||
|
||||
((sacked & TCPCB_LOST) && (sacked & TCPCB_SACKED_RETRANS)))
|
||||
tp->lost += tcp_skb_pcount(skb);
|
||||
}
|
||||
|
||||
static void tcp_skb_mark_lost(struct tcp_sock *tp, struct sk_buff *skb)
|
||||
{
|
||||
if (!(TCP_SKB_CB(skb)->sacked & (TCPCB_LOST|TCPCB_SACKED_ACKED))) {
|
||||
tcp_verify_retransmit_hint(tp, skb);
|
||||
|
||||
tp->lost_out += tcp_skb_pcount(skb);
|
||||
tcp_sum_lost(tp, skb);
|
||||
TCP_SKB_CB(skb)->sacked |= TCPCB_LOST;
|
||||
}
|
||||
}
|
||||
@@ -913,6 +932,7 @@ void tcp_skb_mark_lost_uncond_verify(struct tcp_sock *tp, struct sk_buff *skb)
|
||||
{
|
||||
tcp_verify_retransmit_hint(tp, skb);
|
||||
|
||||
tcp_sum_lost(tp, skb);
|
||||
if (!(TCP_SKB_CB(skb)->sacked & (TCPCB_LOST|TCPCB_SACKED_ACKED))) {
|
||||
tp->lost_out += tcp_skb_pcount(skb);
|
||||
TCP_SKB_CB(skb)->sacked |= TCPCB_LOST;
|
||||
@@ -1094,6 +1114,7 @@ struct tcp_sacktag_state {
|
||||
*/
|
||||
struct skb_mstamp first_sackt;
|
||||
struct skb_mstamp last_sackt;
|
||||
struct rate_sample *rate;
|
||||
int flag;
|
||||
};
|
||||
|
||||
@@ -1261,6 +1282,7 @@ static bool tcp_shifted_skb(struct sock *sk, struct sk_buff *skb,
|
||||
tcp_sacktag_one(sk, state, TCP_SKB_CB(skb)->sacked,
|
||||
start_seq, end_seq, dup_sack, pcount,
|
||||
&skb->skb_mstamp);
|
||||
tcp_rate_skb_delivered(sk, skb, state->rate);
|
||||
|
||||
if (skb == tp->lost_skb_hint)
|
||||
tp->lost_cnt_hint += pcount;
|
||||
@@ -1311,6 +1333,9 @@ static bool tcp_shifted_skb(struct sock *sk, struct sk_buff *skb,
|
||||
tcp_advance_highest_sack(sk, skb);
|
||||
|
||||
tcp_skb_collapse_tstamp(prev, skb);
|
||||
if (unlikely(TCP_SKB_CB(prev)->tx.delivered_mstamp.v64))
|
||||
TCP_SKB_CB(prev)->tx.delivered_mstamp.v64 = 0;
|
||||
|
||||
tcp_unlink_write_queue(skb, sk);
|
||||
sk_wmem_free_skb(sk, skb);
|
||||
|
||||
@@ -1540,6 +1565,7 @@ static struct sk_buff *tcp_sacktag_walk(struct sk_buff *skb, struct sock *sk,
|
||||
dup_sack,
|
||||
tcp_skb_pcount(skb),
|
||||
&skb->skb_mstamp);
|
||||
tcp_rate_skb_delivered(sk, skb, state->rate);
|
||||
|
||||
if (!before(TCP_SKB_CB(skb)->seq,
|
||||
tcp_highest_sack_seq(tp)))
|
||||
@@ -1622,8 +1648,10 @@ tcp_sacktag_write_queue(struct sock *sk, const struct sk_buff *ack_skb,
|
||||
|
||||
found_dup_sack = tcp_check_dsack(sk, ack_skb, sp_wire,
|
||||
num_sacks, prior_snd_una);
|
||||
if (found_dup_sack)
|
||||
if (found_dup_sack) {
|
||||
state->flag |= FLAG_DSACKING_ACK;
|
||||
tp->delivered++; /* A spurious retransmission is delivered */
|
||||
}
|
||||
|
||||
/* Eliminate too old ACKs, but take into
|
||||
* account more or less fresh ones, they can
|
||||
@@ -1890,6 +1918,7 @@ void tcp_enter_loss(struct sock *sk)
|
||||
struct sk_buff *skb;
|
||||
bool new_recovery = icsk->icsk_ca_state < TCP_CA_Recovery;
|
||||
bool is_reneg; /* is receiver reneging on SACKs? */
|
||||
bool mark_lost;
|
||||
|
||||
/* Reduce ssthresh if it has not yet been made inside this window. */
|
||||
if (icsk->icsk_ca_state <= TCP_CA_Disorder ||
|
||||
@@ -1923,8 +1952,12 @@ void tcp_enter_loss(struct sock *sk)
|
||||
if (skb == tcp_send_head(sk))
|
||||
break;
|
||||
|
||||
mark_lost = (!(TCP_SKB_CB(skb)->sacked & TCPCB_SACKED_ACKED) ||
|
||||
is_reneg);
|
||||
if (mark_lost)
|
||||
tcp_sum_lost(tp, skb);
|
||||
TCP_SKB_CB(skb)->sacked &= (~TCPCB_TAGBITS)|TCPCB_SACKED_ACKED;
|
||||
if (!(TCP_SKB_CB(skb)->sacked&TCPCB_SACKED_ACKED) || is_reneg) {
|
||||
if (mark_lost) {
|
||||
TCP_SKB_CB(skb)->sacked &= ~TCPCB_SACKED_ACKED;
|
||||
TCP_SKB_CB(skb)->sacked |= TCPCB_LOST;
|
||||
tp->lost_out += tcp_skb_pcount(skb);
|
||||
@@ -2329,10 +2362,9 @@ static void DBGUNDO(struct sock *sk, const char *msg)
|
||||
}
|
||||
#if IS_ENABLED(CONFIG_IPV6)
|
||||
else if (sk->sk_family == AF_INET6) {
|
||||
struct ipv6_pinfo *np = inet6_sk(sk);
|
||||
pr_debug("Undo %s %pI6/%u c%u l%u ss%u/%u p%u\n",
|
||||
msg,
|
||||
&np->daddr, ntohs(inet->inet_dport),
|
||||
&sk->sk_v6_daddr, ntohs(inet->inet_dport),
|
||||
tp->snd_cwnd, tcp_left_out(tp),
|
||||
tp->snd_ssthresh, tp->prior_ssthresh,
|
||||
tp->packets_out);
|
||||
@@ -2503,6 +2535,9 @@ static inline void tcp_end_cwnd_reduction(struct sock *sk)
|
||||
{
|
||||
struct tcp_sock *tp = tcp_sk(sk);
|
||||
|
||||
if (inet_csk(sk)->icsk_ca_ops->cong_control)
|
||||
return;
|
||||
|
||||
/* Reset cwnd to ssthresh in CWR or Recovery (unless it's undone) */
|
||||
if (inet_csk(sk)->icsk_ca_state == TCP_CA_CWR ||
|
||||
(tp->undo_marker && tp->snd_ssthresh < TCP_INFINITE_SSTHRESH)) {
|
||||
@@ -2879,67 +2914,13 @@ static void tcp_fastretrans_alert(struct sock *sk, const int acked,
|
||||
*rexmit = REXMIT_LOST;
|
||||
}
|
||||
|
||||
/* Kathleen Nichols' algorithm for tracking the minimum value of
|
||||
* a data stream over some fixed time interval. (E.g., the minimum
|
||||
* RTT over the past five minutes.) It uses constant space and constant
|
||||
* time per update yet almost always delivers the same minimum as an
|
||||
* implementation that has to keep all the data in the window.
|
||||
*
|
||||
* The algorithm keeps track of the best, 2nd best & 3rd best min
|
||||
* values, maintaining an invariant that the measurement time of the
|
||||
* n'th best >= n-1'th best. It also makes sure that the three values
|
||||
* are widely separated in the time window since that bounds the worse
|
||||
* case error when that data is monotonically increasing over the window.
|
||||
*
|
||||
* Upon getting a new min, we can forget everything earlier because it
|
||||
* has no value - the new min is <= everything else in the window by
|
||||
* definition and it's the most recent. So we restart fresh on every new min
|
||||
* and overwrites 2nd & 3rd choices. The same property holds for 2nd & 3rd
|
||||
* best.
|
||||
*/
|
||||
static void tcp_update_rtt_min(struct sock *sk, u32 rtt_us)
|
||||
{
|
||||
const u32 now = tcp_time_stamp, wlen = sysctl_tcp_min_rtt_wlen * HZ;
|
||||
struct rtt_meas *m = tcp_sk(sk)->rtt_min;
|
||||
struct rtt_meas rttm = {
|
||||
.rtt = likely(rtt_us) ? rtt_us : jiffies_to_usecs(1),
|
||||
.ts = now,
|
||||
};
|
||||
u32 elapsed;
|
||||
struct tcp_sock *tp = tcp_sk(sk);
|
||||
u32 wlen = sysctl_tcp_min_rtt_wlen * HZ;
|
||||
|
||||
/* Check if the new measurement updates the 1st, 2nd, or 3rd choices */
|
||||
if (unlikely(rttm.rtt <= m[0].rtt))
|
||||
m[0] = m[1] = m[2] = rttm;
|
||||
else if (rttm.rtt <= m[1].rtt)
|
||||
m[1] = m[2] = rttm;
|
||||
else if (rttm.rtt <= m[2].rtt)
|
||||
m[2] = rttm;
|
||||
|
||||
elapsed = now - m[0].ts;
|
||||
if (unlikely(elapsed > wlen)) {
|
||||
/* Passed entire window without a new min so make 2nd choice
|
||||
* the new min & 3rd choice the new 2nd. So forth and so on.
|
||||
*/
|
||||
m[0] = m[1];
|
||||
m[1] = m[2];
|
||||
m[2] = rttm;
|
||||
if (now - m[0].ts > wlen) {
|
||||
m[0] = m[1];
|
||||
m[1] = rttm;
|
||||
if (now - m[0].ts > wlen)
|
||||
m[0] = rttm;
|
||||
}
|
||||
} else if (m[1].ts == m[0].ts && elapsed > wlen / 4) {
|
||||
/* Passed a quarter of the window without a new min so
|
||||
* take 2nd choice from the 2nd quarter of the window.
|
||||
*/
|
||||
m[2] = m[1] = rttm;
|
||||
} else if (m[2].ts == m[1].ts && elapsed > wlen / 2) {
|
||||
/* Passed half the window without a new min so take the 3rd
|
||||
* choice from the last half of the window.
|
||||
*/
|
||||
m[2] = rttm;
|
||||
}
|
||||
minmax_running_min(&tp->rtt_min, wlen, tcp_time_stamp,
|
||||
rtt_us ? : jiffies_to_usecs(1));
|
||||
}
|
||||
|
||||
static inline bool tcp_ack_update_rtt(struct sock *sk, const int flag,
|
||||
@@ -3102,10 +3083,11 @@ static void tcp_ack_tstamp(struct sock *sk, struct sk_buff *skb,
|
||||
*/
|
||||
static int tcp_clean_rtx_queue(struct sock *sk, int prior_fackets,
|
||||
u32 prior_snd_una, int *acked,
|
||||
struct tcp_sacktag_state *sack)
|
||||
struct tcp_sacktag_state *sack,
|
||||
struct skb_mstamp *now)
|
||||
{
|
||||
const struct inet_connection_sock *icsk = inet_csk(sk);
|
||||
struct skb_mstamp first_ackt, last_ackt, now;
|
||||
struct skb_mstamp first_ackt, last_ackt;
|
||||
struct tcp_sock *tp = tcp_sk(sk);
|
||||
u32 prior_sacked = tp->sacked_out;
|
||||
u32 reord = tp->packets_out;
|
||||
@@ -3137,7 +3119,6 @@ static int tcp_clean_rtx_queue(struct sock *sk, int prior_fackets,
|
||||
acked_pcount = tcp_tso_acked(sk, skb);
|
||||
if (!acked_pcount)
|
||||
break;
|
||||
|
||||
fully_acked = false;
|
||||
} else {
|
||||
/* Speedup tcp_unlink_write_queue() and next loop */
|
||||
@@ -3173,6 +3154,7 @@ static int tcp_clean_rtx_queue(struct sock *sk, int prior_fackets,
|
||||
|
||||
tp->packets_out -= acked_pcount;
|
||||
pkts_acked += acked_pcount;
|
||||
tcp_rate_skb_delivered(sk, skb, sack->rate);
|
||||
|
||||
/* Initial outgoing SYN's get put onto the write_queue
|
||||
* just like anything else we transmit. It is not
|
||||
@@ -3205,16 +3187,15 @@ static int tcp_clean_rtx_queue(struct sock *sk, int prior_fackets,
|
||||
if (skb && (TCP_SKB_CB(skb)->sacked & TCPCB_SACKED_ACKED))
|
||||
flag |= FLAG_SACK_RENEGING;
|
||||
|
||||
skb_mstamp_get(&now);
|
||||
if (likely(first_ackt.v64) && !(flag & FLAG_RETRANS_DATA_ACKED)) {
|
||||
seq_rtt_us = skb_mstamp_us_delta(&now, &first_ackt);
|
||||
ca_rtt_us = skb_mstamp_us_delta(&now, &last_ackt);
|
||||
seq_rtt_us = skb_mstamp_us_delta(now, &first_ackt);
|
||||
ca_rtt_us = skb_mstamp_us_delta(now, &last_ackt);
|
||||
}
|
||||
if (sack->first_sackt.v64) {
|
||||
sack_rtt_us = skb_mstamp_us_delta(&now, &sack->first_sackt);
|
||||
ca_rtt_us = skb_mstamp_us_delta(&now, &sack->last_sackt);
|
||||
sack_rtt_us = skb_mstamp_us_delta(now, &sack->first_sackt);
|
||||
ca_rtt_us = skb_mstamp_us_delta(now, &sack->last_sackt);
|
||||
}
|
||||
|
||||
sack->rate->rtt_us = ca_rtt_us; /* RTT of last (S)ACKed packet, or -1 */
|
||||
rtt_update = tcp_ack_update_rtt(sk, flag, seq_rtt_us, sack_rtt_us,
|
||||
ca_rtt_us);
|
||||
|
||||
@@ -3242,7 +3223,7 @@ static int tcp_clean_rtx_queue(struct sock *sk, int prior_fackets,
|
||||
tp->fackets_out -= min(pkts_acked, tp->fackets_out);
|
||||
|
||||
} else if (skb && rtt_update && sack_rtt_us >= 0 &&
|
||||
sack_rtt_us > skb_mstamp_us_delta(&now, &skb->skb_mstamp)) {
|
||||
sack_rtt_us > skb_mstamp_us_delta(now, &skb->skb_mstamp)) {
|
||||
/* Do not re-arm RTO if the sack RTT is measured from data sent
|
||||
* after when the head was last (re)transmitted. Otherwise the
|
||||
* timeout may continue to extend in loss recovery.
|
||||
@@ -3333,8 +3314,15 @@ static inline bool tcp_may_raise_cwnd(const struct sock *sk, const int flag)
|
||||
* information. All transmission or retransmission are delayed afterwards.
|
||||
*/
|
||||
static void tcp_cong_control(struct sock *sk, u32 ack, u32 acked_sacked,
|
||||
int flag)
|
||||
int flag, const struct rate_sample *rs)
|
||||
{
|
||||
const struct inet_connection_sock *icsk = inet_csk(sk);
|
||||
|
||||
if (icsk->icsk_ca_ops->cong_control) {
|
||||
icsk->icsk_ca_ops->cong_control(sk, rs);
|
||||
return;
|
||||
}
|
||||
|
||||
if (tcp_in_cwnd_reduction(sk)) {
|
||||
/* Reduce cwnd if state mandates */
|
||||
tcp_cwnd_reduction(sk, acked_sacked, flag);
|
||||
@@ -3579,17 +3567,21 @@ static int tcp_ack(struct sock *sk, const struct sk_buff *skb, int flag)
|
||||
struct inet_connection_sock *icsk = inet_csk(sk);
|
||||
struct tcp_sock *tp = tcp_sk(sk);
|
||||
struct tcp_sacktag_state sack_state;
|
||||
struct rate_sample rs = { .prior_delivered = 0 };
|
||||
u32 prior_snd_una = tp->snd_una;
|
||||
u32 ack_seq = TCP_SKB_CB(skb)->seq;
|
||||
u32 ack = TCP_SKB_CB(skb)->ack_seq;
|
||||
bool is_dupack = false;
|
||||
u32 prior_fackets;
|
||||
int prior_packets = tp->packets_out;
|
||||
u32 prior_delivered = tp->delivered;
|
||||
u32 delivered = tp->delivered;
|
||||
u32 lost = tp->lost;
|
||||
int acked = 0; /* Number of packets newly acked */
|
||||
int rexmit = REXMIT_NONE; /* Flag to (re)transmit to recover losses */
|
||||
struct skb_mstamp now;
|
||||
|
||||
sack_state.first_sackt.v64 = 0;
|
||||
sack_state.rate = &rs;
|
||||
|
||||
/* We very likely will need to access write queue head. */
|
||||
prefetchw(sk->sk_write_queue.next);
|
||||
@@ -3612,6 +3604,8 @@ static int tcp_ack(struct sock *sk, const struct sk_buff *skb, int flag)
|
||||
if (after(ack, tp->snd_nxt))
|
||||
goto invalid_ack;
|
||||
|
||||
skb_mstamp_get(&now);
|
||||
|
||||
if (icsk->icsk_pending == ICSK_TIME_EARLY_RETRANS ||
|
||||
icsk->icsk_pending == ICSK_TIME_LOSS_PROBE)
|
||||
tcp_rearm_rto(sk);
|
||||
@@ -3622,6 +3616,7 @@ static int tcp_ack(struct sock *sk, const struct sk_buff *skb, int flag)
|
||||
}
|
||||
|
||||
prior_fackets = tp->fackets_out;
|
||||
rs.prior_in_flight = tcp_packets_in_flight(tp);
|
||||
|
||||
/* ts_recent update must be made after we are sure that the packet
|
||||
* is in window.
|
||||
@@ -3677,7 +3672,7 @@ static int tcp_ack(struct sock *sk, const struct sk_buff *skb, int flag)
|
||||
|
||||
/* See if we can take anything off of the retransmit queue. */
|
||||
flag |= tcp_clean_rtx_queue(sk, prior_fackets, prior_snd_una, &acked,
|
||||
&sack_state);
|
||||
&sack_state, &now);
|
||||
|
||||
if (tcp_ack_is_dubious(sk, flag)) {
|
||||
is_dupack = !(flag & (FLAG_SND_UNA_ADVANCED | FLAG_NOT_DUP));
|
||||
@@ -3694,7 +3689,10 @@ static int tcp_ack(struct sock *sk, const struct sk_buff *skb, int flag)
|
||||
|
||||
if (icsk->icsk_pending == ICSK_TIME_RETRANS)
|
||||
tcp_schedule_loss_probe(sk);
|
||||
tcp_cong_control(sk, ack, tp->delivered - prior_delivered, flag);
|
||||
delivered = tp->delivered - delivered; /* freshly ACKed or SACKed */
|
||||
lost = tp->lost - lost; /* freshly marked lost */
|
||||
tcp_rate_gen(sk, delivered, lost, &now, &rs);
|
||||
tcp_cong_control(sk, ack, delivered, flag, &rs);
|
||||
tcp_xmit_recovery(sk, rexmit);
|
||||
return 1;
|
||||
|
||||
@@ -5951,7 +5949,7 @@ int tcp_rcv_state_process(struct sock *sk, struct sk_buff *skb)
|
||||
* so release it.
|
||||
*/
|
||||
if (req) {
|
||||
tp->total_retrans = req->num_retrans;
|
||||
inet_csk(sk)->icsk_retransmits = 0;
|
||||
reqsk_fastopen_remove(sk, req, false);
|
||||
} else {
|
||||
/* Make sure socket is routed, for correct metrics. */
|
||||
@@ -5993,7 +5991,8 @@ int tcp_rcv_state_process(struct sock *sk, struct sk_buff *skb)
|
||||
} else
|
||||
tcp_init_metrics(sk);
|
||||
|
||||
tcp_update_pacing_rate(sk);
|
||||
if (!inet_csk(sk)->icsk_ca_ops->cong_control)
|
||||
tcp_update_pacing_rate(sk);
|
||||
|
||||
/* Prevent spurious tcp_cwnd_restart() on first data packet */
|
||||
tp->lsndtime = tcp_time_stamp;
|
||||
@@ -6326,6 +6325,7 @@ int tcp_conn_request(struct request_sock_ops *rsk_ops,
|
||||
|
||||
tmp_opt.tstamp_ok = tmp_opt.saw_tstamp;
|
||||
tcp_openreq_init(req, &tmp_opt, skb, sk);
|
||||
inet_rsk(req)->no_srccheck = inet_sk(sk)->transparent;
|
||||
|
||||
/* Note: tcp_v6_init_req() might override ir_iif for link locals */
|
||||
inet_rsk(req)->ir_iif = inet_request_bound_dev_if(sk, skb);
|
||||
|
||||
@@ -1196,7 +1196,6 @@ static void tcp_v4_init_req(struct request_sock *req,
|
||||
|
||||
sk_rcv_saddr_set(req_to_sk(req), ip_hdr(skb)->daddr);
|
||||
sk_daddr_set(req_to_sk(req), ip_hdr(skb)->saddr);
|
||||
ireq->no_srccheck = inet_sk(sk_listener)->transparent;
|
||||
ireq->opt = tcp_v4_save_options(skb);
|
||||
}
|
||||
|
||||
|
||||
@@ -464,7 +464,7 @@ struct sock *tcp_create_openreq_child(const struct sock *sk,
|
||||
|
||||
newtp->srtt_us = 0;
|
||||
newtp->mdev_us = jiffies_to_usecs(TCP_TIMEOUT_INIT);
|
||||
newtp->rtt_min[0].rtt = ~0U;
|
||||
minmax_reset(&newtp->rtt_min, tcp_time_stamp, ~0U);
|
||||
newicsk->icsk_rto = TCP_TIMEOUT_INIT;
|
||||
|
||||
newtp->packets_out = 0;
|
||||
@@ -487,6 +487,9 @@ struct sock *tcp_create_openreq_child(const struct sock *sk,
|
||||
newtp->snd_cwnd = TCP_INIT_CWND;
|
||||
newtp->snd_cwnd_cnt = 0;
|
||||
|
||||
/* There's a bubble in the pipe until at least the first ACK. */
|
||||
newtp->app_limited = ~0U;
|
||||
|
||||
tcp_init_xmit_timers(newsk);
|
||||
newtp->write_seq = newtp->pushed_seq = treq->snt_isn + 1;
|
||||
|
||||
|
||||
@@ -90,12 +90,6 @@ struct sk_buff *tcp_gso_segment(struct sk_buff *skb,
|
||||
goto out;
|
||||
}
|
||||
|
||||
/* GSO partial only requires splitting the frame into an MSS
|
||||
* multiple and possibly a remainder. So update the mss now.
|
||||
*/
|
||||
if (features & NETIF_F_GSO_PARTIAL)
|
||||
mss = skb->len - (skb->len % mss);
|
||||
|
||||
copy_destructor = gso_skb->destructor == tcp_wfree;
|
||||
ooo_okay = gso_skb->ooo_okay;
|
||||
/* All segments but the first should have ooo_okay cleared */
|
||||
@@ -108,6 +102,13 @@ struct sk_buff *tcp_gso_segment(struct sk_buff *skb,
|
||||
/* Only first segment might have ooo_okay set */
|
||||
segs->ooo_okay = ooo_okay;
|
||||
|
||||
/* GSO partial and frag_list segmentation only requires splitting
|
||||
* the frame into an MSS multiple and possibly a remainder, both
|
||||
* cases return a GSO skb. So update the mss now.
|
||||
*/
|
||||
if (skb_is_gso(segs))
|
||||
mss *= skb_shinfo(segs)->gso_segs;
|
||||
|
||||
delta = htonl(oldlen + (thlen + mss));
|
||||
|
||||
skb = segs;
|
||||
|
||||
+82
-36
@@ -734,9 +734,16 @@ static void tcp_tsq_handler(struct sock *sk)
|
||||
{
|
||||
if ((1 << sk->sk_state) &
|
||||
(TCPF_ESTABLISHED | TCPF_FIN_WAIT1 | TCPF_CLOSING |
|
||||
TCPF_CLOSE_WAIT | TCPF_LAST_ACK))
|
||||
tcp_write_xmit(sk, tcp_current_mss(sk), tcp_sk(sk)->nonagle,
|
||||
TCPF_CLOSE_WAIT | TCPF_LAST_ACK)) {
|
||||
struct tcp_sock *tp = tcp_sk(sk);
|
||||
|
||||
if (tp->lost_out > tp->retrans_out &&
|
||||
tp->snd_cwnd > tcp_packets_in_flight(tp))
|
||||
tcp_xmit_retransmit_queue(sk);
|
||||
|
||||
tcp_write_xmit(sk, tcp_current_mss(sk), tp->nonagle,
|
||||
0, GFP_ATOMIC);
|
||||
}
|
||||
}
|
||||
/*
|
||||
* One tasklet per cpu tries to send more skbs.
|
||||
@@ -918,6 +925,7 @@ static int tcp_transmit_skb(struct sock *sk, struct sk_buff *skb, int clone_it,
|
||||
skb_mstamp_get(&skb->skb_mstamp);
|
||||
TCP_SKB_CB(skb)->tx.in_flight = TCP_SKB_CB(skb)->end_seq
|
||||
- tp->snd_una;
|
||||
tcp_rate_skb_sent(sk, skb);
|
||||
|
||||
if (unlikely(skb_cloned(skb)))
|
||||
skb = pskb_copy(skb, gfp_mask);
|
||||
@@ -1213,6 +1221,9 @@ int tcp_fragment(struct sock *sk, struct sk_buff *skb, u32 len,
|
||||
tcp_set_skb_tso_segs(skb, mss_now);
|
||||
tcp_set_skb_tso_segs(buff, mss_now);
|
||||
|
||||
/* Update delivered info for the new segment */
|
||||
TCP_SKB_CB(buff)->tx = TCP_SKB_CB(skb)->tx;
|
||||
|
||||
/* If this packet has been sent out already, we must
|
||||
* adjust the various packet counters.
|
||||
*/
|
||||
@@ -1358,6 +1369,7 @@ int tcp_mss_to_mtu(struct sock *sk, int mss)
|
||||
}
|
||||
return mtu;
|
||||
}
|
||||
EXPORT_SYMBOL(tcp_mss_to_mtu);
|
||||
|
||||
/* MTU probing init per socket */
|
||||
void tcp_mtup_init(struct sock *sk)
|
||||
@@ -1545,7 +1557,8 @@ static bool tcp_nagle_check(bool partial, const struct tcp_sock *tp,
|
||||
/* Return how many segs we'd like on a TSO packet,
|
||||
* to send one TSO packet per ms
|
||||
*/
|
||||
static u32 tcp_tso_autosize(const struct sock *sk, unsigned int mss_now)
|
||||
u32 tcp_tso_autosize(const struct sock *sk, unsigned int mss_now,
|
||||
int min_tso_segs)
|
||||
{
|
||||
u32 bytes, segs;
|
||||
|
||||
@@ -1557,10 +1570,23 @@ static u32 tcp_tso_autosize(const struct sock *sk, unsigned int mss_now)
|
||||
* This preserves ACK clocking and is consistent
|
||||
* with tcp_tso_should_defer() heuristic.
|
||||
*/
|
||||
segs = max_t(u32, bytes / mss_now, sysctl_tcp_min_tso_segs);
|
||||
segs = max_t(u32, bytes / mss_now, min_tso_segs);
|
||||
|
||||
return min_t(u32, segs, sk->sk_gso_max_segs);
|
||||
}
|
||||
EXPORT_SYMBOL(tcp_tso_autosize);
|
||||
|
||||
/* Return the number of segments we want in the skb we are transmitting.
|
||||
* See if congestion control module wants to decide; otherwise, autosize.
|
||||
*/
|
||||
static u32 tcp_tso_segs(struct sock *sk, unsigned int mss_now)
|
||||
{
|
||||
const struct tcp_congestion_ops *ca_ops = inet_csk(sk)->icsk_ca_ops;
|
||||
u32 tso_segs = ca_ops->tso_segs_goal ? ca_ops->tso_segs_goal(sk) : 0;
|
||||
|
||||
return tso_segs ? :
|
||||
tcp_tso_autosize(sk, mss_now, sysctl_tcp_min_tso_segs);
|
||||
}
|
||||
|
||||
/* Returns the portion of skb which can be sent right away */
|
||||
static unsigned int tcp_mss_split_point(const struct sock *sk,
|
||||
@@ -1966,12 +1992,14 @@ static int tcp_mtu_probe(struct sock *sk)
|
||||
len = 0;
|
||||
tcp_for_write_queue_from_safe(skb, next, sk) {
|
||||
copy = min_t(int, skb->len, probe_size - len);
|
||||
if (nskb->ip_summed)
|
||||
if (nskb->ip_summed) {
|
||||
skb_copy_bits(skb, 0, skb_put(nskb, copy), copy);
|
||||
else
|
||||
nskb->csum = skb_copy_and_csum_bits(skb, 0,
|
||||
skb_put(nskb, copy),
|
||||
copy, nskb->csum);
|
||||
} else {
|
||||
__wsum csum = skb_copy_and_csum_bits(skb, 0,
|
||||
skb_put(nskb, copy),
|
||||
copy, 0);
|
||||
nskb->csum = csum_block_add(nskb->csum, csum, len);
|
||||
}
|
||||
|
||||
if (skb->len <= copy) {
|
||||
/* We've eaten all the data from this skb.
|
||||
@@ -2020,6 +2048,39 @@ static int tcp_mtu_probe(struct sock *sk)
|
||||
return -1;
|
||||
}
|
||||
|
||||
/* TCP Small Queues :
|
||||
* Control number of packets in qdisc/devices to two packets / or ~1 ms.
|
||||
* (These limits are doubled for retransmits)
|
||||
* This allows for :
|
||||
* - better RTT estimation and ACK scheduling
|
||||
* - faster recovery
|
||||
* - high rates
|
||||
* Alas, some drivers / subsystems require a fair amount
|
||||
* of queued bytes to ensure line rate.
|
||||
* One example is wifi aggregation (802.11 AMPDU)
|
||||
*/
|
||||
static bool tcp_small_queue_check(struct sock *sk, const struct sk_buff *skb,
|
||||
unsigned int factor)
|
||||
{
|
||||
unsigned int limit;
|
||||
|
||||
limit = max(2 * skb->truesize, sk->sk_pacing_rate >> 10);
|
||||
limit = min_t(u32, limit, sysctl_tcp_limit_output_bytes);
|
||||
limit <<= factor;
|
||||
|
||||
if (atomic_read(&sk->sk_wmem_alloc) > limit) {
|
||||
set_bit(TSQ_THROTTLED, &tcp_sk(sk)->tsq_flags);
|
||||
/* It is possible TX completion already happened
|
||||
* before we set TSQ_THROTTLED, so we must
|
||||
* test again the condition.
|
||||
*/
|
||||
smp_mb__after_atomic();
|
||||
if (atomic_read(&sk->sk_wmem_alloc) > limit)
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/* This routine writes packets to the network. It advances the
|
||||
* send_head. This happens as incoming acks open up the remote
|
||||
* window for us.
|
||||
@@ -2057,7 +2118,7 @@ static bool tcp_write_xmit(struct sock *sk, unsigned int mss_now, int nonagle,
|
||||
}
|
||||
}
|
||||
|
||||
max_segs = tcp_tso_autosize(sk, mss_now);
|
||||
max_segs = tcp_tso_segs(sk, mss_now);
|
||||
while ((skb = tcp_send_head(sk))) {
|
||||
unsigned int limit;
|
||||
|
||||
@@ -2106,29 +2167,8 @@ static bool tcp_write_xmit(struct sock *sk, unsigned int mss_now, int nonagle,
|
||||
unlikely(tso_fragment(sk, skb, limit, mss_now, gfp)))
|
||||
break;
|
||||
|
||||
/* TCP Small Queues :
|
||||
* Control number of packets in qdisc/devices to two packets / or ~1 ms.
|
||||
* This allows for :
|
||||
* - better RTT estimation and ACK scheduling
|
||||
* - faster recovery
|
||||
* - high rates
|
||||
* Alas, some drivers / subsystems require a fair amount
|
||||
* of queued bytes to ensure line rate.
|
||||
* One example is wifi aggregation (802.11 AMPDU)
|
||||
*/
|
||||
limit = max(2 * skb->truesize, sk->sk_pacing_rate >> 10);
|
||||
limit = min_t(u32, limit, sysctl_tcp_limit_output_bytes);
|
||||
|
||||
if (atomic_read(&sk->sk_wmem_alloc) > limit) {
|
||||
set_bit(TSQ_THROTTLED, &tp->tsq_flags);
|
||||
/* It is possible TX completion already happened
|
||||
* before we set TSQ_THROTTLED, so we must
|
||||
* test again the condition.
|
||||
*/
|
||||
smp_mb__after_atomic();
|
||||
if (atomic_read(&sk->sk_wmem_alloc) > limit)
|
||||
break;
|
||||
}
|
||||
if (tcp_small_queue_check(sk, skb, 0))
|
||||
break;
|
||||
|
||||
if (unlikely(tcp_transmit_skb(sk, skb, 1, gfp)))
|
||||
break;
|
||||
@@ -2605,7 +2645,8 @@ int __tcp_retransmit_skb(struct sock *sk, struct sk_buff *skb, int segs)
|
||||
* copying overhead: fragmentation, tunneling, mangling etc.
|
||||
*/
|
||||
if (atomic_read(&sk->sk_wmem_alloc) >
|
||||
min(sk->sk_wmem_queued + (sk->sk_wmem_queued >> 2), sk->sk_sndbuf))
|
||||
min_t(u32, sk->sk_wmem_queued + (sk->sk_wmem_queued >> 2),
|
||||
sk->sk_sndbuf))
|
||||
return -EAGAIN;
|
||||
|
||||
if (skb_still_in_host_queue(sk, skb))
|
||||
@@ -2774,7 +2815,7 @@ void tcp_xmit_retransmit_queue(struct sock *sk)
|
||||
last_lost = tp->snd_una;
|
||||
}
|
||||
|
||||
max_segs = tcp_tso_autosize(sk, tcp_current_mss(sk));
|
||||
max_segs = tcp_tso_segs(sk, tcp_current_mss(sk));
|
||||
tcp_for_write_queue_from(skb, sk) {
|
||||
__u8 sacked;
|
||||
int segs;
|
||||
@@ -2828,10 +2869,13 @@ begin_fwd:
|
||||
if (sacked & (TCPCB_SACKED_ACKED|TCPCB_SACKED_RETRANS))
|
||||
continue;
|
||||
|
||||
if (tcp_small_queue_check(sk, skb, 1))
|
||||
return;
|
||||
|
||||
if (tcp_retransmit_skb(sk, skb, segs))
|
||||
return;
|
||||
|
||||
NET_INC_STATS(sock_net(sk), mib_idx);
|
||||
NET_ADD_STATS(sock_net(sk), mib_idx, tcp_skb_pcount(skb));
|
||||
|
||||
if (tcp_in_cwnd_reduction(sk))
|
||||
tp->prr_out += tcp_skb_pcount(skb);
|
||||
@@ -3568,6 +3612,8 @@ int tcp_rtx_synack(const struct sock *sk, struct request_sock *req)
|
||||
if (!res) {
|
||||
__TCP_INC_STATS(sock_net(sk), TCP_MIB_RETRANSSEGS);
|
||||
__NET_INC_STATS(sock_net(sk), LINUX_MIB_TCPSYNRETRANS);
|
||||
if (unlikely(tcp_passive_fastopen(sk)))
|
||||
tcp_sk(sk)->total_retrans++;
|
||||
}
|
||||
return res;
|
||||
}
|
||||
|
||||
@@ -0,0 +1,186 @@
|
||||
#include <net/tcp.h>
|
||||
|
||||
/* The bandwidth estimator estimates the rate at which the network
|
||||
* can currently deliver outbound data packets for this flow. At a high
|
||||
* level, it operates by taking a delivery rate sample for each ACK.
|
||||
*
|
||||
* A rate sample records the rate at which the network delivered packets
|
||||
* for this flow, calculated over the time interval between the transmission
|
||||
* of a data packet and the acknowledgment of that packet.
|
||||
*
|
||||
* Specifically, over the interval between each transmit and corresponding ACK,
|
||||
* the estimator generates a delivery rate sample. Typically it uses the rate
|
||||
* at which packets were acknowledged. However, the approach of using only the
|
||||
* acknowledgment rate faces a challenge under the prevalent ACK decimation or
|
||||
* compression: packets can temporarily appear to be delivered much quicker
|
||||
* than the bottleneck rate. Since it is physically impossible to do that in a
|
||||
* sustained fashion, when the estimator notices that the ACK rate is faster
|
||||
* than the transmit rate, it uses the latter:
|
||||
*
|
||||
* send_rate = #pkts_delivered/(last_snd_time - first_snd_time)
|
||||
* ack_rate = #pkts_delivered/(last_ack_time - first_ack_time)
|
||||
* bw = min(send_rate, ack_rate)
|
||||
*
|
||||
* Notice the estimator essentially estimates the goodput, not always the
|
||||
* network bottleneck link rate when the sending or receiving is limited by
|
||||
* other factors like applications or receiver window limits. The estimator
|
||||
* deliberately avoids using the inter-packet spacing approach because that
|
||||
* approach requires a large number of samples and sophisticated filtering.
|
||||
*
|
||||
* TCP flows can often be application-limited in request/response workloads.
|
||||
* The estimator marks a bandwidth sample as application-limited if there
|
||||
* was some moment during the sampled window of packets when there was no data
|
||||
* ready to send in the write queue.
|
||||
*/
|
||||
|
||||
/* Snapshot the current delivery information in the skb, to generate
|
||||
* a rate sample later when the skb is (s)acked in tcp_rate_skb_delivered().
|
||||
*/
|
||||
void tcp_rate_skb_sent(struct sock *sk, struct sk_buff *skb)
|
||||
{
|
||||
struct tcp_sock *tp = tcp_sk(sk);
|
||||
|
||||
/* In general we need to start delivery rate samples from the
|
||||
* time we received the most recent ACK, to ensure we include
|
||||
* the full time the network needs to deliver all in-flight
|
||||
* packets. If there are no packets in flight yet, then we
|
||||
* know that any ACKs after now indicate that the network was
|
||||
* able to deliver those packets completely in the sampling
|
||||
* interval between now and the next ACK.
|
||||
*
|
||||
* Note that we use packets_out instead of tcp_packets_in_flight(tp)
|
||||
* because the latter is a guess based on RTO and loss-marking
|
||||
* heuristics. We don't want spurious RTOs or loss markings to cause
|
||||
* a spuriously small time interval, causing a spuriously high
|
||||
* bandwidth estimate.
|
||||
*/
|
||||
if (!tp->packets_out) {
|
||||
tp->first_tx_mstamp = skb->skb_mstamp;
|
||||
tp->delivered_mstamp = skb->skb_mstamp;
|
||||
}
|
||||
|
||||
TCP_SKB_CB(skb)->tx.first_tx_mstamp = tp->first_tx_mstamp;
|
||||
TCP_SKB_CB(skb)->tx.delivered_mstamp = tp->delivered_mstamp;
|
||||
TCP_SKB_CB(skb)->tx.delivered = tp->delivered;
|
||||
TCP_SKB_CB(skb)->tx.is_app_limited = tp->app_limited ? 1 : 0;
|
||||
}
|
||||
|
||||
/* When an skb is sacked or acked, we fill in the rate sample with the (prior)
|
||||
* delivery information when the skb was last transmitted.
|
||||
*
|
||||
* If an ACK (s)acks multiple skbs (e.g., stretched-acks), this function is
|
||||
* called multiple times. We favor the information from the most recently
|
||||
* sent skb, i.e., the skb with the highest prior_delivered count.
|
||||
*/
|
||||
void tcp_rate_skb_delivered(struct sock *sk, struct sk_buff *skb,
|
||||
struct rate_sample *rs)
|
||||
{
|
||||
struct tcp_sock *tp = tcp_sk(sk);
|
||||
struct tcp_skb_cb *scb = TCP_SKB_CB(skb);
|
||||
|
||||
if (!scb->tx.delivered_mstamp.v64)
|
||||
return;
|
||||
|
||||
if (!rs->prior_delivered ||
|
||||
after(scb->tx.delivered, rs->prior_delivered)) {
|
||||
rs->prior_delivered = scb->tx.delivered;
|
||||
rs->prior_mstamp = scb->tx.delivered_mstamp;
|
||||
rs->is_app_limited = scb->tx.is_app_limited;
|
||||
rs->is_retrans = scb->sacked & TCPCB_RETRANS;
|
||||
|
||||
/* Find the duration of the "send phase" of this window: */
|
||||
rs->interval_us = skb_mstamp_us_delta(
|
||||
&skb->skb_mstamp,
|
||||
&scb->tx.first_tx_mstamp);
|
||||
|
||||
/* Record send time of most recently ACKed packet: */
|
||||
tp->first_tx_mstamp = skb->skb_mstamp;
|
||||
}
|
||||
/* Mark off the skb delivered once it's sacked to avoid being
|
||||
* used again when it's cumulatively acked. For acked packets
|
||||
* we don't need to reset since it'll be freed soon.
|
||||
*/
|
||||
if (scb->sacked & TCPCB_SACKED_ACKED)
|
||||
scb->tx.delivered_mstamp.v64 = 0;
|
||||
}
|
||||
|
||||
/* Update the connection delivery information and generate a rate sample. */
|
||||
void tcp_rate_gen(struct sock *sk, u32 delivered, u32 lost,
|
||||
struct skb_mstamp *now, struct rate_sample *rs)
|
||||
{
|
||||
struct tcp_sock *tp = tcp_sk(sk);
|
||||
u32 snd_us, ack_us;
|
||||
|
||||
/* Clear app limited if bubble is acked and gone. */
|
||||
if (tp->app_limited && after(tp->delivered, tp->app_limited))
|
||||
tp->app_limited = 0;
|
||||
|
||||
/* TODO: there are multiple places throughout tcp_ack() to get
|
||||
* current time. Refactor the code using a new "tcp_acktag_state"
|
||||
* to carry current time, flags, stats like "tcp_sacktag_state".
|
||||
*/
|
||||
if (delivered)
|
||||
tp->delivered_mstamp = *now;
|
||||
|
||||
rs->acked_sacked = delivered; /* freshly ACKed or SACKed */
|
||||
rs->losses = lost; /* freshly marked lost */
|
||||
/* Return an invalid sample if no timing information is available. */
|
||||
if (!rs->prior_mstamp.v64) {
|
||||
rs->delivered = -1;
|
||||
rs->interval_us = -1;
|
||||
return;
|
||||
}
|
||||
rs->delivered = tp->delivered - rs->prior_delivered;
|
||||
|
||||
/* Model sending data and receiving ACKs as separate pipeline phases
|
||||
* for a window. Usually the ACK phase is longer, but with ACK
|
||||
* compression the send phase can be longer. To be safe we use the
|
||||
* longer phase.
|
||||
*/
|
||||
snd_us = rs->interval_us; /* send phase */
|
||||
ack_us = skb_mstamp_us_delta(now, &rs->prior_mstamp); /* ack phase */
|
||||
rs->interval_us = max(snd_us, ack_us);
|
||||
|
||||
/* Normally we expect interval_us >= min-rtt.
|
||||
* Note that rate may still be over-estimated when a spuriously
|
||||
* retransmistted skb was first (s)acked because "interval_us"
|
||||
* is under-estimated (up to an RTT). However continuously
|
||||
* measuring the delivery rate during loss recovery is crucial
|
||||
* for connections suffer heavy or prolonged losses.
|
||||
*/
|
||||
if (unlikely(rs->interval_us < tcp_min_rtt(tp))) {
|
||||
if (!rs->is_retrans)
|
||||
pr_debug("tcp rate: %ld %d %u %u %u\n",
|
||||
rs->interval_us, rs->delivered,
|
||||
inet_csk(sk)->icsk_ca_state,
|
||||
tp->rx_opt.sack_ok, tcp_min_rtt(tp));
|
||||
rs->interval_us = -1;
|
||||
return;
|
||||
}
|
||||
|
||||
/* Record the last non-app-limited or the highest app-limited bw */
|
||||
if (!rs->is_app_limited ||
|
||||
((u64)rs->delivered * tp->rate_interval_us >=
|
||||
(u64)tp->rate_delivered * rs->interval_us)) {
|
||||
tp->rate_delivered = rs->delivered;
|
||||
tp->rate_interval_us = rs->interval_us;
|
||||
tp->rate_app_limited = rs->is_app_limited;
|
||||
}
|
||||
}
|
||||
|
||||
/* If a gap is detected between sends, mark the socket application-limited. */
|
||||
void tcp_rate_check_app_limited(struct sock *sk)
|
||||
{
|
||||
struct tcp_sock *tp = tcp_sk(sk);
|
||||
|
||||
if (/* We have less than one packet to send. */
|
||||
tp->write_seq - tp->snd_nxt < tp->mss_cache &&
|
||||
/* Nothing in sending host's qdisc queues or NIC tx queue. */
|
||||
sk_wmem_alloc_get(sk) < SKB_TRUESIZE(1) &&
|
||||
/* We are not limited by CWND. */
|
||||
tcp_packets_in_flight(tp) < tp->snd_cwnd &&
|
||||
/* All lost packets have been retransmitted. */
|
||||
tp->lost_out <= tp->retrans_out)
|
||||
tp->app_limited =
|
||||
(tp->delivered + tcp_packets_in_flight(tp)) ? : 1;
|
||||
}
|
||||
@@ -192,6 +192,8 @@ static int tcp_write_timeout(struct sock *sk)
|
||||
if (tp->syn_data && icsk->icsk_retransmits == 1)
|
||||
NET_INC_STATS(sock_net(sk),
|
||||
LINUX_MIB_TCPFASTOPENACTIVEFAIL);
|
||||
} else if (!tp->syn_data && !tp->syn_fastopen) {
|
||||
sk_rethink_txhash(sk);
|
||||
}
|
||||
retry_until = icsk->icsk_syn_retries ? : net->ipv4.sysctl_tcp_syn_retries;
|
||||
syn_set = true;
|
||||
@@ -213,6 +215,8 @@ static int tcp_write_timeout(struct sock *sk)
|
||||
tcp_mtu_probing(icsk, sk);
|
||||
|
||||
dst_negative_advice(sk);
|
||||
} else {
|
||||
sk_rethink_txhash(sk);
|
||||
}
|
||||
|
||||
retry_until = net->ipv4.sysctl_tcp_retries2;
|
||||
@@ -384,6 +388,7 @@ static void tcp_fastopen_synack_timer(struct sock *sk)
|
||||
*/
|
||||
inet_rtx_syn_ack(sk, req);
|
||||
req->num_timeout++;
|
||||
icsk->icsk_retransmits++;
|
||||
inet_csk_reset_xmit_timer(sk, ICSK_TIME_RETRANS,
|
||||
TCP_TIMEOUT_INIT << req->num_timeout, TCP_RTO_MAX);
|
||||
}
|
||||
|
||||
@@ -21,7 +21,7 @@ static struct sk_buff *__skb_udp_tunnel_segment(struct sk_buff *skb,
|
||||
__be16 new_protocol, bool is_ipv6)
|
||||
{
|
||||
int tnl_hlen = skb_inner_mac_header(skb) - skb_transport_header(skb);
|
||||
bool remcsum, need_csum, offload_csum, ufo;
|
||||
bool remcsum, need_csum, offload_csum, ufo, gso_partial;
|
||||
struct sk_buff *segs = ERR_PTR(-EINVAL);
|
||||
struct udphdr *uh = udp_hdr(skb);
|
||||
u16 mac_offset = skb->mac_header;
|
||||
@@ -88,6 +88,8 @@ static struct sk_buff *__skb_udp_tunnel_segment(struct sk_buff *skb,
|
||||
goto out;
|
||||
}
|
||||
|
||||
gso_partial = !!(skb_shinfo(segs)->gso_type & SKB_GSO_PARTIAL);
|
||||
|
||||
outer_hlen = skb_tnl_header_len(skb);
|
||||
udp_offset = outer_hlen - tnl_hlen;
|
||||
skb = segs;
|
||||
@@ -117,7 +119,7 @@ static struct sk_buff *__skb_udp_tunnel_segment(struct sk_buff *skb,
|
||||
* will be using a length value equal to only one MSS sized
|
||||
* segment instead of the entire frame.
|
||||
*/
|
||||
if (skb_is_gso(skb)) {
|
||||
if (gso_partial) {
|
||||
uh->len = htons(skb_shinfo(skb)->gso_size +
|
||||
SKB_GSO_CB(skb)->data_offset +
|
||||
skb->head - (unsigned char *)uh);
|
||||
|
||||
+64
-30
@@ -112,6 +112,27 @@ static inline u32 cstamp_delta(unsigned long cstamp)
|
||||
return (cstamp - INITIAL_JIFFIES) * 100UL / HZ;
|
||||
}
|
||||
|
||||
static inline s32 rfc3315_s14_backoff_init(s32 irt)
|
||||
{
|
||||
/* multiply 'initial retransmission time' by 0.9 .. 1.1 */
|
||||
u64 tmp = (900000 + prandom_u32() % 200001) * (u64)irt;
|
||||
do_div(tmp, 1000000);
|
||||
return (s32)tmp;
|
||||
}
|
||||
|
||||
static inline s32 rfc3315_s14_backoff_update(s32 rt, s32 mrt)
|
||||
{
|
||||
/* multiply 'retransmission timeout' by 1.9 .. 2.1 */
|
||||
u64 tmp = (1900000 + prandom_u32() % 200001) * (u64)rt;
|
||||
do_div(tmp, 1000000);
|
||||
if ((s32)tmp > mrt) {
|
||||
/* multiply 'maximum retransmission time' by 0.9 .. 1.1 */
|
||||
tmp = (900000 + prandom_u32() % 200001) * (u64)mrt;
|
||||
do_div(tmp, 1000000);
|
||||
}
|
||||
return (s32)tmp;
|
||||
}
|
||||
|
||||
#ifdef CONFIG_SYSCTL
|
||||
static int addrconf_sysctl_register(struct inet6_dev *idev);
|
||||
static void addrconf_sysctl_unregister(struct inet6_dev *idev);
|
||||
@@ -187,6 +208,7 @@ static struct ipv6_devconf ipv6_devconf __read_mostly = {
|
||||
.dad_transmits = 1,
|
||||
.rtr_solicits = MAX_RTR_SOLICITATIONS,
|
||||
.rtr_solicit_interval = RTR_SOLICITATION_INTERVAL,
|
||||
.rtr_solicit_max_interval = RTR_SOLICITATION_MAX_INTERVAL,
|
||||
.rtr_solicit_delay = MAX_RTR_SOLICITATION_DELAY,
|
||||
.use_tempaddr = 0,
|
||||
.temp_valid_lft = TEMP_VALID_LIFETIME,
|
||||
@@ -232,6 +254,7 @@ static struct ipv6_devconf ipv6_devconf_dflt __read_mostly = {
|
||||
.dad_transmits = 1,
|
||||
.rtr_solicits = MAX_RTR_SOLICITATIONS,
|
||||
.rtr_solicit_interval = RTR_SOLICITATION_INTERVAL,
|
||||
.rtr_solicit_max_interval = RTR_SOLICITATION_MAX_INTERVAL,
|
||||
.rtr_solicit_delay = MAX_RTR_SOLICITATION_DELAY,
|
||||
.use_tempaddr = 0,
|
||||
.temp_valid_lft = TEMP_VALID_LIFETIME,
|
||||
@@ -3687,7 +3710,7 @@ static void addrconf_rs_timer(unsigned long data)
|
||||
if (idev->if_flags & IF_RA_RCVD)
|
||||
goto out;
|
||||
|
||||
if (idev->rs_probes++ < idev->cnf.rtr_solicits) {
|
||||
if (idev->rs_probes++ < idev->cnf.rtr_solicits || idev->cnf.rtr_solicits < 0) {
|
||||
write_unlock(&idev->lock);
|
||||
if (!ipv6_get_lladdr(dev, &lladdr, IFA_F_TENTATIVE))
|
||||
ndisc_send_rs(dev, &lladdr,
|
||||
@@ -3696,11 +3719,13 @@ static void addrconf_rs_timer(unsigned long data)
|
||||
goto put;
|
||||
|
||||
write_lock(&idev->lock);
|
||||
idev->rs_interval = rfc3315_s14_backoff_update(
|
||||
idev->rs_interval, idev->cnf.rtr_solicit_max_interval);
|
||||
/* The wait after the last probe can be shorter */
|
||||
addrconf_mod_rs_timer(idev, (idev->rs_probes ==
|
||||
idev->cnf.rtr_solicits) ?
|
||||
idev->cnf.rtr_solicit_delay :
|
||||
idev->cnf.rtr_solicit_interval);
|
||||
idev->rs_interval);
|
||||
} else {
|
||||
/*
|
||||
* Note: we do not support deprecated "all on-link"
|
||||
@@ -3949,7 +3974,7 @@ static void addrconf_dad_completed(struct inet6_ifaddr *ifp)
|
||||
send_mld = ifp->scope == IFA_LINK && ipv6_lonely_lladdr(ifp);
|
||||
send_rs = send_mld &&
|
||||
ipv6_accept_ra(ifp->idev) &&
|
||||
ifp->idev->cnf.rtr_solicits > 0 &&
|
||||
ifp->idev->cnf.rtr_solicits != 0 &&
|
||||
(dev->flags&IFF_LOOPBACK) == 0;
|
||||
read_unlock_bh(&ifp->idev->lock);
|
||||
|
||||
@@ -3971,10 +3996,11 @@ static void addrconf_dad_completed(struct inet6_ifaddr *ifp)
|
||||
|
||||
write_lock_bh(&ifp->idev->lock);
|
||||
spin_lock(&ifp->lock);
|
||||
ifp->idev->rs_interval = rfc3315_s14_backoff_init(
|
||||
ifp->idev->cnf.rtr_solicit_interval);
|
||||
ifp->idev->rs_probes = 1;
|
||||
ifp->idev->if_flags |= IF_RS_SENT;
|
||||
addrconf_mod_rs_timer(ifp->idev,
|
||||
ifp->idev->cnf.rtr_solicit_interval);
|
||||
addrconf_mod_rs_timer(ifp->idev, ifp->idev->rs_interval);
|
||||
spin_unlock(&ifp->lock);
|
||||
write_unlock_bh(&ifp->idev->lock);
|
||||
}
|
||||
@@ -4891,6 +4917,8 @@ static inline void ipv6_store_devconf(struct ipv6_devconf *cnf,
|
||||
array[DEVCONF_RTR_SOLICITS] = cnf->rtr_solicits;
|
||||
array[DEVCONF_RTR_SOLICIT_INTERVAL] =
|
||||
jiffies_to_msecs(cnf->rtr_solicit_interval);
|
||||
array[DEVCONF_RTR_SOLICIT_MAX_INTERVAL] =
|
||||
jiffies_to_msecs(cnf->rtr_solicit_max_interval);
|
||||
array[DEVCONF_RTR_SOLICIT_DELAY] =
|
||||
jiffies_to_msecs(cnf->rtr_solicit_delay);
|
||||
array[DEVCONF_FORCE_MLD_VERSION] = cnf->force_mld_version;
|
||||
@@ -4961,18 +4989,18 @@ static inline size_t inet6_if_nlmsg_size(void)
|
||||
}
|
||||
|
||||
static inline void __snmp6_fill_statsdev(u64 *stats, atomic_long_t *mib,
|
||||
int items, int bytes)
|
||||
int bytes)
|
||||
{
|
||||
int i;
|
||||
int pad = bytes - sizeof(u64) * items;
|
||||
int pad = bytes - sizeof(u64) * ICMP6_MIB_MAX;
|
||||
BUG_ON(pad < 0);
|
||||
|
||||
/* Use put_unaligned() because stats may not be aligned for u64. */
|
||||
put_unaligned(items, &stats[0]);
|
||||
for (i = 1; i < items; i++)
|
||||
put_unaligned(ICMP6_MIB_MAX, &stats[0]);
|
||||
for (i = 1; i < ICMP6_MIB_MAX; i++)
|
||||
put_unaligned(atomic_long_read(&mib[i]), &stats[i]);
|
||||
|
||||
memset(&stats[items], 0, pad);
|
||||
memset(&stats[ICMP6_MIB_MAX], 0, pad);
|
||||
}
|
||||
|
||||
static inline void __snmp6_fill_stats64(u64 *stats, void __percpu *mib,
|
||||
@@ -5005,7 +5033,7 @@ static void snmp6_fill_stats(u64 *stats, struct inet6_dev *idev, int attrtype,
|
||||
offsetof(struct ipstats_mib, syncp));
|
||||
break;
|
||||
case IFLA_INET6_ICMP6STATS:
|
||||
__snmp6_fill_statsdev(stats, idev->stats.icmpv6dev->mibs, ICMP6_MIB_MAX, bytes);
|
||||
__snmp6_fill_statsdev(stats, idev->stats.icmpv6dev->mibs, bytes);
|
||||
break;
|
||||
}
|
||||
}
|
||||
@@ -5099,7 +5127,7 @@ static int inet6_set_iftoken(struct inet6_dev *idev, struct in6_addr *token)
|
||||
return -EINVAL;
|
||||
if (!ipv6_accept_ra(idev))
|
||||
return -EINVAL;
|
||||
if (idev->cnf.rtr_solicits <= 0)
|
||||
if (idev->cnf.rtr_solicits == 0)
|
||||
return -EINVAL;
|
||||
|
||||
write_lock_bh(&idev->lock);
|
||||
@@ -5128,8 +5156,10 @@ update_lft:
|
||||
|
||||
if (update_rs) {
|
||||
idev->if_flags |= IF_RS_SENT;
|
||||
idev->rs_interval = rfc3315_s14_backoff_init(
|
||||
idev->cnf.rtr_solicit_interval);
|
||||
idev->rs_probes = 1;
|
||||
addrconf_mod_rs_timer(idev, idev->cnf.rtr_solicit_interval);
|
||||
addrconf_mod_rs_timer(idev, idev->rs_interval);
|
||||
}
|
||||
|
||||
/* Well, that's kinda nasty ... */
|
||||
@@ -5466,20 +5496,6 @@ int addrconf_sysctl_forward(struct ctl_table *ctl, int write,
|
||||
return ret;
|
||||
}
|
||||
|
||||
static
|
||||
int addrconf_sysctl_hop_limit(struct ctl_table *ctl, int write,
|
||||
void __user *buffer, size_t *lenp, loff_t *ppos)
|
||||
{
|
||||
struct ctl_table lctl;
|
||||
int min_hl = 1, max_hl = 255;
|
||||
|
||||
lctl = *ctl;
|
||||
lctl.extra1 = &min_hl;
|
||||
lctl.extra2 = &max_hl;
|
||||
|
||||
return proc_dointvec_minmax(&lctl, write, buffer, lenp, ppos);
|
||||
}
|
||||
|
||||
static
|
||||
int addrconf_sysctl_mtu(struct ctl_table *ctl, int write,
|
||||
void __user *buffer, size_t *lenp, loff_t *ppos)
|
||||
@@ -5713,6 +5729,9 @@ int addrconf_sysctl_ignore_routes_with_linkdown(struct ctl_table *ctl,
|
||||
return ret;
|
||||
}
|
||||
|
||||
static const int one = 1;
|
||||
static const int two_five_five = 255;
|
||||
|
||||
static const struct ctl_table addrconf_sysctl[] = {
|
||||
{
|
||||
.procname = "forwarding",
|
||||
@@ -5726,7 +5745,9 @@ static const struct ctl_table addrconf_sysctl[] = {
|
||||
.data = &ipv6_devconf.hop_limit,
|
||||
.maxlen = sizeof(int),
|
||||
.mode = 0644,
|
||||
.proc_handler = addrconf_sysctl_hop_limit,
|
||||
.proc_handler = proc_dointvec_minmax,
|
||||
.extra1 = (void *)&one,
|
||||
.extra2 = (void *)&two_five_five,
|
||||
},
|
||||
{
|
||||
.procname = "mtu",
|
||||
@@ -5777,6 +5798,13 @@ static const struct ctl_table addrconf_sysctl[] = {
|
||||
.mode = 0644,
|
||||
.proc_handler = proc_dointvec_jiffies,
|
||||
},
|
||||
{
|
||||
.procname = "router_solicitation_max_interval",
|
||||
.data = &ipv6_devconf.rtr_solicit_max_interval,
|
||||
.maxlen = sizeof(int),
|
||||
.mode = 0644,
|
||||
.proc_handler = proc_dointvec_jiffies,
|
||||
},
|
||||
{
|
||||
.procname = "router_solicitation_delay",
|
||||
.data = &ipv6_devconf.rtr_solicit_delay,
|
||||
@@ -6044,8 +6072,14 @@ static int __addrconf_sysctl_register(struct net *net, char *dev_name,
|
||||
|
||||
for (i = 0; table[i].data; i++) {
|
||||
table[i].data += (char *)p - (char *)&ipv6_devconf;
|
||||
table[i].extra1 = idev; /* embedded; no ref */
|
||||
table[i].extra2 = net;
|
||||
/* If one of these is already set, then it is not safe to
|
||||
* overwrite either of them: this makes proc_dointvec_minmax
|
||||
* usable.
|
||||
*/
|
||||
if (!table[i].extra1 && !table[i].extra2) {
|
||||
table[i].extra1 = idev; /* embedded; no ref */
|
||||
table[i].extra2 = net;
|
||||
}
|
||||
}
|
||||
|
||||
snprintf(path, sizeof(path), "net/ipv6/conf/%s", dev_name);
|
||||
|
||||
+1
-2
@@ -648,7 +648,6 @@ static int ip6gre_xmit_other(struct sk_buff *skb, struct net_device *dev)
|
||||
encap_limit = t->parms.encap_limit;
|
||||
|
||||
memcpy(&fl6, &t->fl.u.ip6, sizeof(fl6));
|
||||
fl6.flowi6_proto = skb->protocol;
|
||||
|
||||
err = gre_handle_offloads(skb, !!(t->parms.o_flags & TUNNEL_CSUM));
|
||||
if (err)
|
||||
@@ -1239,7 +1238,7 @@ static void ip6gre_netlink_parms(struct nlattr *data[],
|
||||
parms->encap_limit = nla_get_u8(data[IFLA_GRE_ENCAP_LIMIT]);
|
||||
|
||||
if (data[IFLA_GRE_FLOWINFO])
|
||||
parms->flowinfo = nla_get_u32(data[IFLA_GRE_FLOWINFO]);
|
||||
parms->flowinfo = nla_get_be32(data[IFLA_GRE_FLOWINFO]);
|
||||
|
||||
if (data[IFLA_GRE_FLAGS])
|
||||
parms->flags = nla_get_u32(data[IFLA_GRE_FLAGS]);
|
||||
|
||||
@@ -69,6 +69,7 @@ static struct sk_buff *ipv6_gso_segment(struct sk_buff *skb,
|
||||
int offset = 0;
|
||||
bool encap, udpfrag;
|
||||
int nhoff;
|
||||
bool gso_partial;
|
||||
|
||||
skb_reset_network_header(skb);
|
||||
nhoff = skb_network_header(skb) - skb_mac_header(skb);
|
||||
@@ -101,9 +102,11 @@ static struct sk_buff *ipv6_gso_segment(struct sk_buff *skb,
|
||||
if (IS_ERR(segs))
|
||||
goto out;
|
||||
|
||||
gso_partial = !!(skb_shinfo(segs)->gso_type & SKB_GSO_PARTIAL);
|
||||
|
||||
for (skb = segs; skb; skb = skb->next) {
|
||||
ipv6h = (struct ipv6hdr *)(skb_mac_header(skb) + nhoff);
|
||||
if (skb_is_gso(skb))
|
||||
if (gso_partial)
|
||||
payload_len = skb_shinfo(skb)->gso_size +
|
||||
SKB_GSO_CB(skb)->data_offset +
|
||||
skb->head - (unsigned char *)(ipv6h + 1);
|
||||
|
||||
+15
-4
@@ -321,11 +321,9 @@ static int vti6_rcv(struct sk_buff *skb)
|
||||
goto discard;
|
||||
}
|
||||
|
||||
XFRM_TUNNEL_SKB_CB(skb)->tunnel.ip6 = t;
|
||||
|
||||
rcu_read_unlock();
|
||||
|
||||
return xfrm6_rcv(skb);
|
||||
return xfrm6_rcv_tnl(skb, t);
|
||||
}
|
||||
rcu_read_unlock();
|
||||
return -EINVAL;
|
||||
@@ -340,6 +338,7 @@ static int vti6_rcv_cb(struct sk_buff *skb, int err)
|
||||
struct net_device *dev;
|
||||
struct pcpu_sw_netstats *tstats;
|
||||
struct xfrm_state *x;
|
||||
struct xfrm_mode *inner_mode;
|
||||
struct ip6_tnl *t = XFRM_TUNNEL_SKB_CB(skb)->tunnel.ip6;
|
||||
u32 orig_mark = skb->mark;
|
||||
int ret;
|
||||
@@ -357,7 +356,19 @@ static int vti6_rcv_cb(struct sk_buff *skb, int err)
|
||||
}
|
||||
|
||||
x = xfrm_input_state(skb);
|
||||
family = x->inner_mode->afinfo->family;
|
||||
|
||||
inner_mode = x->inner_mode;
|
||||
|
||||
if (x->sel.family == AF_UNSPEC) {
|
||||
inner_mode = xfrm_ip2inner_mode(x, XFRM_MODE_SKB_CB(skb)->protocol);
|
||||
if (inner_mode == NULL) {
|
||||
XFRM_INC_STATS(dev_net(skb->dev),
|
||||
LINUX_MIB_XFRMINSTATEMODEERROR);
|
||||
return -EINVAL;
|
||||
}
|
||||
}
|
||||
|
||||
family = inner_mode->afinfo->family;
|
||||
|
||||
skb->mark = be32_to_cpu(t->parms.i_key);
|
||||
ret = xfrm_policy_check(NULL, XFRM_POLICY_IN, skb, family);
|
||||
|
||||
+8
-4
@@ -2239,6 +2239,7 @@ static int __ip6mr_fill_mroute(struct mr6_table *mrt, struct sk_buff *skb,
|
||||
struct rta_mfc_stats mfcs;
|
||||
struct nlattr *mp_attr;
|
||||
struct rtnexthop *nhp;
|
||||
unsigned long lastuse;
|
||||
int ct;
|
||||
|
||||
/* If cache is unresolved, don't try to parse IIF and OIF */
|
||||
@@ -2269,12 +2270,14 @@ static int __ip6mr_fill_mroute(struct mr6_table *mrt, struct sk_buff *skb,
|
||||
|
||||
nla_nest_end(skb, mp_attr);
|
||||
|
||||
lastuse = READ_ONCE(c->mfc_un.res.lastuse);
|
||||
lastuse = time_after_eq(jiffies, lastuse) ? jiffies - lastuse : 0;
|
||||
|
||||
mfcs.mfcs_packets = c->mfc_un.res.pkt;
|
||||
mfcs.mfcs_bytes = c->mfc_un.res.bytes;
|
||||
mfcs.mfcs_wrong_if = c->mfc_un.res.wrong_if;
|
||||
if (nla_put_64bit(skb, RTA_MFC_STATS, sizeof(mfcs), &mfcs, RTA_PAD) ||
|
||||
nla_put_u64_64bit(skb, RTA_EXPIRES,
|
||||
jiffies_to_clock_t(c->mfc_un.res.lastuse),
|
||||
nla_put_u64_64bit(skb, RTA_EXPIRES, jiffies_to_clock_t(lastuse),
|
||||
RTA_PAD))
|
||||
return -EMSGSIZE;
|
||||
|
||||
@@ -2282,8 +2285,8 @@ static int __ip6mr_fill_mroute(struct mr6_table *mrt, struct sk_buff *skb,
|
||||
return 1;
|
||||
}
|
||||
|
||||
int ip6mr_get_route(struct net *net,
|
||||
struct sk_buff *skb, struct rtmsg *rtm, int nowait)
|
||||
int ip6mr_get_route(struct net *net, struct sk_buff *skb, struct rtmsg *rtm,
|
||||
int nowait, u32 portid)
|
||||
{
|
||||
int err;
|
||||
struct mr6_table *mrt;
|
||||
@@ -2328,6 +2331,7 @@ int ip6mr_get_route(struct net *net,
|
||||
return -ENOMEM;
|
||||
}
|
||||
|
||||
NETLINK_CB(skb2).portid = portid;
|
||||
skb_reset_transport_header(skb2);
|
||||
|
||||
skb_put(skb2, sizeof(struct ipv6hdr));
|
||||
|
||||
@@ -190,7 +190,7 @@ static struct nf_loginfo trace_loginfo = {
|
||||
.u = {
|
||||
.log = {
|
||||
.level = LOGLEVEL_WARNING,
|
||||
.logflags = NF_LOG_MASK,
|
||||
.logflags = NF_LOG_DEFAULT_MASK,
|
||||
},
|
||||
},
|
||||
};
|
||||
|
||||
@@ -115,7 +115,7 @@ static unsigned int ipv6_helper(void *priv,
|
||||
help = nfct_help(ct);
|
||||
if (!help)
|
||||
return NF_ACCEPT;
|
||||
/* rcu_read_lock()ed by nf_hook_slow */
|
||||
/* rcu_read_lock()ed by nf_hook_thresh */
|
||||
helper = rcu_dereference(help->helper);
|
||||
if (!helper)
|
||||
return NF_ACCEPT;
|
||||
|
||||
@@ -165,7 +165,7 @@ icmpv6_error_message(struct net *net, struct nf_conn *tmpl,
|
||||
return -NF_ACCEPT;
|
||||
}
|
||||
|
||||
/* rcu_read_lock()ed by nf_hook_slow */
|
||||
/* rcu_read_lock()ed by nf_hook_thresh */
|
||||
inproto = __nf_ct_l4proto_find(PF_INET6, origtuple.dst.protonum);
|
||||
|
||||
/* Ordinarily, we'd expect the inverted tupleproto, but it's
|
||||
|
||||
@@ -30,7 +30,7 @@ static struct nf_loginfo default_loginfo = {
|
||||
.u = {
|
||||
.log = {
|
||||
.level = LOGLEVEL_NOTICE,
|
||||
.logflags = NF_LOG_MASK,
|
||||
.logflags = NF_LOG_DEFAULT_MASK,
|
||||
},
|
||||
},
|
||||
};
|
||||
@@ -52,7 +52,7 @@ static void dump_ipv6_packet(struct nf_log_buf *m,
|
||||
if (info->type == NF_LOG_TYPE_LOG)
|
||||
logflags = info->u.log.logflags;
|
||||
else
|
||||
logflags = NF_LOG_MASK;
|
||||
logflags = NF_LOG_DEFAULT_MASK;
|
||||
|
||||
ih = skb_header_pointer(skb, ip6hoff, sizeof(_ip6h), &_ip6h);
|
||||
if (ih == NULL) {
|
||||
@@ -84,7 +84,7 @@ static void dump_ipv6_packet(struct nf_log_buf *m,
|
||||
}
|
||||
|
||||
/* Max length: 48 "OPT (...) " */
|
||||
if (logflags & XT_LOG_IPOPT)
|
||||
if (logflags & NF_LOG_IPOPT)
|
||||
nf_log_buf_add(m, "OPT ( ");
|
||||
|
||||
switch (currenthdr) {
|
||||
@@ -121,7 +121,7 @@ static void dump_ipv6_packet(struct nf_log_buf *m,
|
||||
case IPPROTO_ROUTING:
|
||||
case IPPROTO_HOPOPTS:
|
||||
if (fragment) {
|
||||
if (logflags & XT_LOG_IPOPT)
|
||||
if (logflags & NF_LOG_IPOPT)
|
||||
nf_log_buf_add(m, ")");
|
||||
return;
|
||||
}
|
||||
@@ -129,7 +129,7 @@ static void dump_ipv6_packet(struct nf_log_buf *m,
|
||||
break;
|
||||
/* Max Length */
|
||||
case IPPROTO_AH:
|
||||
if (logflags & XT_LOG_IPOPT) {
|
||||
if (logflags & NF_LOG_IPOPT) {
|
||||
struct ip_auth_hdr _ahdr;
|
||||
const struct ip_auth_hdr *ah;
|
||||
|
||||
@@ -161,7 +161,7 @@ static void dump_ipv6_packet(struct nf_log_buf *m,
|
||||
hdrlen = (hp->hdrlen+2)<<2;
|
||||
break;
|
||||
case IPPROTO_ESP:
|
||||
if (logflags & XT_LOG_IPOPT) {
|
||||
if (logflags & NF_LOG_IPOPT) {
|
||||
struct ip_esp_hdr _esph;
|
||||
const struct ip_esp_hdr *eh;
|
||||
|
||||
@@ -194,7 +194,7 @@ static void dump_ipv6_packet(struct nf_log_buf *m,
|
||||
nf_log_buf_add(m, "Unknown Ext Hdr %u", currenthdr);
|
||||
return;
|
||||
}
|
||||
if (logflags & XT_LOG_IPOPT)
|
||||
if (logflags & NF_LOG_IPOPT)
|
||||
nf_log_buf_add(m, ") ");
|
||||
|
||||
currenthdr = hp->nexthdr;
|
||||
@@ -277,7 +277,7 @@ static void dump_ipv6_packet(struct nf_log_buf *m,
|
||||
}
|
||||
|
||||
/* Max length: 15 "UID=4294967295 " */
|
||||
if ((logflags & XT_LOG_UID) && recurse)
|
||||
if ((logflags & NF_LOG_UID) && recurse)
|
||||
nf_log_dump_sk_uid_gid(m, skb->sk);
|
||||
|
||||
/* Max length: 16 "MARK=0xFFFFFFFF " */
|
||||
@@ -295,7 +295,7 @@ static void dump_ipv6_mac_header(struct nf_log_buf *m,
|
||||
if (info->type == NF_LOG_TYPE_LOG)
|
||||
logflags = info->u.log.logflags;
|
||||
|
||||
if (!(logflags & XT_LOG_MACDECODE))
|
||||
if (!(logflags & NF_LOG_MACDECODE))
|
||||
goto fallback;
|
||||
|
||||
switch (dev->type) {
|
||||
|
||||
@@ -22,9 +22,7 @@ static unsigned int nft_do_chain_ipv6(void *priv,
|
||||
{
|
||||
struct nft_pktinfo pkt;
|
||||
|
||||
/* malformed packet, drop it */
|
||||
if (nft_set_pktinfo_ipv6(&pkt, skb, state) < 0)
|
||||
return NF_DROP;
|
||||
nft_set_pktinfo_ipv6(&pkt, skb, state);
|
||||
|
||||
return nft_do_chain(&pkt, priv);
|
||||
}
|
||||
@@ -102,7 +100,10 @@ static int __init nf_tables_ipv6_init(void)
|
||||
{
|
||||
int ret;
|
||||
|
||||
nft_register_chain_type(&filter_ipv6);
|
||||
ret = nft_register_chain_type(&filter_ipv6);
|
||||
if (ret < 0)
|
||||
return ret;
|
||||
|
||||
ret = register_pernet_subsys(&nf_tables_ipv6_net_ops);
|
||||
if (ret < 0)
|
||||
nft_unregister_chain_type(&filter_ipv6);
|
||||
|
||||
@@ -31,10 +31,9 @@ static unsigned int nf_route_table_hook(void *priv,
|
||||
struct in6_addr saddr, daddr;
|
||||
u_int8_t hop_limit;
|
||||
u32 mark, flowlabel;
|
||||
int err;
|
||||
|
||||
/* malformed packet, drop it */
|
||||
if (nft_set_pktinfo_ipv6(&pkt, skb, state) < 0)
|
||||
return NF_DROP;
|
||||
nft_set_pktinfo_ipv6(&pkt, skb, state);
|
||||
|
||||
/* save source/dest address, mark, hoplimit, flowlabel, priority */
|
||||
memcpy(&saddr, &ipv6_hdr(skb)->saddr, sizeof(saddr));
|
||||
@@ -46,13 +45,16 @@ static unsigned int nf_route_table_hook(void *priv,
|
||||
flowlabel = *((u32 *)ipv6_hdr(skb));
|
||||
|
||||
ret = nft_do_chain(&pkt, priv);
|
||||
if (ret != NF_DROP && ret != NF_QUEUE &&
|
||||
if (ret != NF_DROP && ret != NF_STOLEN &&
|
||||
(memcmp(&ipv6_hdr(skb)->saddr, &saddr, sizeof(saddr)) ||
|
||||
memcmp(&ipv6_hdr(skb)->daddr, &daddr, sizeof(daddr)) ||
|
||||
skb->mark != mark ||
|
||||
ipv6_hdr(skb)->hop_limit != hop_limit ||
|
||||
flowlabel != *((u_int32_t *)ipv6_hdr(skb))))
|
||||
return ip6_route_me_harder(state->net, skb) == 0 ? ret : NF_DROP;
|
||||
flowlabel != *((u_int32_t *)ipv6_hdr(skb)))) {
|
||||
err = ip6_route_me_harder(state->net, skb);
|
||||
if (err < 0)
|
||||
ret = NF_DROP_ERR(err);
|
||||
}
|
||||
|
||||
return ret;
|
||||
}
|
||||
|
||||
+22
-8
@@ -30,6 +30,11 @@
|
||||
#include <net/transp_v6.h>
|
||||
#include <net/ipv6.h>
|
||||
|
||||
#define MAX4(a, b, c, d) \
|
||||
max_t(u32, max_t(u32, a, b), max_t(u32, c, d))
|
||||
#define SNMP_MIB_MAX MAX4(UDP_MIB_MAX, TCP_MIB_MAX, \
|
||||
IPSTATS_MIB_MAX, ICMP_MIB_MAX)
|
||||
|
||||
static int sockstat6_seq_show(struct seq_file *seq, void *v)
|
||||
{
|
||||
struct net *net = seq->private;
|
||||
@@ -191,25 +196,34 @@ static void snmp6_seq_show_item(struct seq_file *seq, void __percpu *pcpumib,
|
||||
atomic_long_t *smib,
|
||||
const struct snmp_mib *itemlist)
|
||||
{
|
||||
unsigned long buff[SNMP_MIB_MAX];
|
||||
int i;
|
||||
unsigned long val;
|
||||
|
||||
for (i = 0; itemlist[i].name; i++) {
|
||||
val = pcpumib ?
|
||||
snmp_fold_field(pcpumib, itemlist[i].entry) :
|
||||
atomic_long_read(smib + itemlist[i].entry);
|
||||
seq_printf(seq, "%-32s\t%lu\n", itemlist[i].name, val);
|
||||
if (pcpumib) {
|
||||
memset(buff, 0, sizeof(unsigned long) * SNMP_MIB_MAX);
|
||||
|
||||
snmp_get_cpu_field_batch(buff, itemlist, pcpumib);
|
||||
for (i = 0; itemlist[i].name; i++)
|
||||
seq_printf(seq, "%-32s\t%lu\n",
|
||||
itemlist[i].name, buff[i]);
|
||||
} else {
|
||||
for (i = 0; itemlist[i].name; i++)
|
||||
seq_printf(seq, "%-32s\t%lu\n", itemlist[i].name,
|
||||
atomic_long_read(smib + itemlist[i].entry));
|
||||
}
|
||||
}
|
||||
|
||||
static void snmp6_seq_show_item64(struct seq_file *seq, void __percpu *mib,
|
||||
const struct snmp_mib *itemlist, size_t syncpoff)
|
||||
{
|
||||
u64 buff64[SNMP_MIB_MAX];
|
||||
int i;
|
||||
|
||||
memset(buff64, 0, sizeof(unsigned long) * SNMP_MIB_MAX);
|
||||
|
||||
snmp_get_cpu_field64_batch(buff64, itemlist, mib, syncpoff);
|
||||
for (i = 0; itemlist[i].name; i++)
|
||||
seq_printf(seq, "%-32s\t%llu\n", itemlist[i].name,
|
||||
snmp_fold_field64(mib, itemlist[i].entry, syncpoff));
|
||||
seq_printf(seq, "%-32s\t%llu\n", itemlist[i].name, buff64[i]);
|
||||
}
|
||||
|
||||
static int snmp6_seq_show(struct seq_file *seq, void *v)
|
||||
|
||||
+17
-5
@@ -1147,15 +1147,16 @@ static struct rt6_info *ip6_pol_route_input(struct net *net, struct fib6_table *
|
||||
return ip6_pol_route(net, table, fl6->flowi6_iif, fl6, flags);
|
||||
}
|
||||
|
||||
static struct dst_entry *ip6_route_input_lookup(struct net *net,
|
||||
struct net_device *dev,
|
||||
struct flowi6 *fl6, int flags)
|
||||
struct dst_entry *ip6_route_input_lookup(struct net *net,
|
||||
struct net_device *dev,
|
||||
struct flowi6 *fl6, int flags)
|
||||
{
|
||||
if (rt6_need_strict(&fl6->daddr) && dev->type != ARPHRD_PIMREG)
|
||||
flags |= RT6_LOOKUP_F_IFACE;
|
||||
|
||||
return fib6_rule_lookup(net, fl6, flags, ip6_pol_route_input);
|
||||
}
|
||||
EXPORT_SYMBOL_GPL(ip6_route_input_lookup);
|
||||
|
||||
void ip6_route_input(struct sk_buff *skb)
|
||||
{
|
||||
@@ -1991,9 +1992,18 @@ static struct rt6_info *ip6_route_info_create(struct fib6_config *cfg)
|
||||
if (!(gwa_type & IPV6_ADDR_UNICAST))
|
||||
goto out;
|
||||
|
||||
if (cfg->fc_table)
|
||||
if (cfg->fc_table) {
|
||||
grt = ip6_nh_lookup_table(net, cfg, gw_addr);
|
||||
|
||||
if (grt) {
|
||||
if (grt->rt6i_flags & RTF_GATEWAY ||
|
||||
(dev && dev != grt->dst.dev)) {
|
||||
ip6_rt_put(grt);
|
||||
grt = NULL;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (!grt)
|
||||
grt = rt6_lookup(net, gw_addr, NULL,
|
||||
cfg->fc_ifindex, 1);
|
||||
@@ -3206,7 +3216,9 @@ static int rt6_fill_node(struct net *net,
|
||||
if (iif) {
|
||||
#ifdef CONFIG_IPV6_MROUTE
|
||||
if (ipv6_addr_is_multicast(&rt->rt6i_dst.addr)) {
|
||||
int err = ip6mr_get_route(net, skb, rtm, nowait);
|
||||
int err = ip6mr_get_route(net, skb, rtm, nowait,
|
||||
portid);
|
||||
|
||||
if (err <= 0) {
|
||||
if (!nowait) {
|
||||
if (err == 0)
|
||||
|
||||
+11
-5
@@ -21,9 +21,10 @@ int xfrm6_extract_input(struct xfrm_state *x, struct sk_buff *skb)
|
||||
return xfrm6_extract_header(skb);
|
||||
}
|
||||
|
||||
int xfrm6_rcv_spi(struct sk_buff *skb, int nexthdr, __be32 spi)
|
||||
int xfrm6_rcv_spi(struct sk_buff *skb, int nexthdr, __be32 spi,
|
||||
struct ip6_tnl *t)
|
||||
{
|
||||
XFRM_TUNNEL_SKB_CB(skb)->tunnel.ip6 = NULL;
|
||||
XFRM_TUNNEL_SKB_CB(skb)->tunnel.ip6 = t;
|
||||
XFRM_SPI_SKB_CB(skb)->family = AF_INET6;
|
||||
XFRM_SPI_SKB_CB(skb)->daddroff = offsetof(struct ipv6hdr, daddr);
|
||||
return xfrm_input(skb, nexthdr, spi, 0);
|
||||
@@ -49,13 +50,18 @@ int xfrm6_transport_finish(struct sk_buff *skb, int async)
|
||||
return -1;
|
||||
}
|
||||
|
||||
int xfrm6_rcv(struct sk_buff *skb)
|
||||
int xfrm6_rcv_tnl(struct sk_buff *skb, struct ip6_tnl *t)
|
||||
{
|
||||
return xfrm6_rcv_spi(skb, skb_network_header(skb)[IP6CB(skb)->nhoff],
|
||||
0);
|
||||
0, t);
|
||||
}
|
||||
EXPORT_SYMBOL(xfrm6_rcv_tnl);
|
||||
|
||||
int xfrm6_rcv(struct sk_buff *skb)
|
||||
{
|
||||
return xfrm6_rcv_tnl(skb, NULL);
|
||||
}
|
||||
EXPORT_SYMBOL(xfrm6_rcv);
|
||||
|
||||
int xfrm6_input_addr(struct sk_buff *skb, xfrm_address_t *daddr,
|
||||
xfrm_address_t *saddr, u8 proto)
|
||||
{
|
||||
|
||||
@@ -236,7 +236,7 @@ static int xfrm6_tunnel_rcv(struct sk_buff *skb)
|
||||
__be32 spi;
|
||||
|
||||
spi = xfrm6_tunnel_spi_lookup(net, (const xfrm_address_t *)&iph->saddr);
|
||||
return xfrm6_rcv_spi(skb, IPPROTO_IPV6, spi);
|
||||
return xfrm6_rcv_spi(skb, IPPROTO_IPV6, spi, NULL);
|
||||
}
|
||||
|
||||
static int xfrm6_tunnel_err(struct sk_buff *skb, struct inet6_skb_parm *opt,
|
||||
|
||||
+2
-3
@@ -832,7 +832,7 @@ static int irda_accept(struct socket *sock, struct socket *newsock, int flags)
|
||||
struct sock *sk = sock->sk;
|
||||
struct irda_sock *new, *self = irda_sk(sk);
|
||||
struct sock *newsk;
|
||||
struct sk_buff *skb;
|
||||
struct sk_buff *skb = NULL;
|
||||
int err;
|
||||
|
||||
err = irda_create(sock_net(sk), newsock, sk->sk_protocol, 0);
|
||||
@@ -897,7 +897,6 @@ static int irda_accept(struct socket *sock, struct socket *newsock, int flags)
|
||||
err = -EPERM; /* value does not seem to make sense. -arnd */
|
||||
if (!new->tsap) {
|
||||
pr_debug("%s(), dup failed!\n", __func__);
|
||||
kfree_skb(skb);
|
||||
goto out;
|
||||
}
|
||||
|
||||
@@ -916,7 +915,6 @@ static int irda_accept(struct socket *sock, struct socket *newsock, int flags)
|
||||
/* Clean up the original one to keep it in listen state */
|
||||
irttp_listen(self->tsap);
|
||||
|
||||
kfree_skb(skb);
|
||||
sk->sk_ack_backlog--;
|
||||
|
||||
newsock->state = SS_CONNECTED;
|
||||
@@ -924,6 +922,7 @@ static int irda_accept(struct socket *sock, struct socket *newsock, int flags)
|
||||
irda_connect_response(new);
|
||||
err = 0;
|
||||
out:
|
||||
kfree_skb(skb);
|
||||
release_sock(sk);
|
||||
return err;
|
||||
}
|
||||
|
||||
@@ -261,10 +261,16 @@ void __ieee80211_start_rx_ba_session(struct sta_info *sta,
|
||||
.timeout = timeout,
|
||||
.ssn = start_seq_num,
|
||||
};
|
||||
|
||||
int i, ret = -EOPNOTSUPP;
|
||||
u16 status = WLAN_STATUS_REQUEST_DECLINED;
|
||||
|
||||
if (tid >= IEEE80211_FIRST_TSPEC_TSID) {
|
||||
ht_dbg(sta->sdata,
|
||||
"STA %pM requests BA session on unsupported tid %d\n",
|
||||
sta->sta.addr, tid);
|
||||
goto end_no_lock;
|
||||
}
|
||||
|
||||
if (!sta->sta.ht_cap.ht_supported) {
|
||||
ht_dbg(sta->sdata,
|
||||
"STA %pM erroneously requests BA session on tid %d w/o QoS\n",
|
||||
|
||||
@@ -584,6 +584,9 @@ int ieee80211_start_tx_ba_session(struct ieee80211_sta *pubsta, u16 tid,
|
||||
ieee80211_hw_check(&local->hw, TX_AMPDU_SETUP_IN_HW))
|
||||
return -EINVAL;
|
||||
|
||||
if (WARN_ON(tid >= IEEE80211_FIRST_TSPEC_TSID))
|
||||
return -EINVAL;
|
||||
|
||||
ht_dbg(sdata, "Open BA session requested for %pM tid %u\n",
|
||||
pubsta->addr, tid);
|
||||
|
||||
|
||||
@@ -1232,7 +1232,7 @@ struct ieee80211_local {
|
||||
spinlock_t tim_lock;
|
||||
unsigned long num_sta;
|
||||
struct list_head sta_list;
|
||||
struct rhashtable sta_hash;
|
||||
struct rhltable sta_hash;
|
||||
struct timer_list sta_cleanup;
|
||||
int sta_generation;
|
||||
|
||||
|
||||
@@ -757,6 +757,7 @@ static void hwmp_perr_frame_process(struct ieee80211_sub_if_data *sdata,
|
||||
sta = next_hop_deref_protected(mpath);
|
||||
if (mpath->flags & MESH_PATH_ACTIVE &&
|
||||
ether_addr_equal(ta, sta->sta.addr) &&
|
||||
!(mpath->flags & MESH_PATH_FIXED) &&
|
||||
(!(mpath->flags & MESH_PATH_SN_VALID) ||
|
||||
SN_GT(target_sn, mpath->sn) || target_sn == 0)) {
|
||||
mpath->flags &= ~MESH_PATH_ACTIVE;
|
||||
@@ -1023,7 +1024,7 @@ void mesh_path_start_discovery(struct ieee80211_sub_if_data *sdata)
|
||||
goto enddiscovery;
|
||||
|
||||
spin_lock_bh(&mpath->state_lock);
|
||||
if (mpath->flags & MESH_PATH_DELETED) {
|
||||
if (mpath->flags & (MESH_PATH_DELETED | MESH_PATH_FIXED)) {
|
||||
spin_unlock_bh(&mpath->state_lock);
|
||||
goto enddiscovery;
|
||||
}
|
||||
|
||||
@@ -826,7 +826,7 @@ void mesh_path_fix_nexthop(struct mesh_path *mpath, struct sta_info *next_hop)
|
||||
mpath->metric = 0;
|
||||
mpath->hop_count = 0;
|
||||
mpath->exp_time = 0;
|
||||
mpath->flags |= MESH_PATH_FIXED;
|
||||
mpath->flags = MESH_PATH_FIXED | MESH_PATH_SN_VALID;
|
||||
mesh_path_activate(mpath);
|
||||
spin_unlock_bh(&mpath->state_lock);
|
||||
mesh_path_tx_pending(mpath);
|
||||
|
||||
+2
-5
@@ -4004,7 +4004,7 @@ static void __ieee80211_rx_handle_packet(struct ieee80211_hw *hw,
|
||||
__le16 fc;
|
||||
struct ieee80211_rx_data rx;
|
||||
struct ieee80211_sub_if_data *prev;
|
||||
struct rhash_head *tmp;
|
||||
struct rhlist_head *tmp;
|
||||
int err = 0;
|
||||
|
||||
fc = ((struct ieee80211_hdr *)skb->data)->frame_control;
|
||||
@@ -4047,13 +4047,10 @@ static void __ieee80211_rx_handle_packet(struct ieee80211_hw *hw,
|
||||
goto out;
|
||||
} else if (ieee80211_is_data(fc)) {
|
||||
struct sta_info *sta, *prev_sta;
|
||||
const struct bucket_table *tbl;
|
||||
|
||||
prev_sta = NULL;
|
||||
|
||||
tbl = rht_dereference_rcu(local->sta_hash.tbl, &local->sta_hash);
|
||||
|
||||
for_each_sta_info(local, tbl, hdr->addr2, sta, tmp) {
|
||||
for_each_sta_info(local, hdr->addr2, sta, tmp) {
|
||||
if (!prev_sta) {
|
||||
prev_sta = sta;
|
||||
continue;
|
||||
|
||||
+23
-33
@@ -67,12 +67,10 @@
|
||||
|
||||
static const struct rhashtable_params sta_rht_params = {
|
||||
.nelem_hint = 3, /* start small */
|
||||
.insecure_elasticity = true, /* Disable chain-length checks. */
|
||||
.automatic_shrinking = true,
|
||||
.head_offset = offsetof(struct sta_info, hash_node),
|
||||
.key_offset = offsetof(struct sta_info, addr),
|
||||
.key_len = ETH_ALEN,
|
||||
.hashfn = sta_addr_hash,
|
||||
.max_size = CONFIG_MAC80211_STA_HASH_MAX_SIZE,
|
||||
};
|
||||
|
||||
@@ -80,8 +78,8 @@ static const struct rhashtable_params sta_rht_params = {
|
||||
static int sta_info_hash_del(struct ieee80211_local *local,
|
||||
struct sta_info *sta)
|
||||
{
|
||||
return rhashtable_remove_fast(&local->sta_hash, &sta->hash_node,
|
||||
sta_rht_params);
|
||||
return rhltable_remove(&local->sta_hash, &sta->hash_node,
|
||||
sta_rht_params);
|
||||
}
|
||||
|
||||
static void __cleanup_single_sta(struct sta_info *sta)
|
||||
@@ -157,19 +155,22 @@ static void cleanup_single_sta(struct sta_info *sta)
|
||||
sta_info_free(local, sta);
|
||||
}
|
||||
|
||||
struct rhlist_head *sta_info_hash_lookup(struct ieee80211_local *local,
|
||||
const u8 *addr)
|
||||
{
|
||||
return rhltable_lookup(&local->sta_hash, addr, sta_rht_params);
|
||||
}
|
||||
|
||||
/* protected by RCU */
|
||||
struct sta_info *sta_info_get(struct ieee80211_sub_if_data *sdata,
|
||||
const u8 *addr)
|
||||
{
|
||||
struct ieee80211_local *local = sdata->local;
|
||||
struct rhlist_head *tmp;
|
||||
struct sta_info *sta;
|
||||
struct rhash_head *tmp;
|
||||
const struct bucket_table *tbl;
|
||||
|
||||
rcu_read_lock();
|
||||
tbl = rht_dereference_rcu(local->sta_hash.tbl, &local->sta_hash);
|
||||
|
||||
for_each_sta_info(local, tbl, addr, sta, tmp) {
|
||||
for_each_sta_info(local, addr, sta, tmp) {
|
||||
if (sta->sdata == sdata) {
|
||||
rcu_read_unlock();
|
||||
/* this is safe as the caller must already hold
|
||||
@@ -190,14 +191,11 @@ struct sta_info *sta_info_get_bss(struct ieee80211_sub_if_data *sdata,
|
||||
const u8 *addr)
|
||||
{
|
||||
struct ieee80211_local *local = sdata->local;
|
||||
struct rhlist_head *tmp;
|
||||
struct sta_info *sta;
|
||||
struct rhash_head *tmp;
|
||||
const struct bucket_table *tbl;
|
||||
|
||||
rcu_read_lock();
|
||||
tbl = rht_dereference_rcu(local->sta_hash.tbl, &local->sta_hash);
|
||||
|
||||
for_each_sta_info(local, tbl, addr, sta, tmp) {
|
||||
for_each_sta_info(local, addr, sta, tmp) {
|
||||
if (sta->sdata == sdata ||
|
||||
(sta->sdata->bss && sta->sdata->bss == sdata->bss)) {
|
||||
rcu_read_unlock();
|
||||
@@ -263,8 +261,8 @@ void sta_info_free(struct ieee80211_local *local, struct sta_info *sta)
|
||||
static int sta_info_hash_add(struct ieee80211_local *local,
|
||||
struct sta_info *sta)
|
||||
{
|
||||
return rhashtable_insert_fast(&local->sta_hash, &sta->hash_node,
|
||||
sta_rht_params);
|
||||
return rhltable_insert(&local->sta_hash, &sta->hash_node,
|
||||
sta_rht_params);
|
||||
}
|
||||
|
||||
static void sta_deliver_ps_frames(struct work_struct *wk)
|
||||
@@ -453,9 +451,9 @@ static int sta_info_insert_check(struct sta_info *sta)
|
||||
is_multicast_ether_addr(sta->sta.addr)))
|
||||
return -EINVAL;
|
||||
|
||||
/* Strictly speaking this isn't necessary as we hold the mutex, but
|
||||
* the rhashtable code can't really deal with that distinction. We
|
||||
* do require the mutex for correctness though.
|
||||
/* The RCU read lock is required by rhashtable due to
|
||||
* asynchronous resize/rehash. We also require the mutex
|
||||
* for correctness.
|
||||
*/
|
||||
rcu_read_lock();
|
||||
lockdep_assert_held(&sdata->local->sta_mtx);
|
||||
@@ -1043,16 +1041,11 @@ static void sta_info_cleanup(unsigned long data)
|
||||
round_jiffies(jiffies + STA_INFO_CLEANUP_INTERVAL));
|
||||
}
|
||||
|
||||
u32 sta_addr_hash(const void *key, u32 length, u32 seed)
|
||||
{
|
||||
return jhash(key, ETH_ALEN, seed);
|
||||
}
|
||||
|
||||
int sta_info_init(struct ieee80211_local *local)
|
||||
{
|
||||
int err;
|
||||
|
||||
err = rhashtable_init(&local->sta_hash, &sta_rht_params);
|
||||
err = rhltable_init(&local->sta_hash, &sta_rht_params);
|
||||
if (err)
|
||||
return err;
|
||||
|
||||
@@ -1068,7 +1061,7 @@ int sta_info_init(struct ieee80211_local *local)
|
||||
void sta_info_stop(struct ieee80211_local *local)
|
||||
{
|
||||
del_timer_sync(&local->sta_cleanup);
|
||||
rhashtable_destroy(&local->sta_hash);
|
||||
rhltable_destroy(&local->sta_hash);
|
||||
}
|
||||
|
||||
|
||||
@@ -1138,17 +1131,14 @@ struct ieee80211_sta *ieee80211_find_sta_by_ifaddr(struct ieee80211_hw *hw,
|
||||
const u8 *localaddr)
|
||||
{
|
||||
struct ieee80211_local *local = hw_to_local(hw);
|
||||
struct rhlist_head *tmp;
|
||||
struct sta_info *sta;
|
||||
struct rhash_head *tmp;
|
||||
const struct bucket_table *tbl;
|
||||
|
||||
tbl = rht_dereference_rcu(local->sta_hash.tbl, &local->sta_hash);
|
||||
|
||||
/*
|
||||
* Just return a random station if localaddr is NULL
|
||||
* ... first in list.
|
||||
*/
|
||||
for_each_sta_info(local, tbl, addr, sta, tmp) {
|
||||
for_each_sta_info(local, addr, sta, tmp) {
|
||||
if (localaddr &&
|
||||
!ether_addr_equal(sta->sdata->vif.addr, localaddr))
|
||||
continue;
|
||||
@@ -1617,7 +1607,6 @@ ieee80211_sta_ps_deliver_response(struct sta_info *sta,
|
||||
|
||||
sta_info_recalc_tim(sta);
|
||||
} else {
|
||||
unsigned long tids = sta->txq_buffered_tids & driver_release_tids;
|
||||
int tid;
|
||||
|
||||
/*
|
||||
@@ -1647,7 +1636,8 @@ ieee80211_sta_ps_deliver_response(struct sta_info *sta,
|
||||
return;
|
||||
|
||||
for (tid = 0; tid < ARRAY_SIZE(sta->sta.txq); tid++) {
|
||||
if (!(tids & BIT(tid)) || txq_has_queue(sta->sta.txq[tid]))
|
||||
if (!(driver_release_tids & BIT(tid)) ||
|
||||
txq_has_queue(sta->sta.txq[tid]))
|
||||
continue;
|
||||
|
||||
sta_info_recalc_tim(sta);
|
||||
|
||||
+7
-12
@@ -455,7 +455,7 @@ struct sta_info {
|
||||
/* General information, mostly static */
|
||||
struct list_head list, free_list;
|
||||
struct rcu_head rcu_head;
|
||||
struct rhash_head hash_node;
|
||||
struct rhlist_head hash_node;
|
||||
u8 addr[ETH_ALEN];
|
||||
struct ieee80211_local *local;
|
||||
struct ieee80211_sub_if_data *sdata;
|
||||
@@ -638,6 +638,9 @@ rcu_dereference_protected_tid_tx(struct sta_info *sta, int tid)
|
||||
*/
|
||||
#define STA_INFO_CLEANUP_INTERVAL (10 * HZ)
|
||||
|
||||
struct rhlist_head *sta_info_hash_lookup(struct ieee80211_local *local,
|
||||
const u8 *addr);
|
||||
|
||||
/*
|
||||
* Get a STA info, must be under RCU read lock.
|
||||
*/
|
||||
@@ -647,17 +650,9 @@ struct sta_info *sta_info_get(struct ieee80211_sub_if_data *sdata,
|
||||
struct sta_info *sta_info_get_bss(struct ieee80211_sub_if_data *sdata,
|
||||
const u8 *addr);
|
||||
|
||||
u32 sta_addr_hash(const void *key, u32 length, u32 seed);
|
||||
|
||||
#define _sta_bucket_idx(_tbl, _a) \
|
||||
rht_bucket_index(_tbl, sta_addr_hash(_a, ETH_ALEN, (_tbl)->hash_rnd))
|
||||
|
||||
#define for_each_sta_info(local, tbl, _addr, _sta, _tmp) \
|
||||
rht_for_each_entry_rcu(_sta, _tmp, tbl, \
|
||||
_sta_bucket_idx(tbl, _addr), \
|
||||
hash_node) \
|
||||
/* compare address and run code only if it matches */ \
|
||||
if (ether_addr_equal(_sta->addr, (_addr)))
|
||||
#define for_each_sta_info(local, _addr, _sta, _tmp) \
|
||||
rhl_for_each_entry_rcu(_sta, _tmp, \
|
||||
sta_info_hash_lookup(local, _addr), hash_node)
|
||||
|
||||
/*
|
||||
* Get STA info by index, BROKEN!
|
||||
|
||||
@@ -746,8 +746,8 @@ void ieee80211_tx_status(struct ieee80211_hw *hw, struct sk_buff *skb)
|
||||
struct ieee80211_tx_info *info = IEEE80211_SKB_CB(skb);
|
||||
__le16 fc;
|
||||
struct ieee80211_supported_band *sband;
|
||||
struct rhlist_head *tmp;
|
||||
struct sta_info *sta;
|
||||
struct rhash_head *tmp;
|
||||
int retry_count;
|
||||
int rates_idx;
|
||||
bool send_to_cooked;
|
||||
@@ -755,7 +755,6 @@ void ieee80211_tx_status(struct ieee80211_hw *hw, struct sk_buff *skb)
|
||||
struct ieee80211_bar *bar;
|
||||
int shift = 0;
|
||||
int tid = IEEE80211_NUM_TIDS;
|
||||
const struct bucket_table *tbl;
|
||||
|
||||
rates_idx = ieee80211_tx_get_rates(hw, info, &retry_count);
|
||||
|
||||
@@ -764,9 +763,7 @@ void ieee80211_tx_status(struct ieee80211_hw *hw, struct sk_buff *skb)
|
||||
sband = local->hw.wiphy->bands[info->band];
|
||||
fc = hdr->frame_control;
|
||||
|
||||
tbl = rht_dereference_rcu(local->sta_hash.tbl, &local->sta_hash);
|
||||
|
||||
for_each_sta_info(local, tbl, hdr->addr1, sta, tmp) {
|
||||
for_each_sta_info(local, hdr->addr1, sta, tmp) {
|
||||
/* skip wrong virtual interface */
|
||||
if (!ether_addr_equal(hdr->addr2, sta->sdata->vif.addr))
|
||||
continue;
|
||||
|
||||
+8
-4
@@ -3447,13 +3447,17 @@ begin:
|
||||
skb_queue_splice_tail(&tx.skbs, &txqi->frags);
|
||||
}
|
||||
|
||||
if (skb && skb_has_frag_list(skb) &&
|
||||
!ieee80211_hw_check(&local->hw, TX_FRAG_LIST)) {
|
||||
if (skb_linearize(skb)) {
|
||||
ieee80211_free_txskb(&local->hw, skb);
|
||||
goto begin;
|
||||
}
|
||||
}
|
||||
|
||||
out:
|
||||
spin_unlock_bh(&fq->lock);
|
||||
|
||||
if (skb && skb_has_frag_list(skb) &&
|
||||
!ieee80211_hw_check(&local->hw, TX_FRAG_LIST))
|
||||
skb_linearize(skb);
|
||||
|
||||
return skb;
|
||||
}
|
||||
EXPORT_SYMBOL(ieee80211_tx_dequeue);
|
||||
|
||||
@@ -663,6 +663,7 @@ ieee802154_if_add(struct ieee802154_local *local, const char *name,
|
||||
|
||||
/* TODO check this */
|
||||
SET_NETDEV_DEV(ndev, &local->phy->dev);
|
||||
dev_net_set(ndev, wpan_phy_net(local->hw.phy));
|
||||
sdata = netdev_priv(ndev);
|
||||
ndev->ieee802154_ptr = &sdata->wpan_dev;
|
||||
memcpy(sdata->name, ndev->name, IFNAMSIZ);
|
||||
|
||||
+7
-2
@@ -101,11 +101,16 @@ ieee802154_subif_frame(struct ieee802154_sub_if_data *sdata,
|
||||
sdata->dev->stats.rx_bytes += skb->len;
|
||||
|
||||
switch (mac_cb(skb)->type) {
|
||||
case IEEE802154_FC_TYPE_BEACON:
|
||||
case IEEE802154_FC_TYPE_ACK:
|
||||
case IEEE802154_FC_TYPE_MAC_CMD:
|
||||
goto fail;
|
||||
|
||||
case IEEE802154_FC_TYPE_DATA:
|
||||
return ieee802154_deliver_skb(skb);
|
||||
default:
|
||||
pr_warn("ieee802154: bad frame received (type = %d)\n",
|
||||
mac_cb(skb)->type);
|
||||
pr_warn_ratelimited("ieee802154: bad frame received "
|
||||
"(type = %d)\n", mac_cb(skb)->type);
|
||||
goto fail;
|
||||
}
|
||||
|
||||
|
||||
+1
-9
@@ -1,9 +1,6 @@
|
||||
#ifndef MPLS_INTERNAL_H
|
||||
#define MPLS_INTERNAL_H
|
||||
|
||||
struct mpls_shim_hdr {
|
||||
__be32 label_stack_entry;
|
||||
};
|
||||
#include <net/mpls.h>
|
||||
|
||||
struct mpls_entry_decoded {
|
||||
u32 label;
|
||||
@@ -93,11 +90,6 @@ struct mpls_route { /* next hop label forwarding entry */
|
||||
|
||||
#define endfor_nexthops(rt) }
|
||||
|
||||
static inline struct mpls_shim_hdr *mpls_hdr(const struct sk_buff *skb)
|
||||
{
|
||||
return (struct mpls_shim_hdr *)skb_network_header(skb);
|
||||
}
|
||||
|
||||
static inline struct mpls_shim_hdr mpls_entry_encode(u32 label, unsigned ttl, unsigned tc, bool bos)
|
||||
{
|
||||
struct mpls_shim_hdr result;
|
||||
|
||||
+16
-6
@@ -170,6 +170,7 @@ struct ncsi_package;
|
||||
|
||||
#define NCSI_PACKAGE_SHIFT 5
|
||||
#define NCSI_PACKAGE_INDEX(c) (((c) >> NCSI_PACKAGE_SHIFT) & 0x7)
|
||||
#define NCSI_RESERVED_CHANNEL 0x1f
|
||||
#define NCSI_CHANNEL_INDEX(c) ((c) & ((1 << NCSI_PACKAGE_SHIFT) - 1))
|
||||
#define NCSI_TO_CHANNEL(p, c) (((p) << NCSI_PACKAGE_SHIFT) | (c))
|
||||
|
||||
@@ -186,9 +187,15 @@ struct ncsi_channel {
|
||||
struct ncsi_channel_mode modes[NCSI_MODE_MAX];
|
||||
struct ncsi_channel_filter *filters[NCSI_FILTER_MAX];
|
||||
struct ncsi_channel_stats stats;
|
||||
struct timer_list timer; /* Link monitor timer */
|
||||
bool enabled; /* Timer is enabled */
|
||||
unsigned int timeout; /* Times of timeout */
|
||||
struct {
|
||||
struct timer_list timer;
|
||||
bool enabled;
|
||||
unsigned int state;
|
||||
#define NCSI_CHANNEL_MONITOR_START 0
|
||||
#define NCSI_CHANNEL_MONITOR_RETRY 1
|
||||
#define NCSI_CHANNEL_MONITOR_WAIT 2
|
||||
#define NCSI_CHANNEL_MONITOR_WAIT_MAX 5
|
||||
} monitor;
|
||||
struct list_head node;
|
||||
struct list_head link;
|
||||
};
|
||||
@@ -206,7 +213,8 @@ struct ncsi_package {
|
||||
struct ncsi_request {
|
||||
unsigned char id; /* Request ID - 0 to 255 */
|
||||
bool used; /* Request that has been assigned */
|
||||
bool driven; /* Drive state machine */
|
||||
unsigned int flags; /* NCSI request property */
|
||||
#define NCSI_REQ_FLAG_EVENT_DRIVEN 1
|
||||
struct ncsi_dev_priv *ndp; /* Associated NCSI device */
|
||||
struct sk_buff *cmd; /* Associated NCSI command packet */
|
||||
struct sk_buff *rsp; /* Associated NCSI response packet */
|
||||
@@ -258,6 +266,7 @@ struct ncsi_dev_priv {
|
||||
struct list_head packages; /* List of packages */
|
||||
struct ncsi_request requests[256]; /* Request table */
|
||||
unsigned int request_id; /* Last used request ID */
|
||||
#define NCSI_REQ_START_IDX 1
|
||||
unsigned int pending_req_num; /* Number of pending requests */
|
||||
struct ncsi_package *active_package; /* Currently handled package */
|
||||
struct ncsi_channel *active_channel; /* Currently handled channel */
|
||||
@@ -274,7 +283,7 @@ struct ncsi_cmd_arg {
|
||||
unsigned char package; /* Destination package ID */
|
||||
unsigned char channel; /* Detination channel ID or 0x1f */
|
||||
unsigned short payload; /* Command packet payload length */
|
||||
bool driven; /* Drive the state machine? */
|
||||
unsigned int req_flags; /* NCSI request properties */
|
||||
union {
|
||||
unsigned char bytes[16]; /* Command packet specific data */
|
||||
unsigned short words[8];
|
||||
@@ -313,7 +322,8 @@ void ncsi_find_package_and_channel(struct ncsi_dev_priv *ndp,
|
||||
unsigned char id,
|
||||
struct ncsi_package **np,
|
||||
struct ncsi_channel **nc);
|
||||
struct ncsi_request *ncsi_alloc_request(struct ncsi_dev_priv *ndp, bool driven);
|
||||
struct ncsi_request *ncsi_alloc_request(struct ncsi_dev_priv *ndp,
|
||||
unsigned int req_flags);
|
||||
void ncsi_free_request(struct ncsi_request *nr);
|
||||
struct ncsi_dev *ncsi_find_dev(struct net_device *dev);
|
||||
int ncsi_process_next_channel(struct ncsi_dev_priv *ndp);
|
||||
|
||||
+27
-10
@@ -53,7 +53,9 @@ static int ncsi_aen_handler_lsc(struct ncsi_dev_priv *ndp,
|
||||
struct ncsi_aen_lsc_pkt *lsc;
|
||||
struct ncsi_channel *nc;
|
||||
struct ncsi_channel_mode *ncm;
|
||||
unsigned long old_data;
|
||||
bool chained;
|
||||
int state;
|
||||
unsigned long old_data, data;
|
||||
unsigned long flags;
|
||||
|
||||
/* Find the NCSI channel */
|
||||
@@ -62,20 +64,27 @@ static int ncsi_aen_handler_lsc(struct ncsi_dev_priv *ndp,
|
||||
return -ENODEV;
|
||||
|
||||
/* Update the link status */
|
||||
ncm = &nc->modes[NCSI_MODE_LINK];
|
||||
lsc = (struct ncsi_aen_lsc_pkt *)h;
|
||||
|
||||
spin_lock_irqsave(&nc->lock, flags);
|
||||
ncm = &nc->modes[NCSI_MODE_LINK];
|
||||
old_data = ncm->data[2];
|
||||
ncm->data[2] = ntohl(lsc->status);
|
||||
data = ntohl(lsc->status);
|
||||
ncm->data[2] = data;
|
||||
ncm->data[4] = ntohl(lsc->oem_status);
|
||||
if (!((old_data ^ ncm->data[2]) & 0x1) ||
|
||||
!list_empty(&nc->link))
|
||||
|
||||
chained = !list_empty(&nc->link);
|
||||
state = nc->state;
|
||||
spin_unlock_irqrestore(&nc->lock, flags);
|
||||
|
||||
if (!((old_data ^ data) & 0x1) || chained)
|
||||
return 0;
|
||||
if (!(nc->state == NCSI_CHANNEL_INACTIVE && (ncm->data[2] & 0x1)) &&
|
||||
!(nc->state == NCSI_CHANNEL_ACTIVE && !(ncm->data[2] & 0x1)))
|
||||
if (!(state == NCSI_CHANNEL_INACTIVE && (data & 0x1)) &&
|
||||
!(state == NCSI_CHANNEL_ACTIVE && !(data & 0x1)))
|
||||
return 0;
|
||||
|
||||
if (!(ndp->flags & NCSI_DEV_HWA) &&
|
||||
nc->state == NCSI_CHANNEL_ACTIVE)
|
||||
state == NCSI_CHANNEL_ACTIVE)
|
||||
ndp->flags |= NCSI_DEV_RESHUFFLE;
|
||||
|
||||
ncsi_stop_channel_monitor(nc);
|
||||
@@ -97,13 +106,21 @@ static int ncsi_aen_handler_cr(struct ncsi_dev_priv *ndp,
|
||||
if (!nc)
|
||||
return -ENODEV;
|
||||
|
||||
spin_lock_irqsave(&nc->lock, flags);
|
||||
if (!list_empty(&nc->link) ||
|
||||
nc->state != NCSI_CHANNEL_ACTIVE)
|
||||
nc->state != NCSI_CHANNEL_ACTIVE) {
|
||||
spin_unlock_irqrestore(&nc->lock, flags);
|
||||
return 0;
|
||||
}
|
||||
spin_unlock_irqrestore(&nc->lock, flags);
|
||||
|
||||
ncsi_stop_channel_monitor(nc);
|
||||
spin_lock_irqsave(&nc->lock, flags);
|
||||
nc->state = NCSI_CHANNEL_INVISIBLE;
|
||||
spin_unlock_irqrestore(&nc->lock, flags);
|
||||
|
||||
spin_lock_irqsave(&ndp->lock, flags);
|
||||
xchg(&nc->state, NCSI_CHANNEL_INACTIVE);
|
||||
nc->state = NCSI_CHANNEL_INACTIVE;
|
||||
list_add_tail_rcu(&nc->link, &ndp->channel_queue);
|
||||
spin_unlock_irqrestore(&ndp->lock, flags);
|
||||
|
||||
|
||||
+1
-1
@@ -272,7 +272,7 @@ static struct ncsi_request *ncsi_alloc_command(struct ncsi_cmd_arg *nca)
|
||||
struct sk_buff *skb;
|
||||
struct ncsi_request *nr;
|
||||
|
||||
nr = ncsi_alloc_request(ndp, nca->driven);
|
||||
nr = ncsi_alloc_request(ndp, nca->req_flags);
|
||||
if (!nr)
|
||||
return NULL;
|
||||
|
||||
|
||||
+127
-71
@@ -132,6 +132,7 @@ static void ncsi_report_link(struct ncsi_dev_priv *ndp, bool force_down)
|
||||
struct ncsi_dev *nd = &ndp->ndev;
|
||||
struct ncsi_package *np;
|
||||
struct ncsi_channel *nc;
|
||||
unsigned long flags;
|
||||
|
||||
nd->state = ncsi_dev_state_functional;
|
||||
if (force_down) {
|
||||
@@ -142,14 +143,21 @@ static void ncsi_report_link(struct ncsi_dev_priv *ndp, bool force_down)
|
||||
nd->link_up = 0;
|
||||
NCSI_FOR_EACH_PACKAGE(ndp, np) {
|
||||
NCSI_FOR_EACH_CHANNEL(np, nc) {
|
||||
spin_lock_irqsave(&nc->lock, flags);
|
||||
|
||||
if (!list_empty(&nc->link) ||
|
||||
nc->state != NCSI_CHANNEL_ACTIVE)
|
||||
nc->state != NCSI_CHANNEL_ACTIVE) {
|
||||
spin_unlock_irqrestore(&nc->lock, flags);
|
||||
continue;
|
||||
}
|
||||
|
||||
if (nc->modes[NCSI_MODE_LINK].data[2] & 0x1) {
|
||||
spin_unlock_irqrestore(&nc->lock, flags);
|
||||
nd->link_up = 1;
|
||||
goto report;
|
||||
}
|
||||
|
||||
spin_unlock_irqrestore(&nc->lock, flags);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -163,43 +171,55 @@ static void ncsi_channel_monitor(unsigned long data)
|
||||
struct ncsi_package *np = nc->package;
|
||||
struct ncsi_dev_priv *ndp = np->ndp;
|
||||
struct ncsi_cmd_arg nca;
|
||||
bool enabled;
|
||||
unsigned int timeout;
|
||||
bool enabled, chained;
|
||||
unsigned int monitor_state;
|
||||
unsigned long flags;
|
||||
int ret;
|
||||
int state, ret;
|
||||
|
||||
spin_lock_irqsave(&nc->lock, flags);
|
||||
timeout = nc->timeout;
|
||||
enabled = nc->enabled;
|
||||
state = nc->state;
|
||||
chained = !list_empty(&nc->link);
|
||||
enabled = nc->monitor.enabled;
|
||||
monitor_state = nc->monitor.state;
|
||||
spin_unlock_irqrestore(&nc->lock, flags);
|
||||
|
||||
if (!enabled || !list_empty(&nc->link))
|
||||
if (!enabled || chained)
|
||||
return;
|
||||
if (nc->state != NCSI_CHANNEL_INACTIVE &&
|
||||
nc->state != NCSI_CHANNEL_ACTIVE)
|
||||
if (state != NCSI_CHANNEL_INACTIVE &&
|
||||
state != NCSI_CHANNEL_ACTIVE)
|
||||
return;
|
||||
|
||||
if (!(timeout % 2)) {
|
||||
switch (monitor_state) {
|
||||
case NCSI_CHANNEL_MONITOR_START:
|
||||
case NCSI_CHANNEL_MONITOR_RETRY:
|
||||
nca.ndp = ndp;
|
||||
nca.package = np->id;
|
||||
nca.channel = nc->id;
|
||||
nca.type = NCSI_PKT_CMD_GLS;
|
||||
nca.driven = false;
|
||||
nca.req_flags = 0;
|
||||
ret = ncsi_xmit_cmd(&nca);
|
||||
if (ret) {
|
||||
netdev_err(ndp->ndev.dev, "Error %d sending GLS\n",
|
||||
ret);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
if (timeout + 1 >= 3) {
|
||||
break;
|
||||
case NCSI_CHANNEL_MONITOR_WAIT ... NCSI_CHANNEL_MONITOR_WAIT_MAX:
|
||||
break;
|
||||
default:
|
||||
if (!(ndp->flags & NCSI_DEV_HWA) &&
|
||||
nc->state == NCSI_CHANNEL_ACTIVE)
|
||||
state == NCSI_CHANNEL_ACTIVE) {
|
||||
ncsi_report_link(ndp, true);
|
||||
ndp->flags |= NCSI_DEV_RESHUFFLE;
|
||||
}
|
||||
|
||||
spin_lock_irqsave(&nc->lock, flags);
|
||||
nc->state = NCSI_CHANNEL_INVISIBLE;
|
||||
spin_unlock_irqrestore(&nc->lock, flags);
|
||||
|
||||
spin_lock_irqsave(&ndp->lock, flags);
|
||||
xchg(&nc->state, NCSI_CHANNEL_INACTIVE);
|
||||
nc->state = NCSI_CHANNEL_INACTIVE;
|
||||
list_add_tail_rcu(&nc->link, &ndp->channel_queue);
|
||||
spin_unlock_irqrestore(&ndp->lock, flags);
|
||||
ncsi_process_next_channel(ndp);
|
||||
@@ -207,10 +227,9 @@ static void ncsi_channel_monitor(unsigned long data)
|
||||
}
|
||||
|
||||
spin_lock_irqsave(&nc->lock, flags);
|
||||
nc->timeout = timeout + 1;
|
||||
nc->enabled = true;
|
||||
nc->monitor.state++;
|
||||
spin_unlock_irqrestore(&nc->lock, flags);
|
||||
mod_timer(&nc->timer, jiffies + HZ * (1 << (nc->timeout / 2)));
|
||||
mod_timer(&nc->monitor.timer, jiffies + HZ);
|
||||
}
|
||||
|
||||
void ncsi_start_channel_monitor(struct ncsi_channel *nc)
|
||||
@@ -218,12 +237,12 @@ void ncsi_start_channel_monitor(struct ncsi_channel *nc)
|
||||
unsigned long flags;
|
||||
|
||||
spin_lock_irqsave(&nc->lock, flags);
|
||||
WARN_ON_ONCE(nc->enabled);
|
||||
nc->timeout = 0;
|
||||
nc->enabled = true;
|
||||
WARN_ON_ONCE(nc->monitor.enabled);
|
||||
nc->monitor.enabled = true;
|
||||
nc->monitor.state = NCSI_CHANNEL_MONITOR_START;
|
||||
spin_unlock_irqrestore(&nc->lock, flags);
|
||||
|
||||
mod_timer(&nc->timer, jiffies + HZ * (1 << (nc->timeout / 2)));
|
||||
mod_timer(&nc->monitor.timer, jiffies + HZ);
|
||||
}
|
||||
|
||||
void ncsi_stop_channel_monitor(struct ncsi_channel *nc)
|
||||
@@ -231,14 +250,14 @@ void ncsi_stop_channel_monitor(struct ncsi_channel *nc)
|
||||
unsigned long flags;
|
||||
|
||||
spin_lock_irqsave(&nc->lock, flags);
|
||||
if (!nc->enabled) {
|
||||
if (!nc->monitor.enabled) {
|
||||
spin_unlock_irqrestore(&nc->lock, flags);
|
||||
return;
|
||||
}
|
||||
nc->enabled = false;
|
||||
nc->monitor.enabled = false;
|
||||
spin_unlock_irqrestore(&nc->lock, flags);
|
||||
|
||||
del_timer_sync(&nc->timer);
|
||||
del_timer_sync(&nc->monitor.timer);
|
||||
}
|
||||
|
||||
struct ncsi_channel *ncsi_find_channel(struct ncsi_package *np,
|
||||
@@ -267,8 +286,9 @@ struct ncsi_channel *ncsi_add_channel(struct ncsi_package *np, unsigned char id)
|
||||
nc->id = id;
|
||||
nc->package = np;
|
||||
nc->state = NCSI_CHANNEL_INACTIVE;
|
||||
nc->enabled = false;
|
||||
setup_timer(&nc->timer, ncsi_channel_monitor, (unsigned long)nc);
|
||||
nc->monitor.enabled = false;
|
||||
setup_timer(&nc->monitor.timer,
|
||||
ncsi_channel_monitor, (unsigned long)nc);
|
||||
spin_lock_init(&nc->lock);
|
||||
INIT_LIST_HEAD(&nc->link);
|
||||
for (index = 0; index < NCSI_CAP_MAX; index++)
|
||||
@@ -405,7 +425,8 @@ void ncsi_find_package_and_channel(struct ncsi_dev_priv *ndp,
|
||||
* be same. Otherwise, the bogus response might be replied. So
|
||||
* the available IDs are allocated in round-robin fashion.
|
||||
*/
|
||||
struct ncsi_request *ncsi_alloc_request(struct ncsi_dev_priv *ndp, bool driven)
|
||||
struct ncsi_request *ncsi_alloc_request(struct ncsi_dev_priv *ndp,
|
||||
unsigned int req_flags)
|
||||
{
|
||||
struct ncsi_request *nr = NULL;
|
||||
int i, limit = ARRAY_SIZE(ndp->requests);
|
||||
@@ -413,30 +434,31 @@ struct ncsi_request *ncsi_alloc_request(struct ncsi_dev_priv *ndp, bool driven)
|
||||
|
||||
/* Check if there is one available request until the ceiling */
|
||||
spin_lock_irqsave(&ndp->lock, flags);
|
||||
for (i = ndp->request_id; !nr && i < limit; i++) {
|
||||
for (i = ndp->request_id; i < limit; i++) {
|
||||
if (ndp->requests[i].used)
|
||||
continue;
|
||||
|
||||
nr = &ndp->requests[i];
|
||||
nr->used = true;
|
||||
nr->driven = driven;
|
||||
if (++ndp->request_id >= limit)
|
||||
ndp->request_id = 0;
|
||||
nr->flags = req_flags;
|
||||
ndp->request_id = i + 1;
|
||||
goto found;
|
||||
}
|
||||
|
||||
/* Fail back to check from the starting cursor */
|
||||
for (i = 0; !nr && i < ndp->request_id; i++) {
|
||||
for (i = NCSI_REQ_START_IDX; i < ndp->request_id; i++) {
|
||||
if (ndp->requests[i].used)
|
||||
continue;
|
||||
|
||||
nr = &ndp->requests[i];
|
||||
nr->used = true;
|
||||
nr->driven = driven;
|
||||
if (++ndp->request_id >= limit)
|
||||
ndp->request_id = 0;
|
||||
nr->flags = req_flags;
|
||||
ndp->request_id = i + 1;
|
||||
goto found;
|
||||
}
|
||||
spin_unlock_irqrestore(&ndp->lock, flags);
|
||||
|
||||
found:
|
||||
spin_unlock_irqrestore(&ndp->lock, flags);
|
||||
return nr;
|
||||
}
|
||||
|
||||
@@ -458,7 +480,7 @@ void ncsi_free_request(struct ncsi_request *nr)
|
||||
nr->cmd = NULL;
|
||||
nr->rsp = NULL;
|
||||
nr->used = false;
|
||||
driven = nr->driven;
|
||||
driven = !!(nr->flags & NCSI_REQ_FLAG_EVENT_DRIVEN);
|
||||
spin_unlock_irqrestore(&ndp->lock, flags);
|
||||
|
||||
if (driven && cmd && --ndp->pending_req_num == 0)
|
||||
@@ -508,10 +530,11 @@ static void ncsi_suspend_channel(struct ncsi_dev_priv *ndp)
|
||||
struct ncsi_package *np = ndp->active_package;
|
||||
struct ncsi_channel *nc = ndp->active_channel;
|
||||
struct ncsi_cmd_arg nca;
|
||||
unsigned long flags;
|
||||
int ret;
|
||||
|
||||
nca.ndp = ndp;
|
||||
nca.driven = true;
|
||||
nca.req_flags = NCSI_REQ_FLAG_EVENT_DRIVEN;
|
||||
switch (nd->state) {
|
||||
case ncsi_dev_state_suspend:
|
||||
nd->state = ncsi_dev_state_suspend_select;
|
||||
@@ -527,7 +550,7 @@ static void ncsi_suspend_channel(struct ncsi_dev_priv *ndp)
|
||||
nca.package = np->id;
|
||||
if (nd->state == ncsi_dev_state_suspend_select) {
|
||||
nca.type = NCSI_PKT_CMD_SP;
|
||||
nca.channel = 0x1f;
|
||||
nca.channel = NCSI_RESERVED_CHANNEL;
|
||||
if (ndp->flags & NCSI_DEV_HWA)
|
||||
nca.bytes[0] = 0;
|
||||
else
|
||||
@@ -544,7 +567,7 @@ static void ncsi_suspend_channel(struct ncsi_dev_priv *ndp)
|
||||
nd->state = ncsi_dev_state_suspend_deselect;
|
||||
} else if (nd->state == ncsi_dev_state_suspend_deselect) {
|
||||
nca.type = NCSI_PKT_CMD_DP;
|
||||
nca.channel = 0x1f;
|
||||
nca.channel = NCSI_RESERVED_CHANNEL;
|
||||
nd->state = ncsi_dev_state_suspend_done;
|
||||
}
|
||||
|
||||
@@ -556,7 +579,9 @@ static void ncsi_suspend_channel(struct ncsi_dev_priv *ndp)
|
||||
|
||||
break;
|
||||
case ncsi_dev_state_suspend_done:
|
||||
xchg(&nc->state, NCSI_CHANNEL_INACTIVE);
|
||||
spin_lock_irqsave(&nc->lock, flags);
|
||||
nc->state = NCSI_CHANNEL_INACTIVE;
|
||||
spin_unlock_irqrestore(&nc->lock, flags);
|
||||
ncsi_process_next_channel(ndp);
|
||||
|
||||
break;
|
||||
@@ -574,10 +599,11 @@ static void ncsi_configure_channel(struct ncsi_dev_priv *ndp)
|
||||
struct ncsi_channel *nc = ndp->active_channel;
|
||||
struct ncsi_cmd_arg nca;
|
||||
unsigned char index;
|
||||
unsigned long flags;
|
||||
int ret;
|
||||
|
||||
nca.ndp = ndp;
|
||||
nca.driven = true;
|
||||
nca.req_flags = NCSI_REQ_FLAG_EVENT_DRIVEN;
|
||||
switch (nd->state) {
|
||||
case ncsi_dev_state_config:
|
||||
case ncsi_dev_state_config_sp:
|
||||
@@ -590,7 +616,7 @@ static void ncsi_configure_channel(struct ncsi_dev_priv *ndp)
|
||||
else
|
||||
nca.bytes[0] = 1;
|
||||
nca.package = np->id;
|
||||
nca.channel = 0x1f;
|
||||
nca.channel = NCSI_RESERVED_CHANNEL;
|
||||
ret = ncsi_xmit_cmd(&nca);
|
||||
if (ret)
|
||||
goto error;
|
||||
@@ -675,10 +701,12 @@ static void ncsi_configure_channel(struct ncsi_dev_priv *ndp)
|
||||
goto error;
|
||||
break;
|
||||
case ncsi_dev_state_config_done:
|
||||
spin_lock_irqsave(&nc->lock, flags);
|
||||
if (nc->modes[NCSI_MODE_LINK].data[2] & 0x1)
|
||||
xchg(&nc->state, NCSI_CHANNEL_ACTIVE);
|
||||
nc->state = NCSI_CHANNEL_ACTIVE;
|
||||
else
|
||||
xchg(&nc->state, NCSI_CHANNEL_INACTIVE);
|
||||
nc->state = NCSI_CHANNEL_INACTIVE;
|
||||
spin_unlock_irqrestore(&nc->lock, flags);
|
||||
|
||||
ncsi_start_channel_monitor(nc);
|
||||
ncsi_process_next_channel(ndp);
|
||||
@@ -707,18 +735,25 @@ static int ncsi_choose_active_channel(struct ncsi_dev_priv *ndp)
|
||||
found = NULL;
|
||||
NCSI_FOR_EACH_PACKAGE(ndp, np) {
|
||||
NCSI_FOR_EACH_CHANNEL(np, nc) {
|
||||
spin_lock_irqsave(&nc->lock, flags);
|
||||
|
||||
if (!list_empty(&nc->link) ||
|
||||
nc->state != NCSI_CHANNEL_INACTIVE)
|
||||
nc->state != NCSI_CHANNEL_INACTIVE) {
|
||||
spin_unlock_irqrestore(&nc->lock, flags);
|
||||
continue;
|
||||
}
|
||||
|
||||
if (!found)
|
||||
found = nc;
|
||||
|
||||
ncm = &nc->modes[NCSI_MODE_LINK];
|
||||
if (ncm->data[2] & 0x1) {
|
||||
spin_unlock_irqrestore(&nc->lock, flags);
|
||||
found = nc;
|
||||
goto out;
|
||||
}
|
||||
|
||||
spin_unlock_irqrestore(&nc->lock, flags);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -797,7 +832,7 @@ static void ncsi_probe_channel(struct ncsi_dev_priv *ndp)
|
||||
int ret;
|
||||
|
||||
nca.ndp = ndp;
|
||||
nca.driven = true;
|
||||
nca.req_flags = NCSI_REQ_FLAG_EVENT_DRIVEN;
|
||||
switch (nd->state) {
|
||||
case ncsi_dev_state_probe:
|
||||
nd->state = ncsi_dev_state_probe_deselect;
|
||||
@@ -807,7 +842,7 @@ static void ncsi_probe_channel(struct ncsi_dev_priv *ndp)
|
||||
|
||||
/* Deselect all possible packages */
|
||||
nca.type = NCSI_PKT_CMD_DP;
|
||||
nca.channel = 0x1f;
|
||||
nca.channel = NCSI_RESERVED_CHANNEL;
|
||||
for (index = 0; index < 8; index++) {
|
||||
nca.package = index;
|
||||
ret = ncsi_xmit_cmd(&nca);
|
||||
@@ -823,7 +858,7 @@ static void ncsi_probe_channel(struct ncsi_dev_priv *ndp)
|
||||
/* Select all possible packages */
|
||||
nca.type = NCSI_PKT_CMD_SP;
|
||||
nca.bytes[0] = 1;
|
||||
nca.channel = 0x1f;
|
||||
nca.channel = NCSI_RESERVED_CHANNEL;
|
||||
for (index = 0; index < 8; index++) {
|
||||
nca.package = index;
|
||||
ret = ncsi_xmit_cmd(&nca);
|
||||
@@ -876,7 +911,7 @@ static void ncsi_probe_channel(struct ncsi_dev_priv *ndp)
|
||||
nca.type = NCSI_PKT_CMD_SP;
|
||||
nca.bytes[0] = 1;
|
||||
nca.package = ndp->active_package->id;
|
||||
nca.channel = 0x1f;
|
||||
nca.channel = NCSI_RESERVED_CHANNEL;
|
||||
ret = ncsi_xmit_cmd(&nca);
|
||||
if (ret)
|
||||
goto error;
|
||||
@@ -884,12 +919,12 @@ static void ncsi_probe_channel(struct ncsi_dev_priv *ndp)
|
||||
nd->state = ncsi_dev_state_probe_cis;
|
||||
break;
|
||||
case ncsi_dev_state_probe_cis:
|
||||
ndp->pending_req_num = 32;
|
||||
ndp->pending_req_num = NCSI_RESERVED_CHANNEL;
|
||||
|
||||
/* Clear initial state */
|
||||
nca.type = NCSI_PKT_CMD_CIS;
|
||||
nca.package = ndp->active_package->id;
|
||||
for (index = 0; index < 0x20; index++) {
|
||||
for (index = 0; index < NCSI_RESERVED_CHANNEL; index++) {
|
||||
nca.channel = index;
|
||||
ret = ncsi_xmit_cmd(&nca);
|
||||
if (ret)
|
||||
@@ -933,7 +968,7 @@ static void ncsi_probe_channel(struct ncsi_dev_priv *ndp)
|
||||
/* Deselect the active package */
|
||||
nca.type = NCSI_PKT_CMD_DP;
|
||||
nca.package = ndp->active_package->id;
|
||||
nca.channel = 0x1f;
|
||||
nca.channel = NCSI_RESERVED_CHANNEL;
|
||||
ret = ncsi_xmit_cmd(&nca);
|
||||
if (ret)
|
||||
goto error;
|
||||
@@ -987,11 +1022,14 @@ int ncsi_process_next_channel(struct ncsi_dev_priv *ndp)
|
||||
goto out;
|
||||
}
|
||||
|
||||
old_state = xchg(&nc->state, NCSI_CHANNEL_INVISIBLE);
|
||||
list_del_init(&nc->link);
|
||||
|
||||
spin_unlock_irqrestore(&ndp->lock, flags);
|
||||
|
||||
spin_lock_irqsave(&nc->lock, flags);
|
||||
old_state = nc->state;
|
||||
nc->state = NCSI_CHANNEL_INVISIBLE;
|
||||
spin_unlock_irqrestore(&nc->lock, flags);
|
||||
|
||||
ndp->active_channel = nc;
|
||||
ndp->active_package = nc->package;
|
||||
|
||||
@@ -1006,7 +1044,7 @@ int ncsi_process_next_channel(struct ncsi_dev_priv *ndp)
|
||||
break;
|
||||
default:
|
||||
netdev_err(ndp->ndev.dev, "Invalid state 0x%x on %d:%d\n",
|
||||
nc->state, nc->package->id, nc->id);
|
||||
old_state, nc->package->id, nc->id);
|
||||
ncsi_report_link(ndp, false);
|
||||
return -EINVAL;
|
||||
}
|
||||
@@ -1070,7 +1108,7 @@ static int ncsi_inet6addr_event(struct notifier_block *this,
|
||||
return NOTIFY_OK;
|
||||
|
||||
nca.ndp = ndp;
|
||||
nca.driven = false;
|
||||
nca.req_flags = 0;
|
||||
nca.package = np->id;
|
||||
nca.channel = nc->id;
|
||||
nca.dwords[0] = nc->caps[NCSI_CAP_MC].cap;
|
||||
@@ -1118,7 +1156,7 @@ struct ncsi_dev *ncsi_register_dev(struct net_device *dev,
|
||||
/* Initialize private NCSI device */
|
||||
spin_lock_init(&ndp->lock);
|
||||
INIT_LIST_HEAD(&ndp->packages);
|
||||
ndp->request_id = 0;
|
||||
ndp->request_id = NCSI_REQ_START_IDX;
|
||||
for (i = 0; i < ARRAY_SIZE(ndp->requests); i++) {
|
||||
ndp->requests[i].id = i;
|
||||
ndp->requests[i].ndp = ndp;
|
||||
@@ -1149,9 +1187,7 @@ EXPORT_SYMBOL_GPL(ncsi_register_dev);
|
||||
int ncsi_start_dev(struct ncsi_dev *nd)
|
||||
{
|
||||
struct ncsi_dev_priv *ndp = TO_NCSI_DEV_PRIV(nd);
|
||||
struct ncsi_package *np;
|
||||
struct ncsi_channel *nc;
|
||||
int old_state, ret;
|
||||
int ret;
|
||||
|
||||
if (nd->state != ncsi_dev_state_registered &&
|
||||
nd->state != ncsi_dev_state_functional)
|
||||
@@ -1163,15 +1199,6 @@ int ncsi_start_dev(struct ncsi_dev *nd)
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* Reset channel's state and start over */
|
||||
NCSI_FOR_EACH_PACKAGE(ndp, np) {
|
||||
NCSI_FOR_EACH_CHANNEL(np, nc) {
|
||||
old_state = xchg(&nc->state, NCSI_CHANNEL_INACTIVE);
|
||||
WARN_ON_ONCE(!list_empty(&nc->link) ||
|
||||
old_state == NCSI_CHANNEL_INVISIBLE);
|
||||
}
|
||||
}
|
||||
|
||||
if (ndp->flags & NCSI_DEV_HWA)
|
||||
ret = ncsi_enable_hwa(ndp);
|
||||
else
|
||||
@@ -1181,6 +1208,35 @@ int ncsi_start_dev(struct ncsi_dev *nd)
|
||||
}
|
||||
EXPORT_SYMBOL_GPL(ncsi_start_dev);
|
||||
|
||||
void ncsi_stop_dev(struct ncsi_dev *nd)
|
||||
{
|
||||
struct ncsi_dev_priv *ndp = TO_NCSI_DEV_PRIV(nd);
|
||||
struct ncsi_package *np;
|
||||
struct ncsi_channel *nc;
|
||||
bool chained;
|
||||
int old_state;
|
||||
unsigned long flags;
|
||||
|
||||
/* Stop the channel monitor and reset channel's state */
|
||||
NCSI_FOR_EACH_PACKAGE(ndp, np) {
|
||||
NCSI_FOR_EACH_CHANNEL(np, nc) {
|
||||
ncsi_stop_channel_monitor(nc);
|
||||
|
||||
spin_lock_irqsave(&nc->lock, flags);
|
||||
chained = !list_empty(&nc->link);
|
||||
old_state = nc->state;
|
||||
nc->state = NCSI_CHANNEL_INACTIVE;
|
||||
spin_unlock_irqrestore(&nc->lock, flags);
|
||||
|
||||
WARN_ON_ONCE(chained ||
|
||||
old_state == NCSI_CHANNEL_INVISIBLE);
|
||||
}
|
||||
}
|
||||
|
||||
ncsi_report_link(ndp, true);
|
||||
}
|
||||
EXPORT_SYMBOL_GPL(ncsi_stop_dev);
|
||||
|
||||
void ncsi_unregister_dev(struct ncsi_dev *nd)
|
||||
{
|
||||
struct ncsi_dev_priv *ndp = TO_NCSI_DEV_PRIV(nd);
|
||||
|
||||
+2
-2
@@ -317,12 +317,12 @@ static int ncsi_rsp_handler_gls(struct ncsi_request *nr)
|
||||
ncm->data[3] = ntohl(rsp->other);
|
||||
ncm->data[4] = ntohl(rsp->oem_status);
|
||||
|
||||
if (nr->driven)
|
||||
if (nr->flags & NCSI_REQ_FLAG_EVENT_DRIVEN)
|
||||
return 0;
|
||||
|
||||
/* Reset the channel monitor if it has been enabled */
|
||||
spin_lock_irqsave(&nc->lock, flags);
|
||||
nc->timeout = 0;
|
||||
nc->monitor.state = NCSI_CHANNEL_MONITOR_START;
|
||||
spin_unlock_irqrestore(&nc->lock, flags);
|
||||
|
||||
return 0;
|
||||
|
||||
@@ -71,8 +71,9 @@ obj-$(CONFIG_NF_DUP_NETDEV) += nf_dup_netdev.o
|
||||
|
||||
# nf_tables
|
||||
nf_tables-objs += nf_tables_core.o nf_tables_api.o nf_tables_trace.o
|
||||
nf_tables-objs += nft_immediate.o nft_cmp.o nft_lookup.o nft_dynset.o
|
||||
nf_tables-objs += nft_immediate.o nft_cmp.o nft_range.o
|
||||
nf_tables-objs += nft_bitwise.o nft_byteorder.o nft_payload.o
|
||||
nf_tables-objs += nft_lookup.o nft_dynset.o
|
||||
|
||||
obj-$(CONFIG_NF_TABLES) += nf_tables.o
|
||||
obj-$(CONFIG_NF_TABLES_INET) += nf_tables_inet.o
|
||||
|
||||
+146
-61
@@ -22,6 +22,7 @@
|
||||
#include <linux/proc_fs.h>
|
||||
#include <linux/mutex.h>
|
||||
#include <linux/slab.h>
|
||||
#include <linux/rcupdate.h>
|
||||
#include <net/net_namespace.h>
|
||||
#include <net/sock.h>
|
||||
|
||||
@@ -61,33 +62,55 @@ EXPORT_SYMBOL(nf_hooks_needed);
|
||||
#endif
|
||||
|
||||
static DEFINE_MUTEX(nf_hook_mutex);
|
||||
#define nf_entry_dereference(e) \
|
||||
rcu_dereference_protected(e, lockdep_is_held(&nf_hook_mutex))
|
||||
|
||||
static struct list_head *nf_find_hook_list(struct net *net,
|
||||
const struct nf_hook_ops *reg)
|
||||
static struct nf_hook_entry *nf_hook_entry_head(struct net *net,
|
||||
const struct nf_hook_ops *reg)
|
||||
{
|
||||
struct list_head *hook_list = NULL;
|
||||
struct nf_hook_entry *hook_head = NULL;
|
||||
|
||||
if (reg->pf != NFPROTO_NETDEV)
|
||||
hook_list = &net->nf.hooks[reg->pf][reg->hooknum];
|
||||
hook_head = nf_entry_dereference(net->nf.hooks[reg->pf]
|
||||
[reg->hooknum]);
|
||||
else if (reg->hooknum == NF_NETDEV_INGRESS) {
|
||||
#ifdef CONFIG_NETFILTER_INGRESS
|
||||
if (reg->dev && dev_net(reg->dev) == net)
|
||||
hook_list = ®->dev->nf_hooks_ingress;
|
||||
hook_head =
|
||||
nf_entry_dereference(
|
||||
reg->dev->nf_hooks_ingress);
|
||||
#endif
|
||||
}
|
||||
return hook_list;
|
||||
return hook_head;
|
||||
}
|
||||
|
||||
struct nf_hook_entry {
|
||||
const struct nf_hook_ops *orig_ops;
|
||||
struct nf_hook_ops ops;
|
||||
};
|
||||
/* must hold nf_hook_mutex */
|
||||
static void nf_set_hooks_head(struct net *net, const struct nf_hook_ops *reg,
|
||||
struct nf_hook_entry *entry)
|
||||
{
|
||||
switch (reg->pf) {
|
||||
case NFPROTO_NETDEV:
|
||||
/* We already checked in nf_register_net_hook() that this is
|
||||
* used from ingress.
|
||||
*/
|
||||
rcu_assign_pointer(reg->dev->nf_hooks_ingress, entry);
|
||||
break;
|
||||
default:
|
||||
rcu_assign_pointer(net->nf.hooks[reg->pf][reg->hooknum],
|
||||
entry);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
int nf_register_net_hook(struct net *net, const struct nf_hook_ops *reg)
|
||||
{
|
||||
struct list_head *hook_list;
|
||||
struct nf_hook_entry *hooks_entry;
|
||||
struct nf_hook_entry *entry;
|
||||
struct nf_hook_ops *elem;
|
||||
|
||||
if (reg->pf == NFPROTO_NETDEV &&
|
||||
(reg->hooknum != NF_NETDEV_INGRESS ||
|
||||
!reg->dev || dev_net(reg->dev) != net))
|
||||
return -EINVAL;
|
||||
|
||||
entry = kmalloc(sizeof(*entry), GFP_KERNEL);
|
||||
if (!entry)
|
||||
@@ -95,19 +118,30 @@ int nf_register_net_hook(struct net *net, const struct nf_hook_ops *reg)
|
||||
|
||||
entry->orig_ops = reg;
|
||||
entry->ops = *reg;
|
||||
|
||||
hook_list = nf_find_hook_list(net, reg);
|
||||
if (!hook_list) {
|
||||
kfree(entry);
|
||||
return -ENOENT;
|
||||
}
|
||||
entry->next = NULL;
|
||||
|
||||
mutex_lock(&nf_hook_mutex);
|
||||
list_for_each_entry(elem, hook_list, list) {
|
||||
if (reg->priority < elem->priority)
|
||||
break;
|
||||
hooks_entry = nf_hook_entry_head(net, reg);
|
||||
|
||||
if (hooks_entry && hooks_entry->orig_ops->priority > reg->priority) {
|
||||
/* This is the case where we need to insert at the head */
|
||||
entry->next = hooks_entry;
|
||||
hooks_entry = NULL;
|
||||
}
|
||||
list_add_rcu(&entry->ops.list, elem->list.prev);
|
||||
|
||||
while (hooks_entry &&
|
||||
reg->priority >= hooks_entry->orig_ops->priority &&
|
||||
nf_entry_dereference(hooks_entry->next)) {
|
||||
hooks_entry = nf_entry_dereference(hooks_entry->next);
|
||||
}
|
||||
|
||||
if (hooks_entry) {
|
||||
entry->next = nf_entry_dereference(hooks_entry->next);
|
||||
rcu_assign_pointer(hooks_entry->next, entry);
|
||||
} else {
|
||||
nf_set_hooks_head(net, reg, entry);
|
||||
}
|
||||
|
||||
mutex_unlock(&nf_hook_mutex);
|
||||
#ifdef CONFIG_NETFILTER_INGRESS
|
||||
if (reg->pf == NFPROTO_NETDEV && reg->hooknum == NF_NETDEV_INGRESS)
|
||||
@@ -122,24 +156,33 @@ EXPORT_SYMBOL(nf_register_net_hook);
|
||||
|
||||
void nf_unregister_net_hook(struct net *net, const struct nf_hook_ops *reg)
|
||||
{
|
||||
struct list_head *hook_list;
|
||||
struct nf_hook_entry *entry;
|
||||
struct nf_hook_ops *elem;
|
||||
|
||||
hook_list = nf_find_hook_list(net, reg);
|
||||
if (!hook_list)
|
||||
return;
|
||||
struct nf_hook_entry *hooks_entry;
|
||||
|
||||
mutex_lock(&nf_hook_mutex);
|
||||
list_for_each_entry(elem, hook_list, list) {
|
||||
entry = container_of(elem, struct nf_hook_entry, ops);
|
||||
if (entry->orig_ops == reg) {
|
||||
list_del_rcu(&entry->ops.list);
|
||||
break;
|
||||
}
|
||||
hooks_entry = nf_hook_entry_head(net, reg);
|
||||
if (hooks_entry->orig_ops == reg) {
|
||||
nf_set_hooks_head(net, reg,
|
||||
nf_entry_dereference(hooks_entry->next));
|
||||
goto unlock;
|
||||
}
|
||||
while (hooks_entry && nf_entry_dereference(hooks_entry->next)) {
|
||||
struct nf_hook_entry *next =
|
||||
nf_entry_dereference(hooks_entry->next);
|
||||
struct nf_hook_entry *nnext;
|
||||
|
||||
if (next->orig_ops != reg) {
|
||||
hooks_entry = next;
|
||||
continue;
|
||||
}
|
||||
nnext = nf_entry_dereference(next->next);
|
||||
rcu_assign_pointer(hooks_entry->next, nnext);
|
||||
hooks_entry = next;
|
||||
break;
|
||||
}
|
||||
|
||||
unlock:
|
||||
mutex_unlock(&nf_hook_mutex);
|
||||
if (&elem->list == hook_list) {
|
||||
if (!hooks_entry) {
|
||||
WARN(1, "nf_unregister_net_hook: hook not found!\n");
|
||||
return;
|
||||
}
|
||||
@@ -151,10 +194,10 @@ void nf_unregister_net_hook(struct net *net, const struct nf_hook_ops *reg)
|
||||
static_key_slow_dec(&nf_hooks_needed[reg->pf][reg->hooknum]);
|
||||
#endif
|
||||
synchronize_net();
|
||||
nf_queue_nf_hook_drop(net, &entry->ops);
|
||||
nf_queue_nf_hook_drop(net, hooks_entry);
|
||||
/* other cpu might still process nfqueue verdict that used reg */
|
||||
synchronize_net();
|
||||
kfree(entry);
|
||||
kfree(hooks_entry);
|
||||
}
|
||||
EXPORT_SYMBOL(nf_unregister_net_hook);
|
||||
|
||||
@@ -188,19 +231,17 @@ EXPORT_SYMBOL(nf_unregister_net_hooks);
|
||||
|
||||
static LIST_HEAD(nf_hook_list);
|
||||
|
||||
int nf_register_hook(struct nf_hook_ops *reg)
|
||||
static int _nf_register_hook(struct nf_hook_ops *reg)
|
||||
{
|
||||
struct net *net, *last;
|
||||
int ret;
|
||||
|
||||
rtnl_lock();
|
||||
for_each_net(net) {
|
||||
ret = nf_register_net_hook(net, reg);
|
||||
if (ret && ret != -ENOENT)
|
||||
goto rollback;
|
||||
}
|
||||
list_add_tail(®->list, &nf_hook_list);
|
||||
rtnl_unlock();
|
||||
|
||||
return 0;
|
||||
rollback:
|
||||
@@ -210,19 +251,34 @@ rollback:
|
||||
break;
|
||||
nf_unregister_net_hook(net, reg);
|
||||
}
|
||||
return ret;
|
||||
}
|
||||
|
||||
int nf_register_hook(struct nf_hook_ops *reg)
|
||||
{
|
||||
int ret;
|
||||
|
||||
rtnl_lock();
|
||||
ret = _nf_register_hook(reg);
|
||||
rtnl_unlock();
|
||||
|
||||
return ret;
|
||||
}
|
||||
EXPORT_SYMBOL(nf_register_hook);
|
||||
|
||||
void nf_unregister_hook(struct nf_hook_ops *reg)
|
||||
static void _nf_unregister_hook(struct nf_hook_ops *reg)
|
||||
{
|
||||
struct net *net;
|
||||
|
||||
rtnl_lock();
|
||||
list_del(®->list);
|
||||
for_each_net(net)
|
||||
nf_unregister_net_hook(net, reg);
|
||||
}
|
||||
|
||||
void nf_unregister_hook(struct nf_hook_ops *reg)
|
||||
{
|
||||
rtnl_lock();
|
||||
_nf_unregister_hook(reg);
|
||||
rtnl_unlock();
|
||||
}
|
||||
EXPORT_SYMBOL(nf_unregister_hook);
|
||||
@@ -246,6 +302,26 @@ err:
|
||||
}
|
||||
EXPORT_SYMBOL(nf_register_hooks);
|
||||
|
||||
/* Caller MUST take rtnl_lock() */
|
||||
int _nf_register_hooks(struct nf_hook_ops *reg, unsigned int n)
|
||||
{
|
||||
unsigned int i;
|
||||
int err = 0;
|
||||
|
||||
for (i = 0; i < n; i++) {
|
||||
err = _nf_register_hook(®[i]);
|
||||
if (err)
|
||||
goto err;
|
||||
}
|
||||
return err;
|
||||
|
||||
err:
|
||||
if (i > 0)
|
||||
_nf_unregister_hooks(reg, i);
|
||||
return err;
|
||||
}
|
||||
EXPORT_SYMBOL(_nf_register_hooks);
|
||||
|
||||
void nf_unregister_hooks(struct nf_hook_ops *reg, unsigned int n)
|
||||
{
|
||||
while (n-- > 0)
|
||||
@@ -253,10 +329,17 @@ void nf_unregister_hooks(struct nf_hook_ops *reg, unsigned int n)
|
||||
}
|
||||
EXPORT_SYMBOL(nf_unregister_hooks);
|
||||
|
||||
unsigned int nf_iterate(struct list_head *head,
|
||||
struct sk_buff *skb,
|
||||
/* Caller MUST take rtnl_lock */
|
||||
void _nf_unregister_hooks(struct nf_hook_ops *reg, unsigned int n)
|
||||
{
|
||||
while (n-- > 0)
|
||||
_nf_unregister_hook(®[n]);
|
||||
}
|
||||
EXPORT_SYMBOL(_nf_unregister_hooks);
|
||||
|
||||
unsigned int nf_iterate(struct sk_buff *skb,
|
||||
struct nf_hook_state *state,
|
||||
struct nf_hook_ops **elemp)
|
||||
struct nf_hook_entry **entryp)
|
||||
{
|
||||
unsigned int verdict;
|
||||
|
||||
@@ -264,20 +347,23 @@ unsigned int nf_iterate(struct list_head *head,
|
||||
* The caller must not block between calls to this
|
||||
* function because of risk of continuing from deleted element.
|
||||
*/
|
||||
list_for_each_entry_continue_rcu((*elemp), head, list) {
|
||||
if (state->thresh > (*elemp)->priority)
|
||||
while (*entryp) {
|
||||
if (state->thresh > (*entryp)->ops.priority) {
|
||||
*entryp = rcu_dereference((*entryp)->next);
|
||||
continue;
|
||||
}
|
||||
|
||||
/* Optimization: we don't need to hold module
|
||||
reference here, since function can't sleep. --RR */
|
||||
repeat:
|
||||
verdict = (*elemp)->hook((*elemp)->priv, skb, state);
|
||||
verdict = (*entryp)->ops.hook((*entryp)->ops.priv, skb, state);
|
||||
if (verdict != NF_ACCEPT) {
|
||||
#ifdef CONFIG_NETFILTER_DEBUG
|
||||
if (unlikely((verdict & NF_VERDICT_MASK)
|
||||
> NF_MAX_VERDICT)) {
|
||||
NFDEBUG("Evil return from %p(%u).\n",
|
||||
(*elemp)->hook, state->hook);
|
||||
(*entryp)->ops.hook, state->hook);
|
||||
*entryp = rcu_dereference((*entryp)->next);
|
||||
continue;
|
||||
}
|
||||
#endif
|
||||
@@ -285,25 +371,23 @@ repeat:
|
||||
return verdict;
|
||||
goto repeat;
|
||||
}
|
||||
*entryp = rcu_dereference((*entryp)->next);
|
||||
}
|
||||
return NF_ACCEPT;
|
||||
}
|
||||
|
||||
|
||||
/* Returns 1 if okfn() needs to be executed by the caller,
|
||||
* -EPERM for NF_DROP, 0 otherwise. */
|
||||
* -EPERM for NF_DROP, 0 otherwise. Caller must hold rcu_read_lock. */
|
||||
int nf_hook_slow(struct sk_buff *skb, struct nf_hook_state *state)
|
||||
{
|
||||
struct nf_hook_ops *elem;
|
||||
struct nf_hook_entry *entry;
|
||||
unsigned int verdict;
|
||||
int ret = 0;
|
||||
|
||||
/* We may already have this, but read-locks nest anyway */
|
||||
rcu_read_lock();
|
||||
|
||||
elem = list_entry_rcu(state->hook_list, struct nf_hook_ops, list);
|
||||
entry = rcu_dereference(state->hook_entries);
|
||||
next_hook:
|
||||
verdict = nf_iterate(state->hook_list, skb, state, &elem);
|
||||
verdict = nf_iterate(skb, state, &entry);
|
||||
if (verdict == NF_ACCEPT || verdict == NF_STOP) {
|
||||
ret = 1;
|
||||
} else if ((verdict & NF_VERDICT_MASK) == NF_DROP) {
|
||||
@@ -312,8 +396,10 @@ next_hook:
|
||||
if (ret == 0)
|
||||
ret = -EPERM;
|
||||
} else if ((verdict & NF_VERDICT_MASK) == NF_QUEUE) {
|
||||
int err = nf_queue(skb, elem, state,
|
||||
verdict >> NF_VERDICT_QBITS);
|
||||
int err;
|
||||
|
||||
RCU_INIT_POINTER(state->hook_entries, entry);
|
||||
err = nf_queue(skb, state, verdict >> NF_VERDICT_QBITS);
|
||||
if (err < 0) {
|
||||
if (err == -ESRCH &&
|
||||
(verdict & NF_VERDICT_FLAG_QUEUE_BYPASS))
|
||||
@@ -321,7 +407,6 @@ next_hook:
|
||||
kfree_skb(skb);
|
||||
}
|
||||
}
|
||||
rcu_read_unlock();
|
||||
return ret;
|
||||
}
|
||||
EXPORT_SYMBOL(nf_hook_slow);
|
||||
@@ -441,7 +526,7 @@ static int __net_init netfilter_net_init(struct net *net)
|
||||
|
||||
for (i = 0; i < ARRAY_SIZE(net->nf.hooks); i++) {
|
||||
for (h = 0; h < NF_MAX_HOOKS; h++)
|
||||
INIT_LIST_HEAD(&net->nf.hooks[i][h]);
|
||||
RCU_INIT_POINTER(net->nf.hooks[i][h], NULL);
|
||||
}
|
||||
|
||||
#ifdef CONFIG_PROC_FS
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user