br_if.c 15 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626627628629630631632633634635636637638639640641642643644645646647648649650651652653654655656657658659660661662663664665666667668669670671672673674675676677678679680681682
  1. /*
  2. * Userspace interface
  3. * Linux ethernet bridge
  4. *
  5. * Authors:
  6. * Lennert Buytenhek <buytenh@gnu.org>
  7. *
  8. * This program is free software; you can redistribute it and/or
  9. * modify it under the terms of the GNU General Public License
  10. * as published by the Free Software Foundation; either version
  11. * 2 of the License, or (at your option) any later version.
  12. */
  13. #include <linux/kernel.h>
  14. #include <linux/netdevice.h>
  15. #include <linux/etherdevice.h>
  16. #include <linux/netpoll.h>
  17. #include <linux/ethtool.h>
  18. #include <linux/if_arp.h>
  19. #include <linux/module.h>
  20. #include <linux/init.h>
  21. #include <linux/rtnetlink.h>
  22. #include <linux/if_ether.h>
  23. #include <linux/slab.h>
  24. #include <net/dsa.h>
  25. #include <net/sock.h>
  26. #include <linux/if_vlan.h>
  27. #include <net/switchdev.h>
  28. #include "br_private.h"
  29. /*
  30. * Determine initial path cost based on speed.
  31. * using recommendations from 802.1d standard
  32. *
  33. * Since driver might sleep need to not be holding any locks.
  34. */
  35. static int port_cost(struct net_device *dev)
  36. {
  37. struct ethtool_link_ksettings ecmd;
  38. if (!__ethtool_get_link_ksettings(dev, &ecmd)) {
  39. switch (ecmd.base.speed) {
  40. case SPEED_10000:
  41. return 2;
  42. case SPEED_1000:
  43. return 4;
  44. case SPEED_100:
  45. return 19;
  46. case SPEED_10:
  47. return 100;
  48. }
  49. }
  50. /* Old silly heuristics based on name */
  51. if (!strncmp(dev->name, "lec", 3))
  52. return 7;
  53. if (!strncmp(dev->name, "plip", 4))
  54. return 2500;
  55. return 100; /* assume old 10Mbps */
  56. }
  57. /* Check for port carrier transitions. */
  58. void br_port_carrier_check(struct net_bridge_port *p, bool *notified)
  59. {
  60. struct net_device *dev = p->dev;
  61. struct net_bridge *br = p->br;
  62. if (!(p->flags & BR_ADMIN_COST) &&
  63. netif_running(dev) && netif_oper_up(dev))
  64. p->path_cost = port_cost(dev);
  65. *notified = false;
  66. if (!netif_running(br->dev))
  67. return;
  68. spin_lock_bh(&br->lock);
  69. if (netif_running(dev) && netif_oper_up(dev)) {
  70. if (p->state == BR_STATE_DISABLED) {
  71. br_stp_enable_port(p);
  72. *notified = true;
  73. }
  74. } else {
  75. if (p->state != BR_STATE_DISABLED) {
  76. br_stp_disable_port(p);
  77. *notified = true;
  78. }
  79. }
  80. spin_unlock_bh(&br->lock);
  81. }
  82. static void br_port_set_promisc(struct net_bridge_port *p)
  83. {
  84. int err = 0;
  85. if (br_promisc_port(p))
  86. return;
  87. err = dev_set_promiscuity(p->dev, 1);
  88. if (err)
  89. return;
  90. br_fdb_unsync_static(p->br, p);
  91. p->flags |= BR_PROMISC;
  92. }
  93. static void br_port_clear_promisc(struct net_bridge_port *p)
  94. {
  95. int err;
  96. /* Check if the port is already non-promisc or if it doesn't
  97. * support UNICAST filtering. Without unicast filtering support
  98. * we'll end up re-enabling promisc mode anyway, so just check for
  99. * it here.
  100. */
  101. if (!br_promisc_port(p) || !(p->dev->priv_flags & IFF_UNICAST_FLT))
  102. return;
  103. /* Since we'll be clearing the promisc mode, program the port
  104. * first so that we don't have interruption in traffic.
  105. */
  106. err = br_fdb_sync_static(p->br, p);
  107. if (err)
  108. return;
  109. dev_set_promiscuity(p->dev, -1);
  110. p->flags &= ~BR_PROMISC;
  111. }
  112. /* When a port is added or removed or when certain port flags
  113. * change, this function is called to automatically manage
  114. * promiscuity setting of all the bridge ports. We are always called
  115. * under RTNL so can skip using rcu primitives.
  116. */
  117. void br_manage_promisc(struct net_bridge *br)
  118. {
  119. struct net_bridge_port *p;
  120. bool set_all = false;
  121. /* If vlan filtering is disabled or bridge interface is placed
  122. * into promiscuous mode, place all ports in promiscuous mode.
  123. */
  124. if ((br->dev->flags & IFF_PROMISC) || !br_vlan_enabled(br->dev))
  125. set_all = true;
  126. list_for_each_entry(p, &br->port_list, list) {
  127. if (set_all) {
  128. br_port_set_promisc(p);
  129. } else {
  130. /* If the number of auto-ports is <= 1, then all other
  131. * ports will have their output configuration
  132. * statically specified through fdbs. Since ingress
  133. * on the auto-port becomes forwarding/egress to other
  134. * ports and egress configuration is statically known,
  135. * we can say that ingress configuration of the
  136. * auto-port is also statically known.
  137. * This lets us disable promiscuous mode and write
  138. * this config to hw.
  139. */
  140. if (br->auto_cnt == 0 ||
  141. (br->auto_cnt == 1 && br_auto_port(p)))
  142. br_port_clear_promisc(p);
  143. else
  144. br_port_set_promisc(p);
  145. }
  146. }
  147. }
  148. static void nbp_update_port_count(struct net_bridge *br)
  149. {
  150. struct net_bridge_port *p;
  151. u32 cnt = 0;
  152. list_for_each_entry(p, &br->port_list, list) {
  153. if (br_auto_port(p))
  154. cnt++;
  155. }
  156. if (br->auto_cnt != cnt) {
  157. br->auto_cnt = cnt;
  158. br_manage_promisc(br);
  159. }
  160. }
  161. static void nbp_delete_promisc(struct net_bridge_port *p)
  162. {
  163. /* If port is currently promiscuous, unset promiscuity.
  164. * Otherwise, it is a static port so remove all addresses
  165. * from it.
  166. */
  167. dev_set_allmulti(p->dev, -1);
  168. if (br_promisc_port(p))
  169. dev_set_promiscuity(p->dev, -1);
  170. else
  171. br_fdb_unsync_static(p->br, p);
  172. }
  173. static void release_nbp(struct kobject *kobj)
  174. {
  175. struct net_bridge_port *p
  176. = container_of(kobj, struct net_bridge_port, kobj);
  177. kfree(p);
  178. }
  179. static struct kobj_type brport_ktype = {
  180. #ifdef CONFIG_SYSFS
  181. .sysfs_ops = &brport_sysfs_ops,
  182. #endif
  183. .release = release_nbp,
  184. };
  185. static void destroy_nbp(struct net_bridge_port *p)
  186. {
  187. struct net_device *dev = p->dev;
  188. p->br = NULL;
  189. p->dev = NULL;
  190. dev_put(dev);
  191. kobject_put(&p->kobj);
  192. }
  193. static void destroy_nbp_rcu(struct rcu_head *head)
  194. {
  195. struct net_bridge_port *p =
  196. container_of(head, struct net_bridge_port, rcu);
  197. destroy_nbp(p);
  198. }
  199. static unsigned get_max_headroom(struct net_bridge *br)
  200. {
  201. unsigned max_headroom = 0;
  202. struct net_bridge_port *p;
  203. list_for_each_entry(p, &br->port_list, list) {
  204. unsigned dev_headroom = netdev_get_fwd_headroom(p->dev);
  205. if (dev_headroom > max_headroom)
  206. max_headroom = dev_headroom;
  207. }
  208. return max_headroom;
  209. }
  210. static void update_headroom(struct net_bridge *br, int new_hr)
  211. {
  212. struct net_bridge_port *p;
  213. list_for_each_entry(p, &br->port_list, list)
  214. netdev_set_rx_headroom(p->dev, new_hr);
  215. br->dev->needed_headroom = new_hr;
  216. }
  217. /* Delete port(interface) from bridge is done in two steps.
  218. * via RCU. First step, marks device as down. That deletes
  219. * all the timers and stops new packets from flowing through.
  220. *
  221. * Final cleanup doesn't occur until after all CPU's finished
  222. * processing packets.
  223. *
  224. * Protected from multiple admin operations by RTNL mutex
  225. */
  226. static void del_nbp(struct net_bridge_port *p)
  227. {
  228. struct net_bridge *br = p->br;
  229. struct net_device *dev = p->dev;
  230. sysfs_remove_link(br->ifobj, p->dev->name);
  231. nbp_delete_promisc(p);
  232. spin_lock_bh(&br->lock);
  233. br_stp_disable_port(p);
  234. spin_unlock_bh(&br->lock);
  235. br_ifinfo_notify(RTM_DELLINK, NULL, p);
  236. list_del_rcu(&p->list);
  237. if (netdev_get_fwd_headroom(dev) == br->dev->needed_headroom)
  238. update_headroom(br, get_max_headroom(br));
  239. netdev_reset_rx_headroom(dev);
  240. nbp_vlan_flush(p);
  241. br_fdb_delete_by_port(br, p, 0, 1);
  242. switchdev_deferred_process();
  243. nbp_update_port_count(br);
  244. netdev_upper_dev_unlink(dev, br->dev);
  245. dev->priv_flags &= ~IFF_BRIDGE_PORT;
  246. netdev_rx_handler_unregister(dev);
  247. br_multicast_del_port(p);
  248. kobject_uevent(&p->kobj, KOBJ_REMOVE);
  249. kobject_del(&p->kobj);
  250. br_netpoll_disable(p);
  251. call_rcu(&p->rcu, destroy_nbp_rcu);
  252. }
  253. /* Delete bridge device */
  254. void br_dev_delete(struct net_device *dev, struct list_head *head)
  255. {
  256. struct net_bridge *br = netdev_priv(dev);
  257. struct net_bridge_port *p, *n;
  258. list_for_each_entry_safe(p, n, &br->port_list, list) {
  259. del_nbp(p);
  260. }
  261. br_recalculate_neigh_suppress_enabled(br);
  262. br_fdb_delete_by_port(br, NULL, 0, 1);
  263. cancel_delayed_work_sync(&br->gc_work);
  264. br_sysfs_delbr(br->dev);
  265. unregister_netdevice_queue(br->dev, head);
  266. }
  267. /* find an available port number */
  268. static int find_portno(struct net_bridge *br)
  269. {
  270. int index;
  271. struct net_bridge_port *p;
  272. unsigned long *inuse;
  273. inuse = kcalloc(BITS_TO_LONGS(BR_MAX_PORTS), sizeof(unsigned long),
  274. GFP_KERNEL);
  275. if (!inuse)
  276. return -ENOMEM;
  277. set_bit(0, inuse); /* zero is reserved */
  278. list_for_each_entry(p, &br->port_list, list) {
  279. set_bit(p->port_no, inuse);
  280. }
  281. index = find_first_zero_bit(inuse, BR_MAX_PORTS);
  282. kfree(inuse);
  283. return (index >= BR_MAX_PORTS) ? -EXFULL : index;
  284. }
  285. /* called with RTNL but without bridge lock */
  286. static struct net_bridge_port *new_nbp(struct net_bridge *br,
  287. struct net_device *dev)
  288. {
  289. struct net_bridge_port *p;
  290. int index, err;
  291. index = find_portno(br);
  292. if (index < 0)
  293. return ERR_PTR(index);
  294. p = kzalloc(sizeof(*p), GFP_KERNEL);
  295. if (p == NULL)
  296. return ERR_PTR(-ENOMEM);
  297. p->br = br;
  298. dev_hold(dev);
  299. p->dev = dev;
  300. p->path_cost = port_cost(dev);
  301. p->priority = 0x8000 >> BR_PORT_BITS;
  302. p->port_no = index;
  303. p->flags = BR_LEARNING | BR_FLOOD | BR_MCAST_FLOOD | BR_BCAST_FLOOD;
  304. br_init_port(p);
  305. br_set_state(p, BR_STATE_DISABLED);
  306. br_stp_port_timer_init(p);
  307. err = br_multicast_add_port(p);
  308. if (err) {
  309. dev_put(dev);
  310. kfree(p);
  311. p = ERR_PTR(err);
  312. }
  313. return p;
  314. }
  315. int br_add_bridge(struct net *net, const char *name)
  316. {
  317. struct net_device *dev;
  318. int res;
  319. dev = alloc_netdev(sizeof(struct net_bridge), name, NET_NAME_UNKNOWN,
  320. br_dev_setup);
  321. if (!dev)
  322. return -ENOMEM;
  323. dev_net_set(dev, net);
  324. dev->rtnl_link_ops = &br_link_ops;
  325. res = register_netdev(dev);
  326. if (res)
  327. free_netdev(dev);
  328. return res;
  329. }
  330. int br_del_bridge(struct net *net, const char *name)
  331. {
  332. struct net_device *dev;
  333. int ret = 0;
  334. rtnl_lock();
  335. dev = __dev_get_by_name(net, name);
  336. if (dev == NULL)
  337. ret = -ENXIO; /* Could not find device */
  338. else if (!(dev->priv_flags & IFF_EBRIDGE)) {
  339. /* Attempt to delete non bridge device! */
  340. ret = -EPERM;
  341. }
  342. else if (dev->flags & IFF_UP) {
  343. /* Not shutdown yet. */
  344. ret = -EBUSY;
  345. }
  346. else
  347. br_dev_delete(dev, NULL);
  348. rtnl_unlock();
  349. return ret;
  350. }
  351. /* MTU of the bridge pseudo-device: ETH_DATA_LEN or the minimum of the ports */
  352. static int br_mtu_min(const struct net_bridge *br)
  353. {
  354. const struct net_bridge_port *p;
  355. int ret_mtu = 0;
  356. list_for_each_entry(p, &br->port_list, list)
  357. if (!ret_mtu || ret_mtu > p->dev->mtu)
  358. ret_mtu = p->dev->mtu;
  359. return ret_mtu ? ret_mtu : ETH_DATA_LEN;
  360. }
  361. void br_mtu_auto_adjust(struct net_bridge *br)
  362. {
  363. ASSERT_RTNL();
  364. /* if the bridge MTU was manually configured don't mess with it */
  365. if (br->mtu_set_by_user)
  366. return;
  367. /* change to the minimum MTU and clear the flag which was set by
  368. * the bridge ndo_change_mtu callback
  369. */
  370. dev_set_mtu(br->dev, br_mtu_min(br));
  371. br->mtu_set_by_user = false;
  372. }
  373. static void br_set_gso_limits(struct net_bridge *br)
  374. {
  375. unsigned int gso_max_size = GSO_MAX_SIZE;
  376. u16 gso_max_segs = GSO_MAX_SEGS;
  377. const struct net_bridge_port *p;
  378. list_for_each_entry(p, &br->port_list, list) {
  379. gso_max_size = min(gso_max_size, p->dev->gso_max_size);
  380. gso_max_segs = min(gso_max_segs, p->dev->gso_max_segs);
  381. }
  382. br->dev->gso_max_size = gso_max_size;
  383. br->dev->gso_max_segs = gso_max_segs;
  384. }
  385. /*
  386. * Recomputes features using slave's features
  387. */
  388. netdev_features_t br_features_recompute(struct net_bridge *br,
  389. netdev_features_t features)
  390. {
  391. struct net_bridge_port *p;
  392. netdev_features_t mask;
  393. if (list_empty(&br->port_list))
  394. return features;
  395. mask = features;
  396. features &= ~NETIF_F_ONE_FOR_ALL;
  397. list_for_each_entry(p, &br->port_list, list) {
  398. features = netdev_increment_features(features,
  399. p->dev->features, mask);
  400. }
  401. features = netdev_add_tso_features(features, mask);
  402. return features;
  403. }
  404. /* called with RTNL */
  405. int br_add_if(struct net_bridge *br, struct net_device *dev,
  406. struct netlink_ext_ack *extack)
  407. {
  408. struct net_bridge_port *p;
  409. int err = 0;
  410. unsigned br_hr, dev_hr;
  411. bool changed_addr;
  412. /* Don't allow bridging non-ethernet like devices, or DSA-enabled
  413. * master network devices since the bridge layer rx_handler prevents
  414. * the DSA fake ethertype handler to be invoked, so we do not strip off
  415. * the DSA switch tag protocol header and the bridge layer just return
  416. * RX_HANDLER_CONSUMED, stopping RX processing for these frames.
  417. */
  418. if ((dev->flags & IFF_LOOPBACK) ||
  419. dev->type != ARPHRD_ETHER || dev->addr_len != ETH_ALEN ||
  420. !is_valid_ether_addr(dev->dev_addr) ||
  421. netdev_uses_dsa(dev))
  422. return -EINVAL;
  423. /* No bridging of bridges */
  424. if (dev->netdev_ops->ndo_start_xmit == br_dev_xmit) {
  425. NL_SET_ERR_MSG(extack,
  426. "Can not enslave a bridge to a bridge");
  427. return -ELOOP;
  428. }
  429. /* Device has master upper dev */
  430. if (netdev_master_upper_dev_get(dev))
  431. return -EBUSY;
  432. /* No bridging devices that dislike that (e.g. wireless) */
  433. if (dev->priv_flags & IFF_DONT_BRIDGE) {
  434. NL_SET_ERR_MSG(extack,
  435. "Device does not allow enslaving to a bridge");
  436. return -EOPNOTSUPP;
  437. }
  438. p = new_nbp(br, dev);
  439. if (IS_ERR(p))
  440. return PTR_ERR(p);
  441. call_netdevice_notifiers(NETDEV_JOIN, dev);
  442. err = dev_set_allmulti(dev, 1);
  443. if (err)
  444. goto put_back;
  445. err = kobject_init_and_add(&p->kobj, &brport_ktype, &(dev->dev.kobj),
  446. SYSFS_BRIDGE_PORT_ATTR);
  447. if (err)
  448. goto err1;
  449. err = br_sysfs_addif(p);
  450. if (err)
  451. goto err2;
  452. err = br_netpoll_enable(p);
  453. if (err)
  454. goto err3;
  455. err = netdev_rx_handler_register(dev, br_handle_frame, p);
  456. if (err)
  457. goto err4;
  458. dev->priv_flags |= IFF_BRIDGE_PORT;
  459. err = netdev_master_upper_dev_link(dev, br->dev, NULL, NULL, extack);
  460. if (err)
  461. goto err5;
  462. err = nbp_switchdev_mark_set(p);
  463. if (err)
  464. goto err6;
  465. dev_disable_lro(dev);
  466. list_add_rcu(&p->list, &br->port_list);
  467. nbp_update_port_count(br);
  468. netdev_update_features(br->dev);
  469. br_hr = br->dev->needed_headroom;
  470. dev_hr = netdev_get_fwd_headroom(dev);
  471. if (br_hr < dev_hr)
  472. update_headroom(br, dev_hr);
  473. else
  474. netdev_set_rx_headroom(dev, br_hr);
  475. if (br_fdb_insert(br, p, dev->dev_addr, 0))
  476. netdev_err(dev, "failed insert local address bridge forwarding table\n");
  477. err = nbp_vlan_init(p);
  478. if (err) {
  479. netdev_err(dev, "failed to initialize vlan filtering on this port\n");
  480. goto err7;
  481. }
  482. spin_lock_bh(&br->lock);
  483. changed_addr = br_stp_recalculate_bridge_id(br);
  484. if (netif_running(dev) && netif_oper_up(dev) &&
  485. (br->dev->flags & IFF_UP))
  486. br_stp_enable_port(p);
  487. spin_unlock_bh(&br->lock);
  488. br_ifinfo_notify(RTM_NEWLINK, NULL, p);
  489. if (changed_addr)
  490. call_netdevice_notifiers(NETDEV_CHANGEADDR, br->dev);
  491. br_mtu_auto_adjust(br);
  492. br_set_gso_limits(br);
  493. kobject_uevent(&p->kobj, KOBJ_ADD);
  494. return 0;
  495. err7:
  496. list_del_rcu(&p->list);
  497. br_fdb_delete_by_port(br, p, 0, 1);
  498. nbp_update_port_count(br);
  499. err6:
  500. netdev_upper_dev_unlink(dev, br->dev);
  501. err5:
  502. dev->priv_flags &= ~IFF_BRIDGE_PORT;
  503. netdev_rx_handler_unregister(dev);
  504. err4:
  505. br_netpoll_disable(p);
  506. err3:
  507. sysfs_remove_link(br->ifobj, p->dev->name);
  508. err2:
  509. kobject_put(&p->kobj);
  510. p = NULL; /* kobject_put frees */
  511. err1:
  512. dev_set_allmulti(dev, -1);
  513. put_back:
  514. dev_put(dev);
  515. kfree(p);
  516. return err;
  517. }
  518. /* called with RTNL */
  519. int br_del_if(struct net_bridge *br, struct net_device *dev)
  520. {
  521. struct net_bridge_port *p;
  522. bool changed_addr;
  523. p = br_port_get_rtnl(dev);
  524. if (!p || p->br != br)
  525. return -EINVAL;
  526. /* Since more than one interface can be attached to a bridge,
  527. * there still maybe an alternate path for netconsole to use;
  528. * therefore there is no reason for a NETDEV_RELEASE event.
  529. */
  530. del_nbp(p);
  531. br_mtu_auto_adjust(br);
  532. br_set_gso_limits(br);
  533. spin_lock_bh(&br->lock);
  534. changed_addr = br_stp_recalculate_bridge_id(br);
  535. spin_unlock_bh(&br->lock);
  536. if (changed_addr)
  537. call_netdevice_notifiers(NETDEV_CHANGEADDR, br->dev);
  538. netdev_update_features(br->dev);
  539. return 0;
  540. }
  541. void br_port_flags_change(struct net_bridge_port *p, unsigned long mask)
  542. {
  543. struct net_bridge *br = p->br;
  544. if (mask & BR_AUTO_MASK)
  545. nbp_update_port_count(br);
  546. if (mask & BR_NEIGH_SUPPRESS)
  547. br_recalculate_neigh_suppress_enabled(br);
  548. }