代码改变世界

六 vxlan端口创建

2017-03-23 12:18  yrpapa  阅读(795)  评论(0)    收藏  举报

一 vxlan端口创建

static int ovs_vport_cmd_new(struct sk_buff *skb, struct genl_info *info)
{
    struct nlattr **a = info->attrs;
    struct ovs_header *ovs_header = info->userhdr;
    struct vport_parms parms;
    struct sk_buff *reply;
    struct vport *vport;
    struct datapath *dp;
    u32 port_no;
    int err;

    if (!a[OVS_VPORT_ATTR_NAME] || !a[OVS_VPORT_ATTR_TYPE] ||
        !a[OVS_VPORT_ATTR_UPCALL_PID])
        return -EINVAL;

    port_no = a[OVS_VPORT_ATTR_PORT_NO]
        ? nla_get_u32(a[OVS_VPORT_ATTR_PORT_NO]) : 0;
    if (port_no >= DP_MAX_PORTS)
        return -EFBIG;

    reply = ovs_vport_cmd_alloc_info();
    if (!reply)
        return -ENOMEM;

    ovs_lock();
restart:
    dp = get_dp(sock_net(skb->sk), ovs_header->dp_ifindex);
    err = -ENODEV;
    if (!dp)
        goto exit_unlock_free;

    if (port_no) {
        vport = ovs_vport_ovsl(dp, port_no);// 根据端口号,查找dp(网桥)关联的ports数组,找到vport结构体
        err = -EBUSY;
        if (vport)
            goto exit_unlock_free;
    } else {
        for (port_no = 1; ; port_no++) { // 查找空闲端口号
            if (port_no >= DP_MAX_PORTS) {
                err = -EFBIG;
                goto exit_unlock_free;
            }
            vport = ovs_vport_ovsl(dp, port_no); 
            if (!vport)
                break;
        }
    }

    parms.name = nla_data(a[OVS_VPORT_ATTR_NAME]);
    parms.type = nla_get_u32(a[OVS_VPORT_ATTR_TYPE]);
    parms.options = a[OVS_VPORT_ATTR_OPTIONS];
    parms.dp = dp;
    parms.port_no = port_no;
    parms.upcall_portids = a[OVS_VPORT_ATTR_UPCALL_PID];

    printk("datapath.c-->ovs_vport_cmd_new\n");

    vport = new_vport(&parms); // 创建端口
    err = PTR_ERR(vport);
    if (IS_ERR(vport)) {
        if (err == -EAGAIN)
            goto restart;
        goto exit_unlock_free;
    }

    err = ovs_vport_cmd_fill_info(vport, reply, info->snd_portid,
                      info->snd_seq, 0, OVS_VPORT_CMD_NEW);
    BUG_ON(err < 0);

    if (netdev_get_fwd_headroom(vport->dev) > dp->max_headroom)
        update_headroom(dp);
    else
        netdev_set_rx_headroom(vport->dev, dp->max_headroom);

    ovs_unlock();

    ovs_notify(&dp_vport_genl_family, &ovs_dp_vport_multicast_group, reply, info);
    return 0;

exit_unlock_free:
    ovs_unlock();
    kfree_skb(reply);
    return err;
}
static struct vport *new_vport(const struct vport_parms *parms)
{
    struct vport *vport;

    printk("datapath.c-->create[%s]\n", parms->name);

    vport = ovs_vport_add(parms);
    if (!IS_ERR(vport)) {
        struct datapath *dp = parms->dp;
        struct hlist_head *head = vport_hash_bucket(dp, vport->port_no);

        hlist_add_head_rcu(&vport->dp_hash_node, head); // 插入dp的ports
    }
    return vport;
}
struct vport *ovs_vport_add(const struct vport_parms *parms)
{
    struct vport_ops *ops;
    struct vport *vport;
        
    printk("vport.c-->create device[%s] entrypoint", parms->name);

    ops = ovs_vport_lookup(parms); // 根据参数查找创建端口的操作,对于vxlan来说,就用vport-vxlan.c的vxlan_create
    if (ops) {
        struct hlist_head *bucket;

        if (!try_module_get(ops->owner))
            return ERR_PTR(-EAFNOSUPPORT);

        vport = ops->create(parms);
        if (IS_ERR(vport)) {
            module_put(ops->owner);
            return vport;
        }   

        bucket = hash_bucket(ovs_dp_get_net(vport->dp),
                     ovs_vport_name(vport));
        hlist_add_head_rcu(&vport->hash_node, bucket);
        return vport;
    }   

    /* Unlock to attempt module load and return -EAGAIN if load
     * was successful as we need to restart the port addition
     * workflow.
     */
    ovs_unlock();
    request_module("vport-type-%d", parms->type);
    ovs_lock();

    if (!ovs_vport_lookup(parms))
        return ERR_PTR(-EAFNOSUPPORT);
    else
        return ERR_PTR(-EAGAIN);
}
static struct vport_ops ovs_vxlan_netdev_vport_ops = { 
    .type           = OVS_VPORT_TYPE_VXLAN,
    .create         = vxlan_create,
    .destroy        = ovs_netdev_tunnel_destroy,
    .get_options        = vxlan_get_options,
#ifndef USE_UPSTREAM_TUNNEL
    .fill_metadata_dst  = vxlan_fill_metadata_dst,
#endif
    .send           = vxlan_xmit,
};
static struct vport *vxlan_create(const struct vport_parms *parms)
{
    struct vport *vport;

    vport = vxlan_tnl_create(parms);
    if (IS_ERR(vport))
        return vport;

    return ovs_netdev_link(vport, parms->name); // 注册设备rx_handler函数
}
static struct vport *vxlan_tnl_create(const struct vport_parms *parms)
{
    struct net *net = ovs_dp_get_net(parms->dp);
    struct nlattr *options = parms->options;
    struct net_device *dev;
    struct vport *vport;
    struct nlattr *a;
    int err;
    struct vxlan_config conf = {
        .no_share = true,
        .flags = VXLAN_F_COLLECT_METADATA | VXLAN_F_UDP_ZERO_CSUM6_RX,
        /* Don't restrict the packets that can be sent by MTU */
        .mtu = IP_MAX_MTU,
    };

    if (!options) {
        err = -EINVAL;
        goto error;
    }

    a = nla_find_nested(options, OVS_TUNNEL_ATTR_DST_PORT);
    if (a && nla_len(a) == sizeof(u16)) {
        conf.dst_port = htons(nla_get_u16(a));
    } else {
        /* Require destination port from userspace. */
        err = -EINVAL;
        goto error;
    }

    vport = ovs_vport_alloc(0, &ovs_vxlan_netdev_vport_ops, parms);
    if (IS_ERR(vport))
        return vport;

    a = nla_find_nested(options, OVS_TUNNEL_ATTR_EXTENSION);
    if (a) {
        err = vxlan_configure_exts(vport, a, &conf);
        if (err) {
            ovs_vport_free(vport);
            goto error;
        }
    }

    rtnl_lock();
    dev = vxlan_dev_create(net, parms->name, NET_NAME_USER, &conf); //创建net_device设备 
if (IS_ERR(dev)) { rtnl_unlock(); ovs_vport_free(vport); return ERR_CAST(dev); } err = dev_change_flags(dev, dev->flags | IFF_UP); //会调用设备驱动的open函数 if (err < 0) { rtnl_delete_link(dev); rtnl_unlock(); ovs_vport_free(vport); goto error; } rtnl_unlock(); return vport; error: return ERR_PTR(err); }
struct net_device *rpl_vxlan_dev_create(struct net *net, const char *name,
                    u8 name_assign_type,
                    struct vxlan_config *conf)
{
    struct nlattr *tb[IFLA_MAX + 1];
    struct net_device *dev;
    int err;

    memset(&tb, 0, sizeof(tb));

    dev = rtnl_create_link(net, name, name_assign_type,
                   &vxlan_link_ops, tb);
    if (IS_ERR(dev))
        return dev;

    err = vxlan_dev_configure(net, dev, conf);
    if (err < 0) {
        free_netdev(dev);
        return ERR_PTR(err);
    }

    err = rtnl_configure_link(dev, NULL);
    if (err < 0) {
        LIST_HEAD(list_kill);

        vxlan_dellink(dev, &list_kill);
        unregister_netdevice_many(&list_kill);
        return ERR_PTR(err);
    }

    return dev;
}   
static int vxlan_dev_configure(struct net *src_net, struct net_device *dev,
                   struct vxlan_config *conf)
{
    struct vxlan_net *vn = net_generic(src_net, vxlan_net_id);
    struct vxlan_dev *vxlan = netdev_priv(dev), *tmp;
    struct vxlan_rdst *dst = &vxlan->default_dst;
    unsigned short needed_headroom = ETH_HLEN;
    int err;
    bool use_ipv6 = false;
    __be16 default_port = vxlan->cfg.dst_port;
    struct net_device *lowerdev = NULL;

    if (conf->flags & VXLAN_F_GPE) {
        if (conf->flags & ~VXLAN_F_ALLOWED_GPE)
            return -EINVAL;
        /* For now, allow GPE only together with COLLECT_METADATA.
         * This can be relaxed later; in such case, the other side
         * of the PtP link will have to be provided.
         */
        if (!(conf->flags & VXLAN_F_COLLECT_METADATA))
            return -EINVAL;

        vxlan_raw_setup(dev);
    } else {
        vxlan_ether_setup(dev);
    }

    vxlan->net = src_net;

    dst->remote_vni = conf->vni;

    memcpy(&dst->remote_ip, &conf->remote_ip, sizeof(conf->remote_ip));

    /* Unless IPv6 is explicitly requested, assume IPv4 */
    if (!dst->remote_ip.sa.sa_family)
        dst->remote_ip.sa.sa_family = AF_INET;

    if (dst->remote_ip.sa.sa_family == AF_INET6 ||
        vxlan->cfg.saddr.sa.sa_family == AF_INET6) {
        if (!IS_ENABLED(CONFIG_IPV6))
            return -EPFNOSUPPORT;
        use_ipv6 = true;
        vxlan->flags |= VXLAN_F_IPV6;
    }

    if (conf->label && !use_ipv6) {
        pr_info("label only supported in use with IPv6\n");
        return -EINVAL;
    }

    if (conf->remote_ifindex) {
        lowerdev = __dev_get_by_index(src_net, conf->remote_ifindex);
        dst->remote_ifindex = conf->remote_ifindex;

        if (!lowerdev) {
            pr_info("ifindex %d does not exist\n", dst->remote_ifindex);
            return -ENODEV;
        }

#if IS_ENABLED(CONFIG_IPV6)
        if (use_ipv6) {
            struct inet6_dev *idev = __in6_dev_get(lowerdev);
            if (idev && idev->cnf.disable_ipv6) {
                pr_info("IPv6 is disabled via sysctl\n");
                return -EPERM;
            }
        }
#endif

        if (!conf->mtu)
            dev->mtu = lowerdev->mtu - (use_ipv6 ? VXLAN6_HEADROOM : VXLAN_HEADROOM);

        needed_headroom = lowerdev->hard_header_len;
    }

    if (conf->mtu) {
        err = __vxlan_change_mtu(dev, lowerdev, dst, conf->mtu, false);
        if (err)
            return err;
    }

    if (use_ipv6 || conf->flags & VXLAN_F_COLLECT_METADATA)
        needed_headroom += VXLAN6_HEADROOM;
    else
        needed_headroom += VXLAN_HEADROOM;
    dev->needed_headroom = needed_headroom;

    memcpy(&vxlan->cfg, conf, sizeof(*conf));
    if (!vxlan->cfg.dst_port) {
        if (conf->flags & VXLAN_F_GPE)
            vxlan->cfg.dst_port = 4790; /* IANA assigned VXLAN-GPE port */
        else
            vxlan->cfg.dst_port = default_port;
    }
    vxlan->flags |= conf->flags;

    if (!vxlan->cfg.age_interval)
        vxlan->cfg.age_interval = FDB_AGE_DEFAULT;

    list_for_each_entry(tmp, &vn->vxlan_list, next) {
        if (tmp->cfg.vni == conf->vni &&
            (tmp->default_dst.remote_ip.sa.sa_family == AF_INET6 ||
             tmp->cfg.saddr.sa.sa_family == AF_INET6) == use_ipv6 &&
            tmp->cfg.dst_port == vxlan->cfg.dst_port &&
            (tmp->flags & VXLAN_F_RCV_FLAGS) ==
            (vxlan->flags & VXLAN_F_RCV_FLAGS))
        return -EEXIST;
    }

    dev->ethtool_ops = &vxlan_ethtool_ops;

    /* create an fdb entry for a valid default destination */
    if (!vxlan_addr_any(&vxlan->default_dst.remote_ip)) {
        err = vxlan_fdb_create(vxlan, all_zeros_mac,
                       &vxlan->default_dst.remote_ip,
                       NUD_REACHABLE|NUD_PERMANENT,
                       NLM_F_EXCL|NLM_F_CREATE,
                       vxlan->cfg.dst_port,
                       vxlan->default_dst.remote_vni,
                       vxlan->default_dst.remote_ifindex,
                       NTF_SELF);
        if (err)
            return err;
    }

    err = register_netdevice(dev);
    if (err) {
        vxlan_fdb_delete_default(vxlan);
        return err;
    }

    list_add(&vxlan->next, &vn->vxlan_list);

    return 0;
static int vxlan_open(struct net_device *dev)
{
    struct vxlan_dev *vxlan = netdev_priv(dev);
    int ret; 

    ret = vxlan_sock_add(vxlan);
    if (ret < 0) 
        return ret; 

    if (vxlan_addr_multicast(&vxlan->default_dst.remote_ip)) {
        ret = vxlan_igmp_join(vxlan);
        if (ret == -EADDRINUSE)
            ret = 0; 
        if (ret) {
            vxlan_sock_release(vxlan);
            return ret; 
        }
    }    

    if (vxlan->cfg.age_interval)
        mod_timer(&vxlan->age_timer, jiffies + FDB_AGE_INTERVAL);

    return ret; 
}
static int vxlan_sock_add(struct vxlan_dev *vxlan)
{
    bool ipv6 = vxlan->flags & VXLAN_F_IPV6;
    bool metadata = vxlan->flags & VXLAN_F_COLLECT_METADATA;
    int ret = 0;

    RCU_INIT_POINTER(vxlan->vn4_sock, NULL);
#if IS_ENABLED(CONFIG_IPV6)
    RCU_INIT_POINTER(vxlan->vn6_sock, NULL);
    if (ipv6 || metadata)
        ret = __vxlan_sock_add(vxlan, true);
#endif
    if (!ret && (!ipv6 || metadata))
        ret = __vxlan_sock_add(vxlan, false);
    if (ret < 0)
        vxlan_sock_release(vxlan);
    return ret;
}
static int __vxlan_sock_add(struct vxlan_dev *vxlan, bool ipv6)
{
    struct vxlan_net *vn = net_generic(vxlan->net, vxlan_net_id);
    struct vxlan_sock *vs = NULL;

    if (!vxlan->cfg.no_share) {
        spin_lock(&vn->sock_lock);
        vs = vxlan_find_sock(vxlan->net, ipv6 ? AF_INET6 : AF_INET,
                     vxlan->cfg.dst_port, vxlan->flags);
        if (vs && !atomic_add_unless(&vs->refcnt, 1, 0)) {
            spin_unlock(&vn->sock_lock);
            return -EBUSY;
        }
        spin_unlock(&vn->sock_lock);
    }
    if (!vs)
        vs = vxlan_socket_create(vxlan->net, ipv6,
                     vxlan->cfg.dst_port, vxlan->flags);
    if (IS_ERR(vs))
        return PTR_ERR(vs);
#if IS_ENABLED(CONFIG_IPV6)
    if (ipv6)
        rcu_assign_pointer(vxlan->vn6_sock, vs);
    else
#endif
        rcu_assign_pointer(vxlan->vn4_sock, vs);
    vxlan_vs_add_dev(vs, vxlan); // vxlan加入???列表
    return 0;
}
static struct vxlan_sock *vxlan_socket_create(struct net *net, bool ipv6,
                          __be16 port, u32 flags)
{
    struct vxlan_net *vn = net_generic(net, vxlan_net_id);
    struct vxlan_sock *vs;
    struct socket *sock;
    unsigned int h;
    struct udp_tunnel_sock_cfg tunnel_cfg;

    vs = kzalloc(sizeof(*vs), GFP_KERNEL);
    if (!vs)
        return ERR_PTR(-ENOMEM);

    for (h = 0; h < VNI_HASH_SIZE; ++h)
        INIT_HLIST_HEAD(&vs->vni_list[h]);

    sock = vxlan_create_sock(net, ipv6, port, flags);
    if (IS_ERR(sock)) {
        pr_info("Cannot bind port %d, err=%ld\n", ntohs(port),
            PTR_ERR(sock));
        kfree(vs);
        return ERR_CAST(sock);
    }

    vs->sock = sock;
    atomic_set(&vs->refcnt, 1);
    vs->flags = (flags & VXLAN_F_RCV_FLAGS);

#ifdef HAVE_UDP_OFFLOAD
    vs->udp_offloads.port = port;
    vs->udp_offloads.callbacks.gro_receive  = vxlan_gro_receive;
    vs->udp_offloads.callbacks.gro_complete = vxlan_gro_complete;
#endif

    spin_lock(&vn->sock_lock);
    hlist_add_head_rcu(&vs->hlist, vs_head(net, port));
static struct vxlan_sock *vxlan_socket_create(struct net *net, bool ipv6,
                          __be16 port, u32 flags)
{
    struct vxlan_net *vn = net_generic(net, vxlan_net_id);
    struct vxlan_sock *vs; 
    struct socket *sock;
    unsigned int h;
    struct udp_tunnel_sock_cfg tunnel_cfg;

    vs = kzalloc(sizeof(*vs), GFP_KERNEL);
    if (!vs)
        return ERR_PTR(-ENOMEM);

    for (h = 0; h < VNI_HASH_SIZE; ++h)
        INIT_HLIST_HEAD(&vs->vni_list[h]);

    sock = vxlan_create_sock(net, ipv6, port, flags);
    if (IS_ERR(sock)) {
        pr_info("Cannot bind port %d, err=%ld\n", ntohs(port),
            PTR_ERR(sock));
        kfree(vs);
        return ERR_CAST(sock);
    }

    vs->sock = sock;
    atomic_set(&vs->refcnt, 1);
    vs->flags = (flags & VXLAN_F_RCV_FLAGS);

#ifdef HAVE_UDP_OFFLOAD
    vs->udp_offloads.port = port;
    vs->udp_offloads.callbacks.gro_receive  = vxlan_gro_receive;
    vs->udp_offloads.callbacks.gro_complete = vxlan_gro_complete;
#endif

    spin_lock(&vn->sock_lock);
static struct socket *vxlan_create_sock(struct net *net, bool ipv6,
                    __be16 port, u32 flags)
{
    struct socket *sock;
    struct udp_port_cfg udp_conf;
    int err;

    memset(&udp_conf, 0, sizeof(udp_conf));

    if (ipv6) {
        udp_conf.family = AF_INET6;
        udp_conf.use_udp6_rx_checksums =
            !(flags & VXLAN_F_UDP_ZERO_CSUM6_RX);
        udp_conf.ipv6_v6only = 1;
    } else {
        udp_conf.family = AF_INET;
    }

    udp_conf.local_udp_port = port;

    /* Open UDP socket */
    err = udp_sock_create(net, &udp_conf, &sock);
    if (err < 0)
        return ERR_PTR(err);

    return sock;
}
static inline int udp_sock_create(struct net *net,
                                  struct udp_port_cfg *cfg,
                                  struct socket **sockp)
{   
        if (cfg->family == AF_INET)
                return udp_sock_create4(net, cfg, sockp);

        if (cfg->family == AF_INET6)
                return udp_sock_create6(net, cfg, sockp);

        return -EPFNOSUPPORT;
}
int rpl_udp_sock_create4(struct net *net, struct udp_port_cfg *cfg,
             struct socket **sockp)
{
    int err;
    struct socket *sock = NULL;
    struct sockaddr_in udp_addr;

    err = sock_create_kern(net, AF_INET, SOCK_DGRAM, 0, &sock);
    if (err < 0)
        goto error;

    udp_addr.sin_family = AF_INET;
    udp_addr.sin_addr = cfg->local_ip;
    udp_addr.sin_port = cfg->local_udp_port;
    err = kernel_bind(sock, (struct sockaddr *)&udp_addr,
            sizeof(udp_addr));
    if (err < 0)
        goto error;

    if (cfg->peer_udp_port) {
        udp_addr.sin_family = AF_INET;
        udp_addr.sin_addr = cfg->peer_ip;
        udp_addr.sin_port = cfg->peer_udp_port;
        err = kernel_connect(sock, (struct sockaddr *)&udp_addr,
                sizeof(udp_addr), 0);
        if (err < 0)
            goto error;
    }
#ifdef HAVE_SK_NO_CHECK_TX
    sock->sk->sk_no_check_tx = !cfg->use_udp_checksums;
#endif
    *sockp = sock;
    return 0;

error: 
    if (sock) {
        kernel_sock_shutdown(sock, SHUT_RDWR);
        sock_release(sock);
    }
    *sockp = NULL;
    return err;
}

设置encap_rcv方法(vxlan在四层收到数据包的处理方法)

void rpl_setup_udp_tunnel_sock(struct net *net, struct socket *sock,
                   struct udp_tunnel_sock_cfg *cfg)
{
    struct sock *sk = sock->sk;
    
    /* Disable multicast loopback */
    inet_sk(sk)->mc_loop = 0;
    
    rcu_assign_sk_user_data(sk, cfg->sk_user_data);

    udp_sk(sk)->encap_type = cfg->encap_type;
    udp_sk(sk)->encap_rcv = cfg->encap_rcv;
#if LINUX_VERSION_CODE >= KERNEL_VERSION(3,9,0)
    udp_sk(sk)->encap_destroy = cfg->encap_destroy;
#endif
#ifdef HAVE_UDP_TUNNEL_SOCK_CFG_GRO_RECEIVE
    udp_sk(sk)->gro_receive = cfg->gro_receive; 
    udp_sk(sk)->gro_complete = cfg->gro_complete; 
#endif          
                
    udp_tunnel_encap_enable(sock);
}       

创建vxlan端口,在内核中主要做了以下几件事情:

1、创建net_device设备对象,通过该对象可以获得vxlan_dev以及vxlan配置信息;

2、创建vxlan socket,并设置encap_rcv函数;

3、注册vxlan offload到内核,实现vxlan报文的gro,根据UDP端口号来判断;


内核收到报文后,gro流程中,udp gro receive后,根据UDP的端口可以找到vxlan gro receive,该函数作为内外层的桥梁实现报文的gro,报文聚合后通过netif_receive_skb_internal上送到协议栈,协议栈一直处理到udp socket收包,然后交给vxlan_udp_encap_recv函数进行处理。