diff --git a/include/net/ip_vs.h b/include/net/ip_vs.h index b2680d368..68eb85e49 100644 --- a/include/net/ip_vs.h +++ b/include/net/ip_vs.h @@ -32,7 +32,7 @@ #include #endif #include /* Netw namespace */ - +#include #define IP_VS_HDR_INVERSE 1 #define IP_VS_HDR_ICMP 2 @@ -1712,6 +1712,15 @@ struct bpf_sym_addrs { const struct file_operations *bpf_prog_fops; }; +struct net *ip_vs_skb_net(struct sk_buff *skb); +enum { + IPVS_ORIGIN_MODE, + IPVS_BPF_MODE, + IPVS_SHARE_NS_MODE, + IPVS_MAX_MODE +}; +extern unsigned int ipvs_mode; +extern struct net init_net; extern struct bpf_sym_addrs resolve_addrs; extern struct bpf_map *conntrack_map; extern bool bpf_mode_on; @@ -1738,5 +1747,4 @@ extern struct cidrs __rcu *non_masq_cidrs; #define IP_VS_SVC_TAB_BITS 8 #define IP_VS_SVC_TAB_SIZE (1 << IP_VS_SVC_TAB_BITS) extern struct hlist_head ip_vs_svc_table[IP_VS_SVC_TAB_SIZE]; - #endif /* _NET_IP_VS_H */ diff --git a/net/netfilter/ipvs/ip_vs_conn.c b/net/netfilter/ipvs/ip_vs_conn.c index 4c2a3d674..cff457cc0 100644 --- a/net/netfilter/ipvs/ip_vs_conn.c +++ b/net/netfilter/ipvs/ip_vs_conn.c @@ -54,15 +54,26 @@ static int ip_vs_conn_tab_bits = CONFIG_IP_VS_TAB_BITS; module_param_named(conn_tab_bits, ip_vs_conn_tab_bits, int, 0444); MODULE_PARM_DESC(conn_tab_bits, "Set connections' hash size"); -bool bpf_mode_on; -module_param_named(mode, bpf_mode_on, bool, 0444); -MODULE_PARM_DESC(mode, "set bpf mode in IPVS"); -EXPORT_SYMBOL_GPL(bpf_mode_on); +unsigned int ipvs_mode; +module_param_named(mode, ipvs_mode, uint, 0444); +MODULE_PARM_DESC(mode, "set mode in IPVS"); +EXPORT_SYMBOL_GPL(ipvs_mode); /* size and mask values */ int ip_vs_conn_tab_size __read_mostly; static int ip_vs_conn_tab_mask __read_mostly; +/* retrieve origin net in skb for xmit + * local-out: ip_queue_xmit->skb_dst_set_noref + * local-in: ip_route_input_slow set it + */ +struct net *ip_vs_skb_net(struct sk_buff *skb) +{ + if (skb_dst(skb)) + return dev_net(skb_dst(skb)->dev); + return NULL; +} + /* * Connection hash table: for input and output packets lookups of IPVS */ @@ -241,7 +252,7 @@ static void ip_vs_unlink_bpf(struct ip_vs_conn *cp) struct bpf_map *map; int err = 0; - if (!bpf_mode_on) + if (ipvs_mode != IPVS_BPF_MODE) return; k.sip = cp->caddr.ip; @@ -319,7 +330,7 @@ static inline bool ip_vs_conn_unlink(struct ip_vs_conn *cp) hlist_del_rcu(&cp->c_list); cp->flags &= ~IP_VS_CONN_F_HASHED; ret = true; - if (bpf_mode_on) + if (ipvs_mode == IPVS_BPF_MODE) ip_vs_unlink_bpf(cp); } } else @@ -1113,7 +1124,7 @@ static bool ip_vs_conn_new_bpf(struct ip_vs_dest *dest, BUILD_BUG_ON(sizeof(atomic_t) != 4); - if (!bpf_mode_on) + if (ipvs_mode != IPVS_BPF_MODE) return true; svc = rcu_dereference(dest->svc); @@ -1251,7 +1262,7 @@ ip_vs_conn_new(const struct ip_vs_conn_param *p, int dest_af, p->protocol); int skip = 0; - if (bpf_mode_on) { + if (ipvs_mode == IPVS_BPF_MODE) { if (!ip_vs_conn_new_bpf(dest, flags, p, &skip)) return NULL; } @@ -1270,7 +1281,7 @@ ip_vs_conn_new(const struct ip_vs_conn_param *p, int dest_af, cp->protocol = p->protocol; ip_vs_addr_set(p->af, &cp->caddr, p->caddr); cp->cport = p->cport; - if (bpf_mode_on) + if (ipvs_mode == IPVS_BPF_MODE) cp->skip_bpf = skip; /* proto should only be IPPROTO_IP if p->vaddr is a fwmark */ ip_vs_addr_set(p->protocol == IPPROTO_IP ? AF_UNSPEC : p->af, @@ -1795,7 +1806,7 @@ int __init ip_vs_conn_init(void) spin_lock_init(&__ip_vs_conntbl_lock_array[idx].l); } - if (bpf_mode_on) { + if (ipvs_mode == IPVS_BPF_MODE) { for (idx = 0; idx < BPF_CONN_LOCKS; idx++) spin_lock_init(&bpf_conntrack_locks[idx]); } diff --git a/net/netfilter/ipvs/ip_vs_core.c b/net/netfilter/ipvs/ip_vs_core.c index 05ea6446c..61f3f30df 100644 --- a/net/netfilter/ipvs/ip_vs_core.c +++ b/net/netfilter/ipvs/ip_vs_core.c @@ -747,11 +747,18 @@ static int ip_route_me_harder2(struct net *net, struct sk_buff *skb, static int ip_vs_route_me_harder(struct netns_ipvs *ipvs, int af, struct sk_buff *skb, unsigned int hooknum) { - if (!bpf_mode_on && !sysctl_snat_reroute(ipvs)) + struct net *net; + + if (ipvs_mode != IPVS_BPF_MODE && !sysctl_snat_reroute(ipvs)) return 0; /* Reroute replies only to remote clients (FORWARD and LOCAL_OUT) */ if (NF_INET_LOCAL_IN == hooknum) return 0; + + net = ipvs->net; + if (ipvs_mode == IPVS_SHARE_NS_MODE) + net = ip_vs_skb_net(skb); + #ifdef CONFIG_IP_VS_IPV6 if (af == AF_INET6) { struct dst_entry *dst = skb_dst(skb); @@ -761,9 +768,9 @@ static int ip_vs_route_me_harder(struct netns_ipvs *ipvs, int af, return 1; } else #endif - if (!bpf_mode_on) { + if (ipvs_mode != IPVS_BPF_MODE) { if (!(skb_rtable(skb)->rt_flags & RTCF_LOCAL) && - ip_route_me_harder(ipvs->net, skb, RTN_LOCAL) != 0) + ip_route_me_harder(net, skb, RTN_LOCAL) != 0) return 1; } else { if (ip_route_me_harder2(ipvs->net, skb, RTN_LOCAL) != 0) @@ -930,12 +937,12 @@ static int handle_response_icmp(int af, struct sk_buff *skb, else ip_vs_update_conntrack(skb, cp, 0); - if (bpf_mode_on) + if (ipvs_mode == IPVS_BPF_MODE) (*(resolve_addrs.ip_finish_output))(cp->ipvs->net, skb->sk, skb); ignore_cp: - if (bpf_mode_on) + if (ipvs_mode == IPVS_BPF_MODE) verdict = NF_STOLEN; else verdict = NF_ACCEPT; @@ -1354,14 +1361,14 @@ handle_response(int af, struct sk_buff *skb, struct ip_vs_proto_data *pd, else ip_vs_update_conntrack(skb, cp, 0); - if (bpf_mode_on) + if (ipvs_mode == IPVS_BPF_MODE) (*(resolve_addrs.ip_finish_output))(cp->ipvs->net, skb->sk, skb); ip_vs_conn_put(cp); LeaveFunction(11); - if (bpf_mode_on) + if (ipvs_mode == IPVS_BPF_MODE) return NF_STOLEN; else return NF_ACCEPT; @@ -1372,6 +1379,12 @@ drop: return NF_STOLEN; } +static void switch_netns(struct netns_ipvs **ipvs, struct sk_buff *skb) +{ + if (ipvs_mode == IPVS_SHARE_NS_MODE) + *ipvs = net_ipvs(&init_net); +} + /* * Check if outgoing packet belongs to the established ip_vs_conn. * bpf: previously, local-in, forward, and local-out may call here! @@ -1388,6 +1401,8 @@ ip_vs_out(struct netns_ipvs *ipvs, unsigned int hooknum, struct sk_buff *skb, in EnterFunction(11); + switch_netns(&ipvs, skb); + /* Already marked as IPVS request or reply? */ if (skb->ipvs_property) return NF_ACCEPT; @@ -1402,7 +1417,7 @@ ip_vs_out(struct netns_ipvs *ipvs, unsigned int hooknum, struct sk_buff *skb, in } /* In bpf mode, this is null */ - if (!bpf_mode_on && unlikely(!skb_dst(skb))) + if (ipvs_mode != IPVS_BPF_MODE && unlikely(!skb_dst(skb))) return NF_ACCEPT; if (!ipvs->enable) @@ -1631,6 +1646,8 @@ ip_vs_in_icmp(struct netns_ipvs *ipvs, struct sk_buff *skb, int *related, unsigned int offset, offset2, ihl, verdict; bool ipip, new_cp = false; + switch_netns(&ipvs, skb); + *related = 1; /* reassemble IP fragments */ @@ -1915,6 +1932,8 @@ ip_vs_in(struct netns_ipvs *ipvs, unsigned int hooknum, struct sk_buff *skb, int int conn_reuse_mode; struct sock *sk; + switch_netns(&ipvs, skb); + /* Already marked as IPVS request or reply? */ if (skb->ipvs_property) return NF_ACCEPT; @@ -1926,7 +1945,7 @@ ip_vs_in(struct netns_ipvs *ipvs, unsigned int hooknum, struct sk_buff *skb, int */ if (unlikely((skb->pkt_type != PACKET_HOST && hooknum != NF_INET_LOCAL_OUT) || - (!bpf_mode_on && !skb_dst(skb)))) { + (ipvs_mode != IPVS_BPF_MODE && !skb_dst(skb)))) { ip_vs_fill_iph_skb(af, skb, false, &iph); IP_VS_DBG_BUF(12, "packet type=%d proto=%d daddr=%s" " ignored in hook %u\n", @@ -1995,7 +2014,9 @@ ip_vs_in(struct netns_ipvs *ipvs, unsigned int hooknum, struct sk_buff *skb, int * as zero rs may be killed already. * Only enable in bpf mode currently. Shall promote to IPVS mode later. */ - if (no_route_to_host_fix && bpf_mode_on && cp && cp->dest && + if (no_route_to_host_fix && + ipvs_mode == IPVS_BPF_MODE && + cp && cp->dest && unlikely(!atomic_read(&cp->dest->weight)) && is_new_conn(skb, &iph) && !iph.fragoffs && conn_reuse_mode == 0) { @@ -2118,7 +2139,8 @@ ip_vs_remote_request4(void *priv, struct sk_buff *skb, * the defrag may impact performance greatly! Lukily, this is not the * case for us! */ - if (bpf_mode_on && unlikely(ip_is_fragment(ip_hdr(skb)))) { + if (ipvs_mode == IPVS_BPF_MODE && + unlikely(ip_is_fragment(ip_hdr(skb)))) { if (ip_vs_gather_frags(net_ipvs(state->net), skb, IP_DEFRAG_VS_IN)) /* return 0 will call skb_free in nf_hook */ @@ -2354,6 +2376,113 @@ static const struct nf_hook_ops ip_vs_bpf_ops[] = { */ }; +static const struct nf_hook_ops ip_vs_ns_ops[] = { + /* After packet filtering, change source only for VS/NAT */ + { + .hook = ip_vs_reply4, + .pf = NFPROTO_IPV4, + .hooknum = NF_INET_LOCAL_IN, + .priority = NF_IP_PRI_NAT_SRC - 2, + }, + { + .hook = ip_vs_remote_request4, + .pf = NFPROTO_IPV4, + .hooknum = NF_INET_LOCAL_IN, + .priority = NF_IP_PRI_NAT_SRC - 1, + }, + + /* Delete IPVS nf local_out hook to handle response packet. + * Consider following steps: + * 1. curl vip:vport on a vm. packet is (nodeip:tmport->vip:vport) + * 2. Ipvs does DNAT and choose a POD on this vm. + * packet is (nodeip:tmpport->rsip:rsport) + * 3. The POD replies. packet is (rsip:rsport -> nodeip:tmpport) + * In nf local-out, ipvs ip_vs_local_reply4 + * does reverse DNAT, and modifies the packet to be + * (VIP:VPORT->nodeip:tmpport) + * 4. The packet go out of the POD's ENI to the iaas switch. + * 5. Iaas switch will drop the packet as it expects the source + * to be the ENI's ip. + * Any side effect to delete the hook? + * If a client out of the cluster accesses the service on a cvm, + * and the cvm choose a process runs on default net ns as the target, + * Break!. However, It doesn't matter as we haven't such case. + */ + + /* After mangle, schedule and forward local requests */ + { + .hook = ip_vs_local_request4, + .pf = NFPROTO_IPV4, + .hooknum = NF_INET_LOCAL_OUT, + .priority = NF_IP_PRI_NAT_DST + 2, + }, + /* After packet filtering (but before ip_vs_out_icmp), catch icmp + * destined for 0.0.0.0/0, which is for incoming IPVS connections + */ + { + .hook = ip_vs_forward_icmp, + .pf = NFPROTO_IPV4, + .hooknum = NF_INET_FORWARD, + .priority = 99, + }, + /* After packet filtering, change source only for VS/NAT */ + { + .hook = ip_vs_reply4, + .pf = NFPROTO_IPV4, + .hooknum = NF_INET_FORWARD, + .priority = 100, + }, +#ifdef CONFIG_IP_VS_IPV6 + /* After packet filtering, change source only for VS/NAT */ + { + .hook = ip_vs_reply6, + .pf = NFPROTO_IPV6, + .hooknum = NF_INET_LOCAL_IN, + .priority = NF_IP6_PRI_NAT_SRC - 2, + }, + /* After packet filtering, forward packet through VS/DR, VS/TUN, + * or VS/NAT(change destination), so that filtering rules can be + * applied to IPVS + */ + { + .hook = ip_vs_remote_request6, + .pf = NFPROTO_IPV6, + .hooknum = NF_INET_LOCAL_IN, + .priority = NF_IP6_PRI_NAT_SRC - 1, + }, + /* Before ip_vs_in, change source only for VS/NAT */ + { + .hook = ip_vs_local_reply6, + .pf = NFPROTO_IPV6, + .hooknum = NF_INET_LOCAL_OUT, + .priority = NF_IP6_PRI_NAT_DST + 1, + }, + /* After mangle, schedule and forward local requests */ + { + .hook = ip_vs_local_request6, + .pf = NFPROTO_IPV6, + .hooknum = NF_INET_LOCAL_OUT, + .priority = NF_IP6_PRI_NAT_DST + 2, + }, + /* After packet filtering (but before ip_vs_out_icmp), catch icmp + * destined for 0.0.0.0/0, which is for incoming IPVS connections + */ + { + .hook = ip_vs_forward_icmp_v6, + .pf = NFPROTO_IPV6, + .hooknum = NF_INET_FORWARD, + .priority = 99, + }, + /* After packet filtering, change source only for VS/NAT */ + { + .hook = ip_vs_reply6, + .pf = NFPROTO_IPV6, + .hooknum = NF_INET_FORWARD, + .priority = 100, + }, +#endif +}; + /* * Initialize IP Virtual Server netns mem. */ @@ -2362,6 +2491,8 @@ static int __net_init __ip_vs_init(struct net *net) struct netns_ipvs *ipvs; int ret; + if (ipvs_mode >= IPVS_MAX_MODE) + return -EINVAL; ipvs = net_generic(net, ip_vs_net_id); if (ipvs == NULL) return -ENOMEM; @@ -2392,12 +2523,19 @@ static int __net_init __ip_vs_init(struct net *net) if (ip_vs_sync_net_init(ipvs) < 0) goto sync_fail; - if (!bpf_mode_on) { + if (ipvs_mode == IPVS_ORIGIN_MODE) { ret = nf_register_net_hooks(net, ip_vs_ops, ARRAY_SIZE(ip_vs_ops)); if (ret < 0) goto hook_fail; - } else { + } + if (ipvs_mode == IPVS_SHARE_NS_MODE) { + ret = nf_register_net_hooks(net, ip_vs_ns_ops, + ARRAY_SIZE(ip_vs_ns_ops)); + if (ret < 0) + goto hook_fail; + } + if (ipvs_mode == IPVS_BPF_MODE) { ret = nf_register_net_hooks(net, ip_vs_bpf_ops, ARRAY_SIZE(ip_vs_bpf_ops)); if (ret < 0) @@ -2418,7 +2556,6 @@ static int __net_init __ip_vs_init(struct net *net) goto hook_fail; } } - return 0; /* * Error handling @@ -2445,10 +2582,14 @@ static void __net_exit __ip_vs_cleanup(struct net *net) { struct netns_ipvs *ipvs = net_ipvs(net); - if (!bpf_mode_on) - nf_unregister_net_hooks(net, ip_vs_ops, - ARRAY_SIZE(ip_vs_ops)); - else { + if (ipvs_mode == IPVS_ORIGIN_MODE) + nf_unregister_net_hooks(net, ip_vs_ops, ARRAY_SIZE(ip_vs_ops)); + + if (ipvs_mode == IPVS_SHARE_NS_MODE) + nf_unregister_net_hooks(net, ip_vs_ns_ops, + ARRAY_SIZE(ip_vs_ns_ops)); + + if (ipvs_mode == IPVS_BPF_MODE) { nf_unregister_net_hooks(net, ip_vs_bpf_ops, ARRAY_SIZE(ip_vs_bpf_ops)); free_percpu(ipvs->bpf_stat); @@ -2494,7 +2635,8 @@ struct bpf_sym_addrs resolve_addrs; static int __init ip_vs_init(void) { int ret; - if (bpf_mode_on) { + + if (ipvs_mode == IPVS_BPF_MODE) { resolve_addrs.ip_finish_output = (output_t)kallsyms_lookup_name("ip_finish_output"); if (!resolve_addrs.ip_finish_output) { @@ -2530,7 +2672,7 @@ static int __init ip_vs_init(void) } } - pr_info("bpf_mode_on is %d\n", bpf_mode_on); + pr_info("ipvs_mode is %d\n", ipvs_mode); ret = ip_vs_control_init(); if (ret < 0) { pr_err("can't setup control.\n"); @@ -2584,7 +2726,7 @@ static void __exit ip_vs_cleanup(void) ip_vs_conn_cleanup(); ip_vs_protocol_cleanup(); ip_vs_control_cleanup(); - if (bpf_mode_on) + if (ipvs_mode == IPVS_BPF_MODE) ip_vs_bpf_put(); pr_info("ipvs unloaded.\n"); } diff --git a/net/netfilter/ipvs/ip_vs_ctl.c b/net/netfilter/ipvs/ip_vs_ctl.c index f9eacf700..aca3d5ffc 100644 --- a/net/netfilter/ipvs/ip_vs_ctl.c +++ b/net/netfilter/ipvs/ip_vs_ctl.c @@ -1191,7 +1191,7 @@ ip_vs_add_service(struct netns_ipvs *ipvs, struct ip_vs_service_user_kern *u, ip_vs_use_count_inc(); /* in bpf mode, avoid loopback traffic */ - if (bpf_mode_on && strcmp(u->sched_name, "wrr") != 0 && + if (ipvs_mode == IPVS_BPF_MODE && strcmp(u->sched_name, "wrr") != 0 && strcmp(u->sched_name, "rr") != 0 && strcmp(u->sched_name, "lc") != 0 && strcmp(u->sched_name, "wlc") != 0) { diff --git a/net/netfilter/ipvs/ip_vs_lc.c b/net/netfilter/ipvs/ip_vs_lc.c index 196368d10..deac4aa17 100644 --- a/net/netfilter/ipvs/ip_vs_lc.c +++ b/net/netfilter/ipvs/ip_vs_lc.c @@ -47,7 +47,8 @@ ip_vs_lc_schedule(struct ip_vs_service *svc, const struct sk_buff *skb, list_for_each_entry_rcu(dest, &svc->destinations, n_list) { if ((dest->flags & IP_VS_DEST_F_OVERLOAD) || atomic_read(&dest->weight) == 0 || - (bpf_mode_on && dest->addr.ip == iph->saddr.ip)) + (ipvs_mode == IPVS_BPF_MODE && + dest->addr.ip == iph->saddr.ip)) continue; doh = ip_vs_dest_conn_overhead(dest); if (!least || doh < loh) { diff --git a/net/netfilter/ipvs/ip_vs_rr.c b/net/netfilter/ipvs/ip_vs_rr.c index 92884d950..ff6c18c1f 100644 --- a/net/netfilter/ipvs/ip_vs_rr.c +++ b/net/netfilter/ipvs/ip_vs_rr.c @@ -76,7 +76,7 @@ ip_vs_rr_schedule(struct ip_vs_service *svc, const struct sk_buff *skb, if (!(dest->flags & IP_VS_DEST_F_OVERLOAD) && atomic_read(&dest->weight) > 0) { /* HIT */ - if (!bpf_mode_on) + if (ipvs_mode != IPVS_BPF_MODE) goto out; else if (dest->addr.ip != iph->saddr.ip) goto out; diff --git a/net/netfilter/ipvs/ip_vs_wlc.c b/net/netfilter/ipvs/ip_vs_wlc.c index eb3d97a15..301dc80ee 100644 --- a/net/netfilter/ipvs/ip_vs_wlc.c +++ b/net/netfilter/ipvs/ip_vs_wlc.c @@ -56,8 +56,9 @@ ip_vs_wlc_schedule(struct ip_vs_service *svc, const struct sk_buff *skb, list_for_each_entry_rcu(dest, &svc->destinations, n_list) { if (!(dest->flags & IP_VS_DEST_F_OVERLOAD) && atomic_read(&dest->weight) > 0) { - if (!bpf_mode_on || - (bpf_mode_on && dest->addr.ip != iph->saddr.ip)) { + if (ipvs_mode != IPVS_BPF_MODE || + (ipvs_mode == IPVS_BPF_MODE && + dest->addr.ip != iph->saddr.ip)) { least = dest; loh = ip_vs_dest_conn_overhead(least); goto nextstage; @@ -74,7 +75,8 @@ ip_vs_wlc_schedule(struct ip_vs_service *svc, const struct sk_buff *skb, /* in bpf mode, avoid loopback traffic */ list_for_each_entry_continue_rcu(dest, &svc->destinations, n_list) { if ((dest->flags & IP_VS_DEST_F_OVERLOAD) || - (bpf_mode_on && dest->addr.ip == iph->saddr.ip)) + (ipvs_mode == IPVS_BPF_MODE && + dest->addr.ip == iph->saddr.ip)) continue; doh = ip_vs_dest_conn_overhead(dest); if ((__s64)loh * atomic_read(&dest->weight) > diff --git a/net/netfilter/ipvs/ip_vs_wrr.c b/net/netfilter/ipvs/ip_vs_wrr.c index 545a716b4..2a354be74 100644 --- a/net/netfilter/ipvs/ip_vs_wrr.c +++ b/net/netfilter/ipvs/ip_vs_wrr.c @@ -185,7 +185,7 @@ ip_vs_wrr_schedule(struct ip_vs_service *svc, const struct sk_buff *skb, n_list) { if (!(dest->flags & IP_VS_DEST_F_OVERLOAD) && atomic_read(&dest->weight) >= mark->cw) { - if (!bpf_mode_on) + if (ipvs_mode != IPVS_BPF_MODE) goto found; else if (dest->addr.ip != iph->saddr.ip) goto found; diff --git a/net/netfilter/ipvs/ip_vs_xmit.c b/net/netfilter/ipvs/ip_vs_xmit.c index 59912b8f3..f03f700c0 100644 --- a/net/netfilter/ipvs/ip_vs_xmit.c +++ b/net/netfilter/ipvs/ip_vs_xmit.c @@ -316,7 +316,13 @@ __ip_vs_get_out_rt(struct netns_ipvs *ipvs, int skb_af, struct sk_buff *skb, int mtu; int local, noref = 1; - if (dest) { + if (ipvs_mode == IPVS_SHARE_NS_MODE) + net = ip_vs_skb_net(skb); + if (!net) + return -1; + + /* when share netns, the cache will error */ + if (dest && ipvs_mode != IPVS_SHARE_NS_MODE) { dest_dst = __ip_vs_dst_check(dest); if (likely(dest_dst)) rt = (struct rtable *) dest_dst->dst_cache; @@ -365,7 +371,7 @@ __ip_vs_get_out_rt(struct netns_ipvs *ipvs, int skb_af, struct sk_buff *skb, /* In bpf mode, this check always return false.Don't call it to avoid * access of skb->dst */ - if (!bpf_mode_on && + if (ipvs_mode != IPVS_BPF_MODE && unlikely(crosses_local_route_boundary(skb_af, skb, rt_mode, local))) { IP_VS_DBG_RL("We are crossing local and non-local addresses" @@ -373,7 +379,10 @@ __ip_vs_get_out_rt(struct netns_ipvs *ipvs, int skb_af, struct sk_buff *skb, goto err_put; } - if (unlikely(local)) { + /* traffic to local address shall route to lo dev + * so that traffic from a POD can choose itself as rs. + */ + if (ipvs_mode != IPVS_SHARE_NS_MODE && unlikely(local)) { /* skb to local stack, preserve old route */ if (!noref) ip_rt_put(rt); @@ -407,7 +416,7 @@ __ip_vs_get_out_rt(struct netns_ipvs *ipvs, int skb_af, struct sk_buff *skb, skb_dst_set(skb, &rt->dst); /* In bpf mode, like ip_output, set output dev */ - if (bpf_mode_on) + if (ipvs_mode == IPVS_BPF_MODE) skb->dev = skb_dst(skb)->dev; return local; @@ -618,6 +627,11 @@ static inline int ip_vs_nat_send_or_cont(int pf, struct sk_buff *skb, struct ip_vs_conn *cp, int local) { int ret = NF_STOLEN; + struct net *net; + + net = cp->ipvs->net; + if (ipvs_mode == IPVS_SHARE_NS_MODE) + net = ip_vs_skb_net(skb); skb->ipvs_property = 1; if (likely(!(cp->flags & IP_VS_CONN_F_NFCT))) @@ -634,7 +648,7 @@ static inline int ip_vs_nat_send_or_cont(int pf, struct sk_buff *skb, if (!local) { skb_forward_csum(skb); - NF_HOOK(pf, NF_INET_LOCAL_OUT, cp->ipvs->net, NULL, skb, + NF_HOOK(pf, NF_INET_LOCAL_OUT, net, NULL, skb, NULL, skb_dst(skb)->dev, dst_output); } else ret = NF_ACCEPT; @@ -647,6 +661,11 @@ static inline int ip_vs_send_or_cont(int pf, struct sk_buff *skb, struct ip_vs_conn *cp, int local) { int ret = NF_STOLEN; + struct net *net; + + net = cp->ipvs->net; + if (ipvs_mode == IPVS_SHARE_NS_MODE) + net = ip_vs_skb_net(skb); skb->ipvs_property = 1; if (likely(!(cp->flags & IP_VS_CONN_F_NFCT))) @@ -654,7 +673,7 @@ static inline int ip_vs_send_or_cont(int pf, struct sk_buff *skb, if (!local) { ip_vs_drop_early_demux_sk(skb); skb_forward_csum(skb); - NF_HOOK(pf, NF_INET_LOCAL_OUT, cp->ipvs->net, NULL, skb, + NF_HOOK(pf, NF_INET_LOCAL_OUT, net, NULL, skb, NULL, skb_dst(skb)->dev, dst_output); } else ret = NF_ACCEPT; @@ -760,13 +779,11 @@ ip_vs_nat_xmit(struct sk_buff *skb, struct ip_vs_conn *cp, IP_VS_DBG(10, "filled cport=%d\n", ntohs(*p)); } - /* In ipvs mode, ip_route_input_slow will set me to 1 for - * local_in pkt from nic! For pkt local-out this is not set! + /* originally, this was set in ip_route_input_slow * In bpf mode, this is not useful since local rs is not allowed */ - if (!bpf_mode_on) - was_input = rt_is_input_route(skb_rtable(skb)); - else + was_input = rt_is_input_route(skb_rtable(skb)); + if (ipvs_mode == IPVS_BPF_MODE) was_input = 1; local = __ip_vs_get_out_rt(cp->ipvs, cp->af, skb, cp->dest, cp->daddr.ip, IP_VS_RT_MODE_LOCAL | @@ -774,7 +791,7 @@ ip_vs_nat_xmit(struct sk_buff *skb, struct ip_vs_conn *cp, IP_VS_RT_MODE_RDR, NULL, ipvsh); if (local < 0) goto tx_error; - if (bpf_mode_on && local == 1) { + if (ipvs_mode == IPVS_BPF_MODE && local == 1) { pr_err("shall not route to local rs in bpf mode\n"); BPF_STAT_INC(cp->ipvs, BPF_XMIT_LOCAL_RS); goto tx_error; @@ -833,7 +850,7 @@ ip_vs_nat_xmit(struct sk_buff *skb, struct ip_vs_conn *cp, /* Another hack: avoid icmp_send in ip_fragment */ skb->ignore_df = 1; - if (!bpf_mode_on) { + if (ipvs_mode != IPVS_BPF_MODE) { rc = ip_vs_nat_send_or_cont(NFPROTO_IPV4, skb, cp, local); } else { /* used by bpf egress to construct the key! @@ -1325,7 +1342,7 @@ ip_vs_icmp_xmit(struct sk_buff *skb, struct ip_vs_conn *cp, /* * mangle and send the packet here (only for VS/NAT) */ - if (!bpf_mode_on) + if (ipvs_mode != IPVS_BPF_MODE) was_input = rt_is_input_route(skb_rtable(skb)); else was_input = 1; @@ -1378,7 +1395,7 @@ ip_vs_icmp_xmit(struct sk_buff *skb, struct ip_vs_conn *cp, /* Another hack: avoid icmp_send in ip_fragment */ skb->ignore_df = 1; - if (!bpf_mode_on) { + if (ipvs_mode != IPVS_BPF_MODE) { rc = ip_vs_nat_send_or_cont(NFPROTO_IPV4, skb, cp, local); } else { /* used by bpf egress to construct the key!