Skip to content

BPF ​

BPF 的组件名称为 bpf, 相关内核选项为

ini
# General setup
# -> Enable bpf() system call
CONFIG_BPF_SYSCALL=y

# General setup
# -> Control Group support
CONFIG_CGROUPS=y

# General setup
# -> Control Group support
# -> Support for eBPF programs attached to cgroups
CONFIG_CGROUP_BPF=y

# Networking support
# -> Networking options
# -> enable BPF STREAM_PARSER
CONFIG_BPF_STREAM_PARSER=y

检查工具为

bash
#!/bin/bash

source ./kernel_config_utils.sh
source ./bpf_utils.sh

bpf_check() {
    is_error=false
    if ! kernel_config_check CONFIG_BPF_SYSCALL; then
        is_error=true
    fi
    if ! kernel_config_check CONFIG_CGROUPS; then
        is_error=true
    fi
    if ! kernel_config_check CONFIG_CGROUP_BPF; then
        is_error=true
    fi
    if ! kernel_config_check CONFIG_BPF_STREAM_PARSER; then
        is_error=true
    fi
    if [ "$is_error" == "true" ]; then
        return 1
    fi

    map_types=$(cat << EOF
    hash
    array
    sockmap
    sockhash
EOF
)
    prog_types=$(cat << EOF
    socket_filter
    perf_event
    sock_ops
EOF
)
    helpers=$(cat << EOF
    bpf_map_lookup_elem
EOF
)

    is_error=false
    for map_type in $map_types; do
        if ! bpf_map_type_check $map_type; then
            is_error=true
        fi
    done
    for prog_type in $prog_types; do
        if bpf_prog_type_check $prog_type; then
            for helper in $helpers; do
                if ! bpf_prog_helper_check $prog_type $helper; then
                    is_error=true
                fi
            done
        else
            is_error=true
        fi
    done
    if [ "$is_error" == "true" ]; then
        return 1
    fi

    return 0
}

bpf 系统调用 ​

bpf 系统调用原型如下:

c
int bpf(int cmd, union bpf_attr *attr, unsigned int size);

在内核源码的 kernel/bpf/syscall.c 文件中 SYSCALL_DEFINE3 宏调用处实现. cmd 对应的 C 类型为 enum bpf_cmd, 在内核源码的 include/uapi/linux/bpf.h 文件中定义. 引入 cmd 的 Linux 版本和提交如下表所示:

cmd用途版本提交号
BPF_TOKEN_CREATEv6.935f96de04127d332a5c5e8a155d31f452f88c76d
BPF_LINK_GET_FD_BY_IDbpf_link 查找v5.82d602c8cf40d65d4a7ac34fe18648d8778e6e594
BPF_LINK_GET_NEXT_IDbpf_link 遍历v5.82d602c8cf40d65d4a7ac34fe18648d8778e6e594

BPF 类型 ​

内核源码的 include/linux/bpf_types.h 定义了 BPF_PROG_TYPE_*, BPF_MAP_TYPE_*, BPF_LINK_TYPE_*. 引入 BPF_PROG_TYPE 的 Linux 版本和提交如下表所示:

BPF_PROG_TYPE版本提交号
BPF_PROG_TYPE_TRACINGtracingv5.5f1b9509c2fb0ef4db8d22dac9aef8e856a5d81f6

BPF helper 函数 ​

内核源码的 include/uapi/linux/bpf.h:___BPF_FUNC_MAPPER 列举了 BPF helper 函数和对应编号, enum bpf_func_id 为对应的 C 类型. 引入 BPF helper 函数的编号, Linux 版本和提交号如下表所示:

BPF helper 函数编号版本提交号
bpf_trace_vprintk177v5.1610aceb629e198429c849d5e995c3bb1ba7a9aaa3

BPF TC 挂载点 ​

BPF TC 挂载点的组件名称为 bpf_tc, 相关内核选项为

ini
# Networking support
# -> Networking options
# -> QoS and/or fair queueing
# -> Ingress/classifier-action Qdisc
CONFIG_NET_SCH_INGRESS=y

# Networking support
# -> Networking options
# -> QoS and/or fair queueing
# -> Actions
CONFIG_NET_CLS_ACT=y

# Networking support
# -> Networking options
# -> QoS and/or fair queueing
# -> Actions
# -> BPF based action
CONFIG_NET_ACT_BPF=y

# Networking support
# -> Networking options
# -> QoS and/or fair queueing
# -> BPF-based classifier
CONFIG_NET_CLS_BPF=y

检查工具为

bash
#!/bin/bash

source ./kernel_config_utils.sh
source ./bpf_utils.sh
source ./cmd_utils.sh
source ./net_utils.sh

bpf_tc_object_check() {
    local ifname=$1
    local bpf_tc_object=$2
    local dir=$3
    local handle="0x1"
    local pref="0x66"
    local a="dev $ifname $dir protocol all handle $handle pref $pref bpf"

    if ${TC_CMD} filter get $a > /dev/null 2>&1; then
        echo "find before attach: ifname '$ifname', bpf_tc_object '$bpf_tc_object', dir '$dir'"
        return 1
    fi
    if ! ${TC_CMD} filter add $a object-file $bpf_tc_object section tc; then
        echo "attach failed: ifname '$ifname', object '$bpf_tc_object', dir '$dir'"
        is_error=true
        return 1
    fi
    if ! ${TC_CMD} filter get $a > /dev/null 2>&1; then
        echo "not find after attach: ifname '$ifname', bpf_tc_object '$bpf_tc_object', dir '$dir'"
        return 1
    fi
    if ! ${TC_CMD} filter delete $a; then
        echo "detach failed: ifname '$ifname', object '$bpf_tc_object', dir '$dir'"
        is_error=true
        return 1
    fi
    if ${TC_CMD} filter get $a > /dev/null 2>&1; then
        echo "find after detach: ifname '$ifname', bpf_tc_object '$bpf_tc_object', dir '$dir'"
        return 1
    fi
}
bpf_tc_check() {
    is_error=false
    if ! kernel_config_check CONFIG_NET_SCH_INGRESS sch_ingress; then
        is_error=true
    fi
    if ! kernel_config_check CONFIG_NET_CLS_ACT; then
        is_error=true
    fi
    if ! kernel_config_check CONFIG_NET_ACT_BPF act_bpf; then
        is_error=true
    fi
    if ! kernel_config_check CONFIG_NET_CLS_BPF cls_bpf; then
        is_error=true
    fi
    if [ "$is_error" == "true" ]; then
        return 1
    fi

    prog_types="sched_act sched_cls"
    helpers=$(cat << EOF
    bpf_map_lookup_elem
    bpf_redirect
    bpf_clone_redirect
EOF
)

    for prog_type in $prog_types; do
        if bpf_prog_type_check $prog_type; then
            for helper in $helpers; do
                if ! bpf_prog_helper_check $prog_type $helper; then
                    is_error=true
                fi
            done
        else
            is_error=true
        fi
    done
    if [ "$is_error" == "true" ]; then
        return 1
    fi

    local ifname="dodo"

    netif_new $ifname
    while :
    do
        if ! ${TC_CMD} qdisc add dev $ifname clsact; then
            echo "add qdisc clsact failed: ifname '$ifname'"
            is_error=true
            break
        fi

        local bpf_tc_object="/opt/sumports/lib/bpf_tc.o"

        if ! bpf_tc_object_check $ifname $bpf_tc_object ingress; then
            is_error=true
            break
        fi
        if ! bpf_tc_object_check $ifname $bpf_tc_object egress; then
            is_error=true
            break
        fi

        break
    done
    netif_delete $ifname
    if [ "$is_error" == "true" ]; then
        return 1
    fi

    return 0
}