summaryrefslogtreecommitdiff
diff options
context:
space:
mode:
-rw-r--r--CMakeLists.txt9
-rw-r--r--cmake/config/ipcp/common.cmake6
-rw-r--r--cmake/config/ipcp/eth.cmake22
-rw-r--r--cmake/config/ipcp/udp.cmake16
-rw-r--r--cmake/config/irmd.cmake2
-rw-r--r--cmake/config/lib.cmake139
-rw-r--r--cmake/config/lib/common.cmake58
-rw-r--r--cmake/config/lib/crypt.cmake41
-rw-r--r--cmake/config/lib/frct.cmake43
-rw-r--r--cmake/config/lib/poa.cmake37
-rw-r--r--cmake/config/lib/ssm.cmake (renamed from cmake/config/ssm.cmake)22
-rw-r--r--cmake/dependencies.cmake12
-rw-r--r--cmake/dependencies/eth/netmap.cmake24
-rw-r--r--cmake/dependencies/udp/ddns.cmake31
-rw-r--r--doc/man/ouroboros-tutorial.743
-rw-r--r--doc/man/ouroboros.8132
-rw-r--r--include/ouroboros/ipcp-dev.h113
-rw-r--r--include/ouroboros/ipcp.h97
-rw-r--r--include/ouroboros/irm.h25
-rw-r--r--include/ouroboros/protobuf.h30
-rw-r--r--include/ouroboros/qos.h6
-rw-r--r--include/ouroboros/serdes-irm.h17
-rw-r--r--irmd.conf.in95
-rw-r--r--src/ipcpd/CMakeLists.txt5
-rw-r--r--src/ipcpd/broadcast/main.c13
-rw-r--r--src/ipcpd/common/connmgr.c93
-rw-r--r--src/ipcpd/common/connmgr.h17
-rw-r--r--src/ipcpd/config.h.in25
-rw-r--r--src/ipcpd/eth/CMakeLists.txt21
-rw-r--r--src/ipcpd/eth/dix.c26
-rw-r--r--src/ipcpd/eth/eth.c2468
-rw-r--r--src/ipcpd/ipcp.c305
-rw-r--r--src/ipcpd/ipcp.h13
-rw-r--r--src/ipcpd/local/CMakeLists.txt1
-rw-r--r--src/ipcpd/local/main.c32
-rw-r--r--src/ipcpd/local/reg.c217
-rw-r--r--src/ipcpd/local/reg.h (renamed from src/ipcpd/eth/llc.c)27
-rw-r--r--src/ipcpd/shim-data.c582
-rw-r--r--src/ipcpd/shim-data.h109
-rw-r--r--src/ipcpd/udp/CMakeLists.txt14
-rw-r--r--src/ipcpd/udp/udp.c1333
-rw-r--r--src/ipcpd/udp/udp4.c42
-rw-r--r--src/ipcpd/udp/udp6.c42
-rw-r--r--src/ipcpd/unicast/fa.c8
-rw-r--r--src/ipcpd/unicast/main.c9
-rw-r--r--src/irmd/config.h.in6
-rw-r--r--src/irmd/configfile.c553
-rw-r--r--src/irmd/ipcp.c129
-rw-r--r--src/irmd/ipcp.h71
-rw-r--r--src/irmd/irmd.h53
-rw-r--r--src/irmd/main.c610
-rw-r--r--src/irmd/oap.h29
-rw-r--r--src/irmd/oap/cli.c38
-rw-r--r--src/irmd/oap/tests/common.c19
-rw-r--r--src/irmd/reg/flow.c3
-rw-r--r--src/irmd/reg/flow.h2
-rw-r--r--src/irmd/reg/name.c1
-rw-r--r--src/irmd/reg/reg.c73
-rw-r--r--src/irmd/reg/reg.h251
-rw-r--r--src/irmd/reg/tests/reg_test.c2
-rw-r--r--src/lib/CMakeLists.txt17
-rw-r--r--src/lib/config.h.in19
-rw-r--r--src/lib/dev.c135
-rw-r--r--src/lib/irm.c127
-rw-r--r--src/lib/pb/ipcp.proto7
-rw-r--r--src/lib/pb/ipcp_config.proto27
-rw-r--r--src/lib/pb/irm.proto9
-rw-r--r--src/lib/pb/model.proto13
-rw-r--r--src/lib/poa/addr.c142
-rw-r--r--src/lib/poa/eth.c1987
-rw-r--r--src/lib/poa/poa.c2515
-rw-r--r--src/lib/poa/poa.h364
-rw-r--r--src/lib/poa/udp.c633
-rw-r--r--src/lib/protobuf.c386
-rw-r--r--src/lib/serdes-irm.c133
-rw-r--r--src/lib/tests/CMakeLists.txt7
-rw-r--r--src/lib/tests/poa_test.c307
-rw-r--r--src/tools/CMakeLists.txt4
-rw-r--r--src/tools/irm/irm_ipcp.c2
-rw-r--r--src/tools/irm/irm_ipcp_bootstrap.c148
-rw-r--r--src/tools/irm/irm_ipcp_connect.c90
-rw-r--r--src/tools/irm/irm_ipcp_create.c15
-rw-r--r--src/tools/irm/irm_ipcp_enroll.c58
-rw-r--r--src/tools/irm/irm_ipcp_list.c23
-rw-r--r--src/tools/irm/irm_ipcp_poa.c98
-rw-r--r--src/tools/irm/irm_ipcp_poa_attach.c153
-rw-r--r--src/tools/irm/irm_ipcp_poa_detach.c153
-rw-r--r--src/tools/irm/irm_ipcp_poa_list.c136
-rw-r--r--src/tools/irm/irm_ops.h12
-rw-r--r--src/tools/irm/irm_utils.c233
-rw-r--r--src/tools/irm/irm_utils.h41
91 files changed, 10267 insertions, 5989 deletions
diff --git a/CMakeLists.txt b/CMakeLists.txt
index bfabd711..e624c22c 100644
--- a/CMakeLists.txt
+++ b/CMakeLists.txt
@@ -51,15 +51,16 @@ include(config/global)
include(dependencies)
-include(config/lib)
-include(config/ssm)
+include(config/lib/common)
+include(config/lib/crypt)
+include(config/lib/poa)
+include(config/lib/frct)
+include(config/lib/ssm)
include(config/irmd)
include(config/ipcp/common)
include(config/ipcp/unicast)
include(config/ipcp/broadcast)
include(config/ipcp/local)
-include(config/ipcp/eth)
-include(config/ipcp/udp)
include(tests)
include(include)
diff --git a/cmake/config/ipcp/common.cmake b/cmake/config/ipcp/common.cmake
index 7dbc252b..79c822a3 100644
--- a/cmake/config/ipcp/common.cmake
+++ b/cmake/config/ipcp/common.cmake
@@ -49,10 +49,4 @@ if(HAVE_FUSE)
if(IPCP_FLOW_STATS)
message(STATUS "IPCP flow statistics enabled")
endif()
-
- set(IPCP_ETH_FLOW_STATS FALSE CACHE BOOL
- "Enable ipcpd-eth flow statistics via RIB")
- if(IPCP_ETH_FLOW_STATS)
- message(STATUS "ipcpd-eth flow statistics enabled")
- endif()
endif()
diff --git a/cmake/config/ipcp/eth.cmake b/cmake/config/ipcp/eth.cmake
deleted file mode 100644
index 6a044d4b..00000000
--- a/cmake/config/ipcp/eth.cmake
+++ /dev/null
@@ -1,22 +0,0 @@
-# Ethernet IPCP configuration options for Ouroboros
-# Options for eth-llc and eth-dix IPCPs
-
-set(IPCP_ETH_LLC_TARGET ipcpd-eth-llc)
-set(IPCP_ETH_DIX_TARGET ipcpd-eth-dix)
-
-set(IPCP_ETH_RD_THR 1 CACHE STRING
- "Number of reader threads in Ethernet IPCP")
-set(IPCP_ETH_WR_THR 1 CACHE STRING
- "Number of writer threads in Ethernet IPCP")
-set(IPCP_ETH_QDISC_BYPASS false CACHE BOOL
- "Bypass the Qdisc in the kernel when using raw sockets")
-set(IPCP_ETH_SNDBUF 4096 CACHE STRING
- "Raw socket SO_SNDBUF in bytes (floored to one frame); 0 = kernel default")
-set(IPCP_ETH_RCVBUF 0 CACHE STRING
- "Raw socket SO_RCVBUF in bytes; 0 = leave kernel default (rmem_default)")
-set(IPCP_ETH_LO_MTU 9000 CACHE STRING
- "Restrict Ethernet MTU over loopback interfaces")
-set(IPCP_ETH_MGMT_FRAME_SIZE 9000 CACHE STRING
- "Management frame buffer size for Ethernet IPCPs")
-set(IPCP_ETH_MPL 100 CACHE STRING
- "Default maximum packet lifetime for the Ethernet IPCPs, in ms")
diff --git a/cmake/config/ipcp/udp.cmake b/cmake/config/ipcp/udp.cmake
deleted file mode 100644
index af84a844..00000000
--- a/cmake/config/ipcp/udp.cmake
+++ /dev/null
@@ -1,16 +0,0 @@
-# UDP IPCP configuration options for Ouroboros
-# Options for udp4 and udp6 IPCPs
-
-set(IPCP_UDP4_TARGET ipcpd-udp4)
-set(IPCP_UDP6_TARGET ipcpd-udp6)
-
-set(IPCP_UDP_RD_THR 3 CACHE STRING
- "Number of reader threads in UDP IPCPs")
-set(IPCP_UDP_WR_THR 3 CACHE STRING
- "Number of writer threads in UDP IPCPs")
-set(IPCP_UDP_MPL 5000 CACHE STRING
- "Default maximum packet lifetime for the UDP IPCPs, in ms")
-set(IPCP_UDP4_MTU 1472 CACHE STRING
- "Fallback UDP4 layer MTU when getsockopt(IP_MTU) is unavailable, in bytes")
-set(IPCP_UDP6_MTU 1452 CACHE STRING
- "Fallback UDP6 layer MTU when getsockopt(IPV6_MTU) is unavailable, in bytes")
diff --git a/cmake/config/irmd.cmake b/cmake/config/irmd.cmake
index 79e24bae..bbb73bec 100644
--- a/cmake/config/irmd.cmake
+++ b/cmake/config/irmd.cmake
@@ -11,7 +11,7 @@ set(ENROLL_TIMEOUT 20000 CACHE STRING
set(REG_TIMEOUT 20000 CACHE STRING
"Timeout for registering a name (ms)")
set(QUERY_TIMEOUT 2000 CACHE STRING
- "Timeout to query a name with an IPCP (ms); must exceed shim retry budget")
+ "Timeout to query a name with an IPCP (ms); must exceed PoA retries")
set(CONNECT_TIMEOUT 20000 CACHE STRING
"Timeout to connect an IPCP to another IPCP (ms)")
set(FLOW_ALLOC_TIMEOUT 20000 CACHE STRING
diff --git a/cmake/config/lib.cmake b/cmake/config/lib.cmake
deleted file mode 100644
index 81a7d6ba..00000000
--- a/cmake/config/lib.cmake
+++ /dev/null
@@ -1,139 +0,0 @@
-# Library configuration options for Ouroboros
-# Options affecting libouroboros-common, libouroboros-dev, libouroboros-irm
-
-# Flow limits
-set(SYS_MAX_FLOWS 10240 CACHE STRING
- "Maximum number of total flows for this system")
-set(PROC_MAX_FLOWS 4096 CACHE STRING
- "Maximum number of flows in an application")
-set(PROC_RES_FDS 64 CACHE STRING
- "Number of reserved flow descriptors per application")
-set(PROC_MAX_FQUEUES 32 CACHE STRING
- "Maximum number of flow sets per application")
-
-# Threading
-if(NOT APPLE)
- set(PTHREAD_COND_CLOCK "CLOCK_MONOTONIC" CACHE STRING
- "Clock to use for condition variable timing")
-else()
- set(PTHREAD_COND_CLOCK "CLOCK_REALTIME" CACHE INTERNAL
- "Clock to use for condition variable timing")
-endif()
-
-# Timeouts
-set(SOCKET_TIMEOUT 500 CACHE STRING
- "Default timeout for responses from IPCPs (ms)")
-
-# QoS settings
-set(QOS_DISABLE_CRC TRUE CACHE BOOL
- "Ignores ber setting on all QoS cubes")
-
-include(utils/CPUUtils)
-detect_pclmul()
-detect_pmull()
-if(HAVE_PCLMUL)
- message(STATUS "CRC-64/NVMe backend: PCLMUL (x86 SSE4.1+PCLMUL)")
-elseif(HAVE_PMULL)
- message(STATUS "CRC-64/NVMe backend: PMULL (aarch64 crypto)")
-else()
- message(STATUS "CRC-64/NVMe backend: byte table (no acceleration)")
-endif()
-
-# Delta-t protocol timers (Watson bound: 3*MPL + A + R).
-# MPL is reported per IPCP (IPCP_*_MPL); A and R are FRCT-wide.
-set(DELTA_T_ACK 1000 CACHE STRING
- "Maximum time to acknowledge a packet (ms)")
-set(DELTA_T_RTX 30000 CACHE STRING
- "Maximum time to retransmit a packet (ms)")
-
-# FRCT configuration
-set(FRCT_REORDER_QUEUE_SIZE 128 CACHE STRING
- "Size of the reordering queue, must be a power of 2")
-set(FRCT_START_WINDOW 128 CACHE STRING
- "Start window, must be a power of 2")
-set(FRCT_LINUX_RTT_ESTIMATOR TRUE CACHE BOOL
- "Use Linux RTT estimator formula instead of the TCP RFC formula")
-set(FRCT_RTO_MDEV_MULTIPLIER 2 CACHE STRING
- "Multiplier for deviation term in the RTO: RTO = sRTT + (mdev << X)")
-set(FRCT_RTO_INC_FACTOR 0 CACHE STRING
- "Divisor for RTO increase after timeout: RTO += RTX >> X, 0: Karn/Partridge")
-set(FRCT_RTO_MIN 250 CACHE STRING
- "Hard floor for Retransmission Timeout (RTO) for FRCT (us)")
-set(FRCT_TICK_TIME 5000 CACHE STRING
- "Tick time for FRCT activity (retransmission, acknowledgments) (us)")
-set(FRCT_DEBUG_STDOUT FALSE CACHE BOOL
- "Print FRCT final counters to stdout at flow teardown")
-
-# Retransmission (RXM) configuration
-set(RXM_MIN_RESOLUTION 20 CACHE STRING
- "Minimum retransmission delay (ns), as a power to 2")
-set(RXM_WHEEL_MULTIPLIER 4 CACHE STRING
- "Factor for retransmission wheel levels as a power to 2")
-set(RXM_WHEEL_LEVELS 3 CACHE STRING
- "Number of levels in the retransmission wheel")
-set(RXM_WHEEL_SLOTS_PER_LEVEL 256 CACHE STRING
- "Number of slots per level in the retransmission wheel, must be a power of 2")
-
-# Acknowledgment wheel configuration
-set(ACK_WHEEL_SLOTS 256 CACHE STRING
- "Number of slots in the acknowledgment wheel, must be a power of 2")
-set(ACK_WHEEL_RESOLUTION 18 CACHE STRING
- "Minimum acknowledgment delay (ns), as a power to 2")
-
-# Thread pool manager (TPM) debugging
-set(TPM_DEBUG_REPORT_INTERVAL 0 CACHE STRING
- "Interval at wich the TPM will report long running threads (s), 0 disables")
-set(TPM_DEBUG_ABORT_TIMEOUT 0 CACHE STRING
- "TPM abort process after a thread reaches this timeout (s), 0 disables")
-
-# Encryption
-set(KEY_LEAF_BITS 20 CACHE STRING
- "Packets per leaf key as a power of two (2^20 = AEAD-safe default)")
-set(KEY_NODE_BITS 6 CACHE STRING
- "Leaf keys per node key, power of two (2^6 = 64; leak compartment)")
-set(KEY_NODE_COUNT 128 CACHE STRING
- "Node keys per batch (N); <= 4096, the 12-bit on-wire node index")
-set(KEY_REKEY_WATERMARK 4 CACHE STRING
- "Re-key when this many node keys remain; 0 disables the count trigger")
-set(KEY_REPLAY_WINDOW 2048 CACHE STRING
- "RX replay window in packets; power of two, >= 128")
-set(KEY_REKEY_WM_CHECK_BITS 16 CACHE STRING
- "Re-key watermark is consulted once per 2^n flow writes")
-if(NOT KEY_REPLAY_WINDOW MATCHES "^[0-9]+$")
- message(FATAL_ERROR "KEY_REPLAY_WINDOW must be a positive integer")
-endif()
-math(EXPR _krw_p2 "${KEY_REPLAY_WINDOW} & (${KEY_REPLAY_WINDOW} - 1)")
-if(KEY_REPLAY_WINDOW LESS 128 OR NOT _krw_p2 EQUAL 0)
- message(FATAL_ERROR "KEY_REPLAY_WINDOW must be a power of two >= 128")
-endif()
-
-# Re-key must finish within its lead window - KEY_REKEY_WATERMARK node keys
-# worth of packets - before the batch exhausts and TX fails closed. dev.c only
-# evaluates the watermark once per FLOW_WM_CHECK writes, so a lead below ~2x
-# that leaves a high-rate flow no room to complete the exchange. Production
-# defaults are vast; this guards under-sized (test) geometries.
-if(KEY_REKEY_WATERMARK GREATER 0)
- math(EXPR _rk_wm_check "1 << ${KEY_REKEY_WM_CHECK_BITS}")
- math(EXPR _rk_lead
- "${KEY_REKEY_WATERMARK} << (${KEY_LEAF_BITS} + ${KEY_NODE_BITS})")
- math(EXPR _rk_min "2 * ${_rk_wm_check}")
- if(_rk_lead LESS _rk_min)
- message(WARNING
- "Re-key lead is ${_rk_lead} packets vs the watermark check interval "
- "${_rk_wm_check}; a high-rate flow may exhaust its key batch before the "
- "re-key completes (TX fails closed until it does). Raise KEY_LEAF_BITS, "
- "KEY_NODE_BITS, or KEY_REKEY_WATERMARK.")
- endif()
-endif()
-
-# Flow statistics (requires FUSE)
-if(HAVE_FUSE)
- set(PROC_FLOW_STATS TRUE CACHE BOOL
- "Enable flow statistics tracking for application flows")
- if(PROC_FLOW_STATS)
- message(STATUS "Application flow statistics enabled")
- else()
- message(STATUS "Application flow statistics disabled")
- endif()
-endif()
-
diff --git a/cmake/config/lib/common.cmake b/cmake/config/lib/common.cmake
new file mode 100644
index 00000000..ebdc3196
--- /dev/null
+++ b/cmake/config/lib/common.cmake
@@ -0,0 +1,58 @@
+# Library configuration options for Ouroboros Options affecting libouroboros-common,
+# libouroboros-dev and libouroboros-irm as a whole; per-subsystem options sit beside this file
+
+# Flow limits
+set(SYS_MAX_FLOWS 10240 CACHE STRING
+ "Maximum number of total flows for this system")
+set(PROC_MAX_FLOWS 4096 CACHE STRING
+ "Maximum number of flows in an application")
+set(PROC_RES_FDS 64 CACHE STRING
+ "Number of reserved flow descriptors per application")
+set(PROC_MAX_FQUEUES 32 CACHE STRING
+ "Maximum number of flow sets per application")
+
+# Threading
+if(NOT APPLE)
+ set(PTHREAD_COND_CLOCK "CLOCK_MONOTONIC" CACHE STRING
+ "Clock to use for condition variable timing")
+else()
+ set(PTHREAD_COND_CLOCK "CLOCK_REALTIME" CACHE INTERNAL
+ "Clock to use for condition variable timing")
+endif()
+
+# Timeouts
+set(SOCKET_TIMEOUT 500 CACHE STRING
+ "Default timeout for responses from IPCPs (ms)")
+
+# QoS settings
+set(QOS_DISABLE_CRC TRUE CACHE BOOL
+ "Ignores ber setting on all QoS cubes")
+
+include(utils/CPUUtils)
+detect_pclmul()
+detect_pmull()
+if(HAVE_PCLMUL)
+ message(STATUS "CRC-64/NVMe backend: PCLMUL (x86 SSE4.1+PCLMUL)")
+elseif(HAVE_PMULL)
+ message(STATUS "CRC-64/NVMe backend: PMULL (aarch64 crypto)")
+else()
+ message(STATUS "CRC-64/NVMe backend: byte table (no acceleration)")
+endif()
+
+# Thread pool manager (TPM) debugging
+set(TPM_DEBUG_REPORT_INTERVAL 0 CACHE STRING
+ "Interval at wich the TPM will report long running threads (s), 0 disables")
+set(TPM_DEBUG_ABORT_TIMEOUT 0 CACHE STRING
+ "TPM abort process after a thread reaches this timeout (s), 0 disables")
+
+# Flow statistics (requires FUSE)
+if(HAVE_FUSE)
+ set(PROC_FLOW_STATS TRUE CACHE BOOL
+ "Enable flow statistics tracking for application flows")
+ if(PROC_FLOW_STATS)
+ message(STATUS "Application flow statistics enabled")
+ else()
+ message(STATUS "Application flow statistics disabled")
+ endif()
+endif()
+
diff --git a/cmake/config/lib/crypt.cmake b/cmake/config/lib/crypt.cmake
new file mode 100644
index 00000000..45fdaaed
--- /dev/null
+++ b/cmake/config/lib/crypt.cmake
@@ -0,0 +1,41 @@
+# Encryption configuration for Ouroboros Options for the key schedule and packet protection in
+# src/lib/crypt/
+
+# Encryption
+set(KEY_LEAF_BITS 20 CACHE STRING
+ "Packets per leaf key as a power of two (2^20 = AEAD-safe default)")
+set(KEY_NODE_BITS 6 CACHE STRING
+ "Leaf keys per node key, power of two (2^6 = 64; leak compartment)")
+set(KEY_NODE_COUNT 128 CACHE STRING
+ "Node keys per batch (N); <= 4096, the 12-bit on-wire node index")
+set(KEY_REKEY_WATERMARK 4 CACHE STRING
+ "Re-key when this many node keys remain; 0 disables the count trigger")
+set(KEY_REPLAY_WINDOW 2048 CACHE STRING
+ "RX replay window in packets; power of two, >= 128")
+set(KEY_REKEY_WM_CHECK_BITS 16 CACHE STRING
+ "Re-key watermark is consulted once per 2^n flow writes")
+if(NOT KEY_REPLAY_WINDOW MATCHES "^[0-9]+$")
+ message(FATAL_ERROR "KEY_REPLAY_WINDOW must be a positive integer")
+endif()
+math(EXPR _krw_p2 "${KEY_REPLAY_WINDOW} & (${KEY_REPLAY_WINDOW} - 1)")
+if(KEY_REPLAY_WINDOW LESS 128 OR NOT _krw_p2 EQUAL 0)
+ message(FATAL_ERROR "KEY_REPLAY_WINDOW must be a power of two >= 128")
+endif()
+
+# Re-key must finish within its lead window - KEY_REKEY_WATERMARK node keys worth of packets -
+# before the batch exhausts and TX fails closed. dev.c only evaluates the watermark once per
+# FLOW_WM_CHECK writes, so a lead below ~2x that leaves a high-rate flow no room to complete the
+# exchange. Production defaults are vast; this guards under-sized (test) geometries.
+if(KEY_REKEY_WATERMARK GREATER 0)
+ math(EXPR _rk_wm_check "1 << ${KEY_REKEY_WM_CHECK_BITS}")
+ math(EXPR _rk_lead
+ "${KEY_REKEY_WATERMARK} << (${KEY_LEAF_BITS} + ${KEY_NODE_BITS})")
+ math(EXPR _rk_min "2 * ${_rk_wm_check}")
+ if(_rk_lead LESS _rk_min)
+ message(WARNING
+ "Re-key lead is ${_rk_lead} packets vs the watermark check interval "
+ "${_rk_wm_check}; a high-rate flow may exhaust its key batch before the "
+ "re-key completes (TX fails closed until it does). Raise KEY_LEAF_BITS, "
+ "KEY_NODE_BITS, or KEY_REKEY_WATERMARK.")
+ endif()
+endif()
diff --git a/cmake/config/lib/frct.cmake b/cmake/config/lib/frct.cmake
new file mode 100644
index 00000000..1bd60aa0
--- /dev/null
+++ b/cmake/config/lib/frct.cmake
@@ -0,0 +1,43 @@
+# FRCT configuration for Ouroboros Options for the flow and retransmission control protocol in
+# src/lib/frct.c
+
+# Delta-t protocol timers (Watson bound: 3*MPL + A + R).
+# MPL is reported per IPCP (IPCP_*_MPL); A and R are FRCT-wide.
+set(DELTA_T_ACK 1000 CACHE STRING
+ "Maximum time to acknowledge a packet (ms)")
+set(DELTA_T_RTX 32000 CACHE STRING
+ "Maximum time to retransmit a packet (ms)")
+
+# FRCT configuration
+set(FRCT_REORDER_QUEUE_SIZE 128 CACHE STRING
+ "Size of the reordering queue, must be a power of 2")
+set(FRCT_START_WINDOW 128 CACHE STRING
+ "Start window, must be a power of 2")
+set(FRCT_LINUX_RTT_ESTIMATOR TRUE CACHE BOOL
+ "Use Linux RTT estimator formula instead of the TCP RFC formula")
+set(FRCT_RTO_MDEV_MULTIPLIER 2 CACHE STRING
+ "Multiplier for deviation term in the RTO: RTO = sRTT + (mdev << X)")
+set(FRCT_RTO_INC_FACTOR 0 CACHE STRING
+ "Divisor for RTO increase after timeout: RTO += RTX >> X, 0: Karn/Partridge")
+set(FRCT_RTO_MIN 250 CACHE STRING
+ "Hard floor for Retransmission Timeout (RTO) for FRCT (us)")
+set(FRCT_TICK_TIME 5000 CACHE STRING
+ "Tick time for FRCT activity (retransmission, acknowledgments) (us)")
+set(FRCT_DEBUG_STDOUT FALSE CACHE BOOL
+ "Print FRCT final counters to stdout at flow teardown")
+
+# Retransmission (RXM) configuration
+set(RXM_MIN_RESOLUTION 20 CACHE STRING
+ "Minimum retransmission delay (ns), as a power to 2")
+set(RXM_WHEEL_MULTIPLIER 4 CACHE STRING
+ "Factor for retransmission wheel levels as a power to 2")
+set(RXM_WHEEL_LEVELS 3 CACHE STRING
+ "Number of levels in the retransmission wheel")
+set(RXM_WHEEL_SLOTS_PER_LEVEL 256 CACHE STRING
+ "Number of slots per level in the retransmission wheel, must be a power of 2")
+
+# Acknowledgment wheel configuration
+set(ACK_WHEEL_SLOTS 256 CACHE STRING
+ "Number of slots in the acknowledgment wheel, must be a power of 2")
+set(ACK_WHEEL_RESOLUTION 18 CACHE STRING
+ "Minimum acknowledgment delay (ns), as a power to 2")
diff --git a/cmake/config/lib/poa.cmake b/cmake/config/lib/poa.cmake
new file mode 100644
index 00000000..c1ca10a8
--- /dev/null
+++ b/cmake/config/lib/poa.cmake
@@ -0,0 +1,37 @@
+# Flow Point of attachment (poa) configuration for Ouroboros
+
+set(POA_MGMT_FRAME_SIZE 16384 CACHE STRING
+ "Maximum size of a flow endpoint management frame, in bytes")
+
+# Management frames are retried or repeated, so a send never waits long.
+set(POA_MGMT_SND_TIMEO 100 CACHE STRING
+ "Deadline for sending a management frame, in ms")
+
+# Flows on one PoA share its transmit queue, so congestion avoidance keys its link estimator on the
+# PoA and needs a bound on their number.
+set(POA_MAX_POAS 16 CACHE STRING
+ "Maximum number of points of attachment per IPCP")
+
+# UDP endpoints
+set(POA_UDP_MPL 5000 CACHE STRING
+ "Default maximum packet lifetime for UDP flow endpoints, in ms")
+set(POA_UDP4_MTU 1472 CACHE STRING
+ "Fallback UDP4 endpoint MTU when getsockopt(IP_MTU) is unavailable")
+set(POA_UDP6_MTU 1452 CACHE STRING
+ "Fallback UDP6 endpoint MTU when getsockopt(IPV6_MTU) is unavailable")
+set(POA_UDP_RD_BUF 65535 CACHE STRING
+ "UDP endpoint receive buffer in bytes. Bounds the advertised MTU")
+
+# Ethernet endpoints
+set(POA_ETH_MPL 100 CACHE STRING
+ "Default maximum packet lifetime for Ethernet flow endpoints, in ms")
+set(POA_ETH_QDISC_BYPASS false CACHE BOOL
+ "Bypass the Qdisc in the kernel when using raw sockets")
+set(POA_ETH_SNDBUF 0 CACHE STRING
+ "Raw socket SO_SNDBUF in bytes (floored to one frame). 0 = kernel default")
+set(POA_ETH_RCVBUF 0 CACHE STRING
+ "Raw socket SO_RCVBUF in bytes. 0 = kernel default")
+set(POA_ETH_LO_MTU 9000 CACHE STRING
+ "Restrict Ethernet flow endpoint MTU over loopback interfaces")
+set(POA_ETH_RD_BUF 16384 CACHE STRING
+ "Cap on the Ethernet endpoint receive buffer and MTU, in bytes")
diff --git a/cmake/config/ssm.cmake b/cmake/config/lib/ssm.cmake
index 589171ea..a9ac35c9 100644
--- a/cmake/config/ssm.cmake
+++ b/cmake/config/lib/ssm.cmake
@@ -1,6 +1,5 @@
-# Secure Shared Memory (SSM) pool configuration for Ouroboros
-# This file defines the allocation parameters for the secure shared memory
-# pool allocator
+# Secure Shared Memory (SSM) pool configuration for Ouroboros This file defines the allocation
+# parameters for the secure shared memory pool allocator
# Shared memory pool naming configuration
set(SSM_PREFIX "ouroboros" CACHE STRING
@@ -21,6 +20,8 @@ set(SSM_PK_BUFF_TAILSPACE 32 CACHE STRING
"Bytes of tailspace to reserve for future tails")
set(SSM_RBUFF_SIZE 1024 CACHE STRING
"Number of slots in a flow's rbuff ring; must be a power of 2")
+set(SSM_RBUFF_TXQ_DELAY 10 CACHE STRING
+ "Queueing delay a flow's tx ring may hold (ms); 0 is unlimited")
set(SSM_RBUFF_PREFIX "/${SHM_PREFIX}.rbuff." CACHE INTERNAL
"Prefix for rbuff POSIX shared memory filenames")
set(SSM_FLOW_SET_PREFIX "/${SHM_PREFIX}.set." CACHE INTERNAL
@@ -32,8 +33,8 @@ set(SSM_POOL_SHARDS 4 CACHE STRING
set(SSM_POOL_RECLAIM_AGE_S 60 CACHE STRING
"Minimum age in seconds before a block is presumed stale and reclaimed")
-# Global Shared Packet Pool (GSPP) - for privileged processes
-# Shared by all processes in 'ouroboros' group (~60 MB total)
+# Global Shared Packet Pool (GSPP) - for privileged processes Shared by all processes in 'ouroboros'
+# group (~60 MB total)
set(SSM_GSPP_256_BLOCKS 1024 CACHE STRING
"GSPP: Number of 256B blocks")
set(SSM_GSPP_512_BLOCKS 2048 CACHE STRING
@@ -53,8 +54,8 @@ set(SSM_GSPP_256K_BLOCKS 32 CACHE STRING
set(SSM_GSPP_1M_BLOCKS 16 CACHE STRING
"GSPP: Number of 1MB blocks")
-# Per-User Pool (PUP) - for unprivileged applications
-# Each unprivileged app gets its own smaller pool (~7.5 MB total)
+# Per-User Pool (PUP) - for unprivileged applications Each unprivileged app gets its own smaller
+# pool (~7.5 MB total)
set(SSM_PUP_256_BLOCKS 512 CACHE STRING
"PUP: Number of 256B blocks")
set(SSM_PUP_512_BLOCKS 512 CACHE STRING
@@ -147,10 +148,9 @@ message(STATUS " Blocks: ${SSM_PUP_256_BLOCKS}, ${SSM_PUP_512_BLOCKS}, "
"${SSM_PUP_16K_BLOCKS}, ${SSM_PUP_64K_BLOCKS}, ${SSM_PUP_256K_BLOCKS}, "
"${SSM_PUP_1M_BLOCKS}")
-# FRCT reorder queue must fit in every enabled size class. If RQ_SIZE
-# >= any backing pool, the receiver advertises a window the pool
-# cannot back; np1_flow_write fails under load and a single dropped
-# fragment wedges the flow. Auto-zeroed classes are skipped.
+# FRCT reorder queue must fit in every enabled size class. If RQ_SIZE >= any backing pool, the
+# receiver advertises a window the pool cannot back; np1_flow_write fails under load and a single
+# dropped fragment wedges the flow. Auto-zeroed classes are skipped.
foreach(_class 256 512 1K 2K)
if(SSM_PUP_${_class}_BLOCKS GREATER 0
AND NOT FRCT_REORDER_QUEUE_SIZE LESS SSM_PUP_${_class}_BLOCKS)
diff --git a/cmake/dependencies.cmake b/cmake/dependencies.cmake
index ff44ad68..93c9c0de 100644
--- a/cmake/dependencies.cmake
+++ b/cmake/dependencies.cmake
@@ -17,19 +17,19 @@ include(dependencies/crypt/libgcrypt)
# IRMd
include(dependencies/irmd/libtoml)
-# Ethernet IPCP backends
+# Ethernet PoA backends. netmap is last: it overrides the others.
include(dependencies/eth/rawsockets)
include(dependencies/eth/bpf)
include(dependencies/eth/netmap)
+
+# The Ethernet PoA needs a backend to send and receive frames.
if(HAVE_RAW_SOCKETS OR HAVE_BPF OR HAVE_NETMAP)
- set(HAVE_ETH TRUE CACHE INTERNAL "Ethernet IPCP support available")
+ set(HAVE_ETH TRUE)
else()
- unset(HAVE_ETH CACHE)
+ message(STATUS "No Ethernet backend, Ethernet PoAs disabled")
+ unset(HAVE_ETH)
endif()
-# UDP IPCP
-include(dependencies/udp/ddns)
-
# Coverage tools
include(dependencies/coverage/gcov)
include(dependencies/coverage/lcov)
diff --git a/cmake/dependencies/eth/netmap.cmake b/cmake/dependencies/eth/netmap.cmake
index 94ecd634..b31c4d03 100644
--- a/cmake/dependencies/eth/netmap.cmake
+++ b/cmake/dependencies/eth/netmap.cmake
@@ -1,18 +1,22 @@
-# netmap support (optional acceleration)
+# netmap support (kernel bypass). Explicit opt-in.
find_path(NETMAP_C_INCLUDE_DIR
net/netmap_user.h
HINTS /usr/include /usr/local/include)
mark_as_advanced(NETMAP_C_INCLUDE_DIR)
-if(NOT HAVE_RAW_SOCKETS AND NOT HAVE_BPF AND NETMAP_C_INCLUDE_DIR)
- set(DISABLE_NETMAP FALSE CACHE BOOL
- "Disable netmap support for ETH IPCPs")
- if(NOT DISABLE_NETMAP)
- message(STATUS "Netmap support for Ethernet IPCPs enabled")
- set(HAVE_NETMAP TRUE)
- else()
- message(STATUS "Netmap support for Ethernet IPCPs disabled by user")
- unset(HAVE_NETMAP)
+set(ENABLE_NETMAP FALSE CACHE BOOL
+ "Use netmap for Ethernet PoAs, overriding raw sockets or BPF")
+
+if(ENABLE_NETMAP)
+ if(NOT NETMAP_C_INCLUDE_DIR)
+ message(FATAL_ERROR "ENABLE_NETMAP is set, but netmap was not found.")
endif()
+
+ message(STATUS "Netmap support for Ethernet PoAs enabled")
+
+ set(HAVE_NETMAP TRUE)
+
+ unset(HAVE_RAW_SOCKETS)
+ unset(HAVE_BPF)
endif()
diff --git a/cmake/dependencies/udp/ddns.cmake b/cmake/dependencies/udp/ddns.cmake
deleted file mode 100644
index e8208e47..00000000
--- a/cmake/dependencies/udp/ddns.cmake
+++ /dev/null
@@ -1,31 +0,0 @@
-# DDNS (Dynamic DNS) support detection
-# Requires nsupdate and nslookup tools
-
-find_program(NSUPDATE_EXECUTABLE
- NAMES nsupdate
- DOC "The nsupdate tool that enables DDNS")
-
-find_program(NSLOOKUP_EXECUTABLE
- NAMES nslookup
- DOC "The nslookup tool that resolves DNS names")
-
-mark_as_advanced(NSLOOKUP_EXECUTABLE NSUPDATE_EXECUTABLE)
-
-if(NSLOOKUP_EXECUTABLE AND NSUPDATE_EXECUTABLE)
- set(DISABLE_DDNS FALSE CACHE BOOL "Disable DDNS support")
- if(NOT DISABLE_DDNS)
- message(STATUS "DDNS support enabled")
- set(HAVE_DDNS TRUE CACHE INTERNAL "Dynamic DNS support available")
- else()
- message(STATUS "DDNS support disabled by user")
- unset(HAVE_DDNS CACHE)
- endif()
-else()
- if(NSLOOKUP_EXECUTABLE)
- message(STATUS "Install nsupdate to enable DDNS support")
- elseif(NSUPDATE_EXECUTABLE)
- message(STATUS "Install nslookup to enable DDNS support")
- else()
- message(STATUS "Install nslookup and nsupdate to enable DDNS support")
- endif()
-endif()
diff --git a/doc/man/ouroboros-tutorial.7 b/doc/man/ouroboros-tutorial.7
index 1fc02a02..54abba5b 100644
--- a/doc/man/ouroboros-tutorial.7
+++ b/doc/man/ouroboros-tutorial.7
@@ -47,37 +47,56 @@ The output should be
.SH PINGING A SERVER APPLICATION OVER THE LOOPBACK ADAPTER
With a running irmd, let's create an IPC process. For this tutorial,
-we will create and bootstrap an eth-llc IPCP over the loopback
-interface. To observe what's going on, open another terminal
-window. Note that "ipcp bootstrap" will create an IPCP if an IPCP by
-that name does not yet exist (See \fBouroboros\fR(8)).
+we will create a unicast IPCP, attach it to the loopback interface and
+bootstrap it. To observe what's going on, open another terminal
+window.
.RS 4
-$ irm ipcp bootstrap type eth-llc name llc layer llc if lo
+$ irm ipcp create name lan type unicast
.RE
.RS 4
==23918== irmd(II): Created IPCP 23932.
-.br
-==23932== ipcpd/eth-llc(II): Using raw socket device.
-.br
-==23918== irmd(II): Bootstrapped IPCP 23932 in layer llc.
+.RE
+
+An IPCP needs a \fIpoint of attachment\fR: the transmission technology
+it sends and receives on. Attach it to the loopback interface (See
+\fBouroboros\fR(8)).
+
+.RS 4
+$ irm ipcp poa attach name lan eth dev lo
+.RE
+
+.RS 4
+==23918== irmd(II): Attached IPCP 23932.
+.RE
+
+.RS 4
+$ irm ipcp bootstrap name lan layer lan
+.RE
+
+.RS 4
+==23918== irmd(II): Bootstrapped IPCP 23932.
.RE
Now that we have the IPCP bootstrapped, it can act as a local network
layer that can provide full connectivity between all processes in the
system. Let's test it using the oping application. First, let's choose
-a name for the server ("my.oping.server") and register in the llc
+a name for the server ("my.oping.server") and register it in the lan
layer.
.RS 4
-$ irm reg name my.oping.server layer llc
+$ irm name create my.oping.server
+.br
+$ irm name register my.oping.server layer lan
.RE
The IRMd should respond with
.RS 4
-==23918== irmd(II): Registered my.oping.server in llc as 716016b1.
+==23918== irmd(II): Created new name: my.oping.server.
+.br
+==23918== irmd(II): Registered my.oping.server with IPCP 23932 as 716016b1.
.RE
Now start a server of oping in the background (or in a different
diff --git a/doc/man/ouroboros.8 b/doc/man/ouroboros.8
index 759b1433..4c83e5a9 100644
--- a/doc/man/ouroboros.8
+++ b/doc/man/ouroboros.8
@@ -98,10 +98,13 @@ accessed by other processes.
In order to enroll an IPC process in a layer, some other member will
have to be reachable over a lower layer. IPCPs that wrap a legacy
transmission technology are all bootstrapped and thus need not enroll
-as they work directly over a physical connection. Ouroboros currently
-supports IPCPs over shared memory (local), L2 (eth-llc and eth-dix)
-and L3 (udp). The unicast and broadcast layers require connections to
-be established between IPCP components for its operation.
+as they work directly over a physical connection. A unicast or
+broadcast IPCP attaches to a transmission technology itself, as a
+\fIpoint of attachment\fR (PoA); Ouroboros currently supports PoAs on
+L2 (Ethernet) and L3 (UDP/IPv4 and UDP/IPv6), next to IPCPs over
+shared memory (local). The unicast and broadcast layers require
+connections to be established between IPCP components for its
+operation.
\fBConnecting the management components\fR using \fImanagement
flows\fR allows management information to be sent between IPCPs so
@@ -123,14 +126,6 @@ creates an IPCP process of type \fItype\fR in the system with name
.PP
\fBlocal\fR - create a loopback IPCP.
.PP
-\fBeth-llc\fR - create an IPCP that attaches to Ethernet using LLC frames.
-.PP
-\fBeth-dix\fR - create an IPCP that attaches to Ethernet using DIX frames.
-.PP
-\fBudp4\fR - create an IPCP that attaches to a UDP/IPv4 socket.
-.PP
-\fBudp6\fR - create an IPCP that attaches to a UDP/IPv6 socket.
-.PP
\fBunicast\fR - create a unicast IPCP that uses lower level layers.
.PP
\fBbroadcast\fR - create a broadcast IPCP that uses lower level layers.
@@ -161,69 +156,9 @@ Values for [\fIparam\fR] are dependent on \fItype\fR:
default: SHA3_256.
.RE
-.PP
-\fBeth-llc\fR
-.RS 4
-.PP
-dev \fIinterface\fR specifies the interface to bind the IPCP to.
-.PP
-[hash \fIpolicy\fR] specifies the hash function used for the directory,
-.br
-\fIpolicy\fR: SHA3_224, SHA3_256, SHA3_384, SHA3_512.
-.br
-default: SHA3_256.
-.RE
-.PP
-\fBeth-dix\fR
-.RS 4
-.PP
-dev \fIinterface\fR specifies the interface to bind the IPCP to.
-.PP
-[ethertype \fIethertype\fR] specifies the ethertype used for the layer.
-.br
-default: 0xA000.
-.PP
-[hash \fIpolicy\fR] specifies the hash function used for the directory,
-.br
-\fIpolicy\fR: SHA3_224, SHA3_256, SHA3_384, SHA3_512.
-.br
-default: SHA3_256.
-.RE
-.PP
-\fBudp4\fR
-.RS 4
-.PP
-ip \fIip\fR specifies the local IPv4 address to bind to
-.PP
-[dns \fIdns\fR] specifies an optional DDNS server that will be used for
-the directory.
-.PP
-[port \fIport\fR] specifies a UDP port that is used for sending and
-receiving ouroboros traffic. This must be the same for the entire UDP4
-layer. Parallel UDP4 layers should use different ports. This UDP port
-needs to be forwarded if the server is behind a NAT and wants to
-receive incoming requests.
-.br
-default: 3435
-.RE
-.PP
-\fBudp6\fR
-.RS 4
-.PP
-ip \fIip\fR specifies the local IPv6 address to bind to
-.PP
-[dns \fIdns\fR] specifies an optional DDNS server that will be used for
-the directory.
-.PP
-[port \fIport\fR] specifies a UDP port that is used for sending and
-receiving ouroboros traffic. This must be the same for the entire UDP6
-layer. Parallel UDP6 layers should use different ports.
-.br
-default: 3435
-.RE
.PP
\fBunicast\fR
@@ -293,6 +228,38 @@ default: SHA3_256.
.RE
.PP
+\fBirm ipcp poa attach\fR name \fIname\fR \fIpoa\fR
+.RS 4
+attaches the IPCP with name \fIname\fR to a point of attachment.
+Repeat to attach more than one. Exactly one \fIpoa\fR is given:
+.PP
+udp \fIip\fR[:\fIport\fR] attaches to a local IPv4 or IPv6 address.
+IPv6 addresses need brackets when a port is given: [\fIip\fR]:\fIport\fR.
+.br
+default port: 3435.
+.PP
+eth dev \fIdevice\fR [ethertype \fIethertype\fR] attaches to an Ethernet
+device.
+.br
+default: 0xA000.
+.RE
+
+.PP
+\fBirm ipcp poa detach\fR name \fIname\fR \fIpoa\fR
+.RS 4
+releases a point of attachment that the IPCP with name \fIname\fR is
+attached to. Flows over it are brought down first. \fIpoa\fR takes the
+same values as for \fBattach\fR.
+.RE
+
+.PP
+\fBirm ipcp poa list\fR name \fIname\fR
+.RS 4
+lists the points of attachment that the IPCP with name \fIname\fR is
+attached to.
+.RE
+
+.PP
\fBirm ipcp enroll\fR name \fIname\fR [type \fItype\fR] [dst \fIdst\fR] \
[layer \fIlayer\fR] [\fIautobind\fR]
.RS 4
@@ -308,6 +275,15 @@ the layer name is a shorthand for the destination name being the same
as the layer name.
.PP
[autobind] will automatically bind this IPCP to its name and the layer name.
+.PP
+[udp \fIip\fR[:\fIport\fR]] enrols over a point of attachment this
+IPCP is attached to, at a peer reachable at this IPv4 or IPv6 address
+or host name.
+.PP
+[eth [dev \fIdevice\fR] [ethertype \fIethertype\fR]] enrols over a
+point of attachment this IPCP is attached to, resolving the peer by a
+name query over Ethernet. [dev \fIdevice\fR] restricts the query to
+one attached device.
.RE
\fBirm ipcp connect\fR name \fIname\fR component \fIcomponent\fR dst
@@ -316,6 +292,18 @@ as the layer name.
connects a \fIcomponent\fR (\fBdt\fR or \fBmgmt\fR) of a unicast or
broadcast IPCP with name \fIname\fR to that component of the
destination IPCP within the same layer.
+.PP
+[qos \fIqos\fR] specifies the QoS cube for a \fBdt\fR flow: raw, safe,
+rt, rt-safe or msg.
+.PP
+[udp \fIip\fR[:\fIport\fR]] connects over a point of attachment
+this IPCP is attached to, to a peer reachable at this IPv4 or IPv6
+address or host name.
+.PP
+[eth [dev \fIdevice\fR] [ethertype \fIethertype\fR]] connects over a
+point of attachment this IPCP is attached to, resolving the peer by a
+name query over Ethernet. [dev \fIdevice\fR] restricts the query to
+one attached device.
.RE
\fBirm ipcp disconnect\fR name \fIname\fR component \fIcomponent\fR dst
diff --git a/include/ouroboros/ipcp-dev.h b/include/ouroboros/ipcp-dev.h
index d00d6f08..8677adf0 100644
--- a/include/ouroboros/ipcp-dev.h
+++ b/include/ouroboros/ipcp-dev.h
@@ -24,55 +24,106 @@
#define OUROBOROS_LIB_IPCP_DEV_H
#include <ouroboros/ipcp.h>
+#include <ouroboros/qos.h>
#include <ouroboros/qoscube.h>
#include <ouroboros/ssm_pool.h>
#include <ouroboros/utils.h>
+#include <stdbool.h>
#include <stdint.h>
+#include <time.h>
-int ipcp_create_r(const struct ipcp_info * info);
+int ipcp_create_r(const struct ipcp_info * info);
-int ipcp_flow_req_arr(const buffer_t * dst,
- qosspec_t qs,
- time_t mpl,
- uint32_t mtu,
- const buffer_t * data);
+int ipcp_flow_req_arr(const buffer_t * dst,
+ qosspec_t qs,
+ time_t mpl,
+ uint32_t mtu,
+ const buffer_t * data);
-int ipcp_flow_update_arr(int flow_id,
- const buffer_t * data);
-
-int ipcp_flow_alloc_reply(int fd,
- int response,
- time_t mpl,
- uint32_t mtu,
+int ipcp_flow_update_arr(int flow_id,
const buffer_t * data);
-int ipcp_flow_read(int fd,
- struct ssm_pk_buff ** spb);
+int ipcp_flow_alloc_reply(int fd,
+ int response,
+ time_t mpl,
+ uint32_t mtu,
+ const buffer_t * data);
+
+int ipcp_flow_read(int fd,
+ struct ssm_pk_buff ** spb);
+
+int ipcp_flow_write(int fd,
+ struct ssm_pk_buff * spb);
+
+int np1_flow_read(int fd,
+ struct ssm_pk_buff ** spb,
+ struct ssm_pool * pool);
+
+int np1_flow_write(int fd,
+ struct ssm_pk_buff * spb,
+ struct ssm_pool * pool);
+
+int ipcp_flow_dealloc(int fd);
+
+int ipcp_flow_fini(int fd);
+
+int ipcp_flow_get_qoscube(int fd,
+ qoscube_t * cube);
+
+size_t ipcp_flow_queued(int fd);
-int ipcp_flow_write(int fd,
- struct ssm_pk_buff * spb);
+/*
+ * Identifies the transmit queue behind the flow. Flows attached to
+ * one point of attachment share a queue and answer the same id.
+ */
+int ipcp_flow_queue_id(int fd);
+
+int ipcp_spb_reserve(struct ssm_pk_buff ** spb,
+ size_t len);
+
+void ipcp_spb_release(struct ssm_pk_buff * spb);
+
+/* PoA lifecycle; an IPCP owns its PoAs, applications have none. */
+int poa_init(const char * name);
-int np1_flow_read(int fd,
- struct ssm_pk_buff ** spb,
- struct ssm_pool * pool);
+int poa_start(void);
-int np1_flow_write(int fd,
- struct ssm_pk_buff * spb,
- struct ssm_pool * pool);
+void poa_stop(void);
-int ipcp_flow_dealloc(int fd);
+void poa_fini(void);
-int ipcp_flow_fini(int fd);
+/* Also answer name queries for the layer; set at bootstrap/enroll. */
+int poa_set_layer(const char * layer);
-int ipcp_flow_get_qoscube(int fd,
- qoscube_t * cube);
+/* Attach or release a PoA; the type in the spec picks the transport. */
+int poa_attach(const struct poa_spec * poa);
-size_t ipcp_flow_queued(int fd);
+int poa_detach(const struct poa_spec * poa);
+
+/* The PoAs this process has attached. */
+ssize_t poa_list(struct poa_spec * specs,
+ size_t max);
+
+/* Allocate a flow to dst over the PoA that carries addr. */
+int poa_flow_alloc(const char * dst,
+ const struct poa_addr * addr,
+ qosspec_t * qs,
+ const struct timespec * timeo);
+
+/* Resolve dst on the attached PoAs; no flow is created. */
+int poa_query(const char * dst,
+ const struct timespec * timeo,
+ struct poa_addr * addr);
+
+/* PoA counterparts of the ipcp_flow_* operations */
+int poa_flow_alloc_resp(int flow_id,
+ int response,
+ const buffer_t * data);
-int ipcp_spb_reserve(struct ssm_pk_buff ** spb,
- size_t len);
+int poa_flow_update(int flow_id,
+ const buffer_t * data);
-void ipcp_spb_release(struct ssm_pk_buff * spb);
+int poa_flow_dealloc(int flow_id);
#endif /* OUROBOROS_LIB_IPCP_DEV_H */
diff --git a/include/ouroboros/ipcp.h b/include/ouroboros/ipcp.h
index 135b8fcb..4da3b81d 100644
--- a/include/ouroboros/ipcp.h
+++ b/include/ouroboros/ipcp.h
@@ -48,13 +48,19 @@ enum ipcp_type { /* IRMd uses order to select an IPCP for flow allocation. */
IPCP_LOCAL = 0,
IPCP_UNICAST,
IPCP_BROADCAST,
- IPCP_ETH_LLC,
- IPCP_ETH_DIX,
- IPCP_UDP4,
- IPCP_UDP6,
IPCP_INVALID
};
+/* The medium a PoA sits on. */
+enum poa_type {
+ POA_UDP4 = 0,
+ POA_UDP6,
+ /* A name to resolve; the family is DNS's pick. Port in udp4. */
+ POA_UDP,
+ POA_ETH,
+ POA_INVALID
+};
+
struct ipcp_info {
enum ipcp_type type;
pid_t pid;
@@ -256,23 +262,52 @@ static const struct uni_config default_uni_config = {
.cong_avoid = CA_MB_ECN
};
-struct eth_config {
+#define POA_UDP_PORT 3435 /* default UDP PoA port */
+#define POA_ETHERTYPE 0xA000 /* default Ethertype */
+#define POA_MAC_SIZE 6
+#define POA_HOST_STRLEN 255
+
+struct eth_poa {
char dev[DEV_NAME_SIZE + 1];
- uint16_t ethertype; /* DIX only*/
+ uint16_t ethertype;
+ uint8_t mac[POA_MAC_SIZE];
};
-struct udp4_config {
+struct udp4_poa {
struct in_addr ip_addr;
- struct in_addr dns_addr;
uint16_t port;
};
-struct udp6_config {
+struct udp6_poa {
struct in6_addr ip_addr;
- struct in6_addr dns_addr;
uint16_t port;
};
+/* PoA a unicast or broadcast IPCP attaches to. */
+struct poa_spec {
+ enum poa_type type;
+ union {
+ struct udp4_poa udp4;
+ struct udp6_poa udp6;
+ struct eth_poa eth;
+ };
+};
+
+
+/* Peer PoA to connect to. */
+struct poa_addr {
+ enum poa_type type;
+ char hostname[POA_HOST_STRLEN + 1]; /* resolved by IRMd */
+ union {
+ struct udp4_poa udp4;
+ struct udp6_poa udp6;
+ struct {
+ struct eth_poa src;
+ struct eth_poa dst;
+ } eth;
+ };
+};
+
/* Layers */
struct layer_info {
char name[LAYER_NAME_SIZE + 1];
@@ -285,58 +320,22 @@ struct ipcp_config {
struct layer_info layer_info;
enum ipcp_type type;
- union {
- struct uni_config unicast;
- struct udp4_config udp4;
- struct udp6_config udp6;
- struct eth_config eth;
- };
+ struct uni_config unicast;
};
/* default configurations */
static const struct ipcp_config local_default_conf = {
- .type = IPCP_LOCAL,
- .layer_info = {
- .dir_hash_algo = DIR_HASH_SHA3_256
- }
-};
-
-static const struct ipcp_config eth_dix_default_conf = {
- .type = IPCP_ETH_DIX,
.layer_info = {
.dir_hash_algo = DIR_HASH_SHA3_256
},
- .eth = {
- .ethertype=0xA000,
- }
-};
-
-static const struct ipcp_config eth_llc_default_conf = {
- .type = IPCP_ETH_LLC,
- .layer_info = {
- .dir_hash_algo = DIR_HASH_SHA3_256
- }
-};
-
-static const struct ipcp_config udp4_default_conf = {
- .type = IPCP_UDP4,
- .udp4 = {
- .port = 3435
- }
-};
-
-static const struct ipcp_config udp6_default_conf = {
- .type = IPCP_UDP6,
- .udp6 = {
- .port = 3435
- }
+ .type = IPCP_LOCAL
};
static const struct ipcp_config uni_default_conf = {
- .type = IPCP_UNICAST,
.layer_info = {
.dir_hash_algo = DIR_HASH_SHA3_256
},
+ .type = IPCP_UNICAST,
.unicast = {
.dt = {
.addr_size = 4,
diff --git a/include/ouroboros/irm.h b/include/ouroboros/irm.h
index 7cb71c21..3e8a24af 100644
--- a/include/ouroboros/irm.h
+++ b/include/ouroboros/irm.h
@@ -46,16 +46,29 @@ int irm_destroy_ipcp(pid_t pid);
ssize_t irm_list_ipcps(struct ipcp_list_info ** ipcps);
-int irm_enroll_ipcp(pid_t pid,
- const char * dst);
+int irm_enroll_ipcp(pid_t pid,
+ const char * dst,
+ const struct poa_addr * addr);
int irm_bootstrap_ipcp(pid_t pid,
const struct ipcp_config * conf);
-int irm_connect_ipcp(pid_t pid,
- const char * dst,
- const char * component,
- qosspec_t qs);
+/* Attach an IPCP to a medium; repeat for more than one PoA. */
+int irm_attach_ipcp(pid_t pid,
+ const struct poa_spec * poa);
+
+int irm_detach_ipcp(pid_t pid,
+ const struct poa_spec * poa);
+
+/* Caller frees *poas. */
+ssize_t irm_list_poas(pid_t pid,
+ struct poa_spec ** poas);
+
+int irm_connect_ipcp(pid_t pid,
+ const char * dst,
+ const char * component,
+ qosspec_t qs,
+ const struct poa_addr * addr);
int irm_disconnect_ipcp(pid_t pid,
const char * dst,
diff --git a/include/ouroboros/protobuf.h b/include/ouroboros/protobuf.h
index 951e8fd6..7937f86b 100644
--- a/include/ouroboros/protobuf.h
+++ b/include/ouroboros/protobuf.h
@@ -24,9 +24,9 @@
#define OUROBOROS_LIB_PROTOBUF_H
#include <ouroboros/flow.h>
-#include <ouroboros/qos.h>
#include <ouroboros/ipcp.h>
#include <ouroboros/irm.h>
+#include <ouroboros/qos.h>
#include <ouroboros/serdes-irm.h>
#include <ouroboros/serdes-oep.h>
@@ -37,9 +37,10 @@ typedef RoutingConfigMsg routing_config_msg_t;
typedef DtConfigMsg dt_config_msg_t;
typedef DirConfigMsg dir_config_msg_t;
typedef DirDhtConfigMsg dir_dht_config_msg_t;
-typedef EthConfigMsg eth_config_msg_t;
-typedef Udp4ConfigMsg udp4_config_msg_t;
-typedef Udp6ConfigMsg udp6_config_msg_t;
+typedef EthPoaMsg eth_poa_msg_t;
+typedef Udp4PoaMsg udp4_poa_msg_t;
+typedef PoaSpecMsg poa_spec_msg_t;
+typedef Udp6PoaMsg udp6_poa_msg_t;
typedef UniConfigMsg uni_config_msg_t;
#include "ipcp.pb-c.h"
@@ -56,6 +57,7 @@ typedef FlowInfoMsg flow_info_msg_t;
typedef NameInfoMsg name_info_msg_t;
typedef LayerInfoMsg layer_info_msg_t;
typedef QosspecMsg qosspec_msg_t;
+typedef PoaAddrMsg poa_addr_msg_t;
#include "enroll.pb-c.h"
typedef EnrollReqMsg enroll_req_msg_t;
@@ -79,6 +81,10 @@ layer_info_msg_t * layer_info_s_to_msg(const struct layer_info * s);
struct layer_info layer_info_msg_to_s(const layer_info_msg_t * msg);
+poa_addr_msg_t * poa_addr_s_to_msg(const struct poa_addr * s);
+
+struct poa_addr poa_addr_msg_to_s(const poa_addr_msg_t * msg);
+
ipcp_info_msg_t * ipcp_info_s_to_msg(const struct ipcp_info * s);
struct ipcp_info ipcp_info_msg_to_s(const ipcp_info_msg_t * msg);
@@ -91,20 +97,24 @@ uni_config_msg_t * uni_config_s_to_msg(const struct uni_config * s);
struct uni_config uni_config_msg_to_s(const uni_config_msg_t * msg);
-eth_config_msg_t * eth_config_s_to_msg(const struct eth_config * s);
+eth_poa_msg_t * eth_poa_s_to_msg(const struct eth_poa * s);
-struct eth_config eth_config_msg_to_s(const eth_config_msg_t * msg);
+struct eth_poa eth_poa_msg_to_s(const eth_poa_msg_t * msg);
-udp4_config_msg_t * udp4_config_s_to_msg(const struct udp4_config * s);
+udp4_poa_msg_t * udp4_poa_s_to_msg(const struct udp4_poa * s);
-struct udp4_config udp4_config_msg_to_s(const udp4_config_msg_t * msg);
+struct udp4_poa udp4_poa_msg_to_s(const udp4_poa_msg_t * msg);
-udp6_config_msg_t * udp6_config_s_to_msg(const struct udp6_config * s);
+udp6_poa_msg_t * udp6_poa_s_to_msg(const struct udp6_poa * s);
-struct udp6_config udp6_config_msg_to_s(const udp6_config_msg_t * msg);
+struct udp6_poa udp6_poa_msg_to_s(const udp6_poa_msg_t * msg);
ipcp_config_msg_t * ipcp_config_s_to_msg(const struct ipcp_config * s);
+poa_spec_msg_t * poa_spec_s_to_msg(const struct poa_spec * s);
+
+struct poa_spec poa_spec_msg_to_s(const poa_spec_msg_t * msg);
+
struct ipcp_config ipcp_config_msg_to_s(const ipcp_config_msg_t * msg);
/* QoS */
diff --git a/include/ouroboros/qos.h b/include/ouroboros/qos.h
index 7980ad00..26fc2637 100644
--- a/include/ouroboros/qos.h
+++ b/include/ouroboros/qos.h
@@ -30,9 +30,9 @@
/* qos_spec.service: framing / reliability class. */
enum qos_service {
- SVC_RAW = 0, /* No FRCT; best-effort raw messages */
- SVC_MESSAGE = 1, /* FRCT, reliable ordered messages */
- SVC_STREAM = 2, /* FRCT, reliable ordered byte stream */
+ SVC_RAW = 0, /* No FRCT; best-effort raw messages */
+ SVC_MESSAGE = 1, /* FRCT, ordered messages, optional reliability */
+ SVC_STREAM = 2, /* FRCT, reliable ordered byte stream */
};
typedef struct qos_spec {
diff --git a/include/ouroboros/serdes-irm.h b/include/ouroboros/serdes-irm.h
index a5854d5b..bc3c46d2 100644
--- a/include/ouroboros/serdes-irm.h
+++ b/include/ouroboros/serdes-irm.h
@@ -52,6 +52,23 @@ int ipcp_flow_req_arr__irm_req_ser(buffer_t * buf,
const struct flow_info * flow,
const buffer_t * data);
+int poa_flow_alloc__irm_req_ser(buffer_t * buf,
+ const struct flow_info * flow,
+ const char * dst);
+
+int poa_flow_alloc_r__irm_req_ser(buffer_t * buf,
+ const struct flow_info * flow,
+ const buffer_t * data,
+ int response);
+
+int ipcp_poa_flow_req_arr__irm_req_ser(buffer_t * buf,
+ const struct flow_info * flow,
+ const buffer_t * data);
+
+int poa_flow__irm_result_des(buffer_t * buf,
+ struct flow_info * flow,
+ buffer_t * data);
+
int ipcp_flow_update_arr__irm_req_ser(buffer_t * buf,
const struct flow_info * flow,
const buffer_t * data);
diff --git a/irmd.conf.in b/irmd.conf.in
index 02cb1da3..c3cfe5e5 100644
--- a/irmd.conf.in
+++ b/irmd.conf.in
@@ -7,10 +7,6 @@
#
# [name.<name of service>] add a new name to the system.
# [local.<name for IPCP>] add a new local IPCP to the system.
-# [eth-llc.<name for IPCP>] add a new IPCP over Ethernet (LLC) to the system.
-# [eth-dix.<name for IPCP>] add a new IPCP over Ethernet (DIX) to the system.
-# [udp4.<name for IPCP>] add a new IPCP over UDP/IPv4 to the system.
-# [udp6.<name for IPCP>] add a new IPCP over UDP/IPv6 to the system.
# [broadcast.<name of IPCP>] add a new broadcast IPCP to the system.
# [unicast.<name of IPCP>] add a new unicast IPCP to the system.
#
@@ -32,6 +28,49 @@
# enrol=<dst>: Enrol with this neigbor (broadcast and unicast only).
# conn=[<dst>]: Create mgmt and data flows with these neighbors.
# reg=[<names>]: Register these names with the IPCP.
+#
+# enrol, and each element of conn, also accept a table instead of a bare
+# name, to reach that peer over one specific point of attachment:
+#
+# {dst=<name>, eth={}} Resolve dst by query over every
+# attached eth PoA.
+# {dst=<name>, eth={dev=<device>}} Same, pinned to one device. eth
+# accepts an optional
+# ethertype=<ethertype>, default
+# 0xA000.
+# {dst=<name>, udp=<addr>[:<port>]} Literal peer; IPv4, bracketed IPv6
+# ("[::1]:3435"), bare IPv6, or a
+# host name the IRMd resolves. Port
+# defaults to 3435.
+#
+# A table names at most one of eth or udp; dst is required.
+#
+# A unicast or broadcast IPCP binds its own points of attachment. udp
+# and eth are each an array of strings or tables, one udp for both
+# address families:
+#
+# udp = [<addr>, ...] <addr> is a literal IPv4 or
+# IPv6 address, with an optional
+# port ("<addr>:<port>",
+# bracketed for IPv6:
+# "[::1]:3435"). Port defaults
+# to 3435. A table,
+# {addr=<addr>[:<port>]}, is
+# also accepted.
+# eth = [<device>, ...] <device> is a bare device
+# name, or a table
+# {dev=<device>[, ethertype=
+# <ethertype>]}. Ethertype
+# defaults to 0xA000.
+#
+# Because a table element is accepted, the array-of-tables spelling
+# still works too, e.g. [[unicast.lan1.udp]] with addr="<addr>" as a
+# separate stanza; repeat the stanza, or the array, to attach more
+# than one PoA.
+#
+# Unlike conn/enrol above, a PoA binds locally: only literal addresses
+# are accepted here, not host names.
+#
# Bootstrap options can be set as in the examples below.
# For more details on the configuration options for each of the IPCP types,
# please refer to the Ouroboros man page.
@@ -57,11 +96,11 @@ prog=["@INSTALL_DIR@/oping"] # Defaults to [].
args=["--listen"] # Defaults to disabled. Autostart server with these args.
lb="round-robin" # Defaults to spill (load-balancing options: spill, round-robin).
# server_sec_file=/path/to/sec.conf Default: @OUROBOROS_SRV_CRT_DIR@/<name>/sec.conf
-# server_crt_file=/path/to/crt.pem Default: @OUROBOROS_SRV_CRT_DIR@/<name>/crt.pem
-# server_key_file=/path/to/key.pem Default: @OUROBOROS_SRV_CRT_DIR@/<name>/key.pem
+# server_crt_file=/path/to/crt.pem Default: @OUROBOROS_SRV_CRT_DIR@/<name>/crt.pem
+# server_key_file=/path/to/key.pem Default: @OUROBOROS_SRV_CRT_DIR@/<name>/key.pem
# client_sec_file=/path/to/sec.conf Default: @OUROBOROS_CLI_CRT_DIR@/<name>/sec.conf
-# client_crt_file=/path/to/crt.pem Default: @OUROBOROS_CLI_CRT_DIR@/<name>/crt.pem
-# client_key_file=/path/to/key.pem Default: @OUROBOROS_CLI_CRT_DIR@/<name>/key.pem
+# client_crt_file=/path/to/crt.pem Default: @OUROBOROS_CLI_CRT_DIR@/<name>/crt.pem
+# client_key_file=/path/to/key.pem Default: @OUROBOROS_CLI_CRT_DIR@/<name>/key.pem
[local.local1]
bootstrap="local1" # Defaults to not set.
@@ -70,34 +109,6 @@ bootstrap="local1" # Defaults to not set.
# NAMES KNOWN
reg=["bc1", "LAN"] # Defaults to [].
-[eth-llc.ethl]
-bootstrap="ethl1" # Defaults to not set.
- # BOOTSTRAP CONFIGURATION
-dev="lo"
-# hash="SHA3_224" # Defaults to SHA3_256.
-
-[eth-dix.eth1]
-bootstrap="eth1" # Defaults to not set.
- # BOOTSTRAP CONFIGURATION
-dev="lo"
-# ethertype=0xA007 # Defaults to 0xA000.
-# hash="SHA3_224" # Defaults to SHA3_256.
-reg=["lan1"]
-
-[udp4.udp1]
-bootstrap="udp" # Defaults to not set.
- # BOOTSTRAP CONFIGURATION
-ip="127.0.0.1"
-# port=9000 # Defaults to 3435.
-# dns="127.0.0.1" # Requires a DDNS server. Disables DDNS support if not set.
-
-[udp6.udp2]
-bootstrap="udp2" # Defaults to not set.
- # BOOTSTRAP CONFIGURATION
-ip="::1"
-# port=9000 # Defaults to 3435.
-# dns="::1" # Requires a DDNS server. Disables DDNS support if not set.
-
[broadcast.bc1]
bootstrap="broadcast" # Defaults to not set.
# autobind=true # Defaults to false.
@@ -130,7 +141,19 @@ autobind=true # Defaults to false.
# NAMES KNOWN
reg=["oping"] # Defaults to [].
+ # POINTS OF ATTACHMENT
+udp = ["127.0.0.1", "127.0.0.1:3436"] # Two PoAs; port defaults to 3435.
+# udp = ["127.0.0.1:9000"] # Or pick a port explicitly.
+
+[[unicast.lan1.eth]]
+dev="lo"
+# ethertype=0xA007 # Defaults to 0xA000.
+
[unicast.lan2]
enrol="LAN" # Defaults to not set.
+# enrol={dst="LAN", eth={dev="eth0"}} # Or enrol over a specific PoA.
conn=["lan1"] # Defaults to [].
+# conn=["lan1",
+# {dst="lan3", eth={}},
+# {dst="lan4", udp="10.0.0.1:3435"}] # Or connect over specific PoAs.
autobind=true # Defaults to false.
diff --git a/src/ipcpd/CMakeLists.txt b/src/ipcpd/CMakeLists.txt
index 609da54a..a84e5369 100644
--- a/src/ipcpd/CMakeLists.txt
+++ b/src/ipcpd/CMakeLists.txt
@@ -4,7 +4,6 @@
# Common sources shared by all IPCPs (absolute paths for subdirectories)
set(IPCP_SOURCES
${CMAKE_CURRENT_SOURCE_DIR}/ipcp.c
- ${CMAKE_CURRENT_SOURCE_DIR}/shim-data.c
)
set(COMMON_SOURCES
@@ -24,7 +23,3 @@ configure_file("${CMAKE_CURRENT_SOURCE_DIR}/config.h.in"
add_subdirectory(local)
add_subdirectory(broadcast)
add_subdirectory(unicast)
-if(HAVE_ETH)
- add_subdirectory(eth)
-endif()
-add_subdirectory(udp)
diff --git a/src/ipcpd/broadcast/main.c b/src/ipcpd/broadcast/main.c
index d18cac82..22b34a76 100644
--- a/src/ipcpd/broadcast/main.c
+++ b/src/ipcpd/broadcast/main.c
@@ -37,6 +37,7 @@
#include <ouroboros/logs.h>
#include <ouroboros/notifier.h>
#include <ouroboros/np1_flow.h>
+#include <ouroboros/qos.h>
#include <ouroboros/random.h>
#include <ouroboros/rib.h>
#include <ouroboros/time.h>
@@ -100,12 +101,14 @@ static void stop_components(void)
enroll_stop();
}
-static int broadcast_ipcp_enroll(const char * dst,
- struct layer_info * info)
+static int broadcast_ipcp_enroll(const char * dst,
+ const struct poa_addr * addr,
+ struct layer_info * info)
{
struct ipcp_config * conf;
- struct conn conn;
- uint8_t id[ENROLL_ID_LEN];
+ struct conn conn;
+ uint8_t id[ENROLL_ID_LEN];
+ qosspec_t qs = qos_msg;
if (random_buffer(id, ENROLL_ID_LEN) < 0) {
log_err("Failed to generate enrollment ID.");
@@ -114,7 +117,7 @@ static int broadcast_ipcp_enroll(const char * dst,
log_info_id(id, "Requesting enrollment.");
- if (connmgr_alloc(COMPID_ENROLL, dst, NULL, &conn) < 0) {
+ if (connmgr_alloc(COMPID_ENROLL, dst, &qs, addr, &conn) < 0) {
log_err_id(id, "Failed to get connection.");
goto fail_id;
}
diff --git a/src/ipcpd/common/connmgr.c b/src/ipcpd/common/connmgr.c
index 6384a45c..e0ad80cb 100644
--- a/src/ipcpd/common/connmgr.c
+++ b/src/ipcpd/common/connmgr.c
@@ -26,6 +26,7 @@
#include <ouroboros/dev.h>
#include <ouroboros/errno.h>
#include <ouroboros/fccntl.h>
+#include <ouroboros/ipcp-dev.h>
#include <ouroboros/list.h>
#include <ouroboros/logs.h>
#include <ouroboros/notifier.h>
@@ -39,6 +40,10 @@
#include <stdlib.h>
#include <string.h>
+#define CONNMGR_ETH_PROBE_TIMEO 20 /* ms, one query attempt */
+#define CONNMGR_ETH_RETRY_TIMEO 1500 /* ms, the remaining tries */
+#define CONNMGR_DHT_TIMEO 1000 /* ms, bounded lower-layer */
+
struct conn_el {
struct list_head next;
struct conn conn;
@@ -60,6 +65,14 @@ struct {
pthread_t acceptor;
} connmgr;
+static bool is_eth_query(const struct poa_addr * addr)
+{
+ static const uint8_t zero[POA_MAC_SIZE] = { 0 };
+
+ return addr->type == POA_ETH &&
+ memcmp(addr->eth.dst.mac, zero, POA_MAC_SIZE) == 0;
+}
+
static int get_id_by_name(const char * name)
{
enum comp_id i;
@@ -121,6 +134,7 @@ static int add_comp_conn(enum comp_id id,
return 0;
}
+/* qs is also an in-parameter, and flow_accept writes it back. */
static void * flow_acceptor(void * o)
{
int fd;
@@ -323,9 +337,10 @@ void connmgr_comp_fini(enum comp_id id)
memset(&connmgr.comps[id].info, 0, sizeof(connmgr.comps[id].info));
}
-int connmgr_ipcp_connect(const char * dst,
- const char * component,
- qosspec_t qs)
+int connmgr_ipcp_connect(const char * dst,
+ const char * component,
+ qosspec_t qs,
+ const struct poa_addr * addr)
{
struct conn_el * ce;
int id;
@@ -334,6 +349,11 @@ int connmgr_ipcp_connect(const char * dst,
assert(dst);
assert(component);
+ if (qs.service == SVC_STREAM) {
+ log_err("No stream service on component flows.");
+ return -ENOTSUP;
+ }
+
ce = malloc(sizeof(*ce));
if (ce == NULL) {
log_err("Out of memory.");
@@ -348,7 +368,7 @@ int connmgr_ipcp_connect(const char * dst,
pthread_cleanup_push(free, ce);
- ret = connmgr_alloc(id, dst, &qs, &ce->conn);
+ ret = connmgr_alloc(id, dst, &qs, addr, &ce->conn);
pthread_cleanup_pop(false);
@@ -414,10 +434,46 @@ int connmgr_ipcp_disconnect(const char * dst,
return 0;
}
-int connmgr_alloc(enum comp_id id,
- const char * dst,
- qosspec_t * qs,
- struct conn * conn)
+/*
+ * Without an address, a peer may be on the wire or reachable through
+ * the layer below. A PoA query is cheap and creates no flow, so it
+ * goes first; the layer below gets a bounded try before the query
+ * retries, and the last try is unbounded.
+ */
+static int alloc_any(const char * dst,
+ qosspec_t * qs)
+{
+ struct timespec probe = TIMESPEC_INIT_MS(CONNMGR_ETH_PROBE_TIMEO);
+ struct timespec retry = TIMESPEC_INIT_MS(CONNMGR_ETH_RETRY_TIMEO);
+ struct timespec below = TIMESPEC_INIT_MS(CONNMGR_DHT_TIMEO);
+ struct poa_addr addr;
+ int fd;
+
+ if (poa_query(dst, &probe, &addr) == 0) {
+ fd = poa_flow_alloc(dst, &addr, qs, NULL);
+ if (fd >= 0)
+ return fd;
+ }
+
+ fd = flow_alloc(dst, qs, &below);
+ if (fd >= 0)
+ return fd;
+
+ if (poa_query(dst, &retry, &addr) == 0) {
+ fd = poa_flow_alloc(dst, &addr, qs, NULL);
+ if (fd >= 0)
+ return fd;
+ }
+
+ return flow_alloc(dst, qs, NULL);
+}
+
+/* A literal peer address bypasses the layer below. */
+int connmgr_alloc(enum comp_id id,
+ const char * dst,
+ qosspec_t * qs,
+ const struct poa_addr * addr,
+ struct conn * conn)
{
struct comp * comp;
int fd;
@@ -428,7 +484,26 @@ int connmgr_alloc(enum comp_id id,
comp = connmgr.comps + id;
- fd = flow_alloc(dst, qs, NULL);
+ if (addr != NULL)
+ fd = poa_flow_alloc(dst, addr, qs, NULL);
+ else
+ fd = alloc_any(dst, qs);
+
+ if (fd == -EPERM && addr != NULL) {
+ log_err("No PoA attached to reach %s.", dst);
+ goto fail_alloc;
+ }
+
+ if (fd == -EINVAL && addr != NULL) {
+ log_err("More than one PoA could reach %s", dst);
+ goto fail_alloc;
+ }
+
+ if (fd == -ETIMEDOUT && addr != NULL && is_eth_query(addr)) {
+ log_err("No answer to name query for %s.", dst);
+ goto fail_alloc;
+ }
+
if (fd < 0) {
log_err("Failed to allocate flow to %s.", dst);
goto fail_alloc;
diff --git a/src/ipcpd/common/connmgr.h b/src/ipcpd/common/connmgr.h
index f48ecd1b..86f80fb0 100644
--- a/src/ipcpd/common/connmgr.h
+++ b/src/ipcpd/common/connmgr.h
@@ -24,6 +24,7 @@
#define OUROBOROS_IPCPD_COMMON_CONNMGR_H
#include <ouroboros/cep.h>
+#include <ouroboros/ipcp.h>
#include <ouroboros/qos.h>
#include "comp.h"
@@ -53,17 +54,19 @@ int connmgr_comp_init(enum comp_id id,
void connmgr_comp_fini(enum comp_id id);
-int connmgr_ipcp_connect(const char * dst,
- const char * component,
- qosspec_t qs);
+int connmgr_ipcp_connect(const char * dst,
+ const char * component,
+ qosspec_t qs,
+ const struct poa_addr * addr);
int connmgr_ipcp_disconnect(const char * dst,
const char * component);
-int connmgr_alloc(enum comp_id id,
- const char * dst,
- qosspec_t * qs,
- struct conn * conn);
+int connmgr_alloc(enum comp_id id,
+ const char * dst,
+ qosspec_t * qs,
+ const struct poa_addr * addr,
+ struct conn * conn);
int connmgr_dealloc(enum comp_id id,
struct conn * conn);
diff --git a/src/ipcpd/config.h.in b/src/ipcpd/config.h.in
index 517016cc..55b0120c 100644
--- a/src/ipcpd/config.h.in
+++ b/src/ipcpd/config.h.in
@@ -25,6 +25,7 @@
#define SYS_MAX_FLOWS @SYS_MAX_FLOWS@
#define PROC_RES_FDS @PROC_RES_FDS@
#define PROC_MAX_FLOWS @PROC_MAX_FLOWS@
+#define POA_MAX_POAS @POA_MAX_POAS@
#define SOCKET_TIMEOUT @SOCKET_TIMEOUT@
#define CONNECT_TIMEOUT @CONNECT_TIMEOUT@
@@ -54,7 +55,6 @@
#cmakedefine DISABLE_CORE_LOCK
#cmakedefine BUILD_CONTAINER
#cmakedefine IPCP_FLOW_STATS
-#cmakedefine IPCP_ETH_FLOW_STATS
#cmakedefine IPCP_DEBUG_LOCAL
#ifdef CONFIG_OUROBOROS_DEBUG
#cmakedefine DEBUG_PROTO_DHT
@@ -62,29 +62,6 @@
#cmakedefine DEBUG_PROTO_LS
#endif
-/* udp */
-#cmakedefine HAVE_DDNS
-#define NSUPDATE_EXEC "@NSUPDATE_EXECUTABLE@"
-#define NSLOOKUP_EXEC "@NSLOOKUP_EXECUTABLE@"
-#define IPCP_UDP_RD_THR @IPCP_UDP_RD_THR@
-#define IPCP_UDP_WR_THR @IPCP_UDP_WR_THR@
-#define IPCP_UDP_MPL @IPCP_UDP_MPL@
-#define IPCP_UDP4_MTU @IPCP_UDP4_MTU@
-#define IPCP_UDP6_MTU @IPCP_UDP6_MTU@
-
-/* eth */
-#cmakedefine HAVE_NETMAP
-#cmakedefine HAVE_BPF
-#cmakedefine HAVE_RAW_SOCKETS
-#cmakedefine IPCP_ETH_QDISC_BYPASS
-#define IPCP_ETH_RD_THR @IPCP_ETH_RD_THR@
-#define IPCP_ETH_WR_THR @IPCP_ETH_WR_THR@
-#define IPCP_ETH_LO_MTU @IPCP_ETH_LO_MTU@
-#define IPCP_ETH_MGMT_FRAME_SIZE @IPCP_ETH_MGMT_FRAME_SIZE@
-#define IPCP_ETH_MPL @IPCP_ETH_MPL@
-#define IPCP_ETH_SNDBUF @IPCP_ETH_SNDBUF@
-#define IPCP_ETH_RCVBUF @IPCP_ETH_RCVBUF@
-
/* local */
#define IPCP_LOCAL_MPL @IPCP_LOCAL_MPL@
#define IPCP_LOCAL_MTU @IPCP_LOCAL_MTU@
diff --git a/src/ipcpd/eth/CMakeLists.txt b/src/ipcpd/eth/CMakeLists.txt
deleted file mode 100644
index 5a36352d..00000000
--- a/src/ipcpd/eth/CMakeLists.txt
+++ /dev/null
@@ -1,21 +0,0 @@
-# Ethernet IPCPs build configuration (LLC and DIX)
-# HAVE_ETH detection is in cmake/dependencies.cmake
-
-add_executable(${IPCP_ETH_LLC_TARGET} llc.c ${IPCP_SOURCES})
-add_executable(${IPCP_ETH_DIX_TARGET} dix.c ${IPCP_SOURCES})
-
-foreach(target ${IPCP_ETH_LLC_TARGET} ${IPCP_ETH_DIX_TARGET})
- target_include_directories(${target} PRIVATE ${IPCP_INCLUDE_DIRS})
- if(HAVE_BPF AND NOT APPLE)
- target_include_directories(${target} PRIVATE ${BPF_C_INCLUDE_DIR})
- endif()
- if(HAVE_NETMAP AND NOT APPLE)
- target_compile_options(${target} PRIVATE -std=c99)
- target_include_directories(${target} PRIVATE ${NETMAP_C_INCLUDE_DIR})
- endif()
- target_link_libraries(${target} PRIVATE ouroboros-dev)
- ouroboros_target_debug_definitions(${target})
-endforeach()
-
-install(TARGETS ${IPCP_ETH_LLC_TARGET} ${IPCP_ETH_DIX_TARGET}
- RUNTIME DESTINATION ${CMAKE_INSTALL_SBINDIR})
diff --git a/src/ipcpd/eth/dix.c b/src/ipcpd/eth/dix.c
deleted file mode 100644
index cf8253bd..00000000
--- a/src/ipcpd/eth/dix.c
+++ /dev/null
@@ -1,26 +0,0 @@
-/*
- * Ouroboros - Copyright (C) 2016 - 2026
- *
- * IPC processes over Ethernet - DIX
- *
- * Dimitri Staessens <dimitri@ouroboros.rocks>
- * Sander Vrijders <sander@ouroboros.rocks>
- *
- * This program is free software; you can redistribute it and/or modify
- * it under the terms of the GNU General Public License version 2 as
- * published by the Free Software Foundation.
- *
- * This program is distributed in the hope that it will be useful,
- * but WITHOUT ANY WARRANTY; without even the implied warranty of
- * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
- * GNU General Public License for more details.
- *
- * You should have received a copy of the GNU General Public License
- * along with this program; if not, write to the Free Software
- * Foundation, Inc., http://www.fsf.org/about/contact/.
- */
-
-#define BUILD_ETH_DIX
-#define OUROBOROS_PREFIX "ipcpd/eth-dix"
-
-#include "eth.c"
diff --git a/src/ipcpd/eth/eth.c b/src/ipcpd/eth/eth.c
deleted file mode 100644
index 2d3bd249..00000000
--- a/src/ipcpd/eth/eth.c
+++ /dev/null
@@ -1,2468 +0,0 @@
-/*
- * Ouroboros - Copyright (C) 2016 - 2026
- *
- * IPC processes over Ethernet
- *
- * Dimitri Staessens <dimitri@ouroboros.rocks>
- * Sander Vrijders <sander@ouroboros.rocks>
- *
- * This program is free software; you can redistribute it and/or modify
- * it under the terms of the GNU General Public License version 2 as
- * published by the Free Software Foundation.
- *
- * This program is distributed in the hope that it will be useful,
- * but WITHOUT ANY WARRANTY; without even the implied warranty of
- * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
- * GNU General Public License for more details.
- *
- * You should have received a copy of the GNU General Public License
- * along with this program; if not, write to the Free Software
- * Foundation, Inc., http://www.fsf.org/about/contact/.
- */
-
-#if !defined(BUILD_ETH_DIX) && !defined(BUILD_ETH_LLC)
-#error Define BUILD_ETH_DIX or BUILD_ETH_LLC to build an Ethernet IPCP
-#endif
-
-#if defined(__APPLE__)
-#define _BSD_SOURCE
-#define _DARWIN_C_SOURCE
-#elif defined(__FreeBSD__)
-#define __BSD_VISIBLE 1
-#elif defined (__linux__) || defined (__CYGWIN__)
-#define _DEFAULT_SOURCE
-#else
-#define _POSIX_C_SOURCE 200112L
-#endif
-
-#include "config.h"
-
-#include <ouroboros/atomics.h>
-#include <ouroboros/endian.h>
-#include <ouroboros/hash.h>
-#include <ouroboros/errno.h>
-#include <ouroboros/list.h>
-#include <ouroboros/utils.h>
-#include <ouroboros/bitmap.h>
-#include <ouroboros/crc8.h>
-#include <ouroboros/dev.h>
-#include <ouroboros/ipcp-dev.h>
-#include <ouroboros/fqueue.h>
-#include <ouroboros/logs.h>
-#include <ouroboros/np1_flow.h>
-#include <ouroboros/time.h>
-#include <ouroboros/fccntl.h>
-#include <ouroboros/pthread.h>
-#include <ouroboros/rib.h>
-
-#ifndef IPCP_ETH_FLOW_STATS
-#undef FETCH_ADD_RELAXED
-#define FETCH_ADD_RELAXED(p, v) ((void) 0)
-#undef FETCH_SUB_RELAXED
-#define FETCH_SUB_RELAXED(p, v) ((void) 0)
-#endif
-
-#include "ipcp.h"
-#include "np1.h"
-#include "shim-data.h"
-
-#include <signal.h>
-#include <stdlib.h>
-#include <fcntl.h>
-#include <unistd.h>
-#include <string.h>
-#include <sys/socket.h>
-#include <sys/types.h>
-#include <sys/ioctl.h>
-
-#include <net/if.h>
-#include <netinet/in.h>
-
-#ifdef __linux__
-#include <linux/if_packet.h>
-#include <linux/if_ether.h>
-#include <linux/netlink.h>
-#include <linux/rtnetlink.h>
-#include <linux/sockios.h>
-#endif
-
-#ifdef __FreeBSD__
-#include <net/if_dl.h>
-#include <netinet/if_ether.h>
-#include <ifaddrs.h>
-#endif
-
-#ifdef __APPLE__
-#include <net/if_dl.h>
-#include <ifaddrs.h>
-#endif
-
-#include <poll.h>
-#include <sys/mman.h>
-
-#if defined(HAVE_NETMAP)
- #define NETMAP_WITH_LIBS
- #include <net/netmap_user.h>
-#elif defined(HAVE_BPF)
- #define BPF_DEV_MAX 256
- #define BPF_BLEN sysconf(_SC_PAGESIZE)
- #include <net/bpf.h>
-#endif
-
-#define MAC_FMT "%02x:%02x:%02x:%02x:%02x:%02x"
-#define MAC_VAL(a) \
- (uint8_t)(a)[0], (uint8_t)(a)[1], (uint8_t)(a)[2], \
- (uint8_t)(a)[3], (uint8_t)(a)[4], (uint8_t)(a)[5]
-
-
-#ifndef ETH_MAX_MTU /* In if_ether.h as of Linux 4.10. */
- #define ETH_MAX_MTU 0xFFFFU
-#endif /* ETH_MAX_MTU */
-#ifdef BUILD_ETH_DIX
- #define ETH_MTU eth_data.mtu
- #define ETH_MTU_MAX ETH_MAX_MTU
-#else
- #define ETH_MTU eth_data.mtu
- #define ETH_MTU_MAX 1500
-#endif /* BUILD_ETH_DIX */
-
-#define ETH_TYPE_LENGTH_SIZE sizeof(uint16_t)
-#define ETH_HEADER_SIZE (2 * MAC_SIZE + ETH_TYPE_LENGTH_SIZE)
-
-#if defined(BUILD_ETH_DIX)
-#define THIS_TYPE IPCP_ETH_DIX
-#define MGMT_EID 0
-#define DIX_EID_SIZE sizeof(uint16_t)
-#define DIX_LENGTH_SIZE sizeof(uint16_t)
-#define DIX_HCS_SIZE CRC8_HASH_LEN
-#define DIX_HEADER_SIZE (DIX_EID_SIZE + DIX_LENGTH_SIZE + DIX_HCS_SIZE)
-#define ETH_HEADER_TOT_SIZE (ETH_HEADER_SIZE + DIX_HEADER_SIZE)
-#define MAX_EIDS (1 << (8 * DIX_EID_SIZE))
-#define ETH_MAX_PACKET_SIZE (ETH_MTU - DIX_HEADER_SIZE)
-#define ETH_FRAME_SIZE (ETH_HEADER_SIZE + ETH_MTU_MAX)
-#elif defined(BUILD_ETH_LLC)
-#define THIS_TYPE IPCP_ETH_LLC
-#define MGMT_SAP 0x01
-#define LLC_FIELDS_SIZE 3
-#define LLC_HCS_SIZE CRC8_HASH_LEN
-#define LLC_HEADER_SIZE (LLC_FIELDS_SIZE + LLC_HCS_SIZE)
-#define ETH_HEADER_TOT_SIZE (ETH_HEADER_SIZE + LLC_HEADER_SIZE)
-#define MAX_SAPS 64
-#define ETH_MAX_PACKET_SIZE (ETH_MTU - LLC_HEADER_SIZE)
-#define ETH_FRAME_SIZE (ETH_HEADER_SIZE + ETH_MTU_MAX)
-#endif
-
-#define NAME_QUERY_TIMEO 1900 /* ms total budget */
-#define NAME_QUERY_RETRIES 3 /* retransmits, 4 attempts total */
-#define MGMT_TIMEO 100 /* ms */
-#define MGMT_FRAME_SIZE IPCP_ETH_MGMT_FRAME_SIZE
-#define ETH_RIB_PATH "eth"
-
-#define FLOW_REQ 0
-#define FLOW_REPLY 1
-#define NAME_QUERY_REQ 2
-#define NAME_QUERY_REPLY 3
-#define FLOW_IRM_UPDATE 4
-
-struct mgmt_msg {
-#if defined(BUILD_ETH_DIX)
- uint16_t seid;
- uint16_t deid;
-#elif defined(BUILD_ETH_LLC)
- uint8_t ssap;
- uint8_t dsap;
- /* QoS here for alignment */
- uint8_t code;
- uint8_t availability;
-#endif
- /* QoS parameters from spec, aligned */
- uint32_t loss;
- uint64_t bandwidth;
- uint32_t ber;
- uint32_t max_gap;
- uint32_t delay;
- uint32_t timeout;
- int32_t response;
- uint8_t service;
-#if defined (BUILD_ETH_DIX)
- uint8_t code;
- uint8_t availability;
-#endif
-} __attribute__((packed));
-
-struct eth_frame {
- uint8_t dst_hwaddr[MAC_SIZE];
- uint8_t src_hwaddr[MAC_SIZE];
-#if defined(BUILD_ETH_DIX)
- uint16_t ethertype;
- uint16_t eid;
- uint16_t length;
-#elif defined(BUILD_ETH_LLC)
- uint16_t length;
- uint8_t dsap;
- uint8_t ssap;
- uint8_t cf;
-#endif
- uint8_t hcs;
- uint8_t payload;
-} __attribute__((packed));
-
-struct ef {
-#if defined(BUILD_ETH_DIX)
- int32_t r_eid;
-#elif defined(BUILD_ETH_LLC)
- int8_t sap;
- int8_t r_sap;
-#endif
- uint8_t r_addr[MAC_SIZE];
-#ifdef IPCP_ETH_FLOW_STATS
- struct {
- time_t stamp;
- size_t p_rcv;
- size_t b_rcv;
- size_t p_dlv_f;
- size_t p_snd;
- size_t b_snd;
- size_t p_snd_f;
- } stat;
-#endif
-};
-
-struct mgmt_frame {
- struct list_head next;
- uint8_t r_addr[MAC_SIZE];
- uint8_t buf[MGMT_FRAME_SIZE];
- size_t len;
-};
-
-struct {
- struct shim_data * shim_data;
-
- int mtu;
-#ifdef __linux__
- int if_idx;
-#endif
-#if defined(HAVE_NETMAP)
- struct nm_desc * nmd;
- uint8_t hw_addr[MAC_SIZE];
- struct pollfd poll_in;
- struct pollfd poll_out;
-#elif defined(HAVE_BPF)
- int bpf;
- uint8_t hw_addr[MAC_SIZE];
-#elif defined(HAVE_RAW_SOCKETS)
- int s_fd;
- struct sockaddr_ll device;
-#endif /* HAVE_NETMAP */
-#if defined (BUILD_ETH_DIX)
- uint16_t ethertype;
-#elif defined(BUILD_ETH_LLC)
- struct bmp * saps;
- int * ef_to_fd;
-#endif
- struct ef * fd_to_ef;
- fset_t * np1_flows;
- pthread_rwlock_t flows_lock;
-#ifdef IPCP_ETH_FLOW_STATS
- struct {
- size_t n_flows;
- size_t n_rcv;
- size_t n_snd;
- size_t n_mgmt_rcv;
- size_t n_mgmt_snd;
- size_t n_bad_id;
- size_t n_dlv_f;
- size_t n_buf_f;
- size_t n_rcv_f;
- size_t n_snd_f;
- size_t kern_rcv;
- size_t kern_drp;
- } stat;
-#endif
-
- pthread_t packet_writer[IPCP_ETH_WR_THR];
- pthread_t packet_reader[IPCP_ETH_RD_THR];
-
-#ifdef __linux__
- pthread_t if_monitor;
-#endif
-
- /* Handle mgmt frames in a different thread */
- pthread_t mgmt_handler;
- pthread_mutex_t mgmt_lock;
- pthread_cond_t mgmt_cond;
- struct list_head mgmt_frames;
-} eth_data;
-
-static int eth_data_init(void)
-{
- int i;
- int ret = -ENOMEM;
- pthread_condattr_t cattr;
-
- eth_data.fd_to_ef =
- malloc(sizeof(*eth_data.fd_to_ef) * SYS_MAX_FLOWS);
- if (eth_data.fd_to_ef == NULL)
- goto fail_fd_to_ef;
-
-#ifdef BUILD_ETH_LLC
- eth_data.ef_to_fd =
- malloc(sizeof(*eth_data.ef_to_fd) * MAX_SAPS);
- if (eth_data.ef_to_fd == NULL)
- goto fail_ef_to_fd;
-
- for (i = 0; i < MAX_SAPS; ++i)
- eth_data.ef_to_fd[i] = -1;
-
- eth_data.saps = bmp_create(MAX_SAPS, 2);
- if (eth_data.saps == NULL)
- goto fail_saps;
-#endif
- eth_data.np1_flows = fset_create();
- if (eth_data.np1_flows == NULL)
- goto fail_np1_flows;
-
- for (i = 0; i < SYS_MAX_FLOWS; ++i) {
-#if defined(BUILD_ETH_DIX)
- eth_data.fd_to_ef[i].r_eid = -1;
-#elif defined(BUILD_ETH_LLC)
- eth_data.fd_to_ef[i].sap = -1;
- eth_data.fd_to_ef[i].r_sap = -1;
-#endif
- memset(&eth_data.fd_to_ef[i].r_addr, 0, MAC_SIZE);
-#ifdef IPCP_ETH_FLOW_STATS
- memset(&eth_data.fd_to_ef[i].stat, 0,
- sizeof(eth_data.fd_to_ef[i].stat));
-#endif
- }
-#ifdef IPCP_ETH_FLOW_STATS
- memset(&eth_data.stat, 0, sizeof(eth_data.stat));
-#endif
-
- eth_data.shim_data = shim_data_create();
- if (eth_data.shim_data == NULL)
- goto fail_shim_data;
-
- ret = -1;
-
- if (pthread_rwlock_init(&eth_data.flows_lock, NULL))
- goto fail_flows_lock;
-
- if (pthread_mutex_init(&eth_data.mgmt_lock, NULL))
- goto fail_mgmt_lock;
-
- if (pthread_condattr_init(&cattr))
- goto fail_condattr;
-
-#ifndef __APPLE__
- pthread_condattr_setclock(&cattr, PTHREAD_COND_CLOCK);
-#endif
-
- if (pthread_cond_init(&eth_data.mgmt_cond, &cattr))
- goto fail_mgmt_cond;
-
- pthread_condattr_destroy(&cattr);
-
- list_head_init(&eth_data.mgmt_frames);
-
- return 0;
-
- fail_mgmt_cond:
- pthread_condattr_destroy(&cattr);
- fail_condattr:
- pthread_mutex_destroy(&eth_data.mgmt_lock);
- fail_mgmt_lock:
- pthread_rwlock_destroy(&eth_data.flows_lock);
- fail_flows_lock:
- shim_data_destroy(eth_data.shim_data);
- fail_shim_data:
- fset_destroy(eth_data.np1_flows);
- fail_np1_flows:
-#ifdef BUILD_ETH_LLC
- bmp_destroy(eth_data.saps);
- fail_saps:
- free(eth_data.ef_to_fd);
- fail_ef_to_fd:
-#endif
- free(eth_data.fd_to_ef);
- fail_fd_to_ef:
- return ret;
-}
-
-static void eth_data_fini(void)
-{
-#if defined(HAVE_NETMAP)
- nm_close(eth_data.nmd);
-#elif defined(HAVE_BPF)
- close(eth_data.bpf);
-#elif defined(HAVE_RAW_SOCKETS)
- close(eth_data.s_fd);
-#endif
- pthread_cond_destroy(&eth_data.mgmt_cond);
- pthread_mutex_destroy(&eth_data.mgmt_lock);
- pthread_rwlock_destroy(&eth_data.flows_lock);
- shim_data_destroy(eth_data.shim_data);
- fset_destroy(eth_data.np1_flows);
-#ifdef BUILD_ETH_LLC
- bmp_destroy(eth_data.saps);
- free(eth_data.ef_to_fd);
-#endif
- free(eth_data.fd_to_ef);
-}
-
-#ifdef IPCP_ETH_FLOW_STATS
-static int eth_rib_read(const char * path,
- char * buf,
- size_t len)
-{
- struct ef * flow;
- int fd;
- char tmstr[RIB_TM_STRLEN];
- struct tm * tm;
- time_t stamp;
- char * entry;
-
- entry = strstr(path, RIB_SEPARATOR) + 1;
- assert(entry);
-
- if (len < 2048)
- return 0;
-
- buf[0] = '\0';
-
- if (strcmp(entry, "summary") == 0) {
- int n;
-#if defined(HAVE_RAW_SOCKETS)
- int rcvbuf = 0;
- int sndbuf = 0;
- int queued = 0;
- socklen_t optlen = sizeof(rcvbuf);
-# if defined(__linux__)
- int outq = 0;
- struct tpacket_stats tp_stats;
- socklen_t tp_len = sizeof(tp_stats);
-# endif
-
- getsockopt(eth_data.s_fd, SOL_SOCKET,
- SO_RCVBUF, &rcvbuf, &optlen);
- optlen = sizeof(sndbuf);
- getsockopt(eth_data.s_fd, SOL_SOCKET,
- SO_SNDBUF, &sndbuf, &optlen);
- ioctl(eth_data.s_fd, FIONREAD, &queued);
-# if defined(__linux__)
- ioctl(eth_data.s_fd, SIOCOUTQ, &outq);
- if (getsockopt(eth_data.s_fd, SOL_PACKET,
- PACKET_STATISTICS,
- &tp_stats, &tp_len) == 0) {
- FETCH_ADD_RELAXED(&eth_data.stat.kern_rcv,
- tp_stats.tp_packets);
- FETCH_ADD_RELAXED(&eth_data.stat.kern_drp,
- tp_stats.tp_drops);
- }
-# endif
-#endif
- n = sprintf(buf,
- "Active flows: %20zu\n"
- "Total frames received: %20zu\n"
- "Total frames sent: %20zu\n"
- "Management frames received: %20zu\n"
- "Management frames sent: %20zu\n"
- "Bad EID/SAP frames: %20zu\n"
- "Delivery (N+1) failures: %20zu\n"
- "Buffer alloc failures: %20zu\n"
- "Frame read failures: %20zu\n"
- "Frame send failures: %20zu\n",
- LOAD_RELAXED(&eth_data.stat.n_flows),
- LOAD_RELAXED(&eth_data.stat.n_rcv),
- LOAD_RELAXED(&eth_data.stat.n_snd),
- LOAD_RELAXED(&eth_data.stat.n_mgmt_rcv),
- LOAD_RELAXED(&eth_data.stat.n_mgmt_snd),
- LOAD_RELAXED(&eth_data.stat.n_bad_id),
- LOAD_RELAXED(&eth_data.stat.n_dlv_f),
- LOAD_RELAXED(&eth_data.stat.n_buf_f),
- LOAD_RELAXED(&eth_data.stat.n_rcv_f),
- LOAD_RELAXED(&eth_data.stat.n_snd_f));
-#if defined(HAVE_RAW_SOCKETS)
- n += sprintf(buf + n,
- "Socket rcvbuf (bytes): %20d\n"
- "Socket sndbuf (bytes): %20d\n"
- "Socket ingress (bytes): %20d\n",
- rcvbuf, sndbuf, queued);
-# if defined(__linux__)
- n += sprintf(buf + n,
- "Socket egress (bytes): %20d\n"
- "Kernel frames received: %20zu\n"
- "Kernel frames dropped: %20zu\n",
- outq,
- LOAD_RELAXED(&eth_data.stat.kern_rcv),
- LOAD_RELAXED(&eth_data.stat.kern_drp));
-# endif
-#endif
- return n;
- }
-
- fd = atoi(entry);
-
- if (fd < 0 || fd >= SYS_MAX_FLOWS)
- return -1;
-
- flow = &eth_data.fd_to_ef[fd];
-
- pthread_rwlock_rdlock(&eth_data.flows_lock);
-
- stamp = flow->stat.stamp;
- if (stamp == 0) {
- pthread_rwlock_unlock(&eth_data.flows_lock);
- return 0;
- }
-
- pthread_rwlock_unlock(&eth_data.flows_lock);
-
- tm = gmtime(&stamp);
- strftime(tmstr, sizeof(tmstr), RIB_TM_FORMAT, tm);
-
- sprintf(buf,
- "Flow established at: %20s\n"
- "Sent (packets): %20zu\n"
- "Sent (bytes): %20zu\n"
- "Send failed (packets): %20zu\n"
- "Received (packets): %20zu\n"
- "Received (bytes): %20zu\n"
- "Delivery (N+1) failures: %20zu\n",
- tmstr,
- LOAD_RELAXED(&flow->stat.p_snd),
- LOAD_RELAXED(&flow->stat.b_snd),
- LOAD_RELAXED(&flow->stat.p_snd_f),
- LOAD_RELAXED(&flow->stat.p_rcv),
- LOAD_RELAXED(&flow->stat.b_rcv),
- LOAD_RELAXED(&flow->stat.p_dlv_f));
-
- return strlen(buf);
-}
-
-static int eth_rib_readdir(char *** buf)
-{
- char entry[RIB_PATH_LEN + 1];
- size_t i;
- int idx = 0;
- int n_entries;
-
- pthread_rwlock_rdlock(&eth_data.flows_lock);
-
- n_entries = (int) LOAD_RELAXED(&eth_data.stat.n_flows) + 1;
-
- *buf = malloc(sizeof(**buf) * n_entries);
- if (*buf == NULL)
- goto fail_entries;
-
- (*buf)[idx] = malloc(strlen("summary") + 1);
- if ((*buf)[idx] == NULL)
- goto fail_entry;
-
- strcpy((*buf)[idx++], "summary");
-
- for (i = 0; i < SYS_MAX_FLOWS && idx < n_entries; ++i) {
- if (eth_data.fd_to_ef[i].stat.stamp == 0)
- continue;
-
- sprintf(entry, "%zu", i);
-
- (*buf)[idx] = malloc(strlen(entry) + 1);
- if ((*buf)[idx] == NULL)
- goto fail_entry;
-
- strcpy((*buf)[idx++], entry);
- }
-
- pthread_rwlock_unlock(&eth_data.flows_lock);
-
- return idx;
-
- fail_entry:
- while (idx-- > 0)
- free((*buf)[idx]);
-
- free(*buf);
- fail_entries:
- pthread_rwlock_unlock(&eth_data.flows_lock);
- return -ENOMEM;
-}
-
-static int eth_rib_getattr(const char * path,
- struct rib_attr * attr)
-{
- int fd;
- char * entry;
- struct ef * flow;
-
- entry = strstr(path, RIB_SEPARATOR) + 1;
- assert(entry);
-
- if (strcmp(entry, "summary") == 0) {
- attr->size = 2048;
- attr->mtime = 0;
- return 0;
- }
-
- fd = atoi(entry);
-
- if (fd < 0 || fd >= SYS_MAX_FLOWS) {
- attr->size = 0;
- attr->mtime = 0;
- return 0;
- }
-
- flow = &eth_data.fd_to_ef[fd];
-
- pthread_rwlock_rdlock(&eth_data.flows_lock);
-
- if (flow->stat.stamp != 0) {
- attr->size = 2048;
- attr->mtime = flow->stat.stamp;
- } else {
- attr->size = 0;
- attr->mtime = 0;
- }
-
- pthread_rwlock_unlock(&eth_data.flows_lock);
-
- return 0;
-}
-
-static struct rib_ops eth_r_ops = {
- .read = eth_rib_read,
- .readdir = eth_rib_readdir,
- .getattr = eth_rib_getattr
-};
-#endif /* IPCP_ETH_FLOW_STATS */
-
-#ifdef BUILD_ETH_LLC
-static uint8_t reverse_bits(uint8_t b)
-{
- b = (b & 0xF0) >> 4 | (b & 0x0F) << 4;
- b = (b & 0xCC) >> 2 | (b & 0x33) << 2;
- b = (b & 0xAA) >> 1 | (b & 0x55) << 1;
-
- return b;
-}
-#endif
-
-/* Pass a buffer that contains space for the header. */
-static int eth_ipcp_send_frame(const uint8_t * dst_addr,
-#if defined(BUILD_ETH_DIX)
- uint16_t deid,
-#elif defined(BUILD_ETH_LLC)
- uint8_t dsap,
- uint8_t ssap,
-#endif
- const uint8_t * frame,
- size_t len)
-{
- uint32_t frame_len = 0;
-#ifdef BUILD_ETH_LLC
- uint8_t cf = 0x03;
-#endif
- struct eth_frame * e_frame;
-#ifdef HAVE_RAW_SOCKETS
- fd_set fds;
-
- FD_ZERO(&fds);
-#endif
-
- assert(frame);
-
- if (len > (size_t) ETH_MAX_PACKET_SIZE)
- return -1;
-
- e_frame = (struct eth_frame *) frame;
-
- memcpy(e_frame->dst_hwaddr, dst_addr, MAC_SIZE);
- memcpy(e_frame->src_hwaddr,
-#if defined(HAVE_NETMAP) || defined(HAVE_BPF)
- eth_data.hw_addr,
-#elif defined(HAVE_RAW_SOCKETS)
- eth_data.device.sll_addr,
-#endif /* HAVE_NETMAP */
- MAC_SIZE);
-#if defined(BUILD_ETH_DIX)
- e_frame->ethertype = eth_data.ethertype;
- e_frame->eid = htons(deid);
- e_frame->length = htons(len);
- mem_hash(HASH_CRC8, &e_frame->hcs,
- (uint8_t *) &e_frame->eid,
- DIX_EID_SIZE + DIX_LENGTH_SIZE);
- frame_len = ETH_HEADER_TOT_SIZE + len;
-#elif defined(BUILD_ETH_LLC)
- e_frame->length = htons(LLC_HEADER_SIZE + len);
- e_frame->dsap = dsap;
- e_frame->ssap = ssap;
- e_frame->cf = cf;
- mem_hash(HASH_CRC8, &e_frame->hcs,
- (uint8_t *) &e_frame->dsap,
- LLC_FIELDS_SIZE);
- frame_len = ETH_HEADER_TOT_SIZE + len;
-#endif
-
-#if defined(HAVE_NETMAP)
- if (poll(&eth_data.poll_out, 1, -1) < 0)
- return -1;
-
- if (nm_inject(eth_data.nmd, frame, frame_len) != (int) frame_len) {
- log_dbg("Failed to send message.");
- return -1;
- }
-#elif defined(HAVE_BPF)
- if (write(eth_data.bpf, frame, frame_len) < 0) {
- log_dbg("Failed to send message.");
- return -1;
- }
-
-#elif defined(HAVE_RAW_SOCKETS)
- FD_SET(eth_data.s_fd, &fds);
- if (select(eth_data.s_fd + 1, NULL, &fds, NULL, NULL) < 0) {
- log_dbg("Select() failed: %s.", strerror(errno));
- return -1;
- }
- assert(FD_ISSET(eth_data.s_fd, &fds));
-
- if (sendto(eth_data.s_fd, frame, frame_len, 0,
- (struct sockaddr *) &eth_data.device,
- sizeof(eth_data.device)) <= 0) {
- log_dbg("Failed to send message: %s.", strerror(errno));
- return -1;
- }
-#endif /* HAVE_NETMAP */
-
- FETCH_ADD_RELAXED(&eth_data.stat.n_snd, 1);
-
- return 0;
-}
-
-static int eth_ipcp_alloc(const uint8_t * dst_addr,
-#if defined(BUILD_ETH_DIX)
- uint16_t eid,
-#elif defined(BUILD_ETH_LLC)
- uint8_t ssap,
-#endif
- const uint8_t * hash,
- qosspec_t qs,
- const buffer_t * data)
-{
- uint8_t * buf;
- struct mgmt_msg * msg;
- size_t len;
- int ret;
-
- len = sizeof(*msg) + ipcp_dir_hash_len();
-
- buf = malloc(len + ETH_HEADER_TOT_SIZE + data->len);
- if (buf == NULL)
- return -1;
-
- memset(buf, 0, len + ETH_HEADER_TOT_SIZE + data->len);
-
- msg = (struct mgmt_msg *) (buf + ETH_HEADER_TOT_SIZE);
- msg->code = FLOW_REQ;
-#if defined(BUILD_ETH_DIX)
- msg->seid = htons(eid);
-#elif defined(BUILD_ETH_LLC)
- msg->ssap = ssap;
-#endif
-
- msg->delay = hton32(qs.delay);
- msg->bandwidth = hton64(qs.bandwidth);
- msg->availability = qs.availability;
- msg->loss = hton32(qs.loss);
- msg->ber = hton32(qs.ber);
- msg->service = qs.service;
- msg->max_gap = hton32(qs.max_gap);
- msg->timeout = hton32(qs.timeout);
-
- memcpy(msg + 1, hash, ipcp_dir_hash_len());
- if (data->len > 0)
- memcpy(buf + len + ETH_HEADER_TOT_SIZE, data->data, data->len);
-
- ret = eth_ipcp_send_frame(dst_addr,
-#if defined(BUILD_ETH_DIX)
- MGMT_EID,
-#elif defined(BUILD_ETH_LLC)
- reverse_bits(MGMT_SAP),
- reverse_bits(MGMT_SAP),
-#endif
- buf, len + data->len);
- free(buf);
-
- if (ret == 0)
- FETCH_ADD_RELAXED(&eth_data.stat.n_mgmt_snd, 1);
-
- return ret;
-}
-
-static int eth_ipcp_alloc_resp(uint8_t * dst_addr,
-#if defined(BUILD_ETH_DIX)
- uint16_t seid,
- uint16_t deid,
-#elif defined(BUILD_ETH_LLC)
- uint8_t ssap,
- uint8_t dsap,
-#endif
- int response,
- const buffer_t * data)
-{
- struct mgmt_msg * msg;
- uint8_t * buf;
-
- buf = malloc(sizeof(*msg) + ETH_HEADER_TOT_SIZE + data->len);
- if (buf == NULL)
- return -1;
-
- memset(buf, 0, sizeof(*msg) + ETH_HEADER_TOT_SIZE + data->len);
-
- msg = (struct mgmt_msg *) (buf + ETH_HEADER_TOT_SIZE);
-
- msg->code = FLOW_REPLY;
-#if defined(BUILD_ETH_DIX)
- msg->seid = htons(seid);
- msg->deid = htons(deid);
-#elif defined(BUILD_ETH_LLC)
- msg->ssap = ssap;
- msg->dsap = dsap;
-#endif
- msg->response = hton32(response);
-
- if (data->len > 0)
- memcpy(msg + 1, data->data, data->len);
-
- if (eth_ipcp_send_frame(dst_addr,
-#if defined(BUILD_ETH_DIX)
- MGMT_EID,
-#elif defined(BUILD_ETH_LLC)
- reverse_bits(MGMT_SAP),
- reverse_bits(MGMT_SAP),
-#endif
- buf, sizeof(*msg) + data->len)) {
- free(buf);
- return -1;
- }
-
- FETCH_ADD_RELAXED(&eth_data.stat.n_mgmt_snd, 1);
-
- free(buf);
-
- return 0;
-}
-
-static int eth_ipcp_flow_update(int fd,
- const buffer_t * data)
-{
- struct mgmt_msg * msg;
- struct ef * flow;
- uint8_t * buf;
- uint8_t r_addr[MAC_SIZE];
- int ret;
-
- buf = malloc(sizeof(*msg) + ETH_HEADER_TOT_SIZE + data->len);
- if (buf == NULL)
- return -1;
-
- memset(buf, 0, sizeof(*msg) + ETH_HEADER_TOT_SIZE + data->len);
-
- msg = (struct mgmt_msg *) (buf + ETH_HEADER_TOT_SIZE);
-
- msg->code = FLOW_IRM_UPDATE;
-
- pthread_rwlock_rdlock(&eth_data.flows_lock);
-
- flow = &eth_data.fd_to_ef[fd];
-#if defined(BUILD_ETH_DIX)
- msg->seid = htons((uint16_t) fd);
- msg->deid = htons((uint16_t) flow->r_eid);
-#elif defined(BUILD_ETH_LLC)
- msg->ssap = flow->sap;
- msg->dsap = (uint8_t) flow->r_sap;
-#endif
- memcpy(r_addr, flow->r_addr, MAC_SIZE);
-
- pthread_rwlock_unlock(&eth_data.flows_lock);
-
- if (data->len > 0)
- memcpy(msg + 1, data->data, data->len);
-
- ret = eth_ipcp_send_frame(r_addr,
-#if defined(BUILD_ETH_DIX)
- MGMT_EID,
-#elif defined(BUILD_ETH_LLC)
- reverse_bits(MGMT_SAP),
- reverse_bits(MGMT_SAP),
-#endif
- buf, sizeof(*msg) + data->len);
- free(buf);
-
- if (ret == 0)
- FETCH_ADD_RELAXED(&eth_data.stat.n_mgmt_snd, 1);
-
- return ret;
-}
-
-static int eth_ipcp_req(uint8_t * r_addr,
-#if defined(BUILD_ETH_DIX)
- uint16_t r_eid,
-#elif defined(BUILD_ETH_LLC)
- uint8_t r_sap,
-#endif
- const uint8_t * dst,
- qosspec_t qs,
- const buffer_t * data)
-{
- int fd;
-
- fd = ipcp_wait_flow_req_arr(dst, qs, IPCP_ETH_MPL,
- ETH_MAX_PACKET_SIZE, data);
- if (fd < 0) {
- log_err("Could not get new flow from IRMd.");
- return -1;
- }
-
- pthread_rwlock_wrlock(&eth_data.flows_lock);
-#if defined(BUILD_ETH_DIX)
- eth_data.fd_to_ef[fd].r_eid = r_eid;
-#elif defined(BUILD_ETH_LLC)
- eth_data.fd_to_ef[fd].r_sap = r_sap;
-#endif
- memcpy(eth_data.fd_to_ef[fd].r_addr, r_addr, MAC_SIZE);
-
- pthread_rwlock_unlock(&eth_data.flows_lock);
-
-#if defined(BUILD_ETH_DIX)
- log_dbg("New flow request, fd %d, remote endpoint %d.", fd, r_eid);
-#elif defined(BUILD_ETH_LLC)
- log_dbg("New flow request, fd %d, remote SAP %d.", fd, r_sap);
-#endif
- return 0;
-}
-
-static int eth_ipcp_alloc_reply(uint8_t * r_addr,
-#if defined(BUILD_ETH_DIX)
- uint16_t seid,
- uint16_t deid,
-#elif defined(BUILD_ETH_LLC)
- uint8_t ssap,
- int dsap,
-#endif
- int response,
- const buffer_t * data)
-{
- int ret = 0;
- int fd = -1;
- time_t mpl = IPCP_ETH_MPL;
-
- pthread_rwlock_wrlock(&eth_data.flows_lock);
-
-#if defined(BUILD_ETH_DIX)
- fd = deid;
-#elif defined(BUILD_ETH_LLC)
- fd = eth_data.ef_to_fd[dsap];
-#endif
- if (fd < 0) {
- pthread_rwlock_unlock(&eth_data.flows_lock);
- log_err("No flow found with that SAP.");
- return -1; /* -EFLOWNOTFOUND */
- }
-
- if (response) {
-#ifdef BUILD_ETH_LLC
- bmp_release(eth_data.saps, eth_data.fd_to_ef[fd].sap);
-#endif
- } else {
-#if defined(BUILD_ETH_DIX)
- eth_data.fd_to_ef[fd].r_eid = seid;
-#elif defined(BUILD_ETH_LLC)
- eth_data.fd_to_ef[fd].r_sap = ssap;
-#endif
- memcpy(eth_data.fd_to_ef[fd].r_addr, r_addr, MAC_SIZE);
- }
-
- pthread_rwlock_unlock(&eth_data.flows_lock);
-
-#if defined(BUILD_ETH_DIX)
- log_dbg("Flow reply, fd %d, src eid %d, dst eid %d.", fd, seid, deid);
-#elif defined(BUILD_ETH_LLC)
- log_dbg("Flow reply, fd %d, SSAP %d, DSAP %d.", fd, ssap, dsap);
-#endif
- if ((ret = ipcp_flow_alloc_reply(fd, response, mpl,
- ETH_MAX_PACKET_SIZE, data)) < 0) {
- log_err("Failed to reply to flow allocation.");
- return -1;
- }
-
- return ret;
-}
-
-static int eth_ipcp_name_query_req(const uint8_t * hash,
- uint8_t * r_addr)
-{
- uint8_t * buf;
- struct mgmt_msg * msg;
- size_t len;
-
- if (shim_data_reg_has(eth_data.shim_data, hash)) {
- len = sizeof(*msg) + ipcp_dir_hash_len();
-
- buf = malloc(len + ETH_HEADER_TOT_SIZE);
- if (buf == NULL)
- return -1;
-
- memset(buf, 0, len + ETH_HEADER_TOT_SIZE);
-
- msg = (struct mgmt_msg *) (buf + ETH_HEADER_TOT_SIZE);
- msg->code = NAME_QUERY_REPLY;
-
- memcpy(msg + 1, hash, ipcp_dir_hash_len());
-
- if (eth_ipcp_send_frame(r_addr,
-#if defined(BUILD_ETH_DIX)
- MGMT_EID,
-#elif defined(BUILD_ETH_LLC)
- reverse_bits(MGMT_SAP),
- reverse_bits(MGMT_SAP),
-#endif
- buf, len)) {
- log_err("Failed to send management frame.");
- free(buf);
- return -1;
- }
-
- FETCH_ADD_RELAXED(&eth_data.stat.n_mgmt_snd, 1);
-
- free(buf);
- }
-
- return 0;
-}
-
-static int eth_ipcp_name_query_reply(const uint8_t * hash,
- uint8_t * r_addr)
-{
- struct addr addr;
-
- memcpy(&addr.mac, r_addr, MAC_SIZE);
-
- shim_data_dir_add_entry(eth_data.shim_data, hash, addr);
-
- shim_data_dir_query_respond(eth_data.shim_data, hash);
-
- return 0;
-}
-
-static int eth_ipcp_flow_update_arr(const uint8_t * buf,
- size_t len)
-{
- struct mgmt_msg * msg;
- buffer_t data;
- int fd;
- int flow_id;
-
- msg = (struct mgmt_msg *) buf;
-
- data.data = (uint8_t *) buf + sizeof(*msg);
- data.len = len - sizeof(*msg);
-
- pthread_rwlock_rdlock(&eth_data.flows_lock);
-#if defined(BUILD_ETH_DIX)
- fd = ntohs(msg->deid);
-#elif defined(BUILD_ETH_LLC)
- fd = eth_data.ef_to_fd[msg->dsap];
-#endif
- pthread_rwlock_unlock(&eth_data.flows_lock);
-
- if (fd < 0 || fd >= SYS_MAX_FLOWS) {
- log_err("Flow update for unknown endpoint.");
- return -1;
- }
-
- flow_id = np1_flow_id(fd);
- if (flow_id < 0)
- return -1;
-
- if (ipcp_flow_update_arr(flow_id, &data) < 0) {
- log_err("Failed to relay flow update on fd %d.", fd);
- return -1;
- }
-
- return 0;
-}
-
-static int eth_ipcp_mgmt_frame(const uint8_t * buf,
- size_t len,
- uint8_t * r_addr)
-{
- struct mgmt_msg * msg;
- size_t msg_len;
- qosspec_t qs;
- buffer_t data;
-
- if (len < sizeof(*msg))
- return -1;
-
- msg = (struct mgmt_msg *) buf;
-
- switch (msg->code) {
- case FLOW_REQ:
- msg_len = sizeof(*msg) + ipcp_dir_hash_len();
-
- if (len < msg_len)
- return -1;
-
- qs.delay = ntoh32(msg->delay);
- qs.bandwidth = ntoh64(msg->bandwidth);
- qs.availability = msg->availability;
- qs.loss = ntoh32(msg->loss);
- qs.ber = ntoh32(msg->ber);
- qs.service = msg->service;
- qs.max_gap = ntoh32(msg->max_gap);
- qs.timeout = ntoh32(msg->timeout);
-
- data.data = (uint8_t *) buf + msg_len;
- data.len = len - msg_len;
-
- if (shim_data_reg_has(eth_data.shim_data,
- buf + sizeof(*msg))) {
- eth_ipcp_req(r_addr,
-#if defined(BUILD_ETH_DIX)
- ntohs(msg->seid),
-#elif defined(BUILD_ETH_LLC)
- msg->ssap,
-#endif
- buf + sizeof(*msg),
- qs,
- &data);
- }
- break;
- case FLOW_REPLY:
- data.data = (uint8_t *) buf + sizeof(*msg);
- data.len = len - sizeof(*msg);
-
- eth_ipcp_alloc_reply(r_addr,
-#if defined(BUILD_ETH_DIX)
- ntohs(msg->seid),
- ntohs(msg->deid),
-#elif defined(BUILD_ETH_LLC)
- msg->ssap,
- msg->dsap,
-#endif
- ntoh32(msg->response),
- &data);
- break;
- case FLOW_IRM_UPDATE:
- eth_ipcp_flow_update_arr(buf, len);
- break;
- case NAME_QUERY_REQ:
- if (len < sizeof(*msg) + ipcp_dir_hash_len())
- return -1;
- eth_ipcp_name_query_req(buf + sizeof(*msg), r_addr);
- break;
- case NAME_QUERY_REPLY:
- if (len < sizeof(*msg) + ipcp_dir_hash_len())
- return -1;
- eth_ipcp_name_query_reply(buf + sizeof(*msg), r_addr);
- break;
- default:
- log_err("Unknown message received %d.", msg->code);
- return -1;
- }
-
- return 0;
-}
-
-static void * eth_ipcp_mgmt_handler(void * o)
-{
- (void) o;
-
- pthread_cleanup_push(__cleanup_mutex_unlock, &eth_data.mgmt_lock);
-
- while (true) {
- int ret = 0;
- struct timespec timeout = TIMESPEC_INIT_MS(MGMT_TIMEO);
- struct timespec abstime;
- struct mgmt_frame * frame = NULL;
-
- clock_gettime(PTHREAD_COND_CLOCK, &abstime);
- ts_add(&abstime, &timeout, &abstime);
-
- pthread_mutex_lock(&eth_data.mgmt_lock);
-
- while (list_is_empty(&eth_data.mgmt_frames) &&
- ret != -ETIMEDOUT)
- ret = -pthread_cond_timedwait(&eth_data.mgmt_cond,
- &eth_data.mgmt_lock,
- &abstime);
- if (ret != -ETIMEDOUT)
- frame = list_first_entry((&eth_data.mgmt_frames),
- struct mgmt_frame, next);
- if (frame != NULL)
- list_del(&frame->next);
-
- pthread_mutex_unlock(&eth_data.mgmt_lock);
-
- if (frame == NULL)
- continue;
-
- eth_ipcp_mgmt_frame(frame->buf, frame->len, frame->r_addr);
-
- free(frame);
- }
-
- pthread_cleanup_pop(false);
-
- return (void *) 0;
-}
-
-static void * eth_ipcp_packet_reader(void * o)
-{
- uint8_t br_addr[MAC_SIZE];
-#if defined(BUILD_ETH_DIX)
- uint16_t deid;
-#elif defined(BUILD_ETH_LLC)
- uint8_t dsap;
- uint8_t ssap;
-#endif
- uint16_t length;
- int fd;
- uint8_t * buf;
-#if defined(HAVE_NETMAP)
- struct nm_pkthdr hdr;
-#else
- struct ssm_pk_buff * spb;
- fd_set fds;
- int frame_len;
-#endif
-#if defined(HAVE_RAW_SOCKETS)
- struct sockaddr_ll src;
- socklen_t slen;
-#endif
- size_t eth_len;
- uint8_t hcs;
- struct eth_frame * e_frame;
- struct mgmt_frame * frame;
-
- (void) o;
-
- ipcp_lock_to_core();
-
- memset(br_addr, 0xff, MAC_SIZE * sizeof(uint8_t));
-
- while (true) {
-#if defined(HAVE_NETMAP)
- if (poll(&eth_data.poll_in, 1, -1) < 0)
- continue;
- if (eth_data.poll_in.revents == 0) /* TIMED OUT */
- continue;
-
- buf = nm_nextpkt(eth_data.nmd, &hdr);
- if (buf == NULL) {
- log_dbg("Bad read from netmap device.");
- continue;
- }
-#else
- FD_ZERO(&fds);
- #if defined(HAVE_BPF)
- FD_SET(eth_data.bpf, &fds);
- if (select(eth_data.bpf + 1, &fds, NULL, NULL, NULL))
- continue;
- assert(FD_ISSET(eth_data.bpf, &fds));
- if (ipcp_spb_reserve(&spb, BPF_LEN))
- continue;
- buf = ssm_pk_buff_head(spb);
- frame_len = read(eth_data.bpf, buf, BPF_BLEN);
- #elif defined(HAVE_RAW_SOCKETS)
- FD_SET(eth_data.s_fd, &fds);
- if (select(eth_data.s_fd + 1, &fds, NULL, NULL, NULL) < 0)
- continue;
- assert(FD_ISSET(eth_data.s_fd, &fds));
- if (ipcp_spb_reserve(&spb, ETH_MTU)) {
- FETCH_ADD_RELAXED(&eth_data.stat.n_buf_f, 1);
- continue;
- }
- buf = ssm_pk_buff_push(spb, ETH_HEADER_TOT_SIZE);
- if (buf == NULL) {
- log_dbg("Failed to allocate header.");
- ipcp_spb_release(spb);
- FETCH_ADD_RELAXED(&eth_data.stat.n_buf_f, 1);
- continue;
- }
- slen = sizeof(src);
- /* MSG_DONTWAIT: RD_THR>1 race-loser bails with EAGAIN. */
- frame_len = recvfrom(eth_data.s_fd, buf,
- ETH_MTU + ETH_HEADER_TOT_SIZE,
- MSG_DONTWAIT,
- (struct sockaddr *) &src, &slen);
- #endif
- if (frame_len == 0) {
- ipcp_spb_release(spb);
- continue; /* Spurious */
- }
-
- if (frame_len < 0) {
- ipcp_spb_release(spb);
-
- if (errno == EAGAIN || errno == EWOULDBLOCK)
- continue;
-
- log_dbg("Failed to rcv frame: %s.", strerror(errno));
- FETCH_ADD_RELAXED(&eth_data.stat.n_rcv_f, 1);
- continue;
- }
-#endif
-
-#if defined(HAVE_NETMAP)
- eth_len = hdr.len;
-#elif defined(HAVE_BPF)
- eth_len = ((struct bpf_hdr *) buf)->bh_caplen;
-#else
- eth_len = (size_t) frame_len;
-#endif
- /* Defense in depth: reject before parsing dereferences. */
- if (eth_len < ETH_HEADER_TOT_SIZE)
- goto fail_frame;
-
-#if defined(HAVE_RAW_SOCKETS)
- /* Drop our own egress. */
- if (src.sll_pkttype == PACKET_OUTGOING)
- goto fail_frame;
-#endif
-
-#if defined(HAVE_BPF) && !defined(HAVE_NETMAP)
- e_frame = (struct eth_frame *)
- (buf + ((struct bpf_hdr *) buf)->bh_hdrlen);
-#else
- e_frame = (struct eth_frame *) buf;
-#endif
- assert(e_frame->dst_hwaddr);
-
-#if !defined(HAVE_BPF)
- #if defined(HAVE_NETMAP)
- if (memcmp(eth_data.hw_addr,
- #elif defined(HAVE_RAW_SOCKETS)
- if (memcmp(eth_data.device.sll_addr,
- #endif /* HAVE_NETMAP */
- e_frame->dst_hwaddr,
- MAC_SIZE) &&
- memcmp(br_addr, e_frame->dst_hwaddr, MAC_SIZE)) {
- FETCH_ADD_RELAXED(&eth_data.stat.n_bad_id, 1);
- goto fail_frame;
- }
-#endif
- length = ntohs(e_frame->length);
-#if defined(BUILD_ETH_DIX)
- if (e_frame->ethertype != eth_data.ethertype)
- goto fail_frame;
-
- if (length > ETH_MTU)
- goto fail_frame;
-
- deid = ntohs(e_frame->eid);
-#elif defined (BUILD_ETH_LLC)
- if (length > 0x05FF) /* DIX */
- goto fail_frame;
-
- if (length < LLC_HEADER_SIZE || length > ETH_MTU)
- goto fail_frame;
-
- length -= LLC_HEADER_SIZE;
-
- dsap = reverse_bits(e_frame->dsap);
- ssap = reverse_bits(e_frame->ssap);
-#endif
-
- if (eth_len < ETH_HEADER_TOT_SIZE + (size_t) length)
- goto fail_frame;
-
-#if defined(BUILD_ETH_DIX)
- mem_hash(HASH_CRC8, &hcs,
- (uint8_t *) &e_frame->eid,
- DIX_EID_SIZE + DIX_LENGTH_SIZE);
-#elif defined(BUILD_ETH_LLC)
- mem_hash(HASH_CRC8, &hcs,
- (uint8_t *) &e_frame->dsap,
- LLC_FIELDS_SIZE);
-#endif
- if (hcs != e_frame->hcs)
- goto fail_frame;
-
-#if defined(BUILD_ETH_DIX)
- if (deid == MGMT_EID) {
-#elif defined (BUILD_ETH_LLC)
- if (ssap == MGMT_SAP && dsap == MGMT_SAP) {
-#endif
- ipcp_spb_release(spb); /* No need for the N+1 buffer. */
-
- if (length > MGMT_FRAME_SIZE) {
- log_warn("Management frame size %u exceeds %u.",
- length, MGMT_FRAME_SIZE);
- continue;
- }
-
- frame = malloc(sizeof(*frame));
- if (frame == NULL) {
- log_err("Failed to allocate frame.");
- continue;
- }
-
- memcpy(frame->buf, &e_frame->payload, length);
- memcpy(frame->r_addr, e_frame->src_hwaddr, MAC_SIZE);
- frame->len = length;
-
- pthread_mutex_lock(&eth_data.mgmt_lock);
- list_add(&frame->next, &eth_data.mgmt_frames);
- pthread_cond_signal(&eth_data.mgmt_cond);
- pthread_mutex_unlock(&eth_data.mgmt_lock);
- FETCH_ADD_RELAXED(&eth_data.stat.n_rcv, 1);
- FETCH_ADD_RELAXED(&eth_data.stat.n_mgmt_rcv, 1);
- } else {
- pthread_rwlock_rdlock(&eth_data.flows_lock);
-
-#if defined(BUILD_ETH_DIX)
- fd = deid;
-#elif defined(BUILD_ETH_LLC)
- fd = eth_data.ef_to_fd[dsap];
-#endif
- if (fd < 0) {
- pthread_rwlock_unlock(&eth_data.flows_lock);
- FETCH_ADD_RELAXED(&eth_data.stat.n_bad_id, 1);
- goto fail_frame;
- }
-
-#ifdef BUILD_ETH_LLC
- if (eth_data.fd_to_ef[fd].r_sap != ssap
- || memcmp(eth_data.fd_to_ef[fd].r_addr,
- e_frame->src_hwaddr, MAC_SIZE)) {
- pthread_rwlock_unlock(&eth_data.flows_lock);
- FETCH_ADD_RELAXED(&eth_data.stat.n_bad_id, 1);
- goto fail_frame;
- }
-#endif
- FETCH_ADD_RELAXED(&eth_data.fd_to_ef[fd].stat.p_rcv, 1);
- FETCH_ADD_RELAXED(&eth_data.fd_to_ef[fd].stat.b_rcv,
- length);
- FETCH_ADD_RELAXED(&eth_data.stat.n_rcv, 1);
- pthread_rwlock_unlock(&eth_data.flows_lock);
-
-#ifndef HAVE_NETMAP
- ssm_pk_buff_pop(spb, ETH_HEADER_TOT_SIZE);
- ssm_pk_buff_truncate(spb, length);
-#else
- if (ipcp_spb_reserve(&spb, length))
- continue;
-
- buf = ssm_pk_buff_head(spb);
- memcpy(buf, &e_frame->payload, length);
-#endif
- if (np1_flow_write(fd, spb, NP1_GET_POOL(fd)) < 0) {
- ipcp_spb_release(spb);
- FETCH_ADD_RELAXED(
- &eth_data.fd_to_ef[fd].stat.p_dlv_f,
- 1);
- FETCH_ADD_RELAXED(&eth_data.stat.n_dlv_f, 1);
- }
-
- continue;
- fail_frame:
-#ifndef HAVE_NETMAP
- ipcp_spb_release(spb);
-#endif
- }
- }
-
- return (void *) 0;
-}
-
-static void cleanup_writer(void * o)
-{
- fqueue_destroy((fqueue_t *) o);
-}
-
-static void * eth_ipcp_packet_writer(void * o)
-{
- int fd;
- struct ssm_pk_buff * spb;
- size_t len;
-#if defined(BUILD_ETH_DIX)
- uint16_t deid;
-#elif defined(BUILD_ETH_LLC)
- uint8_t dsap;
- uint8_t ssap;
-#endif
- uint8_t r_addr[MAC_SIZE];
-
- fqueue_t * fq;
-
- fq = fqueue_create();
- if (fq == NULL)
- return (void *) -1;
-
- (void) o;
-
- ipcp_lock_to_core();
-
- pthread_cleanup_push(cleanup_writer, fq);
-
- while (true) {
- fevent(eth_data.np1_flows, fq, NULL);
- while ((fd = fqueue_next(fq)) >= 0) {
- if (fqueue_type(fq) != FLOW_PKT)
- continue;
-
- if (np1_flow_read(fd, &spb, NP1_GET_POOL(fd))) {
- log_dbg("Bad read from fd %d.", fd);
- continue;
- }
-
- len = ssm_pk_buff_len(spb);
-
- if (ssm_pk_buff_push(spb, ETH_HEADER_TOT_SIZE)
- == NULL) {
- log_dbg("Failed to allocate header.");
- ipcp_spb_release(spb);
- FETCH_ADD_RELAXED(&eth_data.stat.n_buf_f, 1);
- continue;
- }
-
- pthread_rwlock_rdlock(&eth_data.flows_lock);
-#if defined(BUILD_ETH_DIX)
- deid = eth_data.fd_to_ef[fd].r_eid;
-#elif defined(BUILD_ETH_LLC)
- dsap = reverse_bits(eth_data.fd_to_ef[fd].r_sap);
- ssap = reverse_bits(eth_data.fd_to_ef[fd].sap);
-#endif
- memcpy(r_addr,
- eth_data.fd_to_ef[fd].r_addr,
- MAC_SIZE);
-
- pthread_rwlock_unlock(&eth_data.flows_lock);
-
- if (eth_ipcp_send_frame(r_addr,
-#if defined(BUILD_ETH_DIX)
- deid,
-#elif defined(BUILD_ETH_LLC)
- dsap, ssap,
-#endif
- ssm_pk_buff_head(spb),
- len)) {
- log_dbg("Failed to send frame.");
- FETCH_ADD_RELAXED(
- &eth_data.fd_to_ef[fd].stat.p_snd_f,
- 1);
- FETCH_ADD_RELAXED(&eth_data.stat.n_snd_f, 1);
- } else {
- FETCH_ADD_RELAXED(
- &eth_data.fd_to_ef[fd].stat.p_snd,
- 1);
- FETCH_ADD_RELAXED(
- &eth_data.fd_to_ef[fd].stat.b_snd,
- len);
- }
- ipcp_spb_release(spb);
- }
- }
-
- pthread_cleanup_pop(true);
-
- return (void *) 1;
-}
-
-#ifdef __linux__
-static int open_netlink_socket(void)
-{
- struct sockaddr_nl sa;
- int fd;
-
- memset(&sa, 0, sizeof(sa));
- sa.nl_family = AF_NETLINK;
- sa.nl_pid = getpid();
- sa.nl_groups = RTMGRP_LINK;
-
- fd = socket(AF_NETLINK, SOCK_RAW, NETLINK_ROUTE);
- if (fd < 0)
- return -1;
-
- if (bind(fd, (struct sockaddr *) &sa, sizeof(sa))) {
- close(fd);
- return -1;
- }
-
- return fd;
-}
-
-static void change_flows_state(bool up)
-{
- int i;
- uint32_t flags;
-
- pthread_rwlock_rdlock(&eth_data.flows_lock);
-
-#if defined(BUILD_ETH_DIX)
- for (i = 0; i < SYS_MAX_FLOWS; ++i)
- if (eth_data.fd_to_ef[i].r_eid != -1) {
- fccntl(i, FLOWGFLAGS, &flags);
- if (up)
- fccntl(i, FLOWSFLAGS, flags & ~FLOWFDOWN);
- else
- fccntl(i, FLOWSFLAGS, flags | FLOWFDOWN);
- }
-#elif defined(BUILD_ETH_LLC)
- for (i = 0; i < MAX_SAPS; i++)
- if (eth_data.ef_to_fd[i] != -1) {
- fccntl(eth_data.ef_to_fd[i], FLOWGFLAGS, &flags);
- if (up)
- fccntl(eth_data.ef_to_fd[i],
- FLOWSFLAGS, flags & ~FLOWFDOWN);
- else
- fccntl(eth_data.ef_to_fd[i],
- FLOWSFLAGS, flags | FLOWFDOWN);
- }
-#endif
-
- pthread_rwlock_unlock(&eth_data.flows_lock);
-}
-
-static void * eth_ipcp_if_monitor(void * o)
-{
- int fd;
- int status;
- char buf[4096];
- struct iovec iov = {buf, sizeof(buf)};
- struct sockaddr_nl snl;
- struct msghdr msg = {(void *) &snl, sizeof(snl),
- &iov, 1, NULL, 0, 0};
- struct nlmsghdr * h;
- struct ifinfomsg * ifi;
-
- (void ) o;
-
- fd = open_netlink_socket();
- if (fd < 0) {
- log_err("Failed to open socket.");
- return (void *) -1;
- }
-
- pthread_cleanup_push(__cleanup_close_ptr, &fd);
-
- while (true) {
- status = recvmsg(fd, &msg, 0);
- if (status < 0)
- continue;
-
- for (h = (struct nlmsghdr *) buf;
- NLMSG_OK(h, (unsigned int) status);
- h = NLMSG_NEXT(h, status)) {
-
- /* Finish reading */
- if (h->nlmsg_type == NLMSG_DONE)
- break;
-
- /* Message is some kind of error */
- if (h->nlmsg_type == NLMSG_ERROR)
- continue;
-
- /* Only interested in link up/down */
- if (h->nlmsg_type != RTM_NEWLINK)
- continue;
-
- ifi = NLMSG_DATA(h);
-
- /* Not our interface */
- if (ifi->ifi_index != eth_data.if_idx)
- continue;
-
- if (ifi->ifi_flags & IFF_UP) {
- change_flows_state(true);
- log_dbg("Interface up.");
- } else {
- change_flows_state(false);
- log_dbg("Interface down.");
- }
- }
- }
-
- pthread_cleanup_pop(true);
-
- return (void *) 0;
-}
-#endif
-
-#if defined (HAVE_BPF) && !defined(HAVE_NETMAP)
-static int open_bpf_device(void)
-{
- char dev[32];
- size_t i = 0;
-
- for (i = 0; i < BPF_DEV_MAX; i++) {
- int fd = -1;
-
- snprintf(dev, sizeof(dev), "/dev/bpf%zu", i);
-
- fd = open(dev, O_RDWR);
- if (fd > -1)
- return fd;
- }
-
- return -1;
-}
-#endif
-
-#if defined(__FreeBSD__) || defined(__APPLE__)
-static int ifr_hwaddr_from_ifaddrs(struct ifreq * ifr)
-{
- struct ifaddrs * ifaddr;
- struct ifaddrs * ifa;
- int idx;
-
- if (getifaddrs(&ifaddr) < 0) {
- log_err("Could not get interfaces.");
- goto fail_ifaddrs;
- }
-
- for (ifa = ifaddr, idx = 0; ifa != NULL; ifa = ifa->ifa_next, ++idx) {
- if (strcmp(ifa->ifa_name, ifr->ifr_name) == 0)
- break;
- }
-
- if (ifa == NULL) {
- log_err("Interface not found.");
- goto fail_ifa;
- }
-
- memcpy(&ifr->ifr_addr, ifa->ifa_addr, sizeof(*ifa->ifa_addr));
-
- log_dbg("Interface %s hwaddr " MAC_FMT ".", ifr->ifr_name,
- MAC_VAL(ifr->ifr_addr.sa_data));
-
- freeifaddrs(ifaddr);
-
- return 0;
- fail_ifa:
- freeifaddrs(ifaddr);
- fail_ifaddrs:
- return -1;
-
-}
-#elif defined(__linux__)
-static int ifr_hwaddr_from_socket(struct ifreq * ifr)
-{
- int skfd;
-
- skfd = socket(AF_UNIX, SOCK_STREAM, 0);
- if (skfd < 0) {
- log_err("Failed to open socket.");
- goto fail_socket;
- }
-
- if (ioctl(skfd, SIOCGIFHWADDR, ifr)) {
- log_err("Failed to get hwaddr.");
- goto fail_ifr;
- }
-
- log_dbg("Interface %s hwaddr " MAC_FMT ".", ifr->ifr_name,
- MAC_VAL(ifr->ifr_hwaddr.sa_data));
-
- close(skfd);
-
- return 0;
-
- fail_ifr:
- close(skfd);
- fail_socket:
- return -1;
-}
-#endif
-
-static int eth_ifr_hwaddr(struct ifreq * ifr)
-{
-#if defined(__FreeBSD__) || defined(__APPLE__)
- return ifr_hwaddr_from_ifaddrs(ifr);
-#elif defined(__linux__)
- return ifr_hwaddr_from_socket(ifr);
-#else
- return -1;
-#endif
-}
-
-static int eth_ifr_mtu(struct ifreq * ifr)
-{
- int skfd;
-
- skfd = socket(AF_UNIX, SOCK_STREAM, 0);
- if (skfd < 0) {
- log_err("Failed to open socket.");
- goto fail_socket;
- }
-
- if (ioctl(skfd, SIOCGIFMTU, ifr) < 0) {
- log_err("Failed to get MTU.");
- goto fail_mtu;
- }
- close(skfd);
-
- return 0;
-
- fail_mtu:
- close(skfd);
- fail_socket:
- return -1;
-}
-
-static int eth_set_mtu(struct ifreq * ifr)
-{
- if (eth_ifr_mtu(ifr) < 0) {
- log_err("Failed to get interface MTU.");
- return -1;
- }
-
- log_dbg("Device MTU is %d.", ifr->ifr_mtu);
-
- eth_data.mtu = MIN((int) ETH_MTU_MAX, ifr->ifr_mtu);
- if (memcmp(ifr->ifr_name, "lo", 2) == 0 &&
- eth_data.mtu > IPCP_ETH_LO_MTU) {
- log_dbg("Using loopback interface. MTU restricted to %d.",
- IPCP_ETH_LO_MTU);
- eth_data.mtu = IPCP_ETH_LO_MTU;
- }
-
- log_dbg("Layer MTU is %d.", eth_data.mtu);
-
- return 0;
-}
-#if defined(HAVE_NETMAP)
-static int eth_init_nmd(struct ifreq * ifr)
-{
- strcpy(ifn, "netmap:");
- strcat(ifn, ifr->ifr_name);
-
- eth_data.nmd = nm_open(ifn, NULL, 0, NULL);
- if (eth_data.nmd == NULL) {
- log_err("Failed to open netmap device.");
- goto fail_nmd;
- }
-
- memset(&eth_data.poll_in, 0, sizeof(eth_data.poll_in));
- memset(&eth_data.poll_out, 0, sizeof(eth_data.poll_out));
-
- eth_data.poll_in.fd = NETMAP_FD(eth_data.nmd);
- eth_data.poll_in.events = POLLIN;
- eth_data.poll_out.fd = NETMAP_FD(eth_data.nmd);
- eth_data.poll_out.events = POLLOUT;
-
- log_info("Using netmap device.");
-
- return 0;
- fail_nmd:
- return -1;
-}
-#elif defined (HAVE_BPF)
-static int eth_init_bpf(struct ifreq * ifr)
-{
- int enable = 1;
- int disable = 0;
- int blen;
-
- eth_data.bpf = open_bpf_device();
- if (eth_data.bpf < 0) {
- log_err("Failed to open bpf device.");
- goto fail_bpf;
- }
-
- ioctl(eth_data.bpf, BIOCGBLEN, &blen);
- if (BPF_BLEN < blen) {
- log_err("BPF buffer too small (is: %ld must be: %d).",
- BPF_BLEN, blen);
- goto fail_device;
- }
-
- if (ioctl(eth_data.bpf, BIOCSETIF, ifr) < 0) {
- log_err("Failed to set interface.");
- goto fail_device;
- }
-
- if (ioctl(eth_data.bpf, BIOCSHDRCMPLT, &enable) < 0) {
- log_err("Failed to set BIOCSHDRCMPLT.");
- goto fail_device;
- }
-
- if (ioctl(eth_data.bpf, BIOCSSEESENT, &disable) < 0) {
- log_err("Failed to set BIOCSSEESENT.");
- goto fail_device;
- }
-
- if (ioctl(eth_data.bpf, BIOCIMMEDIATE, &enable) < 0) {
- log_err("Failed to set BIOCIMMEDIATE.");
- goto fail_device;
- }
-
- log_info("Using Berkeley Packet Filter.");
-
- return 0;
-
- fail_device:
- close(eth_data.bpf);
- fail_bpf:
- return -1;
-}
-#elif defined(HAVE_RAW_SOCKETS)
-#define SOCKOPT()
-static int eth_init_raw_socket(struct ifreq * ifr)
-{
- int idx;
- int sndbuf;
- int rcvbuf;
-#if defined(IPCP_ETH_QDISC_BYPASS)
- int qdisc_bypass = 1;
-#endif /* ENABLE_QDISC_BYPASS */
-
- idx = if_nametoindex(ifr->ifr_name);
- if (idx == 0) {
- log_err("Failed to retrieve interface index.");
- return -1;
- }
-
- memset(&(eth_data.device), 0, sizeof(eth_data.device));
- eth_data.device.sll_ifindex = idx;
- eth_data.device.sll_family = AF_PACKET;
- memcpy(eth_data.device.sll_addr, ifr->ifr_hwaddr.sa_data, MAC_SIZE);
- eth_data.device.sll_halen = MAC_SIZE;
- eth_data.device.sll_protocol = htons(ETH_P_ALL);
-#if defined (BUILD_ETH_DIX)
- eth_data.s_fd = socket(AF_PACKET, SOCK_RAW, eth_data.ethertype);
-#elif defined (BUILD_ETH_LLC)
- eth_data.s_fd = socket(AF_PACKET, SOCK_RAW, htons(ETH_P_802_2));
-#endif
- if (eth_data.s_fd < 0) {
- log_err("Failed to create socket.");
- goto fail_socket;
- }
-
-#if defined(IPCP_ETH_QDISC_BYPASS)
- if (setsockopt(eth_data.s_fd, SOL_PACKET, PACKET_QDISC_BYPASS,
- &qdisc_bypass, sizeof(qdisc_bypass))) {
- log_info("Qdisc bypass not supported.");
- }
-#endif
-
- sndbuf = IPCP_ETH_SNDBUF;
- if (sndbuf > 0) {
- /* Never below one frame or sendto() cannot fit an SDU. */
- sndbuf = MAX(sndbuf, (int) (ETH_HEADER_TOT_SIZE + ETH_MTU));
- if (setsockopt(eth_data.s_fd, SOL_SOCKET, SO_SNDBUF,
- &sndbuf, sizeof(sndbuf)))
- log_info("Failed to set SO_SNDBUF to %d.", sndbuf);
- }
-
- rcvbuf = IPCP_ETH_RCVBUF;
- if (rcvbuf > 0) {
- /* SO_RCVBUFFORCE bypasses rmem_max; SO_RCVBUF is fallback. */
- if (setsockopt(eth_data.s_fd, SOL_SOCKET, SO_RCVBUFFORCE,
- &rcvbuf, sizeof(rcvbuf))
- && setsockopt(eth_data.s_fd, SOL_SOCKET, SO_RCVBUF,
- &rcvbuf, sizeof(rcvbuf)))
- log_info("Failed to set SO_RCVBUF to %d.", rcvbuf);
- }
-
- if (bind(eth_data.s_fd, (struct sockaddr *) &eth_data.device,
- sizeof(eth_data.device)) < 0) {
- log_err("Failed to bind socket to interface.");
- goto fail_device;
- }
-#ifdef __linux__
- eth_data.if_idx = idx;
-#endif
- log_info("Using raw socket device.");
-
- return 0;
- fail_device:
- close(eth_data.s_fd);
- fail_socket:
- return -1;
-}
-#endif
-
-static int eth_ipcp_bootstrap(struct ipcp_config * conf)
-{
- struct ifreq ifr;
- int i;
-#if defined(HAVE_NETMAP)
- char ifn[IFNAMSIZ];
-#endif /* HAVE_NETMAP */
-
- assert(conf);
- assert(conf->type == THIS_TYPE);
-
- memset(&ifr, 0, sizeof(ifr));
- strcpy(ifr.ifr_name, conf->eth.dev);
-
- if (strlen(conf->eth.dev) >= IFNAMSIZ) {
- log_err("Invalid device name: %s.", conf->eth.dev);
- return -1;
- }
-#ifdef BUILD_ETH_DIX
- if (conf->eth.ethertype < 0x0600 || conf->eth.ethertype == 0xFFFF) {
- log_err("Invalid Ethertype: %d.", conf->eth.ethertype);
- return -1;
- }
- eth_data.ethertype = htons(conf->eth.ethertype);
-#endif
- if (eth_set_mtu(&ifr) < 0) {
- log_err("Failed to set MTU.");
- return -1;
- }
-
- if (eth_ifr_hwaddr(&ifr) < 0) {
- log_err("Failed to get hardware addr.");
- return -1;
- }
-#if defined(HAVE_NETMAP) || defined(HAVE_BPF)
- memcpy(eth_data.hw_addr, LLADDR((struct sockaddr_dl *) &ifr.ifr_addr),
- MAC_SIZE);
-#endif
-#if defined(HAVE_NETMAP)
- if (eth_init_nmd(&ifr) < 0) {
- log_err("Failed to initialize netmap device.");
- return -1;
- }
-#elif defined(HAVE_BPF) /* !HAVE_NETMAP */
- if (eth_init_bpf(&ifr) < 0) {
- log_err("Failed to initialize BPF device.");
- return -1;
- }
-#elif defined(HAVE_RAW_SOCKETS)
- if (eth_init_raw_socket(&ifr) < 0) {
- log_err("Failed to initialize raw socket device.");
- return -1;
- }
-#endif /* HAVE_NETMAP */
-#ifdef IPCP_ETH_FLOW_STATS
- if (rib_reg(ETH_RIB_PATH, &eth_r_ops)) {
- log_err("Failed to register RIB.");
- goto fail_rib_reg;
- }
-#endif
-#if defined(__linux__)
- if (pthread_create(&eth_data.if_monitor, NULL,
- eth_ipcp_if_monitor, NULL)) {
- log_err("Failed to create monitor thread: %s.",
- strerror(errno));
- goto fail_monitor;
- }
-#endif
- if (pthread_create(&eth_data.mgmt_handler, NULL,
- eth_ipcp_mgmt_handler, NULL)) {
- log_err("Failed to create mgmt handler thread: %s.",
- strerror(errno));
- goto fail_mgmt_handler;
- }
-
- for (i = 0; i < IPCP_ETH_RD_THR; i++) {
- if (pthread_create(&eth_data.packet_reader[i], NULL,
- eth_ipcp_packet_reader, NULL)) {
- log_err("Failed to create packet reader thread: %s",
- strerror(errno));
- goto fail_packet_reader;
- }
- }
-
- for (i = 0; i < IPCP_ETH_WR_THR; i++) {
- if (pthread_create(&eth_data.packet_writer[i], NULL,
- eth_ipcp_packet_writer, NULL)) {
- log_err("Failed to create packet writer thread: %s",
- strerror(errno));
- goto fail_packet_writer;
- }
- }
-
-#if defined(BUILD_ETH_DIX)
- log_dbg("Bootstrapped IPCP over DIX Ethernet with pid %d "
- "and Ethertype 0x%X.", getpid(), conf->eth.ethertype);
-#elif defined(BUILD_ETH_LLC)
- log_dbg("Bootstrapped IPCP over Ethernet with LLC with pid %d.",
- getpid());
-#endif
- return 0;
-
- fail_packet_writer:
- while (i-- > 0) {
- pthread_cancel(eth_data.packet_writer[i]);
- pthread_join(eth_data.packet_writer[i], NULL);
- }
- i = IPCP_ETH_RD_THR;
- fail_packet_reader:
- while (i-- > 0) {
- pthread_cancel(eth_data.packet_reader[i]);
- pthread_join(eth_data.packet_reader[i], NULL);
- }
- pthread_cancel(eth_data.mgmt_handler);
- pthread_join(eth_data.mgmt_handler, NULL);
- fail_mgmt_handler:
-#if defined(__linux__)
- pthread_cancel(eth_data.if_monitor);
- pthread_join(eth_data.if_monitor, NULL);
-#endif
-#if defined(__linux__)
- fail_monitor:
-#endif
-#ifdef IPCP_ETH_FLOW_STATS
- rib_unreg(ETH_RIB_PATH);
- fail_rib_reg:
-#endif
-#if defined(HAVE_NETMAP)
- nm_close(eth_data.nmd);
-#elif defined(HAVE_BPF)
- close(eth_data.bpf);
-#elif defined(HAVE_RAW_SOCKETS)
- close(eth_data.s_fd);
-#endif
- return -1;
-}
-
-static int eth_ipcp_reg(const uint8_t * hash)
-{
- if (shim_data_reg_add_entry(eth_data.shim_data, hash)) {
- log_err("Failed to add " HASH_FMT32 " to local registry.",
- HASH_VAL32(hash));
- return -1;
- }
-
- return 0;
-}
-
-static int eth_ipcp_unreg(const uint8_t * hash)
-{
- shim_data_reg_del_entry(eth_data.shim_data, hash);
-
- return 0;
-}
-
-static int eth_ipcp_query(const uint8_t * hash)
-{
- uint8_t r_addr[MAC_SIZE];
- struct timespec timeout;
- struct dir_query * query;
- int ret;
- int attempt;
- uint8_t * buf;
- struct mgmt_msg * msg;
- size_t len;
- long per_ms;
-
- if (shim_data_dir_has(eth_data.shim_data, hash))
- return 0;
-
- len = sizeof(*msg) + ipcp_dir_hash_len();
-
- buf = malloc(len + ETH_HEADER_TOT_SIZE);
- if (buf == NULL)
- return -1;
-
- memset(buf, 0, len + ETH_HEADER_TOT_SIZE);
-
- msg = (struct mgmt_msg *) (buf + ETH_HEADER_TOT_SIZE);
- msg->code = NAME_QUERY_REQ;
-
- memcpy(msg + 1, hash, ipcp_dir_hash_len());
-
- memset(r_addr, 0xff, MAC_SIZE);
-
- per_ms = NAME_QUERY_TIMEO / (NAME_QUERY_RETRIES + 1);
-
- ret = -1;
- for (attempt = 0; attempt <= NAME_QUERY_RETRIES; ++attempt) {
- query = shim_data_dir_query_create(eth_data.shim_data, hash);
- if (query == NULL) {
- ret = -1;
- break;
- }
-
- if (eth_ipcp_send_frame(r_addr,
-#if defined(BUILD_ETH_DIX)
- MGMT_EID,
-#elif defined(BUILD_ETH_LLC)
- reverse_bits(MGMT_SAP),
- reverse_bits(MGMT_SAP),
-#endif
- buf, len)) {
- log_err("Failed to send management frame.");
- shim_data_dir_query_destroy(eth_data.shim_data,
- query);
- ret = -1;
- break;
- }
-
- FETCH_ADD_RELAXED(&eth_data.stat.n_mgmt_snd, 1);
-
- timeout.tv_sec = per_ms / 1000;
- timeout.tv_nsec = (per_ms % 1000) * 1000000L;
-
- ret = shim_data_dir_query_wait(query, &timeout);
-
- shim_data_dir_query_destroy(eth_data.shim_data, query);
-
- if (ret != -ETIMEDOUT)
- break;
- }
-
- free(buf);
-
- return ret;
-}
-
-static int eth_ipcp_flow_alloc(int fd,
- const uint8_t * hash,
- qosspec_t qs,
- const buffer_t * data)
-{
-#ifdef BUILD_ETH_LLC
- uint8_t ssap = 0;
-#endif
- uint8_t r_addr[MAC_SIZE];
- struct addr addr;
-
- assert(hash);
-
- if (!shim_data_dir_has(eth_data.shim_data, hash)) {
- log_err("Destination "HASH_FMT32 "unreachable.",
- HASH_VAL32(hash));
- return -1;
- }
-
- addr = shim_data_dir_get_addr(eth_data.shim_data, hash);
- memcpy(r_addr, &addr.mac, MAC_SIZE);
-
-#ifdef BUILD_ETH_LLC
- pthread_rwlock_wrlock(&eth_data.flows_lock);
- ssap = bmp_allocate(eth_data.saps);
- if (!bmp_is_id_valid(eth_data.saps, ssap)) {
- pthread_rwlock_unlock(&eth_data.flows_lock);
- log_err("Failed to allocate SSAP.");
- return -1;
- }
-
- eth_data.fd_to_ef[fd].sap = ssap;
- eth_data.ef_to_fd[ssap] = fd;
- pthread_rwlock_unlock(&eth_data.flows_lock);
-#endif
-
- if (eth_ipcp_alloc(r_addr,
-#if defined(BUILD_ETH_DIX)
- fd,
-#elif defined(BUILD_ETH_LLC)
- ssap,
-#endif
- hash,
- qs,
- data) < 0) {
-#ifdef BUILD_ETH_LLC
- pthread_rwlock_wrlock(&eth_data.flows_lock);
- bmp_release(eth_data.saps, eth_data.fd_to_ef[fd].sap);
- eth_data.fd_to_ef[fd].sap = -1;
- eth_data.ef_to_fd[ssap] = -1;
- pthread_rwlock_unlock(&eth_data.flows_lock);
- log_err("Failed to allocate with peer.");
-#endif
- return -1;
- }
-
- fset_add(eth_data.np1_flows, fd);
-#ifdef IPCP_ETH_FLOW_STATS
- pthread_rwlock_wrlock(&eth_data.flows_lock);
- memset(&eth_data.fd_to_ef[fd].stat, 0,
- sizeof(eth_data.fd_to_ef[fd].stat));
- eth_data.fd_to_ef[fd].stat.stamp = time(NULL);
- FETCH_ADD_RELAXED(&eth_data.stat.n_flows, 1);
- pthread_rwlock_unlock(&eth_data.flows_lock);
-#endif
-#if defined(BUILD_ETH_LLC)
- log_dbg("Assigned SAP %d for fd %d.", ssap, fd);
-#endif
- return 0;
-}
-
-static int eth_ipcp_flow_alloc_resp(int fd,
- int response,
- const buffer_t * data)
-{
-#if defined(BUILD_ETH_DIX)
- uint16_t r_eid;
-#elif defined(BUILD_ETH_LLC)
- uint8_t ssap;
- uint8_t r_sap;
-#endif
- uint8_t r_addr[MAC_SIZE];
-
- if (ipcp_wait_flow_resp(fd) < 0) {
- log_err("Failed to wait for flow response.");
- return -1;
- }
-
- pthread_rwlock_wrlock(&eth_data.flows_lock);
-#if defined(BUILD_ETH_DIX)
- r_eid = eth_data.fd_to_ef[fd].r_eid;
-#elif defined(BUILD_ETH_LLC)
- ssap = bmp_allocate(eth_data.saps);
- if (!bmp_is_id_valid(eth_data.saps, ssap)) {
- pthread_rwlock_unlock(&eth_data.flows_lock);
- log_err("Failed to allocate SSAP.");
- return -1;
- }
-
- eth_data.fd_to_ef[fd].sap = ssap;
- r_sap = eth_data.fd_to_ef[fd].r_sap;
- eth_data.ef_to_fd[ssap] = fd;
-#endif
- memcpy(r_addr, eth_data.fd_to_ef[fd].r_addr, MAC_SIZE);
-
- pthread_rwlock_unlock(&eth_data.flows_lock);
-
- if (eth_ipcp_alloc_resp(r_addr,
-#if defined(BUILD_ETH_DIX)
- fd, r_eid,
-#elif defined(BUILD_ETH_LLC)
- ssap, r_sap,
-#endif
- response,
- data) < 0) {
-#ifdef BUILD_ETH_LLC
- pthread_rwlock_wrlock(&eth_data.flows_lock);
- bmp_release(eth_data.saps, eth_data.fd_to_ef[fd].sap);
- pthread_rwlock_unlock(&eth_data.flows_lock);
-#endif
- log_err("Failed to respond to peer.");
- return -1;
- }
-
- fset_add(eth_data.np1_flows, fd);
-#ifdef IPCP_ETH_FLOW_STATS
- pthread_rwlock_wrlock(&eth_data.flows_lock);
- memset(&eth_data.fd_to_ef[fd].stat, 0,
- sizeof(eth_data.fd_to_ef[fd].stat));
- eth_data.fd_to_ef[fd].stat.stamp = time(NULL);
- FETCH_ADD_RELAXED(&eth_data.stat.n_flows, 1);
- pthread_rwlock_unlock(&eth_data.flows_lock);
-#endif
-#if defined(BUILD_ETH_LLC)
- log_dbg("Assigned SAP %d for fd %d.", ssap, fd);
-#endif
- return 0;
-}
-
-static int eth_ipcp_flow_dealloc(int fd)
-{
-#ifdef BUILD_ETH_LLC
- uint8_t sap;
-#endif
- ipcp_flow_fini(fd);
-
- fset_del(eth_data.np1_flows, fd);
-
- pthread_rwlock_wrlock(&eth_data.flows_lock);
-
-#if defined(BUILD_ETH_DIX)
- eth_data.fd_to_ef[fd].r_eid = -1;
-#elif defined BUILD_ETH_LLC
- sap = eth_data.fd_to_ef[fd].sap;
- bmp_release(eth_data.saps, sap);
- eth_data.fd_to_ef[fd].sap = -1;
- eth_data.fd_to_ef[fd].r_sap = -1;
- eth_data.ef_to_fd[sap] = -1;
-#endif
- memset(&eth_data.fd_to_ef[fd].r_addr, 0, MAC_SIZE);
-
-#ifdef IPCP_ETH_FLOW_STATS
- memset(&eth_data.fd_to_ef[fd].stat, 0,
- sizeof(eth_data.fd_to_ef[fd].stat));
- FETCH_SUB_RELAXED(&eth_data.stat.n_flows, 1);
-#endif
-
- pthread_rwlock_unlock(&eth_data.flows_lock);
-
- ipcp_flow_dealloc(fd);
-
- return 0;
-}
-
-static struct ipcp_ops eth_ops = {
- .ipcp_bootstrap = eth_ipcp_bootstrap,
- .ipcp_enroll = NULL,
- .ipcp_connect = NULL,
- .ipcp_disconnect = NULL,
- .ipcp_reg = eth_ipcp_reg,
- .ipcp_unreg = eth_ipcp_unreg,
- .ipcp_query = eth_ipcp_query,
- .ipcp_flow_alloc = eth_ipcp_flow_alloc,
- .ipcp_flow_join = NULL,
- .ipcp_flow_alloc_resp = eth_ipcp_flow_alloc_resp,
- .ipcp_flow_dealloc = eth_ipcp_flow_dealloc,
- .ipcp_flow_update = eth_ipcp_flow_update
-};
-
-int main(int argc,
- char * argv[])
-{
- int i;
-
- if (eth_data_init() < 0) {
-#if defined(BUILD_ETH_DIX)
- log_err("Failed to init eth-llc data.");
-#elif defined(BUILD_ETH_LLC)
- log_err("Failed to init eth-dix data.");
-#endif
- goto fail_data_init;
- }
-
- if (ipcp_init(argc, argv, &eth_ops, THIS_TYPE) < 0) {
- log_err("Failed to initialize IPCP.");
- goto fail_init;
- }
-
- if (ipcp_start() < 0) {
- log_err("Failed to start IPCP.");
- goto fail_start;
- }
-
- ipcp_sigwait();
-
- if (ipcp_get_state() == IPCP_SHUTDOWN) {
- for (i = 0; i < IPCP_ETH_WR_THR; ++i)
- pthread_cancel(eth_data.packet_writer[i]);
- for (i = 0; i < IPCP_ETH_RD_THR; ++i)
- pthread_cancel(eth_data.packet_reader[i]);
-
- pthread_cancel(eth_data.mgmt_handler);
-#ifdef __linux__
- pthread_cancel(eth_data.if_monitor);
-#endif
- for (i = 0; i < IPCP_ETH_WR_THR; ++i)
- pthread_join(eth_data.packet_writer[i], NULL);
- for (i = 0; i < IPCP_ETH_RD_THR; ++i)
- pthread_join(eth_data.packet_reader[i], NULL);
-
- pthread_join(eth_data.mgmt_handler, NULL);
-#ifdef __linux__
- pthread_join(eth_data.if_monitor, NULL);
-#endif
-#ifdef IPCP_ETH_FLOW_STATS
- rib_unreg(ETH_RIB_PATH);
-#endif
- }
-
- ipcp_stop();
-
- ipcp_fini();
-
- eth_data_fini();
-
- exit(EXIT_SUCCESS);
-
- fail_start:
- ipcp_fini();
- fail_init:
- eth_data_fini();
- fail_data_init:
- exit(EXIT_FAILURE);
-}
diff --git a/src/ipcpd/ipcp.c b/src/ipcpd/ipcp.c
index dcee4b9c..b25b7d04 100644
--- a/src/ipcpd/ipcp.c
+++ b/src/ipcpd/ipcp.c
@@ -36,6 +36,7 @@
#define OUROBOROS_PREFIX "ipcpd/ipcp"
#define IPCP_INFO "info"
#define ALLOC_TIMEOUT 50 /* ms */
+#define HAS_POA (ipcpd.type != IPCP_LOCAL)
#include <ouroboros/bitmap.h>
#include <ouroboros/dev.h>
@@ -54,6 +55,7 @@
#include "ipcp.h"
#include "np1.h"
+#include <arpa/inet.h>
#include <signal.h>
#include <string.h>
#include <sys/socket.h>
@@ -69,14 +71,14 @@
#define CLOCK_REALTIME_COARSE CLOCK_REALTIME
#endif
+/* Fits "eth <dev> 0x<type>", the longest PoA rendering. */
+/* Matches src/tools/irm/irm_utils.h; keep in sync. */
+#define POA_STRLEN (DEV_NAME_SIZE + 11)
+
static char * ipcp_type_str[] = {
"local",
"unicast",
- "broadcast",
- "eth-llc",
- "eth-dix",
- "udp4",
- "udp6"
+ "broadcast"
};
static char * dir_hash_str[] = {
@@ -237,14 +239,6 @@ static int ipcp_rib_read(const char * path,
strcpy(buf, "unicast\n");
else if (ipcpd.type == IPCP_BROADCAST)
strcpy(buf, "broadcast\n");
- else if (ipcpd.type == IPCP_ETH_LLC)
- strcpy(buf, "eth-llc\n");
- else if (ipcpd.type == IPCP_ETH_DIX)
- strcpy(buf, "eth-dix\n");
- else if (ipcpd.type == IPCP_UDP4)
- strcpy(buf, "udp4\n");
- else if (ipcpd.type == IPCP_UDP6)
- strcpy(buf, "udp6\n");
else
strcpy(buf, "bug\n");
}
@@ -316,8 +310,7 @@ static void * acceptloop(void * o)
(void) o;
- while (ipcp_get_state() != IPCP_SHUTDOWN &&
- ipcp_get_state() != IPCP_INIT) {
+ while (ipcp_get_state() != IPCP_SHUTDOWN) {
struct cmd * cmd;
csockfd = accept(ipcpd.sockfd, 0, 0);
@@ -471,11 +464,6 @@ static void do_bootstrap(ipcp_config_msg_t * conf_msg,
conf = ipcp_config_msg_to_s(conf_msg);
switch(conf.type) { /* FIXED algorithms */
- case IPCP_UDP4:
- /* FALLTHRU */
- case IPCP_UDP6:
- conf.layer_info.dir_hash_algo = (enum pol_dir_hash) HASH_MD5;
- break;
case IPCP_BROADCAST:
conf.layer_info.dir_hash_algo = DIR_HASH_SHA3_256;
break;
@@ -493,6 +481,8 @@ static void do_bootstrap(ipcp_config_msg_t * conf_msg,
strcpy(ipcpd.layer_name, info->name);
ipcpd.dir_hash_algo = (enum hash_algo) info->dir_hash_algo;
+ if (poa_set_layer(info->name) < 0)
+ log_warn("Failed to set layer name for PoA queries.");
ret_msg->layer_info = layer_info_s_to_msg(info);
ipcp_set_state(IPCP_OPERATIONAL);
@@ -503,10 +493,13 @@ static void do_bootstrap(ipcp_config_msg_t * conf_msg,
ipcp_dir_hash_len());
}
-static void do_enroll(const char * dst,
- ipcp_msg_t * ret_msg)
+static void do_enroll(const char * dst,
+ const poa_addr_msg_t * peer,
+ ipcp_msg_t * ret_msg)
{
struct layer_info info;
+ struct poa_addr addr;
+ struct poa_addr * pa = NULL;
log_info("Enrolling with %s...", dst);
@@ -524,7 +517,16 @@ static void do_enroll(const char * dst,
return;
}
- ret_msg->result = ipcpd.ops->ipcp_enroll(dst, &info);
+ if (peer != NULL) {
+ addr = poa_addr_msg_to_s(peer);
+ if (addr.type == POA_INVALID || addr.type == POA_UDP) {
+ ret_msg->result = -EINVAL;
+ return;
+ }
+ pa = &addr;
+ }
+
+ ret_msg->result = ipcpd.ops->ipcp_enroll(dst, pa, &info);
if (ret_msg->result < 0) {
log_err("Failed to bootstrap IPCP.");
return;
@@ -532,6 +534,8 @@ static void do_enroll(const char * dst,
strcpy(ipcpd.layer_name, info.name);
ipcpd.dir_hash_algo = (enum hash_algo) info.dir_hash_algo;
+ if (poa_set_layer(info.name) < 0)
+ log_warn("Failed to set layer name for PoA queries.");
ret_msg->layer_info = layer_info_s_to_msg(&info);
ipcp_set_state(IPCP_OPERATIONAL);
@@ -542,11 +546,136 @@ static void do_enroll(const char * dst,
ipcp_dir_hash_len());
}
-static void do_connect(const char * dst,
- const char * comp,
- qosspec_t qs,
- ipcp_msg_t * ret_msg)
+/* Bounded so one oversized reply cannot be built; 64 is generous. */
+#define POA_LIST_MAX 64
+
+static void do_list_poas(ipcp_msg_t * ret_msg)
+{
+ struct poa_spec eps[POA_LIST_MAX];
+ ssize_t n;
+ ssize_t i;
+
+ if (ipcpd.type != IPCP_UNICAST && ipcpd.type != IPCP_BROADCAST) {
+ ret_msg->result = -ENOTSUP;
+ return;
+ }
+
+ n = poa_list(eps, POA_LIST_MAX);
+ if (n < 0) {
+ ret_msg->result = (int) n;
+ return;
+ }
+
+ if (n > POA_LIST_MAX) {
+ log_warn("Listing %d of %zd PoAs.", POA_LIST_MAX, n);
+
+ n = POA_LIST_MAX;
+ }
+
+ if (n == 0) {
+ ret_msg->result = 0;
+ return;
+ }
+
+ ret_msg->poas = malloc(n * sizeof(*ret_msg->poas));
+ if (ret_msg->poas == NULL) {
+ ret_msg->result = -ENOMEM;
+ return;
+ }
+
+ for (i = 0; i < n; i++) {
+ ret_msg->poas[i] = poa_spec_s_to_msg(&eps[i]);
+ if (ret_msg->poas[i] == NULL) {
+ ret_msg->result = -ENOMEM;
+ return;
+ }
+ ret_msg->n_poas++;
+ }
+
+ ret_msg->result = 0;
+}
+
+/* Names a PoA the way "irm ipcp poa list" prints it. */
+/* Matches src/tools/irm/irm_utils.c; keep in sync. */
+static void poa_spec_str(const struct poa_spec * poa,
+ char * buf,
+ size_t len)
+{
+ char addr[INET6_ADDRSTRLEN];
+
+ switch (poa->type) {
+ case POA_UDP4:
+ if (inet_ntop(AF_INET, &poa->udp4.ip_addr,
+ addr, sizeof(addr)) == NULL)
+ break;
+
+ snprintf(buf, len, "udp4 %s:%u", addr, poa->udp4.port);
+ return;
+ case POA_UDP6:
+ if (inet_ntop(AF_INET6, &poa->udp6.ip_addr,
+ addr, sizeof(addr)) == NULL)
+ break;
+
+ snprintf(buf, len, "udp6 [%s]:%u", addr, poa->udp6.port);
+ return;
+ case POA_ETH:
+ snprintf(buf, len, "eth %s 0x%04X", poa->eth.dev,
+ poa->eth.ethertype);
+ return;
+ default:
+ break;
+ }
+
+ snprintf(buf, len, "(unknown)");
+}
+
+static void do_attach(poa_spec_msg_t * msg,
+ ipcp_msg_t * ret_msg)
{
+ struct poa_spec poa;
+ char str[POA_STRLEN + 1];
+
+ poa = poa_spec_msg_to_s(msg);
+
+ poa_spec_str(&poa, str, sizeof(str));
+
+ ret_msg->result = poa_attach(&poa);
+ if (ret_msg->result < 0) {
+ log_err("Failed to attach %s.", str);
+ return;
+ }
+
+ log_info("Attached %s.", str);
+}
+
+static void do_detach(poa_spec_msg_t * msg,
+ ipcp_msg_t * ret_msg)
+{
+ struct poa_spec poa;
+ char str[POA_STRLEN + 1];
+
+ poa = poa_spec_msg_to_s(msg);
+
+ poa_spec_str(&poa, str, sizeof(str));
+
+ ret_msg->result = poa_detach(&poa);
+ if (ret_msg->result < 0) {
+ log_err("Failed to detach %s.", str);
+ return;
+ }
+
+ log_info("Detached %s.", str);
+}
+
+static void do_connect(const char * dst,
+ const char * comp,
+ qosspec_t qs,
+ const poa_addr_msg_t * peer,
+ ipcp_msg_t * ret_msg)
+{
+ struct poa_addr addr;
+ struct poa_addr * pa = NULL;
+
log_info("Connecting %s to %s...", comp, dst);
if (ipcpd.ops->ipcp_connect == NULL) {
@@ -555,7 +684,16 @@ static void do_connect(const char * dst,
return;
}
- ret_msg->result = ipcpd.ops->ipcp_connect(dst, comp, qs);
+ if (peer != NULL) {
+ addr = poa_addr_msg_to_s(peer);
+ if (addr.type == POA_INVALID || addr.type == POA_UDP) {
+ ret_msg->result = -EINVAL;
+ return;
+ }
+ pa = &addr;
+ }
+
+ ret_msg->result = ipcpd.ops->ipcp_connect(dst, comp, qs, pa);
log_info("Finished connecting.");
}
@@ -734,9 +872,14 @@ static void do_flow_join(pid_t pid,
log_info("Finished joining layer " HASH_FMT32 ".", HASH_VAL32(dst));
}
+/*
+ * The IRMd says whether the flow is on a PoA, as it may not be known
+ * here yet. PoA flows answer without the IPCP type's flow machinery.
+ */
static void do_flow_alloc_resp(int resp,
int flow_id,
uid_t uid,
+ bool is_poa,
const buffer_t * data,
ipcp_msg_t * ret_msg)
{
@@ -745,13 +888,6 @@ static void do_flow_alloc_resp(int resp,
log_info("Responding %d to alloc on flow_id %d.", resp, flow_id);
- if (ipcpd.ops->ipcp_flow_alloc_resp == NULL) {
- log_err("Failed to respond on flow %d: operation unsupported.",
- flow_id);
- ret_msg->result = -ENOTSUP;
- return;
- }
-
if (ipcp_get_state() != IPCP_OPERATIONAL) {
log_err("Failed to respond to flow %d:"
"IPCP in state <%s>, need <%s>.",
@@ -762,6 +898,20 @@ static void do_flow_alloc_resp(int resp,
return;
}
+ if (is_poa) {
+ ret_msg->result = poa_flow_alloc_resp(flow_id, resp, data);
+ log_info("Finished responding %d on PoA flow %d.",
+ ret_msg->result, flow_id);
+ return;
+ }
+
+ if (ipcpd.ops->ipcp_flow_alloc_resp == NULL) {
+ log_err("Failed to respond on flow %d: operation unsupported.",
+ flow_id);
+ ret_msg->result = -ENOTSUP;
+ return;
+ }
+
fd = np1_flow_resp(flow_id, resp);
if (fd < 0) {
log_warn("Flow_id %d is not known.", flow_id);
@@ -788,18 +938,13 @@ static void do_flow_alloc_resp(int resp,
static void do_flow_dealloc(int flow_id,
int timeo_sec,
+ bool is_poa,
ipcp_msg_t * ret_msg)
{
int fd;
log_info("Deallocating flow %d.", flow_id);
- if (ipcpd.ops->ipcp_flow_dealloc == NULL) {
- log_err("Failed to dealloc: operation unsupported.");
- ret_msg->result = -ENOTSUP;
- return;
- }
-
if (ipcp_get_state() != IPCP_OPERATIONAL) {
log_err("Failed to enroll: IPCP in state <%s>, need <%s>.",
ipcp_state_str[ipcp_get_state()],
@@ -808,6 +953,20 @@ static void do_flow_dealloc(int flow_id,
return;
}
+ if (is_poa) {
+ ret_msg->result = poa_flow_dealloc(flow_id);
+
+ log_info("Finished deallocating PoA flow %d.", flow_id);
+ return;
+ }
+
+ if (ipcpd.ops->ipcp_flow_dealloc == NULL) {
+ log_err("Failed to dealloc: operation unsupported.");
+
+ ret_msg->result = -ENOTSUP;
+ return;
+ }
+
fd = np1_flow_dealloc(flow_id, timeo_sec);
if (fd < 0) {
log_warn("Could not deallocate flow_id %d.", flow_id);
@@ -822,18 +981,25 @@ static void do_flow_dealloc(int flow_id,
static void do_flow_update(int flow_id,
const buffer_t * data,
+ bool is_poa,
ipcp_msg_t * ret_msg)
{
int fd;
- if (ipcpd.ops->ipcp_flow_update == NULL) {
- log_err("Failed to update flow: operation unsupported.");
- ret_msg->result = -ENOTSUP;
+ if (ipcp_get_state() != IPCP_OPERATIONAL) {
+ ret_msg->result = -EIPCPSTATE;
return;
}
- if (ipcp_get_state() != IPCP_OPERATIONAL) {
- ret_msg->result = -EIPCPSTATE;
+ if (is_poa) {
+ ret_msg->result = poa_flow_update(flow_id, data);
+ return;
+ }
+
+ if (ipcpd.ops->ipcp_flow_update == NULL) {
+ log_err("Failed to update flow: operation unsupported.");
+
+ ret_msg->result = -ENOTSUP;
return;
}
@@ -881,6 +1047,7 @@ static void * mainloop(void * o)
free(cmd);
if (msg == NULL) {
+ log_err("Failed to unpack command message.");
close(sfd);
continue;
}
@@ -897,11 +1064,23 @@ static void * mainloop(void * o)
do_bootstrap(msg->conf, &ret_msg);
break;
case IPCP_MSG_CODE__IPCP_ENROLL:
- do_enroll(msg->dst, &ret_msg);
+ do_enroll(msg->dst, msg->peer, &ret_msg);
+ break;
+ case IPCP_MSG_CODE__IPCP_ATTACH:
+ assert(HAS_POA);
+ do_attach(msg->poa, &ret_msg);
+ break;
+ case IPCP_MSG_CODE__IPCP_DETACH:
+ assert(HAS_POA);
+ do_detach(msg->poa, &ret_msg);
+ break;
+ case IPCP_MSG_CODE__IPCP_LIST_POAS:
+ do_list_poas(&ret_msg);
break;
case IPCP_MSG_CODE__IPCP_CONNECT:
qs = qos_spec_msg_to_s(msg->qosspec);
- do_connect(msg->dst, msg->comp, qs, &ret_msg);
+ do_connect(msg->dst, msg->comp, qs, msg->peer,
+ &ret_msg);
break;
case IPCP_MSG_CODE__IPCP_DISCONNECT:
do_disconnect(msg->dst, msg->comp, &ret_msg);
@@ -940,17 +1119,20 @@ static void * mainloop(void * o)
data.len = msg->pk.len;
data.data = msg->pk.data;
do_flow_alloc_resp(msg->response, msg->flow_id,
- msg->uid, &data, &ret_msg);
+ msg->uid, msg->is_poa,
+ &data, &ret_msg);
break;
case IPCP_MSG_CODE__IPCP_FLOW_DEALLOC:
- do_flow_dealloc(msg->flow_id, msg->timeo_sec, &ret_msg);
+ do_flow_dealloc(msg->flow_id, msg->timeo_sec,
+ msg->is_poa, &ret_msg);
break;
case IPCP_MSG_CODE__IPCP_FLOW_UPDATE:
assert(msg->pk.len > 0 ? msg->pk.data != NULL
: msg->pk.data == NULL);
data.len = msg->pk.len;
data.data = msg->pk.data;
- do_flow_update(msg->flow_id, &data, &ret_msg);
+ do_flow_update(msg->flow_id, &data, msg->is_poa,
+ &ret_msg);
break;
default:
ret_msg.result = -1;
@@ -988,7 +1170,7 @@ static void * mainloop(void * o)
if (write(sfd, buffer.data, buffer.len) == -1)
log_warn("Failed to send reply message");
- pthread_cleanup_pop(true); /* close sfd */
+ pthread_cleanup_pop(true); /* close sfd */
pthread_cleanup_pop(true); /* free buffer.data */
tpm_end_work(ipcpd.tpm);
@@ -1101,6 +1283,11 @@ int ipcp_init(int argc,
goto fail_rib_reg;
}
+ if (poa_init(ipcpd.name) < 0) {
+ log_err("Failed to initialize PoAs.");
+ goto fail_poa_init;
+ }
+
list_head_init(&ipcpd.cmds);
ipcpd.tpm = tpm_create(IPCP_MIN_THREADS, IPCP_ADD_THREADS,
@@ -1124,6 +1311,8 @@ int ipcp_init(int argc,
return 0;
fail_tpm_create:
+ poa_fini();
+ fail_poa_init:
rib_unreg(IPCP_INFO);
fail_rib_reg:
rib_fini();
@@ -1149,6 +1338,7 @@ int ipcp_init(int argc,
return -1;
}
+/* Enrolment runs over a PoA, so poa_start() precedes any RPC. */
int ipcp_start(void)
{
sigset_t sigset;
@@ -1169,6 +1359,11 @@ int ipcp_start(void)
ipcp_set_state(IPCP_BOOT);
+ if (poa_start() < 0) {
+ log_err("Failed to start PoAs.");
+ goto fail_poa_start;
+ }
+
if (tpm_start(ipcpd.tpm)) {
log_err("Failed to start threadpool manager.");
goto fail_tpm_start;
@@ -1192,6 +1387,8 @@ int ipcp_start(void)
fail_acceptor:
tpm_stop(ipcpd.tpm);
fail_tpm_start:
+ poa_stop();
+ fail_poa_start:
tpm_destroy(ipcpd.tpm);
ipcp_set_state(IPCP_INIT);
ipcp_create_r(&info);
@@ -1263,6 +1460,8 @@ void ipcp_stop(void)
tpm_stop(ipcpd.tpm);
+ poa_stop();
+
ipcp_set_state(IPCP_INIT);
}
@@ -1271,6 +1470,8 @@ void ipcp_fini(void)
tpm_destroy(ipcpd.tpm);
+ poa_fini();
+
rib_unreg(IPCP_INFO);
rib_fini();
diff --git a/src/ipcpd/ipcp.h b/src/ipcpd/ipcp.h
index 210157ec..e0aab291 100644
--- a/src/ipcpd/ipcp.h
+++ b/src/ipcpd/ipcp.h
@@ -36,15 +36,18 @@
#define ipcp_dir_hash_strlen() (ipcp_dir_hash_len() * 2)
+/* Attach or release one flow PoA on a unicast or broadcast IPCP. */
struct ipcp_ops {
int (* ipcp_bootstrap)(struct ipcp_config * conf);
- int (* ipcp_enroll)(const char * dst,
- struct layer_info * info);
+ int (* ipcp_enroll)(const char * dst,
+ const struct poa_addr * addr,
+ struct layer_info * info);
- int (* ipcp_connect)(const char * dst,
- const char * component,
- qosspec_t qs);
+ int (* ipcp_connect)(const char * dst,
+ const char * component,
+ qosspec_t qs,
+ const struct poa_addr * addr);
int (* ipcp_disconnect)(const char * dst,
const char * component);
diff --git a/src/ipcpd/local/CMakeLists.txt b/src/ipcpd/local/CMakeLists.txt
index 91f300a3..af433d01 100644
--- a/src/ipcpd/local/CMakeLists.txt
+++ b/src/ipcpd/local/CMakeLists.txt
@@ -2,6 +2,7 @@
add_executable(${IPCP_LOCAL_TARGET}
main.c
+ reg.c
${IPCP_SOURCES}
)
diff --git a/src/ipcpd/local/main.c b/src/ipcpd/local/main.c
index c0aeb51e..69eac8a6 100644
--- a/src/ipcpd/local/main.c
+++ b/src/ipcpd/local/main.c
@@ -42,7 +42,7 @@
#include "ipcp.h"
#include "np1.h"
-#include "shim-data.h"
+#include "reg.h"
#include <string.h>
#include <stdlib.h>
@@ -53,14 +53,14 @@
#define THIS_TYPE IPCP_LOCAL
struct {
- struct shim_data * shim_data;
+ struct reg * reg;
- int in_out[SYS_MAX_FLOWS];
- fset_t * flows;
- fqueue_t * fq;
+ int in_out[SYS_MAX_FLOWS];
+ fset_t * flows;
+ fqueue_t * fq;
- pthread_rwlock_t lock;
- pthread_t packet_loop;
+ pthread_rwlock_t lock;
+ pthread_t packet_loop;
} local_data;
static int local_data_init(void)
@@ -77,9 +77,9 @@ static int local_data_init(void)
if (local_data.fq == NULL)
goto fail_fqueue;
- local_data.shim_data = shim_data_create();
- if (local_data.shim_data == NULL)
- goto fail_shim_data;
+ local_data.reg = reg_create();
+ if (local_data.reg == NULL)
+ goto fail_reg;
if (pthread_rwlock_init(&local_data.lock, NULL) < 0)
goto fail_rwlock_init;
@@ -87,8 +87,8 @@ static int local_data_init(void)
return 0;
fail_rwlock_init:
- shim_data_destroy(local_data.shim_data);
- fail_shim_data:
+ reg_destroy(local_data.reg);
+ fail_reg:
fqueue_destroy(local_data.fq);
fail_fqueue:
fset_destroy(local_data.flows);
@@ -98,7 +98,7 @@ static int local_data_init(void)
static void local_data_fini(void){
pthread_rwlock_destroy(&local_data.lock);
- shim_data_destroy(local_data.shim_data);
+ reg_destroy(local_data.reg);
fqueue_destroy(local_data.fq);
fset_destroy(local_data.flows);
}
@@ -166,7 +166,7 @@ static int local_ipcp_bootstrap(struct ipcp_config * conf)
static int local_ipcp_reg(const uint8_t * hash)
{
- if (shim_data_reg_add_entry(local_data.shim_data, hash)) {
+ if (reg_add(local_data.reg, hash) < 0) {
log_err("Failed to add " HASH_FMT32 " to local registry.",
HASH_VAL32(hash));
return -1;
@@ -177,7 +177,7 @@ static int local_ipcp_reg(const uint8_t * hash)
static int local_ipcp_unreg(const uint8_t * hash)
{
- shim_data_reg_del_entry(local_data.shim_data, hash);
+ reg_del(local_data.reg, hash);
log_info("Unregistered " HASH_FMT32 ".", HASH_VAL32(hash));
@@ -188,7 +188,7 @@ static int local_ipcp_query(const uint8_t * hash)
{
int ret;
- ret = (shim_data_reg_has(local_data.shim_data, hash) ? 0 : -1);
+ ret = (reg_has(local_data.reg, hash) ? 0 : -1);
return ret;
}
diff --git a/src/ipcpd/local/reg.c b/src/ipcpd/local/reg.c
new file mode 100644
index 00000000..36f19b16
--- /dev/null
+++ b/src/ipcpd/local/reg.c
@@ -0,0 +1,217 @@
+/*
+ * Ouroboros - Copyright (C) 2016 - 2026
+ *
+ * Names registered with the local IPCP
+ *
+ * Dimitri Staessens <dimitri@ouroboros.rocks>
+ * Sander Vrijders <sander@ouroboros.rocks>
+ *
+ * This program is free software; you can redistribute it and/or modify
+ * it under the terms of the GNU General Public License version 2 as
+ * published by the Free Software Foundation.
+ *
+ * This program is distributed in the hope that it will be useful,
+ * but WITHOUT ANY WARRANTY; without even the implied warranty of
+ * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+ * GNU General Public License for more details.
+ *
+ * You should have received a copy of the GNU General Public License
+ * along with this program; if not, write to the Free Software
+ * Foundation, Inc., http://www.fsf.org/about/contact/.
+ */
+
+#if defined(__linux__) || defined(__CYGWIN__)
+#define _DEFAULT_SOURCE
+#else
+#define _POSIX_C_SOURCE 200112L
+#endif
+
+#define OUROBOROS_PREFIX "local-reg"
+
+#include <ouroboros/hash.h>
+#include <ouroboros/list.h>
+#include <ouroboros/logs.h>
+
+#include "reg.h"
+#include "ipcp.h"
+
+#include <assert.h>
+#include <pthread.h>
+#include <stdlib.h>
+#include <string.h>
+
+struct reg_entry {
+ struct list_head list;
+ uint8_t * hash;
+};
+
+struct reg {
+ struct list_head names;
+ pthread_rwlock_t lock;
+};
+
+static struct reg_entry * reg_entry_create(uint8_t * hash)
+{
+ struct reg_entry * entry;
+
+ entry = malloc(sizeof(*entry));
+ if (entry == NULL)
+ return NULL;
+
+ list_head_init(&entry->list);
+
+ entry->hash = hash;
+
+ return entry;
+}
+
+static void reg_entry_destroy(struct reg_entry * entry)
+{
+ assert(entry);
+
+ free(entry->hash);
+ free(entry);
+}
+
+/* Call with the lock held. */
+static struct reg_entry * reg_find(struct reg * reg,
+ const uint8_t * hash)
+{
+ struct list_head * p;
+
+ list_for_each(p, &reg->names) {
+ struct reg_entry * e;
+
+ e = list_entry(p, struct reg_entry, list);
+ if (memcmp(e->hash, hash, ipcp_dir_hash_len()) == 0)
+ return e;
+ }
+
+ return NULL;
+}
+
+struct reg * reg_create(void)
+{
+ struct reg * reg;
+
+ reg = malloc(sizeof(*reg));
+ if (reg == NULL)
+ goto fail_malloc;
+
+ list_head_init(&reg->names);
+
+ if (pthread_rwlock_init(&reg->lock, NULL) < 0)
+ goto fail_lock;
+
+ return reg;
+
+ fail_lock:
+ free(reg);
+ fail_malloc:
+ return NULL;
+}
+
+void reg_destroy(struct reg * reg)
+{
+ if (reg == NULL)
+ return;
+
+ pthread_rwlock_wrlock(&reg->lock);
+
+ while (!list_is_empty(&reg->names)) {
+ struct reg_entry * e;
+
+ e = list_first_entry(&reg->names, struct reg_entry, list);
+
+ list_del(&e->list);
+
+ reg_entry_destroy(e);
+ }
+
+ pthread_rwlock_unlock(&reg->lock);
+
+ pthread_rwlock_destroy(&reg->lock);
+
+ free(reg);
+}
+
+int reg_add(struct reg * reg,
+ const uint8_t * hash)
+{
+ struct reg_entry * entry;
+ uint8_t * dup;
+
+ assert(reg);
+ assert(hash);
+
+ pthread_rwlock_wrlock(&reg->lock);
+
+ if (reg_find(reg, hash) != NULL) {
+ pthread_rwlock_unlock(&reg->lock);
+ log_dbg(HASH_FMT32 " was already registered.",
+ HASH_VAL32(hash));
+ return 0;
+ }
+
+ dup = ipcp_hash_dup(hash);
+ if (dup == NULL)
+ goto fail;
+
+ entry = reg_entry_create(dup);
+ if (entry == NULL) {
+ free(dup);
+ goto fail;
+ }
+
+ list_add(&entry->list, &reg->names);
+
+ pthread_rwlock_unlock(&reg->lock);
+
+ return 0;
+
+ fail:
+ pthread_rwlock_unlock(&reg->lock);
+ return -1;
+}
+
+int reg_del(struct reg * reg,
+ const uint8_t * hash)
+{
+ struct reg_entry * e;
+
+ if (reg == NULL)
+ return -1;
+
+ pthread_rwlock_wrlock(&reg->lock);
+
+ e = reg_find(reg, hash);
+ if (e == NULL) {
+ pthread_rwlock_unlock(&reg->lock);
+ return 0; /* nothing to do */
+ }
+
+ list_del(&e->list);
+
+ pthread_rwlock_unlock(&reg->lock);
+
+ reg_entry_destroy(e);
+
+ return 0;
+}
+
+bool reg_has(struct reg * reg,
+ const uint8_t * hash)
+{
+ bool ret;
+
+ assert(reg);
+ assert(hash);
+
+ pthread_rwlock_rdlock(&reg->lock);
+
+ ret = reg_find(reg, hash) != NULL;
+
+ pthread_rwlock_unlock(&reg->lock);
+
+ return ret;
+}
diff --git a/src/ipcpd/eth/llc.c b/src/ipcpd/local/reg.h
index a772e86e..2c6142bb 100644
--- a/src/ipcpd/eth/llc.c
+++ b/src/ipcpd/local/reg.h
@@ -1,7 +1,7 @@
/*
* Ouroboros - Copyright (C) 2016 - 2026
*
- * IPC processes over Ethernet - LLC
+ * Names registered with the local IPCP
*
* Dimitri Staessens <dimitri@ouroboros.rocks>
* Sander Vrijders <sander@ouroboros.rocks>
@@ -20,7 +20,26 @@
* Foundation, Inc., http://www.fsf.org/about/contact/.
*/
-#define BUILD_ETH_LLC
-#define OUROBOROS_PREFIX "ipcpd/eth-llc"
+#ifndef OUROBOROS_IPCPD_LOCAL_REG_H
+#define OUROBOROS_IPCPD_LOCAL_REG_H
-#include "eth.c"
+#include <stdbool.h>
+#include <stdint.h>
+
+/* The hashes of the names registered with this IPCP. */
+struct reg;
+
+struct reg * reg_create(void);
+
+void reg_destroy(struct reg * reg);
+
+int reg_add(struct reg * reg,
+ const uint8_t * hash);
+
+int reg_del(struct reg * reg,
+ const uint8_t * hash);
+
+bool reg_has(struct reg * reg,
+ const uint8_t * hash);
+
+#endif /* OUROBOROS_IPCPD_LOCAL_REG_H */
diff --git a/src/ipcpd/shim-data.c b/src/ipcpd/shim-data.c
deleted file mode 100644
index 90a676da..00000000
--- a/src/ipcpd/shim-data.c
+++ /dev/null
@@ -1,582 +0,0 @@
-/*
- * Ouroboros - Copyright (C) 2016 - 2026
- *
- * IPC process utilities
- *
- * Dimitri Staessens <dimitri@ouroboros.rocks>
- * Sander Vrijders <sander@ouroboros.rocks>
- *
- * This program is free software; you can redistribute it and/or modify
- * it under the terms of the GNU General Public License version 2 as
- * published by the Free Software Foundation.
- *
- * This program is distributed in the hope that it will be useful,
- * but WITHOUT ANY WARRANTY; without even the implied warranty of
- * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
- * GNU General Public License for more details.
- *
- * You should have received a copy of the GNU General Public License
- * along with this program; if not, write to the Free Software
- * Foundation, Inc., http://www.fsf.org/about/contact/.
- */
-
-#if defined(__linux__) || defined(__CYGWIN__)
-#define _DEFAULT_SOURCE
-#else
-#define _POSIX_C_SOURCE 200112L
-#endif
-
-#include "config.h"
-
-#define OUROBOROS_PREFIX "shim-data"
-
-#include <ouroboros/errno.h>
-#include <ouroboros/hash.h>
-#include <ouroboros/list.h>
-#include <ouroboros/logs.h>
-#include <ouroboros/time.h>
-
-#include "shim-data.h"
-#include "ipcp.h"
-
-#include <assert.h>
-#include <stdlib.h>
-#include <string.h>
-
-struct reg_entry {
- struct list_head list;
- uint8_t * hash;
-};
-
-struct dir_entry {
- struct list_head list;
- uint8_t * hash;
- struct addr addr;
-};
-
-static void destroy_dir_query(struct dir_query * query)
-{
- assert(query);
-
- pthread_mutex_lock(&query->lock);
-
- switch (query->state) {
- case QUERY_INIT:
- query->state = QUERY_DONE;
- break;
- case QUERY_PENDING:
- query->state = QUERY_DESTROY;
- pthread_cond_broadcast(&query->cond);
- break;
- case QUERY_RESPONSE:
- case QUERY_DONE:
- break;
- case QUERY_DESTROY:
- pthread_mutex_unlock(&query->lock);
- return;
- default:
- assert(false);
- return;
- }
-
- while (query->state != QUERY_DONE)
- pthread_cond_wait(&query->cond, &query->lock);
-
- pthread_mutex_unlock(&query->lock);
-
- pthread_cond_destroy(&query->cond);
- pthread_mutex_destroy(&query->lock);
-
- free(query->hash);
- free(query);
-}
-
-static struct reg_entry * reg_entry_create(uint8_t * hash)
-{
- struct reg_entry * entry = malloc(sizeof(*entry));
- if (entry == NULL)
- return NULL;
-
- assert(hash);
-
- entry->hash = hash;
-
- return entry;
-}
-
-static void reg_entry_destroy(struct reg_entry * entry)
-{
- assert(entry);
-
- free(entry->hash);
- free(entry);
-}
-
-static struct dir_entry * dir_entry_create(uint8_t * hash,
- struct addr addr)
-{
- struct dir_entry * entry = malloc(sizeof(*entry));
- if (entry == NULL)
- return NULL;
-
- assert(hash);
-
- entry->addr = addr;
- entry->hash = hash;
-
- return entry;
-}
-
-static void dir_entry_destroy(struct dir_entry * entry)
-{
- assert(entry);
-
- free(entry->hash);
- free(entry);
-}
-
-struct shim_data * shim_data_create(void)
-{
- struct shim_data * sd;
-
- sd = malloc(sizeof(*sd));
- if (sd == NULL)
- return NULL;
-
- /* init the lists */
- list_head_init(&sd->registry);
- list_head_init(&sd->directory);
- list_head_init(&sd->dir_queries);
-
- /* init the locks */
- if (pthread_rwlock_init(&sd->reg_lock, NULL) < 0)
- goto fail_reg_lock_init;
-
- if (pthread_rwlock_init(&sd->dir_lock, NULL) < 0)
- goto fail_dir_lock_init;
-
- if (pthread_mutex_init(&sd->dir_queries_lock, NULL) < 0)
- goto fail_mutex_init;
-
- return sd;
-
- fail_mutex_init:
- pthread_rwlock_destroy(&sd->dir_lock);
- fail_dir_lock_init:
- pthread_rwlock_destroy(&sd->reg_lock);
- fail_reg_lock_init:
- return NULL;
-}
-
-static void clear_registry(struct shim_data * data)
-{
- struct list_head * h;
- struct list_head * t;
-
- assert(data);
-
- list_for_each_safe(h, t, &data->registry) {
- struct reg_entry * e = list_entry(h, struct reg_entry, list);
- list_del(&e->list);
- reg_entry_destroy(e);
- }
-}
-
-static void clear_directory(struct shim_data * data)
-{
- struct list_head * h;
- struct list_head * t;
-
- assert(data);
-
- list_for_each_safe(h, t, &data->directory) {
- struct dir_entry * e = list_entry(h, struct dir_entry, list);
- list_del(&e->list);
- dir_entry_destroy(e);
- }
-}
-
-static void clear_dir_queries(struct shim_data * data)
-{
- struct list_head * h;
- struct list_head * t;
-
- assert(data);
-
- list_for_each_safe(h, t, &data->dir_queries) {
- struct dir_query * e = list_entry(h, struct dir_query, next);
- list_del(&e->next);
- destroy_dir_query(e);
- }
-}
-
-void shim_data_destroy(struct shim_data * data)
-{
- if (data == NULL)
- return;
-
- /* clear the lists */
- pthread_rwlock_wrlock(&data->reg_lock);
- clear_registry(data);
- pthread_rwlock_unlock(&data->reg_lock);
-
- pthread_rwlock_wrlock(&data->dir_lock);
- clear_directory(data);
- pthread_rwlock_unlock(&data->dir_lock);
-
- pthread_mutex_lock(&data->dir_queries_lock);
- clear_dir_queries(data);
- pthread_mutex_unlock(&data->dir_queries_lock);
-
- pthread_rwlock_destroy(&data->dir_lock);
- pthread_rwlock_destroy(&data->reg_lock);
- pthread_mutex_destroy(&data->dir_queries_lock);
-
- free(data);
-}
-
-static struct reg_entry * find_reg_entry_by_hash(struct shim_data * data,
- const uint8_t * hash)
-{
- struct list_head * h;
-
- assert(data);
- assert(hash);
-
- list_for_each(h, &data->registry) {
- struct reg_entry * e = list_entry(h, struct reg_entry, list);
- if (!memcmp(e->hash, hash, ipcp_dir_hash_len()))
- return e;
- }
-
- return NULL;
-}
-
-static struct dir_entry * find_dir_entry(struct shim_data * data,
- const uint8_t * hash,
- struct addr addr)
-{
- struct list_head * h;
- list_for_each(h, &data->directory) {
- struct dir_entry * e = list_entry(h, struct dir_entry, list);
- if (memcmp(&e->addr, &addr, sizeof(addr)) != 0)
- continue;
-
- if (memcmp(e->hash, hash, ipcp_dir_hash_len()) == 0)
- return e;
- }
-
- return NULL;
-}
-
-static struct dir_entry * find_dir_entry_any(struct shim_data * data,
- const uint8_t * hash)
-{
- struct list_head * h;
- list_for_each(h, &data->directory) {
- struct dir_entry * e = list_entry(h, struct dir_entry, list);
- if (!memcmp(e->hash, hash, ipcp_dir_hash_len()))
- return e;
- }
-
- return NULL;
-}
-
-int shim_data_reg_add_entry(struct shim_data * data,
- const uint8_t * hash)
-{
- struct reg_entry * entry;
- uint8_t * hash_dup;
-
- assert(data);
- assert(hash);
-
- pthread_rwlock_wrlock(&data->reg_lock);
-
- if (find_reg_entry_by_hash(data, hash)) {
- pthread_rwlock_unlock(&data->reg_lock);
- log_dbg(HASH_FMT32 " was already in the directory.",
- HASH_VAL32(hash));
- return 0;
- }
-
- hash_dup = ipcp_hash_dup(hash);
- if (hash_dup == NULL) {
- pthread_rwlock_unlock(&data->reg_lock);
- return -1;
- }
-
- entry = reg_entry_create(hash_dup);
- if (entry == NULL) {
- pthread_rwlock_unlock(&data->reg_lock);
- return -1;
- }
-
- list_add(&entry->list, &data->registry);
-
- pthread_rwlock_unlock(&data->reg_lock);
-
- return 0;
-}
-
-int shim_data_reg_del_entry(struct shim_data * data,
- const uint8_t * hash)
-{
- struct reg_entry * e;
- if (data == NULL)
- return -1;
-
- pthread_rwlock_wrlock(&data->reg_lock);
-
- e = find_reg_entry_by_hash(data, hash);
- if (e == NULL) {
- pthread_rwlock_unlock(&data->reg_lock);
- return 0; /* nothing to do */
- }
-
- list_del(&e->list);
-
- pthread_rwlock_unlock(&data->reg_lock);
-
- reg_entry_destroy(e);
-
- return 0;
-}
-
-bool shim_data_reg_has(struct shim_data * data,
- const uint8_t * hash)
-{
- bool ret = false;
-
- assert(data);
- assert(hash);
-
- pthread_rwlock_rdlock(&data->reg_lock);
-
- ret = (find_reg_entry_by_hash(data, hash) != NULL);
-
- pthread_rwlock_unlock(&data->reg_lock);
-
- return ret;
-}
-
-int shim_data_dir_add_entry(struct shim_data * data,
- const uint8_t * hash,
- struct addr addr)
-{
- struct dir_entry * entry;
- uint8_t * entry_hash;
-
- assert(data);
- assert(hash);
-
- pthread_rwlock_wrlock(&data->dir_lock);
-
- if (find_dir_entry(data, hash, addr) != NULL) {
- pthread_rwlock_unlock(&data->dir_lock);
- return -1;
- }
-
- entry_hash = ipcp_hash_dup(hash);
- if (entry_hash == NULL) {
- pthread_rwlock_unlock(&data->dir_lock);
- return -1;
- }
-
- entry = dir_entry_create(entry_hash, addr);
- if (entry == NULL) {
- pthread_rwlock_unlock(&data->dir_lock);
- return -1;
- }
-
- list_add(&entry->list,&data->directory);
-
- pthread_rwlock_unlock(&data->dir_lock);
-
- return 0;
-}
-
-int shim_data_dir_del_entry(struct shim_data * data,
- const uint8_t * hash,
- struct addr addr)
-{
- struct dir_entry * e;
- if (data == NULL)
- return -1;
-
- pthread_rwlock_wrlock(&data->dir_lock);
-
- e = find_dir_entry(data, hash, addr);
- if (e == NULL) {
- pthread_rwlock_unlock(&data->dir_lock);
- return 0; /* nothing to do */
- }
-
- list_del(&e->list);
-
- pthread_rwlock_unlock(&data->dir_lock);
-
- dir_entry_destroy(e);
-
- return 0;
-}
-
-bool shim_data_dir_has(struct shim_data * data,
- const uint8_t * hash)
-{
- bool ret = false;
-
- pthread_rwlock_rdlock(&data->dir_lock);
-
- ret = (find_dir_entry_any(data, hash) != NULL);
-
- pthread_rwlock_unlock(&data->dir_lock);
-
- return ret;
-}
-
-struct addr shim_data_dir_get_addr(struct shim_data * data,
- const uint8_t * hash)
-{
- struct dir_entry * entry;
- struct addr addr = {0};
-
- pthread_rwlock_rdlock(&data->dir_lock);
-
- entry = find_dir_entry_any(data, hash);
- if (entry == NULL) {
- pthread_rwlock_unlock(&data->dir_lock);
- log_warn("No address for " HASH_FMT32 ".", HASH_VAL32(hash));
- return addr; /* undefined behaviour, 0 may be a valid address */
- }
-
- addr = entry->addr;
-
- pthread_rwlock_unlock(&data->dir_lock);
-
- return addr;
-}
-
-struct dir_query * shim_data_dir_query_create(struct shim_data * data,
- const uint8_t * hash)
-{
- struct dir_query * query;
- pthread_condattr_t cattr;
-
- query = malloc(sizeof(*query));
- if (query == NULL)
- return NULL;
-
- query->hash = ipcp_hash_dup(hash);
- if (query->hash == NULL) {
- free(query);
- return NULL;
- }
-
- query->state = QUERY_INIT;
-
- pthread_condattr_init(&cattr);
-#ifndef __APPLE__
- pthread_condattr_setclock(&cattr, PTHREAD_COND_CLOCK);
-#endif
- pthread_cond_init(&query->cond, &cattr);
- pthread_mutex_init(&query->lock, NULL);
-
- list_head_init(&query->next);
-
- pthread_mutex_lock(&data->dir_queries_lock);
- list_add(&query->next, &data->dir_queries);
- pthread_mutex_unlock(&data->dir_queries_lock);
-
- return query;
-}
-
-void shim_data_dir_query_respond(struct shim_data * data,
- const uint8_t * hash)
-{
- struct dir_query * e = NULL;
- struct list_head * pos;
- bool found = false;
-
- pthread_mutex_lock(&data->dir_queries_lock);
-
- list_for_each(pos, &data->dir_queries) {
- e = list_entry(pos, struct dir_query, next);
-
- if (memcmp(e->hash, hash, ipcp_dir_hash_len()) == 0) {
- found = true;
- break;
- }
- }
-
- if (!found) {
- pthread_mutex_unlock(&data->dir_queries_lock);
- return;
- }
-
- pthread_mutex_lock(&e->lock);
-
- if (e->state != QUERY_PENDING) {
- pthread_mutex_unlock(&e->lock);
- pthread_mutex_unlock(&data->dir_queries_lock);
- return;
- }
-
- e->state = QUERY_RESPONSE;
- pthread_cond_broadcast(&e->cond);
-
- while (e->state == QUERY_RESPONSE)
- pthread_cond_wait(&e->cond, &e->lock);
-
- pthread_mutex_unlock(&e->lock);
-
- pthread_mutex_unlock(&data->dir_queries_lock);
-}
-
-void shim_data_dir_query_destroy(struct shim_data * data,
- struct dir_query * query)
-{
- pthread_mutex_lock(&data->dir_queries_lock);
-
- list_del(&query->next);
- destroy_dir_query(query);
-
- pthread_mutex_unlock(&data->dir_queries_lock);
-}
-
-int shim_data_dir_query_wait(struct dir_query * query,
- const struct timespec * timeout)
-{
- struct timespec abstime;
- int ret = 0;
-
- assert(query);
- assert(timeout);
-
- clock_gettime(PTHREAD_COND_CLOCK, &abstime);
- ts_add(&abstime, timeout, &abstime);
-
- pthread_mutex_lock(&query->lock);
-
- if (query->state != QUERY_INIT) {
- pthread_mutex_unlock(&query->lock);
- return -EINVAL;
- }
-
- query->state = QUERY_PENDING;
-
- while (query->state == QUERY_PENDING && ret != -ETIMEDOUT)
- ret = -pthread_cond_timedwait(&query->cond,
- &query->lock,
- &abstime);
-
- if (query->state == QUERY_DESTROY)
- ret = -1;
-
- query->state = QUERY_DONE;
- pthread_cond_broadcast(&query->cond);
-
- pthread_mutex_unlock(&query->lock);
-
- return ret;
-}
diff --git a/src/ipcpd/shim-data.h b/src/ipcpd/shim-data.h
deleted file mode 100644
index fbadb4d4..00000000
--- a/src/ipcpd/shim-data.h
+++ /dev/null
@@ -1,109 +0,0 @@
-/*
- * Ouroboros - Copyright (C) 2016 - 2026
- *
- * Utitilies for building IPC processes
- *
- * Dimitri Staessens <dimitri@ouroboros.rocks>
- * Sander Vrijders <sander@ouroboros.rocks>
- *
- * This program is free software; you can redistribute it and/or modify
- * it under the terms of the GNU General Public License version 2 as
- * published by the Free Software Foundation.
- *
- * This program is distributed in the hope that it will be useful,
- * but WITHOUT ANY WARRANTY; without even the implied warranty of
- * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
- * GNU General Public License for more details.
- *
- * You should have received a copy of the GNU General Public License
- * along with this program; if not, write to the Free Software
- * Foundation, Inc., http://www.fsf.org/about/contact/.
- */
-
-#ifndef OUROBOROS_IPCPD_IPCP_DATA_H
-#define OUROBOROS_IPCPD_IPCP_DATA_H
-
-#include <ouroboros/list.h>
-
-#include <pthread.h>
-#include <stdint.h>
-#include <netinet/in.h>
-#include <sys/types.h>
-
-#define MAC_SIZE 6
-
-enum dir_query_state {
- QUERY_INIT = 0,
- QUERY_PENDING,
- QUERY_RESPONSE,
- QUERY_DONE,
- QUERY_DESTROY
-};
-
-struct dir_query {
- struct list_head next;
- uint8_t * hash;
- enum dir_query_state state;
-
- pthread_mutex_t lock;
- pthread_cond_t cond;
-};
-
-struct addr {
- union {
- uint8_t mac[MAC_SIZE];
- struct in_addr ip4;
- struct in6_addr ip6;
- };
-};
-
-struct shim_data {
- struct list_head registry;
- pthread_rwlock_t reg_lock;
-
- struct list_head directory;
- pthread_rwlock_t dir_lock;
-
- struct list_head dir_queries;
- pthread_mutex_t dir_queries_lock;
-};
-
-struct shim_data * shim_data_create(void);
-
-void shim_data_destroy(struct shim_data * data);
-
-int shim_data_reg_add_entry(struct shim_data * data,
- const uint8_t * hash);
-
-int shim_data_reg_del_entry(struct shim_data * data,
- const uint8_t * hash);
-
-bool shim_data_reg_has(struct shim_data * data,
- const uint8_t * hash);
-
-int shim_data_dir_add_entry(struct shim_data * data,
- const uint8_t * hash,
- struct addr addr);
-
-int shim_data_dir_del_entry(struct shim_data * data,
- const uint8_t * hash,
- struct addr addr);
-
-bool shim_data_dir_has(struct shim_data * data,
- const uint8_t * hash);
-
-struct addr shim_data_dir_get_addr(struct shim_data * data,
- const uint8_t * hash);
-
-struct dir_query * shim_data_dir_query_create(struct shim_data * data,
- const uint8_t * hash);
-
-void shim_data_dir_query_destroy(struct shim_data * data,
- struct dir_query * query);
-
-void shim_data_dir_query_respond(struct shim_data * data,
- const uint8_t * hash);
-
-int shim_data_dir_query_wait(struct dir_query * query,
- const struct timespec * timeout);
-#endif /* OUROBOROS_IPCPD_SHIM_DATA_H */
diff --git a/src/ipcpd/udp/CMakeLists.txt b/src/ipcpd/udp/CMakeLists.txt
deleted file mode 100644
index a98f0919..00000000
--- a/src/ipcpd/udp/CMakeLists.txt
+++ /dev/null
@@ -1,14 +0,0 @@
-# UDP IPCPs build configuration (UDP4 and UDP6)
-# DDNS detection is in cmake/dependencies/udp/ddns.cmake
-
-add_executable(${IPCP_UDP4_TARGET} udp4.c ${IPCP_SOURCES})
-add_executable(${IPCP_UDP6_TARGET} udp6.c ${IPCP_SOURCES})
-
-foreach(target ${IPCP_UDP4_TARGET} ${IPCP_UDP6_TARGET})
- target_include_directories(${target} PRIVATE ${IPCP_INCLUDE_DIRS})
- target_link_libraries(${target} PRIVATE ouroboros-dev)
- ouroboros_target_debug_definitions(${target})
-endforeach()
-
-install(TARGETS ${IPCP_UDP4_TARGET} ${IPCP_UDP6_TARGET}
- RUNTIME DESTINATION ${CMAKE_INSTALL_SBINDIR})
diff --git a/src/ipcpd/udp/udp.c b/src/ipcpd/udp/udp.c
deleted file mode 100644
index db57e2f4..00000000
--- a/src/ipcpd/udp/udp.c
+++ /dev/null
@@ -1,1333 +0,0 @@
-/*
- * Ouroboros - Copyright (C) 2016 - 2026
- *
- * IPC process over UDP
- *
- * Dimitri Staessens <dimitri@ouroboros.rocks>
- * Sander Vrijders <sander@ouroboros.rocks>
- *
- * This program is free software; you can redistribute it and/or modify
- * it under the terms of the GNU General Public License version 2 as
- * published by the Free Software Foundation.
- *
- * This program is distributed in the hope that it will be useful,
- * but WITHOUT ANY WARRANTY; without even the implied warranty of
- * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
- * GNU General Public License for more details.
- *
- * You should have received a copy of the GNU General Public License
- * along with this program; if not, write to the Free Software
- * Foundation, Inc., http://www.fsf.org/about/contact/.
- */
-
-#include "config.h"
-
-#include <ouroboros/bitmap.h>
-#include <ouroboros/endian.h>
-#include <ouroboros/hash.h>
-#include <ouroboros/list.h>
-#include <ouroboros/utils.h>
-#include <ouroboros/dev.h>
-#include <ouroboros/ipcp-dev.h>
-#include <ouroboros/np1_flow.h>
-#include <ouroboros/fqueue.h>
-#include <ouroboros/errno.h>
-#include <ouroboros/logs.h>
-#include <ouroboros/pthread.h>
-
-#include "ipcp.h"
-#include "np1.h"
-#include "shim-data.h"
-
-#include <string.h>
-#include <sys/socket.h>
-#include <sys/select.h>
-#include <arpa/inet.h>
-#include <netdb.h>
-#include <netinet/in.h>
-#include <signal.h>
-#include <stdlib.h>
-#include <sys/wait.h>
-#include <fcntl.h>
-#include <unistd.h>
-#if defined(__linux__)
-#include <netinet/ip.h>
-#endif
-
-#define FLOW_REQ 1
-#define FLOW_REPLY 2
-#define FLOW_IRM_UPDATE 3
-
-#define OUR_HEADER_LEN sizeof(uint32_t) /* adds eid */
-
-#define IPCP_UDP_BUF_SIZE IPCP_UDP_MAX_PACKET_SIZE
-#define IPCP_UDP_MSG_SIZE IPCP_UDP_MAX_PACKET_SIZE
-
-#define DNS_TTL 86400
-
-#define SADDR ((struct sockaddr *) &udp_data.s_saddr)
-#define SADDR_SIZE (sizeof(udp_data.s_saddr))
-#define LOCAL_IP (udp_data.s_saddr.sin_addr.s_addr)
-
-#define MGMT_EID 0
-#define MGMT_FRAME_SIZE (sizeof(struct mgmt_msg))
-#define MGMT_FRAME_BUF_SIZE 2048
-
-#ifdef __linux__
-#define SENDTO_FLAGS MSG_CONFIRM
-#else
-#define SENDTO_FLAGS 0
-#endif
-
-/* Keep order for alignment. */
-struct mgmt_msg {
- uint32_t eid;
- uint32_t s_eid;
- uint32_t d_eid;
- int32_t response;
- uint64_t bandwidth;
- uint32_t delay;
- uint32_t loss;
- uint32_t ber;
- uint32_t max_gap;
- uint32_t timeout;
- uint8_t code;
- /* QoS parameters from spec */
- uint8_t availability;
- uint8_t service;
-} __attribute__((packed));
-
-struct mgmt_frame {
- struct list_head next;
- struct __SOCKADDR r_saddr;
- uint8_t buf[MGMT_FRAME_BUF_SIZE];
- size_t len;
-};
-
-/* UDP flow */
-struct uf {
- int d_eid;
- struct __SOCKADDR r_saddr;
-};
-
-struct {
- struct shim_data * shim_data;
-
- struct __ADDR dns_addr;
- struct __SOCKADDR s_saddr;
- int s_fd;
-
- fset_t * np1_flows;
- struct uf fd_to_uf[SYS_MAX_FLOWS];
- pthread_rwlock_t flows_lock;
-
- pthread_t packet_writer[IPCP_UDP_WR_THR];
- pthread_t packet_reader[IPCP_UDP_RD_THR];
-
- /* Handle mgmt frames in a different thread */
- pthread_t mgmt_handler;
- pthread_mutex_t mgmt_lock;
- pthread_cond_t mgmt_cond;
- struct list_head mgmt_frames;
-} udp_data;
-
-static const char * __inet_ntop(const struct __ADDR * addr,
- char * buf)
-{
- return inet_ntop(__AF, addr, buf, __ADDRSTRLEN);
-}
-
-#if defined(BUILD_IPCP_UDP4)
-#define UDP_MTU_FALLBACK IPCP_UDP4_MTU
-#define UDP_IP_OVERHEAD 28U /* IPv4 + UDP */
-#else
-#define UDP_MTU_FALLBACK IPCP_UDP6_MTU
-#define UDP_IP_OVERHEAD 48U /* IPv6 + UDP */
-#endif
-
-static uint32_t udp_query_mtu(const struct __SOCKADDR * saddr)
-{
-#if defined(__linux__) && (defined(IP_MTU) || defined(IPV6_MTU))
- int sock;
- int mtu = 0;
- socklen_t len = sizeof(mtu);
-
- sock = socket(__AF, SOCK_DGRAM, IPPROTO_UDP);
- if (sock < 0)
- return UDP_MTU_FALLBACK;
-
- if (connect(sock, (const struct sockaddr *) saddr,
- sizeof(*saddr)) < 0)
- goto fallback;
-
-#if defined(BUILD_IPCP_UDP4) && defined(IP_MTU)
- if (getsockopt(sock, IPPROTO_IP, IP_MTU, &mtu, &len) < 0)
- goto fallback;
-#elif defined(BUILD_IPCP_UDP6) && defined(IPV6_MTU)
- if (getsockopt(sock, IPPROTO_IPV6, IPV6_MTU, &mtu, &len) < 0)
- goto fallback;
-#else
- goto fallback;
-#endif
- close(sock);
-
- if (mtu <= (int) UDP_IP_OVERHEAD)
- return UDP_MTU_FALLBACK;
-
- return (uint32_t) mtu - UDP_IP_OVERHEAD;
-
- fallback:
- close(sock);
-#else
- (void) saddr;
-#endif
- return UDP_MTU_FALLBACK;
-}
-
-static int udp_data_init(void)
-{
- int i;
- pthread_condattr_t cattr;
-
- if (pthread_rwlock_init(&udp_data.flows_lock, NULL))
- goto fail_rwlock_init;
-
- if (pthread_condattr_init(&cattr))
- goto fail_condattr;
-#ifndef __APPLE__
- pthread_condattr_setclock(&cattr, PTHREAD_COND_CLOCK);
-#endif
- if (pthread_cond_init(&udp_data.mgmt_cond, &cattr))
- goto fail_mgmt_cond;
-
- if (pthread_mutex_init(&udp_data.mgmt_lock, NULL))
- goto fail_mgmt_lock;
-
- for (i = 0; i < SYS_MAX_FLOWS; ++i)
- udp_data.fd_to_uf[i].d_eid = -1;
-
- udp_data.np1_flows = fset_create();
- if (udp_data.np1_flows == NULL)
- goto fail_fset;
-
- udp_data.shim_data = shim_data_create();
- if (udp_data.shim_data == NULL)
- goto fail_data;
-
- pthread_condattr_destroy(&cattr);
-
- list_head_init(&udp_data.mgmt_frames);
-
- return 0;
-
- fail_data:
- fset_destroy(udp_data.np1_flows);
- fail_fset:
- pthread_mutex_destroy(&udp_data.mgmt_lock);
- fail_mgmt_lock:
- pthread_cond_destroy(&udp_data.mgmt_cond);
- fail_mgmt_cond:
- pthread_condattr_destroy(&cattr);
- fail_condattr:
- pthread_rwlock_destroy(&udp_data.flows_lock);
- fail_rwlock_init:
- return -1;
-}
-
-static void udp_data_fini(void)
-{
- shim_data_destroy(udp_data.shim_data);
-
- fset_destroy(udp_data.np1_flows);
-
- pthread_rwlock_destroy(&udp_data.flows_lock);
- pthread_cond_destroy(&udp_data.mgmt_cond);
- pthread_mutex_destroy(&udp_data.mgmt_lock);
-}
-
-static int udp_ipcp_port_alloc(const struct __SOCKADDR * r_saddr,
- uint32_t s_eid,
- const uint8_t * dst,
- qosspec_t qs,
- const buffer_t * data)
-{
- uint8_t * buf;
- struct mgmt_msg * msg;
- size_t len;
-
- assert(data->len > 0 ? data->data != NULL : data->data == NULL);
-
- len = sizeof(*msg) + ipcp_dir_hash_len();
-
- buf = malloc(len + data->len);
- if (buf == NULL)
- return -1;
-
- memset(buf, 0, len + data->len);
-
- msg = (struct mgmt_msg *) buf;
- msg->eid = hton32(MGMT_EID);
- msg->code = FLOW_REQ;
- msg->s_eid = hton32(s_eid);
- msg->delay = hton32(qs.delay);
- msg->bandwidth = hton64(qs.bandwidth);
- msg->availability = qs.availability;
- msg->loss = hton32(qs.loss);
- msg->ber = hton32(qs.ber);
- msg->service = qs.service;
- msg->max_gap = hton32(qs.max_gap);
- msg->timeout = hton32(qs.timeout);
-
- memcpy(msg + 1, dst, ipcp_dir_hash_len());
- if (data->len > 0)
- memcpy(buf + len, data->data, data->len);
-
- if (sendto(udp_data.s_fd, msg, len + data->len,
- SENDTO_FLAGS,
- (const struct sockaddr *) r_saddr, sizeof(*r_saddr)) < 0) {
- log_err("Failed to send flow allocation request: %s.",
- strerror(errno));
- free(buf);
- return -1;
- }
-
- free(buf);
-
- return 0;
-}
-
-static int udp_ipcp_port_alloc_resp(const struct __SOCKADDR * r_saddr,
- uint32_t s_eid,
- uint32_t d_eid,
- int32_t response,
- const buffer_t * data)
-{
- struct mgmt_msg * msg;
-
- msg = malloc(sizeof(*msg) + data->len);
- if (msg == NULL)
- return -1;
-
- memset(msg, 0, sizeof(*msg) + data->len);
-
- msg->eid = hton32(MGMT_EID);
- msg->code = FLOW_REPLY;
- msg->s_eid = hton32(s_eid);
- msg->d_eid = hton32(d_eid);
- msg->response = hton32(response);
-
- if (data->len > 0)
- memcpy(msg + 1, data->data, data->len);
-
- if (sendto(udp_data.s_fd, msg, sizeof(*msg) + data->len,
- SENDTO_FLAGS,
- (const struct sockaddr *) r_saddr, sizeof(*r_saddr)) < 0 ) {
- free(msg);
- return -1;
- }
-
- free(msg);
-
- return 0;
-}
-
-static int udp_ipcp_flow_update(int fd,
- const buffer_t * data)
-{
- struct mgmt_msg * msg;
- struct __SOCKADDR r_saddr;
- uint32_t d_eid;
-
- msg = malloc(sizeof(*msg) + data->len);
- if (msg == NULL)
- return -1;
-
- memset(msg, 0, sizeof(*msg) + data->len);
-
- pthread_rwlock_rdlock(&udp_data.flows_lock);
-
- r_saddr = udp_data.fd_to_uf[fd].r_saddr;
- d_eid = (uint32_t) udp_data.fd_to_uf[fd].d_eid;
-
- pthread_rwlock_unlock(&udp_data.flows_lock);
-
- msg->eid = hton32(MGMT_EID);
- msg->code = FLOW_IRM_UPDATE;
- msg->s_eid = hton32(d_eid);
- msg->d_eid = hton32((uint32_t) fd);
-
- if (data->len > 0)
- memcpy(msg + 1, data->data, data->len);
-
- if (sendto(udp_data.s_fd, msg, sizeof(*msg) + data->len,
- SENDTO_FLAGS,
- (const struct sockaddr *) &r_saddr,
- sizeof(r_saddr)) < 0) {
- log_err("Failed to send flow update: %s.", strerror(errno));
- free(msg);
- return -1;
- }
-
- free(msg);
-
- return 0;
-}
-
-static int udp_ipcp_port_req(struct __SOCKADDR * c_saddr,
- int d_eid,
- const uint8_t * dst,
- qosspec_t qs,
- const buffer_t * data)
-{
- int fd;
-
- fd = ipcp_wait_flow_req_arr(dst, qs, IPCP_UDP_MPL,
- udp_query_mtu(c_saddr), data);
- if (fd < 0) {
- log_err("Could not get new flow from IRMd.");
- return -1;
- }
-
- pthread_rwlock_wrlock(&udp_data.flows_lock);
-
- udp_data.fd_to_uf[fd].r_saddr = *c_saddr;
- udp_data.fd_to_uf[fd].d_eid = d_eid;
-
- pthread_rwlock_unlock(&udp_data.flows_lock);
-
- log_dbg("Pending allocation request, fd %d, remote eid %d.",
- fd, d_eid);
-
- return 0;
-}
-
-static int udp_ipcp_port_alloc_reply(const struct __SOCKADDR * saddr,
- uint32_t s_eid,
- uint32_t d_eid,
- int32_t response,
- const buffer_t * data)
-{
- time_t mpl = IPCP_UDP_MPL;
-
- pthread_rwlock_wrlock(&udp_data.flows_lock);
-
- if (memcmp(&udp_data.fd_to_uf[s_eid].r_saddr, saddr, sizeof(*saddr))) {
- char ipstr[__ADDRSTRLEN];
- pthread_rwlock_unlock(&udp_data.flows_lock);
- #ifdef BUILD_IPCP_UDP4
- __inet_ntop(&saddr->sin_addr, ipstr);
- #else
- __inet_ntop(&saddr->sin6_addr, ipstr);
- #endif
- log_err("Flow allocation reply for %u from wrong source %s.",
- s_eid, ipstr);
- return -1;
- }
-
- if (response == 0)
- udp_data.fd_to_uf[s_eid].d_eid = d_eid;
-
- pthread_rwlock_unlock(&udp_data.flows_lock);
-
- if (ipcp_flow_alloc_reply(s_eid, response, mpl,
- udp_query_mtu(saddr), data) < 0) {
- log_err("Failed to reply to flow allocation.");
- return -1;
- }
-
- log_dbg("Flow allocation completed on eids (%d, %d).",
- s_eid, d_eid);
-
- return 0;
-}
-
-static int udp_ipcp_flow_update_arr(const uint8_t * buf,
- size_t len)
-{
- struct mgmt_msg * msg;
- buffer_t data;
- int fd;
- int flow_id;
-
- msg = (struct mgmt_msg *) buf;
-
- fd = (int) ntoh32(msg->s_eid);
- if (fd < 0 || fd >= SYS_MAX_FLOWS) {
- log_err("Flow update for invalid eid %d.", fd);
- return -1;
- }
-
- data.len = len - sizeof(*msg);
- data.data = (uint8_t *) buf + sizeof(*msg);
-
- flow_id = np1_flow_id(fd);
- if (flow_id < 0)
- return -1;
-
- if (ipcp_flow_update_arr(flow_id, &data) < 0) {
- log_err("Failed to relay flow update on fd %d.", fd);
- return -1;
- }
-
- return 0;
-}
-
-static int udp_ipcp_mgmt_frame(struct __SOCKADDR c_saddr,
- const uint8_t * buf,
- size_t len)
-{
- struct mgmt_msg * msg;
- size_t msg_len;
- qosspec_t qs;
- buffer_t data;
-
- /* Defence against malformed/corrupted wire input. */
- if (len < sizeof(*msg))
- return -1;
-
- msg = (struct mgmt_msg *) buf;
-
- switch (msg->code) {
- case FLOW_REQ:
- msg_len = sizeof(*msg) + ipcp_dir_hash_len();
-
- if (len < msg_len)
- return -1;
-
- data.len = len - msg_len;
- data.data = (uint8_t *) buf + msg_len;
-
-
- qs.delay = ntoh32(msg->delay);
- qs.bandwidth = ntoh64(msg->bandwidth);
- qs.availability = msg->availability;
- qs.loss = ntoh32(msg->loss);
- qs.ber = ntoh32(msg->ber);
- qs.service = msg->service;
- qs.max_gap = ntoh32(msg->max_gap);
- qs.timeout = ntoh32(msg->timeout);
-
- return udp_ipcp_port_req(&c_saddr, ntoh32(msg->s_eid),
- (uint8_t *) (msg + 1), qs,
- &data);
- case FLOW_REPLY:
- data.len = len - sizeof(*msg);
- data.data = (uint8_t *) buf + sizeof(*msg);
-
- return udp_ipcp_port_alloc_reply(&c_saddr,
- ntoh32(msg->s_eid),
- ntoh32(msg->d_eid),
- ntoh32(msg->response),
- &data);
- case FLOW_IRM_UPDATE:
- return udp_ipcp_flow_update_arr(buf, len);
- default:
- log_err("Unknown message received %d.", msg->code);
- return -1;
- }
-}
-
-static void * udp_ipcp_mgmt_handler(void * o)
-{
- (void) o;
-
- pthread_cleanup_push(__cleanup_mutex_unlock, &udp_data.mgmt_lock);
-
- while (true) {
- struct mgmt_frame * frame;
-
- pthread_mutex_lock(&udp_data.mgmt_lock);
-
- while (list_is_empty(&udp_data.mgmt_frames))
- pthread_cond_wait(&udp_data.mgmt_cond,
- &udp_data.mgmt_lock);
-
- frame = list_first_entry((&udp_data.mgmt_frames),
- struct mgmt_frame, next);
- assert(frame != NULL);
- list_del(&frame->next);
-
- pthread_mutex_unlock(&udp_data.mgmt_lock);
-
- udp_ipcp_mgmt_frame(frame->r_saddr, frame->buf, frame->len);
-
- free(frame);
- }
-
- pthread_cleanup_pop(false);
-
- return (void *) 0;
-}
-
-static void * udp_ipcp_packet_reader(void * o)
-{
- uint8_t buf[IPCP_UDP_MAX_PACKET_SIZE];
- uint8_t * data;
- ssize_t n;
- uint32_t eid;
- uint32_t * eid_p;
-
- (void) o;
-
- ipcp_lock_to_core();
-
- data = buf + sizeof(uint32_t);
- eid_p = (uint32_t *) buf;
-
- while (true) {
- struct mgmt_frame * frame;
- struct __SOCKADDR r_saddr;
- socklen_t len;
- struct ssm_pk_buff * spb;
- uint8_t * head;
-
- len = sizeof(r_saddr);
-
- n = recvfrom(udp_data.s_fd, buf, IPCP_UDP_MAX_PACKET_SIZE, 0,
- (struct sockaddr *) &r_saddr, &len);
- if (n < 0)
- continue;
-
- if (n == 0)
- log_dbg("Got a 0 frame.");
-
- if ((size_t) n < sizeof(eid)) {
- log_dbg("Dropped bad frame.");
- continue;
- }
-
- eid = ntoh32(*eid_p);
-
- /* pass onto mgmt queue */
- if (eid == MGMT_EID) {
- if ((size_t) n < MGMT_FRAME_SIZE) {
- log_warn("Dropped runt mgmt frame.");
- continue;
- }
-
- frame = malloc(sizeof(*frame));
- if (frame == NULL)
- continue;
-
- memcpy(frame->buf, buf, n);
- memcpy(&frame->r_saddr, &r_saddr, sizeof(r_saddr));
- frame->len = n;
-
- pthread_mutex_lock(&udp_data.mgmt_lock);
- list_add(&frame->next, &udp_data.mgmt_frames);
- pthread_cond_signal(&udp_data.mgmt_cond);
- pthread_mutex_unlock(&udp_data.mgmt_lock);
- continue;
- }
-
- n-= sizeof(eid);
-
- if (ipcp_spb_reserve(&spb, n))
- continue;
-
- head = ssm_pk_buff_head(spb);
- memcpy(head, data, n);
- if (np1_flow_write(eid, spb, NP1_GET_POOL(eid)) < 0)
- ipcp_spb_release(spb);
- }
-
- return (void *) 0;
-}
-
-static void cleanup_fqueue(void * fq)
-{
- fqueue_destroy((fqueue_t *) fq);
-}
-
-static void cleanup_spb(void * spb)
-{
- ipcp_spb_release((struct ssm_pk_buff *) spb);
-}
-
-static void * udp_ipcp_packet_writer(void * o)
-{
- fqueue_t * fq;
-
- fq = fqueue_create();
- if (fq == NULL)
- return (void *) -1;
-
- (void) o;
-
- ipcp_lock_to_core();
-
- pthread_cleanup_push(cleanup_fqueue, fq);
-
- while (true) {
- struct __SOCKADDR saddr;
- int eid;
- int fd;
- fevent(udp_data.np1_flows, fq, NULL);
- while ((fd = fqueue_next(fq)) >= 0) {
- struct ssm_pk_buff * spb;
- uint8_t * buf;
- uint16_t len;
-
- if (fqueue_type(fq) != FLOW_PKT)
- continue;
-
- if (np1_flow_read(fd, &spb, NP1_GET_POOL(fd))) {
- log_dbg("Bad read from fd %d.", fd);
- continue;
- }
-
- len = ssm_pk_buff_len(spb);
- if (len > IPCP_UDP_MAX_PACKET_SIZE) {
- log_dbg("Packet length exceeds MTU.");
- ipcp_spb_release(spb);
- continue;
- }
-
- buf = ssm_pk_buff_push(spb, OUR_HEADER_LEN);
- if (buf == NULL) {
- log_dbg("Failed to allocate header.");
- ipcp_spb_release(spb);
- continue;
- }
-
- pthread_rwlock_rdlock(&udp_data.flows_lock);
-
- eid = hton32(udp_data.fd_to_uf[fd].d_eid);
- saddr = udp_data.fd_to_uf[fd].r_saddr;
-
- pthread_rwlock_unlock(&udp_data.flows_lock);
-
- memcpy(buf, &eid, sizeof(eid));
-
- pthread_cleanup_push(cleanup_spb, spb);
-
- if (sendto(udp_data.s_fd, buf, len + OUR_HEADER_LEN,
- SENDTO_FLAGS,
- (const struct sockaddr *) &saddr,
- sizeof(saddr)) < 0)
- log_err("Failed to send packet.");
-
- pthread_cleanup_pop(true);
- }
- }
-
- pthread_cleanup_pop(true);
-
- return (void *) 1;
-}
-
-static bool is_addr_specified(const struct __ADDR * addr)
-{
-#ifdef BUILD_IPCP_UDP4
- return addr->s_addr != 0;
-#else
- return !IN6_IS_ADDR_UNSPECIFIED(addr);
-#endif
-}
-
-static int udp_ipcp_bootstrap(struct ipcp_config * conf)
-{
- char ipstr[__ADDRSTRLEN];
- char dnsstr[__ADDRSTRLEN];
- int i = 1;
-#ifdef BUILD_IPCP_UDP4
- struct udp4_config * udp;
- udp = &conf->udp4;
-#else
- struct udp6_config * udp;
- udp = &conf->udp6;
-#endif
-
- assert(conf != NULL);
- assert(conf->type == THIS_TYPE);
- assert(conf->layer_info.dir_hash_algo == (enum pol_dir_hash) HASH_MD5);
-
- if (__inet_ntop(&udp->ip_addr, ipstr) == NULL) {
- log_err("Failed to convert IP address.");
- return -1;
- }
-
- if (is_addr_specified(&udp->dns_addr)) {
- if (__inet_ntop(&udp->dns_addr, dnsstr) == NULL) {
- log_err("Failed to convert DNS address.");
- return -1;
- }
-#ifndef HAVE_DDNS
- log_warn("DNS disabled at compile time, address ignored.");
-#endif
- } else {
- strcpy(dnsstr, "not set");
- }
-
- /* UDP listen server */
- udp_data.s_fd = socket(__AF, SOCK_DGRAM, IPPROTO_UDP);
- if (udp_data.s_fd < 0) {
- log_err("Can't create socket: %s", strerror(errno));
- goto fail_socket;
- }
-
- memset((char *) &udp_data.s_saddr, 0, sizeof(udp_data.s_saddr));
-#ifdef BUILD_IPCP_UDP4
- udp_data.s_saddr.sin_family = AF_INET;
- udp_data.s_saddr.sin_addr = udp->ip_addr;
- udp_data.s_saddr.sin_port = htons(udp->port);
-#else
- udp_data.s_saddr.sin6_family = AF_INET6;
- udp_data.s_saddr.sin6_addr = udp->ip_addr;
- udp_data.s_saddr.sin6_port = htons(udp->port);
-#endif
- if (bind(udp_data.s_fd, SADDR, SADDR_SIZE) < 0) {
- log_err("Couldn't bind to %s:%d. %s.",
- ipstr, udp->port, strerror(errno));
- goto fail_bind;
- }
-
- udp_data.dns_addr = udp->dns_addr;
-
- if (pthread_create(&udp_data.mgmt_handler, NULL,
- udp_ipcp_mgmt_handler, NULL)) {
- log_err("Failed to create management thread.");
- goto fail_bind;
- }
-
- for (i = 0; i < IPCP_UDP_RD_THR; ++i) {
- if (pthread_create(&udp_data.packet_reader[i], NULL,
- udp_ipcp_packet_reader, NULL)) {
- log_err("Failed to create reader thread.");
- goto fail_packet_reader;
- }
- }
-
- for (i = 0; i < IPCP_UDP_WR_THR; ++i) {
- if (pthread_create(&udp_data.packet_writer[i], NULL,
- udp_ipcp_packet_writer, NULL)) {
- log_err("Failed to create writer thread.");
- goto fail_packet_writer;
- }
- }
-
- log_dbg("Bootstrapped " TYPE_STR " with pid %d.", getpid());
- log_dbg("Bound to IP address %s.", ipstr);
- log_dbg("Using port %u.", udp->port);
- if (is_addr_specified(&udp_data.dns_addr))
- log_dbg("DNS server address is %s.", dnsstr);
- else
- log_dbg("DNS server not in use.");
-
- return 0;
-
- fail_packet_writer:
- while (i-- > 0) {
- pthread_cancel(udp_data.packet_writer[i]);
- pthread_join(udp_data.packet_writer[i], NULL);
- }
- i = IPCP_UDP_RD_THR;
- fail_packet_reader:
- while (i-- > 0) {
- pthread_cancel(udp_data.packet_reader[i]);
- pthread_join(udp_data.packet_reader[i], NULL);
- }
- pthread_cancel(udp_data.mgmt_handler);
- pthread_join(udp_data.mgmt_handler, NULL);
- fail_bind:
- close(udp_data.s_fd);
- fail_socket:
- return -1;
-}
-
-#ifdef HAVE_DDNS
-/* FIXME: Dependency on nsupdate to be removed in the end */
-/* NOTE: Disgusted with this crap */
-static int ddns_send(char * cmd)
-{
- pid_t pid;
- int wstatus;
- int pipe_fd[2];
- char * argv[] = {NSUPDATE_EXEC, 0};
- char * envp[] = {0};
-
- if (pipe(pipe_fd)) {
- log_err("Failed to create pipe: %s.", strerror(errno));
- return -1;
- }
-
- pid = fork();
- if (pid == -1) {
- log_err("Failed to fork: %s.", strerror(errno));
- close(pipe_fd[0]);
- close(pipe_fd[1]);
- return -1;
- }
-
- if (pid == 0) {
- close(pipe_fd[1]);
- dup2(pipe_fd[0], 0);
- execve(argv[0], &argv[0], envp);
- log_err("Failed to execute: %s", strerror(errno));
- exit(1);
- }
-
- close(pipe_fd[0]);
-
- if (write(pipe_fd[1], cmd, strlen(cmd)) == -1) {
- log_err("Failed to communicate with nsupdate: %s.",
- strerror(errno));
- close(pipe_fd[1]);
- return -1;
- }
-
- waitpid(pid, &wstatus, 0);
- if (WIFEXITED(wstatus) && WEXITSTATUS(wstatus) == 0)
- log_dbg("Succesfully communicated with DNS server.");
- else
- log_err("Failed to register with DNS server.");
-
- close(pipe_fd[1]);
-
- return 0;
-}
-
-static struct __ADDR ddns_resolve(char * name,
- struct __ADDR dns_addr)
-{
- pid_t pid = -1;
- int wstatus;
- int pipe_fd[2];
- char dnsstr[__ADDRSTRLEN];
- char buf[IPCP_UDP_BUF_SIZE];
- ssize_t count = 0;
- char * substr = NULL;
- char * substr2 = NULL;
- char * addr_str = "Address:";
- struct __ADDR ip_addr = __ADDR_ANY_INIT;
-
- if (__inet_ntop(&dns_addr, dnsstr) == NULL)
- return ip_addr;
-
- if (pipe(pipe_fd)) {
- log_err("Failed to create pipe: %s.", strerror(errno));
- return ip_addr;
- }
-
- pid = fork();
- if (pid == -1) {
- log_err("Failed to fork: %s.", strerror(errno));
- close(pipe_fd[0]);
- close(pipe_fd[1]);
- return ip_addr;
- }
-
- if (pid == 0) {
- char * argv[] = {NSLOOKUP_EXEC, name, dnsstr, 0};
- char * envp[] = {0};
-
- close(pipe_fd[0]);
- dup2(pipe_fd[1], 1);
- execve(argv[0], &argv[0], envp);
- log_err("Failed to execute: %s", strerror(errno));
- exit(1);
- }
-
- close(pipe_fd[1]);
-
- count = read(pipe_fd[0], buf, IPCP_UDP_BUF_SIZE - 1);
- if (count <= 0) {
- log_err("Failed to communicate with nslookup.");
- close(pipe_fd[0]);
- return ip_addr;
- }
-
- close(pipe_fd[0]);
-
- waitpid(pid, &wstatus, 0);
- if (WIFEXITED(wstatus) && WEXITSTATUS(wstatus) == 0 &&
- count != IPCP_UDP_BUF_SIZE - 1)
- log_dbg("Succesfully communicated with nslookup.");
- else
- log_err("Failed to resolve DNS address.");
-
- buf[count] = '\0';
- substr = strtok(buf, "\n");
- while (substr != NULL) {
- substr2 = substr;
- substr = strtok(NULL, "\n");
- }
-
- if (substr2 == NULL || strstr(substr2, addr_str) == NULL) {
- log_err("Failed to resolve DNS address.");
- return ip_addr;
- }
-
- if (inet_pton(__AF, substr2 + strlen(addr_str) + 1, &ip_addr) != 1) {
- log_err("Failed to resolve DNS address.");
- assert(!is_addr_specified(&ip_addr));
- return ip_addr;
- }
-
- return ip_addr;
-}
-#endif
-
-static int udp_ipcp_reg(const uint8_t * hash)
-{
-#ifdef HAVE_DDNS
- char ipstr[__ADDRSTRLEN];
- char dnsstr[__ADDRSTRLEN];
- char cmd[1000];
- struct __ADDR dns_addr;
- struct __ADDR ip_addr;
-#endif
- char * hashstr;
-
- hashstr = malloc(ipcp_dir_hash_strlen() + 1);
- if (hashstr == NULL) {
- log_err("Failed to malloc hashstr.");
- return -1;
- }
-
- assert(hash);
-
- ipcp_hash_str(hashstr, hash);
-
- if (shim_data_reg_add_entry(udp_data.shim_data, hash)) {
- log_err("Failed to add " HASH_FMT32 " to local registry.",
- HASH_VAL32(hash));
- free(hashstr);
- return -1;
- }
-
-#ifdef HAVE_DDNS
- /* register application with DNS server */
-
- dns_addr = udp_data.dns_addr;
-
- if (is_addr_specified(&dns_addr)) {
-#ifdef BUILD_IPCP_UDP4
- ip_addr = udp_data.s_saddr.sin_addr;
-#else
- ip_addr = udp_data.s_saddr.sin6_addr;
-#endif
- if (__inet_ntop(&ip_addr, ipstr) == NULL) {
- log_err("Failed to convert IP address to string.");
- free(hashstr);
- return -1;
- }
-
- if (__inet_ntop(&dns_addr, dnsstr) == NULL) {
- log_err("Failed to convert DNS address to string.");
- free(hashstr);
- return -1;
- }
-
- sprintf(cmd, "server %s\nupdate add %s %d A %s\nsend\nquit\n",
- dnsstr, hashstr, DNS_TTL, ipstr);
-
- if (ddns_send(cmd)) {
- log_err("Failed to send DDNS message.");
- shim_data_reg_del_entry(udp_data.shim_data, hash);
- free(hashstr);
- return -1;
- }
- }
-#endif
- free(hashstr);
-
- return 0;
-}
-
-static int udp_ipcp_unreg(const uint8_t * hash)
-{
-#ifdef HAVE_DDNS
- char dnsstr[__ADDRSTRLEN];
- /* max DNS name length + max IP length + max command length */
- char cmd[100];
- struct __ADDR dns_addr;
-#endif
- char * hashstr;
-
- assert(hash);
-
- hashstr = malloc(ipcp_dir_hash_strlen() + 1);
- if (hashstr == NULL) {
- log_err("Failed to malloc hashstr.");
- return -1;
- }
-
- ipcp_hash_str(hashstr, hash);
-
-#ifdef HAVE_DDNS
- /* unregister application with DNS server */
-
- dns_addr = udp_data.dns_addr;
-
- if (is_addr_specified(&dns_addr)) {
- if (__inet_ntop(&dns_addr, dnsstr) == NULL) {
- log_err("Failed to convert DNS address to string.");
- free(hashstr);
- return -1;
- }
- sprintf(cmd, "server %s\nupdate delete %s A\nsend\nquit\n",
- dnsstr, hashstr);
-
- ddns_send(cmd);
- }
-#endif
-
- shim_data_reg_del_entry(udp_data.shim_data, hash);
-
- free(hashstr);
-
- return 0;
-}
-
-static int udp_ipcp_query(const uint8_t * hash)
-{
- struct addr addr = {};
- char * hashstr;
- struct addrinfo hints;
- struct addrinfo * ai;
-#ifdef HAVE_DDNS
- struct __ADDR dns_addr = __ADDR_ANY_INIT;
- struct __ADDR ip_addr = __ADDR_ANY_INIT;
-#endif
- assert(hash);
-
- hashstr = malloc(ipcp_dir_hash_strlen() + 1);
- if (hashstr == NULL) {
- log_err("Failed to malloc hashstr.");
- return -ENOMEM;
- }
-
- ipcp_hash_str(hashstr, hash);
-
- if (shim_data_dir_has(udp_data.shim_data, hash)) {
- free(hashstr);
- return 0;
- }
-
-#ifdef HAVE_DDNS
- dns_addr = udp_data.dns_addr;
-
- if (is_addr_specified(&dns_addr)) {
- ip_addr = ddns_resolve(hashstr, dns_addr);
- if (!is_addr_specified(&ip_addr)) {
- log_err("Could not resolve %s.", hashstr);
- free(hashstr);
- return -1;
- }
- } else {
-#endif
- memset(&hints, 0, sizeof(hints));
-
- hints.ai_family = __AF;
- if (getaddrinfo(hashstr, NULL, &hints, &ai) != 0) {
- log_err("Could not resolve %s: %s.", hashstr,
- gai_strerror(errno));
- free(hashstr);
- return -1;
- }
-
- if (ai->ai_family != __AF) {
- log_err("Wrong addres family for %s.", hashstr);
- freeaddrinfo(ai);
- free(hashstr);
- return -1;
- }
-
- #ifdef BUILD_IPCP_UDP4
- addr.ip4 = ((struct sockaddr_in *) (ai->ai_addr))->sin_addr;
- #else
- addr.ip6 = ((struct sockaddr_in6 *) (ai->ai_addr))->sin6_addr;
- #endif
- freeaddrinfo(ai);
-#ifdef HAVE_DDNS
- }
-#endif
- if (shim_data_dir_add_entry(udp_data.shim_data, hash, addr)) {
- log_err("Failed to add directory entry.");
- free(hashstr);
- return -1;
- }
-
- free(hashstr);
-
- return 0;
-}
-
-static int udp_ipcp_flow_alloc(int fd,
- const uint8_t * dst,
- qosspec_t qs,
- const buffer_t * data)
-{
- struct __SOCKADDR r_saddr; /* Server address */
- struct __ADDR ip_addr;
- struct addr addr;
- char ipstr[__ADDRSTRLEN];
-
- (void) qs;
-
- assert(dst);
-
- if (!shim_data_dir_has(udp_data.shim_data, dst)) {
- log_err("Could not resolve destination.");
- return -1;
- }
-
- addr = shim_data_dir_get_addr(udp_data.shim_data, dst);
-#ifdef BUILD_IPCP_UDP4
- ip_addr = addr.ip4;
-#else
- ip_addr = addr.ip6;
-#endif
- if (__inet_ntop(&ip_addr, ipstr) == NULL) {
- log_err("Could not convert IP address.");
- return -1;
- }
-
- log_dbg("Destination " HASH_FMT32 " resolved at IP %s.",
- HASH_VAL32(dst), ipstr);
-
- memset((char *) &r_saddr, 0, sizeof(r_saddr));
-#ifdef BUILD_IPCP_UDP4
- r_saddr.sin_family = AF_INET;
- r_saddr.sin_addr = addr.ip4;
- r_saddr.sin_port = udp_data.s_saddr.sin_port;
-#else
- r_saddr.sin6_family = AF_INET6;
- r_saddr.sin6_addr = addr.ip6;
- r_saddr.sin6_port = udp_data.s_saddr.sin6_port;
-#endif
-
- if (udp_ipcp_port_alloc(&r_saddr, fd, dst, qs, data) < 0) {
- log_err("Could not allocate port.");
- return -1;
- }
-
- pthread_rwlock_wrlock(&udp_data.flows_lock);
-
- udp_data.fd_to_uf[fd].d_eid = -1;
- udp_data.fd_to_uf[fd].r_saddr = r_saddr;
-
- pthread_rwlock_unlock(&udp_data.flows_lock);
-
- fset_add(udp_data.np1_flows, fd);
-
- return 0;
-}
-
-static int udp_ipcp_flow_alloc_resp(int fd,
- int resp,
- const buffer_t * data)
-{
- struct __SOCKADDR saddr;
- int d_eid;
-
- if (ipcp_wait_flow_resp(fd) < 0) {
- log_err("Failed to wait for flow response.");
- return -1;
- }
-
- pthread_rwlock_rdlock(&udp_data.flows_lock);
-
- saddr = udp_data.fd_to_uf[fd].r_saddr;
- d_eid = udp_data.fd_to_uf[fd].d_eid;
-
- pthread_rwlock_unlock(&udp_data.flows_lock);
-
- if (udp_ipcp_port_alloc_resp(&saddr, d_eid, fd, resp, data) < 0) {
- fset_del(udp_data.np1_flows, fd);
- log_err("Failed to respond to flow request.");
- return -1;
- }
-
- fset_add(udp_data.np1_flows, fd);
-
- return 0;
-}
-
-static int udp_ipcp_flow_dealloc(int fd)
-{
- ipcp_flow_fini(fd);
-
- fset_del(udp_data.np1_flows, fd);
-
- pthread_rwlock_wrlock(&udp_data.flows_lock);
-
- udp_data.fd_to_uf[fd].d_eid = -1;
- memset(&udp_data.fd_to_uf[fd].r_saddr, 0, SADDR_SIZE);
-
- pthread_rwlock_unlock(&udp_data.flows_lock);
-
- ipcp_flow_dealloc(fd);
-
- return 0;
-}
-
-static struct ipcp_ops udp_ops = {
- .ipcp_bootstrap = udp_ipcp_bootstrap,
- .ipcp_enroll = NULL,
- .ipcp_connect = NULL,
- .ipcp_disconnect = NULL,
- .ipcp_reg = udp_ipcp_reg,
- .ipcp_unreg = udp_ipcp_unreg,
- .ipcp_query = udp_ipcp_query,
- .ipcp_flow_alloc = udp_ipcp_flow_alloc,
- .ipcp_flow_join = NULL,
- .ipcp_flow_alloc_resp = udp_ipcp_flow_alloc_resp,
- .ipcp_flow_dealloc = udp_ipcp_flow_dealloc,
- .ipcp_flow_update = udp_ipcp_flow_update
-};
-
-int main(int argc,
- char * argv[])
-{
- int i;
-
-
- if (udp_data_init() < 0) {
- log_err("Failed to init udp data.");
- goto fail_data_init;
- }
-
- if (ipcp_init(argc, argv, &udp_ops, THIS_TYPE) < 0) {
- log_err("Failed to initialize IPCP.");
- goto fail_init;
- }
-
- if (ipcp_start() < 0) {
- log_err("Failed to start IPCP.");
- goto fail_start;
- }
-
- ipcp_sigwait();
-
- if (ipcp_get_state() == IPCP_SHUTDOWN) {
- for (i = 0; i < IPCP_UDP_WR_THR; ++i)
- pthread_cancel(udp_data.packet_writer[i]);
- for (i = 0; i < IPCP_UDP_RD_THR; ++i)
- pthread_cancel(udp_data.packet_reader[i]);
- pthread_cancel(udp_data.mgmt_handler);
-
- for (i = 0; i < IPCP_UDP_WR_THR; ++i)
- pthread_join(udp_data.packet_writer[i], NULL);
- for (i = 0; i < IPCP_UDP_RD_THR; ++i)
- pthread_join(udp_data.packet_reader[i], NULL);
- pthread_join(udp_data.mgmt_handler, NULL);
- close(udp_data.s_fd);
- }
-
- ipcp_stop();
-
- ipcp_fini();
-
- udp_data_fini();
-
- exit(EXIT_SUCCESS);
-
- fail_start:
- ipcp_fini();
- fail_init:
- udp_data_fini();
- fail_data_init:
- exit(EXIT_FAILURE);
-}
diff --git a/src/ipcpd/udp/udp4.c b/src/ipcpd/udp/udp4.c
deleted file mode 100644
index ff57bc09..00000000
--- a/src/ipcpd/udp/udp4.c
+++ /dev/null
@@ -1,42 +0,0 @@
-/*
- * Ouroboros - Copyright (C) 2016 - 2026
- *
- * IPC process over UDP/IPv4
- *
- * Dimitri Staessens <dimitri@ouroboros.rocks>
- * Sander Vrijders <sander@ouroboros.rocks>
- *
- * This program is free software; you can redistribute it and/or modify
- * it under the terms of the GNU General Public License version 2 as
- * published by the Free Software Foundation.
- *
- * This program is distributed in the hope that it will be useful,
- * but WITHOUT ANY WARRANTY; without even the implied warranty of
- * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
- * GNU General Public License for more details.
- *
- * You should have received a copy of the GNU General Public License
- * along with this program; if not, write to the Free Software
- * Foundation, Inc., http://www.fsf.org/about/contact/.
- */
-
-#if defined(__linux__) || defined(__CYGWIN__)
-#define _DEFAULT_SOURCE
-#else
-#define _POSIX_C_SOURCE 200112L
-#endif
-
-#include <ouroboros/ipcp-dev.h>
-
-#define BUILD_IPCP_UDP4
-#define THIS_TYPE IPCP_UDP4
-#define TYPE_STR "IPCP over UDP/IPv4"
-#define OUROBOROS_PREFIX "ipcpd/udp4"
-#define IPCP_UDP_MAX_PACKET_SIZE 8980
-#define __AF AF_INET
-#define __ADDRSTRLEN INET_ADDRSTRLEN
-#define __SOCKADDR sockaddr_in
-#define __ADDR in_addr
-#define __ADDR_ANY_INIT { .s_addr = INADDR_ANY }
-
-#include "udp.c"
diff --git a/src/ipcpd/udp/udp6.c b/src/ipcpd/udp/udp6.c
deleted file mode 100644
index 2ceb95f0..00000000
--- a/src/ipcpd/udp/udp6.c
+++ /dev/null
@@ -1,42 +0,0 @@
-/*
- * Ouroboros - Copyright (C) 2016 - 2026
- *
- * IPC process over UDP/IPv6
- *
- * Dimitri Staessens <dimitri@ouroboros.rocks>
- * Sander Vrijders <sander@ouroboros.rocks>
- *
- * This program is free software; you can redistribute it and/or modify
- * it under the terms of the GNU General Public License version 2 as
- * published by the Free Software Foundation.
- *
- * This program is distributed in the hope that it will be useful,
- * but WITHOUT ANY WARRANTY; without even the implied warranty of
- * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
- * GNU General Public License for more details.
- *
- * You should have received a copy of the GNU General Public License
- * along with this program; if not, write to the Free Software
- * Foundation, Inc., http://www.fsf.org/about/contact/.
- */
-
-#if defined(__linux__) || defined(__CYGWIN__)
-#define _DEFAULT_SOURCE
-#else
-#define _POSIX_C_SOURCE 200112L
-#endif
-
-#include <ouroboros/ipcp-dev.h>
-
-#define BUILD_IPCP_UDP6
-#define THIS_TYPE IPCP_UDP6
-#define TYPE_STR "IPCP over UDP/IPv6"
-#define OUROBOROS_PREFIX "ipcpd/udp6"
-#define IPCP_UDP_MAX_PACKET_SIZE 8952
-#define __AF AF_INET6
-#define __ADDRSTRLEN INET6_ADDRSTRLEN
-#define __SOCKADDR sockaddr_in6
-#define __ADDR in6_addr
-#define __ADDR_ANY_INIT IN6ADDR_ANY_INIT
-
-#include "udp.c"
diff --git a/src/ipcpd/unicast/fa.c b/src/ipcpd/unicast/fa.c
index 596b101b..744eb316 100644
--- a/src/ipcpd/unicast/fa.c
+++ b/src/ipcpd/unicast/fa.c
@@ -130,8 +130,8 @@ struct fa_flow {
size_t u_snd; /* Flow updates sent */
size_t u_rcv; /* Flow updates received */
#endif
- uint64_t s_eid; /* Local endpoint id */
- uint64_t r_eid; /* Remote endpoint id */
+ uint64_t s_eid; /* Local PoA id */
+ uint64_t r_eid; /* Remote PoA id */
uint64_t r_addr; /* Remote address */
void * ctx; /* Congestion avoidance context */
uint64_t fair; /* SFQ virtual finish tag (bytes) */
@@ -208,8 +208,8 @@ static int fa_rib_read(const char * path,
sprintf(buf,
"Flow established at: %20s\n"
"Remote address: %20s\n"
- "Local endpoint ID: %20s\n"
- "Remote endpoint ID: %20s\n"
+ "Local PoA ID: %20s\n"
+ "Remote PoA ID: %20s\n"
"Sent (packets): %20zu\n"
"Sent (bytes): %20zu\n"
"Send failed (packets): %20zu\n"
diff --git a/src/ipcpd/unicast/main.c b/src/ipcpd/unicast/main.c
index 86cb1f06..3ee53401 100644
--- a/src/ipcpd/unicast/main.c
+++ b/src/ipcpd/unicast/main.c
@@ -35,6 +35,7 @@
#include <ouroboros/ipcp-dev.h>
#include <ouroboros/logs.h>
#include <ouroboros/notifier.h>
+#include <ouroboros/qos.h>
#include <ouroboros/random.h>
#include <ouroboros/rib.h>
#include <ouroboros/time.h>
@@ -175,12 +176,14 @@ static void stop_components(void)
ipcp_set_state(IPCP_BOOT);
}
-static int unicast_ipcp_enroll(const char * dst,
- struct layer_info * info)
+static int unicast_ipcp_enroll(const char * dst,
+ const struct poa_addr * addr,
+ struct layer_info * info)
{
struct ipcp_config * conf;
struct conn conn;
uint8_t id[ENROLL_ID_LEN];
+ qosspec_t qs = qos_msg;
if (random_buffer(id, ENROLL_ID_LEN) < 0) {
log_err("Failed to generate enrollment ID.");
@@ -189,7 +192,7 @@ static int unicast_ipcp_enroll(const char * dst,
log_info_id(id, "Requesting enrollment.");
- if (connmgr_alloc(COMPID_ENROLL, dst, NULL, &conn) < 0) {
+ if (connmgr_alloc(COMPID_ENROLL, dst, &qs, addr, &conn) < 0) {
log_err_id(id, "Failed to get connection.");
goto fail_id;
}
diff --git a/src/irmd/config.h.in b/src/irmd/config.h.in
index 53f80ca4..2440e180 100644
--- a/src/irmd/config.h.in
+++ b/src/irmd/config.h.in
@@ -21,10 +21,6 @@
*/
-#define IPCP_UDP4_EXEC "@IPCP_UDP4_TARGET@"
-#define IPCP_UDP6_EXEC "@IPCP_UDP6_TARGET@"
-#define IPCP_ETH_LLC_EXEC "@IPCP_ETH_LLC_TARGET@"
-#define IPCP_ETH_DIX_EXEC "@IPCP_ETH_DIX_TARGET@"
#define IPCP_UNICAST_EXEC "@IPCP_UNICAST_TARGET@"
#define IPCP_BROADCAST_EXEC "@IPCP_BROADCAST_TARGET@"
#define IPCP_LOCAL_EXEC "@IPCP_LOCAL_TARGET@"
@@ -75,6 +71,8 @@
#define OUROBOROS_CLI_CRT_DIR "@OUROBOROS_CLI_CRT_DIR@"
#define OUROBOROS_CHAIN_DIR "@OUROBOROS_UNTRUSTED_DIR@"
+/* Endpoint peers are keyed on their address, not on a service name. */
+
#define IRMD_PKILL_TIMEOUT @IRMD_PKILL_TIMEOUT@
#cmakedefine DISABLE_DIRECT_IPC
diff --git a/src/irmd/configfile.c b/src/irmd/configfile.c
index e2e1e554..de02242f 100644
--- a/src/irmd/configfile.c
+++ b/src/irmd/configfile.c
@@ -43,6 +43,7 @@
#include <assert.h>
#include <errno.h>
#include <inttypes.h>
+#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <toml.h>
@@ -92,171 +93,173 @@ static int toml_local(toml_table_t * table,
return toml_hash(table, &conf->layer_info);
}
-static int toml_eth_dev(toml_table_t * table,
- struct eth_config * conf)
-{
- toml_datum_t dev;
- dev = toml_string_in(table, "dev");
- if (!dev.ok) {
- log_err("Missing device.");
- return -1;
- }
+/* Defined with the peer helpers below; shared grammar for both paths. */
+static int toml_udp_str(const char * str,
+ char * host,
+ int * port);
- if (strlen(dev.u.s) > DEV_NAME_SIZE) {
- log_err("Device name too long: %s", dev.u.s);
- free(dev.u.s);
+/* Attach the PoAs an IPCP is given; one call per PoA. */
+static int poa_eth_set(struct poa_spec * poa,
+ const char * dev,
+ uint16_t ethertype)
+{
+ if (strlen(dev) > DEV_NAME_SIZE) {
+ log_err("Invalid device name %s.", dev);
return -1;
}
- strcpy(conf->dev, dev.u.s);
- free(dev.u.s);
+ memset(poa, 0, sizeof(*poa));
+
+ poa->type = POA_ETH;
+ poa->eth.ethertype = ethertype;
+
+ strcpy(poa->eth.dev, dev);
return 0;
}
-static int toml_eth_llc(toml_table_t * table,
- struct ipcp_config * conf)
+static int toml_poa_eth(toml_table_t * tbl,
+ struct poa_spec * poa)
{
- *conf = eth_llc_default_conf;
-
- if (toml_hash(table, &conf->layer_info) < 0)
- return -1;
-
- return toml_eth_dev(table, &conf->eth);
-}
+ toml_datum_t dev;
+ toml_datum_t et;
+ uint16_t ethertype = POA_ETHERTYPE;
+ int ret = -1;
+ dev = toml_string_in(tbl, "dev");
+ if (!dev.ok) {
+ log_err("An eth PoA needs a device.");
+ goto fail;
+ }
-static int toml_ethertype(toml_table_t * table,
- struct eth_config * conf)
-{
- toml_datum_t ethertype;
+ et = toml_int_in(tbl, "ethertype");
+ if (et.ok) {
+ if (et.u.i < 0 || et.u.i > 0xFFFF) {
+ log_err("Invalid ethertype.");
+ goto fail;
+ }
- ethertype = toml_int_in(table, "ethertype");
- if (ethertype.ok)
- conf->ethertype = ethertype.u.i;
+ ethertype = et.u.i;
+ }
- if (conf->ethertype < 0x0600 || conf->ethertype == 0xFFFF)
- return -1;
+ ret = poa_eth_set(poa, dev.u.s, ethertype);
+ fail:
+ if (dev.ok)
+ free(dev.u.s);
- return 0;
+ return ret;
}
-static int toml_eth_dix(toml_table_t * table,
- struct ipcp_config * conf)
+/* A PoA is bound locally: only literal addresses are accepted here. */
+static int toml_poa_udp(const char * str,
+ struct poa_spec * poa)
{
- *conf = eth_dix_default_conf;
+ char host[POA_HOST_STRLEN + 1];
+ int port;
- if (toml_hash(table, &conf->layer_info) < 0)
+ if (toml_udp_str(str, host, &port) < 0)
return -1;
- if (toml_eth_dev(table, &conf->eth) < 0)
- return -1;
+ memset(poa, 0, sizeof(*poa));
- if (toml_ethertype(table, &conf->eth) < 0) {
- log_err("Ethertype not in valid range.");
- return -1;
+ if (inet_pton(AF_INET, host, &poa->udp4.ip_addr.s_addr) == 1) {
+ poa->type = POA_UDP4;
+ poa->udp4.port = port;
+ return 0;
}
- return 0;
-}
-
-static int toml_udp4(toml_table_t * table,
- struct ipcp_config * conf)
-{
- struct udp4_config * udp4;
- toml_datum_t ip;
- toml_datum_t port;
- toml_datum_t dns;
-
- *conf = udp4_default_conf;
- udp4 = &conf->udp4;
-
- ip = toml_string_in(table, "ip");
- if (!ip.ok) {
- log_err("No IP address specified!");
- goto fail_ip;
+ if (inet_pton(AF_INET6, host, &poa->udp6.ip_addr) == 1) {
+ poa->type = POA_UDP6;
+ poa->udp6.port = port;
+ return 0;
}
- if (inet_pton (AF_INET, ip.u.s, &udp4->ip_addr.s_addr) != 1) {
- log_err("Failed to parse IPv4 address %s.", ip.u.s);
- goto fail_addr;
- }
+ log_err("Invalid IP address %s.", host);
- port = toml_int_in(table, "port");
- if (port.ok)
- udp4->port = port.u.i;
+ return -1;
+}
- dns = toml_string_in(table, "dns");
- if (dns.ok) {
- if (inet_pton(AF_INET, dns.u.s, &udp4->dns_addr.s_addr) < 0) {
- log_err("Failed to parse DNS address %s.", ip.u.s);
- goto fail_dns;
+/* Attaches every PoA in the "udp" or "eth" array; string or table. */
+static int toml_poa_array(toml_table_t * table,
+ pid_t pid,
+ const char * key)
+{
+ struct poa_spec poa;
+ toml_array_t * arr;
+ int n;
+ int i;
+ bool is_eth;
+
+ arr = toml_array_in(table, key);
+ if (arr == NULL) {
+ /* A table here would otherwise attach nothing, silently. */
+ if (toml_table_in(table, key) != NULL) {
+ log_err("A %s PoA is an array: %s = [...] "
+ "or [[%s]].", key, key, key);
+ return -1;
}
- free(dns.u.s);
+ return 0;
}
- free(ip.u.s);
+ is_eth = strcmp(key, "eth") == 0;
- return 0;
+ n = toml_array_nelem(arr);
- fail_dns:
- free(dns.u.s);
- fail_addr:
- free(ip.u.s);
- fail_ip:
- return -1;
-}
+ for (i = 0; i < n; i++) {
+ toml_datum_t s;
+ int ret;
-static int toml_udp6(toml_table_t * table,
- struct ipcp_config * conf)
-{
- struct in6_addr ip6;
- struct in6_addr dns6;
- toml_datum_t ip;
- toml_datum_t port;
- toml_datum_t dns;
-
- *conf = udp6_default_conf;
- ip6 = conf->udp6.ip_addr;
- dns6 = conf->udp6.dns_addr;
-
- ip = toml_string_in(table, "ip");
- if (!ip.ok) {
- log_err("No IP address specified!");
- goto fail_ip;
- }
+ s = toml_string_at(arr, i);
+ if (s.ok) {
+ if (is_eth)
+ ret = poa_eth_set(&poa, s.u.s,
+ POA_ETHERTYPE);
+ else
+ ret = toml_poa_udp(s.u.s, &poa);
- if (inet_pton (AF_INET6, ip.u.s, &ip6.s6_addr) != 1) {
- log_err("Failed to parse IPv4 address %s.", ip.u.s);
- goto fail_addr;
- }
+ free(s.u.s);
+ } else {
+ toml_table_t * d = toml_table_at(arr, i);
- port = toml_int_in(table, "port");
- if (port.ok)
- conf->udp6.port = port.u.i;
+ if (d == NULL) {
+ log_err("Invalid %s PoA entry.", key);
+ return -1;
+ }
+
+ if (is_eth) {
+ ret = toml_poa_eth(d, &poa);
+ } else {
+ toml_datum_t addr = toml_string_in(d, "addr");
+
+ if (!addr.ok) {
+ log_err("No addr for udp PoA.");
+ return -1;
+ }
- dns = toml_string_in(table, "dns");
- if (dns.ok) {
- if (inet_pton(AF_INET6, dns.u.s, &dns6.s6_addr) < 0) {
- log_err("Failed to parse DNS address %s.", ip.u.s);
- goto fail_dns;
+ ret = toml_poa_udp(addr.u.s, &poa);
+ free(addr.u.s);
+ }
}
- free(dns.u.s);
- }
+ if (ret < 0)
+ return -1;
- free(ip.u.s);
+ if (attach_ipcp(pid, &poa, true) < 0)
+ return -1;
+ }
return 0;
+}
- fail_dns:
- free(dns.u.s);
- fail_addr:
- free(ip.u.s);
- fail_ip:
- return -1;
+static int toml_attach(toml_table_t * table,
+ pid_t pid)
+{
+ if (toml_poa_array(table, pid, "udp") < 0)
+ return -1;
+
+ return toml_poa_array(table, pid, "eth");
}
static int toml_broadcast(toml_table_t * table,
@@ -265,15 +268,13 @@ static int toml_broadcast(toml_table_t * table,
(void) table;
(void) conf;
- /* Nothing to do here. */
-
return 0;
}
#define BETWEEN(a, b, c) ((a) >= (b) && (a) <= (c))
#define DHT(conf, x) (conf)->dht.params.x
static int toml_dir(toml_table_t * table,
- struct dir_config * conf)
+ struct dir_config * conf)
{
toml_datum_t dir;
toml_datum_t alpha;
@@ -594,6 +595,207 @@ static int toml_register(toml_table_t * table,
return ret;
}
+/* Fills in a peer's eth PoA; dst MAC stays zero, the resolve marker. */
+static int toml_peer_eth(toml_table_t * tbl,
+ struct poa_addr * addr)
+{
+ toml_datum_t dev;
+ toml_datum_t et;
+ int ret = -1;
+
+ dev = toml_string_in(tbl, "dev");
+
+ memset(addr, 0, sizeof(*addr));
+
+ addr->type = POA_ETH;
+ addr->eth.src.ethertype = POA_ETHERTYPE;
+ addr->eth.dst.ethertype = POA_ETHERTYPE;
+
+ et = toml_int_in(tbl, "ethertype");
+ if (et.ok) {
+ if (et.u.i < 0 || et.u.i > 0xFFFF) {
+ log_err("Invalid ethertype.");
+ goto fail;
+ }
+
+ addr->eth.src.ethertype = et.u.i;
+ addr->eth.dst.ethertype = et.u.i;
+ }
+
+ if (dev.ok) {
+ if (strlen(dev.u.s) > DEV_NAME_SIZE) {
+ log_err("Invalid device name %s.", dev.u.s);
+ goto fail;
+ }
+
+ strcpy(addr->eth.src.dev, dev.u.s);
+ }
+
+ ret = 0;
+ fail:
+ if (dev.ok)
+ free(dev.u.s);
+
+ return ret;
+}
+
+/* Same grammar as the CLI's udp keyword; see irm_utils.c parse_udp_str. */
+static int toml_udp_str(const char * str,
+ char * host,
+ int * port)
+{
+ struct in6_addr v6;
+ char buf[POA_HOST_STRLEN + 1];
+ char * p;
+ char * end;
+ long n;
+
+ *port = POA_UDP_PORT;
+
+ if (strlen(str) > POA_HOST_STRLEN)
+ goto fail;
+
+ strcpy(buf, str);
+
+ if (buf[0] == '[') {
+ p = strchr(buf, ']');
+ if (p == NULL)
+ goto fail;
+
+ *p++ = '\0';
+
+ strcpy(host, buf + 1);
+
+ if (*p == '\0')
+ return 0;
+
+ if (*p != ':')
+ goto fail;
+
+ ++p;
+ } else if (inet_pton(AF_INET6, buf, &v6) == 1) {
+ strcpy(host, buf);
+ return 0;
+ } else {
+ p = strrchr(buf, ':');
+ if (p == NULL) {
+ strcpy(host, buf);
+ return 0;
+ }
+
+ *p++ = '\0';
+
+ strcpy(host, buf);
+ }
+
+ n = strtol(p, &end, 10);
+ if (*p == '\0' || *end != '\0' || n < 1 || n > 65535)
+ goto fail;
+
+ *port = (int) n;
+
+ return 0;
+ fail:
+ log_err("Invalid UDP address: %s.", str);
+ return -1;
+}
+
+/* Fills in a peer's udp PoA; an unresolved name is left for the IRMd. */
+static int toml_peer_udp(const char * str,
+ struct poa_addr * addr)
+{
+ char host[POA_HOST_STRLEN + 1];
+ int port;
+
+ if (toml_udp_str(str, host, &port) < 0)
+ return -1;
+
+ memset(addr, 0, sizeof(*addr));
+
+ if (inet_pton(AF_INET, host, &addr->udp4.ip_addr) == 1) {
+ addr->type = POA_UDP4;
+ addr->udp4.port = port;
+ return 0;
+ }
+
+ if (inet_pton(AF_INET6, host, &addr->udp6.ip_addr) == 1) {
+ addr->type = POA_UDP6;
+ addr->udp6.port = port;
+ return 0;
+ }
+
+ addr->type = POA_UDP;
+ addr->udp4.port = port;
+
+ strcpy(addr->hostname, host);
+
+ return 0;
+}
+
+/*
+ * An entry is a name, or a table naming a PoA to reach it over.
+ * On entry, *paddr already points at the caller's struct poa_addr to
+ * fill in. Set to NULL wherever there is no PoA to dial: a bare
+ * dst-only table (recursive lookup) or any parse failure.
+ */
+static int toml_peer(toml_table_t * tbl,
+ char * dst,
+ struct poa_addr ** paddr)
+{
+ struct poa_addr * addr = *paddr;
+ toml_table_t * eth;
+ toml_datum_t name;
+ toml_datum_t udp;
+ int ret = -1;
+
+ name = toml_string_in(tbl, "dst");
+ if (!name.ok) {
+ log_err("PoA table entry has no dst.");
+
+ *paddr = NULL;
+ return -1;
+ }
+
+ if (strlen(name.u.s) > LAYER_NAME_SIZE) {
+ log_err("Destination name too long: %s.", name.u.s);
+ free(name.u.s);
+
+ *paddr = NULL;
+ return -1;
+ }
+
+ strcpy(dst, name.u.s);
+ free(name.u.s);
+
+ eth = toml_table_in(tbl, "eth");
+ udp = toml_string_in(tbl, "udp");
+ if (eth != NULL && udp.ok) {
+ log_err("A PoA is eth or udp, not both.");
+
+ *paddr = NULL;
+ goto fail;
+ }
+
+ if (eth == NULL && !udp.ok) {
+ *paddr = NULL;
+ ret = 0;
+ goto fail;
+ }
+
+ if (eth != NULL)
+ ret = toml_peer_eth(eth, addr);
+ else
+ ret = toml_peer_udp(udp.u.s, addr);
+
+ if (ret < 0)
+ *paddr = NULL;
+ fail:
+ if (udp.ok)
+ free(udp.u.s);
+
+ return ret;
+}
+
static int toml_connect(toml_table_t * table,
pid_t pid)
{
@@ -605,21 +807,40 @@ static int toml_connect(toml_table_t * table,
if (conn == NULL)
return 0;
- for (i=0; ret == 0; i++) {
- toml_datum_t dst;
- qosspec_t qs = qos_raw;
+ for (i = 0; ret == 0; i++) {
+ toml_datum_t dst;
+ toml_table_t * tbl;
+ struct poa_addr addr;
+ struct poa_addr * paddr = &addr;
+ char buf[LAYER_NAME_SIZE + 1];
+ const char * d;
+ qosspec_t qs = qos_raw;
dst = toml_string_at(conn, i);
- if (!dst.ok)
- break;
+ if (dst.ok) {
+ d = dst.u.s;
+ paddr = NULL;
+ } else {
+ tbl = toml_table_at(conn, i);
+ if (tbl == NULL)
+ break;
+
+ ret = toml_peer(tbl, buf, &paddr);
+ if (ret < 0)
+ break;
+
+ d = buf;
+ }
+
+ log_dbg("Connecting %d to %s", pid, d);
- log_dbg("Connecting %d to %s", pid, dst.u.s);
+ ret = connect_ipcp_resolve(pid, d, MGMT_COMP, qs, paddr);
- ret = connect_ipcp(pid, dst.u.s, MGMT_COMP, qs);
if (ret == 0)
- ret = connect_ipcp(pid, dst.u.s, DT_COMP, qs);
+ ret = connect_ipcp_resolve(pid, d, DT_COMP, qs, paddr);
- free(dst.u.s);
+ if (dst.ok)
+ free(dst.u.s);
}
return ret;
@@ -629,9 +850,11 @@ static int toml_ipcp(toml_table_t * table,
struct ipcp_info * info,
struct ipcp_config * conf)
{
- toml_datum_t bootstrap;
- toml_datum_t enrol;
- int ret;
+ toml_datum_t bootstrap;
+ toml_datum_t enrol;
+ toml_table_t * enrol_tbl;
+ bool have_enrol;
+ int ret;
log_dbg("Found IPCP %s in configuration file.", info->name);
@@ -643,21 +866,49 @@ static int toml_ipcp(toml_table_t * table,
bootstrap = toml_string_in(table, "bootstrap");
enrol = toml_string_in(table, "enrol");
- if (bootstrap.ok && enrol.ok) {
+ enrol_tbl = enrol.ok ? NULL : toml_table_in(table, "enrol");
+
+ have_enrol = enrol.ok || enrol_tbl != NULL;
+ if (bootstrap.ok && have_enrol) {
log_err("Ignoring bootstrap for IPCP %s.", info->name);
free(bootstrap.u.s);
bootstrap.ok = false;
}
- if (!bootstrap.ok && !enrol.ok) {
+ if (!bootstrap.ok && !have_enrol) {
log_dbg("Nothing more to do for %s.", info->name);
return 0;
}
- if (enrol.ok) {
+ /* Endpoints come first: enrolment reaches the peer over one. */
+ if (toml_attach(table, info->pid) < 0) {
+ log_err("Failed to attach PoAs for %s.", info->name);
+ return -1;
+ }
+
+ if (have_enrol) {
struct layer_info layer;
- ret = enroll_ipcp(info->pid, enrol.u.s);
- free(enrol.u.s);
+ struct poa_addr addr;
+ struct poa_addr * paddr = &addr;
+ char buf[LAYER_NAME_SIZE + 1];
+ const char * dst;
+
+ if (enrol.ok) {
+ dst = enrol.u.s;
+ paddr = NULL;
+ } else {
+ if (toml_peer(enrol_tbl, buf, &paddr) < 0) {
+ log_err("Invalid enrol table for %s.",
+ info->name);
+ return -1;
+ }
+ dst = buf;
+ }
+
+ ret = enroll_ipcp_resolve(info->pid, dst, paddr);
+
+ if (enrol.ok)
+ free(enrol.u.s);
if (ret < 0) {
log_err("Failed to enrol %s.", info->name);
return -1;
@@ -694,18 +945,6 @@ static int toml_ipcp(toml_table_t * table,
case IPCP_LOCAL:
ret = toml_local(table, conf);
break;
- case IPCP_ETH_DIX:
- ret = toml_eth_dix(table, conf);
- break;
- case IPCP_ETH_LLC:
- ret = toml_eth_llc(table, conf);
- break;
- case IPCP_UDP4:
- ret = toml_udp4(table, conf);
- break;
- case IPCP_UDP6:
- ret = toml_udp6(table, conf);
- break;
case IPCP_BROADCAST:
ret = toml_broadcast(table, conf);
break;
@@ -1031,14 +1270,6 @@ static int toml_toplevel(toml_table_t * table,
return toml_name_list(subtable);
else if (strcmp(key, "local") == 0)
return toml_ipcp_list(subtable, IPCP_LOCAL);
- else if (strcmp(key, "eth-dix") == 0)
- return toml_ipcp_list(subtable, IPCP_ETH_DIX);
- else if (strcmp(key, "eth-llc") == 0)
- return toml_ipcp_list(subtable, IPCP_ETH_LLC);
- else if (strcmp(key, "udp4") == 0)
- return toml_ipcp_list(subtable, IPCP_UDP4);
- else if (strcmp(key, "udp6") == 0)
- return toml_ipcp_list(subtable, IPCP_UDP6);
else if (strcmp(key, "broadcast") == 0)
return toml_ipcp_list(subtable, IPCP_BROADCAST);
else if (strcmp(key, "unicast") == 0)
diff --git a/src/irmd/ipcp.c b/src/irmd/ipcp.c
index 7eccfc80..cd662221 100644
--- a/src/irmd/ipcp.c
+++ b/src/irmd/ipcp.c
@@ -34,6 +34,7 @@
#include <ouroboros/utils.h>
#include "ipcp.h"
+#include "reg/reg.h"
#include <fcntl.h>
#include <pthread.h>
@@ -72,6 +73,10 @@ static char * str_ipcp_cmd(int code)
return "alloc_resp";
case IPCP_MSG_CODE__IPCP_FLOW_DEALLOC:
return "dealloc";
+ case IPCP_MSG_CODE__IPCP_FLOW_UPDATE:
+ return "flow_update";
+ case IPCP_MSG_CODE__IPCP_REPLY:
+ return "reply";
default:
assert(false);
return "unknown";
@@ -196,7 +201,9 @@ int ipcp_bootstrap(pid_t pid,
msg.conf = ipcp_config_s_to_msg(conf);
recv_msg = send_recv_ipcp_msg(pid, &msg);
+
ipcp_config_msg__free_unpacked(msg.conf, NULL);
+
if (recv_msg == NULL)
return -EIPCP;
@@ -225,9 +232,88 @@ int ipcp_bootstrap(pid_t pid,
return ret;
}
-int ipcp_enroll(pid_t pid,
- const char * dst,
- struct layer_info * info)
+ssize_t ipcp_list_poas(pid_t pid,
+ struct poa_spec ** eps)
+{
+ ipcp_msg_t msg = IPCP_MSG__INIT;
+ ipcp_msg_t * recv_msg;
+ size_t nr;
+ size_t i;
+
+ if (eps == NULL)
+ return -EINVAL;
+
+ *eps = NULL;
+
+ msg.code = IPCP_MSG_CODE__IPCP_LIST_POAS;
+
+ recv_msg = send_recv_ipcp_msg(pid, &msg);
+ if (recv_msg == NULL)
+ return -EIPCP;
+
+ nr = recv_msg->n_poas;
+ if (nr == 0) {
+ ipcp_msg__free_unpacked(recv_msg, NULL);
+ return 0;
+ }
+
+ *eps = malloc(nr * sizeof(**eps));
+ if (*eps == NULL) {
+ ipcp_msg__free_unpacked(recv_msg, NULL);
+ return -ENOMEM;
+ }
+
+ for (i = 0; i < nr; i++)
+ (*eps)[i] = poa_spec_msg_to_s(recv_msg->poas[i]);
+
+ ipcp_msg__free_unpacked(recv_msg, NULL);
+
+ return (ssize_t) nr;
+}
+
+int ipcp_attach(pid_t pid,
+ const struct poa_spec * poa,
+ bool attach)
+{
+ ipcp_msg_t msg = IPCP_MSG__INIT;
+ ipcp_msg_t * recv_msg;
+ int ret;
+
+ if (poa == NULL)
+ return -EINVAL;
+
+ if (attach)
+ msg.code = IPCP_MSG_CODE__IPCP_ATTACH;
+ else
+ msg.code = IPCP_MSG_CODE__IPCP_DETACH;
+
+ msg.poa = poa_spec_s_to_msg(poa);
+ if (msg.poa == NULL)
+ return -EINVAL;
+
+ recv_msg = send_recv_ipcp_msg(pid, &msg);
+
+ poa_spec_msg__free_unpacked(msg.poa, NULL);
+
+ if (recv_msg == NULL)
+ return -EIPCP;
+
+ if (!recv_msg->has_result) {
+ ipcp_msg__free_unpacked(recv_msg, NULL);
+ return -EIPCP;
+ }
+
+ ret = recv_msg->result;
+
+ ipcp_msg__free_unpacked(recv_msg, NULL);
+
+ return ret;
+}
+
+int ipcp_enroll(pid_t pid,
+ const char * dst,
+ const struct poa_addr * addr,
+ struct layer_info * info)
{
ipcp_msg_t msg = IPCP_MSG__INIT;
ipcp_msg_t * recv_msg;
@@ -239,7 +325,16 @@ int ipcp_enroll(pid_t pid,
msg.code = IPCP_MSG_CODE__IPCP_ENROLL;
msg.dst = (char *) dst;
+ if (addr != NULL) {
+ msg.peer = poa_addr_s_to_msg(addr);
+ if (msg.peer == NULL)
+ return -ENOMEM;
+ }
+
recv_msg = send_recv_ipcp_msg(pid, &msg);
+
+ if (msg.peer != NULL)
+ poa_addr_msg__free_unpacked(msg.peer, NULL);
if (recv_msg == NULL)
return -EIPCP;
@@ -267,10 +362,11 @@ int ipcp_enroll(pid_t pid,
return 0;
}
-int ipcp_connect(pid_t pid,
- const char * dst,
- const char * component,
- qosspec_t qs)
+int ipcp_connect(pid_t pid,
+ const char * dst,
+ const char * component,
+ qosspec_t qs,
+ const struct poa_addr * addr)
{
ipcp_msg_t msg = IPCP_MSG__INIT;
ipcp_msg_t * recv_msg;
@@ -283,8 +379,21 @@ int ipcp_connect(pid_t pid,
msg.pid = pid;
msg.qosspec = qos_spec_s_to_msg(&qs);
+ if (addr != NULL) {
+ msg.peer = poa_addr_s_to_msg(addr);
+ if (msg.peer == NULL) {
+ free(msg.qosspec);
+ return -ENOMEM;
+ }
+ }
+
recv_msg = send_recv_ipcp_msg(pid, &msg);
+
+ if (msg.peer != NULL)
+ poa_addr_msg__free_unpacked(msg.peer, NULL);
+
free(msg.qosspec);
+
if (recv_msg == NULL)
return -EIPCP;
@@ -457,6 +566,8 @@ int ipcp_flow_update(const struct flow_info * flow,
msg.has_pk = true;
msg.pk.data = data.data;
msg.pk.len = data.len;
+ msg.has_is_poa = true;
+ msg.is_poa = reg_flow_is_poa(flow->id);
recv_msg = send_recv_ipcp_msg(flow->n_1_pid, &msg);
if (recv_msg == NULL) {
@@ -538,6 +649,8 @@ int ipcp_flow_alloc_resp(const struct flow_info * flow,
msg.has_pk = response == 0;
msg.pk.data = data.data;
msg.pk.len = data.len;
+ msg.has_is_poa = true;
+ msg.is_poa = reg_flow_is_poa(flow->id);
recv_msg = send_recv_ipcp_msg(flow->n_1_pid, &msg);
if (recv_msg == NULL)
@@ -567,6 +680,8 @@ int ipcp_flow_dealloc(pid_t pid,
msg.flow_id = flow_id;
msg.has_timeo_sec = true;
msg.timeo_sec = timeo;
+ msg.has_is_poa = true;
+ msg.is_poa = reg_flow_is_poa(flow_id);
recv_msg = send_recv_ipcp_msg(pid, &msg);
if (recv_msg == NULL)
diff --git a/src/irmd/ipcp.h b/src/irmd/ipcp.h
index 8d06623c..1f257e7d 100644
--- a/src/irmd/ipcp.h
+++ b/src/irmd/ipcp.h
@@ -27,48 +27,57 @@
#ifndef OUROBOROS_IRMD_IPCP_H
#define OUROBOROS_IRMD_IPCP_H
-int ipcp_enroll(pid_t pid,
- const char * dst,
- struct layer_info * info);
+int ipcp_enroll(pid_t pid,
+ const char * dst,
+ const struct poa_addr * addr,
+ struct layer_info * info);
-int ipcp_bootstrap(pid_t pid,
- struct ipcp_config * conf,
- struct layer_info * info);
+int ipcp_attach(pid_t pid,
+ const struct poa_spec * poa,
+ bool attach);
-int ipcp_connect(pid_t pid,
- const char * dst,
- const char * component,
- qosspec_t qs);
+ssize_t ipcp_list_poas(pid_t pid,
+ struct poa_spec ** eps);
-int ipcp_disconnect(pid_t pid,
- const char * dst,
- const char * component);
+int ipcp_bootstrap(pid_t pid,
+ struct ipcp_config * conf,
+ struct layer_info * info);
-int ipcp_reg(pid_t pid,
- const buffer_t hash);
+int ipcp_connect(pid_t pid,
+ const char * dst,
+ const char * component,
+ qosspec_t qs,
+ const struct poa_addr * addr);
-int ipcp_unreg(pid_t pid,
+int ipcp_disconnect(pid_t pid,
+ const char * dst,
+ const char * component);
+
+int ipcp_reg(pid_t pid,
const buffer_t hash);
-int ipcp_query(pid_t pid,
- const buffer_t dst);
+int ipcp_unreg(pid_t pid,
+ const buffer_t hash);
+
+int ipcp_query(pid_t pid,
+ const buffer_t dst);
-int ipcp_flow_alloc(const struct flow_info * flow,
- const buffer_t hash,
- const buffer_t data);
+int ipcp_flow_alloc(const struct flow_info * flow,
+ const buffer_t hash,
+ const buffer_t data);
-int ipcp_flow_join(const struct flow_info * flow,
- const buffer_t dst);
+int ipcp_flow_join(const struct flow_info * flow,
+ const buffer_t dst);
-int ipcp_flow_alloc_resp(const struct flow_info * flow,
- int response,
- const buffer_t data);
+int ipcp_flow_alloc_resp(const struct flow_info * flow,
+ int response,
+ const buffer_t data);
-int ipcp_flow_dealloc(pid_t pid,
- int flow_id,
- time_t timeo);
+int ipcp_flow_dealloc(pid_t pid,
+ int flow_id,
+ time_t timeo);
-int ipcp_flow_update(const struct flow_info * flow,
- const buffer_t data);
+int ipcp_flow_update(const struct flow_info * flow,
+ const buffer_t data);
#endif /* OUROBOROS_IRMD_IPCP_H */
diff --git a/src/irmd/irmd.h b/src/irmd/irmd.h
index f88378ad..9d42e248 100644
--- a/src/irmd/irmd.h
+++ b/src/irmd/irmd.h
@@ -26,29 +26,48 @@
#include <ouroboros/ipcp.h>
#include <ouroboros/irm.h>
-int create_ipcp(struct ipcp_info * info);
+int create_ipcp(struct ipcp_info * info);
-int bootstrap_ipcp(pid_t pid,
- struct ipcp_config * conf);
+int bootstrap_ipcp(pid_t pid,
+ struct ipcp_config * conf);
-int enroll_ipcp(pid_t pid,
- const char * dst);
+int enroll_ipcp(pid_t pid,
+ const char * dst,
+ const struct poa_addr * addr);
-int connect_ipcp(pid_t pid,
- const char * dst,
- const char * component,
- qosspec_t qs);
+int enroll_ipcp_resolve(pid_t pid,
+ const char * dst,
+ struct poa_addr * addr);
-int name_create(struct name_info * info);
+int attach_ipcp(pid_t pid,
+ const struct poa_spec * poa,
+ bool attach);
-int name_reg(const char * name,
- pid_t pid);
+ssize_t list_poas(pid_t pid,
+ struct poa_spec ** eps);
-int bind_process(pid_t pid,
- const char * name);
+int connect_ipcp(pid_t pid,
+ const char * dst,
+ const char * component,
+ qosspec_t qs,
+ const struct poa_addr * addr);
-int bind_program(char ** exec,
- const char * name,
- uint8_t flags);
+int connect_ipcp_resolve(pid_t pid,
+ const char * dst,
+ const char * component,
+ qosspec_t qs,
+ struct poa_addr * addr);
+
+int name_create(struct name_info * info);
+
+int name_reg(const char * name,
+ pid_t pid);
+
+int bind_process(pid_t pid,
+ const char * name);
+
+int bind_program(char ** exec,
+ const char * name,
+ uint8_t flags);
#endif /* OUROBOROS_IRMD_H*/
diff --git a/src/irmd/main.c b/src/irmd/main.c
index 19be4ab9..e587a552 100644
--- a/src/irmd/main.c
+++ b/src/irmd/main.c
@@ -61,6 +61,7 @@
#include <dirent.h>
#include <grp.h>
+#include <netdb.h>
#include <pwd.h>
#include <signal.h>
#include <spawn.h>
@@ -225,18 +226,6 @@ static pid_t spawn_ipcp(struct ipcp_info * info)
case IPCP_BROADCAST:
exec_name = IPCP_BROADCAST_EXEC;
break;
- case IPCP_UDP4:
- exec_name = IPCP_UDP4_EXEC;
- break;
- case IPCP_UDP6:
- exec_name = IPCP_UDP6_EXEC;
- break;
- case IPCP_ETH_LLC:
- exec_name = IPCP_ETH_LLC_EXEC;
- break;
- case IPCP_ETH_DIX:
- exec_name = IPCP_ETH_DIX_EXEC;
- break;
case IPCP_LOCAL:
exec_name = IPCP_LOCAL_EXEC;
break;
@@ -367,9 +356,6 @@ int bootstrap_ipcp(pid_t pid,
goto fail;
}
- if (conf->type == IPCP_UDP4 || conf->type == IPCP_UDP6)
- conf->layer_info.dir_hash_algo = (enum pol_dir_hash) HASH_MD5;
-
if (ipcp_bootstrap(pid, conf, &layer)) {
log_err("Could not bootstrap IPCP.");
goto fail;
@@ -389,8 +375,55 @@ int bootstrap_ipcp(pid_t pid,
return -1;
}
-int enroll_ipcp(pid_t pid,
- const char * dst)
+ssize_t list_poas(pid_t pid,
+ struct poa_spec ** eps)
+{
+ struct ipcp_info info;
+
+ info.pid = pid;
+ if (reg_get_ipcp(&info, NULL) < 0) {
+ log_err("Could not find IPCP %d.", pid);
+ return -1;
+ }
+
+ if (info.type != IPCP_UNICAST && info.type != IPCP_BROADCAST)
+ return 0; /* nothing it could be attached to */
+
+ return ipcp_list_poas(pid, eps);
+}
+
+/* Only the types that can carry a PoA may take one. */
+int attach_ipcp(pid_t pid,
+ const struct poa_spec * poa,
+ bool attach)
+{
+ struct ipcp_info info;
+
+ info.pid = pid;
+ if (reg_get_ipcp(&info, NULL) < 0) {
+ log_err("Could not find IPCP %d.", pid);
+ return -1;
+ }
+
+ if (info.type != IPCP_UNICAST && info.type != IPCP_BROADCAST) {
+ log_err("IPCP %d does not support PoAs.", pid);
+ return -1;
+ }
+
+ if (ipcp_attach(pid, poa, attach) < 0) {
+ log_err("Could not %s IPCP %d.",
+ attach ? "attach" : "detach", pid);
+ return -1;
+ }
+
+ log_info("%s IPCP %d.", attach ? "Attached" : "Detached", pid);
+
+ return 0;
+}
+
+int enroll_ipcp(pid_t pid,
+ const char * dst,
+ const struct poa_addr * addr)
{
struct layer_info layer;
struct ipcp_info info;
@@ -402,7 +435,7 @@ int enroll_ipcp(pid_t pid,
goto fail;
}
- if (ipcp_enroll(pid, dst, &layer) < 0) {
+ if (ipcp_enroll(pid, dst, addr, &layer) < 0) {
log_err("Could not enroll IPCP %d.", pid);
goto fail;
}
@@ -421,10 +454,11 @@ int enroll_ipcp(pid_t pid,
return -1;
}
-int connect_ipcp(pid_t pid,
- const char * dst,
- const char * component,
- qosspec_t qs)
+int connect_ipcp(pid_t pid,
+ const char * dst,
+ const char * component,
+ qosspec_t qs,
+ const struct poa_addr * addr)
{
struct ipcp_info info;
@@ -442,7 +476,7 @@ int connect_ipcp(pid_t pid,
log_dbg("Connecting %s to %s.", component, dst);
- if (ipcp_connect(pid, dst, component, qs)) {
+ if (ipcp_connect(pid, dst, component, qs, addr) < 0) {
log_err("Could not connect IPCP %d to %s.", pid, dst);
return -EPERM;
}
@@ -453,6 +487,172 @@ int connect_ipcp(pid_t pid,
return 0;
}
+/* Resolve the dial name, if any, and fill in one record. */
+static int poa_addr_resolve(struct poa_addr * addr,
+ struct addrinfo ** res)
+{
+ struct addrinfo hints;
+
+ *res = NULL;
+
+ if (addr == NULL)
+ return 0;
+
+ if (addr->hostname[0] == '\0')
+ return addr->type == POA_UDP ? -EINVAL : 0;
+
+ if (addr->type != POA_UDP)
+ return -EINVAL;
+
+ memset(&hints, 0, sizeof(hints));
+
+ hints.ai_family = AF_UNSPEC;
+ hints.ai_socktype = SOCK_DGRAM;
+ if (getaddrinfo(addr->hostname, NULL, &hints, res) != 0) {
+ log_err("Failed to resolve %s.", addr->hostname);
+ return -EADDRNOTAVAIL;
+ }
+
+ return 0;
+}
+
+static void poa_addr_from_ai(struct poa_addr * addr,
+ const struct addrinfo * ai,
+ uint16_t port)
+{
+ struct sockaddr_in * in;
+ struct sockaddr_in6 * in6;
+
+ if (ai->ai_family == AF_INET) {
+ in = (struct sockaddr_in *) ai->ai_addr;
+ addr->type = POA_UDP4;
+ addr->udp4.ip_addr = in->sin_addr;
+ addr->udp4.port = port;
+ } else {
+ in6 = (struct sockaddr_in6 *) ai->ai_addr;
+ addr->type = POA_UDP6;
+ addr->udp6.ip_addr = in6->sin6_addr;
+ addr->udp6.port = port;
+ }
+}
+
+/* Skip families without an attached PoA; keep all if none reported. */
+static void poa_families(pid_t pid,
+ bool * v4,
+ bool * v6)
+{
+ struct poa_spec * eps;
+ ssize_t n;
+ ssize_t i;
+
+ eps = NULL;
+
+ *v4 = false;
+ *v6 = false;
+
+ n = list_poas(pid, &eps);
+ for (i = 0; i < n; i++) {
+ if (eps[i].type == POA_UDP4)
+ *v4 = true;
+
+ if (eps[i].type == POA_UDP6)
+ *v6 = true;
+ }
+
+ free(eps);
+
+ if (!*v4 && !*v6) {
+ *v4 = true;
+ *v6 = true;
+ }
+}
+
+/* Try each resolved record in order until an enrolment succeeds. */
+int enroll_ipcp_resolve(pid_t pid,
+ const char * dst,
+ struct poa_addr * addr)
+{
+ struct addrinfo * res;
+ struct addrinfo * ai;
+ uint16_t port;
+ bool v4;
+ bool v6;
+ int ret;
+
+ ret = poa_addr_resolve(addr, &res);
+ if (ret < 0)
+ return ret;
+
+ if (res == NULL)
+ return enroll_ipcp(pid, dst, addr);
+
+ port = addr->udp4.port; /* POA_UDP parks it there */
+
+ poa_families(pid, &v4, &v6);
+
+ ret = -EADDRNOTAVAIL;
+
+ for (ai = res; ai != NULL; ai = ai->ai_next) {
+ if ((ai->ai_family == AF_INET && !v4)
+ || (ai->ai_family == AF_INET6 && !v6))
+ continue;
+
+ poa_addr_from_ai(addr, ai, port);
+
+ ret = enroll_ipcp(pid, dst, addr);
+ if (ret == 0)
+ break;
+ }
+
+ freeaddrinfo(res);
+
+ return ret;
+}
+
+/* Try each resolved record in order until a connect succeeds. */
+int connect_ipcp_resolve(pid_t pid,
+ const char * dst,
+ const char * component,
+ qosspec_t qs,
+ struct poa_addr * addr)
+{
+ struct addrinfo * res;
+ struct addrinfo * ai;
+ uint16_t port;
+ bool v4;
+ bool v6;
+ int ret;
+
+ ret = poa_addr_resolve(addr, &res);
+ if (ret < 0)
+ return ret;
+
+ if (res == NULL)
+ return connect_ipcp(pid, dst, component, qs, addr);
+
+ port = addr->udp4.port; /* POA_UDP parks it there */
+
+ poa_families(pid, &v4, &v6);
+
+ ret = -EADDRNOTAVAIL;
+
+ for (ai = res; ai != NULL; ai = ai->ai_next) {
+ if ((ai->ai_family == AF_INET && !v4)
+ || (ai->ai_family == AF_INET6 && !v6))
+ continue;
+
+ poa_addr_from_ai(addr, ai, port);
+
+ ret = connect_ipcp(pid, dst, component, qs, addr);
+ if (ret == 0)
+ break;
+ }
+
+ freeaddrinfo(res);
+
+ return ret;
+}
+
static int disconnect_ipcp(pid_t pid,
const char * dst,
const char * component)
@@ -882,6 +1082,22 @@ static void __cleanup_flow(void * flow)
reg_destroy_flow(((struct flow_info *) flow)->id);
}
+/*
+ * A PoA flow is secured as the IPCP that owns it: trust is in the
+ * peer itself. Without a security config for that name the flow
+ * stays in the clear.
+ */
+static void poa_name_info(const char * name,
+ struct name_info * info)
+{
+ if (reg_get_name_info(name, info) == 0)
+ return;
+
+ memset(info, 0, sizeof(*info));
+
+ strcpy(info->name, name);
+}
+
static int flow_accept(struct flow_info * flow,
buffer_t * data,
struct timespec * abstime,
@@ -942,7 +1158,9 @@ static int flow_accept(struct flow_info * flow,
goto fail_oap;
}
- if (reg_get_name_info(name, &info) < 0) {
+ if (reg_flow_is_poa(flow->id)) {
+ poa_name_info(name, &info);
+ } else if (reg_get_name_info(name, &info) < 0) {
log_err("Failed to get name info for %s.", name);
err = -ENAME;
goto fail_oap;
@@ -1262,7 +1480,7 @@ static int flow_alloc_direct(const char * dst,
return -EAGAIN;
}
- if (oap_cli_prepare(&ctx, info, &req_hdr, *data, false) < 0) {
+ if (oap_cli_prepare(&ctx, info, NULL, &req_hdr, *data, false) < 0) {
log_err("Failed to prepare OAP for %s.", dst);
return -EBADF;
}
@@ -1295,7 +1513,7 @@ static int flow_alloc_direct(const char * dst,
return -ETIMEDOUT;
}
- err = oap_cli_complete(ctx, info, resp_hdr, data, sk, NULL, NULL);
+ err = oap_cli_complete(ctx, resp_hdr, data, sk, NULL, NULL);
if (err < 0) {
log_err("OAP completion failed for %s.", dst);
freebuf(resp_hdr);
@@ -1378,7 +1596,7 @@ static int flow_alloc(const char * dst,
goto fail_prepare;
}
- if (oap_cli_prepare(&ctx, &info, &req_hdr, *data, false) < 0) {
+ if (oap_cli_prepare(&ctx, &info, NULL, &req_hdr, *data, false) < 0) {
log_err("Failed to prepare OAP request for %s.", dst);
err = -EBADF;
goto fail_prepare;
@@ -1410,7 +1628,7 @@ static int flow_alloc(const char * dst,
goto fail_peer;
}
- err = oap_cli_complete(ctx, &info, resp_hdr, data, sk, NULL, &peer_crt);
+ err = oap_cli_complete(ctx, resp_hdr, data, sk, NULL, &peer_crt);
if (err < 0) {
log_err("OAP completion failed for %s.", dst);
goto fail_complete;
@@ -1444,6 +1662,199 @@ static int flow_alloc(const char * dst,
return err;
}
+/* Creates the flow and prepares the key exchange the IPCP will carry. */
+/* The N-1 is only known once the flow exists. */
+static int poa_alloc(struct flow_info * flow,
+ const char * dst,
+ buffer_t * data)
+{
+ struct ipcp_info ipcp;
+ struct layer_info layer;
+ struct name_info info;
+ buffer_t req_hdr = BUF_INIT;
+ void * ctx;
+ int err;
+
+ ipcp.pid = flow->n_pid;
+ if (reg_get_ipcp(&ipcp, &layer) < 0) {
+ log_err("No IPCP with pid %d.", flow->n_pid);
+
+ err = -EIPCP;
+ goto fail_flow;
+ }
+
+ flow->n_1_pid = 0;
+ if (reg_create_flow(flow) < 0) {
+ log_err("Failed to create PoA flow.");
+
+ err = -EBADF;
+ goto fail_flow;
+ }
+
+ reg_set_name_for_flow_id(ipcp.name, flow->id);
+
+ if (reg_flow_set_poa(flow->id) < 0) {
+ err = -EBADF;
+ goto fail_prepare;
+ }
+
+ flow->uid = reg_get_proc_uid(flow->n_pid);
+ flow->n_1_pid = flow->n_pid; /* the IPCP is its own N-1 */
+ if (reg_prepare_flow_alloc(flow) < 0) {
+ log_err("Failed to prepare PoA flow allocation.");
+
+ err = -EBADF;
+ goto fail_prepare;
+ }
+
+ poa_name_info(ipcp.name, &info);
+
+ if (oap_cli_prepare(&ctx, &info, dst, &req_hdr, *data, false) < 0) {
+ log_err("Failed to prepare OAP request for %s.", ipcp.name);
+
+ err = -EBADF;
+ goto fail_prepare;
+ }
+
+ if (reg_flow_set_oap_ctx(flow->id, ctx) < 0) {
+ err = -EBADF;
+ goto fail_ctx;
+ }
+
+ log_info("Allocating PoA flow %d for %d as %s.",
+ flow->id, flow->n_pid, ipcp.name);
+
+ *data = req_hdr;
+
+ return 0;
+
+ fail_ctx:
+ freebuf(req_hdr);
+ oap_ctx_free(ctx);
+ fail_prepare:
+ reg_destroy_flow(flow->id);
+ fail_flow:
+ return err;
+}
+
+/* Completes the key exchange once the IPCP has the peer's response. */
+/*
+ * We present the PoA's own credentials and expect the peer to present
+ * dst's: the address only says how to reach it. oap_cli_complete
+ * frees the ctx on every path.
+ */
+static int poa_complete(struct flow_info * flow,
+ int response,
+ buffer_t * data,
+ struct crypt_sk * sk)
+{
+ struct name_info info;
+ buffer_t peer_crt = BUF_INIT;
+ buffer_t out = BUF_INIT;
+ buffer_t empty = BUF_INIT;
+ char name[NAME_SIZE + 1];
+ void * ctx;
+ int err;
+
+ ctx = reg_flow_take_oap_ctx(flow->id);
+ if (ctx == NULL) {
+ log_err("No pending PoA flow %d.", flow->id);
+ return -EBADF;
+ }
+
+ if (response < 0) {
+ log_dbg("PoA flow %d refused: %d.", flow->id, response);
+
+ err = response;
+ goto fail;
+ }
+
+ if (reg_get_name_for_flow_id(name, flow->id) < 0) {
+ err = -EBADF;
+ goto fail;
+ }
+
+ poa_name_info(name, &info);
+
+ err = oap_cli_complete(ctx, *data, &out, sk, NULL, &peer_crt);
+ if (err < 0) {
+ log_err("OAP completion failed for %s.", name);
+
+ ctx = NULL;
+ goto fail;
+ }
+
+ if (sk->nid != NID_undef)
+ reg_flow_set_rekey(flow->id, true, peer_crt);
+
+ flow->state = FLOW_ALLOCATED;
+ if (reg_respond_alloc(flow, &empty, 0) < 0) {
+ log_err("Failed to update PoA flow %d.", flow->id);
+
+ err = -EBADF;
+ goto fail_crt;
+ }
+
+ log_info("PoA flow %d allocated to %s.", flow->id, name);
+
+ freebuf(peer_crt);
+ freebuf(out);
+
+ return 0;
+
+ fail_crt:
+ freebuf(peer_crt);
+ freebuf(out);
+ fail:
+ oap_ctx_free(ctx);
+ reg_destroy_flow(flow->id);
+ return err;
+}
+
+/* A peer requested a flow on a PoA of this IPCP. */
+static int poa_req_arr(struct flow_info * flow,
+ buffer_t * data)
+{
+ struct ipcp_info ipcp;
+ struct layer_info layer;
+ int ret;
+
+ ipcp.pid = flow->n_1_pid;
+ if (reg_get_ipcp(&ipcp, &layer) < 0) {
+ log_err("No IPCP with pid %d.", flow->n_1_pid);
+
+ ret = -EIPCP;
+ goto fail;
+ }
+
+ log_info("PoA flow request arrived for %s.", ipcp.name);
+
+ ret = wait_for_accept(ipcp.name);
+ if (ret < 0) {
+ log_err("No active process for %s.", ipcp.name);
+ goto fail;
+ }
+
+ flow->id = ret;
+ flow->state = FLOW_ALLOCATED;
+ if (reg_flow_set_poa(flow->id) < 0) {
+ ret = -EBADF;
+ goto fail;
+ }
+
+ reg_set_name_for_flow_id(ipcp.name, flow->id);
+
+ ret = reg_respond_accept(flow, data);
+ if (ret < 0) {
+ log_err("Failed to respond to PoA flow %d.", flow->id);
+ goto fail;
+ }
+
+ return 0;
+ fail:
+ return ret;
+}
+
static int flow_alloc_reply(struct flow_info * flow,
int response,
buffer_t * data)
@@ -1596,6 +2007,11 @@ static int rekey_name_info(int flow_id,
if (reg_get_name_for_flow_id(name, flow_id) < 0)
return -1;
+ if (reg_flow_is_poa(flow_id)) {
+ poa_name_info(name, info);
+ return 0;
+ }
+
return reg_get_name_info(name, info);
}
@@ -1667,7 +2083,7 @@ static void rekey_do_initiate(struct list_head * tbl,
goto fail;
}
- if (oap_cli_prepare(&ctx, &name, &req, data, true) < 0) {
+ if (oap_cli_prepare(&ctx, &name, NULL, &req, data, true) < 0) {
log_err("Failed to prepare re-key for flow %d.", flow_id);
goto fail;
}
@@ -1758,7 +2174,7 @@ static void rekey_do_complete(struct list_head * tbl,
reg_flow_get_peer_crt(flow_id, &crt);
/* oap_cli_complete frees the ctx on every path. */
- if (oap_cli_complete(e->ctx, &info, buf, &data, &sk, &crt, NULL) < 0) {
+ if (oap_cli_complete(e->ctx, buf, &data, &sk, &crt, NULL) < 0) {
log_warn("Failed to complete re-key for flow %d.", flow_id);
e->ctx = NULL;
goto finish_clear;
@@ -1933,7 +2349,7 @@ static void rekey_do_direct(int flow_id)
return;
}
- if (oap_cli_prepare(&ctx, &info, &req, data, true) < 0) {
+ if (oap_cli_prepare(&ctx, &info, NULL, &req, data, true) < 0) {
log_err("Failed to prepare re-key for flow %d.", flow_id);
reg_flow_clear_in_flight(flow_id);
return;
@@ -2307,6 +2723,10 @@ static irm_msg_t * do_command_msg(irm_msg_t * msg,
pid_t cpid;
irm_msg_t * ret_msg;
buffer_t data;
+ struct poa_addr poa_addr;
+ struct poa_addr * pa;
+ struct poa_spec poa_spec;
+ struct poa_spec * eps = NULL;
memset(&flow, 0, sizeof(flow));
@@ -2350,11 +2770,74 @@ static irm_msg_t * do_command_msg(irm_msg_t * msg,
res = bootstrap_ipcp(msg->pid, &conf);
break;
case IRM_MSG_CODE__IRM_ENROLL_IPCP:
- res = enroll_ipcp(msg->pid, msg->dst);
+ if (msg->peer != NULL) {
+ poa_addr = poa_addr_msg_to_s(msg->peer);
+ if (poa_addr.type == POA_INVALID) {
+ res = -EINVAL;
+ break;
+ }
+ }
+
+ if (msg->conf != NULL)
+ conf = ipcp_config_msg_to_s(msg->conf);
+ res = enroll_ipcp_resolve(msg->pid, msg->dst,
+ msg->peer != NULL ? &poa_addr : NULL);
+ break;
+ case IRM_MSG_CODE__IRM_ATTACH_IPCP:
+ poa_spec = poa_spec_msg_to_s(msg->poa);
+ if (poa_spec.type == POA_INVALID) {
+ res = -EINVAL;
+ break;
+ }
+
+ res = attach_ipcp(msg->pid, &poa_spec, true);
+ break;
+ case IRM_MSG_CODE__IRM_DETACH_IPCP:
+ poa_spec = poa_spec_msg_to_s(msg->poa);
+ if (poa_spec.type == POA_INVALID) {
+ res = -EINVAL;
+ break;
+ }
+
+ res = attach_ipcp(msg->pid, &poa_spec, false);
+ break;
+ case IRM_MSG_CODE__IRM_LIST_POAS:
+ res = list_poas(msg->pid, &eps);
+ if (res > 0) {
+ ssize_t i;
+ ret_msg->poas = malloc(res * sizeof(*ret_msg->poas));
+ if (ret_msg->poas == NULL) {
+ free(eps);
+
+ res = -ENOMEM;
+ break;
+ }
+ for (i = 0; i < res; i++) {
+ ret_msg->poas[i] = poa_spec_s_to_msg(&eps[i]);
+ if (ret_msg->poas[i] == NULL)
+ break;
+ ret_msg->n_poas++;
+ }
+ res = i == res ? 0 : -ENOMEM;
+ }
+ free(eps);
break;
case IRM_MSG_CODE__IRM_CONNECT_IPCP:
flow.qs = qos_spec_msg_to_s(msg->qosspec);
- res = connect_ipcp(msg->pid, msg->dst, msg->comp, flow.qs);
+
+ pa = NULL;
+
+ if (msg->peer != NULL) {
+ poa_addr = poa_addr_msg_to_s(msg->peer);
+ if (poa_addr.type == POA_INVALID) {
+ res = -EINVAL;
+ break;
+ }
+ pa = &poa_addr;
+ }
+
+ res = connect_ipcp_resolve(msg->pid, msg->dst, msg->comp,
+ flow.qs, pa);
break;
case IRM_MSG_CODE__IRM_DISCONNECT_IPCP:
res = disconnect_ipcp(msg->pid, msg->dst, msg->comp);
@@ -2499,6 +2982,67 @@ static irm_msg_t * do_command_msg(irm_msg_t * msg,
if (res == 0)
ret_msg->flow_info = flow_info_s_to_msg(&flow);
break;
+ case IRM_MSG_CODE__IRM_POA_FLOW_ALLOC:
+ flow = flow_info_msg_to_s(msg->flow_info);
+
+ clrbuf(data);
+
+ res = poa_alloc(&flow, msg->dst, &data);
+ if (res == 0) {
+ ret_msg->flow_info = flow_info_s_to_msg(&flow);
+ ret_msg->has_pk = true;
+ ret_msg->pk.len = data.len;
+ ret_msg->pk.data = data.data;
+
+ clrbuf(data);
+ }
+ break;
+ case IRM_MSG_CODE__IRM_POA_FLOW_ALLOC_R:
+ data.len = msg->pk.len;
+ data.data = msg->pk.data;
+ msg->pk.data = NULL; /* pass data */
+ msg->pk.len = 0;
+ flow = flow_info_msg_to_s(msg->flow_info);
+ sk.key = kbuf;
+ res = poa_complete(&flow, msg->response, &data, &sk);
+
+ freebuf(data);
+
+ if (res != 0)
+ break;
+
+ ret_msg->flow_info = flow_info_s_to_msg(&flow);
+ ret_msg->has_cipher_nid = true;
+ ret_msg->cipher_nid = sk.nid;
+
+ if (sk.nid == NID_undef)
+ break;
+
+ hbuf = malloc(SYMMKEYSZ);
+ if (hbuf == NULL) {
+ log_err("Failed to malloc key buf");
+
+ res = -ENOMEM;
+ break;
+ }
+
+ memcpy(hbuf, kbuf, SYMMKEYSZ);
+
+ ret_msg->sym_key.data = hbuf;
+ ret_msg->sym_key.len = SYMMKEYSZ;
+ ret_msg->has_sym_key = true;
+ break;
+ case IRM_MSG_CODE__IPCP_POA_FLOW_REQ_ARR:
+ data.len = msg->pk.len;
+ data.data = msg->pk.data;
+ msg->pk.data = NULL; /* pass data */
+ msg->pk.len = 0;
+ flow = flow_info_msg_to_s(msg->flow_info);
+
+ res = poa_req_arr(&flow, &data);
+ if (res == 0)
+ ret_msg->flow_info = flow_info_s_to_msg(&flow);
+ break;
case IRM_MSG_CODE__IPCP_FLOW_ALLOC_REPLY:
data.len = msg->pk.len;
data.data = msg->pk.data;
diff --git a/src/irmd/oap.h b/src/irmd/oap.h
index 86f11e21..e9d7511b 100644
--- a/src/irmd/oap.h
+++ b/src/irmd/oap.h
@@ -40,21 +40,21 @@ int oap_auth_add_ca_crt(void * crt);
int oap_auth_add_chain_crt(void * crt);
/*
-* Prepare OAP request header for server, returns context
-* Passes client data for srv, returns srv data for client
-* rekey forces ephemeral server-encap KEX (no client-encap; preserves FS/PCS)
-*/
+ * Prepares the request header and returns the context. info holds the
+ * credentials we present, peer the name the server certificate must
+ * carry (NULL expects info->name); rekey forces an ephemeral KEX.
+ */
int oap_cli_prepare(void ** ctx,
const struct name_info * info,
+ const char * peer,
buffer_t * req_buf,
buffer_t data,
bool rekey);
/*
- * Server processes header, creates response header, returns secret key.
- * data is in/out: input=srv data to send, output=cli data received.
- * rekey drops the cert and verifies against cached_crt; peer_crt (or NULL)
- * receives a copy of the peer cert to cache at the initial handshake.
+ * Answers the request header and returns the secret key. data is
+ * in/out; rekey verifies against cached_crt, and peer_crt takes a copy
+ * of the peer cert to cache at the initial handshake.
*/
int oap_srv_process(const struct name_info * info,
buffer_t req_buf,
@@ -70,13 +70,12 @@ int oap_srv_process(const struct name_info * info,
* cached_crt verifies a cert-less re-key; peer_crt (or NULL) receives a
* copy of the peer cert to cache at the initial handshake.
*/
-int oap_cli_complete(void * ctx,
- const struct name_info * info,
- buffer_t rsp_buf,
- buffer_t * data,
- struct crypt_sk * sk,
- const buffer_t * cached_crt,
- buffer_t * peer_crt);
+int oap_cli_complete(void * ctx,
+ buffer_t rsp_buf,
+ buffer_t * data,
+ struct crypt_sk * sk,
+ const buffer_t * cached_crt,
+ buffer_t * peer_crt);
/* Free OAP state (on failure before complete) */
void oap_ctx_free(void * ctx);
diff --git a/src/irmd/oap/cli.c b/src/irmd/oap/cli.c
index ebfcd71f..2203596f 100644
--- a/src/irmd/oap/cli.c
+++ b/src/irmd/oap/cli.c
@@ -50,6 +50,7 @@
struct oap_cli_ctx {
uint8_t __id[OAP_ID_SIZE];
buffer_t id;
+ char peer[NAME_SIZE + 1]; /* expected server name */
uint8_t kex_buf[CRYPT_KEY_BUFSZ];
uint8_t req_hash[MAX_HASH_SIZE];
size_t req_hash_len;
@@ -241,6 +242,7 @@ static int do_client_kex_prepare(const char * server_name,
int oap_cli_prepare(void ** ctx,
const struct name_info * info,
+ const char * peer,
buffer_t * req_buf,
buffer_t data,
bool rekey)
@@ -259,6 +261,11 @@ int oap_cli_prepare(void ** ctx,
clrbuf(*req_buf);
*ctx = NULL;
+ if (peer != NULL && strlen(peer) > NAME_SIZE) {
+ log_err("Peer name too long.");
+ return -EINVAL;
+ }
+
/* Allocate ctx to carry between prepare and complete */
s = malloc(sizeof(*s));
if (s == NULL) {
@@ -269,13 +276,15 @@ int oap_cli_prepare(void ** ctx,
memset(s, 0, sizeof(*s));
OAP_CLI_CTX_INIT(s);
+ strcpy(s->peer, peer != NULL ? peer : info->name);
+
/* Generate session ID */
if (random_buffer(s->__id, OAP_ID_SIZE) < 0) {
log_err("Failed to generate OAP session ID.");
goto fail_id;
}
- log_dbg_id(s->id.data, "Preparing OAP request for %s.", info->name);
+ log_dbg_id(s->id.data, "Preparing OAP request for %s.", s->peer);
/* Load client credentials */
if (load_cli_credentials(info, &pkp, &crt) < 0) {
@@ -315,7 +324,7 @@ int oap_cli_prepare(void ** ctx,
oap_hdr_init(&s->local_hdr, s->id, s->kex_buf, data, s->scfg.c.nid);
- if (do_client_kex_prepare(info->name, s) < 0) {
+ if (do_client_kex_prepare(s->peer, s) < 0) {
log_err_id(s->id.data, "Failed to prepare client KEX.");
goto fail_kex;
}
@@ -482,7 +491,8 @@ static int do_client_kex_complete(struct oap_cli_ctx * s,
}
SET_KEX_CIPHER(scfg, peer_hdr->cipher_str);
- if (crypt_validate_nid(scfg->c.nid) < 0) {
+
+ if (crypt_cipher_rank(scfg->c.nid) < 0) {
log_err_id(id, "Server cipher '%s' not supported.",
peer_hdr->cipher_str);
return -ENOTSUP;
@@ -518,13 +528,12 @@ static int do_client_kex_complete(struct oap_cli_ctx * s,
return do_client_kex_complete_dhe(s, peer_hdr, sk);
}
-int oap_cli_complete(void * ctx,
- const struct name_info * info,
- buffer_t rsp_buf,
- buffer_t * data,
- struct crypt_sk * sk,
- const buffer_t * cached_crt,
- buffer_t * peer_crt)
+int oap_cli_complete(void * ctx,
+ buffer_t rsp_buf,
+ buffer_t * data,
+ struct crypt_sk * sk,
+ const buffer_t * cached_crt,
+ buffer_t * peer_crt)
{
struct oap_cli_ctx * s = ctx;
struct oap_hdr peer_hdr;
@@ -538,7 +547,6 @@ int oap_cli_complete(void * ctx,
int rc;
assert(ctx != NULL);
- assert(info != NULL);
assert(data != NULL);
assert(sk != NULL);
@@ -550,7 +558,7 @@ int oap_cli_complete(void * ctx,
id = s->id.data;
- log_dbg_id(id, "Completing OAP for %s.", info->name);
+ log_dbg_id(id, "Completing OAP for %s.", s->peer);
/* Decode response header using client's md_nid for hash length */
if (oap_hdr_decode(&peer_hdr, rsp_buf, s->req_md_nid, false) < 0) {
@@ -616,9 +624,9 @@ int oap_cli_complete(void * ctx,
}
/* Verify peer certificate name matches expected destination */
- if (peer_hdr.crt.len > 0 && strcmp(peer, info->name) != 0) {
+ if (peer_hdr.crt.len > 0 && strcmp(peer, s->peer) != 0) {
log_err_id(id, "Peer crt for '%s' does not match '%s'.",
- peer, info->name);
+ peer, s->peer);
goto fail_oap;
}
@@ -667,7 +675,7 @@ int oap_cli_complete(void * ctx,
goto fail_oap;
}
- log_info_id(id, "OAP completed for %s.", info->name);
+ log_info_id(id, "OAP completed for %s.", s->peer);
freebuf(peer_hdr.sealed_pt);
diff --git a/src/irmd/oap/tests/common.c b/src/irmd/oap/tests/common.c
index 16d52c63..b65f3997 100644
--- a/src/irmd/oap/tests/common.c
+++ b/src/irmd/oap/tests/common.c
@@ -159,7 +159,7 @@ void oap_test_teardown(struct oap_test_ctx * ctx)
if (ctx->cli.state != NULL) {
res.key = ctx->cli.key;
- oap_cli_complete(ctx->cli.state, &ctx->cli.info, dummy,
+ oap_cli_complete(ctx->cli.state, dummy,
&ctx->data, &res, NULL, NULL);
ctx->cli.state = NULL;
}
@@ -179,8 +179,8 @@ void oap_test_teardown(struct oap_test_ctx * ctx)
int oap_cli_prepare_ctx(struct oap_test_ctx * ctx)
{
- return oap_cli_prepare(&ctx->cli.state, &ctx->cli.info, &ctx->req_hdr,
- ctx->data, ctx->rekey);
+ return oap_cli_prepare(&ctx->cli.state, &ctx->cli.info, NULL,
+ &ctx->req_hdr, ctx->data, ctx->rekey);
}
int oap_srv_process_ctx(struct oap_test_ctx * ctx)
@@ -203,8 +203,7 @@ int oap_cli_complete_ctx(struct oap_test_ctx * ctx)
struct crypt_sk res = { .nid = NID_undef, .key = ctx->cli.key };
int ret;
- ret = oap_cli_complete(ctx->cli.state, &ctx->cli.info, ctx->resp_hdr,
- &ctx->data, &res,
+ ret = oap_cli_complete(ctx->cli.state, ctx->resp_hdr, &ctx->data, &res,
ctx->rekey ? &ctx->cli_crt : NULL,
ctx->rekey ? NULL : &ctx->cli_crt);
ctx->cli.state = NULL;
@@ -527,7 +526,7 @@ int roundtrip_kex_only(void)
goto fail;
}
- if (oap_cli_prepare(&cli_state, &cli_info, &req_hdr,
+ if (oap_cli_prepare(&cli_state, &cli_info, NULL, &req_hdr,
data, false) < 0) {
printf("Client prepare failed.\n");
goto fail_cleanup;
@@ -545,7 +544,7 @@ int roundtrip_kex_only(void)
res.key = cli_key;
- if (oap_cli_complete(cli_state, &cli_info, resp_hdr, &data, &res,
+ if (oap_cli_complete(cli_state, resp_hdr, &data, &res,
NULL, NULL) < 0) {
printf("Client complete failed.\n");
cli_state = NULL;
@@ -575,8 +574,8 @@ int roundtrip_kex_only(void)
fail_cleanup:
if (cli_state != NULL) {
res.key = cli_key;
- oap_cli_complete(cli_state, &cli_info, resp_hdr, &data,
- &res, NULL, NULL);
+
+ oap_cli_complete(cli_state, resp_hdr, &data, &res, NULL, NULL);
}
freebuf(resp_hdr);
freebuf(req_hdr);
@@ -655,7 +654,7 @@ int corrupted_response(const char * root_ca,
res.key = ctx.cli.key;
- if (oap_cli_complete(ctx.cli.state, &ctx.cli.info, ctx.resp_hdr,
+ if (oap_cli_complete(ctx.cli.state, ctx.resp_hdr,
&ctx.data, &res, NULL, NULL) == 0) {
printf("Client should reject corrupted response.\n");
ctx.cli.state = NULL;
diff --git a/src/irmd/reg/flow.c b/src/irmd/reg/flow.c
index 8be2dfc7..63c9199b 100644
--- a/src/irmd/reg/flow.c
+++ b/src/irmd/reg/flow.c
@@ -126,6 +126,9 @@ static int create_rbuffs(struct reg_flow * flow,
assert(flow->n_1_rb == NULL);
flow->info.n_1_pid = info->n_1_pid;
+ if (flow->poa)
+ return 0;
+
flow->n_1_rb = ssm_rbuff_create(info->n_1_pid, info->id);
if (flow->n_1_rb == NULL)
goto fail_n_1_rb;
diff --git a/src/irmd/reg/flow.h b/src/irmd/reg/flow.h
index 166bed61..c7021a0f 100644
--- a/src/irmd/reg/flow.h
+++ b/src/irmd/reg/flow.h
@@ -48,6 +48,8 @@ struct reg_flow {
char name[NAME_SIZE + 1];
bool direct;
+ bool poa; /* transport is a point of attachment */
+ void * oap_ctx; /* key exchange, prepare -> complete */
/* Tier-2 re-key state (encrypted flows only) */
struct {
diff --git a/src/irmd/reg/name.c b/src/irmd/reg/name.c
index a3621fc3..08426033 100644
--- a/src/irmd/reg/name.c
+++ b/src/irmd/reg/name.c
@@ -1,4 +1,3 @@
-
/*
* Ouroboros - Copyright (C) 2016 - 2026
*
diff --git a/src/irmd/reg/reg.c b/src/irmd/reg/reg.c
index ebf3959d..a302fa15 100644
--- a/src/irmd/reg/reg.c
+++ b/src/irmd/reg/reg.c
@@ -2119,6 +2119,77 @@ bool reg_flow_is_direct(int flow_id)
return ret;
}
+bool reg_flow_is_poa(int flow_id)
+{
+ struct reg_flow * flow;
+ bool ret;
+
+ pthread_mutex_lock(&reg.mtx);
+
+ flow = __reg_get_flow(flow_id);
+
+ ret = flow != NULL && flow->poa;
+
+ pthread_mutex_unlock(&reg.mtx);
+
+ return ret;
+}
+
+int reg_flow_set_poa(int flow_id)
+{
+ struct reg_flow * flow;
+ int ret = -1;
+
+ pthread_mutex_lock(&reg.mtx);
+
+ flow = __reg_get_flow(flow_id);
+ if (flow != NULL) {
+ flow->poa = true;
+ ret = 0;
+ }
+
+ pthread_mutex_unlock(&reg.mtx);
+
+ return ret;
+}
+
+int reg_flow_set_oap_ctx(int flow_id,
+ void * ctx)
+{
+ struct reg_flow * flow;
+ int ret = -1;
+
+ pthread_mutex_lock(&reg.mtx);
+
+ flow = __reg_get_flow(flow_id);
+ if (flow != NULL) {
+ flow->oap_ctx = ctx;
+ ret = 0;
+ }
+
+ pthread_mutex_unlock(&reg.mtx);
+
+ return ret;
+}
+
+void * reg_flow_take_oap_ctx(int flow_id)
+{
+ struct reg_flow * flow;
+ void * ctx = NULL;
+
+ pthread_mutex_lock(&reg.mtx);
+
+ flow = __reg_get_flow(flow_id);
+ if (flow != NULL) {
+ ctx = flow->oap_ctx;
+ flow->oap_ctx = NULL;
+ }
+
+ pthread_mutex_unlock(&reg.mtx);
+
+ return ctx;
+}
+
void reg_flow_set_rekey(int flow_id,
bool initiator,
buffer_t peer_crt)
@@ -2607,7 +2678,7 @@ void reg_notify_flow(int flow_id,
pthread_mutex_unlock(&reg.mtx);
}
-/* Wake both endpoints of a direct flow (acceptor and allocator). */
+/* Wake both PoAs of a direct flow (acceptor and allocator). */
void reg_notify_flow_peers(int flow_id,
int event)
{
diff --git a/src/irmd/reg/reg.h b/src/irmd/reg/reg.h
index 8a313d46..6882532c 100644
--- a/src/irmd/reg/reg.h
+++ b/src/irmd/reg/reg.h
@@ -33,138 +33,147 @@
#include "pool.h"
-int reg_init(void);
+int reg_init(void);
-void reg_clear(void);
+void reg_clear(void);
-void reg_fini(void);
+void reg_fini(void);
-int reg_create_flow(struct flow_info * info);
+int reg_create_flow(struct flow_info * info);
-int reg_destroy_flow(int flow_id);
+int reg_destroy_flow(int flow_id);
-bool reg_has_flow(int flow_id);
+bool reg_has_flow(int flow_id);
-int reg_create_proc(const struct proc_info * info);
+int reg_create_proc(const struct proc_info * info);
/* Use this for all processes, including ipcps */
-int reg_destroy_proc(pid_t pid);
+int reg_destroy_proc(pid_t pid);
-bool reg_has_proc(pid_t pid);
+bool reg_has_proc(pid_t pid);
-bool reg_is_proc_privileged(pid_t pid);
+bool reg_is_proc_privileged(pid_t pid);
-int reg_prepare_pool(uid_t uid,
- gid_t gid);
+int reg_prepare_pool(uid_t uid,
+ gid_t gid);
-uid_t reg_get_proc_uid(pid_t pid);
+uid_t reg_get_proc_uid(pid_t pid);
-void reg_kill_all_proc(int signal);
+void reg_kill_all_proc(int signal);
-pid_t reg_get_dead_proc(void);
+pid_t reg_get_dead_proc(void);
-int reg_create_spawned(pid_t pid);
+int reg_create_spawned(pid_t pid);
-bool reg_has_spawned(pid_t pid);
+bool reg_has_spawned(pid_t pid);
-void reg_kill_all_spawned(int signal);
+void reg_kill_all_spawned(int signal);
-int reg_first_spawned(void);
+int reg_first_spawned(void);
-int reg_bind_proc(const char * name,
- pid_t proc);
+int reg_bind_proc(const char * name,
+ pid_t proc);
-int reg_unbind_proc(const char * name,
- pid_t proc);
+int reg_unbind_proc(const char * name,
+ pid_t proc);
-int reg_create_ipcp(const struct ipcp_info * info);
+int reg_create_ipcp(const struct ipcp_info * info);
-bool reg_has_ipcp(pid_t pid);
+bool reg_has_ipcp(pid_t pid);
-int reg_set_layer_for_ipcp(struct ipcp_info * info,
- const struct layer_info * layer);
+int reg_set_layer_for_ipcp(struct ipcp_info * info,
+ const struct layer_info * layer);
-int reg_get_ipcp(struct ipcp_info * info,
- struct layer_info * layer);
+int reg_get_ipcp(struct ipcp_info * info,
+ struct layer_info * layer);
-int reg_get_ipcp_by_layer(struct ipcp_info * info,
- struct layer_info * layer);
+int reg_get_ipcp_by_layer(struct ipcp_info * info,
+ struct layer_info * layer);
/* TODO don't rely on protobuf here */
-int reg_list_ipcps(ipcp_list_msg_t *** msg);
+int reg_list_ipcps(ipcp_list_msg_t *** msg);
-int reg_create_name(const struct name_info * info);
+int reg_create_name(const struct name_info * info);
-int reg_destroy_name(const char * name);
+int reg_destroy_name(const char * name);
-bool reg_has_name(const char * name);
+bool reg_has_name(const char * name);
-int reg_get_name_info(const char * name,
- struct name_info * info);
+int reg_get_name_info(const char * name,
+ struct name_info * info);
-int reg_get_name_for_hash(char * buf,
- enum hash_algo algo,
- const uint8_t * hash);
+int reg_get_name_for_hash(char * buf,
+ enum hash_algo algo,
+ const uint8_t * hash);
-int reg_get_name_for_flow_id(char * buf,
- int flow_id);
+int reg_get_name_for_flow_id(char * buf,
+ int flow_id);
-void reg_set_name_for_flow_id(const char * name,
- int flow_id);
+void reg_set_name_for_flow_id(const char * name,
+ int flow_id);
/* TODO don't rely on protobuf here */
-int reg_list_names(name_info_msg_t *** names);
+int reg_list_names(name_info_msg_t *** names);
-int reg_create_prog(const struct prog_info * info);
+int reg_create_prog(const struct prog_info * info);
-int reg_destroy_prog(const char * name);
+int reg_destroy_prog(const char * name);
-bool reg_has_prog(const char * name);
+bool reg_has_prog(const char * name);
-int reg_get_exec(const char * name,
- char *** exec);
+int reg_get_exec(const char * name,
+ char *** exec);
-int reg_bind_prog(const char * name,
- char ** exec,
- uint8_t flags);
+int reg_bind_prog(const char * name,
+ char ** exec,
+ uint8_t flags);
-int reg_unbind_prog(const char * name,
- const char * prog);
+int reg_unbind_prog(const char * name,
+ const char * prog);
-int reg_prepare_flow_alloc(struct flow_info * info);
+int reg_prepare_flow_alloc(struct flow_info * info);
-int reg_wait_flow_allocated(struct flow_info * info,
+int reg_wait_flow_allocated(struct flow_info * info,
+ buffer_t * pbuf,
+ const struct timespec * abstime);
+
+int reg_respond_alloc(struct flow_info * info,
+ buffer_t * pbuf,
+ int response);
+
+int reg_prepare_flow_accept(struct flow_info * info);
+
+int reg_wait_flow_accepted(struct flow_info * info,
buffer_t * pbuf,
const struct timespec * abstime);
-int reg_respond_alloc(struct flow_info * info,
- buffer_t * pbuf,
- int response);
+int reg_wait_flow_accepting(const char * name,
+ const struct timespec * abstime);
-int reg_prepare_flow_accept(struct flow_info * info);
+int reg_respond_accept(struct flow_info * info,
+ buffer_t * pbuf);
-int reg_wait_flow_accepted(struct flow_info * info,
- buffer_t * pbuf,
- const struct timespec * abstime);
+int reg_prepare_flow_direct(struct flow_info * info,
+ buffer_t * pbuf,
+ uid_t alloc_uid);
-int reg_wait_flow_accepting(const char * name,
- const struct timespec * abstime);
+int reg_respond_flow_direct(int flow_id,
+ buffer_t * pbuf);
+
+int reg_wait_flow_direct(int flow_id,
+ buffer_t * pbuf,
+ const struct timespec * abstime);
-int reg_respond_accept(struct flow_info * info,
- buffer_t * pbuf);
+bool reg_flow_is_direct(int flow_id);
-int reg_prepare_flow_direct(struct flow_info * info,
- buffer_t * pbuf,
- uid_t alloc_uid);
+bool reg_flow_is_poa(int flow_id);
-int reg_respond_flow_direct(int flow_id,
- buffer_t * pbuf);
+int reg_flow_set_poa(int flow_id);
-int reg_wait_flow_direct(int flow_id,
- buffer_t * pbuf,
- const struct timespec * abstime);
+int reg_flow_set_oap_ctx(int flow_id,
+ void * ctx);
-bool reg_flow_is_direct(int flow_id);
+void * reg_flow_take_oap_ctx(int flow_id);
/* Per-flow snapshot for the re-key timer */
struct rekey_info {
@@ -176,70 +185,70 @@ struct rekey_info {
bool direct;
};
-void reg_flow_set_rekey(int flow_id,
- bool initiator,
- buffer_t peer_crt);
+void reg_flow_set_rekey(int flow_id,
+ bool initiator,
+ buffer_t peer_crt);
-int reg_flow_get_peer_crt(int flow_id,
- buffer_t * crt);
+int reg_flow_get_peer_crt(int flow_id,
+ buffer_t * crt);
-int reg_flow_get_epoch(int flow_id);
+int reg_flow_get_epoch(int flow_id);
-bool reg_flow_rekey_pending(int flow_id);
+bool reg_flow_rekey_pending(int flow_id);
-pid_t reg_flow_get_n_1_pid(int flow_id);
+pid_t reg_flow_get_n_1_pid(int flow_id);
-int reg_flow_snapshot_rekey_due(struct rekey_info * snap,
- int max);
+int reg_flow_snapshot_rekey_due(struct rekey_info * snap,
+ int max);
-void reg_flow_clear_in_flight(int flow_id);
+void reg_flow_clear_in_flight(int flow_id);
-bool reg_flow_rekey_begin(int flow_id);
+bool reg_flow_rekey_begin(int flow_id);
-bool reg_flow_rekey_should_yield(int flow_id);
+bool reg_flow_rekey_should_yield(int flow_id);
-int reg_flow_store_pending(int flow_id,
- const uint8_t * seed,
- uint8_t epoch,
- bool initiator);
+int reg_flow_store_pending(int flow_id,
+ const uint8_t * seed,
+ uint8_t epoch,
+ bool initiator);
-int reg_flow_store_pending_direct(int flow_id,
- const uint8_t * seed,
- uint8_t epoch);
+int reg_flow_store_pending_direct(int flow_id,
+ const uint8_t * seed,
+ uint8_t epoch);
-int reg_flow_take_pending(int flow_id,
- uid_t uid,
- pid_t cpid,
- uint8_t * seed,
- uint8_t * epoch,
- bool * initiator);
+int reg_flow_take_pending(int flow_id,
+ uid_t uid,
+ pid_t cpid,
+ uint8_t * seed,
+ uint8_t * epoch,
+ bool * initiator);
-bool reg_flow_rekey_arr_admit(int flow_id,
- pid_t n_1_pid,
- bool is_req);
+bool reg_flow_rekey_arr_admit(int flow_id,
+ pid_t n_1_pid,
+ bool is_req);
-void reg_flow_rekey_arr_done(int flow_id,
- bool is_req);
+void reg_flow_rekey_arr_done(int flow_id,
+ bool is_req);
-bool reg_flow_owned_by(int flow_id,
- uid_t uid);
+bool reg_flow_owned_by(int flow_id,
+ uid_t uid);
-void reg_notify_flow(int flow_id,
- int event);
+void reg_notify_flow(int flow_id,
+ int event);
-void reg_notify_flow_peers(int flow_id,
- int event);
+void reg_notify_flow_peers(int flow_id,
+ int event);
-void reg_dealloc_flow(struct flow_info * info);
+void reg_dealloc_flow(struct flow_info * info);
-void reg_dealloc_flow_resp(struct flow_info * info);
+void reg_dealloc_flow_resp(struct flow_info * info);
-int reg_wait_proc(pid_t pid,
- const struct timespec * abstime);
+int reg_wait_proc(pid_t pid,
+ const struct timespec * abstime);
-int reg_wait_ipcp_boot(struct ipcp_info * ipcp,
- const struct timespec * abstime);
+int reg_wait_ipcp_boot(struct ipcp_info * ipcp,
+ const struct timespec * abstime);
-int reg_respond_ipcp(const struct ipcp_info * info);
+int reg_respond_ipcp(const struct ipcp_info * info);
#endif /* OUROBOROS_IRMD_REG_H */
diff --git a/src/irmd/reg/tests/reg_test.c b/src/irmd/reg/tests/reg_test.c
index a8c1b1fa..ab57241c 100644
--- a/src/irmd/reg/tests/reg_test.c
+++ b/src/irmd/reg/tests/reg_test.c
@@ -746,7 +746,7 @@ static int test_reg_direct_flow_success(void)
reg_dealloc_flow(&info);
if (info.state != FLOW_DEALLOC_PENDING) {
- printf("Same endpoint dealloc changed state.\n");
+ printf("Same PoA dealloc changed state.\n");
goto fail;
}
diff --git a/src/lib/CMakeLists.txt b/src/lib/CMakeLists.txt
index 3abf39d0..48064ce4 100644
--- a/src/lib/CMakeLists.txt
+++ b/src/lib/CMakeLists.txt
@@ -95,9 +95,9 @@ endif()
if(HAVE_LIBURCU)
target_link_libraries(ouroboros-common PRIVATE Urcu::Urcu)
- # urcu headers require C99; override the global -std=c89 for this TU only.
- set_source_files_properties(crypt/keyrot.c PROPERTIES
- COMPILE_OPTIONS "-std=gnu99")
+ # urcu headers require C99; override the global -std=c89 for these TUs.
+ set_source_files_properties(crypt/keyrot.c dev.c poa/addr.c poa/eth.c
+ poa/udp.c PROPERTIES COMPILE_OPTIONS "-std=gnu99")
endif()
install(TARGETS ouroboros-common
@@ -107,8 +107,14 @@ install(TARGETS ouroboros-common
set(SOURCE_FILES_DEV
cep.c
dev.c
+ poa/addr.c
+ poa/udp.c
)
+if(HAVE_ETH)
+ list(APPEND SOURCE_FILES_DEV poa/eth.c)
+endif()
+
add_library(ouroboros-dev SHARED
${SOURCE_FILES_DEV}
${CEP_PROTO_SRCS})
@@ -131,6 +137,11 @@ target_include_directories(ouroboros-dev
target_link_libraries(ouroboros-dev PUBLIC ouroboros-common)
+if(HAVE_LIBURCU)
+ # dev.c calls urcu directly; do not rely on transitive linkage.
+ target_link_libraries(ouroboros-dev PRIVATE Urcu::Urcu)
+endif()
+
install(TARGETS ouroboros-dev
EXPORT OuroborosTargets
LIBRARY DESTINATION ${CMAKE_INSTALL_LIBDIR})
diff --git a/src/lib/config.h.in b/src/lib/config.h.in
index 4ebaa194..3c6985c3 100644
--- a/src/lib/config.h.in
+++ b/src/lib/config.h.in
@@ -79,6 +79,25 @@
#define PROC_RES_FDS @PROC_RES_FDS@
#define PROC_MAX_FQUEUES @PROC_MAX_FQUEUES@
+/* Flow endpoints */
+#define POA_MGMT_FRAME_SIZE @POA_MGMT_FRAME_SIZE@
+#define POA_MGMT_SND_TIMEO @POA_MGMT_SND_TIMEO@
+#define POA_MAX_POAS @POA_MAX_POAS@
+#define POA_UDP_MPL @POA_UDP_MPL@
+#define POA_UDP4_MTU @POA_UDP4_MTU@
+#define POA_UDP6_MTU @POA_UDP6_MTU@
+#define POA_UDP_RD_BUF @POA_UDP_RD_BUF@
+#cmakedefine HAVE_ETH
+#cmakedefine HAVE_RAW_SOCKETS
+#cmakedefine HAVE_BPF
+#cmakedefine HAVE_NETMAP
+#cmakedefine POA_ETH_QDISC_BYPASS
+#define POA_ETH_MPL @POA_ETH_MPL@
+#define POA_ETH_SNDBUF @POA_ETH_SNDBUF@
+#define POA_ETH_RCVBUF @POA_ETH_RCVBUF@
+#define POA_ETH_LO_MTU @POA_ETH_LO_MTU@
+#define POA_ETH_RD_BUF @POA_ETH_RD_BUF@
+
/* Default Delta-t parameters */
#cmakedefine FRCT_LINUX_RTT_ESTIMATOR
#define DELT_A (@DELTA_T_ACK@) /* ms */
diff --git a/src/lib/dev.c b/src/lib/dev.c
index 3fb8d831..eb706691 100644
--- a/src/lib/dev.c
+++ b/src/lib/dev.c
@@ -28,6 +28,9 @@
#include "config.h"
#include "ssm.h"
+#include "poa/poa.h"
+
+#define OUROBOROS_PREFIX "libouroboros"
#include <ouroboros/atomics.h>
#include <ouroboros/bitmap.h>
@@ -45,6 +48,7 @@
#include <ouroboros/ipcp-dev.h>
#include <ouroboros/list.h>
#include <ouroboros/local-dev.h>
+#include <ouroboros/logs.h>
#include <ouroboros/np1_flow.h>
#include <ouroboros/pthread.h>
#include <ouroboros/random.h>
@@ -63,8 +67,10 @@
#ifdef HAVE_LIBGCRYPT
#include <gcrypt.h>
#endif
+#include <arpa/inet.h>
#include <stdarg.h>
#include <stdbool.h>
+#include <stddef.h>
#include <inttypes.h>
#include <stdio.h>
#include <stdlib.h>
@@ -98,8 +104,8 @@ struct flow {
ssize_t part_idx;
struct crypt_ctx * crypt;
- int headsz; /* Selector */
- int tailsz; /* Tag + CRC */
+ int headsz; /* Selector */
+ int tailsz; /* Tag + CRC */
struct timespec rk_grace; /* TX-promote deadline */
struct timespec rk_attempt; /* Last re-key attempt */
@@ -116,6 +122,8 @@ struct flow {
struct timespec rcv_timeo;
struct frcti * frcti;
+
+ struct poa_flow * poa; /* NULL for shared memory flows */
};
struct flow_set {
@@ -696,6 +704,9 @@ static void do_flow_fini(int fd)
{
assert(fd >= 0 && fd < PROC_MAX_FLOWS);
+ if (proc.flows[fd].poa != NULL)
+ poa_flow_detach(proc.flows[fd].poa);
+
if (proc.flows[fd].frcti != NULL)
frcti_destroy(proc.flows[fd].frcti);
@@ -750,9 +761,11 @@ static __inline__ size_t flow_user_mtu(const struct flow * flow,
return raw > hdr ? raw - hdr : 0;
}
+/* A PoA flow transmits on its own socket; it has no tx ring. */
static int flow_init(struct flow_info * info,
struct crypt_sk * sk,
- time_t rtt_hint)
+ time_t rtt_hint,
+ struct poa_flow * pf)
{
struct timespec now;
struct flow * flow;
@@ -777,13 +790,15 @@ static int flow_init(struct flow_info * info,
if (flow->rx_rb == NULL)
goto fail_rx_rb;
- flow->tx_rb = ssm_rbuff_open(info->n_1_pid, info->id);
- if (flow->tx_rb == NULL)
- goto fail_tx_rb;
+ if (pf == NULL) {
+ flow->tx_rb = ssm_rbuff_open(info->n_1_pid, info->id);
+ if (flow->tx_rb == NULL)
+ goto fail_tx_rb;
- flow->set = ssm_flow_set_open(info->n_1_pid);
- if (flow->set == NULL)
- goto fail_set;
+ flow->set = ssm_flow_set_open(info->n_1_pid);
+ if (flow->set == NULL)
+ goto fail_set;
+ }
flow->oflags = FLOWFDEFAULT;
flow->part_idx = NO_PART;
@@ -792,11 +807,14 @@ static int flow_init(struct flow_info * info,
flow->crypt = NULL;
flow->headsz = 0;
flow->tailsz = 0;
+ flow->poa = pf;
if (IS_ENCRYPTED(sk)) {
flow->crypt = crypt_create_ctx(sk);
- if (flow->crypt == NULL)
+ if (flow->crypt == NULL) {
+ err = -ECRYPT;
goto fail_crypt;
+ }
flow->headsz = crypt_get_headsz(flow->crypt);
flow->tailsz = crypt_get_tagsz(flow->crypt);
}
@@ -815,6 +833,9 @@ static int flow_init(struct flow_info * info,
proc.id_to_fd[info->id].fd = fd;
+ if (pf != NULL)
+ poa_flow_attach(pf, info->id, flow->rx_rb);
+
flow_set_state(&proc.id_to_fd[info->id], FLOW_ALLOCATED);
pthread_rwlock_unlock(&proc.lock);
@@ -824,9 +845,11 @@ static int flow_init(struct flow_info * info,
fail_frcti:
crypt_destroy_ctx(flow->crypt);
fail_crypt:
- ssm_flow_set_close(flow->set);
+ if (flow->set != NULL)
+ ssm_flow_set_close(flow->set);
fail_set:
- ssm_rbuff_close(flow->tx_rb);
+ if (flow->tx_rb != NULL)
+ ssm_rbuff_close(flow->tx_rb);
fail_tx_rb:
ssm_rbuff_close(flow->rx_rb);
fail_rx_rb:
@@ -1077,14 +1100,20 @@ static void fini(void)
__attribute__((section(INIT_SECTION))) __typeof__(init) * __init = init;
__attribute__((section(FINI_SECTION))) __typeof__(fini) * __fini = fini;
+/*
+ * A PoA flow is announced by its peer before the accept; from the
+ * reply on, the peer may transmit, so the flow must be able to
+ * receive.
+ */
int flow_accept(qosspec_t * qs,
const struct timespec * timeo)
{
struct flow_info flow;
- struct crypt_sk crypt;
+ struct crypt_sk crypt;
uint8_t buf[SOCK_BUF_SIZE];
buffer_t msg = {SOCK_BUF_SIZE, buf};
uint8_t key[SYMMKEYSZ];
+ struct poa_flow * pf;
int fd;
int err;
@@ -1116,8 +1145,13 @@ int flow_accept(qosspec_t * qs,
if (err < 0)
return err;
- /* No RTT in accept; rtt_hint=0 bootstraps from first ACK. */
- fd = flow_init(&flow, &crypt, 0);
+ pf = poa_flow_take_pending(flow.id);
+
+ fd = flow_init(&flow, &crypt, 0, pf);
+ if (fd >= 0)
+ poa_flow_ready(pf);
+ else if (pf != NULL)
+ poa_flow_detach(pf);
crypt_secure_clear(key, SYMMKEYSZ);
@@ -1173,7 +1207,7 @@ int flow_alloc(const char * dst,
if (err < 0)
return err;
- fd = flow_init(&flow, &crypt, ts_diff_ns(&t1, &t0));
+ fd = flow_init(&flow, &crypt, ts_diff_ns(&t1, &t0), NULL);
crypt_secure_clear(key, SYMMKEYSZ);
@@ -1214,7 +1248,7 @@ int flow_join(const char * dst,
if (err < 0)
return err;
- fd = flow_init(&flow, &crypt, 0);
+ fd = flow_init(&flow, &crypt, 0, NULL);
crypt_secure_clear(key, SYMMKEYSZ);
@@ -1275,7 +1309,8 @@ int flow_dealloc(int fd)
pthread_cleanup_push(__cleanup_rwlock_unlock, &proc.lock);
- ssm_rbuff_fini(flow->tx_rb);
+ if (flow->tx_rb != NULL)
+ ssm_rbuff_fini(flow->tx_rb);
pthread_cleanup_pop(true);
@@ -1428,7 +1463,11 @@ int fccntl(int fd,
break;
case FLOWGTXQLEN:
qlen = va_arg(l, size_t *);
- *qlen = ssm_rbuff_queued(flow->tx_rb);
+
+ if (flow->poa != NULL)
+ *qlen = poa_flow_qpkts(flow->poa);
+ else
+ *qlen = ssm_rbuff_queued(flow->tx_rb);
break;
case FLOWGMTU:
maxp = va_arg(l, size_t *);
@@ -1456,16 +1495,18 @@ int fccntl(int fd,
if (flow->oflags & FLOWFDOWN) {
ssm_rbuff_set_bits(flow->rx_rb, RB_FLOWDOWN);
- ssm_rbuff_set_bits(flow->tx_rb, RB_FLOWDOWN);
- ssm_flow_set_notify(flow->set,
- flow->info.id,
- FLOW_DOWN);
+ if (flow->tx_rb != NULL)
+ ssm_rbuff_set_bits(flow->tx_rb, RB_FLOWDOWN);
+ if (flow->set != NULL)
+ ssm_flow_set_notify(flow->set, flow->info.id,
+ FLOW_DOWN);
} else {
ssm_rbuff_clr_bits(flow->rx_rb, RB_FLOWDOWN);
- ssm_rbuff_clr_bits(flow->tx_rb, RB_FLOWDOWN);
- ssm_flow_set_notify(flow->set,
- flow->info.id,
- FLOW_UP);
+ if (flow->tx_rb != NULL)
+ ssm_rbuff_clr_bits(flow->tx_rb, RB_FLOWDOWN);
+ if (flow->set != NULL)
+ ssm_flow_set_notify(flow->set, flow->info.id,
+ FLOW_UP);
}
break;
@@ -1599,6 +1640,9 @@ static int flow_tx_spb(struct flow * flow,
goto enomem;
}
+ if (flow->poa != NULL)
+ return poa_flow_tx(flow->poa, spb, block, abstime);
+
if (!block)
ret = ssm_rbuff_write(flow->tx_rb, idx);
else
@@ -2450,7 +2494,7 @@ int np1_flow_alloc(pid_t n_pid,
/* np1 flow: n_1_pid is the upper. */
flow.n_1_pid = n_pid;
- return flow_init(&flow, &crypt, 0);
+ return flow_init(&flow, &crypt, 0, NULL);
}
int np1_flow_dealloc(int flow_id,
@@ -2584,7 +2628,7 @@ int ipcp_flow_req_arr(const buffer_t * dst,
crypt.nid = NID_undef;
- return flow_init(&flow, &crypt, 0);
+ return flow_init(&flow, &crypt, 0, NULL);
}
int ipcp_flow_update_arr(int flow_id,
@@ -2853,11 +2897,13 @@ int ipcp_flow_fini(int fd)
}
ssm_rbuff_set_bits(proc.flows[fd].rx_rb, RB_FLOWDOWN);
- ssm_rbuff_set_bits(proc.flows[fd].tx_rb, RB_FLOWDOWN);
- ssm_flow_set_notify(proc.flows[fd].set,
- proc.flows[fd].info.id,
- FLOW_DEALLOC);
+ if (proc.flows[fd].tx_rb != NULL)
+ ssm_rbuff_set_bits(proc.flows[fd].tx_rb, RB_FLOWDOWN);
+
+ if (proc.flows[fd].set != NULL)
+ ssm_flow_set_notify(proc.flows[fd].set, proc.flows[fd].info.id,
+ FLOW_DEALLOC);
rx_rb = proc.flows[fd].rx_rb;
@@ -2892,9 +2938,30 @@ size_t ipcp_flow_queued(int fd)
assert(fd >= 0 && fd < PROC_MAX_FLOWS);
assert(proc.flows[fd].info.id >= 0);
+ if (proc.flows[fd].poa != NULL)
+ return poa_flow_qlen(proc.flows[fd].poa);
+
return ssm_rbuff_queued(proc.flows[fd].tx_rb);
}
+int ipcp_flow_queue_id(int fd)
+{
+ int qid;
+
+ assert(fd >= 0 && fd < PROC_MAX_FLOWS);
+ assert(proc.flows[fd].info.id >= 0);
+
+ if (proc.flows[fd].poa == NULL)
+ return fd;
+
+ /* An unidentified PoA answers for itself, never for an fd. */
+ qid = poa_flow_qid(proc.flows[fd].poa);
+ if (qid < 0 || qid >= POA_MAX_POAS)
+ return fd;
+
+ return PROC_MAX_FLOWS + qid;
+}
+
int local_flow_transfer(int src_fd,
int dst_fd,
struct ssm_pool * src_pool,
@@ -2960,3 +3027,5 @@ int local_flow_transfer(int src_fd,
return ret;
}
+
+#include "poa/poa.c"
diff --git a/src/lib/irm.c b/src/lib/irm.c
index c62701aa..5d53841f 100644
--- a/src/lib/irm.c
+++ b/src/lib/irm.c
@@ -118,6 +118,8 @@ int irm_bootstrap_ipcp(pid_t pid,
msg.has_pid = true;
msg.pid = pid;
msg.conf = ipcp_config_s_to_msg(conf);
+ if (msg.conf == NULL)
+ return -ENOMEM;
recv_msg = send_recv_irm_msg(&msg);
ipcp_config_msg__free_unpacked(msg.conf, NULL);
@@ -135,10 +137,11 @@ int irm_bootstrap_ipcp(pid_t pid,
return ret;
}
-int irm_connect_ipcp(pid_t pid,
- const char * dst,
- const char * component,
- qosspec_t qs)
+int irm_connect_ipcp(pid_t pid,
+ const char * dst,
+ const char * component,
+ qosspec_t qs,
+ const struct poa_addr * addr)
{
irm_msg_t msg = IRM_MSG__INIT;
irm_msg_t * recv_msg;
@@ -152,7 +155,18 @@ int irm_connect_ipcp(pid_t pid,
msg.pid = pid;
msg.qosspec = qos_spec_s_to_msg(&qs);
+ if (addr != NULL) {
+ msg.peer = poa_addr_s_to_msg(addr);
+ if (msg.peer == NULL) {
+ qosspec_msg__free_unpacked(msg.qosspec, NULL);
+ return -ENOMEM;
+ }
+ }
+
recv_msg = send_recv_irm_msg(&msg);
+
+ if (msg.peer != NULL)
+ poa_addr_msg__free_unpacked(msg.peer, NULL);
qosspec_msg__free_unpacked(msg.qosspec, NULL);
if (recv_msg == NULL)
@@ -245,8 +259,99 @@ ssize_t irm_list_ipcps(struct ipcp_list_info ** ipcps)
return nr;
}
-int irm_enroll_ipcp(pid_t pid,
- const char * dst)
+static int irm_attach_detach(pid_t pid,
+ const struct poa_spec * poa,
+ int code)
+{
+ irm_msg_t msg = IRM_MSG__INIT;
+ irm_msg_t * recv_msg;
+ int ret;
+
+ if (pid == -1 || poa == NULL)
+ return -EINVAL;
+
+ msg.code = code;
+ msg.has_pid = true;
+ msg.pid = pid;
+ msg.poa = poa_spec_s_to_msg(poa);
+ if (msg.poa == NULL)
+ return -EINVAL;
+
+ recv_msg = send_recv_irm_msg(&msg);
+
+ poa_spec_msg__free_unpacked(msg.poa, NULL);
+
+ if (recv_msg == NULL)
+ return -EIRMD;
+
+ if (!recv_msg->has_result) {
+ irm_msg__free_unpacked(recv_msg, NULL);
+ return -EIRMD;
+ }
+
+ ret = recv_msg->result;
+
+ irm_msg__free_unpacked(recv_msg, NULL);
+
+ return ret;
+}
+
+ssize_t irm_list_poas(pid_t pid,
+ struct poa_spec ** poas)
+{
+ irm_msg_t msg = IRM_MSG__INIT;
+ irm_msg_t * recv_msg;
+ size_t nr;
+ size_t i;
+
+ if (pid == -1 || poas == NULL)
+ return -EINVAL;
+
+ *poas = NULL;
+
+ msg.code = IRM_MSG_CODE__IRM_LIST_POAS;
+ msg.has_pid = true;
+ msg.pid = pid;
+
+ recv_msg = send_recv_irm_msg(&msg);
+ if (recv_msg == NULL)
+ return -EIRMD;
+
+ nr = recv_msg->n_poas;
+ if (nr == 0) {
+ irm_msg__free_unpacked(recv_msg, NULL);
+ return 0;
+ }
+
+ *poas = malloc(nr * sizeof(**poas));
+ if (*poas == NULL) {
+ irm_msg__free_unpacked(recv_msg, NULL);
+ return -ENOMEM;
+ }
+
+ for (i = 0; i < nr; i++)
+ (*poas)[i] = poa_spec_msg_to_s(recv_msg->poas[i]);
+
+ irm_msg__free_unpacked(recv_msg, NULL);
+
+ return (ssize_t) nr;
+}
+
+int irm_attach_ipcp(pid_t pid,
+ const struct poa_spec * poa)
+{
+ return irm_attach_detach(pid, poa, IRM_MSG_CODE__IRM_ATTACH_IPCP);
+}
+
+int irm_detach_ipcp(pid_t pid,
+ const struct poa_spec * poa)
+{
+ return irm_attach_detach(pid, poa, IRM_MSG_CODE__IRM_DETACH_IPCP);
+}
+
+int irm_enroll_ipcp(pid_t pid,
+ const char * dst,
+ const struct poa_addr * addr)
{
irm_msg_t msg = IRM_MSG__INIT;
irm_msg_t * recv_msg = NULL;
@@ -260,7 +365,17 @@ int irm_enroll_ipcp(pid_t pid,
msg.pid = pid;
msg.dst = (char *) dst;
+ if (addr != NULL) {
+ msg.peer = poa_addr_s_to_msg(addr);
+ if (msg.peer == NULL)
+ return -ENOMEM;
+ }
+
recv_msg = send_recv_irm_msg(&msg);
+
+ if (msg.peer != NULL)
+ poa_addr_msg__free_unpacked(msg.peer, NULL);
+
if (recv_msg == NULL)
return -EIRMD;
diff --git a/src/lib/pb/ipcp.proto b/src/lib/pb/ipcp.proto
index afee4f91..298a15e2 100644
--- a/src/lib/pb/ipcp.proto
+++ b/src/lib/pb/ipcp.proto
@@ -40,6 +40,9 @@ enum ipcp_msg_code {
IPCP_DISCONNECT = 11;
IPCP_REPLY = 12;
IPCP_FLOW_UPDATE = 13;
+ IPCP_ATTACH = 14;
+ IPCP_DETACH = 15;
+ IPCP_LIST_POAS = 16;
}
message ipcp_msg {
@@ -58,4 +61,8 @@ message ipcp_msg {
optional sint32 mpl = 13; /* MPL in ms. */
optional int32 result = 14;
optional uint32 uid = 15; /* 0 = GSPP, >0 = PUP uid */
+ optional poa_addr_msg peer = 16; /* peer PoA address */
+ optional bool is_poa = 17; /* flow rides a PoA */
+ optional poa_spec_msg poa = 18; /* PoA to attach */
+ repeated poa_spec_msg poas = 19; /* PoAs attached */
}
diff --git a/src/lib/pb/ipcp_config.proto b/src/lib/pb/ipcp_config.proto
index a111b6d8..f6d313ac 100644
--- a/src/lib/pb/ipcp_config.proto
+++ b/src/lib/pb/ipcp_config.proto
@@ -66,29 +66,32 @@ message uni_config_msg {
required uint32 cong_avoid = 4;
}
-message eth_config_msg {
+message eth_poa_msg {
required string dev = 1;
required uint32 ethertype = 2;
+ optional bytes mac = 3;
}
-message udp4_config_msg {
- required uint32 ip_addr = 1;
- required uint32 port = 2;
- required uint32 dns_addr = 3; /* set to 0 if unused */
+message udp4_poa_msg {
+ required uint32 ip_addr = 1;
+ required uint32 port = 2;
}
-message udp6_config_msg {
- required bytes ip_addr = 1;
- required uint32 port = 2;
- required bytes dns_addr = 3; /* set to NULL if unused */
+message udp6_poa_msg {
+ required bytes ip_addr = 1;
+ required uint32 port = 2;
}
+/* Exactly one field is set; presence is the PoA type. */
+message poa_spec_msg {
+ optional udp4_poa_msg udp4 = 1;
+ optional udp6_poa_msg udp6 = 2;
+ optional eth_poa_msg eth = 3;
+}
+
message ipcp_config_msg {
required layer_info_msg layer_info = 1;
required uint32 ipcp_type = 2;
optional uni_config_msg unicast = 3;
- optional udp4_config_msg udp4 = 4;
- optional udp6_config_msg udp6 = 5;
- optional eth_config_msg eth = 6;
}
diff --git a/src/lib/pb/irm.proto b/src/lib/pb/irm.proto
index f54bc9ea..8f594fa3 100644
--- a/src/lib/pb/irm.proto
+++ b/src/lib/pb/irm.proto
@@ -55,6 +55,12 @@ enum irm_msg_code {
IRM_REPLY = 27;
IRM_FLOW_UPDATE = 28;
IPCP_FLOW_UPDATE_ARR = 29;
+ IRM_POA_FLOW_ALLOC = 30;
+ IRM_POA_FLOW_ALLOC_R = 31;
+ IPCP_POA_FLOW_REQ_ARR = 32;
+ IRM_ATTACH_IPCP = 33;
+ IRM_DETACH_IPCP = 34;
+ IRM_LIST_POAS = 35;
}
message timespec_msg {
@@ -101,4 +107,7 @@ message irm_msg {
optional uint32 generation = 28; /* re-key batch generation */
optional bool rekey = 29; /* re-key watermark trigger */
optional bool rk_initiator = 30; /* re-key proof-holder side */
+ optional poa_addr_msg peer = 32; /* peer PoA address */
+ optional poa_spec_msg poa = 33; /* PoA to attach */
+ repeated poa_spec_msg poas = 34; /* PoAs attached */
}
diff --git a/src/lib/pb/model.proto b/src/lib/pb/model.proto
index 4c1564a5..e00f3e12 100644
--- a/src/lib/pb/model.proto
+++ b/src/lib/pb/model.proto
@@ -58,6 +58,19 @@ message layer_info_msg {
required uint32 dir_hash_algo = 2;
}
+/* Address of a flow PoA peer. */
+message poa_addr_msg {
+ required uint32 type = 1; /* enum poa_type */
+ optional uint32 ip4 = 2;
+ optional bytes ip6 = 3;
+ optional uint32 port = 4;
+ optional bytes mac = 5;
+ optional uint32 ethertype = 6;
+ optional string dev = 7; /* our device, eth only */
+ optional string hostname = 8; /* IRMd resolves, UDP only */
+ optional bytes src_mac = 9; /* our MAC, eth only */
+}
+
message ipcp_info_msg {
required uint32 type = 1;
required string name = 2;
diff --git a/src/lib/poa/addr.c b/src/lib/poa/addr.c
new file mode 100644
index 00000000..d8811244
--- /dev/null
+++ b/src/lib/poa/addr.c
@@ -0,0 +1,142 @@
+/*
+ * Ouroboros - Copyright (C) 2016 - 2026
+ *
+ * Points of attachment (PoA) - addresses and management messages
+ *
+ * Dimitri Staessens <dimitri@ouroboros.rocks>
+ * Sander Vrijders <sander@ouroboros.rocks>
+ *
+ * This library is free software; you can redistribute it and/or
+ * modify it under the terms of the GNU Lesser General Public License
+ * version 2.1 as published by the Free Software Foundation.
+ *
+ * This library is distributed in the hope that it will be useful,
+ * but WITHOUT ANY WARRANTY; without even the implied warranty of
+ * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
+ * Lesser General Public License for more details.
+ *
+ * You should have received a copy of the GNU Lesser General Public
+ * License along with this library; if not, write to the Free Software
+ * Foundation, Inc., http://www.fsf.org/about/contact/.
+ */
+
+#if defined(__linux__) || defined(__CYGWIN__)
+#ifndef _DEFAULT_SOURCE /* Test include source */
+#define _DEFAULT_SOURCE
+#endif
+#endif
+
+#include "config.h"
+
+#include <ouroboros/endian.h>
+#include <ouroboros/errno.h>
+
+#include "poa.h"
+
+#include <arpa/inet.h>
+#include <stdio.h>
+#include <string.h>
+
+int poa_addr_cmp(const struct poa_addr * a,
+ const struct poa_addr * b)
+{
+ if (a->type != b->type)
+ return 1;
+
+ switch (a->type) {
+ case POA_UDP4:
+ if (a->udp4.port != b->udp4.port)
+ return 1;
+ return memcmp(&a->udp4.ip_addr, &b->udp4.ip_addr,
+ sizeof(a->udp4.ip_addr));
+ case POA_UDP6:
+ if (a->udp6.port != b->udp6.port)
+ return 1;
+ return memcmp(&a->udp6.ip_addr, &b->udp6.ip_addr,
+ sizeof(a->udp6.ip_addr));
+ case POA_ETH:
+ if (a->eth.dst.ethertype != b->eth.dst.ethertype)
+ return 1;
+ return memcmp(a->eth.dst.mac, b->eth.dst.mac, POA_MAC_SIZE);
+ default:
+ return 1;
+ }
+}
+
+/* Display/RIB entry name, e.g. "udp4.<ip>.<port>". */
+int poa_addr_name(const struct poa_addr * a,
+ char * buf,
+ size_t len)
+{
+ char ip[INET6_ADDRSTRLEN];
+ int ret;
+
+ switch (a->type) {
+ case POA_UDP4:
+ if (inet_ntop(AF_INET, &a->udp4.ip_addr,
+ ip, sizeof(ip)) == NULL)
+ return -EINVAL;
+ ret = snprintf(buf, len, "udp4.%s.%u", ip, a->udp4.port);
+ break;
+ case POA_UDP6:
+ if (inet_ntop(AF_INET6, &a->udp6.ip_addr,
+ ip, sizeof(ip)) == NULL)
+ return -EINVAL;
+ ret = snprintf(buf, len, "udp6.%s.%u", ip, a->udp6.port);
+ break;
+ case POA_ETH:
+ ret = snprintf(buf, len, "eth.%s.%04X",
+ a->eth.src.dev, a->eth.src.ethertype);
+ break;
+ default:
+ return -EINVAL;
+ }
+
+ if (ret < 0 || (size_t) ret >= len)
+ return -EMSGSIZE;
+
+ return 0;
+}
+
+void poa_mgmt_msg_ser(struct poa_mgmt_msg * msg,
+ uint8_t code,
+ uint32_t s_eid,
+ uint32_t d_eid,
+ qosspec_t qs,
+ int response,
+ size_t data_len)
+{
+ memset(msg, 0, sizeof(*msg));
+
+ msg->code = code;
+ msg->s_eid = hton32(s_eid);
+ msg->d_eid = hton32(d_eid);
+ msg->response = hton32(response);
+ msg->data_len = hton16((uint16_t) data_len);
+
+ if (code != POA_FLOW_REQ)
+ return;
+
+ msg->bandwidth = hton64(qs.bandwidth);
+ msg->delay = hton32(qs.delay);
+ msg->loss = hton32(qs.loss);
+ msg->ber = hton32(qs.ber);
+ msg->max_gap = hton32(qs.max_gap);
+ msg->timeout = hton32(qs.timeout);
+ msg->availability = qs.availability;
+ msg->service = qs.service;
+}
+
+void poa_mgmt_msg_qos(const struct poa_mgmt_msg * msg,
+ qosspec_t * qs)
+{
+ qs->bandwidth = ntoh64(msg->bandwidth);
+ qs->delay = ntoh32(msg->delay);
+ qs->loss = ntoh32(msg->loss);
+ qs->ber = ntoh32(msg->ber);
+ qs->max_gap = ntoh32(msg->max_gap);
+ qs->timeout = ntoh32(msg->timeout);
+ qs->availability = msg->availability;
+ qs->service = msg->service;
+}
+
diff --git a/src/lib/poa/eth.c b/src/lib/poa/eth.c
new file mode 100644
index 00000000..5e959ef6
--- /dev/null
+++ b/src/lib/poa/eth.c
@@ -0,0 +1,1987 @@
+/*
+ * Ouroboros - Copyright (C) 2016 - 2026
+ *
+ * Points of attachment (PoA) - Ethernet transport
+ *
+ * Dimitri Staessens <dimitri@ouroboros.rocks>
+ * Sander Vrijders <sander@ouroboros.rocks>
+ *
+ * This library is free software; you can redistribute it and/or
+ * modify it under the terms of the GNU Lesser General Public License
+ * version 2.1 as published by the Free Software Foundation.
+ *
+ * This library is distributed in the hope that it will be useful,
+ * but WITHOUT ANY WARRANTY; without even the implied warranty of
+ * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
+ * Lesser General Public License for more details.
+ *
+ * You should have received a copy of the GNU Lesser General Public
+ * License along with this library; if not, write to the Free Software
+ * Foundation, Inc., http://www.fsf.org/about/contact/.
+ */
+
+#if defined(__APPLE__)
+#define _BSD_SOURCE
+#define _DARWIN_C_SOURCE
+#elif defined(__FreeBSD__)
+#define __BSD_VISIBLE 1
+#elif defined(__linux__) || defined(__CYGWIN__)
+#ifndef _DEFAULT_SOURCE
+#define _DEFAULT_SOURCE
+#endif
+#else
+#ifndef _POSIX_C_SOURCE
+#define _POSIX_C_SOURCE 200809L
+#endif
+#endif
+
+#include "config.h"
+
+#define OUROBOROS_PREFIX "poa-eth"
+
+#include <ouroboros/endian.h>
+#include <ouroboros/errno.h>
+#include <ouroboros/hash.h>
+#include <ouroboros/logs.h>
+#include <ouroboros/pthread.h>
+#include <ouroboros/sockets.h>
+#include <ouroboros/time.h>
+
+#include "poa.h"
+
+#ifdef HAVE_RAW_SOCKETS
+#include <net/if.h>
+#include <netinet/in.h>
+#include <linux/if_ether.h>
+#include <linux/if_packet.h>
+#include <linux/netlink.h>
+#include <linux/gen_stats.h>
+#include <linux/pkt_sched.h>
+#include <linux/rtnetlink.h>
+#include <linux/sockios.h>
+#include <sys/ioctl.h>
+#include <sys/socket.h>
+#include <sys/uio.h>
+
+#include <ifaddrs.h>
+#elif defined(HAVE_BPF)
+#include <net/bpf.h>
+#include <net/if.h>
+#include <net/if_dl.h>
+#include <sys/ioctl.h>
+#include <sys/socket.h>
+#include <sys/uio.h>
+
+#include <fcntl.h>
+#include <ifaddrs.h>
+#elif defined(HAVE_NETMAP)
+#define NETMAP_WITH_LIBS
+#include <net/netmap_user.h>
+#include <net/if.h>
+#include <sys/ioctl.h>
+#include <sys/socket.h>
+
+#include <poll.h>
+#ifndef __linux__
+#include <net/if_dl.h>
+#include <ifaddrs.h>
+#endif
+#endif
+
+#include <errno.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <unistd.h>
+
+#define ETH_TYPE_LEN_SIZE sizeof(uint16_t)
+#define ETH_HDR_SIZE (2 * POA_MAC_SIZE + ETH_TYPE_LEN_SIZE)
+#define POA_HDR_SIZE (sizeof(struct poa_hdr))
+#define ETH_HDR_TOT_SIZE (ETH_HDR_SIZE + POA_HDR_SIZE)
+
+#define BPF_DEV_MAX 256 /* /dev/bpfN units to try */
+/* Wait for the link to come back before reading it again. */
+#define ETH_DOWN_TIMEO 100 /* ms */
+/* Budget for a qdisc dump; the send path must not wait on netlink. */
+#define POA_QDISC_TIMEO 5 /* ms */
+
+#ifndef ETH_MAX_MTU /* In if_ether.h as of Linux 4.10. */
+#define ETH_MAX_MTU 0xFFFFU
+#endif
+
+struct poa_hdr {
+ uint16_t eid;
+ uint16_t len;
+ uint8_t hcs;
+} __attribute__((packed));
+
+struct eth_hdr {
+ uint8_t dst[POA_MAC_SIZE];
+ uint8_t src[POA_MAC_SIZE];
+ uint16_t ethertype;
+ struct poa_hdr poa;
+} __attribute__((packed));
+
+
+struct eth_priv {
+ struct poa * poa;
+
+ int s_fd; /* raw socket or bpf device */
+ uint16_t ethertype; /* network order */
+ uint8_t hw_addr[POA_MAC_SIZE];
+ uint32_t mtu; /* device payload capacity */
+ int ifindex; /* link id; 0 where unknown */
+
+ pthread_t reader;
+ bool running;
+#ifdef HAVE_RAW_SOCKETS
+ struct sockaddr_ll device;
+#elif defined(HAVE_BPF)
+ size_t blen; /* bpf device buffer size */
+#elif defined(HAVE_NETMAP)
+ struct nm_desc * nmd;
+ struct pollfd poll_in;
+ struct pollfd poll_out;
+#endif
+ /* The kernel zeroes its counters on read, so accumulate. */
+ size_t kern_rcv;
+ size_t kern_drp;
+#ifdef HAVE_RAW_SOCKETS
+ /* Qdisc depth: netlink descriptor, its try-lock and cache. */
+ int nl_fd;
+ uint8_t nl_busy;
+ size_t nl_pkt;
+ size_t nl_byt;
+ uint64_t nl_time;
+#endif
+};
+
+struct eth_query {
+ struct list_head next;
+ uint8_t hash[POA_QUERY_HLEN];
+ /* The resolve's constraints; replies must satisfy them. */
+ uint16_t ethertype;
+ char c_dev[DEV_NAME_SIZE + 1];
+ uint8_t c_mac[POA_MAC_SIZE];
+ bool replied;
+ char dev[DEV_NAME_SIZE + 1];
+ uint8_t src_mac[POA_MAC_SIZE];
+ uint8_t mac[POA_MAC_SIZE];
+ uint16_t r_ethertype;
+};
+
+static struct {
+ pthread_once_t once;
+
+ struct llist pending;
+ pthread_mutex_t mtx;
+ pthread_cond_t cond;
+} queries = { .once = PTHREAD_ONCE_INIT };
+
+static void eth_hdr_ser(const struct eth_priv * priv,
+ struct eth_hdr * hdr,
+ const uint8_t * dst,
+ uint32_t eid,
+ size_t len)
+{
+ memcpy(hdr->dst, dst, POA_MAC_SIZE);
+ memcpy(hdr->src, priv->hw_addr, POA_MAC_SIZE);
+
+ hdr->ethertype = priv->ethertype;
+ hdr->poa.eid = htons((uint16_t) eid);
+ hdr->poa.len = htons((uint16_t) len);
+ mem_hash(HASH_CRC8, &hdr->poa.hcs,
+ (uint8_t *) &hdr->poa.eid, 2 * sizeof(uint16_t));
+}
+
+/* Oriented from us to the sender: our PoA, then their MAC. */
+static void frame_to_addr(const struct eth_priv * priv,
+ const struct eth_hdr * hdr,
+ struct poa_addr * addr)
+{
+ memset(addr, 0, sizeof(*addr));
+
+ addr->type = priv->poa->type;
+
+ addr->eth.src = priv->poa->local.eth.src;
+
+ memcpy(addr->eth.dst.mac, hdr->src, POA_MAC_SIZE);
+
+ addr->eth.dst.ethertype = ntohs(priv->ethertype);
+}
+
+static bool frame_is_for_us(const struct eth_priv * priv,
+ const uint8_t * dst)
+{
+ static const uint8_t bc[POA_MAC_SIZE] =
+ { 0xff, 0xff, 0xff, 0xff, 0xff, 0xff };
+
+ if (memcmp(dst, priv->hw_addr, POA_MAC_SIZE) == 0)
+ return true;
+
+ return memcmp(dst, bc, POA_MAC_SIZE) == 0;
+}
+
+/*
+ * Parse and validate a frame; on success *eid and *plen locate the
+ * payload. Errors are silent: anyone can spray an interface.
+ */
+static int frame_parse(const struct eth_priv * priv,
+ const uint8_t * buf,
+ size_t len,
+ uint32_t * eid,
+ size_t * plen)
+{
+ const struct eth_hdr * hdr = (const struct eth_hdr *) buf;
+ uint8_t hcs;
+
+ if (len < ETH_HDR_TOT_SIZE)
+ return -1;
+
+ if (!frame_is_for_us(priv, hdr->dst))
+ return -1;
+
+ if (hdr->ethertype != priv->ethertype)
+ return -1;
+
+ mem_hash(HASH_CRC8, &hcs,
+ (const uint8_t *) &hdr->poa.eid, 2 * sizeof(uint16_t));
+
+ if (hcs != hdr->poa.hcs)
+ return -1;
+
+ *eid = ntohs(hdr->poa.eid);
+
+ *plen = ntohs(hdr->poa.len);
+ if (*plen > len - ETH_HDR_TOT_SIZE)
+ return -1;
+
+ return 0;
+}
+
+#define ETH_QUERY_TIMEO 1900 /* ms total budget */
+#define ETH_QUERY_RETRIES 3 /* retransmits, 4 attempts total */
+
+/* A resolve in progress; a reply fills every matching entry. */
+static void queries_init(void)
+{
+ pthread_condattr_t cattr;
+
+ llist_init(&queries.pending);
+
+ pthread_mutex_init(&queries.mtx, NULL);
+
+ pthread_condattr_init(&cattr);
+#ifndef __APPLE__
+ pthread_condattr_setclock(&cattr, PTHREAD_COND_CLOCK);
+#endif
+ pthread_cond_init(&queries.cond, &cattr);
+ pthread_condattr_destroy(&cattr);
+}
+
+static void eth_query_cleanup(void * o)
+{
+ struct eth_query * q = (struct eth_query *) o;
+
+ pthread_mutex_lock(&queries.mtx);
+ llist_del(&q->next, &queries.pending);
+ pthread_mutex_unlock(&queries.mtx);
+}
+
+static void eth_query_arr(struct poa * poa,
+ const struct poa_addr * src,
+ const uint8_t * hash)
+{
+ uint8_t buf[sizeof(struct poa_mgmt_msg)
+ + POA_QUERY_HLEN];
+ struct poa_mgmt_msg * msg = (struct poa_mgmt_msg *) buf;
+
+ if (!poa_has_name(hash))
+ return;
+
+ poa_mgmt_msg_ser(msg, POA_NAME_REPLY, 0, 0, qos_raw, 0, POA_QUERY_HLEN);
+
+ memcpy(buf + sizeof(*msg), hash, POA_QUERY_HLEN);
+
+ if (poa->ops->poa_send_mgmt(poa, src, buf, sizeof(buf)) < 0)
+ return; /* the requester retransmits */
+
+ POA_STAT_BUMP(poa, rep_tx);
+}
+
+static bool mac_is_zero(const uint8_t * mac)
+{
+ static const uint8_t zero[POA_MAC_SIZE] = { 0 };
+
+ return memcmp(mac, zero, POA_MAC_SIZE) == 0;
+}
+
+static bool eth_query_is_match(const struct eth_query * q,
+ const struct poa * poa)
+{
+ if (q->ethertype != 0 &&
+ q->ethertype != poa->local.eth.src.ethertype)
+ return false;
+
+ if (q->c_dev[0] != '\0' &&
+ strcmp(q->c_dev, poa->local.eth.src.dev) != 0)
+ return false;
+
+ if (mac_is_zero(q->c_mac))
+ return true;
+
+ return memcmp(q->c_mac, poa->local.eth.src.mac, POA_MAC_SIZE) == 0;
+}
+
+static void eth_reply_arr(struct poa * poa,
+ const struct poa_addr * src,
+ const uint8_t * hash)
+{
+ struct list_head * p;
+
+ pthread_once(&queries.once, queries_init);
+
+ pthread_mutex_lock(&queries.mtx);
+
+ list_for_each(p, &queries.pending.list) {
+ struct eth_query * q;
+
+ q = list_entry(p, struct eth_query, next);
+ if (q->replied || memcmp(q->hash, hash, POA_QUERY_HLEN) != 0)
+ continue;
+
+ if (!eth_query_is_match(q, poa))
+ continue;
+
+ memcpy(q->mac, src->eth.dst.mac, POA_MAC_SIZE);
+ memcpy(q->src_mac, src->eth.src.mac, POA_MAC_SIZE);
+ strcpy(q->dev, src->eth.src.dev);
+
+ q->r_ethertype = poa->local.eth.src.ethertype;
+
+ q->replied = true;
+ }
+
+ pthread_cond_broadcast(&queries.cond);
+
+ pthread_mutex_unlock(&queries.mtx);
+}
+
+/* Name query frames are handled in the transport. */
+static void eth_rx_mgmt(struct poa * poa,
+ const struct poa_addr * src,
+ const uint8_t * buf,
+ size_t len)
+{
+ const struct poa_mgmt_msg * msg;
+ const uint8_t * hash;
+
+ msg = (const struct poa_mgmt_msg *) buf;
+ if (len < sizeof(*msg)) {
+ poa_rx_mgmt(poa, src, buf, len);
+ return;
+ }
+
+ if (msg->code != POA_NAME_QUERY && msg->code != POA_NAME_REPLY) {
+ poa_rx_mgmt(poa, src, buf, len);
+ return;
+ }
+
+ if (ntoh16(msg->data_len) != POA_QUERY_HLEN)
+ return; /* malformed */
+
+ if (len < sizeof(*msg) + POA_QUERY_HLEN)
+ return; /* malformed */
+
+ hash = buf + sizeof(*msg);
+
+ if (msg->code == POA_NAME_QUERY) {
+ POA_STAT_BUMP(poa, qry_rx);
+ eth_query_arr(poa, src, hash);
+ } else {
+ POA_STAT_BUMP(poa, rep_rx);
+ eth_reply_arr(poa, src, hash);
+ }
+}
+
+static int eth_query_send(const uint8_t * hash,
+ const struct poa_addr * addr)
+{
+ uint8_t buf[sizeof(struct poa_mgmt_msg)
+ + POA_QUERY_HLEN];
+ struct poa_mgmt_msg * msg = (struct poa_mgmt_msg *) buf;
+ struct poa_addr bcast;
+
+ bcast = *addr;
+
+ memset(bcast.eth.dst.mac, 0xff, POA_MAC_SIZE);
+
+ poa_mgmt_msg_ser(msg, POA_NAME_QUERY, 0, 0, qos_raw, 0, POA_QUERY_HLEN);
+
+ memcpy(buf + sizeof(*msg), hash, POA_QUERY_HLEN);
+
+ return poa_bcast_mgmt(&bcast, buf, sizeof(buf));
+}
+
+/* Complete addr for dst by broadcast query; the poa_query op. */
+static int eth_query(const char * dst,
+ const struct timespec * timeo,
+ struct poa_addr * addr)
+{
+ struct timespec dflt = TIMESPEC_INIT_MS(ETH_QUERY_TIMEO);
+ struct timespec rintv = TIMESPEC_INIT_MS(ETH_QUERY_TIMEO
+ / (ETH_QUERY_RETRIES + 1));
+ struct eth_query q;
+ struct timespec abstime;
+ struct timespec now;
+ struct timespec dl;
+ time_t rintv_ns = (time_t) TS_TO_UINT64(rintv);
+ time_t diff;
+ uint8_t hash[POA_QUERY_HLEN];
+ int n;
+ int err = -ETIMEDOUT;
+
+ if (strlen(addr->eth.src.dev) > DEV_NAME_SIZE)
+ return -EINVAL;
+
+ addr->eth.src.ethertype = addr->eth.dst.ethertype;
+
+ /* The destination is set; a zero ethertype cannot be sent. */
+ if (!mac_is_zero(addr->eth.dst.mac))
+ return addr->eth.dst.ethertype < 0x0600 ? -EINVAL : 0;
+
+ str_hash(HASH_SHA3_256, hash, dst);
+
+ pthread_once(&queries.once, queries_init);
+
+ memset(&q, 0, sizeof(q));
+ memcpy(q.hash, hash, POA_QUERY_HLEN);
+
+ q.ethertype = addr->eth.dst.ethertype;
+
+ memcpy(q.c_mac, addr->eth.src.mac, POA_MAC_SIZE);
+ strcpy(q.c_dev, addr->eth.src.dev);
+
+ pthread_mutex_lock(&queries.mtx);
+ llist_add(&q.next, &queries.pending);
+ pthread_mutex_unlock(&queries.mtx);
+
+ pthread_cleanup_push(eth_query_cleanup, &q);
+
+ clock_gettime(PTHREAD_COND_CLOCK, &now);
+ ts_add(&now, timeo != NULL ? timeo : &dflt, &abstime);
+
+ while (ts_diff_ns(&now, &abstime) < 0) {
+ n = eth_query_send(hash, addr);
+ if (n < 0) {
+ err = n;
+ break;
+ }
+
+ if (n == 0) {
+ err = -EPERM;
+ break;
+ }
+
+ ts_add(&now, &rintv, &dl);
+
+ if (ts_diff_ns(&dl, &abstime) > 0)
+ dl = abstime;
+
+ pthread_mutex_lock(&queries.mtx);
+
+ pthread_cleanup_push(__cleanup_mutex_unlock, &queries.mtx);
+
+ while (!q.replied) {
+ if (pthread_cond_timedwait(&queries.cond, &queries.mtx,
+ &dl) == ETIMEDOUT)
+ break;
+ }
+
+ if (q.replied) {
+ memcpy(addr->eth.dst.mac, q.mac, POA_MAC_SIZE);
+ memcpy(addr->eth.src.mac, q.src_mac, POA_MAC_SIZE);
+ strcpy(addr->eth.src.dev, q.dev);
+
+ addr->eth.dst.ethertype = q.r_ethertype;
+ addr->eth.src.ethertype = q.r_ethertype;
+
+ err = 0;
+ }
+
+ pthread_cleanup_pop(true);
+
+ if (err == 0)
+ break;
+
+ clock_gettime(PTHREAD_COND_CLOCK, &now);
+
+ diff = ts_diff_ns(&now, &abstime);
+ if (diff > -rintv_ns) /* skip the runt attempt */
+ break;
+ }
+
+ pthread_cleanup_pop(true);
+
+ return err;
+}
+
+static void eth_spec(const struct poa * poa,
+ struct poa_spec * spec)
+{
+ spec->type = poa->type;
+ spec->eth = poa->local.eth.src;
+}
+
+static bool eth_has_id(const struct poa * poa,
+ const struct poa_spec * spec)
+{
+ if (strnlen(spec->eth.dev, sizeof(spec->eth.dev)) > DEV_NAME_SIZE)
+ return false;
+
+ if (strcmp(poa->local.eth.src.dev, spec->eth.dev) != 0)
+ return false;
+
+ return poa->local.eth.src.ethertype == spec->eth.ethertype;
+}
+
+/* Our end of the link; an unnamed one matches any. */
+static bool eth_is_src(const struct poa * poa,
+ const struct eth_poa * src)
+{
+ if (src->dev[0] != '\0' &&
+ strcmp(poa->local.eth.src.dev, src->dev) != 0)
+ return false;
+
+ if (mac_is_zero(src->mac))
+ return true;
+
+ return memcmp(poa->local.eth.src.mac, src->mac,
+ POA_MAC_SIZE) == 0;
+}
+
+/* An ethertype of 0 in dst matches any, for broadcast queries. */
+static bool eth_match(const struct poa * poa,
+ const struct poa_addr * dst)
+{
+ uint16_t et = dst->eth.dst.ethertype;
+
+ if (et != 0 && et != poa->local.eth.src.ethertype)
+ return false;
+
+ return eth_is_src(poa, &dst->eth.src);
+}
+
+static bool eth_link_match(const struct poa * poa,
+ int id)
+{
+ struct eth_priv * priv = (struct eth_priv *) poa->priv;
+
+ return priv->ifindex == id;
+}
+
+#ifdef HAVE_RAW_SOCKETS
+
+/* MSG_DONTWAIT: the reader blocks on this socket. */
+static int eth_sendv(struct eth_priv * priv,
+ const uint8_t * dst,
+ uint32_t eid,
+ const uint8_t * body,
+ size_t len,
+ bool block,
+ const struct timespec * abstime)
+{
+ struct eth_hdr hdr;
+ struct msghdr msg;
+ struct iovec iov[2];
+ int ret;
+
+ if (len > priv->mtu - POA_HDR_SIZE)
+ return -EMSGSIZE;
+
+ eth_hdr_ser(priv, &hdr, dst, eid, len);
+
+ iov[0].iov_base = &hdr;
+ iov[0].iov_len = ETH_HDR_TOT_SIZE;
+ iov[1].iov_base = (void *) body;
+ iov[1].iov_len = len;
+
+ memset(&msg, 0, sizeof(msg));
+
+ msg.msg_name = &priv->device;
+ msg.msg_namelen = sizeof(priv->device);
+ msg.msg_iov = iov;
+ msg.msg_iovlen = len > 0 ? 2 : 1;
+ while (sendmsg(priv->s_fd, &msg, MSG_DONTWAIT) < 0) {
+ if (errno != EAGAIN && errno != EWOULDBLOCK)
+ return -EIO;
+
+ if (!block)
+ return -EAGAIN;
+
+ ret = poa_wait_out(priv->s_fd, abstime);
+ if (ret < 0)
+ return ret;
+ }
+
+ return 0;
+}
+
+static void * eth_reader(void * o)
+{
+ struct poa * poa = (struct poa *) o;
+ struct eth_priv * priv = (struct eth_priv *) poa->priv;
+ struct timespec down = TIMESPEC_INIT_MS(ETH_DOWN_TIMEO);
+ uint8_t * buf;
+ size_t bufsz;
+
+ bufsz = ETH_HDR_SIZE + priv->mtu;
+
+ buf = malloc(bufsz);
+ if (buf == NULL)
+ return (void *) -1;
+
+ pthread_cleanup_push(free, buf);
+
+ while (true) {
+ struct ssm_pk_buff * spb;
+ struct sockaddr_ll from;
+ socklen_t flen;
+ struct poa_addr src;
+ ssize_t n;
+ uint32_t eid;
+ size_t plen;
+ const uint8_t * body;
+
+ flen = sizeof(from);
+
+ n = recvfrom(priv->s_fd, buf, bufsz, 0,
+ (struct sockaddr *) &from, &flen);
+
+ if (n < 0) {
+ if (errno == EINTR)
+ continue;
+
+ POA_STAT_BUMP(poa, rcv_fail);
+
+ if (errno == ENETDOWN) {
+ nanosleep(&down, NULL);
+ continue;
+ }
+
+ log_err("Reader on %s stopped: %s.",
+ poa->local.eth.src.dev,
+ strerror(errno));
+ break;
+ }
+
+ if (from.sll_pkttype == PACKET_OUTGOING)
+ continue;
+
+ if (frame_parse(priv, buf, (size_t) n, &eid, &plen) < 0)
+ continue;
+
+ body = buf + ETH_HDR_TOT_SIZE;
+
+ if (eid == POA_MGMT_EID) {
+ frame_to_addr(priv, (struct eth_hdr *) buf, &src);
+ eth_rx_mgmt(poa, &src, body, plen);
+ continue;
+ }
+
+ if (poa_spb_reserve(&spb, plen) < 0) {
+ POA_STAT_BUMP(poa, buf_fail);
+ continue;
+ }
+
+ memcpy(ssm_pk_buff_head(spb), body, plen);
+
+ poa_rx_pkt(poa, eid, spb);
+ }
+
+ pthread_cleanup_pop(true);
+
+ return (void *) 0;
+}
+
+/*
+ * One netlink socket for the whole subsystem: RTMGRP_LINK delivers the
+ * events of every interface anyway, so a socket per PoA only added
+ * discards.
+ */
+int poa_monitor_open(void)
+{
+ struct sockaddr_nl sa;
+ int fd;
+
+ memset(&sa, 0, sizeof(sa));
+
+ sa.nl_family = AF_NETLINK;
+ sa.nl_groups = RTMGRP_LINK;
+
+ fd = socket(AF_NETLINK, SOCK_RAW, NETLINK_ROUTE);
+ if (fd < 0)
+ return -1;
+
+ if (bind(fd, (struct sockaddr *) &sa, sizeof(sa)) < 0) {
+ close(fd);
+ return -1;
+ }
+
+ return fd;
+}
+
+void poa_monitor_read(int fd)
+{
+ uint8_t buf[4096];
+ struct nlmsghdr * h;
+ ssize_t n;
+
+ n = recv(fd, buf, sizeof(buf), 0);
+ if (n < 0)
+ return;
+
+ for (h = (struct nlmsghdr *) buf;
+ NLMSG_OK(h, (unsigned int) n);
+ h = NLMSG_NEXT(h, n)) {
+ struct ifinfomsg * ifi;
+ unsigned int usable;
+ bool up;
+ size_t cnt;
+
+ if (h->nlmsg_type == NLMSG_DONE)
+ break;
+
+ if (h->nlmsg_type != RTM_NEWLINK)
+ continue;
+
+ ifi = NLMSG_DATA(h);
+
+ usable = ifi->ifi_flags & (IFF_UP | IFF_RUNNING);
+ up = usable == (IFF_UP | IFF_RUNNING);
+
+ cnt = poa_link_updown(ifi->ifi_index, up);
+ if (cnt > 0)
+ log_info("Link %d %s, %zu flows.", ifi->ifi_index,
+ up ? "up" : "down", cnt);
+ }
+}
+
+static int eth_dev_info(const char * dev,
+ int * idx,
+ uint8_t * mac,
+ uint32_t * mtu)
+{
+ struct ifreq ifr;
+ int fd;
+
+ if (strlen(dev) >= IFNAMSIZ)
+ return -EINVAL;
+
+ *idx = if_nametoindex(dev);
+ if (*idx == 0) {
+ log_err("Failed to find device %s.", dev);
+ return -ENODEV;
+ }
+
+ fd = socket(AF_UNIX, SOCK_STREAM, 0);
+ if (fd < 0)
+ return -EIO;
+
+ memset(&ifr, 0, sizeof(ifr));
+ strcpy(ifr.ifr_name, dev);
+
+ if (ioctl(fd, SIOCGIFHWADDR, &ifr) < 0) {
+ log_err("Failed to get hardware address of %s.", dev);
+ goto fail_ioctl;
+ }
+
+ memcpy(mac, ifr.ifr_hwaddr.sa_data, POA_MAC_SIZE);
+
+ if (ioctl(fd, SIOCGIFMTU, &ifr) < 0) {
+ log_err("Failed to get MTU of %s.", dev);
+ goto fail_ioctl;
+ }
+
+ close(fd);
+
+ *mtu = MIN(MIN(ETH_MAX_MTU, POA_ETH_RD_BUF), (uint32_t) ifr.ifr_mtu);
+ if (memcmp(dev, "lo", 2) == 0 && *mtu > POA_ETH_LO_MTU)
+ *mtu = POA_ETH_LO_MTU;
+
+ return 0;
+
+ fail_ioctl:
+ close(fd);
+ return -EIO;
+}
+
+/* SO_RCVBUFFORCE bypasses rmem_max; SO_RCVBUF is the fallback. */
+static void eth_set_rcvbuf(int fd,
+ int rcvbuf)
+{
+ if (setsockopt(fd, SOL_SOCKET, SO_RCVBUFFORCE,
+ &rcvbuf, sizeof(rcvbuf)) == 0)
+ return;
+
+ if (setsockopt(fd, SOL_SOCKET, SO_RCVBUF, &rcvbuf, sizeof(rcvbuf)) < 0)
+ log_info("Failed to set SO_RCVBUF to %d.", rcvbuf);
+}
+
+/* The send buffer holds at least one frame, or sendmsg cannot fit a PDU. */
+static int eth_socket(struct eth_priv * priv,
+ int idx,
+ const uint8_t * mac)
+{
+ int sndbuf;
+ int rcvbuf;
+#ifdef POA_ETH_QDISC_BYPASS
+ int bypass = 1;
+#endif
+
+ memset(&priv->device, 0, sizeof(priv->device));
+
+ priv->device.sll_ifindex = idx;
+ priv->device.sll_family = AF_PACKET;
+ priv->device.sll_halen = POA_MAC_SIZE;
+ priv->device.sll_protocol = htons(ETH_P_ALL);
+
+ memcpy(priv->device.sll_addr, mac, POA_MAC_SIZE);
+ memcpy(priv->hw_addr, mac, POA_MAC_SIZE);
+
+ priv->s_fd = socket(AF_PACKET, SOCK_RAW, priv->ethertype);
+ if (priv->s_fd < 0) {
+ log_err("Failed to create socket: %s.", strerror(errno));
+ return -1;
+ }
+
+#ifdef POA_ETH_QDISC_BYPASS
+ if (setsockopt(priv->s_fd, SOL_PACKET, PACKET_QDISC_BYPASS,
+ &bypass, sizeof(bypass)) < 0)
+ log_info("Qdisc bypass not supported.");
+#endif
+
+ sndbuf = POA_ETH_SNDBUF;
+ if (sndbuf > 0) {
+ sndbuf = MAX(sndbuf, (int) (ETH_HDR_SIZE + priv->mtu));
+
+ if (setsockopt(priv->s_fd, SOL_SOCKET, SO_SNDBUF,
+ &sndbuf, sizeof(sndbuf)) < 0)
+ log_info("Failed to set SO_SNDBUF to %d.", sndbuf);
+ }
+
+ rcvbuf = POA_ETH_RCVBUF;
+ if (rcvbuf > 0)
+ eth_set_rcvbuf(priv->s_fd, rcvbuf);
+
+ if (bind(priv->s_fd, (struct sockaddr *) &priv->device,
+ sizeof(priv->device)) < 0) {
+ log_err("Failed to bind socket to %d.", idx);
+ close(priv->s_fd);
+ return -1;
+ }
+
+ return 0;
+}
+
+/*
+ * Pull qlen and backlog from the nested TCA_STATS2. The top-level
+ * TCA_STATS shares TCA_STATS_QUEUE's id but carries a wholly
+ * different struct, so descend first; never match on id alone.
+ */
+static void eth_qdisc_parse(struct rtattr * rta,
+ size_t rlen,
+ size_t * byt,
+ size_t * pkt)
+{
+ struct gnet_stats_queue q;
+ struct rtattr * in;
+ size_t ilen;
+
+ for (; RTA_OK(rta, rlen); rta = RTA_NEXT(rta, rlen)) {
+ if (rta->rta_type != TCA_STATS2)
+ continue;
+
+ in = (struct rtattr *) RTA_DATA(rta);
+ ilen = RTA_PAYLOAD(rta);
+ for (; RTA_OK(in, ilen); in = RTA_NEXT(in, ilen)) {
+ if (in->rta_type != TCA_STATS_QUEUE)
+ continue;
+
+ if (RTA_PAYLOAD(in) < sizeof(q))
+ continue;
+
+ memcpy(&q, RTA_DATA(in), sizeof(q));
+
+ *byt = q.backlog;
+ *pkt = q.qlen;
+ }
+ }
+}
+
+/*
+ * Egress backlog of the device's root qdisc, in bytes and packets,
+ * as the kernel queues them. The caller owns the netlink descriptor
+ * and serialises the query. An early end of dump reports failure.
+ */
+static int eth_qdisc_backlog(int fd,
+ int ifindex,
+ size_t * byt,
+ size_t * pkt)
+{
+ struct {
+ struct nlmsghdr nh;
+ struct tcmsg tc;
+ } req;
+ struct nlmsghdr * nh;
+ struct rtattr * rta;
+ struct tcmsg * tc;
+ char buf[16384];
+ ssize_t len;
+ int ret = -1;
+
+ if (fd < 0)
+ goto fail;
+
+ memset(&req, 0, sizeof(req));
+
+ req.nh.nlmsg_len = NLMSG_LENGTH(sizeof(req.tc));
+ req.nh.nlmsg_type = RTM_GETQDISC;
+ req.nh.nlmsg_flags = NLM_F_REQUEST | NLM_F_DUMP;
+ req.nh.nlmsg_seq = 1;
+ req.tc.tcm_family = AF_UNSPEC;
+ req.tc.tcm_ifindex = ifindex;
+ if (send(fd, &req, req.nh.nlmsg_len, 0) < 0)
+ goto fail;
+
+ *byt = 0;
+ *pkt = 0;
+
+ while ((len = recv(fd, buf, sizeof(buf), 0)) > 0) {
+ nh = (struct nlmsghdr *) buf;
+ for (; NLMSG_OK(nh, len); nh = NLMSG_NEXT(nh, len)) {
+ if (nh->nlmsg_type == NLMSG_DONE)
+ goto done;
+
+ if (nh->nlmsg_type == NLMSG_ERROR)
+ goto fail;
+
+ if (nh->nlmsg_type != RTM_NEWQDISC)
+ continue;
+
+ tc = (struct tcmsg *) NLMSG_DATA(nh);
+ if (tc->tcm_ifindex != ifindex)
+ continue;
+
+ if (tc->tcm_parent != TC_H_ROOT)
+ continue;
+
+ rta = (struct rtattr *)
+ ((char *) tc + NLMSG_ALIGN(sizeof(*tc)));
+
+ eth_qdisc_parse(rta, nh->nlmsg_len
+ - NLMSG_LENGTH(sizeof(*tc)),
+ byt, pkt);
+ }
+ }
+
+ goto fail; /* early end: a zero would read as empty */
+ done:
+ ret = 0;
+ fail:
+ return ret;
+}
+
+/*
+ * A netlink descriptor for the qdisc query. SO_RCVTIMEO bounds the
+ * dump: this is read from the send path, and a reply that never
+ * arrives must not park a sender thread.
+ */
+static int eth_qdisc_open(void)
+{
+ struct sockaddr_nl sa;
+ struct timeval tv = TIMEVAL_INIT_MS(POA_QDISC_TIMEO);
+ int fd;
+
+ fd = socket(AF_NETLINK, SOCK_RAW | SOCK_CLOEXEC, NETLINK_ROUTE);
+ if (fd < 0)
+ return -1;
+
+ memset(&sa, 0, sizeof(sa));
+
+ sa.nl_family = AF_NETLINK;
+ if (bind(fd, (struct sockaddr *) &sa, sizeof(sa)) < 0)
+ goto fail;
+
+ if (setsockopt(fd, SOL_SOCKET, SO_RCVTIMEO, &tv, sizeof(tv)) < 0)
+ goto fail;
+
+ return fd;
+ fail:
+ close(fd);
+
+ return -1;
+}
+
+/*
+ * Queue depth in packets, straight from the qdisc. Gated like the
+ * transport's own depth and skipped when another sender is already
+ * asking, so the send path never waits on netlink.
+ */
+static int eth_qpkts(struct poa * poa,
+ size_t * pkts,
+ size_t * byts)
+{
+ struct eth_priv * priv = (struct eth_priv *) poa->priv;
+ struct timespec now;
+ uint64_t ns;
+ size_t byt;
+ size_t pkt;
+
+ if (priv->nl_fd < 0)
+ return -1;
+
+ clock_gettime(PTHREAD_COND_CLOCK, &now);
+
+ ns = TS_TO_UINT64(now);
+ if (ns - LOAD_RELAXED(&priv->nl_time) < POA_QLEN_GATE)
+ goto cached;
+
+ if (__atomic_exchange_n(&priv->nl_busy, 1, __ATOMIC_ACQUIRE) != 0)
+ goto cached;
+
+ if (eth_qdisc_backlog(priv->nl_fd, priv->ifindex, &byt, &pkt) == 0) {
+ STORE_RELAXED(&priv->nl_pkt, pkt);
+ STORE_RELAXED(&priv->nl_byt, byt);
+ STORE_RELAXED(&priv->nl_time, ns);
+ }
+
+ __atomic_store_n(&priv->nl_busy, 0, __ATOMIC_RELEASE);
+ cached:
+ if (LOAD_RELAXED(&priv->nl_time) == 0) /* nothing measured yet */
+ return -1;
+
+ *pkts = LOAD_RELAXED(&priv->nl_pkt);
+ *byts = LOAD_RELAXED(&priv->nl_byt);
+
+ return 0;
+}
+
+static int eth_attach(struct poa * poa,
+ const struct poa_spec * spec)
+{
+ const struct eth_poa * c = &spec->eth;
+ struct eth_priv * priv;
+ uint8_t mac[POA_MAC_SIZE];
+ uint32_t mtu;
+ int idx;
+ int err;
+
+ if (strnlen(c->dev, sizeof(c->dev)) > DEV_NAME_SIZE)
+ return -EINVAL;
+
+ if (c->ethertype < 0x0600 || c->ethertype == 0xFFFF) {
+ log_err("Invalid Ethertype 0x%x.", c->ethertype);
+ return -EINVAL;
+ }
+
+ priv = malloc(sizeof(*priv));
+ if (priv == NULL)
+ return -ENOMEM;
+
+ memset(priv, 0, sizeof(*priv));
+
+ priv->poa = poa;
+ priv->nl_fd = eth_qdisc_open(); /* optional; socket fallback */
+
+ priv->ethertype = htons(c->ethertype);
+
+ err = eth_dev_info(c->dev, &idx, mac, &mtu);
+ if (err < 0)
+ goto fail_conf;
+
+ priv->mtu = mtu;
+ if (eth_socket(priv, idx, mac) < 0)
+ goto fail_conf;
+
+ poa->priv = priv;
+ poa->local.type = poa->type;
+ poa->local.eth.src.ethertype = c->ethertype;
+ priv->ifindex = idx;
+
+ memcpy(poa->local.eth.src.mac, mac, POA_MAC_SIZE);
+ strcpy(poa->local.eth.src.dev, c->dev);
+
+ return 0;
+
+ fail_conf:
+ if (priv->nl_fd >= 0)
+ close(priv->nl_fd);
+
+ free(priv);
+
+ return -EIO;
+}
+
+static void eth_detach(struct poa * poa)
+{
+ struct eth_priv * priv = (struct eth_priv *) poa->priv;
+
+ if (priv == NULL)
+ return;
+
+ close(priv->s_fd);
+
+ if (priv->nl_fd >= 0)
+ close(priv->nl_fd);
+
+ free(priv);
+
+ poa->priv = NULL;
+}
+
+static uint32_t eth_mtu(struct poa * poa,
+ const struct poa_addr * dst)
+{
+ struct eth_priv * priv = (struct eth_priv *) poa->priv;
+
+ (void) dst;
+
+ return priv->mtu - POA_HDR_SIZE;
+}
+
+/* All flows on the PoA share the socket, so this is aggregate. */
+static size_t eth_qlen(struct poa * poa)
+{
+#ifdef SIOCOUTQ
+ struct eth_priv * priv = (struct eth_priv *) poa->priv;
+ int qlen;
+
+ qlen = 0;
+ if (ioctl(priv->s_fd, SIOCOUTQ, &qlen) < 0)
+ return 0;
+
+ return (size_t) qlen;
+#else
+ (void) poa;
+
+ return 0;
+#endif
+}
+
+/*
+ * PACKET_STATISTICS zeroes the kernel counters on read, so totals
+ * accumulate here; relaxed atomics allow concurrent RIB reads.
+ * The qdisc depth rides the sender's guarded path: one dump at a time.
+ */
+static int eth_rib(struct poa * poa,
+ char * buf,
+ size_t len)
+{
+ struct eth_priv * priv = (struct eth_priv *) poa->priv;
+ struct tpacket_stats ts;
+ socklen_t optlen;
+ size_t sndbuf = 0;
+ size_t rcvbuf = 0;
+ size_t qd_byt;
+ size_t qd_pkt;
+ int val;
+ int size;
+
+ optlen = sizeof(val);
+ if (getsockopt(priv->s_fd, SOL_SOCKET, SO_SNDBUF, &val, &optlen) == 0)
+ sndbuf = (size_t) val;
+
+ optlen = sizeof(val);
+ if (getsockopt(priv->s_fd, SOL_SOCKET, SO_RCVBUF, &val, &optlen) == 0)
+ rcvbuf = (size_t) val;
+
+ optlen = sizeof(ts);
+
+ if (getsockopt(priv->s_fd, SOL_PACKET, PACKET_STATISTICS,
+ &ts, &optlen) == 0) {
+ FETCH_ADD_RELAXED(&priv->kern_rcv, ts.tp_packets);
+ FETCH_ADD_RELAXED(&priv->kern_drp, ts.tp_drops);
+ }
+
+ if (eth_qpkts(poa, &qd_pkt, &qd_byt) < 0) {
+ qd_byt = 0;
+ qd_pkt = 0;
+ }
+
+ size = snprintf(buf, len,
+ "Socket sndbuf (bytes): %zu\n"
+ "Socket rcvbuf (bytes): %zu\n"
+ "Kernel packets received: %zu\n"
+ "Kernel packets dropped: %zu\n"
+ "Qdisc backlog (bytes): %zu\n"
+ "Qdisc backlog (packets): %zu\n",
+ sndbuf, rcvbuf,
+ LOAD_RELAXED(&priv->kern_rcv),
+ LOAD_RELAXED(&priv->kern_drp),
+ qd_byt, qd_pkt);
+ if (size < 0 || (size_t) size >= len)
+ return -1;
+
+ return size;
+}
+
+#elif defined(HAVE_BPF)
+
+/*
+ * BSD and macOS reach the link layer through a cloned /dev/bpf
+ * device, bound to an interface with BIOCSETIF. One device per PoA.
+ */
+
+static int eth_sendv(struct eth_priv * priv,
+ const uint8_t * dst,
+ uint32_t eid,
+ const uint8_t * body,
+ size_t len,
+ bool block,
+ const struct timespec * abstime)
+{
+ struct eth_hdr hdr;
+ struct iovec iov[2];
+ int n;
+ int ret;
+
+ if (len > priv->mtu - POA_HDR_SIZE)
+ return -EMSGSIZE;
+
+ eth_hdr_ser(priv, &hdr, dst, eid, len);
+
+ iov[0].iov_base = &hdr;
+ iov[0].iov_len = ETH_HDR_TOT_SIZE;
+ iov[1].iov_base = (void *) body;
+ iov[1].iov_len = len;
+ n = len > 0 ? 2 : 1;
+ while (writev(priv->s_fd, iov, n) < 0) {
+ if (errno != EAGAIN && errno != EWOULDBLOCK)
+ return -EIO;
+
+ if (!block)
+ return -EAGAIN;
+
+ ret = poa_wait_out(priv->s_fd, abstime);
+ if (ret < 0)
+ return ret;
+ }
+
+ return 0;
+}
+
+/* One read yields a batch of BPF_WORDALIGN'ed frames; walk all of them. */
+static void * eth_reader(void * o)
+{
+ struct poa * poa = (struct poa *) o;
+ struct eth_priv * priv = (struct eth_priv *) poa->priv;
+ uint8_t * buf;
+
+ buf = malloc(priv->blen);
+ if (buf == NULL)
+ return (void *) -1;
+
+ pthread_cleanup_push(free, buf);
+
+ while (true) {
+ uint8_t * p;
+ uint8_t * end;
+ ssize_t n;
+
+ n = read(priv->s_fd, buf, priv->blen);
+ if (n < 0) {
+ if (errno == EINTR)
+ continue;
+
+ POA_STAT_BUMP(poa, rcv_fail);
+ break;
+ }
+
+ p = buf;
+ end = buf + n;
+ while (p + sizeof(struct bpf_hdr) <= end) {
+ struct bpf_hdr * bh = (struct bpf_hdr *) p;
+ struct ssm_pk_buff * spb;
+ struct poa_addr src;
+ const uint8_t * frame;
+ const uint8_t * body;
+ uint32_t eid;
+ size_t plen;
+
+ frame = p + bh->bh_hdrlen;
+ if (frame + bh->bh_caplen > end)
+ break;
+
+ if (frame_parse(priv, frame, bh->bh_caplen,
+ &eid, &plen) < 0)
+ goto next;
+
+ body = frame + ETH_HDR_TOT_SIZE;
+
+ if (eid == POA_MGMT_EID) {
+ frame_to_addr(priv,
+ (const struct eth_hdr *) frame,
+ &src);
+ eth_rx_mgmt(poa, &src, body, plen);
+ goto next;
+ }
+
+ if (poa_spb_reserve(&spb, plen) < 0) {
+ POA_STAT_BUMP(poa, buf_fail);
+ goto next;
+ }
+
+ memcpy(ssm_pk_buff_head(spb), body, plen);
+
+ poa_rx_pkt(poa, eid, spb);
+ next:
+ p += BPF_WORDALIGN(bh->bh_hdrlen + bh->bh_caplen);
+ }
+ }
+
+ pthread_cleanup_pop(true);
+
+ return (void *) 0;
+}
+
+static int eth_dev_info(const char * dev,
+ uint8_t * mac,
+ uint32_t * mtu)
+{
+ struct ifaddrs * ifas;
+ struct ifaddrs * ifa;
+ struct ifreq ifr;
+ int fd;
+ int found = 0;
+
+ if (getifaddrs(&ifas) < 0) {
+ log_err("Failed to list interfaces.");
+ return -1;
+ }
+
+ for (ifa = ifas; ifa != NULL; ifa = ifa->ifa_next) {
+ struct sockaddr_dl * dl;
+
+ if (ifa->ifa_addr == NULL)
+ continue;
+
+ if (ifa->ifa_addr->sa_family != AF_LINK)
+ continue;
+
+ if (strcmp(ifa->ifa_name, dev) != 0)
+ continue;
+
+ dl = (struct sockaddr_dl *) ifa->ifa_addr;
+ if (dl->sdl_alen != POA_MAC_SIZE)
+ continue;
+
+ memcpy(mac, LLADDR(dl), POA_MAC_SIZE);
+
+ found = 1;
+ break;
+ }
+
+ freeifaddrs(ifas);
+
+ if (!found) {
+ log_err("No such device: %s.", dev);
+ return -1;
+ }
+
+ fd = socket(AF_INET, SOCK_DGRAM, 0);
+ if (fd < 0)
+ return -1;
+
+ memset(&ifr, 0, sizeof(ifr));
+
+ strcpy(ifr.ifr_name, dev);
+
+ if (ioctl(fd, SIOCGIFMTU, &ifr) < 0) {
+ log_err("Failed to get MTU of %s.", dev);
+ close(fd);
+ return -1;
+ }
+
+ close(fd);
+
+ *mtu = MIN(MIN(ETH_MAX_MTU, POA_ETH_RD_BUF), (uint32_t) ifr.ifr_mtu);
+ if (memcmp(dev, "lo", 2) == 0 && *mtu > POA_ETH_LO_MTU)
+ *mtu = POA_ETH_LO_MTU;
+
+ return 0;
+}
+
+static int eth_bpf_open(void)
+{
+ char dev[32];
+ size_t i;
+
+ for (i = 0; i < BPF_DEV_MAX; ++i) {
+ int fd;
+
+ sprintf(dev, "/dev/bpf%zu", i);
+
+ fd = open(dev, O_RDWR);
+ if (fd >= 0)
+ return fd;
+ }
+
+ return -1;
+}
+
+/*
+ * BIOCSHDRCMPLT: we fill in the source address ourselves.
+ * BIOCSSEESENT: our own egress must not come back at us.
+ * BIOCIMMEDIATE: deliver on arrival, do not wait for a full buffer.
+ */
+static int eth_bpf_setup(struct eth_priv * priv,
+ const char * dev)
+{
+ struct ifreq ifr;
+ int enable = 1;
+ int disable = 0;
+ int blen = 0;
+
+ memset(&ifr, 0, sizeof(ifr));
+
+ strcpy(ifr.ifr_name, dev);
+
+ if (ioctl(priv->s_fd, BIOCSETIF, &ifr) < 0) {
+ log_err("Failed to bind bpf device to %s.", dev);
+ return -1;
+ }
+
+ if (ioctl(priv->s_fd, BIOCGBLEN, &blen) < 0 || blen <= 0) {
+ log_err("Failed to get the bpf buffer length.");
+ return -1;
+ }
+
+ priv->blen = (size_t) blen;
+ if (ioctl(priv->s_fd, BIOCSHDRCMPLT, &enable) < 0) {
+ log_err("Failed to set BIOCSHDRCMPLT.");
+ return -1;
+ }
+
+ if (ioctl(priv->s_fd, BIOCSSEESENT, &disable) < 0) {
+ log_err("Failed to set BIOCSSEESENT.");
+ return -1;
+ }
+
+ if (ioctl(priv->s_fd, BIOCIMMEDIATE, &enable) < 0) {
+ log_err("Failed to set BIOCIMMEDIATE.");
+ return -1;
+ }
+
+ return 0;
+}
+
+static int eth_attach(struct poa * poa,
+ const struct poa_spec * spec)
+{
+ const struct eth_poa * c = &spec->eth;
+ struct eth_priv * priv;
+ uint8_t mac[POA_MAC_SIZE];
+ uint32_t mtu;
+
+ if (strnlen(c->dev, sizeof(c->dev)) > DEV_NAME_SIZE)
+ return -EINVAL;
+
+ if (c->ethertype < 0x0600 || c->ethertype == 0xFFFF) {
+ log_err("Invalid Ethertype 0x%x.", c->ethertype);
+ return -EINVAL;
+ }
+
+ priv = malloc(sizeof(*priv));
+ if (priv == NULL)
+ return -ENOMEM;
+
+ memset(priv, 0, sizeof(*priv));
+
+ priv->poa = poa;
+ priv->s_fd = -1;
+
+ priv->ethertype = htons(c->ethertype);
+
+ if (eth_dev_info(c->dev, mac, &mtu) < 0)
+ goto fail_conf;
+
+ priv->mtu = mtu;
+
+ memcpy(priv->hw_addr, mac, POA_MAC_SIZE);
+
+ priv->s_fd = eth_bpf_open();
+ if (priv->s_fd < 0) {
+ log_err("Failed to open a bpf device.");
+ goto fail_conf;
+ }
+
+ if (eth_bpf_setup(priv, c->dev) < 0)
+ goto fail_device;
+
+ poa->priv = priv;
+ poa->local.type = poa->type;
+ poa->local.eth.src.ethertype = c->ethertype;
+
+ memcpy(poa->local.eth.src.mac, mac, POA_MAC_SIZE);
+ strcpy(poa->local.eth.src.dev, c->dev);
+
+ log_info("Using Berkeley Packet Filter on %s.", c->dev);
+
+ return 0;
+
+ fail_device:
+ close(priv->s_fd);
+ fail_conf:
+ free(priv);
+
+ return -EIO;
+}
+
+static void eth_detach(struct poa * poa)
+{
+ struct eth_priv * priv = (struct eth_priv *) poa->priv;
+
+ if (priv == NULL)
+ return;
+
+ close(priv->s_fd);
+
+ free(priv);
+
+ poa->priv = NULL;
+}
+
+static uint32_t eth_mtu(struct poa * poa,
+ const struct poa_addr * dst)
+{
+ struct eth_priv * priv = (struct eth_priv *) poa->priv;
+
+ (void) dst;
+
+ return priv->mtu - POA_HDR_SIZE;
+}
+
+/* A bpf device has no send queue to report; mb-ECN cannot mark here. */
+static size_t eth_qlen(struct poa * poa)
+{
+ (void) poa;
+
+ return 0;
+}
+
+/* The bpf device buffer is all the receive queue there is. */
+static int eth_rib(struct poa * poa,
+ char * buf,
+ size_t len)
+{
+ struct eth_priv * priv = (struct eth_priv *) poa->priv;
+ struct bpf_stat bs;
+ int size;
+
+ if (ioctl(priv->s_fd, BIOCGSTATS, &bs) == 0) {
+ FETCH_ADD_RELAXED(&priv->kern_rcv, bs.bs_recv);
+ FETCH_ADD_RELAXED(&priv->kern_drp, bs.bs_drop);
+ }
+
+ size = snprintf(buf, len,
+ "Socket rcvbuf (bytes): %zu\n"
+ "Kernel packets received: %zu\n"
+ "Kernel packets dropped: %zu\n",
+ priv->blen,
+ LOAD_RELAXED(&priv->kern_rcv),
+ LOAD_RELAXED(&priv->kern_drp));
+ if (size < 0 || (size_t) size >= len)
+ return -1;
+
+ return size;
+}
+
+#elif defined(HAVE_NETMAP)
+
+/*
+ * netmap gives one PoA the whole port: reads copy out of the NIC
+ * ring, writes inject into it.
+ */
+
+/* nm_inject takes one contiguous frame, so the header is copied in. */
+static int eth_sendv(struct eth_priv * priv,
+ const uint8_t * dst,
+ uint32_t eid,
+ const uint8_t * body,
+ size_t len,
+ bool block,
+ const struct timespec * abstime)
+{
+ uint8_t * frame;
+ size_t flen;
+ int ret;
+
+ if (len > priv->mtu - POA_HDR_SIZE)
+ return -EMSGSIZE;
+
+ flen = ETH_HDR_TOT_SIZE + len;
+
+ frame = malloc(flen);
+ if (frame == NULL)
+ return -ENOMEM;
+
+ eth_hdr_ser(priv, (struct eth_hdr *) frame, dst, eid, len);
+
+ if (len > 0)
+ memcpy(frame + ETH_HDR_TOT_SIZE, body, len);
+
+ if (block)
+ ret = poa_wait_out(priv->poll_out.fd, abstime);
+ else
+ ret = poll(&priv->poll_out, 1, 0) > 0 ? 0 : -EAGAIN;
+
+ if (ret < 0)
+ goto fail;
+
+ ret = nm_inject(priv->nmd, frame, flen) == (int) flen ? 0 : -EIO;
+ fail:
+ free(frame);
+
+ return ret;
+}
+
+/* A slot stays owned by the ring, so each frame is copied out. */
+static void * eth_reader(void * o)
+{
+ struct poa * poa = (struct poa *) o;
+ struct eth_priv * priv = (struct eth_priv *) poa->priv;
+
+ while (true) {
+ struct ssm_pk_buff * spb;
+ struct poa_addr src;
+ struct nm_pkthdr hdr;
+ const uint8_t * frame;
+ const uint8_t * body;
+ uint32_t eid;
+ size_t plen;
+
+ if (poll(&priv->poll_in, 1, -1) < 0) {
+ if (errno == EINTR)
+ continue;
+
+ POA_STAT_BUMP(poa, rcv_fail);
+ break;
+ }
+
+ if (priv->poll_in.revents == 0)
+ continue;
+
+ frame = nm_nextpkt(priv->nmd, &hdr);
+ if (frame == NULL)
+ continue;
+
+ if (frame_parse(priv, frame, hdr.len, &eid, &plen) < 0)
+ continue;
+
+ body = frame + ETH_HDR_TOT_SIZE;
+
+ if (eid == POA_MGMT_EID) {
+ frame_to_addr(priv, (const struct eth_hdr *) frame,
+ &src);
+ eth_rx_mgmt(poa, &src, body, plen);
+ continue;
+ }
+
+ if (poa_spb_reserve(&spb, plen) < 0) {
+ POA_STAT_BUMP(poa, buf_fail);
+ continue;
+ }
+
+ memcpy(ssm_pk_buff_head(spb), body, plen);
+
+ poa_rx_pkt(poa, eid, spb);
+ }
+
+ return (void *) 0;
+}
+
+static int eth_dev_info(const char * dev,
+ uint8_t * mac,
+ uint32_t * mtu)
+{
+ struct ifreq ifr;
+#ifndef __linux__
+ struct ifaddrs * ifas;
+ struct ifaddrs * ifa;
+ int found = 0;
+#endif
+ int fd;
+
+ if (strlen(dev) >= IFNAMSIZ)
+ return -EINVAL;
+
+ fd = socket(AF_INET, SOCK_DGRAM, 0);
+ if (fd < 0)
+ return -EIO;
+
+ memset(&ifr, 0, sizeof(ifr));
+
+ strcpy(ifr.ifr_name, dev);
+
+#ifdef __linux__
+ if (ioctl(fd, SIOCGIFHWADDR, &ifr) < 0) {
+ log_err("Failed to get hardware address of %s.", dev);
+ goto fail_ioctl;
+ }
+
+ memcpy(mac, ifr.ifr_hwaddr.sa_data, POA_MAC_SIZE);
+#else
+ if (getifaddrs(&ifas) < 0)
+ goto fail_ioctl;
+
+ for (ifa = ifas; ifa != NULL; ifa = ifa->ifa_next) {
+ struct sockaddr_dl * dl;
+
+ if (ifa->ifa_addr == NULL)
+ continue;
+
+ if (ifa->ifa_addr->sa_family != AF_LINK)
+ continue;
+
+ if (strcmp(ifa->ifa_name, dev) != 0)
+ continue;
+
+ dl = (struct sockaddr_dl *) ifa->ifa_addr;
+ if (dl->sdl_alen != POA_MAC_SIZE)
+ continue;
+
+ memcpy(mac, LLADDR(dl), POA_MAC_SIZE);
+
+ found = 1;
+ break;
+ }
+
+ freeifaddrs(ifas);
+
+ if (!found) {
+ log_err("No hardware address for %s.", dev);
+ goto fail_ioctl;
+ }
+#endif
+ if (ioctl(fd, SIOCGIFMTU, &ifr) < 0) {
+ log_err("Failed to get MTU of %s.", dev);
+ goto fail_ioctl;
+ }
+
+ close(fd);
+
+ *mtu = MIN(MIN(ETH_MAX_MTU, POA_ETH_RD_BUF), (uint32_t) ifr.ifr_mtu);
+ if (memcmp(dev, "lo", 2) == 0 && *mtu > POA_ETH_LO_MTU)
+ *mtu = POA_ETH_LO_MTU;
+
+ return 0;
+
+ fail_ioctl:
+ close(fd);
+
+ return -EIO;
+}
+
+static int eth_nm_open(struct eth_priv * priv,
+ const char * dev)
+{
+ char ifn[IFNAMSIZ + sizeof("netmap:")];
+
+ strcpy(ifn, "netmap:");
+ strcat(ifn, dev);
+
+ priv->nmd = nm_open(ifn, NULL, 0, NULL);
+ if (priv->nmd == NULL) {
+ log_err("Failed to open netmap device for %s.", dev);
+ return -1;
+ }
+
+ memset(&priv->poll_in, 0, sizeof(priv->poll_in));
+ memset(&priv->poll_out, 0, sizeof(priv->poll_out));
+
+ priv->poll_in.fd = NETMAP_FD(priv->nmd);
+ priv->poll_in.events = POLLIN;
+ priv->poll_out.fd = NETMAP_FD(priv->nmd);
+ priv->poll_out.events = POLLOUT;
+
+ return 0;
+}
+
+static int eth_attach(struct poa * poa,
+ const struct poa_spec * spec)
+{
+ const struct eth_poa * c = &spec->eth;
+ struct eth_priv * priv;
+ uint8_t mac[POA_MAC_SIZE];
+ uint32_t mtu;
+
+ if (strnlen(c->dev, sizeof(c->dev)) > DEV_NAME_SIZE)
+ return -EINVAL;
+
+ if (c->ethertype < 0x0600 || c->ethertype == 0xFFFF) {
+ log_err("Invalid Ethertype 0x%x.", c->ethertype);
+ return -EINVAL;
+ }
+
+ priv = malloc(sizeof(*priv));
+ if (priv == NULL)
+ return -ENOMEM;
+
+ memset(priv, 0, sizeof(*priv));
+
+ priv->poa = poa;
+
+ priv->ethertype = htons(c->ethertype);
+
+ if (eth_dev_info(c->dev, mac, &mtu) < 0)
+ goto fail_conf;
+
+ priv->mtu = mtu;
+
+ memcpy(priv->hw_addr, mac, POA_MAC_SIZE);
+
+ if (eth_nm_open(priv, c->dev) < 0)
+ goto fail_conf;
+
+ poa->priv = priv;
+ poa->local.type = poa->type;
+ poa->local.eth.src.ethertype = c->ethertype;
+
+ memcpy(poa->local.eth.src.mac, mac, POA_MAC_SIZE);
+ strcpy(poa->local.eth.src.dev, c->dev);
+
+ log_info("Using netmap on %s.", c->dev);
+
+ return 0;
+
+ fail_conf:
+ free(priv);
+
+ return -EIO;
+}
+
+static void eth_detach(struct poa * poa)
+{
+ struct eth_priv * priv = (struct eth_priv *) poa->priv;
+
+ if (priv == NULL)
+ return;
+
+ nm_close(priv->nmd);
+
+ free(priv);
+
+ poa->priv = NULL;
+}
+
+static uint32_t eth_mtu(struct poa * poa,
+ const struct poa_addr * dst)
+{
+ struct eth_priv * priv = (struct eth_priv *) poa->priv;
+
+ (void) dst;
+
+ return priv->mtu - POA_HDR_SIZE;
+}
+
+/* The ring is drained by the NIC; there is no queue to report. */
+static size_t eth_qlen(struct poa * poa)
+{
+ (void) poa;
+
+ return 0;
+}
+
+#endif /* HAVE_RAW_SOCKETS */
+
+#ifndef HAVE_RAW_SOCKETS
+
+/* Only netlink reports link events; no other backend has a monitor. */
+int poa_monitor_open(void)
+{
+ return -1;
+}
+
+void poa_monitor_read(int fd)
+{
+ (void) fd;
+}
+
+#endif
+
+/* One reader per socket, so a flow cannot be reordered on receive. */
+static int eth_start(struct poa * poa)
+{
+ struct eth_priv * priv = (struct eth_priv *) poa->priv;
+
+ if (pthread_create(&priv->reader, NULL, eth_reader, poa) != 0)
+ return -1;
+
+ priv->running = true;
+
+ return 0;
+}
+
+static void eth_stop(struct poa * poa)
+{
+ struct eth_priv * priv = (struct eth_priv *) poa->priv;
+
+ if (!priv->running)
+ return;
+
+ pthread_cancel(priv->reader);
+ pthread_join(priv->reader, NULL);
+
+ priv->running = false;
+}
+
+static int eth_send(struct poa * poa,
+ const struct poa_addr * dst,
+ uint32_t eid,
+ struct ssm_pk_buff * spb,
+ bool block,
+ const struct timespec * abstime)
+{
+ return eth_sendv((struct eth_priv *) poa->priv, dst->eth.dst.mac, eid,
+ ssm_pk_buff_head(spb), ssm_pk_buff_len(spb),
+ block, abstime);
+}
+
+static int eth_send_mgmt(struct poa * poa,
+ const struct poa_addr * dst,
+ const uint8_t * buf,
+ size_t len)
+{
+ struct timespec timeo = TIMESPEC_INIT_MS(POA_MGMT_SND_TIMEO);
+ struct timespec abstime;
+
+ clock_gettime(PTHREAD_COND_CLOCK, &abstime);
+ ts_add(&abstime, &timeo, &abstime);
+
+ return eth_sendv((struct eth_priv *) poa->priv, dst->eth.dst.mac,
+ POA_MGMT_EID, buf, len, true, &abstime);
+}
+
+const struct poa_ops eth_poa_ops = {
+ .poa_attach = eth_attach,
+ .poa_detach = eth_detach,
+ .poa_start = eth_start,
+ .poa_stop = eth_stop,
+ .poa_send = eth_send,
+ .poa_send_mgmt = eth_send_mgmt,
+ .poa_query = eth_query,
+ .poa_mtu = eth_mtu,
+ .poa_qlen = eth_qlen,
+#ifdef HAVE_RAW_SOCKETS
+ .poa_qpkts = eth_qpkts,
+#endif
+#ifndef HAVE_NETMAP
+ .poa_rib = eth_rib,
+#endif
+ .poa_spec = eth_spec,
+ .poa_has_id = eth_has_id,
+ .poa_match = eth_match,
+ .poa_link_match = eth_link_match,
+ .mpl = POA_ETH_MPL
+};
diff --git a/src/lib/poa/poa.c b/src/lib/poa/poa.c
new file mode 100644
index 00000000..3ad17c4f
--- /dev/null
+++ b/src/lib/poa/poa.c
@@ -0,0 +1,2515 @@
+/*
+ * Ouroboros - Copyright (C) 2016 - 2026
+ *
+ * Points of attachment (PoA) - transport independent core
+ *
+ * Included by dev.c; uses dev.c statics (proc, flow_init, ...).
+ *
+ * Dimitri Staessens <dimitri@ouroboros.rocks>
+ * Sander Vrijders <sander@ouroboros.rocks>
+ *
+ * This library is free software; you can redistribute it and/or
+ * modify it under the terms of the GNU Lesser General Public License
+ * version 2.1 as published by the Free Software Foundation.
+ *
+ * This library is distributed in the hope that it will be useful,
+ * but WITHOUT ANY WARRANTY; without even the implied warranty of
+ * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
+ * Lesser General Public License for more details.
+ *
+ * You should have received a copy of the GNU Lesser General Public
+ * License along with this library; if not, write to the Free Software
+ * Foundation, Inc., http://www.fsf.org/about/contact/.
+ */
+
+#define POA_MAX_EIDS PROC_MAX_FLOWS
+#define POA_ALLOC_TIMEO 10000 /* ms, overall FLOW_REQ deadline */
+#define POA_RETRY_TIMEO 300 /* ms, FLOW_REQ retransmit period */
+/* Must fit a certificate chain: post-quantum ones are large. */
+#define POA_MGMT_BUFSZ POA_MGMT_FRAME_SIZE
+#define POA_MGMT_QMAX 64 /* queued management frames per PoA */
+#define POA_PEND_TIMEO 10 /* s, reap a request that never completes */
+#define POA_SWEEP_TIMEO 1000 /* ms, sweep interval */
+#define POA_SWEEP_MAX 16 /* requests reaped per sweep */
+#define POA_DEFER_MAX 64 /* replies waiting for their flow id */
+
+/* EWMA over 8 samples. */
+#define POA_AVG_SHIFT 3
+/* Queue cost is sampled every 64th packet: qlen is a syscall. */
+#define POA_COST_MASK 63
+/* Reuse a qlen read for this long; the mark moves on doublings. */
+
+#define POA_RIB "poa"
+/* Fits the RIB labels below with 20-digit counters. */
+#define POA_RIB_STRLEN 2048
+
+enum poa_flow_state {
+ POA_FLOW_NULL = 0,
+ POA_FLOW_PENDING,
+ POA_FLOW_ALLOCATED,
+ POA_FLOW_DEAD
+};
+
+enum poa_state {
+ POA_NULL = 0,
+ POA_INIT, /* poa_init(); this process may attach */
+ POA_RUNNING, /* the threads are up */
+ POA_OPERATIONAL /* bootstrapped or enrolled in a layer */
+};
+
+struct poa_flow {
+ struct poa * poa;
+
+ int flow_id;
+ uint32_t eid;
+ uint32_t r_eid;
+ struct poa_addr r_addr;
+
+ enum poa_flow_state state;
+
+ struct ssm_rbuff * rx_rb;
+
+ struct list_head pend; /* on poa->pending while unanswered */
+ struct timespec t0;
+
+ /* Our answer, kept to re-send when a request is repeated. */
+ bool answered;
+ bool answer_sent;
+
+ /* Handed to a caller that will attach it; not the sweeper's. */
+ bool claimed;
+ int answer;
+ buffer_t answer_data;
+
+ /* Handshake rendezvous with the reader thread. */
+ pthread_mutex_t mtx;
+ pthread_cond_t cond;
+ bool replied;
+ bool pending;
+ int response;
+ buffer_t resp_data;
+};
+
+/*
+ * An answer can be ready before the flow it answers has an id, and
+ * the caller must not be kept waiting for one: it answers to the IRMd,
+ * which gives up long before we would.
+ */
+struct poa_deferred {
+ struct list_head next;
+ struct timespec t0;
+ int flow_id;
+ int response;
+ buffer_t data;
+};
+
+struct poa_mgmt_frame {
+ struct list_head next;
+ struct poa * poa;
+ struct poa_addr src;
+ size_t len;
+ uint8_t buf[POA_MGMT_BUFSZ];
+};
+
+/*
+ * A detach is performed by the management thread, so that it cannot
+ * run alongside the accept side. The caller waits for the result.
+ */
+struct poa_detach_req {
+ struct list_head next;
+ struct poa_spec spec;
+ int result;
+ bool done;
+};
+
+static struct {
+ struct list_head list;
+
+ struct poa_flow * id_to_pf[SYS_MAX_FLOWS];
+
+ struct llist deferred;
+
+ pthread_mutex_t mtx; /* guards id_to_pf */
+ pthread_cond_t cond;
+
+ /* One management thread and one link monitor for all PoAs. */
+ struct llist mgmt_frames;
+ struct list_head detach;
+ pthread_mutex_t mgmt_mtx;
+ pthread_cond_t mgmt_cond; /* work for the handler */
+ pthread_cond_t done_cond; /* a detach has completed */
+ pthread_t mgmt_handler;
+ bool mgmt_stop;
+
+ pthread_t monitor;
+ int mon_fd;
+
+ enum poa_state state;
+
+ /* Hashes of the names this process answers queries for. */
+ uint8_t name_hash[POA_QUERY_HLEN];
+ uint8_t layer_hash[POA_QUERY_HLEN];
+
+ pthread_rwlock_t lock; /* the PoAs and their flows */
+} poas;
+
+static int mgmt_send(struct poa * poa,
+ const struct poa_addr * dst,
+ uint8_t code,
+ uint32_t s_eid,
+ uint32_t d_eid,
+ qosspec_t qs,
+ int response,
+ const buffer_t * data)
+{
+ uint8_t buf[POA_MGMT_BUFSZ];
+ struct poa_mgmt_msg * msg = (struct poa_mgmt_msg *) buf;
+ size_t len;
+
+ len = sizeof(*msg);
+ if (data != NULL && data->len > 0) {
+ if (len + data->len > sizeof(buf))
+ return -EMSGSIZE;
+ memcpy(buf + len, data->data, data->len);
+
+ len += data->len;
+ }
+
+ poa_mgmt_msg_ser(msg, code, s_eid, d_eid, qs, response,
+ data != NULL ? data->len : 0);
+
+ POA_STAT_BUMP(poa, mgmt_tx);
+
+ return poa->ops->poa_send_mgmt(poa, dst, buf, len);
+}
+
+/* Caller holds poas.lock for writing. */
+static struct poa_flow * pf_create(struct poa * poa,
+ const struct poa_addr * r_addr)
+{
+ struct poa_flow * pf;
+ pthread_condattr_t cattr;
+ int eid;
+
+ pf = malloc(sizeof(*pf));
+ if (pf == NULL)
+ goto fail_malloc;
+
+ memset(pf, 0, sizeof(*pf));
+
+ if (pthread_mutex_init(&pf->mtx, NULL) != 0)
+ goto fail_mtx;
+
+ if (pthread_condattr_init(&cattr) != 0)
+ goto fail_cond;
+#ifndef __APPLE__
+ pthread_condattr_setclock(&cattr, PTHREAD_COND_CLOCK);
+#endif
+ if (pthread_cond_init(&pf->cond, &cattr) != 0) {
+ pthread_condattr_destroy(&cattr);
+ goto fail_cond;
+ }
+
+ pthread_condattr_destroy(&cattr);
+
+ eid = bmp_allocate(poa->eids);
+ if (!bmp_is_id_valid(poa->eids, eid))
+ goto fail_eid;
+
+ pf->poa = poa;
+ pf->eid = (uint32_t) eid;
+ pf->flow_id = -1;
+ pf->state = POA_FLOW_PENDING;
+ pf->r_addr = *r_addr;
+
+ clock_gettime(PTHREAD_COND_CLOCK, &pf->t0);
+
+ rcu_wrlock(&poa->guard);
+
+ rcu_publish(pf);
+ rcu_assign(poa->eid_to_pf[eid], pf);
+
+ rcu_wrunlock(&poa->guard);
+
+ list_add_tail(&pf->pend, &poa->flows);
+
+ POA_STAT_BUMP(poa, n_flows);
+
+ return pf;
+
+ fail_eid:
+ pthread_cond_destroy(&pf->cond);
+ fail_cond:
+ pthread_mutex_destroy(&pf->mtx);
+ fail_mtx:
+ free(pf);
+ fail_malloc:
+ return NULL;
+}
+
+/* Caller holds poas.mtx. */
+static void deferred_purge(int flow_id)
+{
+ struct list_head * p;
+ struct list_head * h;
+
+ list_for_each_safe(p, h, &poas.deferred.list) {
+ struct poa_deferred * d;
+
+ d = list_entry(p, struct poa_deferred, next);
+ if (d->flow_id != flow_id)
+ continue;
+
+ llist_del(&d->next, &poas.deferred);
+ freebuf(d->data);
+ free(d);
+ }
+}
+
+/* The id may have moved to a newer flow; only its owner clears it. */
+static void pf_destroy(struct poa_flow * pf)
+{
+ struct poa * poa = pf->poa;
+
+ pthread_rwlock_wrlock(&poas.lock);
+
+ rcu_wrlock(&poa->guard);
+
+ rcu_assign(poa->eid_to_pf[pf->eid], NULL);
+ rcu_assign(pf->rx_rb, NULL);
+
+ rcu_wrunlock(&poa->guard);
+
+ bmp_release(poa->eids, pf->eid);
+
+ if (!list_is_empty(&pf->pend)) {
+ list_del(&pf->pend);
+ POA_STAT_SUB(poa, n_flows, 1);
+ }
+
+ pthread_rwlock_unlock(&poas.lock);
+
+ pthread_mutex_lock(&poas.mtx);
+
+ if (pf->flow_id >= 0 && poas.id_to_pf[pf->flow_id] == pf) {
+ poas.id_to_pf[pf->flow_id] = NULL;
+
+ deferred_purge(pf->flow_id);
+ }
+
+ pthread_mutex_unlock(&poas.mtx);
+
+ rcu_reclaim(&poa->guard); /* a receive may still hold pf */
+
+ freebuf(pf->resp_data);
+ freebuf(pf->answer_data);
+
+ pthread_cond_destroy(&pf->cond);
+ pthread_mutex_destroy(&pf->mtx);
+
+ free(pf);
+}
+
+/*
+ * Keeps the answer for a repeat, then sends it. An accept must wait
+ * until the flow can receive; a refusal needs no receiver.
+ */
+static int pf_answer(struct poa_flow * pf,
+ int response,
+ const buffer_t * data)
+{
+ int err;
+
+ pthread_rwlock_wrlock(&poas.lock);
+
+ freebuf(pf->answer_data);
+
+ if (data != NULL && data->len > 0) {
+ pf->answer_data.data = malloc(data->len);
+ if (pf->answer_data.data != NULL) {
+ memcpy(pf->answer_data.data, data->data, data->len);
+
+ pf->answer_data.len = data->len;
+ }
+ }
+
+ pf->answer = response;
+ pf->answered = true;
+ if (response == 0 && pf->state != POA_FLOW_ALLOCATED) {
+ pthread_rwlock_unlock(&poas.lock);
+ return 0;
+ }
+
+ pf->answer_sent = true;
+
+ pthread_rwlock_unlock(&poas.lock);
+
+ err = mgmt_send(pf->poa, &pf->r_addr, POA_FLOW_REPLY, pf->eid,
+ pf->r_eid, qos_raw, response, data);
+
+ if (err == -ETIMEDOUT || err == -EAGAIN)
+ err = 0; /* stored; a repeat request resends it */
+
+ return err;
+}
+
+/* Takes an answer left for a flow that had no id yet. */
+static struct poa_deferred * deferred_take(int flow_id)
+{
+ struct list_head * p;
+ struct list_head * h;
+
+ list_for_each_safe(p, h, &poas.deferred.list) {
+ struct poa_deferred * d;
+
+ d = list_entry(p, struct poa_deferred, next);
+ if (d->flow_id != flow_id)
+ continue;
+
+ llist_del(&d->next, &poas.deferred);
+
+ return d;
+ }
+
+ return NULL;
+}
+
+/* Publishes the flow_id so an answer can find this flow. */
+static void pf_set_flow_id(struct poa_flow * pf,
+ int flow_id)
+{
+ struct poa_deferred * d;
+
+ pthread_mutex_lock(&poas.mtx);
+
+ pf->flow_id = flow_id;
+ poas.id_to_pf[flow_id] = pf;
+
+ d = deferred_take(flow_id);
+
+ pthread_cond_broadcast(&poas.cond);
+
+ pthread_mutex_unlock(&poas.mtx);
+
+ if (d != NULL) {
+ pf_answer(pf, d->response, &d->data);
+ freebuf(d->data);
+ free(d);
+ }
+}
+
+/*
+ * Between the request arriving and the accept returning, the flow has
+ * an id but no fd yet; flow_init claims the PoA here.
+ */
+static void pf_set_pending(struct poa_flow * pf)
+{
+ pthread_mutex_lock(&poas.mtx);
+
+ pf->pending = true;
+
+ pthread_mutex_unlock(&poas.mtx);
+}
+
+static void pf_clr_pending(struct poa_flow * pf)
+{
+ pthread_mutex_lock(&poas.mtx);
+
+ pf->pending = false;
+
+ pthread_mutex_unlock(&poas.mtx);
+}
+
+/* A process that attaches no PoA has nothing pending. */
+struct poa_flow * poa_flow_take_pending(int flow_id)
+{
+ struct poa_flow * pf;
+
+ if (poas.state == POA_NULL)
+ return NULL;
+
+ if (flow_id < 0 || flow_id >= SYS_MAX_FLOWS)
+ return NULL;
+
+ pthread_mutex_lock(&poas.mtx);
+
+ pf = poas.id_to_pf[flow_id];
+ if (pf != NULL && pf->pending)
+ pf->pending = false;
+ else
+ pf = NULL;
+
+ pthread_mutex_unlock(&poas.mtx);
+
+ return pf;
+}
+
+static struct poa_flow * pf_get(int flow_id)
+{
+ struct poa_flow * pf;
+
+ if (flow_id < 0 || flow_id >= SYS_MAX_FLOWS)
+ return NULL;
+
+ pthread_mutex_lock(&poas.mtx);
+
+ pf = poas.id_to_pf[flow_id];
+
+ pthread_mutex_unlock(&poas.mtx);
+
+ return pf;
+}
+
+#ifdef PROC_FLOW_STATS
+
+/* Caller holds poas.lock. */
+static struct poa * poa_by_rib_name(const char * name)
+{
+ struct list_head * p;
+
+ list_for_each(p, &poas.list) {
+ struct poa * poa = list_entry(p, struct poa, next);
+
+ if (strcmp(poa->name, name) == 0)
+ return poa;
+ }
+
+ return NULL;
+}
+
+static int poa_rib_read(const char * path,
+ char * buf,
+ size_t len)
+{
+ struct poa * poa;
+ const char * entry;
+ size_t qlen;
+ size_t avg;
+ size_t cost;
+ int size;
+ int ret;
+
+ entry = strstr(path, RIB_SEPARATOR) + 1;
+
+ if (len < POA_RIB_STRLEN)
+ return -1;
+
+ pthread_rwlock_rdlock(&poas.lock);
+
+ poa = poa_by_rib_name(entry);
+ if (poa == NULL)
+ goto fail;
+
+ qlen = poa->ops->poa_qlen(poa);
+ avg = poa->avg_len;
+ cost = poa->avg_len + poa->avg_ovh;
+
+ size = snprintf(buf, len,
+ "Active flows: %zu\n"
+ "Packets received: %zu\n"
+ "Bytes received: %zu\n"
+ "Packets sent: %zu\n"
+ "Bytes sent: %zu\n"
+ "Management frames rcvd: %zu\n"
+ "Management frames sent: %zu\n"
+ "Bad EID packets: %zu\n"
+ "Delivery (N+1) failures: %zu\n"
+ "Buffer alloc failures: %zu\n"
+ "Packet read failures: %zu\n"
+ "Packet send failures: %zu\n"
+ "Name queries sent: %zu\n"
+ "Name queries received: %zu\n"
+ "Name replies sent: %zu\n"
+ "Name replies received: %zu\n"
+ "Queued (transport): %zu\n"
+ "Queued (packets): %zu\n"
+ "Mean packet size (bytes): %zu\n"
+ "Mean packet cost: %zu\n",
+ POA_STAT_LOAD(poa, n_flows),
+ POA_STAT_LOAD(poa, rx_pkts),
+ POA_STAT_LOAD(poa, rx_bytes),
+ POA_STAT_LOAD(poa, tx_pkts),
+ POA_STAT_LOAD(poa, tx_bytes),
+ POA_STAT_LOAD(poa, mgmt_rx),
+ POA_STAT_LOAD(poa, mgmt_tx),
+ POA_STAT_LOAD(poa, bad_eid),
+ POA_STAT_LOAD(poa, dlv_fail),
+ POA_STAT_LOAD(poa, buf_fail),
+ POA_STAT_LOAD(poa, rcv_fail),
+ POA_STAT_LOAD(poa, snd_fail),
+ POA_STAT_LOAD(poa, qry_tx),
+ POA_STAT_LOAD(poa, qry_rx),
+ POA_STAT_LOAD(poa, rep_tx),
+ POA_STAT_LOAD(poa, rep_rx),
+ qlen, cost > 0 ? qlen / cost : 0, avg, cost);
+ if (size < 0 || (size_t) size >= len)
+ goto fail;
+
+ if (poa->ops->poa_rib != NULL) {
+ ret = poa->ops->poa_rib(poa, buf + size, len - size);
+ if (ret < 0)
+ goto fail;
+
+ size += ret;
+ }
+
+ pthread_rwlock_unlock(&poas.lock);
+
+ return size;
+
+ fail:
+ pthread_rwlock_unlock(&poas.lock);
+
+ return -1;
+}
+
+static int poa_rib_readdir(char *** buf)
+{
+ struct list_head * p;
+ size_t n = 0;
+ int idx = 0;
+
+ pthread_rwlock_rdlock(&poas.lock);
+
+ list_for_each(p, &poas.list)
+ ++n;
+
+ if (n == 0) {
+ *buf = NULL;
+ goto no_poas;
+ }
+
+ *buf = malloc(sizeof(**buf) * n);
+ if (*buf == NULL)
+ goto fail_entries;
+
+ list_for_each(p, &poas.list) {
+ struct poa * poa = list_entry(p, struct poa, next);
+
+ (*buf)[idx] = malloc(strlen(poa->name) + 1);
+ if ((*buf)[idx] == NULL)
+ goto fail_entry;
+
+ strcpy((*buf)[idx++], poa->name);
+ }
+ no_poas:
+ pthread_rwlock_unlock(&poas.lock);
+
+ return idx;
+
+ fail_entry:
+ while (idx-- > 0)
+ free((*buf)[idx]);
+
+ free(*buf);
+ fail_entries:
+ pthread_rwlock_unlock(&poas.lock);
+
+ return -ENOMEM;
+}
+
+static int poa_rib_getattr(const char * path,
+ struct rib_attr * attr)
+{
+ (void) path;
+
+ attr->size = POA_RIB_STRLEN;
+ attr->mtime = 0;
+
+ return 0;
+}
+
+static struct rib_ops poa_r_ops = {
+ .read = poa_rib_read,
+ .readdir = poa_rib_readdir,
+ .getattr = poa_rib_getattr
+};
+
+#endif /* PROC_FLOW_STATS */
+
+int poa_init(const char * name)
+{
+ pthread_condattr_t cattr;
+
+ assert(name != NULL);
+
+ memset(&poas, 0, sizeof(poas));
+
+ str_hash(HASH_SHA3_256, poas.name_hash, name);
+
+ poas.mon_fd = -1;
+ if (pthread_mutex_init(&poas.mtx, NULL) != 0)
+ goto fail_mtx;
+
+ if (pthread_cond_init(&poas.cond, NULL) != 0)
+ goto fail_cond;
+
+ if (pthread_mutex_init(&poas.mgmt_mtx, NULL) != 0)
+ goto fail_mgmt_mtx;
+
+ if (pthread_condattr_init(&cattr) != 0)
+ goto fail_cattr;
+#ifndef __APPLE__
+ pthread_condattr_setclock(&cattr, PTHREAD_COND_CLOCK);
+#endif
+ if (pthread_cond_init(&poas.mgmt_cond, &cattr) != 0) {
+ pthread_condattr_destroy(&cattr);
+ goto fail_cattr;
+ }
+
+ pthread_condattr_destroy(&cattr);
+
+ if (pthread_cond_init(&poas.done_cond, NULL) != 0)
+ goto fail_done_cond;
+
+ if (pthread_rwlock_init(&poas.lock, NULL) != 0)
+ goto fail_lock;
+
+ list_head_init(&poas.list);
+ llist_init(&poas.deferred);
+ llist_init(&poas.mgmt_frames);
+ list_head_init(&poas.detach);
+
+ poas.state = POA_INIT;
+
+#ifdef PROC_FLOW_STATS
+ if (rib_reg(POA_RIB, &poa_r_ops) < 0)
+ goto fail_rib;
+#endif
+ return 0;
+
+#ifdef PROC_FLOW_STATS
+ fail_rib:
+ pthread_rwlock_destroy(&poas.lock);
+#endif
+
+ fail_lock:
+ pthread_cond_destroy(&poas.done_cond);
+ fail_done_cond:
+ pthread_cond_destroy(&poas.mgmt_cond);
+ fail_cattr:
+ pthread_mutex_destroy(&poas.mgmt_mtx);
+ fail_mgmt_mtx:
+ pthread_cond_destroy(&poas.cond);
+ fail_cond:
+ pthread_mutex_destroy(&poas.mtx);
+ fail_mtx:
+ return -1;
+}
+
+int poa_set_layer(const char * layer)
+{
+ if (layer == NULL)
+ return -EINVAL;
+
+ pthread_rwlock_wrlock(&poas.lock);
+
+ str_hash(HASH_SHA3_256, poas.layer_hash, layer);
+
+ poas.state = POA_OPERATIONAL;
+
+ pthread_rwlock_unlock(&poas.lock);
+
+ return 0;
+}
+
+bool poa_has_name(const uint8_t * hash)
+{
+ bool match = false;
+
+ pthread_rwlock_rdlock(&poas.lock);
+
+ if (poas.state >= POA_INIT)
+ match = memcmp(hash, poas.name_hash, POA_QUERY_HLEN) == 0;
+
+ if (!match && poas.state >= POA_OPERATIONAL)
+ match = memcmp(hash, poas.layer_hash, POA_QUERY_HLEN) == 0;
+
+ pthread_rwlock_unlock(&poas.lock);
+
+ return match;
+}
+
+int poa_spb_reserve(struct ssm_pk_buff ** spb,
+ size_t len)
+{
+ return ssm_pool_alloc_b(proc.pool, len, NULL, spb, NULL) < 0 ? -1 : 0;
+}
+
+void poa_spb_release(struct ssm_pk_buff * spb)
+{
+ ssm_pool_remove(proc.pool, ssm_pk_buff_get_off(spb));
+}
+
+void poa_rx_pkt(struct poa * poa,
+ uint32_t eid,
+ struct ssm_pk_buff * spb)
+{
+ struct poa_flow * pf;
+ struct ssm_rbuff * rx_rb;
+ size_t len;
+
+ len = ssm_pk_buff_len(spb); /* the ring write takes it over */
+
+ if (eid >= poa->n_eids) {
+ POA_STAT_BUMP(poa, bad_eid);
+ poa_spb_release(spb);
+ return;
+ }
+
+ rcu_rdlock(&poa->guard);
+
+ pf = rcu_deref(poa->eid_to_pf[eid]);
+
+ rcu_consume(pf);
+
+ if (pf == NULL)
+ goto fail_eid;
+
+ rx_rb = rcu_deref(pf->rx_rb);
+
+ rcu_consume(rx_rb);
+
+ if (rx_rb == NULL)
+ goto fail_eid;
+
+ if (ssm_rbuff_write(rx_rb, ssm_pk_buff_get_off(spb)) < 0) {
+ POA_STAT_BUMP(poa, dlv_fail);
+ rcu_rdunlock(&poa->guard);
+ poa_spb_release(spb);
+ return;
+ }
+
+ POA_STAT_BUMP(poa, rx_pkts);
+ POA_STAT_ADD(poa, rx_bytes, len);
+
+ ssm_flow_set_notify(proc.fqset, pf->flow_id, FLOW_PKT);
+
+ rcu_rdunlock(&poa->guard);
+
+ return;
+
+ fail_eid:
+ POA_STAT_BUMP(poa, bad_eid);
+
+ rcu_rdunlock(&poa->guard);
+
+ poa_spb_release(spb);
+}
+
+static int poa_flow_req_arr(struct poa_flow * pf,
+ qosspec_t qs,
+ uint32_t mtu,
+ const buffer_t * data)
+{
+ struct flow_info flow;
+ uint8_t buf[SOCK_BUF_SIZE];
+ buffer_t msg = {SOCK_BUF_SIZE, buf};
+ buffer_t out = BUF_INIT;
+ int err;
+
+ memset(&flow, 0, sizeof(flow));
+
+ flow.n_pid = getpid();
+ flow.n_1_pid = getpid();
+ flow.qs = qs;
+ flow.mpl = pf->poa->mpl;
+ flow.mtu = mtu;
+ if (ipcp_poa_flow_req_arr__irm_req_ser(&msg, &flow, data) < 0)
+ return -ENOMEM;
+
+ pf_set_pending(pf);
+
+ err = send_recv_msg(&msg);
+ if (err < 0)
+ goto fail;
+
+ err = poa_flow__irm_result_des(&msg, &flow, &out);
+ if (err < 0)
+ goto fail;
+
+ freebuf(out);
+
+ if (flow.id < 0 || flow.id >= SYS_MAX_FLOWS) {
+ err = -EBADF;
+ goto fail;
+ }
+
+ pf_set_flow_id(pf, flow.id);
+
+ return 0;
+ fail:
+ pf_clr_pending(pf);
+ return err;
+}
+
+static void handle_flow_req(struct poa * poa,
+ const struct poa_addr * src,
+ const struct poa_mgmt_msg * msg,
+ const uint8_t * data,
+ size_t data_len)
+{
+ struct list_head * p;
+ struct poa_flow * pf = NULL;
+ qosspec_t qs;
+ buffer_t buf;
+ buffer_t answer_data = BUF_INIT;
+ uint32_t r_eid;
+ uint32_t mtu;
+ uint32_t eid = 0;
+ int answer = 0;
+ bool found = false;
+ bool served = false;
+
+ r_eid = ntoh32(msg->s_eid);
+
+ poa_mgmt_msg_qos(msg, &qs);
+
+ pthread_rwlock_wrlock(&poas.lock);
+
+ list_for_each(p, &poa->flows) {
+ pf = list_entry(p, struct poa_flow, pend);
+ if (pf->r_eid != r_eid || poa_addr_cmp(&pf->r_addr, src) != 0)
+ continue;
+
+ answer = pf->answer;
+ eid = pf->eid;
+ served = true;
+
+ found = pf->answer_sent;
+ if (found && pf->answer_data.len > 0) {
+ answer_data.data = malloc(pf->answer_data.len);
+ if (answer_data.data != NULL) {
+ memcpy(answer_data.data, pf->answer_data.data,
+ pf->answer_data.len);
+ answer_data.len = pf->answer_data.len;
+ }
+ }
+ break;
+ }
+
+ if (!served) {
+ pf = pf_create(poa, src);
+ if (pf != NULL)
+ pf->r_eid = r_eid;
+ }
+
+ pthread_rwlock_unlock(&poas.lock);
+
+ if (found) {
+ mgmt_send(poa, src, POA_FLOW_REPLY, eid, r_eid, qos_raw, answer,
+ answer_data.len > 0 ? &answer_data : NULL);
+ freebuf(answer_data);
+ }
+
+ if (served || pf == NULL)
+ return;
+
+ buf.len = data_len;
+ buf.data = (uint8_t *) data;
+ mtu = poa->ops->poa_mtu(poa, src);
+ if (poa_flow_req_arr(pf, qs, mtu, &buf) < 0) {
+ mgmt_send(poa, src, POA_FLOW_REPLY, pf->eid, r_eid, qos_raw,
+ -1, NULL);
+ pf_destroy(pf);
+ }
+}
+
+static void handle_flow_reply(struct poa * poa,
+ const struct poa_addr * src,
+ const struct poa_mgmt_msg * msg,
+ const uint8_t * data,
+ size_t data_len)
+{
+ struct poa_flow * pf;
+ uint32_t eid;
+
+ eid = ntoh32(msg->d_eid);
+
+ pthread_rwlock_rdlock(&poas.lock);
+
+ pf = eid < poa->n_eids ? poa->eid_to_pf[eid] : NULL;
+ if (pf == NULL || pf->state != POA_FLOW_PENDING) {
+ pthread_rwlock_unlock(&poas.lock);
+ return;
+ }
+
+ if (poa_addr_cmp(&pf->r_addr, src) != 0) {
+ pthread_rwlock_unlock(&poas.lock);
+ return;
+ }
+
+ pthread_mutex_lock(&pf->mtx);
+
+ if (pf->replied) {
+ pthread_mutex_unlock(&pf->mtx);
+ pthread_rwlock_unlock(&poas.lock);
+ return;
+ }
+
+ if (data_len > 0) {
+ pf->resp_data.data = malloc(data_len);
+ if (pf->resp_data.data != NULL) {
+ memcpy(pf->resp_data.data, data, data_len);
+
+ pf->resp_data.len = data_len;
+ }
+ }
+
+ pf->r_eid = ntoh32(msg->s_eid);
+ pf->response = ntoh32(msg->response);
+ pf->replied = true;
+
+ pthread_cond_broadcast(&pf->cond);
+
+ pthread_mutex_unlock(&pf->mtx);
+
+ pthread_rwlock_unlock(&poas.lock);
+}
+
+static void handle_flow_update(struct poa * poa,
+ const struct poa_addr * src,
+ const struct poa_mgmt_msg * msg,
+ const uint8_t * data,
+ size_t data_len)
+{
+ struct poa_flow * pf;
+ buffer_t buf;
+ uint32_t eid;
+
+ eid = ntoh32(msg->d_eid);
+
+ pthread_rwlock_rdlock(&poas.lock);
+
+ pf = eid < poa->n_eids ? poa->eid_to_pf[eid] : NULL;
+ if (pf == NULL || pf->state != POA_FLOW_ALLOCATED) {
+ pthread_rwlock_unlock(&poas.lock);
+ return;
+ }
+
+ if (poa_addr_cmp(&pf->r_addr, src) != 0) {
+ pthread_rwlock_unlock(&poas.lock);
+ return;
+ }
+
+ eid = (uint32_t) pf->flow_id;
+
+ pthread_rwlock_unlock(&poas.lock);
+
+ buf.len = data_len;
+ buf.data = (uint8_t *) data;
+
+ ipcp_flow_update_arr((int) eid, &buf);
+}
+
+static void mgmt_frame_handle(struct poa_mgmt_frame * frame)
+{
+ const struct poa_mgmt_msg * msg;
+ const uint8_t * data;
+ size_t data_len;
+
+ msg = (const struct poa_mgmt_msg *) frame->buf;
+ if (frame->len < sizeof(*msg))
+ return;
+
+ data_len = ntoh16(msg->data_len);
+ if (data_len > frame->len - sizeof(*msg))
+ return;
+
+ data = frame->buf + sizeof(*msg);
+
+ switch (msg->code) {
+ case POA_FLOW_REQ:
+ handle_flow_req(frame->poa, &frame->src, msg, data, data_len);
+ break;
+ case POA_FLOW_REPLY:
+ handle_flow_reply(frame->poa, &frame->src, msg, data, data_len);
+ break;
+ case POA_FLOW_UPDATE:
+ handle_flow_update(frame->poa, &frame->src, msg, data,
+ data_len);
+ break;
+ default:
+ break;
+ }
+}
+
+static bool pf_steal(struct poa_flow * pf)
+{
+ bool stolen = false;
+
+ pthread_mutex_lock(&poas.mtx);
+
+ if (pf->pending) {
+ pf->pending = false;
+ poas.id_to_pf[pf->flow_id] = NULL;
+
+ deferred_purge(pf->flow_id);
+
+ stolen = true;
+ }
+
+ pthread_mutex_unlock(&poas.mtx);
+
+ return stolen;
+}
+
+static void sweep_pending(void)
+{
+ struct poa_flow * dead[POA_SWEEP_MAX];
+ struct list_head * p;
+ struct list_head * q;
+ struct timespec now;
+ size_t n = 0;
+ size_t i;
+
+ clock_gettime(PTHREAD_COND_CLOCK, &now);
+
+ pthread_rwlock_rdlock(&poas.lock);
+
+ list_for_each(p, &poas.list) {
+ struct poa * poa = list_entry(p, struct poa, next);
+
+ list_for_each(q, &poa->flows) {
+ struct poa_flow * pf;
+
+ if (n == POA_SWEEP_MAX)
+ break;
+
+ pf = list_entry(q, struct poa_flow, pend);
+ if (pf->state != POA_FLOW_PENDING || pf->claimed)
+ continue;
+
+ if (now.tv_sec - pf->t0.tv_sec < POA_PEND_TIMEO)
+ continue;
+
+ if (pf->flow_id >= 0 && !pf_steal(pf))
+ continue;
+
+ dead[n++] = pf;
+ }
+
+ if (n == POA_SWEEP_MAX)
+ break;
+ }
+
+ pthread_rwlock_unlock(&poas.lock);
+
+ for (i = 0; i < n; ++i)
+ pf_destroy(dead[i]);
+
+ pthread_mutex_lock(&poas.mtx);
+
+ list_for_each_safe(p, q, &poas.deferred.list) {
+ struct poa_deferred * d;
+
+ d = list_entry(p, struct poa_deferred, next);
+ if (now.tv_sec - d->t0.tv_sec < POA_PEND_TIMEO)
+ continue;
+
+ llist_del(&d->next, &poas.deferred);
+ freebuf(d->data);
+ free(d);
+ }
+
+ pthread_mutex_unlock(&poas.mtx);
+}
+
+void poa_rx_mgmt(struct poa * poa,
+ const struct poa_addr * src,
+ const uint8_t * buf,
+ size_t len)
+{
+ struct poa_mgmt_frame * frame;
+
+ if (len < sizeof(struct poa_mgmt_msg) || len > POA_MGMT_BUFSZ)
+ return;
+
+ POA_STAT_BUMP(poa, mgmt_rx);
+
+ frame = malloc(offsetof(struct poa_mgmt_frame, buf) + len);
+ if (frame == NULL)
+ return;
+
+ frame->poa = poa;
+ frame->src = *src;
+ frame->len = len;
+
+ memcpy(frame->buf, buf, len);
+
+ pthread_mutex_lock(&poas.mgmt_mtx);
+
+ if (poa->n_mgmt >= POA_MGMT_QMAX) {
+ pthread_mutex_unlock(&poas.mgmt_mtx);
+ free(frame);
+ return;
+ }
+
+ ++poa->n_mgmt;
+
+ llist_add_tail(&frame->next, &poas.mgmt_frames);
+
+ pthread_cond_signal(&poas.mgmt_cond);
+
+ pthread_mutex_unlock(&poas.mgmt_mtx);
+}
+
+static void mgmt_frames_purge(const struct poa * poa)
+{
+ struct list_head * p;
+ struct list_head * h;
+
+ pthread_mutex_lock(&poas.mgmt_mtx);
+
+ list_for_each_safe(p, h, &poas.mgmt_frames.list) {
+ struct poa_mgmt_frame * frame;
+
+ frame = list_entry(p, struct poa_mgmt_frame, next);
+ if (frame->poa != poa)
+ continue;
+
+ --frame->poa->n_mgmt;
+
+ llist_del(&frame->next, &poas.mgmt_frames);
+
+ free(frame);
+ }
+
+ pthread_mutex_unlock(&poas.mgmt_mtx);
+}
+
+static bool poa_has_id(const struct poa * poa,
+ const struct poa_spec * spec)
+{
+ if (poa->type != spec->type)
+ return false;
+
+ return poa->ops->poa_has_id(poa, spec);
+}
+
+/*
+ * The PoA carrying dst, as the transport judges it. -EPERM if none
+ * matches, -EINVAL if several do. Caller holds poas.lock.
+ */
+static int poa_lookup(const struct poa_addr * dst,
+ struct poa ** out)
+{
+ struct list_head * p;
+ struct poa * found = NULL;
+
+ list_for_each(p, &poas.list) {
+ struct poa * poa = list_entry(p, struct poa, next);
+
+ if (poa->type != dst->type)
+ continue;
+
+ if (!poa->ops->poa_match(poa, dst))
+ continue;
+
+ if (found != NULL) /* nothing given, two candidates */
+ return -EINVAL;
+
+ found = poa;
+ }
+
+ if (found == NULL)
+ return -EPERM;
+
+ *out = found;
+
+ return 0;
+}
+
+/* Sends are deadlined, bounding the lock hold on a full queue. */
+int poa_bcast_mgmt(const struct poa_addr * dst,
+ const uint8_t * buf,
+ size_t len)
+{
+ struct list_head * p;
+ int n = 0;
+
+ pthread_rwlock_rdlock(&poas.lock);
+
+ list_for_each(p, &poas.list) {
+ struct poa * poa = list_entry(p, struct poa, next);
+
+ if (poa->type != dst->type)
+ continue;
+
+ if (!poa->ops->poa_match(poa, dst))
+ continue;
+
+ if (poa->ops->poa_send_mgmt(poa, dst, buf, len) < 0)
+ continue;
+
+ /* All management broadcasts are name queries. */
+ POA_STAT_BUMP(poa, qry_tx);
+ ++n;
+ }
+
+ pthread_rwlock_unlock(&poas.lock);
+
+ return n;
+}
+
+static bool deadline_is_malformed(const struct timespec * timeo)
+{
+ if (timeo == NULL)
+ return false;
+
+ if (timeo->tv_sec < 0 || timeo->tv_nsec < 0)
+ return true;
+
+ return timeo->tv_nsec >= BILLION;
+}
+
+/*
+ * Complete addr for dst on any backend that can query. The ops are
+ * collected under poas.lock but called outside it: a query blocks up
+ * to its deadline and takes the lock again to broadcast. The tables
+ * are static, so nothing dangles; a struct poa cannot be carried
+ * across the unlock. The deadline applies per backend.
+ */
+int poa_query(const char * dst,
+ const struct timespec * timeo,
+ struct poa_addr * addr)
+{
+ const struct poa_ops * cand[POA_MAX_POAS];
+ enum poa_type type[POA_MAX_POAS];
+ struct list_head * p;
+ size_t n = 0;
+ size_t i;
+ int err = -ENOTSUP;
+
+ if (dst == NULL || addr == NULL)
+ return -EINVAL;
+
+ if (deadline_is_malformed(timeo))
+ return -EINVAL;
+
+ pthread_rwlock_rdlock(&poas.lock);
+
+ list_for_each(p, &poas.list) {
+ struct poa * poa = list_entry(p, struct poa, next);
+
+ if (poa->ops->poa_query == NULL)
+ continue;
+
+ for (i = 0; i < n && cand[i] != poa->ops; i++)
+ ;
+ if (i < n)
+ continue;
+
+ /* One type per backend: eth. Revisit if that changes. */
+ cand[n] = poa->ops;
+ type[n++] = poa->type;
+ }
+
+ pthread_rwlock_unlock(&poas.lock);
+
+ for (i = 0; i < n; i++) {
+ memset(addr, 0, sizeof(*addr));
+
+ addr->type = type[i];
+
+ err = cand[i]->poa_query(dst, timeo, addr);
+ if (err == 0)
+ return 0;
+ }
+
+ return err;
+}
+
+static int poa_check(const struct poa_addr * dst)
+{
+ struct poa * poa;
+ int err;
+
+ pthread_rwlock_rdlock(&poas.lock);
+
+ err = poa_lookup(dst, &poa);
+
+ pthread_rwlock_unlock(&poas.lock);
+
+ return err;
+}
+
+static int poa_alloc(const struct poa_addr * dst,
+ qosspec_t qs,
+ const buffer_t * req,
+ buffer_t * resp,
+ struct poa_flow ** pf_out,
+ uint32_t * mtu,
+ const struct timespec * timeo)
+{
+ struct timespec dflt = TIMESPEC_INIT_MS(POA_ALLOC_TIMEO);
+ struct timespec rintv = TIMESPEC_INIT_MS(POA_RETRY_TIMEO);
+ struct poa_flow * pf;
+ struct poa * poa;
+ struct timespec abstime;
+ struct timespec now;
+ struct timespec retry;
+ int err;
+
+ pthread_rwlock_wrlock(&poas.lock);
+
+ err = poa_lookup(dst, &poa);
+ if (err < 0) {
+ pthread_rwlock_unlock(&poas.lock);
+ return err;
+ }
+
+ pf = pf_create(poa, dst);
+ if (pf == NULL) {
+ pthread_rwlock_unlock(&poas.lock);
+ return -ENOMEM;
+ }
+
+ pf->claimed = true;
+
+ pthread_rwlock_unlock(&poas.lock);
+
+ clock_gettime(PTHREAD_COND_CLOCK, &abstime);
+ ts_add(&abstime, timeo != NULL ? timeo : &dflt, &abstime);
+
+ pthread_mutex_lock(&pf->mtx);
+
+ while (!pf->replied) {
+ pthread_mutex_unlock(&pf->mtx);
+
+ err = mgmt_send(poa, dst, POA_FLOW_REQ, pf->eid, 0, qs, 0, req);
+
+ pthread_mutex_lock(&pf->mtx);
+
+ if (err < 0 && err != -ETIMEDOUT && err != -EAGAIN) {
+ err = -EIO;
+ goto fail;
+ }
+
+ if (pf->replied)
+ break;
+
+ clock_gettime(PTHREAD_COND_CLOCK, &now);
+ ts_add(&now, &rintv, &retry);
+
+ if (ts_diff_ns(&retry, &abstime) > 0)
+ retry = abstime;
+
+ pthread_cond_timedwait(&pf->cond, &pf->mtx, &retry);
+
+ if (pf->replied)
+ break;
+
+ clock_gettime(PTHREAD_COND_CLOCK, &now);
+
+ if (ts_diff_ns(&now, &abstime) >= 0) {
+ err = -ETIMEDOUT;
+ goto fail;
+ }
+ }
+
+ if (pf->response != 0) {
+ err = -ECONNREFUSED;
+ goto fail;
+ }
+
+ *resp = pf->resp_data;
+ pf->resp_data.len = 0;
+ pf->resp_data.data = NULL;
+
+ pthread_mutex_unlock(&pf->mtx);
+
+ *mtu = poa->ops->poa_mtu(poa, dst);
+ *pf_out = pf;
+
+ return 0;
+
+ fail:
+ pthread_mutex_unlock(&pf->mtx);
+ pf_destroy(pf);
+ return err;
+}
+
+static void poa_alloc_fail(struct poa_flow * pf)
+{
+ pf_destroy(pf);
+}
+
+void poa_flow_attach(struct poa_flow * pf,
+ int flow_id,
+ struct ssm_rbuff * rx_rb)
+{
+ struct poa * poa = pf->poa;
+
+ if (pf->flow_id != flow_id)
+ pf_set_flow_id(pf, flow_id);
+
+ pthread_rwlock_wrlock(&poas.lock);
+
+ pf->state = POA_FLOW_ALLOCATED;
+
+ rcu_wrlock(&poa->guard);
+
+ rcu_publish(rx_rb);
+ rcu_assign(pf->rx_rb, rx_rb);
+
+ rcu_wrunlock(&poa->guard);
+
+ pthread_rwlock_unlock(&poas.lock);
+}
+
+void poa_flow_detach(struct poa_flow * pf)
+{
+ struct poa * poa = pf->poa;
+
+ pthread_rwlock_wrlock(&poas.lock);
+
+ pf->state = POA_FLOW_DEAD;
+
+ rcu_wrlock(&poa->guard);
+
+ rcu_assign(pf->rx_rb, NULL);
+
+ rcu_wrunlock(&poa->guard);
+
+ pthread_rwlock_unlock(&poas.lock);
+
+ pf_destroy(pf);
+}
+
+static size_t flows_updown(struct poa * poa,
+ bool up)
+{
+ struct list_head * p;
+ size_t n = 0;
+
+ list_for_each(p, &poa->flows) {
+ struct poa_flow * pf;
+ struct flow * flow;
+
+ pf = list_entry(p, struct poa_flow, pend);
+ if (pf->state != POA_FLOW_ALLOCATED || pf->flow_id < 0)
+ continue;
+
+ flow = &proc.flows[proc.id_to_fd[pf->flow_id].fd];
+ if (flow->info.id != pf->flow_id)
+ continue;
+
+ if (((flow->oflags & FLOWFDOWN) != 0) != !up)
+ ++n;
+
+ if (up) {
+ flow->oflags &= ~FLOWFDOWN;
+
+ ssm_rbuff_clr_bits(flow->rx_rb, RB_FLOWDOWN);
+ } else {
+ flow->oflags |= FLOWFDOWN;
+
+ ssm_rbuff_set_bits(flow->rx_rb, RB_FLOWDOWN);
+ }
+
+ ssm_flow_set_notify(proc.fqset, pf->flow_id,
+ up ? FLOW_UP : FLOW_DOWN);
+ }
+
+ return n;
+}
+
+size_t poa_link_updown(int id,
+ bool up)
+{
+ struct list_head * p;
+ size_t n = 0;
+
+ pthread_rwlock_wrlock(&proc.lock);
+ pthread_rwlock_rdlock(&poas.lock);
+
+ list_for_each(p, &poas.list) {
+ struct poa * poa = list_entry(p, struct poa, next);
+
+ if (poa->ops->poa_link_match == NULL)
+ continue;
+
+ if (!poa->ops->poa_link_match(poa, id))
+ continue;
+
+ n += flows_updown(poa, up);
+ }
+
+ pthread_rwlock_unlock(&poas.lock);
+ pthread_rwlock_unlock(&proc.lock);
+
+ return n;
+}
+
+static size_t poa_ewma(size_t avg,
+ size_t sz)
+{
+ if (avg == 0)
+ return sz;
+
+ avg = avg + (sz >> POA_AVG_SHIFT) - (avg >> POA_AVG_SHIFT);
+
+ return avg == 0 ? 1 : avg;
+}
+
+static void poa_avg_len_update(struct poa * poa,
+ size_t sz)
+{
+ STORE_RELAXED(&poa->avg_len, poa_ewma(LOAD_RELAXED(&poa->avg_len), sz));
+}
+
+static size_t poa_qlen(struct poa * poa)
+{
+ struct timespec now;
+ uint64_t ns;
+ size_t qlen;
+
+ clock_gettime(PTHREAD_COND_CLOCK, &now);
+
+ ns = TS_TO_UINT64(now);
+ if (ns - LOAD_RELAXED(&poa->q_time) < POA_QLEN_GATE)
+ return LOAD_RELAXED(&poa->q_cache);
+
+ qlen = poa->ops->poa_qlen(poa);
+
+ STORE_RELAXED(&poa->q_cache, qlen);
+ STORE_RELAXED(&poa->q_time, ns);
+
+ return qlen;
+}
+
+static void poa_cost_sample(struct poa * poa,
+ size_t before,
+ size_t len)
+{
+ size_t after;
+
+ after = poa->ops->poa_qlen(poa);
+ if (after <= before)
+ return; /* drained; nothing to learn */
+
+ after -= before;
+ if (after < len || after > (len << 2) + 1024)
+ return;
+
+ STORE_RELAXED(&poa->avg_ovh,
+ poa_ewma(LOAD_RELAXED(&poa->avg_ovh), after - len));
+}
+
+int poa_flow_tx(struct poa_flow * pf,
+ struct ssm_pk_buff * spb,
+ bool block,
+ const struct timespec * abstime)
+{
+ struct poa * poa = pf->poa;
+ size_t len = ssm_pk_buff_len(spb);
+ size_t before = 0;
+ bool sample;
+ int ret;
+
+ sample = (LOAD_RELAXED(&poa->n_tx) & POA_COST_MASK) == 0;
+ if (sample)
+ before = poa->ops->poa_qlen(poa);
+
+ ret = poa->ops->poa_send(poa, &pf->r_addr, pf->r_eid, spb, block,
+ abstime);
+ if (ret < 0) { /* the caller releases the buffer */
+ POA_STAT_BUMP(poa, snd_fail);
+ return ret;
+ }
+
+ POA_STAT_BUMP(poa, tx_pkts);
+ POA_STAT_ADD(poa, tx_bytes, len);
+
+ FETCH_ADD_RELAXED(&poa->n_tx, 1);
+
+ poa_avg_len_update(poa, len);
+
+ if (sample)
+ poa_cost_sample(poa, before, len);
+
+ poa_spb_release(spb);
+
+ return 0;
+}
+
+size_t poa_flow_qlen(const struct poa_flow * pf)
+{
+ struct poa * poa = pf->poa;
+ uint64_t bytes;
+ size_t cost;
+ size_t pkts;
+ size_t byts;
+
+ if (poa->ops->poa_qpkts != NULL &&
+ poa->ops->poa_qpkts(poa, &pkts, &byts) == 0)
+ return byts;
+
+ cost = LOAD_RELAXED(&poa->avg_len) + LOAD_RELAXED(&poa->avg_ovh);
+ if (LOAD_RELAXED(&poa->avg_ovh) == 0 || cost == 0)
+ return poa_qlen(poa); /* overstated beats false empty */
+
+ bytes = (uint64_t) poa_qlen(poa) * LOAD_RELAXED(&poa->avg_len);
+
+ return (size_t) (bytes / cost);
+}
+
+size_t poa_flow_qpkts(const struct poa_flow * pf)
+{
+ struct poa * poa = pf->poa;
+ size_t cost;
+ size_t pkts;
+ size_t byts;
+
+ if (poa->ops->poa_qpkts != NULL &&
+ poa->ops->poa_qpkts(poa, &pkts, &byts) == 0)
+ return pkts;
+
+ cost = LOAD_RELAXED(&poa->avg_len) + LOAD_RELAXED(&poa->avg_ovh);
+ if (LOAD_RELAXED(&poa->avg_ovh) == 0 || cost == 0)
+ return 0;
+
+ return poa_qlen(poa) / cost;
+}
+
+size_t poa_flow_mean_len(const struct poa_flow * pf)
+{
+ return LOAD_RELAXED(&pf->poa->avg_len);
+}
+
+int poa_flow_qid(const struct poa_flow * pf)
+{
+ return pf->poa->qid;
+}
+
+void poa_flow_ready(struct poa_flow * pf)
+{
+ buffer_t data;
+ int answer;
+
+ if (pf == NULL)
+ return;
+
+ clrbuf(data);
+
+ pthread_rwlock_wrlock(&poas.lock);
+
+ if (!pf->answered || pf->answer_sent) {
+ pthread_rwlock_unlock(&poas.lock);
+ return;
+ }
+
+ answer = pf->answer;
+
+ if (pf->answer_data.len > 0) {
+ data.data = malloc(pf->answer_data.len);
+ if (data.data != NULL) {
+ memcpy(data.data, pf->answer_data.data,
+ pf->answer_data.len);
+ data.len = pf->answer_data.len;
+ }
+ }
+
+ pf->answer_sent = true;
+
+ pthread_rwlock_unlock(&poas.lock);
+
+ mgmt_send(pf->poa, &pf->r_addr, POA_FLOW_REPLY, pf->eid,
+ pf->r_eid, qos_raw, answer, &data);
+
+ freebuf(data);
+}
+
+int poa_flow_alloc_resp(int flow_id,
+ int response,
+ const buffer_t * data)
+{
+ struct poa_deferred * d;
+ struct poa_flow * pf;
+
+ if (flow_id < 0 || flow_id >= SYS_MAX_FLOWS)
+ return -EPERM;
+
+ pthread_mutex_lock(&poas.mtx);
+
+ pf = poas.id_to_pf[flow_id];
+ if (pf != NULL) {
+ pthread_mutex_unlock(&poas.mtx);
+ return pf_answer(pf, response, data);
+ }
+
+ if (poas.deferred.len >= POA_DEFER_MAX) {
+ pthread_mutex_unlock(&poas.mtx);
+ return -ENOMEM;
+ }
+
+ d = malloc(sizeof(*d));
+ if (d == NULL) {
+ pthread_mutex_unlock(&poas.mtx);
+ return -ENOMEM;
+ }
+
+ memset(d, 0, sizeof(*d));
+
+ clock_gettime(PTHREAD_COND_CLOCK, &d->t0);
+
+ d->flow_id = flow_id;
+ d->response = response;
+
+ if (data != NULL && data->len > 0) {
+ d->data.data = malloc(data->len);
+ if (d->data.data == NULL) {
+ free(d);
+ pthread_mutex_unlock(&poas.mtx);
+ return -ENOMEM;
+ }
+ memcpy(d->data.data, data->data, data->len);
+
+ d->data.len = data->len;
+ }
+
+ llist_add_tail(&d->next, &poas.deferred);
+
+ pthread_mutex_unlock(&poas.mtx);
+
+ return 0;
+}
+
+int poa_flow_update(int flow_id,
+ const buffer_t * data)
+{
+ struct poa_flow * pf;
+
+ pf = pf_get(flow_id);
+ if (pf == NULL)
+ return -EPERM;
+
+ return mgmt_send(pf->poa, &pf->r_addr, POA_FLOW_UPDATE, pf->eid,
+ pf->r_eid, qos_raw, 0, data);
+}
+
+/* The PoA state is released when the flow itself is torn down. */
+int poa_flow_dealloc(int flow_id)
+{
+ (void) flow_id;
+
+ return 0;
+}
+
+/* PoA id 0 = management channel. */
+static struct poa * poa_create(enum poa_type type,
+ const struct poa_ops * ops,
+ size_t n_eids)
+{
+ struct poa * poa;
+
+ poa = malloc(sizeof(*poa));
+ if (poa == NULL)
+ goto fail_malloc;
+
+ memset(poa, 0, sizeof(*poa));
+
+ poa->eid_to_pf = malloc(sizeof(*poa->eid_to_pf) * n_eids);
+ if (poa->eid_to_pf == NULL)
+ goto fail_map;
+
+ memset(poa->eid_to_pf, 0, sizeof(*poa->eid_to_pf) * n_eids);
+
+ poa->eids = bmp_create(n_eids - 1, 1);
+ if (poa->eids == NULL)
+ goto fail_bmp;
+
+ if (rcu_guard_init(&poa->guard) != 0)
+ goto fail_guard;
+
+ list_head_init(&poa->next);
+ list_head_init(&poa->flows);
+
+ poa->type = type;
+ poa->ops = ops;
+ poa->mpl = ops->mpl;
+ poa->n_eids = n_eids;
+ poa->qid = -1;
+
+ return poa;
+
+ fail_guard:
+ bmp_destroy(poa->eids);
+ fail_bmp:
+ free(poa->eid_to_pf);
+ fail_map:
+ free(poa);
+ fail_malloc:
+ return NULL;
+}
+
+static void poa_destroy(struct poa * poa)
+{
+ rcu_guard_fini(&poa->guard);
+
+ bmp_destroy(poa->eids);
+
+ free(poa->eid_to_pf);
+ free(poa);
+}
+
+static void poa_teardown(struct poa * poa)
+{
+ if (poas.state >= POA_RUNNING)
+ poa->ops->poa_stop(poa);
+
+ mgmt_frames_purge(poa);
+
+ poa->ops->poa_detach(poa);
+
+ poa_destroy(poa);
+}
+
+static void poa_detach_all(void)
+{
+ pthread_rwlock_wrlock(&poas.lock);
+
+ while (!list_is_empty(&poas.list)) {
+ struct poa * poa;
+
+ poa = list_first_entry(&poas.list, struct poa, next);
+
+ list_del(&poa->next);
+
+ pthread_rwlock_unlock(&poas.lock);
+
+ poa_teardown(poa);
+
+ pthread_rwlock_wrlock(&poas.lock);
+ }
+
+ pthread_rwlock_unlock(&poas.lock);
+}
+
+static int poa_do_detach(const struct poa_detach_req * req)
+{
+ struct list_head * p;
+ struct poa * found = NULL;
+
+ pthread_rwlock_wrlock(&poas.lock);
+
+ list_for_each(p, &poas.list) {
+ struct poa * poa = list_entry(p, struct poa, next);
+
+ if (!poa_has_id(poa, &req->spec))
+ continue;
+
+ found = poa;
+ break;
+ }
+
+ if (found == NULL) {
+ pthread_rwlock_unlock(&poas.lock);
+ return -ENOENT;
+ }
+
+ if (!list_is_empty(&found->flows)) {
+ pthread_rwlock_unlock(&poas.lock);
+ return -EBUSY;
+ }
+
+ list_del(&found->next);
+
+ pthread_rwlock_unlock(&poas.lock);
+
+ poa_teardown(found);
+
+ return 0;
+}
+
+static int poa_del(const struct poa_spec * spec)
+{
+ struct poa_detach_req req;
+ int cs;
+ int ret;
+
+ memset(&req, 0, sizeof(req));
+
+ req.spec = *spec;
+
+ pthread_setcancelstate(PTHREAD_CANCEL_DISABLE, &cs);
+
+ if (poas.state < POA_RUNNING) { /* set before workers run */
+ ret = poa_do_detach(&req);
+ goto out;
+ }
+
+ pthread_mutex_lock(&poas.mgmt_mtx);
+
+ if (poas.mgmt_stop) { /* stopped: poa_fini reaps these */
+ pthread_mutex_unlock(&poas.mgmt_mtx);
+
+ ret = -EBUSY;
+ goto out;
+ }
+
+ list_add_tail(&req.next, &poas.detach);
+
+ pthread_cond_signal(&poas.mgmt_cond);
+
+ while (!req.done)
+ pthread_cond_wait(&poas.done_cond, &poas.mgmt_mtx);
+
+ pthread_mutex_unlock(&poas.mgmt_mtx);
+
+ ret = req.result;
+ out:
+ pthread_setcancelstate(cs, NULL);
+
+ return ret;
+}
+
+static __inline__ bool mgmt_idle(void)
+{
+ if (poas.mgmt_stop)
+ return false;
+
+ if (!llist_is_empty(&poas.mgmt_frames))
+ return false;
+
+ return list_is_empty(&poas.detach);
+}
+
+static void detach_run(void)
+{
+ while (!list_is_empty(&poas.detach)) {
+ struct poa_detach_req * req;
+
+ req = list_first_entry(&poas.detach, struct poa_detach_req,
+ next);
+ list_del(&req->next);
+
+ pthread_mutex_unlock(&poas.mgmt_mtx);
+
+ req->result = poa_do_detach(req);
+
+ pthread_mutex_lock(&poas.mgmt_mtx);
+
+ req->done = true;
+
+ pthread_cond_broadcast(&poas.done_cond);
+ }
+}
+
+static void * mgmt_handler(void * o)
+{
+ struct timespec intv = TIMESPEC_INIT_MS(POA_SWEEP_TIMEO);
+
+ (void) o;
+
+ while (true) {
+ struct poa_mgmt_frame * frame;
+ struct timespec abstime;
+
+ pthread_mutex_lock(&poas.mgmt_mtx);
+
+ detach_run();
+
+ while (mgmt_idle()) {
+ clock_gettime(PTHREAD_COND_CLOCK, &abstime);
+ ts_add(&abstime, &intv, &abstime);
+
+ if (pthread_cond_timedwait(&poas.mgmt_cond,
+ &poas.mgmt_mtx,
+ &abstime) == ETIMEDOUT) {
+ pthread_mutex_unlock(&poas.mgmt_mtx);
+ sweep_pending();
+ pthread_mutex_lock(&poas.mgmt_mtx);
+ }
+ }
+
+ if (poas.mgmt_stop) {
+ detach_run(); /* nobody else serves these */
+
+ pthread_mutex_unlock(&poas.mgmt_mtx);
+ break;
+ }
+
+ if (llist_is_empty(&poas.mgmt_frames)) {
+ pthread_mutex_unlock(&poas.mgmt_mtx);
+ continue;
+ }
+
+ frame = llist_first_entry(&poas.mgmt_frames,
+ struct poa_mgmt_frame, next);
+ llist_del(&frame->next, &poas.mgmt_frames);
+
+ --frame->poa->n_mgmt;
+
+ pthread_mutex_unlock(&poas.mgmt_mtx);
+
+ mgmt_frame_handle(frame);
+
+ free(frame);
+ }
+
+ return (void *) 0;
+}
+
+#ifndef HAVE_ETH
+/* Only the Ethernet transport reports link events. */
+int poa_monitor_open(void)
+{
+ return -1;
+}
+
+void poa_monitor_read(int fd)
+{
+ (void) fd;
+}
+#endif
+
+static void * poa_monitor(void * o)
+{
+ (void) o;
+
+ while (true)
+ poa_monitor_read(poas.mon_fd);
+
+ return (void *) 0;
+}
+
+static int monitor_start(void)
+{
+ poas.mon_fd = poa_monitor_open();
+ if (poas.mon_fd < 0)
+ return 0;
+
+ if (pthread_create(&poas.monitor, NULL, poa_monitor, NULL) == 0)
+ return 0;
+
+ close(poas.mon_fd);
+
+ poas.mon_fd = -1;
+
+ return -1;
+}
+
+static void monitor_stop(void)
+{
+ if (poas.mon_fd < 0)
+ return;
+
+ pthread_cancel(poas.monitor);
+ pthread_join(poas.monitor, NULL);
+
+ close(poas.mon_fd);
+
+ poas.mon_fd = -1;
+}
+
+int poa_start(void)
+{
+ struct list_head * p;
+ struct list_head * h;
+
+ if (poas.state == POA_NULL)
+ return 0;
+
+ if (pthread_create(&poas.mgmt_handler, NULL, mgmt_handler, NULL) != 0)
+ goto fail_mgmt;
+
+ if (monitor_start() < 0)
+ goto fail_monitor;
+
+ pthread_rwlock_rdlock(&poas.lock);
+
+ list_for_each(p, &poas.list) {
+ struct poa * poa = list_entry(p, struct poa, next);
+
+ if (poa->ops->poa_start(poa) < 0)
+ goto fail_reader;
+ }
+
+ poas.state = POA_RUNNING;
+
+ pthread_rwlock_unlock(&poas.lock);
+
+ return 0;
+
+ fail_reader:
+ list_for_each(h, &poas.list) {
+ struct poa * poa = list_entry(h, struct poa, next);
+
+ if (h == p)
+ break;
+
+ poa->ops->poa_stop(poa);
+ }
+
+ pthread_rwlock_unlock(&poas.lock);
+
+ monitor_stop();
+ fail_monitor:
+ pthread_mutex_lock(&poas.mgmt_mtx);
+
+ poas.mgmt_stop = true;
+
+ pthread_cond_broadcast(&poas.mgmt_cond);
+ pthread_mutex_unlock(&poas.mgmt_mtx);
+
+ pthread_join(poas.mgmt_handler, NULL);
+ fail_mgmt:
+ return -1;
+}
+
+void poa_stop(void)
+{
+ struct list_head * p;
+
+ if (poas.state < POA_RUNNING)
+ return;
+
+ pthread_mutex_lock(&poas.mgmt_mtx);
+
+ poas.mgmt_stop = true;
+
+ pthread_cond_broadcast(&poas.mgmt_cond);
+ pthread_mutex_unlock(&poas.mgmt_mtx);
+
+ pthread_join(poas.mgmt_handler, NULL);
+
+ monitor_stop();
+
+ pthread_rwlock_rdlock(&poas.lock);
+
+ list_for_each(p, &poas.list) {
+ struct poa * poa = list_entry(p, struct poa, next);
+
+ poa->ops->poa_stop(poa);
+ }
+
+ pthread_rwlock_unlock(&poas.lock);
+
+ poas.state = POA_INIT;
+}
+
+void poa_fini(void)
+{
+ if (poas.state == POA_NULL)
+ return;
+
+ poa_stop();
+
+ poas.state = POA_NULL;
+
+ poa_detach_all();
+
+#ifdef PROC_FLOW_STATS
+ rib_unreg(POA_RIB);
+#endif
+
+ pthread_rwlock_destroy(&poas.lock);
+ pthread_cond_destroy(&poas.done_cond);
+ pthread_cond_destroy(&poas.mgmt_cond);
+ pthread_mutex_destroy(&poas.mgmt_mtx);
+ pthread_cond_destroy(&poas.cond);
+ pthread_mutex_destroy(&poas.mtx);
+}
+
+/*
+ * Lowest queue id no attached PoA holds; detaching frees it by
+ * leaving the list. Caller holds poas.lock.
+ */
+static int poa_qid_alloc(void)
+{
+ struct list_head * p;
+ bool used[POA_MAX_POAS];
+ int i;
+
+ memset(used, 0, sizeof(used));
+
+ list_for_each(p, &poas.list) {
+ struct poa * poa = list_entry(p, struct poa, next);
+
+ if (poa->qid >= 0 && poa->qid < POA_MAX_POAS)
+ used[poa->qid] = true;
+ }
+
+ for (i = 0; i < POA_MAX_POAS; i++)
+ if (!used[i])
+ return i;
+
+ return -1;
+}
+
+static int poa_add(const struct poa_spec * spec,
+ const struct poa_ops * ops,
+ size_t n_eids)
+{
+ struct list_head * p;
+ struct poa * poa;
+ int err;
+
+ pthread_rwlock_rdlock(&poas.lock);
+
+ list_for_each(p, &poas.list) {
+ poa = list_entry(p, struct poa, next);
+ if (poa_has_id(poa, spec)) {
+ pthread_rwlock_unlock(&poas.lock);
+ return -EPERM;
+ }
+ }
+
+ pthread_rwlock_unlock(&poas.lock);
+
+ poa = poa_create(spec->type, ops, n_eids);
+ if (poa == NULL)
+ return -ENOMEM;
+
+ err = poa->ops->poa_attach(poa, spec);
+ if (err < 0)
+ goto fail_bind;
+
+ err = poa_addr_name(&poa->local, poa->name, sizeof(poa->name));
+ if (err < 0)
+ goto fail_start;
+
+ err = -1;
+
+ pthread_rwlock_wrlock(&poas.lock);
+
+ poa->qid = poa_qid_alloc();
+ if (poa->qid < 0) {
+ pthread_rwlock_unlock(&poas.lock);
+ goto fail_start;
+ }
+
+ if (poas.state >= POA_RUNNING && poa->ops->poa_start(poa) < 0) {
+ pthread_rwlock_unlock(&poas.lock);
+ goto fail_start;
+ }
+
+ list_add_tail(&poa->next, &poas.list);
+
+ pthread_rwlock_unlock(&poas.lock);
+
+ return 0;
+
+ fail_start:
+ poa->ops->poa_detach(poa);
+ fail_bind:
+ poa_destroy(poa);
+ return err;
+}
+
+/* The single place a type is bound to its transport. */
+int poa_attach(const struct poa_spec * poa)
+{
+ if (poa == NULL)
+ return -EINVAL;
+
+ switch (poa->type) {
+ case POA_UDP4:
+ /* FALLTHRU */
+ case POA_UDP6:
+ return poa_add(poa, &udp_poa_ops, POA_MAX_EIDS);
+ case POA_ETH:
+#ifdef HAVE_ETH
+ return poa_add(poa, &eth_poa_ops, POA_MAX_EIDS);
+#else
+ return -ENOTSUP;
+#endif
+ default:
+ return -ENOTSUP;
+ }
+}
+
+int poa_detach(const struct poa_spec * poa)
+{
+ if (poa == NULL)
+ return -EINVAL;
+
+ return poa_del(poa);
+}
+
+ssize_t poa_list(struct poa_spec * specs,
+ size_t max)
+{
+ struct list_head * p;
+ size_t n = 0;
+
+ if (specs == NULL)
+ return -EINVAL;
+
+ pthread_rwlock_rdlock(&poas.lock);
+
+ list_for_each(p, &poas.list) {
+ struct poa * poa = list_entry(p, struct poa, next);
+
+ if (n++ >= max)
+ continue;
+
+ memset(specs, 0, sizeof(*specs));
+
+ poa->ops->poa_spec(poa, specs);
+
+ specs++;
+ }
+
+ pthread_rwlock_unlock(&poas.lock);
+
+ return (ssize_t) n;
+}
+
+/*
+ * Complete peer for dst on the backend serving its type. The ops are
+ * borrowed under poas.lock and called outside it (see poa_query); if
+ * every PoA of the type detaches in between, the query's broadcast
+ * reaches nothing and reports -EPERM, as the lookup would.
+ */
+static int poa_peer_resolve(const char * dst,
+ struct poa_addr * peer)
+{
+ const struct poa_ops * ops = NULL;
+ struct list_head * p;
+
+ pthread_rwlock_rdlock(&poas.lock);
+
+ list_for_each(p, &poas.list) {
+ struct poa * poa = list_entry(p, struct poa, next);
+
+ if (poa->type == peer->type) {
+ ops = poa->ops;
+ break;
+ }
+ }
+
+ pthread_rwlock_unlock(&poas.lock);
+
+ if (ops == NULL) /* nothing could carry the flow */
+ return -EPERM;
+
+ if (ops->poa_query == NULL) /* these addresses arrive complete */
+ return 0;
+
+ return ops->poa_query(dst, NULL, peer);
+}
+
+/*
+ * Three steps: the IRMd creates the flow and prepares the key exchange,
+ * the PoA handshakes with the peer, the IRMd completes the
+ * exchange and hands us the key.
+ */
+int poa_flow_alloc(const char * dst,
+ const struct poa_addr * addr,
+ qosspec_t * qs,
+ const struct timespec * timeo)
+{
+ struct flow_info flow;
+ struct poa_flow * pf;
+ struct poa_addr peer;
+ struct crypt_sk crypt;
+ struct timespec t0;
+ struct timespec t1;
+ uint8_t key[SYMMKEYSZ];
+ uint8_t buf[SOCK_BUF_SIZE];
+ buffer_t msg = {SOCK_BUF_SIZE, buf};
+ buffer_t req;
+ buffer_t resp;
+ uint32_t mtu = 0;
+ int err;
+
+ if (addr == NULL)
+ return -EINVAL;
+
+ if (qs != NULL && qs->service == SVC_STREAM && qs->loss != 0)
+ return -EINVAL;
+
+ peer = *addr;
+
+ err = poa_peer_resolve(dst, &peer);
+ if (err < 0)
+ return err;
+
+ addr = &peer;
+
+ err = poa_check(addr);
+ if (err < 0)
+ return err;
+
+ memset(&flow, 0, sizeof(flow));
+
+ flow.n_pid = getpid();
+ flow.n_1_pid = getpid();
+ flow.qs = qs == NULL ? qos_raw : *qs;
+ if (poa_flow_alloc__irm_req_ser(&msg, &flow, dst) < 0)
+ return -ENOMEM;
+
+ err = send_recv_msg(&msg);
+ if (err < 0)
+ return err;
+
+ clrbuf(req);
+ clrbuf(resp);
+
+ err = poa_flow__irm_result_des(&msg, &flow, &req);
+ if (err < 0)
+ return err;
+
+ clock_gettime(PTHREAD_COND_CLOCK, &t0);
+
+ err = poa_alloc(addr, flow.qs, &req, &resp, &pf, &mtu, timeo);
+
+ freebuf(req);
+
+ if (err < 0)
+ goto fail_alloc;
+
+ clock_gettime(PTHREAD_COND_CLOCK, &t1);
+
+ flow.mtu = mtu;
+ flow.mpl = pf->poa->mpl;
+ msg.len = SOCK_BUF_SIZE;
+ msg.data = buf;
+ if (poa_flow_alloc_r__irm_req_ser(&msg, &flow, &resp, 0) < 0) {
+ err = -ENOMEM;
+ goto fail_resp;
+ }
+
+ freebuf(resp);
+
+ err = send_recv_msg(&msg);
+ if (err < 0)
+ goto fail_msg;
+
+ crypt.key = key;
+ crypt.epoch = 0;
+ crypt.role = CRYPT_ROLE_INIT;
+
+ err = flow__irm_result_des(&msg, &flow, &crypt);
+ if (err < 0)
+ goto fail_msg;
+
+ err = flow_init(&flow, &crypt, ts_diff_ns(&t1, &t0), pf);
+
+ crypt_secure_clear(key, SYMMKEYSZ);
+
+ if (err < 0)
+ goto fail_msg;
+
+ if (qs != NULL)
+ *qs = flow.qs;
+
+ return err;
+
+ fail_resp:
+ freebuf(resp);
+ fail_msg:
+ poa_alloc_fail(pf);
+ return err;
+ fail_alloc:
+ msg.len = SOCK_BUF_SIZE;
+ msg.data = buf;
+ if (poa_flow_alloc_r__irm_req_ser(&msg, &flow, NULL, err) == 0)
+ send_recv_msg(&msg);
+
+ return err;
+}
+
diff --git a/src/lib/poa/poa.h b/src/lib/poa/poa.h
new file mode 100644
index 00000000..014986a3
--- /dev/null
+++ b/src/lib/poa/poa.h
@@ -0,0 +1,364 @@
+/*
+ * Ouroboros - Copyright (C) 2016 - 2026
+ *
+ * Points of attachment (PoA) - internal API
+ *
+ * Dimitri Staessens <dimitri@ouroboros.rocks>
+ * Sander Vrijders <sander@ouroboros.rocks>
+ *
+ * This library is free software; you can redistribute it and/or
+ * modify it under the terms of the GNU Lesser General Public License
+ * version 2.1 as published by the Free Software Foundation.
+ *
+ * This library is distributed in the hope that it will be useful,
+ * but WITHOUT ANY WARRANTY; without even the implied warranty of
+ * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
+ * Lesser General Public License for more details.
+ *
+ * You should have received a copy of the GNU Lesser General Public
+ * License along with this library; if not, write to the Free Software
+ * Foundation, Inc., http://www.fsf.org/about/contact/.
+ */
+
+#ifndef OUROBOROS_LIB_POA_POA_H
+#define OUROBOROS_LIB_POA_POA_H
+
+#include <ouroboros/atomics.h>
+#include <ouroboros/ipcp-dev.h>
+#include <ouroboros/list.h>
+#include <ouroboros/qos.h>
+#include <ouroboros/rcu.h>
+#include <ouroboros/ssm_flow_set.h>
+#include <ouroboros/ssm_pool.h>
+#include <ouroboros/ssm_rbuff.h>
+#include <ouroboros/time.h>
+#include <ouroboros/utils.h>
+
+#include <errno.h>
+#include <limits.h>
+#include <poll.h>
+#include <pthread.h>
+#include <stdbool.h>
+#include <stdint.h>
+
+#define POA_MGMT_EID 0 /* reserved for the mgmt channel */
+
+#define POA_FLOW_REQ 1
+#define POA_FLOW_REPLY 2
+#define POA_FLOW_UPDATE 3
+#define POA_NAME_QUERY 4
+#define POA_NAME_REPLY 5
+
+#define POA_QUERY_HLEN 32 /* SHA3-256, the query hash algorithm */
+
+/* Fits "udp6.<ip6>.<port>", the longest display/RIB entry name. */
+#define POA_NAME_STRLEN 63
+
+/* Management message; every transport frames it the same way. */
+struct poa_mgmt_msg {
+ uint8_t code;
+ uint8_t resv[3];
+ uint32_t s_eid;
+ uint32_t d_eid;
+ int32_t response;
+ uint64_t bandwidth;
+ uint32_t delay;
+ uint32_t loss;
+ uint32_t ber;
+ uint32_t max_gap;
+ uint32_t timeout;
+ uint8_t availability;
+ uint8_t service;
+ uint16_t data_len;
+} __attribute__((packed));
+
+struct poa;
+struct poa_flow;
+
+#ifdef PROC_FLOW_STATS
+#define POA_STAT_BUMP(poa, field) FETCH_ADD_RELAXED(&(poa)->stat.field, 1)
+#define POA_STAT_ADD(poa, field, v) FETCH_ADD_RELAXED(&(poa)->stat.field, (v))
+#define POA_STAT_SUB(poa, field, v) FETCH_SUB_RELAXED(&(poa)->stat.field, (v))
+#define POA_STAT_LOAD(poa, field) LOAD_RELAXED(&(poa)->stat.field)
+#else
+#define POA_STAT_BUMP(poa, field) ((void) (poa))
+#define POA_STAT_ADD(poa, field, v) ((void) (poa), (void) (v))
+#define POA_STAT_SUB(poa, field, v) ((void) (poa), (void) (v))
+#define POA_STAT_LOAD(poa, field) ((void) (poa), (size_t) 0)
+#endif
+
+struct poa_stat {
+ size_t n_flows; /* gauge; the RIB reads it without poas.lock */
+ size_t rx_pkts; /* packets delivered to a flow */
+ size_t rx_bytes; /* payload bytes delivered */
+ size_t tx_pkts; /* packets handed to the transport */
+ size_t tx_bytes; /* payload bytes handed to the transport */
+ size_t mgmt_rx; /* frames queued for the mgmt handler */
+ size_t mgmt_tx; /* management frames sent */
+ size_t bad_eid; /* no flow on the EID a peer sent */
+ size_t dlv_fail; /* the rx ring above would not take it */
+ size_t buf_fail; /* no buffer to receive into */
+ size_t rcv_fail; /* transport read failed; the reader exits */
+ size_t snd_fail; /* transport send failed */
+ size_t qry_tx; /* name queries broadcast */
+ size_t qry_rx; /* name queries received */
+ size_t rep_tx; /* name replies sent, query matched */
+ size_t rep_rx; /* name replies received */
+};
+
+/* Spacing between transmit-depth samples; a depth costs a syscall. */
+#define POA_QLEN_GATE (100 * 1000) /* ns */
+
+
+/* Transport operations; public poa_X() dispatches to ops->poa_X. */
+struct poa_ops {
+ /* Parse own arm of the spec; validate; fill local and priv. */
+ int (* poa_attach)(struct poa * poa,
+ const struct poa_spec * spec);
+
+ void (* poa_detach)(struct poa * poa);
+
+ int (* poa_start)(struct poa * poa);
+
+ void (* poa_stop)(struct poa * poa);
+
+ /* Full queue: -EAGAIN unless block; then wait, to abstime if set. */
+ int (* poa_send)(struct poa * poa,
+ const struct poa_addr * dst,
+ uint32_t eid,
+ struct ssm_pk_buff * spb,
+ bool block,
+ const struct timespec * abstime);
+
+ int (* poa_send_mgmt)(struct poa * poa,
+ const struct poa_addr * dst,
+ const uint8_t * buf,
+ size_t len);
+
+ int (* poa_query)(const char * dst,
+ const struct timespec * timeo,
+ struct poa_addr * addr);
+
+ uint32_t (* poa_mtu)(struct poa * poa,
+ const struct poa_addr * dst);
+
+ /* Bytes queued in the transmit path of the PoA. */
+ size_t (* poa_qlen)(struct poa * poa);
+
+ /* Depth from the queue itself; NULL infers it from qlen. */
+ int (* poa_qpkts)(struct poa * poa,
+ size_t * pkts,
+ size_t * byts);
+
+ int (* poa_rib)(struct poa * poa,
+ char * buf,
+ size_t len);
+
+ /* Identity as a spec, e.g. for poa_list. */
+ void (* poa_spec)(const struct poa * poa,
+ struct poa_spec * spec);
+
+ /* Same identity as spec? Caller matched poa->type already. */
+ bool (* poa_has_id)(const struct poa * poa,
+ const struct poa_spec * spec);
+
+ /* Carries dst? Caller matched poa->type already. */
+ bool (* poa_match)(const struct poa * poa,
+ const struct poa_addr * dst);
+
+ /*
+ * Flows ride the link this id names; NULL: no link events.
+ * Ids are meaningful only to the backend whose monitor
+ * produced them; a single backend owns the monitor.
+ */
+ bool (* poa_link_match)(const struct poa * poa,
+ int id);
+
+ /* Maximum packet lifetime in the transport, seconds. */
+ time_t mpl;
+};
+struct poa {
+ struct list_head next;
+
+ enum poa_type type;
+ const struct poa_ops * ops;
+ void * priv;
+
+ struct poa_addr local; /* what peers dial us on */
+
+ /* Display/RIB entry name for local, e.g. "udp4.<ip>.<port>". */
+ char name[POA_NAME_STRLEN + 1];
+
+ time_t mpl;
+
+ /* Identifies the transmit queue the flows on this PoA share. */
+ int qid;
+
+ /* Mean sent packet size (bytes), EWMA over the send path. */
+ size_t avg_len;
+ /* Cost of one packet in the queue, in the transport's terms. */
+ size_t avg_ovh;
+ size_t n_tx;
+
+ /* Last queue depth read, and when, in the transport's terms. */
+ size_t q_cache;
+ uint64_t q_time;
+
+ /* Queued management frames, capped; poas.mgmt_mtx guards. */
+ size_t n_mgmt;
+
+#ifdef PROC_FLOW_STATS
+ struct poa_stat stat;
+#endif
+
+ struct bmp * eids;
+ struct poa_flow ** eid_to_pf;
+ size_t n_eids;
+
+ struct list_head flows; /* live flows, for repeats */
+
+ /* Keeps a flow and its ring alive under the receive path. */
+ struct rcu_guard guard;
+};
+
+/* poa/poa.c is part of the dev.c translation unit. */
+int poa_init(const char * name);
+
+int poa_start(void);
+
+void poa_stop(void);
+
+void poa_fini(void);
+
+/* Also answer name queries for the layer once enrolled. */
+int poa_set_layer(const char * layer);
+
+/* Hash of a name this process answers queries for? */
+bool poa_has_name(const uint8_t * hash);
+
+int poa_flow_tx(struct poa_flow * pf,
+ struct ssm_pk_buff * spb,
+ bool block,
+ const struct timespec * abstime);
+
+size_t poa_flow_qlen(const struct poa_flow * pf);
+
+size_t poa_flow_qpkts(const struct poa_flow * pf);
+
+int poa_flow_qid(const struct poa_flow * pf);
+
+size_t poa_flow_mean_len(const struct poa_flow * pf);
+
+void poa_flow_attach(struct poa_flow * pf,
+ int flow_id,
+ struct ssm_rbuff * rx_rb);
+
+void poa_flow_ready(struct poa_flow * pf);
+
+void poa_flow_detach(struct poa_flow * pf);
+
+struct poa_flow * poa_flow_take_pending(int flow_id);
+
+/* Addresses and management messages (poa/addr.c). */
+int poa_addr_cmp(const struct poa_addr * a,
+ const struct poa_addr * b);
+
+/* Display/RIB entry name, e.g. "udp4.<ip>.<port>". */
+int poa_addr_name(const struct poa_addr * a,
+ char * buf,
+ size_t len);
+
+void poa_mgmt_msg_ser(struct poa_mgmt_msg * msg,
+ uint8_t code,
+ uint32_t s_eid,
+ uint32_t d_eid,
+ qosspec_t qs,
+ int response,
+ size_t data_len);
+
+void poa_mgmt_msg_qos(const struct poa_mgmt_msg * msg,
+ qosspec_t * qs);
+
+/* Called by the transports. */
+void poa_rx_pkt(struct poa * poa,
+ uint32_t eid,
+ struct ssm_pk_buff * spb);
+
+void poa_rx_mgmt(struct poa * poa,
+ const struct poa_addr * src,
+ const uint8_t * buf,
+ size_t len);
+
+/* Reserve a buffer for a received packet, with transport headroom. */
+int poa_spb_reserve(struct ssm_pk_buff ** spb,
+ size_t len);
+
+void poa_spb_release(struct ssm_pk_buff * spb);
+
+/*
+ * All flows on PoAs whose poa_link_match reports this link id go up
+ * or down with it. Returns the number of flows whose state changed.
+ */
+size_t poa_link_updown(int id,
+ bool up);
+
+/*
+ * Link monitor: one socket for the whole subsystem, opened by
+ * poa_start(). Returns -1 where the transport has no monitor.
+ */
+int poa_monitor_open(void);
+
+/* Reads one batch of link events; cancellation point. */
+void poa_monitor_read(int fd);
+
+/* Broadcast a mgmt frame on every PoA matching dst; # sent. */
+int poa_bcast_mgmt(const struct poa_addr * dst,
+ const uint8_t * buf,
+ size_t len);
+
+/* Transport op tables. */
+extern const struct poa_ops udp_poa_ops;
+extern const struct poa_ops eth_poa_ops;
+
+/*
+ * Waits for a descriptor to take another packet, up to abstime.
+ * A NULL deadline waits indefinitely. Transports call this when
+ * their send reports the transmit queue full.
+ */
+static __inline__ int poa_wait_out(int fd,
+ const struct timespec * abstime)
+{
+ struct pollfd pfd;
+ struct timespec now;
+ long ms = -1;
+ bool clamped = false;
+ int ret;
+
+ if (abstime != NULL) {
+ clock_gettime(PTHREAD_COND_CLOCK, &now);
+
+ if (ts_diff_ns(abstime, &now) <= 0)
+ return -ETIMEDOUT;
+
+ ms = ts_diff_ms(abstime, &now) + 1; /* sub-ms must wait */
+ if (ms > INT_MAX) { /* poll takes an int */
+ ms = INT_MAX;
+ clamped = true;
+ }
+ }
+
+ pfd.fd = fd;
+ pfd.events = POLLOUT;
+ pfd.revents = 0;
+
+ ret = poll(&pfd, 1, (int) ms);
+ if (ret < 0)
+ return errno == EINTR ? 0 : -EIO;
+
+ if (ret == 0)
+ return clamped ? 0 : -ETIMEDOUT; /* clamped: retry */
+
+ return 0;
+}
+
+#endif /* OUROBOROS_LIB_POA_POA_H */
diff --git a/src/lib/poa/udp.c b/src/lib/poa/udp.c
new file mode 100644
index 00000000..6753347a
--- /dev/null
+++ b/src/lib/poa/udp.c
@@ -0,0 +1,633 @@
+/*
+ * Ouroboros - Copyright (C) 2016 - 2026
+ *
+ * Points of attachment (PoA) - UDP transport
+ *
+ * Dimitri Staessens <dimitri@ouroboros.rocks>
+ * Sander Vrijders <sander@ouroboros.rocks>
+ *
+ * This library is free software; you can redistribute it and/or
+ * modify it under the terms of the GNU Lesser General Public License
+ * version 2.1 as published by the Free Software Foundation.
+ *
+ * This library is distributed in the hope that it will be useful,
+ * but WITHOUT ANY WARRANTY; without even the implied warranty of
+ * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
+ * Lesser General Public License for more details.
+ *
+ * You should have received a copy of the GNU Lesser General Public
+ * License along with this library; if not, write to the Free Software
+ * Foundation, Inc., http://www.fsf.org/about/contact/.
+ */
+
+#if defined(__APPLE__)
+#define _BSD_SOURCE
+#define _DARWIN_C_SOURCE
+#elif defined(__FreeBSD__)
+#define __BSD_VISIBLE 1
+#elif defined(__linux__) || defined(__CYGWIN__)
+#ifndef _DEFAULT_SOURCE
+#define _DEFAULT_SOURCE
+#endif
+#else
+#ifndef _POSIX_C_SOURCE
+#define _POSIX_C_SOURCE 200809L
+#endif
+#endif
+
+#include "config.h"
+
+#define OUROBOROS_PREFIX "poa-udp"
+
+#include <ouroboros/endian.h>
+#include <ouroboros/errno.h>
+#include <ouroboros/logs.h>
+#include <ouroboros/time.h>
+#include <ouroboros/utils.h>
+
+#include "poa.h"
+
+#ifdef __linux__
+#include <linux/sockios.h>
+#endif
+#include <arpa/inet.h>
+#include <netinet/in.h>
+#include <sys/ioctl.h>
+#include <sys/socket.h>
+#include <sys/uio.h>
+
+#include <errno.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <unistd.h>
+
+#define UDP_HDR_LEN sizeof(uint32_t) /* PoA id */
+#define UDP_MAX_PAYLOAD (POA_UDP_RD_BUF - UDP_HDR_LEN)
+/* The reader buffer must fit a full mgmt frame at any tuning. */
+#define UDP_MAX_PACKET MAX(POA_UDP_RD_BUF, POA_MGMT_FRAME_SIZE + UDP_HDR_LEN)
+#define UDP_IP4_OVERH 28U /* IPv4 + UDP */
+#define UDP_IP6_OVERH 48U /* IPv6 + UDP */
+/* Wait for the link to come back before reading it again. */
+#define UDP_DOWN_TIMEO 100 /* ms */
+
+union udp_saddr {
+ struct sockaddr sa;
+ struct sockaddr_in in;
+ struct sockaddr_in6 in6;
+};
+
+struct udp_priv {
+ int s_fd;
+ int af;
+ union udp_saddr s_saddr;
+ pthread_t reader;
+ bool running;
+};
+
+static socklen_t saddr_len(int af)
+{
+ if (af == AF_INET)
+ return sizeof(struct sockaddr_in);
+
+ return sizeof(struct sockaddr_in6);
+}
+
+static void addr_to_saddr(const struct poa_addr * addr,
+ union udp_saddr * saddr)
+{
+ memset(saddr, 0, sizeof(*saddr));
+
+ if (addr->type == POA_UDP4) {
+ saddr->in.sin_family = AF_INET;
+ saddr->in.sin_addr = addr->udp4.ip_addr;
+ saddr->in.sin_port = htons(addr->udp4.port);
+ } else {
+ saddr->in6.sin6_family = AF_INET6;
+ saddr->in6.sin6_addr = addr->udp6.ip_addr;
+ saddr->in6.sin6_port = htons(addr->udp6.port);
+ }
+}
+
+static void saddr_to_addr(const union udp_saddr * saddr,
+ struct poa_addr * addr)
+{
+ memset(addr, 0, sizeof(*addr));
+
+ if (saddr->sa.sa_family == AF_INET) {
+ addr->type = POA_UDP4;
+ addr->udp4.ip_addr = saddr->in.sin_addr;
+ addr->udp4.port = ntohs(saddr->in.sin_port);
+ } else {
+ addr->type = POA_UDP6;
+ addr->udp6.ip_addr = saddr->in6.sin6_addr;
+ addr->udp6.port = ntohs(saddr->in6.sin6_port);
+ }
+}
+
+/* A datagram longer than the buffer arrives truncated: drop it. */
+static void * udp_reader(void * o)
+{
+ struct poa * poa = (struct poa *) o;
+ struct udp_priv * priv = (struct udp_priv *) poa->priv;
+ struct timespec down = TIMESPEC_INIT_MS(UDP_DOWN_TIMEO);
+ uint8_t * buf;
+
+ buf = malloc(UDP_MAX_PACKET);
+ if (buf == NULL)
+ return (void *) -1;
+
+ pthread_cleanup_push(free, buf);
+
+ while (true) {
+ struct ssm_pk_buff * spb;
+ union udp_saddr r_saddr;
+ struct poa_addr src;
+ struct msghdr mh;
+ struct iovec iov;
+ ssize_t n;
+ uint32_t eid;
+ size_t plen;
+
+ iov.iov_base = buf;
+ iov.iov_len = UDP_MAX_PACKET;
+
+ memset(&mh, 0, sizeof(mh));
+
+ mh.msg_name = &r_saddr;
+ mh.msg_namelen = sizeof(r_saddr);
+ mh.msg_iov = &iov;
+ mh.msg_iovlen = 1;
+
+ n = recvmsg(priv->s_fd, &mh, 0);
+ if (n < 0) {
+ if (errno == EINTR)
+ continue;
+
+ POA_STAT_BUMP(poa, rcv_fail);
+
+ if (errno == ENETDOWN) {
+ nanosleep(&down, NULL);
+ continue;
+ }
+
+ log_err("Reader stopped: %s.", strerror(errno));
+ break;
+ }
+
+ if ((mh.msg_flags & MSG_TRUNC) != 0) {
+ POA_STAT_BUMP(poa, buf_fail);
+ continue;
+ }
+
+ if ((size_t) n < UDP_HDR_LEN)
+ continue;
+
+ eid = ntoh32(*(uint32_t *) buf);
+ plen = (size_t) n - UDP_HDR_LEN;
+
+ saddr_to_addr(&r_saddr, &src);
+
+ if (eid == POA_MGMT_EID) {
+ poa_rx_mgmt(poa, &src, buf + UDP_HDR_LEN, plen);
+ continue;
+ }
+
+ if (poa_spb_reserve(&spb, plen) < 0) {
+ POA_STAT_BUMP(poa, buf_fail);
+ continue;
+ }
+
+ memcpy(ssm_pk_buff_head(spb), buf + UDP_HDR_LEN, plen);
+
+ poa_rx_pkt(poa, eid, spb);
+ }
+
+ pthread_cleanup_pop(true);
+
+ return (void *) 0;
+}
+
+/* Reads the bound address back: an ephemeral port is only known after. */
+static int udp_attach(struct poa * poa,
+ const struct poa_spec * spec)
+{
+ struct udp_priv * priv;
+ socklen_t len;
+ int af;
+
+ af = spec->type == POA_UDP4 ? AF_INET : AF_INET6;
+
+ priv = malloc(sizeof(*priv));
+ if (priv == NULL)
+ return -ENOMEM;
+
+ memset(priv, 0, sizeof(*priv));
+
+ priv->af = af;
+ priv->s_fd = socket(af, SOCK_DGRAM, IPPROTO_UDP);
+ if (priv->s_fd < 0) {
+ log_err("Failed to create socket: %s.", strerror(errno));
+ goto fail_socket;
+ }
+
+ if (af == AF_INET) {
+ priv->s_saddr.in.sin_family = AF_INET;
+ priv->s_saddr.in.sin_addr = spec->udp4.ip_addr;
+ priv->s_saddr.in.sin_port = htons(spec->udp4.port);
+ } else {
+ int on = 1;
+
+ if (setsockopt(priv->s_fd, IPPROTO_IPV6, IPV6_V6ONLY,
+ &on, sizeof(on)) < 0) {
+ log_err("Failed to set IPV6_V6ONLY: %s.",
+ strerror(errno));
+ goto fail_bind;
+ }
+
+ priv->s_saddr.in6.sin6_family = AF_INET6;
+ priv->s_saddr.in6.sin6_addr = spec->udp6.ip_addr;
+ priv->s_saddr.in6.sin6_port = htons(spec->udp6.port);
+ }
+
+ if (bind(priv->s_fd, &priv->s_saddr.sa, saddr_len(af)) < 0) {
+ log_err("Failed to bind: %s.", strerror(errno));
+ goto fail_bind;
+ }
+
+ poa->priv = priv;
+ len = saddr_len(af);
+ if (getsockname(priv->s_fd, &priv->s_saddr.sa, &len) < 0)
+ log_warn("Failed to read the bound address: %s.",
+ strerror(errno));
+
+ saddr_to_addr(&priv->s_saddr, &poa->local);
+
+ return 0;
+
+ fail_bind:
+ close(priv->s_fd);
+ fail_socket:
+ poa->priv = NULL;
+
+ free(priv);
+
+ return -EIO;
+}
+
+static void udp_detach(struct poa * poa)
+{
+ struct udp_priv * priv = (struct udp_priv *) poa->priv;
+
+ if (priv == NULL)
+ return;
+
+ close(priv->s_fd);
+
+ free(priv);
+
+ poa->priv = NULL;
+}
+
+/* One reader per socket, so a flow cannot be reordered on receive. */
+static int udp_start(struct poa * poa)
+{
+ struct udp_priv * priv = (struct udp_priv *) poa->priv;
+
+ if (pthread_create(&priv->reader, NULL, udp_reader, poa) != 0)
+ return -1;
+
+ priv->running = true;
+
+ return 0;
+}
+
+static void udp_stop(struct poa * poa)
+{
+ struct udp_priv * priv = (struct udp_priv *) poa->priv;
+
+ if (!priv->running)
+ return;
+
+ pthread_cancel(priv->reader);
+ pthread_join(priv->reader, NULL);
+
+ priv->running = false;
+}
+
+/*
+ * The PoA id is a separate iovec, so the buffer needs no headroom.
+ * MSG_DONTWAIT: the reader blocks on this socket.
+ */
+static int udp_sendv(struct udp_priv * priv,
+ const struct poa_addr * dst,
+ uint32_t eid,
+ const uint8_t * body,
+ size_t len,
+ bool block,
+ const struct timespec * abstime)
+{
+ union udp_saddr saddr;
+ struct msghdr msg;
+ struct iovec iov[2];
+ uint32_t hdr;
+ int ret;
+
+ if (len > UDP_MAX_PAYLOAD)
+ return -EMSGSIZE;
+
+ addr_to_saddr(dst, &saddr);
+
+ hdr = hton32(eid);
+
+ iov[0].iov_base = &hdr;
+ iov[0].iov_len = sizeof(hdr);
+ iov[1].iov_base = (void *) body;
+ iov[1].iov_len = len;
+
+ memset(&msg, 0, sizeof(msg));
+
+ msg.msg_name = &saddr;
+ msg.msg_namelen = saddr_len(priv->af);
+ msg.msg_iov = iov;
+ msg.msg_iovlen = len > 0 ? 2 : 1;
+ while (sendmsg(priv->s_fd, &msg, MSG_DONTWAIT) < 0) {
+ if (errno != EAGAIN && errno != EWOULDBLOCK)
+ return -EIO;
+
+ if (!block)
+ return -EAGAIN;
+
+ ret = poa_wait_out(priv->s_fd, abstime);
+ if (ret < 0)
+ return ret;
+ }
+
+ return 0;
+}
+
+static int udp_send(struct poa * poa,
+ const struct poa_addr * dst,
+ uint32_t eid,
+ struct ssm_pk_buff * spb,
+ bool block,
+ const struct timespec * abstime)
+{
+ return udp_sendv((struct udp_priv *) poa->priv, dst, eid,
+ ssm_pk_buff_head(spb), ssm_pk_buff_len(spb),
+ block, abstime);
+}
+
+static int udp_send_mgmt(struct poa * poa,
+ const struct poa_addr * dst,
+ const uint8_t * buf,
+ size_t len)
+{
+ struct timespec timeo = TIMESPEC_INIT_MS(POA_MGMT_SND_TIMEO);
+ struct timespec abstime;
+
+ clock_gettime(PTHREAD_COND_CLOCK, &abstime);
+ ts_add(&abstime, &timeo, &abstime);
+
+ return udp_sendv((struct udp_priv *) poa->priv, dst, POA_MGMT_EID,
+ buf, len, true, &abstime);
+}
+
+/* The PoA id header eats into the usable MTU. */
+static uint32_t udp_mtu(struct poa * poa,
+ const struct poa_addr * dst)
+{
+ struct udp_priv * priv = (struct udp_priv *) poa->priv;
+ uint32_t fallback;
+ uint32_t overh;
+#if defined(__linux__) && (defined(IP_MTU) || defined(IPV6_MTU))
+ union udp_saddr saddr;
+ socklen_t len;
+ int sock;
+ int mtu = 0;
+#endif
+ if (priv->af == AF_INET) {
+ fallback = POA_UDP4_MTU;
+ overh = UDP_IP4_OVERH;
+ } else {
+ fallback = POA_UDP6_MTU;
+ overh = UDP_IP6_OVERH;
+ }
+
+ fallback -= UDP_HDR_LEN;
+ if (fallback > UDP_MAX_PAYLOAD)
+ fallback = UDP_MAX_PAYLOAD;
+
+#if defined(__linux__) && (defined(IP_MTU) || defined(IPV6_MTU))
+
+ addr_to_saddr(dst, &saddr);
+
+ sock = socket(priv->af, SOCK_DGRAM, IPPROTO_UDP);
+ if (sock < 0)
+ return fallback;
+
+ if (connect(sock, &saddr.sa, saddr_len(priv->af)) < 0)
+ goto fallback;
+
+ len = sizeof(mtu);
+
+#if defined(IP_MTU)
+ if (priv->af == AF_INET) {
+ if (getsockopt(sock, IPPROTO_IP, IP_MTU, &mtu, &len) < 0)
+ goto fallback;
+ }
+#endif
+#if defined(IPV6_MTU)
+ if (priv->af == AF_INET6) {
+ if (getsockopt(sock, IPPROTO_IPV6, IPV6_MTU, &mtu, &len) < 0)
+ goto fallback;
+ }
+#endif
+ close(sock);
+
+ if (mtu <= (int) (overh + UDP_HDR_LEN))
+ return fallback;
+
+ return MIN((uint32_t) mtu - overh - UDP_HDR_LEN, UDP_MAX_PAYLOAD);
+
+ fallback:
+ close(sock);
+#else
+ (void) dst;
+ (void) overh;
+#endif
+ return fallback;
+}
+
+/* All flows on the PoA share the socket, so this is aggregate. */
+static size_t udp_qlen(struct poa * poa)
+{
+#if defined(__linux__) && defined(SIOCOUTQ)
+ struct udp_priv * priv = (struct udp_priv *) poa->priv;
+ int qlen;
+
+ qlen = 0;
+ if (ioctl(priv->s_fd, SIOCOUTQ, &qlen) < 0)
+ return 0;
+
+ return (size_t) qlen;
+#else
+ (void) poa;
+
+ return 0;
+#endif
+}
+
+/* The kernel keeps no per-socket drop count for UDP. */
+static int udp_rib(struct poa * poa,
+ char * buf,
+ size_t len)
+{
+ struct udp_priv * priv = (struct udp_priv *) poa->priv;
+ socklen_t optlen;
+ size_t sndbuf = 0;
+ size_t rcvbuf = 0;
+ int val;
+ int size;
+
+ optlen = sizeof(val);
+ if (getsockopt(priv->s_fd, SOL_SOCKET, SO_SNDBUF, &val, &optlen) == 0)
+ sndbuf = (size_t) val;
+
+ optlen = sizeof(val);
+ if (getsockopt(priv->s_fd, SOL_SOCKET, SO_RCVBUF, &val, &optlen) == 0)
+ rcvbuf = (size_t) val;
+
+ size = snprintf(buf, len,
+ "Socket sndbuf (bytes): %zu\n"
+ "Socket rcvbuf (bytes): %zu\n",
+ sndbuf, rcvbuf);
+ if (size < 0 || (size_t) size >= len)
+ return -1;
+
+ return size;
+}
+
+/*
+ * Asks the kernel which address it would send from: connect() does the
+ * real route lookup and sends nothing, so this honours the default
+ * route, metrics and policy rules alike.
+ */
+static int udp_src_addr(const struct poa_addr * dst,
+ struct poa_addr * src)
+{
+ union udp_saddr saddr;
+ socklen_t len;
+ int af;
+ int fd;
+
+ if (dst->type != POA_UDP4 && dst->type != POA_UDP6)
+ return -EINVAL;
+
+ af = dst->type == POA_UDP4 ? AF_INET : AF_INET6;
+
+ addr_to_saddr(dst, &saddr);
+
+ fd = socket(af, SOCK_DGRAM, IPPROTO_UDP);
+ if (fd < 0)
+ return -EIO;
+
+ if (connect(fd, &saddr.sa, saddr_len(af)) < 0)
+ goto fail;
+
+ len = saddr_len(af);
+ if (getsockname(fd, &saddr.sa, &len) < 0)
+ goto fail;
+
+ close(fd);
+
+ saddr_to_addr(&saddr, src);
+
+ return 0;
+
+ fail:
+ close(fd);
+
+ return -EIO;
+}
+
+static void udp_spec(const struct poa * poa,
+ struct poa_spec * spec)
+{
+ spec->type = poa->type;
+
+ if (poa->type == POA_UDP4)
+ spec->udp4 = poa->local.udp4;
+ else
+ spec->udp6 = poa->local.udp6;
+}
+
+static bool udp_has_id(const struct poa * poa,
+ const struct poa_spec * spec)
+{
+ if (poa->type == POA_UDP4) {
+ if (poa->local.udp4.port != spec->udp4.port)
+ return false;
+
+ return memcmp(&poa->local.udp4.ip_addr,
+ &spec->udp4.ip_addr,
+ sizeof(spec->udp4.ip_addr)) == 0;
+ }
+
+ if (poa->local.udp6.port != spec->udp6.port)
+ return false;
+
+ return memcmp(&poa->local.udp6.ip_addr, &spec->udp6.ip_addr,
+ sizeof(spec->udp6.ip_addr)) == 0;
+}
+
+static bool udp_addr_is_any(const struct poa_addr * addr)
+{
+ static const struct in6_addr any6 = IN6ADDR_ANY_INIT;
+
+ if (addr->type == POA_UDP4)
+ return addr->udp4.ip_addr.s_addr == htonl(INADDR_ANY);
+
+ return memcmp(&addr->udp6.ip_addr, &any6, sizeof(any6)) == 0;
+}
+
+/*
+ * Our end of the flow: the IP the kernel would send to dst from. A
+ * PoA bound to the wildcard is the catch-all and a failed probe
+ * matches any. Ports are not compared: the probe's is ephemeral.
+ */
+static bool udp_match(const struct poa * poa,
+ const struct poa_addr * dst)
+{
+ struct poa_addr src;
+
+ if (udp_addr_is_any(&poa->local))
+ return true;
+
+ if (udp_src_addr(dst, &src) < 0)
+ return true;
+
+ if (poa->type == POA_UDP4)
+ return memcmp(&poa->local.udp4.ip_addr,
+ &src.udp4.ip_addr,
+ sizeof(src.udp4.ip_addr)) == 0;
+
+ return memcmp(&poa->local.udp6.ip_addr, &src.udp6.ip_addr,
+ sizeof(src.udp6.ip_addr)) == 0;
+}
+
+const struct poa_ops udp_poa_ops = {
+ .poa_attach = udp_attach,
+ .poa_detach = udp_detach,
+ .poa_start = udp_start,
+ .poa_stop = udp_stop,
+ .poa_send = udp_send,
+ .poa_send_mgmt = udp_send_mgmt,
+ .poa_mtu = udp_mtu,
+ .poa_qlen = udp_qlen,
+ .poa_rib = udp_rib,
+ .poa_spec = udp_spec,
+ .poa_has_id = udp_has_id,
+ .poa_match = udp_match,
+ .mpl = POA_UDP_MPL
+};
diff --git a/src/lib/protobuf.c b/src/lib/protobuf.c
index 6beae000..aa247755 100644
--- a/src/lib/protobuf.c
+++ b/src/lib/protobuf.c
@@ -22,6 +22,7 @@
#define _DEFAULT_SOURCE
+#include <ouroboros/errno.h>
#include <ouroboros/protobuf.h>
#include <ouroboros/crypt.h>
#include <ouroboros/proc.h>
@@ -215,6 +216,166 @@ struct layer_info layer_info_msg_to_s(const layer_info_msg_t * msg)
return s;
}
+
+static bool mac_is_zero(const uint8_t * mac)
+{
+ static const uint8_t zero[POA_MAC_SIZE] = { 0 };
+
+ return memcmp(mac, zero, POA_MAC_SIZE) == 0;
+}
+
+poa_addr_msg_t * poa_addr_s_to_msg(const struct poa_addr * s)
+{
+ poa_addr_msg_t * msg;
+
+ assert(s != NULL);
+
+ msg = malloc(sizeof(*msg));
+ if (msg == NULL)
+ goto fail_malloc;
+
+ poa_addr_msg__init(msg);
+
+ msg->type = s->type;
+
+ if (s->hostname[0] != '\0') {
+ msg->hostname = strdup(s->hostname);
+ if (msg->hostname == NULL)
+ goto fail_msg;
+ }
+
+ switch (s->type) {
+ case POA_UDP4:
+ msg->has_ip4 = true;
+ msg->ip4 = s->udp4.ip_addr.s_addr;
+ msg->has_port = true;
+ msg->port = s->udp4.port;
+ break;
+ case POA_UDP6:
+ msg->has_ip6 = true;
+ msg->ip6.len = sizeof(s->udp6.ip_addr);
+ msg->ip6.data = malloc(msg->ip6.len);
+ if (msg->ip6.data == NULL)
+ goto fail_msg;
+ memcpy(msg->ip6.data, &s->udp6.ip_addr, msg->ip6.len);
+
+ msg->has_port = true;
+ msg->port = s->udp6.port;
+ break;
+ case POA_ETH:
+ msg->has_ethertype = true;
+ msg->ethertype = s->eth.dst.ethertype;
+ msg->has_mac = true;
+ msg->mac.len = POA_MAC_SIZE;
+ msg->mac.data = malloc(POA_MAC_SIZE);
+ if (msg->mac.data == NULL)
+ goto fail_msg;
+ memcpy(msg->mac.data, s->eth.dst.mac, POA_MAC_SIZE);
+
+ if (!mac_is_zero(s->eth.src.mac)) {
+ msg->has_src_mac = true;
+ msg->src_mac.len = POA_MAC_SIZE;
+ msg->src_mac.data = malloc(POA_MAC_SIZE);
+ if (msg->src_mac.data == NULL)
+ goto fail_msg;
+ memcpy(msg->src_mac.data, s->eth.src.mac, POA_MAC_SIZE);
+ }
+
+ if (s->eth.src.dev[0] != '\0') {
+ msg->dev = strdup(s->eth.src.dev);
+ if (msg->dev == NULL)
+ goto fail_msg;
+ }
+ break;
+ case POA_UDP:
+ msg->has_port = true;
+ msg->port = s->udp4.port;
+ break;
+ default:
+ goto fail_msg;
+ }
+
+ return msg;
+
+ fail_msg:
+ poa_addr_msg__free_unpacked(msg, NULL);
+ fail_malloc:
+ return NULL;
+}
+
+struct poa_addr poa_addr_msg_to_s(const poa_addr_msg_t * msg)
+{
+ struct poa_addr s;
+
+ assert(msg != NULL);
+
+ memset(&s, 0, sizeof(s));
+
+ s.type = POA_INVALID;
+
+ if (msg->hostname != NULL) {
+ if (strlen(msg->hostname) > POA_HOST_STRLEN)
+ return s;
+ strcpy(s.hostname, msg->hostname);
+ }
+
+ switch (msg->type) {
+ case POA_UDP4:
+ if (!msg->has_ip4 || !msg->has_port)
+ return s;
+
+ s.udp4.ip_addr.s_addr = msg->ip4;
+ s.udp4.port = msg->port;
+ break;
+ case POA_UDP6:
+ if (!msg->has_ip6 || !msg->has_port)
+ return s;
+
+ if (msg->ip6.len != sizeof(s.udp6.ip_addr)
+ || msg->ip6.data == NULL)
+ return s;
+
+ memcpy(&s.udp6.ip_addr, msg->ip6.data, msg->ip6.len);
+
+ s.udp6.port = msg->port;
+ break;
+ case POA_ETH:
+ if (!msg->has_ethertype)
+ return s;
+
+ s.eth.src.ethertype = msg->ethertype;
+ s.eth.dst.ethertype = msg->ethertype;
+
+ if (msg->mac.len != POA_MAC_SIZE || msg->mac.data == NULL)
+ return s;
+
+ memcpy(s.eth.dst.mac, msg->mac.data, POA_MAC_SIZE);
+
+ if (msg->src_mac.len == POA_MAC_SIZE)
+ memcpy(s.eth.src.mac, msg->src_mac.data, POA_MAC_SIZE);
+
+ if (msg->dev != NULL) {
+ if (strlen(msg->dev) > DEV_NAME_SIZE)
+ return s;
+ strcpy(s.eth.src.dev, msg->dev);
+ }
+ break;
+ case POA_UDP:
+ if (msg->hostname == NULL || msg->hostname[0] == '\0'
+ || !msg->has_port)
+ return s;
+
+ s.udp4.port = msg->port;
+ break;
+ default:
+ return s;
+ }
+
+ s.type = msg->type;
+
+ return s;
+}
+
ipcp_info_msg_t * ipcp_info_s_to_msg(const struct ipcp_info * s)
{
ipcp_info_msg_t * msg;
@@ -389,22 +550,6 @@ struct dt_config dt_config_msg_to_s(const dt_config_msg_t * msg)
return s;
}
-struct dir_dht_config dir_dht_config_msg_to_s(const dir_dht_config_msg_t * msg)
-{
- struct dir_dht_config s;
-
- assert(msg != NULL);
-
- s.params.alpha = msg->alpha;
- s.params.k = msg->k;
- s.params.t_expire = msg->t_expire;
- s.params.t_refresh = msg->t_refresh;
- s.params.t_replicate = msg->t_replicate;
- s.peer = msg->peer;
-
- return s;
-}
-
dir_dht_config_msg_t * dir_dht_config_s_to_msg(const struct dir_dht_config * s)
{
dir_dht_config_msg_t * msg;
@@ -427,22 +572,18 @@ dir_dht_config_msg_t * dir_dht_config_s_to_msg(const struct dir_dht_config * s)
return msg;
}
-struct dir_config dir_config_msg_to_s(const dir_config_msg_t * msg)
+struct dir_dht_config dir_dht_config_msg_to_s(const dir_dht_config_msg_t * msg)
{
- struct dir_config s;
+ struct dir_dht_config s;
assert(msg != NULL);
- switch (msg->pol) {
- case DIR_DHT:
- s.dht = dir_dht_config_msg_to_s(msg->dht);
- break;
- default:
- /* No checks here */
- break;
- }
-
- s.pol = msg->pol;
+ s.params.alpha = msg->alpha;
+ s.params.k = msg->k;
+ s.params.t_expire = msg->t_expire;
+ s.params.t_refresh = msg->t_refresh;
+ s.params.t_replicate = msg->t_replicate;
+ s.peer = msg->peer;
return s;
}
@@ -479,6 +620,25 @@ dir_config_msg_t * dir_config_s_to_msg(const struct dir_config * s)
return NULL;
}
+struct dir_config dir_config_msg_to_s(const dir_config_msg_t * msg)
+{
+ struct dir_config s;
+
+ assert(msg != NULL);
+
+ switch (msg->pol) {
+ case DIR_DHT:
+ s.dht = dir_dht_config_msg_to_s(msg->dht);
+ break;
+ default:
+ /* No checks here */
+ break;
+ }
+
+ s.pol = msg->pol;
+
+ return s;
+}
uni_config_msg_t * uni_config_s_to_msg(const struct uni_config * s)
{
@@ -525,9 +685,9 @@ struct uni_config uni_config_msg_to_s(const uni_config_msg_t * msg)
return s;
}
-udp4_config_msg_t * udp4_config_s_to_msg(const struct udp4_config * s)
+udp4_poa_msg_t * udp4_poa_s_to_msg(const struct udp4_poa * s)
{
- udp4_config_msg_t * msg;
+ udp4_poa_msg_t * msg;
assert(s != NULL);
@@ -535,32 +695,30 @@ udp4_config_msg_t * udp4_config_s_to_msg(const struct udp4_config * s)
if (msg == NULL)
return NULL;
- udp4_config_msg__init(msg);
+ udp4_poa_msg__init(msg);
- msg->ip_addr = s->ip_addr.s_addr;
- msg->dns_addr = s->dns_addr.s_addr;
- msg->port = s->port;
+ msg->ip_addr = s->ip_addr.s_addr;
+ msg->port = s->port;
return msg;
}
-struct udp4_config udp4_config_msg_to_s(const udp4_config_msg_t * msg)
+struct udp4_poa udp4_poa_msg_to_s(const udp4_poa_msg_t * msg)
{
- struct udp4_config s;
+ struct udp4_poa s;
assert(msg != NULL);
- s.ip_addr.s_addr = msg->ip_addr;
- s.dns_addr.s_addr = msg->dns_addr;
- s.port = msg->port;
+ s.ip_addr.s_addr = msg->ip_addr;
+ s.port = msg->port;
return s;
}
#define IN6_LEN (size_t) sizeof(struct in6_addr)
-udp6_config_msg_t * udp6_config_s_to_msg(const struct udp6_config * s)
+udp6_poa_msg_t * udp6_poa_s_to_msg(const struct udp6_poa * s)
{
- udp6_config_msg_t * msg;
+ udp6_poa_msg_t * msg;
assert(s != NULL);
@@ -568,7 +726,7 @@ udp6_config_msg_t * udp6_config_s_to_msg(const struct udp6_config * s)
if (msg == NULL)
goto fail_malloc;
- udp6_config_msg__init(msg);
+ udp6_poa_msg__init(msg);
msg->ip_addr.data = malloc(IN6_LEN);
if (msg->ip_addr.data == NULL)
@@ -577,42 +735,33 @@ udp6_config_msg_t * udp6_config_s_to_msg(const struct udp6_config * s)
msg->ip_addr.len = IN6_LEN;
memcpy(msg->ip_addr.data, &s->ip_addr.s6_addr, IN6_LEN);
- msg->dns_addr.data = malloc(IN6_LEN);
- if (msg->dns_addr.data == NULL)
- goto fail_msg;
-
- msg->dns_addr.len = IN6_LEN;
- memcpy(msg->dns_addr.data, &s->dns_addr.s6_addr, IN6_LEN);
-
msg->port = s->port;
return msg;
fail_msg:
- udp6_config_msg__free_unpacked(msg, NULL);
+ udp6_poa_msg__free_unpacked(msg, NULL);
fail_malloc:
return NULL;
}
-struct udp6_config udp6_config_msg_to_s(const udp6_config_msg_t * msg)
+struct udp6_poa udp6_poa_msg_to_s(const udp6_poa_msg_t * msg)
{
- struct udp6_config s;
+ struct udp6_poa s;
assert(msg != NULL);
assert(msg->ip_addr.len == IN6_LEN);
- assert(msg->dns_addr.len == IN6_LEN);
memcpy(&s.ip_addr.s6_addr, msg->ip_addr.data, IN6_LEN);
- memcpy(&s.dns_addr.s6_addr, msg->dns_addr.data, IN6_LEN);
s.port = msg->port;
return s;
}
-eth_config_msg_t * eth_config_s_to_msg(const struct eth_config * s)
+eth_poa_msg_t * eth_poa_s_to_msg(const struct eth_poa * s)
{
- eth_config_msg_t * msg;
+ eth_poa_msg_t * msg;
assert(s != NULL);
@@ -620,7 +769,7 @@ eth_config_msg_t * eth_config_s_to_msg(const struct eth_config * s)
if (msg == NULL)
goto fail_malloc;
- eth_config_msg__init(msg);
+ eth_poa_msg__init(msg);
msg->dev = strdup(s->dev);
if (msg->dev == NULL)
@@ -628,28 +777,115 @@ eth_config_msg_t * eth_config_s_to_msg(const struct eth_config * s)
msg->ethertype = s->ethertype;
+ msg->has_mac = true;
+ msg->mac.len = POA_MAC_SIZE;
+ msg->mac.data = malloc(POA_MAC_SIZE);
+ if (msg->mac.data == NULL)
+ goto fail_msg;
+
+ memcpy(msg->mac.data, s->mac, POA_MAC_SIZE);
+
return msg;
fail_msg:
- eth_config_msg__free_unpacked(msg, NULL);
+ eth_poa_msg__free_unpacked(msg, NULL);
fail_malloc:
return NULL;
}
-struct eth_config eth_config_msg_to_s(const eth_config_msg_t * msg)
+struct eth_poa eth_poa_msg_to_s(const eth_poa_msg_t * msg)
{
- struct eth_config s;
+ struct eth_poa s;
assert(msg != NULL);
assert(strlen(msg->dev) <= DEV_NAME_SIZE);
+ memset(&s, 0, sizeof(s));
+
strcpy(s.dev, msg->dev);
+
s.ethertype = msg->ethertype;
+ if (msg->mac.len == POA_MAC_SIZE)
+ memcpy(s.mac, msg->mac.data, POA_MAC_SIZE);
+
return s;
}
+poa_spec_msg_t * poa_spec_s_to_msg(const struct poa_spec * s)
+{
+ poa_spec_msg_t * msg;
+
+ assert(s != NULL);
+
+ msg = malloc(sizeof(*msg));
+ if (msg == NULL)
+ goto fail_malloc;
+
+ poa_spec_msg__init(msg);
+
+ switch (s->type) {
+ case POA_UDP4:
+ msg->udp4 = udp4_poa_s_to_msg(&s->udp4);
+ if (msg->udp4 == NULL)
+ goto fail_msg;
+ break;
+ case POA_UDP6:
+ msg->udp6 = udp6_poa_s_to_msg(&s->udp6);
+ if (msg->udp6 == NULL)
+ goto fail_msg;
+ break;
+ case POA_ETH:
+ msg->eth = eth_poa_s_to_msg(&s->eth);
+ if (msg->eth == NULL)
+ goto fail_msg;
+ break;
+ default:
+ goto fail_msg;
+ }
+
+ return msg;
+
+ fail_msg:
+ poa_spec_msg__free_unpacked(msg, NULL);
+ fail_malloc:
+ return NULL;
+}
+
+struct poa_spec poa_spec_msg_to_s(const poa_spec_msg_t * msg)
+{
+ struct poa_spec s;
+
+ memset(&s, 0, sizeof(s));
+
+ s.type = POA_INVALID;
+
+ if (msg == NULL)
+ return s;
+
+ if (msg->udp4 != NULL) {
+ s.type = POA_UDP4;
+ s.udp4 = udp4_poa_msg_to_s(msg->udp4);
+ } else if (msg->udp6 != NULL) {
+ if (msg->udp6->ip_addr.len != IN6_LEN
+ || msg->udp6->ip_addr.data == NULL)
+ return s;
+
+ s.type = POA_UDP6;
+ s.udp6 = udp6_poa_msg_to_s(msg->udp6);
+ } else if (msg->eth != NULL) {
+ if (msg->eth->dev == NULL
+ || strlen(msg->eth->dev) > DEV_NAME_SIZE)
+ return s;
+
+ s.type = POA_ETH;
+ s.eth = eth_poa_msg_to_s(msg->eth);
+ }
+
+ return s;
+}
+
ipcp_config_msg_t * ipcp_config_s_to_msg(const struct ipcp_config * s)
{
ipcp_config_msg_t * msg;
@@ -672,23 +908,6 @@ ipcp_config_msg_t * ipcp_config_s_to_msg(const struct ipcp_config * s)
break;
case IPCP_BROADCAST:
break;
- case IPCP_ETH_LLC:
- /* FALLTHRU */
- case IPCP_ETH_DIX:
- msg->eth = eth_config_s_to_msg(&s->eth);
- if (msg->eth == NULL)
- goto fail_msg;
- break;
- case IPCP_UDP4:
- msg->udp4 = udp4_config_s_to_msg(&s->udp4);
- if (msg->udp4 == NULL)
- goto fail_msg;
- break;
- case IPCP_UDP6:
- msg->udp6 = udp6_config_s_to_msg(&s->udp6);
- if (msg->udp6 == NULL)
- goto fail_msg;
- break;
default:
/* No checks here */
break;
@@ -714,6 +933,8 @@ struct ipcp_config ipcp_config_msg_to_s(const ipcp_config_msg_t * msg)
assert(msg != NULL);
+ memset(&s, 0, sizeof(s));
+
s.type = msg->ipcp_type;
s.layer_info = layer_info_msg_to_s(msg->layer_info);
@@ -724,17 +945,6 @@ struct ipcp_config ipcp_config_msg_to_s(const ipcp_config_msg_t * msg)
case IPCP_UNICAST:
s.unicast = uni_config_msg_to_s(msg->unicast);
break;
- case IPCP_ETH_LLC:
- /* FALLTHRU */
- case IPCP_ETH_DIX:
- s.eth = eth_config_msg_to_s(msg->eth);
- break;
- case IPCP_UDP4:
- s.udp4 = udp4_config_msg_to_s(msg->udp4);
- break;
- case IPCP_UDP6:
- s.udp6 = udp6_config_msg_to_s(msg->udp6);
- break;
case IPCP_BROADCAST:
break;
default:
diff --git a/src/lib/serdes-irm.c b/src/lib/serdes-irm.c
index 1d9b4dec..20b25a1b 100644
--- a/src/lib/serdes-irm.c
+++ b/src/lib/serdes-irm.c
@@ -441,6 +441,7 @@ int proc_exit__irm_req_ser(buffer_t * buf)
return -ENOMEM;
}
+/* data is borrowed from the caller; detach before free. */
int ipcp_flow_req_arr__irm_req_ser(buffer_t * buf,
const buffer_t * dst,
const struct flow_info * flow,
@@ -494,6 +495,138 @@ int ipcp_flow_req_arr__irm_req_ser(buffer_t * buf,
return -ENOMEM;
}
+static int __ep_flow_ser(buffer_t * buf,
+ const struct flow_info * flow,
+ const buffer_t * data,
+ int response,
+ const char * dst,
+ int code)
+{
+ irm_msg_t * msg;
+ size_t len;
+ int err = -ENOMEM;
+
+ msg = malloc(sizeof(*msg));
+ if (msg == NULL)
+ goto fail_malloc;
+
+ irm_msg__init(msg);
+
+ msg->code = code;
+ msg->flow_info = flow_info_s_to_msg(flow);
+ if (msg->flow_info == NULL)
+ goto fail_msg;
+
+ if (dst != NULL) {
+ msg->dst = strdup(dst);
+ if (msg->dst == NULL)
+ goto fail_msg;
+ }
+
+ if (data != NULL) {
+ msg->has_pk = true;
+ msg->pk.len = data->len;
+ msg->pk.data = data->data;
+ }
+
+ msg->has_response = true;
+ msg->response = response;
+
+ len = irm_msg__get_packed_size(msg);
+ if (len == 0 || len > buf->len)
+ goto fail_msg;
+
+ buf->len = len;
+
+ irm_msg__pack(msg, buf->data);
+
+ err = 0;
+ fail_msg:
+ msg->pk.len = 0;
+ msg->pk.data = NULL;
+
+ irm_msg__free_unpacked(msg, NULL);
+ fail_malloc:
+ return err;
+}
+
+int poa_flow_alloc__irm_req_ser(buffer_t * buf,
+ const struct flow_info * flow,
+ const char * dst)
+{
+ return __ep_flow_ser(buf, flow, NULL, 0, dst,
+ IRM_MSG_CODE__IRM_POA_FLOW_ALLOC);
+}
+
+int poa_flow_alloc_r__irm_req_ser(buffer_t * buf,
+ const struct flow_info * flow,
+ const buffer_t * data,
+ int response)
+{
+ return __ep_flow_ser(buf, flow, data, response, NULL,
+ IRM_MSG_CODE__IRM_POA_FLOW_ALLOC_R);
+}
+
+int ipcp_poa_flow_req_arr__irm_req_ser(buffer_t * buf,
+ const struct flow_info * flow,
+ const buffer_t * data)
+{
+ return __ep_flow_ser(buf, flow, data, 0, NULL,
+ IRM_MSG_CODE__IPCP_POA_FLOW_REQ_ARR);
+}
+
+int poa_flow__irm_result_des(buffer_t * buf,
+ struct flow_info * flow,
+ buffer_t * data)
+{
+ irm_msg_t * msg;
+ int err;
+
+ msg = irm_msg__unpack(NULL, buf->len, buf->data);
+ if (msg == NULL) {
+ err = -EIRMD;
+ goto fail_msg;
+ }
+
+ if (!msg->has_result) {
+ err = -EIRMD;
+ goto fail;
+ }
+
+ if (msg->result < 0) {
+ err = msg->result;
+ goto fail;
+ }
+
+ if (msg->flow_info == NULL) {
+ err = -EBADF;
+ goto fail;
+ }
+
+ *flow = flow_info_msg_to_s(msg->flow_info);
+
+ clrbuf(*data);
+
+ if (msg->has_pk && msg->pk.len > 0) {
+ data->data = malloc(msg->pk.len);
+ if (data->data == NULL) {
+ err = -ENOMEM;
+ goto fail;
+ }
+ memcpy(data->data, msg->pk.data, msg->pk.len);
+
+ data->len = msg->pk.len;
+ }
+
+ irm_msg__free_unpacked(msg, NULL);
+
+ return 0;
+ fail:
+ irm_msg__free_unpacked(msg, NULL);
+ fail_msg:
+ return err;
+}
+
int ipcp_flow_update_arr__irm_req_ser(buffer_t * buf,
const struct flow_info * flow,
const buffer_t * data)
diff --git a/src/lib/tests/CMakeLists.txt b/src/lib/tests/CMakeLists.txt
index 002d94af..1f2e9ba2 100644
--- a/src/lib/tests/CMakeLists.txt
+++ b/src/lib/tests/CMakeLists.txt
@@ -11,6 +11,7 @@ create_test_sourcelist(${PARENT_DIR}_tests test_suite.c
bitmap_test.c
btree_test.c
crypt_test.c
+ poa_test.c
hash_test.c
kex_test.c
kex_test_ml_kem.c
@@ -25,6 +26,12 @@ create_test_sourcelist(${PARENT_DIR}_tests test_suite.c
add_executable(${PARENT_DIR}_test ${${PARENT_DIR}_tests})
+if(HAVE_LIBURCU)
+ # poa_test.c pulls in poa.h, whose urcu guard needs C99.
+ set_source_files_properties(poa_test.c PROPERTIES
+ COMPILE_OPTIONS "-std=gnu99")
+endif()
+
target_include_directories(${PARENT_DIR}_test PRIVATE
${CMAKE_SOURCE_DIR}/src/lib)
diff --git a/src/lib/tests/poa_test.c b/src/lib/tests/poa_test.c
new file mode 100644
index 00000000..99886769
--- /dev/null
+++ b/src/lib/tests/poa_test.c
@@ -0,0 +1,307 @@
+/*
+ * Ouroboros - Copyright (C) 2016 - 2026
+ *
+ * Flow PoA tests
+ *
+ * Dimitri Staessens <dimitri@ouroboros.rocks>
+ * Sander Vrijders <sander@ouroboros.rocks>
+ *
+ * This program is free software; you can redistribute it and/or modify
+ * it under the terms of the GNU General Public License version 2 as
+ * published by the Free Software Foundation.
+ *
+ * This program is distributed in the hope that it will be useful,
+ * but WITHOUT ANY WARRANTY; without even the implied warranty of
+ * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+ * GNU General Public License for more details.
+ *
+ * You should have received a copy of the GNU General Public License
+ * along with this program; if not, write to the Free Software
+ * Foundation, Inc., http://www.fsf.org/about/contact/.
+ */
+
+#if defined(__linux__) || defined(__CYGWIN__)
+#define _DEFAULT_SOURCE
+#else
+#define _POSIX_C_SOURCE 200809L
+#endif
+
+#include <test/test.h>
+
+#include "poa/addr.c"
+#ifdef HAVE_RAW_SOCKETS
+#include "poa/eth.c"
+#endif
+
+#include <arpa/inet.h>
+#include <stdio.h>
+
+#ifdef HAVE_RAW_SOCKETS
+static const uint8_t eth_our_mac[POA_MAC_SIZE] =
+ { 0x02, 0x00, 0x00, 0x00, 0x00, 0x01 };
+static const uint8_t eth_far_mac[POA_MAC_SIZE] =
+ { 0x02, 0x00, 0x00, 0x00, 0x00, 0x02 };
+static const uint8_t eth_bc_mac[POA_MAC_SIZE] =
+ { 0xff, 0xff, 0xff, 0xff, 0xff, 0xff };
+#endif
+
+/* PoA core stubs: the reader threads never run in this test. */
+void poa_rx_pkt(struct poa * poa,
+ uint32_t eid,
+ struct ssm_pk_buff * spb)
+{
+ (void) poa;
+ (void) eid;
+ (void) spb;
+}
+
+void poa_rx_mgmt(struct poa * poa,
+ const struct poa_addr * src,
+ const uint8_t * buf,
+ size_t len)
+{
+ (void) poa;
+ (void) src;
+ (void) buf;
+ (void) len;
+}
+
+int poa_spb_reserve(struct ssm_pk_buff ** spb,
+ size_t len)
+{
+ (void) spb;
+ (void) len;
+
+ return -1;
+}
+
+size_t poa_link_updown(int ifindex,
+ bool up)
+{
+ (void) ifindex;
+ (void) up;
+
+ return 0;
+}
+
+bool poa_has_name(const uint8_t * hash)
+{
+ (void) hash;
+
+ return false;
+}
+
+int poa_bcast_mgmt(const struct poa_addr * dst,
+ const uint8_t * buf,
+ size_t len)
+{
+ (void) dst;
+ (void) buf;
+ (void) len;
+
+ return 0;
+}
+
+static void udp4_addr(struct poa_addr * addr,
+ const char * ip,
+ uint16_t port)
+{
+ memset(addr, 0, sizeof(*addr));
+
+ addr->type = POA_UDP4;
+ addr->udp4.port = port;
+
+ inet_pton(AF_INET, ip, &addr->udp4.ip_addr);
+}
+
+static int test_poa_addr_cmp(void)
+{
+ struct poa_addr a;
+ struct poa_addr b;
+
+ TEST_START();
+
+ udp4_addr(&a, "10.0.0.10", 3435);
+ udp4_addr(&b, "10.0.0.10", 3435);
+
+ if (poa_addr_cmp(&a, &b) != 0) {
+ printf("Identical addresses did not match.\n");
+ goto fail;
+ }
+
+ udp4_addr(&b, "10.0.0.10", 3436);
+
+ if (poa_addr_cmp(&a, &b) == 0) {
+ printf("Addresses with a different port matched.\n");
+ goto fail;
+ }
+
+ udp4_addr(&b, "10.0.0.11", 3435);
+
+ if (poa_addr_cmp(&a, &b) == 0) {
+ printf("Addresses with a different ip matched.\n");
+ goto fail;
+ }
+
+ b.type = POA_UDP6;
+ if (poa_addr_cmp(&a, &b) == 0) {
+ printf("Addresses of a different type matched.\n");
+ goto fail;
+ }
+
+ TEST_SUCCESS();
+
+ return TEST_RC_SUCCESS;
+ fail:
+ TEST_FAIL();
+ return TEST_RC_FAIL;
+}
+
+static int test_poa_mgmt_msg_qos(void)
+{
+ struct poa_mgmt_msg msg;
+ qosspec_t qs;
+
+ TEST_START();
+
+ poa_mgmt_msg_ser(&msg, POA_FLOW_REQ, 1, 2, qos_stream, 0, 0);
+
+ if (msg.code != POA_FLOW_REQ) {
+ printf("Wrong code in management message.\n");
+ goto fail;
+ }
+
+ if (ntoh32(msg.s_eid) != 1 || ntoh32(msg.d_eid) != 2) {
+ printf("Wrong PoA ids in management message.\n");
+ goto fail;
+ }
+
+ memset(&qs, 0, sizeof(qs));
+
+ poa_mgmt_msg_qos(&msg, &qs);
+
+ if (memcmp(&qs, &qos_stream, sizeof(qs)) != 0) {
+ printf("QoS did not survive the management message.\n");
+ goto fail;
+ }
+
+ TEST_SUCCESS();
+
+ return TEST_RC_SUCCESS;
+ fail:
+ TEST_FAIL();
+ return TEST_RC_FAIL;
+}
+
+#ifdef HAVE_RAW_SOCKETS
+
+static void eth_fake_priv(struct eth_priv * priv,
+ struct poa * e)
+{
+ memset(priv, 0, sizeof(*priv));
+ memset(e, 0, sizeof(*e));
+
+ e->type = POA_ETH;
+
+ strcpy(e->local.eth.src.dev, "test0");
+
+ priv->poa = e;
+ priv->ethertype = htons(0xA000);
+ priv->mtu = 1500;
+
+ memcpy(priv->hw_addr, eth_our_mac, POA_MAC_SIZE);
+}
+
+static int test_eth_frame(void)
+{
+ struct eth_priv priv;
+ struct poa e;
+ uint8_t buf[64];
+ struct eth_hdr * hdr = (struct eth_hdr *) buf;
+ const char * body = "hello";
+ size_t n;
+ size_t plen;
+ uint32_t eid;
+
+ TEST_START();
+
+ eth_fake_priv(&priv, &e);
+
+ eth_hdr_ser(&priv, hdr, eth_our_mac, 7, strlen(body));
+
+ memcpy(buf + ETH_HDR_TOT_SIZE, body, strlen(body));
+
+ n = ETH_HDR_TOT_SIZE + strlen(body);
+ if (frame_parse(&priv, buf, n, &eid, &plen) < 0) {
+ printf("Failed to parse a valid frame.\n");
+ goto fail;
+ }
+
+ if (eid != 7 || plen != strlen(body)) {
+ printf("Expected eid 7 len %zu, got %u len %zu.\n",
+ strlen(body), eid, plen);
+ goto fail;
+ }
+
+ hdr->poa.hcs ^= 0xFF;
+
+ if (frame_parse(&priv, buf, n, &eid, &plen) == 0) {
+ printf("Accepted a corrupt header checksum.\n");
+ goto fail;
+ }
+
+ hdr->poa.hcs ^= 0xFF;
+ hdr->ethertype ^= 0xFF;
+
+ if (frame_parse(&priv, buf, n, &eid, &plen) == 0) {
+ printf("Accepted a foreign Ethertype.\n");
+ goto fail;
+ }
+
+ hdr->ethertype ^= 0xFF;
+
+ if (frame_parse(&priv, buf, ETH_HDR_SIZE + 2, &eid, &plen) == 0) {
+ printf("Accepted a truncated frame.\n");
+ goto fail;
+ }
+
+ eth_hdr_ser(&priv, hdr, eth_far_mac, 7, strlen(body));
+
+ if (frame_parse(&priv, buf, n, &eid, &plen) == 0) {
+ printf("Accepted a frame for another host.\n");
+ goto fail;
+ }
+
+ eth_hdr_ser(&priv, hdr, eth_bc_mac, 7, strlen(body));
+
+ if (frame_parse(&priv, buf, n, &eid, &plen) < 0) {
+ printf("Rejected a broadcast frame.\n");
+ goto fail;
+ }
+
+ TEST_SUCCESS();
+
+ return TEST_RC_SUCCESS;
+ fail:
+ TEST_FAIL();
+ return TEST_RC_FAIL;
+}
+
+#endif /* HAVE_RAW_SOCKETS */
+
+int poa_test(int argc,
+ char ** argv)
+{
+ int ret = 0;
+
+ (void) argc;
+ (void) argv;
+
+ ret |= test_poa_addr_cmp();
+ ret |= test_poa_mgmt_msg_qos();
+#ifdef HAVE_RAW_SOCKETS
+ ret |= test_eth_frame();
+#endif
+
+ return ret;
+}
diff --git a/src/tools/CMakeLists.txt b/src/tools/CMakeLists.txt
index 6b418838..bb601733 100644
--- a/src/tools/CMakeLists.txt
+++ b/src/tools/CMakeLists.txt
@@ -15,6 +15,10 @@ set(IRM_SOURCES
irm/irm_ipcp_destroy.c
irm/irm_ipcp_bootstrap.c
irm/irm_ipcp_enroll.c
+ irm/irm_ipcp_poa.c
+ irm/irm_ipcp_poa_attach.c
+ irm/irm_ipcp_poa_detach.c
+ irm/irm_ipcp_poa_list.c
irm/irm_ipcp_list.c
irm/irm_ipcp_connect.c
irm/irm_ipcp_disconnect.c
diff --git a/src/tools/irm/irm_ipcp.c b/src/tools/irm/irm_ipcp.c
index 34458a20..5c0db9cf 100644
--- a/src/tools/irm/irm_ipcp.c
+++ b/src/tools/irm/irm_ipcp.c
@@ -46,6 +46,7 @@ static void usage(void)
printf("Usage: irm ipcp [OPERATION]\n\n"
"where OPERATION in {create destroy\n"
" bootstrap enroll\n"
+ " poa\n"
" connect disconnect\n"
" list\n"
" help}\n");
@@ -68,6 +69,7 @@ static const struct cmd {
{ "destroy", do_destroy_ipcp },
{ "bootstrap", do_bootstrap_ipcp },
{ "enroll", do_enroll_ipcp },
+ { "poa", poa_cmd },
{ "connect", do_connect_ipcp },
{ "disconnect", do_disconnect_ipcp },
{ "list", do_list_ipcp},
diff --git a/src/tools/irm/irm_ipcp_bootstrap.c b/src/tools/irm/irm_ipcp_bootstrap.c
index cc8bf8fa..bbcf2312 100644
--- a/src/tools/irm/irm_ipcp_bootstrap.c
+++ b/src/tools/irm/irm_ipcp_bootstrap.c
@@ -53,10 +53,6 @@
#define UNICAST "unicast"
#define BROADCAST "broadcast"
-#define IP_UDP4 "udp4"
-#define IP_UDP6 "udp6"
-#define ETH_LLC "eth-llc"
-#define ETH_DIX "eth-dix"
#define LOCAL "local"
#define MD5 "MD5"
@@ -76,18 +72,13 @@
#define DT(x) default_dt_config.x
#define DHT(x) default_dht_config.params.x
#define UNI(x) default_uni_config.x
-#define DIX(x) eth_dix_default_conf.eth.x
-#define LLC(x) eth_llc_default_conf.eth.x
-#define UD4(x) udp4_default_conf.udp4.x
-#define UD6(x) udp6_default_conf.udp6.x
static char * usage_str = \
"Usage: irm ipcp bootstrap\n"
" name <ipcp name>\n"
" layer <layer name>\n"
" [type [TYPE]]\n"
- "where TYPE in {" UNICAST " " BROADCAST " " LOCAL " "
- IP_UDP4 " " IP_UDP6 " " ETH_LLC " " ETH_DIX "},\n\n"
+ "where TYPE in {" UNICAST " " BROADCAST " " LOCAL "},\n\n"
"if TYPE == " UNICAST "\n"
" [addr_auth <ADDRESS_POLICY> (default: %s)]\n"
" [directory <DIRECTORY_POLICY> (default: %s)]\n"
@@ -117,28 +108,6 @@ static char * usage_str = \
" [ls_t_recalc <pff recalc interval (s)> (default: %ld)]\n"
" [ls_t_update <LSA update interval (s)> (default: %ld)]\n"
" [ls_t_timeo <link timeout (s)> (default: %ld)]\n\n"
- "if TYPE == " IP_UDP4 "\n"
- " ip <IP address in dotted notation>\n"
- " [port <UDP port> (default: %d)]\n"
- " [dns <DDNS IPv4 address in dotted notation>"
- " (default: none)]\n\n"
- "if TYPE == " IP_UDP6 "\n"
- " ip <IPv6 address>\n"
- " [port <UDP port> (default: %d)]\n"
- " [dns <DDNS IPv6 address>"
- " (default: none)]\n\n"
-
- "if TYPE == " ETH_LLC "\n"
- " dev <interface name>\n"
- " [hash [ALGORITHM] (default: %s)]\n"
- "where ALGORITHM in {" SHA3_224 " " SHA3_256 " "
- SHA3_384 " " SHA3_512 "}\n\n"
- "if TYPE == " ETH_DIX "\n"
- " dev <interface name>\n"
- " [ethertype <ethertype> (default: 0x%4X)]\n"
- " [hash [ALGORITHM] (default: %s)]\n"
- "where ALGORITHM in {" SHA3_224 " " SHA3_256 " "
- SHA3_384 " " SHA3_512 "}\n\n"
"if TYPE == " LOCAL "\n"
" [hash [ALGORITHM] (default: %s)]\n"
"where ALGORITHM in {" SHA3_224 " " SHA3_256 " "
@@ -160,15 +129,6 @@ static void usage(void)
/* ls */
default_ls_config.t_recalc, default_ls_config.t_update,
default_ls_config.t_timeo,
- /* udp4 */
- UD4(port),
- /* udp6 */
- UD6(port),
- /* eth_llc */
- SHA3_256,
- /* eth_dix */
- DIX(ethertype),
- SHA3_256,
/* local */
SHA3_256,
/* broadcast */
@@ -190,19 +150,9 @@ int do_bootstrap_ipcp(int argc,
enum pol_addr_auth addr_auth_type = UNI(addr_auth_type);
enum pol_cong_avoid cong_avoid = UNI(cong_avoid);
enum pol_dir_hash hash_algo = DIR_HASH_SHA3_256;
- char * ipstr = NULL;
- char * dnsstr = NULL;
- struct in_addr ip4_addr = {.s_addr = INADDR_ANY};
- struct in_addr dns4_addr = UD4(dns_addr);
- int port4 = UD4(port);
- struct in6_addr ip6_addr = IN6ADDR_ANY_INIT;
- struct in6_addr dns6_addr = UD6(dns_addr);
- int port6 = UD6(port);
char * ipcp_type = NULL;
enum ipcp_type type = IPCP_INVALID;
char * layer = NULL;
- char * dev = NULL;
- uint16_t ethertype = DIX(ethertype);
struct ipcp_list_info * ipcps;
ssize_t len = 0;
int i = 0;
@@ -211,6 +161,14 @@ int do_bootstrap_ipcp(int argc,
while (argc > 0) {
cargs = 2;
+ if (matches(*argv, "autobind") == 0)
+ cargs = 1;
+
+ if (argc < cargs) {
+ usage();
+ return -1;
+ }
+
if (matches(*argv, "type") == 0) {
ipcp_type = *(argv + 1);
} else if (matches(*argv, "layer") == 0) {
@@ -228,24 +186,6 @@ int do_bootstrap_ipcp(int argc,
hash_algo = DIR_HASH_SHA3_512;
else
goto unknown_param;
- } else if (matches(*argv, "ip") == 0) {
- ipstr = *(argv + 1);
- } else if (matches(*argv, "dns") == 0) {
- dnsstr = *(argv + 1);
- } else if (matches(*argv, "device") == 0) {
- dev = *(argv + 1);
- } else if (matches(*argv, "ethertype") == 0) {
- /* NOTE: We might do some more checks on strtol. */
- if (matches(*(argv + 1), "0x") == 0)
- ethertype = strtol(*(argv + 1), NULL, 0);
- else
- ethertype = strtol(*(argv + 1), NULL, 16);
- if (ethertype < 0x0600 || ethertype >= 0xFFFF) {
- printf("Invalid Ethertype: \"%s\".\n"
- "Recommended range: 0xA000-0xEFFF.\n",
- *(argv + 1));
- return -1;
- }
} else if (matches(*argv, "addr") == 0) {
addr_size = atoi(*(argv + 1));
} else if (matches(*argv, "eid") == 0) {
@@ -254,9 +194,6 @@ int do_bootstrap_ipcp(int argc,
max_ttl = atoi(*(argv + 1));
} else if (matches(*argv, "rtt") == 0) {
max_rtt = atoi(*(argv + 1));
- } else if (matches(*argv, "port") == 0) {
- port4 = atoi(*(argv + 1));
- port6 = port4;
} else if (matches(*argv, "autobind") == 0) {
autobind = true;
cargs = 1;
@@ -334,55 +271,11 @@ int do_bootstrap_ipcp(int argc,
type = IPCP_UNICAST;
else if (matches(ipcp_type, BROADCAST) == 0)
type = IPCP_BROADCAST;
- else if (matches(ipcp_type, IP_UDP4) == 0)
- type = IPCP_UDP4;
- else if (matches(ipcp_type, IP_UDP6) == 0)
- type = IPCP_UDP6;
- else if (matches(ipcp_type, ETH_DIX) == 0)
- type = IPCP_ETH_DIX;
- else if (matches(ipcp_type, ETH_LLC) == 0)
- type = IPCP_ETH_LLC;
else if (matches(ipcp_type, LOCAL) == 0)
type = IPCP_LOCAL;
else goto fail_usage;
}
- if (type == IPCP_UDP4) {
- if (inet_pton (AF_INET, ipstr, &ip4_addr) != 1) {
- printf("Invalid IPv4 address: \"%s\".\n", ipstr);
- goto fail_usage;
- }
-
- if (ip4_addr.s_addr == INADDR_ANY) {
- printf("Cannot use IPv4 address: \"%s\".\n", ipstr);
- goto fail_usage;
- }
-
- if (dnsstr != NULL &&
- inet_pton(AF_INET, dnsstr, &dns4_addr) != 1) {
- printf("Invalid DNS IPv4 address: \"%s\".\n", dnsstr);
- goto fail_usage;
- }
- }
-
- if (type == IPCP_UDP6) {
- if (inet_pton(AF_INET6, ipstr, &ip6_addr) != 1) {
- printf("Invalid IPv6 address: \"%s\".\n", ipstr);
- goto fail_usage;
- }
-
- if (IN6_IS_ADDR_UNSPECIFIED(&ip6_addr)) {
- printf("Cannot use IPv6 address: \"%s\".\n", ipstr);
- goto fail_usage;
- }
-
- if (dnsstr != NULL &&
- inet_pton(AF_INET6, dnsstr, &dns6_addr) != 1) {
- printf("Invalid DNS IPv6 address: \"%s\".\n", dnsstr);
- goto fail_usage;
- }
- }
-
if (pid == -1) {
if (ipcp_type == NULL) {
printf("No IPCPs matching %s found.\n\n", ipcp);
@@ -432,29 +325,6 @@ int do_bootstrap_ipcp(int argc,
conf.unicast.cong_avoid = cong_avoid;
conf.unicast.dir = dir_config;
break;
- case IPCP_UDP4:
- conf.udp4.ip_addr = ip4_addr;
- conf.udp4.dns_addr = dns4_addr;
- conf.udp4.port = port4;
- break;
- case IPCP_UDP6:
- conf.udp6.ip_addr = ip6_addr;
- conf.udp6.dns_addr = dns6_addr;
- conf.udp6.port = port6;
- break;
- case IPCP_ETH_DIX:
- conf.eth.ethertype = ethertype;
- /* FALLTHRU */
- case IPCP_ETH_LLC:
- if (dev == NULL)
- goto fail_usage;
- if (strlen(dev) > DEV_NAME_SIZE) {
- printf("Device name too long.\n\n");
- goto fail_usage;
- }
-
- strcpy(conf.eth.dev, dev);
- break;
case IPCP_BROADCAST:
/* FALLTHRU */
case IPCP_LOCAL:
diff --git a/src/tools/irm/irm_ipcp_connect.c b/src/tools/irm/irm_ipcp_connect.c
index fb21faec..66646e7d 100644
--- a/src/tools/irm/irm_ipcp_connect.c
+++ b/src/tools/irm/irm_ipcp_connect.c
@@ -44,6 +44,7 @@
#include <stdio.h>
#include <stdlib.h>
+#include <arpa/inet.h>
#include <string.h>
#define DT "dt"
@@ -55,10 +56,14 @@ static void usage(void)
" name <ipcp name>\n"
" dst <name of destination IPCP>\n"
" [component [COMPONENT]]\n"
- "where COMPONENT in {" DT " " MGMT "}\n\n"
+ " [udp [UDP_PEER]]\n"
+ " [eth [dev <device>]"
+ " [ethertype <ethertype>]]\n"
+ "where COMPONENT in {" DT " " MGMT "}\n"
+ "and UDP_PEER is <IP address or host>[:<port>]\n\n"
"if COMPONENT == " DT "\n"
- " [qos [QOS]\n"
- "where QOS in {raw, best, voice, video, data}\n");
+ " [qos [QOS]]\n"
+ "where QOS in {raw, safe, rt, rt-safe, msg}\n");
}
int do_connect_ipcp(int argc,
@@ -69,13 +74,29 @@ int do_connect_ipcp(int argc,
char * comp = "*";
char * component = NULL;
char * qos = NULL;
+ char * udpstr = NULL;
+ char * devstr = NULL;
+ uint16_t ethertype = POA_ETHERTYPE;
+ bool eth = false;
+ struct poa_addr addr;
+ struct poa_addr * pa = NULL;
struct ipcp_list_info * ipcps;
ssize_t len = 0;
pid_t pid = -1;
ssize_t i;
+ int cargs;
qosspec_t qs = qos_raw;
while (argc > 0) {
+ cargs = 2;
+ if (strcmp(*argv, "eth") == 0)
+ cargs = 1;
+
+ if (argc < cargs) {
+ usage();
+ return -1;
+ }
+
if (matches(*argv, "name") == 0) {
ipcp = *(argv + 1);
} else if (matches(*argv, "dst") == 0) {
@@ -84,21 +105,66 @@ int do_connect_ipcp(int argc,
comp = *(argv + 1);
} else if (matches(*argv, "qos") == 0) {
qos = *(argv + 1);
+ } else if (strcmp(*argv, "udp") == 0) {
+ udpstr = *(argv + 1);
+ } else if (strcmp(*argv, "dev") == 0) {
+ devstr = *(argv + 1);
+ } else if (strcmp(*argv, "ethertype") == 0) {
+ if (parse_ethertype(*(argv + 1), &ethertype) < 0) {
+ printf("Invalid ethertype: \"%s\".\n",
+ *(argv + 1));
+ return -1;
+ }
+ } else if (strcmp(*argv, "eth") == 0) {
+ eth = true;
+ cargs = 1;
} else {
printf("\"%s\" is unknown, try \"irm "
"ipcp connect\".\n", *argv);
return -1;
}
- argc -= 2;
- argv += 2;
+ argc -= cargs;
+ argv += cargs;
}
- if (ipcp == NULL || dst == NULL || comp == NULL) {
+ if (ipcp == NULL || comp == NULL) {
usage();
return -1;
}
+ memset(&addr, 0, sizeof(addr));
+
+ if (udpstr != NULL) {
+ if (poa_addr_set_udp(&addr, udpstr) < 0)
+ return -1;
+ pa = &addr;
+ }
+
+ if (eth) {
+ if (udpstr != NULL) {
+ printf("Connect via udp or eth, not both.\n");
+ return -1;
+ }
+
+ if (dst == NULL) {
+ usage();
+ return -1;
+ }
+
+ if (poa_addr_set_eth(&addr, devstr, ethertype) < 0)
+ return -1;
+ pa = &addr;
+ }
+
+ if (dst == NULL && pa == NULL) {
+ usage();
+ return -1;
+ }
+
+ if (dst == NULL)
+ dst = "";
+
if (qos != NULL) {
if (strcmp(qos, "raw") == 0)
qs = qos_raw;
@@ -110,9 +176,11 @@ int do_connect_ipcp(int argc,
qs = qos_rt_safe;
else if (strcmp(qos, "msg") == 0)
qs = qos_msg;
- else if (strcmp(qos, "stream") == 0)
- qs = qos_stream;
- else
+ else if (strcmp(qos, "stream") == 0) {
+ printf("Stream QoS is not allowed on "
+ "IPCP component flows.\n");
+ return -1;
+ } else
printf("Unknown QoS cube, defaulting to raw.\n");
}
@@ -129,13 +197,13 @@ int do_connect_ipcp(int argc,
if (wildcard_match(comp, MGMT) == 0) {
component = MGMT_COMP;
/* FIXME: move to qos_msg when stable */
- if (irm_connect_ipcp(pid, dst, component, qos_raw))
+ if (irm_connect_ipcp(pid, dst, component, qos_raw, pa))
return -1;
}
if (wildcard_match(comp, DT) == 0) {
component = DT_COMP;
- if (irm_connect_ipcp(pid, dst, component, qs))
+ if (irm_connect_ipcp(pid, dst, component, qs, pa))
return -1;
}
diff --git a/src/tools/irm/irm_ipcp_create.c b/src/tools/irm/irm_ipcp_create.c
index c6b2074b..15cfedde 100644
--- a/src/tools/irm/irm_ipcp_create.c
+++ b/src/tools/irm/irm_ipcp_create.c
@@ -46,10 +46,6 @@
#define UNICAST "unicast"
#define BROADCAST "broadcast"
-#define UDP4 "udp4"
-#define UDP6 "udp6"
-#define ETH_LLC "eth-llc"
-#define ETH_DIX "eth-dix"
#define LOCAL "local"
static void usage(void)
@@ -57,8 +53,7 @@ static void usage(void)
printf("Usage: irm ipcp create\n"
" name <ipcp name>\n"
" type [TYPE]\n\n"
- "where TYPE in {" UNICAST " " BROADCAST " " LOCAL " "
- UDP4 " " UDP6 " " ETH_LLC " " ETH_DIX "}\n");
+ "where TYPE in {" UNICAST " " BROADCAST " " LOCAL "}\n");
}
int do_create_ipcp(int argc,
@@ -93,16 +88,8 @@ int do_create_ipcp(int argc,
type = IPCP_UNICAST;
else if (strcmp(ipcp_type, BROADCAST) == 0)
type = IPCP_BROADCAST;
- else if (strcmp(ipcp_type, UDP4) == 0)
- type = IPCP_UDP4;
- else if (strcmp(ipcp_type, UDP6) == 0)
- type = IPCP_UDP6;
else if (strcmp(ipcp_type, LOCAL) == 0)
type = IPCP_LOCAL;
- else if (strcmp(ipcp_type, ETH_LLC) == 0)
- type = IPCP_ETH_LLC;
- else if (strcmp(ipcp_type, ETH_DIX) == 0)
- type = IPCP_ETH_DIX;
else {
printf("IPCP type \"%s\" is unknown.\n", ipcp_type);
usage();
diff --git a/src/tools/irm/irm_ipcp_enroll.c b/src/tools/irm/irm_ipcp_enroll.c
index 350b536e..54fed022 100644
--- a/src/tools/irm/irm_ipcp_enroll.c
+++ b/src/tools/irm/irm_ipcp_enroll.c
@@ -44,6 +44,7 @@
#include "irm_ops.h"
#include "irm_utils.h"
+#include <arpa/inet.h>
#include <string.h>
#define UNICAST "unicast"
@@ -57,7 +58,10 @@ static void usage(void)
" [dst <destination to enroll with>]\n"
" [type [TYPE], default = " UNICAST "]\n"
" [autobind]\n"
- "where TYPE in {" UNICAST " " BROADCAST "}\n");
+ " [udp [UDP_PEER]]\n"
+ " [eth [dev <device>] [ethertype <ethertype>]]\n"
+ "where TYPE in {" UNICAST " " BROADCAST "}\n"
+ "and UDP_PEER is <IP address or host>[:<port>]\n");
}
static int get_layer_name(const char * ipcp,
@@ -86,6 +90,12 @@ int do_enroll_ipcp(int argc,
char * ipcp = NULL;
char * layer = NULL;
char * dst = NULL;
+ char * udpstr = NULL;
+ char * devstr = NULL;
+ uint16_t ethertype = POA_ETHERTYPE;
+ bool eth = false;
+ struct poa_addr addr;
+ struct poa_addr * pa = NULL;
struct ipcp_list_info * ipcps;
pid_t pid = -1;
ssize_t len = 0;
@@ -97,14 +107,37 @@ int do_enroll_ipcp(int argc,
while (argc > 0) {
cargs = 2;
+ if (strcmp(*argv, "eth") == 0)
+ cargs = 1;
+ else if (matches(*argv, "autobind") == 0)
+ cargs = 1;
+
+ if (argc < cargs) {
+ usage();
+ return -1;
+ }
+
if (matches(*argv, "name") == 0) {
ipcp = *(argv + 1);
} else if (matches(*argv, "type") == 0) {
ipcp_type = *(argv + 1);
} else if (matches(*argv, "layer") == 0) {
layer = *(argv + 1);
- } else if (matches(*argv, "dst") == 0) {
+ } else if (strcmp(*argv, "dst") == 0) {
dst = *(argv + 1);
+ } else if (strcmp(*argv, "udp") == 0) {
+ udpstr = *(argv + 1);
+ } else if (strcmp(*argv, "dev") == 0) {
+ devstr = *(argv + 1);
+ } else if (strcmp(*argv, "ethertype") == 0) {
+ if (parse_ethertype(*(argv + 1), &ethertype) < 0) {
+ printf("Invalid ethertype: \"%s\".\n",
+ *(argv + 1));
+ return -1;
+ }
+ } else if (strcmp(*argv, "eth") == 0) {
+ eth = true;
+ cargs = 1;
} else if (matches(*argv, "autobind") == 0) {
autobind = true;
cargs = 1;
@@ -126,6 +159,25 @@ int do_enroll_ipcp(int argc,
if (dst == NULL)
dst = layer;
+ memset(&addr, 0, sizeof(addr));
+
+ if (udpstr != NULL) {
+ if (poa_addr_set_udp(&addr, udpstr) < 0)
+ return -1;
+ pa = &addr;
+ }
+
+ if (eth) {
+ if (udpstr != NULL) {
+ printf("Enroll via udp or eth, not both.\n");
+ return -1;
+ }
+
+ if (poa_addr_set_eth(&addr, devstr, ethertype) < 0)
+ return -1;
+ pa = &addr;
+ }
+
if (strcmp(ipcp_type, UNICAST) == 0)
type = IPCP_UNICAST;
else if (strcmp(ipcp_type, BROADCAST) == 0)
@@ -155,7 +207,7 @@ int do_enroll_ipcp(int argc,
pid = ipcps[i].pid;
- if (irm_enroll_ipcp(pid, dst)) {
+ if (irm_enroll_ipcp(pid, dst, pa) < 0) {
printf("Failed to enroll IPCP.\n");
goto fail;
}
diff --git a/src/tools/irm/irm_ipcp_list.c b/src/tools/irm/irm_ipcp_list.c
index a211a02b..60154efb 100644
--- a/src/tools/irm/irm_ipcp_list.c
+++ b/src/tools/irm/irm_ipcp_list.c
@@ -48,10 +48,6 @@
#define UNICAST "unicast"
#define BROADCAST "broadcast"
-#define UDP4 "udp4"
-#define UDP6 "udp6"
-#define ETH_LLC "eth-llc"
-#define ETH_DIX "eth-dix"
#define LOCAL "local"
static void usage(void)
@@ -60,8 +56,7 @@ static void usage(void)
" [name <ipcp name>]\n"
" [layer <layer_name>]\n\n"
" [type [TYPE]]\n\n"
- "where TYPE = {" UNICAST " " LOCAL " "
- UDP4 " " UDP6 " " ETH_LLC " " ETH_DIX "}\n");
+ "where TYPE = {" UNICAST " " BROADCAST " " LOCAL "}\n");
}
static char * str_type(enum ipcp_type type)
@@ -71,14 +66,6 @@ static char * str_type(enum ipcp_type type)
return UNICAST;
case IPCP_BROADCAST:
return BROADCAST;
- case IPCP_ETH_LLC:
- return ETH_LLC;
- case IPCP_ETH_DIX:
- return ETH_DIX;
- case IPCP_UDP4:
- return UDP4;
- case IPCP_UDP6:
- return UDP6;
case IPCP_LOCAL:
return LOCAL;
default:
@@ -116,16 +103,8 @@ int do_list_ipcp(int argc,
type = IPCP_UNICAST;
else if (strcmp(ipcp_type, BROADCAST) == 0)
type = IPCP_BROADCAST;
- else if (strcmp(ipcp_type, UDP4) == 0)
- type = IPCP_UDP4;
- else if (strcmp(ipcp_type, UDP6) == 0)
- type = IPCP_UDP6;
else if (strcmp(ipcp_type, LOCAL) == 0)
type = IPCP_LOCAL;
- else if (strcmp(ipcp_type, ETH_LLC) == 0)
- type = IPCP_ETH_LLC;
- else if (strcmp(ipcp_type, ETH_DIX) == 0)
- type = IPCP_ETH_DIX;
else {
usage();
return -1;
diff --git a/src/tools/irm/irm_ipcp_poa.c b/src/tools/irm/irm_ipcp_poa.c
new file mode 100644
index 00000000..cd939020
--- /dev/null
+++ b/src/tools/irm/irm_ipcp_poa.c
@@ -0,0 +1,98 @@
+/*
+ * Ouroboros - Copyright (C) 2016 - 2026
+ *
+ * Connect components of unicast or broadcast IPC processes
+ *
+ * Dimitri Staessens <dimitri@ouroboros.rocks>
+ * Sander Vrijders <sander@ouroboros.rocks>
+ *
+ * Redistribution and use in source and binary forms, with or without
+ * modification, are permitted provided that the following conditions
+ * are met:
+ *
+ * 1. Redistributions of source code must retain the above copyright
+ * notice, this list of conditions and the following disclaimer.
+ *
+ * 2. Redistributions in binary form must reproduce the above
+ * copyright notice, this list of conditions and the following
+ * disclaimer in the documentation and/or other materials provided
+ * with the distribution.
+ *
+ * 3. Neither the name of the copyright holder nor the names of its
+ * contributors may be used to endorse or promote products derived
+ * from this software without specific prior written permission.
+ *
+ * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS
+ * "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
+ * LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS
+ * FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE
+ * COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT,
+ * INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES
+ * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
+ * SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
+ * HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT,
+ * STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
+ * ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED
+ * OF THE POSSIBILITY OF SUCH DAMAGE.
+ */
+
+#include "irm_ops.h"
+#include "irm_utils.h"
+
+#include <stdio.h>
+#include <string.h>
+
+static void usage(void)
+{
+ printf("Usage: irm ipcp poa [OPERATION]\n\n"
+ "where OPERATION in {attach detach list help}\n");
+}
+
+static int do_help(int argc,
+ char ** argv)
+{
+ (void) argc;
+ (void) argv;
+
+ usage();
+
+ return 0;
+}
+
+static const struct cmd {
+ const char * cmd;
+ int (* func)(int argc, char ** argv);
+} cmds[] = {
+ { "attach", do_attach_ipcp },
+ { "detach", do_detach_ipcp },
+ { "list", do_list_poas },
+ { "help", do_help },
+ { NULL, NULL }
+};
+
+static int do_cmd(const char * argv0,
+ int argc,
+ char ** argv)
+{
+ const struct cmd * c;
+
+ for (c = cmds; c->cmd != NULL; ++c)
+ if (matches(argv0, c->cmd) == 0)
+ return c->func(argc - 1, argv + 1);
+
+ fprintf(stderr, "\"%s\" is unknown, try \"irm ipcp poa help\".\n",
+ argv0);
+
+ return -1;
+}
+
+int poa_cmd(int argc,
+ char ** argv)
+{
+ if (argc < 1) {
+ usage();
+ return -1;
+ }
+
+ return do_cmd(argv[0], argc, argv);
+}
diff --git a/src/tools/irm/irm_ipcp_poa_attach.c b/src/tools/irm/irm_ipcp_poa_attach.c
new file mode 100644
index 00000000..74a83344
--- /dev/null
+++ b/src/tools/irm/irm_ipcp_poa_attach.c
@@ -0,0 +1,153 @@
+/*
+ * Ouroboros - Copyright (C) 2016 - 2026
+ *
+ * Connect components of unicast or broadcast IPC processes
+ *
+ * Dimitri Staessens <dimitri@ouroboros.rocks>
+ * Sander Vrijders <sander@ouroboros.rocks>
+ *
+ * Redistribution and use in source and binary forms, with or without
+ * modification, are permitted provided that the following conditions
+ * are met:
+ *
+ * 1. Redistributions of source code must retain the above copyright
+ * notice, this list of conditions and the following disclaimer.
+ *
+ * 2. Redistributions in binary form must reproduce the above
+ * copyright notice, this list of conditions and the following
+ * disclaimer in the documentation and/or other materials provided
+ * with the distribution.
+ *
+ * 3. Neither the name of the copyright holder nor the names of its
+ * contributors may be used to endorse or promote products derived
+ * from this software without specific prior written permission.
+ *
+ * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS
+ * "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
+ * LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS
+ * FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE
+ * COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT,
+ * INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES
+ * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
+ * SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
+ * HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT,
+ * STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
+ * ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED
+ * OF THE POSSIBILITY OF SUCH DAMAGE.
+ */
+
+#include <ouroboros/irm.h>
+
+#include "irm_ops.h"
+#include "irm_utils.h"
+
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+
+static void usage(void)
+{
+ printf("Usage: irm ipcp poa attach\n"
+ " name <ipcp name>\n"
+ " [udp [UDP_POA]]\n"
+ " [eth [ETH_POA]]\n"
+ "where exactly one of udp or eth is given\n"
+ "and UDP_POA is <local IP address>[:<port>]\n"
+ "and ETH_POA is dev <device> [ethertype <ethertype>]\n");
+}
+
+int do_attach_ipcp(int argc,
+ char ** argv)
+{
+ char * ipcp = NULL;
+ char * udpstr = NULL;
+ char * devstr = NULL;
+ uint16_t ethertype = POA_ETHERTYPE;
+ bool eth = false;
+ struct poa_spec poa;
+ char str[POA_STRLEN + 1];
+ struct ipcp_list_info * ipcps;
+ ssize_t len = 0;
+ pid_t pid = -1;
+ ssize_t i;
+ int cargs;
+
+ while (argc > 0) {
+ cargs = 2;
+
+ if (strcmp(*argv, "eth") == 0)
+ cargs = 1;
+
+ if (argc < cargs) {
+ usage();
+ return -1;
+ }
+
+ if (matches(*argv, "name") == 0) {
+ ipcp = *(argv + 1);
+ } else if (strcmp(*argv, "udp") == 0) {
+ udpstr = *(argv + 1);
+ } else if (strcmp(*argv, "dev") == 0) {
+ devstr = *(argv + 1);
+ } else if (strcmp(*argv, "ethertype") == 0) {
+ if (parse_ethertype(*(argv + 1), &ethertype) < 0) {
+ printf("Invalid ethertype: \"%s\".\n",
+ *(argv + 1));
+ return -1;
+ }
+ } else if (strcmp(*argv, "eth") == 0) {
+ eth = true;
+ cargs = 1;
+ } else {
+ printf("\"%s\" is unknown, try \"irm ipcp "
+ "poa attach\".\n", *argv);
+ return -1;
+ }
+
+ argc -= cargs;
+ argv += cargs;
+ }
+
+ if (ipcp == NULL) {
+ usage();
+ return -1;
+ }
+
+ if (eth && udpstr != NULL) {
+ printf("A PoA is udp or eth, not both.\n");
+ return -1;
+ }
+
+ if (eth && devstr == NULL) {
+ printf("An eth PoA needs a device.\n");
+ return -1;
+ }
+
+ if (!eth && udpstr == NULL) {
+ usage();
+ return -1;
+ }
+
+ if (poa_spec_set(&poa, udpstr, devstr, ethertype) < 0)
+ return -1;
+
+ len = irm_list_ipcps(&ipcps);
+ for (i = 0; i < len; i++)
+ if (strcmp(ipcps[i].name, ipcp) == 0)
+ pid = ipcps[i].pid;
+
+ free(ipcps);
+
+ if (pid == -1) {
+ printf("No such IPCP: \"%s\".\n", ipcp);
+ return -1;
+ }
+
+ if (irm_attach_ipcp(pid, &poa) < 0) {
+ poa_spec_str(&poa, str, sizeof(str));
+ printf("Failed to attach PoA %s on IPCP %s.\n", str, ipcp);
+ return -1;
+ }
+
+ return 0;
+}
diff --git a/src/tools/irm/irm_ipcp_poa_detach.c b/src/tools/irm/irm_ipcp_poa_detach.c
new file mode 100644
index 00000000..ce6bef13
--- /dev/null
+++ b/src/tools/irm/irm_ipcp_poa_detach.c
@@ -0,0 +1,153 @@
+/*
+ * Ouroboros - Copyright (C) 2016 - 2026
+ *
+ * Connect components of unicast or broadcast IPC processes
+ *
+ * Dimitri Staessens <dimitri@ouroboros.rocks>
+ * Sander Vrijders <sander@ouroboros.rocks>
+ *
+ * Redistribution and use in source and binary forms, with or without
+ * modification, are permitted provided that the following conditions
+ * are met:
+ *
+ * 1. Redistributions of source code must retain the above copyright
+ * notice, this list of conditions and the following disclaimer.
+ *
+ * 2. Redistributions in binary form must reproduce the above
+ * copyright notice, this list of conditions and the following
+ * disclaimer in the documentation and/or other materials provided
+ * with the distribution.
+ *
+ * 3. Neither the name of the copyright holder nor the names of its
+ * contributors may be used to endorse or promote products derived
+ * from this software without specific prior written permission.
+ *
+ * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS
+ * "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
+ * LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS
+ * FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE
+ * COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT,
+ * INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES
+ * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
+ * SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
+ * HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT,
+ * STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
+ * ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED
+ * OF THE POSSIBILITY OF SUCH DAMAGE.
+ */
+
+#include <ouroboros/irm.h>
+
+#include "irm_ops.h"
+#include "irm_utils.h"
+
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+
+static void usage(void)
+{
+ printf("Usage: irm ipcp poa detach\n"
+ " name <ipcp name>\n"
+ " [udp [UDP_POA]]\n"
+ " [eth [ETH_POA]]\n"
+ "where exactly one of udp or eth is given\n"
+ "and UDP_POA is <local IP address>[:<port>]\n"
+ "and ETH_POA is dev <device> [ethertype <ethertype>]\n");
+}
+
+int do_detach_ipcp(int argc,
+ char ** argv)
+{
+ char * ipcp = NULL;
+ char * udpstr = NULL;
+ char * devstr = NULL;
+ uint16_t ethertype = POA_ETHERTYPE;
+ bool eth = false;
+ struct poa_spec poa;
+ char str[POA_STRLEN + 1];
+ struct ipcp_list_info * ipcps;
+ ssize_t len = 0;
+ pid_t pid = -1;
+ ssize_t i;
+ int cargs;
+
+ while (argc > 0) {
+ cargs = 2;
+
+ if (strcmp(*argv, "eth") == 0)
+ cargs = 1;
+
+ if (argc < cargs) {
+ usage();
+ return -1;
+ }
+
+ if (matches(*argv, "name") == 0) {
+ ipcp = *(argv + 1);
+ } else if (strcmp(*argv, "udp") == 0) {
+ udpstr = *(argv + 1);
+ } else if (strcmp(*argv, "dev") == 0) {
+ devstr = *(argv + 1);
+ } else if (strcmp(*argv, "ethertype") == 0) {
+ if (parse_ethertype(*(argv + 1), &ethertype) < 0) {
+ printf("Invalid ethertype: \"%s\".\n",
+ *(argv + 1));
+ return -1;
+ }
+ } else if (strcmp(*argv, "eth") == 0) {
+ eth = true;
+ cargs = 1;
+ } else {
+ printf("\"%s\" is unknown, try \"irm ipcp "
+ "poa detach\".\n", *argv);
+ return -1;
+ }
+
+ argc -= cargs;
+ argv += cargs;
+ }
+
+ if (ipcp == NULL) {
+ usage();
+ return -1;
+ }
+
+ if (eth && udpstr != NULL) {
+ printf("A PoA is udp or eth, not both.\n");
+ return -1;
+ }
+
+ if (eth && devstr == NULL) {
+ printf("An eth PoA needs a device.\n");
+ return -1;
+ }
+
+ if (!eth && udpstr == NULL) {
+ usage();
+ return -1;
+ }
+
+ if (poa_spec_set(&poa, udpstr, devstr, ethertype) < 0)
+ return -1;
+
+ len = irm_list_ipcps(&ipcps);
+ for (i = 0; i < len; i++)
+ if (strcmp(ipcps[i].name, ipcp) == 0)
+ pid = ipcps[i].pid;
+
+ free(ipcps);
+
+ if (pid == -1) {
+ printf("No such IPCP: \"%s\".\n", ipcp);
+ return -1;
+ }
+
+ if (irm_detach_ipcp(pid, &poa) < 0) {
+ poa_spec_str(&poa, str, sizeof(str));
+ printf("Failed to detach PoA %s on IPCP %s.\n", str, ipcp);
+ return -1;
+ }
+
+ return 0;
+}
diff --git a/src/tools/irm/irm_ipcp_poa_list.c b/src/tools/irm/irm_ipcp_poa_list.c
new file mode 100644
index 00000000..8797c3c1
--- /dev/null
+++ b/src/tools/irm/irm_ipcp_poa_list.c
@@ -0,0 +1,136 @@
+/*
+ * Ouroboros - Copyright (C) 2016 - 2026
+ *
+ * List the points of attachment of an IPC process
+ *
+ * Dimitri Staessens <dimitri@ouroboros.rocks>
+ * Sander Vrijders <sander@ouroboros.rocks>
+ *
+ * Redistribution and use in source and binary forms, with or without
+ * modification, are permitted provided that the following conditions
+ * are met:
+ *
+ * 1. Redistributions of source code must retain the above copyright
+ * notice, this list of conditions and the following disclaimer.
+ *
+ * 2. Redistributions in binary form must reproduce the above
+ * copyright notice, this list of conditions and the following
+ * disclaimer in the documentation and/or other materials provided
+ * with the distribution.
+ *
+ * 3. Neither the name of the copyright holder nor the names of its
+ * contributors may be used to endorse or promote products derived
+ * from this software without specific prior written permission.
+ *
+ * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS
+ * "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
+ * LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS
+ * FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE
+ * COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT,
+ * INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES
+ * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
+ * SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
+ * HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT,
+ * STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
+ * ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED
+ * OF THE POSSIBILITY OF SUCH DAMAGE.
+ */
+
+#include <ouroboros/irm.h>
+
+#include "irm_ops.h"
+#include "irm_utils.h"
+
+#include <arpa/inet.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+
+static void usage(void)
+{
+ printf("Usage: irm ipcp poa list\n"
+ " name <ipcp name>\n");
+}
+
+static void print_poa(const struct poa_spec * poa)
+{
+ char buf[INET6_ADDRSTRLEN];
+
+ switch (poa->type) {
+ case POA_UDP4:
+ if (inet_ntop(AF_INET, &poa->udp4.ip_addr, buf,
+ sizeof(buf)) == NULL)
+ return;
+ printf("%-6s %s:%u\n", "udp4", buf, poa->udp4.port);
+ break;
+ case POA_UDP6:
+ if (inet_ntop(AF_INET6, &poa->udp6.ip_addr, buf,
+ sizeof(buf)) == NULL)
+ return;
+ printf("%-6s [%s]:%u\n", "udp6", buf, poa->udp6.port);
+ break;
+ case POA_ETH:
+ printf("%-6s %s 0x%04X %02x:%02x:%02x:%02x:%02x:%02x\n",
+ "eth", poa->eth.dev, poa->eth.ethertype,
+ poa->eth.mac[0], poa->eth.mac[1], poa->eth.mac[2],
+ poa->eth.mac[3], poa->eth.mac[4], poa->eth.mac[5]);
+ break;
+ default:
+ break;
+ }
+}
+
+int do_list_poas(int argc,
+ char ** argv)
+{
+ char * ipcp = NULL;
+ struct poa_spec * poas = NULL;
+ struct ipcp_list_info * ipcps;
+ ssize_t len = 0;
+ pid_t pid = -1;
+ ssize_t n;
+ ssize_t i;
+
+ while (argc > 1) {
+ if (matches(*argv, "name") == 0) {
+ ipcp = *(argv + 1);
+ } else {
+ printf("\"%s\" is unknown, try \"irm ipcp "
+ "poa list\".\n", *argv);
+ return -1;
+ }
+
+ argc -= 2;
+ argv += 2;
+ }
+
+ if (ipcp == NULL) {
+ usage();
+ return -1;
+ }
+
+ len = irm_list_ipcps(&ipcps);
+ for (i = 0; i < len; i++)
+ if (strcmp(ipcps[i].name, ipcp) == 0)
+ pid = ipcps[i].pid;
+
+ free(ipcps);
+
+ if (pid == -1) {
+ printf("No such IPCP: \"%s\".\n", ipcp);
+ return -1;
+ }
+
+ n = irm_list_poas(pid, &poas);
+ if (n < 0) {
+ printf("Failed to list PoAs of %s.\n", ipcp);
+ return -1;
+ }
+
+ for (i = 0; i < n; i++)
+ print_poa(&poas[i]);
+
+ free(poas);
+
+ return 0;
+}
diff --git a/src/tools/irm/irm_ops.h b/src/tools/irm/irm_ops.h
index 195c5cbc..7f6c65cd 100644
--- a/src/tools/irm/irm_ops.h
+++ b/src/tools/irm/irm_ops.h
@@ -51,6 +51,18 @@ int do_bootstrap_ipcp(int argc,
int do_enroll_ipcp(int argc,
char ** argv);
+int poa_cmd(int argc,
+ char ** argv);
+
+int do_attach_ipcp(int argc,
+ char ** argv);
+
+int do_detach_ipcp(int argc,
+ char ** argv);
+
+int do_list_poas(int argc,
+ char ** argv);
+
int do_connect_ipcp(int argc,
char ** argv);
diff --git a/src/tools/irm/irm_utils.c b/src/tools/irm/irm_utils.c
index 69873097..c43accec 100644
--- a/src/tools/irm/irm_utils.c
+++ b/src/tools/irm/irm_utils.c
@@ -77,10 +77,21 @@
*/
-#include <string.h>
+#if defined(__linux__) || defined(__CYGWIN__)
+#define _DEFAULT_SOURCE
+#else
+#define _POSIX_C_SOURCE 200809L
+#endif
+
+#include <ouroboros/ipcp.h>
#include "irm_utils.h"
+#include <arpa/inet.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+
int matches(const char * cmd,
const char * pattern)
{
@@ -123,3 +134,223 @@ int wildcard_match(const char * pattern,
}
}
}
+
+/* Splits "<addr>[:<port>]"; bare IPv6 needs no brackets. */
+static int parse_udp_str(const char * str,
+ char * host,
+ int * port)
+{
+ struct in6_addr v6;
+ char buf[POA_HOST_STRLEN + 1];
+ char * p;
+ char * end;
+ long n;
+
+ *port = POA_UDP_PORT;
+
+ if (strlen(str) > POA_HOST_STRLEN)
+ goto fail;
+
+ strcpy(buf, str);
+
+ if (buf[0] == '[') {
+ p = strchr(buf, ']');
+ if (p == NULL)
+ goto fail;
+
+ *p++ = '\0';
+
+ strcpy(host, buf + 1);
+
+ if (*p == '\0')
+ return 0;
+
+ if (*p != ':')
+ goto fail;
+
+ ++p;
+ } else if (inet_pton(AF_INET6, buf, &v6) == 1) {
+ strcpy(host, buf);
+ return 0;
+ } else {
+ p = strrchr(buf, ':');
+ if (p == NULL) {
+ strcpy(host, buf);
+ return 0;
+ }
+
+ *p++ = '\0';
+
+ strcpy(host, buf);
+ }
+
+ n = strtol(p, &end, 10);
+ if (*p == '\0' || *end != '\0' || n < 1 || n > 65535)
+ goto fail;
+
+ *port = (int) n;
+
+ return 0;
+ fail:
+ printf("Invalid UDP address: \"%s\".\n", str);
+ return -1;
+}
+
+/* An unresolved name is left for the IRMd, which picks the family. */
+int poa_addr_set_udp(struct poa_addr * addr,
+ const char * str)
+{
+ char host[POA_HOST_STRLEN + 1];
+ int port;
+
+ if (parse_udp_str(str, host, &port) < 0)
+ return -1;
+
+ if (inet_pton(AF_INET, host, &addr->udp4.ip_addr) == 1) {
+ addr->type = POA_UDP4;
+ addr->udp4.port = port;
+ return 0;
+ }
+
+ if (inet_pton(AF_INET6, host, &addr->udp6.ip_addr) == 1) {
+ addr->type = POA_UDP6;
+ addr->udp6.port = port;
+ return 0;
+ }
+
+ addr->type = POA_UDP;
+ addr->udp4.port = port;
+
+ strcpy(addr->hostname, host);
+
+ return 0;
+}
+
+/*
+ * Parses a hex ethertype; rejects garbage and out-of-range values.
+ * Overflow clamps to LONG_MAX and lands in the range check.
+ */
+int parse_ethertype(const char * str,
+ uint16_t * ethertype)
+{
+ char * end;
+ long val;
+
+ val = strtol(str, &end, 16);
+
+ if (end == str || *end != '\0')
+ return -1;
+
+ if (val < 0 || val > 0xFFFF)
+ return -1;
+
+ *ethertype = (uint16_t) val;
+
+ return 0;
+}
+
+int poa_addr_set_eth(struct poa_addr * addr,
+ const char * devstr,
+ uint16_t ethertype)
+{
+ addr->type = POA_ETH;
+
+ addr->eth.src.ethertype = ethertype;
+ addr->eth.dst.ethertype = ethertype;
+
+ if (devstr != NULL) {
+ if (strlen(devstr) > DEV_NAME_SIZE) {
+ printf("Invalid device name: \"%s\".\n", devstr);
+ return -1;
+ }
+
+ strcpy(addr->eth.src.dev, devstr);
+ }
+
+ return 0;
+}
+
+/* Matches src/ipcpd/ipcp.c; keep in sync. */
+void poa_spec_str(const struct poa_spec * poa,
+ char * buf,
+ size_t len)
+{
+ char addr[INET6_ADDRSTRLEN];
+
+ switch (poa->type) {
+ case POA_UDP4:
+ if (inet_ntop(AF_INET, &poa->udp4.ip_addr,
+ addr, sizeof(addr)) == NULL)
+ break;
+
+ snprintf(buf, len, "udp4 %s:%u", addr, poa->udp4.port);
+ return;
+ case POA_UDP6:
+ if (inet_ntop(AF_INET6, &poa->udp6.ip_addr,
+ addr, sizeof(addr)) == NULL)
+ break;
+
+ snprintf(buf, len, "udp6 [%s]:%u", addr, poa->udp6.port);
+ return;
+ case POA_ETH:
+ snprintf(buf, len, "eth %s 0x%04X", poa->eth.dev,
+ poa->eth.ethertype);
+ return;
+ default:
+ break;
+ }
+
+ snprintf(buf, len, "(unknown)");
+}
+
+int poa_spec_set(struct poa_spec * poa,
+ const char * udpstr,
+ const char * devstr,
+ uint16_t ethertype)
+{
+ char host[POA_HOST_STRLEN + 1];
+ int port;
+
+ memset(poa, 0, sizeof(*poa));
+
+ if ((udpstr != NULL) + (devstr != NULL) > 1) {
+ printf("A PoA is an address or a device.\n");
+ return -1;
+ }
+
+ if (udpstr != NULL) {
+ if (parse_udp_str(udpstr, host, &port) < 0)
+ return -1;
+
+ if (inet_pton(AF_INET, host, &poa->udp4.ip_addr) == 1) {
+ poa->type = POA_UDP4;
+ poa->udp4.port = port;
+ return 0;
+ }
+
+ if (inet_pton(AF_INET6, host, &poa->udp6.ip_addr) == 1) {
+ poa->type = POA_UDP6;
+ poa->udp6.port = port;
+ return 0;
+ }
+
+ printf("Invalid IP address: \"%s\".\n", udpstr);
+ return -1;
+ }
+
+ if (devstr != NULL) {
+ if (strlen(devstr) > DEV_NAME_SIZE) {
+ printf("Invalid device name: \"%s\".\n", devstr);
+ return -1;
+ }
+
+ poa->type = POA_ETH;
+ poa->eth.ethertype = ethertype;
+
+ strcpy(poa->eth.dev, devstr);
+
+ return 0;
+ }
+
+ return -1;
+}
diff --git a/src/tools/irm/irm_utils.h b/src/tools/irm/irm_utils.h
index c6d4bf18..3760b4e4 100644
--- a/src/tools/irm/irm_utils.h
+++ b/src/tools/irm/irm_utils.h
@@ -79,10 +79,43 @@
#ifndef OUROBOROS_TOOLS_IRM_UTILS_H
#define OUROBOROS_TOOLS_IRM_UTILS_H
-int matches(const char * cmd,
- const char * pattern);
+#include <ouroboros/ipcp.h>
-int wildcard_match(const char * pattern,
- const char * string);
+#include <arpa/inet.h>
+
+#include <stdint.h>
+
+int matches(const char * cmd,
+ const char * pattern);
+
+int wildcard_match(const char * pattern,
+ const char * string);
+
+/* Fill one PoA to attach or detach: an address or a device. */
+int poa_spec_set(struct poa_spec * poa,
+ const char * udpstr,
+ const char * devstr,
+ uint16_t ethertype);
+
+/* Fill a peer PoA address; a host name is resolved by the IRMd. */
+int poa_addr_set_udp(struct poa_addr * addr,
+ const char * str);
+
+/* Fits "eth <dev> 0x<type>", the longest PoA rendering. */
+/* Matches src/ipcpd/ipcp.c; keep in sync. */
+#define POA_STRLEN (DEV_NAME_SIZE + 11)
+
+/* Names a PoA the way the IPCP logs it. */
+void poa_spec_str(const struct poa_spec * poa,
+ char * buf,
+ size_t len);
+
+int poa_addr_set_eth(struct poa_addr * addr,
+ const char * devstr,
+ uint16_t ethertype);
+
+/* Parses a hex ethertype; rejects garbage and out-of-range values. */
+int parse_ethertype(const char * str,
+ uint16_t * ethertype);
#endif /* OUROBOROS_TOOLS_IRM_UTILS_H */