Include liburing in static build
This commit is contained in:
@@ -20,7 +20,7 @@ RUN echo 'deb http://deb.debian.org/debian bullseye-backports main' >> /etc/apt/
|
||||
|
||||
RUN apt-get update
|
||||
RUN apt-get -y install etcd qemu-system-x86 qemu-block-extra qemu-utils fio libasan5 \
|
||||
liburing1 liburing-dev libgoogle-perftools-dev devscripts libjerasure-dev cmake libibverbs-dev libisal-dev
|
||||
libgoogle-perftools-dev devscripts libjerasure-dev cmake libibverbs-dev libisal-dev
|
||||
RUN apt-get -y build-dep fio qemu=`dpkg -s qemu-system-x86|grep ^Version:|awk '{print $2}'`
|
||||
RUN apt-get update && apt-get -y install jq lp-solve sudo nfs-common fdisk parted
|
||||
RUN apt-get --download-only source fio qemu=`dpkg -s qemu-system-x86|grep ^Version:|awk '{print $2}'`
|
||||
|
||||
Vendored
+1
-1
@@ -2,7 +2,7 @@ Source: vitastor
|
||||
Section: admin
|
||||
Priority: optional
|
||||
Maintainer: Vitaliy Filippov <vitalif@yourcmc.ru>
|
||||
Build-Depends: debhelper, liburing-dev (>= 0.6), g++ (>= 8), libstdc++6 (>= 8),
|
||||
Build-Depends: debhelper, g++ (>= 8), libstdc++6 (>= 8),
|
||||
linux-libc-dev, libgoogle-perftools-dev, libjerasure-dev, libgf-complete-dev,
|
||||
libibverbs-dev, libisal-dev, cmake, pkg-config, libnl-3-dev, libnl-genl-3-dev,
|
||||
node-bindings <!nocheck>, node-gyp, node-nan
|
||||
|
||||
Vendored
+1
-1
@@ -26,7 +26,7 @@ RUN if [ "$REL" = "buster" -o "$REL" = "bullseye" -o "$REL" = "bookworm" ]; then
|
||||
echo 'APT::Install-Suggests false;' >> /etc/apt/apt.conf
|
||||
|
||||
RUN apt-get update
|
||||
RUN DEBIAN_FRONTEND=noninteractive TZ=Europe/Moscow apt-get -y install fio liburing-dev libgoogle-perftools-dev devscripts
|
||||
RUN DEBIAN_FRONTEND=noninteractive TZ=Europe/Moscow apt-get -y install fio libgoogle-perftools-dev devscripts
|
||||
RUN DEBIAN_FRONTEND=noninteractive TZ=Europe/Moscow apt-get -y build-dep qemu
|
||||
# To build a custom version
|
||||
#RUN cp /root/packages/qemu-orig/* /root
|
||||
|
||||
Vendored
+1
-1
@@ -25,7 +25,7 @@ RUN set -e -x; \
|
||||
echo 'APT::Install-Suggests false;' >> /etc/apt/apt.conf
|
||||
|
||||
RUN apt-get update && \
|
||||
apt-get -y install fio liburing-dev libgoogle-perftools-dev devscripts libjerasure-dev cmake \
|
||||
apt-get -y install fio libgoogle-perftools-dev devscripts libjerasure-dev cmake \
|
||||
libibverbs-dev librdmacm-dev libisal-dev libnl-3-dev libnl-genl-3-dev curl nodejs npm node-nan node-bindings && \
|
||||
apt-get -y build-dep fio && \
|
||||
apt-get --download-only source fio
|
||||
|
||||
@@ -49,7 +49,6 @@
|
||||
recommended because io_uring is a relatively new technology and there is
|
||||
at least one bug which reproduces with io_uring and HP SmartArray
|
||||
controllers in 5.4
|
||||
- liburing 0.4 or newer
|
||||
- lp_solve
|
||||
- etcd 3.4.15 or newer. Earlier versions won't work because of various bugs,
|
||||
for example [#12402](https://github.com/etcd-io/etcd/pull/12402).
|
||||
|
||||
@@ -48,7 +48,6 @@
|
||||
- Ядро Linux 5.4 или новее, для поддержки io_uring. Рекомендуется даже 5.8,
|
||||
так как io_uring - относительно новый интерфейс и в версиях до 5.8 встречались
|
||||
некоторые баги, например, зависание с io_uring и контроллером HP SmartArray
|
||||
- liburing 0.4 или новее
|
||||
- lp_solve
|
||||
- etcd 3.4.15 или новее. Более старые версии не будут работать из-за разных багов,
|
||||
например, [#12402](https://github.com/etcd-io/etcd/pull/12402).
|
||||
|
||||
@@ -15,7 +15,7 @@
|
||||
- gcc and g++ 8 or newer, clang 10 or newer, or other compiler with C++11 plus
|
||||
designated initializers support from C++20
|
||||
- CMake
|
||||
- liburing, jerasure headers and libraries
|
||||
- jerasure headers and libraries
|
||||
- ISA-L, libibverbs and librdmacm headers and libraries (optional)
|
||||
- tcmalloc (google-perftools-dev)
|
||||
|
||||
|
||||
@@ -15,7 +15,7 @@
|
||||
- gcc и g++ >= 8, либо clang >= 10, либо другой компилятор с поддержкой C++11 плюс
|
||||
назначенных инициализаторов (designated initializers) из C++20
|
||||
- CMake
|
||||
- Заголовки и библиотеки liburing, jerasure
|
||||
- Заголовки и библиотеки jerasure
|
||||
- Опционально - заголовки и библиотеки ISA-L, libibverbs, librdmacm
|
||||
- tcmalloc (google-perftools-dev)
|
||||
|
||||
|
||||
@@ -21,17 +21,3 @@ RUN rpm --nomd5 -i fio*.src.rpm
|
||||
RUN rm -f /etc/yum.repos.d/CentOS-Media.repo
|
||||
RUN cd ~/rpmbuild/SPECS && yum-builddep -y fio.spec
|
||||
RUN yum -y install cmake3
|
||||
|
||||
ADD https://vitastor.io/rpms/liburing-el7/liburing-0.7-2.el7.src.rpm /root
|
||||
|
||||
RUN set -e; \
|
||||
rpm -i liburing*.src.rpm; \
|
||||
cd ~/rpmbuild/SPECS/; \
|
||||
. /opt/rh/devtoolset-9/enable; \
|
||||
rpmbuild -ba liburing.spec; \
|
||||
mkdir -p /root/packages/liburing-el7; \
|
||||
rm -rf /root/packages/liburing-el7/*; \
|
||||
cp ~/rpmbuild/RPMS/*/liburing* /root/packages/liburing-el7/; \
|
||||
cp ~/rpmbuild/SRPMS/liburing* /root/packages/liburing-el7/
|
||||
|
||||
RUN rpm -i `ls /root/packages/liburing-el7/liburing-*.x86_64.rpm | grep -v debug`
|
||||
|
||||
@@ -7,7 +7,6 @@ License: Vitastor Network Public License 1.1
|
||||
URL: https://vitastor.io/
|
||||
Source0: vitastor-2.2.3.el7.tar.gz
|
||||
|
||||
BuildRequires: liburing-devel >= 0.6
|
||||
BuildRequires: gperftools-devel
|
||||
BuildRequires: devtoolset-9-gcc-c++
|
||||
BuildRequires: rh-nodejs12
|
||||
@@ -35,8 +34,6 @@ size with configurable redundancy (replication or erasure codes/XOR).
|
||||
Summary: Vitastor - OSD
|
||||
Requires: libJerasure2
|
||||
Requires: libisa-l
|
||||
Requires: liburing >= 0.6
|
||||
Requires: liburing < 2
|
||||
Requires: vitastor-client = %{version}-%{release}
|
||||
Requires: util-linux
|
||||
Requires: parted
|
||||
@@ -60,8 +57,6 @@ scheduling cluster-level operations.
|
||||
|
||||
%package -n vitastor-client
|
||||
Summary: Vitastor - client
|
||||
Requires: liburing >= 0.6
|
||||
Requires: liburing < 2
|
||||
|
||||
|
||||
%description -n vitastor-client
|
||||
|
||||
@@ -17,17 +17,3 @@ RUN dnf -y install gcc-toolset-9 gcc-toolset-9-gcc-c++ gperftools-devel \
|
||||
RUN dnf download --source fio
|
||||
RUN rpm --nomd5 -i fio*.src.rpm
|
||||
RUN cd ~/rpmbuild/SPECS && dnf builddep -y --enablerepo=powertools --spec fio.spec
|
||||
|
||||
ADD https://vitastor.io/rpms/liburing-el7/liburing-0.7-2.el7.src.rpm /root
|
||||
|
||||
RUN set -e; \
|
||||
rpm -i liburing*.src.rpm; \
|
||||
cd ~/rpmbuild/SPECS/; \
|
||||
. /opt/rh/gcc-toolset-9/enable; \
|
||||
rpmbuild -ba liburing.spec; \
|
||||
mkdir -p /root/packages/liburing-el8; \
|
||||
rm -rf /root/packages/liburing-el8/*; \
|
||||
cp ~/rpmbuild/RPMS/*/liburing* /root/packages/liburing-el8/; \
|
||||
cp ~/rpmbuild/SRPMS/liburing* /root/packages/liburing-el8/
|
||||
|
||||
RUN rpm -i `ls /root/packages/liburing-el8/liburing-*.x86_64.rpm | grep -v debug`
|
||||
|
||||
@@ -7,7 +7,6 @@ License: Vitastor Network Public License 1.1
|
||||
URL: https://vitastor.io/
|
||||
Source0: vitastor-2.2.3.el8.tar.gz
|
||||
|
||||
BuildRequires: liburing-devel >= 0.6
|
||||
BuildRequires: gperftools-devel
|
||||
BuildRequires: gcc-toolset-9-gcc-c++
|
||||
BuildRequires: nodejs >= 10
|
||||
@@ -34,8 +33,6 @@ size with configurable redundancy (replication or erasure codes/XOR).
|
||||
Summary: Vitastor - OSD
|
||||
Requires: libJerasure2
|
||||
Requires: libisa-l
|
||||
Requires: liburing >= 0.6
|
||||
Requires: liburing < 2
|
||||
Requires: vitastor-client = %{version}-%{release}
|
||||
Requires: util-linux
|
||||
Requires: parted
|
||||
@@ -58,8 +55,6 @@ scheduling cluster-level operations.
|
||||
|
||||
%package -n vitastor-client
|
||||
Summary: Vitastor - client
|
||||
Requires: liburing >= 0.6
|
||||
Requires: liburing < 2
|
||||
|
||||
|
||||
%description -n vitastor-client
|
||||
|
||||
@@ -10,7 +10,7 @@ WORKDIR /root
|
||||
RUN sed -i 's/enabled=0/enabled=1/' /etc/yum.repos.d/*.repo
|
||||
RUN dnf -y install epel-release dnf-plugins-core
|
||||
RUN dnf -y install https://vitastor.io/rpms/centos/9/vitastor-release-1.0-1.el9.noarch.rpm
|
||||
RUN dnf -y install gcc-c++ gperftools-devel fio nodejs rpm-build jerasure-devel libisa-l-devel gf-complete-devel rdma-core-devel libarchive liburing-devel cmake libnl3-devel
|
||||
RUN dnf -y install gcc-c++ gperftools-devel fio nodejs rpm-build jerasure-devel libisa-l-devel gf-complete-devel rdma-core-devel libarchive cmake libnl3-devel
|
||||
RUN dnf download --source fio
|
||||
RUN rpm --nomd5 -i fio*.src.rpm
|
||||
RUN cd ~/rpmbuild/SPECS && dnf builddep -y --spec fio.spec
|
||||
|
||||
@@ -7,7 +7,6 @@ License: Vitastor Network Public License 1.1
|
||||
URL: https://vitastor.io/
|
||||
Source0: vitastor-2.2.3.el9.tar.gz
|
||||
|
||||
BuildRequires: liburing-devel >= 0.6
|
||||
BuildRequires: gperftools-devel
|
||||
BuildRequires: gcc-c++
|
||||
BuildRequires: nodejs >= 10
|
||||
|
||||
+11
-3
@@ -12,6 +12,7 @@ set(WITH_QEMU false CACHE BOOL "Build QEMU driver inside Vitastor source tree")
|
||||
set(WITH_FIO true CACHE BOOL "Build FIO driver")
|
||||
set(QEMU_PLUGINDIR qemu CACHE STRING "QEMU plugin directory suffix (qemu-kvm on RHEL)")
|
||||
set(WITH_ASAN false CACHE BOOL "Build with AddressSanitizer")
|
||||
set(WITH_SYSTEM_LIBURING false CACHE BOOL "Use system liburing")
|
||||
if("${CMAKE_INSTALL_PREFIX}" MATCHES "^/usr/local/?$")
|
||||
if(EXISTS "/etc/debian_version")
|
||||
set(CMAKE_INSTALL_LIBDIR "lib/${CMAKE_LIBRARY_ARCHITECTURE}")
|
||||
@@ -20,7 +21,7 @@ if("${CMAKE_INSTALL_PREFIX}" MATCHES "^/usr/local/?$")
|
||||
endif()
|
||||
|
||||
add_definitions(-DVITASTOR_VERSION="2.2.3")
|
||||
add_definitions(-D_LARGEFILE64_SOURCE -D_FILE_OFFSET_BITS=64 -Wall -Wno-sign-compare -Wno-comment -Wno-parentheses -Wno-pointer-arith -fdiagnostics-color=always -fno-omit-frame-pointer -I ${CMAKE_SOURCE_DIR}/src)
|
||||
add_definitions(-D_GNU_SOURCE -D_LARGEFILE64_SOURCE -D_FILE_OFFSET_BITS=64 -Wall -Wno-sign-compare -Wno-comment -Wno-parentheses -Wno-pointer-arith -fdiagnostics-color=always -fno-omit-frame-pointer -I ${CMAKE_SOURCE_DIR}/src)
|
||||
add_link_options(-fno-omit-frame-pointer)
|
||||
if (${WITH_ASAN})
|
||||
add_definitions(-fsanitize=address)
|
||||
@@ -49,7 +50,6 @@ endmacro(install_symlink)
|
||||
check_include_file("linux/nbd-netlink.h" HAVE_NBD_NETLINK_H)
|
||||
|
||||
find_package(PkgConfig)
|
||||
pkg_check_modules(LIBURING REQUIRED liburing)
|
||||
if (${WITH_QEMU})
|
||||
pkg_check_modules(GLIB REQUIRED glib-2.0)
|
||||
endif (${WITH_QEMU})
|
||||
@@ -66,6 +66,15 @@ if (RDMACM_LIBRARIES)
|
||||
add_definitions(-DWITH_RDMACM)
|
||||
endif (RDMACM_LIBRARIES)
|
||||
|
||||
if (${WITH_SYSTEM_LIBURING})
|
||||
pkg_check_modules(LIBURING REQUIRED liburing>=2.10)
|
||||
include_directories(${LIBURING_INCLUDE_DIRS})
|
||||
else()
|
||||
include_directories(${CMAKE_SOURCE_DIR}/src/liburing/include)
|
||||
add_subdirectory(liburing)
|
||||
set(LIBURING_LIBRARIES uring)
|
||||
endif (${WITH_SYSTEM_LIBURING})
|
||||
|
||||
add_custom_target(build_tests)
|
||||
add_custom_target(test
|
||||
COMMAND
|
||||
@@ -86,7 +95,6 @@ include_directories(
|
||||
${CMAKE_SOURCE_DIR}/src/test
|
||||
${CMAKE_SOURCE_DIR}/src/util
|
||||
/usr/include/jerasure
|
||||
${LIBURING_INCLUDE_DIRS}
|
||||
${IBVERBS_INCLUDE_DIRS}
|
||||
)
|
||||
|
||||
|
||||
@@ -5,7 +5,6 @@
|
||||
|
||||
#ifdef VITASTOR_SOURCE_TREE
|
||||
#define BUILD_DSO
|
||||
#define _GNU_SOURCE
|
||||
#endif
|
||||
#include "qemu/osdep.h"
|
||||
#include "qemu/main-loop.h"
|
||||
|
||||
@@ -0,0 +1,10 @@
|
||||
cmake_minimum_required(VERSION 2.8.12)
|
||||
|
||||
project(vitastor)
|
||||
|
||||
# liburing.a
|
||||
add_library(uring STATIC
|
||||
queue.c register.c setup.c syscall.c version.c
|
||||
)
|
||||
target_compile_options(uring PUBLIC -fPIC)
|
||||
target_compile_definitions(uring PUBLIC _GNU_SOURCE _LARGEFILE_SOURCE _FILE_OFFSET_BITS=64)
|
||||
@@ -0,0 +1,20 @@
|
||||
Copyright 2020 Jens Axboe
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a
|
||||
copy of this software and associated documentation files (the
|
||||
"Software"), to deal in the Software without restriction, including
|
||||
without limitation the rights to use, copy, modify, merge, publish,
|
||||
distribute, sublicense, and/or sell copies of the Software, and to
|
||||
permit persons to whom the Software is furnished to do so, subject to
|
||||
the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included
|
||||
in all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS
|
||||
OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT.
|
||||
IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY
|
||||
CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT,
|
||||
TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE
|
||||
SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
@@ -0,0 +1,25 @@
|
||||
liburing
|
||||
--------
|
||||
|
||||
This is a copy if liburing 2.11 from https://github.com/axboe/liburing/
|
||||
|
||||
For more info on io_uring, please see:
|
||||
|
||||
https://kernel.dk/io_uring.pdf
|
||||
|
||||
Subscribe to io-uring@vger.kernel.org for io_uring related discussions
|
||||
and development for both kernel and userspace. The list is archived here:
|
||||
|
||||
https://lore.kernel.org/io-uring/
|
||||
|
||||
License
|
||||
-------
|
||||
|
||||
"This repo" refers to the current directory as it's a copy. :-)
|
||||
|
||||
All software contained within this repo is dual licensed LGPL and MIT, see
|
||||
COPYING and LICENSE, except for a header coming from the kernel which is
|
||||
dual licensed GPL with a Linux-syscall-note exception and MIT, see
|
||||
COPYING.GPL and <https://spdx.org/licenses/Linux-syscall-note.html>.
|
||||
|
||||
Jens Axboe 2022-05-19
|
||||
@@ -0,0 +1,47 @@
|
||||
/* SPDX-License-Identifier: MIT */
|
||||
|
||||
#ifndef LIBURING_ARCH_AARCH64_LIB_H
|
||||
#define LIBURING_ARCH_AARCH64_LIB_H
|
||||
|
||||
#include <elf.h>
|
||||
#include "../../syscall.h"
|
||||
|
||||
static inline long __get_page_size(void)
|
||||
{
|
||||
Elf64_Off buf[2];
|
||||
long ret = 4096;
|
||||
int fd;
|
||||
|
||||
fd = __sys_open("/proc/self/auxv", O_RDONLY, 0);
|
||||
if (fd < 0)
|
||||
return ret;
|
||||
|
||||
while (1) {
|
||||
ssize_t x;
|
||||
|
||||
x = __sys_read(fd, buf, sizeof(buf));
|
||||
if (x < (long) sizeof(buf))
|
||||
break;
|
||||
|
||||
if (buf[0] == AT_PAGESZ) {
|
||||
ret = buf[1];
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
__sys_close(fd);
|
||||
return ret;
|
||||
}
|
||||
|
||||
static inline long get_page_size(void)
|
||||
{
|
||||
static long cache_val;
|
||||
|
||||
if (cache_val)
|
||||
return cache_val;
|
||||
|
||||
cache_val = __get_page_size();
|
||||
return cache_val;
|
||||
}
|
||||
|
||||
#endif /* #ifndef LIBURING_ARCH_AARCH64_LIB_H */
|
||||
@@ -0,0 +1,91 @@
|
||||
/* SPDX-License-Identifier: MIT */
|
||||
|
||||
#ifndef LIBURING_ARCH_AARCH64_SYSCALL_H
|
||||
#define LIBURING_ARCH_AARCH64_SYSCALL_H
|
||||
|
||||
#if defined(__aarch64__)
|
||||
|
||||
#define __do_syscallN(...) ({ \
|
||||
__asm__ volatile ( \
|
||||
"svc 0" \
|
||||
: "=r"(x0) \
|
||||
: __VA_ARGS__ \
|
||||
: "memory", "cc"); \
|
||||
(long) x0; \
|
||||
})
|
||||
|
||||
#define __do_syscall0(__n) ({ \
|
||||
register long x8 __asm__("x8") = __n; \
|
||||
register long x0 __asm__("x0"); \
|
||||
\
|
||||
__do_syscallN("r" (x8)); \
|
||||
})
|
||||
|
||||
#define __do_syscall1(__n, __a) ({ \
|
||||
register long x8 __asm__("x8") = __n; \
|
||||
register __typeof__(__a) x0 __asm__("x0") = __a; \
|
||||
\
|
||||
__do_syscallN("r" (x8), "0" (x0)); \
|
||||
})
|
||||
|
||||
#define __do_syscall2(__n, __a, __b) ({ \
|
||||
register long x8 __asm__("x8") = __n; \
|
||||
register __typeof__(__a) x0 __asm__("x0") = __a; \
|
||||
register __typeof__(__b) x1 __asm__("x1") = __b; \
|
||||
\
|
||||
__do_syscallN("r" (x8), "0" (x0), "r" (x1)); \
|
||||
})
|
||||
|
||||
#define __do_syscall3(__n, __a, __b, __c) ({ \
|
||||
register long x8 __asm__("x8") = __n; \
|
||||
register __typeof__(__a) x0 __asm__("x0") = __a; \
|
||||
register __typeof__(__b) x1 __asm__("x1") = __b; \
|
||||
register __typeof__(__c) x2 __asm__("x2") = __c; \
|
||||
\
|
||||
__do_syscallN("r" (x8), "0" (x0), "r" (x1), "r" (x2)); \
|
||||
})
|
||||
|
||||
#define __do_syscall4(__n, __a, __b, __c, __d) ({ \
|
||||
register long x8 __asm__("x8") = __n; \
|
||||
register __typeof__(__a) x0 __asm__("x0") = __a; \
|
||||
register __typeof__(__b) x1 __asm__("x1") = __b; \
|
||||
register __typeof__(__c) x2 __asm__("x2") = __c; \
|
||||
register __typeof__(__d) x3 __asm__("x3") = __d; \
|
||||
\
|
||||
__do_syscallN("r" (x8), "0" (x0), "r" (x1), "r" (x2), "r" (x3));\
|
||||
})
|
||||
|
||||
#define __do_syscall5(__n, __a, __b, __c, __d, __e) ({ \
|
||||
register long x8 __asm__("x8") = __n; \
|
||||
register __typeof__(__a) x0 __asm__("x0") = __a; \
|
||||
register __typeof__(__b) x1 __asm__("x1") = __b; \
|
||||
register __typeof__(__c) x2 __asm__("x2") = __c; \
|
||||
register __typeof__(__d) x3 __asm__("x3") = __d; \
|
||||
register __typeof__(__e) x4 __asm__("x4") = __e; \
|
||||
\
|
||||
__do_syscallN("r" (x8), "0" (x0), "r" (x1), "r" (x2), "r" (x3), \
|
||||
"r"(x4)); \
|
||||
})
|
||||
|
||||
#define __do_syscall6(__n, __a, __b, __c, __d, __e, __f) ({ \
|
||||
register long x8 __asm__("x8") = __n; \
|
||||
register __typeof__(__a) x0 __asm__("x0") = __a; \
|
||||
register __typeof__(__b) x1 __asm__("x1") = __b; \
|
||||
register __typeof__(__c) x2 __asm__("x2") = __c; \
|
||||
register __typeof__(__d) x3 __asm__("x3") = __d; \
|
||||
register __typeof__(__e) x4 __asm__("x4") = __e; \
|
||||
register __typeof__(__f) x5 __asm__("x5") = __f; \
|
||||
\
|
||||
__do_syscallN("r" (x8), "0" (x0), "r" (x1), "r" (x2), "r" (x3), \
|
||||
"r" (x4), "r"(x5)); \
|
||||
})
|
||||
|
||||
#include "../syscall-defs.h"
|
||||
|
||||
#else /* #if defined(__aarch64__) */
|
||||
|
||||
#include "../generic/syscall.h"
|
||||
|
||||
#endif /* #if defined(__aarch64__) */
|
||||
|
||||
#endif /* #ifndef LIBURING_ARCH_AARCH64_SYSCALL_H */
|
||||
@@ -0,0 +1,17 @@
|
||||
/* SPDX-License-Identifier: MIT */
|
||||
|
||||
#ifndef LIBURING_ARCH_GENERIC_LIB_H
|
||||
#define LIBURING_ARCH_GENERIC_LIB_H
|
||||
|
||||
static inline long get_page_size(void)
|
||||
{
|
||||
long page_size;
|
||||
|
||||
page_size = sysconf(_SC_PAGESIZE);
|
||||
if (page_size < 0)
|
||||
page_size = 4096;
|
||||
|
||||
return page_size;
|
||||
}
|
||||
|
||||
#endif /* #ifndef LIBURING_ARCH_GENERIC_LIB_H */
|
||||
@@ -0,0 +1,100 @@
|
||||
/* SPDX-License-Identifier: MIT */
|
||||
|
||||
#ifndef LIBURING_ARCH_GENERIC_SYSCALL_H
|
||||
#define LIBURING_ARCH_GENERIC_SYSCALL_H
|
||||
|
||||
#include <fcntl.h>
|
||||
|
||||
static inline int __sys_io_uring_register(unsigned int fd, unsigned int opcode,
|
||||
const void *arg, unsigned int nr_args)
|
||||
{
|
||||
int ret;
|
||||
ret = syscall(__NR_io_uring_register, fd, opcode, arg, nr_args);
|
||||
return (ret < 0) ? -errno : ret;
|
||||
}
|
||||
|
||||
static inline int __sys_io_uring_setup(unsigned int entries,
|
||||
struct io_uring_params *p)
|
||||
{
|
||||
int ret;
|
||||
ret = syscall(__NR_io_uring_setup, entries, p);
|
||||
return (ret < 0) ? -errno : ret;
|
||||
}
|
||||
|
||||
static inline int __sys_io_uring_enter2(unsigned int fd, unsigned int to_submit,
|
||||
unsigned int min_complete,
|
||||
unsigned int flags, void *arg,
|
||||
size_t sz)
|
||||
{
|
||||
int ret;
|
||||
ret = syscall(__NR_io_uring_enter, fd, to_submit, min_complete, flags,
|
||||
arg, sz);
|
||||
return (ret < 0) ? -errno : ret;
|
||||
}
|
||||
|
||||
static inline int __sys_io_uring_enter(unsigned int fd, unsigned int to_submit,
|
||||
unsigned int min_complete,
|
||||
unsigned int flags, sigset_t *sig)
|
||||
{
|
||||
return __sys_io_uring_enter2(fd, to_submit, min_complete, flags, sig,
|
||||
_NSIG / 8);
|
||||
}
|
||||
|
||||
static inline int __sys_open(const char *pathname, int flags, mode_t mode)
|
||||
{
|
||||
int ret;
|
||||
ret = open(pathname, flags, mode);
|
||||
return (ret < 0) ? -errno : ret;
|
||||
}
|
||||
|
||||
static inline ssize_t __sys_read(int fd, void *buffer, size_t size)
|
||||
{
|
||||
ssize_t ret;
|
||||
ret = read(fd, buffer, size);
|
||||
return (ret < 0) ? -errno : ret;
|
||||
}
|
||||
|
||||
static inline void *__sys_mmap(void *addr, size_t length, int prot, int flags,
|
||||
int fd, off_t offset)
|
||||
{
|
||||
void *ret;
|
||||
ret = mmap(addr, length, prot, flags, fd, offset);
|
||||
return (ret == MAP_FAILED) ? ERR_PTR(-errno) : ret;
|
||||
}
|
||||
|
||||
static inline int __sys_munmap(void *addr, size_t length)
|
||||
{
|
||||
int ret;
|
||||
ret = munmap(addr, length);
|
||||
return (ret < 0) ? -errno : ret;
|
||||
}
|
||||
|
||||
static inline int __sys_madvise(void *addr, size_t length, int advice)
|
||||
{
|
||||
int ret;
|
||||
ret = madvise(addr, length, advice);
|
||||
return (ret < 0) ? -errno : ret;
|
||||
}
|
||||
|
||||
static inline int __sys_getrlimit(int resource, struct rlimit *rlim)
|
||||
{
|
||||
int ret;
|
||||
ret = getrlimit(resource, rlim);
|
||||
return (ret < 0) ? -errno : ret;
|
||||
}
|
||||
|
||||
static inline int __sys_setrlimit(int resource, const struct rlimit *rlim)
|
||||
{
|
||||
int ret;
|
||||
ret = setrlimit(resource, rlim);
|
||||
return (ret < 0) ? -errno : ret;
|
||||
}
|
||||
|
||||
static inline int __sys_close(int fd)
|
||||
{
|
||||
int ret;
|
||||
ret = close(fd);
|
||||
return (ret < 0) ? -errno : ret;
|
||||
}
|
||||
|
||||
#endif /* #ifndef LIBURING_ARCH_GENERIC_SYSCALL_H */
|
||||
@@ -0,0 +1,48 @@
|
||||
/* SPDX-License-Identifier: MIT */
|
||||
|
||||
#ifndef LIBURING_ARCH_RISCV64_LIB_H
|
||||
#define LIBURING_ARCH_RISCV64_LIB_H
|
||||
|
||||
#include <elf.h>
|
||||
#include <sys/auxv.h>
|
||||
#include "../../syscall.h"
|
||||
|
||||
static inline long __get_page_size(void)
|
||||
{
|
||||
Elf64_Off buf[2];
|
||||
long ret = 4096;
|
||||
int fd;
|
||||
|
||||
fd = __sys_open("/proc/self/auxv", O_RDONLY, 0);
|
||||
if (fd < 0)
|
||||
return ret;
|
||||
|
||||
while (1) {
|
||||
ssize_t x;
|
||||
|
||||
x = __sys_read(fd, buf, sizeof(buf));
|
||||
if (x < (long) sizeof(buf))
|
||||
break;
|
||||
|
||||
if (buf[0] == AT_PAGESZ) {
|
||||
ret = buf[1];
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
__sys_close(fd);
|
||||
return ret;
|
||||
}
|
||||
|
||||
static inline long get_page_size(void)
|
||||
{
|
||||
static long cache_val;
|
||||
|
||||
if (cache_val)
|
||||
return cache_val;
|
||||
|
||||
cache_val = __get_page_size();
|
||||
return cache_val;
|
||||
}
|
||||
|
||||
#endif /* #ifndef LIBURING_ARCH_RISCV64_LIB_H */
|
||||
@@ -0,0 +1,100 @@
|
||||
/* SPDX-License-Identifier: MIT */
|
||||
|
||||
#ifndef LIBURING_ARCH_RISCV64_SYSCALL_H
|
||||
#define LIBURING_ARCH_RISCV64_SYSCALL_H
|
||||
|
||||
#if defined(__riscv) && __riscv_xlen == 64
|
||||
|
||||
#define __do_syscallM(...) ({ \
|
||||
__asm__ volatile ( \
|
||||
"ecall" \
|
||||
: "=r"(a0) \
|
||||
: __VA_ARGS__ \
|
||||
: "memory", "a1"); \
|
||||
(long) a0; \
|
||||
})
|
||||
|
||||
#define __do_syscallN(...) ({ \
|
||||
__asm__ volatile ( \
|
||||
"ecall" \
|
||||
: "=r"(a0) \
|
||||
: __VA_ARGS__ \
|
||||
: "memory"); \
|
||||
(long) a0; \
|
||||
})
|
||||
|
||||
#define __do_syscall0(__n) ({ \
|
||||
register long a7 __asm__("a7") = __n; \
|
||||
register long a0 __asm__("a0"); \
|
||||
\
|
||||
__do_syscallM("r" (a7)); \
|
||||
})
|
||||
|
||||
#define __do_syscall1(__n, __a) ({ \
|
||||
register long a7 __asm__("a7") = __n; \
|
||||
register __typeof__(__a) a0 __asm__("a0") = __a; \
|
||||
\
|
||||
__do_syscallM("r" (a7), "0" (a0)); \
|
||||
})
|
||||
|
||||
#define __do_syscall2(__n, __a, __b) ({ \
|
||||
register long a7 __asm__("a7") = __n; \
|
||||
register __typeof__(__a) a0 __asm__("a0") = __a; \
|
||||
register __typeof__(__b) a1 __asm__("a1") = __b; \
|
||||
\
|
||||
__do_syscallN("r" (a7), "0" (a0), "r" (a1)); \
|
||||
})
|
||||
|
||||
#define __do_syscall3(__n, __a, __b, __c) ({ \
|
||||
register long a7 __asm__("a7") = __n; \
|
||||
register __typeof__(__a) a0 __asm__("a0") = __a; \
|
||||
register __typeof__(__b) a1 __asm__("a1") = __b; \
|
||||
register __typeof__(__c) a2 __asm__("a2") = __c; \
|
||||
\
|
||||
__do_syscallN("r" (a7), "0" (a0), "r" (a1), "r" (a2)); \
|
||||
})
|
||||
|
||||
#define __do_syscall4(__n, __a, __b, __c, __d) ({ \
|
||||
register long a7 __asm__("a7") = __n; \
|
||||
register __typeof__(__a) a0 __asm__("a0") = __a; \
|
||||
register __typeof__(__b) a1 __asm__("a1") = __b; \
|
||||
register __typeof__(__c) a2 __asm__("a2") = __c; \
|
||||
register __typeof__(__d) a3 __asm__("a3") = __d; \
|
||||
\
|
||||
__do_syscallN("r" (a7), "0" (a0), "r" (a1), "r" (a2), "r" (a3));\
|
||||
})
|
||||
|
||||
#define __do_syscall5(__n, __a, __b, __c, __d, __e) ({ \
|
||||
register long a7 __asm__("a7") = __n; \
|
||||
register __typeof__(__a) a0 __asm__("a0") = __a; \
|
||||
register __typeof__(__b) a1 __asm__("a1") = __b; \
|
||||
register __typeof__(__c) a2 __asm__("a2") = __c; \
|
||||
register __typeof__(__d) a3 __asm__("a3") = __d; \
|
||||
register __typeof__(__e) a4 __asm__("a4") = __e; \
|
||||
\
|
||||
__do_syscallN("r" (a7), "0" (a0), "r" (a1), "r" (a2), "r" (a3), \
|
||||
"r"(a4)); \
|
||||
})
|
||||
|
||||
#define __do_syscall6(__n, __a, __b, __c, __d, __e, __f) ({ \
|
||||
register long a7 __asm__("a7") = __n; \
|
||||
register __typeof__(__a) a0 __asm__("a0") = __a; \
|
||||
register __typeof__(__b) a1 __asm__("a1") = __b; \
|
||||
register __typeof__(__c) a2 __asm__("a2") = __c; \
|
||||
register __typeof__(__d) a3 __asm__("a3") = __d; \
|
||||
register __typeof__(__e) a4 __asm__("a4") = __e; \
|
||||
register __typeof__(__f) a5 __asm__("a5") = __f; \
|
||||
\
|
||||
__do_syscallN("r" (a7), "0" (a0), "r" (a1), "r" (a2), "r" (a3), \
|
||||
"r" (a4), "r"(a5)); \
|
||||
})
|
||||
|
||||
#include "../syscall-defs.h"
|
||||
|
||||
#else /* #if defined(__riscv) && __riscv_xlen == 64 */
|
||||
|
||||
#include "../generic/syscall.h"
|
||||
|
||||
#endif /* #if defined(__riscv) && __riscv_xlen == 64 */
|
||||
|
||||
#endif /* #ifndef LIBURING_ARCH_RISCV64_SYSCALL_H */
|
||||
@@ -0,0 +1,94 @@
|
||||
/* SPDX-License-Identifier: MIT */
|
||||
|
||||
#ifndef LIBURING_ARCH_SYSCALL_DEFS_H
|
||||
#define LIBURING_ARCH_SYSCALL_DEFS_H
|
||||
|
||||
#include <fcntl.h>
|
||||
|
||||
static inline int __sys_open(const char *pathname, int flags, mode_t mode)
|
||||
{
|
||||
/*
|
||||
* Some architectures don't have __NR_open, but __NR_openat.
|
||||
*/
|
||||
#ifdef __NR_open
|
||||
return (int) __do_syscall3(__NR_open, pathname, flags, mode);
|
||||
#else
|
||||
return (int) __do_syscall4(__NR_openat, AT_FDCWD, pathname, flags, mode);
|
||||
#endif
|
||||
}
|
||||
|
||||
static inline ssize_t __sys_read(int fd, void *buffer, size_t size)
|
||||
{
|
||||
return (ssize_t) __do_syscall3(__NR_read, fd, buffer, size);
|
||||
}
|
||||
|
||||
static inline void *__sys_mmap(void *addr, size_t length, int prot, int flags,
|
||||
int fd, off_t offset)
|
||||
{
|
||||
int nr;
|
||||
|
||||
#if defined(__NR_mmap2)
|
||||
nr = __NR_mmap2;
|
||||
offset >>= 12;
|
||||
#else
|
||||
nr = __NR_mmap;
|
||||
#endif
|
||||
return (void *) __do_syscall6(nr, addr, length, prot, flags, fd, offset);
|
||||
}
|
||||
|
||||
static inline int __sys_munmap(void *addr, size_t length)
|
||||
{
|
||||
return (int) __do_syscall2(__NR_munmap, addr, length);
|
||||
}
|
||||
|
||||
static inline int __sys_madvise(void *addr, size_t length, int advice)
|
||||
{
|
||||
return (int) __do_syscall3(__NR_madvise, addr, length, advice);
|
||||
}
|
||||
|
||||
static inline int __sys_getrlimit(int resource, struct rlimit *rlim)
|
||||
{
|
||||
return (int) __do_syscall2(__NR_getrlimit, resource, rlim);
|
||||
}
|
||||
|
||||
static inline int __sys_setrlimit(int resource, const struct rlimit *rlim)
|
||||
{
|
||||
return (int) __do_syscall2(__NR_setrlimit, resource, rlim);
|
||||
}
|
||||
|
||||
static inline int __sys_close(int fd)
|
||||
{
|
||||
return (int) __do_syscall1(__NR_close, fd);
|
||||
}
|
||||
|
||||
static inline int __sys_io_uring_register(unsigned int fd, unsigned int opcode,
|
||||
const void *arg, unsigned int nr_args)
|
||||
{
|
||||
return (int) __do_syscall4(__NR_io_uring_register, fd, opcode, arg,
|
||||
nr_args);
|
||||
}
|
||||
|
||||
static inline int __sys_io_uring_setup(unsigned int entries,
|
||||
struct io_uring_params *p)
|
||||
{
|
||||
return (int) __do_syscall2(__NR_io_uring_setup, entries, p);
|
||||
}
|
||||
|
||||
static inline int __sys_io_uring_enter2(unsigned int fd, unsigned int to_submit,
|
||||
unsigned int min_complete,
|
||||
unsigned int flags, void *arg,
|
||||
size_t sz)
|
||||
{
|
||||
return (int) __do_syscall6(__NR_io_uring_enter, fd, to_submit,
|
||||
min_complete, flags, arg, sz);
|
||||
}
|
||||
|
||||
static inline int __sys_io_uring_enter(unsigned int fd, unsigned int to_submit,
|
||||
unsigned int min_complete,
|
||||
unsigned int flags, sigset_t *sig)
|
||||
{
|
||||
return __sys_io_uring_enter2(fd, to_submit, min_complete, flags, sig,
|
||||
_NSIG / 8);
|
||||
}
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1,11 @@
|
||||
/* SPDX-License-Identifier: MIT */
|
||||
|
||||
#ifndef LIBURING_ARCH_X86_LIB_H
|
||||
#define LIBURING_ARCH_X86_LIB_H
|
||||
|
||||
static inline long get_page_size(void)
|
||||
{
|
||||
return 4096;
|
||||
}
|
||||
|
||||
#endif /* #ifndef LIBURING_ARCH_X86_LIB_H */
|
||||
@@ -0,0 +1,296 @@
|
||||
/* SPDX-License-Identifier: MIT */
|
||||
|
||||
#ifndef LIBURING_ARCH_X86_SYSCALL_H
|
||||
#define LIBURING_ARCH_X86_SYSCALL_H
|
||||
|
||||
#if defined(__x86_64__)
|
||||
/**
|
||||
* Note for syscall registers usage (x86-64):
|
||||
* - %rax is the syscall number.
|
||||
* - %rax is also the return value.
|
||||
* - %rdi is the 1st argument.
|
||||
* - %rsi is the 2nd argument.
|
||||
* - %rdx is the 3rd argument.
|
||||
* - %r10 is the 4th argument (**yes it's %r10, not %rcx!**).
|
||||
* - %r8 is the 5th argument.
|
||||
* - %r9 is the 6th argument.
|
||||
*
|
||||
* `syscall` instruction will clobber %r11 and %rcx.
|
||||
*
|
||||
* After the syscall returns to userspace:
|
||||
* - %r11 will contain %rflags.
|
||||
* - %rcx will contain the return address.
|
||||
*
|
||||
* IOW, after the syscall returns to userspace:
|
||||
* %r11 == %rflags and %rcx == %rip.
|
||||
*/
|
||||
|
||||
#define __do_syscall0(NUM) ({ \
|
||||
intptr_t rax; \
|
||||
\
|
||||
__asm__ volatile( \
|
||||
"syscall" \
|
||||
: "=a"(rax) /* %rax */ \
|
||||
: "a"(NUM) /* %rax */ \
|
||||
: "rcx", "r11", "memory" \
|
||||
); \
|
||||
rax; \
|
||||
})
|
||||
|
||||
#define __do_syscall1(NUM, ARG1) ({ \
|
||||
intptr_t rax; \
|
||||
\
|
||||
__asm__ volatile( \
|
||||
"syscall" \
|
||||
: "=a"(rax) /* %rax */ \
|
||||
: "a"((NUM)), /* %rax */ \
|
||||
"D"((ARG1)) /* %rdi */ \
|
||||
: "rcx", "r11", "memory" \
|
||||
); \
|
||||
rax; \
|
||||
})
|
||||
|
||||
#define __do_syscall2(NUM, ARG1, ARG2) ({ \
|
||||
intptr_t rax; \
|
||||
\
|
||||
__asm__ volatile( \
|
||||
"syscall" \
|
||||
: "=a"(rax) /* %rax */ \
|
||||
: "a"((NUM)), /* %rax */ \
|
||||
"D"((ARG1)), /* %rdi */ \
|
||||
"S"((ARG2)) /* %rsi */ \
|
||||
: "rcx", "r11", "memory" \
|
||||
); \
|
||||
rax; \
|
||||
})
|
||||
|
||||
#define __do_syscall3(NUM, ARG1, ARG2, ARG3) ({ \
|
||||
intptr_t rax; \
|
||||
\
|
||||
__asm__ volatile( \
|
||||
"syscall" \
|
||||
: "=a"(rax) /* %rax */ \
|
||||
: "a"((NUM)), /* %rax */ \
|
||||
"D"((ARG1)), /* %rdi */ \
|
||||
"S"((ARG2)), /* %rsi */ \
|
||||
"d"((ARG3)) /* %rdx */ \
|
||||
: "rcx", "r11", "memory" \
|
||||
); \
|
||||
rax; \
|
||||
})
|
||||
|
||||
#define __do_syscall4(NUM, ARG1, ARG2, ARG3, ARG4) ({ \
|
||||
intptr_t rax; \
|
||||
register __typeof__(ARG4) __r10 __asm__("r10") = (ARG4); \
|
||||
\
|
||||
__asm__ volatile( \
|
||||
"syscall" \
|
||||
: "=a"(rax) /* %rax */ \
|
||||
: "a"((NUM)), /* %rax */ \
|
||||
"D"((ARG1)), /* %rdi */ \
|
||||
"S"((ARG2)), /* %rsi */ \
|
||||
"d"((ARG3)), /* %rdx */ \
|
||||
"r"(__r10) /* %r10 */ \
|
||||
: "rcx", "r11", "memory" \
|
||||
); \
|
||||
rax; \
|
||||
})
|
||||
|
||||
#define __do_syscall5(NUM, ARG1, ARG2, ARG3, ARG4, ARG5) ({ \
|
||||
intptr_t rax; \
|
||||
register __typeof__(ARG4) __r10 __asm__("r10") = (ARG4); \
|
||||
register __typeof__(ARG5) __r8 __asm__("r8") = (ARG5); \
|
||||
\
|
||||
__asm__ volatile( \
|
||||
"syscall" \
|
||||
: "=a"(rax) /* %rax */ \
|
||||
: "a"((NUM)), /* %rax */ \
|
||||
"D"((ARG1)), /* %rdi */ \
|
||||
"S"((ARG2)), /* %rsi */ \
|
||||
"d"((ARG3)), /* %rdx */ \
|
||||
"r"(__r10), /* %r10 */ \
|
||||
"r"(__r8) /* %r8 */ \
|
||||
: "rcx", "r11", "memory" \
|
||||
); \
|
||||
rax; \
|
||||
})
|
||||
|
||||
#define __do_syscall6(NUM, ARG1, ARG2, ARG3, ARG4, ARG5, ARG6) ({ \
|
||||
intptr_t rax; \
|
||||
register __typeof__(ARG4) __r10 __asm__("r10") = (ARG4); \
|
||||
register __typeof__(ARG5) __r8 __asm__("r8") = (ARG5); \
|
||||
register __typeof__(ARG6) __r9 __asm__("r9") = (ARG6); \
|
||||
\
|
||||
__asm__ volatile( \
|
||||
"syscall" \
|
||||
: "=a"(rax) /* %rax */ \
|
||||
: "a"((NUM)), /* %rax */ \
|
||||
"D"((ARG1)), /* %rdi */ \
|
||||
"S"((ARG2)), /* %rsi */ \
|
||||
"d"((ARG3)), /* %rdx */ \
|
||||
"r"(__r10), /* %r10 */ \
|
||||
"r"(__r8), /* %r8 */ \
|
||||
"r"(__r9) /* %r9 */ \
|
||||
: "rcx", "r11", "memory" \
|
||||
); \
|
||||
rax; \
|
||||
})
|
||||
|
||||
#include "../syscall-defs.h"
|
||||
|
||||
#else /* #if defined(__x86_64__) */
|
||||
|
||||
#ifdef CONFIG_NOLIBC
|
||||
/**
|
||||
* Note for syscall registers usage (x86, 32-bit):
|
||||
* - %eax is the syscall number.
|
||||
* - %eax is also the return value.
|
||||
* - %ebx is the 1st argument.
|
||||
* - %ecx is the 2nd argument.
|
||||
* - %edx is the 3rd argument.
|
||||
* - %esi is the 4th argument.
|
||||
* - %edi is the 5th argument.
|
||||
* - %ebp is the 6th argument.
|
||||
*/
|
||||
|
||||
#define __do_syscall0(NUM) ({ \
|
||||
intptr_t eax; \
|
||||
\
|
||||
__asm__ volatile( \
|
||||
"int $0x80" \
|
||||
: "=a"(eax) /* %eax */ \
|
||||
: "a"(NUM) /* %eax */ \
|
||||
: "memory" \
|
||||
); \
|
||||
eax; \
|
||||
})
|
||||
|
||||
#define __do_syscall1(NUM, ARG1) ({ \
|
||||
intptr_t eax; \
|
||||
\
|
||||
__asm__ volatile( \
|
||||
"int $0x80" \
|
||||
: "=a"(eax) /* %eax */ \
|
||||
: "a"(NUM), /* %eax */ \
|
||||
"b"((ARG1)) /* %ebx */ \
|
||||
: "memory" \
|
||||
); \
|
||||
eax; \
|
||||
})
|
||||
|
||||
#define __do_syscall2(NUM, ARG1, ARG2) ({ \
|
||||
intptr_t eax; \
|
||||
\
|
||||
__asm__ volatile( \
|
||||
"int $0x80" \
|
||||
: "=a" (eax) /* %eax */ \
|
||||
: "a"(NUM), /* %eax */ \
|
||||
"b"((ARG1)), /* %ebx */ \
|
||||
"c"((ARG2)) /* %ecx */ \
|
||||
: "memory" \
|
||||
); \
|
||||
eax; \
|
||||
})
|
||||
|
||||
#define __do_syscall3(NUM, ARG1, ARG2, ARG3) ({ \
|
||||
intptr_t eax; \
|
||||
\
|
||||
__asm__ volatile( \
|
||||
"int $0x80" \
|
||||
: "=a" (eax) /* %eax */ \
|
||||
: "a"(NUM), /* %eax */ \
|
||||
"b"((ARG1)), /* %ebx */ \
|
||||
"c"((ARG2)), /* %ecx */ \
|
||||
"d"((ARG3)) /* %edx */ \
|
||||
: "memory" \
|
||||
); \
|
||||
eax; \
|
||||
})
|
||||
|
||||
#define __do_syscall4(NUM, ARG1, ARG2, ARG3, ARG4) ({ \
|
||||
intptr_t eax; \
|
||||
\
|
||||
__asm__ volatile( \
|
||||
"int $0x80" \
|
||||
: "=a" (eax) /* %eax */ \
|
||||
: "a"(NUM), /* %eax */ \
|
||||
"b"((ARG1)), /* %ebx */ \
|
||||
"c"((ARG2)), /* %ecx */ \
|
||||
"d"((ARG3)), /* %edx */ \
|
||||
"S"((ARG4)) /* %esi */ \
|
||||
: "memory" \
|
||||
); \
|
||||
eax; \
|
||||
})
|
||||
|
||||
#define __do_syscall5(NUM, ARG1, ARG2, ARG3, ARG4, ARG5) ({ \
|
||||
intptr_t eax; \
|
||||
\
|
||||
__asm__ volatile( \
|
||||
"int $0x80" \
|
||||
: "=a" (eax) /* %eax */ \
|
||||
: "a"(NUM), /* %eax */ \
|
||||
"b"((ARG1)), /* %ebx */ \
|
||||
"c"((ARG2)), /* %ecx */ \
|
||||
"d"((ARG3)), /* %edx */ \
|
||||
"S"((ARG4)), /* %esi */ \
|
||||
"D"((ARG5)) /* %edi */ \
|
||||
: "memory" \
|
||||
); \
|
||||
eax; \
|
||||
})
|
||||
|
||||
|
||||
/*
|
||||
* On i386, the 6th argument of syscall goes in %ebp. However, both Clang
|
||||
* and GCC cannot use %ebp in the clobber list and in the "r" constraint
|
||||
* without using -fomit-frame-pointer. To make it always available for
|
||||
* any kind of compilation, the below workaround is implemented:
|
||||
*
|
||||
* 1) Push the 6-th argument.
|
||||
* 2) Push %ebp.
|
||||
* 3) Load the 6-th argument from 4(%esp) to %ebp.
|
||||
* 4) Do the syscall (int $0x80).
|
||||
* 5) Pop %ebp (restore the old value of %ebp).
|
||||
* 6) Add %esp by 4 (undo the stack pointer).
|
||||
*
|
||||
* WARNING:
|
||||
* Don't use register variables for __do_syscall6(), there is a known
|
||||
* GCC bug that results in an endless loop.
|
||||
*
|
||||
* BugLink: https://gcc.gnu.org/bugzilla/show_bug.cgi?id=105032
|
||||
*
|
||||
*/
|
||||
#define __do_syscall6(NUM, ARG1, ARG2, ARG3, ARG4, ARG5, ARG6) ({ \
|
||||
intptr_t eax = (intptr_t)(NUM); \
|
||||
intptr_t arg6 = (intptr_t)(ARG6); /* Always in memory */ \
|
||||
__asm__ volatile ( \
|
||||
"pushl %[_arg6]\n\t" \
|
||||
"pushl %%ebp\n\t" \
|
||||
"movl 4(%%esp),%%ebp\n\t" \
|
||||
"int $0x80\n\t" \
|
||||
"popl %%ebp\n\t" \
|
||||
"addl $4,%%esp" \
|
||||
: "+a"(eax) /* %eax */ \
|
||||
: "b"(ARG1), /* %ebx */ \
|
||||
"c"(ARG2), /* %ecx */ \
|
||||
"d"(ARG3), /* %edx */ \
|
||||
"S"(ARG4), /* %esi */ \
|
||||
"D"(ARG5), /* %edi */ \
|
||||
[_arg6]"m"(arg6) /* memory */ \
|
||||
: "memory", "cc" \
|
||||
); \
|
||||
eax; \
|
||||
})
|
||||
|
||||
#include "../syscall-defs.h"
|
||||
|
||||
#else /* #ifdef CONFIG_NOLIBC */
|
||||
|
||||
#include "../generic/syscall.h"
|
||||
|
||||
#endif /* #ifdef CONFIG_NOLIBC */
|
||||
|
||||
#endif /* #if defined(__x86_64__) */
|
||||
|
||||
#endif /* #ifndef LIBURING_ARCH_X86_SYSCALL_H */
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,81 @@
|
||||
/* SPDX-License-Identifier: MIT */
|
||||
#ifndef LIBURING_BARRIER_H
|
||||
#define LIBURING_BARRIER_H
|
||||
|
||||
/*
|
||||
From the kernel documentation file refcount-vs-atomic.rst:
|
||||
|
||||
A RELEASE memory ordering guarantees that all prior loads and
|
||||
stores (all po-earlier instructions) on the same CPU are completed
|
||||
before the operation. It also guarantees that all po-earlier
|
||||
stores on the same CPU and all propagated stores from other CPUs
|
||||
must propagate to all other CPUs before the release operation
|
||||
(A-cumulative property). This is implemented using
|
||||
:c:func:`smp_store_release`.
|
||||
|
||||
An ACQUIRE memory ordering guarantees that all post loads and
|
||||
stores (all po-later instructions) on the same CPU are
|
||||
completed after the acquire operation. It also guarantees that all
|
||||
po-later stores on the same CPU must propagate to all other CPUs
|
||||
after the acquire operation executes. This is implemented using
|
||||
:c:func:`smp_acquire__after_ctrl_dep`.
|
||||
*/
|
||||
|
||||
#ifdef __cplusplus
|
||||
#include <atomic>
|
||||
|
||||
template <typename T>
|
||||
static inline void IO_URING_WRITE_ONCE(T &var, T val)
|
||||
{
|
||||
std::atomic_store_explicit(reinterpret_cast<std::atomic<T> *>(&var),
|
||||
val, std::memory_order_relaxed);
|
||||
}
|
||||
template <typename T>
|
||||
static inline T IO_URING_READ_ONCE(const T &var)
|
||||
{
|
||||
return std::atomic_load_explicit(
|
||||
reinterpret_cast<const std::atomic<T> *>(&var),
|
||||
std::memory_order_relaxed);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
static inline void io_uring_smp_store_release(T *p, T v)
|
||||
{
|
||||
std::atomic_store_explicit(reinterpret_cast<std::atomic<T> *>(p), v,
|
||||
std::memory_order_release);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
static inline T io_uring_smp_load_acquire(const T *p)
|
||||
{
|
||||
return std::atomic_load_explicit(
|
||||
reinterpret_cast<const std::atomic<T> *>(p),
|
||||
std::memory_order_acquire);
|
||||
}
|
||||
|
||||
static inline void io_uring_smp_mb()
|
||||
{
|
||||
std::atomic_thread_fence(std::memory_order_seq_cst);
|
||||
}
|
||||
#else
|
||||
#include <stdatomic.h>
|
||||
|
||||
#define IO_URING_WRITE_ONCE(var, val) \
|
||||
atomic_store_explicit((_Atomic __typeof__(var) *)&(var), \
|
||||
(val), memory_order_relaxed)
|
||||
#define IO_URING_READ_ONCE(var) \
|
||||
atomic_load_explicit((_Atomic __typeof__(var) *)&(var), \
|
||||
memory_order_relaxed)
|
||||
|
||||
#define io_uring_smp_store_release(p, v) \
|
||||
atomic_store_explicit((_Atomic __typeof__(*(p)) *)(p), (v), \
|
||||
memory_order_release)
|
||||
#define io_uring_smp_load_acquire(p) \
|
||||
atomic_load_explicit((_Atomic __typeof__(*(p)) *)(p), \
|
||||
memory_order_acquire)
|
||||
|
||||
#define io_uring_smp_mb() \
|
||||
atomic_thread_fence(memory_order_seq_cst)
|
||||
#endif
|
||||
|
||||
#endif /* defined(LIBURING_BARRIER_H) */
|
||||
@@ -0,0 +1,13 @@
|
||||
/* SPDX-License-Identifier: MIT */
|
||||
#ifndef LIBURING_COMPAT_H
|
||||
#define LIBURING_COMPAT_H
|
||||
|
||||
#include <linux/time_types.h>
|
||||
/* <linux/time_types.h> is included above and not needed again */
|
||||
#define UAPI_LINUX_IO_URING_H_SKIP_LINUX_TIME_TYPES_H 1
|
||||
|
||||
#include <linux/openat2.h>
|
||||
|
||||
#include <linux/blkdev.h>
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1,979 @@
|
||||
/* SPDX-License-Identifier: (GPL-2.0 WITH Linux-syscall-note) OR MIT */
|
||||
/*
|
||||
* Header file for the io_uring interface.
|
||||
*
|
||||
* Copyright (C) 2019 Jens Axboe
|
||||
* Copyright (C) 2019 Christoph Hellwig
|
||||
*/
|
||||
#ifndef LINUX_IO_URING_H
|
||||
#define LINUX_IO_URING_H
|
||||
|
||||
#include <linux/fs.h>
|
||||
#include <linux/types.h>
|
||||
/*
|
||||
* this file is shared with liburing and that has to autodetect
|
||||
* if linux/time_types.h is available or not, it can
|
||||
* define UAPI_LINUX_IO_URING_H_SKIP_LINUX_TIME_TYPES_H
|
||||
* if linux/time_types.h is not available
|
||||
*/
|
||||
#ifndef UAPI_LINUX_IO_URING_H_SKIP_LINUX_TIME_TYPES_H
|
||||
#include <linux/time_types.h>
|
||||
#endif
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
/*
|
||||
* IO submission data structure (Submission Queue Entry)
|
||||
*/
|
||||
struct io_uring_sqe {
|
||||
__u8 opcode; /* type of operation for this sqe */
|
||||
__u8 flags; /* IOSQE_ flags */
|
||||
__u16 ioprio; /* ioprio for the request */
|
||||
__s32 fd; /* file descriptor to do IO on */
|
||||
union {
|
||||
__u64 off; /* offset into file */
|
||||
__u64 addr2;
|
||||
struct {
|
||||
__u32 cmd_op;
|
||||
__u32 __pad1;
|
||||
};
|
||||
};
|
||||
union {
|
||||
__u64 addr; /* pointer to buffer or iovecs */
|
||||
__u64 splice_off_in;
|
||||
struct {
|
||||
__u32 level;
|
||||
__u32 optname;
|
||||
};
|
||||
};
|
||||
__u32 len; /* buffer size or number of iovecs */
|
||||
union {
|
||||
__kernel_rwf_t rw_flags;
|
||||
__u32 fsync_flags;
|
||||
__u16 poll_events; /* compatibility */
|
||||
__u32 poll32_events; /* word-reversed for BE */
|
||||
__u32 sync_range_flags;
|
||||
__u32 msg_flags;
|
||||
__u32 timeout_flags;
|
||||
__u32 accept_flags;
|
||||
__u32 cancel_flags;
|
||||
__u32 open_flags;
|
||||
__u32 statx_flags;
|
||||
__u32 fadvise_advice;
|
||||
__u32 splice_flags;
|
||||
__u32 rename_flags;
|
||||
__u32 unlink_flags;
|
||||
__u32 hardlink_flags;
|
||||
__u32 xattr_flags;
|
||||
__u32 msg_ring_flags;
|
||||
__u32 uring_cmd_flags;
|
||||
__u32 waitid_flags;
|
||||
__u32 futex_flags;
|
||||
__u32 install_fd_flags;
|
||||
__u32 nop_flags;
|
||||
};
|
||||
__u64 user_data; /* data to be passed back at completion time */
|
||||
/* pack this to avoid bogus arm OABI complaints */
|
||||
union {
|
||||
/* index into fixed buffers, if used */
|
||||
__u16 buf_index;
|
||||
/* for grouped buffer selection */
|
||||
__u16 buf_group;
|
||||
} __attribute__((packed));
|
||||
/* personality to use, if used */
|
||||
__u16 personality;
|
||||
union {
|
||||
__s32 splice_fd_in;
|
||||
__u32 file_index;
|
||||
__u32 zcrx_ifq_idx;
|
||||
__u32 optlen;
|
||||
struct {
|
||||
__u16 addr_len;
|
||||
__u16 __pad3[1];
|
||||
};
|
||||
};
|
||||
union {
|
||||
struct {
|
||||
__u64 addr3;
|
||||
__u64 __pad2[1];
|
||||
};
|
||||
__u64 optval;
|
||||
/*
|
||||
* If the ring is initialized with IORING_SETUP_SQE128, then
|
||||
* this field is used for 80 bytes of arbitrary command data
|
||||
*/
|
||||
__u8 cmd[0];
|
||||
};
|
||||
};
|
||||
|
||||
/*
|
||||
* If sqe->file_index is set to this for opcodes that instantiate a new
|
||||
* direct descriptor (like openat/openat2/accept), then io_uring will allocate
|
||||
* an available direct descriptor instead of having the application pass one
|
||||
* in. The picked direct descriptor will be returned in cqe->res, or -ENFILE
|
||||
* if the space is full.
|
||||
*/
|
||||
#define IORING_FILE_INDEX_ALLOC (~0U)
|
||||
|
||||
enum io_uring_sqe_flags_bit {
|
||||
IOSQE_FIXED_FILE_BIT,
|
||||
IOSQE_IO_DRAIN_BIT,
|
||||
IOSQE_IO_LINK_BIT,
|
||||
IOSQE_IO_HARDLINK_BIT,
|
||||
IOSQE_ASYNC_BIT,
|
||||
IOSQE_BUFFER_SELECT_BIT,
|
||||
IOSQE_CQE_SKIP_SUCCESS_BIT,
|
||||
};
|
||||
|
||||
/*
|
||||
* sqe->flags
|
||||
*/
|
||||
/* use fixed fileset */
|
||||
#define IOSQE_FIXED_FILE (1U << IOSQE_FIXED_FILE_BIT)
|
||||
/* issue after inflight IO */
|
||||
#define IOSQE_IO_DRAIN (1U << IOSQE_IO_DRAIN_BIT)
|
||||
/* links next sqe */
|
||||
#define IOSQE_IO_LINK (1U << IOSQE_IO_LINK_BIT)
|
||||
/* like LINK, but stronger */
|
||||
#define IOSQE_IO_HARDLINK (1U << IOSQE_IO_HARDLINK_BIT)
|
||||
/* always go async */
|
||||
#define IOSQE_ASYNC (1U << IOSQE_ASYNC_BIT)
|
||||
/* select buffer from sqe->buf_group */
|
||||
#define IOSQE_BUFFER_SELECT (1U << IOSQE_BUFFER_SELECT_BIT)
|
||||
/* don't post CQE if request succeeded */
|
||||
#define IOSQE_CQE_SKIP_SUCCESS (1U << IOSQE_CQE_SKIP_SUCCESS_BIT)
|
||||
|
||||
/*
|
||||
* io_uring_setup() flags
|
||||
*/
|
||||
#define IORING_SETUP_IOPOLL (1U << 0) /* io_context is polled */
|
||||
#define IORING_SETUP_SQPOLL (1U << 1) /* SQ poll thread */
|
||||
#define IORING_SETUP_SQ_AFF (1U << 2) /* sq_thread_cpu is valid */
|
||||
#define IORING_SETUP_CQSIZE (1U << 3) /* app defines CQ size */
|
||||
#define IORING_SETUP_CLAMP (1U << 4) /* clamp SQ/CQ ring sizes */
|
||||
#define IORING_SETUP_ATTACH_WQ (1U << 5) /* attach to existing wq */
|
||||
#define IORING_SETUP_R_DISABLED (1U << 6) /* start with ring disabled */
|
||||
#define IORING_SETUP_SUBMIT_ALL (1U << 7) /* continue submit on error */
|
||||
/*
|
||||
* Cooperative task running. When requests complete, they often require
|
||||
* forcing the submitter to transition to the kernel to complete. If this
|
||||
* flag is set, work will be done when the task transitions anyway, rather
|
||||
* than force an inter-processor interrupt reschedule. This avoids interrupting
|
||||
* a task running in userspace, and saves an IPI.
|
||||
*/
|
||||
#define IORING_SETUP_COOP_TASKRUN (1U << 8)
|
||||
/*
|
||||
* If COOP_TASKRUN is set, get notified if task work is available for
|
||||
* running and a kernel transition would be needed to run it. This sets
|
||||
* IORING_SQ_TASKRUN in the sq ring flags. Not valid with COOP_TASKRUN.
|
||||
*/
|
||||
#define IORING_SETUP_TASKRUN_FLAG (1U << 9)
|
||||
#define IORING_SETUP_SQE128 (1U << 10) /* SQEs are 128 byte */
|
||||
#define IORING_SETUP_CQE32 (1U << 11) /* CQEs are 32 byte */
|
||||
/*
|
||||
* Only one task is allowed to submit requests
|
||||
*/
|
||||
#define IORING_SETUP_SINGLE_ISSUER (1U << 12)
|
||||
|
||||
/*
|
||||
* Defer running task work to get events.
|
||||
* Rather than running bits of task work whenever the task transitions
|
||||
* try to do it just before it is needed.
|
||||
*/
|
||||
#define IORING_SETUP_DEFER_TASKRUN (1U << 13)
|
||||
|
||||
/*
|
||||
* Application provides the memory for the rings
|
||||
*/
|
||||
#define IORING_SETUP_NO_MMAP (1U << 14)
|
||||
|
||||
/*
|
||||
* Register the ring fd in itself for use with
|
||||
* IORING_REGISTER_USE_REGISTERED_RING; return a registered fd index rather
|
||||
* than an fd.
|
||||
*/
|
||||
#define IORING_SETUP_REGISTERED_FD_ONLY (1U << 15)
|
||||
|
||||
/*
|
||||
* Removes indirection through the SQ index array.
|
||||
*/
|
||||
#define IORING_SETUP_NO_SQARRAY (1U << 16)
|
||||
|
||||
/* Use hybrid poll in iopoll process */
|
||||
#define IORING_SETUP_HYBRID_IOPOLL (1U << 17)
|
||||
|
||||
enum io_uring_op {
|
||||
IORING_OP_NOP,
|
||||
IORING_OP_READV,
|
||||
IORING_OP_WRITEV,
|
||||
IORING_OP_FSYNC,
|
||||
IORING_OP_READ_FIXED,
|
||||
IORING_OP_WRITE_FIXED,
|
||||
IORING_OP_POLL_ADD,
|
||||
IORING_OP_POLL_REMOVE,
|
||||
IORING_OP_SYNC_FILE_RANGE,
|
||||
IORING_OP_SENDMSG,
|
||||
IORING_OP_RECVMSG,
|
||||
IORING_OP_TIMEOUT,
|
||||
IORING_OP_TIMEOUT_REMOVE,
|
||||
IORING_OP_ACCEPT,
|
||||
IORING_OP_ASYNC_CANCEL,
|
||||
IORING_OP_LINK_TIMEOUT,
|
||||
IORING_OP_CONNECT,
|
||||
IORING_OP_FALLOCATE,
|
||||
IORING_OP_OPENAT,
|
||||
IORING_OP_CLOSE,
|
||||
IORING_OP_FILES_UPDATE,
|
||||
IORING_OP_STATX,
|
||||
IORING_OP_READ,
|
||||
IORING_OP_WRITE,
|
||||
IORING_OP_FADVISE,
|
||||
IORING_OP_MADVISE,
|
||||
IORING_OP_SEND,
|
||||
IORING_OP_RECV,
|
||||
IORING_OP_OPENAT2,
|
||||
IORING_OP_EPOLL_CTL,
|
||||
IORING_OP_SPLICE,
|
||||
IORING_OP_PROVIDE_BUFFERS,
|
||||
IORING_OP_REMOVE_BUFFERS,
|
||||
IORING_OP_TEE,
|
||||
IORING_OP_SHUTDOWN,
|
||||
IORING_OP_RENAMEAT,
|
||||
IORING_OP_UNLINKAT,
|
||||
IORING_OP_MKDIRAT,
|
||||
IORING_OP_SYMLINKAT,
|
||||
IORING_OP_LINKAT,
|
||||
IORING_OP_MSG_RING,
|
||||
IORING_OP_FSETXATTR,
|
||||
IORING_OP_SETXATTR,
|
||||
IORING_OP_FGETXATTR,
|
||||
IORING_OP_GETXATTR,
|
||||
IORING_OP_SOCKET,
|
||||
IORING_OP_URING_CMD,
|
||||
IORING_OP_SEND_ZC,
|
||||
IORING_OP_SENDMSG_ZC,
|
||||
IORING_OP_READ_MULTISHOT,
|
||||
IORING_OP_WAITID,
|
||||
IORING_OP_FUTEX_WAIT,
|
||||
IORING_OP_FUTEX_WAKE,
|
||||
IORING_OP_FUTEX_WAITV,
|
||||
IORING_OP_FIXED_FD_INSTALL,
|
||||
IORING_OP_FTRUNCATE,
|
||||
IORING_OP_BIND,
|
||||
IORING_OP_LISTEN,
|
||||
IORING_OP_RECV_ZC,
|
||||
IORING_OP_EPOLL_WAIT,
|
||||
IORING_OP_READV_FIXED,
|
||||
IORING_OP_WRITEV_FIXED,
|
||||
|
||||
/* this goes last, obviously */
|
||||
IORING_OP_LAST,
|
||||
};
|
||||
|
||||
/*
|
||||
* sqe->uring_cmd_flags top 8bits aren't available for userspace
|
||||
* IORING_URING_CMD_FIXED use registered buffer; pass this flag
|
||||
* along with setting sqe->buf_index.
|
||||
*/
|
||||
#define IORING_URING_CMD_FIXED (1U << 0)
|
||||
#define IORING_URING_CMD_MASK IORING_URING_CMD_FIXED
|
||||
|
||||
|
||||
/*
|
||||
* sqe->fsync_flags
|
||||
*/
|
||||
#define IORING_FSYNC_DATASYNC (1U << 0)
|
||||
|
||||
/*
|
||||
* sqe->timeout_flags
|
||||
*/
|
||||
#define IORING_TIMEOUT_ABS (1U << 0)
|
||||
#define IORING_TIMEOUT_UPDATE (1U << 1)
|
||||
#define IORING_TIMEOUT_BOOTTIME (1U << 2)
|
||||
#define IORING_TIMEOUT_REALTIME (1U << 3)
|
||||
#define IORING_LINK_TIMEOUT_UPDATE (1U << 4)
|
||||
#define IORING_TIMEOUT_ETIME_SUCCESS (1U << 5)
|
||||
#define IORING_TIMEOUT_MULTISHOT (1U << 6)
|
||||
#define IORING_TIMEOUT_CLOCK_MASK (IORING_TIMEOUT_BOOTTIME | IORING_TIMEOUT_REALTIME)
|
||||
#define IORING_TIMEOUT_UPDATE_MASK (IORING_TIMEOUT_UPDATE | IORING_LINK_TIMEOUT_UPDATE)
|
||||
/*
|
||||
* sqe->splice_flags
|
||||
* extends splice(2) flags
|
||||
*/
|
||||
#define SPLICE_F_FD_IN_FIXED (1U << 31) /* the last bit of __u32 */
|
||||
|
||||
/*
|
||||
* POLL_ADD flags. Note that since sqe->poll_events is the flag space, the
|
||||
* command flags for POLL_ADD are stored in sqe->len.
|
||||
*
|
||||
* IORING_POLL_ADD_MULTI Multishot poll. Sets IORING_CQE_F_MORE if
|
||||
* the poll handler will continue to report
|
||||
* CQEs on behalf of the same SQE.
|
||||
*
|
||||
* IORING_POLL_UPDATE Update existing poll request, matching
|
||||
* sqe->addr as the old user_data field.
|
||||
*
|
||||
* IORING_POLL_LEVEL Level triggered poll.
|
||||
*/
|
||||
#define IORING_POLL_ADD_MULTI (1U << 0)
|
||||
#define IORING_POLL_UPDATE_EVENTS (1U << 1)
|
||||
#define IORING_POLL_UPDATE_USER_DATA (1U << 2)
|
||||
#define IORING_POLL_ADD_LEVEL (1U << 3)
|
||||
|
||||
/*
|
||||
* ASYNC_CANCEL flags.
|
||||
*
|
||||
* IORING_ASYNC_CANCEL_ALL Cancel all requests that match the given key
|
||||
* IORING_ASYNC_CANCEL_FD Key off 'fd' for cancelation rather than the
|
||||
* request 'user_data'
|
||||
* IORING_ASYNC_CANCEL_ANY Match any request
|
||||
* IORING_ASYNC_CANCEL_FD_FIXED 'fd' passed in is a fixed descriptor
|
||||
* IORING_ASYNC_CANCEL_USERDATA Match on user_data, default for no other key
|
||||
* IORING_ASYNC_CANCEL_OP Match request based on opcode
|
||||
*/
|
||||
#define IORING_ASYNC_CANCEL_ALL (1U << 0)
|
||||
#define IORING_ASYNC_CANCEL_FD (1U << 1)
|
||||
#define IORING_ASYNC_CANCEL_ANY (1U << 2)
|
||||
#define IORING_ASYNC_CANCEL_FD_FIXED (1U << 3)
|
||||
#define IORING_ASYNC_CANCEL_USERDATA (1U << 4)
|
||||
#define IORING_ASYNC_CANCEL_OP (1U << 5)
|
||||
|
||||
/*
|
||||
* send/sendmsg and recv/recvmsg flags (sqe->ioprio)
|
||||
*
|
||||
* IORING_RECVSEND_POLL_FIRST If set, instead of first attempting to send
|
||||
* or receive and arm poll if that yields an
|
||||
* -EAGAIN result, arm poll upfront and skip
|
||||
* the initial transfer attempt.
|
||||
*
|
||||
* IORING_RECV_MULTISHOT Multishot recv. Sets IORING_CQE_F_MORE if
|
||||
* the handler will continue to report
|
||||
* CQEs on behalf of the same SQE.
|
||||
*
|
||||
* IORING_RECVSEND_FIXED_BUF Use registered buffers, the index is stored in
|
||||
* the buf_index field.
|
||||
*
|
||||
* IORING_SEND_ZC_REPORT_USAGE
|
||||
* If set, SEND[MSG]_ZC should report
|
||||
* the zerocopy usage in cqe.res
|
||||
* for the IORING_CQE_F_NOTIF cqe.
|
||||
* 0 is reported if zerocopy was actually possible.
|
||||
* IORING_NOTIF_USAGE_ZC_COPIED if data was copied
|
||||
* (at least partially).
|
||||
*
|
||||
* IORING_RECVSEND_BUNDLE Used with IOSQE_BUFFER_SELECT. If set, send or
|
||||
* recv will grab as many buffers from the buffer
|
||||
* group ID given and send them all. The completion
|
||||
* result will be the number of buffers send, with
|
||||
* the starting buffer ID in cqe->flags as per
|
||||
* usual for provided buffer usage. The buffers
|
||||
* will be contiguous from the starting buffer ID.
|
||||
*/
|
||||
#define IORING_RECVSEND_POLL_FIRST (1U << 0)
|
||||
#define IORING_RECV_MULTISHOT (1U << 1)
|
||||
#define IORING_RECVSEND_FIXED_BUF (1U << 2)
|
||||
#define IORING_SEND_ZC_REPORT_USAGE (1U << 3)
|
||||
#define IORING_RECVSEND_BUNDLE (1U << 4)
|
||||
|
||||
/*
|
||||
* cqe.res for IORING_CQE_F_NOTIF if
|
||||
* IORING_SEND_ZC_REPORT_USAGE was requested
|
||||
*
|
||||
* It should be treated as a flag, all other
|
||||
* bits of cqe.res should be treated as reserved!
|
||||
*/
|
||||
#define IORING_NOTIF_USAGE_ZC_COPIED (1U << 31)
|
||||
|
||||
/*
|
||||
* accept flags stored in sqe->ioprio
|
||||
*/
|
||||
#define IORING_ACCEPT_MULTISHOT (1U << 0)
|
||||
#define IORING_ACCEPT_DONTWAIT (1U << 1)
|
||||
#define IORING_ACCEPT_POLL_FIRST (1U << 2)
|
||||
|
||||
/*
|
||||
* IORING_OP_MSG_RING command types, stored in sqe->addr
|
||||
*/
|
||||
enum io_uring_msg_ring_flags {
|
||||
IORING_MSG_DATA, /* pass sqe->len as 'res' and off as user_data */
|
||||
IORING_MSG_SEND_FD, /* send a registered fd to another ring */
|
||||
};
|
||||
|
||||
/*
|
||||
* IORING_OP_MSG_RING flags (sqe->msg_ring_flags)
|
||||
*
|
||||
* IORING_MSG_RING_CQE_SKIP Don't post a CQE to the target ring. Not
|
||||
* applicable for IORING_MSG_DATA, obviously.
|
||||
*/
|
||||
#define IORING_MSG_RING_CQE_SKIP (1U << 0)
|
||||
/* Pass through the flags from sqe->file_index to cqe->flags */
|
||||
#define IORING_MSG_RING_FLAGS_PASS (1U << 1)
|
||||
|
||||
/*
|
||||
* IORING_OP_FIXED_FD_INSTALL flags (sqe->install_fd_flags)
|
||||
*
|
||||
* IORING_FIXED_FD_NO_CLOEXEC Don't mark the fd as O_CLOEXEC
|
||||
*/
|
||||
#define IORING_FIXED_FD_NO_CLOEXEC (1U << 0)
|
||||
|
||||
/*
|
||||
* IORING_OP_NOP flags (sqe->nop_flags)
|
||||
*
|
||||
* IORING_NOP_INJECT_RESULT Inject result from sqe->result
|
||||
*/
|
||||
#define IORING_NOP_INJECT_RESULT (1U << 0)
|
||||
|
||||
/*
|
||||
* IO completion data structure (Completion Queue Entry)
|
||||
*/
|
||||
struct io_uring_cqe {
|
||||
__u64 user_data; /* sqe->user_data value passed back */
|
||||
__s32 res; /* result code for this event */
|
||||
__u32 flags;
|
||||
|
||||
/*
|
||||
* If the ring is initialized with IORING_SETUP_CQE32, then this field
|
||||
* contains 16-bytes of padding, doubling the size of the CQE.
|
||||
*/
|
||||
__u64 big_cqe[];
|
||||
};
|
||||
|
||||
/*
|
||||
* cqe->flags
|
||||
*
|
||||
* IORING_CQE_F_BUFFER If set, the upper 16 bits are the buffer ID
|
||||
* IORING_CQE_F_MORE If set, parent SQE will generate more CQE entries
|
||||
* IORING_CQE_F_SOCK_NONEMPTY If set, more data to read after socket recv
|
||||
* IORING_CQE_F_NOTIF Set for notification CQEs. Can be used to distinct
|
||||
* them from sends.
|
||||
* IORING_CQE_F_BUF_MORE If set, the buffer ID set in the completion will get
|
||||
* more completions. In other words, the buffer is being
|
||||
* partially consumed, and will be used by the kernel for
|
||||
* more completions. This is only set for buffers used via
|
||||
* the incremental buffer consumption, as provided by
|
||||
* a ring buffer setup with IOU_PBUF_RING_INC. For any
|
||||
* other provided buffer type, all completions with a
|
||||
* buffer passed back is automatically returned to the
|
||||
* application.
|
||||
*/
|
||||
#define IORING_CQE_F_BUFFER (1U << 0)
|
||||
#define IORING_CQE_F_MORE (1U << 1)
|
||||
#define IORING_CQE_F_SOCK_NONEMPTY (1U << 2)
|
||||
#define IORING_CQE_F_NOTIF (1U << 3)
|
||||
#define IORING_CQE_F_BUF_MORE (1U << 4)
|
||||
|
||||
#define IORING_CQE_BUFFER_SHIFT 16
|
||||
|
||||
/*
|
||||
* Magic offsets for the application to mmap the data it needs
|
||||
*/
|
||||
#define IORING_OFF_SQ_RING 0ULL
|
||||
#define IORING_OFF_CQ_RING 0x8000000ULL
|
||||
#define IORING_OFF_SQES 0x10000000ULL
|
||||
#define IORING_OFF_PBUF_RING 0x80000000ULL
|
||||
#define IORING_OFF_PBUF_SHIFT 16
|
||||
#define IORING_OFF_MMAP_MASK 0xf8000000ULL
|
||||
|
||||
/*
|
||||
* Filled with the offset for mmap(2)
|
||||
*/
|
||||
struct io_sqring_offsets {
|
||||
__u32 head;
|
||||
__u32 tail;
|
||||
__u32 ring_mask;
|
||||
__u32 ring_entries;
|
||||
__u32 flags;
|
||||
__u32 dropped;
|
||||
__u32 array;
|
||||
__u32 resv1;
|
||||
__u64 user_addr;
|
||||
};
|
||||
|
||||
/*
|
||||
* sq_ring->flags
|
||||
*/
|
||||
#define IORING_SQ_NEED_WAKEUP (1U << 0) /* needs io_uring_enter wakeup */
|
||||
#define IORING_SQ_CQ_OVERFLOW (1U << 1) /* CQ ring is overflown */
|
||||
#define IORING_SQ_TASKRUN (1U << 2) /* task should enter the kernel */
|
||||
|
||||
struct io_cqring_offsets {
|
||||
__u32 head;
|
||||
__u32 tail;
|
||||
__u32 ring_mask;
|
||||
__u32 ring_entries;
|
||||
__u32 overflow;
|
||||
__u32 cqes;
|
||||
__u32 flags;
|
||||
__u32 resv1;
|
||||
__u64 user_addr;
|
||||
};
|
||||
|
||||
/*
|
||||
* cq_ring->flags
|
||||
*/
|
||||
|
||||
/* disable eventfd notifications */
|
||||
#define IORING_CQ_EVENTFD_DISABLED (1U << 0)
|
||||
|
||||
/*
|
||||
* io_uring_enter(2) flags
|
||||
*/
|
||||
#define IORING_ENTER_GETEVENTS (1U << 0)
|
||||
#define IORING_ENTER_SQ_WAKEUP (1U << 1)
|
||||
#define IORING_ENTER_SQ_WAIT (1U << 2)
|
||||
#define IORING_ENTER_EXT_ARG (1U << 3)
|
||||
#define IORING_ENTER_REGISTERED_RING (1U << 4)
|
||||
#define IORING_ENTER_ABS_TIMER (1U << 5)
|
||||
#define IORING_ENTER_EXT_ARG_REG (1U << 6)
|
||||
#define IORING_ENTER_NO_IOWAIT (1U << 7)
|
||||
|
||||
/*
|
||||
* Passed in for io_uring_setup(2). Copied back with updated info on success
|
||||
*/
|
||||
struct io_uring_params {
|
||||
__u32 sq_entries;
|
||||
__u32 cq_entries;
|
||||
__u32 flags;
|
||||
__u32 sq_thread_cpu;
|
||||
__u32 sq_thread_idle;
|
||||
__u32 features;
|
||||
__u32 wq_fd;
|
||||
__u32 resv[3];
|
||||
struct io_sqring_offsets sq_off;
|
||||
struct io_cqring_offsets cq_off;
|
||||
};
|
||||
|
||||
/*
|
||||
* io_uring_params->features flags
|
||||
*/
|
||||
#define IORING_FEAT_SINGLE_MMAP (1U << 0)
|
||||
#define IORING_FEAT_NODROP (1U << 1)
|
||||
#define IORING_FEAT_SUBMIT_STABLE (1U << 2)
|
||||
#define IORING_FEAT_RW_CUR_POS (1U << 3)
|
||||
#define IORING_FEAT_CUR_PERSONALITY (1U << 4)
|
||||
#define IORING_FEAT_FAST_POLL (1U << 5)
|
||||
#define IORING_FEAT_POLL_32BITS (1U << 6)
|
||||
#define IORING_FEAT_SQPOLL_NONFIXED (1U << 7)
|
||||
#define IORING_FEAT_EXT_ARG (1U << 8)
|
||||
#define IORING_FEAT_NATIVE_WORKERS (1U << 9)
|
||||
#define IORING_FEAT_RSRC_TAGS (1U << 10)
|
||||
#define IORING_FEAT_CQE_SKIP (1U << 11)
|
||||
#define IORING_FEAT_LINKED_FILE (1U << 12)
|
||||
#define IORING_FEAT_REG_REG_RING (1U << 13)
|
||||
#define IORING_FEAT_RECVSEND_BUNDLE (1U << 14)
|
||||
#define IORING_FEAT_MIN_TIMEOUT (1U << 15)
|
||||
#define IORING_FEAT_RW_ATTR (1U << 16)
|
||||
#define IORING_FEAT_NO_IOWAIT (1U << 17)
|
||||
|
||||
/*
|
||||
* io_uring_register(2) opcodes and arguments
|
||||
*/
|
||||
enum io_uring_register_op {
|
||||
IORING_REGISTER_BUFFERS = 0,
|
||||
IORING_UNREGISTER_BUFFERS = 1,
|
||||
IORING_REGISTER_FILES = 2,
|
||||
IORING_UNREGISTER_FILES = 3,
|
||||
IORING_REGISTER_EVENTFD = 4,
|
||||
IORING_UNREGISTER_EVENTFD = 5,
|
||||
IORING_REGISTER_FILES_UPDATE = 6,
|
||||
IORING_REGISTER_EVENTFD_ASYNC = 7,
|
||||
IORING_REGISTER_PROBE = 8,
|
||||
IORING_REGISTER_PERSONALITY = 9,
|
||||
IORING_UNREGISTER_PERSONALITY = 10,
|
||||
IORING_REGISTER_RESTRICTIONS = 11,
|
||||
IORING_REGISTER_ENABLE_RINGS = 12,
|
||||
|
||||
/* extended with tagging */
|
||||
IORING_REGISTER_FILES2 = 13,
|
||||
IORING_REGISTER_FILES_UPDATE2 = 14,
|
||||
IORING_REGISTER_BUFFERS2 = 15,
|
||||
IORING_REGISTER_BUFFERS_UPDATE = 16,
|
||||
|
||||
/* set/clear io-wq thread affinities */
|
||||
IORING_REGISTER_IOWQ_AFF = 17,
|
||||
IORING_UNREGISTER_IOWQ_AFF = 18,
|
||||
|
||||
/* set/get max number of io-wq workers */
|
||||
IORING_REGISTER_IOWQ_MAX_WORKERS = 19,
|
||||
|
||||
/* register/unregister io_uring fd with the ring */
|
||||
IORING_REGISTER_RING_FDS = 20,
|
||||
IORING_UNREGISTER_RING_FDS = 21,
|
||||
|
||||
/* register ring based provide buffer group */
|
||||
IORING_REGISTER_PBUF_RING = 22,
|
||||
IORING_UNREGISTER_PBUF_RING = 23,
|
||||
|
||||
/* sync cancelation API */
|
||||
IORING_REGISTER_SYNC_CANCEL = 24,
|
||||
|
||||
/* register a range of fixed file slots for automatic slot allocation */
|
||||
IORING_REGISTER_FILE_ALLOC_RANGE = 25,
|
||||
|
||||
/* return status information for a buffer group */
|
||||
IORING_REGISTER_PBUF_STATUS = 26,
|
||||
|
||||
/* set/clear busy poll settings */
|
||||
IORING_REGISTER_NAPI = 27,
|
||||
IORING_UNREGISTER_NAPI = 28,
|
||||
|
||||
IORING_REGISTER_CLOCK = 29,
|
||||
|
||||
/* clone registered buffers from source ring to current ring */
|
||||
IORING_REGISTER_CLONE_BUFFERS = 30,
|
||||
|
||||
/* send MSG_RING without having a ring */
|
||||
IORING_REGISTER_SEND_MSG_RING = 31,
|
||||
|
||||
/* register a netdev hw rx queue for zerocopy */
|
||||
IORING_REGISTER_ZCRX_IFQ = 32,
|
||||
|
||||
/* resize CQ ring */
|
||||
IORING_REGISTER_RESIZE_RINGS = 33,
|
||||
|
||||
IORING_REGISTER_MEM_REGION = 34,
|
||||
|
||||
/* this goes last */
|
||||
IORING_REGISTER_LAST,
|
||||
|
||||
/* flag added to the opcode to use a registered ring fd */
|
||||
IORING_REGISTER_USE_REGISTERED_RING = 1U << 31
|
||||
};
|
||||
|
||||
/* io-wq worker categories */
|
||||
enum io_wq_type {
|
||||
IO_WQ_BOUND,
|
||||
IO_WQ_UNBOUND,
|
||||
};
|
||||
|
||||
/* deprecated, see struct io_uring_rsrc_update */
|
||||
struct io_uring_files_update {
|
||||
__u32 offset;
|
||||
__u32 resv;
|
||||
__aligned_u64 /* __s32 * */ fds;
|
||||
};
|
||||
|
||||
enum {
|
||||
/* initialise with user provided memory pointed by user_addr */
|
||||
IORING_MEM_REGION_TYPE_USER = 1,
|
||||
};
|
||||
|
||||
struct io_uring_region_desc {
|
||||
__u64 user_addr;
|
||||
__u64 size;
|
||||
__u32 flags;
|
||||
__u32 id;
|
||||
__u64 mmap_offset;
|
||||
__u64 __resv[4];
|
||||
};
|
||||
|
||||
enum {
|
||||
/* expose the region as registered wait arguments */
|
||||
IORING_MEM_REGION_REG_WAIT_ARG = 1,
|
||||
};
|
||||
|
||||
struct io_uring_mem_region_reg {
|
||||
__u64 region_uptr; /* struct io_uring_region_desc * */
|
||||
__u64 flags;
|
||||
__u64 __resv[2];
|
||||
};
|
||||
|
||||
/*
|
||||
* Register a fully sparse file space, rather than pass in an array of all
|
||||
* -1 file descriptors.
|
||||
*/
|
||||
#define IORING_RSRC_REGISTER_SPARSE (1U << 0)
|
||||
|
||||
struct io_uring_rsrc_register {
|
||||
__u32 nr;
|
||||
__u32 flags;
|
||||
__u64 resv2;
|
||||
__aligned_u64 data;
|
||||
__aligned_u64 tags;
|
||||
};
|
||||
|
||||
struct io_uring_rsrc_update {
|
||||
__u32 offset;
|
||||
__u32 resv;
|
||||
__aligned_u64 data;
|
||||
};
|
||||
|
||||
struct io_uring_rsrc_update2 {
|
||||
__u32 offset;
|
||||
__u32 resv;
|
||||
__aligned_u64 data;
|
||||
__aligned_u64 tags;
|
||||
__u32 nr;
|
||||
__u32 resv2;
|
||||
};
|
||||
|
||||
/* Skip updating fd indexes set to this value in the fd table */
|
||||
#define IORING_REGISTER_FILES_SKIP (-2)
|
||||
|
||||
#define IO_URING_OP_SUPPORTED (1U << 0)
|
||||
|
||||
struct io_uring_probe_op {
|
||||
__u8 op;
|
||||
__u8 resv;
|
||||
__u16 flags; /* IO_URING_OP_* flags */
|
||||
__u32 resv2;
|
||||
};
|
||||
|
||||
struct io_uring_probe {
|
||||
__u8 last_op; /* last opcode supported */
|
||||
__u8 ops_len; /* length of ops[] array below */
|
||||
__u16 resv;
|
||||
__u32 resv2[3];
|
||||
struct io_uring_probe_op ops[];
|
||||
};
|
||||
|
||||
struct io_uring_restriction {
|
||||
__u16 opcode;
|
||||
union {
|
||||
__u8 register_op; /* IORING_RESTRICTION_REGISTER_OP */
|
||||
__u8 sqe_op; /* IORING_RESTRICTION_SQE_OP */
|
||||
__u8 sqe_flags; /* IORING_RESTRICTION_SQE_FLAGS_* */
|
||||
};
|
||||
__u8 resv;
|
||||
__u32 resv2[3];
|
||||
};
|
||||
|
||||
struct io_uring_clock_register {
|
||||
__u32 clockid;
|
||||
__u32 __resv[3];
|
||||
};
|
||||
|
||||
enum {
|
||||
IORING_REGISTER_SRC_REGISTERED = (1U << 0),
|
||||
IORING_REGISTER_DST_REPLACE = (1U << 1),
|
||||
};
|
||||
|
||||
struct io_uring_clone_buffers {
|
||||
__u32 src_fd;
|
||||
__u32 flags;
|
||||
__u32 src_off;
|
||||
__u32 dst_off;
|
||||
__u32 nr;
|
||||
__u32 pad[3];
|
||||
};
|
||||
|
||||
struct io_uring_buf {
|
||||
__u64 addr;
|
||||
__u32 len;
|
||||
__u16 bid;
|
||||
__u16 resv;
|
||||
};
|
||||
|
||||
struct io_uring_buf_ring {
|
||||
union {
|
||||
/*
|
||||
* To avoid spilling into more pages than we need to, the
|
||||
* ring tail is overlaid with the io_uring_buf->resv field.
|
||||
*/
|
||||
struct {
|
||||
__u64 resv1;
|
||||
__u32 resv2;
|
||||
__u16 resv3;
|
||||
__u16 tail;
|
||||
};
|
||||
struct io_uring_buf bufs[0];
|
||||
};
|
||||
};
|
||||
|
||||
/*
|
||||
* Flags for IORING_REGISTER_PBUF_RING.
|
||||
*
|
||||
* IOU_PBUF_RING_MMAP: If set, kernel will allocate the memory for the ring.
|
||||
* The application must not set a ring_addr in struct
|
||||
* io_uring_buf_reg, instead it must subsequently call
|
||||
* mmap(2) with the offset set as:
|
||||
* IORING_OFF_PBUF_RING | (bgid << IORING_OFF_PBUF_SHIFT)
|
||||
* to get a virtual mapping for the ring.
|
||||
* IOU_PBUF_RING_INC: If set, buffers consumed from this buffer ring can be
|
||||
* consumed incrementally. Normally one (or more) buffers
|
||||
* are fully consumed. With incremental consumptions, it's
|
||||
* feasible to register big ranges of buffers, and each
|
||||
* use of it will consume only as much as it needs. This
|
||||
* requires that both the kernel and application keep
|
||||
* track of where the current read/recv index is at.
|
||||
*/
|
||||
enum io_uring_register_pbuf_ring_flags {
|
||||
IOU_PBUF_RING_MMAP = 1,
|
||||
IOU_PBUF_RING_INC = 2,
|
||||
};
|
||||
|
||||
/* argument for IORING_(UN)REGISTER_PBUF_RING */
|
||||
struct io_uring_buf_reg {
|
||||
__u64 ring_addr;
|
||||
__u32 ring_entries;
|
||||
__u16 bgid;
|
||||
__u16 flags;
|
||||
__u64 resv[3];
|
||||
};
|
||||
|
||||
/* argument for IORING_REGISTER_PBUF_STATUS */
|
||||
struct io_uring_buf_status {
|
||||
__u32 buf_group; /* input */
|
||||
__u32 head; /* output */
|
||||
__u32 resv[8];
|
||||
};
|
||||
|
||||
/* argument for IORING_(UN)REGISTER_NAPI */
|
||||
struct io_uring_napi {
|
||||
__u32 busy_poll_to;
|
||||
__u8 prefer_busy_poll;
|
||||
__u8 pad[3];
|
||||
__u64 resv;
|
||||
};
|
||||
|
||||
/*
|
||||
* io_uring_restriction->opcode values
|
||||
*/
|
||||
enum io_uring_register_restriction_op {
|
||||
/* Allow an io_uring_register(2) opcode */
|
||||
IORING_RESTRICTION_REGISTER_OP = 0,
|
||||
|
||||
/* Allow an sqe opcode */
|
||||
IORING_RESTRICTION_SQE_OP = 1,
|
||||
|
||||
/* Allow sqe flags */
|
||||
IORING_RESTRICTION_SQE_FLAGS_ALLOWED = 2,
|
||||
|
||||
/* Require sqe flags (these flags must be set on each submission) */
|
||||
IORING_RESTRICTION_SQE_FLAGS_REQUIRED = 3,
|
||||
|
||||
IORING_RESTRICTION_LAST
|
||||
};
|
||||
|
||||
enum {
|
||||
IORING_REG_WAIT_TS = (1U << 0),
|
||||
};
|
||||
|
||||
/*
|
||||
* Argument for io_uring_enter(2) with
|
||||
* IORING_GETEVENTS | IORING_ENTER_EXT_ARG_REG set, where the actual argument
|
||||
* is an index into a previously registered fixed wait region described by
|
||||
* the below structure.
|
||||
*/
|
||||
struct io_uring_reg_wait {
|
||||
struct __kernel_timespec ts;
|
||||
__u32 min_wait_usec;
|
||||
__u32 flags;
|
||||
__u64 sigmask;
|
||||
__u32 sigmask_sz;
|
||||
__u32 pad[3];
|
||||
__u64 pad2[2];
|
||||
};
|
||||
|
||||
/*
|
||||
* Argument for io_uring_enter(2) with IORING_GETEVENTS | IORING_ENTER_EXT_ARG
|
||||
*/
|
||||
struct io_uring_getevents_arg {
|
||||
__u64 sigmask;
|
||||
__u32 sigmask_sz;
|
||||
__u32 min_wait_usec;
|
||||
__u64 ts;
|
||||
};
|
||||
|
||||
/*
|
||||
* Argument for IORING_REGISTER_SYNC_CANCEL
|
||||
*/
|
||||
struct io_uring_sync_cancel_reg {
|
||||
__u64 addr;
|
||||
__s32 fd;
|
||||
__u32 flags;
|
||||
struct __kernel_timespec timeout;
|
||||
__u8 opcode;
|
||||
__u8 pad[7];
|
||||
__u64 pad2[3];
|
||||
};
|
||||
|
||||
/*
|
||||
* Argument for IORING_REGISTER_FILE_ALLOC_RANGE
|
||||
* The range is specified as [off, off + len)
|
||||
*/
|
||||
struct io_uring_file_index_range {
|
||||
__u32 off;
|
||||
__u32 len;
|
||||
__u64 resv;
|
||||
};
|
||||
|
||||
struct io_uring_recvmsg_out {
|
||||
__u32 namelen;
|
||||
__u32 controllen;
|
||||
__u32 payloadlen;
|
||||
__u32 flags;
|
||||
};
|
||||
|
||||
/*
|
||||
* Argument for IORING_OP_URING_CMD when file is a socket
|
||||
*/
|
||||
enum io_uring_socket_op {
|
||||
SOCKET_URING_OP_SIOCINQ = 0,
|
||||
SOCKET_URING_OP_SIOCOUTQ,
|
||||
SOCKET_URING_OP_GETSOCKOPT,
|
||||
SOCKET_URING_OP_SETSOCKOPT,
|
||||
};
|
||||
|
||||
/* Zero copy receive refill queue entry */
|
||||
struct io_uring_zcrx_rqe {
|
||||
__u64 off;
|
||||
__u32 len;
|
||||
__u32 __pad;
|
||||
};
|
||||
|
||||
struct io_uring_zcrx_cqe {
|
||||
__u64 off;
|
||||
__u64 __pad;
|
||||
};
|
||||
|
||||
/* The bit from which area id is encoded into offsets */
|
||||
#define IORING_ZCRX_AREA_SHIFT 48
|
||||
#define IORING_ZCRX_AREA_MASK (~(((__u64)1 << IORING_ZCRX_AREA_SHIFT) - 1))
|
||||
|
||||
struct io_uring_zcrx_offsets {
|
||||
__u32 head;
|
||||
__u32 tail;
|
||||
__u32 rqes;
|
||||
__u32 __resv2;
|
||||
__u64 __resv[2];
|
||||
};
|
||||
|
||||
enum io_uring_zcrx_area_flags {
|
||||
IORING_ZCRX_AREA_DMABUF = 1,
|
||||
};
|
||||
|
||||
struct io_uring_zcrx_area_reg {
|
||||
__u64 addr;
|
||||
__u64 len;
|
||||
__u64 rq_area_token;
|
||||
__u32 flags;
|
||||
__u32 dmabuf_fd;
|
||||
__u64 __resv2[2];
|
||||
};
|
||||
|
||||
/*
|
||||
* Argument for IORING_REGISTER_ZCRX_IFQ
|
||||
*/
|
||||
struct io_uring_zcrx_ifq_reg {
|
||||
__u32 if_idx;
|
||||
__u32 if_rxq;
|
||||
__u32 rq_entries;
|
||||
__u32 flags;
|
||||
|
||||
__u64 area_ptr; /* pointer to struct io_uring_zcrx_area_reg */
|
||||
__u64 region_ptr; /* struct io_uring_region_desc * */
|
||||
|
||||
struct io_uring_zcrx_offsets offsets;
|
||||
__u32 zcrx_id;
|
||||
__u32 __resv2;
|
||||
__u64 __resv[3];
|
||||
};
|
||||
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1,8 @@
|
||||
/* SPDX-License-Identifier: MIT */
|
||||
#ifndef LIBURING_VERSION_H
|
||||
#define LIBURING_VERSION_H
|
||||
|
||||
#define IO_URING_VERSION_MAJOR 2
|
||||
#define IO_URING_VERSION_MINOR 11
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1,39 @@
|
||||
/* SPDX-License-Identifier: MIT */
|
||||
#ifndef LIBURING_SANITIZE_H
|
||||
#define LIBURING_SANITIZE_H
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
struct io_uring;
|
||||
struct iovec;
|
||||
|
||||
#if defined(CONFIG_USE_SANITIZER)
|
||||
void liburing_sanitize_ring(struct io_uring *ring);
|
||||
void liburing_sanitize_address(const void *addr);
|
||||
void liburing_sanitize_region(const void *addr, unsigned int len);
|
||||
void liburing_sanitize_iovecs(const struct iovec *iovecs, unsigned nr);
|
||||
#else
|
||||
#define __maybe_unused __attribute__((__unused__))
|
||||
static inline void liburing_sanitize_ring(struct io_uring __maybe_unused *ring)
|
||||
{
|
||||
}
|
||||
static inline void liburing_sanitize_address(const void __maybe_unused *addr)
|
||||
{
|
||||
}
|
||||
static inline void liburing_sanitize_region(const void __maybe_unused *addr,
|
||||
unsigned int __maybe_unused len)
|
||||
{
|
||||
}
|
||||
static inline void liburing_sanitize_iovecs(const struct iovec __maybe_unused *iovecs,
|
||||
unsigned __maybe_unused nr)
|
||||
{
|
||||
}
|
||||
#endif
|
||||
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1,21 @@
|
||||
/* SPDX-License-Identifier: MIT */
|
||||
#ifndef LIBURING_INT_FLAGS
|
||||
#define LIBURING_INT_FLAGS
|
||||
|
||||
#define INT_FLAGS_MASK (IORING_ENTER_REGISTERED_RING | \
|
||||
IORING_ENTER_NO_IOWAIT)
|
||||
|
||||
enum {
|
||||
INT_FLAG_REG_RING = IORING_ENTER_REGISTERED_RING,
|
||||
INT_FLAG_NO_IOWAIT = IORING_ENTER_NO_IOWAIT,
|
||||
INT_FLAG_REG_REG_RING = 1,
|
||||
INT_FLAG_APP_MEM = 2,
|
||||
INT_FLAG_CQ_ENTER = 4,
|
||||
};
|
||||
|
||||
static inline int ring_enter_flags(struct io_uring *ring)
|
||||
{
|
||||
return ring->int_flags & INT_FLAGS_MASK;
|
||||
}
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1,52 @@
|
||||
/* SPDX-License-Identifier: MIT */
|
||||
#ifndef LIBURING_LIB_H
|
||||
#define LIBURING_LIB_H
|
||||
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#if defined(__x86_64__) || defined(__i386__)
|
||||
#include "arch/x86/lib.h"
|
||||
#elif defined(__aarch64__)
|
||||
#include "arch/aarch64/lib.h"
|
||||
#elif defined(__riscv) && __riscv_xlen == 64
|
||||
#include "arch/riscv64/lib.h"
|
||||
#else
|
||||
/*
|
||||
* We don't have nolibc support for this arch. Must use libc!
|
||||
*/
|
||||
#ifdef CONFIG_NOLIBC
|
||||
#error "This arch doesn't support building liburing without libc"
|
||||
#endif
|
||||
/* libc wrappers. */
|
||||
#include "arch/generic/lib.h"
|
||||
#endif
|
||||
|
||||
|
||||
#ifndef offsetof
|
||||
#define offsetof(TYPE, FIELD) ((size_t) &((TYPE *)0)->FIELD)
|
||||
#endif
|
||||
|
||||
#ifndef container_of
|
||||
#define container_of(PTR, TYPE, FIELD) ({ \
|
||||
__typeof__(((TYPE *)0)->FIELD) *__FIELD_PTR = (PTR); \
|
||||
(TYPE *)((char *) __FIELD_PTR - offsetof(TYPE, FIELD)); \
|
||||
})
|
||||
#endif
|
||||
|
||||
#define __maybe_unused __attribute__((__unused__))
|
||||
#define __hot __attribute__((__hot__))
|
||||
#define __cold __attribute__((__cold__))
|
||||
|
||||
#ifdef CONFIG_NOLIBC
|
||||
void *__uring_memset(void *s, int c, size_t n);
|
||||
void *__uring_malloc(size_t len);
|
||||
void __uring_free(void *p);
|
||||
|
||||
#define malloc(LEN) __uring_malloc(LEN)
|
||||
#define free(PTR) __uring_free(PTR)
|
||||
#define memset(PTR, C, LEN) __uring_memset(PTR, C, LEN)
|
||||
#endif
|
||||
|
||||
#endif /* #ifndef LIBURING_LIB_H */
|
||||
@@ -0,0 +1,485 @@
|
||||
/* SPDX-License-Identifier: MIT */
|
||||
#define _POSIX_C_SOURCE 200112L
|
||||
|
||||
#include "lib.h"
|
||||
#include "syscall.h"
|
||||
#include "liburing.h"
|
||||
#include "int_flags.h"
|
||||
#include "liburing/sanitize.h"
|
||||
#include "liburing/io_uring.h"
|
||||
|
||||
/*
|
||||
* Returns true if we're not using SQ thread (thus nobody submits but us)
|
||||
* or if IORING_SQ_NEED_WAKEUP is set, so submit thread must be explicitly
|
||||
* awakened. For the latter case, we set the thread wakeup flag.
|
||||
* If no SQEs are ready for submission, returns false.
|
||||
*/
|
||||
static inline bool sq_ring_needs_enter(struct io_uring *ring,
|
||||
unsigned submit,
|
||||
unsigned *flags)
|
||||
{
|
||||
if (!submit)
|
||||
return false;
|
||||
|
||||
if (!(ring->flags & IORING_SETUP_SQPOLL))
|
||||
return true;
|
||||
|
||||
/*
|
||||
* Ensure the kernel can see the store to the SQ tail before we read
|
||||
* the flags.
|
||||
*/
|
||||
io_uring_smp_mb();
|
||||
|
||||
if (uring_unlikely(IO_URING_READ_ONCE(*ring->sq.kflags) &
|
||||
IORING_SQ_NEED_WAKEUP)) {
|
||||
*flags |= IORING_ENTER_SQ_WAKEUP;
|
||||
return true;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
static inline bool cq_ring_needs_flush(struct io_uring *ring)
|
||||
{
|
||||
return IO_URING_READ_ONCE(*ring->sq.kflags) &
|
||||
(IORING_SQ_CQ_OVERFLOW | IORING_SQ_TASKRUN);
|
||||
}
|
||||
|
||||
static inline bool cq_ring_needs_enter(struct io_uring *ring)
|
||||
{
|
||||
return (ring->int_flags & INT_FLAG_CQ_ENTER) || cq_ring_needs_flush(ring);
|
||||
}
|
||||
|
||||
struct get_data {
|
||||
unsigned submit;
|
||||
unsigned wait_nr;
|
||||
unsigned get_flags;
|
||||
int sz;
|
||||
int has_ts;
|
||||
void *arg;
|
||||
};
|
||||
|
||||
static int _io_uring_get_cqe(struct io_uring *ring,
|
||||
struct io_uring_cqe **cqe_ptr,
|
||||
struct get_data *data)
|
||||
{
|
||||
struct io_uring_cqe *cqe = NULL;
|
||||
bool looped = false;
|
||||
int err = 0;
|
||||
|
||||
do {
|
||||
bool need_enter = false;
|
||||
unsigned flags = ring_enter_flags(ring);
|
||||
unsigned nr_available;
|
||||
int ret;
|
||||
|
||||
ret = __io_uring_peek_cqe(ring, &cqe, &nr_available);
|
||||
if (ret) {
|
||||
if (!err)
|
||||
err = ret;
|
||||
break;
|
||||
}
|
||||
if (!cqe && !data->wait_nr && !data->submit) {
|
||||
/*
|
||||
* If we already looped once, we already entered
|
||||
* the kernel. Since there's nothing to submit or
|
||||
* wait for, don't keep retrying.
|
||||
*/
|
||||
if (looped || !cq_ring_needs_enter(ring)) {
|
||||
if (!err)
|
||||
err = -EAGAIN;
|
||||
break;
|
||||
}
|
||||
need_enter = true;
|
||||
}
|
||||
if (data->wait_nr > nr_available || need_enter) {
|
||||
flags |= IORING_ENTER_GETEVENTS | data->get_flags;
|
||||
need_enter = true;
|
||||
}
|
||||
if (sq_ring_needs_enter(ring, data->submit, &flags))
|
||||
need_enter = true;
|
||||
if (!need_enter)
|
||||
break;
|
||||
if (looped && data->has_ts) {
|
||||
struct io_uring_getevents_arg *arg = data->arg;
|
||||
|
||||
if (!cqe && arg->ts && !err)
|
||||
err = -ETIME;
|
||||
break;
|
||||
}
|
||||
|
||||
ret = __sys_io_uring_enter2(ring->enter_ring_fd, data->submit,
|
||||
data->wait_nr, flags, data->arg,
|
||||
data->sz);
|
||||
if (ret < 0) {
|
||||
if (!err)
|
||||
err = ret;
|
||||
break;
|
||||
}
|
||||
|
||||
data->submit -= ret;
|
||||
if (cqe)
|
||||
break;
|
||||
if (!looped) {
|
||||
looped = true;
|
||||
err = ret;
|
||||
}
|
||||
} while (1);
|
||||
|
||||
*cqe_ptr = cqe;
|
||||
return err;
|
||||
}
|
||||
|
||||
int __io_uring_get_cqe(struct io_uring *ring, struct io_uring_cqe **cqe_ptr,
|
||||
unsigned submit, unsigned wait_nr, sigset_t *sigmask)
|
||||
{
|
||||
struct get_data data = {
|
||||
.submit = submit,
|
||||
.wait_nr = wait_nr,
|
||||
.get_flags = 0,
|
||||
.sz = _NSIG / 8,
|
||||
.arg = sigmask,
|
||||
};
|
||||
|
||||
return _io_uring_get_cqe(ring, cqe_ptr, &data);
|
||||
}
|
||||
|
||||
int io_uring_get_events(struct io_uring *ring)
|
||||
{
|
||||
int flags = IORING_ENTER_GETEVENTS | ring_enter_flags(ring);
|
||||
|
||||
return __sys_io_uring_enter(ring->enter_ring_fd, 0, 0, flags, NULL);
|
||||
}
|
||||
|
||||
static inline bool io_uring_peek_batch_cqe_(struct io_uring *ring,
|
||||
struct io_uring_cqe **cqes,
|
||||
unsigned *count)
|
||||
{
|
||||
unsigned ready = io_uring_cq_ready(ring);
|
||||
unsigned shift;
|
||||
unsigned head;
|
||||
unsigned mask;
|
||||
unsigned last;
|
||||
|
||||
if (!ready)
|
||||
return false;
|
||||
|
||||
shift = io_uring_cqe_shift(ring);
|
||||
head = *ring->cq.khead;
|
||||
mask = ring->cq.ring_mask;
|
||||
if (ready < *count)
|
||||
*count = ready;
|
||||
last = head + *count;
|
||||
for (;head != last; head++)
|
||||
*(cqes++) = &ring->cq.cqes[(head & mask) << shift];
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
/*
|
||||
* Fill in an array of IO completions up to count, if any are available.
|
||||
* Returns the amount of IO completions filled.
|
||||
*/
|
||||
unsigned io_uring_peek_batch_cqe(struct io_uring *ring,
|
||||
struct io_uring_cqe **cqes, unsigned count)
|
||||
{
|
||||
if (io_uring_peek_batch_cqe_(ring, cqes, &count))
|
||||
return count;
|
||||
|
||||
if (!cq_ring_needs_flush(ring))
|
||||
return 0;
|
||||
|
||||
io_uring_get_events(ring);
|
||||
if (!io_uring_peek_batch_cqe_(ring, cqes, &count))
|
||||
return 0;
|
||||
|
||||
return count;
|
||||
}
|
||||
|
||||
/*
|
||||
* Sync internal state with kernel ring state on the SQ side. Returns the
|
||||
* number of pending items in the SQ ring, for the shared ring.
|
||||
*/
|
||||
static unsigned __io_uring_flush_sq(struct io_uring *ring)
|
||||
{
|
||||
struct io_uring_sq *sq = &ring->sq;
|
||||
unsigned tail = sq->sqe_tail;
|
||||
|
||||
if (sq->sqe_head != tail) {
|
||||
sq->sqe_head = tail;
|
||||
/*
|
||||
* Ensure kernel sees the SQE updates before the tail update.
|
||||
*/
|
||||
if (!(ring->flags & IORING_SETUP_SQPOLL))
|
||||
*sq->ktail = tail;
|
||||
else
|
||||
io_uring_smp_store_release(sq->ktail, tail);
|
||||
}
|
||||
/*
|
||||
* This load needs to be atomic, since sq->khead is written concurrently
|
||||
* by the kernel, but it doesn't need to be load_acquire, since the
|
||||
* kernel doesn't store to the submission queue; it advances khead just
|
||||
* to indicate that it's finished reading the submission queue entries
|
||||
* so they're available for us to write to.
|
||||
*/
|
||||
return tail - IO_URING_READ_ONCE(*sq->khead);
|
||||
}
|
||||
|
||||
/*
|
||||
* If we have kernel support for IORING_ENTER_EXT_ARG, then we can use that
|
||||
* more efficiently than queueing an internal timeout command.
|
||||
*/
|
||||
static int io_uring_wait_cqes_new(struct io_uring *ring,
|
||||
struct io_uring_cqe **cqe_ptr,
|
||||
unsigned wait_nr,
|
||||
struct __kernel_timespec *ts,
|
||||
unsigned int min_wait_usec,
|
||||
sigset_t *sigmask)
|
||||
{
|
||||
struct io_uring_getevents_arg arg = {
|
||||
.sigmask = (unsigned long) sigmask,
|
||||
.sigmask_sz = _NSIG / 8,
|
||||
.ts = (unsigned long) ts
|
||||
};
|
||||
struct get_data data = {
|
||||
.wait_nr = wait_nr,
|
||||
.get_flags = IORING_ENTER_EXT_ARG,
|
||||
.sz = sizeof(arg),
|
||||
.has_ts = ts != NULL,
|
||||
.arg = &arg
|
||||
};
|
||||
|
||||
if (min_wait_usec && ring->features & IORING_FEAT_MIN_TIMEOUT)
|
||||
arg.min_wait_usec = min_wait_usec;
|
||||
|
||||
return _io_uring_get_cqe(ring, cqe_ptr, &data);
|
||||
}
|
||||
|
||||
/*
|
||||
* Like io_uring_wait_cqe(), except it accepts a timeout value as well. Note
|
||||
* that an sqe is used internally to handle the timeout. For kernel doesn't
|
||||
* support IORING_FEAT_EXT_ARG, applications using this function must never
|
||||
* set sqe->user_data to LIBURING_UDATA_TIMEOUT!
|
||||
*
|
||||
* For kernels without IORING_FEAT_EXT_ARG (5.10 and older), if 'ts' is
|
||||
* specified, the application need not call io_uring_submit() before
|
||||
* calling this function, as we will do that on its behalf. From this it also
|
||||
* follows that this function isn't safe to use for applications that split SQ
|
||||
* and CQ handling between two threads and expect that to work without
|
||||
* synchronization, as this function manipulates both the SQ and CQ side.
|
||||
*
|
||||
* For kernels with IORING_FEAT_EXT_ARG, no implicit submission is done and
|
||||
* hence this function is safe to use for applications that split SQ and CQ
|
||||
* handling between two threads.
|
||||
*/
|
||||
static int __io_uring_submit_timeout(struct io_uring *ring, unsigned wait_nr,
|
||||
struct __kernel_timespec *ts)
|
||||
{
|
||||
struct io_uring_sqe *sqe;
|
||||
int ret;
|
||||
|
||||
/*
|
||||
* If the SQ ring is full, we may need to submit IO first
|
||||
*/
|
||||
sqe = io_uring_get_sqe(ring);
|
||||
if (!sqe) {
|
||||
ret = io_uring_submit(ring);
|
||||
if (ret < 0)
|
||||
return ret;
|
||||
sqe = io_uring_get_sqe(ring);
|
||||
if (!sqe)
|
||||
return -EAGAIN;
|
||||
}
|
||||
io_uring_prep_timeout(sqe, ts, wait_nr, 0);
|
||||
sqe->user_data = LIBURING_UDATA_TIMEOUT;
|
||||
return __io_uring_flush_sq(ring);
|
||||
}
|
||||
|
||||
int io_uring_wait_cqes(struct io_uring *ring, struct io_uring_cqe **cqe_ptr,
|
||||
unsigned wait_nr, struct __kernel_timespec *ts,
|
||||
sigset_t *sigmask)
|
||||
{
|
||||
int to_submit = 0;
|
||||
|
||||
if (ts) {
|
||||
if (ring->features & IORING_FEAT_EXT_ARG)
|
||||
return io_uring_wait_cqes_new(ring, cqe_ptr, wait_nr,
|
||||
ts, 0, sigmask);
|
||||
to_submit = __io_uring_submit_timeout(ring, wait_nr, ts);
|
||||
if (to_submit < 0)
|
||||
return to_submit;
|
||||
}
|
||||
|
||||
return __io_uring_get_cqe(ring, cqe_ptr, to_submit, wait_nr, sigmask);
|
||||
}
|
||||
|
||||
int io_uring_wait_cqes_min_timeout(struct io_uring *ring,
|
||||
struct io_uring_cqe **cqe_ptr,
|
||||
unsigned wait_nr,
|
||||
struct __kernel_timespec *ts,
|
||||
unsigned int min_wait_usec, sigset_t *sigmask)
|
||||
{
|
||||
return io_uring_wait_cqes_new(ring, cqe_ptr, wait_nr, ts, min_wait_usec,
|
||||
sigmask);
|
||||
}
|
||||
|
||||
int io_uring_submit_and_wait_reg(struct io_uring *ring,
|
||||
struct io_uring_cqe **cqe_ptr,
|
||||
unsigned wait_nr, int reg_index)
|
||||
{
|
||||
unsigned long offset = reg_index * sizeof(struct io_uring_reg_wait);
|
||||
|
||||
struct get_data data = {
|
||||
.submit = __io_uring_flush_sq(ring),
|
||||
.wait_nr = wait_nr,
|
||||
.get_flags = IORING_ENTER_EXT_ARG |
|
||||
IORING_ENTER_EXT_ARG_REG,
|
||||
.sz = sizeof(struct io_uring_reg_wait),
|
||||
.has_ts = true,
|
||||
.arg = (void *) (uintptr_t) offset,
|
||||
};
|
||||
|
||||
if (!(ring->features & IORING_FEAT_EXT_ARG))
|
||||
return -EINVAL;
|
||||
|
||||
return _io_uring_get_cqe(ring, cqe_ptr, &data);
|
||||
}
|
||||
|
||||
static int __io_uring_submit_and_wait_timeout(struct io_uring *ring,
|
||||
struct io_uring_cqe **cqe_ptr, unsigned wait_nr,
|
||||
struct __kernel_timespec *ts,
|
||||
unsigned int min_wait, sigset_t *sigmask)
|
||||
{
|
||||
int to_submit;
|
||||
|
||||
if (ts) {
|
||||
if (ring->features & IORING_FEAT_EXT_ARG) {
|
||||
struct io_uring_getevents_arg arg = {
|
||||
.sigmask = (unsigned long) sigmask,
|
||||
.sigmask_sz = _NSIG / 8,
|
||||
.min_wait_usec = min_wait,
|
||||
.ts = (unsigned long) ts
|
||||
};
|
||||
struct get_data data = {
|
||||
.submit = __io_uring_flush_sq(ring),
|
||||
.wait_nr = wait_nr,
|
||||
.get_flags = IORING_ENTER_EXT_ARG,
|
||||
.sz = sizeof(arg),
|
||||
.has_ts = ts != NULL,
|
||||
.arg = &arg
|
||||
};
|
||||
|
||||
return _io_uring_get_cqe(ring, cqe_ptr, &data);
|
||||
}
|
||||
to_submit = __io_uring_submit_timeout(ring, wait_nr, ts);
|
||||
if (to_submit < 0)
|
||||
return to_submit;
|
||||
} else
|
||||
to_submit = __io_uring_flush_sq(ring);
|
||||
|
||||
return __io_uring_get_cqe(ring, cqe_ptr, to_submit, wait_nr, sigmask);
|
||||
}
|
||||
|
||||
int io_uring_submit_and_wait_min_timeout(struct io_uring *ring,
|
||||
struct io_uring_cqe **cqe_ptr,
|
||||
unsigned wait_nr,
|
||||
struct __kernel_timespec *ts,
|
||||
unsigned min_wait,
|
||||
sigset_t *sigmask)
|
||||
{
|
||||
if (!(ring->features & IORING_FEAT_MIN_TIMEOUT))
|
||||
return -EINVAL;
|
||||
return __io_uring_submit_and_wait_timeout(ring, cqe_ptr, wait_nr, ts,
|
||||
min_wait, sigmask);
|
||||
}
|
||||
|
||||
int io_uring_submit_and_wait_timeout(struct io_uring *ring,
|
||||
struct io_uring_cqe **cqe_ptr,
|
||||
unsigned wait_nr,
|
||||
struct __kernel_timespec *ts,
|
||||
sigset_t *sigmask)
|
||||
{
|
||||
return __io_uring_submit_and_wait_timeout(ring, cqe_ptr, wait_nr, ts, 0,
|
||||
sigmask);
|
||||
}
|
||||
|
||||
/*
|
||||
* See io_uring_wait_cqes() - this function is the same, it just always uses
|
||||
* '1' as the wait_nr.
|
||||
*/
|
||||
int io_uring_wait_cqe_timeout(struct io_uring *ring,
|
||||
struct io_uring_cqe **cqe_ptr,
|
||||
struct __kernel_timespec *ts)
|
||||
{
|
||||
return io_uring_wait_cqes(ring, cqe_ptr, 1, ts, NULL);
|
||||
}
|
||||
|
||||
/*
|
||||
* Submit sqes acquired from io_uring_get_sqe() to the kernel.
|
||||
*
|
||||
* Returns number of sqes submitted
|
||||
*/
|
||||
static int __io_uring_submit(struct io_uring *ring, unsigned submitted,
|
||||
unsigned wait_nr, bool getevents)
|
||||
{
|
||||
bool cq_needs_enter = getevents || wait_nr || cq_ring_needs_enter(ring);
|
||||
unsigned flags = ring_enter_flags(ring);
|
||||
int ret;
|
||||
|
||||
liburing_sanitize_ring(ring);
|
||||
|
||||
if (sq_ring_needs_enter(ring, submitted, &flags) || cq_needs_enter) {
|
||||
if (cq_needs_enter)
|
||||
flags |= IORING_ENTER_GETEVENTS;
|
||||
|
||||
ret = __sys_io_uring_enter(ring->enter_ring_fd, submitted,
|
||||
wait_nr, flags, NULL);
|
||||
} else
|
||||
ret = submitted;
|
||||
|
||||
return ret;
|
||||
}
|
||||
|
||||
static int __io_uring_submit_and_wait(struct io_uring *ring, unsigned wait_nr)
|
||||
{
|
||||
return __io_uring_submit(ring, __io_uring_flush_sq(ring), wait_nr, false);
|
||||
}
|
||||
|
||||
/*
|
||||
* Submit sqes acquired from io_uring_get_sqe() to the kernel.
|
||||
*
|
||||
* Returns number of sqes submitted
|
||||
*/
|
||||
int io_uring_submit(struct io_uring *ring)
|
||||
{
|
||||
return __io_uring_submit_and_wait(ring, 0);
|
||||
}
|
||||
|
||||
/*
|
||||
* Like io_uring_submit(), but allows waiting for events as well.
|
||||
*
|
||||
* Returns number of sqes submitted
|
||||
*/
|
||||
int io_uring_submit_and_wait(struct io_uring *ring, unsigned wait_nr)
|
||||
{
|
||||
return __io_uring_submit_and_wait(ring, wait_nr);
|
||||
}
|
||||
|
||||
int io_uring_submit_and_get_events(struct io_uring *ring)
|
||||
{
|
||||
return __io_uring_submit(ring, __io_uring_flush_sq(ring), 0, true);
|
||||
}
|
||||
|
||||
#ifdef LIBURING_INTERNAL
|
||||
struct io_uring_sqe *io_uring_get_sqe(struct io_uring *ring)
|
||||
{
|
||||
return _io_uring_get_sqe(ring);
|
||||
}
|
||||
#endif
|
||||
|
||||
int __io_uring_sqring_wait(struct io_uring *ring)
|
||||
{
|
||||
int flags = IORING_ENTER_SQ_WAIT | ring_enter_flags(ring);
|
||||
|
||||
return __sys_io_uring_enter(ring->enter_ring_fd, 0, 0, flags, NULL);
|
||||
}
|
||||
@@ -0,0 +1,510 @@
|
||||
/* SPDX-License-Identifier: MIT */
|
||||
#define _POSIX_C_SOURCE 200112L
|
||||
|
||||
#include "lib.h"
|
||||
#include "syscall.h"
|
||||
#include "liburing.h"
|
||||
#include "setup.h"
|
||||
#include "int_flags.h"
|
||||
#include "liburing/io_uring.h"
|
||||
#include "liburing/sanitize.h"
|
||||
|
||||
static inline int do_register(struct io_uring *ring, unsigned int opcode,
|
||||
const void *arg, unsigned int nr_args)
|
||||
{
|
||||
int fd;
|
||||
|
||||
liburing_sanitize_address(arg);
|
||||
|
||||
if (ring->int_flags & INT_FLAG_REG_REG_RING) {
|
||||
opcode |= IORING_REGISTER_USE_REGISTERED_RING;
|
||||
fd = ring->enter_ring_fd;
|
||||
} else {
|
||||
fd = ring->ring_fd;
|
||||
}
|
||||
|
||||
return __sys_io_uring_register(fd, opcode, arg, nr_args);
|
||||
}
|
||||
|
||||
int io_uring_register_buffers_update_tag(struct io_uring *ring, unsigned off,
|
||||
const struct iovec *iovecs,
|
||||
const __u64 *tags,
|
||||
unsigned nr)
|
||||
{
|
||||
liburing_sanitize_iovecs(iovecs, nr);
|
||||
|
||||
struct io_uring_rsrc_update2 up = {
|
||||
.offset = off,
|
||||
.data = (unsigned long)iovecs,
|
||||
.tags = (unsigned long)tags,
|
||||
.nr = nr,
|
||||
};
|
||||
|
||||
return do_register(ring, IORING_REGISTER_BUFFERS_UPDATE, &up, sizeof(up));
|
||||
}
|
||||
|
||||
int io_uring_register_buffers_tags(struct io_uring *ring,
|
||||
const struct iovec *iovecs,
|
||||
const __u64 *tags,
|
||||
unsigned nr)
|
||||
{
|
||||
liburing_sanitize_iovecs(iovecs, nr);
|
||||
|
||||
struct io_uring_rsrc_register reg = {
|
||||
.nr = nr,
|
||||
.data = (unsigned long)iovecs,
|
||||
.tags = (unsigned long)tags,
|
||||
};
|
||||
|
||||
return do_register(ring, IORING_REGISTER_BUFFERS2, ®, sizeof(reg));
|
||||
}
|
||||
|
||||
int io_uring_register_buffers_sparse(struct io_uring *ring, unsigned nr)
|
||||
{
|
||||
struct io_uring_rsrc_register reg = {
|
||||
.flags = IORING_RSRC_REGISTER_SPARSE,
|
||||
.nr = nr,
|
||||
};
|
||||
|
||||
return do_register(ring, IORING_REGISTER_BUFFERS2, ®, sizeof(reg));
|
||||
}
|
||||
|
||||
int io_uring_register_buffers(struct io_uring *ring, const struct iovec *iovecs,
|
||||
unsigned nr_iovecs)
|
||||
{
|
||||
liburing_sanitize_iovecs(iovecs, nr_iovecs);
|
||||
|
||||
return do_register(ring, IORING_REGISTER_BUFFERS, iovecs, nr_iovecs);
|
||||
}
|
||||
|
||||
int io_uring_unregister_buffers(struct io_uring *ring)
|
||||
{
|
||||
return do_register(ring, IORING_UNREGISTER_BUFFERS, NULL, 0);
|
||||
}
|
||||
|
||||
int io_uring_register_files_update_tag(struct io_uring *ring, unsigned off,
|
||||
const int *files, const __u64 *tags,
|
||||
unsigned nr_files)
|
||||
{
|
||||
liburing_sanitize_address(files);
|
||||
liburing_sanitize_address(tags);
|
||||
|
||||
struct io_uring_rsrc_update2 up = {
|
||||
.offset = off,
|
||||
.data = (unsigned long)files,
|
||||
.tags = (unsigned long)tags,
|
||||
.nr = nr_files,
|
||||
};
|
||||
|
||||
return do_register(ring, IORING_REGISTER_FILES_UPDATE2, &up, sizeof(up));
|
||||
}
|
||||
|
||||
/*
|
||||
* Register an update for an existing file set. The updates will start at
|
||||
* 'off' in the original array, and 'nr_files' is the number of files we'll
|
||||
* update.
|
||||
*
|
||||
* Returns number of files updated on success, -ERROR on failure.
|
||||
*/
|
||||
int io_uring_register_files_update(struct io_uring *ring, unsigned off,
|
||||
const int *files, unsigned nr_files)
|
||||
{
|
||||
liburing_sanitize_address(files);
|
||||
|
||||
struct io_uring_rsrc_update up = {
|
||||
.offset = off,
|
||||
.data = (unsigned long) files,
|
||||
};
|
||||
|
||||
return do_register(ring, IORING_REGISTER_FILES_UPDATE, &up, nr_files);
|
||||
}
|
||||
|
||||
static int increase_rlimit_nofile(unsigned nr)
|
||||
{
|
||||
int ret;
|
||||
struct rlimit rlim;
|
||||
|
||||
ret = __sys_getrlimit(RLIMIT_NOFILE, &rlim);
|
||||
if (ret < 0)
|
||||
return ret;
|
||||
|
||||
if (rlim.rlim_cur < nr) {
|
||||
rlim.rlim_cur += nr;
|
||||
__sys_setrlimit(RLIMIT_NOFILE, &rlim);
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
int io_uring_register_files_sparse(struct io_uring *ring, unsigned nr)
|
||||
{
|
||||
struct io_uring_rsrc_register reg = {
|
||||
.flags = IORING_RSRC_REGISTER_SPARSE,
|
||||
.nr = nr,
|
||||
};
|
||||
int ret, did_increase = 0;
|
||||
|
||||
do {
|
||||
ret = do_register(ring, IORING_REGISTER_FILES2, ®, sizeof(reg));
|
||||
if (ret >= 0)
|
||||
break;
|
||||
if (ret == -EMFILE && !did_increase) {
|
||||
did_increase = 1;
|
||||
increase_rlimit_nofile(nr);
|
||||
continue;
|
||||
}
|
||||
break;
|
||||
} while (1);
|
||||
|
||||
return ret;
|
||||
}
|
||||
|
||||
int io_uring_register_files_tags(struct io_uring *ring, const int *files,
|
||||
const __u64 *tags, unsigned nr)
|
||||
{
|
||||
liburing_sanitize_address(files);
|
||||
liburing_sanitize_address(tags);
|
||||
|
||||
struct io_uring_rsrc_register reg = {
|
||||
.nr = nr,
|
||||
.data = (unsigned long)files,
|
||||
.tags = (unsigned long)tags,
|
||||
};
|
||||
int ret, did_increase = 0;
|
||||
|
||||
do {
|
||||
ret = do_register(ring, IORING_REGISTER_FILES2, ®, sizeof(reg));
|
||||
if (ret >= 0)
|
||||
break;
|
||||
if (ret == -EMFILE && !did_increase) {
|
||||
did_increase = 1;
|
||||
increase_rlimit_nofile(nr);
|
||||
continue;
|
||||
}
|
||||
break;
|
||||
} while (1);
|
||||
|
||||
return ret;
|
||||
}
|
||||
|
||||
int io_uring_register_files(struct io_uring *ring, const int *files,
|
||||
unsigned nr_files)
|
||||
{
|
||||
int ret, did_increase = 0;
|
||||
|
||||
liburing_sanitize_address(files);
|
||||
|
||||
do {
|
||||
ret = do_register(ring, IORING_REGISTER_FILES, files, nr_files);
|
||||
if (ret >= 0)
|
||||
break;
|
||||
if (ret == -EMFILE && !did_increase) {
|
||||
did_increase = 1;
|
||||
increase_rlimit_nofile(nr_files);
|
||||
continue;
|
||||
}
|
||||
break;
|
||||
} while (1);
|
||||
|
||||
return ret;
|
||||
}
|
||||
|
||||
int io_uring_unregister_files(struct io_uring *ring)
|
||||
{
|
||||
return do_register(ring, IORING_UNREGISTER_FILES, NULL, 0);
|
||||
}
|
||||
|
||||
int io_uring_register_eventfd(struct io_uring *ring, int event_fd)
|
||||
{
|
||||
return do_register(ring, IORING_REGISTER_EVENTFD, &event_fd, 1);
|
||||
}
|
||||
|
||||
int io_uring_unregister_eventfd(struct io_uring *ring)
|
||||
{
|
||||
return do_register(ring, IORING_UNREGISTER_EVENTFD, NULL, 0);
|
||||
}
|
||||
|
||||
int io_uring_register_eventfd_async(struct io_uring *ring, int event_fd)
|
||||
{
|
||||
return do_register(ring, IORING_REGISTER_EVENTFD_ASYNC, &event_fd, 1);
|
||||
}
|
||||
|
||||
int io_uring_register_probe(struct io_uring *ring, struct io_uring_probe *p,
|
||||
unsigned int nr_ops)
|
||||
{
|
||||
return do_register(ring, IORING_REGISTER_PROBE, p, nr_ops);
|
||||
}
|
||||
|
||||
int io_uring_register_personality(struct io_uring *ring)
|
||||
{
|
||||
return do_register(ring, IORING_REGISTER_PERSONALITY, NULL, 0);
|
||||
}
|
||||
|
||||
int io_uring_unregister_personality(struct io_uring *ring, int id)
|
||||
{
|
||||
return do_register(ring, IORING_UNREGISTER_PERSONALITY, NULL, id);
|
||||
}
|
||||
|
||||
int io_uring_register_restrictions(struct io_uring *ring,
|
||||
struct io_uring_restriction *res,
|
||||
unsigned int nr_res)
|
||||
{
|
||||
return do_register(ring, IORING_REGISTER_RESTRICTIONS, res, nr_res);
|
||||
}
|
||||
|
||||
int io_uring_enable_rings(struct io_uring *ring)
|
||||
{
|
||||
return do_register(ring, IORING_REGISTER_ENABLE_RINGS, NULL, 0);
|
||||
}
|
||||
|
||||
int io_uring_register_iowq_aff(struct io_uring *ring, size_t cpusz,
|
||||
const cpu_set_t *mask)
|
||||
{
|
||||
if (cpusz >= (1U << 31))
|
||||
return -EINVAL;
|
||||
|
||||
return do_register(ring, IORING_REGISTER_IOWQ_AFF, mask, (int) cpusz);
|
||||
}
|
||||
|
||||
int io_uring_unregister_iowq_aff(struct io_uring *ring)
|
||||
{
|
||||
return do_register(ring, IORING_UNREGISTER_IOWQ_AFF, NULL, 0);
|
||||
}
|
||||
|
||||
int io_uring_register_iowq_max_workers(struct io_uring *ring, unsigned int *val)
|
||||
{
|
||||
return do_register(ring, IORING_REGISTER_IOWQ_MAX_WORKERS, val, 2);
|
||||
}
|
||||
|
||||
int io_uring_register_ring_fd(struct io_uring *ring)
|
||||
{
|
||||
struct io_uring_rsrc_update up = {
|
||||
.data = ring->ring_fd,
|
||||
.offset = -1U,
|
||||
};
|
||||
int ret;
|
||||
|
||||
if (ring->int_flags & INT_FLAG_REG_RING)
|
||||
return -EEXIST;
|
||||
|
||||
ret = do_register(ring, IORING_REGISTER_RING_FDS, &up, 1);
|
||||
if (ret == 1) {
|
||||
ring->enter_ring_fd = up.offset;
|
||||
ring->int_flags |= INT_FLAG_REG_RING;
|
||||
if (ring->features & IORING_FEAT_REG_REG_RING) {
|
||||
ring->int_flags |= INT_FLAG_REG_REG_RING;
|
||||
}
|
||||
}
|
||||
return ret;
|
||||
}
|
||||
|
||||
|
||||
int io_uring_unregister_ring_fd(struct io_uring *ring)
|
||||
{
|
||||
struct io_uring_rsrc_update up = {
|
||||
.offset = ring->enter_ring_fd,
|
||||
};
|
||||
int ret;
|
||||
|
||||
if (!(ring->int_flags & INT_FLAG_REG_RING))
|
||||
return -EINVAL;
|
||||
|
||||
ret = do_register(ring, IORING_UNREGISTER_RING_FDS, &up, 1);
|
||||
if (ret == 1) {
|
||||
ring->enter_ring_fd = ring->ring_fd;
|
||||
ring->int_flags &= ~(INT_FLAG_REG_RING | INT_FLAG_REG_REG_RING);
|
||||
}
|
||||
return ret;
|
||||
}
|
||||
|
||||
int io_uring_close_ring_fd(struct io_uring *ring)
|
||||
{
|
||||
if (!(ring->features & IORING_FEAT_REG_REG_RING))
|
||||
return -EOPNOTSUPP;
|
||||
if (!(ring->int_flags & INT_FLAG_REG_RING))
|
||||
return -EINVAL;
|
||||
if (ring->ring_fd == -1)
|
||||
return -EBADF;
|
||||
|
||||
__sys_close(ring->ring_fd);
|
||||
ring->ring_fd = -1;
|
||||
return 1;
|
||||
}
|
||||
|
||||
int io_uring_register_buf_ring(struct io_uring *ring,
|
||||
struct io_uring_buf_reg *reg,
|
||||
unsigned int __maybe_unused flags)
|
||||
{
|
||||
reg->flags |= flags;
|
||||
return do_register(ring, IORING_REGISTER_PBUF_RING, reg, 1);
|
||||
}
|
||||
|
||||
int io_uring_unregister_buf_ring(struct io_uring *ring, int bgid)
|
||||
{
|
||||
struct io_uring_buf_reg reg = { .bgid = bgid };
|
||||
|
||||
return do_register(ring, IORING_UNREGISTER_PBUF_RING, ®, 1);
|
||||
}
|
||||
|
||||
int io_uring_buf_ring_head(struct io_uring *ring, int buf_group, uint16_t *head)
|
||||
{
|
||||
liburing_sanitize_address(head);
|
||||
|
||||
struct io_uring_buf_status buf_status = {
|
||||
.buf_group = buf_group,
|
||||
};
|
||||
int ret;
|
||||
|
||||
ret = do_register(ring, IORING_REGISTER_PBUF_STATUS, &buf_status, 1);
|
||||
if (ret)
|
||||
return ret;
|
||||
*head = buf_status.head;
|
||||
return 0;
|
||||
}
|
||||
|
||||
int io_uring_register_sync_cancel(struct io_uring *ring,
|
||||
struct io_uring_sync_cancel_reg *reg)
|
||||
{
|
||||
return do_register(ring, IORING_REGISTER_SYNC_CANCEL, reg, 1);
|
||||
}
|
||||
|
||||
int io_uring_register_file_alloc_range(struct io_uring *ring,
|
||||
unsigned off, unsigned len)
|
||||
{
|
||||
struct io_uring_file_index_range range = {
|
||||
.off = off,
|
||||
.len = len
|
||||
};
|
||||
|
||||
return do_register(ring, IORING_REGISTER_FILE_ALLOC_RANGE, &range, 0);
|
||||
}
|
||||
|
||||
int io_uring_register_napi(struct io_uring *ring, struct io_uring_napi *napi)
|
||||
{
|
||||
return do_register(ring, IORING_REGISTER_NAPI, napi, 1);
|
||||
}
|
||||
|
||||
int io_uring_unregister_napi(struct io_uring *ring, struct io_uring_napi *napi)
|
||||
{
|
||||
return do_register(ring, IORING_UNREGISTER_NAPI, napi, 1);
|
||||
}
|
||||
|
||||
int io_uring_register_clock(struct io_uring *ring,
|
||||
struct io_uring_clock_register *arg)
|
||||
{
|
||||
return do_register(ring, IORING_REGISTER_CLOCK, arg, 0);
|
||||
}
|
||||
|
||||
int __io_uring_clone_buffers_offset(struct io_uring *dst, struct io_uring *src,
|
||||
unsigned int dst_off, unsigned int src_off,
|
||||
unsigned int nr, unsigned int flags)
|
||||
{
|
||||
struct io_uring_clone_buffers buf = {
|
||||
.src_fd = src->ring_fd,
|
||||
.flags = flags,
|
||||
.src_off = src_off,
|
||||
.dst_off = dst_off,
|
||||
.nr = nr,
|
||||
};
|
||||
|
||||
if (flags & IORING_REGISTER_SRC_REGISTERED &&
|
||||
src->int_flags & INT_FLAG_REG_REG_RING) {
|
||||
buf.src_fd = src->enter_ring_fd;
|
||||
} else {
|
||||
buf.src_fd = src->ring_fd;
|
||||
buf.flags &= ~IORING_REGISTER_SRC_REGISTERED;
|
||||
}
|
||||
|
||||
return do_register(dst, IORING_REGISTER_CLONE_BUFFERS, &buf, 1);
|
||||
}
|
||||
|
||||
int io_uring_clone_buffers_offset(struct io_uring *dst, struct io_uring *src,
|
||||
unsigned int dst_off, unsigned int src_off,
|
||||
unsigned int nr, unsigned int flags)
|
||||
{
|
||||
return __io_uring_clone_buffers_offset(dst, src, dst_off, src_off, nr,
|
||||
flags | IORING_REGISTER_SRC_REGISTERED);
|
||||
}
|
||||
|
||||
int io_uring_clone_buffers(struct io_uring *dst, struct io_uring *src)
|
||||
{
|
||||
return __io_uring_clone_buffers_offset(dst, src, 0, 0, 0, IORING_REGISTER_SRC_REGISTERED);
|
||||
}
|
||||
|
||||
int __io_uring_clone_buffers(struct io_uring *dst, struct io_uring *src,
|
||||
unsigned int flags)
|
||||
{
|
||||
return __io_uring_clone_buffers_offset(dst, src, 0, 0, 0, flags);
|
||||
}
|
||||
|
||||
int io_uring_register_ifq(struct io_uring *ring,
|
||||
struct io_uring_zcrx_ifq_reg *reg)
|
||||
{
|
||||
return do_register(ring, IORING_REGISTER_ZCRX_IFQ, reg, 1);
|
||||
}
|
||||
|
||||
int io_uring_resize_rings(struct io_uring *ring, struct io_uring_params *p)
|
||||
{
|
||||
unsigned sq_head, sq_tail;
|
||||
int ret;
|
||||
|
||||
if (ring->flags & IORING_SETUP_NO_MMAP)
|
||||
return -EINVAL;
|
||||
|
||||
memset(&p->sq_off, 0, sizeof(p->sq_off));
|
||||
memset(&p->cq_off, 0, sizeof(p->cq_off));
|
||||
|
||||
ret = do_register(ring, IORING_REGISTER_RESIZE_RINGS, p, 1);
|
||||
if (ret < 0)
|
||||
goto out;
|
||||
|
||||
sq_head = ring->sq.sqe_head;
|
||||
sq_tail = ring->sq.sqe_tail;
|
||||
io_uring_unmap_rings(&ring->sq, &ring->cq);
|
||||
memset(&ring->sq, 0, sizeof(ring->sq));
|
||||
memset(&ring->cq, 0, sizeof(ring->cq));
|
||||
ret = io_uring_mmap(ring->ring_fd, p, &ring->sq, &ring->cq);
|
||||
if (ret)
|
||||
goto out;
|
||||
|
||||
ring->sq.sqe_head = sq_head;
|
||||
ring->sq.sqe_tail = sq_tail;
|
||||
|
||||
/*
|
||||
* Directly map SQ slots to SQEs
|
||||
*/
|
||||
if (!(p->flags & IORING_SETUP_NO_SQARRAY)) {
|
||||
unsigned index;
|
||||
|
||||
for (index = 0; index < p->sq_entries; index++)
|
||||
ring->sq.array[index] = index;
|
||||
}
|
||||
|
||||
/* clear for next use */
|
||||
out:
|
||||
p->flags = 0;
|
||||
return ret;
|
||||
}
|
||||
|
||||
int io_uring_register_wait_reg(struct io_uring *ring,
|
||||
struct io_uring_reg_wait *reg, int nr)
|
||||
{
|
||||
return -EINVAL;
|
||||
}
|
||||
|
||||
int io_uring_register_region(struct io_uring *ring,
|
||||
struct io_uring_mem_region_reg *reg)
|
||||
{
|
||||
return do_register(ring, IORING_REGISTER_MEM_REGION, reg, 1);
|
||||
}
|
||||
|
||||
int io_uring_set_iowait(struct io_uring *ring, bool enable_iowait)
|
||||
{
|
||||
if (!(ring->features & IORING_FEAT_NO_IOWAIT))
|
||||
return -EOPNOTSUPP;
|
||||
if (enable_iowait)
|
||||
ring->int_flags &= ~INT_FLAG_NO_IOWAIT;
|
||||
else
|
||||
ring->int_flags |= INT_FLAG_NO_IOWAIT;
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,698 @@
|
||||
/* SPDX-License-Identifier: MIT */
|
||||
#define _DEFAULT_SOURCE
|
||||
|
||||
#include "lib.h"
|
||||
#include "syscall.h"
|
||||
#include "liburing.h"
|
||||
#include "int_flags.h"
|
||||
#include "setup.h"
|
||||
#include "liburing/io_uring.h"
|
||||
#include <stdio.h>
|
||||
|
||||
#define KERN_MAX_ENTRIES 32768
|
||||
#define KERN_MAX_CQ_ENTRIES (2 * KERN_MAX_ENTRIES)
|
||||
|
||||
static inline int __fls(int x)
|
||||
{
|
||||
if (!x)
|
||||
return 0;
|
||||
return 8 * sizeof(x) - __builtin_clz(x);
|
||||
}
|
||||
|
||||
static unsigned roundup_pow2(unsigned depth)
|
||||
{
|
||||
return 1U << __fls(depth - 1);
|
||||
}
|
||||
|
||||
static int get_sq_cq_entries(unsigned entries, struct io_uring_params *p,
|
||||
unsigned *sq, unsigned *cq)
|
||||
{
|
||||
unsigned cq_entries;
|
||||
|
||||
if (!entries)
|
||||
return -EINVAL;
|
||||
if (entries > KERN_MAX_ENTRIES) {
|
||||
if (!(p->flags & IORING_SETUP_CLAMP))
|
||||
return -EINVAL;
|
||||
entries = KERN_MAX_ENTRIES;
|
||||
}
|
||||
|
||||
entries = roundup_pow2(entries);
|
||||
if (p->flags & IORING_SETUP_CQSIZE) {
|
||||
if (!p->cq_entries)
|
||||
return -EINVAL;
|
||||
cq_entries = p->cq_entries;
|
||||
if (cq_entries > KERN_MAX_CQ_ENTRIES) {
|
||||
if (!(p->flags & IORING_SETUP_CLAMP))
|
||||
return -EINVAL;
|
||||
cq_entries = KERN_MAX_CQ_ENTRIES;
|
||||
}
|
||||
cq_entries = roundup_pow2(cq_entries);
|
||||
if (cq_entries < entries)
|
||||
return -EINVAL;
|
||||
} else {
|
||||
cq_entries = 2 * entries;
|
||||
}
|
||||
|
||||
*sq = entries;
|
||||
*cq = cq_entries;
|
||||
return 0;
|
||||
}
|
||||
|
||||
void io_uring_unmap_rings(struct io_uring_sq *sq, struct io_uring_cq *cq)
|
||||
{
|
||||
if (sq->ring_sz)
|
||||
__sys_munmap(sq->ring_ptr, sq->ring_sz);
|
||||
if (cq->ring_ptr && cq->ring_sz && cq->ring_ptr != sq->ring_ptr)
|
||||
__sys_munmap(cq->ring_ptr, cq->ring_sz);
|
||||
}
|
||||
|
||||
void io_uring_setup_ring_pointers(struct io_uring_params *p,
|
||||
struct io_uring_sq *sq,
|
||||
struct io_uring_cq *cq)
|
||||
{
|
||||
sq->khead = sq->ring_ptr + p->sq_off.head;
|
||||
sq->ktail = sq->ring_ptr + p->sq_off.tail;
|
||||
sq->kring_mask = sq->ring_ptr + p->sq_off.ring_mask;
|
||||
sq->kring_entries = sq->ring_ptr + p->sq_off.ring_entries;
|
||||
sq->kflags = sq->ring_ptr + p->sq_off.flags;
|
||||
sq->kdropped = sq->ring_ptr + p->sq_off.dropped;
|
||||
if (!(p->flags & IORING_SETUP_NO_SQARRAY))
|
||||
sq->array = sq->ring_ptr + p->sq_off.array;
|
||||
|
||||
cq->khead = cq->ring_ptr + p->cq_off.head;
|
||||
cq->ktail = cq->ring_ptr + p->cq_off.tail;
|
||||
cq->kring_mask = cq->ring_ptr + p->cq_off.ring_mask;
|
||||
cq->kring_entries = cq->ring_ptr + p->cq_off.ring_entries;
|
||||
cq->koverflow = cq->ring_ptr + p->cq_off.overflow;
|
||||
cq->cqes = cq->ring_ptr + p->cq_off.cqes;
|
||||
if (p->cq_off.flags)
|
||||
cq->kflags = cq->ring_ptr + p->cq_off.flags;
|
||||
|
||||
sq->ring_mask = *sq->kring_mask;
|
||||
sq->ring_entries = *sq->kring_entries;
|
||||
cq->ring_mask = *cq->kring_mask;
|
||||
cq->ring_entries = *cq->kring_entries;
|
||||
}
|
||||
|
||||
static size_t params_sqes_size(const struct io_uring_params *p, unsigned sqes)
|
||||
{
|
||||
sqes <<= io_uring_sqe_shift_from_flags(p->flags);
|
||||
return sqes * sizeof(struct io_uring_sqe);
|
||||
}
|
||||
|
||||
static size_t params_cq_size(const struct io_uring_params *p, unsigned cqes)
|
||||
{
|
||||
cqes <<= io_uring_cqe_shift_from_flags(p->flags);
|
||||
return cqes * sizeof(struct io_uring_cqe);
|
||||
}
|
||||
|
||||
int io_uring_mmap(int fd, struct io_uring_params *p, struct io_uring_sq *sq,
|
||||
struct io_uring_cq *cq)
|
||||
{
|
||||
int ret;
|
||||
|
||||
sq->ring_sz = p->sq_off.array + p->sq_entries * sizeof(unsigned);
|
||||
cq->ring_sz = p->cq_off.cqes + params_cq_size(p, p->cq_entries);
|
||||
|
||||
if (p->features & IORING_FEAT_SINGLE_MMAP) {
|
||||
if (cq->ring_sz > sq->ring_sz)
|
||||
sq->ring_sz = cq->ring_sz;
|
||||
cq->ring_sz = sq->ring_sz;
|
||||
}
|
||||
sq->ring_ptr = __sys_mmap(0, sq->ring_sz, PROT_READ | PROT_WRITE,
|
||||
MAP_SHARED | MAP_POPULATE, fd,
|
||||
IORING_OFF_SQ_RING);
|
||||
if (IS_ERR(sq->ring_ptr))
|
||||
return PTR_ERR(sq->ring_ptr);
|
||||
|
||||
if (p->features & IORING_FEAT_SINGLE_MMAP) {
|
||||
cq->ring_ptr = sq->ring_ptr;
|
||||
} else {
|
||||
cq->ring_ptr = __sys_mmap(0, cq->ring_sz, PROT_READ | PROT_WRITE,
|
||||
MAP_SHARED | MAP_POPULATE, fd,
|
||||
IORING_OFF_CQ_RING);
|
||||
if (IS_ERR(cq->ring_ptr)) {
|
||||
ret = PTR_ERR(cq->ring_ptr);
|
||||
cq->ring_ptr = NULL;
|
||||
goto err;
|
||||
}
|
||||
}
|
||||
|
||||
sq->sqes = __sys_mmap(0, params_sqes_size(p, p->sq_entries),
|
||||
PROT_READ | PROT_WRITE, MAP_SHARED | MAP_POPULATE,
|
||||
fd, IORING_OFF_SQES);
|
||||
if (IS_ERR(sq->sqes)) {
|
||||
ret = PTR_ERR(sq->sqes);
|
||||
err:
|
||||
io_uring_unmap_rings(sq, cq);
|
||||
return ret;
|
||||
}
|
||||
|
||||
io_uring_setup_ring_pointers(p, sq, cq);
|
||||
return 0;
|
||||
}
|
||||
|
||||
/*
|
||||
* For users that want to specify sq_thread_cpu or sq_thread_idle, this
|
||||
* interface is a convenient helper for mmap()ing the rings.
|
||||
* Returns -errno on error, or zero on success. On success, 'ring'
|
||||
* contains the necessary information to read/write to the rings.
|
||||
*/
|
||||
__cold int io_uring_queue_mmap(int fd, struct io_uring_params *p,
|
||||
struct io_uring *ring)
|
||||
{
|
||||
memset(ring, 0, sizeof(*ring));
|
||||
return io_uring_mmap(fd, p, &ring->sq, &ring->cq);
|
||||
}
|
||||
|
||||
static size_t io_uring_sqes_size(const struct io_uring *ring)
|
||||
{
|
||||
return (ring->sq.ring_entries << io_uring_sqe_shift(ring)) *
|
||||
sizeof(struct io_uring_sqe);
|
||||
}
|
||||
|
||||
/*
|
||||
* Ensure that the mmap'ed rings aren't available to a child after a fork(2).
|
||||
* This uses madvise(..., MADV_DONTFORK) on the mmap'ed ranges.
|
||||
*/
|
||||
__cold int io_uring_ring_dontfork(struct io_uring *ring)
|
||||
{
|
||||
size_t len;
|
||||
int ret;
|
||||
|
||||
if (!ring->sq.ring_ptr || !ring->sq.sqes || !ring->cq.ring_ptr)
|
||||
return -EINVAL;
|
||||
|
||||
len = io_uring_sqes_size(ring);
|
||||
ret = __sys_madvise(ring->sq.sqes, len, MADV_DONTFORK);
|
||||
if (ret < 0)
|
||||
return ret;
|
||||
|
||||
len = ring->sq.ring_sz;
|
||||
ret = __sys_madvise(ring->sq.ring_ptr, len, MADV_DONTFORK);
|
||||
if (ret < 0)
|
||||
return ret;
|
||||
|
||||
if (ring->cq.ring_ptr != ring->sq.ring_ptr) {
|
||||
len = ring->cq.ring_sz;
|
||||
ret = __sys_madvise(ring->cq.ring_ptr, len, MADV_DONTFORK);
|
||||
if (ret < 0)
|
||||
return ret;
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* FIXME */
|
||||
static size_t huge_page_size = 2 * 1024 * 1024;
|
||||
|
||||
#define KRING_SIZE 64
|
||||
|
||||
/*
|
||||
* Returns negative for error, or number of bytes used in the buffer on success
|
||||
*/
|
||||
static int io_uring_alloc_huge(unsigned entries, struct io_uring_params *p,
|
||||
struct io_uring_sq *sq, struct io_uring_cq *cq,
|
||||
void *buf, size_t buf_size)
|
||||
{
|
||||
unsigned long page_size = get_page_size();
|
||||
unsigned sq_entries, cq_entries;
|
||||
size_t ring_mem, sqes_mem;
|
||||
unsigned long mem_used = 0;
|
||||
void *ptr;
|
||||
int ret;
|
||||
|
||||
ret = get_sq_cq_entries(entries, p, &sq_entries, &cq_entries);
|
||||
if (ret)
|
||||
return ret;
|
||||
|
||||
ring_mem = KRING_SIZE;
|
||||
|
||||
sqes_mem = params_sqes_size(p, sq_entries);
|
||||
if (!(p->flags & IORING_SETUP_NO_SQARRAY))
|
||||
sqes_mem += sq_entries * sizeof(unsigned);
|
||||
sqes_mem = (sqes_mem + page_size - 1) & ~(page_size - 1);
|
||||
|
||||
ring_mem += sqes_mem + params_cq_size(p, cq_entries);
|
||||
mem_used = ring_mem;
|
||||
mem_used = (mem_used + page_size - 1) & ~(page_size - 1);
|
||||
|
||||
/*
|
||||
* A maxed-out number of CQ entries with IORING_SETUP_CQE32 fills a 2MB
|
||||
* huge page by itself, so the SQ entries won't fit in the same huge
|
||||
* page. For SQEs, that shouldn't be possible given KERN_MAX_ENTRIES,
|
||||
* but check that too to future-proof (e.g. against different huge page
|
||||
* sizes). Bail out early so we don't overrun.
|
||||
*/
|
||||
if (!buf && (sqes_mem > huge_page_size || ring_mem > huge_page_size))
|
||||
return -ENOMEM;
|
||||
|
||||
if (buf) {
|
||||
if (mem_used > buf_size)
|
||||
return -ENOMEM;
|
||||
ptr = buf;
|
||||
} else {
|
||||
int map_hugetlb = 0;
|
||||
if (sqes_mem <= page_size)
|
||||
buf_size = page_size;
|
||||
else {
|
||||
buf_size = huge_page_size;
|
||||
map_hugetlb = MAP_HUGETLB;
|
||||
}
|
||||
ptr = __sys_mmap(NULL, buf_size, PROT_READ|PROT_WRITE,
|
||||
MAP_SHARED|MAP_ANONYMOUS|map_hugetlb,
|
||||
-1, 0);
|
||||
if (IS_ERR(ptr))
|
||||
return PTR_ERR(ptr);
|
||||
}
|
||||
|
||||
sq->sqes = ptr;
|
||||
if (mem_used <= buf_size) {
|
||||
sq->ring_ptr = (void *) sq->sqes + sqes_mem;
|
||||
/* clear ring sizes, we have just one mmap() to undo */
|
||||
cq->ring_sz = 0;
|
||||
sq->ring_sz = 0;
|
||||
} else {
|
||||
int map_hugetlb = 0;
|
||||
if (ring_mem <= page_size)
|
||||
buf_size = page_size;
|
||||
else {
|
||||
buf_size = huge_page_size;
|
||||
map_hugetlb = MAP_HUGETLB;
|
||||
}
|
||||
ptr = __sys_mmap(NULL, buf_size, PROT_READ|PROT_WRITE,
|
||||
MAP_SHARED|MAP_ANONYMOUS|map_hugetlb,
|
||||
-1, 0);
|
||||
if (IS_ERR(ptr)) {
|
||||
__sys_munmap(sq->sqes, 1);
|
||||
return PTR_ERR(ptr);
|
||||
}
|
||||
sq->ring_ptr = ptr;
|
||||
sq->ring_sz = buf_size;
|
||||
cq->ring_sz = 0;
|
||||
}
|
||||
|
||||
cq->ring_ptr = (void *) sq->ring_ptr;
|
||||
p->sq_off.user_addr = (unsigned long) sq->sqes;
|
||||
p->cq_off.user_addr = (unsigned long) sq->ring_ptr;
|
||||
return (int) mem_used;
|
||||
}
|
||||
|
||||
int __io_uring_queue_init_params(unsigned entries, struct io_uring *ring,
|
||||
struct io_uring_params *p, void *buf,
|
||||
size_t buf_size)
|
||||
{
|
||||
int fd, ret = 0;
|
||||
unsigned *sq_array;
|
||||
unsigned sq_entries, index;
|
||||
|
||||
memset(ring, 0, sizeof(*ring));
|
||||
|
||||
/*
|
||||
* The kernel does this check already, but checking it here allows us
|
||||
* to avoid handling it below.
|
||||
*/
|
||||
if (p->flags & IORING_SETUP_REGISTERED_FD_ONLY
|
||||
&& !(p->flags & IORING_SETUP_NO_MMAP))
|
||||
return -EINVAL;
|
||||
|
||||
if (p->flags & IORING_SETUP_NO_MMAP) {
|
||||
ret = io_uring_alloc_huge(entries, p, &ring->sq, &ring->cq,
|
||||
buf, buf_size);
|
||||
if (ret < 0)
|
||||
return ret;
|
||||
if (buf)
|
||||
ring->int_flags |= INT_FLAG_APP_MEM;
|
||||
}
|
||||
|
||||
fd = __sys_io_uring_setup(entries, p);
|
||||
if (fd < 0) {
|
||||
if ((p->flags & IORING_SETUP_NO_MMAP) &&
|
||||
!(ring->int_flags & INT_FLAG_APP_MEM)) {
|
||||
__sys_munmap(ring->sq.sqes, 1);
|
||||
io_uring_unmap_rings(&ring->sq, &ring->cq);
|
||||
}
|
||||
return fd;
|
||||
}
|
||||
|
||||
if (!(p->flags & IORING_SETUP_NO_MMAP)) {
|
||||
ret = io_uring_queue_mmap(fd, p, ring);
|
||||
if (ret) {
|
||||
__sys_close(fd);
|
||||
return ret;
|
||||
}
|
||||
} else {
|
||||
io_uring_setup_ring_pointers(p, &ring->sq, &ring->cq);
|
||||
}
|
||||
|
||||
/*
|
||||
* Directly map SQ slots to SQEs
|
||||
*/
|
||||
sq_entries = ring->sq.ring_entries;
|
||||
|
||||
if (!(p->flags & IORING_SETUP_NO_SQARRAY)) {
|
||||
sq_array = ring->sq.array;
|
||||
for (index = 0; index < sq_entries; index++)
|
||||
sq_array[index] = index;
|
||||
}
|
||||
ring->features = p->features;
|
||||
ring->flags = p->flags;
|
||||
ring->enter_ring_fd = fd;
|
||||
if (p->flags & IORING_SETUP_REGISTERED_FD_ONLY) {
|
||||
ring->ring_fd = -1;
|
||||
ring->int_flags |= INT_FLAG_REG_RING | INT_FLAG_REG_REG_RING;
|
||||
} else {
|
||||
ring->ring_fd = fd;
|
||||
}
|
||||
/*
|
||||
* IOPOLL always needs to enter, except if SQPOLL is set as well.
|
||||
* Use an internal flag to check for this.
|
||||
*/
|
||||
if ((ring->flags & (IORING_SETUP_IOPOLL|IORING_SETUP_SQPOLL)) ==
|
||||
IORING_SETUP_IOPOLL)
|
||||
ring->int_flags |= INT_FLAG_CQ_ENTER;
|
||||
|
||||
return ret;
|
||||
}
|
||||
|
||||
static int io_uring_queue_init_try_nosqarr(unsigned entries, struct io_uring *ring,
|
||||
struct io_uring_params *p, void *buf,
|
||||
size_t buf_size)
|
||||
{
|
||||
unsigned flags = p->flags;
|
||||
int ret;
|
||||
|
||||
p->flags |= IORING_SETUP_NO_SQARRAY;
|
||||
ret = __io_uring_queue_init_params(entries, ring, p, buf, buf_size);
|
||||
|
||||
/* don't fallback if explicitly asked for NOSQARRAY */
|
||||
if (ret != -EINVAL || (flags & IORING_SETUP_NO_SQARRAY))
|
||||
return ret;
|
||||
|
||||
p->flags = flags;
|
||||
return __io_uring_queue_init_params(entries, ring, p, buf, buf_size);
|
||||
}
|
||||
|
||||
/*
|
||||
* Like io_uring_queue_init_params(), except it allows the application to pass
|
||||
* in a pre-allocated memory range that is used for the shared data between
|
||||
* the kernel and the application. This includes the sqes array, and the two
|
||||
* rings. The memory must be contiguous, the use case here is that the app
|
||||
* allocates a huge page and passes it in.
|
||||
*
|
||||
* Returns the number of bytes used in the buffer, the app can then reuse
|
||||
* the buffer with the returned offset to put more rings in the same huge
|
||||
* page. Returns -ENOMEM if there's not enough room left in the buffer to
|
||||
* host the ring.
|
||||
*/
|
||||
int io_uring_queue_init_mem(unsigned entries, struct io_uring *ring,
|
||||
struct io_uring_params *p,
|
||||
void *buf, size_t buf_size)
|
||||
{
|
||||
/* should already be set... */
|
||||
p->flags |= IORING_SETUP_NO_MMAP;
|
||||
return io_uring_queue_init_try_nosqarr(entries, ring, p, buf, buf_size);
|
||||
}
|
||||
|
||||
int io_uring_queue_init_params(unsigned entries, struct io_uring *ring,
|
||||
struct io_uring_params *p)
|
||||
{
|
||||
int ret;
|
||||
|
||||
ret = io_uring_queue_init_try_nosqarr(entries, ring, p, NULL, 0);
|
||||
return ret >= 0 ? 0 : ret;
|
||||
}
|
||||
|
||||
/*
|
||||
* Returns -errno on error, or zero on success. On success, 'ring'
|
||||
* contains the necessary information to read/write to the rings.
|
||||
*/
|
||||
__cold int io_uring_queue_init(unsigned entries, struct io_uring *ring,
|
||||
unsigned flags)
|
||||
{
|
||||
struct io_uring_params p;
|
||||
|
||||
memset(&p, 0, sizeof(p));
|
||||
p.flags = flags;
|
||||
|
||||
return io_uring_queue_init_params(entries, ring, &p);
|
||||
}
|
||||
|
||||
__cold void io_uring_queue_exit(struct io_uring *ring)
|
||||
{
|
||||
struct io_uring_sq *sq = &ring->sq;
|
||||
struct io_uring_cq *cq = &ring->cq;
|
||||
|
||||
if (!(ring->int_flags & INT_FLAG_APP_MEM)) {
|
||||
__sys_munmap(sq->sqes, io_uring_sqes_size(ring));
|
||||
io_uring_unmap_rings(sq, cq);
|
||||
}
|
||||
|
||||
/*
|
||||
* Not strictly required, but frees up the slot we used now rather
|
||||
* than at process exit time.
|
||||
*/
|
||||
if (ring->int_flags & INT_FLAG_REG_RING)
|
||||
io_uring_unregister_ring_fd(ring);
|
||||
if (ring->ring_fd != -1)
|
||||
__sys_close(ring->ring_fd);
|
||||
}
|
||||
|
||||
__cold struct io_uring_probe *io_uring_get_probe_ring(struct io_uring *ring)
|
||||
{
|
||||
struct io_uring_probe *probe;
|
||||
size_t len;
|
||||
int r;
|
||||
|
||||
len = sizeof(*probe) + 256 * sizeof(struct io_uring_probe_op);
|
||||
probe = malloc(len);
|
||||
if (!probe)
|
||||
return NULL;
|
||||
memset(probe, 0, len);
|
||||
|
||||
r = io_uring_register_probe(ring, probe, 256);
|
||||
if (r >= 0)
|
||||
return probe;
|
||||
|
||||
free(probe);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
__cold struct io_uring_probe *io_uring_get_probe(void)
|
||||
{
|
||||
struct io_uring ring;
|
||||
struct io_uring_probe *probe;
|
||||
int r;
|
||||
|
||||
r = io_uring_queue_init(2, &ring, 0);
|
||||
if (r < 0)
|
||||
return NULL;
|
||||
|
||||
probe = io_uring_get_probe_ring(&ring);
|
||||
io_uring_queue_exit(&ring);
|
||||
return probe;
|
||||
}
|
||||
|
||||
__cold void io_uring_free_probe(struct io_uring_probe *probe)
|
||||
{
|
||||
free(probe);
|
||||
}
|
||||
|
||||
static size_t rings_size(struct io_uring_params *p, unsigned entries,
|
||||
unsigned cq_entries, long page_size)
|
||||
{
|
||||
size_t pages, sq_size, cq_size;
|
||||
|
||||
/*
|
||||
* CQ ring size is number of pages that we need for the
|
||||
* struct io_uring_cqe entries, which may be 16b (default) or
|
||||
* 32b if the ring is setup with IORING_SETUP_CQE32. We also need
|
||||
* room for the head/tail parts.
|
||||
*/
|
||||
cq_size = params_cq_size(p, cq_entries);
|
||||
cq_size += KRING_SIZE;
|
||||
cq_size = (cq_size + page_size - 1) & ~(page_size - 1);
|
||||
pages = (size_t) cq_size / page_size;
|
||||
|
||||
sq_size = params_sqes_size(p, entries);
|
||||
sq_size = (sq_size + page_size - 1) & ~(page_size - 1);
|
||||
pages += sq_size / page_size;
|
||||
return pages * page_size;
|
||||
}
|
||||
|
||||
ssize_t io_uring_memory_size_params(unsigned entries, struct io_uring_params *p)
|
||||
{
|
||||
unsigned sq, cq;
|
||||
long page_size;
|
||||
ssize_t ret;
|
||||
|
||||
if (!entries)
|
||||
return -EINVAL;
|
||||
if (entries > KERN_MAX_ENTRIES) {
|
||||
if (!(p->flags & IORING_SETUP_CLAMP))
|
||||
return -EINVAL;
|
||||
entries = KERN_MAX_ENTRIES;
|
||||
}
|
||||
|
||||
ret = get_sq_cq_entries(entries, p, &sq, &cq);
|
||||
if (ret)
|
||||
return ret;
|
||||
|
||||
page_size = get_page_size();
|
||||
return rings_size(p, sq, cq, page_size);
|
||||
}
|
||||
|
||||
ssize_t io_uring_memory_size(unsigned entries, unsigned ring_flags)
|
||||
{
|
||||
struct io_uring_params p = { .flags = ring_flags, };
|
||||
|
||||
return io_uring_memory_size_params(entries, &p);
|
||||
}
|
||||
|
||||
/*
|
||||
* Return the required ulimit -l memlock memory required for a given ring
|
||||
* setup, in bytes. May return -errno on error. On newer (5.12+) kernels,
|
||||
* io_uring no longer requires any memlock memory, and hence this function
|
||||
* will return 0 for that case. On older (5.11 and prior) kernels, this will
|
||||
* return the required memory so that the caller can ensure that enough space
|
||||
* is available before setting up a ring with the specified parameters.
|
||||
*/
|
||||
__cold ssize_t io_uring_mlock_size_params(unsigned entries,
|
||||
struct io_uring_params *p)
|
||||
{
|
||||
struct io_uring_params lp;
|
||||
struct io_uring ring;
|
||||
ssize_t ret;
|
||||
|
||||
memset(&lp, 0, sizeof(lp));
|
||||
|
||||
/*
|
||||
* We only really use this inited ring to see if the kernel is newer
|
||||
* or not. Newer kernels don't require memlocked memory. If we fail,
|
||||
* it's most likely because it's an older kernel and we have no
|
||||
* available memlock space. Just continue on, lp.features will still
|
||||
* be zeroed at this point and we'll do the right thing.
|
||||
*/
|
||||
ret = io_uring_queue_init_params(entries, &ring, &lp);
|
||||
if (!ret)
|
||||
io_uring_queue_exit(&ring);
|
||||
|
||||
/*
|
||||
* Native workers imply using cgroup memory accounting, and hence no
|
||||
* memlock memory is needed for the ring allocations.
|
||||
*/
|
||||
if (lp.features & IORING_FEAT_NATIVE_WORKERS)
|
||||
return 0;
|
||||
|
||||
return io_uring_memory_size_params(entries, p);
|
||||
}
|
||||
|
||||
/*
|
||||
* Return required ulimit -l memory space for a given ring setup. See
|
||||
* @io_uring_mlock_size_params().
|
||||
*/
|
||||
__cold ssize_t io_uring_mlock_size(unsigned entries, unsigned flags)
|
||||
{
|
||||
struct io_uring_params p;
|
||||
|
||||
memset(&p, 0, sizeof(p));
|
||||
p.flags = flags;
|
||||
return io_uring_mlock_size_params(entries, &p);
|
||||
}
|
||||
|
||||
#if defined(__hppa__)
|
||||
static struct io_uring_buf_ring *br_setup(struct io_uring *ring,
|
||||
unsigned int nentries, int bgid,
|
||||
unsigned int flags, int *err)
|
||||
{
|
||||
struct io_uring_buf_ring *br;
|
||||
struct io_uring_buf_reg reg;
|
||||
size_t ring_size;
|
||||
off_t off;
|
||||
int lret;
|
||||
|
||||
memset(®, 0, sizeof(reg));
|
||||
reg.ring_entries = nentries;
|
||||
reg.bgid = bgid;
|
||||
reg.flags = IOU_PBUF_RING_MMAP;
|
||||
|
||||
*err = 0;
|
||||
lret = io_uring_register_buf_ring(ring, ®, flags);
|
||||
if (lret) {
|
||||
*err = lret;
|
||||
return NULL;
|
||||
}
|
||||
|
||||
off = IORING_OFF_PBUF_RING | (unsigned long long) bgid << IORING_OFF_PBUF_SHIFT;
|
||||
ring_size = nentries * sizeof(struct io_uring_buf);
|
||||
br = __sys_mmap(NULL, ring_size, PROT_READ | PROT_WRITE,
|
||||
MAP_SHARED | MAP_POPULATE, ring->ring_fd, off);
|
||||
if (IS_ERR(br)) {
|
||||
*err = PTR_ERR(br);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
return br;
|
||||
}
|
||||
#else
|
||||
static struct io_uring_buf_ring *br_setup(struct io_uring *ring,
|
||||
unsigned int nentries, int bgid,
|
||||
unsigned int flags, int *err)
|
||||
{
|
||||
struct io_uring_buf_ring *br;
|
||||
struct io_uring_buf_reg reg;
|
||||
size_t ring_size;
|
||||
int lret;
|
||||
|
||||
memset(®, 0, sizeof(reg));
|
||||
ring_size = nentries * sizeof(struct io_uring_buf);
|
||||
br = __sys_mmap(NULL, ring_size, PROT_READ | PROT_WRITE,
|
||||
MAP_ANONYMOUS | MAP_PRIVATE, -1, 0);
|
||||
if (IS_ERR(br)) {
|
||||
*err = PTR_ERR(br);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
reg.ring_addr = (unsigned long) (uintptr_t) br;
|
||||
reg.ring_entries = nentries;
|
||||
reg.bgid = bgid;
|
||||
|
||||
*err = 0;
|
||||
lret = io_uring_register_buf_ring(ring, ®, flags);
|
||||
if (lret) {
|
||||
__sys_munmap(br, ring_size);
|
||||
*err = lret;
|
||||
br = NULL;
|
||||
}
|
||||
|
||||
return br;
|
||||
}
|
||||
#endif
|
||||
|
||||
struct io_uring_buf_ring *io_uring_setup_buf_ring(struct io_uring *ring,
|
||||
unsigned int nentries,
|
||||
int bgid, unsigned int flags,
|
||||
int *err)
|
||||
{
|
||||
struct io_uring_buf_ring *br;
|
||||
|
||||
br = br_setup(ring, nentries, bgid, flags, err);
|
||||
if (br)
|
||||
io_uring_buf_ring_init(br);
|
||||
|
||||
return br;
|
||||
}
|
||||
|
||||
int io_uring_free_buf_ring(struct io_uring *ring, struct io_uring_buf_ring *br,
|
||||
unsigned int nentries, int bgid)
|
||||
{
|
||||
int ret;
|
||||
|
||||
ret = io_uring_unregister_buf_ring(ring, bgid);
|
||||
if (ret)
|
||||
return ret;
|
||||
|
||||
__sys_munmap(br, nentries * sizeof(struct io_uring_buf));
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,15 @@
|
||||
/* SPDX-License-Identifier: MIT */
|
||||
#ifndef LIBURING_SETUP_H
|
||||
#define LIBURING_SETUP_H
|
||||
|
||||
int __io_uring_queue_init_params(unsigned entries, struct io_uring *ring,
|
||||
struct io_uring_params *p, void *buf,
|
||||
size_t buf_size);
|
||||
void io_uring_unmap_rings(struct io_uring_sq *sq, struct io_uring_cq *cq);
|
||||
int io_uring_mmap(int fd, struct io_uring_params *p, struct io_uring_sq *sq,
|
||||
struct io_uring_cq *cq);
|
||||
void io_uring_setup_ring_pointers(struct io_uring_params *p,
|
||||
struct io_uring_sq *sq,
|
||||
struct io_uring_cq *cq);
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1,29 @@
|
||||
/* SPDX-License-Identifier: MIT */
|
||||
|
||||
#include "syscall.h"
|
||||
#include <liburing.h>
|
||||
|
||||
int io_uring_enter(unsigned int fd, unsigned int to_submit,
|
||||
unsigned int min_complete, unsigned int flags, sigset_t *sig)
|
||||
{
|
||||
return __sys_io_uring_enter(fd, to_submit, min_complete, flags, sig);
|
||||
}
|
||||
|
||||
int io_uring_enter2(unsigned int fd, unsigned int to_submit,
|
||||
unsigned int min_complete, unsigned int flags,
|
||||
void *arg, size_t sz)
|
||||
{
|
||||
return __sys_io_uring_enter2(fd, to_submit, min_complete, flags, arg,
|
||||
sz);
|
||||
}
|
||||
|
||||
int io_uring_setup(unsigned int entries, struct io_uring_params *p)
|
||||
{
|
||||
return __sys_io_uring_setup(entries, p);
|
||||
}
|
||||
|
||||
int io_uring_register(unsigned int fd, unsigned int opcode, const void *arg,
|
||||
unsigned int nr_args)
|
||||
{
|
||||
return __sys_io_uring_register(fd, opcode, arg, nr_args);
|
||||
}
|
||||
@@ -0,0 +1,53 @@
|
||||
/* SPDX-License-Identifier: MIT */
|
||||
#ifndef LIBURING_SYSCALL_H
|
||||
#define LIBURING_SYSCALL_H
|
||||
|
||||
#include <errno.h>
|
||||
#include <signal.h>
|
||||
#include <stdint.h>
|
||||
#include <unistd.h>
|
||||
#include <stdbool.h>
|
||||
#include <sys/mman.h>
|
||||
#include <sys/syscall.h>
|
||||
#include <sys/resource.h>
|
||||
#include <liburing.h>
|
||||
|
||||
/*
|
||||
* Don't put this below the #include "arch/$arch/syscall.h", that
|
||||
* file may need it.
|
||||
*/
|
||||
struct io_uring_params;
|
||||
|
||||
static inline void *ERR_PTR(intptr_t n)
|
||||
{
|
||||
return (void *) n;
|
||||
}
|
||||
|
||||
static inline int PTR_ERR(const void *ptr)
|
||||
{
|
||||
return (int) (intptr_t) ptr;
|
||||
}
|
||||
|
||||
static inline bool IS_ERR(const void *ptr)
|
||||
{
|
||||
return uring_unlikely((uintptr_t) ptr >= (uintptr_t) -4095UL);
|
||||
}
|
||||
|
||||
#if defined(__x86_64__) || defined(__i386__)
|
||||
#include "arch/x86/syscall.h"
|
||||
#elif defined(__aarch64__)
|
||||
#include "arch/aarch64/syscall.h"
|
||||
#elif defined(__riscv) && __riscv_xlen == 64
|
||||
#include "arch/riscv64/syscall.h"
|
||||
#else
|
||||
/*
|
||||
* We don't have native syscall wrappers
|
||||
* for this arch. Must use libc!
|
||||
*/
|
||||
#ifdef CONFIG_NOLIBC
|
||||
#error "This arch doesn't support building liburing without libc"
|
||||
#endif
|
||||
/* libc syscall wrappers. */
|
||||
#include "arch/generic/syscall.h"
|
||||
#endif
|
||||
#endif
|
||||
@@ -0,0 +1,21 @@
|
||||
/* SPDX-License-Identifier: MIT */
|
||||
|
||||
#include "liburing.h"
|
||||
#include "liburing/io_uring_version.h"
|
||||
|
||||
int io_uring_major_version(void)
|
||||
{
|
||||
return IO_URING_VERSION_MAJOR;
|
||||
}
|
||||
|
||||
int io_uring_minor_version(void)
|
||||
{
|
||||
return IO_URING_VERSION_MINOR;
|
||||
}
|
||||
|
||||
bool io_uring_check_version(int major, int minor)
|
||||
{
|
||||
return major > io_uring_major_version() ||
|
||||
(major == io_uring_major_version() &&
|
||||
minor > io_uring_minor_version());
|
||||
}
|
||||
Reference in New Issue
Block a user