diff --git a/docs/README.md b/docs/README.md index ed63d60e..9c015a75 100644 --- a/docs/README.md +++ b/docs/README.md @@ -58,6 +58,7 @@ User-facing MeTTa skills the agent invokes. Each page follows the template **Sig - [reference-channels.md](./reference-channels.md) — IRC, Telegram, Slack, Mattermost, WebSocket, and websearch adapters plus the channel contract - [reference-python-bridges.md](./reference-python-bridges.md) — `lib_llm_ext.py`, `src/helper.py`, `src/skills.pl` - [reference-memory-portability.md](./reference-memory-portability.md) — Operator backup, restore, and archive-transfer workflow +- [reference-gpu.md](./reference-gpu.md) - Experimental NVIDIA GPU access: host setup, building and running a GPU-enabled container ### Plugin API diff --git a/docs/reference-gpu.md b/docs/reference-gpu.md new file mode 100644 index 00000000..50bf7c8e --- /dev/null +++ b/docs/reference-gpu.md @@ -0,0 +1,107 @@ +# Reference - GPU Support (Experimental) + +> **Experimental.** GPU support is under testing and is not included in Omega +> releases yet; it is planned for a future release. Released images and +> `scripts/omega start` do not give the agent GPU access. To try it, build the +> image from the `devices-landlock-fix` branch and start the container with +> your own `docker run` command, as described below. + +## Host setup + +### Native Linux + +1. Install the NVIDIA driver for your distribution and reboot. `nvidia-smi -L` + on the host must list your GPUs. +2. Install the + [NVIDIA Container Toolkit](https://docs.nvidia.com/datacenter/cloud-native/container-toolkit/latest/install-guide.html). +3. Register the toolkit with Docker and restart Docker: + + ```sh + sudo nvidia-ctk runtime configure --runtime=docker + sudo systemctl restart docker + ``` + +4. Check that CDI is enabled in Docker: `docker info --format '{{json .CDISpecDirs}}'` + must print a non-empty list. If it does not, enable it and restart Docker: + + ```sh + sudo nvidia-ctk runtime configure --runtime=docker --cdi.enabled=true + sudo systemctl restart docker + ``` + +5. Check the CDI specification: `nvidia-ctk cdi list` must print + `nvidia.com/gpu=...` entries. Recent toolkit versions keep it up to date with + `nvidia-cdi-refresh.service`; otherwise generate it: + + ```sh + sudo nvidia-ctk cdi generate --output=/var/run/cdi/nvidia.yaml + ``` + + Regenerate the specification after every driver update. + +### Windows with WSL2 + +1. Install the NVIDIA driver on Windows. Do not install a Linux NVIDIA driver + inside the WSL distribution: WSL exposes the Windows driver as `/dev/dxg` + and the libraries in `/usr/lib/wsl/lib`. +2. Install Docker Engine and the NVIDIA Container Toolkit inside the WSL + distribution and follow steps 2-5 of the native Linux setup. + +### SELinux (Fedora, RHEL and derivatives) + +When Docker runs with SELinux labeling, containers may be denied access to the +GPU device nodes even though the nodes exist, and `nvidia-smi` fails inside the +container. Allow containers to use devices: + +```sh +sudo setsebool -P container_use_devices on +``` + +### Other setups + +- **Rootless Docker** needs extra toolkit configuration, see the rootless mode + section of the NVIDIA Container Toolkit install guide. +- **Docker Desktop** and **Podman** are not covered by these instructions. + +## Build the image + +Build the image from the `devices-landlock-fix` branch. Its security policy +allows the agent to use the GPU; images built from other branches do not. + +```sh +git clone https://github.com/singnet/Omega.git +cd Omega +git checkout devices-landlock-fix +docker build -t omega:gpu . +``` + +## Start the container + +Start the container with your own command and pass the GPUs with +`--device nvidia.com/gpu=all`, or a single GPU by its CDI name from +`nvidia-ctk cdi list`, for example `--device nvidia.com/gpu=0`. The other +options match the ones `scripts/omega` uses. Example for Telegram and OpenAI, +with `OPENAI_API_KEY`, `TG_BOT_TOKEN` and `OMEGA_AUTH_SECRET` exported in your +shell: + +```sh +docker rm -f omega +docker run -d -it \ + --name omega \ + --init \ + --security-opt no-new-privileges:true \ + --add-host=host.docker.internal:host-gateway \ + --tmpfs /tmp:size=64m,mode=1777 \ + --tmpfs /var/tmp:size=64m,mode=1777 \ + --tmpfs /run:size=16m,mode=755 \ + --volume omega-memory:/PeTTa/repos/Omega/memory \ + --device nvidia.com/gpu=all \ + -e OPENAI_API_KEY \ + -e TG_BOT_TOKEN \ + -e OMEGA_AUTH_SECRET \ + omega:gpu +``` + +Do not restart this container with `scripts/omega start`: it recreates the +container without the GPU options. `docker stop omega` and `docker start omega` +keep them. diff --git a/profile/policy.py b/profile/policy.py index 981718af..eb11ce96 100644 --- a/profile/policy.py +++ b/profile/policy.py @@ -71,7 +71,7 @@ class FileSystemPolicy: | AccessFs.REMOVE_DIR | AccessFs.MAKE_FIFO | AccessFs.MAKE_SOCK) READ_WRITE_FILE_ACCESS = (AccessFs.READ_FILE | AccessFs.WRITE_FILE | - AccessFs.TRUNCATE) + AccessFs.TRUNCATE | AccessFs.IOCTL_DEV) def __init__(self): self._compatibility = LandLockCompatibility.BEST_EFFORT @@ -116,11 +116,33 @@ def load_dict(self, policy: dict): self._read_only = [Path(f'{p}') for p in ro] self._read_write = [Path(f'{p}') for p in rw] + @staticmethod + def _existing_paths(paths: list[Path]) -> list[Path]: + """Drop policy paths that do not exist on a host machine. + + Landlock cannot add a rule for a missing path, and the policy lists + platform-specific paths such as GPU device nodes that exist only on + some host machines. + + Args: + paths: Paths listed in the policy. + + Returns: + Paths that exist on a host machine. + """ + existing = [p for p in paths if p.exists()] + missing = [str(p) for p in paths if not p.exists()] + if missing: + logger.info(f"Skipped missing policy paths: {missing}") + return existing + def apply(self): - rod = list(filter(lambda p: p.is_dir(), self._read_only)) - rof = list(filter(lambda p: not p.is_dir(), self._read_only)) - rwd = list(filter(lambda p: p.is_dir(), self._read_write)) - rwf = list(filter(lambda p: not p.is_dir(), self._read_write)) + ro = self._existing_paths(self._read_only) + rw = self._existing_paths(self._read_write) + rod = list(filter(lambda p: p.is_dir(), ro)) + rof = list(filter(lambda p: not p.is_dir(), ro)) + rwd = list(filter(lambda p: p.is_dir(), rw)) + rwf = list(filter(lambda p: not p.is_dir(), rw)) strict = self._compatibility == LandLockCompatibility.HARD_REQUIREMENT Landlock(strict=strict) \ diff --git a/profile/policy.yaml b/profile/policy.yaml index 22c3d240..45bd1b4a 100644 --- a/profile/policy.yaml +++ b/profile/policy.yaml @@ -5,7 +5,6 @@ filesystem_policy: - /bin - /usr - /lib - - /proc - /dev/urandom - /etc - /opt @@ -21,5 +20,14 @@ filesystem_policy: - /opt/sentence_transformers - /var/tmp - /dev/shm + # The CUDA driver names its threads via /proc/self/task//comm. + - /proc + # GPU device nodes: /dev/dxg on WSL2, /dev/nvidia* on native Linux. + # Paths missing on the host are skipped. + - /dev/dxg + - /dev/nvidiactl + - /dev/nvidia[0-9]* + - /dev/nvidia-uvm + - /dev/nvidia-uvm-tools landlock: compatibility: best_effort