diff --git a/.devcontainer/Dockerfile b/.devcontainer/Dockerfile
index 8ae30d8..d676eeb 100644
--- a/.devcontainer/Dockerfile
+++ b/.devcontainer/Dockerfile
@@ -42,21 +42,28 @@ RUN apt-get update && \
gfortran && \
# Install necessary CUDA packages
curl -fsSL https://developer.download.nvidia.com/compute/cuda/repos/ubuntu2204/x86_64/cuda-keyring_1.1-1_all.deb -o /tmp/cuda-keyring.deb && \
- dpkg -i /tmp/cuda-keyring.deb && rm /tmp/cuda-keyring.deb && \
- apt-get update && \
+ dpkg -i /tmp/cuda-keyring.deb && rm /tmp/cuda-keyring.deb
+
+ARG CUDA_MAJOR_VERSION=12
+ARG CUDA_MINOR_VERSION=8
+
+# hadolint ignore=DL3008
+RUN apt-get update && \
apt-get install -y -q --no-install-recommends \
- cuda-toolkit-12-4 \
- libcudnn9-dev-cuda-12 \
- libnpp-dev-12-4 \
- && \
+ cuda-toolkit-${CUDA_MAJOR_VERSION}-${CUDA_MINOR_VERSION} \
+ libcudnn9-dev-cuda-${CUDA_MAJOR_VERSION} && \
rm -rf /var/lib/apt/lists/* && apt-get clean
RUN python3 -m venv ${VIRTUAL_ENV} && \
${VIRTUAL_ENV}/bin/pip install --no-cache-dir \
- "pip==26.0.1" \
+ "pip==26.2.1" \
"torch==2.10.0" \
"torchvision==0.25.0" \
- "numpy==1.26.0"
+ "numpy==1.26.4"
+
+# Ensure paths capture the version dynamically
+ENV PATH="/usr/local/cuda/bin:${PATH}"
+ENV LD_LIBRARY_PATH="/usr/local/cuda/lib64:${LD_LIBRARY_PATH}"
# OPENCV W/ CUDA SUPPORT
ENV PYTHON_VERSION=3.10
@@ -74,6 +81,8 @@ RUN git clone --branch ${OPENCV_VERSION} --depth 1 https://github.com/opencv/ope
WORKDIR ${DEPENDENCY_DIR}/opencv/build
RUN ln -s /usr/include/x86_64-linux-gnu/cudnn*.h /usr/local/cuda/include/ && \
ln -s /usr/lib/x86_64-linux-gnu/libcudnn*.so* /usr/local/cuda/lib64/
+
+# hadolint ignore=SC2046
RUN cmake -D CMAKE_BUILD_TYPE=RELEASE \
-D CMAKE_INSTALL_PREFIX="${VIRTUAL_ENV}" \
-D OPENCV_EXTRA_MODULES_PATH="${DEPENDENCY_DIR}/opencv_contrib/modules" \
@@ -84,7 +93,7 @@ RUN cmake -D CMAKE_BUILD_TYPE=RELEASE \
-D WITH_TBB=ON \
-D WITH_V4L=ON \
-D OPENCV_DNN_CUDA=ON \
- -D CUDA_ARCH_BIN=70,75,80,86,89,90 \
+ -D CUDA_ARCH_BIN=80,86,89,90 \
-D CUDA_FAST_MATH=ON \
-D CUDA_TOOLKIT_ROOT_DIR=/usr/local/cuda \
-D CUDNN_INCLUDE_DIR=/usr/include/x86_64-linux-gnu \
@@ -93,8 +102,8 @@ RUN cmake -D CMAKE_BUILD_TYPE=RELEASE \
-D BUILD_opencv_cudaarithm=ON \
-D BUILD_opencv_cudafilters=ON \
-D BUILD_opencv_cudacodec=ON \
- -D WITH_NVCUVID=ON \
- -D WITH_NVCUVENC=ON \
+ -D WITH_NVCUVID=OFF \
+ -D WITH_NVCUVENC=OFF \
-D WITH_VAAPI=ON \
-D WITH_FFMPEG=ON \
-D BUILD_opencv_python3=ON \
@@ -108,12 +117,13 @@ SHELL ["/bin/bash", "-o", "pipefail", "-c"]
RUN mkdir -p /tmp/cv2_deps && \
PY_CV2_SO=$(find "${VIRTUAL_ENV}" -name "cv2*.so" | head -n 1) && \
ldd "${PY_CV2_SO}" | grep "=> /" | awk '{print $3}' | xargs -I '{}' cp -v '{}' /tmp/cv2_deps/ && \
- cp -v /usr/local/cuda/lib64/libnpp*.so* /tmp/cv2_deps/ && \
- cp -v /usr/local/cuda/lib64/libcudart.so* /tmp/cv2_deps/ && \
- cp -P /usr/local/cuda/lib64/libnvrtc.so* /tmp/cv2_deps/ && \
+ # FIX: Robust fallback copying to grab NPP/CUDA libs regardless of whether they land in /usr/local or system path channels
+ (cp -v /usr/local/cuda/lib64/libnpp*.so* /tmp/cv2_deps/ || cp -v /usr/lib/x86_64-linux-gnu/libnpp*.so* /tmp/cv2_deps/) && \
+ (cp -v /usr/local/cuda/lib64/libcudart.so* /tmp/cv2_deps/ || cp -v /usr/lib/x86_64-linux-gnu/libcudart.so* /tmp/cv2_deps/) && \
+ (cp -P /usr/local/cuda/lib64/libnvrtc.so* /tmp/cv2_deps/ || cp -P /usr/lib/x86_64-linux-gnu/libnvrtc.so* /tmp/cv2_deps/) && \
+ (cp -v /usr/local/cuda/lib64/libcublas*.so* /tmp/cv2_deps/ || cp -v /usr/lib/x86_64-linux-gnu/libcublas*.so* /tmp/cv2_deps/) && \
cp -v /usr/lib/x86_64-linux-gnu/libnvidia-encode.so* /tmp/cv2_deps/ || true
-
# Clean up
RUN rm -rf ${DEPENDENCY_DIR}
@@ -121,61 +131,64 @@ RUN rm -rf ${DEPENDENCY_DIR}
FROM openvisualcloud/xeon-ubuntu2204-media-nginx:23.1@sha256:d19eb597dc210134063803630ae2ea1ec84dfd4189138f59551e2f5ed047284a
ARG DEBIAN_FRONTEND=noninteractive
+ARG CUDA_MAJOR_VERSION=12
+ARG CUDA_MINOR_VERSION=8
ENV VIRTUAL_ENV=/opt/venv
ENV PATH="$VIRTUAL_ENV/bin:/usr/local/cuda/bin:${PATH}"
# Install Runtime Libraries, FFmpeg/VA-API headers, and cuDNN
# Includes libva and ffmpeg libraries required for HEVC decoding
# hadolint ignore=DL3008
-RUN apt-get update && apt-get install -y --no-install-recommends \
- ca-certificates \
- curl \
- python3 \
- libgl1 \
- libglib2.0-0 \
- libtbb12 \
- # Critical for Hardware Decoding (HEVC)
- libva2 \
- libva-drm2 \
- libva-x11-2 \
- libavcodec58 \
- libavformat58 \
- libswscale5 \
- libv4l-0 && \
+RUN apt-get update && \
+ apt-get install -y --no-install-recommends \
+ ca-certificates \
+ curl \
+ python3 \
+ libgl1 \
+ libglib2.0-0 \
+ libtbb12 \
+ # Critical for Hardware Decoding (HEVC)
+ libva2 \
+ libva-drm2 \
+ libva-x11-2 \
+ libavcodec58 \
+ libavformat58 \
+ libswscale5 \
+ libv4l-0 && \
# Install necessary CUDA packages
curl -fsSL https://developer.download.nvidia.com/compute/cuda/repos/ubuntu2204/x86_64/cuda-keyring_1.1-1_all.deb -o /tmp/cuda-keyring.deb && \
- dpkg -i /tmp/cuda-keyring.deb && rm /tmp/cuda-keyring.deb && \
+ dpkg -i /tmp/cuda-keyring.deb && rm /tmp/cuda-keyring.deb && \
apt-get update && \
- apt-get install -y -q --no-install-recommends libcudnn9-cuda-12 \
+ apt-get install -y -q --no-install-recommends \
+ libcudnn9-cuda-${CUDA_MAJOR_VERSION} \
libavcodec-dev \
libavformat-dev \
- libavutil-dev nvidia-cuda-dev && \
+ libavutil-dev \
+ cuda-nvcc-${CUDA_MAJOR_VERSION}-${CUDA_MINOR_VERSION} && \
rm -rf /var/lib/apt/lists/*
# Copy the entire pre-compiled virtual environment from the build stage
COPY --from=build ${VIRTUAL_ENV} ${VIRTUAL_ENV}
-# Copy NPP (NVIDIA Performance Primitives) libs - Required for OpenCV CUDA
-# COPY --from=build /usr/local/cuda/lib64/libnpp*.so* /usr/local/cuda/lib64/
+
# Copy additional required CUDA math/parallel libs
RUN mkdir -p /usr/local/cuda/lib64
COPY --from=build /tmp/cv2_deps/* /usr/local/cuda/lib64/
+
+# Tells the host engine to expose all GPUs and utility binaries (like nvidia-smi)
+ENV NVIDIA_VISIBLE_DEVICES=all
+# FIX: Cleaned duplicate line later in file to ensure capabilities are consistently initialized
+ENV NVIDIA_DRIVER_CAPABILITIES=compute,utility,video
ENV LD_LIBRARY_PATH="/usr/local/cuda/lib64:${VIRTUAL_ENV}/lib:${LD_LIBRARY_PATH}"
RUN ldconfig
-ARG DEVICE="CPU"
+ARG DEVICE="GPU"
ENV DEVICE="${DEVICE}"
+ARG DEBUG="0"
+ENV DEBUG="${DEBUG}"
+
# Set the working directory in the container
WORKDIR /home
COPY requirements.txt /home/
-ENV NVIDIA_DRIVER_CAPABILITIES=all
-
-# RUN pip3 install --no-cache-dir "pytest>=9.0.3" && \
-# pip3 install --no-cache-dir --require-hashes -r /home/requirements.CPU.txt --index-url https://download.pytorch.org/whl/cpu --extra-index-url https://pypi.org/simple && \
-# pip3 uninstall -y ultralytics opencv-python opencv-contrib-python opencv-python-headless && \
-# pip3 install --no-cache-dir --require-hashes -r /home/requirements.GPU.txt && \
-# pip3 uninstall -y opencv-python opencv-contrib-python opencv-python-headless
-
-
RUN pip3 install --no-cache-dir --require-hashes -r /home/requirements.txt && \
- pip3 uninstall -y opencv-python opencv-contrib-python opencv-python-headless
\ No newline at end of file
+ pip3 uninstall -y opencv-python opencv-contrib-python opencv-python-headless
diff --git a/.devcontainer/devcontainer.json b/.devcontainer/devcontainer.json
index b44624a..8bc23d7 100644
--- a/.devcontainer/devcontainer.json
+++ b/.devcontainer/devcontainer.json
@@ -2,13 +2,10 @@
// README at: https://github.com/devcontainers/templates/tree/main/src/debian
{
"name": "Pipeline Dev",
- // Or use a Dockerfile or Docker Compose file. More info: https://containers.dev/guide/dockerfile
- // "image": "mcr.microsoft.com/devcontainers/base:bullseye"
+
"build": {
// Path is relative to the devcontainer.json file.
"dockerfile" : "Dockerfile",
- // "context" : "..",
- // "dockerfile" : "../fastapi/Dockerfile",
"context" : "../fastapi",
"args":{
"HTTP_PROXY" : "${localEnv:HTTP_PROXY}",
@@ -16,21 +13,50 @@
"HTTPS_PROXY" : "${localEnv:HTTPS_PROXY}",
"https_proxy" : "${localEnv:https_proxy}",
"NO_PROXY" : "${localEnv:NO_PROXY}",
- "no_proxy" : "${localEnv:no_proxy}",
- }
+ "no_proxy" : "${localEnv:no_proxy}"
+ },
+ // Forces the build phase to use the host network to fix any Ubuntu timeout
+ "options": [
+ "--network=host"
+ ]
},
+
"containerEnv" : {
"HTTP_PROXY" : "${localEnv:HTTP_PROXY}",
"http_proxy" : "${localEnv:http_proxy}",
"HTTPS_PROXY" : "${localEnv:HTTPS_PROXY}",
"https_proxy" : "${localEnv:https_proxy}",
"NO_PROXY" : "${localEnv:NO_PROXY}",
- "no_proxy" : "${localEnv:no_proxy}",
+ "no_proxy" : "${localEnv:no_proxy}"
},
- "runArgs" : [ "--rm", "--gpus", "all", "--ipc=host", "--name", "pipeline_dev", "-p", "8011:8000","--cap-add=SYS_NICE", "--shm-size=8gb"], //, "--net=host", "--privileged", "--shm-size=2gb"],
+ // Native, modern way to pass GPUs to the Dev Container
+ // VS Code will automatically append the --gpus all flag if detected
+ "hostRequirements": {
+ "gpu": "optional"
+ },
+
+ "runArgs" : [
+ // "-p", "8011:8000",
+ // "--rm",
+ // "--gpus", "all",
+ "--restart", "unless-stopped",
+ "--ipc=host",
+ "--name", "pipeline_dev",
+ "--cap-add=SYS_NICE",
+ "--shm-size=8gb",
+ "--memory=16g", // Limits container RAM consumption ceiling
+ "--memory-swap=16g", // Prevents toxic swap disk compilation leaks
+ "--network=host" // Ensures runtime matches the host network
+ ],
+ //, "--privileged", "--shm-size=2gb"],
+
+ // Forwarding ports cleanly via VS Code spec instead of raw "-p"
+ // "forwardPorts": [],
+ "overrideCommand": true,
- "postStartCommand": "apt-get update -o Acquire::Check-Date=false -y && apt-get install -y gdb sudo && cp -rp /workspace/app/*.md /app/ && cp -rp /workspace/app/.vscode /app/ && cp -rp /workspace/app/.gitignore /app/ && cp -rp /workspace/app/.github /app/",
+ "postStartCommand": "apt-get update -o Acquire::Check-Date=false -y && apt-get install -y gdb sudo",
+ // && cp -rp /workspace/app/*.md /home/ && cp -rp /workspace/app/.vscode /home/ && cp -rp /workspace/app/.gitignore /home/ && cp -rp /workspace/app/.github /home/",
// "postAttachCommand": "sudo chown -R ${localEnv:USER} /workspaces",
// Features to add to the dev container. More info: https://containers.dev/features.
@@ -51,7 +77,7 @@
},
"workspaceFolder": "/home",
- "workspaceMount": "source=./,target=/workspace/app,type=bind,consistency=cached",
+ "workspaceMount": "source=${localWorkspaceFolder},target=/workspace/app,type=bind,consistency=cached",
// For training only
// "mounts": [
@@ -60,10 +86,8 @@
// For fastapi
"mounts": [
- "source=./inputs,target=/watch_dir,type=bind",
- "source=./fastapi,target=/home,type=bind"
- // "source=./fastapi/resources,target=/home/resources,type=bind",
- // "source=./fastapi/tests,target=/home/tests,type=bind"
+ "source=${localWorkspaceFolder}/inputs,target=/watch_dir,type=bind",
+ "source=${localWorkspaceFolder}/fastapi,target=/home,type=bind"
],
// Uncomment to connect as root instead. More info: https://aka.ms/dev-containers-non-root.
diff --git a/.github/assets/fastapi/requirements.in b/.github/assets/fastapi/requirements.in
index d7c5f92..1207d22 100644
--- a/.github/assets/fastapi/requirements.in
+++ b/.github/assets/fastapi/requirements.in
@@ -1,29 +1,27 @@
av>=17.0.1
-cucim>=23.10.0
cupy-cuda12x>=13.6.0
fastapi>=0.135.1
nncf>=2.19.0
-numpy>=1.26.4
+numpy==1.26.4
nvidia-cuda-runtime-cu12>=12.8.90
nvidia-cudnn-cu12>=9.10.2.21
-onnx>=1.21.0
+onnx>=1.22.0
onnxruntime-gpu>=1.23.2
onnxslim>=0.1.82
openvino-dev>=2024.6.0
-pip>=26.1
+pip>=26.2
protobuf>=5.29.6,<6 # For VDMS
-pynvvideocodec>=2.1.0
pytest>=9.0.3
requests>=2.33.0
tensorrt_cu12==10.14.1.48.post1
torch==2.10.0
-torchvision==0.25.0
-ultralytics>=8.4.8
+torchvision>=0.25.0
+ultralytics>=8.4.34
uvicorn>=0.42.0
vdms>=0.0.23
wheel>=0.46.3
idna>=3.15
-pillow>=12.2.0
-starlette>=1.0.1
+pillow>=12.3.0
+starlette>=1.3.1
urllib3>=2.7.0
diff --git a/.github/assets/finetune/requirements.in b/.github/assets/finetune/requirements.in
index 0e3d576..6620211 100644
--- a/.github/assets/finetune/requirements.in
+++ b/.github/assets/finetune/requirements.in
@@ -1,19 +1,19 @@
-cucim>=23.10.0
cupy-cuda12x>=13.6.0
-numpy>=1.26.4
+numpy==1.26.4
nvidia-cuda-runtime-cu12>=12.8.90
nvidia-cudnn-cu12>=9.10.2.21
-onnx>=1.21.0
+onnx>=1.22.0
onnxruntime-gpu>=1.23.2
onnxslim>=0.1.82
-pip>=26.1
+pip>=26.2
requests>=2.33.0
tensorrt_cu12==10.14.1.48.post1
torch==2.10.0
-torchvision==0.25.0
-ultralytics>=8.4.8
+torchvision>=0.25.0
+ultralytics>=8.4.34
idna>=3.15
-pillow>=12.2.0
+pillow>=12.3.0
+protobuf>=5.29.6,<6 # For VDMS
urllib3>=2.7.0
diff --git a/.github/assets/frontend/requirements.in b/.github/assets/frontend/requirements.in
index 614cefb..4edff61 100644
--- a/.github/assets/frontend/requirements.in
+++ b/.github/assets/frontend/requirements.in
@@ -1,9 +1,9 @@
-pip>=26.1
+pip>=26.2
ply>=3.11
protobuf>=5.29.6,<6 # For VDMS
psutil>=5.9.0
requests>=2.33.0
-tornado>=6.5.5
+tornado>=6.5.8
vdms>=0.0.23
idna>=3.15
diff --git a/.github/assets/udf/requirements.in b/.github/assets/udf/requirements.in
index 7f89880..ca1a239 100644
--- a/.github/assets/udf/requirements.in
+++ b/.github/assets/udf/requirements.in
@@ -1,8 +1,8 @@
Flask>=3.1.3
inotify>=0.2.12
-numpy>=2.2.6
-opencv-python-headless>=4.13.0.90
-pip>=26.1
+numpy==1.26.4
+opencv-python-headless<4.12
+pip>=26.2
protobuf>=5.29.6,<6 # For VDMS
vdms>=0.0.23
werkzeug>=3.1.6
diff --git a/.github/assets/video/requirements.in b/.github/assets/video/requirements.in
index 787e525..f13618e 100644
--- a/.github/assets/video/requirements.in
+++ b/.github/assets/video/requirements.in
@@ -1,8 +1,8 @@
inotify>=0.2.12
-pip>=26.1
+pip>=26.2
pyyaml>=6.0.3
requests>=2.33.0
-tornado>=6.5.5
+tornado>=6.5.8
idna>=3.15
urllib3>=2.7.0
diff --git a/.gitignore b/.gitignore
index 0a718e2..41cfe6f 100644
--- a/.gitignore
+++ b/.gitignore
@@ -1,15 +1,17 @@
.venv*
-.vscode
*.log
**/__pycache__/
**/_*
+**/* copy*
**/DockerImageTars/
build/*
fastapi/resources/models/intel
fastapi/resources/models/ultralytics
+fastapi/tests/*_data/
fastapi/tests/*_results*
fastapi/tests/*_test_imgs
fastapi/tests/*.mp4
+fastapi/tests/run_test_*.*
finetune/.env
finetune/app/*-Results
inputs/camera_config.yaml
diff --git a/.pre-commit-config.yaml b/.pre-commit-config.yaml
index e5dca08..a3ad3e1 100644
--- a/.pre-commit-config.yaml
+++ b/.pre-commit-config.yaml
@@ -20,12 +20,12 @@ repos:
- id: mixed-line-ending
args: ["--fix=auto"]
- id: trailing-whitespace
- - repo: https://github.com/pre-commit/mirrors-clang-format
- rev: v19.1.7
- hooks:
- - id: clang-format
- types: [c, c++]
- args: [
- "--style=Google",
- "-i"
- ]
+ - repo: https://github.com/pre-commit/mirrors-clang-format
+ rev: v19.1.7
+ hooks:
+ - id: clang-format
+ types: [c, c++]
+ args: [
+ "--style=Google",
+ "-i"
+ ]
diff --git a/README.md b/README.md
index 5228d7b..9fdae79 100644
--- a/README.md
+++ b/README.md
@@ -1,26 +1,23 @@

[](https://scorecard.dev/viewer/?uri=github.com/IntelLabs/Video-Curation-Sample)
-This sample implements libraries for video content analysis, database ingestion, content search and visualization:
-- **Ingest**: Analyze video content and ingest the data into the VDMS.
-- **VDMS**: Store metadata efficiently in a graph-based database.
-- **Visualization**: Visualize content search based on video metadata.
+This sample implements a pipeline focusing on real-time processing of high-resolution (8K) video for object detection.
+The application can process high-resolution video in real-time using a Smart Filtering Pipeline to significantly reduce pixel processing (compute) by *automatically* identifying the regions of interest which is then forwarded to the detection model.
-
-**This is a concept sample in active development.**
-
+Please see [High Resolution Object Detection Pipeline](./doc/pipeline.md) for more details.
-
-## Software Stacks
-This sample is powered by the following software stacks:
-- **NGINX Web Service**:
- - [The NGINX/FFmpeg-based web serving stack](/deployment/Dockerfiles/Xeon/ubuntu-22.04/media/nginx) is used to store and segment video content and serve web services. The software stack is optimized for Intel Xeon Scalable Processors.
-
+
+

+
High-level High Resolution Object Detection Pipeline
+
### License Obligations
- FFmpeg is an open source project licensed under LGPL and GPL. See https://www.ffmpeg.org/legal.html. You are solely responsible for determining if your use of FFmpeg requires any additional licenses. Intel is not responsible for obtaining any such licenses, nor liable for any licensing fees due, in connection with your use of FFmpeg.
-
+
+
+### Datasets & Attributions
+- This project utilizes third-party open datasets. Please see our [Data Attributions](docs/DATASETS.md) for full licensing, copyright details, and citation parameters.
## Install Prerequisites:
@@ -32,8 +29,6 @@ This sample is powered by the following software stacks:
- **Docker Engine**:
- Install [docker engine](https://docs.docker.com/install) and verify you have Docker Compose V2 2.18.0+ setup.
-
-
- Setup docker proxy as follows if you are behind a firewall:
```bash
sudo mkdir -p /etc/systemd/system/docker.service.d
@@ -41,199 +36,38 @@ This sample is powered by the following software stacks:
sudo systemctl daemon-reload
sudo systemctl restart docker
```
-
-
-## Prepare Videos and/or Camera Configurations
-This application processes videos and camera streams by watching the `inputs` directory.
-MP4 videos should be placed in this directory and the application will automatically begin processing it. If interested in using sample videos, see [examples](/doc/cmake.md#examples).
-To configure the application for camera streams, add the camera name and URL to `inputs/camera_config.yaml`.
-The YAML file provides a sample entry.
-
+## Deploy High Resolution Drone Detection using Smart Filtering
+All components for this application are dockerize.
+Scripts are provided to make deployment easier.
-
-## Start/Stop Service using Scripts:
-We have provided scripts to help with the deployment.
-The [start_app.sh](/start_app.sh) script provides everything you need to deploy the service.
-To run the service using GPU, use:
-```bash
-./start_app.sh -e GPU
-```
-This same script can be used to run the service with CPU, resizing video to 640x640, etc.
-The accepted parameters for the start script are below:
-
-| Parameter | Type | Description | Default |
-| :----------------- | :------: | :--------------------------------------------------------------------------------------------- | :-----------: |
-| -h | optional | Print this help message | |
-| -d or --debug | optional | Flag for debugging | "0" |
-| -e or --device | optional | Device to use for inference | CPU |
-| -i or --ingestion | optional | Ingestion type (object, face) | "object,face" |
-| -l or --tars | optional | Flag to load docker images instead of building from Dockerfiles | "0" |
-| -m or --model | optional | Custom YOLO model name (.pt). If not provided model YOLO11n is used. | |
-| -o or --omit-det | optional | By default, object detections are printed. To omit printing detections to screen, enable flag. | False |
-| -z or --resize | optional | Flag to resize video to model input size (640x640) | False |
-
-
-See also [Customize Build Process](doc/cmake.md) for additional information on options.
-
-
-The [stop_app.sh](/stop_app.sh) script is provided to stop the service from a different terminal. To stop and prune the docker images (data doesn't persist), run:
+### Start
+[Optional] To make sure there aren't any running containers for this application, run the following which stops the application and prunes containers:
```bash
-./stop_app.sh -p
+./stop.sh –p
```
-**NOTE:** Remove `-p` if you only want to stop the docker containers without pruning docker builder, containers, volumes, and networks.
-
-
-
-## Manually Build Streaming Sample:
-Instead of running the start/stop using the above scripts, you can run the commands individually as shown below.
-
+To start the application, run the following:
```bash
-mkdir build
-cd build
-cmake ..
-make
+./start_app.sh –m drone_detection
```
-See also [Customize Build Process](doc/cmake.md) for additional options.
-
-### Start/Stop Sample:
+### View Live Detections
+Launch your browser and browse to ```https://:30077```. The sample UI is similar to the following:
-Use the following commands to start/stop services via docker-compose:
+
+

+
+### Shutdown
+To shutdown this application, run the following:
```bash
-make start_docker_compose
-make stop_docker_compose
-```
-
-
-
-
-
-
-## Display Object Detected
-Once the application is started, we have provided a script to display the detections with associated timestamps.
-To use the script, be sure ***NOT*** to use flags `-o` or `--omit-det` so the detection details are available.
-To display this information, in a separate terminal, run `display_detections.sh` with the objects of interest.
-For example, to display when "person" and "car" objects are detected, run:
-```bash
-./display_detections.sh person,car
-```
-
-
-
-## Launch Sample UI:
-Launch your browser and browse to ```https://:30007```. The sample UI is similar to the following:
-
-
-
-
-
-
-***NOTE:*** If you see a browser warning of self-signed certificate, please accept it to proceed to the sample UI.
-
-
-
-## Redeploy Service by Persisting Data
-There may be cases where you want to stop the service but redeploy later with data persisted. In such cases, follow the following steps, assuming service is currently running, i.e. via `./start_app.sh -e CPU -z -d`:
-
-1. [*Optional*] For redeployment, we will save the state of the named volumes. To conserve space in the `lcc_vdms-content` volume, it may be beneficial to delete the `/mnt/tmp` directory, in case some files were not removed. This is done by running:
- ```bash
- docker exec -it lcc_vdms-service_1 rm -rf /mnt/tmp
- ```
-
-
-1. Save docker images for running service:
- ```bash
- ./deployment/DockerImageTars/save_docker_images.sh -e CPU -z
- ```
-
-
- The options used should be the same as those used for running the original service, if available.
- The accepted parameters for `save_docker_images.sh` are below:
-
- | Parameter | Type | Description | Default |
- | :----------------- | :------: | :------------------------------------------------- | :-----------: |
- | -h | optional | Print this help message | |
- | -e or --device | optional | Device to use for inference | CPU |
- | -z or --resize | optional | Flag to resize video to model input size (640x640) | False |
-
-
-1. Stop the running service:
- ```bash
- ./stop_app.sh
- ```
- The stop script stops the service, removes all stopped Docker containers and removes all unused Docker networks.
- We ***DO NOT*** use the `-p` flag here to avoid removing the volumes before backing them up.
-
-
-1. Backup the docker volumes to redeploy later. First rename the `lcc_app-content` volume and remove original:
- ```bash
- ./deployment/DockerImageTars/rename_volume.sh lcc_app-content lcc_app-content_saved
-
- docker volume rm lcc_app-content
- ```
-
- The `lcc_vdms-content` volume is sparse, so to backup, compress the volume to respective location. In this case, we save the file to `deployment/DockerImageTars/resize` and remove original:
- ```bash
- docker run --rm -v lcc_vdms-content:/data \
- -v ${PWD}/deployment/DockerImageTars/resize:/backup \
- ubuntu bash -c "tar czSf /backup/lcc_vdms-content_saved.tar.gz /data"
-
- docker volume rm lcc_vdms-content
- ```
-
-
-1. At this point, images and volumes are saved. You can safely, prune all docker images/volumes if needed via `./stop_app.sh -p`.
-
-1. Prior to redeployment, videos already inserted in previous deployment, which should not be processed again (if any), should be removed from `inputs/`, and any camera configurations should be updated for next deployment.
-
-1. Repopulate the docker volumes to expected name. For `lcc_app-content_saved`, we can simply rename it:
- ```bash
- ./deployment/DockerImageTars/rename_volume.sh lcc_app-content_saved lcc_app-content
-
- docker volume rm lcc_app-content_saved
- ```
-
- The `lcc_vdms-content` volume should be created and populated with compressed data from previous step:
- ```bash
- docker volume create lcc_vdms-content
-
- docker run --rm -v lcc_vdms-content:/data \
- -v ${PWD}/deployment/DockerImageTars/resize:/backup \
- ubuntu bash -c "tar -xvzf /backup/lcc_vdms-content_saved.tar.gz"
- ```
-
-
-1. Restart the service with appropriate options but include `--tars` to use saved images. Here, we're re-deploying using the same flags as the original deployment.
- ```bash
- ./start_app.sh -e CPU -z -d --tars
- ```
-
-
----
-
-## See Also
-
-- [Configuration Options](doc/cmake.md)
-
-
-
-- [Visual Data Management System](https://github.com/intellabs/vdms)
diff --git a/deployment/docker-swarm/docker-compose.yml.m4 b/deployment/docker-swarm/docker-compose.yml.m4
index 2e8e8cb..7a776e7 100644
--- a/deployment/docker-swarm/docker-compose.yml.m4
+++ b/deployment/docker-swarm/docker-compose.yml.m4
@@ -1,13 +1,9 @@
services:
-include(frontend.m4)
-include(udf.m4)
-include(vdms.m4)
include(video.m4)
include(secret.m4)
include(network.m4)
volumes:
app-content:
- vdms-content:
diff --git a/deployment/docker-swarm/video.m4 b/deployment/docker-swarm/video.m4
index 1596d2e..5caca10 100644
--- a/deployment/docker-swarm/video.m4
+++ b/deployment/docker-swarm/video.m4
@@ -101,7 +101,4 @@ define(`PROFILE_GPU', `runtime: nvidia
networks:
- appnet
restart: always
- depends_on:
- - udf-service
- - vdms-service
ifelse(ifdef(`GPU', `yes'), `yes', PROFILE_GPU, PROFILE_DEFAULT)
diff --git a/doc/DATASETS.md b/doc/DATASETS.md
new file mode 100644
index 0000000..4c9b9a4
--- /dev/null
+++ b/doc/DATASETS.md
@@ -0,0 +1,25 @@
+# Dataset and Third-Party Media Attributions
+
+This repository utilizes external datasets and media for training/demo purposes, without storing the actual files.
+
+---
+
+## 1. SynDroneVision Dataset
+* **Source:** [Zenodo (Record 13360116)](https://zenodo.org/records/13360116)
+* **License:** [CC-BY-4.0](https://creativecommons.org/licenses/by/4.0/legalcode.en)
+* **Citation:** See DOI [10.1109/WACV61041.2025.00742](https://doi.org/10.1109/WACV61041.2025.00742) (Lenhard et al., WACV 2025).
+* **Usage:** External data for fine-tuning model for drone detection.
+
+---
+
+## 2. Sample Drone Video (`anduril_swarm.mp4`)
+* **Source:** [droneforge/yolov11-UAV-finetune](https://github.com/droneforge/yolov11-UAV-finetune/blob/main/anduril_swarm.mp4)
+* **Usage:** External sample video for pipeline demonstration.
+
+---
+
+## 3. [Optional] DUT Anti-UAV Dataset
+* **Source:** [wangdongdut/DUT-Anti-UAV](https://github.com/wangdongdut/DUT-Anti-UAV)
+* **License:** [Apache License 2.0](https://github.com/wangdongdut/DUT-Anti-UAV/blob/master/LICENSE)
+* **Citation:** See DOI [10.48550/arXiv.2205.10851](https://doi.org/10.48550/arXiv.2205.10851) (Zhao et al., IEEE T-ITS 2022).
+* **Usage:** External sample video dataset ONLY used for pipeline evaluation via command: `python tests/test_eval.py --type object`. Test script ([test_eval.py](../fastapi/tests/test_eval.py)) evaluates videos 9-20 of the dataset since the camera seems stationary.
diff --git a/doc/Pipeline.png b/doc/Pipeline.png
index 3c72bd2..eb25afb 100644
Binary files a/doc/Pipeline.png and b/doc/Pipeline.png differ
diff --git a/doc/PipelineFlow.png b/doc/PipelineFlow.png
new file mode 100644
index 0000000..1933d9f
Binary files /dev/null and b/doc/PipelineFlow.png differ
diff --git a/doc/PipelineOptions.png b/doc/PipelineOptions.png
deleted file mode 100644
index 00b5d3f..0000000
Binary files a/doc/PipelineOptions.png and /dev/null differ
diff --git a/doc/cmake.md b/doc/cmake.md
index 4c7b591..33f4a5c 100644
--- a/doc/cmake.md
+++ b/doc/cmake.md
@@ -27,10 +27,10 @@ cd inputs
```
#### Deploy Service
-Then use the start script to deploy. An example for CPU is:
+Then use the start script to deploy. An example for GPU is:
```bash
cd ..
-./start_app.sh -e CPU
+./start_app.sh -e GPU
```
@@ -63,21 +63,6 @@ ffmpeg -re -stream_loop -1 -i ${TEST_VIDEO} ${GENERAL_OPTS} \
#### Deploy Service
Then use the start script to deploy. An example for GPU and resizing videos to lower resolution (640x640) is:
```bash
-./start_app.sh -e GPU -z
+./start_app.sh –m drone_detection
```
-
-
diff --git a/doc/finetune.md b/doc/finetune.md
index 7d73a12..a83faea 100644
--- a/doc/finetune.md
+++ b/doc/finetune.md
@@ -2,7 +2,7 @@
This guide provides details on how to fine-tune a YOLO model using Ultralytics on GPU ONLY.
For simplicity, the use-case for this guide is Drone Detection.
-Therefore, the goal is to finetune the YOLO11n model to detect only one class (`drone`).
+Therefore, the goal is to fine-tune the YOLO11n model to detect only one class (`drone`).
## Drone Detection Dataset
@@ -23,10 +23,10 @@ The configurations used for training on 2x NVIDIA A100 80GB PCIe are specified i
Feel free to modify these parameters based on your hardware limitations such as VRAM of GPU.
-## Finetune Script
-The finetune script is used to run training, validation, and also test on a provided video (optional).
+## Fine-tune Script
+The fine-tune script is used to run training, validation, and also test on a provided video (optional).
To make deployment easy, we provide a Dockerfile which has the ideal environment and allow the script to run with deployment.
-The script has a few adjustible arguments, so feel free to modify the call in next section as needed.
+The script has a few adjustable arguments, so feel free to modify the call in next section as needed.
```bash
# Runs default arguments
python finetune.py
@@ -42,7 +42,7 @@ THe following arguments are available:
| --dataset-name DATASET_NAME | `SynDroneVision` | Name of dataset; A directory with this name should be in --data-dir (-d). |
| -l LABELS_STR,
--labels LABELS_STR | `drone` | A comma-separated list of labels (classes) for model. |
| --yaml-name YAML_NAME | `drones` | Name of file (YAML_NAME.yaml) with data specifications. Be sure the `path` in this file correlates with `LOCAL_DATA_DIR` value. |
-| --no-train | | Skip finetune stage |
+| --no-train | | Skip fine-tune stage |
| --test-video TEST_VIDEO | | Test video path for prediction. If not provided, inference is disabled |
@@ -56,7 +56,7 @@ To avoid modifying your system for training, you can use the provided Dockerfile
For easy access of host data, the `inputs` directory containing any input videos, the `finetune/app` directory containing this code, and the parent directory where datasets are stored (i.e. `/data1/datasets`) are mounted to the container.
Please see below for instructions for deploying container via `docker` and `docker compose`.
***NOTE:*** `REPO_DIR` is the path of this repo's main directory. Also be sure to update values in `.env` if using docker compose.
-- **Docker:** For this option, be sure to build container first. You can start the container and finetune script via run command.
+- **Docker:** For this option, be sure to build container first. You can start the container and fine-tune script via run command. If behind proxy, be sure to set them using `--build-arg` and `--env`.
```bash
REPO_DIR=`pwd`
LOCAL_DATA_DIR=/path/to/your/actual/data/directory
@@ -65,7 +65,7 @@ Please see below for instructions for deploying container via `docker` and `dock
cd finetune
docker build -f Dockerfile -t lcc_finetune:latest .
- # Run finetune script default values
+ # Run fine-tune script default values
docker run -it --ipc=host --gpus all \
--name finetune_container \
-v ${REPO_DIR}/inputs:/watch_dir \
diff --git a/doc/pipeline.md b/doc/pipeline.md
index 4236c11..fbbbe08 100644
--- a/doc/pipeline.md
+++ b/doc/pipeline.md
@@ -1,32 +1,35 @@
-# High Resolution Object Detection Pipeline (WIP)
+# High Resolution Object Detection using Smart Filtering
-This pipeline focuses on real-time processing of high-resolution (8K) video for object detection.
-Currently, real-time processing of high-resolution (HR) frames is not possible as large compute power is needed which are typically beyond any HW capabilities.
-There are techniques such as image tiling or resolution reduction, but these workarounds compromisr detail integrity, diminish inference throughput, and can make rendering real-time processing of HR feeds infeasible within HW boundaries.
+This pipeline focuses on real-time processing of high-resolution (8K) video from stationary camera while preserving details to detect potentially small objects, i.e. drones.
+Currently, real-time processing of high-resolution (HR) frames on resource-constrained edge hardware requires scaling up to 2×–15× GPUs.
+There are techniques such as image tiling or resolution reduction, but these workarounds compromise detail integrity, diminish inference throughput, and can make rendering real-time processing of HR feeds infeasible within HW boundaries.
HR video streams typically achieve higher accuracy in object detection, therefore, a technique to reduce compute on HR videos is needed.
To solve this issue, we created a Smart Filtering Pipeline to significantly reduce pixel processing (compute) by *automatically* identifying the regions of interest (ROIs) which is then forwarded to the detection model.
-
-High-level High Resolution Object Detection Pipeline
-
+
+

+
High-level High Resolution Object Detection Pipeline
+
-## Smart Filtering Pipeline
+## Smart Filtering
The Smart Filtering (SF) is a portion of the pipeline which filters HR videos for ROIs which are used in the detection phase instead of processing the entire frame.
This helps reduce the compute while maintaining detection accuracy.
-In smart filtering, we use motion to help identify ROIs, which is ideal for surviellence and applications such as the test use-case, drone detection, with stationary cameras.
+In smart filtering, we use motion to help identify ROIs, which is ideal for surveillance and applications such as the test use-case, drone detection, with stationary cameras.
Keep in mind, if camera is NOT stationary, some background movement may be captured as foreground objects due to background subtraction algorithm.
-
-Flow of Smart Filtering Pipeline
+
+

+
Flow of Smart Filtering Pipeline
+
The current filtering pipeline includes the following steps:
1. **Resize:** Resize full resolution (HR) frame to smaller size for less computation.
1. **Background Subtraction:** Use resized image to obtain mask of foreground objects using background subtraction. With background subtraction, there is an option to include the masks of previous frames which is enabled by default for 3 previous frames.
-1. **Threshold:** Apply thresholding to mask generated by previous step
-1. **Clean Mask:** Dilate the thresholded frame for easier detection of ROIs.
+1. **Threshold:** Apply threshold to mask generated by previous step
+1. **Clean Mask:** Dilate the threshold frame for easier detection of ROIs.
1. **Identify Region of Interests (ROIs):** Retrieve contours from dilated mask
1. Filter contours by defined minimum contour area and obtain ROIs in full resolution coordinates.
1. Sort ROIs by area (largest to smallest)
@@ -37,6 +40,33 @@ The current filtering pipeline includes the following steps:
***NOTE:*** If the video resolution is less than 2K (1920x1080), Smart Filtering is disabled as most systems can support this resolution in real-time.
+### Configurable Parameters
+
+The following are parameters used to fine-tune the Smart Filtering pipeline.
+
+| Category | Variable | Default | Description |
+| ---------------------- | ------------------------------------ | ----------- |------------ |
+| Smart Filtering | SMART_FILTERING_PIXEL_CONSTRAINT | 1920 * 1080 | Smart filtering is enabled if video is larger than 2K. YOLO models typically can efficiently process lower resolution without overloading resources. |
+| Smart Filtering | SMART_FILTERING_WARMUP | False | Warmup all components of pipeline |
+| Background Subtraction | BKGD_SUB_INCLUDE_HISTORY | False | Include temporal history with mask generation. True: dilate masks in history and combine into single current mask via `method`. The final background model mask is generated using the union of combined mask (if true) and current background mask. |
+| Background Subtraction | BKGD_SUB_INCLUDE_HISTORY_DILATE_KERNEL_SIZE | 15 | Kernel used to dilate temporal masks if `BKGD_SUB_INCLUDE_HISTORY` is True. |
+| Background Subtraction | BKGD_SUB_INCLUDE_HISTORY_METHOD | or | Method used to combine mask history. Specify `and` or `or`. |
+| Background Subtraction | BKGD_SUB_INCLUDE_HISTORY_TEMPORAL_SIZE | 3 | Number of previous masks to maintain for temporal mask generation. |
+| Background Subtraction | BKGD_SUB_MOG2_DETECTSHADOWS | False | A boolean to enable/disable shadow detection. |
+| Background Subtraction | BKGD_SUB_MOG2_HISTORY | 2 * TARGET_FPS | The number of previous frames used to build the background model. Higher value: More stable background. It is less likely to be "fooled" by a tree swaying slightly, but it takes much longer for a newly stopped object (like a parked car) to become part of the background. Lower value: The model adapts very quickly. Useful for rapidly changing lighting, but may cause moving objects to "disappear" into the background if they move slowly. |
+| Background Subtraction | BKGD_SUB_MOG2_LR | 1 / BKGD_SUB_MOG2_HISTORY | Learning rate to use for background model |
+| Background Subtraction | BKGD_SUB_MOG2_VARTHRESHOLD | 50 | The Mahalanobis distance threshold to decide whether a pixel is foreground or background. Higher value (Lower sensitivity): This reduces "salt and pepper" noise and ignores small fluctuations, but you might lose the edges of your actual objects (making bounding boxes smaller/fragmented). Lower value (Higher sensitivity): You’ll catch every tiny movement, but you’ll get a lot of noise from compression artifacts or camera sensor grain. |
+| Threshold | THRESHOLD_VALUE | 127 | The threshold value used to classify pixel intensities. |
+| Clean Mask | DILATE_KERNEL_SIZE | 5 | Kernel used to dilate mask AFTER threshold is applied. |
+| Identify ROIs | ROI_BB_FULL_RES_PADDING | MODEL_W * .02 | Padding to add to ROIs to make sure detection captures full box. |
+| Identify ROIs | ROI_CONTAINMENT_THRESH | 0.95 | Input argument to filter_contained_boxes used for removing smaller contained ROIs. Current value is 0.9 (90%). |
+| Identify ROIs | ROI_DISTANCE_THRESH_RATIO | 0.05 | Input argument to merge ROIs is distance less than this parameter. |
+| Identify ROIs | ROI_MAX_RELATIVE_SIZE_RATIO | 1.0 | Ratio of frame width/height to limit ROIs. Currently accepts ROI if width/height less than frame size. |
+| Identify ROIs | ROI_MERGE_SIZE_LIMIT | MODEL_W * 1.25 | The max width/height of merged ROIs. Value is equivalent as MODEL_W (expected imgsz for model.predict) |
+| Identify ROIs | ROI_MIN_AREA_RATIO | 0.01 | Minimum area ratio of pixels for objects/motion. This value is in terms of pixels in resized image. Currently value is set to 41 pixels (1% of width x 1% of height). |
+
+
+
### Testing
The current implementation of the Smart Filtering pipeline is optimized for the test use-case, drone detection.
Drones are typically small in the video frames so if your use-case of interest has different objects, it may be beneficial to test the pipeline results on an existing test video for your use-case.
@@ -44,7 +74,7 @@ Drones are typically small in the video frames so if your use-case of interest h
For this case, we provide [`test_detections.py`](/fastapi/tests/test_detections.py) which annotates ROIs identified by the Smart Filtering pipeline onto each frame of the video for visual inspection.
For testing purposes, you can use VSCode DevContainer (easiest method) or manually deploy the fastapi dockerfile. Using VSCode is straight forward, so here, we will manually deploy the fastapi Dockerfile as it contains the same setup used in the application AND start the test script.
-Here we will build the container, if not available:
+Here we will build the container, if not available. If behind proxy, be sure to set them using `--build-arg`.
```bash
REPO_DIR=`pwd`
@@ -73,9 +103,11 @@ Since this work focuses on HR videos, we converted the test video to 8K using th
To deploy this test, run the following command but modify the name of the test_video.
+If behind proxy, be sure to set them using `--env`.
The results from the test will be saved in `fastapi/tests/test_detections_results/drone_detection/`.
```bash
docker run --rm --ipc=host \
+--user root \
--gpus all --env NVIDIA_DRIVER_CAPABILITIES=all \
--name test_detections \
--env ENABLE_VDMS=False \
@@ -87,7 +119,7 @@ docker run --rm --ipc=host \
-v ${REPO_DIR}/fastapi/resources:/home/resources \
-v ${REPO_DIR}/fastapi/tests:/home/tests \
-v ${REPO_DIR}/fastapi/tests/nginx.conf:/etc/nginx/nginx.conf \
-lcc_fastapi:stream /bin/bash -c "python /home/tests/test_detections.py --source .mp4 --type motion"
+lcc_fastapi:stream /bin/bash -c "python /home/tests/test_detections.py --source --type motion"
```
@@ -95,6 +127,7 @@ If you are not satisfied with the results, feel free to modify/optimize the pipe
The following command starts the detached container.
```bash
docker run -d --rm --ipc=host \
+--user root \
--gpus all --env NVIDIA_DRIVER_CAPABILITIES=all \
--name test_detections \
--env ENABLE_VDMS=False \
@@ -112,13 +145,14 @@ lcc_fastapi:stream tail -f /dev/null
You can attach to the container to debug the filtering as needed.
-For testing other components, please see [Testing Pipeline Components](#testing-pipeline-components) section.
+For testing other components, please see [Test Pipeline Components](#test-pipeline-components) section.
#### VSCode
If using VSCode, the provided `.vscode` directory is already mounted in the above command which includes `launch.json` for debugging.
You can use the provided DevContainer or use the docker extension to connect to the `lcc_fastapi:stream` container by clicking `Attach Visual Studio Code`.
+The DevContainer assumes GPU availability.
Once connected, it will install VSCode in the container.
If you receive an error regarding permissions during installation, you must modify the remote user.
To allow permission for the installation, open the `Command Pallette` and select `Dev Containers: Open Attached Container Configuration File`.
@@ -140,37 +174,7 @@ Edit the file as follows:
```
-### Optimization Areas
-
-For the smart filtering pipeline, the following are potential areas for optimizations.
-
-| Category | Variable | Default | Description |
-| ---------------------- | ------------------------------------ | ----------- |------------ |
-| Smart Filtering | SMART_FILTERING_PIXEL_CONSTRAINT | 1920 * 1080 | Smart filtering is enabled if vidoe is larger than 2K. YOLO models typicaly can efficiently process lower resolution without overloading resources. |
-| Background Subtraction | BKGD_SUB_INCLUDE_HISTORY | True | Include temporal history with mask generation. True: dilate masks in history and combine into single current mask via `method`. The final background model mask is generated using the union of combined mask (if true) and current background mask. |
-| Background Subtraction | BKGD_SUB_INCLUDE_HISTORY_DILATE_KERNEL_SIZE | 15 x 15 | Kernel used to dilate temporal masks if `BKGD_SUB_INCLUDE_HISTORY` is True. |
-| Background Subtraction | BKGD_SUB_INCLUDE_HISTORY_METHOD | or | Method used to combine mask history. Specify `and` or `or`. |
-| Background Subtraction | BKGD_SUB_INCLUDE_HISTORY_TEMPORAL_SIZE | 3 | Number of previous masks to maintain for temporal mask generation. |
-| Background Subtraction | BKGD_SUB_MOG2_DETECTSHADOWS | False | A boolean to enable/disable shadow detection. |
-| Background Subtraction | BKGD_SUB_MOG2_HISTORY | 2 * TARGET_FPS | The number of previous frames used to build the background model. Higher value: More stable background. It is less likely to be "fooled" by a tree swaying slightly, but it takes much longer for a newly stopped object (like a parked car) to become part of the background. Lower value: The model adapts very quickly. Useful for rapidly changing lighting, but may cause moving objects to "disappear" into the background if they move slowly. |
-| Background Subtraction | BKGD_SUB_MOG2_LR | 1 / BKGD_SUB_MOG2_HISTORY | Learning rate to use for background model |
-| Background Subtraction | BKGD_SUB_MOG2_VARTHRESHOLD | 10 | The Mahalanobis distance threshold to decide whether a pixel is foreground or background. Higher value (Lower sensitivity): This reduces "salt and pepper" noise and ignores small fluctuations, but you might lose the edges of your actual objects (making bounding boxes smaller/fragmented). Lower value (Higher sensitivity): You’ll catch every tiny movement, but you’ll get a lot of noise from compression artifacts or camera sensor grain. |
-| Threshold | THRESHOLD_VALUE | 127 | The threshold value used to classify pixel intensities. |
-| Clean Mask | DILATE_KERNEL_SIZE | 5 x 5 | Kernel used to dilate mask AFTER thresholding is performed. |
-| Identify ROIs | ROI_BB_FULL_RES_PADDING | MODEL_W * .02 | Padding to add to ROIs to make sure detection captures full box. |
-| Identify ROIs | ROI_CONTAINMENT_THRESH | 0.95 | Input argument to filter_contained_boxes used for removing smaller contained ROIs. Current value is 0.9 (90%). |
-| Identify ROIs | ROI_DISTANCE_THRESH_RATIO | 0.05 | Input argument to merge ROIs is distance less than this parameter. |
-| Identify ROIs | ROI_MAX_RELATIVE_SIZE_RATIO | 1.0 | Ratio of frame width/height to limit ROIs. Currently accepts ROI if width/height less than frame size. |
-| Identify ROIs | ROI_MERGE_SIZE_LIMIT | MODEL_W * 2 | The max width/height of merged ROIs. Value is equivalent as MODEL_W (expected imgsz for model.predict) |
-| Identify ROIs | ROI_MIN_AREA_RATIO | 0.01 | Minimum area ratio of pixels for objects/motion. This value is in terms of pixels in resized image. Currently value is set to 41 pixels (1% of width x 1% of height). |
-
-
-Once satisfied with annotated results, you can proceed with running the full pipeline.
-***NOTE:*** If you modified any methods within the test script, be sure to update the predefined code to use your changes in the pipeline.
-
-
-
-## Detections using SF and Fine-Tuned YOLO Model
+## Detection Model
This guide assumes a YOLO model or fine-tuned YOLO model is available for detection.
@@ -185,22 +189,60 @@ Please note the model labels are retrieved from the model directly, so the model
-## Testing Pipeline Components
+## Deploy Pipeline
+All components for this application are dockerize.
+Scripts are provided to make deployment easier.
+
+### Start
+[Optional] To make sure there aren't any running containers for this application, run the following which stops the application and prunes containers:
+```bash
+./stop.sh –p
+```
+
+To start the application, run the following:
+```bash
+./start_app.sh –m drone_detection
+```
+
+
+### View Live Detections
+Launch your browser and browse to ```https://:30077```. The sample UI is similar to the following:
+
+
+

+
+
+
+### Shutdown
+To shutdown this application, run the following:
+```bash
+./stop.sh
+```
+
+Or add the necessary flag to stop the application and prune containers:
+```bash
+./stop.sh –p
+```
+
+
+## Test Pipeline Components
We provided multiple test scripts to test different components of the pipeline.
Each of the tests can be run as specified in the [Smart Filtering Pipeline: Testing](#testing) section using the same video `anduril_swarm_8K.mp4` as `SOURCE`.
Here we provide details on each available test.
| Component | Test File | Description |
| --------- | --------- | ----------- |
-| Detection Model | [test_model.py](/fastapi/tests/test_model.py) | Test the model for both CPU and GPU on provided RTSP URL or video file |
-| Smart Filtering | [test_detections.py](/fastapi/tests/test_detections.py) | This test independently evaluates the detection pipeline (with and without Smart Filtering) only and does not include video clip generation or sending metadata to database for querying. |
-| Stream Reader | [test_pipeline.py](/fastapi/tests/test_pipeline.py) | Scenario 1 tests the behavior of Readers when provided an invalid RTSP url.
Scenario 2 reads the RTSP url or video file for a specified duration or until it ends. |
+| Model | [test_model.py](/fastapi/tests/test_model.py) | Test the model for GPU on provided RTSP URL or video file |
+| Stream Readers | [test_readers.py](/fastapi/tests/test_readers.py) | Independently test the stream readers for GPU on provided RTSP URL or video file |
+| Smart Filtering | [test_detections.py](/fastapi/tests/test_detections.py) | Independently test the detection pipeline (with and without Smart Filtering) only. Test does not include video clip generation or sending metadata to database for querying. |
+| Stream Readers | [test_pipeline.py](/fastapi/tests/test_pipeline.py) | Scenario 1 tests the behavior of Readers when provided an invalid RTSP url.
Scenario 2 reads the RTSP url or video file for a specified duration or until it ends. |
| Video Clip Generation | [test_pipeline.py](/fastapi/tests/test_pipeline.py) | Scenario 3 mimics the clip generation within the pipeline. |
| Smart Filtering | [test_pipeline.py](/fastapi/tests/test_pipeline.py) | Scenario 4 tests the entire pipeline and saves output video of results. |
-### Test Models
-This test reads the `SOURCE` from `./inputs` (the local directory or `/watch_dir` if in container), or RTSP server, and creates a video with overlaid detection results for both GPU and CPU model.
+### Test Model
+This test reads the `SOURCE` from `./inputs` (the local directory or `/watch_dir` if in container), or RTSP server, and creates a video with overlaid detection results for GPU model.
+Results are located in `test_model_results/{MODEL_NAME}`.
```bash
python test_models.py -s "${SOURCE}"
```
@@ -216,20 +258,34 @@ The arguments available for this test are as follows:
| -s SOURCE,
--source SOURCE | anduril_swarm_8K.mp4 | Video filename (located in /inputs) or RTSP target stream endpoint |
| --no-custom | - | Enable if using Ultralytics YOLO model |
| -m MODEL_NAME,
--model MODEL_NAME | drone_detection | Name of model. Required if `--no-custom` is enabled. |
-| --type {object,motion} | None | Filter by detection type (object or motion) |
-| --device {cpu,gpu} | None | Filter by device (cpu or gpu) |
+
+
+
+### Test Readers
+This test reads the `SOURCE` from `./inputs` (the local directory or `/watch_dir` if in container), or RTSP server, and creates a video with overlaid detection results for GPU model.
+Results are located in `test_readers_results/{SOURCE_NAME}`.
+```bash
+python test_readers.py -s "${SOURCE}"
+```
+
+The arguments available for this test are as follows:
+| Argument | Default | Description |
+| -------- | ------- | ----------- |
+| -s SOURCE,
--source SOURCE | anduril_swarm_8K.mp4 | Video filename (located in ./inputs) |
+| --debug | False | Enable debug message |
### Test Detections
This test reads the `SOURCE` from `./inputs`, and evaluates the detection pipeline only.
+Results are located in `test_detections_results/{MODEL_NAME}/{SOURCE_NAME}`.
To run the detection test, use the following:
```bash
python test_detections.py --source "${SOURCE}"
```
-You also have control on which scenario, model, etc. to use during testinh.
+You also have control on which scenario, model, etc. to use during testing.
The arguments available are as follows:
| Argument | Default | Description |
| -------- | ------- | ----------- |
@@ -237,7 +293,6 @@ The arguments available are as follows:
| --no-custom | - | Enable if using Ultralytics YOLO model |
| -m MODEL_NAME,
--model MODEL_NAME | drone_detection | Name of model. Required if `--no-custom` is enabled. |
| --type {object,motion} | None | Filter by detection type (object or motion). "motion" shows the ROI while "object" shows detection results. |
-| --device {cpu,gpu,all} | all | Filter by target hardware. |
| --sf | None | Filter test by Smart Filtering pipeline |
| --debug | False | Enable debug message |
| -n DEBUG_FRAME_LIMIT | 100 | Number of frames used for debugging |
@@ -246,13 +301,14 @@ The arguments available are as follows:
### Test Pipeline
This test reads the `SOURCE` from `./inputs` or RTSP server, and uses the stream readers to perform multiple scenarios.
+Results are located in `test_pipeline_results/{MODEL_NAME}/{SOURCE_NAME}`.
To run the full pipeline test, use the following:
```bash
python test_pipeline.py --source "${SOURCE}"
```
-You also have control on which scenario, model, etc. to use during testinh.
+You also have control on which scenario, model, etc. to use during testing.
The arguments available are as follows:
| Argument | Default | Description |
| -------- | ------- | ----------- |
@@ -262,14 +318,13 @@ The arguments available are as follows:
| --no-custom | - | Enable if using Ultralytics YOLO model |
| -m MODEL_NAME,
--model MODEL_NAME | drone_detection | Name of model. Required if `--no-custom` is enabled. |
| --type {object,motion} | None | Filter by detection type (object or motion). "motion" shows the ROI while "object" shows detection results. |
-| --device {cpu,gpu,all} | all | Filter by target hardware. |
| --sf | None | Filter test by Smart Filtering pipeline |
| --debug | False | Enable debug message |
#### Scenario 1: Invalid RTSP URL
-Scenario 1 is a simple test to verify the pipeline ends if an invalid input is provided. For this test, an invalie RTSP url is provided, there are 5 retry attempts, and after a failed connection, the test ends. By default, the test iterates over available devices ("cpu" and "gpu").
+Scenario 1 is a simple test to verify the pipeline ends if an invalid input is provided. For this test, an invalid RTSP url is provided, there are 5 retry attempts, and after a failed connection, the test ends. By default, the test iterates over available devices ("cpu" and "gpu").
#### Scenario 2: Stability & Throughput
@@ -279,34 +334,15 @@ Scenario 2 evaluates the stability and throughput of the readers by continuously
#### Scenario 3: Video Clip Generation
Scenario 3 evaluates the video clip generation process in the pipeline.
In the pipeline, we segment the input video into 10 sec clips for easy retrieval and playback associated with results in the query UI. By default, the test iterates over available devices ("cpu" and "gpu").
-The resulting video clips generated from this test are located in `test_pipeline_results/scenario3_*`.
+The resulting video clips generated from this test are located in `test_pipeline_results/{MODEL_NAME}/{SOURCE_NAME}/scenario3_{DEVICE}`.
#### Scenario 4: Pipeline (Without Sending Metadata)
Scenario 4 mimics the entire pipeline excluding sending the generated metadata to VDMS. By default, the test iterates over available devices ("cpu" and "gpu"), detection type ("object" or "motion"), and with and without SF.
-The resulting video clips and the detection results are located in `test_pipeline_results/scenario4_*`.
+The resulting video clips and the detection results are located in `test_pipeline_results/{MODEL_NAME}/{SOURCE_NAME}/scenario4_{DEVICE}`.
-In the case of testing the Smart Filtering pipeline, to display the ROIs, you can use the following, which will test for both gpu and cpu.
+In the case of testing the Smart Filtering pipeline, to display the ROIs, you can use the following, which will test for GPU.
```bash
-python test_pipeline.py --scenario 4 --source "${SOURCE}" --type motion
+python test_pipeline.py --source "${SOURCE}" --scenario 4 --sf --type motion
```
-
-## Pipeline Deployment
-
-### Deployment
-./stop.sh –p
-./start_app.sh –e GPU –o –m drone_detection
-
-
-### Visualization: View live detection
-GOTO: http://:30077/
-
-### Query: View Page to Query Videos
-GOTO: http://:30007/
-
-
-### Shutdown
-./stop.sh –p
-
-
diff --git a/doc/sample-ui.gif b/doc/sample-ui.gif
index 0138761..eacb03f 100644
Binary files a/doc/sample-ui.gif and b/doc/sample-ui.gif differ
diff --git a/fastapi/.vscode/launch.json b/fastapi/.vscode/launch.json
index 5ea2550..96f4eb1 100644
--- a/fastapi/.vscode/launch.json
+++ b/fastapi/.vscode/launch.json
@@ -5,75 +5,130 @@
"version": "0.2.0",
"configurations": [
{
- "name": "Python Debugger: Pipeline",
+ // Test the model on provided RTSP URL or video file
+ "name": "Eval Pipeline",
"type": "debugpy",
"request": "launch",
- "program": "${workspaceFolder}/tests/test_pipeline.py",
+ "program": "${workspaceFolder}/tests/test_eval.py",
+ "cwd": "${workspaceFolder}/tests",
+ "python": "/opt/venv/bin/python",
"console": "integratedTerminal",
+ "justMyCode": false,
"subProcess": false,
+ "args": [
+ "--type", "object",
+ ]
+ },
+ {
+ // Test the model on provided RTSP URL or video file
+ "name": "Test Model",
+ "type": "debugpy",
+ "request": "launch",
+ "program": "${workspaceFolder}/tests/test_model.py",
+ "cwd": "${workspaceFolder}/tests",
+ "python": "/opt/venv/bin/python",
+ "console": "integratedTerminal",
"justMyCode": false,
+ "subProcess": false,
"args": [
- "--duration", "2",
-
- // AVAILABLE SCENARIO PARAMS
- // Use --scenario to specify tests by number (1 - 4)
- // 1 - 3: --device (cpu or gpu)
- // 4: --device, --sf, --type (object or motion)
-
// VIDEOS EXPECTED TO BE rstp url OR IN inputs/ or /watch_dir
// "--source", "rtsp://:/",
// "--source", ".mp4",
]
},
-
{
- "name": "Python Debugger: Model",
+ // Independently test the stream readers on provided RTSP URL or video file
+ "name": "Test Readers (Independent)",
"type": "debugpy",
"request": "launch",
- "program": "${workspaceFolder}/tests/test_model.py",
+ "program": "${workspaceFolder}/tests/test_readers.py",
+ "cwd": "${workspaceFolder}/tests",
+ "python": "/opt/venv/bin/python",
+ "console": "integratedTerminal",
+ "justMyCode": false,
+ "subProcess": false,
+ "args": [
+ // VIDEOS EXPECTED TO BE rstp url OR IN inputs/ or /watch_dir
+ // "--source", "rtsp://:/",
+ // "--source", ".mp4"
+ ]
+ },
+ {
+ // Independently test the detection pipeline (with and without Smart Filtering) only
+ "name": "Test Detections (Independent)",
+ "type": "debugpy",
+ "request": "launch",
+ "program": "${workspaceFolder}/tests/test_detections.py",
"cwd": "${workspaceFolder}/tests",
+ "python": "/opt/venv/bin/python",
"console": "integratedTerminal",
"justMyCode": false,
"subProcess": false,
+ "env":{
+ "PYTHONTRACEMALLOC":"20",
+ },
"args": [
// VIDEOS EXPECTED TO BE rstp url OR IN inputs/ or /watch_dir
+ // "--source", "rtsp://:/",
// "--source", ".mp4",
+ "--type", "object"
]
},
{
- "name": "Python Debugger: Smart filtering",
+ // Test pipeline with various scenarios
+ "name": "Test Pipeline",
"type": "debugpy",
"request": "launch",
"program": "${workspaceFolder}/tests/test_pipeline.py",
+ "cwd": "${workspaceFolder}/tests",
+ "python": "/opt/venv/bin/python",
"console": "integratedTerminal",
"subProcess": false,
"justMyCode": false,
"args": [
- // SF SCENARIO PARAMS
- // 4: --device, --sf, --type (object or motion)
- "--scenario", "4",
- "--sf",
- "--type", "motion",
-
// VIDEOS EXPECTED TO BE rstp url OR IN inputs/ or /watch_dir
// "--source", "rtsp://:/",
// "--source", ".mp4",
+
+ // AVAILABLE SCENARIO PARAMS
+ // Use --scenario to specify tests by number (1 - 4)
+ // 4: --sf, --type (object or motion)
+
+ "--duration", "1",
+ "--type", "object"
]
},
{
- "name": "Python Debugger: test_detections",
+ // Test Smart Filtering within pipeline to display the ROIs
+ "name": "Test Pipeline: Smart filtering Only (Scenario 4)",
"type": "debugpy",
"request": "launch",
- "program": "${workspaceFolder}/tests/test_detections.py",
+ "program": "${workspaceFolder}/tests/test_pipeline.py",
"cwd": "${workspaceFolder}/tests",
+ "python": "/opt/venv/bin/python",
"console": "integratedTerminal",
- "justMyCode": false,
"subProcess": false,
+ "justMyCode": false,
"args": [
- "--type", "object",
+ // VIDEOS EXPECTED TO BE rstp url OR IN inputs/ or /watch_dir
+ // "--source", "rtsp://:/",
// "--source", ".mp4",
- // "-m", "yolo11n", "--no-custom",
+
+ // SF SCENARIO PARAMS
+ // 4: --sf, --type (object or motion)
+ "--scenario", "4",
+ "--sf",
+ "--type", "object",
+
+ "--duration", "1",
]
+ },
+ {
+ "name": "Python Debugger: Current File",
+ "type": "debugpy",
+ "request": "launch",
+ "program": "${file}",
+ "console": "integratedTerminal"
}
]
}
diff --git a/fastapi/Dockerfile b/fastapi/Dockerfile
index a771678..1649163 100644
--- a/fastapi/Dockerfile
+++ b/fastapi/Dockerfile
@@ -42,21 +42,28 @@ RUN apt-get update && \
gfortran && \
# Install necessary CUDA packages
curl -fsSL https://developer.download.nvidia.com/compute/cuda/repos/ubuntu2204/x86_64/cuda-keyring_1.1-1_all.deb -o /tmp/cuda-keyring.deb && \
- dpkg -i /tmp/cuda-keyring.deb && rm /tmp/cuda-keyring.deb && \
- apt-get update && \
+ dpkg -i /tmp/cuda-keyring.deb && rm /tmp/cuda-keyring.deb
+
+ARG CUDA_MAJOR_VERSION=12
+ARG CUDA_MINOR_VERSION=8
+
+# hadolint ignore=DL3008
+RUN apt-get update && \
apt-get install -y -q --no-install-recommends \
- cuda-toolkit-12-4 \
- libcudnn9-dev-cuda-12 \
- libnpp-dev-12-4 \
- && \
+ cuda-toolkit-${CUDA_MAJOR_VERSION}-${CUDA_MINOR_VERSION} \
+ libcudnn9-dev-cuda-${CUDA_MAJOR_VERSION} && \
rm -rf /var/lib/apt/lists/* && apt-get clean
RUN python3 -m venv ${VIRTUAL_ENV} && \
${VIRTUAL_ENV}/bin/pip install --no-cache-dir \
- "pip==26.0.1" \
+ "pip==26.2.1" \
"torch==2.10.0" \
"torchvision==0.25.0" \
- "numpy==1.26.0"
+ "numpy==1.26.4"
+
+# Ensure paths capture the version dynamically
+ENV PATH="/usr/local/cuda/bin:${PATH}"
+ENV LD_LIBRARY_PATH="/usr/local/cuda/lib64:${LD_LIBRARY_PATH}"
# OPENCV W/ CUDA SUPPORT
ENV PYTHON_VERSION=3.10
@@ -74,6 +81,8 @@ RUN git clone --branch ${OPENCV_VERSION} --depth 1 https://github.com/opencv/ope
WORKDIR ${DEPENDENCY_DIR}/opencv/build
RUN ln -s /usr/include/x86_64-linux-gnu/cudnn*.h /usr/local/cuda/include/ && \
ln -s /usr/lib/x86_64-linux-gnu/libcudnn*.so* /usr/local/cuda/lib64/
+
+# hadolint ignore=SC2046
RUN cmake -D CMAKE_BUILD_TYPE=RELEASE \
-D CMAKE_INSTALL_PREFIX="${VIRTUAL_ENV}" \
-D OPENCV_EXTRA_MODULES_PATH="${DEPENDENCY_DIR}/opencv_contrib/modules" \
@@ -84,7 +93,7 @@ RUN cmake -D CMAKE_BUILD_TYPE=RELEASE \
-D WITH_TBB=ON \
-D WITH_V4L=ON \
-D OPENCV_DNN_CUDA=ON \
- -D CUDA_ARCH_BIN=70,75,80,86,89,90 \
+ -D CUDA_ARCH_BIN=80,86,89,90 \
-D CUDA_FAST_MATH=ON \
-D CUDA_TOOLKIT_ROOT_DIR=/usr/local/cuda \
-D CUDNN_INCLUDE_DIR=/usr/include/x86_64-linux-gnu \
@@ -93,8 +102,8 @@ RUN cmake -D CMAKE_BUILD_TYPE=RELEASE \
-D BUILD_opencv_cudaarithm=ON \
-D BUILD_opencv_cudafilters=ON \
-D BUILD_opencv_cudacodec=ON \
- -D WITH_NVCUVID=ON \
- -D WITH_NVCUVENC=ON \
+ -D WITH_NVCUVID=OFF \
+ -D WITH_NVCUVENC=OFF \
-D WITH_VAAPI=ON \
-D WITH_FFMPEG=ON \
-D BUILD_opencv_python3=ON \
@@ -108,12 +117,13 @@ SHELL ["/bin/bash", "-o", "pipefail", "-c"]
RUN mkdir -p /tmp/cv2_deps && \
PY_CV2_SO=$(find "${VIRTUAL_ENV}" -name "cv2*.so" | head -n 1) && \
ldd "${PY_CV2_SO}" | grep "=> /" | awk '{print $3}' | xargs -I '{}' cp -v '{}' /tmp/cv2_deps/ && \
- cp -v /usr/local/cuda/lib64/libnpp*.so* /tmp/cv2_deps/ && \
- cp -v /usr/local/cuda/lib64/libcudart.so* /tmp/cv2_deps/ && \
- cp -P /usr/local/cuda/lib64/libnvrtc.so* /tmp/cv2_deps/ && \
+ # FIX: Robust fallback copying to grab NPP/CUDA libs regardless of whether they land in /usr/local or system path channels
+ (cp -v /usr/local/cuda/lib64/libnpp*.so* /tmp/cv2_deps/ || cp -v /usr/lib/x86_64-linux-gnu/libnpp*.so* /tmp/cv2_deps/) && \
+ (cp -v /usr/local/cuda/lib64/libcudart.so* /tmp/cv2_deps/ || cp -v /usr/lib/x86_64-linux-gnu/libcudart.so* /tmp/cv2_deps/) && \
+ (cp -P /usr/local/cuda/lib64/libnvrtc.so* /tmp/cv2_deps/ || cp -P /usr/lib/x86_64-linux-gnu/libnvrtc.so* /tmp/cv2_deps/) && \
+ (cp -v /usr/local/cuda/lib64/libcublas*.so* /tmp/cv2_deps/ || cp -v /usr/lib/x86_64-linux-gnu/libcublas*.so* /tmp/cv2_deps/) && \
cp -v /usr/lib/x86_64-linux-gnu/libnvidia-encode.so* /tmp/cv2_deps/ || true
-
# Clean up
RUN rm -rf ${DEPENDENCY_DIR}
@@ -121,45 +131,57 @@ RUN rm -rf ${DEPENDENCY_DIR}
FROM openvisualcloud/xeon-ubuntu2204-media-nginx:23.1@sha256:d19eb597dc210134063803630ae2ea1ec84dfd4189138f59551e2f5ed047284a
ARG DEBIAN_FRONTEND=noninteractive
+ARG CUDA_MAJOR_VERSION=12
+ARG CUDA_MINOR_VERSION=8
ENV VIRTUAL_ENV=/opt/venv
ENV PATH="$VIRTUAL_ENV/bin:/usr/local/cuda/bin:${PATH}"
# Install Runtime Libraries, FFmpeg/VA-API headers, and cuDNN
# Includes libva and ffmpeg libraries required for HEVC decoding
# hadolint ignore=DL3008
-RUN apt-get update && apt-get install -y --no-install-recommends \
- ca-certificates \
- curl \
- python3 \
- libgl1 \
- libglib2.0-0 \
- libtbb12 \
- # Critical for Hardware Decoding (HEVC)
- libva2 \
- libva-drm2 \
- libva-x11-2 \
- libavcodec58 \
- libavformat58 \
- libswscale5 \
- libv4l-0 && \
+RUN apt-get update && \
+ apt-get install -y --no-install-recommends \
+ ca-certificates \
+ curl \
+ python3 \
+ libgl1 \
+ libglib2.0-0 \
+ libtbb12 \
+ # Critical for Hardware Decoding (HEVC)
+ libva2 \
+ libva-drm2 \
+ libva-x11-2 \
+ libavcodec58 \
+ libavformat58 \
+ libswscale5 \
+ libv4l-0 && \
# Install necessary CUDA packages
curl -fsSL https://developer.download.nvidia.com/compute/cuda/repos/ubuntu2204/x86_64/cuda-keyring_1.1-1_all.deb -o /tmp/cuda-keyring.deb && \
- dpkg -i /tmp/cuda-keyring.deb && rm /tmp/cuda-keyring.deb && \
+ dpkg -i /tmp/cuda-keyring.deb && rm /tmp/cuda-keyring.deb && \
apt-get update && \
- apt-get install -y -q --no-install-recommends libcudnn9-cuda-12 && \
+ apt-get install -y -q --no-install-recommends \
+ libcudnn9-cuda-${CUDA_MAJOR_VERSION} \
+ libavcodec-dev \
+ libavformat-dev \
+ libavutil-dev \
+ cuda-nvcc-${CUDA_MAJOR_VERSION}-${CUDA_MINOR_VERSION} && \
rm -rf /var/lib/apt/lists/*
# Copy the entire pre-compiled virtual environment from the build stage
COPY --from=build ${VIRTUAL_ENV} ${VIRTUAL_ENV}
-# Copy NPP (NVIDIA Performance Primitives) libs - Required for OpenCV CUDA
-# COPY --from=build /usr/local/cuda/lib64/libnpp*.so* /usr/local/cuda/lib64/
+
# Copy additional required CUDA math/parallel libs
RUN mkdir -p /usr/local/cuda/lib64
COPY --from=build /tmp/cv2_deps/* /usr/local/cuda/lib64/
+
+# Tells the host engine to expose all GPUs and utility binaries (like nvidia-smi)
+ENV NVIDIA_VISIBLE_DEVICES=all
+# FIX: Cleaned duplicate line later in file to ensure capabilities are consistently initialized
+ENV NVIDIA_DRIVER_CAPABILITIES=compute,utility,video
ENV LD_LIBRARY_PATH="/usr/local/cuda/lib64:${VIRTUAL_ENV}/lib:${LD_LIBRARY_PATH}"
RUN ldconfig
-ARG DEVICE="CPU"
+ARG DEVICE="GPU"
ENV DEVICE="${DEVICE}"
ARG DEBUG="0"
diff --git a/fastapi/entrypoint.sh b/fastapi/entrypoint.sh
index 850f4a3..694b55c 100644
--- a/fastapi/entrypoint.sh
+++ b/fastapi/entrypoint.sh
@@ -1,6 +1,10 @@
#!/bin/bash
set -e
+echo "Sweeping stale shared memory segments from previous runs..."
+# Delete only Python-specific shared memory blocks to avoid messing up other host processes
+rm -f /dev/shm/psm_*
+
# Run the download script using container env vars
echo "Starting model download..."
python /home/include/models.py -o /var/www/cache/model_classes.json
diff --git a/fastapi/include/__init__.py b/fastapi/include/__init__.py
new file mode 100644
index 0000000..c57681f
--- /dev/null
+++ b/fastapi/include/__init__.py
@@ -0,0 +1,78 @@
+import os
+import warnings
+from multiprocessing import resource_tracker
+
+import torch
+
+# ==============================================================================
+# SUPPRESS WARNINGS
+
+# warnings.filterwarnings(
+# "ignore", category=FutureWarning, message=".*reduce_op` is deprecated.*"
+# )
+warnings.filterwarnings("ignore", category=FutureWarning)
+warnings.filterwarnings("ignore", category=DeprecationWarning)
+warnings.filterwarnings("ignore", category=UserWarning, module="torch.distributed")
+warnings.filterwarnings("ignore", message=".*reduce_op.*")
+warnings.filterwarnings("ignore", message=".*expandable_segments.*")
+
+# ==============================================================================
+# MONKEY PATCH
+
+
+def _patch_resource_tracker_for_shm():
+ """
+ Prevents Python's resource_tracker from tracking shared memory segments.
+ Eliminates BPO-39959 / Issue #82300 KeyError crashes.
+ """
+ _orig_register = resource_tracker.register
+ _orig_unregister = resource_tracker.unregister
+
+ def _safe_register(name, rtype):
+ if rtype == "shared_memory":
+ return # Do not register shared memory with tracker daemon
+ return _orig_register(name, rtype)
+
+ def _safe_unregister(name, rtype):
+ if rtype == "shared_memory":
+ return # Do not send unregister signals for shared memory
+ return _orig_unregister(name, rtype)
+
+ resource_tracker.register = _safe_register
+ resource_tracker.unregister = _safe_unregister
+
+
+_patch_resource_tracker_for_shm()
+
+
+# Cache the hardware availability state globally.
+# This prevents internal framework loops from triggering driver/NVML checks per frame.
+_cuda_available = torch.cuda.is_available()
+
+
+def _patched_is_available():
+ return _cuda_available
+
+
+torch.cuda.is_available = _patched_is_available
+
+# ==============================================================================
+# ENVIRONMENT
+
+os.environ["OMP_NUM_THREADS"] = "4" # "4" #"2"
+os.environ["MKL_NUM_THREADS"] = "2"
+os.environ["PYTORCH_ALLOC_CONF"] = "expandable_segments:True"
+
+# Suppress low-delay reference block warnings from OpenCV/PyAV/FFmpeg
+os.environ["OPENCV_FFMPEG_LOGLEVEL"] = "-8"
+os.environ["OPENCV_LOG_LEVEL"] = "OFF"
+
+try:
+ torch.set_num_interop_threads(1)
+ torch.set_num_threads(2)
+except RuntimeError:
+ # Safe graceful fallback if a process-level fork duplicated context maps
+ pass
+
+# Force disable tracking states to stop graph allocation recursion
+torch.set_grad_enabled(False)
diff --git a/fastapi/include/default_configs.py b/fastapi/include/default_configs.py
index 9046094..41b4fba 100644
--- a/fastapi/include/default_configs.py
+++ b/fastapi/include/default_configs.py
@@ -3,7 +3,7 @@
CUSTOM_MODEL_FLAG_DEFAULT = False
DEBUG_DEFAULT = "0"
DEVICE_DEFAULT = "CPU"
-MAX_WORKERS = 4 # 6
+MAX_WORKERS = 4 # 4 # 6
OMIT_DETECTIONS_FLAG_DEFAULT = False
SHARED_OUTPUT_DEFAULT = "/var/www/mp4"
TEST_MODE_DEFAULT = False
@@ -12,18 +12,19 @@
# VIDEO WRITER
CLIP_DURATION_DEFAULT = 10
-TARGET_FPS = 15
+TARGET_FPS = 15.0
# VDMS
DBHOST_DEFAULT = "vdms-service"
DBPORT_DEFAULT = 55555
-ENABLE_QUERYING_DEFAULT = True # False, True
+ENABLE_QUERYING_DEFAULT = False # False, True
INGESTION_DEFAULT = "object" # object,face
UDF_HOST_DEFAULT = "udf-service"
UDF_PORT_DEFAULT = 5011
# MODEL
-DETECTION_THRESHOLD_DEFAULT = 0.25
+DETECTION_THRESHOLD_DEFAULT = 0.1 # 0.25
+IOU_THRESHOLD_DEFAULT = 0.1 # 0.7
DYNAMIC_FLAG_DEFAULT = True
HALF_FLAG_DEFAULT = True
MAX_DETECTIONS = 100
@@ -36,6 +37,7 @@
# SMART FILTERING (RESIZE, BKGD SUB, THRESHOLD, DILATE)
# Enable if resolution > 2K 1920x1080
SMART_FILTERING_PIXEL_CONSTRAINT = 1920 * 1080
+SMART_FILTERING_WARMUP = False # True, False
BKGD_SUB_INCLUDE_HISTORY = False # True, False
BKGD_SUB_INCLUDE_HISTORY_DILATE_KERNEL_SIZE = 15 # 21 #15
BKGD_SUB_INCLUDE_HISTORY_METHOD = "or" # "and", or
@@ -43,12 +45,12 @@
BKGD_SUB_MOG2_DETECTSHADOWS = False
BKGD_SUB_MOG2_HISTORY = int(2 * TARGET_FPS)
BKGD_SUB_MOG2_LR = 1 / BKGD_SUB_MOG2_HISTORY # 0.002 # 1 / BKGD_SUB_MOG2_HISTORY
-BKGD_SUB_MOG2_VARTHRESHOLD = 10
+BKGD_SUB_MOG2_VARTHRESHOLD = 50 # 30 # 20 #10
DILATE_KERNEL_SIZE = 5 # (3, 3)
RESIZE_FLAG_DEFAULT = False
ROI_BB_FULL_RES_PADDING = int(0.02 * MODEL_W) # 10, ~13
ROI_CONTAINMENT_THRESH = 0.95
-ROI_DISTANCE_THRESH_RATIO = 0.05
+ROI_DISTANCE_THRESH_RATIO = 0.05 # 0.10 #.15 # 0.05
ROI_MAX_RELATIVE_SIZE_RATIO = 1 # 1 # 0.8
ROI_MERGE_SIZE_LIMIT = MODEL_W * 1.25
ROI_MIN_AREA_RATIO = 0.01
diff --git a/fastapi/include/detectors.py b/fastapi/include/detectors.py
new file mode 100644
index 0000000..2a93006
--- /dev/null
+++ b/fastapi/include/detectors.py
@@ -0,0 +1,4362 @@
+# ==============================================================================
+# LOGGING
+
+import logging
+import sys
+
+logging.basicConfig(
+ level=logging.INFO,
+ # format="%(asctime)s - %(name)s - %(levelname)s - %(message)s",
+ format="%(asctime)s [%(levelname)s] %(name)s (%(filename)s:%(lineno)d) - %(message)s",
+ handlers=[logging.StreamHandler(sys.stdout)],
+)
+
+main_app_logger = logging.getLogger(__name__)
+
+
+# ==============================================================================
+# IMPORTS
+
+import time
+import traceback
+from collections import deque
+from pathlib import Path
+
+import cupy
+import cupyx.scipy
+import cupyx.scipy.ndimage
+import cv2
+import numpy as np
+import torch
+import torch.nn.functional as F
+from ultralytics.engine.results import Boxes, Results
+from ultralytics.utils.nms import non_max_suppression
+
+sys.path.insert(1, str(Path(__file__).parent.parent))
+from include.models import get_model
+from include.utils import (
+ get_bb_overlay,
+ get_bounds_kernel,
+ merge_boxes_gpu,
+)
+
+# ==============================================================================
+# CONFIGURATIONS
+STREAM_ARG = False
+PADDING_SCALE = 0.5
+
+
+class BaseObjectDetector:
+ def __init__(
+ self,
+ config,
+ device="cuda",
+ timer_enabled=True,
+ resize_hw=(640, 640),
+ frame_hw=(4320, 7680),
+ target_fps=15.0,
+ result_dir="/tmp",
+ run_name=None,
+ debug_frame_limit=-1,
+ **kwargs,
+ ): # , label_source=["drone"]
+ self.device_input = device
+ self.timer_enabled = timer_enabled
+ self.config = config
+ self.target_fps = target_fps
+ self.resize_h, self.resize_w = int(resize_hw[0]), int(resize_hw[1])
+ self.frame_height, self.frame_width = int(frame_hw[0]), int(frame_hw[1])
+ # self.label_source = label_source
+ self.compiled_no_grad_gate = torch.no_grad()
+
+ if run_name is None:
+ run_type = "sf" if self.config.sf_enabled else "yolo"
+ det_type = self.config.DETECTION_TYPE
+ device_name = self.config.DEVICE.lower()
+ run_name = f"{run_type}_{det_type}_{device_name}"
+ self._testMethodName = run_name
+
+ self.result_dir = result_dir
+ # self.debug_frame_limit = debug_frame_limit
+ self.debug_frame_limit = (
+ debug_frame_limit
+ if self.config.DEBUG_FLAG and debug_frame_limit > -1
+ else -1
+ )
+
+ provided_model = kwargs.get("model")
+ self.setup_model(provided_model)
+ self.initialize()
+
+ def model_warmup(self, H=640, W=640):
+ H, W = int(H), int(W)
+ # Move the dummy input creation inside a no_grad block
+ with self.compiled_no_grad_gate:
+ # main_app_logger.info(f"Starting warmup for {self.name}...")
+ dummy_input = torch.zeros((1, 3, H, W)).to(self.device_input)
+
+ # Perform iterations directly on the main thread
+ for i in range(5):
+ dummy_result = self.run_model(
+ dummy_input,
+ imgsz=(H, W),
+ batch=1,
+ device_input=self.device_input,
+ stream=STREAM_ARG,
+ )
+ # if i == 0 and hasattr(self.model, "predictor") and self.model.predictor:
+ # self.cached_predictor = self.model.predictor
+ # self.cached_predictor.profile = False
+
+ del dummy_input, dummy_result
+
+ if hasattr(self.model, "predictor") and self.model.predictor is not None:
+ predictor = self.model.predictor
+
+ # Disable the internal framework statistics stopwatch loops
+ # This wipes out the trailing 2.3ms context synchronization gates completely!
+ predictor.stride = 32
+ predictor.profile = False
+
+ # CRITICAL OPTIMIZATION: Tell the post-processor to skip input image array compilation.
+ # This completely satisfies the internal checks, bypassing convert_torch2numpy_batch
+ # and dropping your 17.99ms PCIe bus device-to-host download stall down to 0 ms.
+ predictor.save_dir = None
+ predictor.args.visualize = False
+ predictor.args.show = False
+ predictor.args.save = False
+ self.cached_predictor = predictor
+
+ # Force GPU to finish before returning
+ # if self.device_input == "cuda":
+ # torch.cuda.synchronize()
+
+ # Pin a lightweight tensor to prevent downstream empty_cache() calls
+ # from destroying the compiled model weight layouts in memory
+ self.__persistent_vram_lock = torch.zeros((1,), device=self.device_input)
+
+ main_app_logger.info("Warmup complete")
+
+ def setup_model(self, provided_model, force_export=False):
+ if (
+ self.frame_width * self.frame_height
+ ) <= self.config.SMART_FILTERING_PIXEL_CONSTRAINT:
+ if "_noSF" not in self.config.model_path:
+ oldpath = Path(self.config.model_path)
+ old_modelname = self.config.MODEL_NAME
+ self.config.MODEL_NAME = f"{old_modelname}_noSF"
+ new_model_name = oldpath.name.replace(
+ old_modelname, self.config.MODEL_NAME
+ )
+ self.config.model_path = str(oldpath.parent / new_model_name)
+
+ if provided_model is not None and not isinstance(provided_model, str):
+ self.model = provided_model
+ self.label_source = [v for k, v in self.model.names.items()]
+ self.label_source_dict = self.model.names
+ else:
+ run_platform_name = "engine" if "cuda" in self.device_input else "openvino"
+ self.model, _, self.label_source = get_model(
+ Path(self.config.model_path).parent,
+ self.config.MODEL_NAME.replace("_noSF", ""),
+ run_platform_name,
+ self.device_input,
+ batch=self.config.MODEL_MAX_BATCH_SIZE,
+ force_export=force_export,
+ sf_enabled=self.config.sf_enabled,
+ model_h=self.resize_h,
+ model_w=self.resize_w,
+ )
+ self.label_source_dict = self.model.names
+
+ # if run_platform_name == "openvino":
+ # self.model.predictor.args.embed = False
+
+ # Warmup Model
+ if not self.config.sf_enabled:
+ self.model_warmup(self.frame_height, self.frame_width)
+ else:
+ self.model_warmup(self.resize_h, self.resize_w)
+
+ def run_model(
+ self,
+ frame,
+ imgsz=(640, 640),
+ batch=1,
+ device_input="cuda",
+ stream=False,
+ ):
+ if (
+ isinstance(frame, torch.Tensor)
+ # and device_input == "cuda"
+ # and getattr(self, "cached_predictor", None) is not None
+ ):
+ if frame.dtype == torch.uint8:
+ # Use standard out-of-place division to prevent byte truncation errors
+ frame = frame.to(dtype=torch.float32).mul(1.0 / 255.0)
+ else:
+ # If it's already a float tensor array block, in-place scaling is safe
+ frame = frame.mul_(1.0 / 255.0)
+ raw_module = None
+
+ if hasattr(self, "cached_predictor") and self.cached_predictor is not None:
+ raw_module = self.cached_predictor.model
+
+ with torch.inference_mode():
+ # Establish your pipeline invariant:
+ # frame = [B,C,H,W], uint8, CUDA, [0,255].
+ # if frame.dtype == torch.uint8:
+ # dtype = (
+ # torch.float16
+ # if getattr(raw_module, "fp16", False)
+ # else torch.float32
+ # )
+ # frame = frame.to(dtype=dtype).mul_(1.0 / 255.0)
+ # elif not frame.is_floating_point():
+ # frame = frame.float()
+
+ if raw_module is not None:
+ # start = torch.cuda.Event(enable_timing=True)
+ # end = torch.cuda.Event(enable_timing=True)
+
+ # start.record()
+ # If input is guaranteed 640x640, don't pad here.
+ raw_output = raw_module(frame)
+ # end.record()
+ # torch.cuda.synchronize()
+ # inference_ms = start.elapsed_time(end)
+
+ # start.record()
+ decoded_results = non_max_suppression(
+ prediction=raw_output,
+ conf_thres=self.config.DETECTION_THRESHOLD,
+ iou_thres=self.config.IOU_THRESHOLD,
+ max_det=self.config.MAX_DETECTIONS,
+ classes=None,
+ )
+ # end.record()
+ # torch.cuda.synchronize()
+ # nms_ms = start.elapsed_time(end)
+
+ # main_app_logger.info(f"Inference: {inference_ms:.3f} ms")
+ # main_app_logger.info(f"NMS: {nms_ms:.3f} ms")
+
+ results = []
+ dummy_img = np.empty(
+ (1, 1, 3),
+ dtype=np.uint8,
+ )
+
+ for pred in decoded_results:
+ if pred is None or pred.shape[0] == 0:
+ pred = torch.empty(
+ (0, 6),
+ dtype=frame.dtype,
+ device=frame.device,
+ )
+
+ pre_calculated_boxes = Boxes(
+ pred,
+ imgsz,
+ )
+
+ # Construct a conforming container using native VRAM references
+ res = Results(
+ orig_img=dummy_img,
+ path="",
+ names=self.label_source_dict,
+ boxes=None, # Bypasses internal allocation thrashing! # pred,
+ )
+ res.boxes = pre_calculated_boxes
+ results.append(res)
+
+ del decoded_results, raw_output
+ return results
+
+ # Fallback to normal Ultralytics path.
+ with torch.inference_mode():
+ if getattr(self, "cached_predictor", None) is not None:
+ return self.cached_predictor(frame)
+ else:
+ static_img_h = int(imgsz[0]) # Maps explicitly to your fixed 640 target
+ static_img_w = int(imgsz[1])
+ return self.model.predict(
+ frame,
+ imgsz=(static_img_h, static_img_w),
+ batch=batch,
+ device=device_input,
+ verbose=False,
+ stream=stream,
+ conf=self.config.DETECTION_THRESHOLD,
+ iou=self.config.IOU_THRESHOLD,
+ max_det=self.config.MAX_DETECTIONS,
+ rect=(batch == 1),
+ profile=False,
+ )
+
+ def format_bbs_and_frame_4_detection_v1(self, bbs_full_res, device_frame):
+ clean_bbs = []
+
+ if self.config.sf_enabled and bbs_full_res is not None:
+ if torch.is_tensor(bbs_full_res):
+ # Check hardware registry constraints dynamically
+ if bbs_full_res.is_cuda:
+ # 🚀 GPU Target: Stage non-blocking fast copy into page-locked host memory
+ # via PCIe DMA background channels to completely bypass lock-stalls
+ # pinned_host_tensor = bbs_full_res.detach()#.to(
+ # # device="cpu",
+ # # non_blocking=True,
+ # # memory_format=torch.contiguous_format,
+ # # )
+ # clean_bbs = pinned_host_tensor.numpy()
+
+ # Moving the data to the host via non_blocking=True allows the PCIe
+ # DMA engine to handle the transfer, freeing up your CPU thread context!
+ cpu_tensor = bbs_full_res.detach().to(
+ device="cpu", non_blocking=True
+ )
+
+ # Extract the numpy array safely without triggering a global stream sync fence
+ clean_bbs = cpu_tensor.numpy()
+ else:
+ # 🚀 CPU Target: Direct view mapping without triggering device-transfer hooks
+ clean_bbs = bbs_full_res.detach().numpy()
+ else:
+ clean_bbs = np.array(bbs_full_res)
+
+ # Layout tracking adjustments (Universal compatibility layer)
+ det_frame = device_frame
+ if torch.is_tensor(det_frame):
+ det_frame = det_frame.contiguous() # byte()
+
+ if det_frame.ndim == 4:
+ det_frame = det_frame.squeeze(0).permute(1, 2, 0)
+ elif det_frame.shape == 3:
+ det_frame = det_frame.permute(1, 2, 0)
+
+ merged = clean_bbs if self.config.sf_enabled else None
+
+ return merged, det_frame
+
+ def format_bbs_and_frame_4_detection(self, bbs_full_res, device_frame):
+ clean_bbs = bbs_full_res if self.config.sf_enabled else None
+
+ # if self.config.sf_enabled and bbs_full_res is not None:
+ # if torch.is_tensor(bbs_full_res):
+ # # Check hardware registry constraints dynamically
+ # if bbs_full_res.is_cuda:
+ # # 🚀 GPU Target: Stage non-blocking fast copy into page-locked host memory
+ # # via PCIe DMA background channels to completely bypass lock-stalls
+ # # pinned_host_tensor = bbs_full_res.detach()#.to(
+ # # # device="cpu",
+ # # # non_blocking=True,
+ # # # memory_format=torch.contiguous_format,
+ # # # )
+ # # clean_bbs = pinned_host_tensor.numpy()
+
+ # # Moving the data to the host via non_blocking=True allows the PCIe
+ # # DMA engine to handle the transfer, freeing up your CPU thread context!
+ # cpu_tensor = bbs_full_res.detach().to(device="cpu", non_blocking=True)
+
+ # # Extract the numpy array safely without triggering a global stream sync fence
+ # clean_bbs = cpu_tensor.numpy()
+ # else:
+ # # 🚀 CPU Target: Direct view mapping without triggering device-transfer hooks
+ # clean_bbs = bbs_full_res.detach().numpy()
+ # else:
+ # clean_bbs = np.array(bbs_full_res)
+
+ # Layout tracking adjustments (Universal compatibility layer)
+ det_frame = device_frame
+ if torch.is_tensor(det_frame):
+ # det_frame = det_frame.contiguous() #byte()
+
+ if det_frame.ndim == 4:
+ det_frame = det_frame.squeeze(0) # .permute(1, 2, 0)
+ # elif det_frame.shape == 3:
+ # det_frame = det_frame.permute(1, 2, 0)
+
+ if not det_frame.is_contiguous():
+ det_frame = det_frame.contiguous()
+
+ return clean_bbs, det_frame
+
+ def motion2metadata(self, merged, frame_count_target):
+ metadata = {}
+ if merged is not None and merged.shape[0] > 0:
+ # merged = merged / self.scales_tensor.cpu().numpy().reshape(-1, 4)
+ merged = merged.div(self.scales_tensor.view(-1, 4))
+
+ # Calculate area for each box: (xmax - xmin) * (ymax - ymin)
+ widths = merged[:, 2] - merged[:, 0]
+ heights = merged[:, 3] - merged[:, 1]
+ areas = widths * heights
+ # max_area = np.max(areas) if np.max(areas) > 0 else 1.0
+
+ if torch.is_tensor(areas):
+ # Use native PyTorch reduction math if areas is a device tensor block
+ raw_max = torch.max(areas).item() if areas.numel() > 0 else 0.0
+ max_area = raw_max if raw_max > 0 else 1.0
+ else:
+ # Fallback standard array parsing logic for host NumPy sequences
+ raw_max = np.max(areas) if len(areas) > 0 else 0.0
+ max_area = raw_max if raw_max > 0 else 1.0
+
+ # Format motion results like detection results for evaluation
+ for i, bb in enumerate(merged):
+ disp_x, disp_y, disp_x2, disp_y2 = bb
+ disp_w = disp_x2 - disp_x
+ disp_h = disp_y2 - disp_y
+
+ if disp_w > 2 and disp_h > 2:
+ obj_id = len(metadata)
+ # num_objs += 1
+ framenum_str = f"{frame_count_target:04d}_{obj_id:04d}"
+ score = areas[i] / max_area
+
+ metadata[framenum_str] = {
+ "frameId": int(frame_count_target),
+ "bbId": framenum_str,
+ "bbox": {
+ "x": int(disp_x),
+ "y": int(disp_y),
+ "height": int(disp_h),
+ "width": int(disp_w),
+ "object": "",
+ "object_det": {
+ "confidence": score,
+ "frameH": int(self.resize_h),
+ "frameW": int(self.resize_w),
+ },
+ },
+ }
+
+ return metadata
+
+ def initialize(self):
+ self.device = self.config.DEVICE
+ self.device_input = self.config.device_input
+ self.disp_w, self.disp_h = self.config.DISPLAY_FRAME_SIZE
+ self.resize_h, self.resize_w = [self.config.MODEL_H, self.config.MODEL_W]
+ # self.fixed_inference_batch = torch.empty(
+ # (
+ # self.config.MODEL_MAX_BATCH_SIZE,
+ # 3,
+ # self.config.MODEL_H,
+ # self.config.MODEL_W,
+ # ),
+ # dtype=torch.half,
+ # device=self.device_input,
+ # )
+
+ # DEBUG FUNCTIONS --------------------------------------------
+ def debug_save_mask(self, frame_source, frame_num, rois=None, gt_boxes=None):
+ debug_dir = self.result_dir / "debug_stages" / self._testMethodName / "mask"
+ debug_dir.mkdir(parents=True, exist_ok=True)
+
+ save_path = (
+ debug_dir / f"frame_{frame_num:04d}_mask.jpg"
+ ) # f"mask_{frame_num:04d}.jpg"
+
+ # cv2.imwrite(
+ # # str(stage_debug_dir / f"frame_{f_num:04d}_stage6_threshold.jpg"),
+ # str(stage_debug_dir / f"frame_{self.frame_count_target:04d}_stagerois_merged.jpg"),
+ # display_frame,
+ # )
+
+ if frame_num > self.config.DEBUG_FRAME_LIMIT:
+ return
+
+ # # Download or copy the data
+ # if hasattr(frame_source, "download"):
+ # img_cpu = frame_source.download()
+ # elif torch.is_tensor(frame_source):
+ # # .contiguous() fixes the horizontal "shredding"/static look
+ # temp = frame_source.squeeze(0) if frame_source.ndim == 4 else frame_source
+ # img_cpu = temp.permute(1, 2, 0).contiguous().cpu().numpy()
+ # else:
+ # # For numpy arrays (like your pinned memory), ensure memory is linear
+ # img_cpu = np.ascontiguousarray(frame_source)
+
+ # # Fix Visibility (Normalization)
+ # # If float, scale to 0-255. If uint8, leave as is to avoid "neon" colors.
+ # if img_cpu.dtype != np.uint8:
+ # if img_cpu.max() <= 1.0:
+ # img_cpu = (img_cpu * 255).clip(0, 255).astype(np.uint8)
+ # else:
+ # img_cpu = img_cpu.astype(np.uint8)
+
+ # Download/Copy the frame
+ if torch.is_tensor(frame_source):
+ # .contiguous() is CRITICAL here to fix the "shredded" look
+ temp = frame_source.squeeze(0) if frame_source.ndim == 4 else frame_source
+ # img_cpu = temp.permute(1, 2, 0).contiguous().cpu().numpy()
+ img_cpu = temp.contiguous().cpu().numpy()
+ img_cpu = cv2.cvtColor(img_cpu, cv2.COLOR_RGB2BGR)
+ elif hasattr(frame_source, "download"):
+ img_cpu = frame_source.download()
+ else:
+ img_cpu = np.ascontiguousarray(frame_source)
+
+ # Fix Shape: restore spatial grid if flattened
+ if img_cpu.ndim == 3 and img_cpu.shape[0] == 1:
+ img_cpu = img_cpu.reshape((self.resize_h, self.resize_w, 3))
+
+ # Fix Visibility: ONLY multiply if it's actually floating point
+ # If uint8 is multiplied by 255, it wraps around and creates "neon" colors
+ if img_cpu.dtype != np.uint8:
+ if img_cpu.max() <= 1.0:
+ img_cpu = (img_cpu * 255).clip(0, 255).astype(np.uint8)
+ else:
+ img_cpu = img_cpu.astype(np.uint8)
+
+ # Color Space: Standardize to BGR for imwrite
+ if len(img_cpu.shape) != 3:
+ img_cpu = cv2.cvtColor(img_cpu, cv2.COLOR_GRAY2BGR)
+
+ h_img, w_img = img_cpu.shape[:2]
+ scale_x = float(w_img) / self.frame_width
+ scale_y = float(h_img) / self.frame_height
+ if gt_boxes is not None:
+ # Factor for converting 8K (original) bb dimensions to display dimensions
+ # disp_w, disp_h = inf_data["mask"].shape[:2] if hasattr(inf_data["mask"], "shape") else inf_data["mask"].size()
+ # scale_display_ox = disp_w / self.frame_width
+ # scale_display_oy = disp_h / self.frame_height
+
+ img_cpu = get_bb_overlay(
+ img_cpu,
+ gt_boxes,
+ (scale_x, scale_y),
+ (w_img, h_img),
+ color=(0, 255, 0), # green
+ )
+
+ # Draw 8K Boxes (Scaled down)
+ if rois is not None and len(rois) > 0:
+ boxes = rois.cpu().tolist() if torch.is_tensor(rois) else rois
+ for box in boxes:
+ x1, y1, x2, y2 = [
+ # int(box[0] * scale_x),
+ # int(box[1] * scale_y),
+ # int(box[2] * scale_x),
+ # int(box[3] * scale_y),
+ max(0, int(box[0] * scale_x)),
+ max(0, int(box[1] * scale_y)),
+ min(w_img - 1, int(box[2] * scale_x)),
+ min(h_img - 1, int(box[3] * scale_y)),
+ ]
+ cv2.rectangle(img_cpu, (x1, y1), (x2, y2), (0, 0, 255), 2)
+
+ # Save to disk
+ cv2.imwrite(str(save_path), img_cpu)
+ del img_cpu
+
+ def debug_save_img_roi(self, frame_source, bbs_full_res, frame_num, gt_boxes=None):
+ debug_dir = self.result_dir / "debug_stages" / self._testMethodName / "img_roi"
+ debug_dir.mkdir(parents=True, exist_ok=True)
+
+ # out_filename = str(debug_dir / f"analysis_{frame_num:04d}.jpg")
+ save_path = debug_dir / f"frame_{frame_num:04d}_analysis.jpg"
+
+ if frame_num > self.config.DEBUG_FRAME_LIMIT:
+ return
+
+ # Download/Copy the frame
+ if torch.is_tensor(frame_source):
+ # .contiguous() is CRITICAL here to fix the "shredded" look
+ temp = frame_source.squeeze(0) if frame_source.ndim == 4 else frame_source
+ img_cpu = temp.permute(1, 2, 0).contiguous().cpu().numpy()
+ elif hasattr(frame_source, "download"):
+ img_cpu = frame_source.download()
+ else:
+ img_cpu = np.ascontiguousarray(frame_source)
+
+ # Fix Shape: restore spatial grid if flattened
+ if img_cpu.ndim == 3 and img_cpu.shape[0] == 1:
+ img_cpu = img_cpu.reshape((self.resize_h, self.resize_w, 3))
+
+ # Fix Visibility: ONLY multiply if it's actually floating point
+ # If uint8 is multiplied by 255, it wraps around and creates "neon" colors
+ if img_cpu.dtype != np.uint8:
+ if img_cpu.max() <= 1.0:
+ img_cpu = (img_cpu * 255).clip(0, 255).astype(np.uint8)
+ else:
+ img_cpu = img_cpu.astype(np.uint8)
+
+ # Color Space: Standardize to BGR for imwrite
+ if len(img_cpu.shape) != 3:
+ img_cpu = cv2.cvtColor(img_cpu, cv2.COLOR_GRAY2BGR)
+
+ # Draw 8K Boxes (Scaled down)
+ h_img, w_img = img_cpu.shape[:2]
+ scale_x = w_img / self.frame_width
+ scale_y = h_img / self.frame_height
+
+ if gt_boxes is not None:
+ # Factor for converting 8K (original) bb dimensions to display dimensions
+ # disp_w, disp_h = inf_data["mask"].shape[:2] if hasattr(inf_data["mask"], "shape") else inf_data["mask"].size()
+ # scale_display_ox = disp_w / self.frame_width
+ # scale_display_oy = disp_h / self.frame_height
+
+ img_cpu = get_bb_overlay(
+ img_cpu,
+ gt_boxes,
+ (scale_x, scale_y),
+ (w_img, h_img),
+ color=(0, 255, 0), # green
+ )
+
+ if bbs_full_res is not None:
+ boxes = (
+ bbs_full_res.cpu().tolist()
+ if torch.is_tensor(bbs_full_res)
+ else bbs_full_res
+ )
+ # Annotate Box to image shape
+ for box in boxes:
+ x1, y1, x2, y2 = [
+ max(0, int(box[0] * scale_x)),
+ max(0, int(box[1] * scale_y)),
+ min(w_img - 1, int(box[2] * scale_x)),
+ min(h_img - 1, int(box[3] * scale_y)),
+ ]
+ img_cpu = cv2.rectangle(img_cpu, (x1, y1), (x2, y2), (0, 0, 255), 2)
+
+ cv2.imwrite(str(save_path), img_cpu)
+ del img_cpu
+
+ def debug_save_crops(self, cropped_batch, frame_num):
+ """Saves the first 5 crops of a batch to the results directory."""
+ debug_dir = self.result_dir / "debug_stages" / self._testMethodName / "crops"
+ # debug_dir = self.result_dir / "debug_stages" / self._testMethodName
+ debug_dir.mkdir(parents=True, exist_ok=True)
+
+ # Only save for the first self.config.DEBUG_FRAME_LIMIT frames to avoid disk bloat
+ if frame_num > self.config.DEBUG_FRAME_LIMIT:
+ return
+
+ for i, crop in enumerate(cropped_batch[: self.config.DEBUG_FRAME_LIMIT]):
+ # Convert GPU Tensor [C, H, W] -> NumPy [H, W, C]
+ if torch.is_tensor(crop):
+ # Reverse normalization (* 255) and permute to BGR
+ # img = (crop.squeeze(0).permute(1, 2, 0) * 255).byte().cpu().numpy()
+ temp = crop.squeeze(0) if crop.ndim == 4 else crop
+ # Flip the color channels from RGB to BGR before any extraction utilities run
+ # Flips only the first axis (channels) from [0, 1, 2] to [2, 1, 0]
+ temp = torch.flip(temp, dims=[0])
+ img = temp.permute(1, 2, 0).contiguous().cpu().numpy()
+ # img = cv2.cvtColor(img, cv2.COLOR_RGB2BGR)
+ else:
+ img = crop
+
+ if img.dtype != np.uint8:
+ if img.max() <= 1.0:
+ img = (img * 255).clip(0, 255).astype(np.uint8)
+ else:
+ img = img.astype(np.uint8)
+
+ cv2.imwrite(str(debug_dir / f"frame_{frame_num}_crop_{i}.jpg"), img)
+
+ del img
+
+ def debug_save_img(self, frame_source, frame_num):
+ debug_dir = self.result_dir / "debug_stages" / self._testMethodName / "img"
+ debug_dir.mkdir(parents=True, exist_ok=True)
+
+ if frame_num > self.config.DEBUG_FRAME_LIMIT:
+ return
+
+ # Download/Copy the frame
+ if torch.is_tensor(frame_source):
+ # .contiguous() is CRITICAL here to fix the "shredded" look
+ temp = frame_source.squeeze(0) if frame_source.ndim == 4 else frame_source
+ img_cpu = temp.permute(1, 2, 0).contiguous().cpu().numpy()
+ elif hasattr(frame_source, "download"):
+ img_cpu = frame_source.download()
+ else:
+ img_cpu = np.ascontiguousarray(frame_source)
+
+ # Fix Shape: restore spatial grid if flattened
+ if img_cpu.ndim == 3 and img_cpu.shape[0] == 1:
+ img_cpu = img_cpu.reshape((self.resize_h, self.resize_w, 3))
+
+ # Fix Visibility: ONLY multiply if it's actually floating point
+ # If uint8 is multiplied by 255, it wraps around and creates "neon" colors
+ if img_cpu.dtype != np.uint8:
+ if img_cpu.max() <= 1.0:
+ img_cpu = (img_cpu * 255).clip(0, 255).astype(np.uint8)
+ else:
+ img_cpu = img_cpu.astype(np.uint8)
+
+ # Color Space: Standardize to BGR for imwrite
+ if len(img_cpu.shape) == 3:
+ pass
+ else:
+ img_cpu = cv2.cvtColor(img_cpu, cv2.COLOR_GRAY2BGR)
+
+ cv2.imwrite(str(debug_dir / f"analysis_{frame_num:04d}.jpg"), img_cpu)
+ del img_cpu
+
+
+class GeneralObjectDetector(BaseObjectDetector):
+ def initialize(self):
+ super().initialize()
+
+ if self.device_input == "cuda":
+ self.gpu_id = 0
+ self.device_index = f"cuda:{self.gpu_id}"
+ self.inference_stream = torch.cuda.Stream()
+
+ self.gpu_float_staging = torch.empty(
+ (1, 3, self.frame_height, self.frame_width),
+ dtype=torch.float16,
+ device=self.device_index,
+ )
+
+ if self.timer_enabled:
+ self.det_start, self.det_end = (
+ torch.cuda.Event(enable_timing=True),
+ torch.cuda.Event(enable_timing=True),
+ )
+ else:
+ self.device_index = "cpu"
+
+ def run(
+ self, device_frame, overall_frame_num, frame_in_clip_count=0, gt_boxes=None
+ ):
+ if self.config.DETECTION_TYPE == "motion":
+ main_app_logger.info(
+ "[SKIP] Invalid type (motion) for GeneralObjectDetector!"
+ )
+ return
+
+ metrics = {
+ "sf_time": 0,
+ "roi_time": 0,
+ "det_time": 0,
+ "bbs": None,
+ "batch_density": 1,
+ }
+ inf_data = {}
+ bbs_full_res = None
+ motion_detected = True # Uses all frames for evaluation
+ inf_data["frameNum"] = overall_frame_num
+
+ # Format BBs for Detection
+ merged, det_frame = self.format_bbs_and_frame_4_detection(
+ bbs_full_res, device_frame
+ )
+
+ try:
+ if (
+ self.config.DEBUG_FLAG
+ and overall_frame_num <= self.config.DEBUG_FRAME_LIMIT
+ ):
+ self.debug_save_mask(
+ det_frame, overall_frame_num, rois=merged, gt_boxes=gt_boxes
+ )
+
+ if not self.config.DISABLE_DETECTION:
+ # --- 3. MODEL INFERENCE TIMING BLOCK ---
+ if self.device_input == "cuda" and self.timer_enabled:
+ self.det_start.record(self.inference_stream)
+ elif self.timer_enabled:
+ t_start = time.perf_counter()
+
+ metadata, _ = self.get_detections(
+ det_frame,
+ frame_in_clip_count,
+ device_input=self.config.device_input,
+ )
+
+ # num_objs = len(metadata.keys())
+
+ if self.device_input == "cuda" and self.timer_enabled:
+ self.det_end.record(self.inference_stream)
+ # self.inference_stream.synchronize()
+
+ # Lock-free check: wait for event completion without blocking the CPU GIL
+ while not self.det_end.query():
+ # Yield GIL to let file-writers and thread-pools work
+ time.sleep(0.001)
+ # self.det_end.synchronize()
+ # Leverages the hardware driver scheduler without thread polling overhead
+ # if hasattr(self, "inference_stream"):
+ # self.inference_stream.synchronize()
+
+ # Full-frame YOLO baseline tracks the elapsed time from t_start on page 20, line 1273
+ metrics["det_time"] = self.det_start.elapsed_time(self.det_end)
+
+ elif self.timer_enabled:
+ # CPU Path Execution: Must use standard wall-clock timing loops to avoid CUDA Event errors
+ metrics["det_time"] = (time.perf_counter() - t_start) * 1000.0
+
+ except Exception:
+ traceback.print_exc()
+ finally:
+ del merged, bbs_full_res
+ return metrics, metadata, det_frame, motion_detected
+
+ @torch.inference_mode()
+ def get_detections(self, frame, frame_id, device_input="cuda"):
+ metadata = {}
+ num_objs = 0
+
+ with torch.inference_mode():
+ if isinstance(frame, torch.Tensor):
+ if frame.ndim == 3 and frame.shape[-1] == 3:
+ with self.compiled_no_grad_gate:
+ # 1. Transform from [4320, 7680, 3] to [3, 4320, 7680]
+ # 2. Unsqueeze(0) converts it to the required [1, 3, 4320, 7680] shape format
+ frame_inference = (
+ frame.permute(2, 0, 1).contiguous().unsqueeze(0)
+ )
+ else:
+ frame_inference = frame.contiguous()
+ else:
+ frame_inference = frame
+
+ H, W = frame_inference.shape[-2:]
+ target_imgsz = (int(H), int(W))
+ scale_display_x = self.resize_w / W # 640 / 8192
+ scale_display_y = self.resize_h / H # 640 / 4608
+ results = self.run_model(
+ frame_inference,
+ imgsz=target_imgsz,
+ batch=1,
+ device_input=device_input,
+ stream=STREAM_ARG,
+ )
+
+ # Extract full resolution detections
+ if results and len(results) > 0:
+ boxes = results[0].boxes
+ if boxes is not None and len(boxes) > 0:
+ # OPTIMIZATION: Download coordinates as a unified batch array block to drop PCIe latency
+ batch_coords = boxes.xywh.cpu().numpy()
+ batch_cls = boxes.cls.cpu().numpy().astype(np.int32)
+ batch_conf = boxes.conf.cpu().numpy()
+ # for idx, box in enumerate(boxes):
+ # # coords = box.xywh[0].cpu().tolist() # [x_center, y_center, width, height]
+ # # cls_id = int(box.cls[0].cpu().item())
+ # # conf = float(box.conf[0].cpu().item())
+ # coords = (
+ # box.xywh.cpu().squeeze().tolist()
+ # ) # Converts [x_center, y_center, w, h] safely
+ # cls_id = int(box.cls.cpu().item())
+ # class_name = self.label_source[cls_id]
+ # confidence = float(box.conf.cpu().item())
+ for idx in range(len(batch_coords)):
+ coords = batch_coords[idx]
+ cls_id = batch_cls[idx]
+ class_name = self.label_source[cls_id]
+ confidence = float(batch_conf[idx])
+
+ # Guard against un-squeezed structural lists
+ if isinstance(coords[0], list):
+ coords = coords[0]
+
+ # Convert center bounds coordinates back to upper-left origin layout standard
+ # and scale to 640x640
+ disp_x = (coords[0] - (coords[2] / 2.0)) * scale_display_x
+ disp_y = (coords[1] - (coords[3] / 2.0)) * scale_display_y
+ disp_w = coords[2] * scale_display_x
+ disp_h = coords[3] * scale_display_y
+
+ # ----------------------------------
+ # TODO: Need function, same for SF and non-SF
+ if disp_w > 2 and disp_h > 2:
+ # Resized
+ object_res = [
+ int(disp_x), # int(abs_x1 * scale_x),
+ int(disp_y), # int(abs_y1 * scale_y),
+ int(disp_h), # int(height * scale_y),
+ int(disp_w), # int(width * scale_x),
+ class_name,
+ confidence,
+ int(self.resize_h),
+ int(self.resize_w),
+ ]
+
+ obj_id = len(metadata)
+ num_objs += 1
+ framenum_str = f"{frame_id:04d}_{obj_id:04d}"
+ metadata[framenum_str] = {
+ "frameId": int(frame_id),
+ "bbId": framenum_str,
+ "bbox": {
+ "x": int(object_res[0]),
+ "y": int(object_res[1]),
+ "height": int(object_res[2]),
+ "width": int(object_res[3]),
+ "object": str(object_res[4]),
+ "object_det": {
+ "confidence": float(object_res[5]),
+ "frameH": int(object_res[6]),
+ "frameW": int(object_res[7]),
+ },
+ },
+ }
+ # ----------------------------------
+
+ if "boxes" in locals():
+ del boxes
+ if "results" in locals():
+ del results
+
+ del frame, frame_inference
+ if "cuda" in device_input:
+ torch.cuda.empty_cache()
+ return metadata, num_objs
+
+
+# Create a dummy object representing the interface
+class GPUHolder:
+ def __init__(self, interface):
+ self.__cuda_array_interface__ = interface
+
+
+class SmartFilteringObjectDetector(BaseObjectDetector):
+ def initialize(self):
+ super().initialize()
+
+ self.scale_x = self.frame_width / self.resize_w
+ self.scale_y = self.frame_height / self.resize_h
+
+ self.min_roi_w = int(self.config.ROI_MIN_AREA_RATIO * self.resize_w)
+ self.min_roi_h = int(self.config.ROI_MIN_AREA_RATIO * self.resize_h)
+ self.max_roi_w = int(self.resize_w * self.config.ROI_MAX_RELATIVE_SIZE_RATIO)
+ self.max_roi_h = int(self.resize_h * self.config.ROI_MAX_RELATIVE_SIZE_RATIO)
+ self.max_cached_elements = 100
+ # Pre-allocate a single static tensor to hold the non-zero indices.
+ # For a 640x640 mask, there are at most 409,600 pixels.
+ # This pre-allocates a static ~3.2MB buffer in VRAM once.
+ self.coords_scratchpad = torch.empty(
+ (self.config.MODEL_H * self.config.MODEL_W, 2),
+ dtype=torch.long,
+ device="cuda",
+ )
+
+ # Pre-allocate a static box container for up to 100 candidate ROIs
+ # Sized [100, 4] for [x1, y1, x2, y2]
+ self.static_boxes_out = torch.empty(
+ (self.max_cached_elements, 4), dtype=torch.float32, device="cuda"
+ )
+
+ # Default Kernels
+ self.dilate_kernel = cv2.getStructuringElement(
+ cv2.MORPH_ELLIPSE,
+ # cv2.MORPH_RECT,
+ (self.config.DILATE_KERNEL_SIZE, self.config.DILATE_KERNEL_SIZE),
+ )
+ # self.dilate_kernel_for_enhanced_mask = np.ones((15,15), np.uint8) # 5, 5) (21, 21)
+ self.dilate_kernel_for_enhanced_mask = cv2.getStructuringElement(
+ cv2.MORPH_ELLIPSE,
+ # cv2.MORPH_RECT,
+ (
+ self.config.BKGD_SUB_INCLUDE_HISTORY_DILATE_KERNEL_SIZE,
+ self.config.BKGD_SUB_INCLUDE_HISTORY_DILATE_KERNEL_SIZE,
+ ),
+ )
+
+ # Determine minimum contour size relative to frame resolution
+ self.min_contour_area = int((self.min_roi_h) * (self.min_roi_h)) # 207
+
+ self.dist_thresh_8k = max(
+ self.config.ROI_DISTANCE_THRESH_RATIO * self.frame_width,
+ self.config.ROI_DISTANCE_THRESH_RATIO * self.frame_height,
+ )
+ multiplier = 2.0 if self.device_input == "cpu" else 1.0
+ self.dist_thresh_640 = (
+ max(
+ self.config.ROI_DISTANCE_THRESH_RATIO * self.resize_w,
+ self.config.ROI_DISTANCE_THRESH_RATIO * self.resize_h,
+ )
+ * multiplier
+ ) # 0.05 * self.resize_w
+
+ if not hasattr(self, "static_canvas_scratch"):
+ self.static_canvas_scratch = torch.full(
+ (640, 640, 3),
+ fill_value=114,
+ dtype=torch.uint8,
+ device=self.device_input,
+ )
+ if self.device_input == "cuda":
+ self.init_gpu_pipeline()
+ else:
+ self.init_cpu_pipeline()
+
+ self.scales_tensor = torch.tensor(
+ [self.scale_x, self.scale_y, self.scale_x, self.scale_y],
+ device=self.device_index,
+ )
+
+ self.fixed_inference_batch = torch.empty(
+ (
+ self.config.MODEL_MAX_BATCH_SIZE,
+ 3,
+ self.config.MODEL_H,
+ self.config.MODEL_W,
+ ),
+ dtype=torch.half,
+ device=self.device_input,
+ )
+
+ def run(
+ self, device_frame, overall_frame_num, frame_in_clip_count=0, gt_boxes=None
+ ):
+ if self.device_input == "cuda":
+ return self.run_gpu_pipeline(
+ device_frame, overall_frame_num, frame_in_clip_count
+ )
+ else:
+ return self.run_cpu_pipeline(
+ device_frame, overall_frame_num, frame_in_clip_count
+ )
+
+ # v1
+ # def get_detections_v1(
+ # self, frame, frame_id, device_input="cuda", merged=None, thickness=2
+ # ):
+ # metadata = {}
+ # if merged is None or len(merged) == 0:
+ # return metadata, 0
+
+ # is_cuda = device_input == "cuda"
+ # num_objs = 0
+
+ # # FIX 1: Pre-cache class property values onto local function registers
+ # # Bypasses repeated slow overhead parsing blocks inside the hot box-loop
+ # max_batch_size = getattr(self.config, "MODEL_MAX_BATCH_SIZE", 64)
+ # target_crop_size = self.config.MODEL_W
+ # resize_w_factor = self.resize_w / self.frame_width
+ # resize_h_factor = self.resize_h / self.frame_height
+ # resize_h_int = int(self.resize_h)
+ # resize_w_int = int(self.resize_w)
+
+ # if is_cuda and isinstance(frame, torch.Tensor):
+ # src_tensor = frame.squeeze(0) if frame.ndim == 4 else frame
+ # if src_tensor.shape[-1] == 3:
+ # src_tensor = src_tensor.permute(2, 0, 1)
+ # src_h, src_w = src_tensor.shape[-2:]
+ # else:
+ # src_tensor = np.asarray(frame)
+ # if src_tensor.ndim == 4:
+ # src_tensor = src_tensor[0]
+ # src_h, src_w = src_tensor.shape[:2]
+
+ # results_pool = []
+ # patch_coordinates = []
+ # patch_idx = 0
+
+ # # -----------------------------------------------------------------
+ # # STEP 1: VECTORIZED SWEEP & DIRECT-WRITE MATRIX POOL
+ # # -----------------------------------------------------------------
+ # roi_patches = []
+ # for box in merged:
+ # if patch_idx >= max_batch_size:
+ # break
+
+ # # FIX 2: Bypassing hasattr() check loop hooks entirely
+ # # Directly extract standard bounding elements as fast local scalars
+ # box_data = box.tolist() if hasattr(box, "tolist") else box
+ # x1_raw, y1_raw, x2_raw, y2_raw = (
+ # box_data[0],
+ # box_data[1],
+ # box_data[2],
+ # box_data[3],
+ # )
+
+ # w_raw = x2_raw - x1_raw
+ # h_raw = y2_raw - y1_raw
+ # cx = (x1_raw + x2_raw) * 0.5
+ # cy = (y1_raw + y2_raw) * 0.5
+
+ # w_cushioned = w_raw + (target_crop_size * 0.1)
+ # h_cushioned = h_raw + (target_crop_size * 0.1)
+
+ # crop_w = max(w_cushioned, target_crop_size)
+ # crop_h = max(h_cushioned, target_crop_size)
+
+ # x1 = cx - (crop_w / 2.0)
+ # y1 = cy - (crop_h / 2.0)
+ # x2 = cx + (crop_w / 2.0)
+ # y2 = cy + (crop_h / 2.0)
+
+ # shift_left = max(0.0, 0.0 - x1)
+ # shift_right = max(0.0, x2 - src_w)
+ # x1 += shift_left - shift_right
+ # x2 += shift_left - shift_right
+
+ # shift_top = max(0.0, 0.0 - y1)
+ # shift_bottom = max(0.0, y2 - src_h)
+ # y1 += shift_top - shift_bottom
+ # y2 += shift_top - shift_bottom
+
+ # x1, y1 = max(0, int(x1)), max(0, int(y1))
+ # x2, y2 = min(src_w, int(x2)), min(src_h, int(y2))
+
+ # box_w, box_h = x2 - x1, y2 - y1
+ # if box_w < 8 or box_h < 8:
+ # continue
+
+ # scale = min(target_crop_size / box_w, target_crop_size / box_h)
+ # nw, nh = int(max(1, box_w * scale)), int(max(1, box_h * scale))
+ # dx = (target_crop_size - nw) // 2
+ # dy = (target_crop_size - nh) // 2
+
+ # if is_cuda and isinstance(src_tensor, torch.Tensor):
+ # # with torch.no_grad():
+ # crop = src_tensor[:, y1:y2, x1:x2]
+ # crop_resized = F.interpolate(
+ # crop.unsqueeze(0),
+ # size=(nh, nw),
+ # mode="nearest",
+ # ).squeeze(0)
+
+ # self.fixed_inference_batch[patch_idx].fill_(114.0)
+ # self.fixed_inference_batch[patch_idx][
+ # :, dy : dy + nh, dx : dx + nw
+ # ].copy_(crop_resized, non_blocking=True)
+
+ # else:
+ # crop = src_tensor[y1:y2, x1:x2]
+ # crop_resized = cv2.resize(
+ # crop, (nw, nh), interpolation=cv2.INTER_NEAREST
+ # )
+ # padded_canvas = np.empty(
+ # (target_crop_size, target_crop_size, 3), dtype=np.uint8
+ # )
+ # padded_canvas.fill(114)
+ # padded_canvas[dy : dy + nh, dx : dx + nw] = crop_resized
+
+ # self.fixed_inference_batch[patch_idx].copy_(
+ # torch.from_numpy(padded_canvas).permute(2, 0, 1), non_blocking=True
+ # )
+
+ # if self.debug_frame_limit > -1:
+ # roi_patches.append(self.fixed_inference_batch[patch_idx])
+ # patch_coordinates.append((x1, y1, box_w, box_h, scale, dx, dy))
+ # patch_idx += 1
+
+ # if patch_idx == 0:
+ # return {}, 0
+
+ # if (
+ # self.debug_frame_limit > -1
+ # ): # and self.config.DEBUG_FLAG and hasattr(self, "debug_save_crops") and len(roi_patches) > 0:
+ # self.debug_save_crops(roi_patches, frame_id)
+ # roi_patches = []
+
+ # # -----------------------------------------------------------------
+ # # STEP 2: INSTANT NON-BLOCKING SUB-SLICE MODEL INGESTION
+ # # -----------------------------------------------------------------
+ # # with torch.inference_mode():
+ # inference_batch = self.fixed_inference_batch[:patch_idx].clone()
+ # batch_res = self.run_model(
+ # inference_batch,
+ # imgsz=(target_crop_size, target_crop_size),
+ # batch=patch_idx,
+ # device_input=device_input,
+ # stream=STREAM_ARG,
+ # )
+ # results_pool.extend(batch_res)
+
+ # if is_cuda and hasattr(self, "inference_stream"):
+ # self.inference_stream.synchronize()
+ # # -----------------------------------------------------------------
+ # # OPTIMIZED STEP 3: TRUE BATCHED DEVICE HANDOFF (0ms LOCKS)
+ # # -----------------------------------------------------------------
+ # main_xyxy_list = []
+ # main_cls_list = []
+ # main_conf_list = []
+ # patch_mapping_indices = []
+
+ # # 1. Asynchronously collect the raw VRAM references from the model pool
+ # for idx, res in enumerate(results_pool):
+ # if res.boxes is not None and len(res.boxes) > 0:
+ # # Gather underlying torch.Tensor GPU pointer segments straight out of memory
+ # main_xyxy_list.append(res.boxes.data[:, 0:4])
+ # main_cls_list.append(res.boxes.data[:, 5])
+ # main_conf_list.append(res.boxes.data[:, 4])
+ # # Track how many objects belong to this patch index to map them back later
+ # patch_mapping_indices.append((idx, len(res.boxes)))
+
+ # if len(main_xyxy_list) == 0:
+ # return {}, 0
+
+ # # if len(main_xyxy_list) > 0:
+ # # with torch.inference_mode():
+ # # 2. Vectorized GPU Concatenation
+ # # Stacks independent slices into uniform, single-pass matrices entirely on the GPU
+ # all_main_xyxy = torch.cat(main_xyxy_list, dim=0)
+ # all_main_clss = torch.cat(main_cls_list, dim=0)
+ # all_main_confs = torch.cat(main_conf_list, dim=0)
+
+ # num_detected = int(all_main_xyxy.shape[0])
+ # num_clss = int(all_main_clss.shape[0])
+ # num_confs = int(all_main_confs.shape[0])
+
+ # if is_cuda:
+ # # Perform fast, non-blocking asynchronous streaming transfers across the PCIe bus
+ # self.pinned_cpu_xyxy[:num_detected].copy_(all_main_xyxy, non_blocking=True)
+ # cuda_int_clss = all_main_clss.to(torch.int32)
+ # self.pinned_cpu_clss[:num_clss].copy_(cuda_int_clss, non_blocking=True)
+ # self.pinned_cpu_confs[:num_confs].copy_(all_main_confs, non_blocking=True)
+
+ # # Synchronize ONLY the specific inference stream handle right before reading on host
+ # # self.inference_stream.synchronize()
+ # if not hasattr(self, "_dma_fence_event"):
+ # self._dma_fence_event = torch.cuda.Event()
+ # self._dma_fence_event.record(self.inference_stream)
+ # self._dma_fence_event.synchronize()
+
+ # all_cpu_xyxy = self.pinned_cpu_xyxy[:num_detected].numpy()
+ # all_cpu_clss = self.pinned_cpu_clss[:num_clss].numpy().flatten()
+ # all_cpu_confs = self.pinned_cpu_confs[:num_confs].numpy().flatten()
+ # else:
+ # all_cpu_xyxy = all_main_xyxy.cpu().numpy()
+ # all_cpu_clss = all_main_clss.cpu().numpy().astype(np.int32).flatten()
+ # all_cpu_confs = all_main_confs.cpu().numpy().flatten()
+
+ # # Global tracking increment initialization
+ # global_box_ptr = 0
+ # total_available_boxes = len(all_cpu_clss)
+
+ # # 4. Map detections back to global metadata layout space using our scalar index pointers
+ # for idx, num_boxes in patch_mapping_indices:
+ # ox1, oy1, box_w, box_h, scale, dx, dy = patch_coordinates[idx]
+
+ # for _ in range(num_boxes):
+ # if global_box_ptr >= total_available_boxes:
+ # break
+
+ # lx1, ly1, lx2, ly2 = all_cpu_xyxy[global_box_ptr]
+ # class_id = int(all_cpu_clss[global_box_ptr])
+ # # if class_id < len(self.label_source):
+ # try:
+ # if class_id >= 0 and class_id < len(self.label_source):
+ # class_name = self.label_source[class_id]
+ # else:
+ # class_name = "unknown"
+ # except Exception:
+ # traceback.print_exc()
+
+ # confidence = float(all_cpu_confs[global_box_ptr])
+
+ # lx1_unpadded = lx1 - dx
+ # ly1_unpadded = ly1 - dy
+ # lx2_unpadded = lx2 - dx
+ # ly2_unpadded = ly2 - dy
+
+ # global_x1 = ox1 + (lx1_unpadded / scale)
+ # global_y1 = oy1 + (ly1_unpadded / scale)
+ # global_x2 = ox1 + (lx2_unpadded / scale)
+ # global_y2 = oy1 + (ly2_unpadded / scale)
+
+ # disp_x = int(global_x1 * resize_w_factor)
+ # disp_y = int(global_y1 * resize_h_factor)
+ # disp_x2 = int(global_x2 * resize_w_factor)
+ # disp_y2 = int(global_y2 * resize_h_factor)
+
+ # disp_w = disp_x2 - disp_x
+ # disp_h = disp_y2 - disp_y
+
+ # if disp_w > 2 and disp_h > 2 and class_name != "unknown":
+ # num_objs += 1
+ # obj_id = num_objs - 1
+ # framenum_str = f"{frame_id:04d}_{obj_id:04d}"
+ # metadata[framenum_str] = {
+ # "frameId": int(frame_id),
+ # "bbId": framenum_str,
+ # "bbox": {
+ # "x": disp_x,
+ # "y": disp_y,
+ # "height": disp_h,
+ # "width": disp_w,
+ # "object": class_name,
+ # "object_det": {
+ # "confidence": confidence,
+ # "frameH": resize_h_int,
+ # "frameW": resize_w_int,
+ # },
+ # },
+ # }
+
+ # global_box_ptr += 1
+
+ # # Clean local tensor references to protect unmanaged memory scopes
+ # crop = None
+ # del (
+ # crop,
+ # patch_coordinates,
+ # results_pool,
+ # main_xyxy_list,
+ # main_cls_list,
+ # main_conf_list,
+ # )
+ # return metadata, num_objs
+
+ # v2
+ @torch.inference_mode()
+ def get_detections_v2(
+ self, frame, frame_id, device_input="cuda", merged=None, thickness=2
+ ):
+ metadata = {}
+ if merged is None or len(merged) == 0:
+ return metadata, 0
+
+ num_boxes = merged.shape[0]
+ if num_boxes == 0:
+ return {}, 0
+
+ target_h, target_w = (self.config.MODEL_H, self.config.MODEL_W)
+
+ # =========================================================================
+ # STEP 1: VECTORIZED MATH FOR CUSHION & LETTERBOX PADDING
+ # =========================================================================
+
+ # 1a. Apply 10% Cushion to original boxes
+ cushion = target_w * 0.1
+
+ x1_t = (merged[:, 0] - cushion).clamp(min=0)
+ y1_t = (merged[:, 1] - cushion).clamp(min=0)
+ x2_t = (merged[:, 2] + cushion).clamp(max=self.frame_width)
+ y2_t = (merged[:, 3] + cushion).clamp(max=self.frame_height)
+
+ # 1b. Calculate scaled dimensions to fit within target size (preserving aspect ratio)
+ box_w = (x2_t - x1_t).clamp(min=1)
+ box_h = (y2_t - y1_t).clamp(min=1)
+
+ scale = torch.min(target_w / box_w, target_h / box_h)
+
+ new_w = (box_w * scale).int()
+ new_h = (box_h * scale).int()
+
+ pad_x = (target_w - new_w) // 2
+ pad_y = (target_h - new_h) // 2
+
+ # =========================================================================
+ # STEP 2: HIGH-SPEED CROPPING & RESIZING LOOP
+ # =========================================================================
+
+ # Ensure frame is CHW and float for F.interpolate
+ if frame.dim() == 3:
+ if frame.shape[2] == 3:
+ frame_chw = frame.permute(2, 0, 1).float()
+ else:
+ frame_chw = frame.float()
+ elif frame.dim() == 4:
+ frame_chw = frame.squeeze(0).float()
+ if frame_chw.shape[2] == 3:
+ frame_chw = frame_chw.permute(2, 0, 1)
+ else:
+ frame_chw = frame.float()
+
+ # Determine correct gray padding value (114 for 0-255 range, 114/255 for 0-1 range)
+ gray_val = (
+ 114.0 / 255.0
+ if getattr(self, "normalize", False) or frame_chw.max() <= 1.0
+ else 114.0
+ )
+
+ padded_batch = torch.full(
+ (num_boxes, 3, target_h, target_w),
+ gray_val,
+ device=frame.device,
+ dtype=torch.float32,
+ )
+
+ # CRITICAL PERFORMANCE FIX: Extract tensors to fast Python lists
+ # This prevents the loop from halting to sync CPU/GPU on every single slice iteration
+ x1_c = x1_t.int().tolist()
+ y1_c = y1_t.int().tolist()
+ x2_c = x2_t.int().tolist()
+ y2_c = y2_t.int().tolist()
+ nw_c = new_w.tolist()
+ nh_c = new_h.tolist()
+ px_c = pad_x.tolist()
+ py_c = pad_y.tolist()
+
+ for i in range(num_boxes):
+ # 1. Slice cushioned box
+ crop = frame_chw[:, y1_c[i] : y2_c[i], x1_c[i] : x2_c[i]].unsqueeze(0)
+
+ # 2. Resize maintaining aspect ratio exactly once
+ crop_resized = F.interpolate(
+ crop, size=(nh_c[i], nw_c[i]), mode="bilinear", align_corners=False
+ )
+
+ # 3. Paste into the center of the padded gray canvas
+ padded_batch[
+ i, :, py_c[i] : py_c[i] + nh_c[i], px_c[i] : px_c[i] + nw_c[i]
+ ] = crop_resized.squeeze(0)
+
+ self.fixed_inference_batch[:num_boxes] = padded_batch.to(
+ self.fixed_inference_batch.dtype
+ )
+
+ # =========================================================================
+ # STEP 3: MODEL INFERENCE
+ # =========================================================================
+ batch_res = self.run_model(
+ self.fixed_inference_batch[:num_boxes],
+ imgsz=(target_h, target_w),
+ batch=num_boxes,
+ device_input=device_input,
+ stream=STREAM_ARG,
+ )
+
+ # =========================================================================
+ # STEP 4: BATCHED COORDINATE TRANSLATION
+ # =========================================================================
+ all_detections = []
+ detection_to_crop_map = []
+
+ for i, res in enumerate(batch_res):
+ if res.boxes is not None and len(res.boxes) > 0:
+ all_detections.append(res.boxes.data)
+ detection_to_crop_map.extend([i] * len(res.boxes))
+
+ if not all_detections:
+ return {}, 0
+
+ all_detections_tensor = torch.cat(all_detections, dim=0)
+ detection_to_crop_map = torch.tensor(
+ detection_to_crop_map, device=merged.device, dtype=torch.long
+ )
+
+ # Coordinates are local to the 640x640 padded input
+ local_boxes = all_detections_tensor[:, :4]
+
+ # Get the mapping variables for each specific detection
+ scale_map = scale[detection_to_crop_map]
+ pad_x_map = pad_x[detection_to_crop_map]
+ pad_y_map = pad_y[detection_to_crop_map]
+ orig_x1_map = x1_t[detection_to_crop_map]
+ orig_y1_map = y1_t[detection_to_crop_map]
+
+ # 4a. Remove padding offset and rescale to original cushioned crop dimensions
+ unpadded_x1 = (local_boxes[:, 0] - pad_x_map) / scale_map
+ unpadded_y1 = (local_boxes[:, 1] - pad_y_map) / scale_map
+ unpadded_x2 = (local_boxes[:, 2] - pad_x_map) / scale_map
+ unpadded_y2 = (local_boxes[:, 3] - pad_y_map) / scale_map
+
+ # 4b. Add original cushioned crop's top-left corner to get global 8K coordinates
+ global_x1 = unpadded_x1 + orig_x1_map
+ global_y1 = unpadded_y1 + orig_y1_map
+ global_x2 = unpadded_x2 + orig_x1_map
+ global_y2 = unpadded_y2 + orig_y1_map
+
+ # 4c. Scale from 8K to 640x640 metadata space
+ meta_scale_x = self.resize_w / self.frame_width
+ meta_scale_y = self.resize_h / self.frame_height
+
+ meta_x1 = global_x1 * meta_scale_x
+ meta_y1 = global_y1 * meta_scale_y
+ meta_x2 = global_x2 * meta_scale_x
+ meta_y2 = global_y2 * meta_scale_y
+
+ # =========================================================================
+ # STEP 5: METADATA FORMATTING
+ # =========================================================================
+ all_confs = all_detections_tensor[:, 4].cpu().numpy()
+ all_cls_ids = all_detections_tensor[:, 5].cpu().numpy().astype(int)
+ meta_boxes_cpu = (
+ torch.stack([meta_x1, meta_y1, meta_x2, meta_y2], dim=1).cpu().numpy()
+ )
+
+ num_objs = 0
+ for i in range(len(all_detections_tensor)):
+ class_id = all_cls_ids[i]
+ if class_id < 0 or class_id >= len(self.label_source):
+ continue
+
+ disp_x, disp_y, disp_x2, disp_y2 = meta_boxes_cpu[i]
+ disp_w = disp_x2 - disp_x
+ disp_h = disp_y2 - disp_y
+
+ if disp_w > 2 and disp_h > 2:
+ num_objs += 1
+ obj_id = num_objs - 1
+ framenum_str = f"{int(frame_id):04d}_{obj_id:04d}"
+ metadata[framenum_str] = {
+ "frameId": int(frame_id),
+ "bbId": framenum_str,
+ "bbox": {
+ "x": int(disp_x),
+ "y": int(disp_y),
+ "height": int(disp_h),
+ "width": int(disp_w),
+ "object": self.label_source[class_id],
+ "object_det": {
+ "confidence": float(all_confs[i]),
+ "frameH": int(self.resize_h),
+ "frameW": int(self.resize_w),
+ },
+ },
+ }
+
+ del all_detections_tensor
+ return metadata, num_objs
+
+ # v3 - reduce memory footprint
+ @torch.inference_mode()
+ def get_detections(
+ self, frame, frame_id, device_input="cuda", merged=None, thickness=2
+ ):
+ metadata = {}
+ if merged is None or len(merged) == 0:
+ return metadata, 0
+
+ num_boxes = merged.shape[0]
+ if num_boxes == 0:
+ return {}, 0
+
+ target_h, target_w = (self.config.MODEL_H, self.config.MODEL_W)
+
+ # =========================================================================
+ # STEP 1: VECTORIZED MATH FOR CUSHION & LETTERBOX PADDING
+ # =========================================================================
+
+ # 1a. Apply 10% Cushion to original boxes
+ cushion = target_w * 0.1
+
+ x1_t = (merged[:, 0] - cushion).clamp(min=0).detach().cpu()
+ y1_t = (merged[:, 1] - cushion).clamp(min=0).detach().cpu()
+ x2_t = (merged[:, 2] + cushion).clamp(max=self.frame_width).detach().cpu()
+ y2_t = (merged[:, 3] + cushion).clamp(max=self.frame_height).detach().cpu()
+
+ # 1b. Calculate scaled dimensions to fit within target size (preserving aspect ratio)
+ box_w = (x2_t - x1_t).clamp(min=1).detach().cpu()
+ box_h = (y2_t - y1_t).clamp(min=1).detach().cpu()
+
+ scale = torch.min(target_w / box_w, target_h / box_h).detach().cpu()
+
+ new_w = (box_w * scale).int()
+ new_h = (box_h * scale).int()
+
+ pad_x = (target_w - new_w) // 2
+ pad_y = (target_h - new_h) // 2
+
+ # =========================================================================
+ # STEP 2: HIGH-SPEED CROPPING & RESIZING LOOP
+ # =========================================================================
+
+ # Ensure frame is CHW and float for F.interpolate
+ if frame.dim() == 3:
+ if frame.shape[2] == 3:
+ # frame_chw = frame.permute(2, 0, 1).float()
+ # Permute is just a "view" metadata change (0ms, 0 VRAM allocation)
+ permuted_view = frame.permute(2, 0, 1)
+
+ # Copy and cast directly into our pre-allocated static float buffer in-place!
+ # This allocates 0 MB of new memory!
+ self.gpu_float_staging.copy_(permuted_view, non_blocking=True)
+
+ # Now divide by 255.0 in-place to normalize (reuses the exact same float32 memory)
+ frame_chw = self.gpu_float_staging[0] # .div_(255.0)
+ else:
+ frame_chw = frame.float()
+ elif frame.dim() == 4:
+ frame_chw = frame.squeeze(0).float()
+ if frame_chw.shape[2] == 3:
+ frame_chw = frame_chw.permute(2, 0, 1)
+ else:
+ frame_chw = frame.float()
+
+ # Determine correct gray padding value (114 for 0-255 range, 114/255 for 0-1 range)
+ gray_val = (
+ 114.0 / 255.0
+ if getattr(self, "normalize", False) or frame_chw.max() <= 1.0
+ else 114.0
+ )
+
+ # padded_batch = torch.full(
+ # (num_boxes, 3, target_h, target_w),
+ # gray_val,
+ # device=frame.device,
+ # dtype=torch.float32,
+ # )
+
+ # CRITICAL PERFORMANCE FIX: Extract tensors to fast Python lists
+ # This prevents the loop from halting to sync CPU/GPU on every single slice iteration
+ x1_c = x1_t.int().tolist()
+ y1_c = y1_t.int().tolist()
+ x2_c = x2_t.int().tolist()
+ y2_c = y2_t.int().tolist()
+ nw_c = new_w.tolist()
+ nh_c = new_h.tolist()
+ px_c = pad_x.tolist()
+ py_c = pad_y.tolist()
+
+ for i in range(num_boxes):
+ # 1. Slice cushioned box
+ crop = frame_chw[:, y1_c[i] : y2_c[i], x1_c[i] : x2_c[i]].unsqueeze(0)
+
+ # 2. Resize maintaining aspect ratio exactly once
+ crop_resized = F.interpolate(
+ crop, size=(nh_c[i], nw_c[i]), mode="bilinear", align_corners=False
+ )
+
+ # 3. Paste into the center of the padded gray canvas
+ # padded_batch[
+ # i, :, py_c[i] : py_c[i] + nh_c[i], px_c[i] : px_c[i] + nw_c[i]
+ # ] = crop_resized.squeeze(0)
+ self.fixed_inference_batch[
+ i, :, py_c[i] : py_c[i] + nh_c[i], px_c[i] : px_c[i] + nw_c[i]
+ ] = crop_resized.squeeze(0)
+
+ # self.fixed_inference_batch[:num_boxes] = padded_batch.to(
+ # self.fixed_inference_batch.dtype
+ # )
+
+ # =========================================================================
+ # STEP 3: MODEL INFERENCE
+ # =========================================================================
+
+ # =========================================================================
+ # STEP 4: BATCHED COORDINATE TRANSLATION
+ # =========================================================================
+ # all_detections = []
+ # detection_to_crop_map = []
+ num_objs = 0
+ meta_scale_x = self.resize_w / self.frame_width
+ meta_scale_y = self.resize_h / self.frame_height
+
+ with torch.no_grad():
+ batch_res = self.run_model(
+ self.fixed_inference_batch[:num_boxes],
+ imgsz=(target_h, target_w),
+ batch=num_boxes,
+ device_input=device_input,
+ stream=STREAM_ARG,
+ )
+
+ for i, res in enumerate(batch_res):
+ if res.boxes is None or len(res.boxes) == 0:
+ del res
+ continue
+
+ # Move data to CPU immediately to free VRAM
+ local_boxes = res.boxes.data.detach().cpu()
+
+ # Get the mapping variables for each specific detection
+ scale_map = scale[i]
+ pad_x_map = pad_x[i]
+ pad_y_map = pad_y[i]
+ orig_x1_map = x1_t[i]
+ orig_y1_map = y1_t[i]
+
+ # 4a. Remove padding offset and rescale to original cushioned crop dimensions
+ unpadded_x1 = (local_boxes[:, 0] - pad_x_map) / scale_map
+ unpadded_y1 = (local_boxes[:, 1] - pad_y_map) / scale_map
+ unpadded_x2 = (local_boxes[:, 2] - pad_x_map) / scale_map
+ unpadded_y2 = (local_boxes[:, 3] - pad_y_map) / scale_map
+
+ # 4b. Add original cushioned crop's top-left corner to get global 8K coordinates
+ global_x1 = unpadded_x1 + orig_x1_map
+ global_y1 = unpadded_y1 + orig_y1_map
+ # global_x2 = unpadded_x2 + orig_x1_map
+ # global_y2 = unpadded_y2 + orig_y1_map
+
+ # 4c. Scale from 8K to 640x640 metadata space
+ meta_x1 = global_x1 * meta_scale_x
+ meta_y1 = global_y1 * meta_scale_y
+ # meta_x2 = global_x2 * meta_scale_x
+ # meta_y2 = global_y2 * meta_scale_y
+ meta_w = (unpadded_x2 - unpadded_x1) * meta_scale_x
+ meta_h = (unpadded_y2 - unpadded_y1) * meta_scale_y
+
+ # =========================================================================
+ # STEP 5: METADATA FORMATTING
+ # =========================================================================
+ # Format metadata
+ for j in range(len(local_boxes)):
+ if meta_w[j] > 2 and meta_h[j] > 2:
+ class_id = int(local_boxes[j, 5])
+ if 0 <= class_id < len(self.label_source):
+ framenum_str = f"{int(frame_id):04d}_{num_objs:04d}"
+ metadata[framenum_str] = {
+ "frameId": int(frame_id),
+ "bbId": framenum_str,
+ "bbox": {
+ "x": int(meta_x1[j]),
+ "y": int(meta_y1[j]),
+ "height": int(meta_h[j]),
+ "width": int(meta_w[j]),
+ "object": self.label_source[class_id],
+ "object_det": {
+ "confidence": float(local_boxes[j, 4]),
+ "frameH": int(self.resize_h),
+ "frameW": int(self.resize_w),
+ },
+ },
+ }
+ num_objs += 1
+
+ # Explicitly delete tensors to free memory inside the loop
+ del res, local_boxes
+
+ del batch_res
+ self.fixed_inference_batch.fill_(gray_val)
+
+ if "cuda" in device_input:
+ torch.cuda.empty_cache()
+ return metadata, num_objs
+
+ # GPU ------------------------------------------------
+
+ def init_gpu_pipeline(self):
+ self.gpu_id = 0
+ self.inference_stream = torch.cuda.Stream()
+
+ if self.timer_enabled:
+ self.sf_start, self.sf_end = (
+ torch.cuda.Event(enable_timing=True),
+ torch.cuda.Event(enable_timing=True),
+ )
+ self.roi_start, self.roi_end = (
+ torch.cuda.Event(enable_timing=True),
+ torch.cuda.Event(enable_timing=True),
+ )
+ self.det_start, self.det_end = (
+ torch.cuda.Event(enable_timing=True),
+ torch.cuda.Event(enable_timing=True),
+ )
+
+ # --- START: FUSED KERNEL COMPILATION ---
+ # This CUDA C++ kernel fuses a 3x3 Box Blur and a binary threshold (value: 50).
+ # A fused dilation is very complex; we apply it separately for correctness.
+ # This still reduces 3 kernel launches (Blur, Thresh, Dilate) to 2 (Fused, Dilate).
+ fused_blur_thresh_kernel_code = r"""
+ extern "C" __global__
+ void fused_blur_thresh(const unsigned char* src, unsigned char* dst, int width, int height, int src_step, int dst_step) {
+ int x = blockIdx.x * blockDim.x + threadIdx.x;
+ int y = blockIdx.y * blockDim.y + threadIdx.y;
+
+ if (x >= width || y >= height) return;
+
+ // 1. Fused 3x3 Box Blur
+ float sum = 0.0f;
+ int count = 0;
+ for (int dy = -1; dy <= 1; ++dy) {
+ for (int dx = -1; dx <= 1; ++dx) {
+ int nx = x + dx;
+ int ny = y + dy;
+ if (nx >= 0 && nx < width && ny >= 0 && ny < height) {
+ sum += src[ny * src_step + nx];
+ count++;
+ }
+ }
+ }
+ float blurred_val = sum / count;
+
+ // 2. Fused Threshold
+ unsigned char output_val = (blurred_val > 50.0f) ? 255 : 0;
+
+ // Write the final result
+ dst[y * dst_step + x] = output_val;
+ }
+ """
+ self.fused_blur_thresh_kernel = cupy.RawKernel(
+ fused_blur_thresh_kernel_code, "fused_blur_thresh"
+ )
+
+ # This CUDA C++ kernel fuses a 3x3 Box Blur, a binary threshold (50), and a 3x3 Dilation.
+ # fused_3_in_1_kernel_code = r'''
+ # #define TILE_DIM 16
+ # #define BLOCK_DIM 16
+
+ # extern "C" __global__
+ # void fused_btd(const unsigned char* src, unsigned char* dst, int width, int height, int src_step, int dst_step) {
+
+ # // Shared memory for the source tile and the intermediate thresholded tile
+ # __shared__ unsigned char s_src_tile[TILE_DIM + 2][TILE_DIM + 2];
+ # __shared__ unsigned char s_thresh_tile[TILE_DIM][TILE_DIM];
+
+ # int tx = threadIdx.x;
+ # int ty = threadIdx.y;
+ # int gx = blockIdx.x * TILE_DIM + tx;
+ # int gy = blockIdx.y * TILE_DIM + ty;
+
+ # // Load source data into shared memory (including a 1-pixel halo)
+ # for (int i = ty; i < TILE_DIM + 2; i += BLOCK_DIM) {
+ # for (int j = tx; j < TILE_DIM + 2; j += BLOCK_DIM) {
+ # int load_x = gx - 1 + j;
+ # int load_y = gy - 1 + i;
+ # if (load_x >= 0 && load_x < width && load_y >= 0 && load_y < height) {
+ # s_src_tile[i][j] = src[load_y * src_step + load_x];
+ # } else {
+ # s_src_tile[i][j] = 0;
+ # }
+ # }
+ # }
+ # __syncthreads();
+
+ # // --- 1. Fused Blur & Threshold ---
+ # if (tx < TILE_DIM && ty < TILE_DIM) {
+ # float sum = 0.0f;
+ # for (int dy = 0; dy <= 2; ++dy) {
+ # for (int dx = 0; dx <= 2; ++dx) {
+ # sum += s_src_tile[ty + dy][tx + dx];
+ # }
+ # }
+ # float blurred_val = sum / 9.0f;
+ # s_thresh_tile[ty][tx] = (blurred_val > 50.0f) ? 255 : 0;
+ # }
+ # __syncthreads();
+
+ # // --- 2. Fused Dilation ---
+ # if (tx < TILE_DIM && ty < TILE_DIM && gx < width && gy < height) {
+ # unsigned char max_val = 0;
+ # for (int dy = -1; dy <= 1; ++dy) {
+ # for (int dx = -1; dx <= 1; ++dx) {
+ # int check_x = tx + dx;
+ # int check_y = ty + dy;
+ # if (check_x >= 0 && check_x < TILE_DIM && check_y >= 0 && check_y < TILE_DIM) {
+ # if (s_thresh_tile[check_y][check_x] > max_val) {
+ # max_val = s_thresh_tile[check_y][check_x];
+ # }
+ # }
+ # }
+ # }
+ # dst[gy * dst_step + gx] = max_val;
+ # }
+ # }
+ # '''
+
+ # fused_3_in_1_kernel_code = r'''
+ # extern "C" __global__
+ # void fused_btd(const unsigned char* src, unsigned char* dst, int width, int height, int src_step, int dst_step) {
+ # int x = blockIdx.x * blockDim.x + threadIdx.x;
+ # int y = blockIdx.y * blockDim.y + threadIdx.y;
+
+ # if (x >= width || y >= height) return;
+
+ # // --- STEP 1 & 2: 3x3 NEIGHBORHOOD BLUR & THRESHOLD ---
+ # // We compute the thresholded values of the local 3x3 neighborhood on the fly
+ # unsigned char local_thresh[3][3];
+
+ # for (int dy = -1; dy <= 1; ++dy) {
+ # for (int dx = -1; dx <= 1; ++dx) {
+ # int nx = x + dx;
+ # int ny = y + dy;
+
+ # // Clamp coordinates to image edges safely
+ # nx = (nx < 0) ? 0 : ((nx >= width) ? width - 1 : nx);
+ # ny = (ny < 0) ? 0 : ((ny >= height) ? height - 1 : ny);
+
+ # // Compute blur for coordinate (nx, ny)
+ # float sum = 0.0f;
+ # int count = 0;
+ # for (int k_dy = -1; k_dy <= 1; ++k_dy) {
+ # for (int k_dx = -1; k_dx <= 1; ++k_dx) {
+ # int k_nx = nx + k_dx;
+ # int k_ny = ny + k_dy;
+
+ # k_nx = (k_nx < 0) ? 0 : ((k_nx >= width) ? width - 1 : k_nx);
+ # k_ny = (k_ny < 0) ? 0 : ((k_ny >= height) ? height - 1 : k_ny);
+
+ # sum += src[k_ny * src_step + k_nx];
+ # count++;
+ # }
+ # }
+ # float blurred_val = sum / (float)count;
+
+ # // Threshold instantly
+ # local_thresh[dy + 1][dx + 1] = (blurred_val > 50.0f) ? 255 : 0;
+ # }
+ # }
+
+ # // --- STEP 3: DILATE ON THRESHOLDED RESULTS ---
+ # unsigned char max_val = 0;
+ # for (int dy = 0; dy < 3; ++dy) {
+ # for (int dx = 0; dx < 3; ++dx) {
+ # if (local_thresh[dy][dx] > max_val) {
+ # max_val = local_thresh[dy][dx];
+ # }
+ # }
+ # }
+
+ # // Write the final dilate value safely to global memory
+ # dst[y * dst_step + x] = max_val;
+ # }
+ # '''
+ fused_3_in_1_kernel_code = r"""
+ extern "C" __global__
+ void fused_btd(const unsigned char* src, unsigned char* dst, int width, int height, int src_step, int dst_step) {
+ int x = blockIdx.x * blockDim.x + threadIdx.x;
+ int y = blockIdx.y * blockDim.y + threadIdx.y;
+
+ if (x >= width || y >= height) return;
+
+ // --- STEP 1 & 2: 17x17 BLUR & THRESHOLD ---
+ // 5x5 local neighborhood for the Dilation phase
+ unsigned char local_thresh[5][5];
+
+ // 5x5 Dilation radius (dy, dx from -2 to 2)
+ for (int dy = -2; dy <= 2; ++dy) {
+ for (int dx = -2; dx <= 2; ++dx) {
+ int nx = x + dx;
+ int ny = y + dy;
+
+ // Clamp coordinates to image edges safely
+ nx = (nx < 0) ? 0 : ((nx >= width) ? width - 1 : nx);
+ ny = (ny < 0) ? 0 : ((ny >= height) ? height - 1 : ny);
+
+ // 5x5 Blur radius (k_dy, k_dx from -8 to 8 (for 17x17))
+ float sum = 0.0f;
+ int count = 0;
+ for (int k_dy = -2; k_dy <= 2; ++k_dy) {
+ for (int k_dx = -8; k_dx <= 8; ++k_dx) {
+ int k_nx = nx + k_dx;
+ int k_ny = ny + k_dy;
+
+ k_nx = (k_nx < 0) ? 0 : ((k_nx >= width) ? width - 1 : k_nx);
+ k_ny = (k_ny < 0) ? 0 : ((k_ny >= height) ? height - 1 : k_ny);
+
+ sum += src[k_ny * src_step + k_nx];
+ count++;
+ }
+ }
+ float blurred_val = sum / (float)count;
+
+ // Threshold instantly (offset dx, dy by +2 to fit in 0-4 array indices)
+ local_thresh[dy + 2][dx + 2] = (blurred_val > 50.0f) ? 255 : 0;
+ }
+ }
+
+ // --- STEP 3: 5x5 DILATE ON THRESHOLDED RESULTS ---
+ unsigned char max_val = 0;
+ for (int dy = 0; dy < 5; ++dy) {
+ for (int dx = 0; dx < 5; ++dx) {
+ if (local_thresh[dy][dx] > max_val) {
+ max_val = local_thresh[dy][dx];
+ }
+ }
+ }
+
+ // Write the final dilate value safely to global memory
+ dst[y * dst_step + x] = max_val;
+ }
+ """
+
+ self.fused_3_in_1_kernel = cupy.RawKernel(fused_3_in_1_kernel_code, "fused_btd")
+ # --- END: FUSED KERNEL COMPILATION ---
+
+ self.prepare_gpu_pipeline()
+
+ def allocate_gpu(self):
+ """
+ Allocates persistent GpuMat buffers and CUDA streams to
+ enable zero-copy GPU processing.
+ """
+ self.device_index = f"cuda:{self.gpu_id}"
+
+ # Safely extract the primitive compiled C++ function pointer out of CuPy's RawKernel
+ if hasattr(get_bounds_kernel, "kernel"):
+ self._raw_bounds_function = get_bounds_kernel.kernel
+ else:
+ # Fallback handle if utilizing an alternate CuPy execution wrapper mapping
+ self._raw_bounds_function = get_bounds_kernel
+
+ ksize = (17, 17)
+ self._cuda_gaussian_filter = cv2.cuda.createGaussianFilter(
+ srcType=cv2.CV_8UC1, dstType=cv2.CV_8UC1, ksize=ksize, sigma1=0
+ )
+
+ self.recycled_resize_mat = cv2.cuda.GpuMat(
+ self.resize_h, self.resize_w, cv2.CV_8UC3
+ )
+ # Pre-allocate double buffers for resizing
+ # self.recycled_resize_mat_A = cv2.cuda.GpuMat(self.resize_h, self.resize_w, cv2.CV_8UC3)
+ # self.recycled_resize_mat_B = cv2.cuda.GpuMat(self.resize_h, self.resize_w, cv2.CV_8UC3)
+
+ # Track which buffer is active
+ self.use_buffer_A = True
+
+ self.raw_mask = cv2.cuda.GpuMat(self.resize_h, self.resize_w, cv2.CV_8UC1)
+ self.thresh_mask = cv2.cuda.GpuMat(self.resize_h, self.resize_w, cv2.CV_8UC1)
+ self.clean_mask = cv2.cuda.GpuMat(self.resize_h, self.resize_w, cv2.CV_8UC1)
+ self.d_blurred = cv2.cuda.GpuMat(self.clean_mask.size(), cv2.CV_8UC1)
+
+ # Compile the kernel once at startup
+ # self._row_bounds_function = cupy.RawKernel(PROPAGATION_KERNEL_CODE, "get_row_bounds_fused")
+
+ # self.fixed_inference_batch = torch.empty(
+ # (
+ # self.config.MODEL_MAX_BATCH_SIZE,
+ # 3,
+ # self.config.MODEL_H,
+ # self.config.MODEL_W,
+ # ),
+ # dtype=torch.half,
+ # device="cuda",
+ # )
+
+ # self.gpu_float_staging = None
+
+ self.gpu_float_staging = torch.empty(
+ (1, 3, self.frame_height, self.frame_width),
+ dtype=torch.float16,
+ device=self.device_index,
+ )
+ # self.stream = cv2.cuda.Stream()
+ # self.ingest_stream = torch.cuda.Stream()
+ # self.inference_stream = torch.cuda.Stream()
+ self.bgs_stream = cv2.cuda.Stream()
+ # self.gpu_fullres_frame = cv2.cuda.GpuMat(
+ # self.frame_height, self.frame_width, cv2.CV_8UC3
+ # )
+ # self.resized_gpumat = cv2.cuda.GpuMat(self.resize_h, self.resize_w, cv2.CV_8UC3)
+ # self.resized_frame = cv2.cuda.GpuMat(self.resize_h, self.resize_w, cv2.CV_8UC3)
+ # self.resized_frame.setTo(0, self.bgs_stream)
+ # self.fgMask = cv2.cuda.GpuMat(self.resize_h, self.resize_w, cv2.CV_8UC1)
+ self.prev_bkgd = cv2.cuda.GpuMat(self.resize_h, self.resize_w, cv2.CV_8UC1)
+ if self.config.BKGD_SUB_INCLUDE_HISTORY_METHOD == "and":
+ self.prev_bkgd.setTo((1,))
+ else:
+ self.prev_bkgd.setTo((0,))
+ self.mask_history = deque(
+ maxlen=self.config.BKGD_SUB_INCLUDE_HISTORY_TEMPORAL_SIZE
+ )
+ self.mask_history.append(self.prev_bkgd)
+
+ self.gpu_threshold_dst_frame = cv2.cuda.GpuMat(
+ self.resize_h, self.resize_w, cv2.CV_8UC1
+ )
+ self.gpu_morphed_frame = cv2.cuda.GpuMat(
+ self.resize_h, self.resize_w, cv2.CV_8UC1
+ )
+
+ self.upload_stream = cv2.cuda.Stream()
+ self.upload_event = cv2.cuda.Event()
+
+ # self.queue_capacity = int(2 * self.target_fps) # 60
+ self.num_buffers = int(2 * self.target_fps) # self.queue_capacity + 5
+ self.gpu_buffer_pool = [
+ cv2.cuda.GpuMat(self.frame_height, self.frame_width, cv2.CV_8UC3)
+ for _ in range(self.num_buffers)
+ ]
+ self.buffer_idx = 0
+
+ self.frame_buffer_pool = [
+ torch.empty(
+ (3, self.frame_height, self.frame_width),
+ dtype=torch.uint8,
+ device="cuda",
+ )
+ for _ in range(2)
+ ]
+ self.pool_idx = 0
+
+ # Create a matching pool of pinned host memory for the 8K frames
+ # self.host_buffer_pool = [
+ # cv2.cuda.HostMem(self.frame_height, self.frame_width, cv2.CV_8UC3)
+ # for _ in range(self.num_buffers)
+ # ]
+
+ # Create continuous buffers to prevent stride artifacts during 8K downloads
+ self.pinned_downloaded_resizedframe_np = cv2.cuda.createContinuous(
+ self.resize_h, self.resize_w, cv2.CV_8UC3
+ )
+ self.pinned_downloaded_frame_np = cv2.cuda.createContinuous(
+ self.resize_h, self.resize_w, cv2.CV_8UC1
+ )
+ cv2.cuda.createContinuous(
+ self.resize_h, self.resize_w, cv2.CV_8UC1, self.gpu_threshold_dst_frame
+ )
+ cv2.cuda.createContinuous(
+ self.resize_h, self.resize_w, cv2.CV_8UC1, self.gpu_morphed_frame
+ )
+
+ # This prevents the AI thread from overwriting the encoder's data.
+ self.gpu_encoder_8k_buf = cv2.cuda.createContinuous(
+ self.frame_height, self.frame_width, cv2.CV_8UC3
+ )
+
+ # Continuous allocation prevents stride/padding artifacts
+ self.gpu_display_frame = cv2.cuda.createContinuous(
+ self.disp_h, self.disp_w, cv2.CV_8UC3
+ )
+
+ # Create a dedicated background stream for encoding tasks
+ self.encode_stream = cv2.cuda.Stream()
+
+ # Allocate a permanent float32/float16 channel layout space directly on VRAM
+ # self.static_gpu_360p = torch.empty(
+ # (1, 3, self.disp_h, self.disp_w),
+ # dtype=torch.float32,
+ # device="cuda",
+ # )
+ # self.static_gpu_byte_bchw = torch.empty(
+ # (1, 3, self.disp_h, self.disp_w),
+ # dtype=torch.uint8,
+ # device="cuda",
+ # ).contiguous()
+
+ # Allocates a fixed memory space directly accessible by your GPU DMA engine
+ self.pinned_cpu_xyxy = torch.empty(
+ (self.config.MAX_DETECTIONS, 4), dtype=torch.float32, device="cpu"
+ ).pin_memory()
+ self.pinned_cpu_clss = torch.empty(
+ (self.config.MAX_DETECTIONS,), dtype=torch.int32, device="cpu"
+ ).pin_memory()
+ self.pinned_cpu_confs = torch.empty(
+ (self.config.MAX_DETECTIONS,), dtype=torch.float32, device="cpu"
+ ).pin_memory()
+
+ self.cupy_structure = cupy.ones((3, 3), dtype=cupy.int32) # , order="C")
+ self._max_labels = 1024 # Cap maximum tracking elements per frame
+ # Pre-allocate static, persistent workspace array caches straight in VRAM
+ self._x1_pool = cupy.zeros((self._max_labels,), dtype=cupy.int32)
+ self._y1_pool = cupy.zeros((self._max_labels,), dtype=cupy.int32)
+ self._x2_pool = cupy.zeros((self._max_labels,), dtype=cupy.int32)
+ self._y2_pool = cupy.zeros((self._max_labels,), dtype=cupy.int32)
+
+ # Pre-allocate the labeled output tensor space to avoid internal allocation churn
+ # Match your target resized canvas dimensions (e.g., 640x640)
+ self._labeled_scratch = cupy.empty(
+ (self.resize_h, self.resize_w), dtype=cupy.int32, order="C"
+ )
+
+ # Create two isolated tracking canvases to handle the ping-pong data stream
+ # self.static_host_canvases = [
+ # np.zeros((self.disp_h, self.disp_w, 3), dtype=np.uint8),
+ # np.zeros((self.disp_h, self.disp_w, 3), dtype=np.uint8),
+ # ]
+ # self.canvas_selector = 0
+
+ # # Register BOTH buffers as page-locked memory
+ # cv2.cuda.registerPageLocked(self.static_host_canvases[0])
+ # cv2.cuda.registerPageLocked(self.static_host_canvases[1])
+
+ def prepare_gpu_pipeline(self):
+ self.allocate_gpu()
+
+ # Subtraction
+ # self.backSub_lock = threading.Lock()
+ # history = int(2 * self.target_fps) # 300 # int(5 * self.target_fps)
+ self.lr = self.config.BKGD_SUB_MOG2_LR # 1 / history
+ self.backSub = cv2.cuda.createBackgroundSubtractorMOG2(
+ history=self.config.BKGD_SUB_MOG2_HISTORY, # Clear ghosts of fast drones in ~2 seconds (2*fps)
+ varThreshold=int(
+ self.config.BKGD_SUB_MOG2_VARTHRESHOLD # 1.15
+ ), # High threshold to ignore "shimmer" and compression noise # default 16
+ # varThreshold=50, #self.config.BKGD_SUB_MOG2_VARTHRESHOLD, # High threshold to ignore "shimmer" and compression noise # default 16
+ # CUDA implementation of MOG2 often requires a higher varThreshold to achieve the same "cleanliness" as the CPU (15-20%)
+ detectShadows=self.config.BKGD_SUB_MOG2_DETECTSHADOWS, # default True
+ )
+ self.opencv_bgs_output = cv2.cuda.GpuMat(
+ self.resize_h, self.resize_w, cv2.CV_8UC1
+ )
+ self.opencv_bgs_dilate_output = cv2.cuda.GpuMat(
+ self.resize_h, self.resize_w, cv2.CV_8UC1
+ )
+
+ # Force the GPU to match the CPU's background criteria limits
+ # self.backSub.setBackgroundRatio(0.05) # Standardize background matching speed
+ # self.backSub.setComplexityReductionThreshold(0.05) # Drop unstable low-variance noise regions
+ # self.backSub.setVarMin(4.0) # High-pass filter to erase micro-vibrations
+
+ self.dilate_filter = cv2.cuda.createMorphologyFilter(
+ cv2.MORPH_DILATE, cv2.CV_8U, self.dilate_kernel
+ )
+ self.dilate_filter_for_enhanced_mask = cv2.cuda.createMorphologyFilter(
+ cv2.MORPH_DILATE, cv2.CV_8UC1, self.dilate_kernel_for_enhanced_mask
+ )
+ # # self.morph_kernel = cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (5, 5))
+ # self.morph_kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (5, 5))
+ # self.morph_filter = cv2.cuda.createMorphologyFilter(
+ # cv2.MORPH_DILATE, cv2.CV_8UC1, self.morph_kernel
+ # )
+ # self.labels_gpu = cv2.cuda.GpuMat(self.resize_h, self.resize_w, cv2.CV_32S)
+ # self.labels_gpu = cv2.cuda.GpuMat(self.resize_h, self.resize_w, cv2.CV_8U)
+ # self.labels_gpu.setTo(0, self.bgs_stream)
+
+ def gpu_warmup(self):
+ """
+ Comprehensive, non-blocking GPU warmup harness.
+ Pre-compiles CuPy Scipy labeling kernels, box-merging logic,
+ and TensorRT engine profiles without disk writing side effects.
+ """
+ # import cv2
+ # import torch
+
+ self.warmup_stream = cv2.cuda.Stream()
+
+ h, w = self.resize_h, self.resize_w
+
+ # 1. Standard OpenCV CUDA Pre-processing Compilation
+ gpu_warmup_input_frame = cv2.cuda_GpuMat(h, w, cv2.CV_8U)
+ if gpu_warmup_input_frame is not None:
+ gpu_warmup_input_frame.setTo(255)
+ gpu_warmup_frame = cv2.cuda_GpuMat(h, w, cv2.CV_8U)
+
+ cv2.cuda.resize(
+ gpu_warmup_input_frame,
+ (w, h),
+ stream=self.warmup_stream,
+ dst=gpu_warmup_frame,
+ interpolation=cv2.INTER_NEAREST,
+ )
+
+ gpu_threshold_dst_frame = cv2.cuda_GpuMat(h, w, cv2.CV_8U)
+ cv2.cuda.threshold(
+ gpu_warmup_frame,
+ self.config.THRESHOLD_VALUE,
+ self.config.THRESHOLD_MAX_VALUE,
+ cv2.THRESH_BINARY,
+ gpu_threshold_dst_frame,
+ self.warmup_stream,
+ )
+
+ gpu_morphed_frame = cv2.cuda_GpuMat(h, w, cv2.CV_8U)
+ self.dilate_filter.apply(
+ gpu_threshold_dst_frame, gpu_morphed_frame, self.warmup_stream
+ )
+ active_stream_ptr = (
+ self.bgs_stream.cudaPtr() if hasattr(self, "bgs_stream") else 0
+ )
+ self.warmup_stream.waitForCompletion()
+
+ mock_mask = cupy.zeros((640, 640), dtype=cupy.uint8)
+ mock_mask[100:150, 100:150] = 1 # Draw a dummy blob
+ # mock_scratch = cupy.empty((640, 640), dtype=cupy.int32)
+ # mock_structure = cupy.ones((3, 3), dtype=cupy.int32)
+
+ # Invoke once to trigger NVRTC compilation and cache the binary module to disk
+ # cupyx.scipy.ndimage.label(mock_mask, structure=self.cupy_structure, output=mock_scratch)
+ with cupy.cuda.ExternalStream(active_stream_ptr):
+ cupyx.scipy.ndimage.label(
+ mock_mask, structure=self.cupy_structure, output=self._labeled_scratch
+ )
+ # cupy.cuda.stream.get_current_stream().synchronize()
+ cupy.cuda.ExternalStream(active_stream_ptr).synchronize()
+ main_app_logger.info("[WARMUP] CuPy SciPy Labeling kernels fully cached.")
+
+ # 🚀 2. PIPELINE EXTENSION: Pre-allocate static canvas memories if missing
+ # if not hasattr(self, "static_canvas_scratch"):
+ # self.static_canvas_scratch = torch.full(
+ # (640, 640, 3), fill_value=114, dtype=torch.uint8, device="cuda"
+ # )
+
+ # 🚀 3. PIPELINE EXTENSION: Pre-compile AI Model and Box reduction pipelines
+ # Capture the original debug configuration state
+ original_debug_flag = getattr(self.config, "DEBUG_FLAG", False)
+
+ try:
+ # Force the debug flag off during warmup loop to completely block debug_save_crops
+ self.config.DEBUG_FLAG = False
+
+ # Construct a synthetic 8K model input canvas block on the GPU device
+ # Adjust dimensions if your raw source video maps differently than [8192, 8192, 3]
+ synthetic_8k_tensor = torch.zeros(
+ (self.frame_height, self.frame_width, 3),
+ dtype=torch.uint8,
+ device="cuda",
+ )
+
+ # Create a standard mock overlapping region block to exercise merge_boxes_gpu
+ # Shapes: [N, 4] -> format (x1, y1, x2, y2)
+ mock_overlapping_boxes = torch.tensor(
+ [
+ [100.0, 100.0, 250.0, 250.0],
+ [102.0, 101.0, 248.0, 252.0], # Overlapping twin bounding box
+ [500.0, 500.0, 640.0, 640.0],
+ ],
+ dtype=torch.float32,
+ device="cuda",
+ )
+
+ # Run exactly 3 dummy steps to guarantee CuPy Scipy features and TensorRT allocations cache
+ main_app_logger.info(
+ "[WARMUP] Priming algorithm pipeline layers asynchronously..."
+ )
+ for _ in range(3):
+ with torch.inference_mode():
+ # Feed through the master detection utility function directly
+ # _, _ = self.get_detections_with_smart_filtering(
+ # frame=synthetic_8k_tensor,
+ # frame_id=0,
+ # device_input="cuda",
+ # merged=mock_overlapping_boxes,
+ # )
+ self.get_detections(
+ frame=synthetic_8k_tensor,
+ frame_id=0,
+ device_input="cuda",
+ merged=mock_overlapping_boxes
+ if self.config.sf_enabled
+ else None,
+ )
+
+ # Hard fence sync to secure compilation layers across active CUDA contexts
+ # torch.cuda.synchronize()
+ main_app_logger.info(
+ "[WARMUP] Complete. All pipeline pathways completely compiled."
+ )
+
+ finally:
+ # Safely restore your original testing workflow debug constraints
+ self.config.DEBUG_FLAG = original_debug_flag
+
+ # Release resources
+ del synthetic_8k_tensor, mock_overlapping_boxes
+ del (
+ gpu_warmup_input_frame,
+ gpu_warmup_frame,
+ gpu_threshold_dst_frame,
+ gpu_morphed_frame,
+ )
+ if hasattr(self, "warmup_stream"):
+ delattr(self, "warmup_stream")
+
+ torch.cuda.empty_cache()
+
+ def cleanup_gpu_v1(self):
+ """
+ Explicitly releases all GPU-allocated memory to prevent
+ VRAM leaks in 8K concurrent streams.
+ """
+ # Iterate through class attributes to explicitly release VRAM.
+ for attr_name in list(self.__dict__.keys()):
+ attr_value = getattr(self, attr_name)
+
+ # Check if the attribute is a GpuMat
+ if isinstance(attr_value, cv2.cuda.GpuMat):
+ # 🏎️ Force the NVIDIA driver to deallocate this specific memory segment
+ attr_value.release()
+ setattr(self, attr_name, None)
+ main_app_logger.info(f"✅ Released GpuMat: {attr_name}")
+
+ # if hasattr(self, "gpu_fullres_frame") and self.gpu_fullres_frame is not None:
+ # try:
+ # self.gpu_fullres_frame.release()
+ # except Exception:
+ # self.gpu_fullres_frame = None
+
+ if hasattr(self, "gpu_morphed_frame") and self.gpu_morphed_frame is not None:
+ try:
+ self.gpu_morphed_frame.release()
+ except Exception:
+ self.gpu_morphed_frame = None
+
+ # if hasattr(self, "labels_gpu") and self.labels_gpu is not None:
+ # try:
+ # self.labels_gpu.release()
+ # except Exception:
+ # self.labels_gpu = None
+
+ if hasattr(self, "gpu_encoder_8k_buf") and self.gpu_encoder_8k_buf is not None:
+ try:
+ self.gpu_encoder_8k_buf.release()
+ except Exception:
+ self.gpu_encoder_8k_buf = None
+
+ if hasattr(self, "gpu_display_frame") and self.gpu_display_frame is not None:
+ try:
+ self.gpu_display_frame.release()
+ except Exception:
+ self.gpu_display_frame = None
+
+ if hasattr(self, "gpu_crop_batch"):
+ for mat in self.gpu_crop_batch:
+ if isinstance(mat, cv2.cuda.GpuMat):
+ mat.release()
+ self.gpu_crop_batch = []
+
+ # if hasattr(self, "stream"):
+ # self.stream.waitForCompletion()
+
+ self.pinned_downloaded_resizedframe_np = None
+ self.gpu_threshold_dst_frame = None
+ self.gpu_morphed_frame = None
+ self.pinned_downloaded_frame_np = None
+
+ # Handle specific buffers (like your Ping-Pong lists)
+ # if hasattr(self, "encode_buffers"):
+ # self.encode_buffers.clear()
+
+ # Clear the BGS history
+ if hasattr(self, "mask_history"):
+ self.mask_history.clear()
+
+ # Optional: Final flush of the CUDA caching allocator
+ if torch.cuda.is_available():
+ torch.cuda.empty_cache()
+
+ def apply_background_subtraction_gpu(
+ self, gpu_frame, include_history=True, method="and", stream=None
+ ):
+ stream = stream if isinstance(stream, cv2.cuda.Stream) else self.bgs_stream
+
+ # raw_mask = self.backSub.apply(
+ # gpu_frame,
+ # # motion_input,
+ # float(self.lr), # 0.005, # float(self.lr),
+ # stream=stream,
+ # )
+ self.backSub.apply(
+ image=gpu_frame,
+ fgmask=self.opencv_bgs_output,
+ learningRate=float(self.lr),
+ stream=stream,
+ )
+ raw_mask = self.opencv_bgs_output
+
+ if include_history:
+ # if self.config.BKGD_SUB_INCLUDE_HISTORY_METHOD == "and":
+ # self.prev_bkgd.setTo((1,))
+ # else:
+ # self.prev_bkgd.setTo((0,))
+ # If the deque is full, pop the oldest GpuMat and explicitly release it.
+ if len(self.mask_history) >= self.mask_history.maxlen:
+ old_mask = self.mask_history.popleft()
+ if old_mask is not None:
+ # It releases the VRAM held by the oldest mask.
+ old_mask.release()
+
+ for m in list(self.mask_history):
+ # Dilate the historical mask on GPU
+ # dilated = self.dilate_filter_for_enhanced_mask.apply(m, stream=stream)
+ self.dilate_filter_for_enhanced_mask.apply(
+ src=m, dst=self.opencv_bgs_dilate_output, stream=stream
+ )
+
+ if method == "or":
+ # Bitwise OR on GPU
+ cv2.cuda.bitwise_or(
+ self.prev_bkgd,
+ self.opencv_bgs_dilate_output,
+ self.prev_bkgd,
+ stream=stream,
+ )
+ else:
+ # Bitwise AND on GPU
+ cv2.cuda.bitwise_and(
+ self.prev_bkgd,
+ self.opencv_bgs_dilate_output,
+ self.prev_bkgd,
+ stream=stream,
+ )
+
+ self.mask_history.append(raw_mask.clone()) # .clone()
+
+ raw_mask = cv2.cuda.bitwise_or(
+ raw_mask, self.prev_bkgd, stream=self.bgs_stream
+ )
+ return raw_mask
+
+ def get_sf_gpu_rois_v1(
+ self, device_frame, overall_frame_num, max_candidates=100, limit_640=1280
+ ):
+ debug_frame_limit = self.debug_frame_limit
+
+ if overall_frame_num <= debug_frame_limit:
+ stage_debug_dir = (
+ self.result_dir / "debug_stages" / self._testMethodName / "roi_stages"
+ )
+ stage_debug_dir.mkdir(parents=True, exist_ok=True)
+
+ # 2. BRIDGE THE PYTORCH TO OPENCV VRAM GAP
+ if torch.is_tensor(device_frame):
+ h_raw, w_raw, ch = device_frame.shape
+ cuda_mem_ptr = device_frame.data_ptr()
+ cv_type = cv2.CV_8UC3 if ch == 3 else cv2.CV_8UC1
+ row_step_bytes = device_frame.stride()[0] * device_frame.element_size()
+ src_gpu_mat = cv2.cuda.createGpuMatFromCudaMemory(
+ h_raw, w_raw, cv_type, cuda_mem_ptr, step=row_step_bytes
+ )
+ else:
+ src_gpu_mat = device_frame
+ ch = src_gpu_mat.channels()
+
+ if overall_frame_num <= debug_frame_limit:
+ src_cpu = src_gpu_mat.download()
+ if ch == 3:
+ src_cpu = cv2.cvtColor(src_cpu, cv2.COLOR_RGB2BGR)
+ cv2.imwrite(
+ str(stage_debug_dir / f"frame_{overall_frame_num:04d}_stage1_src.jpg"),
+ src_cpu,
+ )
+
+ # 3. STRIDED LAYOUT INITIALIZATION
+ # if not hasattr(self, "raw_mask"):
+ # self.recycled_resize_mat = cv2.cuda.GpuMat(self.resize_h, self.resize_w, cv2.CV_8UC3 if ch == 3 else cv2.CV_8UC1)
+ # self.raw_mask = cv2.cuda.GpuMat(self.resize_h, self.resize_w, cv2.CV_8UC1)
+ # self.thresh_mask = cv2.cuda.GpuMat(self.resize_h, self.resize_w, cv2.CV_8UC1)
+ # self.clean_mask = cv2.cuda.GpuMat(self.resize_h, self.resize_w, cv2.CV_8UC1)
+ # self.d_blurred = cv2.cuda.GpuMat(self.clean_mask.size(), cv2.CV_8UC1)
+
+ # 4. RUN ASYNCHRONOUS DOWN-SAMPLING GATE
+ cv2.cuda.resize(
+ src_gpu_mat,
+ dst=self.recycled_resize_mat,
+ dsize=(self.resize_w, self.resize_h),
+ interpolation=cv2.INTER_NEAREST,
+ stream=self.bgs_stream,
+ )
+
+ if overall_frame_num <= debug_frame_limit:
+ self.bgs_stream.waitForCompletion()
+ resize_cpu = self.recycled_resize_mat.download()
+ if ch == 3:
+ resize_cpu = cv2.cvtColor(resize_cpu, cv2.COLOR_RGB2BGR)
+ cv2.imwrite(
+ str(
+ stage_debug_dir / f"frame_{overall_frame_num:04d}_stage2_resize.jpg"
+ ),
+ resize_cpu,
+ )
+
+ # 5. BACKGROUND SUBTRACTION & FILTERING
+ self.raw_mask = self.apply_background_subtraction_gpu(
+ self.recycled_resize_mat,
+ include_history=self.config.BKGD_SUB_INCLUDE_HISTORY,
+ method=self.config.BKGD_SUB_INCLUDE_HISTORY_METHOD,
+ stream=self.bgs_stream,
+ )
+
+ if overall_frame_num <= debug_frame_limit:
+ self.bgs_stream.waitForCompletion()
+ cv2.imwrite(
+ str(
+ stage_debug_dir
+ / f"frame_{overall_frame_num:04d}_stage5_history_mask.jpg"
+ ),
+ self.raw_mask.download(),
+ )
+
+ self._cuda_gaussian_filter.apply(
+ self.raw_mask, self.d_blurred, stream=self.bgs_stream
+ )
+
+ if overall_frame_num <= debug_frame_limit:
+ self.bgs_stream.waitForCompletion()
+ cv2.imwrite(
+ str(
+ stage_debug_dir / f"frame_{overall_frame_num:04d}_stage5b_blur.jpg"
+ ),
+ self.d_blurred.download(),
+ )
+
+ cv2.cuda.threshold(
+ self.d_blurred,
+ 50,
+ self.config.THRESHOLD_MAX_VALUE,
+ cv2.THRESH_BINARY,
+ self.thresh_mask,
+ stream=self.bgs_stream,
+ )
+
+ if overall_frame_num <= debug_frame_limit:
+ self.bgs_stream.waitForCompletion()
+ cv2.imwrite(
+ str(
+ stage_debug_dir
+ / f"frame_{overall_frame_num:04d}_stage6_threshold.jpg"
+ ),
+ self.thresh_mask.download(),
+ )
+
+ self.dilate_filter.apply(self.thresh_mask, self.clean_mask, self.bgs_stream)
+
+ # [STAGE 7 DEBUG] Check Final Dilated Output Mask
+ if overall_frame_num <= debug_frame_limit:
+ self.bgs_stream.waitForCompletion()
+ cv2.imwrite(
+ str(
+ stage_debug_dir
+ / f"frame_{overall_frame_num:04d}_stage8_final_dilatemask.jpg"
+ ),
+ self.clean_mask.download(),
+ )
+
+ # inf_data = {
+ # "mask": self.clean_mask,
+ # "full_frame": device_frame,
+ # "frameNum": overall_frame_num,
+ # }
+ # start = torch.cuda.Event(enable_timing=True)
+ # end = torch.cuda.Event(enable_timing=True)
+
+ # 6. REGION OF INTEREST ANALYSIS
+ limit_8K = 1280
+ limit_640 = limit_8K / self.scale_x
+ # start.record()
+ raw_boxes = self.get_gpu_rois_by_area(
+ overall_frame_num, self.clean_mask, max_candidates=50, limit_640=limit_640
+ )
+ # end.record()
+ # torch.cuda.synchronize()
+ # gpu_rois_by_area_ms = start.elapsed_time(end)
+
+ if raw_boxes.shape[0] < 1:
+ return torch.empty((0, 4), device=self.device_input)
+
+ raw_boxes = self.gpu_roi_padding(raw_boxes, method="uniform")
+ # start.record()
+ clean_640p = merge_boxes_gpu(
+ raw_boxes,
+ gap_limit=self.dist_thresh_640,
+ size_limit=limit_640,
+ max_cached_elements=self.max_cached_elements,
+ )
+ # end.record()
+ # torch.cuda.synchronize()
+ # merge_ms = start.elapsed_time(end)
+
+ # main_app_logger.info(f"gpu_rois_by_area: {gpu_rois_by_area_ms:.3f} ms")
+ # main_app_logger.info(f"merge_boxes_gpu: {merge_ms:.3f} ms")
+
+ if clean_640p.shape[0] < 1:
+ return torch.empty((0, 4), device=self.device_input)
+
+ clean_640p[:, 0].clamp_(min=0, max=self.resize_w - 1)
+ clean_640p[:, 1].clamp_(min=0, max=self.resize_h - 1)
+ clean_640p[:, 2].clamp_(min=0, max=self.resize_w - 1)
+ clean_640p[:, 3].clamp_(min=0, max=self.resize_h - 1)
+
+ # OPTIMIZATION: IN-PLACE RESCALING
+ # Avoid slicing arrays out sequentially into individual variables (xmin, ymin...)
+ # Construct and scale standard boxes directly to prevent memory allocation chattering.
+ standard_boxes = torch.stack(
+ [clean_640p[:, 0], clean_640p[:, 1], clean_640p[:, 2], clean_640p[:, 3]],
+ dim=1,
+ )
+
+ # if standard_boxes is None or standard_boxes.shape[0] == 0:
+ # return torch.empty((0, 4), device=self.device_input, dtype=torch.float32)
+
+ # Multiply inplace and avoid manual `del` garbage sweeps to maintain maximum throughput
+ bbs_full_res = (standard_boxes * self.scales_tensor).detach()
+ return bbs_full_res
+
+ # def get_sf_gpu_rois_v2(
+ # self, device_frame, overall_frame_num, max_candidates=100, limit_640=1280
+ # ):
+ # debug_frame_limit = self.debug_frame_limit
+
+ # if overall_frame_num <= debug_frame_limit:
+ # stage_debug_dir = (
+ # self.result_dir / "debug_stages" / self._testMethodName / "roi_stages"
+ # )
+ # stage_debug_dir.mkdir(parents=True, exist_ok=True)
+
+ # # 2. BRIDGE THE PYTORCH TO OPENCV VRAM GAP
+ # if torch.is_tensor(device_frame):
+ # h_raw, w_raw, ch = device_frame.shape
+ # cuda_mem_ptr = device_frame.data_ptr()
+ # cv_type = cv2.CV_8UC3 if ch == 3 else cv2.CV_8UC1
+ # row_step_bytes = device_frame.stride()[0] * device_frame.element_size()
+ # src_gpu_mat = cv2.cuda.createGpuMatFromCudaMemory(
+ # h_raw, w_raw, cv_type, cuda_mem_ptr, step=row_step_bytes
+ # )
+ # else:
+ # src_gpu_mat = device_frame
+ # ch = src_gpu_mat.channels()
+
+ # if overall_frame_num <= debug_frame_limit:
+ # src_cpu = src_gpu_mat.download()
+ # if ch == 3:
+ # src_cpu = cv2.cvtColor(src_cpu, cv2.COLOR_RGB2BGR)
+ # cv2.imwrite(
+ # str(stage_debug_dir / f"frame_{overall_frame_num:04d}_stage1_src.jpg"),
+ # src_cpu,
+ # )
+ # del src_cpu
+
+ # # 3. STRIDED LAYOUT INITIALIZATION
+ # # if not hasattr(self, "raw_mask"):
+ # # self.recycled_resize_mat = cv2.cuda.GpuMat(self.resize_h, self.resize_w, cv2.CV_8UC3 if ch == 3 else cv2.CV_8UC1)
+ # # self.raw_mask = cv2.cuda.GpuMat(self.resize_h, self.resize_w, cv2.CV_8UC1)
+ # # self.thresh_mask = cv2.cuda.GpuMat(self.resize_h, self.resize_w, cv2.CV_8UC1)
+ # # self.clean_mask = cv2.cuda.GpuMat(self.resize_h, self.resize_w, cv2.CV_8UC1)
+ # # self.d_blurred = cv2.cuda.GpuMat(self.clean_mask.size(), cv2.CV_8UC1)
+
+ # # 4. RUN ASYNCHRONOUS DOWN-SAMPLING GATE
+ # cv2.cuda.resize(
+ # src_gpu_mat,
+ # dst=self.recycled_resize_mat,
+ # dsize=(self.resize_w, self.resize_h),
+ # interpolation=cv2.INTER_NEAREST,
+ # stream=self.bgs_stream,
+ # )
+ # isolated_resize_mat = self.recycled_resize_mat.clone()
+
+ # if overall_frame_num <= debug_frame_limit:
+ # self.bgs_stream.waitForCompletion()
+ # resize_cpu = self.recycled_resize_mat.download()
+ # if ch == 3:
+ # resize_cpu = cv2.cvtColor(resize_cpu, cv2.COLOR_RGB2BGR)
+ # cv2.imwrite(
+ # str(
+ # stage_debug_dir / f"frame_{overall_frame_num:04d}_stage2_resize.jpg"
+ # ),
+ # resize_cpu,
+ # )
+
+ # # 5. BACKGROUND SUBTRACTION & FILTERING
+ # self.raw_mask = self.apply_background_subtraction_gpu(
+ # isolated_resize_mat, # self.recycled_resize_mat,
+ # include_history=self.config.BKGD_SUB_INCLUDE_HISTORY,
+ # method=self.config.BKGD_SUB_INCLUDE_HISTORY_METHOD,
+ # stream=self.bgs_stream,
+ # )
+
+ # if overall_frame_num <= debug_frame_limit:
+ # self.bgs_stream.waitForCompletion()
+ # cv2.imwrite(
+ # str(
+ # stage_debug_dir
+ # / f"frame_{overall_frame_num:04d}_stage5_history_mask.jpg"
+ # ),
+ # self.raw_mask.download(),
+ # )
+
+ # # self._cuda_gaussian_filter.apply(
+ # # self.raw_mask, self.d_blurred, stream=self.bgs_stream
+ # # )
+ # # cv2.cuda.threshold(
+ # # self.d_blurred,
+ # # 50,
+ # # self.config.THRESHOLD_MAX_VALUE,
+ # # cv2.THRESH_BINARY,
+ # # self.thresh_mask,
+ # # stream=self.bgs_stream,
+ # # )
+
+ # # self.dilate_filter.apply(self.thresh_mask, self.clean_mask, self.bgs_stream)
+ # # --- START: FUSED KERNEL EXECUTION ---
+ # # We bridge raw_mask and clean_mask to CuPy arrays (zero-copy)
+ # height, width = self.raw_mask.size()
+
+ # raw_mask_cp = cupy.ndarray(
+ # shape=(height, width),
+ # dtype=cupy.uint8,
+ # memptr=cupy.cuda.MemoryPointer(
+ # cupy.cuda.UnownedMemory(
+ # self.raw_mask.cudaPtr(), self.raw_mask.step * height, self.raw_mask
+ # ),
+ # 0,
+ # ),
+ # strides=(self.raw_mask.step, 1),
+ # )
+
+ # clean_mask_cp = cupy.ndarray(
+ # shape=(height, width),
+ # dtype=cupy.uint8,
+ # memptr=cupy.cuda.MemoryPointer(
+ # cupy.cuda.UnownedMemory(
+ # self.clean_mask.cudaPtr(),
+ # self.clean_mask.step * height,
+ # self.clean_mask,
+ # ),
+ # 0,
+ # ),
+ # strides=(self.clean_mask.step, 1),
+ # )
+
+ # # 3. Launch the Fused CUDA Kernel!
+ # # Grid/Block sizes optimized for RTX A6000 architecture
+ # block_dim = 16
+ # grid_size = (
+ # (width + block_dim - 1) // block_dim,
+ # (height + block_dim - 1) // block_dim,
+ # )
+ # block_size = (block_dim, block_dim)
+
+ # # --- START: FUSED KERNEL EXECUTION ---
+ # # Get the underlying CUDA stream pointer from the OpenCV stream
+ # # self.cupy_bgs_stream = cupy.cuda.ExternalStream(self.bgs_stream.cudaPtr())
+ # # self.fused_blur_thresh_kernel(
+ # # grid_size,
+ # # block_size,
+ # # (
+ # # self.raw_mask.cudaPtr(), # const unsigned char* src
+ # # self.thresh_mask.cudaPtr(), # unsigned char* dst
+ # # width, # int width
+ # # height, # int height
+ # # self.raw_mask.step, # int src_step
+ # # self.thresh_mask.step # int dst_step
+ # # ),
+ # # stream=self.cupy_bgs_stream
+ # # )
+ # # self.dilate_filter.apply(self.thresh_mask, self.clean_mask, self.bgs_stream)
+
+ # # --- END: FUSED KERNEL EXECUTION ---
+
+ # # --- START: 3-in-1 FUSED KERNEL EXECUTION ---
+
+ # # 2. Wait for OpenCV BGS stream to finish before CuPy takes over
+ # # self.bgs_stream.waitForCompletion()
+ # self.cupy_bgs_stream = cupy.cuda.ExternalStream(self.bgs_stream.cudaPtr())
+
+ # # 4. Launch the single fused kernel on CuPy's default stream
+ # self.fused_3_in_1_kernel(
+ # grid_size,
+ # block_size,
+ # (
+ # raw_mask_cp, # src
+ # clean_mask_cp, # dst
+ # width,
+ # height,
+ # self.raw_mask.step,
+ # self.clean_mask.step,
+ # ),
+ # stream=self.cupy_bgs_stream,
+ # )
+
+ # # 5. Ensure CuPy kernel is finished before proceeding
+ # # import cupy as cp
+ # # cupy.cuda.Stream.null.synchronize()
+ # # --- END: FUSED KERNEL EXECUTION ---
+
+ # # [STAGE 7 DEBUG] Check Final Dilated Output Mask
+ # # if overall_frame_num <= debug_frame_limit:
+ # # self.bgs_stream.waitForCompletion()
+ # # cv2.imwrite(
+ # # str(
+ # # stage_debug_dir
+ # # / f"frame_{overall_frame_num:04d}_stage8_final_dilatemask.jpg"
+ # # ),
+ # # self.clean_mask.download(),
+ # # )
+
+ # # inf_data = {
+ # # "mask": self.clean_mask,
+ # # "full_frame": device_frame,
+ # # "frameNum": overall_frame_num,
+ # # }
+ # # start = torch.cuda.Event(enable_timing=True)
+ # # end = torch.cuda.Event(enable_timing=True)
+
+ # # 6. REGION OF INTEREST ANALYSIS
+ # limit_8K = 1280
+ # limit_640 = limit_8K / self.scale_x
+ # # start.record()
+ # raw_boxes = self.get_gpu_rois_by_area(
+ # overall_frame_num, self.clean_mask, max_candidates=50, limit_640=limit_640
+ # )
+ # # end.record()
+ # # torch.cuda.synchronize()
+ # # gpu_rois_by_area_ms = start.elapsed_time(end)
+
+ # if raw_boxes.shape[0] < 1:
+ # return torch.empty((0, 4), device=self.device_input)
+
+ # raw_boxes = self.gpu_roi_padding(raw_boxes, method="uniform")
+ # # start.record()
+ # clean_640p = merge_boxes_gpu(
+ # raw_boxes,
+ # gap_limit=self.dist_thresh_640,
+ # size_limit=limit_640,
+ # max_cached_elements=self.max_cached_elements,
+ # )
+ # # end.record()
+ # # torch.cuda.synchronize()
+ # # merge_ms = start.elapsed_time(end)
+
+ # # main_app_logger.info(f"gpu_rois_by_area: {gpu_rois_by_area_ms:.3f} ms")
+ # # main_app_logger.info(f"merge_boxes_gpu: {merge_ms:.3f} ms")
+
+ # if clean_640p.shape[0] < 1:
+ # return torch.empty((0, 4), device=self.device_input)
+
+ # clean_640p[:, 0].clamp_(min=0, max=self.resize_w - 1)
+ # clean_640p[:, 1].clamp_(min=0, max=self.resize_h - 1)
+ # clean_640p[:, 2].clamp_(min=0, max=self.resize_w - 1)
+ # clean_640p[:, 3].clamp_(min=0, max=self.resize_h - 1)
+
+ # # OPTIMIZATION: IN-PLACE RESCALING
+ # # Avoid slicing arrays out sequentially into individual variables (xmin, ymin...)
+ # # Construct and scale standard boxes directly to prevent memory allocation chattering.
+ # standard_boxes = torch.stack(
+ # [clean_640p[:, 0], clean_640p[:, 1], clean_640p[:, 2], clean_640p[:, 3]],
+ # dim=1,
+ # )
+
+ # # if standard_boxes is None or standard_boxes.shape[0] == 0:
+ # # return torch.empty((0, 4), device=self.device_input, dtype=torch.float32)
+
+ # # Multiply inplace and avoid manual `del` garbage sweeps to maintain maximum throughput
+ # bbs_full_res = (standard_boxes * self.scales_tensor).detach()
+ # return bbs_full_res
+
+ # v3: reduce memory footprint
+ def get_sf_gpu_rois(
+ self, device_frame, overall_frame_num, max_candidates=100, limit_640=1280
+ ):
+ cupy.get_default_memory_pool().free_all_blocks()
+ height, width = self.raw_mask.size()
+
+ raw_mask_cp = cupy.ndarray(
+ shape=(height, width),
+ dtype=cupy.uint8,
+ memptr=cupy.cuda.MemoryPointer(
+ cupy.cuda.UnownedMemory(
+ self.raw_mask.cudaPtr(),
+ self.raw_mask.step * height,
+ self.raw_mask,
+ ),
+ 0,
+ ),
+ strides=(self.raw_mask.step, 1),
+ )
+
+ clean_mask_cp = cupy.ndarray(
+ shape=(height, width),
+ dtype=cupy.uint8,
+ memptr=cupy.cuda.MemoryPointer(
+ cupy.cuda.UnownedMemory(
+ self.clean_mask.cudaPtr(),
+ self.clean_mask.step * height,
+ self.clean_mask,
+ ),
+ 0,
+ ),
+ strides=(self.clean_mask.step, 1),
+ )
+ try:
+ debug_frame_limit = self.debug_frame_limit
+
+ if overall_frame_num <= debug_frame_limit:
+ stage_debug_dir = (
+ self.result_dir
+ / "debug_stages"
+ / self._testMethodName
+ / "roi_stages"
+ )
+ stage_debug_dir.mkdir(parents=True, exist_ok=True)
+
+ # 2. BRIDGE THE PYTORCH TO OPENCV VRAM GAP
+ if torch.is_tensor(device_frame):
+ h_raw, w_raw, ch = device_frame.shape
+ cuda_mem_ptr = device_frame.data_ptr()
+ cv_type = cv2.CV_8UC3 if ch == 3 else cv2.CV_8UC1
+ row_step_bytes = device_frame.stride()[0] * device_frame.element_size()
+ src_gpu_mat = cv2.cuda.createGpuMatFromCudaMemory(
+ h_raw, w_raw, cv_type, cuda_mem_ptr, step=row_step_bytes
+ )
+ else:
+ src_gpu_mat = device_frame
+ ch = src_gpu_mat.channels()
+
+ if overall_frame_num <= debug_frame_limit:
+ src_cpu = src_gpu_mat.download()
+ if ch == 3:
+ src_cpu = cv2.cvtColor(src_cpu, cv2.COLOR_RGB2BGR)
+ cv2.imwrite(
+ str(
+ stage_debug_dir
+ / f"frame_{overall_frame_num:04d}_stage1_src.jpg"
+ ),
+ src_cpu,
+ )
+ del src_cpu
+
+ # 1. Swap active buffers (Double Buffering)
+ # This gives us a 100% thread-safe copy without allocating a single byte of new VRAM!
+ # if self.use_buffer_A:
+ # active_resize_mat = self.recycled_resize_mat_A
+ # self.use_buffer_A = False
+ # else:
+ # active_resize_mat = self.recycled_resize_mat_B
+ # self.use_buffer_A = True
+
+ # 4. RUN ASYNCHRONOUS DOWN-SAMPLING GATE
+ cv2.cuda.resize(
+ src_gpu_mat,
+ dst=self.recycled_resize_mat, # active_resize_mat,
+ dsize=(self.resize_w, self.resize_h),
+ interpolation=cv2.INTER_NEAREST,
+ stream=self.bgs_stream,
+ )
+ # isolated_resize_mat = self.recycled_resize_mat.clone()
+
+ if overall_frame_num <= debug_frame_limit:
+ self.bgs_stream.waitForCompletion()
+ resize_cpu = self.recycled_resize_mat.download()
+ if ch == 3:
+ resize_cpu = cv2.cvtColor(resize_cpu, cv2.COLOR_RGB2BGR)
+ cv2.imwrite(
+ str(
+ stage_debug_dir
+ / f"frame_{overall_frame_num:04d}_stage2_resize.jpg"
+ ),
+ resize_cpu,
+ )
+ del resize_cpu
+
+ # 5. BACKGROUND SUBTRACTION & FILTERING
+ self.raw_mask = self.apply_background_subtraction_gpu(
+ self.recycled_resize_mat,
+ include_history=self.config.BKGD_SUB_INCLUDE_HISTORY,
+ method=self.config.BKGD_SUB_INCLUDE_HISTORY_METHOD,
+ stream=self.bgs_stream,
+ )
+
+ if overall_frame_num <= debug_frame_limit:
+ self.bgs_stream.waitForCompletion()
+ cv2.imwrite(
+ str(
+ stage_debug_dir
+ / f"frame_{overall_frame_num:04d}_stage5_history_mask.jpg"
+ ),
+ self.raw_mask.download(),
+ )
+
+ # --- START: FUSED KERNEL EXECUTION ---
+ # We bridge raw_mask and clean_mask to CuPy arrays (zero-copy)
+ # height, width = self.raw_mask.size()
+
+ # raw_mask_cp = cupy.ndarray(
+ # shape=(height, width),
+ # dtype=cupy.uint8,
+ # memptr=cupy.cuda.MemoryPointer(
+ # cupy.cuda.UnownedMemory(
+ # self.raw_mask.cudaPtr(),
+ # self.raw_mask.step * height,
+ # self.raw_mask,
+ # ),
+ # 0,
+ # ),
+ # strides=(self.raw_mask.step, 1),
+ # )
+
+ # clean_mask_cp = cupy.ndarray(
+ # shape=(height, width),
+ # dtype=cupy.uint8,
+ # memptr=cupy.cuda.MemoryPointer(
+ # cupy.cuda.UnownedMemory(
+ # self.clean_mask.cudaPtr(),
+ # self.clean_mask.step * height,
+ # self.clean_mask,
+ # ),
+ # 0,
+ # ),
+ # strides=(self.clean_mask.step, 1),
+ # )
+
+ # 3. Launch the Fused CUDA Kernel!
+ # Grid/Block sizes optimized for RTX A6000 architecture
+ block_dim = 16
+ grid_size = (
+ (width + block_dim - 1) // block_dim,
+ (height + block_dim - 1) // block_dim,
+ )
+ block_size = (block_dim, block_dim)
+
+ # 2. Wait for OpenCV BGS stream to finish before CuPy takes over
+ # self.bgs_stream.waitForCompletion()
+ self.cupy_bgs_stream = cupy.cuda.ExternalStream(self.bgs_stream.cudaPtr())
+
+ # 4. Launch the single fused kernel on CuPy's default stream
+ self.fused_3_in_1_kernel(
+ grid_size,
+ block_size,
+ (
+ raw_mask_cp, # src
+ clean_mask_cp, # dst
+ width,
+ height,
+ self.raw_mask.step,
+ self.clean_mask.step,
+ ),
+ stream=self.cupy_bgs_stream,
+ )
+
+ # 6. REGION OF INTEREST ANALYSIS
+ limit_8K = 1280
+ limit_640 = limit_8K / self.scale_x
+ # start.record()
+ raw_boxes = self.get_gpu_rois_by_area(
+ overall_frame_num,
+ self.clean_mask,
+ max_candidates=50,
+ limit_640=limit_640,
+ )
+ # end.record()
+ # torch.cuda.synchronize()
+ # gpu_rois_by_area_ms = start.elapsed_time(end)
+
+ if raw_boxes.shape[0] < 1:
+ return torch.empty((0, 4), device=self.device_input)
+
+ raw_boxes = self.gpu_roi_padding(raw_boxes, method="uniform")
+ # start.record()
+ clean_640p = merge_boxes_gpu(
+ raw_boxes,
+ gap_limit=self.dist_thresh_640,
+ size_limit=limit_640,
+ max_cached_elements=self.max_cached_elements,
+ )
+ # end.record()
+ # torch.cuda.synchronize()
+ # merge_ms = start.elapsed_time(end)
+
+ # main_app_logger.info(f"gpu_rois_by_area: {gpu_rois_by_area_ms:.3f} ms")
+ # main_app_logger.info(f"merge_boxes_gpu: {merge_ms:.3f} ms")
+
+ if clean_640p.shape[0] < 1:
+ return torch.empty((0, 4), device=self.device_input)
+
+ clean_640p[:, 0].clamp_(min=0, max=self.resize_w - 1)
+ clean_640p[:, 1].clamp_(min=0, max=self.resize_h - 1)
+ clean_640p[:, 2].clamp_(min=0, max=self.resize_w - 1)
+ clean_640p[:, 3].clamp_(min=0, max=self.resize_h - 1)
+
+ # OPTIMIZATION: IN-PLACE RESCALING
+ # Avoid slicing arrays out sequentially into individual variables (xmin, ymin...)
+ # Construct and scale standard boxes directly to prevent memory allocation chattering.
+ # standard_boxes = torch.stack(
+ # [
+ # clean_640p[:, 0],
+ # clean_640p[:, 1],
+ # clean_640p[:, 2],
+ # clean_640p[:, 3],
+ # ],
+ # dim=1,
+ # )
+ bbs_full_res = (
+ torch.stack(
+ [
+ clean_640p[:, 0],
+ clean_640p[:, 1],
+ clean_640p[:, 2],
+ clean_640p[:, 3],
+ ],
+ dim=1,
+ )
+ * self.scales_tensor
+ ).detach()
+
+ # if standard_boxes is None or standard_boxes.shape[0] == 0:
+ # return torch.empty((0, 4), device=self.device_input, dtype=torch.float32)
+
+ # Multiply inplace and avoid manual `del` garbage sweeps to maintain maximum throughput
+ # bbs_full_res = (standard_boxes * self.scales_tensor).detach()
+ finally:
+ # 2. Release temporary OpenCV GpuMats. Using 'in locals()' is safer.
+ if "raw_mask_cp" in locals():
+ del raw_mask_cp
+ if "clean_mask_cp" in locals():
+ del clean_mask_cp
+ if "src_gpu_mat" in locals():
+ del src_gpu_mat
+ # if "isolated_resize_mat" in locals():
+ # del isolated_resize_mat
+
+ del device_frame
+
+ if "raw_boxes" in locals():
+ del raw_boxes
+ if "clean_640p" in locals():
+ del clean_640p
+
+ # 3. Force CuPy to release all unused memory back to the system.
+ # This is the most critical step for preventing inter-library OOM errors.
+ cupy.get_default_memory_pool().free_all_blocks()
+ return bbs_full_res
+
+ def run_gpu_pipeline(
+ self, device_frame, overall_frame_num, frame_in_clip_count=0, gt_boxes=None
+ ):
+ metrics = {"sf_time": 0, "roi_time": 0, "det_time": 0, "bbs": None}
+ metadata = {}
+
+ # Get ROIs
+ if self.timer_enabled:
+ self.sf_start.record(self.inference_stream)
+
+ bgs_input_frame = (
+ device_frame.byte() if torch.is_tensor(device_frame) else device_frame
+ )
+ bbs_full_res = self.get_sf_gpu_rois(bgs_input_frame, overall_frame_num)
+
+ if self.timer_enabled:
+ self.sf_end.record(self.inference_stream)
+
+ # metrics["bbs"] = bbs_full_res
+ # Detach the matrix coordinates cleanly out of the tracking graph immediately
+ if bbs_full_res is not None:
+ if torch.is_tensor(bbs_full_res):
+ metrics["bbs"] = bbs_full_res.detach().cpu().numpy()
+ elif isinstance(bbs_full_res, np.ndarray):
+ metrics["bbs"] = bbs_full_res
+ else:
+ metrics["bbs"] = np.empty((0, 4), dtype=np.float32)
+
+ merged, det_frame = self.format_bbs_and_frame_4_detection(
+ bbs_full_res, device_frame
+ )
+ motion_detected = len(merged) > 0 if merged is not None else False
+ metrics["batch_density"] = len(merged) if merged is not None else 0
+
+ if (
+ self.config.DEBUG_FLAG
+ and overall_frame_num <= self.config.DEBUG_FRAME_LIMIT
+ ):
+ self.debug_save_mask(
+ det_frame, overall_frame_num, rois=merged, gt_boxes=gt_boxes
+ )
+
+ if not self.config.DISABLE_DETECTION:
+ # --- 3. MODEL INFERENCE TIMING BLOCK ---
+ if self.device_input == "cuda" and self.timer_enabled:
+ self.det_start.record(self.inference_stream)
+ # elif self.timer_enabled:
+ # t_start = time.perf_counter()
+
+ # Get Detection Metadata
+ if self.config.DETECTION_TYPE != "motion": # Object detection
+ metadata, _ = self.get_detections(
+ det_frame,
+ frame_in_clip_count,
+ merged=merged,
+ thickness=self.config.THICKNESS,
+ device_input=self.config.device_input,
+ )
+ # num_objs = len(metadata.keys())
+ else: # Motion: Smart filtering results
+ # 8k bb to 640
+ metadata = self.motion2metadata(merged, frame_in_clip_count)
+
+ if self.device_input == "cuda" and self.timer_enabled:
+ self.det_end.record(self.inference_stream)
+ # self.inference_stream.synchronize()
+
+ # Lock-free check: wait for event completion without blocking the CPU GIL
+ while not self.det_end.query():
+ # Yield GIL to let file-writers and thread-pools work
+ time.sleep(0.001)
+ # self.det_end.synchronize()
+
+ # Full-frame YOLO baseline tracks the elapsed time from t_start on page 20, line 1273
+ metrics["sf_time"] = self.sf_start.elapsed_time(self.sf_end)
+ metrics["det_time"] = self.det_start.elapsed_time(self.det_end)
+ elif self.device_input == "cuda" and self.timer_enabled:
+ metrics["sf_time"] = self.sf_start.elapsed_time(self.sf_end)
+
+ # elif self.timer_enabled:
+ # # CPU Path Execution: Must use standard wall-clock timing loops to avoid CUDA Event errors
+ # metrics["det_time"] = (time.perf_counter() - t_start) * 1000.0
+
+ del merged, bbs_full_res
+ return metrics, metadata, det_frame, motion_detected
+
+ # v1
+ def find_contours_gpu_equivalent(self, mask_gpu_mat, stream=None, limit_640=None):
+ """
+ Refactored allocation-free GPU bounding box contour extraction tree.
+ Bypasses dynamic VRAM creation paths entirely.
+ """
+
+ # start = torch.cuda.Event(enable_timing=True)
+ # end = torch.cuda.Event(enable_timing=True)
+ w, h = mask_gpu_mat.size()
+ ptr = mask_gpu_mat.cudaPtr()
+ pitch_bytes = mask_gpu_mat.step
+
+ mask_cp = cupy.ndarray(
+ (h, w),
+ dtype=cupy.uint8,
+ memptr=cupy.cuda.MemoryPointer(
+ cupy.cuda.UnownedMemory(ptr, pitch_bytes * h, mask_gpu_mat), 0
+ ),
+ strides=(pitch_bytes, 1),
+ )
+
+ stream_ptr = stream.cudaPtr() if stream else 0
+
+ with cupy.cuda.ExternalStream(stream_ptr):
+ if (
+ not hasattr(self, "_labeled_scratch")
+ or self._labeled_scratch.shape != (h, w)
+ or not self._labeled_scratch.flags.c_contiguous
+ ):
+ # Explicitly force order='C' (Row-Major Linear Stride alignment) at initialization.
+ # This completely eliminates internal syncdetect regular expression matching stutters!
+ self._labeled_scratch = cupy.empty((h, w), dtype=cupy.int32, order="C")
+
+ # start = cupy.cuda.Event()
+ # end = cupy.cuda.Event()
+ # start.record()
+ # 2. FAST IN-PLACE LABELING
+ # Because output is guaranteed to be a linear memory layout, SciPy processes it with zero device syncs
+ # TODO: Custom CUDA kernel to replace b/c label algorithm is costly
+ num_labels = cupyx.scipy.ndimage.label(
+ mask_cp, structure=self.cupy_structure, output=self._labeled_scratch
+ )
+ # end.record()
+ # end.synchronize()
+ # ms = cupy.cuda.get_elapsed_time(start, end)
+ # main_app_logger.info(f"cupyx label: {ms:.3f} ms")
+
+ if num_labels == 0:
+ return torch.empty((0, 4), device="cuda")
+
+ # Prevent array overflows if features spike abnormally
+ safe_labels = min(num_labels + 1, self._max_labels)
+
+ # Quick, specialized in-place fills over our persistent pools (0 loop churn)
+ self._x1_pool[:safe_labels].fill(w)
+ self._y1_pool[:safe_labels].fill(h)
+ self._x2_pool[:safe_labels].fill(-1)
+ self._y2_pool[:safe_labels].fill(-1)
+
+ # self._x1_pool[:safe_labels] = w
+ # self._y1_pool[:safe_labels] = h
+ # self._x2_pool[:safe_labels] = -1
+ # self._y2_pool[:safe_labels] = -1
+
+ # Extract references for your custom kernel execution map
+ pitch_elements = self._labeled_scratch.strides[0] // 4
+
+ tpb = (16, 16, 1)
+ bpg = ((w + tpb[0] - 1) // tpb[0], (h + tpb[1] - 1) // tpb[1], 1)
+
+ # 3. DIRECT DRIVER FUNCTION CALL (Maintains your accurate get_bounds logic)
+ self._raw_bounds_function(
+ bpg,
+ tpb,
+ (
+ self._labeled_scratch.data.ptr, # const int* labeled
+ pitch_elements, # int pitch
+ w, # int w
+ h, # int h
+ num_labels, # int num_labels
+ self._x1_pool[:safe_labels].data.ptr, # int* x1
+ self._y1_pool[:safe_labels].data.ptr, # int* y1
+ self._x2_pool[:safe_labels].data.ptr, # int* x2
+ self._y2_pool[:safe_labels].data.ptr, # int* y2
+ ),
+ )
+
+ # 4. ELIMINATE COLUMN_STACK ALLOCATION OVERHEAD (Replaces lines 91-105)
+ # Your previous code allocated memory on the heap every frame via cupy.column_stack.
+ # Instead, map the pre-allocated pool buffers directly into contiguous PyTorch tensor views!
+ x1_t = torch.as_tensor(self._x1_pool[1:safe_labels], device="cuda")
+ y1_t = torch.as_tensor(self._y1_pool[1:safe_labels], device="cuda")
+ x2_t = torch.as_tensor(self._x2_pool[1:safe_labels], device="cuda")
+ y2_t = torch.as_tensor(self._y2_pool[1:safe_labels], device="cuda")
+
+ # Stack tensors natively using unified memory views without copies or host syncs
+ return torch.stack((x1_t, y1_t, x2_t, y2_t), dim=1).float()
+
+ # def find_contours_gpu_equivalent_v3(self, clean_mask, stream=None, limit_640=1280):
+ # """
+ # Extremely fast, zero-allocation GPU boundary extraction.
+ # Bypasses OpenCV's costly contour allocations entirely.
+ # """
+ # # Convert the OpenCV GpuMat directly to a PyTorch GPU Tensor view (Zero-Copy)
+ # # This does not allocate any new VRAM!
+ # # mask_tensor = torch.as_tensor(
+ # # clean_mask,
+ # # device="cuda"
+ # # )
+ # height, width = clean_mask.size()
+
+ # # 1. Grab the raw C++ VRAM pointer from the OpenCV GpuMat
+ # # and wrap it in a CuPy array without copying the data
+ # mask_cp = cupy.ndarray(
+ # shape=(height, width),
+ # dtype=cupy.uint8,
+ # memptr=cupy.cuda.MemoryPointer(
+ # cupy.cuda.UnownedMemory(
+ # clean_mask.cudaPtr(),
+ # clean_mask.step * height,
+ # clean_mask
+ # ),
+ # 0
+ # ),
+ # strides=(clean_mask.step, 1)
+ # )
+ # mask_tensor = from_dlpack(mask_cp.toDlpack())
+
+ # # 1. Extract the coordinates of all active motion pixels (Non-zero coordinates)
+ # # We write directly into our pre-allocated coordinates scratchpad using a view
+ # nz = torch.nonzero(mask_tensor, out=self.coords_scratchpad)
+
+ # if nz.shape[0] == 0:
+ # return torch.empty((0, 4), dtype=torch.float32, device="cuda")
+
+ # # 2. Extract boundaries using vectorized Min/Max operations
+ # # This completely avoids creating temporary contour objects!
+ # y_coords = nz[:, 0]
+ # x_coords = nz[:, 1]
+
+ # # Find the global min/max coordinates of the motion region
+ # x1 = x_coords.min()
+ # y1 = y_coords.min()
+ # x2 = x_coords.max()
+ # y2 = y_coords.max()
+
+ # # 3. Write directly to our pre-allocated static output box
+ # self.static_boxes_out[0, 0] = x1
+ # self.static_boxes_out[0, 1] = y1
+ # self.static_boxes_out[0, 2] = x2
+ # self.static_boxes_out[0, 3] = y2
+
+ # # Return a view of the active boxes (Zero new allocations)
+ # return self.static_boxes_out[:1]
+
+ # def find_contours_gpu_equivalent_v2(self, mask_gpu_mat, stream=None, limit_640=None):
+ # """
+ # Replaces the CuPy version with a 100% PyTorch-native implementation.
+ # This completely eliminates CuPy-PyTorch context-switching overhead
+ # and utilizes parallel GPU reduction via scatter_reduce.
+ # """
+ # # We manually construct the __cuda_array_interface__ to share GPU memory pointers
+ # # without any host-device copies.
+
+ # # 1. Gather GpuMat metadata
+ # height, width = mask_gpu_mat.size()
+ # gpu_ptr = mask_gpu_mat.cudaPtr()
+ # step = mask_gpu_mat.step # Number of bytes per row (stride)
+
+ # # Instantly wrap the pointer into a PyTorch GPU Tensor (0 ms latency!)
+ # holder = GPUHolder({
+ # 'shape': (height, width),
+ # 'typestr': '|u1', # OpenCV's CV_8UC1 is equivalent to uint8
+ # 'data': (gpu_ptr, False), # Pointer address, Read-only=False
+ # 'strides': (step, 1),
+ # 'version': 3
+ # })
+ # mask_tensor = torch.as_tensor(holder, device=self.device_input)
+ # # =========================================================================
+
+ # # 2. Convert mask to boolean for segmentation
+ # binary_mask = mask_tensor > 0
+
+ # # 3. Use OpenCV's highly optimized CPU CC labeling (faster than CuPy for small labels/warmups)
+ # # Or if OpenCV's CUDA CC labeling is available, dispatch there.
+ # # This gives a fast, reliable label grid.
+ # num_labels, labels = cv2.connectedComponents(
+ # binary_mask.cpu().numpy().astype(np.uint8),
+ # connectivity=8
+ # )
+
+ # # If there are only background pixels, return an empty coordinate tensor
+ # if num_labels <= 1:
+ # return torch.empty((0, 4), device=self.device_input, dtype=torch.float32)
+
+ # # 4. Push label grid back to the active GPU
+ # labels_gpu = torch.as_tensor(labels, device=self.device_input)
+
+ # # 5. Pre-allocate coordinate bounds tensors on the GPU
+ # # We use standard 32-bit integers to keep memory operations light
+ # max_val = torch.iinfo(torch.int32).max
+ # x1 = torch.full((num_labels,), max_val, device=self.device_input, dtype=torch.int32)
+ # y1 = torch.full((num_labels,), max_val, device=self.device_input, dtype=torch.int32)
+ # x2 = torch.full((num_labels,), -1, device=self.device_input, dtype=torch.int32)
+ # y2 = torch.full((num_labels,), -1, device=self.device_input, dtype=torch.int32)
+
+ # # 6. Extract raw indices of all active foreground pixels on GPU
+ # # This keeps the coordinates completely in memory
+ # y_coords, x_coords = torch.nonzero(labels_gpu, as_tuple=True)
+ # label_vals = labels_gpu[y_coords, x_coords]
+
+ # # 7. Perform high-speed parallel GPU reductions to find min/max coordinates
+ # # 'amin' and 'amax' are computed concurrently across all threads
+ # x1.scatter_reduce_(0, label_vals, x_coords.int(), reduce="amin", include_self=False)
+ # y1.scatter_reduce_(0, label_vals, y_coords.int(), reduce="amin", include_self=False)
+ # x2.scatter_reduce_(0, label_vals, x_coords.int(), reduce="amax", include_self=False)
+ # y2.scatter_reduce_(0, label_vals, y_coords.int(), reduce="amax", include_self=False)
+
+ # # 8. Stack bounds and discard index 0 (which represents the background)
+ # # Returns shape: (N, 4) -> [x1, y1, x2, y2] matching your downstream merge_boxes format
+ # boxes = torch.stack((x1[1:], y1[1:], x2[1:], y2[1:]), dim=1).float()
+
+ # return boxes
+
+ # def find_contours_gpu_equivalent_v3(self, mask_gpu_mat, stream=None, limit_640=None):
+ # """
+ # 100% GPU-native connected components labeling.
+ # Combines CuPy labeling with PyTorch-native vectorized bounding box reduction.
+ # """
+ # # import cupy
+ # # import cupyx.scipy.ndimage
+ # # import torch
+
+ # # 1. Zero-copy bridge from OpenCV GpuMat to CuPy array on the GPU
+ # height, width = mask_gpu_mat.size()
+
+ # mask_cp = cupy.ndarray(
+ # shape=(height, width),
+ # dtype=cupy.uint8,
+ # memptr=cupy.cuda.MemoryPointer(
+ # cupy.cuda.UnownedMemory(mask_gpu_mat.cudaPtr(), mask_gpu_mat.step * height, mask_gpu_mat), 0
+ # ),
+ # strides=(mask_gpu_mat.step, 1)
+ # )
+
+ # # 2. Perform high-speed connected components labeling on GPU using CuPy
+ # structure = cupy.ones((3, 3), dtype=cupy.int32)
+ # labeled_mask, num_labels = cupyx.scipy.ndimage.label(mask_cp > 0, structure=structure)
+
+ # if num_labels == 0:
+ # return torch.empty((0, 4), device=self.device_input, dtype=torch.float32)
+
+ # # =========================================================================
+ # # 3. Zero-copy bridge from labeled CuPy array to PyTorch GPU Tensor
+ # # =========================================================================
+ # # Since labeled_mask is a CuPy array, we can access its __cuda_array_interface__
+ # # to expose the memory block directly to PyTorch without any copies!
+ # labels_gpu = torch.as_tensor(labeled_mask, device=self.device_input)
+
+ # # 4. Pre-allocate coordinate bounds tensors on the GPU
+ # max_val = torch.iinfo(torch.int32).max
+ # x1 = torch.full((num_labels + 1,), max_val, device=self.device_input, dtype=torch.int32)
+ # y1 = torch.full((num_labels + 1,), max_val, device=self.device_input, dtype=torch.int32)
+ # x2 = torch.full((num_labels + 1,), -1, device=self.device_input, dtype=torch.int32)
+ # y2 = torch.full((num_labels + 1,), -1, device=self.device_input, dtype=torch.int32)
+
+ # # 5. Extract raw indices of all active foreground pixels on the GPU
+ # y_coords, x_coords = torch.nonzero(labels_gpu, as_tuple=True)
+ # label_vals = labels_gpu[y_coords, x_coords].long() # Cast to long for indexing
+
+ # # 6. Parallel GPU reduction to find bounding coordinates for all labels at once
+ # x1.scatter_reduce_(0, label_vals, x_coords.int(), reduce="amin", include_self=False)
+ # y1.scatter_reduce_(0, label_vals, y_coords.int(), reduce="amin", include_self=False)
+ # x2.scatter_reduce_(0, label_vals, x_coords.int(), reduce="amax", include_self=False)
+ # y2.scatter_reduce_(0, label_vals, y_coords.int(), reduce="amax", include_self=False)
+
+ # # 7. Stack bounds, skipping index 0 (the background label)
+ # # Returns shape: (N, 4) -> [x1, y1, x2, y2]
+ # boxes = torch.stack((x1[1:], y1[1:], x2[1:], y2[1:]), dim=1).float()
+
+ # return boxes
+
+ def get_gpu_rois_by_area(self, frameNum, mask, max_candidates=100, limit_640=1280):
+ # Extract true spatial constraints straight from the active mask object footprint
+ if torch.is_tensor(mask):
+ mask_h, mask_w = mask.shape[-2:]
+ elif isinstance(mask, cv2.cuda.GpuMat):
+ # cv2.cuda.GpuMat.size() returns a tuple of (width, height) standard formatting
+ mask_w, mask_h = mask.size()
+ else:
+ mask_h, mask_w = mask.shape[:2]
+
+ # This prevents find_contours_gpu_equivalent from mutating the mask variables used by other threads.
+ if isinstance(mask, cv2.cuda.GpuMat):
+ # .clone() allocates a new C++ memory surface and forces full continuity
+ isolated_kernel_mask = mask # .clone()
+ elif torch.is_tensor(mask):
+ isolated_kernel_mask = mask # .clone().contiguous()
+ else:
+ isolated_kernel_mask = mask # .copy()
+
+ # Get raw boxes from mask (Direct VRAM bridge)
+ boxes_gpu = self.find_contours_gpu_equivalent(
+ isolated_kernel_mask,
+ stream=self.bgs_stream,
+ limit_640=limit_640, # 640*1.5,
+ )
+
+ # --- FIX: ELIMINATE STREAM RACE ---
+ if boxes_gpu is None or len(boxes_gpu) == 0:
+ return torch.empty((0, 4), device=self.device_input)
+
+ # if len(boxes_gpu) > max_candidates: # Adjust threshold based on target max objects
+ # # Prioritize or slice to prevent merge_boxes_gpu from thrashing scatter_reduce_
+ # boxes_gpu = boxes_gpu[:max_candidates]
+
+ # Wrap existing GPU memory as a float tensor (Zero Copy)
+ raw_boxes = torch.as_tensor(boxes_gpu, device=self.device_input).float()
+ # if self.device_input == "cuda":
+ # # Wrap the native device handle and IMMEDIATELY append .clone()
+ # # This allocates a brand new, physically isolated VRAM block to secure the bounding boxes
+ # raw_boxes = (
+ # torch.as_tensor(boxes_gpu, device=self.device_input).float().clone()
+ # )
+ # else:
+ # raw_boxes = torch.as_tensor(boxes_gpu, device=self.device_input).float()
+
+ if raw_boxes is not None and len(raw_boxes) > 0:
+ # Vectorized Pre-Filter (Removes noise blobs before merging)
+ w = raw_boxes[:, 2] - raw_boxes[:, 0]
+ h = raw_boxes[:, 3] - raw_boxes[:, 1]
+ # mask_filter = (w > 2) & (h > 2) & (w * h > self.min_contour_area) & (w < mask_w) & (h < mask_h)
+ mask_filter = (
+ (w <= self.max_roi_w)
+ & (h <= self.max_roi_h)
+ & (w >= self.min_roi_w)
+ & (h >= self.min_roi_h)
+ )
+ # Ensure both structures match dimensions along the indexing axis
+ # if raw_boxes.shape[0] == mask_filter.shape[0] and raw_boxes.shape[0] > 0:
+ # raw_boxes = raw_boxes[mask_filter]
+ # else:
+ # # Short-circuit directly to an empty tensor if shapes don't match or are 0
+ # raw_boxes = torch.empty((0, 4), device=self.device_input, dtype=torch.float32)
+ if raw_boxes.shape[0] > 0:
+ raw_boxes = raw_boxes[mask_filter]
+
+ # Guard rails: Clamp boundaries in-place to stay safely within the 8K master frame
+ # Indices 0 and 2 are X coordinates bounded by frame width; 1 and 3 are Y coordinates bounded by frame height
+ raw_boxes[:, 0].clamp_(min=0, max=self.resize_w - 1)
+ raw_boxes[:, 1].clamp_(min=0, max=self.resize_h - 1)
+ raw_boxes[:, 2].clamp_(min=0, max=self.resize_w - 1)
+ raw_boxes[:, 3].clamp_(min=0, max=self.resize_h - 1)
+
+ # Re-assign back to your pipeline's tracking variable
+ # raw_boxes = padded_tensor
+
+ # Prevents N^2 distance matrix from exploding during high noise
+ # if raw_boxes.shape[0] > max_candidates:
+ # # Prioritize the largest blobs (most likely to be drones)
+ # areas = (raw_boxes[:, 2] - raw_boxes[:, 0]) * (
+ # raw_boxes[:, 3] - raw_boxes[:, 1]
+ # )
+ # flat_areas = areas.view(-1)
+ # _, indices = torch.topk(
+ # flat_areas, k=min(max_candidates, flat_areas.shape[0]), dim=-1
+ # )
+ # # _, indices = torch.topk(areas, max_candidates)
+ # raw_boxes = raw_boxes[indices]
+ num_boxes = raw_boxes.shape[0]
+ if num_boxes > max_candidates:
+ areas = (raw_boxes[:, 2] - raw_boxes[:, 0]) * (
+ raw_boxes[:, 3] - raw_boxes[:, 1]
+ )
+ _, indices = torch.topk(areas, k=max_candidates, dim=0)
+ raw_boxes = raw_boxes[indices]
+ return raw_boxes
+
+ def gpu_roi_padding(self, raw_boxes, method="uniform"):
+ """
+ raw_boxes: in resize space (640)
+ """
+
+ if method == "uniform":
+ # pad using scale factor
+ widths = raw_boxes[:, 2] - raw_boxes[:, 0]
+ heights = raw_boxes[:, 3] - raw_boxes[:, 1]
+ # pad_w = widths * (PADDING_SCALE / 2.0)
+ # pad_h = heights * (PADDING_SCALE / 2.0)
+ # Add a minimum 3px clamp to ensure tiny objects don't lose their margins
+ pad_w = torch.clamp(widths * (PADDING_SCALE / 2.0), min=3.0)
+ pad_h = torch.clamp(heights * (PADDING_SCALE / 2.0), min=3.0)
+ padding_mask = torch.stack([-pad_w, -pad_h, pad_w, pad_h], dim=1)
+
+ # Apply padding to all bounding boxes concurrently via broad-vector math
+ raw_boxes = raw_boxes + padding_mask
+ elif method == "modelsize":
+ # extend box up to MODEL size
+ # if box is larger, keep as-is
+ scaley_8Kto640 = self.config.MODEL_H / self.frame_height
+ scalex_8Kto640 = self.config.MODEL_W / self.frame_width
+ model_w_640 = int(self.config.MODEL_W * scalex_8Kto640)
+ model_h_640 = int(self.config.MODEL_H * scaley_8Kto640)
+
+ widths = raw_boxes[:, 2] - raw_boxes[:, 0]
+ heights = raw_boxes[:, 3] - raw_boxes[:, 1]
+ # if (widths,heights) == (model_w_640,model_h_640) or widths > model_w_640 or heights > model_h_640:
+ # pass # Keep as is
+ # else:
+ x_centers = (raw_boxes[:, 0] + raw_boxes[:, 2]) / 2.0
+ y_centers = (raw_boxes[:, 1] + raw_boxes[:, 3]) / 2.0
+
+ # Determine target dimensions (clamp to target size if smaller)
+ new_widths = torch.clamp(widths, min=model_w_640)
+ new_heights = torch.clamp(heights, min=model_h_640)
+
+ # Calculate new coordinates from the centers
+ new_x1 = x_centers - (new_widths / 2.0)
+ new_y1 = y_centers - (new_heights / 2.0)
+ new_x2 = x_centers + (new_widths / 2.0)
+ new_y2 = y_centers + (new_heights / 2.0)
+
+ # 4. Stack into a tensor matching raw_boxes shape
+ raw_boxes = torch.stack([new_x1, new_y1, new_x2, new_y2], dim=1)
+ elif method == "pixel":
+ # pad using scale factor
+ # widths = raw_boxes[:, 2] - raw_boxes[:, 0]
+ # heights = raw_boxes[:, 3] - raw_boxes[:, 1]
+ # # pad_w = widths * (PADDING_SCALE / 2.0)
+ # # pad_h = heights * (PADDING_SCALE / 2.0)
+ # Add a minimum 3px clamp to ensure tiny objects don't lose their margins
+ # padding_mask = torch.stack([-PADDING_PX, -PADDING_PX, PADDING_PX, PADDING_PX], dim=1)
+
+ # Apply padding to all bounding boxes concurrently via broad-vector math
+ # raw_boxes = raw_boxes + padding_mask
+ # padding = 5 # self.config.ROI_BB_FULL_RES_PADDING / self.scale_x
+ padding_scale = 0.01
+ raw_boxes[:, 0] -= int(padding_scale * self.frame_width)
+ raw_boxes[:, 1] -= int(padding_scale * self.frame_height)
+ raw_boxes[:, 2] += int(padding_scale * self.frame_width)
+ raw_boxes[:, 3] += int(padding_scale * self.frame_height)
+
+ return raw_boxes
+
+ def filter_contained_boxes(self, boxes, overlap_thresh=0.9, max_elements=2000):
+ """
+ True Zero-Copy Anchor Filter. Eliminates torch.stack stream synchronization
+ stalls by utilizing a pre-allocated boolean tracking register map.
+ Supports both CPU and GPU execution profiles natively.
+ """
+ if boxes.shape[0] <= 1:
+ return boxes
+
+ # 1. Extract dimensions safely using vectorized layout views
+ x1, y1, x2, y2 = boxes[:, 0], boxes[:, 1], boxes[:, 2], boxes[:, 3]
+ w = (x2 - x1).clamp(min=0)
+ h = (y2 - y1).clamp(min=0)
+ areas = w * h
+
+ valid_mask = (
+ (w < self.max_roi_w)
+ & (h < self.max_roi_h)
+ & (w >= self.min_roi_w)
+ & (h >= self.min_roi_h)
+ )
+ boxes = boxes[valid_mask]
+ areas = areas[valid_mask]
+
+ N = boxes.shape[0]
+ if N <= 1:
+ return boxes
+
+ # 2. Sort structural metrics into our pre-allocated scratch index register
+ order = self._filter_order_scratch[:N]
+ torch.argsort(areas, stable=False, dim=0, descending=True, out=order)
+
+ # 🚀 REPLACEMENT: Use an in-place boolean tracking mask instead of a Python list
+ # to collect survival flags without triggering heap fragmentation or synchronization locks.
+ if (
+ not hasattr(self, "_filter_survival_mask")
+ or self._filter_survival_mask.shape[0] < max_elements
+ ):
+ self._filter_survival_mask = torch.zeros(
+ (max_elements,), dtype=torch.bool, device=boxes.device
+ )
+
+ survival_mask = self._filter_survival_mask[:N].zero_()
+
+ # 3. User-Space Evaluation Loop
+ while order.numel() > 0:
+ i = order[0]
+ survival_mask[i] = (
+ True # Mark this anchor index as verified on-device instantly
+ )
+
+ if order.numel() == 1:
+ break
+
+ order_tail = order[1:]
+ num_tail = order_tail.numel()
+
+ # Enforce static VRAM outputs using target 'out=' parameter parameters
+ torch.max(
+ boxes[i, 0],
+ boxes[order_tail, 0],
+ out=self._filter_scratch_x1[:num_tail],
+ )
+ torch.max(
+ boxes[i, 1],
+ boxes[order_tail, 1],
+ out=self._filter_scratch_y1[:num_tail],
+ )
+ torch.min(
+ boxes[i, 2],
+ boxes[order_tail, 2],
+ out=self._filter_scratch_x2[:num_tail],
+ )
+ torch.min(
+ boxes[i, 3],
+ boxes[order_tail, 3],
+ out=self._filter_scratch_y2[:num_tail],
+ )
+
+ inter_w = (
+ self._filter_scratch_x2[:num_tail] - self._filter_scratch_x1[:num_tail]
+ ).clamp_(min=0)
+ inter_h = (
+ self._filter_scratch_y2[:num_tail] - self._filter_scratch_y1[:num_tail]
+ ).clamp_(min=0)
+ inter_area = inter_w * inter_h
+
+ torch.div(
+ inter_area,
+ (areas[order_tail] + 1e-6),
+ out=self._filter_scratch_ioa[:num_tail],
+ )
+
+ mask = self._filter_scratch_keep[:num_tail]
+ torch.le(self._filter_scratch_ioa[:num_tail], overlap_thresh, out=mask)
+
+ # Advance tracking window slice cleanly to the filtered child nodes
+ order = order_tail[mask]
+
+ # 🚀 REPLACEMENT: Slice the original tensor layout with the boolean mask directly.
+ # This keeps the execution pipeline loopless and avoids cross-hardware synchronization stalls.
+ return boxes[survival_mask]
+
+ def cpu_roi_padding(self, coords_xywh, method="uniform"):
+ """
+ raw_boxes: in resize space (640)
+ """
+
+ if method == "uniform":
+ # pad using scale factor
+ widths = coords_xywh[:, 2]
+ heights = coords_xywh[:, 3]
+ # pad_w = widths * (PADDING_SCALE / 2.0)
+ # pad_h = heights * (PADDING_SCALE / 2.0)
+ # Add a minimum 3px clamp to ensure tiny objects don't lose their margins
+ pad_w = torch.clamp(widths * (PADDING_SCALE / 2.0), min=3.0)
+ pad_h = torch.clamp(heights * (PADDING_SCALE / 2.0), min=3.0)
+ # padding_mask = torch.stack([-pad_w, -pad_h, pad_w, pad_h], dim=1)
+
+ x1 = coords_xywh[:, 0] - pad_w
+ y1 = coords_xywh[:, 1] - pad_h
+ x2 = (coords_xywh[:, 0] + widths) + pad_w
+ y2 = (coords_xywh[:, 1] + heights) + pad_h
+ raw_boxes = torch.stack([x1, y1, x2, y2], dim=1)
+
+ elif method == "modelsize":
+ # extend box up to MODEL size
+ # if box is larger, keep as-is
+ scaley_8Kto640 = self.config.MODEL_H / self.frame_height
+ scalex_8Kto640 = self.config.MODEL_W / self.frame_width
+ model_w_640 = int(self.config.MODEL_W * scalex_8Kto640)
+ model_h_640 = int(self.config.MODEL_H * scaley_8Kto640)
+
+ widths = coords_xywh[:, 2]
+ heights = coords_xywh[:, 3]
+ # if (widths,heights) == (model_w_640,model_h_640) or widths > model_w_640 or heights > model_h_640:
+ # pass # Keep as is
+ # else:
+ x_centers = (coords_xywh[:, 0] + widths) / 2.0
+ y_centers = (coords_xywh[:, 1] + heights) / 2.0
+
+ # Determine target dimensions (clamp to target size if smaller)
+ new_widths = torch.clamp(widths, min=model_w_640)
+ new_heights = torch.clamp(heights, min=model_h_640)
+
+ # Calculate new coordinates from the centers
+ new_x1 = x_centers - (new_widths / 2.0)
+ new_y1 = y_centers - (new_heights / 2.0)
+ new_x2 = x_centers + (new_widths / 2.0)
+ new_y2 = y_centers + (new_heights / 2.0)
+
+ # 4. Stack into a tensor matching raw_boxes shape
+ raw_boxes = torch.stack([new_x1, new_y1, new_x2, new_y2], dim=1)
+
+ elif method == "pixel":
+ # pad using scale factor
+ # widths = raw_boxes[:, 2] - raw_boxes[:, 0]
+ # heights = raw_boxes[:, 3] - raw_boxes[:, 1]
+ # # pad_w = widths * (PADDING_SCALE / 2.0)
+ # # pad_h = heights * (PADDING_SCALE / 2.0)
+ # Add a minimum 3px clamp to ensure tiny objects don't lose their margins
+ # padding_mask = torch.stack([-PADDING_PX, -PADDING_PX, PADDING_PX, PADDING_PX], dim=1)
+
+ # Apply padding to all bounding boxes concurrently via broad-vector math
+ # raw_boxes = raw_boxes + padding_mask
+ # padding = 5 # self.config.ROI_BB_FULL_RES_PADDING / self.scale_x
+ padding_scale = 0.01
+ widths = coords_xywh[:, 2]
+ heights = coords_xywh[:, 3]
+ pad_w = int(padding_scale * self.frame_width)
+ pad_h = int(padding_scale * self.frame_height)
+ # raw_boxes[:, 2] += int(padding_scale * self.frame_width)
+ # raw_boxes[:, 3] += int(padding_scale * self.frame_height)
+
+ x1 = coords_xywh[:, 0] - pad_w
+ y1 = coords_xywh[:, 1] - pad_h
+ x2 = (coords_xywh[:, 0] + widths) + pad_w
+ y2 = (coords_xywh[:, 1] + heights) + pad_h
+ raw_boxes = torch.stack([x1, y1, x2, y2], dim=1)
+
+ # Clean vectorized boundary clamping on host CPU memory
+ raw_boxes[:, 0].clamp_(min=0, max=self.resize_w - 1)
+ raw_boxes[:, 1].clamp_(min=0, max=self.resize_h - 1)
+ raw_boxes[:, 2].clamp_(min=0, max=self.resize_w - 1)
+ raw_boxes[:, 3].clamp_(min=0, max=self.resize_h - 1)
+
+ return raw_boxes
+
+ # CPU ------------------------------------------------
+ def init_cpu_pipeline(self):
+ self.prepare_cpu_pipeline()
+
+ def allocate_cpu(self):
+ self.device_index = "cpu"
+ if not hasattr(
+ self, "_pinned_small_frame"
+ ): # or self._pinned_small_frame.shape[:2] != (self.resize_h, self.resize_w):
+ self._pinned_small_frame = np.zeros(
+ (self.resize_h, self.resize_w, 3), dtype=np.uint8
+ )
+ self._pinned_fg_mask = np.zeros(
+ (self.resize_h, self.resize_w), dtype=np.uint8
+ )
+ self._pinned_blurred_mask = np.zeros(
+ (self.resize_h, self.resize_w), dtype=np.uint8
+ )
+ self._pinned_threshold_mask = np.zeros(
+ (self.resize_h, self.resize_w), dtype=np.uint8
+ )
+ self._pinned_dilated_mask = np.zeros(
+ (self.resize_h, self.resize_w), dtype=np.uint8
+ )
+
+ # Existing order and boolean validation maps
+ self._filter_scratch_keep = torch.zeros(
+ (self.max_cached_elements,), dtype=torch.bool, device=self.device_input
+ )
+ self._filter_order_scratch = torch.zeros(
+ (self.max_cached_elements,), dtype=torch.long, device=self.device_input
+ )
+
+ # Persistent coordinate layers to absorb inner-loop tensor evaluations safely
+ self._filter_scratch_x1 = torch.zeros(
+ (self.max_cached_elements,), dtype=torch.float32, device=self.device_input
+ )
+ self._filter_scratch_y1 = torch.zeros(
+ (self.max_cached_elements,), dtype=torch.float32, device=self.device_input
+ )
+ self._filter_scratch_x2 = torch.zeros(
+ (self.max_cached_elements,), dtype=torch.float32, device=self.device_input
+ )
+ self._filter_scratch_y2 = torch.zeros(
+ (self.max_cached_elements,), dtype=torch.float32, device=self.device_input
+ )
+ self._filter_scratch_ioa = torch.zeros(
+ (self.max_cached_elements,), dtype=torch.float32, device=self.device_input
+ )
+
+ # pass
+ self.resized_frame = np.zeros((3, self.resize_h, self.resize_w), dtype="uint8")
+ # cv2.cuda.createContinuous(
+ # self.resize_h, self.resize_w, cv2.CV_8UC3
+ # )
+
+ self.fgMask = np.zeros(
+ (self.resize_h, self.resize_w), dtype="uint8"
+ ) # For resize
+
+ if self.config.BKGD_SUB_INCLUDE_HISTORY_METHOD == "and":
+ self.prev_bkgd = np.ones(
+ (self.resize_h, self.resize_w), dtype="uint8"
+ ) # * 255
+ else:
+ self.prev_bkgd = np.zeros((self.resize_h, self.resize_w), dtype="uint8")
+
+ # self.prev_bkgd = np.ones((self.resize_h, self.resize_w), dtype="uint8") * 255
+
+ self.mask_history = deque(
+ maxlen=self.config.BKGD_SUB_INCLUDE_HISTORY_TEMPORAL_SIZE
+ )
+ self.mask_history.append(self.prev_bkgd)
+
+ def prepare_cpu_pipeline(self): # , method="mog2"):
+ self.allocate_cpu()
+
+ # Subtraction
+ self.lr = self.config.BKGD_SUB_MOG2_LR
+ self.backSub = cv2.createBackgroundSubtractorMOG2(
+ history=self.config.BKGD_SUB_MOG2_HISTORY, # Clear ghosts of fast drones in ~2 seconds (2*fps)
+ varThreshold=self.config.BKGD_SUB_MOG2_VARTHRESHOLD, # High threshold to ignore "shimmer" and compression noise # default 16
+ detectShadows=self.config.BKGD_SUB_MOG2_DETECTSHADOWS, # default True
+ )
+ # else:
+ # raise ValueError(f"Provided method ({method}) is not available.")
+
+ def cleanup_cpu_v1(self):
+ """
+ Purges large 8K NumPy buffers and CPU-based AI resources.
+ """
+ self._pinned_small_frame = None
+ self._pinned_fg_mask = None
+ self._pinned_blurred_mask = None
+ self._pinned_threshold_mask = None
+ self._pinned_dilated_mask = None
+
+ # Nullify specific class references to allow Garbage Collection
+ self.executor = None
+ self.clip_executor = None
+ self.reader = None
+ self.latest_processed_frame = None
+
+ # Clear the Ping-Pong buffers (up to 200MB of RAM)
+ # if hasattr(self, "encode_buffers"):
+ # self.encode_buffers.clear()
+
+ # Explicitly nullify large arrays to trigger Garbage Collection
+ self.resized_frame = None
+ self.fgMask = None
+ self.prev_bkgd = None
+
+ # Clear the BGS history
+ if hasattr(self, "mask_history"):
+ self.mask_history.clear()
+
+ def apply_background_subtraction_cpu(
+ self, motion_input, include_history=True, method="and"
+ ):
+ raw_mask = self.backSub.apply(
+ motion_input, learningRate=self.config.BKGD_SUB_MOG2_LR
+ )
+
+ if include_history:
+ # self.prev_bkgd = np.zeros((self.resize_h, self.resize_w), dtype="uint8")
+
+ for m in list(self.mask_history):
+ # Dilate the historical mask on CPU
+ dilated = cv2.dilate(
+ m, self.dilate_kernel_for_enhanced_mask, iterations=1
+ )
+
+ if method == "or":
+ # Bitwise OR on CPU
+ cv2.bitwise_or(self.prev_bkgd, dilated, dst=self.prev_bkgd)
+ else:
+ # Bitwise AND on CPU
+ cv2.bitwise_and(self.prev_bkgd, dilated, dst=self.prev_bkgd)
+
+ self.mask_history.append(raw_mask) # .copy())
+
+ # if (
+ # self.prev_bkgd.max() != self.prev_bkgd.min()
+ # and self.prev_bkgd.max() > 0
+ # ):
+ # combined_mask_bool = (self.fgMask > 0) | (self.prev_bkgd > 0)
+ # self.fgMask = combined_mask_bool.astype(np.uint8) * 255
+ raw_mask = cv2.bitwise_or(raw_mask, self.prev_bkgd)
+ return raw_mask
+
+ def get_sf_cpu_rois(
+ self, device_frame, overall_frame_num, max_candidates=100, limit_640=1280
+ ):
+ # is_debug = self.config.DEBUG_FLAG
+ # test_mode = self.config.TEST_MODE
+ debug_frame_limit = (
+ self.debug_frame_limit
+ if self.config.DEBUG_FLAG and self.debug_frame_limit > -1
+ else -1
+ )
+
+ if overall_frame_num <= debug_frame_limit:
+ stage_debug_dir = (
+ self.result_dir / "debug_stages" / self._testMethodName / "roi_stages"
+ )
+ stage_debug_dir.mkdir(parents=True, exist_ok=True)
+ # f_num = overall_frame_num
+
+ if overall_frame_num <= debug_frame_limit:
+ cv2.imwrite(
+ str(stage_debug_dir / f"frame_{overall_frame_num:04d}_stage1_src.jpg"),
+ device_frame,
+ )
+
+ # target_w, target_h = self.resize_w, self.resize_h
+
+ # Force a zero-allocation resize into our pre-allocated, sequential cache-line matrix buffer
+ cv2.resize(
+ device_frame,
+ (self.resize_w, self.resize_h),
+ dst=self._pinned_small_frame,
+ interpolation=cv2.INTER_NEAREST,
+ )
+ # [STAGE 2 DEBUG] Check Downsampled Frame
+ if overall_frame_num <= debug_frame_limit:
+ cv2.imwrite(
+ str(
+ stage_debug_dir / f"frame_{overall_frame_num:04d}_stage2_resize.jpg"
+ ),
+ self._pinned_small_frame,
+ )
+
+ # --- PHASE 2: STRIPPED SINGLE-THREAD BACKGROUND ARITHMETIC ---
+ # Run background subtraction straight into our static memory address lane
+ # We pass your configured learning rate (BKGD_SUB_MOG2_LR) to lock the temporal parameters
+ # self._pinned_fg_mask = self.backSub.apply(
+ # self._pinned_small_frame,
+ # # dst=self._pinned_fg_mask,
+ # learningRate=self.config.BKGD_SUB_MOG2_LR,
+ # )
+
+ self._pinned_fg_mask = self.apply_background_subtraction_cpu(
+ self._pinned_small_frame,
+ include_history=self.config.BKGD_SUB_INCLUDE_HISTORY,
+ method=self.config.BKGD_SUB_INCLUDE_HISTORY_METHOD,
+ )
+
+ # [STAGE 5 DEBUG] Check Mask after History ORing/ANDing steps
+ if overall_frame_num <= debug_frame_limit:
+ cv2.imwrite(
+ str(
+ stage_debug_dir
+ / f"frame_{overall_frame_num:04d}_stage5_history_mask.jpg"
+ ),
+ self._pinned_fg_mask,
+ )
+
+ # d_blurred = cv2.cuda.GpuMat(raw_mask.size(), cv2.CV_8UC1)
+ # self._cuda_gaussian_filter.apply(raw_mask, d_blurred, stream=self.bgs_stream)
+ ksize = (17, 17)
+ self._pinned_blurred_mask = cv2.GaussianBlur(self._pinned_fg_mask, ksize, 0)
+
+ # [STAGE 6 DEBUG]
+ if overall_frame_num <= debug_frame_limit:
+ cv2.imwrite(
+ str(
+ stage_debug_dir / f"frame_{overall_frame_num:04d}_stage5b_blur.jpg"
+ ),
+ # str(stage_debug_dir / f"frame_{overall_frame_num:04d}_stage6_thresh_final_mask.jpg"),
+ self._pinned_blurred_mask,
+ )
+
+ # 7. MORPHOLOGICAL TRANSFORMATIONS & BINARY FILTERS
+ _, self._pinned_threshold_mask = cv2.threshold(
+ self._pinned_blurred_mask,
+ 50, # self.config.THRESHOLD_VALUE,
+ self.config.THRESHOLD_MAX_VALUE,
+ cv2.THRESH_BINARY,
+ )
+
+ # [STAGE 6 DEBUG] Check Binary Threshold Output
+ if overall_frame_num <= debug_frame_limit:
+ cv2.imwrite(
+ str(
+ stage_debug_dir
+ / f"frame_{overall_frame_num:04d}_stage6_threshold.jpg"
+ ),
+ # str(stage_debug_dir / f"frame_{overall_frame_num:04d}_stage6_thresh_final_mask.jpg"),
+ self._pinned_threshold_mask,
+ )
+
+ # Execute morphology dilation inline within our zero-allocation ring workspace [PDF: 0.1.18]
+ cv2.dilate(
+ self._pinned_threshold_mask,
+ self.dilate_kernel,
+ dst=self._pinned_dilated_mask,
+ iterations=1,
+ )
+ # [STAGE 7 DEBUG] Check Final Dilated Output Mask
+ if overall_frame_num <= debug_frame_limit:
+ cv2.imwrite(
+ str(
+ stage_debug_dir
+ / f"frame_{overall_frame_num:04d}_stage8_final_dilatemask.jpg"
+ ),
+ self._pinned_dilated_mask,
+ )
+
+ # inf_data = {"full_frame": frame, "mask": self._pinned_dilated_mask}
+
+ # REGION OF INTEREST ANALYSIS
+ # get cpu rois by area
+ contours, _ = cv2.findContours(
+ self._pinned_dilated_mask, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE
+ )
+ raw_boxes = [
+ list(cv2.boundingRect(c))
+ for c in contours
+ if cv2.contourArea(c) > self.min_contour_area
+ ]
+
+ if len(raw_boxes) < 1:
+ return torch.empty((0, 4), device=self.device_input)
+
+ if len(raw_boxes) > max_candidates:
+ raw_boxes.sort(key=lambda b: b[2] * b[3], reverse=True)
+ raw_boxes = raw_boxes[:max_candidates]
+
+ with self.compiled_no_grad_gate:
+ # coords_xywh = np.array(raw_boxes_xywh, dtype=np.float32)
+ coords_xywh = torch.tensor(
+ raw_boxes, dtype=torch.float32, device=self.device_input
+ )
+
+ raw_boxes = self.cpu_roi_padding(coords_xywh, method="uniform")
+
+ clean_640p = self.filter_contained_boxes(
+ raw_boxes, overlap_thresh=self.config.ROI_CONTAINMENT_THRESH
+ )
+
+ if clean_640p.shape[0] < 1:
+ # del raw_boxes_640p, clean_640p, raw_boxes_xywh # , raw_boxes
+ return torch.empty((0, 4), device=self.device_input)
+
+ # Scale to 8K space
+ # Sever the PyTorch graph graph entirely by detaching and casting to a standard numpy block
+ bbs_full_res = (clean_640p * self.scales_tensor).detach() # .clone()
+
+ # Explicitly clear the intermediate tensor tensors to drop heap allocations to 0
+ # del raw_boxes_640p, clean_640p, raw_boxes_xywh, scaled_output # , raw_boxes
+
+ return bbs_full_res
+
+ def run_cpu_pipeline(
+ self, device_frame, overall_frame_num, frame_in_clip_count=0, gt_boxes=None
+ ):
+ metrics = {"sf_time": 0, "roi_time": 0, "det_time": 0, "bbs": None}
+
+ # Get ROIs
+ if self.timer_enabled:
+ t_start = time.perf_counter()
+
+ bbs_full_res = self.get_sf_cpu_rois(device_frame, overall_frame_num)
+
+ if self.timer_enabled:
+ metrics["sf_time"] = (time.perf_counter() - t_start) * 1000.0
+
+ # metrics["bbs"] = bbs_full_res
+ if bbs_full_res is not None:
+ if torch.is_tensor(bbs_full_res):
+ metrics["bbs"] = bbs_full_res.detach().cpu().numpy()
+ elif isinstance(bbs_full_res, np.ndarray):
+ metrics["bbs"] = bbs_full_res
+ else:
+ metrics["bbs"] = np.empty((0, 4), dtype=np.float32)
+
+ merged, det_frame = self.format_bbs_and_frame_4_detection(
+ bbs_full_res, device_frame
+ )
+ motion_detected = len(merged) > 0 if merged is not None else False
+ metrics["batch_density"] = len(merged) if merged is not None else 0
+
+ if (
+ self.config.DEBUG_FLAG
+ and overall_frame_num <= self.config.DEBUG_FRAME_LIMIT
+ ):
+ self.debug_save_mask(
+ det_frame, overall_frame_num, rois=merged, gt_boxes=gt_boxes
+ )
+
+ if not self.config.DISABLE_DETECTION:
+ # --- 3. MODEL INFERENCE TIMING BLOCK ---
+ if self.timer_enabled:
+ t_start = time.perf_counter()
+
+ # Get Detection Metadata
+ if self.config.DETECTION_TYPE != "motion": # Object detection
+ metadata, _ = self.get_detections(
+ det_frame,
+ frame_in_clip_count,
+ merged=merged,
+ thickness=self.config.THICKNESS,
+ device_input=self.config.device_input,
+ )
+ # num_objs = len(metadata.keys())
+ else: # Motion: Smart filtering results
+ # 8k bb to 640
+ metadata = self.motion2metadata(merged, self.frame_count_target)
+
+ if self.timer_enabled:
+ metrics["det_time"] = (time.perf_counter() - t_start) * 1000.0
+
+ del merged, bbs_full_res
+ return metrics, metadata, det_frame, motion_detected
+
+ def cleanup_gpu(self):
+ """
+ Explicitly releases all GPU-allocated memory, persistent tensors,
+ and pinned cross-hardware streams to prevent VRAM leaks.
+ """
+ main_app_logger.info("[CLEANUP] Starting comprehensive GPU resource purge...")
+
+ # 1. Clear the persistent VRAM lock tensor to allow full layout collection
+ if hasattr(self, "_BaseObjectDetector__persistent_vram_lock"):
+ self._BaseObjectDetector__persistent_vram_lock = None
+
+ # 2. Iterate through and explicitly release all primitive OpenCV GpuMat layers
+ for attr_name in list(self.__dict__.keys()):
+ attr_value = getattr(self, attr_name)
+ if isinstance(attr_value, cv2.cuda.GpuMat):
+ attr_value.release()
+ setattr(self, attr_name, None)
+ main_app_logger.info(f" [x] Released GpuMat: {attr_name}")
+
+ # 3. Explicitly nullify persistent PyTorch Tensors on GPU/CPU
+ tensor_attributes = [
+ "fixed_inference_batch",
+ "gpu_float_staging",
+ "scales_tensor",
+ "pinned_cpu_xyxy",
+ "pinned_cpu_clss",
+ "pinned_cpu_confs",
+ "static_canvas_scratch",
+ "_labeled_scratch",
+ "_filter_survival_mask",
+ "_filter_scratch_keep",
+ "_filter_order_scratch",
+ "_filter_scratch_x1",
+ "_filter_scratch_y1",
+ "_filter_scratch_x2",
+ "_filter_scratch_y2",
+ "_filter_scratch_ioa",
+ ]
+ for attr in tensor_attributes:
+ if hasattr(self, attr):
+ setattr(self, attr, None)
+
+ # 4. Flush pre-allocated lists and buffer collections
+ if hasattr(self, "gpu_buffer_pool"):
+ for mat in self.gpu_buffer_pool:
+ if isinstance(mat, cv2.cuda.GpuMat):
+ mat.release()
+ self.gpu_buffer_pool = []
+
+ if hasattr(self, "frame_buffer_pool"):
+ self.frame_buffer_pool = []
+
+ if hasattr(self, "mask_history"):
+ self.mask_history.clear()
+
+ # 5. Nullify models and framework engines to drop context weights
+ self.model = None
+ self.cached_predictor = None
+ self.compiled_no_grad_gate = None
+ if hasattr(self, "_raw_bounds_function"):
+ self._raw_bounds_function = None
+
+ # 6. Final unified hardware fence synchronization and allocator sweep
+ if torch.cuda.is_available():
+ torch.cuda.synchronize()
+ torch.cuda.empty_cache()
+ main_app_logger.info(
+ "[CLEANUP] GPU resource tracking layers completely isolated."
+ )
+
+ def cleanup_cpu(self):
+ """
+ Purges large 8K NumPy buffers and host-side tracking structures.
+ """
+ main_app_logger.info("[CLEANUP] Purging host CPU variables...")
+
+ cpu_buffers = [
+ "_pinned_small_frame",
+ "_pinned_fg_mask",
+ "_pinned_blurred_mask",
+ "_pinned_threshold_mask",
+ "_pinned_dilated_mask",
+ "resized_frame",
+ "fgMask",
+ "prev_bkgd",
+ ]
+ for attr in cpu_buffers:
+ if hasattr(self, attr):
+ setattr(self, attr, None)
+
+ if hasattr(self, "mask_history"):
+ self.mask_history.clear()
+
+ # Clean background web workers and thread pool contexts safely
+ self.executor = None
+ self.clip_executor = None
+ self.reader = None
+ self.latest_processed_frame = None
+
+ # Explicitly run standard garbage collection to drop unreferenced heap layers
+ import gc
+
+ gc.collect()
+ main_app_logger.info("[CLEANUP] Host CPU workspace successfully cleared.")
diff --git a/fastapi/include/handlers.py b/fastapi/include/handlers.py
index 9188bcc..f95cd43 100644
--- a/fastapi/include/handlers.py
+++ b/fastapi/include/handlers.py
@@ -1,9 +1,13 @@
+# ==============================================================================
+# IMPORTS
import asyncio
import gc
+import inspect
import json
import logging
import multiprocessing as mp
import os
+import pickle
import queue
import shutil
import subprocess
@@ -11,19 +15,17 @@
import threading
import time
import traceback
-from collections import deque
+import types
from concurrent.futures import ThreadPoolExecutor
from contextlib import asynccontextmanager
from datetime import datetime
-from multiprocessing import shared_memory
from pathlib import Path
-import cupy
-import cupyx.scipy
-import cupyx.scipy.ndimage
import cv2
import numpy as np
+import psutil
import torch
+import torch.cuda._memory_viz as memory_viz
import torch.nn.functional as F
from ultralytics import YOLO
from ultralytics.utils.checks import check_imgsz
@@ -32,34 +34,73 @@
sys.path.insert(1, str(Path(__file__).parent.parent))
from include.default_configs import ENABLE_QUERYING_DEFAULT
-from include.models import get_model
+from include.detectors import GeneralObjectDetector, SmartFilteringObjectDetector
from include.utils import (
- BOUNDS_KERNEL,
- DETECTION_ACCEL_KERNEL,
+ AsyncDisplayVideoWriter,
+ AsyncVideoWriter,
+ DummyProcess,
+ PipelineConfig,
VDMSPool,
- find_contours_gpu_equivalent,
- merge_boxes_cpu,
- merge_boxes_gpu,
+ # safe_unregister_shm,
+ analyze_tracemalloc_snapshot,
+ default_attr_keys,
+ get_bb_overlay,
+ get_metadata_overlay,
+ # find_contours_gpu_equivalent,
+ global_frame_prefetch_worker_v1,
+ metadata2vdms_with_retry,
+ release_native_linux_heap,
+ release_shared_memory,
)
-# ----- SETUP LOGGING -----
+# Force OpenCV to run sequentially to prevent context-switching overhead
+cv2.setNumThreads(0) # Forces OpenCV loops to run strictly sequentially
+
+# ==============================================================================
+# LOGGING
logging.basicConfig(
level=logging.INFO,
- format="%(asctime)s - %(name)s - %(levelname)s - %(message)s",
+ # format="%(asctime)s - %(name)s - %(levelname)s - %(message)s",
+ format="%(asctime)s [%(levelname)s] %(name)s (%(filename)s:%(lineno)d) - %(message)s",
handlers=[logging.StreamHandler(sys.stdout)],
)
# Suppress low-delay reference block warnings from OpenCV/PyAV/FFmpeg
-os.environ["OPENCV_FFMPEG_LOGLEVEL"] = "-8"
-os.environ["OPENCV_LOG_LEVEL"] = "OFF"
+# os.environ["OPENCV_FFMPEG_LOGLEVEL"] = "-8"
+# os.environ["OPENCV_LOG_LEVEL"] = "OFF"
logging.getLogger("libav").setLevel(logging.CRITICAL)
logging.getLogger("libav.hevc").setLevel(logging.CRITICAL)
-
main_app_logger = logging.getLogger(__name__)
+
+# ==============================================================================
+# FUNCTIONS
+
STREAM_ARG = False
+# PADDING_PX = 5 #25
+PADDING_SCALE = 0.5 # 0.2 #3 #0.05 # 0.045 # 0.05
+BASE_PIPELINE_CONFIG = PipelineConfig(
+ SHARED_MODEL=os.getenv("SHARED_MODEL", False),
+ ENABLE_QUERYING=os.getenv("ENABLE_QUERYING", ENABLE_QUERYING_DEFAULT),
+)
+os.environ["OPENCV_FFMPEG_CAPTURE_OPTIONS"] = (
+ "rtsp_transport;tcp;hwaccel;cuda;threads;auto;low_delay;1;probesize;5000000"
+)
+
+ffmpeg_cores = "4,5,6,7" # Manually define cores
+
+# ----- GLOBAL VARIABLES -----
+all_metadata = {}
+
+# Queue for metadata being sent to vdms
+send_metadata_queue = queue.Queue()
+
+# Tracks if both components are finished:
+# {"clip_name": {"video": bool, "meta": bool}}
+clip_completion_tracker = {}
def log_to_logger(message, level="info"):
+ """Safely logs a message to the main application logger."""
try:
if level.lower() == "debug":
main_app_logger.debug(message)
@@ -71,64 +112,143 @@ def log_to_logger(message, level="info"):
pass
-# ----- PIPELINE CONFIGURATION -----
-os.environ["PYTORCH_ALLOC_CONF"] = "expandable_segments:True"
-# Force OpenCV to use a single thread for its operations.
-# This prevents internal OpenCV threads from "racing" against AI logic.
-# cv2.setNumThreads(1)
-
-# Force OpenCV to run sequentially to prevent context-switching overhead
-# cv2.setNumThreads(0)
-cv2.setNumThreads(os.cpu_count() or 4)
-
-
-from include.utils import (
- PipelineConfig,
- PipelineMapping,
- draw_label,
- get_detection_color,
- get_display_frame_in_bytes,
- metadata2vdms_with_retry,
- tensor2opencv,
-)
+from multiprocessing.shared_memory import SharedMemory
-BASE_PIPELINE_CONFIG = PipelineConfig(
- SHARED_MODEL=os.getenv("SHARED_MODEL", False),
- ENABLE_QUERYING=os.getenv("ENABLE_QUERYING", ENABLE_QUERYING_DEFAULT),
-)
+def video_writer_core_loop(
+ write_queue,
+ segment_pattern,
+ fps,
+ width,
+ height,
+ clip_duration,
+ shm_names,
+ frame_bytes,
+):
+ """
+ Original core loop, now running in an isolated process.
+ It manages its own FFmpeg pipe and reads frames from the multiprocessing queue.
+ """
+ # 1. Move your existing FFmpeg command creation here
+ # (Adapt these arguments if your original segment command differed slightly)
+ # command = [
+ # 'ffmpeg', '-y',
+ # '-f', 'rawvideo', '-vcodec', 'rawvideo',
+ # '-s', f'{width}x{height}', '-pix_fmt', 'bgr24', '-r', str(fps),
+ # '-i', '-',
+ # '-c:v', 'libx264', '-preset', 'ultrafast', '-crf', '28',
+ # '-f', 'segment', '-segment_time', '60', '-reset_timestamps', '1',
+ # segment_pattern
+ # ]
+ # clip_duration = int(self.config.CLIP_DURATION)
+ # Attach to the existing shared memory blocks
+ shm_blocks = [SharedMemory(name=name) for name in shm_names]
+
+ frame_buffers = [
+ np.ndarray((height, width, 3), dtype=np.uint8, buffer=shm.buf)
+ for shm in shm_blocks
+ ]
+ # Example: Get the CPU core count and reserve the last few for FFmpeg
+ # core_count = os.cpu_count()
+ # ffmpeg_cores = "4,5,6,7" # Manually define cores
+ command = [
+ "taskset",
+ "-c",
+ ffmpeg_cores,
+ "ffmpeg",
+ "-y",
+ "-f",
+ "rawvideo",
+ "-pix_fmt",
+ "bgr24",
+ "-s",
+ f"{width}x{height}",
+ "-r",
+ str(fps),
+ "-i",
+ "-",
+ "-c:v",
+ "libx264", # "mpeg4", # Or "libx264" if you prefer H.264
+ "-crf",
+ "23",
+ "-f",
+ "mpegts",
+ "-movflags",
+ "faststart",
+ "-force_key_frames",
+ f"expr:gte(t,n_forced*{clip_duration})",
+ "-f",
+ "segment",
+ "-segment_time",
+ f"{clip_duration}",
+ "-reset_timestamps",
+ "1",
+ "-segment_format",
+ "mp4",
+ # CRITICAL: Force fragmented headers so every chunk has a valid moov atom instantly
+ "-segment_format_options",
+ "movflags=frag_keyframe+empty_moov+default_base_moof",
+ segment_pattern,
+ ]
-# Optimizes RTSP ingestion with hardware acceleration and low-delay flags
-os.environ["OPENCV_FFMPEG_CAPTURE_OPTIONS"] = (
- "rtsp_transport;tcp|hwaccel;cuda|threads;auto|low_delay;1|probesize;5000000"
- # "rtsp_transport;tcp|hwaccel;cuda|threads;1|probesize;32|analyzeduration;0"
- # "rtsp_transport;tcp|hwaccel;cuda|threads;2|probesize;32|analyzeduration;0"
- # "rtsp_transport;tcp|hwaccel;cuda|threads;2|probesize;32|analyzeduration;0"
- # "rtsp_transport;tcp|hwaccel;cuda|threads;4|probesize;5000000|analyzeduration;5000000"
- # "rtsp_transport;udp|hwaccel;cuda|threads;8|stimeout;5000000|listen_timeout;5000"
-)
+ # 2. Start the FFmpeg pipe INSIDE the process
+ video_writer = subprocess.Popen(
+ command, stdin=subprocess.PIPE, stderr=subprocess.DEVNULL
+ )
+ try:
+ # while True:
+ # # 3. Read the frame array directly from the queue
+ # frame = write_queue.get()
+
+ # # Poison pill to stop the process cleanly
+ # if frame is None:
+ # break
+
+ # # =================================================================
+ # # 🛠️ SAFE BYTE EXTRACTION
+ # # =================================================================
+ # if torch.is_tensor(frame):
+ # # Move to CPU, convert to NumPy, and get bytes
+ # raw_bytes = frame.detach().cpu().numpy().tobytes()
+ # elif isinstance(frame, np.ndarray):
+ # raw_bytes = frame.tobytes()
+ # else:
+ # raw_bytes = bytes(frame) # Fallback for raw byte strings
+ # # =================================================================
-# ----- GLOBAL VARIABLES -----
-# ENABLE_QUERYING = os.getenv("ENABLE_QUERYING", ENABLE_QUERYING_DEFAULT)
-# if BASE_PIPELINE_CONFIG.ENABLE_QUERYING:
-# Tracks all metadata
-all_metadata = {}
+ # # Write the raw bytes to the FFmpeg pipe
+ # video_writer.stdin.write(raw_bytes)
-# Tracks clip_filename once video re-encoded
-# video_ready_list = {}
+ # # Help garbage collection in the isolated process
+ # del frame, raw_bytes
+ while True:
+ # Get the SLOT INDEX from the queue (not the full frame)
+ slot_idx = write_queue.get()
+ if slot_idx is None:
+ break
-# Queue for metadata being sent to vdms
-send_metadata_queue = queue.Queue()
+ # 3. Access the frame directly from the shared memory buffer
+ frame = frame_buffers[slot_idx]
+ video_writer.stdin.write(frame.tobytes())
-# Tracks if both components are finished:
-# {"clip_name": {"video": bool, "meta": bool}}
-clip_completion_tracker = {}
+ except Exception as e:
+ logging.error(f"[VIDEO WRITER] Loop exception: {e}")
+ finally:
+ video_writer.stdin.close()
+ video_writer.wait()
+ # Clean up process-local attachments
+ for shm in shm_blocks:
+ shm.close()
# ----- FASTAPI APPLICATION STARTUP/SHUTDOWN -----
# The lifespan parameter handles startup and shutdown
async def auto_cleanup_janitor(app):
+ """
+ A background task that runs periodically to find and clean up stale or inactive streams.
+ This acts as a safety net to prevent resource leaks from streams that do not terminate correctly.
+ """
while True:
await asyncio.sleep(10)
now = time.time()
@@ -139,59 +259,92 @@ async def auto_cleanup_janitor(app):
for name, streamer in list(app.state.active_streams.items()):
# streamer = app.state.active_streams.get(name)
if not streamer:
+ app.state.active_streams.pop(name, None)
continue
- ai_backlog = streamer.get_executor_backlog()
- video_backlog = (
- streamer.write_queue.qsize()
- if streamer.config.ENABLE_QUERYING
- else 0
- )
- io_backlog = (
- streamer.io_executor._work_queue.qsize()
- if hasattr(streamer, "io_executor")
- else 0
- )
-
- # Check if the stream is marked inactive OR timed out
- # streamer.active should be False when the video source ends
- is_stale = now - streamer.last_heartbeat > 30
-
- should_remove = False
-
- if not streamer.active and (
- ai_backlog == 0 and video_backlog == 0 and io_backlog == 0
- ):
- should_remove = True # Video ended naturally
- elif is_stale and (
- ai_backlog == 0 and video_backlog == 0 and io_backlog == 0
- ):
- should_remove = True # Browser tab closed/Network lost
- elif now - streamer.last_heartbeat > 90:
- should_remove = True # Hard timeout for hung processes
-
- if should_remove:
- async with app.state.stream_lock:
- if BASE_PIPELINE_CONFIG.DEBUG == "1":
- print(f"CLEANUP: Removing {name} from active_streams")
- streamer.stop()
- app.state.active_streams.pop(name, None)
-
- # --- Synchronization Data Purge ---
- # if BASE_PIPELINE_CONFIG.ENABLE_QUERYING:
- # # Remove trackers older than 5 minutes (300s)
- # stale_keys = [
- # k
- # for k, v in clip_completion_tracker.items()
- # if (now - v.get("start", now)) > 300
- # ]
- # for k in stale_keys:
- # clip_completion_tracker.pop(k, None)
- # all_metadata.pop(k, None)
+ # If the stream handler marked itself inactive or stopped
+ # if getattr(streamer, "_is_stopped", False) or not getattr(streamer, "active", True):
+ # main_app_logger.info(f"JANITOR: Removing dead stream {name}")
+ # app.state.active_streams.pop(name, None)
+ # if not streamer._is_stopped:
+ # loop = asyncio.get_event_loop()
+ # loop.run_in_executor(None, streamer.stop)
+ # continue
+
+ if hasattr(
+ streamer, "last_heartbeat"
+ ): # and (getattr(streamer, "active", False) == "RUNNING"):
+ # # Inactivity heartbeat check (e.g., 5 seconds with no new frames)
+ # if now - getattr(streamer, "last_heartbeat", now) > 15.0:
+ # main_app_logger.info(f"JANITOR: Stream {name} timed out. Evicting.")
+ # app.state.active_streams.pop(name, None)
+ # if not streamer._is_stopped:
+ # loop = asyncio.get_event_loop()
+ # loop.run_in_executor(None, streamer.stop)
+ # continue
+
+ ai_backlog = streamer.get_executor_backlog()
+ video_backlog = (
+ streamer.write_queue.qsize()
+ if streamer.config.ENABLE_QUERYING
+ else 0
+ )
+ # io_backlog = (
+ # streamer.io_executor._work_queue.qsize()
+ # if hasattr(streamer, "io_executor")
+ # else 0
+ # )
+
+ # Check if the stream is marked inactive OR timed out
+ # streamer.active should be False when the video source ends
+ is_stale = now - streamer.last_heartbeat > 30 # No activity for 30s
+ is_hung = (
+ now - streamer.last_heartbeat > 90
+ ) # Hard timeout after 90s
+
+ should_remove = False
+ reason = ""
+
+ # Conditions to remove stream
+ if streamer._is_stopped:
+ should_remove, reason = True, "Handler already stopped"
+ elif not streamer.active and (
+ ai_backlog == 0 and video_backlog == 0 # and io_backlog == 0
+ ):
+ should_remove, reason = True, "Video ended naturally"
+ elif is_stale and (
+ ai_backlog == 0 and video_backlog == 0 # io_backlog == 0
+ ):
+ should_remove, reason = True, "Browser tab closed/Network lost"
+ elif is_hung:
+ should_remove, reason = True, "Hard timeout for hung processes"
+
+ if should_remove:
+ async with app.state.stream_lock:
+ if BASE_PIPELINE_CONFIG.DEBUG == "1":
+ main_app_logger.info(
+ f"CLEANUP: Removing {name} from active_streams: {reason}"
+ )
+ try:
+ if not streamer._is_stopped:
+ streamer.stop()
+ streamer.stop_threads(["process_thread"])
+ app.state.active_streams.pop(name, None)
+ finally:
+ del streamer
+
+ gc.collect() # Final garbage collection
+ if torch.cuda.is_available():
+ torch.cuda.empty_cache()
@asynccontextmanager
async def lifespan(app: FastAPI):
+ """
+ Manages FastAPI startup and shutdown lifecycle.
+ Guarantees release on shutdown.
+ """
+
# --- STARTUP ---
if not hasattr(app.state, "classes"):
app.state.classes = None
@@ -201,27 +354,31 @@ async def lifespan(app: FastAPI):
app.state.status = "Ready"
app.state.stream_lock = asyncio.Lock()
+
+ # Warmup shared model (if configured)
if BASE_PIPELINE_CONFIG.SHARED_MODEL:
app.state.model = YOLO(
BASE_PIPELINE_CONFIG.model_path, verbose=False, task="detect"
)
device_input = "cuda" if BASE_PIPELINE_CONFIG.DEVICE == "GPU" else "cpu"
- print("Starting shared model warmup...")
- dummy_input = torch.zeros(
- (1, 3, BASE_PIPELINE_CONFIG.MODEL_H, BASE_PIPELINE_CONFIG.MODEL_W)
- ).to(device_input)
- for _ in range(20):
- _ = app.state.model(dummy_input, verbose=False)
-
- del dummy_input
- torch.cuda.empty_cache()
- print("Shared model warmup and VRAM purge complete.")
+ main_app_logger.info("Starting shared model warmup...")
+ try:
+ dummy_input = torch.zeros(
+ (1, 3, BASE_PIPELINE_CONFIG.MODEL_H, BASE_PIPELINE_CONFIG.MODEL_W)
+ ).to(device_input)
+ for _ in range(5):
+ _ = app.state.model(dummy_input, verbose=False)
+ finally:
+ del dummy_input
+ if "cuda" in device_input:
+ torch.cuda.empty_cache()
+ main_app_logger.info("Shared model warmup and VRAM purge complete.")
janitor_task = asyncio.create_task(auto_cleanup_janitor(app))
if BASE_PIPELINE_CONFIG.DEBUG == "1":
- print(f"--- APP STARTUP | PID: {os.getpid()} | STATE READY ---")
+ main_app_logger.info(f"--- APP STARTUP | PID: {os.getpid()} | STATE READY ---")
yield
@@ -229,85 +386,24 @@ async def lifespan(app: FastAPI):
janitor_task.cancel()
async with app.state.stream_lock:
for name, streamer in list(app.state.active_streams.items()):
- print(f"Shutting down stream: {name}")
+ main_app_logger.info(f"Shutting down stream: {name}")
streamer.stop() # Custom stop method defined below
+ streamer.stop_threads(["process_thread"])
app.state.active_streams.pop(name, None)
- app.state.status = "Stopped"
-
-
-# ----- INGESTION FUNCTIONS -----
-def nv12_to_rgb_torch(
- nv12_tensor, h, w, is_h264_8k=False, out_buffer=None, is_bgr=True
-):
- """
- Highly optimized NV12/YUV to RGB/BGR conversion.
- Fixes the 'Blue Frame' by applying proper YUV-to-RGB matrix math.
- """
- with torch.no_grad():
- if is_h264_8k:
- # 8K Path: Expects planar data [C, H, W]
- # y: [1, H, W], uv: [1, H, W] (already resized or needs upsampling)
- # y = nv12_tensor[0:1, :, :].half()
- # # If your 8K source has separate U and V, adjust indices [1:2] and [2:3]
- # u = nv12_tensor[1:2, :, :].half()
- # v = nv12_tensor[2:3, :, :].half() # Fallback if interleaved
- # 8K Planar Path: Map the channels directly
- y = nv12_tensor[0:1, :, :].half()
- u = nv12_tensor[1:2, :, :].half()
- v = nv12_tensor[2:3, :, :].half() # Use the 3rd channel!
-
- # No column slicing (0::2) needed if it's already planar.
- # But we must ensure U and V match Y dimensions if they were subsampled.
- if u.shape[-1] != w or u.shape[-2] != h:
- u = F.interpolate(u.unsqueeze(0), size=(h, w), mode="bilinear").squeeze(
- 0
- )
- v = F.interpolate(v.unsqueeze(0), size=(h, w), mode="bilinear").squeeze(
- 0
- )
- else:
- # Standard NV12 Path: Image is [H*1.5, W]
- y = nv12_tensor[:h, :w].unsqueeze(0).half()
- uv = (
- nv12_tensor[h:, :w]
- .reshape(h // 2, w // 2, 2)
- .permute(2, 0, 1)
- .unsqueeze(0)
- .half()
- )
- # Upsample Chroma (4:2:0 -> 4:4:4)
- uv_up = F.interpolate(uv, size=(h, w), mode="nearest")
- u = uv_up[0, 0:1, :, :]
- v = uv_up[0, 1:2, :, :]
-
- # --- YUV to RGB Conversion Math (BT.709) ---
- # 1. Normalize Luma and center Chroma
- y = (y - 16.0) * 1.164
- u = u - 128.0
- v = v - 128.0
-
- # 2. Matrix Multiplication (Coefficients for natural color)
- r = y + 1.793 * v
- g = y - 0.213 * u - 0.533 * v
- b = y + 2.112 * u
-
- # 3. Stack into final order
- if is_bgr:
- colored_img = torch.cat([b, g, r], dim=0)
- else:
- colored_img = torch.cat([r, g, b], dim=0)
-
- # 4. Final Clamp and Format
- colored_img.clamp_(0, 255)
- output = colored_img.to(torch.uint8)
+ del streamer
- if out_buffer is not None:
- out_buffer.copy_(output)
- return out_buffer
+ # Clear all application state
+ app.state.active_streams.clear()
+ if hasattr(app.state, "model"):
+ del app.state.model
- return output
+ gc.collect() # Final garbage collection
+ if torch.cuda.is_available():
+ torch.cuda.empty_cache()
+ app.state.status = "Stopped"
+# ----- INGESTION FUNCTIONS -----
def send_metadata(
VDMS_POOL=None,
DEBUG_FLAG=BASE_PIPELINE_CONFIG.DEBUG_FLAG,
@@ -331,10 +427,18 @@ def send_metadata(
clip_key = ""
width = 0
height = 0
+
while True:
+ queue_details = None
+ clip_metadata = None
+
try:
queue_details = send_metadata_queue.get()
+ # send_metadata_queue.task_done()
if queue_details is None:
+ main_app_logger.info(
+ "[METADATA-DEBUG D-pill] Poison pill received. Terminating send_metadata thread loops cleanly.",
+ )
break
# # (clip_key, clip_filename, width, height, clip_metadata) = queue_details
@@ -343,6 +447,9 @@ def send_metadata(
clip_metadata = all_metadata.pop(clip_key, None)
if clip_metadata:
+ main_app_logger.info(
+ f"sending 2 vdms: {clip_key} with {len(clip_metadata)} items."
+ )
success = metadata2vdms_with_retry(
clip_key,
clip_filename,
@@ -380,7 +487,6 @@ def send_metadata(
f" [CRITICAL] Permanent VDMS failure. Data saved to: {error_path}"
)
- send_metadata_queue.task_done()
else:
main_app_logger.error(
f" [MISSING] Metadata for {clip_key} was lost before upload!"
@@ -388,281 +494,109 @@ def send_metadata(
except Exception as e:
# pass
- print(f"[EXCEPTION] Exception occurred in send_metadata: {e}")
-
-
-# ----- STREAM HANDLERS -----
-def scale_clusters_to_8k(merged_640, frame_w=7680, frame_h=4320):
- # Ratios for 8K projection
- scale_x = frame_w / 640.0
- scale_y = frame_h / 640.0
- final_rois = []
-
- for box in merged_640:
- # Calculate centroid in 640p space
- cx_640 = (box[0] + box[2]) / 2.0
- cy_640 = (box[1] + box[3]) / 2.0
-
- # Map to 8K space with float precision to avoid offset drift
- cx_8k = cx_640 * scale_x
- cy_8k = cy_640 * scale_y
-
- # Center the 640x640 YOLO crop at the 8K centroid
- half = 320
- nx1 = max(0, int(cx_8k - half))
- ny1 = max(0, int(cy_8k - half))
- nx2 = min(frame_w, nx1 + 640)
- ny2 = min(frame_h, ny1 + 640)
-
- # Shift back if clamped at 8K boundaries
- if nx2 == frame_w:
- nx1 = max(0, frame_w - 640)
- if ny2 == frame_h:
- ny1 = max(0, frame_h - 640)
-
- final_rois.append([nx1, ny1, nx1 + 640, ny1 + 640])
-
- return final_rois
-
-
-def rendering_worker(
- queue,
- shared_details,
- ready_idx,
- reader_active_idx,
- frame_lengths,
- signal_queue,
- display_size,
- quality,
-):
- disp_w, disp_h = display_size
- # Attach to both buffers
- shm_names = shared_details["shm_names"]
- worker_shms = [mp.shared_memory.SharedMemory(name=n) for n in shm_names]
- num_shms = len(shm_names)
-
- # Get shm
- # shm_name = shared_details.get("shm_name")
- # try:
- # shm = mp.shared_memory.SharedMemory(name=shm_name)
- # except Exception as e:
- # print(f"[WORKER] SHM attach failed: {e}", flush=True)
- # return
-
- try:
- while True:
- item = queue.get()
- if item is None: # Sentinel value to stop the worker
- break
-
- # frame is display size
- # metadata in resized res
- display_frame, frameNum, metadata_or_bbs, class_list = item
-
- # display_size = (self.resize_h, self.resize_w)
- # display_frame = cv2.resize(frame, display_size, interpolation=cv2.INTER_NEAREST)
-
- scale_display_x = disp_w / 640
- scale_display_y = disp_h / 640
-
- if isinstance(metadata_or_bbs, dict):
- # Case: Object Detection
- display_frame = get_metadata_overlay(
- display_frame,
- metadata_or_bbs,
- class_list,
- (scale_display_x, scale_display_y),
- (disp_w, disp_h),
- )
-
- elif metadata_or_bbs is not None:
- # Case: Motion Detections Only (SF Path)
- display_frame = get_bb_overlay(
- display_frame,
- metadata_or_bbs,
- (scale_display_x, scale_display_y),
- (disp_w, disp_h),
- )
-
- # writer.write(display_frame)
- if frameNum > shared_details["last_id"]: # self.last_delivered_frame_id:
- frame_bytes = get_display_frame_in_bytes(
- display_frame,
- display_size=display_size,
- quality=quality,
- return_bytes=True,
- )
- if frame_bytes:
- # THE HARD GUARD: If the reader is currently touching RAM, skip this write.
- # This prevents the '1-minute' scramble by ensuring zero memory overlap.
- # if signal_queue.full():
- # continue
- frame_len = len(frame_bytes)
-
- forbidden_idx = [ready_idx.value, reader_active_idx.value]
- available_idx = [
- i for i in range(num_shms) if i not in forbidden_idx
- ]
-
- if not available_idx:
- continue
-
- # Write to the buffer that is NOT currently 'ready'
- # write_idx = (shared_details["buffer_idx"] + 1) % 2
- # write_idx = 1 if ready_idx.value == 0 else 0
- # current_ready = ready_idx.value
- # write_idx = (current_ready + 1) % 3
- write_idx = available_idx[0]
- shm = worker_shms[write_idx]
-
- # Zero-copy write to RAM
- shm.buf[:frame_len] = frame_bytes
-
- frame_lengths[write_idx] = frame_len
- # shared_details["buffer_idx"] = write_idx
- ready_idx.value = write_idx
- shared_details["last_id"] = frameNum
- # self.last_frame_id = frameNum
- # self.last_heartbeat = time.time()
- # Signal the FastAPI generator that a new frame is ready
- # self.loop.call_soon_threadsafe(self.frame_ready_event.set)
- # self.mp_frame_ready_event.set()
- # try:
- # signal_queue.put_nowait(True)
- # except Exception:
- # pass
- signal_queue.put(True)
-
- # END While
-
- except Exception as e:
- print(f"[EXCEPTION] Error while rendering display: {e}")
- finally:
- for s in worker_shms:
- s.close()
+ main_app_logger.info(
+ f"[EXCEPTION] Exception occurred in send_metadata: {e}"
+ )
+ traceback.print_exc()
+ finally:
+ if "queue_details" in locals():
+ del queue_details
+ if "clip_metadata" in locals():
+ del clip_metadata
+ send_metadata_queue.task_done()
-def get_metadata_overlay(
- display_frame, metadata_or_bbs, class_list, scale_display, disp_size
-):
- scale_display_x, scale_display_y = scale_display
- disp_w, disp_h = disp_size
- for _, obj in metadata_or_bbs.items():
- bbox = obj["bbox"]
- x = max(0, int(bbox["x"] * scale_display_x))
- y = max(0, int(bbox["y"] * scale_display_y))
- w = min(disp_w, int(bbox["width"] * scale_display_x))
- h = min(disp_h, int(bbox["height"] * scale_display_y))
-
- class_name = bbox["object"]
- class_id = class_list.index(class_name) if class_name in class_list else 0
- confidence = bbox.get("object_det", {}).get("confidence", 0.0)
-
- bb_color = get_detection_color(class_id, is_bgr=True)
- label = f"{class_name} {confidence:.2f}"
-
- cv2.rectangle(display_frame, (x, y), (x + w, y + h), bb_color, 2)
- draw_label(display_frame, label, (x, y), color=bb_color, padding=5)
- return display_frame
-
-
-def get_bb_overlay(display_frame, metadata_or_bbs, scale_display, disp_size):
- scale_display_x, scale_display_y = scale_display
- disp_w, disp_h = disp_size
- for box in metadata_or_bbs:
- if torch.is_tensor(box):
- x1, y1, x2, y2 = box.to(torch.int).cpu().tolist()
- else:
- x1, y1, x2, y2 = map(int, box)
+_KNOWN_HANDLER_METHODS = {}
- x1 = max(0, int(x1 * scale_display_x))
- y1 = max(0, int(y1 * scale_display_y))
- x2 = min(disp_w, int(x2 * scale_display_x))
- y2 = min(disp_h, int(y2 * scale_display_y))
- display_frame = cv2.rectangle(display_frame, (x1, y1), (x2, y2), (0, 0, 255), 2)
- return display_frame
+def get_known_handler_methods(handler_class):
+ if handler_class not in _KNOWN_HANDLER_METHODS:
+ handler_methods = {}
+ for cls in [handler_class, DeviceBaseHandler]:
+ for name, attr in inspect.getmembers(cls, predicate=inspect.isfunction):
+ handler_methods[name] = attr
+ _KNOWN_HANDLER_METHODS[handler_class] = handler_methods
+ return _KNOWN_HANDLER_METHODS[handler_class]
-def test_rendering_worker(queue, display_size, out_path, target_fps):
- """
- Ultra-efficient video saver for TEST_MODE.
- Pipes raw BGR frames directly into an internal FFmpeg engine subshell.
- """
- disp_w, disp_h = display_size
- # Construct optimized MPEG-4 parameters to match main pipeline architecture
- ffmpeg_cmd = [
+def standalone_writer_process(write_queue, fps, width, height, output_path):
+ """Runs completely isolated from the GIL and main AI loop."""
+ # 1. Initialize FFmpeg INSIDE the new process
+ command = [
"ffmpeg",
"-y",
"-f",
"rawvideo",
+ "-vcodec",
+ "rawvideo",
+ "-s",
+ f"{width}x{height}",
"-pix_fmt",
"bgr24",
- "-s",
- f"{disp_w}x{disp_h}",
"-r",
- str(int(target_fps)),
+ str(fps),
"-i",
"-",
"-c:v",
"libx264",
+ "-preset",
+ "ultrafast",
"-crf",
- "23",
- # "-c:v", "mpeg4", # Or "libx264" if you prefer H.264
- # "-qscale:v", "4", # Quality scale (use -crf 23 if using libx264)
- str(out_path),
+ "28",
+ "-pix_fmt",
+ "yuv420p",
+ output_path,
]
- # Spawn background daemon process
- proc = subprocess.Popen(
- ffmpeg_cmd,
- stdin=subprocess.PIPE,
- stdout=subprocess.DEVNULL,
- stderr=subprocess.DEVNULL,
- bufsize=10**7,
- )
+ writer = subprocess.Popen(command, stdin=subprocess.PIPE, stderr=subprocess.DEVNULL)
try:
while True:
- item = queue.get()
- if item is None: # Sentinel value to drain and close the process
+ frame_data = write_queue.get()
+ if frame_data is None: # Poison pill to stop
break
- display_frame, frameNum, metadata_or_bbs, class_list = item
- display_frame = np.ascontiguousarray(display_frame)
- scale_display_x = disp_w / 640
- scale_display_y = disp_h / 640
+ # Write the raw bytes directly to FFmpeg
+ writer.stdin.write(frame_data.tobytes())
- # --- Draw Detection Overlays ---
- if isinstance(metadata_or_bbs, dict):
- # Object Mode (YOLO Structs)
- display_frame = get_metadata_overlay(
- display_frame,
- metadata_or_bbs,
- class_list,
- (scale_display_x, scale_display_y),
- (disp_w, disp_h),
- )
+ except Exception as e:
+ logging.error(f"[WRITER PROCESS] Error: {e}")
+ finally:
+ writer.stdin.close()
+ writer.wait()
- elif metadata_or_bbs is not None:
- # # Motion / Smart Filtering Overlay Path
- display_frame = get_bb_overlay(
- display_frame,
- metadata_or_bbs,
- (scale_display_x, scale_display_y),
- (disp_w, disp_h),
- )
- # Pipe continuous raw contiguous memory block directly into kernel filesystem handles
- proc.stdin.write(np.ascontiguousarray(display_frame).tobytes())
- # queue.task_done()
+# ----- STREAM HANDLERS -----
+def get_test_handler(test_class_self, device): # Resolve concrete handler class type
+ HandlerClass = GPUStreamHandler if device == "gpu" else CPUStreamHandler
+ # HandlerClass.pipeline_fn = test_class_self.__class__.pipeline_fn
+
+ # Dynamically re-bind backend methods to this execution instance
+ handler_classes = [HandlerClass, DeviceBaseHandler]
+ handler_methods = get_known_handler_methods(HandlerClass)
+ for method_name, method_func in handler_methods.items():
+ if method_name in ["pipeline_fn", "config"]: # "run_realtime_inference"
+ continue
+
+ if not hasattr(test_class_self.__class__, method_name):
+ setattr(
+ test_class_self,
+ method_name,
+ types.MethodType(method_func, test_class_self),
+ )
- except Exception as e:
- print(f"[TEST-WORKER-EXCEPTION] Video compilation error: {e}")
- finally:
- if proc.stdin:
- proc.stdin.close()
- proc.wait()
+ orig_methods = list(
+ sorted(
+ set(
+ [
+ k
+ for h in handler_classes + [test_class_self]
+ for k in h.__dict__.keys()
+ ]
+ )
+ )
+ )
+ return test_class_self, orig_methods
class DeviceBaseHandler:
@@ -675,415 +609,531 @@ def __init__(
self.active = True
self.active_streams = active_streams
self.config = config
+
+ sf_name = "SF" if config.sf_enabled else "noSF"
+ self._testMethodName = kwargs.get(
+ "run_name",
+ f"{name}_{sf_name}_{config.DETECTION_TYPE.lower()}_{config.DEVICE.lower()}",
+ )
+
configstr = "\n".join(
[f"\t{k}: {v}" for k, v in config.__dict__.items() if not k.startswith("_")]
)
- log_to_logger(f"PipelineConfig: \n{configstr}\n", level="info")
+ main_app_logger.info(f"PipelineConfig: \n{configstr}\n")
self.loop = asyncio.get_event_loop()
self.frame_ready_event = asyncio.Event()
- self._is_stopped = False # 🛡️ Shutdown guard
- self._stop_lock = threading.Lock() # 🔒 Local lock for this instance
- self.mp_frame_ready_event = mp.Event()
+ self._is_stopped = False
+ self._stop_lock = threading.Lock() # Local lock for this instance
+ self.main_startup_event = mp.Event()
# From global
self.device = self.config.DEVICE
self.device_input = self.config.device_input
- # self.disp_w, self.disp_h = self.config.DISPLAY_FRAME_SIZE
+ self.disp_w, self.disp_h = self.config.DISPLAY_FRAME_SIZE
self.resize_h, self.resize_w = [self.config.MODEL_H, self.config.MODEL_W]
- self.setup_reader(self.config.TARGET_FPS, self.config.CLIP_DURATION)
+ self.setup_reader(
+ self.config.TARGET_FPS,
+ self.config.CLIP_DURATION,
+ startup_event=self.main_startup_event,
+ )
# Kwargs
# clip_duration = kwargs.get("clip_duration", CLIP_DURATION)
self.initialize_variables()
- provided_model = kwargs.get("model")
- self.setup_model(provided_model)
-
- self.prepare_pipeline()
+ # Should be started before calling setup_threads
+ # if hasattr(self.reader, "worker") and not self.reader.worker.is_alive():
+ # self.reader.worker.start()
# Start dedicated inference thread and timers
# self.stat_start_time = time.perf_counter() # timing to display frame
- self.last_heartbeat = time.time()
self.setup_threads()
+ self.last_heartbeat = time.perf_counter()
- def setup_model(self, provided_model, force_export=False):
- if (
- self.frame_width * self.frame_height
- ) <= self.config.SMART_FILTERING_PIXEL_CONSTRAINT:
- if "_noSF" not in self.config.model_path:
- oldpath = Path(self.config.model_path)
- old_modelname = self.config.MODEL_NAME
- self.config.MODEL_NAME = f"{old_modelname}_noSF"
- new_model_name = oldpath.name.replace(
- old_modelname, self.config.MODEL_NAME
- )
- self.config.model_path = str(oldpath.parent / new_model_name)
+ def start(self):
+ """
+ Starts the decoupled ingestion and inference threads in the correct order.
+ """
+ # PRE-SYNC: Ensure GPU is idle before timing starts
+ if self.device_input == "cuda" and torch.cuda.is_available():
+ torch.cuda.synchronize()
- if provided_model is not None and not isinstance(provided_model, str):
- self.model = provided_model
- self.label_source = [v for k, v in self.model.names.items()]
- else:
- # if isinstance(provided_model, str) or provided_model is None:
- # if Path(self.config.model_path).exists():
- # self.model = YOLO(self.config.model_path, verbose=False, task="detect")
- # self.label_source = []
- # for k, v in self.model.names.items():
- # self.label_source.append(v)
- # else:
- run_platform_name = "engine" if "cuda" in self.device_input else "openvino"
- self.model, _, self.label_source = get_model(
- Path(self.config.model_path).parent,
- self.config.MODEL_NAME.replace("_noSF", ""),
- run_platform_name,
- self.device_input,
- batch=self.config.MODEL_MAX_BATCH_SIZE,
- force_export=force_export,
- sf_enabled=self.config.sf_enabled,
- model_h=self.resize_h,
- model_w=self.resize_w,
- )
+ main_app_logger.info(f"[START {self.name}] Starting Threads ...")
- if not self.config.sf_enabled:
- self.model_warmup(self.frame_height, self.frame_width)
- else:
- self.model_warmup(self.resize_h, self.resize_w)
- # else:
- # self.model = provided_model
- # self.label_source = []
- # for k, v in self.model.names.items():
- # self.label_source.append(v)
+ # Start the hardware-decoupled reader first
+ self.reader.start()
- def initialize_variables(self):
- # self.input_fps = self.reader.input_fps
- # self.target_fps = self.reader.target_fps
- # self.step_size = self.input_fps / self.target_fps
- # self.frame_skip = self.reader.frame_skip
- # self.max_frames_per_clip = self.reader.max_frames_per_clip
- # self.frame_interval = self.reader.frame_interval
- # self.frame_width = self.reader.frame_width
- # self.frame_height = self.reader.frame_height
- # self.numFrames = self.reader.numFrames
- # self.duration_s = self.numFrames / self.input_fps
- # self.expected_num_frames = int(self.duration_s * self.target_fps)
- # self.get_frameWH()
- # 1. HARD GUARD: Capture reader values and ensure the connection is active
- self.input_fps = self.reader.input_fps
- self.target_fps = self.reader.target_fps
- self.frame_width = self.reader.frame_width
- self.frame_height = self.reader.frame_height
- self.numFrames = self.reader.numFrames
+ # if self.config.ENABLE_QUERYING:
+ # self._initialize_writer()
- if self.input_fps <= 0 or self.frame_width <= 0 or self.frame_height <= 0:
- main_app_logger.error(
- f"[{self.name}] Stream handler fast-fail triggered. Destination "
- f"unreachable or invalid ({self.source}). Terminating pipeline configuration."
- )
- # Instantly stop background threads to prevent zombie process leakage
- if hasattr(self, "reader") and self.reader is not None:
- self.reader.stop()
+ # Small delay to allow the reader's deque to populate
+ time.sleep(0.1)
- raise RuntimeError(
- f"Failed to initialize stream reader endpoint: {self.source}"
- )
+ # Start the producer and consumer threads
+ if hasattr(self, "process_thread") and not self.process_thread.is_alive():
+ self.process_thread.start()
- # 2. Proceed with calculation mechanics safely only if values are healthy
- self.step_size = self.input_fps / self.target_fps
- self.frame_skip = self.reader.frame_skip
- self.max_frames_per_clip = self.reader.max_frames_per_clip
- self.frame_interval = self.reader.frame_interval
+ if not self.config.DISABLE_DETECTION:
+ if hasattr(self, "render_proc") and hasattr(self.render_proc, "start"):
+ self.render_proc.start()
- self.duration_s = self.numFrames / self.input_fps
- self.expected_num_frames = int(self.duration_s * self.target_fps)
- self.get_frameWH()
+ if hasattr(self, "display_proc") and hasattr(self.display_proc, "start"):
+ self.display_proc.start()
- # Determine minimum contour size relative to frame resolution
- self.min_contour_area = int(
- (self.config.ROI_MIN_AREA_RATIO * self.resize_w)
- * (self.config.ROI_MIN_AREA_RATIO * self.resize_h)
- ) # 207
-
- self.dist_thresh_8k = max(
- self.config.ROI_DISTANCE_THRESH_RATIO * self.frame_width,
- self.config.ROI_DISTANCE_THRESH_RATIO * self.frame_height,
- )
- self.dist_thresh_640 = max(
- self.config.ROI_DISTANCE_THRESH_RATIO * self.resize_w,
- self.config.ROI_DISTANCE_THRESH_RATIO * self.resize_h,
- ) # 0.05 * self.resize_w
- self.scales_tensor = torch.tensor(
- [self.scale_x, self.scale_y, self.scale_x, self.scale_y],
- # device="cpu",
- device=self.device_input,
- )
+ if (
+ self.config.ENABLE_QUERYING
+ and not self.config.TEST_MODE
+ and not self.metadata_thread.is_alive()
+ ):
+ self.metadata_thread.start()
- self.disp_w, self.disp_h = self.config.DISPLAY_FRAME_SIZE
+ if self.config.ENABLE_QUERYING and not self.writer_process.is_alive():
+ self.writer_process.start()
- # Performance Tracking
- self.elapsed_display_time = 0.0
- self.frame_count = 0 # Frame count for videos
- self.frame_count_target = 0
- self.last_delivered_frame_id = -1 # Track what was actually sent
- self.last_frame_id = 0
- self.latest_processed_frame = None
- self.next_process_idx = 0.0
- self.stat_fps = 0
- self.stat_frame_count = 0
- self.total_objects_detected = 0
+ self.last_heartbeat = time.perf_counter()
+ return self
- self.writer_done = True
+ def stop(self):
+ """
+ A robust, sequential shutdown process that gracefully terminates all threads
+ and releases all resources without race conditions, especially for live streams.
+ """
+ # Force the GPU to finish all active kernels, copy operations,
+ # and event processing across all streams.
+ # This guarantees no async operations are holding the memory locked!
+ if "cuda" in str(self.device_input) and torch.cuda.is_available():
+ # main_app_logger.info(f"[STOP {self.name}] Synchronizing all CUDA streams...")
+ torch.cuda.synchronize(self.device_input)
+
+ # Use a lock to prevent this complex function from being called by multiple threads at once.
+ with self._stop_lock:
+ if getattr(self, "_is_stopped", False):
+ return
- # Video Clipping
- # self.video_writer = None
- self.ffmpeg_proc = None # Replaces cv2.VideoWriter completely
- # self.fourcc = cv2.VideoWriter_fourcc(*"mp4v") # avc1, mp4v
- self.clip_id = 0
- # self.clip_filename = ""
- self.clip_filename_pattern = f"{self.config.SHARED_OUTPUT}/{self.name}_%03d.mp4"
- self.clip_key = f"{self.name}_000.mp4"
- # self.tmp_file = ""
- self.frame_in_clip_count = 0
-
- if self.config.ENABLE_QUERYING:
- # Thread-safe queue for the resized frames (640x640)
- # maxlen=300 allows for a 20-second buffer in case of extreme disk lag
- # Non-blocking queue for frames and control signals
- self.write_queue = queue.Queue(maxsize=300)
- if not self.config.TEST_MODE:
- self.send_metadata_queue = queue.Queue()
- self.writer_done = False
- self.stop_writer = threading.Event() if self.config.ENABLE_QUERYING else None
+ main_app_logger.info(
+ f"[STOP {self.name}] Initiating orderly shutdown sequence..."
+ )
- # --- PRE-ALLOCATED ZERO-COPY HARDWARE RING WORKSPACE ---
- self.ring_depth = 4 # 8
- self.gpu_ring_idx = 0
- self.cpu_ring_idx = 0
+ if hasattr(self, "write_queue") and hasattr(self, "writer_process"):
+ try:
+ self.write_queue.put(
+ None, timeout=1.0
+ ) # The poison pill for the process
+ except (queue.Full, AttributeError):
+ pass
- self.pinned_matrices = []
- self.pinned_tensors = []
+ # Signal all active loops to stop accepting new work.
+ self.active = False
+ self.prefetch_active = False
- # Pre-allocate a 4-slot ring buffer for raw 8K BGR frames
- self.ai_ring_depth = 4
- self.ai_ring_idx = 0
- self.frame_stride_bytes = (
- self.resize_w * self.resize_h * 3
- ) # ~99.5 MB per frame
+ if hasattr(self, "active_streams") and isinstance(
+ self.active_streams, dict
+ ):
+ self.active_streams.pop(self.name, None)
- self.ai_shms = []
- self.ai_shm_names = []
- self.ai_pinned_tensors = [] # Explicit property initialization 🚀
+ # Wake up any FastAPI async streaming generators waiting on frame_ready_event
+ if (
+ hasattr(self, "frame_ready_event")
+ and self.frame_ready_event is not None
+ ):
+ try:
+ if hasattr(self, "loop") and self.loop and self.loop.is_running():
+ self.loop.call_soon_threadsafe(self.frame_ready_event.set)
+ else:
+ self.frame_ready_event.set()
+ except Exception:
+ pass
- for i in range(self.ai_ring_depth):
- name = f"shm_ai_640_{self.name}_{i}_{os.getpid()}"
+ main_app_logger.info(f"[STOP {self.name}] Stopping writer event...")
+ self.set_stop_writer_event(remove=False)
- try:
- # Attempt to attach to a lingering zombie segment
- old_shm = shared_memory.SharedMemory(name=name)
- old_shm.close()
- old_shm.unlink() # Permanently destroys the old OS block handle
- main_app_logger.warning(f"Cleaned up residual zombie shared memory block: {name}")
- except FileNotFoundError:
- pass # Block doesn't exist, safe to proceed normal initialization
-
- shm = shared_memory.SharedMemory(
- name=name, create=True, size=self.frame_stride_bytes
+ # Shut down the worker pools. wait=True is crucial.
+ # This blocks until all background tasks have finished.
+ main_app_logger.info(
+ f"[STOP {self.name}] Shutting down worker executors..."
)
- self.ai_shms.append(shm)
- self.ai_shm_names.append(name)
+ self.stop_executors(["clip_executor", "executor"])
- # Map a zero-copy lockless numpy array view straight onto the memory block
- view = np.ndarray(
- (self.resize_h, self.resize_w, 3), dtype=np.uint8, buffer=shm.buf
+ # Join all other background threads.
+ main_app_logger.info(
+ f"[STOP {self.name}] Joining all background threads..."
)
+ self.stop_threads(
+ [
+ "process_thread",
+ "writer_process",
+ "metadata_thread",
+ "display_proc",
+ "render_proc",
+ ]
+ )
+ # if (
+ # hasattr(self, "writer_process") and self.writer_process.is_alive()
+ # ): # actually a process
+ # self.writer_process.join(timeout=3.0)
+ # if self.writer_process.is_alive():
+ # self.writer_process.terminate()
+ self.unregister_pinned_cuda_data()
+ buffer_pools = [
+ "shm_buffer_pool",
+ "gpu_input",
+ "clipper_shm_np_views",
+ "pinned_matrices",
+ "pinned_tensors",
+ ]
+ self.clear_buffer_pools(buffer_pools)
- # Page-lock the host buffer window to maximize PCIe bus transfer bandwidth
- try:
- cv2.cuda.registerPageLocked(view)
- except Exception:
- pass
-
- # Expose a direct matching PyTorch host tensor map to secure high-speed uploads
- self.ai_pinned_tensors.append(torch.from_numpy(view))
+ if hasattr(self, "reader") and self.reader is not None:
+ if hasattr(self.reader, "pinned_views"):
+ self.reader.pinned_views = []
+ if hasattr(self.reader, "_static_gpu_frame_buffer"):
+ self.reader._static_gpu_frame_buffer = None
- # Pre-allocate a 4D FP16 GPU staging canvas to maximize Tensor Core performance
- # if self.device_input == "cuda":
- # self.ai_gpu_staging = torch.empty(
- # (1, 3, self.frame_height, self.frame_width),
- # dtype=torch.float16,
- # device=f"cuda:{self.gpu_id}",
- # )
- # self.preview_gpu_staging = torch.empty(
- # (1, 3, self.frame_height, self.frame_width),
- # dtype=torch.float16,
- # device=f"cuda:{self.gpu_id}",
- # )
+ if hasattr(self, "reader") and self.reader is not None:
+ main_app_logger.info(f"[STOP {self.name}] Stopping reader process...")
+ self.reader.stop()
+ main_app_logger.info(f"[STOP {self.name}] Reader process stopped.")
- # Pre-allocate 640x640 workspace footprint across CPU and GPU spaces
- for _ in range(self.ring_depth):
- mat = np.zeros((self.resize_h, self.resize_w, 3), dtype=np.uint8)
- try:
- cv2.cuda.registerPageLocked(mat)
- except cv2.error:
- pass
- self.pinned_matrices.append(mat)
- self.pinned_tensors.append(torch.from_numpy(mat))
- # Isolate CUDA tasks using a dedicated stream and independent hardware completion barriers
- self.processing_stream = (
- torch.cuda.Stream() if self.device_input == "cuda" else None
- )
- # Pre-allocated hardware events guarantee completely non-blocking stream isolation
- self.slot_events = (
- [torch.cuda.Event() for _ in range(8)]
- if self.device_input == "cuda"
- else None
- )
+ if hasattr(self, "prefetch_threads") and self.prefetch_threads:
+ main_app_logger.info(f"[STOP {self.name}] Stopping prefetch threads...")
+ for val in self.prefetch_threads:
+ self.stop_thread(val)
+ self.prefetch_threads.clear()
+ # delattr(self, "prefetch_threads")
- self.gpu_float_staging = None
- if self.device_input == "cuda":
- self.gpu_float_staging = torch.empty(
- (1, 3, self.frame_height, self.frame_width),
- dtype=torch.float16,
- device=f"cuda:{self.gpu_id}",
+ if hasattr(self, "async_writer") and self.async_writer is not None:
+ main_app_logger.info(f"[STOP {self.name}] Releasing writer...")
+ try:
+ self.async_writer.release()
+ except Exception:
+ pass
+ setattr(self, "async_writer", None)
+
+ # Clear the buffer that holds direct pointers to the shared memory.
+ # This is the most critical step to release the "exported pointers".
+ # buffer_pools = ["shm_buffer_pool", "gpu_input"]
+ # buffer_pools = [
+ # "shm_buffer_pool",
+ # "gpu_input",
+ # "clipper_shm_np_views",
+ # "pinned_matrices",
+ # "pinned_tensors",
+ # ]
+ # self.clear_buffer_pools(buffer_pools)
+ # if hasattr(self, "shm_buffer_pool") and self.shm_buffer_pool is not None:
+ # main_app_logger.info(
+ # f"[STOP {self.name}] Clearing shared memory buffer pool to release pointers..."
+ # )
+ # for i in range(len(self.shm_buffer_pool)):
+ # self.shm_buffer_pool[i] = None
+ # # self.shm_buffer_pool = None
+
+ # if hasattr(self, "gpu_input") and self.gpu_input is not None:
+ # # main_app_logger.info(
+ # # f"[STOP {self.name}] Clearing shared memory buffer pool to release pointers..."
+ # # )
+ # for i in range(len(self.gpu_input)):
+ # self.gpu_input[i] = None
+ # self.gpu_input = None
+ # # self.gpu_input.clear()
+
+ # Force instant garbage collection to drop Python buffer exports from memoryview
+ gc.collect()
+ if "cuda" in str(self.device_input) and torch.cuda.is_available():
+ torch.cuda.empty_cache()
+
+ shm_names = [
+ "clipper_shm_blocks",
+ "reader.shms",
+ "shms",
+ ]
+ main_app_logger.info(f"[STOP {self.name}] Clearing shared memory...")
+ self.clear_shared_memory_list(shm_names)
+
+ # if hasattr(self, "clipper_shm_blocks"):
+ # for shm in self.clipper_shm_blocks:
+ # shm.close()
+ # shm.unlink() # This removes the file from /dev/shm
+
+ # shms_to_close = [] # self.shms
+ # if hasattr(self, "reader") and self.reader is not None:
+ # main_app_logger.info(
+ # f"[STOP {self.name}] Calling reader.stop() to clean up worker process..."
+ # )
+ # self.reader.stop()
+ # if hasattr(self.reader, "shms"):
+ # shms_to_close += self.reader.shms
+ # self.reader = None
+
+ # if hasattr(sys, "exc_info"):
+ # # This clears the three-element tuple (type, value, traceback)
+ # exc_info = sys.exc_info()
+ # if exc_info[2] is not None: # Check if a traceback object exists
+ # traceback.clear_frames(exc_info[2])
+ # # For older python versions, or as an extra measure
+ # if hasattr(sys, "exc_clear"):
+ # sys.exc_clear()
+
+ # gc.collect()
+ # if "cuda" in str(self.device_input) and torch.cuda.is_available():
+ # torch.cuda.empty_cache()
+ # if hasattr(torch.cuda, "ipc_collect"):
+ # torch.cuda.ipc_collect()
+
+ # if shms_to_close:
+ # main_app_logger.info(f"[STOP {self.name}] Closing OS shared memory files...")
+ # for shm in shms_to_close:
+ # try:
+ # shm.close()
+ # shm.unlink()
+ # except Exception:
+ # pass
+ # shms_to_close.clear()
+
+ # =========================================================================
+ # 🕵️♂️ ADVANCED REFERENCE INVESTIGATOR
+ # =========================================================================
+ # import gc
+ # import inspect
+
+ main_app_logger.info("\n" + "=" * 80)
+ main_app_logger.info(
+ "🕵️♂️ [STOP INVESTIGATION] Scanning for active pointers to SharedMemory blocks..."
+ )
+ # main_app_logger.info("=" * 80)
+
+ # if hasattr(self, "reader") and hasattr(self.reader, "shms"):
+ # for idx, shm in enumerate(self.reader.shms):
+ # main_app_logger.info(f"\n[SHM BLOCK {idx}] Name: {shm.name}")
+
+ # # Create a function to recursively find the ultimate owner
+ # def find_owner(obj, depth=0):
+ # if depth > 5: # Safety limit to prevent infinite recursion
+ # return
+
+ # referrers = gc.get_referrers(obj)
+ # for ref in referrers:
+ # # Ignore the current function's local variables
+ # if ref is referrers or ref is locals():
+ # continue
+
+ # ref_type_name = type(ref).__name__
+
+ # # If we find a class instance, we found the owner!
+ # if hasattr(ref, "__dict__"):
+ # # Check if it's a class instance and not just a dict
+ # if not isinstance(ref, (dict, list, tuple, set)):
+ # for k, v in ref.__dict__.items():
+ # if v is obj:
+ # main_app_logger.info(
+ # f"{' ' * depth}└── Found Owner: Class <{type(ref).__name__}> holds reference in attribute '{k}'"
+ # )
+
+ # # If the referrer is a list or tuple, recurse deeper
+ # elif isinstance(ref, (list, tuple)):
+ # main_app_logger.info(
+ # f"{' ' * depth}└── Held by container: <{ref_type_name}> of len {len(ref)}"
+ # )
+ # find_owner(ref, depth + 1)
+
+ # find_owner(shm)
+ # main_app_logger.info("=" * 80 + "\n")
+ # =========================================================================
+
+ # 8. Clean up any remaining handler resources like queues and events.
+ main_app_logger.info(f"[STOP {self.name}] Cleaning up final resources...")
+ self.stop_sync_manager()
+ self.drain_and_close_queues(
+ ["signal_queue", "render_queue", "prefetch_queue", "write_queue"]
)
- # Default Kernels
- self.dilate_kernel = cv2.getStructuringElement(
- cv2.MORPH_ELLIPSE,
- # cv2.MORPH_RECT,
- (self.config.DILATE_KERNEL_SIZE, self.config.DILATE_KERNEL_SIZE),
- )
- # self.dilate_kernel_for_enhanced_mask = np.ones((15,15), np.uint8) # 5, 5) (21, 21)
- self.dilate_kernel_for_enhanced_mask = cv2.getStructuringElement(
- cv2.MORPH_ELLIPSE,
- # cv2.MORPH_RECT,
- (
- self.config.BKGD_SUB_INCLUDE_HISTORY_DILATE_KERNEL_SIZE,
- self.config.BKGD_SUB_INCLUDE_HISTORY_DILATE_KERNEL_SIZE,
- ),
- )
+ remove_attrs = [
+ "_cached_grid_x",
+ "_cached_grid_y",
+ "ffmpeg_proc",
+ "frame_in_clip_count",
+ "latest_processed_frame",
+ "mp_frame_ready_flag",
+ "mp_last_id",
+ "processing_stream",
+ "reader_active_idx",
+ "ready_buffer_idx",
+ "render_queue_backlog_counter",
+ "render_queue_counter",
+ "shm_frame_lengths",
+ "stat_start_time",
+ "write_queue_backlog_counter",
+ "write_queue_counter",
+ "static_gpu_360p",
+ "static_gpu_byte_bchw",
+ "gpu_float_staging",
+ "pinned_matrices",
+ "pinned_tensors",
+ ]
+ main_app_logger.info(f"[STOP {self.name}] Removing attributes...")
+ self.remove_scalar_attributes(remove_attrs)
+
+ # self.stop_events()
+ events = [
+ "frame_ready_event",
+ "queue_data_ready_event",
+ "_d2h_fence",
+ "det_end",
+ "det_start",
+ "frame_ready_event",
+ "main_startup_event",
+ "queue_data_ready_event",
+ "reader_lock",
+ "roi_end",
+ "roi_start",
+ "sf_end",
+ "sf_start",
+ "worker_tracking_lock",
+ ]
+ main_app_logger.info(f"[STOP {self.name}] Stopping events...")
+ self.stop_events(events, keys_to_skip_deletion=default_attr_keys)
+
+ if hasattr(self, "slot_events") and self.slot_events is not None:
+ main_app_logger.info(f"[STOP {self.name}] Stopping slot events...")
+ # Grab local reference and immediately nullify instance attribute
+ events_to_clean = self.slot_events
+ self.slot_events = None
+ if events_to_clean is not None:
+ for ev in list(events_to_clean):
+ try:
+ if isinstance(ev, torch.cuda.Event):
+ # Force internal driver handle release
+ del ev
+ except Exception:
+ pass
+ # self.slot_events = None
- def setup_reader(self, target_fps, clip_duration):
- # if hasattr(self, "reader"):
- # del self.reader
+ if getattr(self, "processor", None):
+ delattr(self, "processor")
- # Add a tiny sleep or garbage collect to ensure the GPU handle is released
- gc.collect()
- torch.cuda.empty_cache() # Clear any remaining context
+ if getattr(self, "evaluator", None):
+ delattr(self, "evaluator")
- # TODO: Further investigate GPU path, bkgd subtraction sensitive to artifacts
- self.gpu_id = 0
- if self.device_input == "cuda": # and not self.is_rtsp:
- from include.readers import GPUHybridReader
+ # self.pinned_matrices.clear()
+ # self.pinned_tensors.clear()
- self.reader = GPUHybridReader(
- source=self.source,
- target_fps=target_fps,
- clip_duration=clip_duration,
- gpu_id=self.gpu_id,
- queue_size=0 if self.config.TEST_MODE else 2,
- )
- else:
- from include.readers import CPUHybridReader
+ self.clean_up_tensors_and_arrays()
- self.reader = CPUHybridReader(
- source=self.source,
- target_fps=target_fps,
- clip_duration=clip_duration,
- queue_size=0 if self.config.TEST_MODE else 2,
- )
+ release_native_linux_heap()
- def prepare_pipeline(self):
- if self.device_input == "cuda":
- self.prepare_gpu_pipeline()
- if len(self.active_streams) == 0:
- self.gpu_warmup()
- else:
- self.prepare_cpu_pipeline()
+ gc.collect()
+ if "cuda" in self.device_input and torch.cuda.is_available():
+ torch.cuda.empty_cache()
+ if hasattr(torch.cuda, "ipc_collect"):
+ torch.cuda.ipc_collect()
+
+ # with self._stop_lock:
+ self._is_stopped = True
+ self.status = "DONE"
+ self.active_streams.pop(self.name, None)
+ main_app_logger.info(f"[STOP {self.name}] Shutdown complete.")
def setup_threads(self):
- # Shared 10MB memory for display
- self.setup_shared_memory()
+ """Overrides handlers.py to bind threads dynamically to the test instance."""
+
+ self._cached_grid_y, self._cached_grid_x = torch.meshgrid(
+ torch.arange(
+ self.resize_h, device=self.device_input
+ ), # Match target tracking resolution
+ torch.arange(self.resize_w, device=self.device_input),
+ indexing="ij",
+ )
+
+ self.setup_shared_memory() # Natively sets up Manager dictionary and buffers
# Executor for Async YOLO tasks and FFmpeg re-encoding
self.executor = ThreadPoolExecutor(max_workers=self.config.MAX_WORKERS)
+
self.clip_executor = ThreadPoolExecutor(max_workers=self.config.MAX_WORKERS)
- print(
+ main_app_logger.info(
f"sf_enabled: {self.config.sf_enabled}\tTEST_MODE: {self.config.TEST_MODE}",
- flush=True,
)
- # Producer: Handles acquisition and AI metadata logs
self.process_thread = threading.Thread(
target=self.run_realtime_inference,
- args=(self.config.sf_enabled,),
+ kwargs={
+ "sf_enabled": self.config.sf_enabled,
+ "gt_enabled": getattr(self, "gt_enabled", False),
+ },
daemon=True,
)
- self.signal_queue = mp.Queue(maxsize=1)
- self.render_queue = mp.Queue(maxsize=5)
+ # # Open up looking-ahead buffer horizons to eliminate 8K queue backpressure stalls
+ # self.render_queue_maxsize = 16 # 4
+ # self.render_queue = queue.Queue(maxsize=self.render_queue_maxsize)
+
+ # self.signal_queue_maxsize = 32 # 8
+ # self.signal_queue = queue.Queue(maxsize=self.signal_queue_maxsize)
if self.config.TEST_MODE:
test_dir = os.getenv(
"TEST_SUITE_RENDER_DIR", str(Path(self.config.SHARED_OUTPUT))
)
os.makedirs(test_dir, exist_ok=True)
- out_path = os.path.join(test_dir, f"{self.name}_detections_output.mp4")
- log_to_logger(
- f"[TEST MODE] Detection results saved to: {out_path}", level="info"
- )
- self.render_proc = threading.Thread(
- target=test_rendering_worker,
- args=(
- self.render_queue,
- (self.disp_w, self.disp_h),
- out_path,
- self.target_fps,
- ),
- daemon=True,
+
+ if hasattr(self, "video_output_name"):
+ video_output_name = self.video_output_name
+ else:
+ if self.source.startswith("rtsp"):
+ short_name = "rtsp"
+ else:
+ short_name = Path(self.source).stem
+ video_output_name = f"{self._testMethodName}_{short_name}.mp4"
+
+ self.output_path = os.path.join(test_dir, video_output_name)
+ # self.output_path = os.path.join(
+ # test_dir, f"{self.name}_detections_output.mp4"
+ # )
+ # log_to_logger(
+ # f"[TEST MODE] Detection results saved to: {self.output_path}",
+ # level="info",
+ # )
+
+ self.async_writer = AsyncVideoWriter(
+ self.output_path,
+ cv2.VideoWriter_fourcc(*"avc1"), # avc1, mp4v
+ float(self.target_fps),
+ (self.disp_w, self.disp_h),
)
# Dummy target alignment to prevent execution signature exceptions
- self.display_proc = threading.Thread(target=lambda: None, daemon=True)
- else:
- self.render_proc = mp.Process(
- target=rendering_worker,
- args=(
- self.render_queue,
- self.shared_details,
- self.ready_buffer_idx,
- self.reader_active_idx,
- self.shm_frame_lengths,
- self.signal_queue,
- (self.disp_w, self.disp_h),
- self.config.DISPLAY_FRAME_QUALITY,
- ),
+ # self.render_proc = threading.Thread(target=lambda: None, daemon=True)
+ # self.display_proc = threading.Thread(target=lambda: None, daemon=True)
+ # self.render_proc = DummyProcess()
+ # self.display_proc = DummyProcess()
+ log_to_logger(
+ f"[TEST MODE] Results saved to: {self.output_path}", level="info"
)
-
- def display_signal_sync():
- while self.active:
- # Wait for signal
- # if self.mp_frame_ready_event.wait(timeout=1.0):
- # self.mp_frame_ready_event.clear()
- try:
- _ = self.signal_queue.get(timeout=1.0)
- # print(f"[DEBUG]: Signal received in FastAPI process for {self.name}", flush=True)
- # Wake FastAPI async loop in main thread
- self.loop.call_soon_threadsafe(self.frame_ready_event.set)
- except queue.Empty:
- continue
-
- self.display_proc = threading.Thread(
- target=display_signal_sync, daemon=True
+ else:
+ self.async_writer = AsyncDisplayVideoWriter(
+ float(self.target_fps),
+ (self.disp_w, self.disp_h),
+ quality=int(self.config.DISPLAY_FRAME_QUALITY),
)
+ self.async_writer.set_handler_context(self)
+ # self.async_writer.start_worker()
+ try:
+ self.loop = asyncio.get_running_loop()
+ except RuntimeError:
+ try:
+ self.loop = asyncio.get_event_loop()
+ except RuntimeError:
+ self.loop = None # Headless fallback context
if self.config.ENABLE_QUERYING:
# NEW: Dedicated I/O pool for Disk/GPU transfers (Higher worker count for 8K)
- self.io_executor = ThreadPoolExecutor(max_workers=8)
+ # self.io_executor = ThreadPoolExecutor(max_workers=8)
# Dedicated FFmpeg pool so re-encoding doesn't slow down live AI
- self.ffmpeg_executor = ThreadPoolExecutor(max_workers=2)
+ # self.ffmpeg_executor = ThreadPoolExecutor(max_workers=2)
if not self.config.TEST_MODE:
# Sends metadata to VDMS
@@ -1103,378 +1153,2103 @@ def display_signal_sync():
)
# Consumer: Handles GPU-to-CPU download and Disk I/O (Writing resized frames to RAM disk)
- self.writer_thread = threading.Thread(
- target=self.video_writer_core_loop,
- args=(self.stop_writer,),
- daemon=True,
- )
+ # self.writer_process = threading.Thread(
+ # target=self.video_writer_core_loop,
+ # args=(self.stop_writer,),
+ # daemon=True,
+ # )
- def setup_shared_memory(self):
- self.manager = mp.Manager()
+ # self.render_proc = threading.Thread(target=lambda: None, daemon=True)
+ # self.display_proc = threading.Thread(target=lambda: None, daemon=True)
+ # self.render_proc = DummyProcess()
+ # self.display_proc = DummyProcess()
- self.shms = []
- shm_names = []
- num_shms = 3
+ def initialize_variables(self):
+ self.prefetch_threads = []
+ # Track active worker threads atomically to protect end-of-stream drainage
+ num_prefetch_workers = 4 if self.device_input == "cpu" else 3
+ self.active_workers_count = num_prefetch_workers
+ self.worker_tracking_lock = threading.Lock()
- # Shared Integer to track which buffer is "Ready" for the UI
- # 'i' for integer, initialized to 0
- self.ready_buffer_idx = mp.Value("i", 0)
- self.reader_active_idx = mp.Value("i", -1)
- self.shm_frame_lengths = mp.Array("i", [0 for _ in range(num_shms)])
+ # Override the atomic tracking count to match our active pool allocation
+ with self.worker_tracking_lock:
+ self.active_workers_count = num_prefetch_workers
- for idx in range(num_shms):
- # self.shm = mp.shared_memory.SharedMemory(create=True, size=10*1024*1024)
- shm_name = f"shm_{self.name}_{idx}_{os.getpid()}"
- # print(f"[DEBUG]: Setting up SHM {shm_name}", flush=True)
- try:
- shm = mp.shared_memory.SharedMemory(
- name=shm_name, create=True, size=10 * 1024 * 1024
- )
- except FileExistsError:
- # Attach to existing memory
- shm = mp.shared_memory.SharedMemory(name=shm_name)
- except Exception as e:
- main_app_logger.error(f"Failed to initialize shared memory: {e}")
- raise
- self.shms.append(shm)
- shm_names.append(shm_name)
+ # self.initialize_run_realtime_inference(read_frame_only, num_prefetch_workers)
- self.shared_details = self.manager.dict()
- self.shared_details["shm_names"] = shm_names
- # self.shared_details["buffer_idx"] = 0
- # self.shared_details["frame_length"] = [0 for _ in range(num_shms)]
- self.shared_details["last_id"] = -1
+ # 1. HARD GUARD: Capture reader values and ensure the connection is active
+ if not hasattr(self, "result_dir"):
+ self.result_dir = self.config.SHARED_OUTPUT
+ self.device_index = f"cuda:{self.gpu_id}" if self.device_input else "cpu"
+ self.input_fps = self.reader.input_fps
+ self.target_fps = self.reader.target_fps
+ self.frame_width = self.reader.frame_width
+ self.frame_height = self.reader.frame_height
+ self.numFrames = self.reader.numFrames
+ # self.min_roi_w = int(self.config.ROI_MIN_AREA_RATIO * self.resize_w)
+ # self.min_roi_h = int(self.config.ROI_MIN_AREA_RATIO * self.resize_h)
+ # self.max_roi_w = int(self.resize_w * self.config.ROI_MAX_RELATIVE_SIZE_RATIO)
+ # self.max_roi_h = int(self.resize_h * self.config.ROI_MAX_RELATIVE_SIZE_RATIO)
+ # self.max_cached_elements = 100
- def start(self):
- """
- Starts the decoupled ingestion and inference threads in the correct order.
- """
- # PRE-SYNC: Ensure GPU is idle before timing starts
- if self.device_input == "cuda":
- if torch.cuda.is_available():
- torch.cuda.synchronize()
+ if self.input_fps <= 0 or self.frame_width <= 0 or self.frame_height <= 0:
+ main_app_logger.error(
+ f"[{self.name}] Stream handler fast-fail triggered. Destination "
+ f"unreachable or invalid ({self.source}). Terminating pipeline configuration."
+ )
+ # Instantly stop background threads to prevent zombie process leakage
+ if hasattr(self, "reader") and self.reader is not None:
+ self.reader.stop()
- # Start the hardware-decoupled reader first
- self.reader.start()
+ raise RuntimeError(
+ f"Failed to initialize stream reader endpoint: {self.source}"
+ )
- if not self.config.DISABLE_DETECTION:
- self.render_proc.start()
+ # 2. Proceed with calculation mechanics safely only if values are healthy
+ self.step_size = (
+ float(self.input_fps) / float(self.target_fps)
+ if hasattr(self, "target_fps")
+ else 1.0
+ )
+ self.frame_skip = self.reader.frame_skip
+ self.max_frames_per_clip = self.reader.max_frames_per_clip
+ self.frame_interval = self.reader.frame_interval
- self.display_proc.start()
+ self.duration_s = self.numFrames / self.input_fps
+ self.expected_num_frames = int(self.duration_s * self.target_fps)
+ self.get_frameWH()
- if self.config.ENABLE_QUERYING:
- self._initialize_writer()
+ # Determine minimum contour size relative to frame resolution
+ # self.min_contour_area = int(
+ # (self.min_roi_h)
+ # * (self.min_roi_h)
+ # ) # 207
+
+ # self.dist_thresh_8k = max(
+ # self.config.ROI_DISTANCE_THRESH_RATIO * self.frame_width,
+ # self.config.ROI_DISTANCE_THRESH_RATIO * self.frame_height,
+ # )
+ # multiplier = 2.0 if self.device_input == "cpu" else 1.0
+ # self.dist_thresh_640 = (
+ # max(
+ # self.config.ROI_DISTANCE_THRESH_RATIO * self.resize_w,
+ # self.config.ROI_DISTANCE_THRESH_RATIO * self.resize_h,
+ # )
+ # * multiplier
+ # ) # 0.05 * self.resize_w
+ # self.scales_tensor = torch.tensor(
+ # [self.scale_x, self.scale_y, self.scale_x, self.scale_y],
+ # # device="cpu",
+ # device=self.device_input,
+ # )
- # Small delay to allow the reader's deque to populate
- time.sleep(0.1)
+ # self.disp_w, self.disp_h = self.config.DISPLAY_FRAME_SIZE
- # Start the producer and consumer threads
- if hasattr(self, "process_thread") and not self.process_thread.is_alive():
- self.process_thread.start()
+ # Performance Tracking
+ self.elapsed_display_time = 0.0
+ self.frame_count = 0 # Frame count for videos
+ self.frame_count_target = 0
+ self.last_delivered_frame_id = -1 # Track what was actually sent
+ self.last_frame_id = 0
+ self.latest_processed_frame = None
+ self.next_process_idx = 0.0
+ self.stat_fps = 0
+ self.stat_frame_count = 0
+ self.abs_frame_num = 0
+ self.total_objects_detected = 0
+ self.is_cuda = self.device.lower() == "gpu" and torch.cuda.is_available()
+ self.frame_in_clip_count = 0
- if (
- self.config.ENABLE_QUERYING
- and not self.config.TEST_MODE
- and not self.metadata_thread.is_alive()
- ):
- self.metadata_thread.start()
+ self.writer_done = True
- if self.config.ENABLE_QUERYING and not self.writer_thread.is_alive():
- self.writer_thread.start()
+ # Video Clipping
+ # self.video_writer = None
+ self.ffmpeg_proc = None # Replaces cv2.VideoWriter completely
+ # self.fourcc = cv2.VideoWriter_fourcc(*"mp4v") # avc1, mp4v
+ # self.clip_id = 0
+ # self.clip_filename = ""
+ self.clip_filename_pattern = f"{self.config.SHARED_OUTPUT}/{self.name}_%03d.mp4"
+ # self.clip_key = f"{self.name}_000.mp4"
+ # self.tmp_file = ""
+ # self.frame_in_clip_count = 0
- return self
+ if self.config.ENABLE_QUERYING:
+ self.gpu_float_staging = torch.empty(
+ (1, 3, self.frame_height, self.frame_width),
+ dtype=torch.float16,
+ device=self.device_index,
+ )
+ # Thread-safe queue for the resized frames (640x640)
+ # maxlen=300 allows for a 20-second buffer in case of extreme disk lag
+ # Non-blocking queue for frames and control signals
+ # self.write_queue = queue.Queue(maxsize=int(self.target_fps)) # 300)
+
+ # self.write_queue = queue.Queue(
+ # maxsize=int(self.target_fps / 2)
+ # if self.device_input == "cpu"
+ # else int(self.target_fps / 2)
+ # ) # 300)
+
+ # Define the depth of your shared memory ring buffer. 16 is a safe number.
+ self.clipper_ring_depth = 16
+ self.clipper_shm_frame_bytes = self.config.MODEL_W * self.config.MODEL_H * 3
+
+ # Create the shared memory blocks
+ self.clipper_shm_blocks = [
+ SharedMemory(create=True, size=self.clipper_shm_frame_bytes)
+ for _ in range(self.clipper_ring_depth)
+ ]
+
+ # Get the names to pass to the new process
+ self.clipper_shm_names = [shm.name for shm in self.clipper_shm_blocks]
+
+ # Create NumPy array views that the main process will use to write data
+ self.clipper_shm_np_views = [
+ np.ndarray(
+ (self.config.MODEL_H, self.config.MODEL_W, 3),
+ dtype=np.uint8,
+ buffer=shm.buf,
+ )
+ for shm in self.clipper_shm_blocks
+ ]
+
+ # An atomic, thread-safe counter to track the next available slot
+ self.clipper_ring_idx = mp.Value("i", 0)
+ self.clipper_idx_lock = (
+ mp.Lock()
+ ) # Lock to prevent race conditions when getting an index
+
+ self.write_queue = mp.Queue(maxsize=self.clipper_ring_depth * 2)
+ # if (
+ # not hasattr(self, "writer_process")
+ # or self.writer_process is None
+ # or not self.writer_process.is_alive()
+ # ):
+ # main_app_logger.info(
+ # " [CLIPPER-INIT] Target worker runtime thread is offline. Provisioning core consumer loop thread...",
+ # )
+ self.writer_process = mp.Process(
+ target=video_writer_core_loop,
+ args=(
+ self.write_queue,
+ self.clip_filename_pattern,
+ self.target_fps,
+ self.config.MODEL_W,
+ self.config.MODEL_H,
+ int(self.config.CLIP_DURATION),
+ # Pass the shared memory details to the new process
+ self.clipper_shm_names,
+ self.clipper_shm_frame_bytes,
+ ),
+ daemon=True,
+ )
+ # self.writer_process.start()
+ if not self.config.TEST_MODE:
+ self.send_metadata_queue = queue.Queue()
+ self.writer_done = False
+ self.stop_writer = threading.Event() if self.config.ENABLE_QUERYING else None
- def stop(self):
- """
- Comprehensive resource release. Safely drains the frame pipelines,
- forces a graceful FFmpeg flush to prevent 'moov atom' index corruption,
- and cleanly unlinks shared memory layers.
- """
- with self._stop_lock:
- if self._is_stopped:
- return # Already stopped by another thread
+ # self._cached_grid_y, self._cached_grid_x = torch.meshgrid(
+ # torch.arange(
+ # self.resize_h, device=self.device_input
+ # ), # Match target tracking resolution
+ # torch.arange(self.resize_w, device=self.device_input),
+ # indexing="ij",
+ # )
- # 1. Instantly pull out of active dashboards to stop inbound traffic
- if self.name in self.active_streams:
- self.active_streams.pop(self.name, None)
+ # self.fixed_inference_batch = torch.empty(
+ # (self.config.MODEL_MAX_BATCH_SIZE, 3, self.config.MODEL_H, self.config.MODEL_W),
+ # dtype=torch.half,
+ # device=self.device_input,
+ # )
+ # Create a deep look-ahead queue buffer that consumes near 0 RAM
+ # because it only holds reference pointers to your 6 SHM slots!
+ # if self.is_rtsp:
+ # prefetch_maxsize = 10
+ prefetch_maxsize = 5 # use 4 or 5; 10 risk memlock
+ # else:
+ # prefetch_maxsize = (
+ # 128 if self.device_input != "cpu" else int(self.target_fps)
+ # )
- self.active = False
- print(
- f"[STOP] Initiating graceful flush shutdown for {self.name}",
- flush=True,
- )
+ self.prefetch_queue = queue.Queue(
+ maxsize=prefetch_maxsize,
+ )
+ # self.prefetch_queue = mp.Queue(maxsize=128)
+ # Initialize a thread-safe signaling handle at class setup (run_realtime_inference)
+ self.queue_data_ready_event = threading.Event()
+ maxsize = (
+ self.prefetch_queue.maxsize
+ if hasattr(self.prefetch_queue, "maxsize")
+ else self.prefetch_queue._maxsize
+ )
+ # self.shm_buffer_pool = [None] * maxsize
+ try:
+ self.shm_buffer_pool = [
+ # (
+ # torch.empty(
+ # (self.frame_height, self.frame_width, 3), dtype=torch.uint8
+ # ).pin_memory(),
+ # None, # event
+ # 0, # frame_num
+ # 0, # abs_frame_num
+ # 0.0, # read_latency
+ # )
+ None
+ for _ in range(maxsize)
+ ]
+ except Exception as e_:
+ traceback.print_exc()
+ main_app_logger.info(f"[INITIALIZATION ERROR] Error occurred: {e_}")
+ self.gpu_input = [
+ torch.empty(
+ (self.frame_height, self.frame_width, 3),
+ device=self.device_input,
+ dtype=torch.uint8,
+ ) # .pin_memory()
+ for _ in range(maxsize)
+ ]
- # 2. Trigger your event handler flags
- if hasattr(self, "stop_writer") and self.stop_writer is not None:
- try:
- self.stop_writer.set()
- except Exception:
- pass
+ self.prefetch_active = True
- # 3. PHASE 1: UNBLOCK CONSUMER CORES (Poison Pill Deliveries First)
- if hasattr(self, "write_queue") and self.write_queue is not None:
+ self.prefetch_threads = []
+
+ # self.shm_buffer_pool = [None] * maxsize
+ # self.shm_buffer_pool = mp.Manager().list([None] * maxsize)
+ # global_shared_pool = mp.Manager().list([None] * maxsize)
+ # self.shm_buffer_pool = global_shared_pool
+
+ # Track active worker threads atomically to protect end-of-stream drainage
+ num_prefetch_workers = 1 # 4 if self.device_input == "cpu" else 3
+ self.active_workers_count = num_prefetch_workers
+ self.worker_tracking_lock = threading.RLock()
+ # self.obj_counter_l-ock = threading.Lock()
+
+ # Offload the pre-fetch layer straight to a daemon thread context
+ # prefetch_thread = threading.Thread(target=frame_prefetch_worker, daemon=True)
+ # prefetch_thread.start()
+ # num_prefetch_workers = 1 if self.device_input == "cpu" else 3
+ # num_prefetch_workers = 3
+
+ # Override the atomic tracking count to match our active pool allocation
+ with self.worker_tracking_lock:
+ self.active_workers_count = num_prefetch_workers
+
+ # This guarantees frames are read from self.reader sequentially
+ # and stay perfectly in order, while letting workers decode in parallel!
+ self.reader_lock = threading.RLock()
+
+ # --- PRE-ALLOCATED ZERO-COPY HARDWARE RING WORKSPACE ---
+ self.ring_depth = 4 # 8 # _async_clipper_worker
+ self.gpu_ring_idx = 0 # _async_clipper_worker
+ self.cpu_ring_idx = 0 # _async_clipper_worker
+
+ self.pinned_matrices = [] # _async_clipper_worker, video_writer_core_loop (CPU)
+ if self.device_input == "cuda":
+ self.pinned_tensors = [] # _async_clipper_worker
+
+ self.static_gpu_byte_bchw = torch.empty(
+ (1, 3, self.disp_h, self.disp_w),
+ dtype=torch.uint8,
+ device="cuda",
+ ).contiguous()
+
+ # Pre-allocate 640x640 workspace footprint across CPU and GPU spaces
+ for _ in range(
+ self.ring_depth
+ ): # _async_clipper_worker, video_writer_core_loop
+ mat = np.zeros((self.resize_h, self.resize_w, 3), dtype=np.uint8)
+ if self.device_input == "cuda":
try:
- # Clear out pending frame backlogs to speed up shutdown execution
- while not self.write_queue.empty():
- try:
- self.write_queue.get_nowait()
- except Exception:
- break
- # Dispatch clean poison pill token to release the consumer thread
- self.write_queue.put(None)
- except Exception:
+ cv2.cuda.registerPageLocked(mat)
+ except cv2.error:
pass
+ self.pinned_tensors.append(torch.from_numpy(mat))
- if hasattr(self, "render_queue") and self.render_queue is not None:
+ self.pinned_matrices.append(mat)
+
+ # self.init_pipeline_shared_memory()
+
+ # if self.device_input == "cuda" and not hasattr(self, "_cuda_gaussian_filter"):
+ # # Large block blur to dissolve high-frequency single-pixel speckles
+ # # ksize=(15, 15)
+ # # ksize=(11, 11)
+ # ksize = (17, 17)
+ # self._cuda_gaussian_filter = cv2.cuda.createGaussianFilter(
+ # srcType=cv2.CV_8UC1, dstType=cv2.CV_8UC1, ksize=ksize, sigma1=0
+ # )
+
+ # if not hasattr(self, "_filter_scratch_keep"):
+ # # Existing order and boolean validation maps
+ # self._filter_scratch_keep = torch.zeros(
+ # (self.max_cached_elements,), dtype=torch.bool, device=self.device_input
+ # )
+ # self._filter_order_scratch = torch.zeros(
+ # (self.max_cached_elements,), dtype=torch.long, device=self.device_input
+ # )
+
+ # # Persistent coordinate layers to absorb inner-loop tensor evaluations safely
+ # self._filter_scratch_x1 = torch.zeros(
+ # (self.max_cached_elements,), dtype=torch.float32, device=self.device_input
+ # )
+ # self._filter_scratch_y1 = torch.zeros(
+ # (self.max_cached_elements,), dtype=torch.float32, device=self.device_input
+ # )
+ # self._filter_scratch_x2 = torch.zeros(
+ # (self.max_cached_elements,), dtype=torch.float32, device=self.device_input
+ # )
+ # self._filter_scratch_y2 = torch.zeros(
+ # (self.max_cached_elements,), dtype=torch.float32, device=self.device_input
+ # )
+ # self._filter_scratch_ioa = torch.zeros(
+ # (self.max_cached_elements,), dtype=torch.float32, device=self.device_input
+ # )
+
+ # Isolate CUDA tasks using a dedicated stream and independent hardware completion barriers
+ self.processing_stream = (
+ torch.cuda.Stream() if self.device_input == "cuda" else None
+ )
+
+ # Cache the context manager instance persistently
+ self.compiled_no_grad_gate = torch.no_grad()
+
+ # Pre-allocated hardware events guarantee completely non-blocking stream isolation
+ self.slot_events = (
+ [torch.cuda.Event() for _ in range(8)]
+ if self.device_input == "cuda"
+ else None
+ )
+
+ # Allocate a permanent float32/float16 channel layout space directly on VRAM
+ if self.device_input == "cuda":
+ self.static_gpu_360p = torch.empty(
+ (1, 3, self.disp_h, self.disp_w),
+ dtype=torch.float32,
+ device=self.device_input,
+ )
+ self.static_gpu_byte_bchw = torch.empty(
+ (1, 3, self.disp_h, self.disp_w),
+ dtype=torch.uint8,
+ device=self.device_input,
+ ).contiguous()
+
+ def setup_reader(self, target_fps, clip_duration, startup_event=None):
+ # if hasattr(self, "reader"):
+ # del self.reader
+
+ self.gpu_id = 0
+
+ # try:
+ determined_queue_size = (
+ # 4 if self.is_rtsp else 2 # (0 if self.config.TEST_MODE else 2)
+ # 4 if self.is_rtsp else 8
+ 4 # max: 5
+ )
+ if self.device_input == "cuda": # and not self.is_rtsp:
+ # Add a tiny sleep or garbage collect to ensure the GPU handle is released
+ # gc.collect()
+ # torch.cuda.empty_cache() # Clear any remaining context
+
+ from include.readers import GPUReader
+
+ self.reader = GPUReader(
+ source=self.source,
+ startup_event=startup_event,
+ target_fps=target_fps,
+ clip_duration=clip_duration,
+ gpu_id=0, # self.gpu_id,
+ queue_size=determined_queue_size,
+ )
+ else:
+ from include.readers import CPUReader
+
+ self.reader = CPUReader(
+ source=self.source,
+ startup_event=startup_event,
+ target_fps=target_fps,
+ clip_duration=clip_duration,
+ queue_size=determined_queue_size,
+ )
+ # except Exception as e:
+ # self.reader = None
+ # raise ValueError(f"Stream reader initialization failure: {e}")
+
+ def setup_shared_memory(self):
+ self.manager = mp.Manager()
+
+ self.shms = []
+ shm_names = []
+ num_shms = 3
+
+ # Shared Integer to track which buffer is "Ready" for the UI
+ # 'i' for integer, initialized to 0
+ self.ready_buffer_idx = mp.Value("i", 0)
+ self.reader_active_idx = mp.Value("i", -1)
+ self.shm_frame_lengths = mp.Array("i", [0 for _ in range(num_shms)])
+ self.mp_last_id = mp.Value("i", -1)
+ self.mp_frame_ready_flag = mp.Value("b", False)
+
+ # --- HIGH-SPEED ATOMIC TELEMETRY BACKLOG REGISTERS ---
+ self.render_queue_backlog_counter = mp.Value("i", 0)
+ self.write_queue_backlog_counter = mp.Value("i", 0)
+
+ display_timestamp = int(time.time_ns())
+
+ for idx in range(num_shms):
+ # self.shm = mp.shared_memory.SharedMemory(create=True, size=10*1024*1024)
+ shm_name = f"shm_{self.name}_{idx}_{os.getpid()}_{display_timestamp}"
+ if shm_name not in shm_names:
try:
- while not self.render_queue.empty():
- try:
- self.render_queue.get_nowait()
- except Exception:
- break
- self.render_queue.put_nowait(None)
- except Exception:
+ SharedMemory(name=shm_name).unlink()
+ except FileNotFoundError:
pass
+ # main_app_logger.info(f"[DEBUG]: Setting up SHM {shm_name}")
+ try:
+ shm = SharedMemory(
+ name=shm_name, create=True, size=10 * 1024 * 1024
+ )
+ # safe_unregister_shm(shm.name)
+ except FileExistsError:
+ # Attach to existing memory
+ shm = SharedMemory(name=shm_name)
+ # safe_unregister_shm(shm.name)
+ except Exception as e:
+ main_app_logger.error(f"Failed to initialize shared memory: {e}")
+ raise
+ self.shms.append(shm)
+ shm_names.append(shm_name)
+
+ # try:
+ # unregister(shm._name, "shared_memory")
+ # # main_app_logger.info(f"[HANDLER] Successfully unregistered {shm.name} from Resource Tracker.")
+ # except Exception:
+ # pass
+
+ self.signal_queue = self.manager.Queue(maxsize=32)
+ self.render_queue = self.manager.Queue(maxsize=10) # old-5
+
+ self.shared_details = self.manager.dict()
+ self.shared_details["shm_names"] = shm_names
+ # self.shared_details["buffer_idx"] = 0
+ # self.shared_details["frame_length"] = [0 for _ in range(num_shms)]
+ self.shared_details["last_id"] = -1
+
+ # PIPELINE FUNCTIONS --------------------------------------------
+
+ def update_frame(self, stat_start_time):
+ # if self.device_input == "cuda":
+ # torch.cuda.synchronize()
+
+ self.stat_frame_count += 1
+ self.elapsed_display_time += time.perf_counter() - stat_start_time
+ # self.elapsed_display_time = time.perf_counter() - stat_start_time
+ # if elapsed > 0.5:
+ self.stat_fps = round(self.stat_frame_count / self.elapsed_display_time, 1)
+
+ @torch.inference_mode()
+ def pipeline_fn(
+ self,
+ device_frame,
+ overall_frame_num,
+ stat_start_time,
+ current_clip_id,
+ gt_boxes=None,
+ read_frame_only=False,
+ ):
+ global all_metadata
+ current_clip_key = f"{self.name}_{current_clip_id:03d}.mp4"
+ current_clip_path = f"{self.config.SHARED_OUTPUT}/{current_clip_key}"
+
+ metadata = {}
+ motion_detected = False
+ metrics = {"sf_time": 0, "roi_time": 0, "det_time": 0, "bbs": None}
- # 4. PHASE 2: GRACEFUL FFmpeg DEFLATION GATE (Bypasses Moov Issues)
- if hasattr(self, "ffmpeg_proc") and self.ffmpeg_proc is not None:
+ # Given input frame, get metadata and metrics
+ try:
+ if read_frame_only:
+ # Get frame, metadata and metrics
+ _, det_frame = self.processor.format_bbs_and_frame_4_detection(
+ [], device_frame
+ )
+
+ else:
+ # --- CLIP GENERATION ---
+ if self.config.ENABLE_QUERYING:
+ if (
+ self.config.DEBUG_FLAG
+ and hasattr(self, "max_frames_per_clip")
+ and (
+ self.frame_in_clip_count % 15 == 0
+ or self.frame_in_clip_count == 1
+ )
+ ):
+ main_app_logger.info(
+ f"[CLIPPER] Frame progress tracking index: {self.frame_in_clip_count}/{self.max_frames_per_clip} (Overall Frame: {overall_frame_num})"
+ )
+
+ self.prep_frame_for_video(device_frame, overall_frame_num)
+
+ if self.frame_in_clip_count > self.max_frames_per_clip:
+ global clip_completion_tracker
+ if current_clip_key not in clip_completion_tracker:
+ clip_completion_tracker[current_clip_key] = {
+ "video": False,
+ "meta": False,
+ "start_time": time.time(),
+ }
+
+ clip_completion_tracker[current_clip_key]["meta"] = True
+ # main_app_logger.info(
+ # f" [BARRIER-SEAL] All metadata extracted for {current_clip_key}. Evaluating convergence...",
+ #
+ # )
+ self._evaluate_barrier_and_dispatch(
+ current_clip_key,
+ current_clip_path,
+ self.resize_w,
+ self.resize_h,
+ )
+
+ self.start_new_clip(current_clip_id)
+
+ # Run model on frame, get metadata and metrics
+ with torch.inference_mode():
+ metrics, metadata, det_frame, motion_detected = self.processor.run(
+ device_frame,
+ overall_frame_num,
+ frame_in_clip_count=self.frame_in_clip_count,
+ gt_boxes=None,
+ )
+
+ num_objs = len(metadata.keys())
+ self.total_objects_detected += num_objs
+
+ if self.config.DEBUG_FLAG:
+ meta_keys = ", ".join(list(metadata.keys()))
+ main_app_logger.info(
+ f"[DEBUG] {current_clip_key} metadata keys: {meta_keys}",
+ )
+
+ if num_objs > 0:
+ if current_clip_key not in all_metadata:
+ all_metadata[current_clip_key] = {"object": {}, "face": {}}
+
+ all_metadata[current_clip_key]["object"].update(metadata)
+
+ # data_to_draw = metadata
+ if self.config.DEBUG_FLAG and overall_frame_num % 50 == 0:
+ main_app_logger.info(
+ f"[METADATA-DEBUG A] Logged tracking structures for Frame #{overall_frame_num} into dictionary key: {current_clip_key}. Number detections: {self.total_objects_detected}",
+ )
+
+ if self.config.TEST_MODE:
+ self.frame2video(
+ det_frame,
+ overall_frame_num,
+ metadata,
+ getattr(self, "label_source", None),
+ stat_start_time,
+ )
+ else:
+ self.frame2output(
+ det_frame,
+ overall_frame_num,
+ metadata,
+ self.label_source,
+ stat_start_time,
+ )
+
+ # with self.obj_counter_lock:
+ # self.num_objs += len(metadata.keys())
+ # except (TypeError, IndexError):
+ # # App is shutting down and processor/reader not available
+ # self.active = False
+
+ except Exception as e_detection:
+ # traceback.print_exc()
+ # if self.active:
+ traceback.print_exc()
+ main_app_logger.info(f"[DETECTION ERROR] Exception: {e_detection}")
+ # self.active = False
+
+ finally:
+ # del inf_data, bbs_full_res, device_frame, det_frame
+ if "det_frame" in locals():
+ del det_frame
+ if "device_frame" in locals():
+ del device_frame
+ # if "bbs_full_res" in locals():
+ # del bbs_full_res
+ # if "inf_data" in locals():
+ # del inf_data
+
+ return metadata, metrics # Skip full detection pass
+
+ def initialize_run_realtime_inference(self, read_frame_only, num_prefetch_workers):
+ # self.duration_target = 30
+ # self.status = "RUNNING"
+ self.processor = self._setup_processor(read_frame_only)
+
+ if hasattr(self, "processor") and hasattr(self.processor, "label_source"):
+ self.label_source = self.processor.label_source
+ # Setup empty list to track background futures natively
+ # self._active_inference_futures = []
+
+ if self.is_cuda and not hasattr(self, "sf_start") and self.config.TEST_MODE:
+ self.sf_start, self.sf_end = (
+ torch.cuda.Event(enable_timing=True),
+ torch.cuda.Event(enable_timing=True),
+ )
+ self.roi_start, self.roi_end = (
+ torch.cuda.Event(enable_timing=True),
+ torch.cuda.Event(enable_timing=True),
+ )
+ self.det_start, self.det_end = (
+ torch.cuda.Event(enable_timing=True),
+ torch.cuda.Event(enable_timing=True),
+ )
+
+ if self.config.TEST_MODE:
+ self.component_stats = {
+ "sf": [],
+ "roi": [],
+ "det": [],
+ "dma_upload": [], # Track PCIe Transfer Latencies
+ "queue_blocked": [], # Track GIL Serialization stalls
+ "batch_sizes": [], # Track Smart Filtering density
+ "thread_backlog": [], # Track thread work pool backlog
+ }
+ self.crops_per_frame_list = []
+
+ # Bind active execution permanently to sub-stream BEFORE entering loop (Bypasses TLS Overhead)
+ # if self.device_input == "cuda" and torch.cuda.is_available():
+ # torch.cuda.set_stream(self.inference_stream)
+ if hasattr(self, "target_fps") and hasattr(self, "duration_target"):
+ self.max_target_frames = int(self.duration_target * float(self.target_fps))
+ elif hasattr(self, "numFrames"):
+ self.max_target_frames = self.numFrames
+ else:
+ self.max_target_frames = float("inf")
+
+ # self.dynamic_limit = max(2, int(0.5 * self.target_fps))
+
+ # Initialize a thread-safe signaling handle at class setup (run_realtime_inference)
+ self.queue_data_ready_event = threading.Event()
+
+ try:
+ for _ in range(num_prefetch_workers):
+ # prefetch_thread = mp.Process(
+ prefetch_thread = threading.Thread(
+ target=lambda: global_frame_prefetch_worker_v1(self),
+ daemon=True,
+ )
+ prefetch_thread.start()
+ self.prefetch_threads.append(prefetch_thread)
+ except Exception as e:
+ main_app_logger.critical(
+ f"[{self.name}] Failed to start prefetch workers: {e}", exc_info=True
+ )
+ traceback.print_exc()
+ raise
+
+ @torch.inference_mode()
+ def run_realtime_inference(
+ self,
+ sf_enabled=True,
+ profiler=None,
+ gt_enabled=False,
+ read_frame_only=False,
+ ):
+ self.status = "RUNNING"
+
+ self.initialize_run_realtime_inference(
+ read_frame_only, self.active_workers_count
+ )
+
+ self.stat_start_time = time.perf_counter()
+ # pipeline_start_time = self.stat_start_time
+ # last_loop_cycle_timestamp = self.stat_start_time
+
+ missing_frame_cnt = 0
+ max_retries = int(self.target_fps)
+ # Release the master startup event at the last possible moment.
+ self.main_startup_event.set()
+ while (
+ self.active # or not self.prefetch_queue.empty()
+ ): # and self.frame_count_target < self.max_target_frames:
+ stat_start_time = time.perf_counter()
+ try:
+ # FRAME RETRIEVAL ---------------------------------------------
try:
- print(
- "[STOP] Closing video pipeline write handles to flush metadata...",
- flush=True,
+ safe_frame, frame_details, should_continue = (
+ self._get_frame_from_queue()
)
- if self.ffmpeg_proc.stdin:
- self.ffmpeg_proc.stdin.close() # Safely alerts FFmpeg to finalize files
- if self.ffmpeg_proc.stderr:
- self.ffmpeg_proc.stderr.close() # Instantly forces readline() to return None and exits thread safely
+ if not should_continue:
+ self.active = False # Signal loop termination
+ break
+ if safe_frame is None: # and self.is_rtsp:
+ missing_frame_cnt += 1
+
+ if missing_frame_cnt >= max_retries:
+ self.active = False # Signal loop termination
+ main_app_logger.info("Too many frames missing. Exiting ...")
+ break
+ continue # Skip to the next iteration if the frame is invalid
+
+ # try:
+ # # ret, slot_idx = self.prefetch_queue.get(block=True, timeout=1.0)
+ # # ret, slot_idx = self.prefetch_queue.get(block=False)
+ # ret, slot_idx = self.prefetch_queue.get(block=True, timeout=1.0)
+ # self.prefetch_queue.task_done()
+
+ # except queue.Empty:
+ # with self.worker_tracking_lock:
+ # if (
+ # self.active_workers_count == 0
+ # and self.prefetch_queue.empty()
+ # ):
+ # self.active = False
+ # break
+ # # If the queue is empty AND the prefetch workers are done, we can exit.
+ # if not self.active:
+ # break
+ # continue
+
+ # # If self.stop() drops the queues, break out of the thread loop natively.
+ # except (OSError, ValueError, AssertionError):
+ # main_app_logger.info(
+ # "[PROCESS THREAD] Ingestion queue disconnected via stop() signal. Breaking loop."
+ # )
+ # self.active = False
+ # break
+
+ # if ret is False or slot_idx == "END_OF_STREAM":
+ # main_app_logger.info(
+ # "[PROCESS THREAD] End of video stream detected. Breaking loop naturally."
+ # )
+ # self.active = False
+ # break
+
+ # if not self.active:
+ # break
+
+ # # if slot_idx == -1 or self.shm_buffer_pool[slot_idx] is None:
+ # # continue
+ # if (
+ # slot_idx == -1
+ # or not hasattr(self, "shm_buffer_pool")
+ # or self.shm_buffer_pool is None
+ # ):
+ # continue
+ # if (
+ # slot_idx >= len(self.shm_buffer_pool)
+ # or self.shm_buffer_pool[slot_idx] is None
+ # ):
+ # continue
+
+ # # Zero-Copy Reference Extraction straight out of the memory slot array
+ # (
+ # raw_shm_frame,
+ # current_event,
+ # frame_num,
+ # abs_frame_num,
+ # true_read_latency_secs,
+ # ) = self.shm_buffer_pool[slot_idx]
+
+ # if frame_num == 0:
+ # main_app_logger.info(
+ # f"[VERIFY - CONSUMER] Main loop is officially processing Frame {frame_num}!",
+ # )
+
+ # if "cuda" in str(self.device_input):
+ # safe_frame = self.gpu_input[slot_idx]
+ # # safe_frame = cpu_tensor.to(self.device_input, non_blocking=True)
+ # safe_frame.copy_(raw_shm_frame)
+ # # self.gpu_input[slot_idx].zero_()
+ # else:
+ # safe_frame = raw_shm_frame
+
+ # # Record the PyTorch CUDA event on the current stream
+ # if current_event is not None and isinstance(
+ # current_event, torch.cuda.Event
+ # ):
+ # # This tells the GPU that the consumer is done reading this slot
+ # current_event.record(torch.cuda.current_stream())
+
+ # # Originally 0-based but make 1-based
+ # frame_num += 1
+ # abs_frame_num += 1
+ # self.abs_frame_num = abs_frame_num
+ missing_frame_cnt = 0
+
+ except queue.Empty:
+ if getattr(self.reader, "reconnect_failed", False):
+ self.active = False
+ break
+ if not self.active:
+ break # Exit on error if shutdown has been initiated
+ time.sleep(0.001)
+ continue
+
+ # FRAME PROCESSING ---------------------------------------------
+ # stat_start_time = self.stat_start_time
+ self.abs_frame_num = frame_details["abs_frame_num"]
+ calculated_clip_id = (
+ frame_details["frame_num"] - 1
+ ) // self.max_frames_per_clip
+ self.frame_count += 1
+ self.frame_count_target += 1
+ self.frame_in_clip_count += 1
+
+ # RUN PIPELINE_FN ---------------------------------------------
+ # run_pipelinefn_start = time.perf_counter()
+ # self.frame_count_target += 1
+ # self.frame_in_clip_count += 1
+
+ # if gt_enabled:
+ # target_boxes_array = self.get_frame_gt_boxes(
+ # abs_frame_num, gt_sequence, gt_boxes
+ # )
+
+ metadata_or_bbs, metrics = self.pipeline_fn(
+ safe_frame,
+ frame_details["frame_num"],
+ stat_start_time,
+ calculated_clip_id,
+ read_frame_only=read_frame_only,
+ )
+
+ # # FRAME POST PROCESSING ---------------------------------------------
+
+ # Explicitly delete frame variables to free their references
+ frame_8k = None
+ del frame_8k, safe_frame
+ if "metadata_or_bbs" in locals():
+ del metadata_or_bbs
+
+ if self.frame_count % 100 == 0:
+ gc.collect()
+
+ # Force a microscopic micro-yield if executing on CPU.
+ # This grants immediate execution priority back to the background cleaning thread,
+ # allowing the garbage collector to evict data from RAM instantly!
+ if self.device_input == "cpu":
+ time.sleep(
+ 0
+ ) # .005) # 1ms yield breaks core processor starvation lock
+
+ # END -> frame processing
+ except torch.cuda.OutOfMemoryError:
+ main_app_logger.info("!" * 70)
+ main_app_logger.info(
+ "[CRITICAL TEST CRASH] GPU MEMORY CEILING HIT INSIDE RUNNER LOOP!"
+ )
+ main_app_logger.info(
+ "Freezing allocation history registers and writing diagnostic log..."
+ )
+ main_app_logger.info("!" * 70)
- # Grant a soft 5-second window for storage layer disk synchronization
- self.ffmpeg_proc.wait(timeout=5.0)
- print(
- " [STOP] FFmpeg closed cleanly with valid indexing atoms.",
- flush=True,
+ try:
+ snapshot_filename = (
+ f"/tmp/test_vram_leak_profile_pid{os.getpid()}.pickle"
)
- except subprocess.TimeoutExpired:
- print(
- " [STOP-WARN] Video flush timed out. Forcing hard termination.",
- flush=True,
+ torch.cuda.memory._dump_snapshot(snapshot_filename)
+ main_app_logger.info(
+ f"[PROFILER SUCCESSFUL] Snapshot profile written to: {snapshot_filename}"
)
- try:
- self.ffmpeg_proc.kill()
- except Exception:
- pass
- except Exception as io_err:
- print(
- f" [STOP-WARN] Error during streaming flush: {io_err}",
- flush=True,
+ main_app_logger.info(
+ "--> Drag and drop this file directly into: https://pytorch.org"
+ )
+ except Exception as dump_err:
+ main_app_logger.info(
+ f"Failed to record profile data snapshot: {dump_err}"
+ )
+
+ # Force safe system unlinking of background workers to clean up OS handles
+ self.active = False
+ if hasattr(self, "reader") and self.reader is not None:
+ self.reader.stop()
+ raise
+
+ except Exception as e:
+ if not self.active or not getattr(self, "prefetch_active", True):
+ main_app_logger.info(
+ "[PROCESS THREAD] System shutdown detected during exception sweep. Exiting thread payload context."
+ )
+ break
+
+ main_app_logger.info(
+ f"[CRITICAL PIPELINE ERROR] Crash on frame: {repr(e)}"
+ )
+ traceback.print_exc()
+ raise e # Let it break so you can see the exact line number!
+
+ # END -> while self.active and self.frame_count_target < self.max_target_frames:
+
+ # PIPELINE POST PROCESSING ---------------------------------------------
+ # )
+
+ self.async_writer.release()
+
+ # CALCULATE PERFORMANCE METRICS ---------------------------------------------
+ main_app_logger.info(
+ f"Execution Finished. Total Output Frames Written: {self.frame_count_target}"
+ )
+
+ # Force early hardware driver sweep before unbinding threads
+ if self.device_input == "cuda" and torch.cuda.is_available():
+ torch.cuda.synchronize()
+ torch.cuda.empty_cache()
+ torch.cuda.ipc_collect()
+
+ gc.collect()
+
+ if profiler is not None:
+ profiler.stop()
+
+ self.stop()
+
+ def get_overlay(
+ self,
+ display_frame,
+ metadata_or_bbs,
+ class_list,
+ scale_display_x,
+ scale_display_y,
+ color=(0, 0, 255),
+ ):
+ if isinstance(metadata_or_bbs, dict):
+ detached_bbs = {
+ k: (v.detach() if torch.is_tensor(v) else v)
+ for k, v in metadata_or_bbs.items()
+ }
+ # Object Mode (YOLO Structs)
+ display_frame = get_metadata_overlay(
+ display_frame,
+ detached_bbs,
+ class_list,
+ (scale_display_x, scale_display_y),
+ (self.disp_w, self.disp_h),
+ is_bgr=True,
+ )
+
+ elif metadata_or_bbs is not None:
+ # # Motion / Smart Filtering Overlay Path
+ display_frame = get_bb_overlay(
+ display_frame,
+ metadata_or_bbs,
+ (scale_display_x, scale_display_y),
+ (self.disp_w, self.disp_h),
+ color=color,
+ )
+
+ return display_frame
+
+ def _get_frame_from_queue(self):
+ """
+ Retrieves the next available frame data from the shared memory prefetch queue.
+
+ This helper encapsulates the logic for handling an empty queue, stream termination
+ signals, and invalid data slots, making the main processing loop cleaner.
+
+ Returns:
+ A tuple containing:
+ - (torch.Tensor or None): The frame tensor if successful, otherwise None.
+ - (dict or None): A dictionary with frame metadata (frame_num, event), or None.
+ - (bool): A flag indicating if the processing loop should continue.
+ """
+ try:
+ # Block for a short timeout to wait for a frame
+ ret, slot_idx = self.prefetch_queue.get(block=True, timeout=1.0)
+ self.prefetch_queue.task_done()
+ except queue.Empty:
+ # If the queue is empty, check if workers are still active before deciding to exit.
+ with self.worker_tracking_lock:
+ if self.active_workers_count == 0 and self.prefetch_queue.empty():
+ main_app_logger.info(
+ "[HELPER] Prefetch queue is empty and all workers are done. Signaling stop."
)
- finally:
- self.ffmpeg_proc = None
- self.video_writer = None
-
- # 5. PHASE 3: SHUTDOWN MULTIPROCESSING DAEMONS
- for proc_attr in ["render_proc", "ai_proc"]:
- proc = getattr(self, proc_attr, None)
- if proc is not None:
+ return None, None, False # Stop
+ return None, None, True # Continue waiting
+ except (OSError, ValueError):
+ # The queue was likely closed by self.stop(), signal to exit.
+ main_app_logger.info("[HELPER] Prefetch queue was closed. Signaling stop.")
+ return None, None, False # Stop
+
+ # Check for end-of-stream signal or invalid data
+ if not ret or slot_idx == "END_OF_STREAM":
+ main_app_logger.info("[HELPER] Received end-of-stream signal.")
+ return None, None, False # Stop
+
+ if (
+ slot_idx == -1
+ or not hasattr(self, "shm_buffer_pool")
+ or self.shm_buffer_pool is None
+ or slot_idx >= len(self.shm_buffer_pool)
+ or self.shm_buffer_pool[slot_idx] is None
+ ):
+ return None, None, True # Continue, skip this invalid slot
+
+ # --- Zero-Copy Reference Extraction ---
+ (
+ raw_shm_frame,
+ current_event,
+ frame_num,
+ abs_frame_num,
+ true_read_latency_secs,
+ ) = self.shm_buffer_pool[slot_idx]
+ # self.shm_buffer_pool[slot_idx] = None # Immediately clear the slot
+
+ if frame_num == 0:
+ main_app_logger.info(
+ f"[VERIFY - CONSUMER] Main loop is officially processing Frame {frame_num}!",
+ )
+
+ # Convert numpy array from shared memory to a torch tensor
+ # cpu_tensor = torch.from_numpy(raw_shm_frame) if isinstance(raw_shm_frame, np.ndarray) else raw_shm_frame
+
+ # # Move to the correct device (GPU or CPU)
+ # if "cuda" in str(self.device_input):
+ # safe_frame = cpu_tensor.to(self.device_input, non_blocking=True)
+ # else:
+ # safe_frame = cpu_tensor
+
+ # if "cuda" in str(self.device_input):
+ # safe_frame = self.gpu_input[slot_idx]
+ # # safe_frame = cpu_tensor.to(self.device_input, non_blocking=True)
+ # safe_frame.copy_(raw_shm_frame)
+ # # self.gpu_input[slot_idx].zero_()
+ # else:
+ safe_frame = raw_shm_frame
+
+ # Record the CUDA event to signal that the consumer is done with this slot
+ if current_event and isinstance(current_event, torch.cuda.Event):
+ current_event.record(torch.cuda.current_stream())
+
+ frame_details = {
+ "frame_num": frame_num + 1,
+ "abs_frame_num": abs_frame_num + 1,
+ "event": current_event,
+ "reader_time": true_read_latency_secs,
+ }
+
+ # Explicitly delete intermediate tensors
+ del raw_shm_frame
+
+ return safe_frame, frame_details, True
+
+ def _setup_processor(self, read_frame_only):
+ """
+ Selects and initializes the appropriate object detector based on the configuration.
+
+ This helper encapsulates the logic for creating the AI model (`processor`),
+ handling potential initialization errors gracefully.
+
+ Args:
+ read_frame_only (bool): If True, no processor is initialized as frames are only being read.
+
+ Returns:
+ An initialized detector instance (e.g., SmartFilteringObjectDetector) or None.
+ """
+ # If we are only reading frames, no AI model is needed.
+ # if read_frame_only:
+ # return None
+
+ # Determine the correct detector class based on the smart filtering configuration.
+ DetectorClass = (
+ SmartFilteringObjectDetector
+ if self.config.sf_enabled
+ else GeneralObjectDetector
+ )
+ detector_name = DetectorClass.__name__
+ main_app_logger.info(f"[{self.name}] Initializing processor: {detector_name}")
+
+ try:
+ # Common configuration for all detectors.
+ debug_frame_limit = (
+ self.config.DEBUG_FRAME_LIMIT if self.config.DEBUG_FLAG else -1
+ )
+
+ processor = DetectorClass(
+ config=self.config,
+ device=self.device_input,
+ timer_enabled=True,
+ resize_hw=(self.resize_h, self.resize_w),
+ frame_hw=(self.frame_height, self.frame_width),
+ target_fps=self.target_fps,
+ result_dir=self.result_dir,
+ run_name=self._testMethodName,
+ debug_frame_limit=debug_frame_limit,
+ )
+ return processor
+ except Exception as e:
+ # If the model fails to load, log the error and halt the pipeline.
+ main_app_logger.critical(
+ f"[{self.name}] CRITICAL: Failed to initialize AI processor '{detector_name}'. Error: {e}",
+ exc_info=True,
+ )
+ # Propagate the error to stop the handler from starting.
+ raise
+
+ def frame2video(
+ self,
+ device_frame,
+ frameNum,
+ metadata_or_bbs,
+ class_list,
+ stat_start_time,
+ gt_boxes=None,
+ ):
+ """
+ Frame is formated and added to video
+ metadata_or_bbs: Expected to be in resize dimensions (640x640)
+ """
+
+ # Factor for converting 640 bb dimensions to display dimensions
+ scale_display_x = self.disp_w / self.resize_w # 640
+ scale_display_y = self.disp_h / self.resize_h # 640
+
+ if self.device_input == "cuda": # and torch.is_tensor(device_frame):
+ with torch.inference_mode():
+ # 2. Fast Inline VRAM Downscaling into our static memory slot
+ # Reshape to (Batch, Channel, Height, Width) seamlessly without duplicating data
+ gpu_tensor = device_frame[None].permute(0, 3, 1, 2)
+
+ resized_tensor = torch.nn.functional.interpolate(
+ gpu_tensor,
+ size=(self.disp_h, self.disp_w),
+ mode="nearest",
+ )
+
+ # Using copy_() avoids creating any runtime allocations or memory-stride drift.
+ self.static_gpu_byte_bchw.copy_(resized_tensor)
+
+ # Safely reshape the pre-allocated, locked GPU memory layout back to standard HWC array topology.
+ # Since the underlying array layout is strictly contiguous, this permute is guaranteed
+ # to be zero-copy on the GPU and free from memory race stalls.
+ bgr_contiguous = self.static_gpu_byte_bchw.squeeze(0).permute(1, 2, 0)
+
+ # 1. Grab the current free pinned slot tracking variables from your reader
+ d2h_idx = self.reader.d2h_selector
+ pinned_tensor_buf = self.reader.d2h_buffers[d2h_idx]
+
+ # # Synchronize only the side download stream layout channel
+ # self.reader.download_stream.synchronize()
+ # Initialize a static download sync event on setup_context if not present
+ # if not hasattr(self, "_d2h_fence"):
+ # self._d2h_fence = torch.cuda.Event()
+
+ with torch.cuda.stream(self.reader.download_stream):
+ pinned_tensor_buf.copy_(
+ bgr_contiguous,
+ non_blocking=True,
+ )
+ # self._d2h_fence.record()
+
+ # self._d2h_fence.synchronize()
+
+ cpu_360p_frame = self.reader.d2h_numpys[d2h_idx]
+
+ # 5. Non-Blocking Push straight to your background AsyncVideoWriter thread pool
+ # We pass a direct .copy() slice so the hot loop can instantly reuse the pinned ring buffer
+ # display_frame = np.array(cpu_360p_frame, copy=True, order="C")
+ display_frame = np.asarray(cpu_360p_frame, order="C")
+ if display_frame.shape[-1] == 3:
+ display_frame = cv2.cvtColor(display_frame, cv2.COLOR_RGB2BGR)
+
+ # --- Draw Detection Overlays ---
+ if gt_boxes is not None:
+ # Factor for converting 8K (original) bb dimensions to display dimensions
+ scale_display_ox = self.disp_w / self.frame_width
+ scale_display_oy = self.disp_h / self.frame_height
+
+ display_frame = self.get_overlay(
+ display_frame,
+ gt_boxes,
+ None,
+ scale_display_ox,
+ scale_display_oy,
+ color=(0, 255, 0),
+ )
+
+ display_frame = self.get_overlay(
+ display_frame,
+ metadata_or_bbs,
+ class_list,
+ scale_display_x,
+ scale_display_y,
+ )
+
+ self.async_writer.write_frame(display_frame)
+ self.update_frame(stat_start_time)
+ # self.canvas_selector = 1 - self.canvas_selector
+ self.reader.d2h_selector = 1 - self.reader.d2h_selector
+
+ if (
+ self.config.DEBUG_FLAG
+ and self.config.TEST_MODE
+ and frameNum <= self.config.DEBUG_FRAME_LIMIT
+ ):
+ stage_debug_dir = (
+ self.result_dir / "debug_stages" / self._testMethodName / "display"
+ )
+ stage_debug_dir.mkdir(parents=True, exist_ok=True)
+ cv2.imwrite(
+ # str(stage_debug_dir / f"frame_{f_num:04d}_stage6_threshold.jpg"),
+ str(stage_debug_dir / f"frame_{frameNum:04d}_final_frame.jpg"),
+ display_frame,
+ )
+
+ else: # CPU
+ # Reusable baseline track for CPU execution mappings
+ display_frame = cv2.resize(
+ device_frame,
+ (self.disp_w, self.disp_h),
+ interpolation=cv2.INTER_NEAREST,
+ )
+
+ # --- Draw Detection Overlays ---
+ if gt_boxes is not None:
+ # Factor for converting 8K (original) bb dimensions to display dimensions
+ scale_display_ox = self.disp_w / self.frame_width
+ scale_display_oy = self.disp_h / self.frame_height
+
+ display_frame = self.get_overlay(
+ display_frame,
+ gt_boxes,
+ None,
+ scale_display_ox,
+ scale_display_oy,
+ color=(0, 255, 0),
+ )
+
+ display_frame = self.get_overlay(
+ display_frame,
+ metadata_or_bbs,
+ class_list,
+ scale_display_x,
+ scale_display_y,
+ )
+ self.async_writer.write_frame(display_frame)
+ self.update_frame(stat_start_time)
+
+ # if "display_frame" in locals():
+ # del display_frame
+ # if "metadata_or_bbs" in locals():
+ # del metadata_or_bbs
+
+ def frame2output(
+ self, device_frame, frame_num, metadata_or_bbs, class_list, stat_start_time
+ ):
+ """
+ Renders drone detection bounding boxes onto the canvas
+ before dispatching to the live UI shared memory stream.
+ """
+ if not self.active:
+ return
+
+ # Factor for converting 640 bb dimensions to display dimensions
+ scale_display_x = self.disp_w / self.resize_w # 640
+ scale_display_y = self.disp_h / self.resize_h # 640
+
+ try:
+ if self.device_input == "cuda": # and torch.is_tensor(device_frame):
+ with torch.inference_mode():
+ # main_app_logger.info(f"device_frame.shape: {device_frame.shape}")
+ # main_app_logger.info(f"\tdevice_frame[None, :].shape: {device_frame[None, :].shape}")
+ # 2. Fast Inline VRAM Downscaling into our static memory slot
+ # Reshape to (Batch, Channel, Height, Width) seamlessly without duplicating data
+ gpu_tensor = device_frame[None, :].permute(0, 3, 1, 2).float()
+
+ resized_tensor = torch.nn.functional.interpolate(
+ gpu_tensor,
+ size=(self.disp_h, self.disp_w),
+ mode="nearest",
+ )
+
+ # CONVERTS TO CPU EARLIER (WORKS but lowers fps)
+ # hwc_byte_tensor = resized_tensor.squeeze(0).permute(1, 2, 0).byte()
+ # cpu_tensor = hwc_byte_tensor.cpu()
+ # bgr_contiguous = cpu_tensor.contiguous()
+
+ # Using copy_() avoids creating any runtime allocations or memory-stride drift.
+ self.static_gpu_byte_bchw.copy_(resized_tensor.byte())
+
+ # Safely reshape the pre-allocated, locked GPU memory layout back to standard HWC array topology.
+ # Since the underlying array layout is strictly contiguous, this permute is guaranteed
+ # to be zero-copy on the GPU and free from memory race stalls.
+ bgr_contiguous = self.static_gpu_byte_bchw.squeeze(0).permute(1, 2, 0)
+
+ # 1. Grab the current free pinned slot tracking variables from your reader
+ d2h_idx = self.reader.d2h_selector
+ pinned_tensor_buf = self.reader.d2h_buffers[d2h_idx]
+
+ # c_idx = self.canvas_selector
+ # current_canvas = self.static_host_canvases[c_idx]
+
+ # host_tensor_view = torch.as_tensor(current_canvas, device="cpu")
+
+ # 3. Non-blocking asynchronous PCIe DMA Download directly into our page-locked canvas
+ # with torch.cuda.stream(self.reader.download_stream):
+ # if (
+ # hasattr(self.reader, "download_stream")
+ # and self.reader.download_stream is not None
+ # ):
+ # torch.cuda.set_stream(self.reader.download_stream)
+ # pinned_tensor = torch.from_numpy(current_canvas).cuda()
+ pinned_tensor_buf.copy_(bgr_contiguous, non_blocking=True)
+
+ # Synchronize only the side download stream layout channel
+ # self.reader.download_stream.synchronize()
+
+ cpu_360p_frame = self.reader.d2h_numpys[d2h_idx]
+
+ # 5. Non-Blocking Push straight to your background AsyncVideoWriter thread pool
+ # We pass a direct .copy() slice so the hot loop can instantly reuse the pinned ring buffer
+ display_frame = np.array(cpu_360p_frame, copy=True, order="C")
+ if display_frame is not None and display_frame.shape[-1] == 3:
+ display_frame = cv2.cvtColor(display_frame, cv2.COLOR_RGB2BGR)
+
+ # --- Draw Detection Overlays ---
+ display_frame = self.get_overlay(
+ display_frame,
+ metadata_or_bbs,
+ class_list,
+ scale_display_x,
+ scale_display_y,
+ )
+
+ self.async_writer.write_frame(display_frame, frame_num)
+ self.update_frame(stat_start_time)
+ # self.canvas_selector = 1 - self.canvas_selector
+ self.reader.d2h_selector = 1 - self.reader.d2h_selector
+
+ else:
+ # Reusable baseline track for CPU execution mappings
+ display_frame = cv2.resize(
+ device_frame,
+ (self.disp_w, self.disp_h),
+ interpolation=cv2.INTER_NEAREST,
+ )
+
+ # --- Draw Detection Overlays ---
+ display_frame = self.get_overlay(
+ display_frame,
+ metadata_or_bbs,
+ class_list,
+ scale_display_x,
+ scale_display_y,
+ )
+ self.async_writer.write_frame(display_frame, frame_num)
+ self.update_frame(stat_start_time)
+
+ except Exception:
+ traceback.print_exc()
+ # return
+
+ if "display_frame" in locals():
+ del display_frame
+ if "metadata_or_bbs" in locals():
+ del metadata_or_bbs
+
+ # CLEANUP --------------------------------------------
+
+ def clean_up_tensors_and_arrays_v1(self):
+ main_app_logger.info("[CLEANUP] Safely stripping active runtime arrays ...")
+
+ # 1. Force disable tracking states to stop graph allocation recursion
+ # torch.set_grad_enabled(False)
+
+ # if hasattr(torch, "jit") and hasattr(torch.jit, "_builtins"):
+ # if isinstance(torch.jit._builtins, dict):
+ # torch.jit._sbuiltins.clear()
+ # else:
+ # torch.jit._builtins = {}
+
+ # 2. Collect references safely using strict type checking
+ # This completely avoids pulling unmanaged proxy objects from the heap
+ all_live_objects = gc.get_objects()
+
+ target_tensors = []
+ target_arrays = []
+
+ for obj in all_live_objects:
+ try:
+ obj_type = type(obj)
+ if isinstance(obj, types.FrameType):
+ frame_info = inspect.getframeinfo(obj)
+ if (
+ "openvino" in frame_info.filename
+ or "openvino.py" in frame_info.filename
+ ):
+ # Clear the frame's local variable dictionary to break circular references
+ obj.f_locals.clear()
+
+ # Check for concrete types to prevent triggering proxy __getattr__ hooks
+ if obj_type is torch.Tensor:
+ target_tensors.append(obj)
+ elif obj_type is np.ndarray:
+ # Explicitly guard size checks to keep it completely stable
+ if obj.base is None and obj.ndim > 0:
+ target_arrays.append(obj)
+ except Exception:
+ traceback.print_exc()
+
+ # 3. Truncate discovered references in-place without triggering deletions
+ reclaimed_tensors = 0
+ for tensor in target_tensors:
+ try:
+ # Truncate raw storage footprint safely
+ tensor.data = torch.empty(0, device=self.device_input)
+ reclaimed_tensors += 1
+ except Exception:
+ traceback.print_exc()
+
+ reclaimed_arrays = 0
+ for arr in target_arrays:
+ try:
+ # Shrink writeable numpy arrays down to 0 bytes safely
+ if arr.flags.writeable:
+ arr.resize((0,), refcheck=False)
+ reclaimed_arrays += 1
+ except (ValueError, SystemError):
+ traceback.print_exc()
+ except Exception:
+ traceback.print_exc()
+
+ main_app_logger.info(
+ f"[CLEANUP] Reclaimed {reclaimed_tensors} tensors and {reclaimed_arrays} arrays safely."
+ )
+
+ # Clean local registers immediately
+ all_live_objects = None
+ target_tensors = None
+ target_arrays = None
+
+ if hasattr(sys, "exc_info"):
+ sys.exc_clear() if hasattr(sys, "exc_clear") else None
+ gc.collect()
+
+ def clean_up_tensors_and_arrays(self):
+ """
+ Safely releases instance-local memory without corrupting global PyTorch tensors.
+ """
+ main_app_logger.info(
+ f"[CLEANUP {getattr(self, 'name', '')}] Releasing local arrays and flushing caches..."
+ )
+
+ # Explicitly release any instance-level tensor references
+ instance_tensor_attrs = [
+ "static_gpu_360p",
+ "static_gpu_byte_bchw",
+ "gpu_float_staging",
+ "scales_tensor",
+ "_cached_grid_x",
+ "_cached_grid_y",
+ ]
+ for attr in instance_tensor_attrs:
+ if hasattr(self, attr):
+ setattr(self, attr, None)
+
+ if hasattr(sys, "exc_info"):
+ try:
+ if hasattr(sys, "exc_clear"):
+ sys.exc_clear()
+ except Exception:
+ pass
+
+ gc.collect()
+ if "cuda" in str(self.device_input) and torch.cuda.is_available():
+ torch.cuda.empty_cache()
+ if hasattr(torch.cuda, "ipc_collect"):
+ torch.cuda.ipc_collect()
+
+ def drain_queues(self, all_queues, wait_time=15.0):
+ # Wait (15 seconds) for your background writers to pull remaining matrices out of the pipe
+ drain_start = time.perf_counter()
+ while time.perf_counter() - drain_start < wait_time:
+ still_has_frames = False
+ for q_name in all_queues:
+ q_val = getattr(self, q_name, None)
+ if q_val is not None:
try:
- if proc.is_alive():
- proc.terminate()
- proc.join(timeout=0.5)
- proc.close()
+ # Accumulate total remaining frame backlogs across your pipelines
+ # if hasattr(q_val, "qsize"):
+ # backlog += q_val.qsize()
+ if not q_val.empty():
+ still_has_frames = True
+ break
except Exception:
pass
- setattr(self, proc_attr, None)
- # 6. PHASE 4: FINAL SEGMENT EVALUATION CONVERGENCE
+ if not still_has_frames:
+ main_app_logger.info(
+ "[TEARDOWN] All background queues are empty. Proceeding to safe close.",
+ )
+ break
+
+ # Yield micro-slices to grant immediate execution priority to background threads
+ time.sleep(0.01)
+
+ def remove_scalar_attributes(self, remove_attrs):
+ for attr in remove_attrs:
+ val = getattr(self, attr, None)
+ if val is not None:
+ # try:
+ # val.clear()
+ # except Exception:
+ # setattr(self, attr, None)
+
+ # # try: delattr(self, attr)
+ # # except AttributeError: pass
+ try:
+ # THE NATIVE SHIELD: Only call clear() if it has the attribute!
+ if hasattr(val, "clear") and callable(getattr(val, "clear")):
+ val.clear()
+ else:
+ # If it's a primitive type like an int, reset it safely
+ if isinstance(val, int):
+ setattr(self, attr, 0)
+ elif isinstance(val, float):
+ setattr(self, attr, 0.0)
+ elif isinstance(val, dict):
+ setattr(self, attr, {})
+ elif torch.is_tensor(val):
+ val.data = torch.empty(0, device=self.device_input)
+ else:
+ setattr(self, attr, None)
+ if hasattr(self, attr):
+ delattr(self, attr)
+ except Exception:
+ # pass
+ traceback.print_exc()
+
+ try:
+ if hasattr(self, attr):
+ delattr(self, attr)
+ except Exception:
+ # pass
+ traceback.print_exc()
+
+ def drain_and_close_queues(self, all_queues):
+ for key in all_queues:
+ val = getattr(self, key, None)
+ if val is None:
+ continue
+ try:
+ while not val.empty():
+ val.get_nowait()
+ except Exception:
+ pass
try:
- final_clip_key = f"{self.name}_{self.clip_id:03d}.mp4"
- final_clip_path = f"{self.config.SHARED_OUTPUT}/{final_clip_key}"
+ if hasattr(val, "close"):
+ if hasattr(val, "cancel_join_thread"):
+ val.cancel_join_thread()
+ val.close()
+ except Exception:
+ pass
+ # setattr(self, key, None)
- # Check if the final truncated or partial segment actually exists before dispatching
+ def stop_events(self, events=[], keys_to_skip_deletion=[]):
+ for event_attr in events:
+ # if hasattr(self, event_attr):
+ tgt_event = getattr(self, event_attr, None)
+ if tgt_event is None:
+ continue
+ try:
+ if hasattr(tgt_event, "_handle"):
+ tgt_event._handle.close()
if (
- os.path.exists(final_clip_path)
- and os.path.getsize(final_clip_path) > 0
+ isinstance(tgt_event, torch.cuda.Event)
+ and event_attr not in keys_to_skip_deletion
):
- print(
- f" [STOP-FLUSH] Registering finalized terminal clip: {final_clip_key}",
- flush=True,
- )
- global clip_completion_tracker
- if final_clip_key not in clip_completion_tracker:
- clip_completion_tracker[final_clip_key] = {
- "video": False,
- "meta": False,
- "start_time": time.time(),
- }
-
- clip_completion_tracker[final_clip_key]["video"] = True
- clip_completion_tracker[final_clip_key]["meta"] = True
- self._evaluate_barrier_and_dispatch(
- final_clip_key, final_clip_path, self.resize_w, self.resize_h
- )
- except Exception as final_flush_err:
- print(
- f" [STOP-WARN] Final segment tracking layer bypass failed: {final_flush_err}",
- flush=True,
- )
+ # Force internal driver handle release
+ del tgt_event
+ except Exception:
+ # pass
+ traceback.print_exc()
+ setattr(self, event_attr, None)
+
+ if hasattr(self, event_attr) and event_attr not in keys_to_skip_deletion:
+ delattr(self, event_attr)
+
+ def stop_thread(self, val):
+ # current_thread_id = threading.get_ident()
+ # val = getattr(self, targeted_key, None)
+ # val = targeted_key_val
+ try:
+ if hasattr(val, "is_alive") and val.is_alive():
+ if val.ident != threading.get_ident():
+ if hasattr(val, "terminate"):
+ val.terminate()
+ val.join(timeout=0.5)
+ except Exception:
+ if val is None:
+ pass
+ traceback.print_exc()
+
+ def stop_threads(self, all_targeted_keys):
+ current_thread_id = threading.get_ident()
+ for attr in all_targeted_keys:
+ val = getattr(self, attr, None)
+ if val is None:
+ continue
+ try:
+ if hasattr(val, "is_alive") and val.is_alive():
+ if getattr(val, "ident", None) != current_thread_id:
+ # if hasattr(val, "terminate"): val.terminate()
+ # val.join(timeout=0.5)
+ if isinstance(val, (threading.Thread, DummyProcess)):
+ # Threads can only be joined cooperatively via flags (self.active = False)
+ val.join(timeout=0.2) # Use a tight, rapid timeout gate
+ else:
+ val.join(timeout=3.0)
+ # OS multi-processing blocks can safely accept termination hooks
+ if val.is_alive() and hasattr(val, "terminate"):
+ val.terminate()
+ except Exception:
+ # pass
+ traceback.print_exc()
- # 8. PHASE 6: UNMAP HARDWARE MEMORY OBJECTS
- if hasattr(self, "shared_details"):
+ # Clean up references immediately to drop tracking counters
+ if getattr(val, "ident", None) != current_thread_id:
+ setattr(self, attr, None)
+
+ def set_stop_writer_event(self, remove=False):
+ if hasattr(self, "stop_writer") and self.stop_writer is not None:
+ try:
+ self.stop_writer.set()
+ except Exception:
+ pass
+ setattr(self, "stop_writer", None)
+
+ if remove:
try:
- # Force-close the internal lock primitive hidden inside the Manager dict proxy
- if hasattr(self.shared_details, "_ctx") and hasattr(
- self.shared_details._ctx, "RLock"
- ):
- lock = self.shared_details._ctx.RLock()
- if hasattr(lock, "_semlock"):
- lock._semlock._close()
- self.shared_details.clear()
+ delattr(self, "stop_writer")
+ except AttributeError:
+ pass
+
+ def stop_executors(self, all_targeted_keys, remove=True):
+ for key in list(all_targeted_keys):
+ val = getattr(self, key, None)
+ if val is None:
+ continue
+ try:
+ val.shutdown(wait=False, cancel_futures=True)
+ # Natively strip out the inner worker thread arrays to drop OS handles
+ if hasattr(val, "_threads"):
+ val._threads.clear()
+ except Exception:
+ try:
+ val.shutdown(wait=False)
+ except Exception:
+ pass
+ setattr(self, key, None)
+ if remove:
+ delattr(self, key)
+
+ def unregister_pinned_cuda_data_v1(self):
+ # Unregister the hidden page-locks inside the ai_pinned_tensors pool
+ if hasattr(self, "ai_pinned_tensors") and self.ai_pinned_tensors:
+ for tensor in list(self.ai_pinned_tensors):
+ try:
+ if torch.is_tensor(tensor):
+ # Extract the shared NumPy view to break the C++ driver lock
+ cv2.cuda.unregisterPageLocked(tensor.numpy())
+ except cv2.error:
+ pass
+ except Exception:
+ pass
+
+ # Unregister the standard pinned_tensors pool page-locks
+ if hasattr(self, "pinned_tensors") and self.pinned_tensors:
+ for tensor in list(self.pinned_tensors):
+ try:
+ if torch.is_tensor(tensor):
+ cv2.cuda.unregisterPageLocked(tensor.numpy())
+ except cv2.error:
+ pass
+ except Exception:
+ pass
+
+ # Unregister any standalone page-locked matrices
+ if hasattr(self, "pinned_matrices") and self.pinned_matrices:
+ for active_mat in list(self.pinned_matrices):
+ if getattr(self, "device_input", "cpu") == "cuda":
+ try:
+ # Wrap explicitly to catch the native OpenCV -217 API Exception!
+ cv2.cuda.unregisterPageLocked(active_mat)
+ except cv2.error:
+ # Catch the pointer registry mismatch safely without breaking the stop() execution stream
+ pass
+ except Exception:
+ pass
+
+ # FORCE PYTORCH C++ ALLOCATOR TO DISSOLVE THE HARDWARE BOUNDARIES
+ if torch.cuda.is_available():
+ torch.cuda.synchronize()
+ torch.cuda.empty_cache()
+
+ def unregister_pinned_cuda_data(self):
+ """
+ Safely unregisters OpenCV page-locked host memory from the CUDA driver
+ before the underlying arrays are released.
+ """
+ if getattr(self, "device_input", "cpu") != "cuda":
+ return
+
+ if hasattr(self, "pinned_matrices") and self.pinned_matrices:
+ for active_mat in list(self.pinned_matrices):
+ if active_mat is not None:
+ try:
+ cv2.cuda.unregisterPageLocked(active_mat)
+ except Exception:
+ pass
+
+ if hasattr(self, "pinned_tensors") and self.pinned_tensors:
+ for tensor in list(self.pinned_tensors):
+ if tensor is not None and torch.is_tensor(tensor):
+ try:
+ cv2.cuda.unregisterPageLocked(tensor.numpy())
+ except Exception:
+ pass
+
+ # Force CUDA driver to finalize unpinning
+ if torch.cuda.is_available():
+ torch.cuda.synchronize()
+ torch.cuda.empty_cache()
+ if hasattr(torch.cuda, "ipc_collect"):
+ torch.cuda.ipc_collect()
+
+ def clear_buffer_pools(self, buffer_pools):
+ for buffer_pool in buffer_pools:
+ val = getattr(self, buffer_pool, None)
+ if val is None:
+ continue
+
+ setattr(self, buffer_pool, None)
+ for i in range(len(val)):
+ val[i] = None
+ if hasattr(val, "clear"):
+ try:
+ val.clear()
except Exception:
pass
- del self.shared_details
- if (
- hasattr(self, "mp_frame_ready_event")
- and self.mp_frame_ready_event is not None
- ):
+ def clear_pinned_data(self):
+ tensor_pools = ["ai_pinned_tensors", "pinned_tensors"]
+ for pool_attr in tensor_pools:
+ pool = getattr(self, pool_attr, None)
+ if pool:
+ for tensor in list(pool):
+ try:
+ if torch.is_tensor(tensor):
+ # Truncate internal storage mapping layer immediately
+ tensor.data = torch.empty(0, device="cpu")
+ except Exception:
+ pass
try:
- # mp.Event allocates an internal ctx.Cond / ctx.Lock boundary pair
- if hasattr(self.mp_frame_ready_event, "_cond"):
- cond = self.mp_frame_ready_event._cond
- if hasattr(cond, "_lock") and hasattr(cond._lock, "_semlock"):
- cond._lock._semlock._close()
+ pool.clear()
except Exception:
pass
- self.mp_frame_ready_event = None
- for q_attr in ["signal_queue", "render_queue"]:
- if hasattr(self, q_attr):
- q = getattr(self, q_attr)
- if q is not None:
- try:
- # 1. Flush and break down standard background worker threads safely
- q.close()
- q.join_thread()
-
- # 2. CRITICAL: Force-close the hidden POSIX lock primitives to clear resource_tracker limits
- if hasattr(q, "_rlock") and q._rlock is not None:
- if hasattr(q._rlock, "_semlock"):
- q._rlock._semlock._close()
-
- if hasattr(q, "_writer") and q._writer is not None:
- if hasattr(q._writer, "_semlock"):
- q._writer._semlock._close()
- except Exception:
- pass
- setattr(self, q_attr, None)
+ # 7. UNPIN HARDWARE MATRICES (From initialize_variables)
+ if hasattr(self, "pinned_matrices") and self.pinned_matrices:
+ try:
+ self.pinned_matrices.clear()
+ except Exception:
+ pass
- for primitive_attr in [
- "ready_buffer_idx",
- "reader_active_idx",
- "shm_frame_lengths",
- ]:
- if hasattr(self, primitive_attr):
- obj = getattr(self, primitive_attr)
- if obj is not None and hasattr(obj, "get_lock"):
- try:
- # Grab the hidden lower-level POSIX context lock map
- lock = obj.get_lock()
- # If the lock handle is bound to an active system descriptor, release it
- if hasattr(lock, "_semlock"):
- lock._semlock._close() # Forces immediate unlinking at the OS level
- except Exception:
- pass
- setattr(self, primitive_attr, None)
+ def clear_shared_memory_list(self, shm_names: list):
+ """
+ Cleans up and unlinks shared memory collections, supporting both
+ direct attributes ("shms") and nested attributes ("reader.shms").
+ """
+ for shm_path in shm_names:
+ curr_obj = self
+ for attr in shm_path.split("."):
+ curr_obj = getattr(curr_obj, attr, None)
+ if curr_obj is None:
+ break
- if hasattr(self, "manager") and self.manager is not None:
- try:
- self.manager.shutdown()
- except Exception:
- pass
- self.manager = None
+ val = curr_obj
+ if not val:
+ continue
- if hasattr(self, "pinned_matrices") and self.pinned_matrices:
- for active_mat in self.pinned_matrices:
+ release_shared_memory(list(val))
+ # for shm in list(val):
+ # if shm is not None:
+ # try:
+ # # Release buffer view if open
+ # if hasattr(shm, "buf") and shm.buf is not None:
+ # shm.buf.release()
+ # except Exception:
+ # pass
+
+ # try:
+ # shm.close()
+ # except Exception:
+ # pass
+
+ # try:
+ # shm.unlink()
+ # except (FileNotFoundError, AttributeError, OSError):
+ # pass
+
+ # # Clear the container list/collection in-place
+ # if hasattr(val, "clear"):
+ # try:
+ # val.clear()
+ # except Exception:
+ # pass
+
+ def clean_up_reader(self):
+ """Halts the background prefetch staging worker completely before draining pipeline queues."""
+ if hasattr(self, "reader") and self.reader is not None:
+ try:
+ # 1. IMMEDIATE HALT: Force the background loop condition to fail instantly
+ self.reader.stopped = True
+ self.prefetch_active = False
+ self.active = False
+
+ # 2. Join the prefetch thread gracefully so it finishes its current read and dies
+ if hasattr(self, "prefetch_threads") and self.prefetch_threads:
+ for th in self.prefetch_threads:
+ if th.is_alive():
+ th.join(timeout=0.2)
+ self.prefetch_threads.clear()
+
+ # if hasattr(self.reader, "print_breakdown"):
+ # self.reader.print_breakdown()
+
+ # 3. Safe, deadlock-free queue drain loop (Now guaranteed to exit cleanly)
+ drain_timeout_start = time.perf_counter()
+ while not self.prefetch_queue.empty():
try:
- cv2.cuda.unregisterPageLocked(active_mat)
+ self.prefetch_queue.get_nowait()
+ self.prefetch_queue.task_done()
except Exception:
- pass
- self.pinned_matrices.clear()
- self.pinned_tensors.clear()
+ break
+ # Hard escape hatch safety guard to prevent hanging if threads ghost
+ if time.perf_counter() - drain_timeout_start > 0.5:
+ break
+
+ # 4. Trigger standard standalone hardware unpinning routines
+ # if hasattr(self.reader, "release_hardware_pins"):
+ # self.reader.release_hardware_pins()
- if hasattr(self, "ai_shms") and self.ai_shms:
- for shm in self.ai_shms:
+ self.reader.stop()
+
+ except Exception as e:
+ main_app_logger.debug(
+ f"Reader object isolation step encountered a soft error: {e}"
+ )
+ finally:
+ self.reader = None
+
+ def unlink_shared_memory(self, all_shm_keys):
+ for key in all_shm_keys:
+ val = getattr(self, key, None)
+ if val is None:
+ continue
+ for shm in list(val):
+ if shm is not None:
try:
+ # Drop direct memoryview trackers instantly
+ if hasattr(shm, "buf") and shm.buf is not None:
+ try:
+ shm.buf.release()
+ except Exception:
+ pass
+
+ # Invalidate private mmap structures to drop pytest stack frame caches
+ if hasattr(shm, "_mmap") and shm._mmap is not None:
+ try:
+ shm._mmap = None
+ except Exception:
+ pass
+
+ # try:
+ # shm.__class__.__del__ = lambda self: None
+ # except Exception:
+ # pass
+
+ # # Unlink text descriptors out of the core tracker process
+ # shm_descriptor = shm._name if shm._name.startswith("/") else f"/{shm._name}"
+ # try:
+ # unregister(shm_descriptor, "shared_memory")
+ # except Exception:
+ # pass
+
shm.close()
shm.unlink()
except Exception:
pass
- self.ai_shms.clear()
- self.ai_shm_names.clear()
+ try:
+ val.clear()
+ except Exception:
+ pass
- if hasattr(self, "shms") and self.shms:
- for shm in self.shms:
- shm.close()
- try:
- shm.unlink()
- except FileNotFoundError:
- pass
- self.shms.clear()
-
- if hasattr(self, "cap") and self.cap is not None:
- self.cap.release()
- self.cap = None
-
- for pool_name in [
- "executor",
- "io_executor",
- "clip_executor",
- "ffmpeg_executor",
- ]:
- if hasattr(self, pool_name) and getattr(self, pool_name) is not None:
+ def stop_sync_manager(self):
+ """Safely shuts down the background Sync Base Manager process daemon."""
+ # if hasattr(self, "shared_details") and self.shared_details is not None:
+ # try: self.shared_details.clear()
+ # except Exception: pass
+
+ if hasattr(self, "manager") and self.manager is not None:
+ try:
+ if hasattr(self.manager, "_allocated"):
+ for memory_block in list(self.manager._allocated):
+ try:
+ if (
+ hasattr(memory_block, "buf")
+ and memory_block.buf is not None
+ ):
+ memory_block.buf.release()
+ memory_block.close()
+ memory_block.unlink()
+ except Exception:
+ pass
try:
- getattr(self, pool_name).shutdown(wait=True)
+ self.manager._allocated.clear()
except Exception:
pass
- setattr(self, pool_name, None)
+ self.manager.shutdown()
- self.reader = None
- self._is_stopped = True
- print(f" [STOP] {self.name} pipeline resources fully released.", flush=True)
-
- def model_warmup(self, H=640, W=640):
- H, W = int(H), int(W)
- # Move the dummy input creation inside a no_grad block
- with torch.no_grad():
- print(f"Starting warmup for {self.name}...", flush=True)
- dummy_input = torch.zeros((1, 3, H, W)).to(self.device_input)
-
- # Perform iterations directly on the main thread
- for i in range(5):
- _ = self.run_model(
- dummy_input,
- imgsz=(H, W),
- batch=1,
- device_input=self.device_input,
- stream=STREAM_ARG,
- )
+ # Added
+ if (
+ hasattr(self.manager, "_process")
+ and self.manager._process is not None
+ ):
+ proc = self.manager._process
+ if proc.is_alive():
+ proc.join(timeout=0.2)
+ if proc.is_alive():
+ proc.terminate()
+ proc.join()
- # Force GPU to finish before returning
- if self.device_input == "cuda":
- torch.cuda.synchronize()
+ except Exception:
+ pass
+ setattr(self, "manager", None)
+
+ def purge_attributes(self):
+ # ─── FORCE LOW-LEVEL OPENVINO C++ HARDWARE PURGE ───
+ if hasattr(self, "model") and self.model is not None:
+ try:
+ # If your backend utilizes an OpenVINO core orchestrator:
+ model_wrapper = self.model
+
+ # Check for common OpenVINO runtime handle names inside your core class
+ for attr_name in ["core", "_core", "ov_core", "runtime"]:
+ if hasattr(model_wrapper, attr_name):
+ ov_core = getattr(model_wrapper, attr_name)
+ if ov_core is not None:
+ # 1. Force OpenVINO to clear internal cached compiled models and models' graphs
+ if hasattr(ov_core, "get_property"):
+ try:
+ # Tells OpenVINO to drop its physical memory caching metrics pools
+ ov_core.set_property({}, {})
+ except Exception:
+ pass
+
+ # 2. Trigger explicit Python C++ boundary deconstruction bindings
+ if hasattr(ov_core, "__del__"):
+ ov_core.__del__()
+
+ setattr(model_wrapper, attr_name, None)
+ except Exception:
+ pass
+
+ # Un-bind dynamic instance methods to break Pytest's reference hold
+ for dynamic_method in ["run_realtime_inference", "pipeline_fn"]:
+ if hasattr(self, dynamic_method):
+ try:
+ # Wiping out the bound method object instantly frees the frame loops
+ # setattr(self, dynamic_method, None)
+ delattr(self, dynamic_method)
+ except Exception:
+ pass
+
+ for cls_attr in ["d2h_buffers", "d2h_numpys", "d2h_selector"]: # "pipeline_fn",
+ if hasattr(self, cls_attr):
+ try:
+ setattr(self, cls_attr, None)
+ except AttributeError:
+ pass
+
+ # Now it is structurally safe to purge attributes without causing cross-thread collisions
+ keys_to_purge = [
+ "__persistent_vram_lock",
+ "_DeviceBaseHandler__persistent_vram_lock", # Wipes out the private lock tensor safely
+ "ai_gpu_staging",
+ "ai_pinned_tensors",
+ "ai_shm_names", # Clear names list references
+ "ai_shms", # Clear handler specific memory arrays
+ # "all_preds",
+ # "all_targets",
+ "bgs_stream",
+ "config", # Force config destruction
+ "evaluator",
+ "executor",
+ "frame_buffer_pool",
+ "gpu_buffer_pool",
+ "gpu_display_frame",
+ "gpu_encoder_8k_buf",
+ "gpu_float_staging",
+ "host_buffer_pool",
+ "inference_stream",
+ "model",
+ "pinned_downloaded_frame_np",
+ "pinned_downloaded_resizedframe_np",
+ "pinned_matrices", # Clear handler registration matrix list
+ "pinned_tensors", # Clear backend tensor mappings
+ "process_thread",
+ "raw_input",
+ "reader",
+ "scales_tensor",
+ "shms", # Clear handler specific memory arrays
+ "static_gpu_360p", # Added to secure structural bounds
+ "static_gpu_byte_bchw", # Added to secure structural bounds
+ # "static_host_canvases",
+ # "io_executor",
+ ]
+ for key in keys_to_purge:
+ if hasattr(self, key):
+ try:
+ delattr(self, key)
+ except AttributeError:
+ pass
- print(f"Warmup complete for {self.name}", flush=True)
+ gc.collect()
+ # HELPER FUNCTIONS --------------------------------------------
def get_executor_backlog(self):
"""Returns the number of tasks currently waiting in the thread pool queue."""
# access the internal queue of the executor
- return self.executor._work_queue.qsize()
+ # return self.executor._work_queue.qsize()
+ if getattr(self, "executor", None) is None:
+ return 0
+ try:
+ # Handle standard ThreadPoolExecutor and custom queue wrappers cleanly
+ if hasattr(self.executor, "_work_queue"):
+ return self.executor._work_queue.qsize()
+ return 0
+ except Exception:
+ return 0
def get_clip_executor_backlog(self):
"""Returns the number of tasks currently waiting in the thread pool queue."""
# access the internal queue of the clip_executor
- return self.clip_executor._work_queue.qsize()
+ # return self.clip_executor._work_queue.qsize()
+ if getattr(self, "clip_executor", None) is None:
+ return 0
+ try:
+ if hasattr(self.clip_executor, "_work_queue"):
+ return self.clip_executor._work_queue.qsize()
+ return 0
+ except Exception:
+ return 0
def check_disk_usage(self, path, min_gb=0.5):
"""Returns True if there is at least min_gb available at path."""
@@ -1484,7 +3259,7 @@ def check_disk_usage(self, path, min_gb=0.5):
free_gb = free / (2**30)
return free_gb > min_gb
except Exception as e:
- print(f"[EXCEPTION] Disk check error: {e}")
+ main_app_logger.info(f"[EXCEPTION] Disk check error: {e}")
return False
# Gets frame W and H details
@@ -1498,1892 +3273,399 @@ def get_frameWH(self):
else:
new_sizeHW = check_imgsz(
[self.frame_height, self.frame_width]
- ) # expects hxw
-
- new_sizeWH = (new_sizeHW[1], new_sizeHW[0])
-
- self.width = new_sizeWH[0]
- self.height = new_sizeWH[1]
-
- # Configure scaling for 8K-to-Model coordinate mapping
- self.resize_h, self.resize_w = [self.config.MODEL_H, self.config.MODEL_W]
- self.scale_x = self.frame_width / self.config.MODEL_W
- self.scale_y = self.frame_height / self.config.MODEL_H
-
- def update_frame(self, stat_start_time):
- self.stat_frame_count += 1
- self.elapsed_display_time += time.perf_counter() - stat_start_time
- # if elapsed > 0.5:
- self.stat_fps = round(self.stat_frame_count / self.elapsed_display_time, 1)
-
- def is_processing(self):
- """Returns True if any part of the pipeline is still active."""
- if not self.reader.stopped:
- return True
-
- if (
- self.process_thread is not None and self.process_thread.is_alive()
- ): # or not self.reader.frame_queue.empty():
- return True
-
- if not self.reader.frame_queue.empty():
- return True
-
- q_size = self.write_queue.qsize()
- if self.process_thread is not None:
- print(
- f"[STATUS] Writer: {self.write_count}/{self.frame_count_target} | Q: {q_size} | Inf: {'Alive' if self.process_thread.is_alive() else 'Dead'}",
- end="\r",
- flush=True,
- )
- else:
- print(
- f"[STATUS] Writer: {self.write_count}/{self.frame_count_target} | Q: {q_size}",
- end="\r",
- flush=True,
- )
-
- if not self.write_queue.empty():
- return True
- if self.write_count < self.frame_count_target:
- # q_size = self.write_queue.qsize()
- # print(f"[DRAIN] Writer: {self.write_count}/{self.frame_count} | Queue: {q_size}", end="\r", flush=True)
- return True
- return False
-
- def run_model(
- self,
- frame,
- imgsz=(BASE_PIPELINE_CONFIG.MODEL_H, BASE_PIPELINE_CONFIG.MODEL_W),
- batch=1,
- device_input="cuda",
- stream=False,
- ):
- # --- DEBUG VERIFICATION ---
- # if torch.is_tensor(frame):
- # # We want to see [N, 3, 640, 640] where N > 1
- # print(f"[DEBUG] Tensor Input Shape: {frame.shape}")
- # elif isinstance(frame, list):
- # print(f"[DEBUG] List Input Length: {len(frame)}")
-
- # print(f"[DEBUG] Stream Mode: {stream} | Requested Batch: {batch}")
- # ---------------------------
- if isinstance(frame, torch.Tensor):
- # Ensure on the right device
- frame = frame.to(device_input)
-
- # Make sure input is multiple of 32
- h, w = frame.shape[-2:]
- pad_h = (32 - h % 32) % 32
- pad_w = (32 - w % 32) % 32
-
- if pad_h > 0 or pad_w > 0:
- # F.pad for 4D tensor (B, C, H, W) uses (left, right, top, bottom)
- frame = F.pad(frame, (0, pad_w, 0, pad_h), value=0)
-
- results = self.model.predict(
- frame,
- imgsz=imgsz,
- batch=batch,
- device=device_input,
- verbose=False,
- stream=stream,
- conf=self.config.DETECTION_THRESHOLD,
- max_det=self.config.MAX_DETECTIONS,
- rect=True, # False, #
- )
- return results
-
- # def _encode_and_signal(self, data, frame_num):
- # """Worker task for JPEG encoding. Optimized for zero-copy GPU transfers."""
- # if data is None:
- # return
-
- # if isinstance(data, cv2.cuda.GpuMat):
- # # GPU PATH: Use existing 8K GPU pointer. Resize on GPU (Instant) 🏎️
- # cv2.cuda.resize(
- # data,
- # (self.disp_w, self.disp_h),
- # stream=self.encode_stream,
- # dst=self.gpu_display_frame,
- # )
- # # Only download the small (640x360) frame, not the 8K frame!
- # display_frame = self.gpu_display_frame.download(self.encode_stream)
- # else:
- # # CPU PATH: Resize and force memory contiguity for faster encoding
- # display_frame = cv2.resize(
- # data, (self.disp_w, self.disp_h), interpolation=cv2.INTER_NEAREST
- # )
- # display_frame = np.ascontiguousarray(display_frame)
-
- # # Drop quality to 35. This reduces the bytes FastAPI has to push to the browser.
- # # success, buffer = cv2.imencode(
- # # ".jpg", display_frame, [cv2.IMWRITE_JPEG_QUALITY, 35]
- # # )
-
- # # if success:
- # # self.latest_processed_frame = buffer.tobytes()
- # # self.last_delivered_frame_id = frame_num
- # # self.last_heartbeat = time.time()
- # # # Non-blocking signal to FastAPI
- # # self.loop.call_soon_threadsafe(self.frame_ready_event.set)
-
- # # Run the high-overhead JPEG compression block within a non-blocking background thread worker
- # def _async_compress_task(img_mat, f_idx):
- # success, buffer = cv2.imencode(".jpg", img_mat, [cv2.IMWRITE_JPEG_QUALITY, 35])
- # if success and self.active:
- # self.latest_processed_frame = buffer.tobytes()
- # self.last_delivered_frame_id = f_idx
- # self.last_heartbeat = time.time()
- # # Safely wake up the FastAPI async loop on the main process thread
- # self.loop.call_soon_threadsafe(self.frame_ready_event.set)
-
- # # Submit directly to your background I/O pool to unblock the inference stream
- # self.io_executor.submit(_async_compress_task, display_frame, frame_num)
-
- def _encode_and_signal(self, data, frame_num):
- """
- Asynchronously processes display outputs by offloading high-overhead
- JPEG compression tasks directly onto the background I/O pool.
- """
- if data is None:
- return
-
- # Check the backlog of your I/O task queue to prevent memory leaks
- if hasattr(self, "io_executor") and self.io_executor._work_queue.qsize() > 4:
- return # Skip displaying this frame to preserve CPU performance
-
- def _async_render_and_compress(frame_data, f_num):
- if isinstance(frame_data, cv2.cuda.GpuMat):
- # GPU Path: Perform hardware-accelerated resizing inside VRAM
- cv2.cuda.resize(
- frame_data,
- (self.disp_w, self.disp_h),
- stream=self.encode_stream,
- dst=self.gpu_display_frame,
- )
- display_frame = self.gpu_display_frame.download(self.encode_stream)
- else:
- # CPU Path: Perform rapid array transformations
- display_frame = cv2.resize(
- frame_data,
- (self.disp_w, self.disp_h),
- interpolation=cv2.INTER_NEAREST,
- )
-
- display_frame = np.ascontiguousarray(display_frame)
- success, buffer = cv2.imencode(
- ".jpg", display_frame, [cv2.IMWRITE_JPEG_QUALITY, 35]
- )
-
- if success and self.active:
- self.latest_processed_frame = buffer.tobytes()
- self.last_delivered_frame_id = f_num
- self.last_heartbeat = time.time()
- # Non-blocking signal to wake up the FastAPI event loop
- self.loop.call_soon_threadsafe(self.frame_ready_event.set)
-
- # Offload the processing overhead entirely from the inference thread pool
- self.io_executor.submit(_async_render_and_compress, data, frame_num)
-
- def update_ui_fallback(self, frame, frame_num):
- # If backlog is very high, drop JPEG quality to 25 to clear the 'pause' faster
- backlog = self.get_executor_backlog()
- adaptive_quality = (
- 20
- if backlog > (self.dynamic_limit * 2)
- else self.config.DISPLAY_FRAME_QUALITY
- )
-
- # FALLBACK: If AI is busy, worker thread encodes raw frame for the UI.
- # This offloads the 40ms CPU cost from the main Producer loop.
- display_frame = cv2.resize(
- frame, (self.disp_w, self.disp_h), interpolation=cv2.INTER_NEAREST
- )
- _, buffer = cv2.imencode(
- ".jpg", display_frame, [cv2.IMWRITE_JPEG_QUALITY, adaptive_quality]
- )
-
- # print(f"DEFAULT DISP frameNum/last_frame_id {frame_num} self.last_delivered_frame_id: {self.last_delivered_frame_id}", flush=True) #\n\tframe_bytes: {frame_bytes}", flush=True)
- self.latest_processed_frame = buffer.tobytes()
- self.last_delivered_frame_id = frame_num
- self.last_frame_id = frame_num
- self.last_heartbeat = time.time()
- # Signal the FastAPI generator that a new frame is ready
- self.loop.call_soon_threadsafe(self.frame_ready_event.set)
-
- def _check_shm_safety(self, threshold_percent=90):
- """
- Scans /dev/shm and deletes the oldest .mp4 files if usage exceeds threshold.
- This prevents the 8K stream from crashing the entire container.
- """
- # Check current usage of the RAM disk
- usage = shutil.disk_usage("/dev/shm")
- percent_used = (usage.used / usage.total) * 100
-
- if percent_used > threshold_percent:
- print(
- f" [CRITICAL] /dev/shm usage at {percent_used:.1f}%. Purging old clips..."
- )
-
- # Get all .mp4 files in /dev/shm sorted by oldest first
- shm_path = Path("/dev/shm")
- clips = sorted(shm_path.glob("*.mp4"), key=lambda x: x.stat().st_mtime)
-
- # Delete files until we are under 70% usage or run out of files
- for clip in clips:
- try:
- # Don't delete the file the current writer is actively using!
- # if str(clip) == self.tmp_file:
- # continue
-
- clip.unlink()
- print(f"[PURGE] Deleted {clip.name} to free RAM.")
-
- # Re-check usage after each deletion
- usage = shutil.disk_usage("/dev/shm")
- if (usage.used / usage.total) * 100 < 70:
- break
- except Exception as e:
- print(f"[EXCEPTION] Could not purge {clip}: {e}")
-
- # def run_realtime_inference(self, sf_enabled):
- # """Producer: Maintains the target FPS and updates clip IDs."""
- # # Calculate a dynamic limit: tolerate 0.5 seconds of lag.
- # # If target_fps is 15, the limit is 7. If target_fps is 30, the limit is 15.
- # self.dynamic_limit = max(2, int(0.5 * self.target_fps))
- # last_frame_time = time.perf_counter()
- # while self.active:
- # # --- FRAME RETRIEVAL ---
- # try:
- # device_frame, frame_num = self.reader.read()
- # if device_frame is None:
- # if self.reader.stopped:
- # self.active = False
-
- # clip_filename = f"{self.config.SHARED_OUTPUT}/{self.name}_{self.clip_id:03d}.mp4"
- # clip_key = Path(clip_filename).name
- # global send_metadata_queue, all_metadata
- # if clip_key in all_metadata:
- # # Signal to process metadata for previous cli
- # if (
- # "send_metadata_queue" in globals()
- # or "send_metadata_queue" in locals()
- # ):
- # try:
- # send_metadata_queue.put(
- # (clip_filename, self.resize_w, self.resize_h)
- # )
- # except Exception as queue_err:
- # print(
- # f"[CLIPPER-WARN] Metadata queue push skipped or unallocated: {queue_err}",
- # flush=True,
- # )
- # break
- # continue
- # except queue.Empty:
- # if getattr(self.reader, "reconnect_failed", False):
- # self.active = False
- # break
- # time.sleep(0.002)
- # continue
-
- # # Keep 8K frame on GPU (Skip CPU conversion for non-target frames)
- # # if self.device_input == "cuda" and not self.reader.is_h264_8k:
- # # with torch.cuda.stream(self.ingest_stream):
- # # device_frame = nv12_to_rgb_torch(
- # # device_frame, self.frame_height, self.frame_width# , is_bgr=False
- # # )
- # # self.ingest_stream.synchronize()
-
- # # if device_frame is not None:
- # self.frame_count += 1
- # is_target_frame = float(frame_num) >= self.next_process_idx
-
- # # Determine if this frame should be AI or Raw based on backlog
- # # But ALWAYS submit to the executor to maintain frame order.
- # backlog = self.get_executor_backlog()
-
- # while backlog > 4 and self.active:
- # time.sleep(0.005)
- # backlog = self.get_executor_backlog()
-
- # def wrapped_fn(*args):
- # if self.device_input == "cuda":
- # with torch.cuda.stream(self.inference_stream):
- # dev_frame, f_num, target_flag = args
- # # FIX: Explicitly crop out any hardware padding columns horizontally
- # # and rows vertically before forcing linear memory contiguity.
- # if dev_frame.ndim >= 2:
- # h_raw, w_raw = dev_frame.shape[-2:]
- # if w_raw != self.frame_width or h_raw != self.frame_height:
- # dev_frame = dev_frame[..., :self.frame_height, :self.frame_width]
-
- # isolated_frame = dev_frame.clone().contiguous()
- # self.pipeline_fn(isolated_frame, f_num, target_flag)
- # else:
- # self.pipeline_fn(*args)
-
- # # Handoff to AI and Writer
- # if self.active:
- # self.executor.submit(
- # # pipeline_fn,
- # wrapped_fn,
- # device_frame,
- # frame_num,
- # is_target_frame,
- # )
-
- # # --- PRECISE CLOCK SYNC ---
- # # This prevents the producer from "lapping" the consumer
- # # and building that jumpy backlog in the first place.
- # elapsed = time.perf_counter() - last_frame_time
- # if elapsed < self.frame_interval:
- # # time.sleep(self.frame_interval - elapsed)
- # # Subtract a small epsilon (0.001) for OS scheduling overhead
- # # sleep_duration = self.frame_interval - elapsed - 0.0025
- # # if sleep_duration > 0.001:
- # # time.sleep(sleep_duration)
- # time.sleep(max(0, self.frame_interval - elapsed - 0.0015))
- # last_frame_time = time.perf_counter()
-
- # # self.update_frame()
- # self.last_heartbeat = time.time()
-
- # self.stop()
-
- def run_realtime_inference(self, sf_enabled):
- """Producer: Maintains the target FPS and updates clip IDs."""
- # Calculate a dynamic limit: tolerate 0.5 seconds of lag.
- # If target_fps is 15, the limit is 7. If target_fps is 30, the limit is 15.
- self.dynamic_limit = max(2, int(0.5 * self.target_fps))
- last_frame_time = time.perf_counter()
- while self.active:
- # --- FRAME RETRIEVAL ---
- try:
- device_frame, frame_num = self.reader.read()
- if device_frame is None:
- if self.reader is None or (
- hasattr(self.reader, "stopped") and self.reader.stopped
- ):
- if self.device_input == "cuda":
- torch.cuda.synchronize()
- self.active = False
-
- clip_filename = f"{self.config.SHARED_OUTPUT}/{self.name}_{self.clip_id:03d}.mp4"
- clip_key = Path(clip_filename).name
- global send_metadata_queue, all_metadata
- if clip_key in all_metadata:
- # Signal to process metadata for previous cli
- if (
- "send_metadata_queue" in globals()
- or "send_metadata_queue" in locals()
- ):
- try:
- send_metadata_queue.put(
- (clip_filename, self.resize_w, self.resize_h)
- )
- except Exception as queue_err:
- print(
- f"[CLIPPER-WARN] Metadata queue push skipped or unallocated: {queue_err}",
- flush=True,
- )
- break
- continue
- except queue.Empty:
- if getattr(self.reader, "reconnect_failed", False):
- self.active = False
- break
- time.sleep(0.002)
- continue
-
- self.stat_start_time = time.perf_counter() # timing to display detection
-
- # Keep 8K frame on GPU (Skip CPU conversion for non-target frames)
- # if self.device_input == "cuda" and not self.reader.is_h264_8k:
- # with torch.cuda.stream(self.ingest_stream):
- # device_frame = nv12_to_rgb_torch(
- # device_frame, self.frame_height, self.frame_width# , is_bgr=False
- # )
- # self.ingest_stream.synchronize()
-
- # if device_frame is not None:
- self.frame_count += 1
- is_target_frame = float(frame_num) >= self.next_process_idx
- # if not self.config.sf_enabled or abs(float(self.input_fps) - float(self.target_fps)) < 0.01:
- # is_target_frame = True
- # else:
- # is_target_frame = float(frame_num) >= self.next_process_idx
-
- # Determine if this frame should be AI or Raw based on backlog
- # But ALWAYS submit to the executor to maintain frame order.
- backlog = self.get_executor_backlog()
-
- while backlog > 4 and self.active:
- time.sleep(0.005)
- backlog = self.get_executor_backlog()
-
- def wrapped_fn(*args):
- if self.device_input == "cuda":
- # Ensure the worker thread switches to your targeted pipeline execution timeline
- torch.cuda.set_stream(self.inference_stream)
- dev_frame, f_num, target_flag, stat_start_time = args
- isolated_device_frame = (
- dev_frame.clone()
- if torch.is_tensor(dev_frame)
- else dev_frame.copy()
- )
-
- self.pipeline_fn(
- isolated_device_frame, f_num, target_flag, stat_start_time
- )
-
- # Force a non-blocking device barrier to ensure operations have fully hit VRAM
- # before releasing the thread context
- self.inference_stream.synchronize()
- else:
- dev_frame, f_num, target_flag, stat_start_time = args
- isolated_device_frame = (
- dev_frame.clone()
- if torch.is_tensor(dev_frame)
- else dev_frame.copy()
- )
- self.pipeline_fn(
- isolated_device_frame, f_num, target_flag, stat_start_time
- )
-
- if is_target_frame: # timing to display detection
- self.next_process_idx += self.step_size
- # Handoff to AI and Writer
- # if self.active:
- # Clone the tensor buffer immediately on the producer thread
- # to prevent upstream overwrite races by the next reader iteration.
- # isolated_device_frame = device_frame.clone() if torch.is_tensor(device_frame) else device_frame.copy()
- self.executor.submit(
- # pipeline_fn,
- wrapped_fn,
- device_frame,
- frame_num,
- is_target_frame,
- self.stat_start_time,
- )
- # else:
- # # Process background execution context for skipped frames
- # self.pipeline_fn(device_frame, frame_num, is_target_frame)
-
- # if self.device_input == "cuda":
- # torch.cuda.synchronize()
-
- # --- PRECISE CLOCK SYNC ---
- # This prevents the producer from "lapping" the consumer
- # and building that jumpy backlog in the first place.
- elapsed = time.perf_counter() - last_frame_time
- if elapsed < self.frame_interval:
- # time.sleep(self.frame_interval - elapsed)
- # Subtract a small epsilon (0.001) for OS scheduling overhead
- sleep_duration = max(0, self.frame_interval - elapsed - 0.0015)
- if sleep_duration > 0.001:
- time.sleep(sleep_duration)
- last_frame_time = time.perf_counter()
-
- # self.update_frame()
- self.last_heartbeat = time.time()
-
- self.stop()
-
- def filter_contained_boxes(self, boxes, overlap_thresh=0.9):
- """
- Vectorized IoA filter: Removes boxes if most of their area is inside another box.
- """
- if boxes.shape[0] <= 1:
- return boxes
-
- # Calculate Areas
- w = (boxes[:, 2] - boxes[:, 0]).clamp(min=0)
- h = (boxes[:, 3] - boxes[:, 1]).clamp(min=0)
- areas = w * h
- valid_mask = (
- (w < (self.resize_w * self.config.ROI_MAX_RELATIVE_SIZE_RATIO))
- & (h < (self.resize_h * self.config.ROI_MAX_RELATIVE_SIZE_RATIO))
- & (w > 0)
- & (h > 0)
- & (areas >= self.min_contour_area)
- )
- boxes = boxes[valid_mask]
- areas = areas[valid_mask]
-
- # Compute all-to-all Intersections [N, N]
- lt = torch.max(boxes.unsqueeze(1)[:, :, :2], boxes.unsqueeze(0)[:, :, :2])
- rb = torch.min(boxes.unsqueeze(1)[:, :, 2:], boxes.unsqueeze(0)[:, :, 2:])
- wh = (rb - lt).clamp(min=0)
- inter_area = wh[:, :, 0] * wh[:, :, 1]
-
- # Intersection over Area (How much of Box A is in Box B)
- # ioa[i, j] = (Box i ∩ Box j) / Area(i)
- ioa = inter_area / (areas.unsqueeze(1) + 1e-6)
-
- # Filter logic:
- # Only remove if Box J is LARGER than Box I and overlap is high
- diag = torch.eye(boxes.shape[0], device=boxes.device, dtype=torch.bool)
- larger_mask = areas.unsqueeze(0) >= areas.unsqueeze(1)
-
- to_remove = (ioa > overlap_thresh) & larger_mask & ~diag
- return boxes[~to_remove.any(dim=1)]
-
- # def get_detections(
- # self,
- # frame_raw, # BGR
- # frameNum, # Added to metadata, should be frames in clip
- # merged=None,
- # thickness=2,
- # device_input="cuda",
- # ):
- # metadata = {}
- # try:
- # # H, W = frame_raw.shape[:2] # Unpack once
- # # FIX: Get H/W correctly for both Numpy [H, W, C] and Tensor [C, H, W]
- # if torch.is_tensor(frame_raw):
- # H, W = frame_raw.shape[-2:] # Gets last two dims
- # else:
- # H, W = frame_raw.shape[:2]
-
- # if merged is None:
- # if torch.is_tensor(frame_raw):
- # # Swap channels (-3 is the C dim) and normalize
- # frame_input = (
- # # frame_raw.float() / 255.0
- # frame_raw.flip(-3).float() / 255.0
- # )
- # if frame_input.ndim == 3:
- # frame_input = frame_input.unsqueeze(0)
- # else:
- # frame_input = frame_raw
-
- # # Run Inference (Keep stream=False as it is stable)
- # results = self.run_model(
- # frame_input, # Should be RGB
- # imgsz=(H, W),
- # batch=1,
- # device_input=device_input,
- # stream=False,
- # )
- # else:
- # cropped_batch = []
- # cropped_coords = []
- # # PREPARE CROPS (Path-specific Optimization)
- # if device_input == "cuda" and torch.is_tensor(frame_raw):
- # # GPU PATH: Zero-copy slicing + Hardware Interpolation
- # for box in merged:
- # x1, y1, x2, y2 = [int(val) for val in box]
- # cropped_coords.append((x1, y1))
- # crop = frame_raw[:, y1:y2, x1:x2].unsqueeze(0)
- # # crop_float = crop.float() / 255.0
- # crop_float = crop.flip(-3).float() / 255.0
- # # F.interpolate on A6000 handles 100+ crops in ~2ms
- # crop_resized = F.interpolate(
- # crop_float,
- # size=(self.resize_h, self.resize_w),
- # mode="bilinear",
- # align_corners=False,
- # )
- # cropped_batch.append(crop_resized.to(torch.half))
-
- # # Consolidate into 4D Tensor for Parallel Hardware Batching
- # input_data = (
- # torch.cat(cropped_batch, dim=0) if cropped_batch else None
- # )
- # else:
- # # CPU PATH: OpenCV Batching
- # foi_cpu = (
- # frame_raw.permute(1, 2, 0).byte().cpu().numpy()
- # if torch.is_tensor(frame_raw)
- # else frame_raw
- # )
- # for box in merged:
- # x1, y1, x2, y2 = [int(val) for val in box]
- # cropped_coords.append((x1, y1))
- # crop = foi_cpu[y1:y2, x1:x2]
- # # Batching on CPU still needs resized inputs
- # crop_resized = cv2.resize(
- # crop,
- # (self.resize_w, self.resize_h),
- # interpolation=cv2.INTER_NEAREST,
- # )
- # cropped_batch.append(crop_resized)
-
- # input_data = (
- # cropped_batch # List of arrays for OpenVINO/CPU batching
- # )
-
- # if cropped_batch == []:
- # print("[DEBUG] Early exit: cropped_batch is empty", flush=True)
- # return metadata, None
-
- # # if self.config.DEBUG_FLAG:
- # # self.debug_save_crops(cropped_batch, frameNum)
-
- # # CHUNKED BATCH INFERENCE
- # # Process in chunks of MODEL_MAX_BATCH_SIZE to stay within TensorRT/OpenVINO profile limits
- # results = []
- # for i in range(0, len(input_data), self.config.MODEL_MAX_BATCH_SIZE):
- # chunk = input_data[i : i + self.config.MODEL_MAX_BATCH_SIZE]
- # chunk_results = self.run_model(
- # chunk, # Should be RGB
- # imgsz=(self.resize_h, self.resize_w),
- # batch=len(chunk),
- # device_input=device_input,
- # stream=False, # stream=False is critical
- # )
- # results.extend(list(chunk_results))
-
- # # Process results and draw 8K-space overlays
- # # Display scales for the final 640x640 stretched output
- # num_objs = 0
- # scale_display_x = self.resize_w / W # 640 / 8192
- # scale_display_y = self.resize_h / H # 640 / 4608
- # for ridx, r in enumerate(list(results)):
- # if r.boxes is None or len(r.boxes) == 0:
- # continue
-
- # # Determine the ROI expansion ratio for this specific crop
- # if merged is not None:
- # x1_8k, y1_8k, x2_8k, y2_8k = merged[ridx]
- # off_x, off_y = x1_8k, y1_8k
- # # Ratio: How many 8K pixels does one inference pixel represent?
- # roi_ratio_x = (x2_8k - x1_8k) / self.resize_w
- # roi_ratio_y = (y2_8k - y1_8k) / self.resize_h
-
- # else:
- # off_x, off_y = 0, 0
- # roi_ratio_x, roi_ratio_y = 1.0, 1.0
-
- # # Move to CPU in one bulk operation per crop
- # boxes = r.boxes.xyxy.cpu().numpy()
- # clss = r.boxes.cls.cpu().numpy().astype(int)
- # confs = r.boxes.conf.cpu().numpy()
-
- # for j in range(len(boxes)):
- # num_objs += 1
-
- # bx1, by1, bx2, by2 = boxes[j]
- # # abs_x1, abs_y1 = off_x + bx1, off_y + by1
- # # abs_x2, abs_y2 = off_x + bx2, off_y + by2
-
- # # # Map to absolute 8K pixels (Scale crop-coords to 8K then add offset)
- # # abs_x1 = off_x + (bx1 * roi_ratio_x)
- # # abs_y1 = off_y + (by1 * roi_ratio_y)
- # # abs_x2 = off_x + (bx2 * roi_ratio_x)
- # # abs_y2 = off_y + (by2 * roi_ratio_y)
-
- # # Map to absolute 8K pixels
- # abs_x1 = off_x + (bx1 * roi_ratio_x)
- # abs_y1 = off_y + (by1 * roi_ratio_y)
- # abs_x2 = off_x + (bx2 * roi_ratio_x)
- # abs_y2 = off_y + (by2 * roi_ratio_y)
-
- # # Map to 640x640 Display pixels (Apply the non-uniform stretch)
- # # disp_x = int(abs_x1 * scale_display_x)
- # # disp_y = int(abs_y1 * scale_display_y)
- # # disp_w = int((abs_x2 - abs_x1) * scale_display_x)
- # # disp_h = int((abs_y2 - abs_y1) * scale_display_y)
- # disp_x = abs_x1 * scale_display_x
- # disp_y = abs_y1 * scale_display_y
- # disp_w = (abs_x2 - abs_x1) * scale_display_x
- # disp_h = (abs_y2 - abs_y1) * scale_display_y
-
- # class_id = clss[j]
- # class_name = self.label_source[class_id]
- # confidence = confs[j]
-
- # if not self.config.OMIT_DETECTIONS_FLAG:
- # timestamp = datetime.now().strftime("%Y-%m-%d %H:%M:%S.%f")[:-3]
- # print(
- # # f"[OBJECT DETECTION] {class_name} detected in frame {frameNum} (Total detected: {current_cnt})",
- # f"[{timestamp}] {self.name} DETECTION on Frame {frameNum}: {class_name} detected",
- # flush=True,
- # )
-
- # # if not self.config.TEST_MODE:
- # # bb_color = get_detection_color(class_id, is_bgr=True)
-
- # # foi = cv2.rectangle(
- # # foi,
- # # (abs_x1, abs_y1),
- # # (abs_x2, abs_y2),
- # # bb_color,
- # # thickness,
- # # )
- # # label = f"{class_name} {confidence:.2f}"
- # # draw_label(foi, label, (abs_x1, abs_y1), color=bb_color, padding=5)
-
- # # height = min(abs_y2, H) - max(0, abs_y1)
- # # width = min(abs_x2, W) - max(0, abs_x1)
-
- # # Resized
- # object_res = [
- # int(disp_x), # int(abs_x1 * scale_x),
- # int(disp_y), # int(abs_y1 * scale_y),
- # int(disp_h), # int(height * scale_y),
- # int(disp_w), # int(width * scale_x),
- # class_name,
- # confidence,
- # int(self.resize_h),
- # int(self.resize_w),
- # ]
-
- # framenum_str = f"{frameNum:04d}_{j:04d}"
- # # if self.config.DEBUG_FLAG:
- # # meta_str = ",".join([str(o) for o in object_res + [framenum_str]])
- # # print(f"[{self.name} METADATA],{meta_str}", flush=True)
-
- # # Full Res
- # metadata[framenum_str] = {
- # "frameId": int(frameNum),
- # "bbId": framenum_str,
- # "bbox": {
- # "x": int(object_res[0]),
- # "y": int(object_res[1]),
- # "height": int(object_res[2]),
- # "width": int(object_res[3]),
- # "object": str(object_res[4]),
- # "object_det": {
- # "confidence": float(object_res[5]),
- # "frameH": int(object_res[6]),
- # "frameW": int(object_res[7]),
- # },
- # },
- # }
- # except Exception as e:
- # print(f"[GET_DETECTION] Exception: {e}\n{traceback.print_exc()}")
- # num_objs = len(metadata.keys())
-
- # if self.config.DEBUG_FLAG:
- # log_to_logger(f"[DEBUG] get_detections returned {num_objs} detections", level="debug")
- # return metadata, None
-
- # # # Queue frame for display (reduce quality for 8K bandwidth)
- # # frame_bytes = get_display_frame_in_bytes(
- # # foi,
- # # display_size=(self.disp_w, self.disp_h),
- # # quality=self.config.DISPLAY_FRAME_QUALITY,
- # # return_bytes=True,
- # # )
-
- # # return metadata, frame_bytes
-
- def get_gpu_rois_by_area(self, mask, max_candidates=100):
- # Extract true spatial constraints straight from the active mask object footprint
- if torch.is_tensor(mask):
- mask_h, mask_w = mask.shape[-2:]
- elif isinstance(mask, cv2.cuda.GpuMat):
- # cv2.cuda.GpuMat.size() returns a tuple of (width, height) standard formatting
- mask_w, mask_h = mask.size()
- else:
- mask_h, mask_w = mask.shape[:2]
-
- # This prevents find_contours_gpu_equivalent from mutating the mask variables used by other threads.
- if isinstance(mask, cv2.cuda.GpuMat):
- # .clone() allocates a new C++ memory surface and forces full continuity
- isolated_kernel_mask = mask.clone()
- elif torch.is_tensor(mask):
- isolated_kernel_mask = mask.clone().contiguous()
- else:
- isolated_kernel_mask = mask.copy()
-
- # Get raw boxes from mask (Direct VRAM bridge)
- boxes_gpu = find_contours_gpu_equivalent(
- isolated_kernel_mask,
- stream=self.bgs_stream,
- limit_640=640 * 1.5,
- )
-
- # --- FIX: ELIMINATE STREAM RACE ---
- if boxes_gpu is None or len(boxes_gpu) == 0:
- return torch.empty((0, 4), device=self.device_input)
-
- # Wrap existing GPU memory as a float tensor (Zero Copy)
- # raw_boxes = torch.as_tensor(boxes_gpu, device=self.device_input).float()
- if self.device_input == "cuda":
- # Wrap the native device handle and IMMEDIATELY append .clone()
- # This allocates a brand new, physically isolated VRAM block to secure the bounding boxes
- raw_boxes = (
- torch.as_tensor(boxes_gpu, device=self.device_input).float().clone()
- )
- else:
- raw_boxes = torch.as_tensor(boxes_gpu, device=self.device_input).float()
-
- # Vectorized Pre-Filter (Removes noise blobs before merging)
- w = raw_boxes[:, 2] - raw_boxes[:, 0]
- h = raw_boxes[:, 3] - raw_boxes[:, 1]
- mask_filter = (w * h > self.min_contour_area) & (w < mask_w) & (h < mask_h)
- raw_boxes = raw_boxes[mask_filter]
-
- # Prevents N^2 distance matrix from exploding during high noise
- if raw_boxes.shape[0] > max_candidates:
- # Prioritize the largest blobs (most likely to be drones)
- areas = (raw_boxes[:, 2] - raw_boxes[:, 0]) * (
- raw_boxes[:, 3] - raw_boxes[:, 1]
- )
- _, indices = torch.topk(areas, max_candidates)
- raw_boxes = raw_boxes[indices]
- return raw_boxes
-
- # def get_gpu_rois(self, frame, frameNum, mask):
- # raw_boxes = self.get_gpu_rois_by_area(mask)
-
- # if raw_boxes.shape[0] < 1:
- # return torch.empty((0, 4), device=self.device_input)
-
- # if raw_boxes.shape[0] > 1:
- # raw_boxes = merge_boxes_gpu(raw_boxes, gap_limit=self.dist_thresh_640)
-
- # clean_640p = self.filter_contained_boxes(
- # raw_boxes, overlap_thresh=self.config.ROI_CONTAINMENT_THRESH
- # )
-
- # if clean_640p.shape[0] < 1:
- # return torch.empty((0, 4), device=self.device_input)
-
- # # Scale to 8K space
- # return clean_640p * self.scales_tensor
+ ) # expects hxw
- def get_cpu_rois(self, frame, frameNum, mask):
- contours, _ = cv2.findContours(mask, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)
- raw_boxes_xywh = [
- list(cv2.boundingRect(c))
- for c in contours
- if cv2.contourArea(c) > self.min_contour_area
- ]
- raw_boxes = [[x, y, x + w, y + h] for x, y, w, h in raw_boxes_xywh]
+ new_sizeWH = (new_sizeHW[1], new_sizeHW[0])
- if len(raw_boxes) < 1:
- return torch.empty((0, 4), device=self.device_input)
+ self.width = new_sizeWH[0]
+ self.height = new_sizeWH[1]
- if len(raw_boxes) > 1:
- raw_boxes = merge_boxes_cpu(raw_boxes, gap_limit=self.dist_thresh_640)
+ # Configure scaling for 8K-to-Model coordinate mapping
+ self.resize_h, self.resize_w = [self.config.MODEL_H, self.config.MODEL_W]
+ self.scale_x = self.frame_width / self.config.MODEL_W
+ self.scale_y = self.frame_height / self.config.MODEL_H
- raw_boxes_640p = torch.tensor(raw_boxes, device=self.device_input).float()
+ def is_processing(self):
+ """Returns True if any part of the pipeline is still active."""
+ if not self.reader.stopped:
+ return True
- clean_640p = self.filter_contained_boxes(
- raw_boxes_640p, overlap_thresh=self.config.ROI_CONTAINMENT_THRESH
- )
+ if (
+ self.process_thread is not None and self.process_thread.is_alive()
+ ): # or not self.reader.frame_queue.empty():
+ return True
- if clean_640p.shape[0] < 1:
- return torch.empty((0, 4), device=self.device_input)
+ if not self.reader.frame_queue.empty():
+ return True
- # Scale to 8K space
- return clean_640p * self.scales_tensor
+ q_size = self.write_queue.qsize()
+ if self.process_thread is not None:
+ main_app_logger.info(
+ f"[STATUS] Writer: {self.write_count}/{self.frame_count_target} | Q: {q_size} | Inf: {'Alive' if self.process_thread.is_alive() else 'Dead'}",
+ end="\r",
+ )
+ else:
+ main_app_logger.info(
+ f"[STATUS] Writer: {self.write_count}/{self.frame_count_target} | Q: {q_size}",
+ end="\r",
+ )
- # Project clusters back to 8K and build centered YOLO windows
- # final_8k_rois = scale_clusters_to_8k(
- # merged_640, frame_w=self.frame_width, frame_h=self.frame_height
- # )
- # return torch.tensor(final_8k_rois, device=self.device_input).float()
+ if not self.write_queue.empty():
+ return True
+ if self.write_count < self.frame_count_target:
+ # q_size = self.write_queue.qsize()
+ # main_app_logger.info(f"[DRAIN] Writer: {self.write_count}/{self.frame_count} | Queue: {q_size}", end="\r")
+ return True
+ return False
- def get_detections(
- self, frame, frame_id, thickness=2, device_input="cuda", merged=None
- ):
+ def _check_shm_safety(self, threshold_percent=90):
"""
- Processes full-frame 8K targets or aggregates smart-filtered bounding-box regions
- into uniform tensor arrays for batched inference execution.
+ Scans /dev/shm and deletes the oldest .mp4 files if usage exceeds threshold.
+ This prevents the 8K stream from crashing the entire container.
"""
- metadata = {}
- is_cuda = device_input == "cuda"
+ # Check current usage of the RAM disk
+ usage = shutil.disk_usage("/dev/shm")
+ percent_used = (usage.used / usage.total) * 100
- # =====================================================================
- # 🚀 PATH 1: FULL RESOLUTION TRACK (sf_enabled = False)
- # =====================================================================
- if merged is None:
- with torch.inference_mode():
- if isinstance(frame, torch.Tensor):
- # # Transpose to channel-first shape format [1, C, H, W] if layout is trailing
- # if frame.ndim == 3 and frame.shape[-1] == 3:
- # frame = frame.permute(2, 0, 1).unsqueeze(0).clone().contiguous()
- # elif frame.ndim == 3:
- # frame = frame.unsqueeze(0).clone().contiguous()
-
- # # Create a memory-contiguous layout view and cast cleanly inside VRAM
- # frame = frame.contiguous()
-
- # Transpose to channel-first shape format [1, C, H, W] if layout is trailing
- if frame.ndim == 3 and frame.shape[-1] == 3:
- # CRITICAL BUG FIX: Appending .clone().contiguous() creates a brand new,
- # physically isolated tensor memory block. This breaks the link to
- # temporary buffers, preventing the VS Code debugger from crashing on evaluation.
- frame = frame.permute(2, 0, 1).unsqueeze(0).clone().contiguous()
- elif frame.ndim == 3:
- frame = frame.unsqueeze(0).clone().contiguous()
- else:
- # Ensure any multi-dimensional batch views are physically packed
- frame = frame.clone().contiguous()
+ if percent_used > threshold_percent:
+ main_app_logger.info(
+ f" [CRITICAL] /dev/shm usage at {percent_used:.1f}%. Purging old clips..."
+ )
- if frame.dtype == torch.uint8:
- frame = frame.to(
- device_input, dtype=torch.float16, non_blocking=True
- )
- frame.div_(255.0) # Safe in-place float normalization
- else:
- frame = frame.to(device_input, non_blocking=True)
+ # Get all .mp4 files in /dev/shm sorted by oldest first
+ shm_path = Path("/dev/shm")
+ clips = sorted(shm_path.glob("*.mp4"), key=lambda x: x.stat().st_mtime)
- img_size = frame.shape[-2:]
- else:
- # CPU / Host NumPy NDArray fallback
- img_size = frame.shape[:2]
-
- # img_size = (self.resize_h, self.resize_w)
- H, W = img_size
- scale_display_x = self.resize_w / W # 640 / 8192
- scale_display_y = self.resize_h / H # 640 / 4608
- results = self.run_model(
- frame,
- imgsz=img_size,
- batch=1,
- device_input=device_input,
- stream=STREAM_ARG,
- )
+ # Delete files until we are under 70% usage or run out of files
+ for clip in clips:
+ try:
+ # Don't delete the file the current writer is actively using!
+ # if str(clip) == self.tmp_file:
+ # continue
- # Extract full resolution detections
- if results and len(results) > 0:
- boxes = results[0].boxes
- if boxes is not None:
- for idx, box in enumerate(boxes):
- # coords = box.xywh[0].cpu().tolist() # [x_center, y_center, width, height]
- # cls_id = int(box.cls[0].cpu().item())
- # conf = float(box.conf[0].cpu().item())
- coords = (
- box.xywh.cpu().squeeze().tolist()
- ) # Converts [x_center, y_center, w, h] safely
- cls_id = int(box.cls.cpu().item())
- class_name = self.label_source[cls_id]
- confidence = float(box.conf.cpu().item())
-
- # Guard against un-squeezed structural lists
- if isinstance(coords[0], list):
- coords = coords[0]
-
- # Convert center bounds coordinates back to upper-left origin layout standard
- # and scale to 640x640
- disp_x = (coords[0] - (coords[2] / 2.0)) * scale_display_x
- disp_y = (coords[1] - (coords[3] / 2.0)) * scale_display_y
- disp_w = coords[2] * scale_display_x
- disp_h = coords[3] * scale_display_y
-
- if disp_w > 2 and disp_h > 2:
- # Resized
- object_res = [
- int(disp_x), # int(abs_x1 * scale_x),
- int(disp_y), # int(abs_y1 * scale_y),
- int(disp_h), # int(height * scale_y),
- int(disp_w), # int(width * scale_x),
- class_name,
- confidence,
- int(self.resize_h),
- int(self.resize_w),
- ]
-
- obj_id = len(metadata)
- framenum_str = f"{frame_id:04d}_{obj_id:04d}"
- metadata[framenum_str] = {
- "frameId": int(frame_id),
- "bbId": framenum_str,
- "bbox": {
- "x": int(object_res[0]),
- "y": int(object_res[1]),
- "height": int(object_res[2]),
- "width": int(object_res[3]),
- "object": str(object_res[4]),
- "object_det": {
- "confidence": float(object_res[5]),
- "frameH": int(object_res[6]),
- "frameW": int(object_res[7]),
- },
- },
- }
- del frame
- return metadata, None
+ clip.unlink()
+ main_app_logger.info(f"[PURGE] Deleted {clip.name} to free RAM.")
- # =====================================================================
- # 📦 PATH 2: SMART FILTER ROIs TRACK (sf_enabled = True)
- # =====================================================================
- # else:
- # roi_patches = []
- # patch_coordinates = []
- # max_batch_size = getattr(self.config, "MODEL_MAX_BATCH_SIZE", 4)
-
- # # 1. Image Canvas Matrix Pre-processing Isolation Gauges
- # if is_cuda and isinstance(frame, torch.Tensor):
- # src_tensor = frame.squeeze(0) if frame.ndim == 4 else frame
- # if src_tensor.shape[-1] == 3:
- # src_tensor = src_tensor.permute(2, 0, 1) # Force [C, H, W]
- # src_h, src_w = src_tensor.shape[-2:]
- # else:
- # # Host fallback tracking pointers
- # src_tensor = np.asarray(frame)
- # src_h, src_w = src_tensor.shape[:2]
-
- # # 2. Extract and Standarize Regions of Interest (ROIs)
- # for box in merged:
- # x1, y1, x2, y2 = map(int, box)
- # x1, y1 = max(0, x1), max(0, y1)
- # x2, y2 = min(src_w, x2), min(src_h, y2)
-
- # if (x2 - x1) < 8 or (y2 - y1) < 8:
- # continue # Filter out noise slices
-
- # if is_cuda and isinstance(src_tensor, torch.Tensor):
- # with torch.no_grad():
- # crop = src_tensor[:, y1:y2, x1:x2]
- # # Bilinear scale interpolation pass on GPU hardware to achieve uniform dimensions
- # if crop.shape[-2:] != (self.resize_h, self.resize_w):
- # crop = F.interpolate(
- # crop.unsqueeze(0).float(),
- # size=(self.resize_h, self.resize_w),
- # mode="bilinear",
- # align_corners=False,
- # ).squeeze(0)
- # roi_patches.append(crop)
- # else:
- # # CPU Path: Crop and resize via OpenCV linear interpolation
- # crop = src_tensor[y1:y2, x1:x2]
- # if crop.shape[:2] != (self.resize_h, self.resize_w):
- # crop = cv2.resize(
- # crop,
- # (self.resize_w, self.resize_h),
- # interpolation=cv2.INTER_LINEAR,
- # )
- # roi_patches.append(crop)
-
- # patch_coordinates.append((x1, y1, x2 - x1, y2 - y1))
-
- # if not roi_patches:
- # return metadata, None
-
- # # 3. Process Patches via Multi-Cam Inference Batch Factory
- # results_pool = []
- # for i in range(0, len(roi_patches), max_batch_size):
- # batch_slices = roi_patches[i : i + max_batch_size]
- # current_batch_len = len(batch_slices)
-
- # if is_cuda and isinstance(batch_slices[0], torch.Tensor):
- # with torch.inference_mode():
- # # Stack patches into a balanced [N, C, MODEL_H, MODEL_W] tensor matrix array
- # inference_batch = torch.stack(batch_slices).to(
- # device_input, dtype=torch.float16, non_blocking=True
- # )
- # inference_batch.div_(
- # 255.0
- # ) # Normalize directly in VRAM page boundaries
-
- # batch_res = self.run_model(
- # inference_batch,
- # imgsz=(self.resize_h, self.resize_w),
- # batch=current_batch_len,
- # device_input=device_input,
- # stream=STREAM_ARG,
- # )
- # results_pool.extend(batch_res)
- # del inference_batch
- # else:
- # # CPU Path Processing Pass Stack
- # with torch.inference_mode():
- # # Standardize NumPy array layouts into single host batch matrix block layouts
- # np_batch = np.stack(batch_slices).astype(np.float32) / 255.0
- # inference_batch = (
- # torch.from_numpy(np_batch)
- # .permute(0, 3, 1, 2)
- # .to(device_input)
- # )
-
- # batch_res = self.run_model(
- # inference_batch,
- # imgsz=(self.resize_h, self.resize_w),
- # batch=current_batch_len,
- # device_input=device_input,
- # stream=STREAM_ARG,
- # )
- # results_pool.extend(batch_res)
- # del inference_batch
-
- # # 4. Map Patch Bounding Boxes back onto the Global 8K Frame Coordinates Map
- # scale_display_x = self.resize_w / self.frame_width # 640 / 8192
- # scale_display_y = self.resize_h / self.frame_height # 640 / 4608
- # for idx, res in enumerate(results_pool):
- # ox, oy, o_width, o_height = patch_coordinates[idx]
- # if res.boxes is not None:
- # for b_box in res.boxes:
- # lx1, ly1, lx2, ly2 = b_box.xyxy[0].cpu().tolist()
- # cls_id = int(b_box.cls[0].cpu().item())
- # confidence = float(b_box.conf[0].cpu().item())
- # class_name = self.label_source[cls_id]
- # # confidence = confs[j]
-
- # # Map relative coordinates proportionally to the original 8K ROI slice layout
- # global_x1 = ox + (lx1 * (o_width / float(self.resize_w)))
- # global_y1 = oy + (ly1 * (o_height / float(self.resize_h)))
- # global_x2 = ox + (lx2 * (o_width / float(self.resize_w)))
- # global_y2 = oy + (ly2 * (o_height / float(self.resize_h)))
- # disp_x = global_x1 * scale_display_x
- # disp_y = global_y1 * scale_display_y
- # disp_w = (global_x2 - global_x1) * scale_display_x
- # disp_h = (global_y2 - global_y1) * scale_display_y
-
- # if disp_w > 0 and disp_h > 0:
- # # Resized
- # object_res = [
- # int(disp_x), # int(abs_x1 * scale_x),
- # int(disp_y), # int(abs_y1 * scale_y),
- # int(disp_h), # int(height * scale_y),
- # int(disp_w), # int(width * scale_x),
- # class_name,
- # confidence,
- # int(self.resize_h),
- # int(self.resize_w),
- # ]
-
- # obj_id = len(metadata)
- # framenum_str = f"{frame_id:04d}_{obj_id:04d}"
- # metadata[framenum_str] = {
- # "frameId": int(frame_id),
- # "bbId": framenum_str,
- # "bbox": {
- # "x": int(object_res[0]),
- # "y": int(object_res[1]),
- # "height": int(object_res[2]),
- # "width": int(object_res[3]),
- # "object": str(object_res[4]),
- # "object_det": {
- # "confidence": float(object_res[5]),
- # "frameH": int(object_res[6]),
- # "frameW": int(object_res[7]),
- # },
- # },
- # }
-
- # return metadata, None
- # =====================================================================
- # PATH 2: SMART FILTER ROIs TRACK (sf_enabled = True) -> PADDED ASPECT PRESERVING 📦
- # =====================================================================
- else:
- roi_patches = []
- patch_coordinates = []
- max_batch_size = getattr(self.config, "MODEL_MAX_BATCH_SIZE", 4)
-
- # 1. Canvas Matrix Pre-processing Isolation
- if is_cuda and isinstance(frame, torch.Tensor):
- src_tensor = frame.squeeze(0) if frame.ndim == 4 else frame
- if src_tensor.shape[-1] == 3:
- src_tensor = src_tensor.permute(2, 0, 1) # Force [C, H, W]
- src_h, src_w = src_tensor.shape[-2:]
- else:
- src_tensor = np.asarray(frame)
- src_h, src_w = src_tensor.shape[:2]
-
- # Target layout constraints
- th, tw = self.resize_h, self.resize_w
-
- # 2. Extract, Aspect-Scale, and Pad Regions of Interest (ROIs)
- for box in merged:
- x1, y1, x2, y2 = map(int, box)
- x1, y1 = max(0, x1), max(0, y1)
- x2, y2 = min(src_w, x2), min(src_h, y2)
-
- box_w, box_h = x2 - x1, y2 - y1
- if box_w < 8 or box_h < 8:
- continue # Filter out invalid spatial artifacts
-
- if is_cuda and isinstance(src_tensor, torch.Tensor):
- with torch.no_grad():
- crop = src_tensor[:, y1:y2, x1:x2]
-
- # Calculate aspect-preserving scale factor
- scale = min(tw / box_w, th / box_h)
- nw, nh = int(box_w * scale), int(box_h * scale)
- nw, nh = max(1, nw), max(1, nh)
-
- # Dynamic scaling: downsample or upsample using clean bilinear grids
- if (box_h, box_w) != (nh, nw):
- crop_resized = F.interpolate(
- crop.unsqueeze(0).float(),
- size=(nh, nw),
- mode="bilinear",
- align_corners=False,
- ).squeeze(0)
- else:
- crop_resized = crop.float()
+ # Re-check usage after each deletion
+ usage = shutil.disk_usage("/dev/shm")
+ if (usage.used / usage.total) * 100 < 70:
+ break
+ except Exception as e:
+ main_app_logger.info(f"[EXCEPTION] Could not purge {clip}: {e}")
- # Force the canvas tracking allocation to occur safely inside your active stream scope context
- with torch.cuda.stream(self.inference_stream):
- # Instantiate a clean, pre-allocated padded evaluation canvas context
- padded_canvas = torch.zeros(
- (3, th, tw), dtype=torch.float32, device=device_input
- )
+ def print_active_gpu_tensor_memory(self):
+ main_app_logger.info(
+ "=" * 60,
+ )
+ main_app_logger.info(
+ "\033[95m[VRAM INVESTIGATOR] Scanning Active GPU Tensors with Sources:\033[0m"
+ )
- # Center the aspect-scaled crop onto the dark zero-padded mask grid
- dx = (tw - nw) // 2
- dy = (th - nh) // 2
- padded_canvas[:, dy : dy + nh, dx : dx + nw] = crop_resized
+ # Capture the exact C++ memory registry structure
+ try:
+ raw_snapshot = torch.cuda.memory._snapshot()
+ segments = raw_snapshot.get("segments", [])
+ except Exception:
+ segments = []
+ main_app_logger.info(
+ "[WARN] Failed parsing native memory context snapshot.",
+ )
- # Convert directly to half-precision to speed up pipeline passes
- roi_patches.append(padded_canvas.to(torch.half))
- else:
- # CPU Path: Aspect-preserving resize and padding via OpenCV
- crop = src_tensor[y1:y2, x1:x2]
- scale = min(tw / box_w, th / box_h)
- nw, nh = max(1, int(box_w * scale)), max(1, int(box_h * scale))
+ # Map raw block storage memory addresses straight to Python frames
+ addr_to_source = {}
+ for seg in segments:
+ for block in seg.get("blocks", []):
+ if block.get("state") == "active_allocated":
+ addr = block.get("address")
+ history = block.get("history", [])
+ if history:
+ # Inspect the deepest frame in the allocation stack
+ frame = history[-1]
+ filename = frame.get("filename", "Unknown")
+ lineno = frame.get("line", 0)
+ func_name = frame.get("name", "unknown_func")
+ addr_to_source[addr] = f"{filename}:{lineno} ({func_name})"
+
+ # Scan the heap via GC and resolve actual backing storage layers
+ leaked_tensors = []
+ total_detected_bytes = 0
+
+ for obj in gc.get_objects():
+ try:
+ if torch.is_tensor(obj) and obj.is_cuda:
+ t_bytes = obj.element_size() * obj.nelement()
+ total_detected_bytes += t_bytes
+ if t_bytes > 0:
+ leaked_tensors.append(obj)
+ except Exception:
+ pass
- crop_resized = cv2.resize(
- crop, (nw, nh), interpolation=cv2.INTER_LINEAR
- )
+ obj = None # Break tracking register references
- # Center-pad the array block with zeros (black)
- padded_canvas = np.zeros((th, tw, 3), dtype=np.uint8)
- dx = (tw - nw) // 2
- dy = (th - nh) // 2
- padded_canvas[dy : dy + nh, dx : dx + nw] = crop_resized
- roi_patches.append(padded_canvas)
-
- # Store the scaling shifts to map detections back to the 8K coordinate grid accurately
- patch_coordinates.append((x1, y1, box_w, box_h, scale, dx, dy))
-
- if not roi_patches:
- return metadata, None
-
- # 3. Process Patches via Multi-Cam Inference Batch Factory
- results_pool = []
- for i in range(0, len(roi_patches), max_batch_size):
- batch_slices = roi_patches[i : i + max_batch_size]
- current_batch_len = len(batch_slices)
-
- if is_cuda and isinstance(batch_slices[0], torch.Tensor):
- with torch.inference_mode():
- torch.cuda.set_stream(self.inference_stream)
- inference_batch = torch.stack(batch_slices).to(
- device_input, dtype=torch.float16, non_blocking=True
- )
- inference_batch.div_(255.0) # In-place GPU normalization
-
- batch_res = self.run_model(
- inference_batch,
- imgsz=(th, tw),
- batch=current_batch_len,
- device_input=device_input,
- stream=STREAM_ARG,
- )
- results_pool.extend(batch_res)
- del inference_batch
- else:
- with torch.inference_mode():
- np_batch = np.stack(batch_slices).astype(np.float32) / 255.0
- inference_batch = (
- torch.from_numpy(np_batch)
- .permute(0, 3, 1, 2)
- .to(device_input)
- )
+ for i, tensor in enumerate(leaked_tensors):
+ t_bytes = tensor.element_size() * tensor.nelement()
- batch_res = self.run_model(
- inference_batch,
- imgsz=(th, tw),
- batch=current_batch_len,
- device_input=device_input,
- stream=STREAM_ARG,
- )
- results_pool.extend(batch_res)
- del inference_batch
+ # --- THE FIX: Extract the address of the underlying storage block ---
+ try:
+ if hasattr(tensor, "untyped_storage"):
+ storage_addr = tensor.untyped_storage().data_ptr()
+ elif hasattr(tensor, "storage") and tensor.storage():
+ storage_addr = tensor.storage().data_ptr()
+ else:
+ storage_addr = tensor.data_ptr()
+ except Exception:
+ storage_addr = tensor.data_ptr()
- # 4. Map Patch Bounding Boxes back onto the Global 8K Frame Coordinates Map
- scale_display_x = tw / float(self.frame_width)
- scale_display_y = th / float(self.frame_height)
+ # Extract absolute code trace locations from our snapshot dictionary
+ source_loc = addr_to_source.get(
+ storage_addr, "Unknown Native C++ Allocation / Model Context"
+ )
- for idx, res in enumerate(results_pool):
- ox, oy, o_width, o_height, scale_f, pad_x, pad_y = patch_coordinates[
- idx
- ]
- if res.boxes is not None and len(res.boxes) > 0:
- all_xyxy = res.boxes.xyxy.cpu().numpy()
- all_clss = res.boxes.cls.cpu().numpy().astype(int)
- all_confs = res.boxes.conf.cpu().numpy().astype(float)
-
- for j in range(len(all_xyxy)):
- lx1, ly1, lx2, ly2 = all_xyxy[j]
- class_name = self.label_source[all_clss[j]]
- confidence = all_confs[j]
-
- # Reverse the centering padding offset values
- lx1_unpadded = lx1 - pad_x
- ly1_unpadded = ly1 - pad_y
- lx2_unpadded = lx2 - pad_x
- ly2_unpadded = ly2 - pad_y
-
- # Reverse the aspect ratio scale shift to map back to absolute 8K coordinates
- global_x1 = ox + (lx1_unpadded / scale_f)
- global_y1 = oy + (ly1_unpadded / scale_f)
- global_x2 = ox + (lx2_unpadded / scale_f)
- global_y2 = oy + (ly2_unpadded / scale_f)
-
- # Project directly onto the 640x640 display monitoring layout canvas
- disp_x = global_x1 * scale_display_x
- disp_y = global_y1 * scale_display_y
- disp_w = (global_x2 - global_x1) * scale_display_x
- disp_h = (global_y2 - global_y1) * scale_display_y
-
- if disp_w > 0 and disp_h > 0:
- object_res = [
- int(disp_x),
- int(disp_y),
- int(disp_h),
- int(disp_w),
- class_name,
- confidence,
- int(th),
- int(tw),
- ]
- obj_id = len(metadata)
- framenum_str = f"{frame_id:04d}_{obj_id:04d}"
- metadata[framenum_str] = {
- "frameId": int(frame_id),
- "bbId": framenum_str,
- "bbox": {
- "x": int(object_res[0]),
- "y": int(object_res[1]),
- "height": int(object_res[2]),
- "width": int(object_res[3]),
- "object": str(object_res[4]),
- "object_det": {
- "confidence": float(object_res[5]),
- "frameH": int(object_res[6]),
- "frameW": int(object_res[7]),
- },
- },
- }
- return metadata, None
-
- # def get_gpu_rois_by_area(self, mask, max_candidates=25): #, limit_640=640*2): #max_candidates=100, limit_640=100):
- # # Get raw boxes from mask (Direct VRAM bridge)
- # boxes_gpu = find_contours_gpu_equivalent(
- # mask,
- # stream=self.bgs_stream,
- # grid_size=32,
- # limit_640=640*2,
- # max_boxes=100, #250,
- # ) # max_candidates)
-
- # if boxes_gpu is None or len(boxes_gpu) == 0:
- # return torch.empty((0, 4), device=self.device_input)
-
- # # Wrap existing GPU memory as a float tensor (Zero Copy)
- # raw_boxes = torch.as_tensor(boxes_gpu, device=self.device_input).float()
-
- # # Vectorized Pre-Filter (Removes noise blobs before merging)
- # w = raw_boxes[:, 2] - raw_boxes[:, 0]
- # h = raw_boxes[:, 3] - raw_boxes[:, 1]
- # mask_filter = (
- # (w * h > self.min_contour_area) & (w < self.resize_w) & (h < self.resize_h)
- # )
- # raw_boxes = raw_boxes[mask_filter]
-
- # # keep_idx = []
- # # for i in range(raw_boxes.shape[0]):
- # # # 1. Convert to Python integers and calculate width/height
- # # x1, y1, x2, y2 = [int(v.item()) for v in raw_boxes[i]]
- # # w, h = x2 - x1, y2 - y1
-
- # # if w <= 0 or h <= 0:
- # # continue
-
- # # try:
- # # # 2. Correct GpuMat ROI constructor: cv2.cuda.GpuMat(parent, (x, y, w, h))
- # # roi_mask = cv2.cuda.GpuMat(mask, (x1, y1, w, h))
-
- # # # 3. Density check (white pixels / area)
- # # # If density < 5%, it's likely the sparse terrain noise from your image
- # # if (cv2.cuda.countNonZero(roi_mask) / (w * h)) > 0.01:
- # # keep_idx.append(i)
- # # except Exception:
- # # # Skips boxes that might be slightly out of mask bounds
- # # continue
-
- # # raw_boxes = raw_boxes[keep_idx] if keep_idx else torch.empty((0, 4), device=self.device_input)
-
- # # Prevents N^2 distance matrix from exploding during high noise
- # if raw_boxes.shape[0] > max_candidates:
- # # Prioritize the largest blobs (most likely to be drones)
- # areas = (raw_boxes[:, 2] - raw_boxes[:, 0]) * (
- # raw_boxes[:, 3] - raw_boxes[:, 1]
- # )
- # _, indices = torch.topk(areas, max_candidates)
- # raw_boxes = raw_boxes[indices]
- # return raw_boxes
-
- def get_gpu_rois(self, frame, frameNum, mask):
- # If more than 20% of the screen is moving, don't bother with crops
- # if current_coverage > 0.6:
- # return torch.tensor([[0, 0, self.frame_width, self.frame_height]], device=self.device_input)
-
- limit_640 = 640 * 2 # 40 # self.config.ROI_MERGE_SIZE_LIMIT / self.scale_x
- raw_boxes = self.get_gpu_rois_by_area(
- mask, max_candidates=50
- ) # , limit_640=limit_640)
- # padding = 5 # self.config.ROI_BB_FULL_RES_PADDING / self.scale_x
- # raw_boxes[:, 0] -= padding
- # raw_boxes[:, 1] -= padding
- # raw_boxes[:, 2] += padding
- # raw_boxes[:, 3] += padding
-
- if raw_boxes.shape[0] < 1:
- return torch.empty((0, 4), device=self.device_input)
-
- if raw_boxes.shape[0] > 1:
- raw_boxes = merge_boxes_gpu(
- raw_boxes,
- gap_limit=self.dist_thresh_640,
- size_limit=limit_640,
+ main_app_logger.info(
+ f" > Tensor {i:3d} | Shape: {str(list(tensor.shape)):<18} | "
+ f"Size: {t_bytes / 1024**2:6.2f} MB | Source: \033[93m{source_loc}\033[0m"
)
- clean_640p = self.filter_contained_boxes(
- raw_boxes, overlap_thresh=self.config.ROI_CONTAINMENT_THRESH
+ del leaked_tensors
+ main_app_logger.info(
+ f"[VRAM INVESTIGATOR] Total Live Tensor Memory: {total_detected_bytes / 1024**2:.2f} MB"
)
+ gc.collect()
+ if (
+ torch.cuda.is_available()
+ ): # self.device_input == "cuda" and torch.cuda.is_available():
+ torch.cuda.synchronize()
+ torch.cuda.empty_cache() # Flush the pool BEFORE the guard snapshots it
+
+ def print_active_shared_memory(self):
+ main_app_logger.info(
+ "=" * 60,
+ )
+ main_app_logger.info(
+ "\033[96m[SHM INVESTIGATOR] Scanning Active OS Shared Memory Filesystem Tables:\033[0m"
+ )
+ try:
+ shm_dir = Path("/dev/shm")
+ if shm_dir.exists():
+ # Extract and inventory all live POSIX memory segments allocated right now
+ shm_files = [
+ f
+ for f in shm_dir.iterdir()
+ if f.is_file()
+ and not f.name.startswith("sem.")
+ and not f.name.startswith("psm")
+ ]
+ main_app_logger.info(
+ f" > Discovered Live OS-Mapped Memory Nodes: {len(shm_files)}"
+ )
- if clean_640p.shape[0] < 1:
- return torch.empty((0, 4), device=self.device_input)
-
- # # Scale to 8K space
- # # return clean_640p * self.scales_tensor
- # # margin = 0.10
- # # offsets = (clean_640p[:, 2:] - clean_640p[:, :2]) * margin
- # # clean_640p[:, :2] -= offsets
- # # clean_640p[:, 2:] += offsets
- # # 1. Add 30-pixel 'breathing room' (in 640p space)
- # # padding = 40 # self.config.ROI_BB_FULL_RES_PADDING / self.scale_x
- # # clean_640p[:, 0] -= padding
- # # clean_640p[:, 1] -= padding
- # # clean_640p[:, 2] += padding
- # # clean_640p[:, 3] += padding
-
- # # 2. Re-merge the padded boxes (connects nearby drones into one clean crop)
- # clean_640p = merge_boxes_gpu(
- # clean_640p,
- # gap_limit=self.dist_thresh_640,
- # size_limit=limit_640, # self.config.ROI_MERGE_SIZE_LIMIT / self.scale_x,
- # )
-
- # # Scale to 8K and clamp
- # clean_full = clean_640p * self.scales_tensor
- # # clean_full[:, [0, 2]] = clean_full[:, [0, 2]].clamp(0, self.frame_width)
- # # clean_full[:, [1, 3]] = clean_full[:, [1, 3]].clamp(0, self.frame_height)
- # return clean_full
-
- xmin = clean_640p[:, 0]
- ymin = clean_640p[:, 1]
- w = clean_640p[:, 2]
- h = clean_640p[:, 3]
-
- xmax = xmin + w
- ymax = ymin + h
-
- # Stack into standard format layout [xmin, ymin, xmax, ymax]
- standard_boxes = torch.stack([xmin, ymin, xmax, ymax], dim=1)
-
- # 4. Scale to absolute 8K workspace dimensions accurately
- return standard_boxes * self.scales_tensor
-
- # def pipeline_fn(self, device_frame, overall_frame_num, is_target_frame):
- # # ─── LAZY CUDA STREAM INITIALIZATION FOR PROCESS ISOLATION ───
- # # Ensures streams are bound directly to the active executing process memory space
- # if getattr(self, "device_input", "cpu") == "cuda":
- # if not hasattr(self, "stream") or self.stream is None:
- # self.stream = cv2.cuda.Stream()
- # if not hasattr(self, "bgs_stream") or self.bgs_stream is None:
- # self.bgs_stream = cv2.cuda.Stream()
- # # ─────────────────────────────────────────────────────────────
-
- # global all_metadata
- # current_clip_id = self.clip_id
- # current_clip_key = f"{self.name}_{current_clip_id:03d}.mp4"
- # current_clip_path = f"{self.config.SHARED_OUTPUT}/{current_clip_key}"
- # # --- MOTION MASK GENERATION GATE ---
- # if self.config.sf_enabled:
- # if self.device_input == "cuda":
- # inf_data = self.rbtd_full_gpu(device_frame)
- # torch.cuda.current_stream().synchronize()
- # if isinstance(inf_data, dict) and "mask" in inf_data:
- # if torch.is_tensor(inf_data["mask"]):
- # inf_data["mask"] = inf_data["mask"].contiguous()
- # else:
- # inf_data = self.rbtd_full_cpu(device_frame)
- # else:
- # inf_data = {}
-
- # # --- PIPELINE AT TARGET RATE ---
- # if not is_target_frame:
- # return
-
- # # if is_target_frame:
- # self.next_process_idx += self.step_size
- # self.frame_count_target += 1 # 1-indexed
- # self.frame_in_clip_count += 1
- # inf_data["frameNum"] = self.frame_count_target
-
- # # --- CLIP GENERATION ---
- # if self.config.ENABLE_QUERYING:
- # if self.config.DEBUG_FLAG and (
- # self.frame_in_clip_count % 15 == 0 or self.frame_in_clip_count == 1
- # ):
- # print(
- # f"[CLIPPER] Frame progress tracking index: {self.frame_in_clip_count}/{self.max_frames_per_clip} (Overall Frame: {overall_frame_num})",
- # flush=True,
- # )
-
- # self.prep_frame_for_video(device_frame, overall_frame_num)
-
- # if self.frame_in_clip_count > self.max_frames_per_clip:
- # global clip_completion_tracker
- # if current_clip_key not in clip_completion_tracker:
- # clip_completion_tracker[current_clip_key] = {
- # "video": False,
- # "meta": False,
- # "start_time": time.time(),
- # }
-
- # clip_completion_tracker[current_clip_key]["meta"] = True
- # print(
- # f" [BARRIER-SEAL] All metadata extracted for {current_clip_key}. Evaluating convergence...",
- # flush=True,
- # )
- # self._evaluate_barrier_and_dispatch(
- # current_clip_key,
- # current_clip_path,
- # self.resize_w,
- # self.resize_h,
- # )
-
- # self.start_new_clip()
-
- # if not self.config.DISABLE_DETECTION:
- # # --- FULL-RESOLUTION ROI EXTRACTION MAPS ---
- # bbs_full_res = None
- # if self.config.sf_enabled:
- # if self.device_input == "cuda":
- # bbs_full_res = self.get_gpu_rois(
- # inf_data["full_frame"],
- # self.frame_count_target,
- # inf_data["mask"],
- # )
- # else:
- # bbs_full_res = self.get_cpu_rois(
- # inf_data["full_frame"],
- # self.frame_count_target,
- # inf_data["mask"],
- # )
-
- # # Isolate raw coordinate matrices out of device graphs to prevent exit race conditions
- # # clean_bbs = []
- # # if self.config.sf_enabled and bbs_full_res is not None:
- # # if torch.is_tensor(bbs_full_res):
- # # clean_bbs = bbs_full_res.detach().cpu().tolist()
- # # elif isinstance(bbs_full_res, list):
- # # clean_bbs = [
- # # b.detach().cpu().tolist() if torch.is_tensor(b) else b
- # # for b in bbs_full_res
- # # ]
- # # else:
- # # clean_bbs = bbs_full_res
- # clean_bbs = []
- # if self.config.sf_enabled and bbs_full_res is not None:
- # if torch.is_tensor(bbs_full_res):
- # clean_bbs = bbs_full_res.detach().cpu().numpy()
- # else:
- # clean_bbs = np.array(bbs_full_res)
-
- # # print(f"[DEBUG] {current_clip_key}: {len(clean_bbs)} ROIs detected!")
-
- # if self.config.DETECTION_TYPE != "motion":
- # # Object Mode: Run YOLO and prepare metadata
- # det_frame = (
- # inf_data["full_frame"]
- # if "full_frame" in inf_data
- # else device_frame
- # ) # RGB
- # merged = clean_bbs if self.config.sf_enabled else None
- # # num_bbs = 0 if merged is None else len(clean_bbs)
- # # print(f"[DEBUG] {current_clip_key} 'merged' num bbs: {num_bbs}")
- # metadata, _ = self.get_detections(
- # det_frame,
- # self.frame_in_clip_count, # self.frame_count_target,
- # merged=merged,
- # thickness=self.config.THICKNESS,
- # device_input=self.config.device_input,
- # )
- # if self.config.DEBUG_FLAG:
- # meta_keys = ", ".join(list(metadata.keys()))
- # print(
- # f"[DEBUG] {current_clip_key} metadata keys: {meta_keys}",
- # flush=True,
- # )
- # if current_clip_key not in all_metadata:
- # all_metadata[current_clip_key] = {"object": {}, "face": {}}
-
- # all_metadata[current_clip_key]["object"].update(metadata)
- # # data_to_draw = metadata
-
- # # print(f"Sending to queue", flush=True)
-
- # display_source = (
- # inf_data["full_frame"]
- # if (inf_data and "full_frame" in inf_data)
- # else device_frame
- # )
-
- # if self.device_input == "cuda":
- # gpu_resized = F.interpolate(
- # display_source.unsqueeze(0).float(),
- # size=(self.disp_h, self.disp_w),
- # mode="bilinear",
- # align_corners=False,
- # ).squeeze(0).contiguous()
- # disp_frame = np.copy(
- # tensor2opencv(
- # gpu_resized, self.config.device_input, is_bgr=True
- # )
- # )
- # else:
- # cpu_resized = cv2.resize(device_frame, (self.disp_w, self.disp_h))
- # disp_frame = np.copy(
- # tensor2opencv(
- # cpu_resized, self.config.device_input, is_bgr=True
- # )
- # )
-
- # data_to_draw = (
- # clean_bbs if self.config.DETECTION_TYPE == "motion" else metadata
- # )
+ for f_path in shm_files:
+ try:
+ f_stat = f_path.stat()
+ size_mb = f_stat.st_size / (1024 * 1024)
- # # PUSH FRAME UNCONDITIONALLY: Ensures the test encoder gets raw frame tokens
- # try:
- # self.render_queue.put_nowait( # put(
- # (
- # disp_frame,
- # inf_data["frameNum"],
- # data_to_draw,
- # self.label_source,
- # )
- # )
- # except queue.Full:
- # pass
+ # Highlight the files using visual color anchors for scannability
+ main_app_logger.info(
+ f" ⚠️ \033[93m[ALIVE SHM NODE]\033[0m File: {f_path.name:<25} | Size: {size_mb:7.2f} MB"
+ )
+ except Exception:
+ pass
+ else:
+ main_app_logger.info(
+ " [ERROR] /dev/shm runtime directory is inaccessible on this host context."
+ )
+ except Exception as e:
+ main_app_logger.info(
+ f" [WARN] Kernel inspection execution pass failed: {e}"
+ )
- # self.update_frame()
+ # gc.collect()
+ # if torch.cuda.is_available():
+ # torch.cuda.synchronize()
+ # torch.cuda.empty_cache() # Flush the pool BEFORE the guard snapshots it
- def pipeline_fn(
- self, device_frame, overall_frame_num, is_target_frame, stat_start_time
+ def calculate_leaked_memory(
+ self, device, video_name, start_allocated, start_reserved
):
- global all_metadata
- current_clip_id = self.clip_id
- current_clip_key = f"{self.name}_{current_clip_id:03d}.mp4"
- current_clip_path = f"{self.config.SHARED_OUTPUT}/{current_clip_key}"
- # --- MOTION MASK GENERATION GATE ---
- if self.config.sf_enabled:
- if self.device_input == "cuda":
- inf_data = self.rbtd_full_gpu(device_frame)
- # torch.cuda.current_stream().synchronize()
- # if isinstance(inf_data, dict) and "mask" in inf_data:
- # if torch.is_tensor(inf_data["mask"]):
- # inf_data["mask"] = inf_data["mask"].contiguous()
- # inf_data = self.rbtd_full_gpu(device_frame)
- else:
- inf_data = self.rbtd_full_cpu(device_frame)
- else:
- inf_data = {}
+ _testMethodName = f"{video_name}_{device}"
+ main_app_logger.info("=" * 60)
+ max_allowed_leak = 1024 * 1024 # 1MB buffer allowance
+ msg = "[LEAKAGE INVESTIGATOR] Scanning memory allocations:\n"
+ if device == "gpu" and torch.cuda.is_available():
+ # torch.cuda.synchronize()
+
+ end_allocated = torch.cuda.memory_allocated(0)
+ end_reserved = torch.cuda.memory_reserved(0)
+
+ leak_allocated = end_allocated - start_allocated
+ leak_reserved = end_reserved - start_reserved
+ # max_allowed_leak = 1024 * 1024 # 1MB buffer allowance
+
+ if leak_allocated > max_allowed_leak:
+ msg += (
+ f"\n🔴 GPU Memory Leak Detected for {_testMethodName}!\n"
+ f"Check for dangling references or missing 'del' statements\n\n"
+ )
+ # else:
+ # msg = "\n"
- # --- PIPELINE AT TARGET RATE ---
- if not is_target_frame:
- return
+ msg += (
+ f"\tPre-Setup Allocation: {start_allocated / 1024**2:.2f} MB\n"
+ f"\tPost-Teardown Allocation: {end_allocated / 1024**2:.2f} MB\n"
+ f"\tNet Leaked VRAM: {leak_allocated / 1024**2:.2f} MB\n"
+ f"\tNet Leaked Reserved Blocks: {leak_reserved / 1024**2:.2f} MB"
+ )
- # if is_target_frame:
- # self.next_process_idx += self.step_size
- self.frame_count_target += 1 # 1-indexed
- self.frame_in_clip_count += 1
- inf_data["frameNum"] = self.frame_count_target
+ # main_app_logger.info(msg, )
+ else:
+ process = psutil.Process(os.getpid())
+ end_rss = process.memory_info().rss
- # --- CLIP GENERATION ---
- if self.config.ENABLE_QUERYING:
- if self.config.DEBUG_FLAG and (
- self.frame_in_clip_count % 15 == 0 or self.frame_in_clip_count == 1
- ):
- print(
- f"[CLIPPER] Frame progress tracking index: {self.frame_in_clip_count}/{self.max_frames_per_clip} (Overall Frame: {overall_frame_num})",
- flush=True,
- )
+ # start_allocated and start_reserved must be populated with baseline RSS in each_test_setup
+ leak_rss = end_rss - start_allocated
- self.prep_frame_for_video(device_frame, overall_frame_num)
-
- if self.frame_in_clip_count > self.max_frames_per_clip:
- global clip_completion_tracker
- if current_clip_key not in clip_completion_tracker:
- clip_completion_tracker[current_clip_key] = {
- "video": False,
- "meta": False,
- "start_time": time.time(),
- }
-
- clip_completion_tracker[current_clip_key]["meta"] = True
- print(
- f" [BARRIER-SEAL] All metadata extracted for {current_clip_key}. Evaluating convergence...",
- flush=True,
- )
- self._evaluate_barrier_and_dispatch(
- current_clip_key,
- current_clip_path,
- self.resize_w,
- self.resize_h,
+ if leak_rss > max_allowed_leak:
+ msg += (
+ f"\n🔴 CPU Memory Leak Detected for {_testMethodName}!\n"
+ f"Check for dangling references or missing 'del' statements\n\n"
)
+ # else:
+ # msg = "\n"
- self.start_new_clip()
+ msg += (
+ f" Pre-Setup Host RAM Allocation: {start_allocated / 1024**2:.2f} MB\n"
+ )
+ msg += f" Post-Teardown Host RAM Allocation: {end_rss / 1024**2:.2f} MB\n"
+ msg += f" Net Leaked Host System Memory: {leak_rss / 1024**2:.2f} MB"
+ main_app_logger.info(msg)
+ main_app_logger.info("=" * 60)
+
+ def diagnostic_profiler(self, device, video_name, start_allocated, start_reserved):
+ _testMethodName = f"{video_name}_{device}"
+ self.calculate_leaked_memory(
+ device, video_name, start_allocated, start_reserved
+ )
- if not self.config.DISABLE_DETECTION:
- # --- FULL-RESOLUTION ROI EXTRACTION MAPS ---
- bbs_full_res = None
- if self.config.sf_enabled:
- if self.device_input == "cuda":
- bbs_full_res = self.get_gpu_rois(
- inf_data["full_frame"],
- self.frame_count_target,
- inf_data["mask"],
- )
- else:
- bbs_full_res = self.get_cpu_rois(
- inf_data["full_frame"],
- self.frame_count_target,
- inf_data["mask"],
- )
+ # main_app_logger.info("=" * 60, )
+ # main_app_logger.info(
+ # f"\n\033[95m[DIAGNOSTICS] Starting Automated Leak Analysis for {_testMethodName}...\033[0m",
+ # ,
+ # )
- # Isolate raw coordinate matrices out of device graphs to prevent exit race conditions
- clean_bbs = []
- if self.config.sf_enabled and bbs_full_res is not None:
- if torch.is_tensor(bbs_full_res):
- clean_bbs = bbs_full_res.detach().cpu().numpy()
- else:
- clean_bbs = np.array(bbs_full_res)
+ # 1. Run objgraph to inspect the Python object reference trees before clearing containers
+ # try:
+ # main_app_logger.info(
+ # "\033[94m[DIAGNOSTICS] Python Object Registry Standings:\033[0m",
+ # )
+ # objgraph.show_most_common_types(limit=10)
+
+ # Check if the metrics arrays are pinning references inside memory
+ # for tracker_attr in ["all_preds", "all_targets"]:
+ # if hasattr(self, tracker_attr):
+ # tgt_list = getattr(self, tracker_attr)
+ # if len(tgt_list) > 0:
+ # graph_path = f"/tmp/backrefs_{tracker_attr}_{device}.png"
+ # main_app_logger.info(
+ # f"\033[93m[WARN] '{tracker_attr}' contains {len(tgt_list)} entries. Generating reference graph to: {graph_path}\033[0m"
+ # )
+ # objgraph.show_backrefs(
+ # [tgt_list], max_depth=3, filename=graph_path
+ # )
+ # except ImportError:
+ # main_app_logger.info(
+ # "\033[91m[DIAGNOSTICS] 'objgraph' package missing. Skipping reference chain mapping. (pip install objgraph)\033[0m"
+ # )
- # print(f"[DEBUG] {current_clip_key}: {len(clean_bbs)} ROIs detected!")
+ # 2. Dump PyTorch Memory Snapshot before clearing VRAM caches
+ if device == "gpu" and torch.cuda.is_available():
+ try:
+ # snapshot_path = f"/tmp/vram_leak_profile_{self._testMethodName}.pickle"
+ snapshot_path = self.output_path.replace(".mp4", "_vram_profile.html")
+ torch.cuda.memory._dump_snapshot(snapshot_path)
+ main_app_logger.info(
+ f"\033[92m[DIAGNOSTICS] VRAM Snapshot Trace generated successfully: {snapshot_path}\033[0m"
+ )
+ main_app_logger.info(
+ "\033[92m--> Upload this file to https://pytorch.org to inspect leak allocation stacks.\033[0m"
+ )
- if self.config.DETECTION_TYPE != "motion":
- # Object Mode: Run YOLO and prepare metadata
- if "full_frame" in inf_data:
- det_frame = inf_data["full_frame"]
- else:
- det_frame = (
- device_frame.clone()
- if torch.is_tensor(device_frame)
- else device_frame.copy()
- )
+ with open(snapshot_path, "rb") as f:
+ snapshot = pickle.load(f)
- merged = clean_bbs if self.config.sf_enabled else None
- # num_bbs = 0 if merged is None else len(clean_bbs)
- # print(f"[DEBUG] {current_clip_key} 'merged' num bbs: {num_bbs}")
- metadata, _ = self.get_detections(
- det_frame,
- self.frame_in_clip_count, # self.frame_count_target,
- merged=merged,
- thickness=self.config.THICKNESS,
- device_input=self.config.device_input,
+ # Print an HTML visualization path map of the allocations
+ html_timeline = memory_viz.trace_plot(snapshot)
+ html_path = snapshot_path.replace(".pickle", ".html")
+ with open(html_path, "w", encoding="utf-8") as f:
+ f.write(html_timeline)
+ except Exception as e:
+ main_app_logger.info(
+ f"\033[91m[DIAGNOSTICS] Failed to generate PyTorch memory snapshot: {e}\033[0m"
)
- if self.config.DEBUG_FLAG:
- meta_keys = ", ".join(list(metadata.keys()))
- print(
- f"[DEBUG] {current_clip_key} metadata keys: {meta_keys}",
- flush=True,
- )
- if current_clip_key not in all_metadata:
- all_metadata[current_clip_key] = {"object": {}, "face": {}}
- all_metadata[current_clip_key]["object"].update(metadata)
- # data_to_draw = metadata
+ def assess_memory(self, device, video_name, start_allocated, start_reserved):
+ gc.collect()
+
+ surviving_arrays = [
+ obj
+ for obj in gc.get_objects()
+ # if isinstance(obj, np.ndarray) and obj.size >= 1 #(1920 * 1080)
+ if type(obj) is np.ndarray and obj.ndim > 0
+ ]
+ analyze_tracemalloc_snapshot()
- # print(f"Sending to queue", flush=True)
+ main_app_logger.info(
+ f"[DIAGNOSTICS] Found {len(surviving_arrays)} uncollected large arrays alive in RAM filesystem."
+ )
- display_source = (
- inf_data["full_frame"]
- if (inf_data and "full_frame" in inf_data)
- else device_frame
+ for i, arr in enumerate(surviving_arrays):
+ referrers = gc.get_referrers(arr)
+ main_app_logger.info(
+ f" > Array {i} | Shape: {arr.shape} | Pinned by {len(referrers)} references:"
)
-
- if self.device_input == "cuda":
- gpu_resized = (
- F.interpolate(
- display_source.unsqueeze(0).float(),
- size=(self.disp_h, self.disp_w),
- mode="bilinear",
- align_corners=False,
+ for ref in referrers:
+ if isinstance(ref, dict):
+ main_app_logger.info(
+ f" - Dict Keys holding this array: {list(ref.keys())[:4]}"
)
- .squeeze(0)
- .contiguous()
- )
- disp_frame = np.copy(
- tensor2opencv(gpu_resized, self.config.device_input, is_bgr=True)
- )
- else:
- cpu_resized = cv2.resize(device_frame, (self.disp_w, self.disp_h))
- disp_frame = np.copy(
- tensor2opencv(cpu_resized, self.config.device_input, is_bgr=True)
- )
+ else:
+ main_app_logger.info(
+ f" - Variable holding object layout: {type(ref)}",
+ )
+ # ───────────────────────────────────────
- data_to_draw = (
- clean_bbs if self.config.DETECTION_TYPE == "motion" else metadata
- )
+ # analyze_tracemalloc_snapshot()
- # PUSH FRAME UNCONDITIONALLY: Ensures the test encoder gets raw frame tokens
- try:
- if (
- hasattr(self, "render_queue")
- and getattr(self, "render_queue", None) is not None
- # and not self.render_queue.full()
- ):
- self.render_queue.put( # put( put_nowait
- (
- disp_frame,
- inf_data["frameNum"],
- data_to_draw,
- self.label_source,
- )
- )
- except queue.Full:
- pass
+ if device == "gpu":
+ self.print_active_gpu_tensor_memory()
+ else:
+ # --- CPU RAM PATH METRICS ---
+ # main_app_logger.info("=" * 60, )
+ # main_app_logger.info("\n[RAM INVESTIGATOR] Scanning Host CPU Memory Standings:", )
+ # process = psutil.Process(os.getpid())
+ # current_rss = process.memory_info().rss / (1024 * 1024) # Host RAM in MB
+ # main_app_logger.info(f" > Current Process Resident Set Size (RSS): {current_rss:.2f} MB", )
+ self.print_active_cpu_tensor_memory()
- self.update_frame(stat_start_time)
+ self.print_active_shared_memory()
- # VIDEO CLIPPING
- def start_new_clip(self):
+ # AUTOMATED DIAGNOSTIC PROFILING PHASE (TRIGGERED ON TEST TEARDOWN)
+ self.diagnostic_profiler(device, video_name, start_allocated, start_reserved)
+
+ # VIDEO CLIPPING --------------------------------------------
+ def start_new_clip(self, clip_id):
"""
Seals the current AI tracking state layout and safely moves the instance metadata references
to the next sequential file block segment index.
"""
global clip_completion_tracker, all_metadata
- # Capture context pointers prior to counter mutation steps
- old_clip_id = self.clip_id
- old_clip_key = f"{self.name}_{old_clip_id:03d}.mp4"
- # old_clip_path = f"{self.config.SHARED_OUTPUT}/{old_clip_key}"
-
- print(
- f" [CLIPPER] Rotating AI context engine tracking timeline layer. Sealing metadata for: {old_clip_key}",
- flush=True,
- )
-
- # Seal the AI processing side of the tracker
- # if old_clip_key not in clip_completion_tracker:
- # clip_completion_tracker[old_clip_key] = {"video": False, "meta": False, "start_time": time.time()}
-
- # clip_completion_tracker[old_clip_key]["meta"] = True
-
- # # Evaluate the barrier in case the video segment completed before the AI loop reached this gate
- # self._evaluate_barrier_and_dispatch(old_clip_key, old_clip_path, self.resize_w, self.resize_h)
-
# Mutate tracker instance metrics parameters for the upcoming segment chunk window
- self.clip_id += 1
+ # self.clip_id += 1
self.frame_in_clip_count = 1
self._check_shm_safety(threshold_percent=90)
log_to_logger(
- f"New clip created: clip frame {self.frame_in_clip_count} ({self.frame_count_target})) of {self.max_frames_per_clip}",
+ f"New clip created: clip frame {self.frame_in_clip_count} of {self.max_frames_per_clip} (Overall target frame: {self.frame_count_target})",
level="info",
)
@@ -3393,86 +3675,282 @@ def prep_frame_for_video(self, device_frame, frame_num):
return
if not hasattr(self, "write_queue") or self.write_queue is None:
- print(
+ main_app_logger.info(
" [CLIPPER-INIT] Missing write_queue footprint. Provisioning runtime workspace buffer...",
- flush=True,
)
- self.write_queue = queue.Queue(maxsize=300)
+ # self.write_queue = queue.Queue(maxsize=300)
+ # write_queue_size = int(self.target_fps / 2) if self.device_input == "cpu" else int(self.target_fps / 2)
+ write_queue_size = int(2 * self.target_fps)
+ self.write_queue = queue.Queue(maxsize=write_queue_size) # 300)
self.writer_done = False
if not self.config.TEST_MODE and (
not hasattr(self, "send_metadata_queue") or self.send_metadata_queue is None
):
- print(
+ main_app_logger.info(
" [CLIPPER-INIT] Binding instance metadata reference array layer dynamically...",
- flush=True,
)
- self.send_metadata_queue = queue.Queue()
+ self.send_metadata_queue = queue.Queue(maxsize=50)
if not hasattr(self, "stop_writer") or self.stop_writer is None:
self.stop_writer = threading.Event()
- if (
- not hasattr(self, "writer_thread")
- or self.writer_thread is None
- or not self.writer_thread.is_alive()
- ):
- print(
- " [CLIPPER-INIT] Target worker runtime thread is offline. Provisioning core consumer loop thread...",
- flush=True,
- )
- self.writer_thread = threading.Thread(
- target=self.video_writer_core_loop,
- args=(self.stop_writer,),
- daemon=True,
- )
- self.writer_thread.start()
+ # if (
+ # not hasattr(self, "writer_process")
+ # or self.writer_process is None
+ # or not self.writer_process.is_alive()
+ # ):
+ # main_app_logger.info(
+ # " [CLIPPER-INIT] Target worker runtime thread is offline. Provisioning core consumer loop thread...",
+ # )
+ # self.writer_process = threading.Thread(
+ # target=self.video_writer_core_loop,
+ # args=(self.stop_writer,),
+ # daemon=True,
+ # )
+ # self.writer_process.start()
+
+ # if getattr(self, "video_writer", None) is None:
+ # main_app_logger.info(
+ # " [CLIPPER-INIT] Downstream execution handle is blank. Initializing FFmpeg subprocess daemon...",
+ # )
+ # self._initialize_writer()
+
+ if not self.active or self._is_stopped:
+ return
+
+ # self.clip_executor.submit(self._async_clipper_worker, device_frame, frame_num)
+ try:
+ if hasattr(self, "clip_executor") and self.clip_executor is not None:
+ self.clip_executor.submit(
+ self._async_clipper_worker, device_frame, frame_num
+ )
+ except RuntimeError as e:
+ # If the executor was shut down in the fraction of a millisecond
+ # between our check and the submit, catch it silently and drop the frame.
+ if "cannot schedule new futures" in str(e):
+ pass
+ else:
+ # If it's a different RuntimeError, we still want to see it
+ raise e
+ except Exception:
+ # Catch other unexpected submission errors
+ # pass
+ traceback.print_exc()
+
+ @torch.inference_mode()
+ def _async_clipper_worker_v1(self, device_frame, frame_num):
+ """
+ A worker that resizes a full-resolution frame and places it in a queue
+ for the video writer. Runs in a separate thread.
+ """
+ # ring_idx = (frame_num - 1) % self.ring_depth
+ try:
+ # If a shutdown has started, attributes might be gone. Exit cleanly.
+ if (
+ not self.active
+ or not hasattr(self, "processing_stream")
+ or self.processing_stream is None
+ ):
+ return
+
+ if self.device_input == "cuda":
+ # with torch.cuda.stream(self.processing_stream):
+ if (
+ hasattr(self, "processing_stream")
+ and self.processing_stream is not None
+ ):
+ torch.cuda.set_stream(self.processing_stream)
+ if device_frame.shape[-1] == 3:
+ gpu_ch_first = device_frame.permute(2, 0, 1).contiguous()
+ else:
+ gpu_ch_first = device_frame.contiguous()
+
+ self.gpu_float_staging[0, :, :, :].copy_(
+ gpu_ch_first, non_blocking=True
+ )
+
+ gpu_resized = F.interpolate(
+ self.gpu_float_staging,
+ size=(self.resize_h, self.resize_w),
+ # mode="bilinear",
+ # align_corners=False,
+ mode="nearest",
+ ).squeeze(0)
+
+ gpu_final = gpu_resized.clamp(0, 255).to(torch.uint8)
+
+ # If the tensor shape follows PyTorch's standard format (Channels, Height, Width),
+ # we reverse the channel order [0, 1, 2] to [2, 1, 0] (RGB -> BGR)
+ if gpu_final.ndim == 3 and gpu_final.shape[0] == 3:
+ gpu_final = gpu_final[[2, 1, 0], :, :]
+
+ gpu_contiguous = gpu_final.permute(1, 2, 0).contiguous()
+
+ active_tensor = self.pinned_tensors[self.gpu_ring_idx]
+ active_tensor.copy_(gpu_contiguous, non_blocking=True)
+
+ if self.slot_events is not None:
+ self.slot_events[self.gpu_ring_idx].record(self.processing_stream)
+
+ # with self.write_queue_backlog_counter.get_lock():
+ # self.write_queue_backlog_counter.value += 1
+ # self.write_queue.put(
+ # {
+ # "ring_slot_idx": self.gpu_ring_idx,
+ # "frame_num": frame_num,
+ # "pipe_handle": self.video_writer,
+ # }
+ # )
+ try:
+ # Put the NumPy array directly into the queue
+ self.write_queue.put_nowait(active_tensor)
+
+ # (Optional) Increment your backlog counter if you still use it
+ with self.write_queue_backlog_counter.get_lock():
+ self.write_queue_backlog_counter.value += 1
+ except queue.Full:
+ pass # Drop the frame if the writer process is lagging
+ self.gpu_ring_idx = (self.gpu_ring_idx + 1) % self.ring_depth
+ else:
+ active_matrix = self.pinned_matrices[self.cpu_ring_idx]
+ cv2.resize(
+ device_frame,
+ (self.resize_w, self.resize_h),
+ dst=active_matrix,
+ interpolation=cv2.INTER_NEAREST,
+ )
+
+ # with self.write_queue_backlog_counter.get_lock():
+ # self.write_queue_backlog_counter.value += 1
+ # self.write_queue.put(
+ # {
+ # "ring_slot_idx": self.cpu_ring_idx,
+ # "frame_num": frame_num,
+ # "pipe_handle": self.video_writer,
+ # }
+ # )
+ try:
+ # Put the NumPy array directly into the queue
+ self.write_queue.put_nowait(active_matrix)
- if getattr(self, "video_writer", None) is None:
- print(
- " [CLIPPER-INIT] Downstream execution handle is blank. Initializing FFmpeg subprocess daemon...",
- flush=True,
+ # (Optional) Increment your backlog counter if you still use it
+ with self.write_queue_backlog_counter.get_lock():
+ self.write_queue_backlog_counter.value += 1
+ except queue.Full:
+ pass # Drop the frame if the writer process is lagging
+
+ self.cpu_ring_idx = (self.cpu_ring_idx + 1) % self.ring_depth
+
+ except Exception as e:
+ main_app_logger.info(
+ f"[CRITICAL-CLIPPER-WORKER] Resizing execution loop dropped: {e}",
)
- self._initialize_writer()
-
- self.clip_executor.submit(self._async_clipper_worker, device_frame, frame_num)
+ traceback.print_exc()
+ finally:
+ # This worker handles large 8K frames. Deleting all local tensor
+ # references here is critical to prevent VRAM accumulation.
+ if "device_frame" in locals():
+ del device_frame
+ if "gpu_ch_first" in locals():
+ del gpu_ch_first
+ if "gpu_resized" in locals():
+ del gpu_resized
+ if "gpu_final" in locals():
+ del gpu_final
+ if "gpu_contiguous" in locals():
+ del gpu_contiguous
+
+ @torch.inference_mode()
def _async_clipper_worker(self, device_frame, frame_num):
+ """
+ A worker that resizes a full-resolution frame and places it in a queue
+ for the video writer. Runs in a separate thread.
+ """
+ # ring_idx = (frame_num - 1) % self.ring_depth
try:
+ # If a shutdown has started, attributes might be gone. Exit cleanly.
+ if (
+ not self.active
+ or not hasattr(self, "processing_stream")
+ or self.processing_stream is None
+ ):
+ return
+
if self.device_input == "cuda":
- with torch.cuda.stream(self.processing_stream):
- if device_frame.shape[-1] == 3:
- gpu_ch_first = device_frame.permute(2, 0, 1).contiguous()
- else:
- gpu_ch_first = device_frame.contiguous()
+ # with torch.cuda.stream(self.processing_stream):
+ if (
+ hasattr(self, "processing_stream")
+ and self.processing_stream is not None
+ ):
+ torch.cuda.set_stream(self.processing_stream)
+ if device_frame.shape[-1] == 3:
+ gpu_ch_first = device_frame.permute(2, 0, 1).contiguous()
+ else:
+ gpu_ch_first = device_frame.contiguous()
- self.gpu_float_staging[0, :, :, :].copy_(
- gpu_ch_first, non_blocking=True
- )
+ self.gpu_float_staging[0, :, :, :].copy_(
+ gpu_ch_first, non_blocking=True
+ )
+
+ gpu_resized = F.interpolate(
+ self.gpu_float_staging,
+ size=(self.resize_h, self.resize_w),
+ # mode="bilinear",
+ # align_corners=False,
+ mode="nearest",
+ ).squeeze(0)
+
+ gpu_final = gpu_resized.clamp(0, 255).to(torch.uint8)
- gpu_resized = F.interpolate(
- self.gpu_float_staging,
- size=(self.resize_h, self.resize_w),
- mode="bilinear",
- align_corners=False,
- ).squeeze(0)
+ # If the tensor shape follows PyTorch's standard format (Channels, Height, Width),
+ # we reverse the channel order [0, 1, 2] to [2, 1, 0] (RGB -> BGR)
+ if gpu_final.ndim == 3 and gpu_final.shape[0] == 3:
+ gpu_final = gpu_final[[2, 1, 0], :, :]
- gpu_final = gpu_resized.clamp(0, 255).to(torch.uint8)
- gpu_contiguous = gpu_final.permute(1, 2, 0).contiguous()
+ gpu_contiguous = gpu_final.permute(1, 2, 0).contiguous()
- active_tensor = self.pinned_tensors[self.gpu_ring_idx]
- active_tensor.copy_(gpu_contiguous, non_blocking=True)
+ active_tensor = self.pinned_tensors[self.gpu_ring_idx]
+ active_tensor.copy_(gpu_contiguous, non_blocking=True)
if self.slot_events is not None:
self.slot_events[self.gpu_ring_idx].record(self.processing_stream)
- self.write_queue.put(
- {
- "ring_slot_idx": self.gpu_ring_idx,
- "frame_num": frame_num,
- "pipe_handle": self.video_writer,
- }
- )
+ # with self.write_queue_backlog_counter.get_lock():
+ # self.write_queue_backlog_counter.value += 1
+ # self.write_queue.put(
+ # {
+ # "ring_slot_idx": self.gpu_ring_idx,
+ # "frame_num": frame_num,
+ # "pipe_handle": self.video_writer,
+ # }
+ # )
+
+ # try:
+ # # Put the NumPy array directly into the queue
+ # self.write_queue.put_nowait(active_tensor)
+
+ # # (Optional) Increment your backlog counter if you still use it
+ # with self.write_queue_backlog_counter.get_lock():
+ # self.write_queue_backlog_counter.value += 1
+ # except queue.Full:
+ # pass # Drop the frame if the writer process is lagging
+
+ # 1. Get the next available ring buffer slot atomically
+ with self.clipper_idx_lock:
+ slot_idx = self.clipper_ring_idx.value
+ self.clipper_ring_idx.value = (
+ self.clipper_ring_idx.value + 1
+ ) % self.clipper_ring_depth
+
+ # 2. Get the NumPy view for that slot and copy the data into it
+ target_buffer = self.clipper_shm_np_views[slot_idx]
+ np.copyto(target_buffer, active_tensor)
+
+ # 3. Put the INTEGER INDEX into the queue (extremely fast)
+ self.write_queue.put_nowait(slot_idx)
+
self.gpu_ring_idx = (self.gpu_ring_idx + 1) % self.ring_depth
else:
active_matrix = self.pinned_matrices[self.cpu_ring_idx]
@@ -3480,25 +3958,65 @@ def _async_clipper_worker(self, device_frame, frame_num):
device_frame,
(self.resize_w, self.resize_h),
dst=active_matrix,
- interpolation=cv2.INTER_LINEAR,
+ interpolation=cv2.INTER_NEAREST,
)
- self.write_queue.put(
- {
- "ring_slot_idx": self.cpu_ring_idx,
- "frame_num": frame_num,
- "pipe_handle": self.video_writer,
- }
- )
+ # with self.write_queue_backlog_counter.get_lock():
+ # self.write_queue_backlog_counter.value += 1
+ # self.write_queue.put(
+ # {
+ # "ring_slot_idx": self.cpu_ring_idx,
+ # "frame_num": frame_num,
+ # "pipe_handle": self.video_writer,
+ # }
+ # )
+
+ # try:
+ # # Put the NumPy array directly into the queue
+ # self.write_queue.put_nowait(active_matrix)
+
+ # # (Optional) Increment your backlog counter if you still use it
+ # with self.write_queue_backlog_counter.get_lock():
+ # self.write_queue_backlog_counter.value += 1
+ # except queue.Full:
+ # pass # Drop the frame if the writer process is lagging
+
+ # 1. Get the next available ring buffer slot atomically
+ with self.clipper_idx_lock:
+ slot_idx = self.clipper_ring_idx.value
+ self.clipper_ring_idx.value = (
+ self.clipper_ring_idx.value + 1
+ ) % self.clipper_ring_depth
+
+ # 2. Get the NumPy view for that slot and copy the data into it
+ target_buffer = self.clipper_shm_np_views[slot_idx]
+ np.copyto(target_buffer, active_matrix)
+
+ # 3. Put the INTEGER INDEX into the queue (extremely fast)
+ self.write_queue.put_nowait(slot_idx)
+
self.cpu_ring_idx = (self.cpu_ring_idx + 1) % self.ring_depth
except Exception as e:
- print(
+ main_app_logger.info(
f"[CRITICAL-CLIPPER-WORKER] Resizing execution loop dropped: {e}",
- flush=True,
)
traceback.print_exc()
+ finally:
+ # This worker handles large 8K frames. Deleting all local tensor
+ # references here is critical to prevent VRAM accumulation.
+ if "device_frame" in locals():
+ del device_frame
+ if "gpu_ch_first" in locals():
+ del gpu_ch_first
+ if "gpu_resized" in locals():
+ del gpu_resized
+ if "gpu_final" in locals():
+ del gpu_final
+ if "gpu_contiguous" in locals():
+ del gpu_contiguous
+
def _initialize_writer(self):
"""
Spawns a persistent background FFmpeg subprocess with native segment-splitting
@@ -3523,9 +4041,13 @@ def _initialize_writer(self):
"-i",
"-",
"-c:v",
- "libx264", #"mpeg4", # Or "libx264" if you prefer H.264
- "-crf","23",
- "-f","mpegts","-movflags","faststart",
+ "libx264", # "mpeg4", # Or "libx264" if you prefer H.264
+ "-crf",
+ "23",
+ "-f",
+ "mpegts",
+ "-movflags",
+ "faststart",
"-force_key_frames",
f"expr:gte(t,n_forced*{clip_duration})",
"-f",
@@ -3542,7 +4064,7 @@ def _initialize_writer(self):
self.clip_filename_pattern,
]
- # print(f" [FFMPEG-INIT] Spawning binary pipeline targeted at: {self.clip_filename_pattern}", flush=True)
+ # main_app_logger.info(f" [FFMPEG-INIT] Spawning binary pipeline targeted at: {self.clip_filename_pattern}")
try:
# log_dir = "/home/logs"
@@ -3560,25 +4082,23 @@ def _initialize_writer(self):
bufsize=0,
)
self.video_writer = self.ffmpeg_proc.stdin
- print(
+ main_app_logger.info(
" [FFMPEG-INIT] Subprocess online. Log stream parser engine initialization sequencing...",
- flush=True,
)
# Fire memory loop monitor
- self.log_parser_thread = threading.Thread(
- target=self._ffmpeg_log_parser_loop, daemon=True
- )
- self.log_parser_thread.start()
+ # self.log_parser_thread = threading.Thread(
+ # target=self._ffmpeg_log_parser_loop, daemon=True
+ # )
+ # self.log_parser_thread.start()
log_to_logger(
f"Persistent native-segmenting FFmpeg writer spawned for stream: {self.name}",
level="info",
)
except Exception as e:
- print(
+ main_app_logger.info(
f"[CRITICAL-FFMPEG] Process spawn aborted at kernel boundary: {e}",
- flush=True,
)
self.video_writer = None
@@ -3657,9 +4177,8 @@ def _ffmpeg_log_parser_loop(self):
) # Rest the polling thread to preserve CPU cycles
if not file_stable:
- print(
+ main_app_logger.info(
f" [PARSER-WARN] IO Flush timeout exceeded for {completed_clip_key}. Forcing dispatch anyway.",
- flush=True,
)
# =========================================================================
@@ -3675,9 +4194,8 @@ def _ffmpeg_log_parser_loop(self):
True
)
if self.config.DEBUG_FLAG:
- print(
+ main_app_logger.info(
f"[PARSER] Memory pipe intercepted completed segment confirmation: {completed_clip_key}",
- flush=True,
)
# Run convergence evaluation pass
@@ -3688,128 +4206,147 @@ def _ffmpeg_log_parser_loop(self):
self.resize_h,
)
except Exception as calc_err:
- print(
+ main_app_logger.info(
f"[PARSER-WARN] Index calculation lookback anomaly skipped: {calc_err}",
- flush=True,
)
except Exception as parse_err:
- print(
+ main_app_logger.info(
f"[PARSER-ERROR] Failed to extract target token patterns out of logging stream: {parse_err}",
- flush=True,
)
continue
if self.config.DEBUG_FLAG:
- print(
+ main_app_logger.info(
" [LOG-PARSER] Memory loop pipe interface closed down smoothly.",
- flush=True,
)
- def video_writer_core_loop(self, stop_evt):
- """
- Thread-safe background min-heap consumer with adaptive sequence hole recovery.
- """
- print(
- " [WRITER-LOOP] Background tracking consumer loop active and polling memory queues...",
- flush=True,
- )
- try:
- while not stop_evt.is_set() or not self.write_queue.empty():
- try:
- data = None
- try:
- data = self.write_queue.get(timeout=0.02)
- except (queue.Empty, AttributeError):
- continue
-
- if data is None:
- continue
-
- control_data = data.get("control")
- if control_data == "FLUSH":
- print(
- " [WRITER-LOOP] Intercepted downstream engine flush token code signature.",
- flush=True,
- )
- if "pipe_handle" in data and data["pipe_handle"]:
- try:
- data["pipe_handle"].close()
- except Exception:
- pass
- self.write_queue.task_done()
- continue
-
- slot_target = data.get("ring_slot_idx")
- sock_handle = data.get("pipe_handle")
- # frame_num = data.get("frame_num")
+ # def video_writer_core_loop(self, stop_evt):
+ # """Thread-safe background min-heap consumer with adaptive sequence hole recovery
- if (
- sock_handle is None
- and getattr(self, "video_writer", None) is not None
- ):
- sock_handle = self.video_writer
-
- if slot_target is not None and sock_handle is not None:
- if self.device_input == "cuda" and self.slot_events is not None:
- self.slot_events[slot_target].synchronize()
-
- if (
- self.ffmpeg_proc is None
- or self.ffmpeg_proc.poll() is not None
- ):
- print(
- " [WRITER-WARN] Downstream execution loop pipe was broken out-of-band. Launching recovery routine...",
- flush=True,
- )
- self._initialize_writer()
- sock_handle = self.video_writer
- if sock_handle is None:
- self.write_queue.task_done()
- continue
+ # and leak-proof lifecycle task tracking signals.
+ # """
+ # main_app_logger.info(
+ # " [WRITER-LOOP] Background tracking consumer loop active and polling memory queues...",
+ # )
+ # try:
+ # # Use a safe helper function to check if the queue is both active and not empty
+ # def is_queue_active_and_populated():
+ # if self.write_queue is None:
+ # return False
+ # try:
+ # return not self.write_queue.empty()
+ # except (AttributeError, ValueError):
+ # # In case the queue was just destroyed by the main thread
+ # return False
+
+ # while not stop_evt.is_set() or not is_queue_active_and_populated():
+ # try:
+ # data = None
+ # try:
+ # data = self.write_queue.get(timeout=0.02)
+ # except (queue.Empty, AttributeError):
+ # continue
+
+ # if data is None:
+ # continue
+
+ # # --- DETACHED PROCESSING LOGIC ENVELOPE ------------------
+ # try:
+ # control_data = data.get("control")
+ # if control_data == "FLUSH":
+ # main_app_logger.info(
+ # " [WRITER-LOOP] Intercepted downstream engine flush token code signature.",
+ # )
+ # if "pipe_handle" in data and data["pipe_handle"]:
+ # try:
+ # data["pipe_handle"].close()
+ # except Exception:
+ # pass
+ # continue
- try:
- raw_buffer_view = memoryview(
- self.pinned_matrices[slot_target]
- )
- sock_handle.write(raw_buffer_view)
- sock_handle.flush()
+ # slot_target = data.get("ring_slot_idx")
+ # sock_handle = data.get("pipe_handle")
- except (OSError, ValueError) as pipe_err:
- print(
- f" [PIPE-ERROR] Write operation dropped on ring index slot {slot_target}: {pipe_err}",
- flush=True,
- )
- pass
+ # if (
+ # sock_handle is None
+ # and getattr(self, "video_writer", None) is not None
+ # ):
+ # sock_handle = self.video_writer
- self.write_queue.task_done()
- else:
- self.write_queue.task_done()
+ # if slot_target is not None and sock_handle is not None:
+ # if (
+ # self.device_input == "cuda"
+ # and self.slot_events is not None
+ # ):
+ # self.slot_events[slot_target].synchronize()
- except Exception as e:
- print(
- f"[WRITER-EXCEPTION] Worker engine cycle processing failure: {e}",
- flush=True,
- )
- continue
+ # if (
+ # self.ffmpeg_proc is None
+ # or self.ffmpeg_proc.poll() is not None
+ # ):
+ # main_app_logger.info(
+ # " [WRITER-WARN] Downstream execution loop pipe was broken out-of-band. Launching recovery routine...",
+ # )
+ # self.initialize_writer()
+ # sock_handle = self.video_writer
+ # if sock_handle is None:
+ # continue
+
+ # try:
+ # raw_buffer_view = memoryview(
+ # self.pinned_matrices[slot_target]
+ # )
+ # sock_handle.write(raw_buffer_view)
+ # sock_handle.flush()
+ # except (OSError, ValueError) as pipe_err:
+ # main_app_logger.info(
+ # f" [PIPE-ERROR] Write operation dropped on ring index slot {slot_target}: {pipe_err}",
+ # )
+ # finally:
+ # if "raw_buffer_view" in locals():
+ # del raw_buffer_view
+
+ # del slot_target, sock_handle
+
+ # with self.write_queue_backlog_counter.get_lock():
+ # self.write_queue_backlog_counter.value -= 1
+ # if (
+ # self.write_queue_backlog_counter.value % 30 == 0
+ # ): # Every ~2 seconds
+ # gc.collect()
+ # if torch.cuda.is_available():
+ # torch.cuda.empty_cache()
+
+ # # --- ENSURE PRECISE TASK TRACKING REGISTRATION ALLOCATION ---
+ # finally:
+ # if "data" in locals():
+ # del data
+ # # This executes exactly once per queue iteration pass, completely
+ # # eliminating "task_done() called too many times" exceptions.
+ # self.write_queue.task_done()
+
+ # except Exception as e:
+ # main_app_logger.info(
+ # f" [WRITER-EXCEPTION] Worker engine cycle processing failure: {e}",
+ # )
+ # continue
- try:
- if hasattr(self, "socket_path") and os.path.exists(self.socket_path):
- os.remove(self.socket_path)
- except Exception:
- pass
+ # if hasattr(self, "socket_path") and os.path.exists(self.socket_path):
+ # try:
+ # os.remove(self.socket_path)
+ # except Exception:
+ # pass
- self.writer_done = True
- print(
- " [WRITER-LOOP] Thread pool queue completely drained. Processing safe exit termination sequence...",
- flush=True,
- )
+ # self.writer_done = True
+ # main_app_logger.info(
+ # " [WRITER-LOOP] Thread pool queue completely drained. Processing safe exit termination sequence...",
+ # )
- except Exception as fatal_err:
- print(
- f"[FATAL-WRITER-CRASH] Unhandled background crash: {fatal_err}",
- flush=True,
- )
- traceback.print_exc()
+ # except Exception as fatal_err:
+ # main_app_logger.info(
+ # f"[FATAL-WRITER-CRASH] Unhandled background crash: {fatal_err}",
+ # )
+ # traceback.print_exc()
def _evaluate_barrier_and_dispatch(self, clip_key, clip_path, frame_w, frame_h):
"""
@@ -3830,14 +4367,23 @@ def _evaluate_barrier_and_dispatch(self, clip_key, clip_path, frame_w, frame_h):
# Check convergence layout: Trigger upload sequence only if both pipelines sealed operations
if tracker["video"] and tracker["meta"]:
if self.config.DEBUG_FLAG:
- print(
+ main_app_logger.info(
f" [BARRIER-CONVERGENCE] Fully synchronized state reached for asset: {clip_key}",
- flush=True,
)
# Extract and unmap metadata tracking payloads safely from shared RAM memory space
- clip_metadata = all_metadata.pop(clip_key, None)
- clip_completion_tracker.pop(clip_key, None)
+ # clip_metadata = all_metadata.pop(clip_key, None)
+ # clip_completion_tracker.pop(clip_key, None)
+ clip_metadata = all_metadata.get(clip_key)
+ # Immediately delete the data from the global dictionaries to release VRAM
+ if clip_key in all_metadata:
+ del all_metadata[clip_key]
+ if clip_key in clip_completion_tracker:
+ del clip_completion_tracker[clip_key]
+
+ gc.collect()
+ if "cuda" in self.device_input:
+ torch.cuda.empty_cache()
if not self.config.TEST_MODE and clip_metadata:
# Re-insert the fully constructed framework into the active VDMS queue worker pool
@@ -3850,14 +4396,12 @@ def _evaluate_barrier_and_dispatch(self, clip_key, clip_path, frame_w, frame_h):
global send_metadata_queue
send_metadata_queue.put((clip_path, frame_w, frame_h))
- print(
+ main_app_logger.info(
f" [BARRIER-INGEST] Unified data packages successfully submitted for DB processing: {clip_key}",
- flush=True,
)
elif not self.config.TEST_MODE:
- print(
+ main_app_logger.info(
f" [BARRIER-WARN] Synchronization completed but all_metadata structure for {clip_key} was empty!",
- flush=True,
)
elif not self.config.TEST_MODE:
waiting_on = (
@@ -3865,185 +4409,12 @@ def _evaluate_barrier_and_dispatch(self, clip_key, clip_path, frame_w, frame_h):
if not tracker["video"]
else "AI frame processing execution"
)
- print(
+ main_app_logger.info(
f" [BARRIER-WAIT] {clip_key}: Milestone checked. Awaiting {waiting_on} before issuing DB ingestion call.",
- flush=True,
)
class GPUStreamHandler(DeviceBaseHandler):
- def allocate_gpu(self):
- """
- Allocates persistent GpuMat buffers and CUDA streams to
- enable zero-copy GPU processing.
- """
- self.stream = cv2.cuda.Stream()
- self.ingest_stream = torch.cuda.Stream()
- self.inference_stream = torch.cuda.Stream()
- self.bgs_stream = cv2.cuda.Stream()
- self.gpu_fullres_frame = cv2.cuda.GpuMat(
- self.frame_height, self.frame_width, cv2.CV_8UC3
- )
- # self.resized_gpumat = cv2.cuda.GpuMat(self.resize_h, self.resize_w, cv2.CV_8UC3)
- self.resized_frame = cv2.cuda.GpuMat(self.resize_h, self.resize_w, cv2.CV_8UC3)
- self.resized_frame.setTo(0, self.bgs_stream)
- self.fgMask = cv2.cuda.GpuMat(self.resize_h, self.resize_w, cv2.CV_8UC1)
- self.prev_bkgd = cv2.cuda.GpuMat(self.resize_h, self.resize_w, cv2.CV_8UC1)
- if self.config.BKGD_SUB_INCLUDE_HISTORY_METHOD == "and":
- self.prev_bkgd.setTo((1,))
- else:
- self.prev_bkgd.setTo((0,))
- self.mask_history = deque(
- maxlen=self.config.BKGD_SUB_INCLUDE_HISTORY_TEMPORAL_SIZE
- )
- self.mask_history.append(self.prev_bkgd)
- self.gpu_threshold_dst_frame = cv2.cuda.GpuMat(
- self.resize_h, self.resize_w, cv2.CV_8UC1
- )
- self.gpu_morphed_frame = cv2.cuda.GpuMat(
- self.resize_h, self.resize_w, cv2.CV_8UC1
- )
-
- self.upload_stream = cv2.cuda.Stream()
- self.upload_event = cv2.cuda.Event()
-
- # self.queue_capacity = int(2 * self.target_fps) # 60
- self.num_buffers = int(2 * self.target_fps) # self.queue_capacity + 5
- self.gpu_buffer_pool = [
- cv2.cuda.GpuMat(self.frame_height, self.frame_width, cv2.CV_8UC3)
- for _ in range(self.num_buffers)
- ]
- self.buffer_idx = 0
-
- self.frame_buffer_pool = [
- torch.empty(
- (3, self.frame_height, self.frame_width),
- dtype=torch.uint8,
- device="cuda",
- )
- for _ in range(2)
- ]
- self.pool_idx = 0
-
- # Create a matching pool of pinned host memory for the 8K frames
- self.host_buffer_pool = [
- cv2.cuda.HostMem(self.frame_height, self.frame_width, cv2.CV_8UC3)
- for _ in range(self.num_buffers)
- ]
-
- # Create continuous buffers to prevent stride artifacts during 8K downloads
- self.pinned_downloaded_resizedframe_np = cv2.cuda.createContinuous(
- self.resize_h, self.resize_w, cv2.CV_8UC3
- )
- self.pinned_downloaded_frame_np = cv2.cuda.createContinuous(
- self.resize_h, self.resize_w, cv2.CV_8UC1
- )
- cv2.cuda.createContinuous(
- self.resize_h, self.resize_w, cv2.CV_8UC1, self.gpu_threshold_dst_frame
- )
- cv2.cuda.createContinuous(
- self.resize_h, self.resize_w, cv2.CV_8UC1, self.gpu_morphed_frame
- )
-
- # This prevents the AI thread from overwriting the encoder's data.
- self.gpu_encoder_8k_buf = cv2.cuda.createContinuous(
- self.frame_height, self.frame_width, cv2.CV_8UC3
- )
-
- # Continuous allocation prevents stride/padding artifacts
- self.gpu_display_frame = cv2.cuda.createContinuous(
- self.disp_h, self.disp_w, cv2.CV_8UC3
- )
-
- # Create a dedicated background stream for encoding tasks
- self.encode_stream = cv2.cuda.Stream()
-
- def prepare_gpu_pipeline(self):
- self.operation_device_map = PipelineMapping(
- resize_device="gpu",
- bkgd_subtraction_device="gpu",
- threshold_device="gpu",
- erodeAndDilate_device="gpu",
- detection_device="gpu",
- ) # rbtd_detection_gpu
-
- self.device_input = "cuda"
-
- self.allocate_gpu()
-
- # Subtraction
- # history = int(2 * self.target_fps) # 300 # int(5 * self.target_fps)
- self.lr = self.config.BKGD_SUB_MOG2_LR # 1 / history
- self.backSub = cv2.cuda.createBackgroundSubtractorMOG2(
- history=self.config.BKGD_SUB_MOG2_HISTORY, # Clear ghosts of fast drones in ~2 seconds (2*fps)
- varThreshold=int(
- 1.15 * self.config.BKGD_SUB_MOG2_VARTHRESHOLD
- ), # High threshold to ignore "shimmer" and compression noise # default 16
- # CUDA implementation of MOG2 often requires a higher varThreshold to achieve the same "cleanliness" as the CPU (15-20%)
- detectShadows=self.config.BKGD_SUB_MOG2_DETECTSHADOWS, # default True
- )
-
- self.dilate_filter = cv2.cuda.createMorphologyFilter(
- cv2.MORPH_DILATE, cv2.CV_8U, self.dilate_kernel
- )
- self.dilate_filter_for_enhanced_mask = cv2.cuda.createMorphologyFilter(
- cv2.MORPH_DILATE, cv2.CV_8UC1, self.dilate_kernel_for_enhanced_mask
- )
- # # self.morph_kernel = cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (5, 5))
- # self.morph_kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (5, 5))
- # self.morph_filter = cv2.cuda.createMorphologyFilter(
- # cv2.MORPH_DILATE, cv2.CV_8UC1, self.morph_kernel
- # )
- # self.labels_gpu = cv2.cuda.GpuMat(self.resize_h, self.resize_w, cv2.CV_32S)
- self.labels_gpu = cv2.cuda.GpuMat(self.resize_h, self.resize_w, cv2.CV_8U)
- self.labels_gpu.setTo(0, self.bgs_stream)
-
- def gpu_warmup(self):
- # WARM UP (Crucial for first-run latency)
- # JIT kernels are compiled on the first call
- h, w = self.resize_h, self.resize_w
-
- self.gpu_warmup_input_frame_np = cv2.cuda.createContinuous(h, w, cv2.CV_8UC3)
-
- if self.gpu_warmup_input_frame_np is not None:
- self.gpu_warmup_input_frame_np[:] = [255, 0, 0]
-
- gpu_warmup_input_frame = cv2.cuda.GpuMat(h, w, cv2.CV_8U)
- gpu_warmup_input_frame.upload(self.gpu_warmup_input_frame_np)
- cv2.cuda.createContinuous(h, w, cv2.CV_8U, gpu_warmup_input_frame)
-
- # Trigger compiler
- gpu_warmup_frame = cv2.cuda.GpuMat(h, w, cv2.CV_8U)
- cv2.cuda.createContinuous(h, w, cv2.CV_8U, gpu_warmup_frame)
-
- cv2.cuda.resize(
- gpu_warmup_input_frame,
- (self.resize_w, self.resize_h),
- stream=self.stream,
- dst=gpu_warmup_frame,
- interpolation=cv2.INTER_NEAREST,
- )
- # Thresholding
- gpu_threshold_dst_frame = cv2.cuda.GpuMat(h, w, cv2.CV_8U)
- cv2.cuda.createContinuous(h, w, cv2.CV_8U, gpu_threshold_dst_frame)
- cv2.cuda.threshold(
- gpu_warmup_frame,
- self.config.THRESHOLD_VALUE,
- self.config.THRESHOLD_MAX_VALUE,
- cv2.THRESH_BINARY,
- gpu_threshold_dst_frame,
- self.stream,
- )
-
- gpu_morphed_frame = cv2.cuda.GpuMat(h, w, cv2.CV_8U)
- cv2.cuda.createContinuous(h, w, cv2.CV_8U, gpu_morphed_frame)
- self.dilate_filter.apply(
- gpu_threshold_dst_frame, gpu_morphed_frame, self.stream
- )
- self.stream.waitForCompletion()
-
- torch.cuda.empty_cache()
-
def cleanup_gpu(self):
"""
Explicitly releases all GPU-allocated memory to prevent
@@ -4058,7 +4429,7 @@ def cleanup_gpu(self):
# 🏎️ Force the NVIDIA driver to deallocate this specific memory segment
attr_value.release()
setattr(self, attr_name, None)
- print(f"✅ Released GpuMat: {attr_name}")
+ main_app_logger.info(f"✅ Released GpuMat: {attr_name}")
if hasattr(self, "gpu_fullres_frame") and self.gpu_fullres_frame is not None:
try:
@@ -4096,8 +4467,8 @@ def cleanup_gpu(self):
mat.release()
self.gpu_crop_batch = []
- if hasattr(self, "stream"):
- self.stream.waitForCompletion()
+ # if hasattr(self, "stream"):
+ # self.stream.waitForCompletion()
self.pinned_downloaded_resizedframe_np = None
self.gpu_threshold_dst_frame = None
@@ -4114,494 +4485,21 @@ def cleanup_gpu(self):
# Optional: Final flush of the CUDA caching allocator
if torch.cuda.is_available():
+ torch.cuda.synchronize()
torch.cuda.empty_cache()
- def apply_background_subtraction_gpu(
- self, include_history=True, method="and", stream=None
- ):
- stream = stream if isinstance(stream, cv2.cuda.Stream) else self.stream
- self.fgMask = self.backSub.apply(
- self.resized_frame, float(self.lr), stream=stream
- )
-
- if include_history:
- # If this is the first run, clone the mask instead of ANDing with an empty/white buffer
- # if len(self.mask_history) < 1:
- # self.prev_bkgd.setTo(255, stream) # Clear the initial white buffer
-
- for m in list(self.mask_history):
- # Dilate the historical mask on GPU
- dilated = self.dilate_filter_for_enhanced_mask.apply(m, stream=stream)
-
- if method == "or":
- # Bitwise OR on GPU
- cv2.cuda.bitwise_or(
- self.prev_bkgd, dilated, self.prev_bkgd, stream=stream
- )
- else:
- # Bitwise AND on GPU
- cv2.cuda.bitwise_and(
- self.prev_bkgd, dilated, self.prev_bkgd, stream=stream
- )
-
- self.mask_history.append(self.fgMask.clone())
-
- min_val, max_val, _, _ = cv2.cuda.minMaxLoc(self.prev_bkgd)
- if max_val != min_val and max_val > 0:
- # bitor = cv2.cuda.bitwise_or(
- # self.fgMask, self.prev_bkgd, stream=stream
- # )
- # bitand = cv2.cuda.bitwise_and(
- # self.fgMask, self.prev_bkgd, stream=stream
- # )
- # not_bitand = cv2.cuda.bitwise_not(self.prev_bkgd, stream=stream)
- # self.fgMask = cv2.cuda.subtract(self.fgMask, bitand, stream=stream)
- self.fgMask = cv2.cuda.bitwise_or(
- self.fgMask, self.prev_bkgd, stream=stream
- )
- # self.fgMask = cv2.cuda.bitwise_or(
- # self.fgMask, self.mask_history[-2], stream=stream
- # )
- # if method == "or":
- # self.fgMask = cv2.cuda.bitwise_and(
- # self.fgMask, self.prev_bkgd, stream=stream
- # )
- # else:
- # self.fgMask = cv2.cuda.bitwise_or(
- # self.fgMask, self.prev_bkgd, stream=stream
- # )
-
- def rbtd_full_gpuv1(self, frame):
- """
- GPU-Accelerated Motion Detection Pipeline (Producer).
-
- This function performs high-speed background subtraction (BGS) on a downscaled
- version of the 8K frame to identify regions of interest (ROIs).
-
- Args:
- frame (np.ndarray): The raw 8K input frame.
- frameNum (float): Chronological timestamp or ID.
-
- Returns:
- dict: Contains the frame ID, the GPU-resident motion mask, and the original 8K frame.
- """
- # stream = self.stream
-
- with torch.no_grad(): # , torch.cuda.stream(self.bgs_stream):
- # frame is [3, 4320, 7680]
- resized_torch = F.interpolate(
- frame.unsqueeze(0).float(),
- size=(self.resize_h, self.resize_w),
- mode="nearest",
- # mode="bilinear",
- # align_corners=False,
- ).squeeze(0)
- # resized_torch = F.interpolate(
- # frame.to(memory_format=torch.channels_last),
- # size=(self.resize_h, self.resize_w),
- # mode="nearest",
- # # align_corners=False
- # )
-
- # CRITICAL: Change format from [3, 640, 640] to [640, 640, 3]
- # .contiguous() is mandatory here to reorganize the actual memory bits
- resized_torch = resized_torch.permute(1, 2, 0).byte().contiguous()
- gpu_mat_view = cv2.cuda.createGpuMatFromCudaMemory(
- self.resize_h, self.resize_w, cv2.CV_8UC3, resized_torch.data_ptr()
- )
-
- # Bridge the TINY 640p frame to OpenCV
- # Moving 8K (100MB) to CPU takes 114ms.
- # Moving 640p (1.2MB) to CPU takes <0.2ms.
- # This preserves your 15 FPS target.
- # small_cpu = resized_torch.permute(1, 2, 0).cpu().numpy()
- # self.resized_frame.upload(small_cpu)
- # gpu_mat_bridge = torch2gpumat(resized_torch.byte())
- gpu_mat_view.copyTo(self.bgs_stream, self.resized_frame)
- self.apply_background_subtraction_gpu(
- include_history=self.config.BKGD_SUB_INCLUDE_HISTORY,
- method=self.config.BKGD_SUB_INCLUDE_HISTORY_METHOD,
- stream=self.bgs_stream,
- )
-
- # self.bgs_stream.waitForCompletion()
- # Clean up the motion mask using Thresholding and Morphology (Dilation)
- # cv2.cuda.threshold(
- # self.fgMask,
- # self.config.THRESHOLD_VALUE,
- # 255,
- # cv2.THRESH_BINARY,
- # dst=self.gpu_threshold_dst_frame,
- # stream=self.bgs_stream,
- # )
- # self.dilate_filter.apply(
- # self.gpu_threshold_dst_frame, dst=self.labels_gpu, stream=self.bgs_stream
- # )
-
- # return {
- # # "frameNum": frameNum, # overall frame
- # "mask": self.labels_gpu, # GpuMat pointer to cleaned mask
- # # "full_frame": frame, # Kept for high-res cropping
- # "full_frame": frame, # current_gpu_frame,
- # }
-
- # if self.config.ENABLE_QUERYING and self.video_writer: # and not self.video_queue.full():
- # self.pinned_downloaded_resizedframe_np = self.resized_frame.download(stream)
- # # self.resized_frame.download(self.stream, self.pinned_downloaded_resizedframe_np)
- # # self.video_writer.write(self.pinned_downloaded_resizedframe_np)
- # self.write_queue.put(self.pinned_downloaded_resizedframe_np.copy())
-
- # if (
- # (self.debug_range[0] * self.target_fps)
- # <= self.frame_count
- # <= (self.debug_range[1] * self.target_fps)
- # ):
- # mask_cpu = self.fgMask.download() # fgMask is a GpuMat from the backend
- # mask_path = f"{self.out_imgdir}/mask_frame_{self.frame_count}.png"
- # cv2.imwrite(mask_path, mask_cpu)
- # print(f"[DEBUG] Saved BGS Mask: {mask_path}")
-
- # --------------
- # NOISE FILTERING: Median Blur (Kills pixel noise)
- # if not hasattr(self, 'median_filter'):
- # self.median_filter = cv2.cuda.createMedianFilter(cv2.CV_8UC1, 5)
-
- # # Syntax: apply(src, dst, stream)
- # self.median_filter.apply(self.fgMask, self.fgMask, self.bgs_stream)
-
- # # MORPHOLOGY: Erode then Dilate (Opening)
- # if not hasattr(self, 'erode_filter'):
- # # 3x3 kernel is sufficient when combined with a median filter
- # kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (3, 3))
-
- # # In OpenCV 4.x, use createMorphologyFilter for both Erode and Dilate
- # self.erode_filter = cv2.cuda.createMorphologyFilter(cv2.MORPH_ERODE, cv2.CV_8UC1, kernel)
- # self.dilate_filter = cv2.cuda.createMorphologyFilter(cv2.MORPH_DILATE, cv2.CV_8UC1, kernel)
-
- # # Shave off noise (Erode)
- # self.erode_filter.apply(self.fgMask, self.fgMask, self.bgs_stream)
-
- # # Restore object size (Dilate)
- # self.dilate_filter.apply(self.fgMask, self.fgMask, self.bgs_stream)
-
- # return {"mask": self.fgMask, "full_frame": frame}
- # --------------
-
- # Clean up the motion mask using Thresholding and Morphology (Dilation)
- # Fused Cleanup (Threshold + Dilate)
- # This replaces: cv2.cuda.threshold AND cv2.cuda.dilate
- w, h = self.fgMask.size()
- pitch = self.fgMask.step
- tpb = (16, 16)
- bpg = ((w + 15) // 16, (h + 15) // 16)
-
- # Bridge GpuMat to CuPy for the kernel
- fg_ptr = self.fgMask.cudaPtr()
- with cupy.cuda.ExternalStream(self.bgs_stream.cudaPtr()):
- fg_cp = cupy.ndarray(
- (h, w),
- dtype=cupy.uint8,
- memptr=cupy.cuda.MemoryPointer(
- cupy.cuda.UnownedMemory(fg_ptr, pitch * h, self), 0
- ),
- strides=(pitch, 1),
- )
-
- # Launch Fused Kernel
- labels_ptr = self.labels_gpu.cudaPtr()
- # Ensure this matches the size of fg_cp
- labels_cp = cupy.ndarray(
- (h, w),
- dtype=cupy.uint8,
- memptr=cupy.cuda.MemoryPointer(
- cupy.cuda.UnownedMemory(labels_ptr, self.labels_gpu.step * h, self),
- 0,
- ),
- strides=(self.labels_gpu.step, 1),
- )
- # with cupy.cuda.ExternalStream(self.stream.cudaPtr()):
- # Launch your kernel on the same physical GPU stream as OpenCV BGS
- DETECTION_ACCEL_KERNEL(
- bpg,
- tpb,
- (fg_cp, labels_cp, pitch, w, h, self.config.THRESHOLD_VALUE),
- )
- # Ensure CuPy is done before the stream returns to OpenCV/PyTorch
- # cupy.cuda.get_current_stream().synchronize()
-
- # if self.debug_range[0] <= self.frame_count <= self.debug_range[1]:
- # after_cpu = self.labels_gpu.download()
- # # Multiply by 50 to make distinct object labels visible
- # visible_labels = (after_cpu * 50).clip(0, 255).astype(np.uint8)
- # cv2.imwrite(
- # f"{self.out_imgdir}/mask_AFTER_frame_{self.frame_count}.png",
- # visible_labels,
- # )
-
- return {
- # "frameNum": frameNum, # overall frame
- "mask": self.labels_gpu, # GpuMat pointer to cleaned mask
- # "full_frame": frame, # Kept for high-res cropping
- "full_frame": frame, # current_gpu_frame,
- }
-
- def rbtd_full_gpu(self, device_frame):
- """
- Asynchronously handles downsampling, conversion, background subtraction, and morphology
- by reshaping PyTorch CUDA Tensors to interleaved layouts before mapping to GpuMat.
- """
- # 1. BRIDGE THE PYTORCH TO OPENCV VRAM GAP (ZERO-COPY & INTERLEAVED)
- if torch.is_tensor(device_frame):
- # Reshape tensor to channel-last layout [H, W, C] to completely prevent the 3x3 grid tiling artifact
- interleaved_frame = device_frame.permute(1, 2, 0).contiguous()
-
- h_raw, w_raw, ch = interleaved_frame.shape
- cuda_mem_ptr = interleaved_frame.data_ptr()
- cv_type = cv2.CV_8UC3 if ch == 3 else cv2.CV_8UC1
-
- src_gpu_mat = cv2.cuda.createGpuMatFromCudaMemory(
- h_raw, w_raw, cv_type, cuda_mem_ptr
- )
- else:
- src_gpu_mat = device_frame
- ch = src_gpu_mat.channels()
-
- # 2. INSTANTIATE GPUMAT CONTROLLERS WITH STRIDED LAYOUT HEADERS
- recycled_resize_mat = cv2.cuda.GpuMat(
- self.resize_h, self.resize_w, cv2.CV_8UC3 if ch == 3 else cv2.CV_8UC1
- )
- gray_resize_mat = cv2.cuda.GpuMat(self.resize_h, self.resize_w, cv2.CV_8UC1)
- raw_mask = cv2.cuda.GpuMat(self.resize_h, self.resize_w, cv2.CV_8UC1)
- thresh_mask = cv2.cuda.GpuMat(self.resize_h, self.resize_w, cv2.CV_8UC1)
- clean_mask = cv2.cuda.GpuMat(self.resize_h, self.resize_w, cv2.CV_8UC1)
-
- # 3. RUN ASYNCHRONOUS DOWN-SAMPLING GATE
- cv2.cuda.resize(
- src_gpu_mat,
- dst=recycled_resize_mat,
- dsize=(self.resize_w, self.resize_h),
- interpolation=cv2.INTER_LINEAR,
- stream=self.bgs_stream,
- )
-
- # 4. COLOR LAYOUT CORRECTION (COLLAPSE CHANNELS SAFELY)
- if recycled_resize_mat.channels() == 3:
- cv2.cuda.cvtColor(
- recycled_resize_mat,
- cv2.COLOR_BGR2GRAY,
- dst=gray_resize_mat,
- stream=self.bgs_stream,
- )
- motion_input = gray_resize_mat
- else:
- motion_input = recycled_resize_mat
-
- # 4. EXECUTE SEGMENTATION METRICS
- raw_mask = self.backSub.apply(
- motion_input,
- 0.005, # float(self.lr),
- stream=self.bgs_stream,
- )
- include_history = self.config.BKGD_SUB_INCLUDE_HISTORY
- method = self.config.BKGD_SUB_INCLUDE_HISTORY_METHOD
- if include_history:
- # If this is the first run, clone the mask instead of ANDing with an empty/white buffer
- # if len(self.mask_history) < 1:
- # self.prev_bkgd.setTo(255, stream) # Clear the initial white buffer
-
- for m in list(self.mask_history):
- # Dilate the historical mask on GPU
- dilated = self.dilate_filter_for_enhanced_mask.apply(
- m, stream=self.bgs_stream
- )
-
- if method == "or":
- # Bitwise OR on GPU
- cv2.cuda.bitwise_or(
- self.prev_bkgd, dilated, self.prev_bkgd, stream=self.bgs_stream
- )
- else:
- # Bitwise AND on GPU
- cv2.cuda.bitwise_and(
- self.prev_bkgd, dilated, self.prev_bkgd, stream=self.bgs_stream
- )
-
- self.mask_history.append(raw_mask.clone())
-
- min_val, max_val, _, _ = cv2.cuda.minMaxLoc(self.prev_bkgd)
- if max_val != min_val and max_val > 0:
- raw_mask = cv2.cuda.bitwise_or(
- raw_mask, self.prev_bkgd, stream=self.bgs_stream
- )
-
- # 5. MORPHOLOGICAL TRANSFORMATIONS BINARY FILTERS
- cv2.cuda.threshold(
- raw_mask,
- self.config.THRESHOLD_VALUE,
- self.config.THRESHOLD_MAX_VALUE,
- cv2.THRESH_BINARY,
- thresh_mask,
- stream=self.bgs_stream,
- )
-
- # cv2.cuda.dilate(
- # thresh_mask,
- # clean_mask,
- # self.dilate_kernel,
- # stream=self.bgs_stream
- # )
- self.dilate_filter.apply(thresh_mask, clean_mask, self.bgs_stream)
-
- # 6. ENFORCE INDEPENDENT WORKSPACE MEMORY VIEWS
- # Allocate an isolated output mask surface container
- isolated_kernel_mask = cv2.cuda.GpuMat(
- self.resize_h, self.resize_w, cv2.CV_8UC1
- )
- clean_mask.copyTo(dst=isolated_kernel_mask, stream=self.bgs_stream)
-
- # return isolated_kernel_mask
- # self.bgs_stream.waitForCompletion()
-
- return {
- "mask": isolated_kernel_mask, # GpuMat pointer to cleaned mask
- "full_frame": device_frame, # current_gpu_frame,
- }
-
- def findContours_gpu(self, mask, method="fused"):
- # self.stream.waitForCompletion()
- h, w = mask.size()
- ptr = mask.cudaPtr()
- pitch = mask.step
- mask_cp = cupy.ndarray(
- (h, w),
- dtype=cupy.uint8,
- memptr=cupy.cuda.MemoryPointer(
- cupy.cuda.UnownedMemory(ptr, pitch * h, self), 0
- ),
- strides=(pitch, 1),
- )
-
- labeled, num_labels = cupyx.scipy.ndimage.label(mask_cp, output=cupy.int32)
- # labeled = labeled.astype(cupy.int32)
- # labeled.strides is in bytes. For int32, we need row-start in bytes.
- labeled_pitch_bytes = labeled.strides[0]
-
- if num_labels == 0:
- return torch.empty((0, 4), device="cuda")
-
- # Pre-allocate bounds with sentinels
- x1, y1 = (
- cupy.full((num_labels + 1,), w, dtype=cupy.int32),
- cupy.full((num_labels + 1,), h, dtype=cupy.int32),
- )
- x2, y2 = (
- cupy.full((num_labels + 1,), -1, dtype=cupy.int32),
- cupy.full((num_labels + 1,), -1, dtype=cupy.int32),
- )
-
- # BOUNDS_KERNEL(((w+15)//16, (h+15)//16), (16, 16), (labeled, w, h, num_labels, x1, y1, x2, y2))
- BOUNDS_KERNEL(
- ((w + 15) // 16, (h + 15) // 16),
- (16, 16),
- (
- labeled.data.ptr,
- labeled_pitch_bytes,
- w,
- h,
- num_labels,
- x1.data.ptr,
- y1.data.ptr,
- x2.data.ptr,
- y2.data.ptr,
- ),
- )
- # cupy.cuda.get_current_stream().synchronize()
- # cupy.cuda.Stream.null.synchronize()
- # Convert to torch
- boxes = torch.stack(
- [
- torch.as_tensor(x1[1:], device="cuda"),
- torch.as_tensor(y1[1:], device="cuda"),
- torch.as_tensor(x2[1:], device="cuda"),
- torch.as_tensor(y2[1:], device="cuda"),
- ],
- dim=1,
- ).float()
-
- # --- CRITICAL FIX: Filter out boxes that weren't updated by the kernel ---
- # A valid box must have x2 >= x1
- valid_mask = boxes[:, 2] >= boxes[:, 0]
- boxes = boxes[valid_mask]
-
- # if boxes.shape[0] > 0:
- # print(f"[GPU] Found {boxes.shape[0]} boxes", flush=True)
- return boxes
-
class CPUStreamHandler(DeviceBaseHandler):
- def allocate_cpu(self):
- # pass
- self.resized_frame = np.zeros((3, self.resize_h, self.resize_w), dtype="uint8")
- # cv2.cuda.createContinuous(
- # self.resize_h, self.resize_w, cv2.CV_8UC3
- # )
-
- self.fgMask = np.zeros(
- (self.resize_h, self.resize_w), dtype="uint8"
- ) # For resize
-
- if self.config.BKGD_SUB_INCLUDE_HISTORY_METHOD == "and":
- self.prev_bkgd = np.ones(
- (self.resize_h, self.resize_w), dtype="uint8"
- ) # * 255
- else:
- self.prev_bkgd = np.zeros((self.resize_h, self.resize_w), dtype="uint8")
-
- # self.prev_bkgd = np.ones((self.resize_h, self.resize_w), dtype="uint8") * 255
-
- self.mask_history = deque(
- maxlen=self.config.BKGD_SUB_INCLUDE_HISTORY_TEMPORAL_SIZE
- )
- self.mask_history.append(self.prev_bkgd)
-
- def prepare_cpu_pipeline(self): # , method="mog2"):
- self.operation_device_map = PipelineMapping() # "full_cpu"
- self.device_input = self.operation_device_map.detection_device
-
- self.allocate_cpu()
-
- # Subtraction
- # if method == "knn":
- # history = 300 # int(5 * self.target_fps)
- # background_thresh = 350
- # NSamples = 10
- # kNNSamples = 2
- # self.lr = 1 / history
-
- # self.backSub = cv2.createBackgroundSubtractorKNN(
- # history=history, # default 500
- # dist2Threshold=background_thresh, # default 400
- # detectShadows=False, # default True
- # )
- # self.backSub.setkNNSamples(kNNSamples)
- # self.backSub.setNSamples(NSamples)
- # elif method == "mog2":
- # history = int(2 * self.target_fps)
- # background_thresh = 10
- self.lr = self.config.BKGD_SUB_MOG2_LR
-
- self.backSub = cv2.createBackgroundSubtractorMOG2(
- history=self.config.BKGD_SUB_MOG2_HISTORY, # Clear ghosts of fast drones in ~2 seconds (2*fps)
- varThreshold=self.config.BKGD_SUB_MOG2_VARTHRESHOLD, # High threshold to ignore "shimmer" and compression noise # default 16
- detectShadows=self.config.BKGD_SUB_MOG2_DETECTSHADOWS, # default True
- )
- # else:
- # raise ValueError(f"Provided method ({method}) is not available.")
-
def cleanup_cpu(self):
"""
Purges large 8K NumPy buffers and CPU-based AI resources.
"""
+ self._pinned_small_frame = None
+ self._pinned_fg_mask = None
+ self._pinned_blurred_mask = None
+ self._pinned_threshold_mask = None
+ self._pinned_dilated_mask = None
+
# Nullify specific class references to allow Garbage Collection
self.executor = None
self.clip_executor = None
@@ -4620,95 +4518,3 @@ def cleanup_cpu(self):
# Clear the BGS history
if hasattr(self, "mask_history"):
self.mask_history.clear()
-
- def apply_background_subtraction_cpu(self, include_history=True, method="and"):
- self.fgMask = self.backSub.apply(
- self.cpu_resized_frame, learningRate=float(self.lr)
- )
-
- if include_history:
- # If this is the first run, clone the mask instead of ANDing with an empty/white buffer
- # if len(self.mask_history) < 1:
- # self.prev_bkgd.setTo(0, stream) # Clear the initial white buffer
-
- for m in list(self.mask_history):
- # Dilate the historical mask on CPU
- dilated = cv2.dilate(
- m, self.dilate_kernel_for_enhanced_mask, iterations=1
- )
-
- if method == "or":
- # Bitwise OR on CPU
- cv2.bitwise_or(self.prev_bkgd, dilated, dst=self.prev_bkgd)
- else:
- # Bitwise AND on CPU
- cv2.bitwise_and(self.prev_bkgd, dilated, dst=self.prev_bkgd)
-
- self.mask_history.append(self.fgMask.copy())
-
- if (
- self.prev_bkgd.max() != self.prev_bkgd.min()
- and self.prev_bkgd.max() > 0
- ):
- combined_mask_bool = (self.fgMask > 0) | (self.prev_bkgd > 0)
- self.fgMask = combined_mask_bool.astype(np.uint8) * 255
-
- def rbtd_full_cpu(self, frame):
- """
- CPU-Based Motion Detection Pipeline (Producer).
-
- Performs background subtraction on the CPU to identify moving objects.
- Ideal for saving VRAM or for environments without high-end NVIDIA GPUs.
-
- Args:
- frame (np.ndarray): The raw 8K input frame.
- frameNum (float): Unique ID for the current frame.
-
- Returns:
- dict: Motion data containing the frame ID, CPU-based mask, and original 8K frame.
- """
- # Resize the 8K frame to a smaller 'model' size (e.g., 640x640)
- # Using INTER_NEAREST as it is the fastest CPU interpolation method.
- # H, W = self.resize_h, self.resize_w
- self.cpu_resized_frame = cv2.resize(
- frame, (self.resize_w, self.resize_h), interpolation=cv2.INTER_NEAREST
- )
- # if self.config.ENABLE_QUERYING and self.video_writer:
- # self.write_queue.put(self.cpu_resized_frame.copy())
-
- # Apply Background Subtraction on CPU
- self.apply_background_subtraction_cpu(
- include_history=self.config.BKGD_SUB_INCLUDE_HISTORY,
- method=self.config.BKGD_SUB_INCLUDE_HISTORY_METHOD,
- )
-
- # ----------------
- # NOISE FILTERING: Median Blur
- # Kills small single-pixel noise specs
- # self.fgMask = cv2.medianBlur(self.fgMask, 5)
-
- # # MORPHOLOGY: Opening (Erode then Dilate)
- # kernel = np.ones((3,3), np.uint8)
-
- # # Remove small specs
- # self.fgMask = cv2.erode(self.fgMask, kernel, iterations=2)
-
- # # Re-expand and connect nearby moving pixels
- # self.fgMask = cv2.dilate(self.fgMask, kernel, iterations=2)
- # return {"mask": self.fgMask, "full_frame": frame}
- # ----------------
-
- # Clean up the motion mask using Thresholding and Morphology (Dilation)
- _, mask = cv2.threshold(
- self.fgMask,
- self.config.THRESHOLD_VALUE,
- self.config.THRESHOLD_MAX_VALUE,
- cv2.THRESH_BINARY,
- )
- mask = cv2.dilate(mask, self.dilate_kernel, iterations=1)
-
- return {
- # "frameNum": frameNum, # overall frame
- "mask": mask,
- "full_frame": frame, # Kept for high-res cropping
- }
diff --git a/fastapi/include/models.py b/fastapi/include/models.py
index e9dc452..3be6e20 100644
--- a/fastapi/include/models.py
+++ b/fastapi/include/models.py
@@ -6,13 +6,13 @@
from pathlib import Path
import tensorrt as trt
+from torch import cuda
+from ultralytics import YOLO
+from ultralytics.utils.checks import check_requirements
sys.path.insert(1, str(Path(__file__).parent.parent))
from include.default_configs import ENABLE_QUERYING_DEFAULT
from include.utils import PipelineConfig, get_freest_gpu, str2bool
-from torch import cuda
-from ultralytics import YOLO
-from ultralytics.utils.checks import check_requirements
# OBJECT DETECTION
@@ -83,9 +83,9 @@ def get_model(
):
final_model_path = f"{model_dir}/{model_name}.pt"
pt_detection_model = YOLO(final_model_path, verbose=False, task="detect")
- label_source = []
- for k, v in pt_detection_model.names.items():
- label_source.append(v)
+
+ # Secure the master label source dictionary reference from the raw baseline weights
+ master_names_dict = pt_detection_model.names
if run_platform == "openvino":
final_model_path = f"{model_dir}/{model_name}_openvino_model/"
@@ -96,7 +96,7 @@ def get_model(
dynamic=dynamic_flag,
device=device_input,
# batch=batch,
- data={"names": pt_detection_model.names},
+ data={"names": master_names_dict},
)
object_detection_model = YOLO(
@@ -105,13 +105,50 @@ def get_model(
task="detect",
)
- # det_ov_model = core.read_model(final_model_path+"yolo11n.xml")
- # ov_config = {hints.performance_mode: hints.PerformanceMode.LATENCY}
- # if device == "GPU":
- # ov_config["GPU_DISABLE_WINOGRAD_CONVOLUTION"] = "YES"
- # compiled_model = core.compile_model(det_ov_model, device, ov_config)
+ import numpy as np
+ import openvino as ov
+ import openvino.preprocess as wp
+ from openvino import properties
+
+ core = ov.Core()
+ det_ov_model = core.read_model(final_model_path + f"{model_name}.xml")
+
+ # --- ULTRA-FAST HIGH-SPEED C++ GRAPH INJECTION ---
+ # We tell the OpenVINO compiler graph to apply the layout shift
+ # and normalization layers natively in C++ before returning data to Python
+ ppp = wp.PrePostProcessor(det_ov_model)
+
+ # Inject an implicit transposition rule directly onto the output layer node
+ # This shifts shapes from [1, 5, 8400] -> [1, 8400, 5] automatically in memory
+ output = ppp.output(0)
+ output.tensor().set_layout(ov.Layout("NCW"))
+ output.model().set_layout(ov.Layout("NWC"))
+
+ # Build the modifications back into the primary model reference structure
+ det_ov_model = ppp.build()
+
+ ov_config = {
+ # properties.hint.performance_mode(): properties.hint.PerformanceMode.THROUGHPUT,
+ properties.hint.performance_mode(): properties.hint.PerformanceMode.LATENCY,
+ properties.streams.num(): 1, # Set strictly to 1 to prevent thread thrashing #2,
+ properties.inference_num_threads(): os.cpu_count() or 4,
+ }
+ compiled_model = core.compile_model(
+ det_ov_model, device_input.upper(), ov_config
+ )
+ dummy_canvas = np.zeros((model_h, model_w, 3), dtype=np.uint8)
+ _ = object_detection_model.predict(dummy_canvas, verbose=False, save=False)
# object_detection_model.predictor.model.ov_compiled_model = compiled_model
+ if (
+ hasattr(object_detection_model, "predictor")
+ and object_detection_model.predictor is not None
+ ):
+ # Re-map the raw execution backend instance to utilize the multi-stream parameters
+ object_detection_model.predictor.model.backend.ov_compiled_model = (
+ compiled_model
+ )
+
elif run_platform == "engine":
final_model_path = f"{model_dir}/{model_name}.engine"
onnx_model_path = f"{model_dir}/{model_name}.onnx"
@@ -159,7 +196,7 @@ def get_model(
dynamic=True,
device=device_input,
simplify=True,
- data={"names": pt_detection_model.names},
+ data={"names": master_names_dict},
)
if hasattr(pt_detection_model.model, "stride"):
@@ -174,7 +211,7 @@ def get_model(
metadata={
"stride": max_stride,
"task": "detect",
- "names": pt_detection_model.names,
+ "names": master_names_dict,
},
)
@@ -200,7 +237,7 @@ def get_model(
device=device_input,
simplify=True,
batch=batch,
- data={"names": pt_detection_model.names},
+ data={"names": master_names_dict},
)
object_detection_model = YOLO(final_model_path, verbose=False, task="detect")
@@ -215,6 +252,13 @@ def get_model(
else:
raise ValueError(f"[!] Model for {run_platform} is not implemented.")
+ # --- SAFELY COMPILING FINAL DEFINITIVE LABELS LIST ---
+ # Interrogate the runtime model instance dictionary first; fallback directly to master reference maps
+ current_names = (
+ getattr(object_detection_model, "names", master_names_dict) or master_names_dict
+ )
+ label_source = [current_names[i] for i in sorted(current_names.keys())]
+
return object_detection_model, final_model_path, label_source
diff --git a/fastapi/include/readers.py b/fastapi/include/readers.py
index 8494b08..c08b7e1 100644
--- a/fastapi/include/readers.py
+++ b/fastapi/include/readers.py
@@ -1,31 +1,46 @@
+# ==============================================================================
+# IMPORTS
import ctypes
+import gc
+import inspect
import logging
+import multiprocessing as mp
import os
-import queue
+import signal
import sys
import threading
import time
-import traceback
+import types
+from collections import deque
+from multiprocessing import Condition, Process, Value
+from multiprocessing.shared_memory import SharedMemory
+from pathlib import Path
import av
import cv2
import numpy as np
import torch
-import torch.nn.functional as F
from ultralytics.utils.checks import check_imgsz
-# from fastapi import FastAPI
+sys.path.insert(1, str(Path(__file__).parent.parent))
+from include.default_configs import ENABLE_QUERYING_DEFAULT
+from include.utils import (
+ PipelineConfig,
+ ResourceTrackerFilter,
+ manual_fps_calculation,
+ str2bool,
+)
+
+# ==============================================================================
+# LOGGING
-# ----- SETUP LOGGING -----
logging.basicConfig(
level=logging.INFO,
- format="%(asctime)s - %(name)s - %(levelname)s - %(message)s",
+ # format="%(asctime)s - %(name)s - %(levelname)s - %(message)s",
+ format="%(asctime)s [%(levelname)s] %(name)s (%(filename)s:%(lineno)d) - %(message)s",
handlers=[logging.StreamHandler(sys.stdout)],
)
-# Suppress low-delay reference block warnings from OpenCV/PyAV/FFmpeg
-os.environ["OPENCV_FFMPEG_LOGLEVEL"] = "-8"
-os.environ["OPENCV_LOG_LEVEL"] = "OFF"
logging.getLogger("libav").setLevel(logging.CRITICAL)
logging.getLogger("libav.hevc").setLevel(logging.CRITICAL)
av.logging.set_level(av.logging.PANIC)
@@ -33,21 +48,11 @@
main_app_logger = logging.getLogger(__name__)
-# ----- PIPELINE CONFIGURATION -----
-os.environ["PYTORCH_ALLOC_CONF"] = "expandable_segments:True"
-# Force OpenCV to use a single thread for its operations.
-# This prevents internal OpenCV threads from "racing" against AI logic.
-# cv2.setNumThreads(1)
+# ==============================================================================
+# PIPELINE CONFIGURATION
-# Force OpenCV to run sequentially to prevent context-switching overhead
cv2.setNumThreads(0)
-# DEVICE = os.getenv("DEVICE", "CPU")
-# device_input = DEVICE.lower() if DEVICE == "CPU" else "cuda"
-
-from include.default_configs import ENABLE_QUERYING_DEFAULT
-from include.utils import PipelineConfig, manual_fps_calculation, str2bool
-
BASE_PIPELINE_CONFIG = PipelineConfig(
CODE_DIR=os.getenv("CODE_DIR", "/home"),
CUSTOM_MODEL_FLAG=str2bool(os.getenv("CUSTOM_MODEL_FLAG", False)),
@@ -58,7 +63,6 @@
INGESTION=os.getenv("INGESTION", "object"),
MODEL_NAME=os.getenv("MODEL_NAME", "yolo11n"),
OMIT_DETECTIONS_FLAG=str2bool(os.getenv("OMIT_DETECTIONS_FLAG", False)),
- # RESIZE_FLAG=str2bool(os.getenv("RESIZE_FLAG", False)),
SHARED_MODEL=os.getenv("SHARED_MODEL", False),
SHARED_OUTPUT=os.getenv("SHARED_OUTPUT", "/var/www/mp4"),
TEST_MODE=str2bool(os.getenv("TEST_MODE", False)),
@@ -67,138 +71,248 @@
UDF_PORT=5011,
)
-# Placeholder for dynamic import for PyNvVideoCodec (GPU package)
-nvc = None
-
-# ----- VIDEO READERS -----
-class BaseReader:
- def __init__(
- self,
- source,
- target_fps=BASE_PIPELINE_CONFIG.TARGET_FPS,
- clip_duration=BASE_PIPELINE_CONFIG.CLIP_DURATION,
- queue_size=2,
- ):
- self.source = source
- self.is_rtsp = str(self.source).startswith("rtsp://")
- self.frame_idx = 0
- self.frame_queue = queue.Queue(maxsize=queue_size) # 2) # 5
- self.stopped = False
- self.reconnect_failed = False
- self.init_error = None
- self.target_fps = (
- float(target_fps) if target_fps not in [None, 0] else target_fps
+def capture_shared_memory_worker(
+ startup_event,
+ source_input,
+ shm_names,
+ frame_shape,
+ running_flag,
+ raw_frame_counter,
+ latest_idx,
+ reader_idx,
+ frame_condition,
+ buffer_occupancy,
+ target_fps,
+ slot_available_event,
+ num_shm_slots,
+):
+ """Isolated background process using triple-buffering pointer rotation to eliminate memcpy."""
+ if str(source_input).lower().startswith("rtsp://"):
+ os.environ["OPENCV_FFMPEG_CAPTURE_OPTIONS"] = (
+ "rtsp_transport;tcp;buffer_size;33554432;threads;32;max_delay;500000"
)
- self.clip_duration = (
- float(clip_duration) if clip_duration not in [None, 0] else clip_duration
+ # "rtsp_transport;tcp;buffer_size;52428800;fifo_size;500000;max_delay;100000;stimeout;2000000"
+ # "rtsp_transport;tcp;buffer_size;15728640;threads;16"
+ elif "OPENCV_FFMPEG_CAPTURE_OPTIONS" in os.environ:
+ del os.environ["OPENCV_FFMPEG_CAPTURE_OPTIONS"]
+
+ if startup_event:
+ startup_event.wait() # Worker will pause here until the main app is ready
+
+ retry_cnt = 0
+ max_retries = 5
+ cap = None
+ is_rtsp_stream = str(source_input).lower().startswith("rtsp://")
+
+ while retry_cnt < max_retries:
+ cap = cv2.VideoCapture(
+ source_input,
+ cv2.CAP_FFMPEG,
+ [cv2.CAP_PROP_HW_ACCELERATION, cv2.VIDEO_ACCELERATION_ANY],
)
+ if cap.isOpened():
+ break
- def start(self):
- threading.Thread(target=self.stream_frames, daemon=True).start()
- return self
+ retry_cnt += 1
+ if not is_rtsp_stream or retry_cnt >= max_retries:
+ main_app_logger.error(
+ f"Critical: Could not open/connect to video resource: {source_input}"
+ )
+ running_flag.value = False
+ return
- def stop(self):
- """Cleanly stop the reader and release resources."""
- self.stopped = True
- self.release()
+ wait_time = retry_cnt * 2
+ main_app_logger.warning(
+ f"RTSP process connection pending... Retry ({retry_cnt}/{max_retries}) in {wait_time} seconds."
+ )
+ time.sleep(wait_time)
+
+ # existing_shm = SharedMemory(name=shm_name)
+ # shared_array = np.ndarray(frame_shape, dtype=np.uint8, buffer=existing_shm.buf)
+ # Map all 3 shared memory regions simultaneously
+ shm_blocks = [SharedMemory(name=name) for name in shm_names]
+ # from multiprocessing import resource_tracker
+ # for shm in shm_blocks:
+ # resource_tracker.unregister(shm._name, "shared_memory")
+ arrays = [
+ np.ndarray(frame_shape, dtype=np.uint8, buffer=shm.buf) for shm in shm_blocks
+ ]
+
+ write_idx = 0
+ worker_frame_num = 0.0
+ worker_next_process_idx = 0.0
+ native_fps = cap.get(cv2.CAP_PROP_FPS)
+ step_size = float(native_fps) / float(target_fps)
+
+ try:
+ while running_flag.value:
+ # 1. Check if this frame is a targeted frame before running heavy decode operations
+ is_target_frame = worker_frame_num >= worker_next_process_idx
+ raw_frame_counter.value = int(worker_frame_num)
+ worker_frame_num += 1.0
+
+ if not is_target_frame:
+ # Fast metadata-only skip: Zero heavy decode compute or CPU/GPU overhead
+ if not cap.grab():
+ running_flag.value = False
+ with frame_condition:
+ frame_condition.notify_all()
+ break
+ # raw_frame_counter.value += 1
+ continue
+
+ ret, frame = cap.read()
+ if not ret or frame is None:
+ running_flag.value = False
+ with frame_condition:
+ frame_condition.notify_all()
+ break
+
+ worker_next_process_idx += step_size
+
+ # Prevent local files from overwriting unread ring buffer slots ---
+ # if not str(source_input).lower().startswith("rtsp://"):
+ # # while buffer_occupancy.value >= 2 and running_flag.value:
+ # # pass
+ # with frame_condition:
+ # while buffer_occupancy.value >= 2 and running_flag.value:
+ # # Drop the thread into a low-power kernel sleep state
+ # frame_condition.wait(timeout=0.01)
+
+ # Prevent local files from overwriting unread ring buffer slots losslessly
+ # if not str(source_input).lower().startswith("rtsp://"):
+ # if buffer_occupancy.value >= 2:
+ # slot_available_event.clear() # Lock the gate
+ # while buffer_occupancy.value >= 2 and running_flag.value:
+ # # Block instantly via kernel context without time-stepping drifts
+ # slot_available_event.wait(timeout=1.0)
+
+ if not str(source_input).lower().startswith("rtsp://"):
+ max_allowed_occupancy = num_shm_slots - 1
+ while (
+ buffer_occupancy.value >= max_allowed_occupancy
+ and running_flag.value
+ ):
+ time.sleep(0.001)
+
+ # Dynamic Ring Buffer Selection: Identify free block
+ curr_latest = latest_idx.value
+ curr_reader = reader_idx.value
+
+ write_idx = None
+ for idx in range(num_shm_slots):
+ if idx != curr_latest and idx != curr_reader:
+ write_idx = idx
+ break
- def read(self):
- try:
- # If the reader is stopped, don't wait a full second;
- # check immediately to speed up the "Draining" phase.
- wait_time = 0.1 if self.stopped else 2.0
- return self.frame_queue.get(timeout=wait_time)
- except Exception:
- return None, None
+ # Fallback if somehow all slots are locked (should theoretically never happen)
+ if write_idx is None:
+ write_idx = (curr_latest + 1) % num_shm_slots
+ arrays[write_idx][:] = frame[:]
-class CPUHybridReader(BaseReader):
- """
- Decouples frame acquisition from processing.
- Uses a background thread to ingest frames into a small deque,
- preventing OpenCV buffer lag.
- """
+ # Notify waiting processing loops without any polling latency
+ with frame_condition:
+ latest_idx.value = write_idx
+ # raw_frame_counter.value += 1
+ buffer_occupancy.value += 1
+ frame_condition.notify_all() # Wake up all waiting reader threads instantly!
+
+ except Exception:
+ running_flag.value = False
+ finally:
+ cap.release()
+
+ # Delete the memoryview-holding arrays first
+ del arrays
+ for shm in shm_blocks:
+ try:
+ shm.close()
+ except Exception:
+ pass
+ del shm_blocks
+ gc.collect()
+
+
+class BaseReader:
def __init__(
self,
source,
+ startup_event=None,
target_fps=BASE_PIPELINE_CONFIG.TARGET_FPS,
clip_duration=BASE_PIPELINE_CONFIG.CLIP_DURATION,
MODEL_W=BASE_PIPELINE_CONFIG.MODEL_W,
MODEL_H=BASE_PIPELINE_CONFIG.MODEL_H,
queue_size=2,
- # as_tensor=False,
):
- super().__init__(
- source,
- target_fps=target_fps,
- clip_duration=clip_duration,
- queue_size=queue_size,
- )
- # self.as_tensor = as_tensor
- target_fps, clip_duration = (self.target_fps, self.clip_duration)
- # options = (
- # {"rtsp_transport": "tcp", "stimeout": "5000000"}
- # if str(self.source).startswith("rtsp")
- # else {}
- # )
+ self.source = source
+ self.is_rtsp = str(self.source).startswith("rtsp://")
+ # self.frame_queue = queue.Queue(maxsize=queue_size)
+ self.frame_queue = deque(maxlen=queue_size)
+ self.stopped = False
+ self.total_input_frames = 0 # Increments by step size (Physical frames)
+ self.target_frames_passed = 0 # Increments sequentially (Target space frames)
+ # self.frame_idx = 0.0
+ self.total_shm_copy_time = 0.0
+ self.total_h2d_time = 0.0
+ self.total_gpu_resize_time = 0.0
+ self.total_d2h_time = 0.0
+ self.total_queue_wait_time = 0.0
self.MODEL_H = MODEL_H
self.MODEL_W = MODEL_W
- self.device = "CPU" # Global from include.utils
- self.target_frame_idx = 0
- self.cap = None
+ self.target_fps = (
+ float(target_fps) if target_fps not in [None, 0] else target_fps
+ )
+ self.clip_duration = (
+ float(clip_duration) if clip_duration not in [None, 0] else clip_duration
+ )
+
+ probe_cap = None
max_retries = 5
retry_cnt = 0
connected = False
+ # while probe_retry < max_retries:
+ # probe_cap = cv2.VideoCapture(self.source, cv2.CAP_FFMPEG)
+ # if probe_cap.isOpened():
+ # break
+ # probe_retry += 1
+ # if not self.is_rtsp:
+ # break
+ # wait_time = probe_retry * 2
+ # main_app_logger.warning(
+ # f"Connection pending... Retry ({probe_retry}/{max_retries}) in {wait_time} seconds."
+ # )
+ # time.sleep(wait_time)
+
while not connected and not self.stopped:
try:
- # Clean up stale capture descriptors safely to clear sockets
- if self.cap is not None:
- try:
- self.cap.release()
- except Exception:
- pass
- self.cap = None
-
- # Safely intercept capture context extraction layers
- # Note: We pass self.source instead of recreating string evaluations
- if not isinstance(self.source, cv2.VideoCapture):
- params = [cv2.CAP_PROP_N_THREADS, 1]
- test_cap = cv2.VideoCapture(
- str(self.source), cv2.CAP_FFMPEG, params=params
- )
- else:
- test_cap = self.source
-
- if test_cap and test_cap.isOpened():
- # Validate that we can extract safe telemetry properties
+ probe_cap = cv2.VideoCapture(self.source, cv2.CAP_FFMPEG)
+ if probe_cap.isOpened():
self.get_fps_and_framecnt(
- test_cap, self.target_fps, self.clip_duration
+ probe_cap, self.target_fps, self.clip_duration
)
- self.frame_width = int(test_cap.get(cv2.CAP_PROP_FRAME_WIDTH))
- self.frame_height = int(test_cap.get(cv2.CAP_PROP_FRAME_HEIGHT))
- self.numFrames = int(test_cap.get(cv2.CAP_PROP_FRAME_COUNT))
+ self.frame_width = int(probe_cap.get(cv2.CAP_PROP_FRAME_WIDTH))
+ self.frame_height = int(probe_cap.get(cv2.CAP_PROP_FRAME_HEIGHT))
+ self.numFrames = int(probe_cap.get(cv2.CAP_PROP_FRAME_COUNT))
if self.input_fps <= 0 or self.frame_width <= 0:
raise RuntimeError(
"VideoCapture opened but returned invalid stream properties."
)
-
- self.cap = test_cap
self.get_frameWH()
connected = True
+ probe_cap.release()
else:
raise RuntimeError(
"OpenCV VideoCapture failed to open target URI resource context."
)
-
except Exception as e:
retry_cnt += 1
self.init_error = str(e)
-
# Exit if local file resource or retry count exceeded
if not self.is_rtsp or retry_cnt >= max_retries:
main_app_logger.error(
@@ -208,56 +322,348 @@ def __init__(
self.stopped = True
# Halt and exit the server immediately
raise RuntimeError(
- f"Critical CPU stream reader initialization failure: {self.init_error}"
+ f"Critical stream reader initialization failure: {self.init_error}"
)
- return
wait_time = retry_cnt * 2
main_app_logger.warning(
- f"CPU connection pending... Retry ({retry_cnt}/{max_retries}) in {wait_time} seconds."
+ f"Connection pending... Retry ({retry_cnt}/{max_retries}) in {wait_time} seconds."
)
time.sleep(wait_time)
- # self.cap = self._create_capture(target_fps, clip_duration)
- self.cap.set(cv2.CAP_PROP_BUFFERSIZE, 1) # Force low latency
- self.cap.set(cv2.CAP_PROP_HW_ACCELERATION, cv2.VIDEO_ACCELERATION_ANY)
-
- # self.frame_queue = deque(maxlen=5) # Keep queue small to stay "real-time"
- # self.frame_queue = queue.Queue(maxsize=5)
- # self.stopped = False
- # self.device = "CPU" # Global from include.utils
- # self.frame_idx = 0
- # self.target_frame_idx = 0
- # self.frame_queue = queue.Queue(maxsize=30)
+ # if probe_cap is None or not probe_cap.isOpened():
+ # self.reconnect_failed = True
+ # self.stopped = True
+ # raise RuntimeError(f"OpenCV failed to open target resource: {self.source}")
- def _create_capture(self, target_fps, clip_duration):
- """Creates a VideoCapture with stable RTSP options."""
- # if not isinstance(self.source, cv2.VideoCapture):
- # self.source = str(self.source)
- # params = [cv2.CAP_PROP_N_THREADS, 1]
- # cap = cv2.VideoCapture(self.source, cv2.CAP_FFMPEG, params=params)
- # else:
- # cap = self.source
- # self.get_fps_and_framecnt(cap, target_fps, clip_duration)
- # self.frame_width = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH))
- # self.frame_height = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT))
- # self.numFrames = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))
+ # self.get_fps_and_framecnt(probe_cap, self.target_fps, self.clip_duration)
+ # self.frame_width = int(probe_cap.get(cv2.CAP_PROP_FRAME_WIDTH))
+ # self.frame_height = int(probe_cap.get(cv2.CAP_PROP_FRAME_HEIGHT))
+ # self.numFrames = int(probe_cap.get(cv2.CAP_PROP_FRAME_COUNT))
# self.get_frameWH()
- # return cap
+ # probe_cap.release()
+
+ self.frame_shape = (self.frame_height, self.frame_width, 3)
+ self.frame_bytes = self.frame_width * self.frame_height * 3
+
+ # Need exactly 3 slots to prevent a lock collision
+ num_shm_slots = max(3, queue_size)
+ self.shms = [
+ SharedMemory(create=True, size=self.frame_bytes)
+ for _ in range(num_shm_slots)
+ ]
+
+ self.running_flag = Value("b", True)
+ self.raw_frame_counter = Value("i", 0)
+ self.latest_idx = Value("i", 0) # Tracks the newest complete frame index
+ self.reader_idx = Value("i", -1) # Locks the frame currently being processed
+ shm_names = [shm.name for shm in self.shms]
+ # atomic counter to track unread slot density ---
+ self.buffer_occupancy = Value("i", 0)
+
+ # Create a shared cross-process condition lock variable
+ self.frame_condition = Condition()
+ self.slot_available_event = mp.Event()
+ self.slot_available_event.set() # Default to unblocked state
+
+ self.worker = Process(
+ target=capture_shared_memory_worker,
+ args=(
+ startup_event,
+ self.source,
+ shm_names,
+ self.frame_shape,
+ self.running_flag,
+ self.raw_frame_counter,
+ self.latest_idx,
+ self.reader_idx,
+ self.frame_condition,
+ self.buffer_occupancy,
+ self.target_fps,
+ self.slot_available_event,
+ num_shm_slots,
+ ),
+ daemon=True,
+ )
+ self.worker.start()
+ time.sleep(2.5 if self.is_rtsp else 0.05)
+
+ if not self.running_flag.value:
+ for shm in self.shms:
+ shm.close()
+ shm.unlink()
+ raise RuntimeError(
+ "Background worker process failed to map the stream link."
+ )
+
+ def read(self):
+ """
+ Optimized Kernel-Native Condition Event queue fetcher.
+ Parks the consumer thread instantly without CPU thrashing or GIL locks,
+ waking up on hardware-driven signals from the producer process.
+ """
+ # Escape check: Stream explicitly halted and queue empty
+ if self.stopped and len(self.frame_queue) == 0:
+ return False, None, None, None, None
+
+ # # 2. Native Hardware Wait Boundary
+ # # If the queue is starved, drop the thread into an un-polled kernel block
+ # if len(self.frame_queue) == 0 and not self.stopped:
+ # with self.frame_condition:
+ # # Re-verify inside the lock to safely block multi-thread race patterns
+ # while len(self.frame_queue) == 0 and not self.stopped:
+ # # Increase watchdog timeout from 2.0s to 10.0s for local file testing
+ # timeout_sec = 2.0 if self.is_rtsp else 10.0
+
+ # # Thread blocks via OS futex without time-stepping drift or context jitter.
+ # # Wakes up immediately when the processing loop fires notify_all().
+ # if not self.frame_condition.wait(timeout=timeout_sec):
+ # # Watchdog trigger: 2 seconds of zero frame activity indicates a dead stream
+ # return (
+ # False,
+ # None,
+ # None,
+ # self.target_frames_passed,
+ # self.total_input_frames,
+ # )
+
+ # Lock-Free Pop Extraction Pass
+ if len(self.frame_queue) > 0:
+ try:
+ return self.frame_queue.popleft()
+ except IndexError:
+ pass # Multi-thread race fallback guard rail
+
+ # Brief non-blocking yield if queue is momentarily empty
+ if not self.stopped:
+ time.sleep(0.001)
+ if len(self.frame_queue) > 0:
+ try:
+ return self.frame_queue.popleft()
+ except IndexError:
+ pass
+
+ return False, None, None, self.target_frames_passed, self.total_input_frames
+
+ def start(self):
+ self.thread = threading.Thread(target=self._processing_loop, daemon=True)
+ self.thread.start()
+ return self
+
+ def stop(self):
+ """Cleanly unregisters the locked memory mapping blocks from the OS page table."""
+ if self.stopped:
+ return
- if isinstance(self.source, cv2.VideoCapture):
- return self.source
- params = [cv2.CAP_PROP_N_THREADS, 1]
- return cv2.VideoCapture(str(self.source), cv2.CAP_FFMPEG, params=params)
+ self.stopped = True
+
+ # Signal the worker process to exit its loop
+ if hasattr(self, "running_flag"):
+ self.running_flag.value = False
+
+ if hasattr(self, "worker") and self.worker is not None:
+ try:
+ if self.worker.is_alive():
+ self.worker.terminate()
+ self.worker.join(timeout=0.5)
+ # Execute an OS-level SIGKILL fallback if process flags hang
+ if self.worker.is_alive() and getattr(self.worker, "pid", None):
+ os.kill(self.worker.pid, signal.SIGKILL)
+ self.worker.join()
+ except Exception:
+ pass
+ self.worker = None
+
+ # Join active processing pools and thread handles safely
+ if hasattr(self, "frame_condition") and self.frame_condition is not None:
+ try:
+ with self.frame_condition:
+ self.frame_condition.notify_all()
+ except Exception:
+ pass
+ self.frame_condition = None
+
+ if hasattr(self, "thread") and self.thread is not None:
+ try:
+ if self.thread.is_alive():
+ self.thread.join(timeout=0.5)
+ except Exception:
+ pass
+ self.thread = None
+
+ # 6. Set and unbind OS-level Event notification primitive file descriptor pipelines
+ if (
+ hasattr(self, "slot_available_event")
+ and self.slot_available_event is not None
+ ):
+ try:
+ self.slot_available_event.set()
+ if (
+ hasattr(self.slot_available_event, "_handle")
+ and self.slot_available_event._handle
+ ):
+ self.slot_available_event._handle.close()
+ except Exception:
+ pass
+ self.slot_available_event = None
+
+ # 7. Drain and empty the frame buffer queue completely
+ if hasattr(self, "frame_queue") and self.frame_queue is not None:
+ try:
+ while len(self.frame_queue) > 0:
+ self.frame_queue.popleft()
+ except Exception:
+ pass
+ self.frame_queue = None
+
+ if hasattr(self, "shms") and self.shms:
+ for shm_block in list(self.shms):
+ try:
+ if hasattr(shm_block, "buf") and shm_block.buf is not None:
+ shm_block.buf.release()
+ except Exception:
+ pass
+ try:
+ shm_block.close()
+ except Exception:
+ pass
+ try:
+ shm_block.unlink()
+ except Exception:
+ pass
+ self.shms.clear()
+
+ if hasattr(self, "bridge_thread"):
+ self.bridge_thread.join(timeout=1.0)
+
+ for primitive_attr in [
+ "running_flag",
+ "buffer_occupancy",
+ "latest_idx",
+ "reader_idx",
+ "raw_frame_counter",
+ ]:
+ if hasattr(self, primitive_attr):
+ setattr(self, primitive_attr, None)
+
+ self.clean_up_tensors_and_arrays()
+
+ def clean_up_tensors_and_arrays(self):
+ main_app_logger.info(
+ "[READER CLEANUP] Safely stripping active runtime arrays ..."
+ )
+ # Collect references safely using strict type checking
+ # This completely avoids pulling unmanaged proxy objects from the heap
+ all_live_objects = gc.get_objects()
+
+ target_tensors = []
+ target_arrays = []
+
+ for obj in all_live_objects:
+ try:
+ obj_type = type(obj)
+ if isinstance(obj, types.FrameType):
+ frame_info = inspect.getframeinfo(obj)
+ if (
+ "openvino" in frame_info.filename
+ or "openvino.py" in frame_info.filename
+ ):
+ # Clear the frame's local variable dictionary to break circular references
+ obj.f_locals.clear()
+
+ # Check for concrete types to prevent triggering proxy __getattr__ hooks
+ if obj_type is torch.Tensor:
+ target_tensors.append(obj)
+ elif obj_type is np.ndarray:
+ # Explicitly guard size checks to keep it completely stable
+ if obj.base is None and obj.ndim > 0:
+ target_arrays.append(obj)
+ except Exception:
+ pass
+
+ # 3. Truncate discovered references in-place without triggering deletions
+ reclaimed_tensors = 0
+ for tensor in target_tensors:
+ try:
+ # Truncate raw storage footprint safely
+ tensor.data = torch.empty(0, device=self.device_input)
+ reclaimed_tensors += 1
+ except Exception:
+ pass
+
+ reclaimed_arrays = 0
+ for arr in target_arrays:
+ try:
+ # Shrink writeable numpy arrays down to 0 bytes safely
+ if arr.flags.writeable:
+ arr.resize((0,), refcheck=False)
+ reclaimed_arrays += 1
+ except (ValueError, SystemError):
+ pass
+
+ main_app_logger.info(
+ f"[READER CLEANUP] Reclaimed {reclaimed_tensors} tensors and {reclaimed_arrays} arrays safely."
+ )
+
+ # Clean local registers immediately
+ all_live_objects = None
+ target_tensors = None
+ target_arrays = None
+ gc.collect()
+
+ def unlink_shared_memory(self, all_shm_keys):
+ for key in all_shm_keys:
+ val = getattr(self, key, None)
+ if val is None:
+ continue
+ for shm in list(val):
+ if shm is not None:
+ try:
+ # Drop direct memoryview trackers instantly
+ if hasattr(shm, "buf") and shm.buf is not None:
+ try:
+ shm.buf.release()
+ except Exception:
+ pass
+
+ # Invalidate private mmap structures to drop pytest stack frame caches
+ if hasattr(shm, "_mmap") and shm._mmap is not None:
+ try:
+ shm._mmap = None
+ except Exception:
+ pass
+
+ # try:
+ # shm.__class__.__del__ = lambda self: None
+ # except Exception:
+ # pass
+
+ # # Unlink text descriptors out of the core tracker process
+ # shm_descriptor = shm._name if shm._name.startswith("/") else f"/{shm._name}"
+ # try:
+ # unregister(shm_descriptor, "shared_memory")
+ # except Exception:
+ # pass
+
+ shm.close()
+ shm.unlink()
+ except Exception:
+ pass
+ try:
+ val.clear()
+ except Exception:
+ pass
# Gets video details
def get_fps_and_framecnt(self, cap, target_fps, clip_duration):
- self.input_fps = int(cap.get(cv2.CAP_PROP_FPS)) # hardware fps
+ self.input_fps = cap.get(cv2.CAP_PROP_FPS) # hardware fps
# print(f"in fps: {self.input_fps} target fps: {target_fps}")
if self.input_fps == 0: # Case when FPS isn't available
self.input_fps = manual_fps_calculation(cap, num_frames=10)
- print(f"new in fps: {self.input_fps}")
-
+ main_app_logger.debug(f"new in fps: {self.input_fps}")
+ self.numFrames = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))
+ if not self.is_rtsp:
+ self.total_input_frames = self.numFrames
# If the stream can't connect, stop immediately instead of calculating.
if self.input_fps <= 0:
raise RuntimeError(
@@ -270,8 +676,9 @@ def get_fps_and_framecnt(self, cap, target_fps, clip_duration):
else self.input_fps
)
+ self.step_size = self.input_fps / self.target_fps
if self.input_fps > 0 and self.target_fps > 0:
- self.frame_skip = max(1, int(self.input_fps / self.target_fps))
+ self.frame_skip = max(1, int(self.step_size))
else:
self.frame_skip = 1
# self.skip_count = self.frame_skip - 1
@@ -294,783 +701,1023 @@ def get_frameWH(self):
[self.frame_height, self.frame_width]
) # expects hxw
- new_sizeWH = (new_sizeHW[1], new_sizeHW[0])
-
- self.width = new_sizeWH[0]
- self.height = new_sizeWH[1]
+ self.width = new_sizeHW[1]
+ self.height = new_sizeHW[0]
# Configure scaling for 8K-to-Model coordinate mapping
self.resize_h, self.resize_w = [self.MODEL_H, self.MODEL_W]
self.scale_x = self.frame_width / self.MODEL_W
self.scale_y = self.frame_height / self.MODEL_H
- def stream_frames(self):
- """
- Continuously grabs frames. Throttles local files to maintain
- the target FPS and manages RTSP reconnections.
- """
- is_rtsp = str(self.source).startswith("rtsp")
- max_retries = 5
- retry_cnt = 0
-
- while not self.stopped:
- try:
- if self.cap is None or not self.cap.isOpened():
- if not is_rtsp:
- self.stopped = True
- break
- try:
- # Logic to recreate the VideoCapture
- # self.cap = self._create_capture(
- # self.target_fps, self.clip_duration
- # )
- # self.cap.set(cv2.CAP_PROP_BUFFERSIZE, 1)
- params = [cv2.CAP_PROP_N_THREADS, 1]
- self.cap = cv2.VideoCapture(
- str(self.source), cv2.CAP_FFMPEG, params=params
- )
- self.cap.set(cv2.CAP_PROP_BUFFERSIZE, 1)
- retry_cnt = 0
- except Exception:
- retry_cnt += 1
- if retry_cnt >= max_retries:
- self.stopped = True
- break
-
- wait_time = retry_cnt * 2
- main_app_logger.warning(
- f"CPU reconnect failed. Retry {retry_cnt} in {wait_time}s"
- )
- time.sleep(wait_time)
- continue
-
- # Fully decode this frame
- ret, frame = self.cap.read()
- if not ret:
- if not is_rtsp:
- self.stopped = True
- break
-
- # If a live stream returns No Frame, don't just die—trigger a reconnect
- main_app_logger.warning("No frame received. Attempting reconnect.")
- if self.cap:
- self.cap.release()
- self.cap = None
- continue
- # if self.as_tensor:
- # frame_tensor = torch.from_numpy(frame)
+class CPUReader(BaseReader):
+ """Asynchronous CPU frame reader and processor utilizing AVX2 optimized OpenCV routines."""
- # # 3. Push to GPU memory immediately
- # # non_blocking=True speeds up the host-to-device transfer
- # frame_tensor = frame_tensor.to("cuda", non_blocking=True)
+ def __init__(
+ self,
+ source,
+ startup_event=None,
+ target_fps=BASE_PIPELINE_CONFIG.TARGET_FPS,
+ clip_duration=BASE_PIPELINE_CONFIG.CLIP_DURATION,
+ MODEL_W=BASE_PIPELINE_CONFIG.MODEL_W,
+ MODEL_H=BASE_PIPELINE_CONFIG.MODEL_H,
+ queue_size=2,
+ ):
+ # self.source_input = source_input
+ # self.frame_queue = queue.Queue(maxsize=30)
+ # self.stopped = False
+ # self.total_shm_copy_time = 0.0
+ # self.total_h2d_time = 0.0
+ # self.total_gpu_resize_time = 0.0
+ # self.total_d2h_time = 0.0
+ super().__init__(
+ source,
+ startup_event=startup_event,
+ target_fps=target_fps,
+ clip_duration=clip_duration,
+ MODEL_W=MODEL_W,
+ MODEL_H=MODEL_H,
+ queue_size=queue_size,
+ )
- # # 4. Rearrange dimensions to PyTorch format: [H, W, C] -> [C, H, W]
- # # .permute() changes layout; .float() converts uint8 to float32 for interpolation
- # frame_tensor = frame_tensor.permute(2, 0, 1).float()
+ # self.MODEL_H = MODEL_H
+ # self.MODEL_W = MODEL_W
+ self.device_index = "cpu"
- # # 5. Add Batch dimension: [C, H, W] -> [1, C, H, W]
- # frame_tensor = frame_tensor.unsqueeze(0)
+ # self.shm = SharedMemory(create=True, size=self.frame_bytes)
+ # self.running_flag = Value('b', True)
- # # 6. Normalize pixel values to [0.0, 1.0] if required by your model
- # frame = frame_tensor / 255.0
+ # self.worker = Process(
+ # target=capture_shared_memory_worker,
+ # args=(self.source_input, self.shm.name, self.frame_shape, self.running_flag),
+ # daemon=True
+ # )
+ # # self.worker.start()
+ # time.sleep(2.5) # Warm up wait window
+
+ # if not self.running_flag.value:
+ # self.shm.close()
+ # self.shm.unlink()
+ # raise RuntimeError("Background worker process failed to map the stream link.")
+
+ # self.shared_array_view = np.ndarray(self.frame_shape, dtype=np.uint8, buffer=self.shm.buf)
+ self.static_buffer_numpy = np.empty(self.frame_shape, dtype=np.uint8)
+
+ # self.thread = threading.Thread(target=self._processing_loop, daemon=True)
+ # self.thread.start()
+
+ @torch.inference_mode()
+ def _processing_loop(self):
+ # target_fps = 15.0
+ # frame_num = 0.0
+ # next_process_idx = 0.0
+ # step_size = float(self.input_fps) / float(self.target_fps)
+ # # self.total_queue_wait_time = 0.0
+
+ # # Base loop pacing on the native camera interval (e.g., 33.3ms for 30 FPS)
+ # inbound_frame_interval = 1.0 / float(self.input_fps)
+ last_processed_shm_idx = -1
+ while not self.stopped: # and self.running_flag.value:
+ # loop_start = time.perf_counter()
+ # is_target_frame = (frame_num >= next_process_idx)
+
+ # frame_num += 1.0
+
+ # if not is_target_frame:
+ # continue
+
+ with self.frame_condition:
+ # Thread sleeps instantly until the background worker calls notify_all()
+ # self.frame_condition.wait(timeout=1.0)
+ # Only wait if the background worker hasn't delivered a new index yet
+ while (
+ self.latest_idx.value == last_processed_shm_idx and not self.stopped
+ ):
+ if not self.frame_condition.wait(timeout=0.5):
+ if (
+ not self.running_flag.value
+ # and self.latest_idx.value == last_processed_shm_idx
+ ):
+ self.stopped = True
+ break
+ # continue
+ # self.frame_condition.wait()
+ if self.stopped:
+ break
- # self.frame_queue.put((frame, self.frame_idx))
- try:
- self.frame_queue.put((frame, self.frame_idx), timeout=1.0)
- except queue.Full:
+ active_idx = self.latest_idx.value
+ self.reader_idx.value = active_idx
+ last_processed_shm_idx = active_idx
+
+ # Signal the worker that a slot has cleared up losslessly ---
+ self.buffer_occupancy.value = max(0, self.buffer_occupancy.value - 1)
+ # self.slot_available_event.set()
+
+ # is_target_frame = (frame_num >= next_process_idx)
+ # Reconstruct source frame position using step spacing ratio
+ # if self.numFrames != 0:
+ # self.total_input_frames = int(
+ # self.target_frames_passed * getattr(self, "frame_skip", 1)
+ # )
+ # if self.is_rtsp:
+
+ target_frames_passed = self.target_frames_passed
+ # total_input_frames = int(
+ # target_frames_passed * getattr(self, "frame_skip", 1)
+ # ) # int(self.raw_frame_counter.value)
+ total_input_frames = round(target_frames_passed * self.step_size)
+
+ # Update counters inside target_fps space loop execution
+ self.total_input_frames = total_input_frames
+
+ # if not is_target_frame:
+ # continue
+
+ # # Check if frame is a target frame before heavy operations
+ # # if frame_num >= next_process_idx:
+ # next_process_idx += step_size
+
+ t_copy = time.perf_counter()
+ # np.copyto(self.static_buffer_numpy, self.shared_array_view)
+ # Zero-Copy Pointer Swap: lock the latest completed buffer index
+ # active_idx = self.latest_idx.value
+ # self.reader_idx.value = active_idx
+ # last_processed_shm_idx = active_idx
+
+ # # Signal the worker that a slot has cleared up losslessly ---
+ # self.buffer_occupancy.value = max(0, self.buffer_occupancy.value - 1)
+
+ # Wrap an array view directly around the locked active SHM region
+ active_shm = self.shms[active_idx]
+ frame_view = np.ndarray(
+ self.frame_shape, dtype=np.uint8, buffer=active_shm.buf
+ )
+ self.total_shm_copy_time += time.perf_counter() - t_copy
+
+ t_queue_block = time.perf_counter()
+ # if not self.is_rtsp or not self.frame_queue.full():
+ # t_queue_block = time.perf_counter()
+ # # self.frame_queue.put((True, cpu_360p_frame))
+ # self.frame_queue.put(
+ # (True, frame_view, target_frames_passed, total_input_frames),
+ # # timeout=1.0,
+ # ) # self.static_buffer_numpy.copy()))
+ # self.total_queue_wait_time += time.perf_counter() - t_queue_block
+
+ if self.is_rtsp:
+ # If the consumer loop drops frames, instantly evict the oldest
+ # matrix view reference to keep frame processing real-time.
+ if len(self.frame_queue) >= self.frame_queue.maxlen:
try:
- # Non-blocking pop to gracefully evict the oldest stale frame token
- self.frame_queue.get_nowait()
- except queue.Empty:
+ self.frame_queue.popleft()
+ except IndexError:
pass
- self.frame_queue.put((frame, self.frame_idx))
- self.target_frame_idx += 1
- self.frame_idx += 1
- except Exception as e:
- main_app_logger.error(f"CPU Reader error: {e}")
- time.sleep(1)
+ self.frame_queue.append(
+ (True, frame_view, None, target_frames_passed, total_input_frames)
+ )
+ else:
+ # Lossless tracking sequence for local testing files
+ # while len(self.frame_queue) >= self.frame_queue.maxlen and not self.stopped:
+ # time.sleep(0.001)
+ safe_cpu_copy = frame_view.copy()
+ self.frame_queue.append(
+ (
+ True,
+ safe_cpu_copy,
+ None,
+ target_frames_passed,
+ total_input_frames,
+ )
+ )
- def release(self):
- print("Closing HybridReader...")
- self.stopped = True
+ self.total_queue_wait_time += time.perf_counter() - t_queue_block
+ self.target_frames_passed += 1
- time.sleep(0.2)
- if self.cap is not None and self.cap.isOpened():
- self.cap.release()
+ # frame_num += 1.0
+ # Keep the reader thread aligned with target ingestion speeds
+ # elapsed = time.perf_counter() - loop_start
+ # time_to_wait = inbound_frame_interval - elapsed
+ # if time_to_wait > 0:
+ # time.sleep(time_to_wait)
-class GPUHybridReader(BaseReader):
- """
- Encapsulates a Zero-Copy 8K video pipeline.
- Bridges PyAV (Demuxing) and NVDEC (Hardware Decoding) directly to PyTorch.
- """
+ # self.frame_queue.put((False, None, target_frames_passed, total_input_frames))
+ # self.frame_idx = frame_num
+ # self.running_flag.value = False
+
+
+class GPUReader(BaseReader):
+ """Asynchronous GPU frame reader leveraging Pinned Host Memory and PyTorch CUDA tensors."""
def __init__(
self,
source,
- gpu_id=0,
+ startup_event=None,
+ gpu_id=1, # 0,
target_fps=BASE_PIPELINE_CONFIG.TARGET_FPS,
clip_duration=BASE_PIPELINE_CONFIG.CLIP_DURATION,
MODEL_W=BASE_PIPELINE_CONFIG.MODEL_W,
MODEL_H=BASE_PIPELINE_CONFIG.MODEL_H,
queue_size=2,
):
- global nvc
+ # self.source_input = source_input
+ # self.frame_queue = queue.Queue(maxsize=30)
+ # self.stopped = False
+ # self.total_shm_copy_time = 0.0
+ # self.total_h2d_time = 0.0
+ # self.total_gpu_resize_time = 0.0
+ # self.total_d2h_time = 0.0
super().__init__(
source,
+ startup_event=startup_event,
target_fps=target_fps,
clip_duration=clip_duration,
+ MODEL_W=MODEL_W,
+ MODEL_H=MODEL_H,
queue_size=queue_size,
)
- if "PyNvVideoCodec" not in sys.modules:
- try:
- import PyNvVideoCodec as nvc
+ self.gpu_id = gpu_id
+ self.device_index = torch.device(f"cuda:{gpu_id}")
- globals()["nvc"] = nvc
+ # self.shm = SharedMemory(create=True, size=self.frame_bytes)
+ # self.running_flag = Value('b', True)
- except ImportError:
- raise ImportError(
- "GPUHybridReader requires PyNvVideoCodec. Please install."
- )
+ # self.worker = Process(
+ # target=capture_shared_memory_worker,
+ # args=(self.source_input, self.shm.name, self.frame_shape, self.running_flag),
+ # daemon=True
+ # )
+ # self.worker.start()
+ # time.sleep(2.5) # Warm up wait window
- self.gpu_id = gpu_id
- self.container = None
- self.nv_dec = None
- self.bsf = None
-
- # --- Initialize CUDA Context & Bridge ---
- torch.cuda.set_device(self.gpu_id)
- torch.cuda.init()
- _ = torch.zeros(1).cuda() # Force context creation
-
- # self.is_opened = self.open(target_fps, clip_duration)
- self.cuda_lib = ctypes.CDLL("libcuda.so.1")
- ctx = ctypes.c_void_p()
- self.cuda_lib.cuCtxGetCurrent(ctypes.byref(ctx))
- self.cuda_ctx_handle = ctx.value if ctx.value is not None else 0
-
- self.cuda_lib.cuCtxSetCurrent(ctypes.c_void_p(self.cuda_ctx_handle))
- target_fps, clip_duration = (self.target_fps, self.clip_duration)
- self.av_options = (
- {
- # "rtsp_transport": "tcp",
- # "stimeout": "2000000", # 2s
- # "probesize": "10000000", # "32000000", # 32MB for 8K
- # "analyzeduration": "5000000",
- # "buffer_size": "10240000", # 10MB socket buffer
- "rtsp_transport": "tcp",
- "stimeout": "2000000",
- "timeout": "2000000",
- "rw_timeout": "2000000",
- "err_detect": "explode",
- "flags": "discardcorrupt",
- "probesize": "32000000",
- "analyzeduration": "32000000",
- }
- if str(self.source).startswith("rtsp")
- else {}
+ # if not self.running_flag.value:
+ # self.shm.close()
+ # self.shm.unlink()
+ # raise RuntimeError("Background worker process failed to map the stream link.")
+
+ # self.shared_array_view = np.ndarray(self.frame_shape, dtype=np.uint8, buffer=self.shm.buf)
+
+ # Reusable hardware space optimizations
+
+ self._static_gpu_frame_buffer = torch.empty(
+ (self.frame_height, self.frame_width, 3),
+ dtype=torch.uint8,
+ device=self.device_index,
+ )
+ self._flipped_static_gpu_frame_buffer = torch.empty_like(
+ self._static_gpu_frame_buffer
+ )
+ # Pre-allocate a second buffer for the BGR conversion.
+ # This is our destination tensor.
+ self._bgr_gpu_frame_buffer = torch.empty(
+ (self.frame_height, self.frame_width, 3),
+ dtype=torch.uint8,
+ device=self.device_index,
)
+ # self.static_buffer_tensor = torch.empty(self.frame_shape, dtype=torch.uint8, pin_memory=True)
+ # self.static_buffer_numpy = self.static_buffer_tensor.numpy()
+ # Reusable double-buffered hardware space optimizations
+ # self.static_buffer_tensors = [
+ # torch.empty(self.frame_shape, dtype=torch.uint8, pin_memory=True),
+ # torch.empty(self.frame_shape, dtype=torch.uint8, pin_memory=True)
+ # ]
+ # self.static_buffer_numpys = [t.numpy() for t in self.static_buffer_tensors]
+ # self.buffer_selector = 0 # Alternates between 0 and 1
+ self.upload_stream = torch.cuda.Stream(device=self.device_index)
+
+ self.download_stream = torch.cuda.Stream(device=self.device_index)
+ self.d2h_buffers = [
+ torch.empty((360, 640, 3), dtype=torch.uint8, pin_memory=True),
+ torch.empty((360, 640, 3), dtype=torch.uint8, pin_memory=True),
+ ]
+ self.d2h_numpys = [b.numpy() for b in self.d2h_buffers]
+ self.d2h_selector = 0
+
+ self.pinned_views = []
+ for shm in self.shms:
+ shm_numpy = np.ndarray(self.frame_shape, dtype=np.uint8, buffer=shm.buf)
+ shm_tensor = torch.from_numpy(shm_numpy)
+ self.pinned_views.append(shm_tensor)
+ # # Reusable hardware space optimizations using OS-level Page-Locking
# try:
- # self.container = av.open(
- # self.source,
- # options={
- # **self.av_options,
- # "err_detect": "ignore_err", # Don't stop demuxing on minor packet errors
- # "flags": "low_delay", # Reduce internal buffering
- # },
- # )
- # self.container.streams.video[0].thread_type = 'AUTO'
- # streams = self.container.streams.get(video=0)
- # self.stream = streams[0] if isinstance(streams, list) else streams
+ # self.cudart = ctypes.CDLL("libcudart.so") # Linux
+ # except OSError:
+ # try:
+ # self.cudart = ctypes.CDLL("libcudart.so.12")
+ # except OSError:
+ # self.cudart = ctypes.CDLL("cudart64_120.dll")
+
+ # self.cudart.cudaHostRegister.argtypes = [ctypes.c_void_p, ctypes.c_size_t, ctypes.c_uint]
+ # self.cudart.cudaHostRegister.restype = ctypes.c_int
+
+ # self.cudart.cudaHostUnregister.argtypes = [ctypes.c_void_p]
+ # self.cudart.cudaHostUnregister.restype = ctypes.c_int
+
+ # self.pinned_views = []
+ # cudaHostRegisterPortable = 0x01 # Visible to all CUDA contexts
+
+ # # Permanently register and create zero-copy tensor views over all 3 SHM allocations
+ # for shm in self.shms:
+ # # 1. Create a ctypes character array mapping directly onto the memoryview buffer
+ # ctypes_array = (ctypes.c_char * self.frame_bytes).from_buffer(shm.buf)
+
+ # # 2. Extract the true virtual memory address pointer from the ctypes overlay
+ # shm_ptr = ctypes.c_void_p(ctypes.addressof(ctypes_array))
+
+ # # 3. Pin the raw memory chunk inside the OS page tracking tables
+ # res = self.cudart.cudaHostRegister(
+ # shm_ptr, self.frame_bytes, cudaHostRegisterPortable
+ # )
+ # # if res != 0:
+ # if res == 712:
+ # # main_app_logger.warning(
+ # # f"[CUDA SHM] Address {hex(ctypes.addressof(ctypes_array))} is already page-locked. Reusing active pin safely."
+ # # )
+ # main_app_logger.info(
+ # f"[CUDA RESTORE] Found stale page-lock at {hex(shm_ptr.value)}. Forcing reset..."
+ # )
+ # # try:
+ # # # Load low-level driver to break the lazy process hold
+ # # # import ctypes
+ # # # cuda_driver = ctypes.CDLL("libcuda.so") #.6")
+ # # # cuda_driver.cuMemHostUnregister(shm_ptr)
+ # # # force_clear_driver_cuda_pin(shm_ptr.value)
+ # # try:
+ # # cuda_driver = ctypes.CDLL("libcuda.so.6")
+ # # except OSError:
+ # # cuda_driver = ctypes.CDLL("libcuda.so")
+
+ # # # Break the registration using the low-level driver layout hook
+ # # cuda_driver.cuMemHostUnregister(shm_ptr)
+ # self.cudart.cudaHostUnregister(shm_ptr)
+ # # Re-register immediately within the new PyTorch runtime context
+ # res = self.cudart.cudaHostRegister(
+ # shm_ptr, self.frame_bytes, cudaHostRegisterPortable
+ # )
+ # # except Exception as e:
+ # # main_app_logger.debug(
+ # # f"Driver pin restoration fallback skipped: {e}"
+ # # )
+ # if res != 0 and res != 712:
+ # raise RuntimeError(f"cudaHostRegister failed with status code: {res}")
+
+ # # 4. Create a high-speed zero-copy NumPy -> Torch wrapper view over that allocation
+ # # shm_numpy = np.ndarray(self.frame_shape, dtype=np.uint8, buffer=shm.buf)
+ # shm_numpy = np.frombuffer(shm._mmap, dtype=np.uint8).reshape(
+ # self.frame_shape
+ # )
+ # shm_tensor = torch.from_numpy(shm_numpy)
+ # self.pinned_views.append(shm_tensor)
+
+ # self.thread = threading.Thread(target=self._processing_loop, daemon=True)
+ # self.thread.start()
+
+ @torch.inference_mode()
+ def _processing_loop_v1(self):
+ last_processed_shm_idx = -1
+ # torch.cuda.synchronize(self.device_index)
+ torch.cuda.synchronize()
+
+ while not self.stopped: # and self.running_flag.value:
+ with self.frame_condition:
+ # Thread sleeps instantly until the background worker calls notify_all()
+ # self.frame_condition.wait(timeout=1.0)
+ # Only wait if the background worker hasn't delivered a new index yet
+ while (
+ self.latest_idx.value == last_processed_shm_idx and not self.stopped
+ ):
+ if not self.frame_condition.wait(timeout=0.5):
+ if (
+ not self.running_flag.value
+ # and self.latest_idx.value == last_processed_shm_idx
+ ):
+ self.stopped = True
+ break
+ # continue
+ # self.frame_condition.wait()
+ if self.stopped:
+ break
- self.get_stream()
+ active_idx = self.latest_idx.value
+ self.reader_idx.value = active_idx
+ last_processed_shm_idx = active_idx
+
+ # Signal the worker that a slot has cleared up losslessly ---
+ self.buffer_occupancy.value = max(0, self.buffer_occupancy.value - 1)
+ # self.slot_available_event.set()
+
+ target_frames_passed = self.target_frames_passed
+ # total_input_frames = int(
+ # target_frames_passed * getattr(self, "frame_skip", 1)
+ # ) # int(self.raw_frame_counter.value)
+ total_input_frames = round(target_frames_passed * self.step_size)
+
+ # Update counters inside target_fps space loop execution
+ self.total_input_frames = total_input_frames
+
+ t_copy = time.perf_counter()
+ # np.copyto(current_numpy_buf, frame_view)
+ current_tensor_view = self.pinned_views[active_idx]
+ self.total_shm_copy_time += time.perf_counter() - t_copy
+
+ # ASYNCHRONOUS PCIe UPLOAD (Only runs for target frames)
+ t_h2d = time.perf_counter()
+
+ self._static_gpu_frame_buffer.copy_(current_tensor_view, non_blocking=False)
+
+ # 2. Native Channel Inversion View (RGB -> BGR)
+ # Instead of calling OpenCV, use standard tensor index slicing.
+ # This keeps the execution entirely in PyTorch's native C++ backend.
+ # bgr_view = self._static_gpu_frame_buffer[:, :, [2, 1, 0]]
+
+ # 3. Fast Contiguity Restoration Block
+ # Rather than letting downstream handlers detect a non-contiguous stride
+ # layout (which causes major latency spikes), enforce memory linearity
+ # asynchronously inside the upload stream context here:
+ # bgr_tensor = bgr_view.contiguous()
+ # bgr_tensor = torch.flip(self._static_gpu_frame_buffer, dims=[2]).contiguous()
+ # torch.cuda.synchronize(self.device_index)
+ self._bgr_gpu_frame_buffer[:, :, 0] = self._static_gpu_frame_buffer[:, :, 2]
+ self._bgr_gpu_frame_buffer[:, :, 1] = self._static_gpu_frame_buffer[:, :, 1]
+ self._bgr_gpu_frame_buffer[:, :, 2] = self._static_gpu_frame_buffer[:, :, 0]
+
+ # The final tensor to be sent is now the pre-allocated BGR buffer.
+ bgr_tensor = self._bgr_gpu_frame_buffer
+ self.total_h2d_time += time.perf_counter() - t_h2d
+
+ # self.buffer_selector = 1 - self.buffer_selector
+
+ # if not self.is_rtsp or not self.frame_queue.full():
+ t_queue_block = time.perf_counter()
+ # Record an event on the stream and make the main thread wait for it asynchronously
+ current_event = torch.cuda.Event()
+ current_event.record(self.upload_stream)
+ # current_event.wait()
+ # ---
+ if self.is_rtsp:
+ if len(self.frame_queue) >= self.frame_queue.maxlen:
+ try:
+ # Pop the oldest item from the queue
+ stale_item = self.frame_queue.popleft()
+ # Extract its tracking event reference
+ stale_event = stale_item[2]
+ if stale_event:
+ # Clear the event reference without blocking the hot loop
+ del stale_event
+ except IndexError:
+ pass
+ self.frame_queue.append(
+ (
+ True,
+ bgr_tensor,
+ current_event,
+ target_frames_passed,
+ total_input_frames,
+ )
+ )
- # Map Codec
- codec_map = {"hevc": nvc.cudaVideoCodec.HEVC, "h264": nvc.cudaVideoCodec.H264}
- self.nvc_codec = codec_map.get(
- self.stream.codec_context.name, nvc.cudaVideoCodec.HEVC
- )
+ else:
+ # LOCAL Lossless Files: Check the previous frame's hardware event
+ # instead of blocking on the current frame's upload status.
+ if len(self.frame_queue) >= self.frame_queue.maxlen:
+ # Retrieve the oldest active item to check its status
+ oldest_frame_set = self.frame_queue[0]
+ oldest_event = oldest_frame_set[2]
+
+ if oldest_event and not oldest_event.query():
+ # Only wait if the hardware lane is completely saturated
+ # oldest_event.synchronize()
+
+ # Yield thread execution priority cooperatively for a microscopic window
+ # to let the PCIe DMA engine finish transferring bytes smoothly
+ time.sleep(0.0005)
+ # 2. Asynchronous Watchdog Check
+ if not oldest_event.query():
+ oldest_event.synchronize() # Hard fence fallback only when fully saturated
+
+ # self.frame_queue.append(
+ # (True, bgr_tensor, current_event, target_frames_passed, total_input_frames)
+ # )
+ with self.frame_condition:
+ self.frame_queue.append(
+ (
+ True,
+ bgr_tensor,
+ current_event,
+ target_frames_passed,
+ total_input_frames,
+ )
+ )
+ self.frame_condition.notify_all() # Instantly wake up your main read lane!
+
+ # ---
+
+ self.total_queue_wait_time += time.perf_counter() - t_queue_block
+ self.target_frames_passed += 1
+ del bgr_tensor, current_tensor_view # bgr_view
+
+ # @torch.inference_mode()
+ # def _processing_loop(self):
+ # last_processed_shm_idx = -1
+ # # torch.cuda.synchronize(self.device_index)
+ # torch.cuda.synchronize()
+
+ # while not self.stopped: # and self.running_flag.value:
+ # with self.frame_condition:
+ # # Thread sleeps instantly until the background worker calls notify_all()
+ # # self.frame_condition.wait(timeout=1.0)
+ # # Only wait if the background worker hasn't delivered a new index yet
+ # while (
+ # self.latest_idx.value == last_processed_shm_idx and not self.stopped
+ # ):
+ # if not self.frame_condition.wait(timeout=0.5):
+ # if (
+ # not self.running_flag.value
+ # # and self.latest_idx.value == last_processed_shm_idx
+ # ):
+ # self.stopped = True
+ # break
+ # # continue
+ # # self.frame_condition.wait()
+ # if self.stopped:
+ # break
+
+ # active_idx = self.latest_idx.value
+ # self.reader_idx.value = active_idx
+ # last_processed_shm_idx = active_idx
+
+ # # Signal the worker that a slot has cleared up losslessly ---
+ # self.buffer_occupancy.value = max(0, self.buffer_occupancy.value - 1)
+ # # self.slot_available_event.set()
+
+ # target_frames_passed = self.target_frames_passed
+ # # total_input_frames = int(
+ # # target_frames_passed * getattr(self, "frame_skip", 1)
+ # # ) # int(self.raw_frame_counter.value)
+ # total_input_frames = round(target_frames_passed * self.step_size)
+
+ # # Update counters inside target_fps space loop execution
+ # self.total_input_frames = total_input_frames
+
+ # t_copy = time.perf_counter()
+ # # np.copyto(current_numpy_buf, frame_view)
+ # current_tensor_view = self.pinned_views[active_idx]
+ # self.total_shm_copy_time += time.perf_counter() - t_copy
+
+ # # ASYNCHRONOUS PCIe UPLOAD (Only runs for target frames)
+ # t_h2d = time.perf_counter()
+
+ # with torch.cuda.stream(self.upload_stream):
+ # self._static_gpu_frame_buffer.copy_(current_tensor_view, non_blocking=True)
+
+ # # Record the hardware completion event directly
+ # current_event = torch.cuda.Event()
+ # current_event.record(self.upload_stream)
+ # self.total_h2d_time += time.perf_counter() - t_h2d
+
+ # # self.buffer_selector = 1 - self.buffer_selector
+
+ # # if not self.is_rtsp or not self.frame_queue.full():
+ # t_queue_block = time.perf_counter()
+ # # Record an event on the stream and make the main thread wait for it asynchronously
+ # # current_event = torch.cuda.Event()
+ # # current_event.record(self.upload_stream)
+ # # current_event.wait()
+ # # ---
+ # if len(self.frame_queue) >= self.frame_queue.maxlen:
+ # try:
+ # self.frame_queue.popleft()
+ # except IndexError:
+ # pass
+
+ # self.frame_queue.append(
+ # (
+ # True,
+ # self._static_gpu_frame_buffer,
+ # current_event,
+ # target_frames_passed,
+ # total_input_frames,
+ # )
+ # )
- # self.stream_width = self.stream.width
- # self.stream_height = self.stream.height
+ # # ---
- self.frame_width = self.true_width
- self.frame_height = self.true_height
+ # self.total_queue_wait_time += time.perf_counter() - t_queue_block
+ # self.target_frames_passed += 1
+ # del bgr_tensor, current_tensor_view # bgr_view
- self.input_fps = self.metadata_fps
+ @torch.inference_mode()
+ def _processing_loop(self):
+ """
+ High-Throughput Asynchronous GPU Frame Ingestion Loop.
- self.target_fps = (
- target_fps
- if target_fps not in [None, 0] and self.input_fps > target_fps
- else self.input_fps
- )
- if self.input_fps > 0 and self.target_fps > 0:
- self.frame_skip = max(1, int(self.input_fps / self.target_fps))
- else:
- self.frame_skip = 1
- self.max_frames_per_clip = (
- None
- if clip_duration is None
- else int(self.target_fps * float(clip_duration))
- )
- self.frame_interval = 1.0 / self.target_fps
+ Transfers decoded 8K BGR frames from shared memory into pre-allocated VRAM buffers
+ over a dedicated CUDA upload stream without blocking the CPU thread.
+ """
+ last_processed_shm_idx = -1
- self.numFrames = self.total_frames
+ # Pre-synchronize the upload stream before entering the hot loop
+ if hasattr(self, "upload_stream") and self.upload_stream is not None:
+ self.upload_stream.synchronize()
- self.raw_input = torch.empty(
- (self.frame_height, self.frame_width, 3), dtype=torch.uint8, device="cuda"
- )
- self.sync_locked = True
+ while not self.stopped:
+ # PULL NEXT READY FRAME INDEX FROM BACKGROUND WORKER -----------------
+ if hasattr(self, "frame_condition") and self.frame_condition is not None:
+ with self.frame_condition:
+ # Wait for the background reader process to deliver a new frame slot
+ while (
+ self.latest_idx.value == last_processed_shm_idx
+ and not self.stopped
+ ):
+ if not self.frame_condition.wait(timeout=0.2):
+ if (
+ not getattr(self, "running_flag", None)
+ or not self.running_flag.value
+ ):
+ self.stopped = True
+ break
+ if self.stopped:
+ break
- def get_stream(self):
- max_retries = 5
- retry_cnt = 0
- connected = False
- is_rtsp = str(self.source).startswith("rtsp")
+ active_idx = self.latest_idx.value
+ self.reader_idx.value = active_idx
+ last_processed_shm_idx = active_idx
- while not connected:
- try:
- self.container = av.open(
- self.source,
- options={
- **self.av_options,
- "err_detect": "ignore_err", # Don't stop demuxing on minor packet errors
- "flags": "low_delay", # Reduce internal buffering
- },
- )
- self.container.streams.video[0].thread_type = "AUTO"
- streams = self.container.streams.get(video=0)
- self.stream = streams[0] if isinstance(streams, list) else streams
+ # Decrement buffer occupancy atomically
+ if hasattr(self, "buffer_occupancy"):
+ self.buffer_occupancy.value = max(0, self.buffer_occupancy.value - 1)
- if self.stream:
- self.stream_width = self.stream.width
- self.stream_height = self.stream.height
- connected = True
+ # Calculate physical vs. target frame indices
+ target_frames_passed = self.target_frames_passed
+ total_input_frames = round(
+ target_frames_passed * getattr(self, "step_size", 1.0)
+ )
+ self.total_input_frames = total_input_frames
- except Exception as e:
- retry_cnt += 1
- self.init_error = str(e)
+ # 2. ZERO-COPY HOST TENSOR RETRIEVAL -----------------------------------
+ current_tensor_view = self.pinned_views[active_idx]
- if not is_rtsp or retry_cnt >= max_retries:
- main_app_logger.error(
- f"Critical: Could not open/connect to {self.source}"
- )
- self.reconnect_failed = True
- self.stopped = True
- # Halt and exit the server immediately
- raise RuntimeError(
- f"Critical GPU stream reader initialization failure: {self.init_error}"
- )
- return
+ # 3. ASYNCHRONOUS PCIe UPLOAD TO GPU (VRAM) ---------------------------
+ t_h2d = time.perf_counter()
- wait_time = retry_cnt * 2
- main_app_logger.warning(
- f"GPU connection pending... Retry ({retry_cnt}/{max_retries}) in {wait_time} seconds."
+ # Execute DMA copy and hardware event record on the dedicated upload stream
+ with torch.cuda.stream(self.upload_stream):
+ # Asynchronous PCIe Host-to-Device transfer (Takes < 0.1ms CPU time)
+ self._static_gpu_frame_buffer.copy_(
+ current_tensor_view, non_blocking=True
)
- time.sleep(wait_time)
- def stream_frames(self):
- """
- Universal background thread for 8K video.
- - RTSP: Reconnects automatically if the stream drops.
- - Files: Processes until EOF and then stops cleanly.
- """
- is_rtsp = str(self.source).startswith("rtsp")
- max_retries = 5
- retry_cnt = 0
+ # Fast Zero-Allocation Color Swap (BGR Device -> RGB Device)
+ # .flip(dims=[2]) creates a fast virtual view.
+ # .copy_() writes it contiguously into the pre-allocated RGB buffer.
+ self._flipped_static_gpu_frame_buffer.copy_(
+ self._static_gpu_frame_buffer.flip(dims=[2]), non_blocking=True
+ )
- # The outer while loop allows RTSP to recover from network hiccups
- while not self.stopped:
- try:
- # Ensure the container and decoder are active
- # For RTSP, if the demuxer loop below exits, we re-verify the connection here
- if self.container is None:
+ # Record hardware completion barrier on the upload stream
+ current_event = torch.cuda.Event()
+ current_event.record(self.upload_stream)
+
+ self.total_h2d_time += time.perf_counter() - t_h2d
+
+ # 4. LOCK-FREE THREAD-SAFE QUEUE INGESTION -----------------------------
+ t_queue_block = time.perf_counter()
+
+ # Evict oldest unread frame if queue is full to preserve real-time latency
+ # if len(self.frame_queue) >= self.frame_queue.maxlen:
+ # try:
+ # self.frame_queue.popleft()
+ # except IndexError:
+ # pass
+
+ # # Push the VRAM buffer and hardware completion event to the consumer queue
+ # self.frame_queue.append(
+ # (
+ # True,
+ # self._static_gpu_frame_buffer,
+ # current_event,
+ # target_frames_passed,
+ # total_input_frames,
+ # )
+ # )
+
+ if getattr(self, "is_rtsp", False):
+ # Live RTSP: Evict oldest frame if consumer falls behind
+ if len(self.frame_queue) >= self.frame_queue.maxlen:
try:
- self.container = av.open(
- self.source,
- options={
- **self.av_options,
- "err_detect": "explode", # Don't stop demuxing on minor packet errors
- "flags": "low_delay", # Reduce internal buffering
- },
- )
- retry_cnt = 0
- except Exception as e:
- retry_cnt += 1
- if retry_cnt >= max_retries:
- # Catch the 404 and stop the thread instead of retrying
- main_app_logger.info(
- f"Stream {self.source} ended or not found. Closing reader."
- )
- self.stopped = True
- break
- wait_time = retry_cnt * 2
- main_app_logger.warning(
- f"Connection failed ({e}). Retry {retry_cnt}/{max_retries} in {wait_time}s..."
- )
- time.sleep(wait_time)
- continue
-
- streams = self.container.streams.get(video=0)
- self.stream = streams[0] if isinstance(streams, list) else streams
-
- # INITIALIZE DECODER if missing (Fixes the NoneType Error)
- if self.nv_dec is None:
- # Sync with PyTorch
- self.cuda_lib.cuCtxSetCurrent(ctypes.c_void_p(self.cuda_ctx_handle))
- torch_stream = torch.cuda.current_stream().cuda_stream
-
- # Detect Codec
- codec_map = {
- "hevc": nvc.cudaVideoCodec.HEVC,
- "h264": nvc.cudaVideoCodec.H264,
- }
- nvc_codec = codec_map.get(
- self.stream.codec_context.name, nvc.cudaVideoCodec.HEVC
+ self.frame_queue.popleft()
+ except IndexError:
+ pass
+ self.frame_queue.append(
+ (
+ True,
+ self._flipped_static_gpu_frame_buffer,
+ current_event,
+ target_frames_passed,
+ total_input_frames,
+ )
+ )
+ else:
+ # Local Video File: Wait cooperatively so ZERO frames are lost
+ while (
+ len(self.frame_queue) >= self.frame_queue.maxlen
+ and not self.stopped
+ ):
+ time.sleep(0.001)
+
+ self.frame_queue.append(
+ (
+ True,
+ self._flipped_static_gpu_frame_buffer,
+ current_event,
+ target_frames_passed,
+ total_input_frames,
)
+ )
- # Professional cards often support 8K HEVC but are capped at 4K for H.264
- if nvc_codec == nvc.cudaVideoCodec.HEVC:
- # Use 8K limits for HEVC
- hw_max_w, hw_max_h = 8192, 4320
- else:
- # Safely default to 4K limits for H.264 and other codecs
- hw_max_w, hw_max_h = 4096, 4096
+ self.total_queue_wait_time += time.perf_counter() - t_queue_block
+ self.target_frames_passed += 1
- # Ensure max dimensions are at least as large as current stream
- hw_max_w = max(hw_max_w, self.stream_width)
- hw_max_h = max(hw_max_h, self.stream_height)
+ # Release local pointer references immediately
+ del current_tensor_view
- # Determine if the library build supports Ultra Low Latency enums.
- try:
- latency_mode = nvc.DisplayDecodeLatencyType.ULTRA_LOW_LATENCY
- except AttributeError:
- latency_mode = nvc.DisplayDecodeLatencyType.NATIVE
-
- self.nv_dec = nvc.CreateDecoder(
- gpuid=self.gpu_id,
- codec=nvc_codec,
- cudacontext=int(self.cuda_ctx_handle),
- cudastream=int(torch_stream),
- usedevicememory=1,
- maxwidth=hw_max_w, # Force 8K profile support
- maxheight=hw_max_h,
- latency=latency_mode,
- )
+ def stop(self):
+ """GPU specialized cleaner layer. Unpins hardware tracking tables, flushes VRAM
- # Re-initialize BitStream Filter for this session
- bsf_map = {"hevc": "hevc_mp4toannexb", "h264": "h264_mp4toannexb"}
- bsf_name = bsf_map.get(self.stream.codec_context.name)
- local_bsf = (
- av.BitStreamFilterContext(bsf_name, self.stream)
- if bsf_name
- else None
- )
+ and synchronizes active side streams BEFORE parsing to the BaseReader.
+ """
+ if self.stopped:
+ return
- target_fps, clip_duration = (self.target_fps, self.clip_duration)
- self.stream_width = self.stream.width
- self.stream_height = self.stream.height
- self.frame_width = self.true_width
- self.frame_height = self.true_height
- self.input_fps = self.metadata_fps
- self.target_fps = (
- target_fps
- if target_fps not in [None, 0] and self.input_fps > target_fps
- else self.input_fps
- )
- if self.input_fps > 0 and self.target_fps > 0:
- self.frame_skip = max(1, int(self.input_fps / self.target_fps))
- else:
- self.frame_skip = 1
- self.max_frames_per_clip = (
- None
- if clip_duration is None
- else int(self.target_fps * float(clip_duration))
- )
- self.frame_interval = 1.0 / self.target_fps
- self.numFrames = self.total_frames
+ self.stopped = True
- self.is_h264_8k = (
- self.stream.codec_context.name == "h264"
- and self.stream_width > 4096
- )
+ # Signal the worker process to exit its loop
+ if hasattr(self, "running_flag"):
+ self.running_flag.value = False
+
+ # 1. Drive standalone GPU hardware unpinning while shm buffer mappings are valid
+ # if hasattr(self, "release_hardware_pinsv2"):
+ # self.release_hardware_pinsv2()
+
+ # 2. Break active PyTorch tensor data reference vectors to drop allocator pin metrics
+ for tensor_attr in [
+ "pinned_views",
+ "d2h_buffers",
+ "d2h_numpys",
+ "_static_gpu_frame_buffer",
+ "_bgr_gpu_frame_buffer",
+ ]:
+ if hasattr(self, tensor_attr):
+ setattr(self, tensor_attr, None)
+
+ # 3. Synchronize and overwrite side hardware execution lanes
+ if hasattr(self, "upload_stream") and self.upload_stream is not None:
+ try:
+ self.upload_stream.synchronize()
+ except Exception:
+ pass
+ self.upload_stream = None
- # --- PATH A: H.264 8K CPU Fallback ---
- if self.is_h264_8k:
- print(
- f"[INFO] Starting CPU-Decode Fallback for {self.source}",
- flush=True,
- )
- for frame in self.container.decode(video=0):
- # if self.stopped:
- # break
-
- # img_array = frame.to_ndarray(format="rgb24")
- img_array = frame.to_ndarray(format="bgr24")
- # gpu_tensor = (
- # torch.from_numpy(img_array).to("cuda").permute(2, 0, 1)
- # )
- self.raw_input.copy_(torch.from_numpy(img_array))
- gpu_tensor = self.raw_input.permute(2, 0, 1)
+ if hasattr(self, "download_stream") and self.download_stream is not None:
+ try:
+ self.download_stream.synchronize()
+ except Exception:
+ pass
+ self.download_stream = None
- # CRITICAL FIX: Break the shared buffer pointer!
- # Clones the frame to a private memory block before queueing it.
- safe_frame = gpu_tensor.clone().contiguous()
+ # 4. Dismantle low-level ctypes CDLL pointers to completely drop unmanaged contexts
+ # self.cudart = None
- self.frame_queue.put((safe_frame, self.frame_idx))
- self.frame_idx += 1
+ # 5. Hand execution over to BaseReader to tear down shared primitives and worker structures
+ super().stop()
- # if not is_rtsp and self.frame_idx >= 500:
- # break # File test limit
+ def release_hardware_pins_v1(self):
+ if hasattr(self, "cudart") and getattr(self, "shms", None):
+ # import ctypes
- # --- PATH B: HEVC Hardware Acceleration (15 FPS Path) ---
- else:
- print(
- f"[INFO] Starting HW-Accelerated Pump for {self.source}",
- flush=True,
+ for shm in self.shms:
+ try:
+ # Re-map the raw structural array pointer identity block
+ ctypes_array = (ctypes.c_char * self.frame_bytes).from_buffer(
+ shm.buf
)
- # Create an explicit stream packet extractor to isolate bitstream errors
- packet_iterator = self.container.demux(self.stream)
+ shm_ptr = ctypes.c_void_p(ctypes.addressof(ctypes_array))
+
+ # Force the CUDA device table driver to drop the OS memory pin
+ res = self.cudart.cudaHostUnregister(shm_ptr)
+ if res != 0:
+ pass # Ignore if already unpinned natively
+ except Exception:
+ pass
+ main_app_logger.info(
+ "\033[92m[CUDA UNPIN] Shared memory pages successfully released from hardware table.\033[0m"
+ )
- while not self.stopped:
- try:
- packet = next(packet_iterator)
- except (ValueError, OSError, StopIteration):
- break
+ def release_hardware_pinsv2(self):
+ """
+ Reverses the CUDA host registration by mapping the exact memory locations
+ and force-unpins them before closing the underlying OS shared memory handles.
+ """
+ # import ctypes
+ # import gc
+ # import sys
- # Check for corruption/validity safely
- is_broken = packet.size == 0 or packet.dts is None
- # is_broken = (
- # is_broken
- # or getattr(packet, "corrupt", False)
- # or getattr(packet, "is_corrupt", False)
- # )
+ # Localized stream filter to swallow intermediate garbage collection signals
- if is_broken:
- # main_app_logger.warning("[WARNING] Corrupt packet detected. Locking sync.")
- # self.sync_locked = True
- continue
-
- # ─── OPTIMIZED HARDWARE SYNCHRONIZATION BARRIER ──────────────────
- if self.sync_locked:
- # Rely explicitly on PyAV header flag definitions instead of packet.size
- if getattr(packet, "is_keyframe", False):
- main_app_logger.info(
- "[SYNC] Valid hardware keyframe intercepted. Unlocking decoder track."
- )
- self.sync_locked = False
- else:
- # Drop incomplete frames to prevent hardware canvas corruption
- continue
- # ─────────────────────────────────────────────────────────────────
- # if self.stopped:
- # break
- if packet.size == 0 or packet.dts is None:
- continue
-
- # Apply Annex B filter
- filtered_packets = (
- local_bsf.filter(packet) if local_bsf else [packet]
+ original_stderr = sys.stderr
+ sys.stderr = ResourceTrackerFilter(original_stderr)
+
+ try:
+ # 2. Drop active processing frame buffers and views to unlock reference paths
+ # if (
+ # hasattr(self, "_static_gpu_frame_buffer")
+ # and self._static_gpu_frame_buffer is not None
+ # ):
+ # self._static_gpu_frame_buffer = None
+
+ # if hasattr(self, "pinned_views") and self.pinned_views:
+ # for idx in range(len(self.pinned_views)):
+ # self.pinned_views[idx] = None
+ # self.pinned_views.clear()
+
+ # # 3. Synchronize background streaming channels
+ # if hasattr(self, "upload_stream") and self.upload_stream is not None:
+ # try:
+ # self.upload_stream.synchronize()
+ # except Exception:
+ # pass
+ # self.upload_stream = None
+
+ # 4. Iterate and forcefully unregister the exact memory allocation pointers
+ if hasattr(self, "cudart") and getattr(self, "shms", None):
+ for shm in list(self.shms):
+ try:
+ if hasattr(shm, "buf") and shm.buf is not None:
+ # Map the exact size layout matching line 969 to capture the real pointer
+ ctypes_array = (
+ ctypes.c_char * self.frame_bytes
+ ).from_buffer(shm.buf)
+ shm_ptr = ctypes.c_void_p(ctypes.addressof(ctypes_array))
+
+ # Direct low-level Driver API fallback execution to wipe lazy driver context maps
+ # if hasattr(self, "force_clear_driver_cuda_pin"):
+ # self.force_clear_driver_cuda_pin(shm_ptr.value)
+
+ # High-level Runtime API eviction
+ self.cudart.cudaHostUnregister(shm_ptr)
+ # if res == 0:
+ # main_app_logger.info(
+ # f"[CUDA UNPIN] Successfully unpinned address context: {hex(shm_ptr.value)}"
+ # )
+
+ # 5. Instantly close and unlink POSIX shared memory allocations to release locks
+ # shm.close()
+ # shm.unlink()
+ except Exception as e:
+ main_app_logger.debug(
+ f"Failed to cleanly unpin shared memory region: {e}"
)
- for filtered_packet in filtered_packets:
- if self.sync_locked:
- continue
-
- if filtered_packet.size < 10:
- continue
-
- pkt_bytes = bytes(filtered_packet)
- # Extract raw memory address from the numpy tuple
- ptr_info = np.frombuffer(
- pkt_bytes, dtype=np.uint8
- ).__array_interface__["data"]
- addr = (
- ptr_info[0]
- if isinstance(ptr_info, (tuple, list))
- else ptr_info
- )
- nvc_packet = nvc.PacketData()
- nvc_packet.bsl_data = int(addr)
- nvc_packet.bsl = filtered_packet.size
+ # self.shms.clear()
- try:
- for decoded_frame in self.nv_dec.Decode(nvc_packet):
- try:
- # Zero-Copy Bridge: Hardware Surface -> PyTorch Tensor
- gpu_tensor = torch.from_dlpack(decoded_frame)
- # 2. CONVERSION: Turn NV12 [YUV] into BGR immediately
- converted_bgr = self.nv12_to_bgr_reader(
- gpu_tensor
- )
-
- # CRITICAL FIX: Force a brand-new, isolated memory block allocation
- # on the GPU so the decoder cannot overwrite this frame's pixels
- # while downstream threads are analyzing the mask.
- safe_frame = converted_bgr.clone().contiguous()
-
- while not self.stopped:
- try:
- self.frame_queue.put(
- (
- safe_frame,
- self.frame_idx,
- ),
- timeout=0.1,
- )
- break
- except queue.Full:
- try:
- self.frame_queue.get_nowait()
- except queue.Empty:
- pass
- torch.cuda.current_stream().synchronize()
- self.frame_idx += 1
- except (
- av.FFmpegError,
- av.InvalidDataError,
- RuntimeError,
- ) as decode_fault:
- main_app_logger.warning(
- f"Isolating corrupt video packet fragment: {decode_fault}"
- )
- if hasattr(decoded_frame, "Unlock"):
- decoded_frame.Unlock()
- continue
- finally:
- # Immediate VRAM Cleanup
- if "gpu_tensor" in locals():
- del gpu_tensor
- if hasattr(decoded_frame, "Unlock"):
- decoded_frame.Unlock()
- del decoded_frame
-
- except Exception as e:
- self.sync_locked = True
- if any(c in str(e) for c in ["700", "208"]):
- self.nv_dec = None
- break
- # if not is_rtsp and self.frame_idx >= 500:
- # break # File test limit
-
- # --- UNIVERSAL EXIT LOGIC ---
- if not is_rtsp:
- print(
- f"[DEBUG] Reached End of File: {self.source}. Exiting thread.",
- flush=True,
- )
- self.stopped = True
- break
- else:
- # If we reach here and it's RTSP, the demuxer stopped yielding
- print(
- "[WARNING] RTSP glitch detected. Closing container and retrying...",
- flush=True,
- )
- if self.container:
- # Explicitly flush streams before closing
- for s in self.container.streams:
- s.codec_context.flush_buffers()
- self.container.close()
- self.container = None
-
- if local_bsf:
+ finally:
+ # 6. Flush native device pools completely
+ gc.collect()
+ if torch.cuda.is_available():
+ torch.cuda.synchronize()
+ torch.cuda.empty_cache()
+ sys.stderr = original_stderr
+ # print(
+ # "\033[92m\n[CUDA UNPIN] Hardware tables fully cleared.\033[0m",
+ # flush=True,
+ # )
+
+ def release_hardware_pins(self):
+ """Forcefully purges child workers and cleanly unpins raw CUDA memory layers."""
+
+ # 1. FORCE-KILL THE BACKGROUND MULTIPROCESSING WORKER FIRST
+ # If left alive, it anchors the page lock inside the Linux kernel tables!
+ if hasattr(self, "worker") and self.worker is not None:
+ try:
+ if self.worker.is_alive():
+ self.worker.terminate()
+ self.worker.join(timeout=0.1)
+
+ # Hard kill fallback if the worker process ignores graceful signals
+ if self.worker.is_alive() and getattr(self.worker, "pid", None):
+ if self.worker.pid > 1:
+ # main_app_logger.info(
+ # f"[KILL] Evicting worker holding lock (PID: {self.worker.pid})"
+ # )
try:
- local_bsf.filter(None)
- except Exception:
- pass
+ os.kill(self.worker.pid, signal.SIGKILL)
+ self.worker.join()
+ except ProcessLookupError:
+ pass # It died naturally right before the signal arrived
+ except Exception:
+ pass
+ self.worker = None
+
+ # 2. Drop active PyTorch frame views to unlock storage layout references
+ if (
+ hasattr(self, "_static_gpu_frame_buffer")
+ and self._static_gpu_frame_buffer is not None
+ ):
+ self._static_gpu_frame_buffer = None
+
+ if hasattr(self, "pinned_views") and self.pinned_views:
+ for idx in range(len(self.pinned_views)):
+ self.pinned_views[idx] = None
+ self.pinned_views.clear()
+
+ # Flush outstanding elements inside your streaming frame buffers queue
+ if hasattr(self, "frame_queue"):
+ try:
+ while len(self.frame_queue) > 0:
+ self.frame_queue.popleft()
+ except Exception:
+ pass
+ gc.collect()
+
+ # 3. Safely unregister host pages via the exact allocation pointers
+ if hasattr(self, "cudart") and getattr(self, "shms", None):
+ for shm in list(self.shms):
+ try:
+ # Reconstruct the exact pointer alignment mapping to clear the driver registration
+ ctypes_array = (ctypes.c_char * self.frame_bytes).from_buffer(
+ shm.buf
+ )
+ shm_ptr = ctypes.c_void_p(ctypes.addressof(ctypes_array))
- # Force close the C++ decoder context layer before resetting the pointer
- if self.nv_dec is not None:
+ # Call high-speed unregister on the exact pointer value
+ self.cudart.cudaHostUnregister(shm_ptr)
+
+ # Release internal mmap references safely before executing close()
+ if hasattr(shm, "buf") and shm.buf is not None:
try:
- if hasattr(self.nv_dec, "Close"):
- self.nv_dec.Close()
+ shm.buf.release()
except Exception:
pass
- del self.nv_dec
- torch.cuda.synchronize()
- torch.cuda.empty_cache()
- self.nv_dec = None
- while not self.frame_queue.empty():
+ if hasattr(shm, "_mmap") and shm._mmap is not None:
try:
- self.frame_queue.get_nowait()
+ shm._mmap = None
except Exception:
- break
- time.sleep(0.5) # Wait before reconnection attempt
+ pass
- except Exception as e:
- print(f"[EXCEPTION] Reader Thread Failure: {e}", flush=True)
- traceback.print_exc()
- if not is_rtsp:
- self.stopped = True
- break
+ # Direct file descriptor removal and unlinking from /dev/shm
+ shm.close()
+ shm.unlink()
+ except Exception:
+ pass
+ self.shms.clear()
- self.sync_locked = True
- if self.nv_dec is not None:
- try:
- # Force a hardware surface close step to release hardware rings instantly
- if hasattr(self.nv_dec, "Close"):
- self.nv_dec.Close()
- except Exception:
- pass
- del self.nv_dec
+ # 4. Flush native memory pools completely back to Host OS
+ gc.collect()
+ if torch.cuda.is_available():
+ torch.cuda.synchronize()
+ torch.cuda.empty_cache()
- # Force CUDA to synchronize and clear driver error state flags safely
- torch.cuda.synchronize()
- torch.cuda.empty_cache()
- self.nv_dec = None
- time.sleep(1.0) # Backoff for RTSP reconnection
- self.stopped = True
+def force_clear_driver_cuda_pin(raw_address_int):
+ """
+ Forcefully drops an unmanaged page-lock registration directly out of the
+ NVIDIA Driver API context using a raw virtual address integer pointer.
+ """
+ if not raw_address_int:
+ return
- def release(self):
- """Safely flushes the decoder and closes the connection."""
- print("Closing HybridReader...")
- self.stopped = True # Signal thread to stop
- time.sleep(0.5)
+ main_app_logger.info(
+ f"[HARDWARE FORCE-CLEAR] Evicting address {hex(raw_address_int)} from driver context..."
+ )
- if self.nv_dec and self.frame_idx > 0:
- try:
- self.nv_dec.Decode(nvc.PacketData()) # Flush
- del self.nv_dec
- self.nv_dec = None
- except Exception:
- pass
- if self.container:
- self.container.close()
- self.container = None
-
- # def nv12_to_bgr_reader(self, nv12_tensor): # , h, w, is_8k=False):
- # """Internal reader helper to convert raw hardware surfaces to BGR."""
- # h, w = self.frame_height, self.frame_width
- # is_8k = self.is_h264_8k
- # with torch.no_grad():
- # if is_8k:
- # # 8K Path: Extract and unzipper interleaved UV
- # y = nv12_tensor[0:1, :, :].half()
- # uv_raw = nv12_tensor[1:2, :, :].half()
- # u = uv_raw[:, :, 0::2]
- # v = uv_raw[:, :, 1::2]
- # u = F.interpolate(u.unsqueeze(0), size=(h, w), mode="nearest").squeeze(
- # 0
- # )
- # v = F.interpolate(v.unsqueeze(0), size=(h, w), mode="nearest").squeeze(
- # 0
- # )
- # else:
- # # Standard NV12 Path
- # y = nv12_tensor[:h, :w].unsqueeze(0).half()
-
- # uv = (
- # nv12_tensor[h:, :w]
- # .reshape(h // 2, w // 2, 2)
- # .permute(2, 0, 1)
- # .unsqueeze(0)
- # .half()
- # )
- # uv_up = F.interpolate(uv, size=(h, w), mode="nearest")
- # u, v = uv_up[0, 0:1, :, :], uv_up[0, 1:2, :, :]
-
- # # BT.709 Math (Natural Color)
- # y = (y - 16.0) * 1.164
- # u, v = u - 128.0, v - 128.0
-
- # r = y + 1.793 * v
- # g = y - 0.213 * u - 0.533 * v
- # b = y + 2.112 * u
-
- # # Stack as BGR [B, G, R] for OpenCV/Browser compatibility
- # return torch.cat([b, g, r], dim=0).clamp(0, 255).to(torch.uint8)
-
- def nv12_to_bgr_reader(self, nv12_tensor):
- """Internal reader helper to convert raw hardware surfaces to BGR."""
- h, w = self.frame_height, self.frame_width
- is_8k = self.is_h264_8k
-
- # EXTRACT TRUE STRIDE: Get the actual hardware allocation width
- # which includes the hidden byte-alignment padding columns.
- stride_w = nv12_tensor.shape[1]
-
- with torch.no_grad():
- if is_8k:
- # 8K Path: Slice using stride_w, then strip padding out cleanly
- y = nv12_tensor[0:1, :, :stride_w].half()
- uv_raw = nv12_tensor[1:2, :, :stride_w].half()
- u = uv_raw[:, :, 0::2]
- v = uv_raw[:, :, 1::2]
- u = F.interpolate(u.unsqueeze(0), size=(h, w), mode="nearest").squeeze(
- 0
- )
- v = F.interpolate(v.unsqueeze(0), size=(h, w), mode="nearest").squeeze(
- 0
- )
- else:
- # Standard NV12 Path
- # 1. Slice using the full hardware stride_w to preserve vertical row boundaries
- y_padded = nv12_tensor[:h, :stride_w].half()
- uv_padded = nv12_tensor[h:, :stride_w].half()
-
- # 2. Crop out the hardware padding columns horizontally to get clean spatial grids
- y = y_padded[:, :w].unsqueeze(0)
-
- uv = (
- uv_padded[:, :w]
- .reshape(h // 2, w // 2, 2)
- .permute(2, 0, 1)
- .unsqueeze(0)
- )
- uv_up = F.interpolate(uv, size=(h, w), mode="nearest")
- u, v = uv_up[0, 0:1, :, :], uv_up[0, 1:2, :, :]
-
- # BT.709 Math (Natural Color)
- y = (y - 16.0) * 1.164
- u, v = u - 128.0, v - 128.0
+ try:
+ # 1. Load the core low-level CUDA Driver API library (Linux fallback)
+ try:
+ cuda_driver = ctypes.CDLL("libcuda.so.6")
+ except OSError:
+ cuda_driver = ctypes.CDLL("libcuda.so") # Alternate system path fallback
- r = y + 1.793 * v
- g = y - 0.213 * u - 0.533 * v
- b = y + 2.112 * u
+ # 2. Reconstruct the raw void pointer from the integer key
+ void_ptr = ctypes.c_void_p(raw_address_int)
- # Stack as BGR [B, G, R]
- out_tensor = torch.cat([b, g, r], dim=0).clamp(0, 255).to(torch.uint8)
+ # 3. Invoke cuMemHostUnregister directly to break the lazy driver lock
+ # Status code 0 indicates absolute success
+ status = cuda_driver.cuMemHostUnregister(void_ptr)
- # CRITICAL PIECE: Force memory to be completely linear and sequentially packed
- # before it enters the queue or gets sliced by the sub-frame window pipeline
- return out_tensor.contiguous()
+ if status == 0:
+ main_app_logger.info(
+ f" [SUCCESS] Address {hex(raw_address_int)} successfully purged from driver context."
+ )
+ else:
+ # Code 101/713 typically means it was already lazily swept by an OS page update
+ main_app_logger.info(
+ f" [INFO] Driver returned status code {status} for address {hex(raw_address_int)}. Cleared."
+ )
- @property
- def true_height(self):
- """
- Returns the logical video height.
- Handles cases where metadata might report the 1.5x NV12 buffer height.
- """
- # if self.nv_dec:
- # return int(self.nv_dec.Height()) # Get from decoder instead of stream
- # return int(self.stream.height) # / 1.5)
- return self.stream.height if self.stream else 0
-
- @property
- def true_width(self):
- # if self.nv_dec:
- # return self.nv_dec.Width()
- # return self.stream.width
- return self.stream.width if self.stream else 0
-
- @property
- def metadata_fps(self):
- """Returns the FPS defined in the video metadata/header."""
- if self.stream and self.stream.average_rate:
- return float(self.stream.average_rate)
- return 0.0
-
- @property
- def total_frames(self):
- """Returns the total number of frames defined in the file metadata."""
- if self.stream:
- # Check nb_frames first
- if self.stream.frames > 0:
- return self.stream.frames
- return 0 # Returns 0 for live streams or files with missing metadata
+ except Exception as error:
+ main_app_logger.info(
+ f" [FAILED] Driver force-clear encountered an error: {error}"
+ )
diff --git a/fastapi/include/utils.py b/fastapi/include/utils.py
index 0cae055..f2e11a1 100644
--- a/fastapi/include/utils.py
+++ b/fastapi/include/utils.py
@@ -1,22 +1,35 @@
# Copyright (C) 2025 Intel Corporation
+# ==============================================================================
+# IMPORTS
+import ctypes
+import gc
+import inspect
+import logging
import os
import queue
import subprocess
+import sys
+import threading
import time
import traceback
+import tracemalloc
+from collections import deque
from dataclasses import dataclass
from math import ceil
+from multiprocessing import resource_tracker
from pathlib import Path
from random import randint
import cupy
-import cupyx.scipy
-import cupyx.scipy.ndimage
import cv2
import numpy as np
import torch
-import torch.nn.functional as F
+from pydantic import BaseModel
+
+import vdms
+
+sys.path.insert(1, str(Path(__file__).parent.parent))
from include.default_configs import (
BKGD_SUB_INCLUDE_HISTORY,
BKGD_SUB_INCLUDE_HISTORY_DILATE_KERNEL_SIZE,
@@ -40,6 +53,7 @@
DISPLAY_FRAME_SIZE,
ENABLE_QUERYING_DEFAULT,
INGESTION_DEFAULT,
+ IOU_THRESHOLD_DEFAULT,
MAX_DETECTIONS,
MAX_WORKERS,
MODEL_H,
@@ -67,15 +81,123 @@
UDF_HOST_DEFAULT,
UDF_PORT_DEFAULT,
)
-from pydantic import BaseModel
-import vdms
+# ==============================================================================
+# LOGGING
+logging.basicConfig(
+ level=logging.INFO,
+ # format="%(asctime)s - %(name)s - %(levelname)s - %(message)s",
+ format="%(asctime)s [%(levelname)s] %(name)s (%(filename)s:%(lineno)d) - %(message)s",
+ handlers=[logging.StreamHandler(sys.stdout)],
+)
+
+main_app_logger = logging.getLogger(__name__)
+
+# ==============================================================================
"""
GENERAL DEFINITIONS/FUNCTIONS
"""
+def get_cpu_partition(num_ffmpeg_cores: int = 4):
+ """
+ Partitions available container cores into FFmpeg cores and Main App cores.
+ """
+ available_cores = sorted(list(os.sched_getaffinity(0)))
+ total_cores = len(available_cores)
+
+ if total_cores <= num_ffmpeg_cores:
+ # Fallback if the container has <= 4 cores allocated
+ ffmpeg_cores = available_cores
+ main_app_cores = available_cores
+ else:
+ # Take the first 4 cores for FFmpeg, reserve the rest for the main app
+ ffmpeg_cores = available_cores[:num_ffmpeg_cores]
+ main_app_cores = available_cores[num_ffmpeg_cores:]
+
+ return ffmpeg_cores, main_app_cores
+
+
+def safe_unregister_shm(shm_name: str):
+ """
+ Unregisters a shared memory block from Python's resource tracker
+ using both normalized and slash-prefixed formats.
+ """
+ clean_name = shm_name.lstrip("/")
+ slash_name = f"/{clean_name}"
+
+ for name in (clean_name, slash_name):
+ try:
+ resource_tracker.unregister(name, "shared_memory")
+ except (KeyError, ValueError, AttributeError, Exception):
+ pass
+
+
+def release_shared_memory(list_of_shms):
+ for shm in list_of_shms:
+ if shm is not None:
+ try:
+ # Release buffer view if open
+ if hasattr(shm, "buf") and shm.buf is not None:
+ shm.buf.release()
+ except Exception:
+ pass
+
+ try:
+ shm.close()
+ except Exception:
+ pass
+
+ try:
+ shm.unlink()
+ except (FileNotFoundError, AttributeError, OSError):
+ pass
+
+ # Clear the container list/collection in-place
+ if hasattr(list_of_shms, "clear"):
+ try:
+ list_of_shms.clear()
+ except Exception:
+ pass
+
+
+def release_native_linux_heap():
+ gc.collect()
+ try:
+ # Load the native C library bindings mapping standard glibc
+ libc = ctypes.CDLL("libc.so.6")
+
+ # malloc_trim(0) forces glibc to completely pack and return all
+ # completely disconnected heap memory boundaries back to the kernel.
+ libc.malloc_trim(0)
+ # print(
+ # "[MALLOC_TRIM] Successfully flushed unmanaged C++ heap to OS.", flush=True
+ # )
+ except Exception as e:
+ print(
+ f"[MALLOC_TRIM] Native system trim execution pass failed: {e}", flush=True
+ )
+
+
+def install_and_load_pip_package(package_name: str, attribute_name=None):
+ import importlib
+
+ name = package_name.split("[")[0]
+ try:
+ # import py_package
+ module = importlib.import_module(name)
+ except ImportError:
+ print(f"{name} package not found. Installing ...")
+ subprocess.check_call([sys.executable, "-m", "pip", "install", package_name])
+ module = importlib.import_module(name)
+
+ if attribute_name:
+ return getattr(module, attribute_name)
+
+ return module
+
+
def get_freest_gpu():
# Queries free memory from nvidia-smi
command = "nvidia-smi --query-gpu=memory.free --format=csv,nounits,noheader"
@@ -110,12 +232,494 @@ def str2bool(in_val):
return False
+def global_frame_prefetch_worker_v1(instance):
+ """
+ Asynchronous Threaded Staging Engine.
+ Leverages thread-level parallelism to interleave heavy I/O file latency
+ completely outside the primary model execution track.
+ """
+ pool_maxsize = (
+ instance.prefetch_queue.maxsize
+ if hasattr(instance.prefetch_queue, "maxsize")
+ else instance.prefetch_queue._maxsize
+ )
+
+ # This ensures no frames are read until the main loop is ready.
+ # if hasattr(instance, "start_processing_event"):
+ # instance.start_processing_event.wait()
+
+ while instance.active and instance.prefetch_active:
+ try:
+ # if instance.prefetch_queue is None:
+ # instance.prefetch_active = False
+ # break
+ # 1. Capture time immediately before hardware decryption/extraction
+ decode_start_time = time.perf_counter()
+
+ # Step 1: Overlap video decoding and frame I/O reads.
+ with instance.reader_lock:
+ payload = instance.reader.read()
+
+ # if payload[1] is None:
+ # # print("[PREFETCH WORKER] Reader returned None, exiting.")
+ # # break
+ # time.sleep(0.005) # Sleep 5ms and retry
+ # continue
+
+ ret, frame_8k, current_event, frame_num, abs_frame_num = payload
+
+ # 2. Isolate raw decompression latency (completely independent of consumer queues)
+ true_read_latency_secs = time.perf_counter() - decode_start_time # * 1000.0
+
+ if not ret or frame_8k is None:
+ # # If RTSP, wait briefly to see if it's just a network delay.
+ # if instance.active and getattr(instance, "is_rtsp", True):
+ # time.sleep(0.005)
+ # continue
+ # else:
+ # print("[PREFETCH WORKER] End of stream or invalid frame, exiting.")
+ # break
+
+ # Allow retries if the reader is still active and has not officially stopped
+ if instance.active and not getattr(instance.reader, "stopped", False):
+ time.sleep(0.005)
+ continue
+ else:
+ main_app_logger.info(
+ "[PREFETCH WORKER] End of stream detected, exiting."
+ )
+ break
+
+ if frame_num == 0:
+ main_app_logger.info(
+ f"[VERIFY - PRODUCER] Decoded Frame {frame_num} (abs: {abs_frame_num}) and writing to slot {frame_num % pool_maxsize}!"
+ )
+
+ # if ret is True and frame_8k is not None:
+ slot_idx = frame_num % pool_maxsize
+
+ # if torch.is_tensor(frame_8k):
+ # safe_frame = frame_8k.clone()
+ # elif isinstance(frame_8k, np.ndarray):
+ # safe_frame = frame_8k.copy()
+ # else:
+ # safe_frame = frame_8k
+
+ # Convert to tensor
+ # Make a safe copy of the raw frame data
+ # if isinstance(frame_8k, np.ndarray):
+ # safe_frame = frame_8k.copy()
+ # else:
+ # safe_frame = frame_8k.clone() if torch.is_tensor(frame_8k) else frame_8k
+
+ # safe_frame = instance.shm_buffer_pool[slot_idx][0]
+
+ # # 2. OVERWRITE IN-PLACE (ZERO ALLOCATIONS!)
+ # if isinstance(frame_8k, np.ndarray):
+ # safe_frame.copy_(torch.from_numpy(frame_8k))
+ # else:
+ # safe_frame.copy_(frame_8k)
+
+ # If frame_8k is already on the GPU (from GPUReader), use it directly!
+ if isinstance(frame_8k, np.ndarray):
+ # CPU fallback: copy to GPU input slot
+ instance.gpu_input[slot_idx].copy_(
+ torch.from_numpy(frame_8k), non_blocking=True
+ )
+ safe_frame = instance.gpu_input[slot_idx]
+ else:
+ # Already on GPU: forward directly with zero extra transfers
+ safe_frame = frame_8k
+
+ # safe_frame = frame_8k
+ # Explicitly destroy the original shared-memory views right here!
+ # del payload, frame_8k # , tensor_frame
+ # print(f"[PREFETCH WORKER] frame_num: {frame_num}\tabs_frame_num: {abs_frame_num}", flush=True)
+
+ # Step 2: Directly write to array pool within the same CUDA context.
+ # 3. EXPAND POOL TUPLE: Append the true background latency value to the array matrix
+ instance.shm_buffer_pool[slot_idx] = (
+ safe_frame,
+ current_event,
+ frame_num,
+ abs_frame_num,
+ true_read_latency_secs, # Decoupled payload metric channel
+ )
+ # del payload, frame_8k
+
+ if hasattr(instance, "queue_data_ready_event"):
+ instance.queue_data_ready_event.set()
+
+ # Step 3: Fast notification down a standard thread-safe queue.
+ # Wakes up the consumer loop in microseconds via low-level OS variables.
+ try:
+ instance.prefetch_queue.put((True, slot_idx), block=True)
+ # except AttributeError:
+ # instance.prefetch_active = False
+ # break
+ except Exception:
+ pass
+ finally:
+ del payload, frame_8k
+ except Exception as e:
+ # Yield minimal execution slicing if an index collision occurs
+ # time.sleep(0.001)
+ # CRITICAL: Print the actual error instead of swallowing it.
+ print(f"[PREFETCH WORKER CRASH] An error occurred: {e}")
+ traceback.print_exc()
+ # Stop the loop on error to prevent silent failures.
+ raise
+
+ # Pipeline breakdown tracker
+ # with instance.worker_tracking_lock:
+ # instance.active_workers_count -= 1
+ # if instance.active_workers_count == 0:
+ # # instance.prefetch_queue.put((False, "END_OF_STREAM"), block=True)
+ # main_app_logger.info(
+ # "[PREFETCH WORKER] All workers finished. Sending END_OF_STREAM."
+ # )
+ # if (
+ # hasattr(instance, "prefetch_queue")
+ # and instance.prefetch_queue is not None
+ # ):
+ # try:
+ # instance.prefetch_queue.put(
+ # (False, "END_OF_STREAM"), block=True, timeout=1.0
+ # )
+ # except Exception:
+ # pass
+ lock_obj = getattr(instance, "worker_tracking_lock", None)
+
+ if lock_obj is not None and hasattr(lock_obj, "__enter__"):
+ with lock_obj:
+ instance.active_workers_count -= 1
+ is_last_worker = instance.active_workers_count == 0
+ else:
+ # Fallback in case the main thread already destroyed the lock during shutdown
+ # (Safe to decrement directly since threads are joining anyway)
+ try:
+ instance.active_workers_count -= 1
+ is_last_worker = instance.active_workers_count == 0
+ except Exception:
+ is_last_worker = False
+
+ # Send END_OF_STREAM if this is the last worker exiting
+ if is_last_worker:
+ main_app_logger.info(
+ "[PREFETCH WORKER] All workers finished. Sending END_OF_STREAM."
+ )
+ if hasattr(instance, "prefetch_queue") and instance.prefetch_queue is not None:
+ try:
+ instance.prefetch_queue.put(
+ (False, "END_OF_STREAM"), block=True, timeout=1.0
+ )
+ except Exception:
+ pass
+
+
+def analyze_tracemalloc_snapshot():
+ """
+ Take and analyze tracemalloc snapshot
+ """
+ system_exclusions = [
+ "importlib",
+ "runpy",
+ "pydev",
+ "typing",
+ "abc",
+ "contextlib",
+ "unittest",
+ "pytest",
+ ]
+
+ main_app_logger.info("=" * 60)
+ # Take snapshot with tracking data enabled
+ snapshot = tracemalloc.take_snapshot()
+ tracemalloc.stop()
+ stats = snapshot.statistics("traceback")
+
+ main_app_logger.info(
+ f"{'ALLOCATED SIZE':<15} | {'OBJECT TYPE':<20} | {'SOURCE LOCATION'}",
+ )
+ main_app_logger.info("-" * 80)
+
+ stat_cnt = 0
+ for stat in stats: # [:10]: # Check the top 10 largest leaks
+ if stat_cnt >= 10:
+ break
+
+ first_frame = stat.traceback[0]
+ source_info = f"{first_frame.filename.split('/')[-1]}:{first_frame.lineno}"
+
+ if any(exc in source_info for exc in system_exclusions):
+ continue
+
+ # Extract the trace block memory identity
+ obj_type_name = "Unknown / Raw C Block"
+
+ # Query the raw trace details to extract live block structures
+ for trace in stat.traceback:
+ # We look for objects instantiated at this file/line in the heap
+ for obj in gc.get_objects():
+ try:
+ # Cross-reference: does this object's line history match the stat line?
+ # Python doesn't save creation history on every object, so we inspect properties:
+ if torch.is_tensor(obj):
+ # Tensors are easily tracked if they have shapes or match your sizes
+ if (
+ obj.is_cuda
+ and (obj.element_size() * obj.nelement()) == stat.size
+ ):
+ obj_type_name = f"torch.Tensor (Shape: {list(obj.shape)})"
+ break
+ elif inspect.isframe(obj):
+ obj_type_name = "Live Frame Context"
+ break
+ elif inspect.isfunction(obj) or inspect.ismethod(obj):
+ obj_type_name = f"Function ({obj.__name__})"
+ break
+ elif isinstance(obj, dict):
+ # Dictionary allocations match structure profiles
+ try:
+ # Guard checks against mutating async dictionary managers
+ if (
+ sys.getsizeof(obj) == stat.size
+ or len(obj) == stat.count
+ ):
+ obj_type_name = "dict Namespace"
+ break
+ except Exception:
+ pass
+ except Exception:
+ pass
+
+ obj = None
+ # Print the compiled breakdown
+
+ main_app_logger.info(
+ f"{stat.size / 1024:<11.1f} KiB | {obj_type_name:<20} | {source_info}"
+ )
+ stat_cnt += 1
+
+
+# from concurrent.futures import ThreadPoolExecutor
+
+
+# def global_frame_prefetch_worker(instance):
+# """GIL-free asynchronous staging engine utilizing a dedicated single-worker pool
+# to execute raw C++ reads completely outside the primary interpreter track.
+# """
+# pool_maxsize = (
+# instance.prefetch_queue.maxsize
+# if hasattr(instance.prefetch_queue, "maxsize")
+# else instance.prefetch_queue._maxsize
+# )
+
+# # A dedicated single-worker context specifically for C++ operations
+# # with ThreadPoolExecutor(max_workers=1) as reader_executor:
+# while instance.active and instance.prefetch_active:
+# try:
+# # 1. Capture time immediately before hardware decryption/extraction
+# decode_start_time = time.perf_counter()
+
+# # Offload ONLY the unmanaged C++ read() call to the executor.
+# # This drops the GIL instantly while the driver retrieves the frame bytes.
+# # future = reader_executor.submit(instance.reader.read)
+# # payload = (
+# # future.result()
+# # ) # Blocking wait until the background C++ task completes
+
+# # Explicitly overwrite the internal future reference
+# # tracking properties to break the C++ object data caching loop immediately
+# # if hasattr(future, "_result"):
+# # future._result = None # Breaks the structural reference hold layout
+# # future = None # Drops local scope tracking handles
+
+# # Overlap video decoding and frame I/O reads natively.
+# # The GIL is dropped directly inside the compiled C++ cv2.VideoCapture layer.
+# with instance.reader_lock:
+# payload = instance.reader.read()
+
+# # Isolate raw decompression latency (completely independent of consumer queues)
+# true_read_latency_sec = (
+# time.perf_counter() - decode_start_time
+# )
+
+# if payload[0] is False:
+# break
+
+# ret, frame_8k, current_event, frame_num, abs_frame_num = payload
+
+
+# if ret is True and frame_8k is not None:
+# slot_idx = frame_num % pool_maxsize
+# if torch.is_tensor(frame_8k):
+# safe_frame = frame_8k.clone()
+# elif isinstance(frame_8k, np.ndarray):
+# safe_frame = frame_8k.copy()
+# else:
+# safe_frame = frame_8k
+
+# # Explicitly destroy the original shared-memory views right here!
+# del payload, frame_8k
+
+# print(f"[PREFETCH WORKER] frame_num: {frame_num}\tabs_frame_num: {abs_frame_num}", flush=True)
+# # Directly write to array pool within the same CUDA context.
+# # EXPAND POOL TUPLE: Append the true background latency value to the array matrix
+# with instance.worker_tracking_lock:
+# instance.shm_buffer_pool[slot_idx] = (
+# safe_frame,
+# current_event,
+# frame_num,
+# abs_frame_num,
+# true_read_latency_sec, # Decoupled payload metric channel
+# )
+
+# if hasattr(instance, "queue_data_ready_event"):
+# instance.queue_data_ready_event.set()
+
+# # Step 3: Fast notification down a standard thread-safe queue.
+# # Wakes up the consumer loop in microseconds via low-level OS variables.
+# instance.prefetch_queue.put((True, slot_idx), block=True)
+# else:
+# # break
+# time.sleep(0.001)
+# continue
+# except Exception:
+# # Yield minimal execution slicing if an index collision occurs
+# time.sleep(0.002)
+# continue
+
+# # Pipeline breakdown tracker
+# # with instance.worker_tracking_lock:
+# # instance.active_workers_count -= 1
+# # if instance.active_workers_count == 0:
+# # instance.prefetch_queue.put((False, "END_OF_STREAM"), block=True)
+
+
+def global_frame_prefetch_worker_process(
+ active, # mp.Value flag
+ prefetch_active, # mp.Value flag
+ prefetch_queue, # mp.Queue mapping descriptor
+ reader_lock, # mp.Lock mutex channel
+ reader, # Picklable reader baseline layout
+ shm_buffer_pool, # Shared pointer memory pool array matrix
+ queue_data_ready_event, # Pass the shared multiprocessing Event channel
+ worker_tracking_lock, # Pass the shared multiprocessing Lock mutex
+ active_workers_count, # Pass the shared atomic integer context
+):
+ """
+ Asynchronous Threaded Staging Engine.
+ Leverages thread-level parallelism to interleave heavy I/O file latency
+ completely outside the primary model execution track.
+ """
+ pool_maxsize = (
+ prefetch_queue.maxsize
+ if hasattr(prefetch_queue, "maxsize")
+ else prefetch_queue._maxsize
+ )
+
+ try:
+ while active.value and prefetch_active.value:
+ try:
+ # 1. Capture time immediately before hardware decryption/extraction
+ decode_start_time = time.perf_counter()
+
+ # Step 1: Overlap video decoding and frame I/O reads.
+ with reader_lock:
+ payload = reader.read()
+
+ # 2. Isolate raw decompression latency (completely independent of consumer queues)
+ true_read_latency_ms = (
+ time.perf_counter() - decode_start_time
+ ) * 1000.0
+
+ if payload is None:
+ break
+
+ ret, frame_8k, current_event, frame_num, abs_frame_num = payload
+
+ if ret is True and frame_8k is not None:
+ slot_idx = frame_num % pool_maxsize
+
+ if torch.is_tensor(frame_8k):
+ safe_frame = frame_8k.clone()
+ elif isinstance(frame_8k, np.ndarray):
+ safe_frame = frame_8k.copy()
+ else:
+ safe_frame = frame_8k
+
+ # Explicitly destroy the original shared-memory views right here!
+ del payload, frame_8k
+
+ # Step 2: Directly write to array pool within the same CUDA context.
+ # 3. EXPAND POOL TUPLE: Append the true background latency value to the array matrix
+ shm_buffer_pool[slot_idx] = (
+ safe_frame,
+ current_event,
+ frame_num,
+ abs_frame_num,
+ true_read_latency_ms, # Decoupled payload metric channel
+ )
+
+ queue_data_ready_event.set()
+
+ prefetch_queue.put((True, slot_idx), block=True)
+ else:
+ break
+ except Exception:
+ # Yield minimal execution slicing if an index collision occurs
+ time.sleep(0.001)
+
+ finally:
+ # Pipeline breakdown tracker
+ with worker_tracking_lock:
+ active_workers_count.value -= 1
+ if active_workers_count.value == 0:
+ prefetch_queue.put((False, "END_OF_STREAM"), block=True)
+
+
PROJECT_PATH = Path(__file__).parent.parent
DEBUG_FLAG_DEFAULT = True if DEBUG_DEFAULT == "1" else False
LOCKTIMEOUT_RETRIES = 5
+default_attr_keys = [
+ "_is_stopped",
+ "_stop_lock",
+ "_testMethodName",
+ "abs_frame_num",
+ "active_workers_count",
+ "active",
+ "baseline_before_start",
+ "component_stats",
+ "config",
+ "crops_per_frame_list",
+ "device_input",
+ "device",
+ "disp_h",
+ "disp_w",
+ "duration_target",
+ "elapsed_display_time",
+ "frame_count_target",
+ "frame_count",
+ "max_target_frames",
+ "output_path",
+ "pcie_throughput_gbps",
+ "prefetch_active",
+ "reader",
+ "resize_h",
+ "resize_w",
+ "stat_fps",
+ "stat_frame_count",
+ "status",
+ "VIDEO_GT_DETAILS",
+ "video_output_name",
+ "vram_efficiency",
+ "worker_tracking_lock",
+]
+
class PipelineConfig:
def __init__(self, **kwargs):
@@ -159,6 +763,7 @@ def __init__(self, **kwargs):
self.DETECTION_THRESHOLD = float(
kwargs.get("DETECTION_THRESHOLD", DETECTION_THRESHOLD_DEFAULT)
)
+ self.IOU_THRESHOLD = float(kwargs.get("IOU_THRESHOLD", IOU_THRESHOLD_DEFAULT))
self.MAX_DETECTIONS = int(kwargs.get("MAX_DETECTIONS", MAX_DETECTIONS))
self.MODEL_H = int(kwargs.get("MODEL_H", MODEL_H))
self.MODEL_W = int(kwargs.get("MODEL_W", MODEL_W))
@@ -264,7 +869,9 @@ def _create_connection(self):
def get_connection(self):
# Borrow a connection (blocks if pool is empty)
- return self.pool.get(block=True, timeout=10)
+ db = self.pool.get(block=True, timeout=10)
+ self.pool.task_done()
+ return db
def return_connection(self, conn):
# Put the connection back for reuse
@@ -513,6 +1120,121 @@ def return_connection(self, conn):
)
+PROPAGATION_KERNEL_CODE = r"""
+extern "C" __global__
+void get_row_bounds_fused(const unsigned char* mask, int pitch, int w, int h,
+ int* x1, int* y1, int* x2, int* y2, int* num_labels) {
+ // Each thread tracks exactly one horizontal row across the frame canvas
+ int y = blockIdx.y * blockDim.y + threadIdx.y;
+
+ if (y < h) {
+ bool row_has_motion = false;
+ int min_x = w;
+ int max_x = -1;
+
+ // Perform a fast linear register sweep across the row (0 global atomics)
+ for (int x = 0; x < w; ++x) {
+ unsigned char pixel = mask[y * pitch + x];
+ if (pixel > 0) {
+ row_has_motion = true;
+ if (x < min_x) min_x = x;
+ if (x > max_x) max_x = x;
+ }
+ }
+
+ // If the row contains motion, commit the boundaries to global memory
+ if (row_has_motion) {
+ // Use an atomic index fetch to smoothly assign contiguous tracking labels
+ int label = atomicAdd(num_labels, 1);
+
+ // Limit to our safe allocation pool ceiling to prevent memory overflows
+ if (label < 256) {
+ x1[label] = min_x;
+ y1[label] = y;
+ x2[label] = max_x + 1;
+ y2[label] = y + 1;
+ }
+ }
+ }
+}
+"""
+
+
+def get_metadata_overlay(
+ display_frame, metadata_or_bbs, class_list, scale_display, disp_size, is_bgr=True
+):
+ scale_display_x, scale_display_y = scale_display
+ disp_w, disp_h = disp_size
+ for frame_str, obj in metadata_or_bbs.items():
+ bbox = obj["bbox"]
+ raw_x = bbox["x"] * scale_display_x
+ raw_y = bbox["y"] * scale_display_y
+ w = bbox["width"] * scale_display_x
+ h = bbox["height"] * scale_display_y
+ x = max(0, int(raw_x))
+ y = max(0, int(raw_y))
+ x2 = min(disp_w - 1, int(raw_x + w))
+ y2 = min(disp_h - 1, int(raw_y + h))
+
+ class_name = bbox["object"]
+ class_id = class_list.index(class_name) if class_name in class_list else 0
+ bb_color = get_detection_color(class_id, is_bgr=is_bgr)
+
+ # print(f'{frame_str} xyxy: ', (x, y), (x2, y2), flush=True)
+ display_frame = cv2.rectangle(display_frame, (x, y), (x2, y2), bb_color, 2)
+
+ if class_name != "":
+ confidence = bbox.get("object_det", {}).get("confidence", 0.0)
+ label = f"{class_name} {confidence:.2f}"
+ draw_label(display_frame, label, (x, y), color=bb_color, padding=5)
+ return display_frame
+
+
+def get_bb_overlay(
+ display_frame, metadata_or_bbs, scale_display, disp_size, color=(0, 0, 255)
+):
+ """
+ Ultra-low latency canvas renderer. Optimized to preserve high FPS.
+ """
+ # Instant escape route for blank inference sequences
+ if metadata_or_bbs is None or len(metadata_or_bbs) == 0:
+ return display_frame
+
+ scale_display_x, scale_display_y = scale_display
+ disp_w, disp_h = disp_size
+
+ # Direct Type Extraction: Handle specific incoming payloads instantly
+ if isinstance(metadata_or_bbs, list):
+ # Acknowledge your track array: Fast extraction of structural list layers
+ if isinstance(metadata_or_bbs[0], dict):
+ boxes = np.array([b["bbox"] for b in metadata_or_bbs], dtype=np.float32)
+ else:
+ boxes = np.array(metadata_or_bbs, dtype=np.float32)
+ elif isinstance(metadata_or_bbs, np.ndarray):
+ boxes = metadata_or_bbs
+ else:
+ # Fallback mechanism handles raw GPU traces securely if passed downstream
+ boxes = metadata_or_bbs.detach().cpu().numpy()
+
+ # Pre-allocate coordinates in a single vectorized sweep
+ x1 = (boxes[:, 0] * scale_display_x).astype(np.int32)
+ y1 = (boxes[:, 1] * scale_display_y).astype(np.int32)
+ x2 = (boxes[:, 2] * scale_display_x).astype(np.int32)
+ y2 = (boxes[:, 3] * scale_display_y).astype(np.int32)
+
+ # Perform low-level C++ drawing inside OpenCV (Zero typing penalties)
+ for i in range(len(boxes)):
+ # Clip bounding box corners safely within resolution bounds
+ rx1 = max(0, x1[i])
+ ry1 = max(0, y1[i])
+ rx2 = min(disp_w - 1, x2[i])
+ ry2 = min(disp_h - 1, y2[i])
+
+ cv2.rectangle(display_frame, (rx1, ry1), (rx2, ry2), color, 2)
+
+ return display_frame
+
+
def tensor2opencv(frame_source, device_input, is_bgr=True, resize_h=640, resize_w=640):
if torch.is_tensor(frame_source):
# .contiguous() is CRITICAL here to fix the "shredded" look
@@ -553,7 +1275,7 @@ def tensor2opencv(frame_source, device_input, is_bgr=True, resize_h=640, resize_
return img_cpu
-def gpumat2cupy(gpu_mat):
+def gpumat2cupyv1(gpu_mat):
"""Bridge OpenCV GpuMat to CuPy without copying data."""
# Get properties from GpuMat
w, h = gpu_mat.size()
@@ -592,6 +1314,28 @@ class Holder:
return cupy.asarray(holder)
+def gpumat2cupy(gpu_mat):
+ """
+ Bridge OpenCV GpuMat to CuPy instantly without copying data or parsing dict envelopes.
+ """
+ w, h = gpu_mat.size()
+ channels = 3 if gpu_mat.type() == cv2.CV_8UC3 else 1
+
+ if channels == 3:
+ shape = (h, w, 3)
+ strides = (gpu_mat.step, 3, 1)
+ else:
+ shape = (h, w)
+ strides = (gpu_mat.step, 1)
+
+ # OPTIMIZATION: Construct a native CuPy array memory wrapper instantly over the raw C++ pointer.
+ # This completely bypasses the Python dictionary interpreter parsing and 'cupy.asarray' overhead.
+ mem = cupy.cuda.UnownedMemory(gpu_mat.cudaPtr(), gpu_mat.step * h, gpu_mat)
+ mptr = cupy.cuda.MemoryPointer(mem, 0)
+
+ return cupy.ndarray(shape=shape, dtype=cupy.uint8, memptr=mptr, strides=strides)
+
+
def torch2gpumat(tensor):
"""
Creates an OpenCV GpuMat pointing to the same memory as a PyTorch tensor.
@@ -646,68 +1390,919 @@ def torch2gpumat(tensor):
# Compile the kernel once
get_bounds_kernel = cupy.RawKernel(BOUNDS_KERNEL_CODE, "get_bounds")
+# BOUNDS_KERNEL_CODE = r"""
+# extern "C" __global__
+# void get_bounds_pure(const int* labeled, int pitch_elements, int w, int h, int num_labels,
+# int* x1, int* y1, int* x2, int* y2) {
+# int x = blockIdx.x * blockDim.x + threadIdx.x;
+# int y = blockIdx.y * blockDim.y + threadIdx.y;
+
+# if (x < w && y < h) {
+# // Safe 32-bit element pitch extraction
+# int label = labeled[y * pitch_elements + x];
+
+# if (label > 0 && label <= num_labels) {
+# atomicMin(&x1[label], x);
+# atomicMin(&y1[label], y);
+# atomicMax(&x2[label], x + 1);
+# atomicMax(&y2[label], y + 1);
+# }
+# }
+# }
+# """
+# get_bounds_kernel = cupy.RawKernel(BOUNDS_KERNEL_CODE, "get_bounds_pure")
+
+
+# def merge_boxes_gpuv1(raw_boxes, gap_limit=10, size_limit=1000):
+# """
+# Refined Parallel Merger with Size Constraints.
+# Prevents merges that would create boxes larger than size_limit.
+# """
+# if raw_boxes.shape[0] <= 1:
+# return raw_boxes
+
+# x1, y1, x2, y2 = raw_boxes.unbind(1)
+
+# # 1. Calculate pairwise gaps (Existing logic)
+# h_gaps = torch.max(
+# torch.zeros(1, device=raw_boxes.device),
+# torch.max(x1.unsqueeze(0) - x2.unsqueeze(1), x1.unsqueeze(1) - x2.unsqueeze(0)),
+# )
+# v_gaps = torch.max(
+# torch.zeros(1, device=raw_boxes.device),
+# torch.max(y1.unsqueeze(0) - y2.unsqueeze(1), y1.unsqueeze(1) - y2.unsqueeze(0)),
+# )
+
+# # 2. NEW: Calculate potential union dimensions for ALL pairs [N, N]
+# # We find the min/max coordinates if box i and box j were merged
+# union_x1 = torch.min(x1.unsqueeze(0), x1.unsqueeze(1))
+# union_y1 = torch.min(y1.unsqueeze(0), y1.unsqueeze(1))
+# union_x2 = torch.max(x2.unsqueeze(0), x2.unsqueeze(1))
+# union_y2 = torch.max(y2.unsqueeze(0), y2.unsqueeze(1))
+
+# union_w = union_x2 - union_x1
+# union_h = union_y2 - union_y1
+
+# # 3. ADJACENCY MASK: Must be close AND the result must be under the limit
+# # This prevents the creation of massive "megaboxes"
+# adj = (
+# (h_gaps < gap_limit)
+# & (v_gaps < gap_limit)
+# & (union_w < size_limit)
+# & (union_h < size_limit)
+# )
+
+# # 4. Parallel Connected Components (Existing logic)
+# components = torch.arange(raw_boxes.shape[0], device=raw_boxes.device)
+# r = 3
+# for _ in range(r):
+# components = torch.max(adj * components, dim=1).values
+
+# unique_ids = components.unique()
+# merged = []
+# for i in unique_ids:
+# mask = components == i
+# merged.append(
+# torch.cat(
+# [raw_boxes[mask, :2].min(0).values, raw_boxes[mask, 2:].max(0).values]
+# )
+# )
+
+# return torch.stack(merged)
+
+
+# last
+def merge_boxes_gpu(raw_boxes, gap_limit=10, size_limit=1280, max_cached_elements=100):
+ """
+ Refined Parallel Merger utilizing static function-attached scratchpads.
+ Maintains a 100% linear VRAM profile and eliminates dynamic allocations.
+ """
+ if raw_boxes.shape[0] <= 1:
+ return raw_boxes
+
+ # 1. ROBUST CACHE CHECK: Verify shape dimension bounds to prevent 1D flat layout leakage
+ if (
+ not hasattr(merge_boxes_gpu, "adj_matrix")
+ or merge_boxes_gpu.adj_matrix.ndim != 2
+ ):
+ merge_boxes_gpu.adj_matrix = torch.zeros(
+ (max_cached_elements, max_cached_elements),
+ dtype=torch.bool,
+ device=raw_boxes.device,
+ )
+ merge_boxes_gpu.components = torch.zeros(
+ (max_cached_elements,), dtype=torch.long, device=raw_boxes.device
+ )
+ merge_boxes_gpu.scratch_out = torch.zeros(
+ (max_cached_elements, 4), dtype=torch.float32, device=raw_boxes.device
+ )
+
+ # Establish safe spatial clip windows depending on current frame tracking load
+ N = min(raw_boxes.shape[0], max_cached_elements)
-def merge_boxes_gpu(raw_boxes, gap_limit=10, size_limit=1000):
+ # 2. VECTORIZED ARITHMETIC WITH VIEWS: Completely replace unbind() and unsqueeze() list loops
+ x1 = raw_boxes[:N, 0]
+ y1 = raw_boxes[:N, 1]
+ x2 = raw_boxes[:N, 2]
+ y2 = raw_boxes[:N, 3]
+
+ # Compute gaps smoothly using in-place operations over zero-copy memory layouts
+ h_gaps = torch.clamp(
+ torch.max(x1.view(N, 1) - x2.view(1, N), x1.view(1, N) - x2.view(N, 1)), min=0
+ )
+ v_gaps = torch.clamp(
+ torch.max(y1.view(N, 1) - y2.view(1, N), y1.view(1, N) - y2.view(N, 1)), min=0
+ )
+
+ # Map target union envelopes natively
+ union_w = torch.max(x2.view(N, 1), x2.view(1, N)) - torch.min(
+ x1.view(N, 1), x1.view(1, N)
+ )
+ union_h = torch.max(y2.view(N, 1), y2.view(1, N)) - torch.min(
+ y1.view(N, 1), y1.view(1, N)
+ )
+
+ # Overwrite adjacency bounds directly into the pre-allocated cache slice
+ adj = merge_boxes_gpu.adj_matrix[:N, :N]
+ adj.copy_(
+ (h_gaps < gap_limit)
+ & (v_gaps < gap_limit)
+ & (union_w < size_limit)
+ & (union_h < size_limit)
+ )
+
+ # 3. ACCELERATED CONNECTED COMPONENTS: Run pointer rotations in place
+ comp = merge_boxes_gpu.components[:N]
+ torch.arange(N, device=raw_boxes.device, out=comp)
+
+ # Unified logical components compression step
+ for _ in range(3):
+ comp.copy_(torch.max(adj * comp, dim=1).values)
+
+ # unique_ids = comp.unique()
+ # num_merged = unique_ids.shape[0]
+
+ # # 4. STATIC MEMORY AGGREGATION: Eliminate the loops appending torch.cat() arrays
+ # out_buffer = merge_boxes_gpu.scratch_out[:num_merged]
+
+ # for idx, i in enumerate(unique_ids):
+ # mask = comp == i
+ # boxes_subset = raw_boxes[:N][mask]
+
+ # # Write structural outputs directly into our static memory block channels
+ # out_buffer[idx, 0] = boxes_subset[:, 0].min()
+ # out_buffer[idx, 1] = boxes_subset[:, 1].min()
+ # out_buffer[idx, 2] = boxes_subset[:, 2].max()
+ # out_buffer[idx, 3] = boxes_subset[:, 3].max()
+
+ # # Return the clean floating-point tensor slice natively
+ # return out_buffer.clone()
+
+ # 1. Initialize an allocation-free O(1) bitmask directly over device registers
+ # N is already our maximum candidate dimension cap mapped for this frame
+ valid_bitmask = torch.zeros((N,), dtype=torch.bool, device=raw_boxes.device)
+
+ # 2. Flag every single active logical group ID index in parallel in a single CUDA command
+ valid_bitmask[comp] = True
+
+ # 3. Pull unique indices instantly using a fast non-zero memory address pass
+ unique_ids = torch.nonzero(valid_bitmask).squeeze(1)
+ num_merged = unique_ids.shape[0]
+
+ # 4. Map the aggregation workspace straight onto our persistent static cache slice
+ out_buffer = merge_boxes_gpu.scratch_out[:num_merged]
+
+ # 5. Extract bounding coordinates cleanly without dynamic vector list loops
+ for idx, i in enumerate(unique_ids):
+ mask = comp == i
+ boxes_subset = raw_boxes[:N][mask]
+
+ if boxes_subset.ndim == 1:
+ # Reshapes a [4] vector back into a valid [1, 4] 2D matrix canvas
+ boxes_subset = boxes_subset.view(1, -1)
+
+ # Write structural outputs directly into our static memory block channels
+ out_buffer[idx, 0] = boxes_subset[:, 0].min()
+ out_buffer[idx, 1] = boxes_subset[:, 1].min()
+ out_buffer[idx, 2] = boxes_subset[:, 2].max()
+ out_buffer[idx, 3] = boxes_subset[:, 3].max()
+
+ # Return the clean floating-point tensor slice natively
+ return out_buffer.clone()
+
+
+def merge_boxes_gpu_8_25(boxes, gap_limit, size_limit=None, max_cached_elements=100):
"""
- Refined Parallel Merger with Size Constraints.
- Prevents merges that would create boxes larger than size_limit.
+ Optimized GPU box merging using a grid-based aggregation algorithm.
+ This avoids the O(n^2) complexity of pairwise distance calculations, making it
+ ideal for high-density scenarios with hundreds or thousands of boxes.
+
+ Args:
+ boxes (torch.Tensor): A tensor of shape (N, 4) with boxes [x1, y1, x2, y2].
+ gap_limit (float): The cell size for the grid. Boxes within the same
+ cell will be merged.
+ size_limit (float, optional): Not used in this version but kept for API compatibility.
+ max_cached_elements (int, optional): Not used but kept for API compatibility.
+
+ Returns:
+ torch.Tensor: A tensor of shape (M, 4) with the merged boxes.
+ """
+ if boxes.shape[0] == 0:
+ return torch.empty((0, 4), device=boxes.device, dtype=boxes.dtype)
+
+ # Assume a 640x640 coordinate space, as this is where the merging happens.
+ # If this changes, these values must be updated.
+ IMAGE_WIDTH = 640
+ IMAGE_HEIGHT = 640
+
+ # --- Grid-Based Aggregation ---
+
+ # 1. Calculate box centers
+ x_centers = (boxes[:, 0] + boxes[:, 2]) / 2.0
+ y_centers = (boxes[:, 1] + boxes[:, 3]) / 2.0
+
+ # 2. Define the grid and assign each box to a grid cell ID
+ grid_w = int(IMAGE_WIDTH / gap_limit) + 1
+
+ # Assign a 1D grid cell index to each box
+ grid_x_indices = (x_centers / gap_limit).long()
+ grid_y_indices = (y_centers / gap_limit).long()
+
+ # Combine 2D grid indices into a single 1D index for scatter_reduce
+ cell_indices = grid_y_indices * grid_w + grid_x_indices
+ num_cells = grid_w * (int(IMAGE_HEIGHT / gap_limit) + 1)
+
+ # 3. Use scatter_reduce to merge boxes in each cell in parallel
+ # We need tensors to hold the min/max coordinates for each cell.
+ # Initialize `merged_x1` and `merged_y1` to a large value.
+ # Initialize `merged_x2` and `merged_y2` to a small value.
+
+ merged_x1 = torch.full(
+ (num_cells,), float("inf"), device=boxes.device, dtype=boxes.dtype
+ )
+ merged_y1 = torch.full(
+ (num_cells,), float("inf"), device=boxes.device, dtype=boxes.dtype
+ )
+ merged_x2 = torch.full(
+ (num_cells,), float("-inf"), device=boxes.device, dtype=boxes.dtype
+ )
+ merged_y2 = torch.full(
+ (num_cells,), float("-inf"), device=boxes.device, dtype=boxes.dtype
+ )
+
+ # Find the min x1 and y1 for all boxes in each cell
+ merged_x1.scatter_reduce_(
+ 0, cell_indices, boxes[:, 0], reduce="amin", include_self=False
+ )
+ merged_y1.scatter_reduce_(
+ 0, cell_indices, boxes[:, 1], reduce="amin", include_self=False
+ )
+
+ # Find the max x2 and y2 for all boxes in each cell
+ merged_x2.scatter_reduce_(
+ 0, cell_indices, boxes[:, 2], reduce="amax", include_self=False
+ )
+ merged_y2.scatter_reduce_(
+ 0, cell_indices, boxes[:, 3], reduce="amax", include_self=False
+ )
+
+ # 4. Filter out the empty cells to get the final merged boxes
+ # A cell is considered populated if its min value is not infinity.
+ valid_cells_mask = merged_x1 != float("inf")
+
+ final_boxes = torch.stack(
+ [
+ merged_x1[valid_cells_mask],
+ merged_y1[valid_cells_mask],
+ merged_x2[valid_cells_mask],
+ merged_y2[valid_cells_mask],
+ ],
+ dim=1,
+ )
+
+ # The grid-based approach might merge boxes that are in the same cell but
+ # not directly adjacent. A second pass on the much smaller set of merged
+ # boxes could be done, but for performance, this initial merge is often sufficient.
+ # For now, we return the direct result of the grid aggregation.
+
+ return final_boxes
+
+
+# last
+def merge_boxes_gpu_v1(
+ raw_boxes, gap_limit=10, size_limit=1000, max_cached_elements=256
+):
+ """
+ Refined Parallel Merger utilizing static function-attached scratchpads.
+ 100% loop-free vectorized tensor reduction matching original accuracy.
"""
if raw_boxes.shape[0] <= 1:
return raw_boxes
- x1, y1, x2, y2 = raw_boxes.unbind(1)
+ # 1. PERSISTENT WORKSPACE CACHE INITIALIZATION
+ if (
+ not hasattr(merge_boxes_gpu, "adj_matrix")
+ or merge_boxes_gpu.adj_matrix.ndim != 2
+ ):
+ merge_boxes_gpu.adj_matrix = torch.zeros(
+ (max_cached_elements, max_cached_elements),
+ dtype=torch.bool,
+ device=raw_boxes.device,
+ )
+ merge_boxes_gpu.components = torch.zeros(
+ (max_cached_elements,), dtype=torch.long, device=raw_boxes.device
+ )
+ merge_boxes_gpu.scratch_out = torch.zeros(
+ (max_cached_elements, 4), dtype=torch.float32, device=raw_boxes.device
+ )
+
+ N = min(raw_boxes.shape[0], max_cached_elements)
+
+ # 2. VECTORIZED ARITHMETIC WITH VIEWS (Your exact matching geometry)
+ x1 = raw_boxes[:N, 0]
+ y1 = raw_boxes[:N, 1]
+ x2 = raw_boxes[:N, 2]
+ y2 = raw_boxes[:N, 3]
- # 1. Calculate pairwise gaps (Existing logic)
- h_gaps = torch.max(
- torch.zeros(1, device=raw_boxes.device),
- torch.max(x1.unsqueeze(0) - x2.unsqueeze(1), x1.unsqueeze(1) - x2.unsqueeze(0)),
+ h_gaps = torch.clamp(
+ torch.max(x1.view(N, 1) - x2.view(1, N), x1.view(1, N) - x2.view(N, 1)), min=0
)
- v_gaps = torch.max(
- torch.zeros(1, device=raw_boxes.device),
- torch.max(y1.unsqueeze(0) - y2.unsqueeze(1), y1.unsqueeze(1) - y2.unsqueeze(0)),
+ v_gaps = torch.clamp(
+ torch.max(y1.view(N, 1) - y2.view(1, N), y1.view(1, N) - y2.view(N, 1)), min=0
)
- # 2. NEW: Calculate potential union dimensions for ALL pairs [N, N]
- # We find the min/max coordinates if box i and box j were merged
- union_x1 = torch.min(x1.unsqueeze(0), x1.unsqueeze(1))
- union_y1 = torch.min(y1.unsqueeze(0), y1.unsqueeze(1))
- union_x2 = torch.max(x2.unsqueeze(0), x2.unsqueeze(1))
- union_y2 = torch.max(y2.unsqueeze(0), y2.unsqueeze(1))
+ union_w = torch.max(x2.view(N, 1), x2.view(1, N)) - torch.min(
+ x1.view(N, 1), x1.view(1, N)
+ )
+ union_h = torch.max(y2.view(N, 1), y2.view(1, N)) - torch.min(
+ y1.view(N, 1), y1.view(1, N)
+ )
- union_w = union_x2 - union_x1
- union_h = union_y2 - union_y1
+ adj = merge_boxes_gpu.adj_matrix[:N, :N]
+ adj.copy_(
+ (h_gaps < gap_limit)
+ & (v_gaps < gap_limit)
+ & (union_w < size_limit)
+ & (union_h < size_limit)
+ )
+
+ # 3. ACCELERATED CONNECTED COMPONENTS
+ comp = merge_boxes_gpu.components[:N]
+ torch.arange(N, device=raw_boxes.device, out=comp)
+
+ # Label propagation iterations
+ for _ in range(3):
+ comp.copy_(torch.max(adj * comp, dim=1).values)
- # 3. ADJACENCY MASK: Must be close AND the result must be under the limit
- # This prevents the creation of massive "megaboxes"
- adj = (
+ # 4. 100% LOOP-FREE VECTORIZED REDUCTION (Replaces lines 102-115)
+ # Remap cluster assignments to continuous indices from 0 to M-1
+ unique_ids, cluster_assignments = torch.unique(comp, return_inverse=True)
+ num_merged = unique_ids.shape[0]
+
+ # Pre-size output buffer view slices
+ out_buffer = merge_boxes_gpu.scratch_out[:num_merged]
+
+ # Vectorized min/max reduction using scatter_reduce
+ # We initialize extreme values to find true minimums and maximums
+ out_buffer[:, 0:2].fill_(float("inf"))
+ # out_buffer[:, 1].fill_(float("inf"))
+ out_buffer[:, 2:4].fill_(float("-inf"))
+ # out_buffer[:, 3].fill_(float("-inf"))
+
+ # Parallel reduction via native C++ PyTorch backend kernels (0ms loop overhead)
+ out_buffer[:, 0].scatter_reduce_(
+ 0, cluster_assignments, x1, reduce="amin", include_self=False
+ )
+ out_buffer[:, 1].scatter_reduce_(
+ 0, cluster_assignments, y1, reduce="amin", include_self=False
+ )
+ out_buffer[:, 2].scatter_reduce_(
+ 0, cluster_assignments, x2, reduce="amax", include_self=False
+ )
+ out_buffer[:, 3].scatter_reduce_(
+ 0, cluster_assignments, y2, reduce="amax", include_self=False
+ )
+
+ # Enforce geometric size limits vectorially
+ widths = out_buffer[:, 2] - out_buffer[:, 0]
+ heights = out_buffer[:, 3] - out_buffer[:, 1]
+
+ width_mask = widths > size_limit
+ height_mask = heights > size_limit
+
+ out_buffer[width_mask, 2] = out_buffer[width_mask, 0] + size_limit
+ out_buffer[height_mask, 3] = out_buffer[height_mask, 1] + size_limit
+
+ return out_buffer.clone()
+
+
+def merge_boxes_gpu_v2(
+ raw_boxes, gap_limit=10, size_limit=1000, max_cached_elements=256
+):
+ """
+ Refined Parallel Merger utilizing static function-attached scratchpads.
+ Maintains a 100% linear VRAM profile and eliminates dynamic allocations.
+ Uses fully vectorized 2D broadcasting to calculate merged bounding boxes
+ simultaneously on the GPU without sequential loops or CPU synchronization stalls.
+ """
+ if raw_boxes.shape[0] <= 1:
+ return raw_boxes
+
+ # 1. ROBUST CACHE CHECK: Verify shape dimension bounds to prevent 1D flat layout leakage
+ if (
+ not hasattr(merge_boxes_gpu, "adj_matrix")
+ or merge_boxes_gpu.adj_matrix.ndim != 2
+ ):
+ merge_boxes_gpu.adj_matrix = torch.zeros(
+ (max_cached_elements, max_cached_elements),
+ dtype=torch.bool,
+ device=raw_boxes.device,
+ )
+ merge_boxes_gpu.components = torch.zeros(
+ (max_cached_elements,), dtype=torch.long, device=raw_boxes.device
+ )
+ merge_boxes_gpu.scratch_out = torch.zeros(
+ (max_cached_elements, 4), dtype=torch.float32, device=raw_boxes.device
+ )
+
+ # Establish safe spatial clip windows depending on current frame tracking load
+ N = min(raw_boxes.shape[0], max_cached_elements)
+
+ # 2. VECTORIZED ARITHMETIC WITH VIEWS: Completely replace unbind() and unsqueeze() list loops
+ x1 = raw_boxes[:N, 0]
+ y1 = raw_boxes[:N, 1]
+ x2 = raw_boxes[:N, 2]
+ y2 = raw_boxes[:N, 3]
+
+ # Create spatial lookup grids via view broadcasting
+ x1_col, x1_row = x1.view(N, 1), x1.view(1, N)
+ y1_col, y1_row = y1.view(N, 1), y1.view(1, N)
+ x2_col, x2_row = x2.view(N, 1), x2.view(1, N)
+ y2_col, y2_row = y2.view(N, 1), y2.view(1, N)
+
+ # Compute distance boundaries directly using hardware vector instructions
+ h_gaps = torch.clamp(
+ torch.maximum(x1_col, x1_row) - torch.minimum(x2_col, x2_row), min=0
+ )
+ v_gaps = torch.clamp(
+ torch.maximum(y1_col, y1_row) - torch.minimum(y2_col, y2_row), min=0
+ )
+
+ # Map target union envelopes without generating intermediate subtraction tensors
+ union_w = torch.maximum(x2_col, x2_row) - torch.minimum(x1_col, x1_row)
+ union_h = torch.maximum(y2_col, y2_row) - torch.minimum(y1_col, y1_row)
+
+ # Overwrite adjacency bounds directly into the pre-allocated cache slice
+ adj = merge_boxes_gpu.adj_matrix[:N, :N]
+ adj.copy_(
(h_gaps < gap_limit)
& (v_gaps < gap_limit)
& (union_w < size_limit)
& (union_h < size_limit)
)
- # 4. Parallel Connected Components (Existing logic)
- components = torch.arange(raw_boxes.shape[0], device=raw_boxes.device)
- r = 3
- for _ in range(r):
- components = torch.max(adj * components, dim=1).values
+ # 3. ACCELERATED CONNECTED COMPONENTS: Run pointer rotations in place
+ comp = merge_boxes_gpu.components[:N]
+ torch.arange(N, device=raw_boxes.device, out=comp)
+
+ # Wrap the destination outputs into a fast component view format
+ comp_idx_scratch = torch.zeros(N, dtype=torch.long, device=raw_boxes.device)
+ output_tuple = (comp, comp_idx_scratch)
+
+ # Unified logical components compression step
+ for _ in range(3):
+ torch.max(torch.where(adj, comp, 0), dim=1, out=output_tuple)
+
+ # 4. VECTORIZED COMPONENT REDUCTION (ZERO-LOOP BRIDGING)
+ # Find the unique cluster assignments entirely on the GPU
+ unique_ids = comp.unique()
+ num_merged = unique_ids.shape[0]
+
+ # Sub-slice our persistent function cache block instantly
+ out_buffer = merge_boxes_gpu.scratch_out[:num_merged]
+ boxes_subset = raw_boxes[:N]
+
+ # Create a 2D broadcasted membership equality grid [Num_Clusters, N_Boxes]
+ # This evaluates cluster assignments for all boxes simultaneously on the GPU!
+ mask_grid = comp.unsqueeze(0) == unique_ids.view(-1, 1)
+
+ # Mask out invalid bounding box entries by casting non-members to extreme float values.
+ # This guarantees unassigned boxes do not interfere with the parallel min/max reduction.
+ x1_masked = torch.where(mask_grid, boxes_subset[:, 0].unsqueeze(0), float("inf"))
+ y1_masked = torch.where(mask_grid, boxes_subset[:, 1].unsqueeze(0), float("inf"))
+ x2_masked = torch.where(mask_grid, boxes_subset[:, 2].unsqueeze(0), float("-inf"))
+ y2_masked = torch.where(mask_grid, boxes_subset[:, 3].unsqueeze(0), float("-inf"))
+
+ # Execute single-pass parallel reductions across dim=1 (the boxes axis)
+ # Bypasses Python loop steps, CPU stalls, and unoptimized scatter commands entirely
+ out_buffer[:, 0] = x1_masked.min(dim=1).values
+ out_buffer[:, 1] = y1_masked.min(dim=1).values
+ out_buffer[:, 2] = x2_masked.max(dim=1).values
+ out_buffer[:, 3] = y2_masked.max(dim=1).values
+
+ # Clean local tensor references inside the function namespace to protect VRAM scope
+ del (
+ mask_grid,
+ comp_idx_scratch,
+ x1_masked,
+ y1_masked,
+ x2_masked,
+ y2_masked,
+ boxes_subset,
+ )
- unique_ids = components.unique()
- merged = []
- for i in unique_ids:
- mask = components == i
- merged.append(
- torch.cat(
- [raw_boxes[mask, :2].min(0).values, raw_boxes[mask, 2:].max(0).values]
+ return out_buffer.clone()
+
+
+def merge_boxes_gpu_v3(
+ raw_boxes, gap_limit=10, size_limit=640, max_cached_elements=1500
+):
+ """
+ Optimized Parallel Merger for pixel-integer motion boxes.
+ Groups close boxes and dynamically expands all resulting blocks up to a uniform
+ size_limit square for optimal YOLO backbone processing.
+
+ Args:
+ raw_boxes (torch.Tensor): Int or Float Tensor of shape [N, 4] -> [x1, y1, x2, y2]
+ gap_limit (int): Proximity merge trigger threshold in absolute pixels.
+ size_limit (int): Target square dimension (width & height) for YOLO input.
+ """
+ if raw_boxes.shape[0] == 0:
+ return raw_boxes
+
+ # Sizing metrics for center-out expansion
+ half_size = float(size_limit) / 2.0
+
+ # 1. DYNAMIC FRAME RESOLUTION DETECTION
+ # Automatically extracts frame limits from the highest observed coordinates in the batch
+ # to avoid hardcoding video resolution dimensions.
+ if raw_boxes.shape[0] > 0:
+ frame_w = int(torch.max(raw_boxes[:, 2]).item())
+ frame_h = int(torch.max(raw_boxes[:, 3]).item())
+ # Safe fallback buffer if tracking objects at the extreme top/left edges
+ frame_w = max(frame_w, 640)
+ frame_h = max(frame_h, 640)
+ else:
+ frame_w, frame_h = 1920, 1080
+
+ # Handle single box edge-case quickly to minimize path latency
+ if raw_boxes.shape[0] == 1:
+ merged_buffer = raw_boxes.clone().float()
+ num_merged = 1
+ else:
+ # 2. ROBUST VRAM CACHE SEEDING
+ if (
+ not hasattr(merge_boxes_gpu, "adj_matrix")
+ or merge_boxes_gpu.adj_matrix.ndim != 2
+ ):
+ merge_boxes_gpu.adj_matrix = torch.zeros(
+ (max_cached_elements, max_cached_elements),
+ dtype=torch.bool,
+ device=raw_boxes.device,
+ )
+ merge_boxes_gpu.components = torch.zeros(
+ (max_cached_elements,), dtype=torch.long, device=raw_boxes.device
)
+ merge_boxes_gpu.scratch_out = torch.zeros(
+ (max_cached_elements, 4), dtype=torch.float32, device=raw_boxes.device
+ )
+
+ N = min(raw_boxes.shape[0], max_cached_elements)
+
+ # 3. PARALLEL GEOMETRIC GAP CALCULATIONS
+ x1 = raw_boxes[:N, 0]
+ y1 = raw_boxes[:N, 1]
+ x2 = raw_boxes[:N, 2]
+ y2 = raw_boxes[:N, 3]
+
+ h_gaps = torch.clamp(
+ torch.max(x1.view(N, 1) - x2.view(1, N), x1.view(1, N) - x2.view(N, 1)),
+ min=0,
+ )
+ v_gaps = torch.clamp(
+ torch.max(y1.view(N, 1) - y2.view(1, N), y1.view(1, N) - y2.view(N, 1)),
+ min=0,
)
- return torch.stack(merged)
+ union_w = torch.max(x2.view(N, 1), x2.view(1, N)) - torch.min(
+ x1.view(N, 1), x1.view(1, N)
+ )
+ union_h = torch.max(y2.view(N, 1), y2.view(1, N)) - torch.min(
+ y1.view(N, 1), y1.view(1, N)
+ )
+
+ # Vectorized generation of adjacency grid state properties
+ adj = merge_boxes_gpu.adj_matrix[:N, :N]
+ adj.copy_(
+ (h_gaps < gap_limit)
+ & (v_gaps < gap_limit)
+ & (union_w <= size_limit)
+ & (union_h <= size_limit)
+ )
+
+ # 4. CONNECTED COMPONENT DISCOVERY VIA GRAPH ITERATIONS
+ comp = merge_boxes_gpu.components[:N]
+ torch.arange(N, device=raw_boxes.device, out=comp)
+
+ for _ in range(3):
+ comp.copy_(torch.max(adj * comp, dim=1).values)
+
+ unique_ids = comp.unique()
+ num_merged = unique_ids.shape[0]
+ merged_buffer = merge_boxes_gpu.scratch_out[:num_merged]
+
+ for idx, i in enumerate(unique_ids):
+ mask = comp == i
+ boxes_subset = raw_boxes[:N][mask]
+
+ merged_buffer[idx, 0] = boxes_subset[:, 0].min()
+ merged_buffer[idx, 1] = boxes_subset[:, 1].min()
+ merged_buffer[idx, 2] = boxes_subset[:, 2].max()
+ merged_buffer[idx, 3] = boxes_subset[:, 3].max()
+
+ # 5. SYMMETRIC CENTROID SQUARE EXPANSION
+ cx = (merged_buffer[:, 0] + merged_buffer[:, 2]) * 0.5
+ cy = (merged_buffer[:, 1] + merged_buffer[:, 3]) * 0.5
+
+ ex1 = cx - half_size
+ ey1 = cy - half_size
+ ex2 = cx + half_size
+ ey2 = cy + half_size
+
+ # 6. EDGE-COLLISION ANCHOR SHIFTING
+ # Instead of destructive clipping (which breaks squares), shift windows inward
+ # when they hit frame edges to guarantee your exact YOLO aspect ratio is maintained.
+ shift_left = torch.clamp(0.0 - ex1, min=0)
+ shift_right = torch.clamp(ex2 - frame_w, min=0)
+ ex1 += shift_left - shift_right
+ ex2 += shift_left - shift_right
+
+ shift_top = torch.clamp(0.0 - ey1, min=0)
+ shift_bottom = torch.clamp(ey2 - frame_h, min=0)
+ ey1 += shift_top - shift_bottom
+ ey2 += shift_top - shift_bottom
+
+ # 7. INTEGER FORMAT OUTPUT AND SECURE BOUNDARY BOUNDING
+ final_output = torch.zeros(
+ (num_merged, 4), dtype=torch.long, device=raw_boxes.device
+ )
+ final_output[:, 0] = torch.clamp(ex1.round().long(), min=0, max=frame_w)
+ final_output[:, 1] = torch.clamp(ey1.round().long(), min=0, max=frame_h)
+ final_output[:, 2] = torch.clamp(ex2.round().long(), min=0, max=frame_w)
+ final_output[:, 3] = torch.clamp(ey2.round().long(), min=0, max=frame_h)
+
+ return final_output
+
+
+def merge_boxes_gpu_v4(
+ raw_boxes, gap_limit=10, size_limit=1000, max_cached_elements=1500
+):
+ """
+ Refined Parallel Merger utilizing static function-attached scratchpads.
+ Maintains a 100% linear VRAM profile and eliminates dynamic allocations.
+ """
+ if raw_boxes.shape[0] <= 1:
+ return raw_boxes
+
+ # ─── ADD THIS HIGH-THROUGHPUT CANDIDATE FILTER GUARD ──────────────────
+ # 1. Compute areas of all boxes instantly in parallel on the GPU
+ # widths = raw_boxes[:, 2] - raw_boxes[:, 0]
+ # heights = raw_boxes[:, 3] - raw_boxes[:, 1]
+ # areas = widths * heights
+
+ # # 2. If the scene is overloaded, prioritize the top 400 largest macro regions
+ # # This prevents noise artifacts from blowing up the [N, N] grid layout!
+ # if raw_boxes.shape[0] > 400:
+ # _, top_indices = torch.topk(areas, k=400, sorted=False)
+ # raw_boxes = raw_boxes[top_indices]
+ # ──────────────────────────────────────────────────────────────────────
+
+ # 1. ROBUST CACHE CHECK: Verify shape dimension bounds to prevent 1D flat layout leakage
+ if (
+ not hasattr(merge_boxes_gpu, "adj_matrix")
+ or merge_boxes_gpu.adj_matrix.ndim != 2
+ ):
+ merge_boxes_gpu.adj_matrix = torch.zeros(
+ (max_cached_elements, max_cached_elements),
+ dtype=torch.bool,
+ device=raw_boxes.device,
+ )
+ merge_boxes_gpu.components = torch.zeros(
+ (max_cached_elements,), dtype=torch.long, device=raw_boxes.device
+ )
+ merge_boxes_gpu.components_idx = torch.zeros(
+ (max_cached_elements,), dtype=torch.long, device=raw_boxes.device
+ )
+ merge_boxes_gpu.scratch_out = torch.zeros(
+ (max_cached_elements, 4), dtype=torch.float32, device=raw_boxes.device
+ )
+
+ # Establish safe spatial clip windows depending on current frame tracking load
+ N = min(raw_boxes.shape[0], max_cached_elements)
+
+ # 2. VECTORIZED ARITHMETIC WITH VIEWS: Completely replace unbind() and unsqueeze() list loops
+ x1 = raw_boxes[:N, 0]
+ y1 = raw_boxes[:N, 1]
+ x2 = raw_boxes[:N, 2]
+ y2 = raw_boxes[:N, 3]
+
+ # # Compute gaps smoothly using in-place operations over zero-copy memory layouts
+ # h_gaps = torch.clamp(
+ # torch.max(x1.view(N, 1) - x2.view(1, N), x1.view(1, N) - x2.view(N, 1)), min=0
+ # )
+ # v_gaps = torch.clamp(
+ # torch.max(y1.view(N, 1) - y2.view(1, N), y1.view(1, N) - y2.view(N, 1)), min=0
+ # )
+
+ # # Map target union envelopes natively
+ # union_w = torch.max(x2.view(N, 1), x2.view(1, N)) - torch.min(
+ # x1.view(N, 1), x1.view(1, N)
+ # )
+ # union_h = torch.max(y2.view(N, 1), y2.view(1, N)) - torch.min(
+ # y1.view(N, 1), y1.view(1, N)
+ # )
+ x1_col, x1_row = x1.view(N, 1), x1.view(1, N)
+ y1_col, y1_row = y1.view(N, 1), y1.view(1, N)
+ x2_col, x2_row = x2.view(N, 1), x2.view(1, N)
+ y2_col, y2_row = y2.view(N, 1), y2.view(1, N)
+
+ # Compute distance boundaries directly.
+ # torch.maximum/minimum maps to direct hardware vector instructions.
+ h_gaps = torch.clamp(
+ torch.maximum(x1_col, x1_row) - torch.minimum(x2_col, x2_row), min=0
+ )
+ v_gaps = torch.clamp(
+ torch.maximum(y1_col, y1_row) - torch.minimum(y2_col, y2_row), min=0
+ )
+
+ # Map target union envelopes without generating intermediate subtraction tensors
+ union_w = torch.maximum(x2_col, x2_row) - torch.minimum(x1_col, x1_row)
+ union_h = torch.maximum(y2_col, y2_row) - torch.minimum(y1_col, y1_row)
+
+ # Overwrite adjacency bounds directly into the pre-allocated cache slice
+ adj = merge_boxes_gpu.adj_matrix[:N, :N]
+ adj.copy_(
+ (h_gaps < gap_limit)
+ & (v_gaps < gap_limit)
+ & (union_w < size_limit)
+ & (union_h < size_limit)
+ )
+ # 3. ACCELERATED CONNECTED COMPONENTS: Run pointer rotations in place
+ comp = merge_boxes_gpu.components[:N]
+ # comp_idx = merge_boxes_gpu.components_idx[:N]
+ comp_idx_scratch = torch.zeros(N, dtype=torch.long, device=raw_boxes.device)
+ torch.arange(N, device=raw_boxes.device, out=comp)
+
+ # Pack the destination outputs into the exact 2-tensor tuple expected by torch.max
+ output_tuple = (comp, comp_idx_scratch)
+
+ # Unified logical components compression step
+ for _ in range(3):
+ # comp.copy_(torch.max(adj * comp, dim=1).values)
+ # Using a logical matrix view avoids any intermediate [N, N] expansions
+ torch.max(torch.where(adj, comp, 0), dim=1, out=output_tuple)
+
+ # unique_ids = comp.unique()
+ # num_merged = unique_ids.shape[0]
+
+ # # 4. STATIC MEMORY AGGREGATION: Eliminate the loops appending torch.cat() arrays
+ # out_buffer = merge_boxes_gpu.scratch_out[:num_merged]
+
+ # for idx, i in enumerate(unique_ids):
+ # mask = comp == i
+ # boxes_subset = raw_boxes[:N][mask]
+
+ # # Write structural outputs directly into our static memory block channels
+ # out_buffer[idx, 0] = boxes_subset[:, 0].min()
+ # out_buffer[idx, 1] = boxes_subset[:, 1].min()
+ # out_buffer[idx, 2] = boxes_subset[:, 2].max()
+ # out_buffer[idx, 3] = boxes_subset[:, 3].max()
+
+ # # Return the clean floating-point tensor slice natively
+ # return out_buffer.clone()
+
+ # 1. Map labels to a clean 0-indexed range without calling .unique() or syncing with the CPU
+ # torch.unique(..., return_inverse=True) handles the grouping array assembly entirely on the GPU
+ # _, inverse_indices = torch.unique(comp, return_inverse=True)
+ # num_merged = inverse_indices.max().item() + 1 # Minimal sync step
+
+ # out_buffer = merge_boxes_gpu.scratch_out[:num_merged]
+ # boxes_subset = raw_boxes[:N]
+
+ # # 2. Pre-fill scratchpad buffers with boundary initialization constants
+ # out_buffer[:, 0:2].fill_(float('inf'))
+ # out_buffer[:, 2:4].fill_(float('-inf'))
+
+ # # 3. Use highly optimized CUDA scatter reduction handles to compute all mins/maxes at once
+ # # This completely eliminates your sequential Python "for idx, i in enumerate" loops!
+ # out_buffer[:, 0:2].scatter_reduce_(0, inverse_indices.view(-1, 1).expand(-1, 2), boxes_subset[:, 0:2], reduce="amin", include_self=False)
+ # out_buffer[:, 2:4].scatter_reduce_(0, inverse_indices.view(-1, 1).expand(-1, 2), boxes_subset[:, 2:4], reduce="amax", include_self=False)
+
+ # return out_buffer.clone()
+
+ # 1. Use maximum theoretical capacity boundaries directly (N candidates max)
+ # This completely skips calculating max item counts on the CPU host!
+ out_buffer = merge_boxes_gpu.scratch_out[:N]
+ boxes_subset = raw_boxes[:N]
+
+ # 2. Reset our fixed scratchpad slice boundaries using parallel operations
+ out_buffer[:, 0:2].fill_(float("inf"))
+ out_buffer[:, 2:4].fill_(float("-inf"))
+
+ # 3. Use 'comp' directly as your index axis mapping variable!
+ # Because 'comp' values fall within [0, N-1], they are already valid scatter addresses.
+ # This executes 100% asynchronously on the GPU without pausing the CPU thread.
+ out_buffer[:, 0:2].scatter_reduce_(
+ 0,
+ comp.view(-1, 1).expand(-1, 2),
+ boxes_subset[:, 0:2],
+ reduce="amin",
+ include_self=False,
+ )
+ out_buffer[:, 2:4].scatter_reduce_(
+ 0,
+ comp.view(-1, 1).expand(-1, 2),
+ boxes_subset[:, 2:4],
+ reduce="amax",
+ include_self=False,
+ )
-def merge_boxes_cpu(boxes, gap_limit=10):
+ # 4. Filter out unwritten rows where the boundaries remain at infinity
+ # valid_mask = out_buffer[:, 0] != float('inf')
+
+ # # return out_buffer[valid_mask].clone()
+ # # Extract the exact non-zero index mappings entirely on the GPU.
+ # # This prevents the CPU from forcing a data shape verification step!
+ # valid_indices = torch.nonzero(valid_mask).squeeze(1)
+
+ # # Index with integers instead of booleans to drop line 1098 down to 0ms
+ # return out_buffer[valid_indices].clone()
+
+ # 4. Filter out unwritten rows where the boundaries remain at infinity
+ valid_mask = out_buffer[:, 0] != float("inf")
+
+ # Sort the boolean mask in descending order entirely on the GPU hardware.
+ # This automatically pushes all valid boxes (True/1) to the top of the matrix
+ # and leaves unwritten slots (False/0) at the bottom.
+ _, sort_indices = torch.sort(valid_mask.long(), descending=True, dim=0)
+
+ # Rearrange the static out_buffer asynchronously without altering its shape [N, 4]
+ out_buffer_sorted = out_buffer[sort_indices]
+
+ # Overwrite the remaining invalid infinity rows to safe zero-pads.
+ # This ensures downstream deep learning blocks don't encounter NaN errors.
+ invalid_rows_mask = ~valid_mask[sort_indices]
+ out_buffer_sorted[invalid_rows_mask] = 0.0
+
+ # Return the static-shaped [N, 4] tensor handle.
+ # Because N is already known by the CPU, line 1098 drops instantly to 0ms!
+ return out_buffer_sorted
+
+
+# def merge_boxes_cpuv1(boxes, gap_limit=10):
+# """
+# Greedy merge in 640x640 space to consolidate swarm fragments.
+# Input: List of [x1, y1, x2, y2] within [0, 640]
+# """
+# if not boxes:
+# return []
+
+# # O(N log N) sort by X for early exit optimization
+# boxes = sorted(boxes, key=lambda x: x[0])
+# merged = []
+
+# while boxes:
+# curr = boxes.pop(0)
+# i = 0
+# while i < len(boxes):
+# test = boxes[i]
+# # Early exit: horizontal gap exceeds limit
+# if test[0] - curr[2] > gap_limit:
+# break
+
+# # Check vertical gap
+# y_dist = max(0, test[1] - curr[3], curr[3] - test[1])
+# if y_dist <= gap_limit:
+# # Expand curr box to include test
+# curr = [
+# min(curr[0], test[0]),
+# min(curr[1], test[1]),
+# max(curr[2], test[2]),
+# max(curr[3], test[3]),
+# ]
+# boxes.pop(i)
+# i = 0 # Re-check boundaries
+# else:
+# i += 1
+# merged.append(curr)
+# return merged
+
+
+def merge_boxes_cpu(boxes, gap_limit=10, size_limit=800):
"""
Greedy merge in 640x640 space to consolidate swarm fragments.
+ Preserves original boxes larger than size_limit, merging others where possible.
Input: List of [x1, y1, x2, y2] within [0, 640]
"""
if not boxes:
@@ -719,9 +2314,24 @@ def merge_boxes_cpu(boxes, gap_limit=10):
while boxes:
curr = boxes.pop(0)
+
+ # --- PRESERVE OVERSIZED CROPS ---
+ # If the current bounding box is already larger than the size limit,
+ # skip merging it with anything else and save it directly to the output pool.
+ if (curr[2] - curr[0]) > size_limit or (curr[3] - curr[1]) > size_limit:
+ merged.append(curr)
+ continue
+
i = 0
while i < len(boxes):
test = boxes[i]
+
+ # If the test candidate box is already larger than the size limit,
+ # do not attempt to merge it into the current tracking cluster.
+ if (test[2] - test[0]) > size_limit or (test[3] - test[1]) > size_limit:
+ i += 1
+ continue
+
# Early exit: horizontal gap exceeds limit
if test[0] - curr[2] > gap_limit:
break
@@ -729,150 +2339,159 @@ def merge_boxes_cpu(boxes, gap_limit=10):
# Check vertical gap
y_dist = max(0, test[1] - curr[3], curr[3] - test[1])
if y_dist <= gap_limit:
- # Expand curr box to include test
- curr = [
- min(curr[0], test[0]),
- min(curr[1], test[1]),
- max(curr[2], test[2]),
- max(curr[3], test[3]),
- ]
+ # Calculate proposed expanded dimensions if merged
+ new_x1 = min(curr[0], test[0])
+ new_y1 = min(curr[1], test[1])
+ new_x2 = max(curr[2], test[2])
+ new_y2 = max(curr[3], test[3])
+
+ # --- BOUNDARY EXPANSION GUARD ---
+ # Reject the merge if combining these small boxes would push
+ # the resulting macro patch past the maximum allowed size limit.
+ if (new_x2 - new_x1) > size_limit or (new_y2 - new_y1) > size_limit:
+ i += 1
+ continue
+
+ # Expand current tracking box safely
+ curr = [new_x1, new_y1, new_x2, new_y2]
boxes.pop(i)
- i = 0 # Re-check boundaries
+ i = 0 # Re-check updated boundaries against remaining boxes
else:
i += 1
merged.append(curr)
return merged
-def find_contours_gpu_equivalent(
- mask_gpu_mat, stream=None, grid_size=16, limit_640=1000, max_boxes=100
-):
- """
- ULTRA-OPTIMIZED: Grid-based Region Proposal.
- Reduces N to prevent merger bottlenecks. Latency: <0.5ms.
- """
- # 1. Zero-copy bridge: GpuMat -> CuPy -> Torch
- mask_cp = gpumat2cupy(mask_gpu_mat)
- mask_tensor = torch.as_tensor(mask_cp, device="cuda")
-
- # # 2. Downsample via Max Pooling (Acts as Denoise + Grouper)
- # # A 32x32 grid on 640x640 creates a 20x20 matrix (400 cells max)
- # pooled = F.max_pool2d(
- # mask_tensor.unsqueeze(0).unsqueeze(0).float(),
- # kernel_size=grid_size,
- # stride=grid_size
- # ).squeeze()
-
- # # 3. Get Indices of Motion
- # indices = torch.nonzero(pooled > 0)
- # A grid_size of 32 has 1,024 pixels.
- # We require at least 5% (approx 50 pixels) to be white to trigger a ROI.
- # This 'math' kills terrain shimmer but keeps solid drone blobs.
- # density_threshold = (grid_size * grid_size) * 0.05
-
- # Use torch.count_nonzero if using AvgPool, or stick to pooled with threshold
- # Since 'pooled' from MaxPool is just the max value (0 or 255),
- # we should use F.avg_pool2d instead to get a density map:
-
- density_map = F.avg_pool2d(
- mask_tensor.unsqueeze(0).unsqueeze(0).float(),
- kernel_size=grid_size,
- stride=grid_size,
- ).squeeze()
-
- # Define your sensitivity (e.g., 5% density)
- # If a block is 5% full of 'white' (255) pixels, the average will be 12.75
- density_threshold = 2 # int((grid_size * grid_size) * 0.01)
-
- # 255 * 0.10 means the grid cell must be 10% white pixels
- indices = torch.nonzero(density_map > density_threshold)
-
- # EARLY EXIT: No motion detected, return empty
- if indices.shape[0] == 0:
- return torch.empty((0, 4), device="cuda")
- # 2. SORT BY DENSITY: Get values for each index
- # We pull the density values for every 'hot' cell
- densities = density_map[indices[:, 0], indices[:, 1]]
-
- # 3. Get the sort order (Descending: highest density first)
- _, sort_order = torch.sort(densities, descending=True)
- indices = indices[sort_order]
-
- # CAP N: If scene is too noisy, take top regions to save the merger
- if indices.shape[0] > max_boxes:
- indices = indices[:max_boxes]
-
- # 4. Map back to 640p Bounding Boxes
- y1, x1 = indices[:, 0] * grid_size, indices[:, 1] * grid_size
- y2, x2 = y1 + grid_size, x1 + grid_size
- raw_boxes = torch.stack([x1, y1, x2, y2], dim=1).float()
-
- # 5. Merge adjacent grid blocks
- # gap_limit=grid_size+2 ensures diagonal/nearby blocks connect
- return merge_boxes_gpu(raw_boxes, gap_limit=grid_size * 2, size_limit=limit_640)
-
-
-def find_contours_gpu_equivalentv1(mask_gpu_mat, stream=None):
- """
- GPU equivalent to cv2.findContours(mask, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)
- Returns: torch.Tensor [N, 4] containing (x1, y1, x2, y2) in analysis space.
- """
- # Bridge OpenCV GpuMat to CuPy (Zero-Copy)
- w, h = mask_gpu_mat.size()
- ptr = mask_gpu_mat.cudaPtr()
- pitch_bytes = mask_gpu_mat.step
-
- mask_cp = cupy.ndarray(
- (h, w),
- dtype=cupy.uint8,
- memptr=cupy.cuda.MemoryPointer(
- cupy.cuda.UnownedMemory(ptr, pitch_bytes * h, mask_gpu_mat), 0
- ),
- strides=(pitch_bytes, 1),
- )
-
- # Use the stream pointer if provided, otherwise default to 0 (Null Stream)
- stream_ptr = stream.cudaPtr() if stream else 0
-
- with cupy.cuda.ExternalStream(stream_ptr):
- # Labeling (Equivalent to finding connected components)
- # labeled is an int32 array where every 'blob' has a unique number
- structure = cupy.array([[1, 1, 1], [1, 1, 1], [1, 1, 1]])
- labeled, num_labels = cupyx.scipy.ndimage.label(mask_cp, structure=structure)
-
- if num_labels == 0:
- return torch.empty((0, 4), device="cuda")
-
- # Setup Bounding Box Buffers
- x1 = cupy.full((num_labels + 1,), w, dtype=cupy.int32)
- y1 = cupy.full((num_labels + 1,), h, dtype=cupy.int32)
- x2 = cupy.full((num_labels + 1,), -1, dtype=cupy.int32)
- y2 = cupy.full((num_labels + 1,), -1, dtype=cupy.int32)
-
- # Run the Bounds Kernel
- # IMPORTANT: Use labeled.strides[0]//4 to get the pitch in elements
- pitch_elements = labeled.strides[0] // 4
- tpb = (16, 16)
- bpg = ((w + tpb[0] - 1) // tpb[0], (h + tpb[1] - 1) // tpb[1])
-
- get_bounds_kernel(
- bpg, tpb, (labeled, pitch_elements, w, h, num_labels, x1, y1, x2, y2)
- )
-
- # Stack and return as Torch Tensor for YOLO/Drawing
- # We skip index 0 as it represents the background (black)
- # boxes = torch.stack(
- # [
- # torch.as_tensor(x1[1:], device="cuda"),
- # torch.as_tensor(y1[1:], device="cuda"),
- # torch.as_tensor(x2[1:], device="cuda"),
- # torch.as_tensor(y2[1:], device="cuda"),
- # ],
- # dim=1,
- # ).float()
- boxes = cupy.column_stack((x1[1:], y1[1:], x2[1:], y2[1:]))
-
- return torch.as_tensor(boxes, device="cuda").float()
+# def find_contours_gpu_equivalent_v1(
+# mask_gpu_mat, stream=None, grid_size=16, limit_640=1000, max_boxes=100
+# ):
+# """
+# ULTRA-OPTIMIZED: Grid-based Region Proposal.
+# Reduces N to prevent merger bottlenecks. Latency: <0.5ms.
+# """
+# # 1. Zero-copy bridge: GpuMat -> CuPy -> Torch
+# mask_cp = gpumat2cupy(mask_gpu_mat)
+# mask_tensor = torch.as_tensor(mask_cp, device="cuda")
+
+# # # 2. Downsample via Max Pooling (Acts as Denoise + Grouper)
+# # # A 32x32 grid on 640x640 creates a 20x20 matrix (400 cells max)
+# # pooled = F.max_pool2d(
+# # mask_tensor.unsqueeze(0).unsqueeze(0).float(),
+# # kernel_size=grid_size,
+# # stride=grid_size
+# # ).squeeze()
+
+# # # 3. Get Indices of Motion
+# # indices = torch.nonzero(pooled > 0)
+# # A grid_size of 32 has 1,024 pixels.
+# # We require at least 5% (approx 50 pixels) to be white to trigger a ROI.
+# # This 'math' kills terrain shimmer but keeps solid drone blobs.
+# # density_threshold = (grid_size * grid_size) * 0.05
+
+# # Use torch.count_nonzero if using AvgPool, or stick to pooled with threshold
+# # Since 'pooled' from MaxPool is just the max value (0 or 255),
+# # we should use F.avg_pool2d instead to get a density map:
+
+# density_map = F.avg_pool2d(
+# mask_tensor.unsqueeze(0).unsqueeze(0).float(),
+# kernel_size=grid_size,
+# stride=grid_size,
+# ).squeeze()
+
+# # Define your sensitivity (e.g., 5% density)
+# # If a block is 5% full of 'white' (255) pixels, the average will be 12.75
+# density_threshold = 2 # int((grid_size * grid_size) * 0.01)
+
+# # 255 * 0.10 means the grid cell must be 10% white pixels
+# indices = torch.nonzero(density_map > density_threshold)
+
+# # EARLY EXIT: No motion detected, return empty
+# if indices.shape[0] == 0:
+# return torch.empty((0, 4), device="cuda")
+# # 2. SORT BY DENSITY: Get values for each index
+# # We pull the density values for every 'hot' cell
+# densities = density_map[indices[:, 0], indices[:, 1]]
+
+# # 3. Get the sort order (Descending: highest density first)
+# _, sort_order = torch.sort(densities, descending=True)
+# indices = indices[sort_order]
+
+# # CAP N: If scene is too noisy, take top regions to save the merger
+# if indices.shape[0] > max_boxes:
+# indices = indices[:max_boxes]
+
+# # 4. Map back to 640p Bounding Boxes
+# y1, x1 = indices[:, 0] * grid_size, indices[:, 1] * grid_size
+# y2, x2 = y1 + grid_size, x1 + grid_size
+# raw_boxes = torch.stack([x1, y1, x2, y2], dim=1).float()
+
+# # 5. Merge adjacent grid blocks
+# # gap_limit=grid_size+2 ensures diagonal/nearby blocks connect
+# # del mask_cp, mask_tensor, density_map, densities, indices
+# return merge_boxes_gpu(raw_boxes, gap_limit=grid_size * 2, size_limit=limit_640)
+
+# Better: moved to handlers.py
+# def find_contours_gpu_equivalent(mask_gpu_mat, stream=None, limit_640=None):
+# """
+# GPU equivalent to cv2.findContours(mask, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)
+# Returns: torch.Tensor [N, 4] containing (x1, y1, x2, y2) in analysis space.
+# """
+# # Bridge OpenCV GpuMat to CuPy (Zero-Copy)
+# w, h = mask_gpu_mat.size()
+# ptr = mask_gpu_mat.cudaPtr()
+# pitch_bytes = mask_gpu_mat.step
+
+# mask_cp = cupy.ndarray(
+# (h, w),
+# dtype=cupy.uint8,
+# memptr=cupy.cuda.MemoryPointer(
+# cupy.cuda.UnownedMemory(ptr, pitch_bytes * h, mask_gpu_mat), 0
+# ),
+# strides=(pitch_bytes, 1),
+# )
+
+# # Use the stream pointer if provided, otherwise default to 0 (Null Stream)
+# stream_ptr = stream.cudaPtr() if stream else 0
+
+# with cupy.cuda.ExternalStream(stream_ptr):
+# # Labeling (Equivalent to finding connected components)
+# # labeled is an int32 array where every 'blob' has a unique number
+# structure = cupy.ones((3, 3), dtype=cupy.int32)
+# labeled, num_labels = cupyx.scipy.ndimage.label(mask_cp, structure=structure)
+
+# if num_labels == 0:
+# return torch.empty((0, 4), device="cuda")
+
+# # Setup Bounding Box Buffers
+# x1 = cupy.full((num_labels + 1,), w, dtype=cupy.int32)
+# y1 = cupy.full((num_labels + 1,), h, dtype=cupy.int32)
+# x2 = cupy.full((num_labels + 1,), -1, dtype=cupy.int32)
+# y2 = cupy.full((num_labels + 1,), -1, dtype=cupy.int32)
+
+# # Run the Bounds Kernel
+# # IMPORTANT: Use labeled.strides[0]//4 to get the pitch in elements
+# pitch_elements = labeled.strides[0] // 4
+# tpb = (16, 16)
+# bpg = ((w + tpb[0] - 1) // tpb[0], (h + tpb[1] - 1) // tpb[1])
+
+# get_bounds_kernel(
+# bpg, tpb, (labeled, pitch_elements, w, h, num_labels, x1, y1, x2, y2)
+# )
+
+# # Stack and return as Torch Tensor for YOLO/Drawing
+# # We skip index 0 as it represents the background (black)
+# # boxes = torch.stack(
+# # [
+# # torch.as_tensor(x1[1:], device="cuda"),
+# # torch.as_tensor(y1[1:], device="cuda"),
+# # torch.as_tensor(x2[1:], device="cuda"),
+# # torch.as_tensor(y2[1:], device="cuda"),
+# # ],
+# # dim=1,
+# # ).float()
+# boxes = cupy.column_stack((x1[1:], y1[1:], x2[1:], y2[1:]))
+
+# return torch.as_tensor(boxes, device="cuda").float()
def get_detection_color(index, is_bgr=False):
@@ -977,7 +2596,7 @@ def format_df_value(value):
return value
-def get_display_frame_in_bytes(
+def get_display_frame_in_bytesv1(
foi, display_size=(960, 540), quality=50, return_bytes=True, device="CPU"
): # Expects BGR
H, W = foi.shape[:2]
@@ -1003,6 +2622,50 @@ def get_display_frame_in_bytes(
return frame_bytes
+def get_display_frame_in_bytes(
+ foi, display_size=(960, 540), quality=50, return_bytes=True, device="CPU"
+):
+ """
+ Safely formats and compresses video frames for browser distribution.
+ Accepts packed BGR array layouts directly.
+ """
+ if foi is None:
+ return None
+
+ # Defensive Guard: If a raw PyTorch CUDA tensor accidentally leaks into this
+ # context path, instantly bring it down safely to a standard contiguous numpy array
+ if torch.is_tensor(foi):
+ if foi.is_cuda:
+ foi = foi.detach().cpu()
+ foi = foi.numpy()
+
+ # CRITICAL SIZING FIX: Explicitly parse display_size as (Width, Height)
+ # to perfectly match OpenCV's internal spatial coordinate expectations
+ target_w, target_h = display_size
+ current_h, current_w = foi.shape[:2]
+
+ # Check matching structural criteria correctly
+ if current_h == target_h and current_w == target_w:
+ display_frame = foi
+ else:
+ # Resize safely without slipping strides or inverting height/width constraints
+ display_frame = cv2.resize(
+ foi, (target_w, target_h), interpolation=cv2.INTER_NEAREST
+ )
+
+ # Encode to crisp, valid JPEG compression matrices
+ ret, buffer = cv2.imencode(
+ ".jpg", display_frame, [int(cv2.IMWRITE_JPEG_QUALITY), quality]
+ )
+
+ if ret and return_bytes:
+ return buffer.tobytes()
+ elif ret:
+ return buffer
+
+ return None
+
+
# Manual FPS calculation if OpenCV reports 0
def manual_fps_calculation(src, num_frames=10):
if isinstance(src, cv2.VideoCapture):
@@ -1032,6 +2695,49 @@ def manual_fps_calculation(src, num_frames=10):
return 0
+def scale_bbox(
+ bbox, origW, origH, targetW=7680, targetH=4320, in_format="xywh", out_format="xywh"
+):
+ if in_format == "xywh":
+ x, y, w, h = bbox
+ x2 = x + w
+ y2 = y + h
+ else:
+ x, y, x2, y2 = bbox
+
+ # Get scale factors
+ scale_x = targetW / origW
+ scale_y = targetH / origH
+
+ # Translate bbox to target resolution
+ # x_target = max(0, min(int(round(x * scale_x)), targetW - 1))
+ # y_target = max(0, min(int(round(y * scale_y)), targetH - 1))
+ # w_target = min(targetW - x_target, int(round(w * scale_x)))
+ # h_target = min(targetH - y_target, int(round(h * scale_y)))
+ x_target = max(0, int(x * scale_x))
+ y_target = max(0, int(y * scale_y))
+ x2_target = min(targetW - 1, int(x2 * scale_x))
+ y2_target = min(targetH - 1, int(y2 * scale_y))
+ w_target = x2_target - x_target
+ h_target = y2_target - y_target
+ if out_format == "xywh":
+ return [x_target, y_target, w_target, h_target]
+ else:
+ return [x_target, y_target, x2_target, y2_target]
+
+
+def scale_bbox_xywh(bbox, origW, origH, targetW=7680, targetH=4320):
+ return scale_bbox(
+ bbox,
+ origW,
+ origH,
+ targetW=targetW,
+ targetH=targetH,
+ in_format="xywh",
+ out_format="xywh",
+ )
+
+
def rgb_to_nv12_torch(rgb_tensor):
"""
Fast GPU conversion from RGB to NV12 using PyTorch.
@@ -1339,46 +3045,60 @@ def merge_boxes_limit(boxes, dist_threshold=25, min_area=32, max_size=640):
return boxes
-def filter_contained_boxes(boxes, containment_thresh=0.90):
- if len(boxes) < 2:
- return boxes
+# def filter_contained_boxes(boxes, containment_thresh=0.90):
+# if len(boxes) < 2:
+# return boxes
- # Convert to NumPy for vectorized math
- objs = np.array(boxes)
- areas = (objs[:, 2] - objs[:, 0]) * (objs[:, 3] - objs[:, 1])
+# # Convert to NumPy for vectorized math
+# objs = np.array(boxes)
+# areas = (objs[:, 2] - objs[:, 0]) * (objs[:, 3] - objs[:, 1])
- # Sort by area descending
- order = areas.argsort()[::-1]
- objs = objs[order]
- areas = areas[order]
+# # Sort by area descending
+# order = areas.argsort()[::-1]
+# objs = objs[order]
+# areas = areas[order]
- keep = []
- idx_list = np.arange(len(objs))
+# keep = []
+# idx_list = np.arange(len(objs))
- while len(idx_list) > 0:
- i = idx_list[0]
- keep.append(objs[i].tolist())
- if len(idx_list) == 1:
- break
+# while len(idx_list) > 0:
+# i = idx_list[0]
+# keep.append(objs[i].tolist())
+# if len(idx_list) == 1:
+# break
+
+# # Vectorized Intersection over Union (IoU) / Containment
+# others = objs[idx_list[1:]]
+# ix1 = np.maximum(objs[i, 0], others[:, 0])
+# iy1 = np.maximum(objs[i, 1], others[:, 1])
+# ix2 = np.minimum(objs[i, 2], others[:, 2])
+# iy2 = np.minimum(objs[i, 3], others[:, 3])
+
+# iw = np.maximum(0, ix2 - ix1)
+# ih = np.maximum(0, iy2 - iy1)
+# inter_area = iw * ih
+
+# # Calculate how much 'others' are contained within 'i'
+# containment = inter_area / areas[idx_list[1:]]
+
+# # Only keep boxes that are NOT mostly contained within the current box
+# idx_list = idx_list[1:][containment < containment_thresh]
- # Vectorized Intersection over Union (IoU) / Containment
- others = objs[idx_list[1:]]
- ix1 = np.maximum(objs[i, 0], others[:, 0])
- iy1 = np.maximum(objs[i, 1], others[:, 1])
- ix2 = np.minimum(objs[i, 2], others[:, 2])
- iy2 = np.minimum(objs[i, 3], others[:, 3])
+# return keep
- iw = np.maximum(0, ix2 - ix1)
- ih = np.maximum(0, iy2 - iy1)
- inter_area = iw * ih
- # Calculate how much 'others' are contained within 'i'
- containment = inter_area / areas[idx_list[1:]]
+class DummyProcess:
+ def start(self):
+ pass
+
+ def join(self, timeout=None):
+ pass
- # Only keep boxes that are NOT mostly contained within the current box
- idx_list = idx_list[1:][containment < containment_thresh]
+ def is_alive(self):
+ return False
- return keep
+ def close(self):
+ pass
@dataclass
@@ -1394,3 +3114,199 @@ class PipelineMapping:
class StreamRequest(BaseModel):
url: str
name: str
+
+
+class ResourceTrackerFilter:
+ def __init__(self, original_stderr):
+ self.stderr = original_stderr
+ self.buffer = ""
+
+ def write(self, data):
+ self.buffer += data
+ # If background resource tracker signals bleed out during collection, suppress them
+ if (
+ "resource_tracker.py" in self.buffer
+ or "KeyError:" in self.buffer
+ or "shm_ai_640" in self.buffer
+ or "cache[rtype].remove(name)" in self.buffer
+ ):
+ if "\n" in data:
+ self.buffer = ""
+ return
+ self.stderr.write(data)
+ self.buffer = ""
+
+ def flush(self):
+ self.stderr.flush()
+
+
+# ==============================================================================
+# TRUE LOCK-FREE ASYNC VIDEO ENCODER WORKER
+# ==============================================================================
+class AsyncVideoWriter:
+ """Handles disk saving operations completely lock-free without thread-signaling overhead."""
+
+ def __init__(self, path, fourcc, fps, size):
+ self.writer = cv2.VideoWriter(path, fourcc, fps, size)
+ # Expand double buffer ring map to mitigate any underlying I/O bursts
+ self.buffer = deque(maxlen=int(fps))
+ self.running = True
+
+ # REMOVE SEVERE BOTTLENECK SIGNALS:
+ # self.frame_ready = threading.Event() <-- Deleted to clear lock.acquire thrashing!
+
+ # Pre-allocate page-locked pinned host buffer allocations
+ self.host_staging_tensor = torch.empty(
+ (size[1], size[0], 3), dtype=torch.uint8, device="cpu"
+ ).pin_memory()
+
+ self.thread = threading.Thread(target=self._write_loop, daemon=True)
+ self.thread.start()
+
+ def _write_loop(self):
+ while self.running or self.buffer:
+ try:
+ # Natively poll the lock-free structure directly without forcing a kernel signal wait block
+ if not self.buffer:
+ # A crisp 2ms sleep keeps the CPU completely cool while preserving
+ # sub-millisecond thread scheduling wake-ups on low priority cores
+ time.sleep(0.005)
+ continue
+
+ frame_payload = self.buffer.popleft()
+ if frame_payload is None:
+ break
+
+ # --- UNTHROTTLED HARDWARE VIEW EXTRACTION ---
+ if torch.is_tensor(frame_payload):
+ if frame_payload.is_cuda:
+ self.host_staging_tensor.copy_(frame_payload, non_blocking=True)
+
+ ctx_event = torch.cuda.Event()
+ ctx_event.record(torch.cuda.current_stream())
+ ctx_event.synchronize()
+ numpy_frame = self.host_staging_tensor.numpy()
+ else:
+ numpy_frame = frame_payload.numpy()
+ else:
+ numpy_frame = np.asarray(frame_payload)
+
+ self.writer.write(numpy_frame)
+
+ except Exception as e:
+ main_app_logger.info(
+ f"[DISK-WRITER-ERROR] Asynchronous frame save failed: {e}",
+ )
+ continue
+
+ def write_frame(self, frame):
+ """Pure lock-free, zero-signal submission track. Fires under 1 microsecond!"""
+ if self.running:
+ # Overwrites elements atomically without triggering standard Python lockouts
+ self.buffer.append(frame)
+ # REMOVE: self.frame_ready.set() <-- GONE! CPU thread remains un-starved!
+
+ def release(self):
+ """Secure drain release prevents early container dropouts."""
+ if torch.cuda.is_available():
+ torch.cuda.synchronize()
+
+ self.running = False
+ self.buffer.append(None)
+
+ if self.thread.is_alive():
+ self.thread.join(timeout=5.0)
+ self.writer.release()
+
+
+class AsyncDisplayVideoWriter:
+ """
+ Lock-free single-element buffer optimized for 8K pipelines.
+ Eliminates queue.Queue lock contention to preserve strict target frame rates.
+ """
+
+ def __init__(self, target_fps, display_size, quality=60):
+ self.target_fps = float(target_fps)
+ self.disp_w, self.disp_h = display_size
+ self.quality = int(quality)
+
+ # 1. 🔄 FIX: Use a lockless deque with maxlen=1 instead of queue.Queue
+ self.buffer = deque(maxlen=1)
+ self.running = True
+ self.handler = None
+
+ self.thread = threading.Thread(target=self._write_loop, daemon=True)
+ self.thread.start()
+
+ def set_handler_context(self, handler_instance):
+ self.handler = handler_instance
+
+ def _write_loop(self):
+ encode_param = [int(cv2.IMWRITE_JPEG_QUALITY), self.quality]
+
+ while self.running:
+ try:
+ # 2. 🔄 FIX: Non-blocking pop from the single-element buffer
+ if not self.buffer:
+ time.sleep(0.015) # 0.005)
+ continue
+
+ display_frame, frame_num = self.buffer.popleft()
+
+ if self.handler is None or not self.handler.active:
+ continue
+
+ if (
+ display_frame.shape[1] != self.disp_w
+ or display_frame.shape[0] != self.disp_h
+ ):
+ display_frame = cv2.resize(
+ display_frame,
+ (self.disp_w, self.disp_h),
+ interpolation=cv2.INTER_NEAREST,
+ )
+
+ ret, jpeg_buf = cv2.imencode(".jpg", display_frame, encode_param)
+ if not ret:
+ continue
+
+ frame_bytes = jpeg_buf.tobytes()
+ frame_len = len(frame_bytes)
+ num_shms = len(self.handler.shms)
+
+ write_idx = (self.handler.ready_buffer_idx.value + 1) % num_shms
+
+ shm_block = self.handler.shms[write_idx]
+ shm_block.buf[:frame_len] = memoryview(frame_bytes)
+
+ self.handler.shm_frame_lengths[write_idx] = frame_len
+ self.handler.ready_buffer_idx.value = write_idx
+ self.handler.shared_details["last_id"] = frame_num
+ self.handler.mp_last_id.value = frame_num
+
+ if hasattr(self.handler, "loop") and self.handler.loop is not None:
+ self.handler.loop.call_soon_threadsafe(
+ self.handler.frame_ready_event.set
+ )
+ else:
+ self.handler.frame_ready_event.set()
+
+ except Exception as e:
+ main_app_logger.info(
+ f"[STREAM-ERROR] Lock-free frame streaming dropped: {e}"
+ )
+ continue
+
+ def write_frame(self, frame, frame_num):
+ """Thread-safe lock-free atomic submission track."""
+ if self.running:
+ # 3. 🔄 FIX: Directly append to overwrite the old entry without thread locks
+ self.buffer.append((frame, frame_num))
+
+ def release(self):
+ self.running = False
+
+ self.buffer.clear()
+
+ if self.thread.is_alive():
+ self.thread.join(timeout=0.5)
diff --git a/fastapi/main.py b/fastapi/main.py
index 3a02e75..0c14d9f 100644
--- a/fastapi/main.py
+++ b/fastapi/main.py
@@ -1,6 +1,13 @@
+# ==============================================================================
+# SUPPRESS WARNINGS
import warnings
warnings.filterwarnings("ignore", message="The value of the smallest subnormal for")
+warnings.filterwarnings(
+ "ignore", category=FutureWarning, message=".*reduce_op` is deprecated.*"
+)
+
+# ==============================================================================
import asyncio
import json
@@ -21,6 +28,8 @@
from fastapi.responses import StreamingResponse
from fastapi.templating import Jinja2Templates
+# ==============================================================================
+
MODEL_CLASSES_FILE = "/var/www/cache/model_classes.json"
RUN_CONFIG = PipelineConfig(
@@ -62,7 +71,7 @@
# Standardizes logs across the application and uvicorn server
logging.basicConfig(
level=logging.INFO,
- format="%(asctime)s - %(name)s - %(levelname)s - %(message)s",
+ format="%(asctime)s [%(levelname)s] %(name)s (%(filename)s:%(lineno)d) - %(message)s",
handlers=[logging.StreamHandler(sys.stdout)],
)
main_app_logger = logging.getLogger("fastapi_app")
@@ -110,36 +119,58 @@ async def stream_video(data: StreamRequest, request: Request):
dict: Status message and the updated list of active stream keys.
"""
url, name = data.url, data.name
+
+ gc.collect() # Final garbage collection
+ if torch.cuda.is_available():
+ torch.cuda.empty_cache()
+
+ # async with request.app.state.stream_lock:
active_streams = request.app.state.active_streams
+ if name in active_streams:
+ print(f"[STREAM_VIDEO]: Clearing old stream for {name}...")
+ active_streams[name].stop()
+ active_streams[name].stop_threads(["process_thread"])
+ active_streams[name] = None
+ del active_streams[name]
+
# Only initialize if the stream isn't already being processed
if name not in active_streams:
- print(f"Starting background worker for {name}...")
-
- # Check if a global model instance should be passed to the handler
- if RUN_CONFIG.SHARED_MODEL:
- handler = VideoStreamHandler(
- url,
- name,
- active_streams,
- config=RUN_CONFIG,
- model=app.state.model,
- )
- else:
- handler = VideoStreamHandler(
- url,
- name,
- active_streams,
- config=RUN_CONFIG,
- )
-
- # Start the background thread (OpenCV capture + AI inference)
- handler.start()
-
- # Register the handler in the global state for cross-endpoint access
- app.state.active_streams[name] = handler
-
- curr_keys = list(app.state.active_streams.keys())
+ print(f"[STREAM_VIDEO]: Starting background worker for {name}...")
+
+ try:
+ # Check if a global model instance should be passed to the handler
+ if RUN_CONFIG.SHARED_MODEL:
+ handler = VideoStreamHandler(
+ url,
+ name,
+ active_streams,
+ config=RUN_CONFIG,
+ model=app.state.model,
+ )
+ else:
+ handler = VideoStreamHandler(
+ url,
+ name,
+ active_streams,
+ config=RUN_CONFIG,
+ )
+
+ # Start the background thread (OpenCV capture + AI inference)
+ handler.start()
+ print("[STREAM_VIDEO]: Worker running...")
+
+ # Register the handler in the global state for cross-endpoint access
+ request.app.state.active_streams[name] = handler
+ except RuntimeError:
+ if "handler" in locals():
+ del handler
+ gc.collect()
+ if torch.cuda.is_available():
+ torch.cuda.empty_cache()
+ print("[STREAM_VIDEO]: Could not start stream. Try again...")
+
+ curr_keys = list(request.app.state.active_streams.keys())
if RUN_CONFIG.DEBUG_FLAG:
print(
f"stream DEBUG VIEW | PID: {os.getpid()} | Looking for: {name} | Found Keys: {curr_keys}"
@@ -174,30 +205,57 @@ async def frame_generator(streamer, request: Request):
Yields frames only when the background worker signals a new frame is ready.
Ensures strict chronological order using frame IDs.
"""
- shm_names = streamer.shared_details["shm_names"]
- reader_shms = [mp.shared_memory.SharedMemory(name=n) for n in shm_names]
+ reader_shms = []
last_sent_id = -1
try:
+ shm_names = streamer.shared_details.get("shm_names", [])
+ reader_shms = [mp.shared_memory.SharedMemory(name=n) for n in shm_names]
while streamer.active:
# Stop the generator immediately if the browser tab is closed
if await request.is_disconnected():
main_app_logger.info(f"Client disconnected from {name}")
break
- # Wait for the background thread to signal that AI processing is complete
- await streamer.frame_ready_event.wait()
- streamer.frame_ready_event.clear() # Reset for the next frame
+ if getattr(streamer, "_is_stopped", False) or not streamer.active:
+ main_app_logger.info(
+ f"Video execution finished naturally for {name}. Closing stream."
+ )
+ async with request.app.state.stream_lock:
+ app.state.active_streams.pop(name, None)
+ # active_streams.pop(name, None)
+ break
+
+ try:
+ # Wait for the background thread to signal that AI processing is complete
+ await asyncio.wait_for(
+ streamer.frame_ready_event.wait(), timeout=2.0
+ )
+ except asyncio.TimeoutError:
+ # Check if stream died or reader stopped
+ if not streamer.active or getattr(streamer, "_is_stopped", False):
+ async with request.app.state.stream_lock:
+ app.state.active_streams.pop(name, None)
+ break
+ continue
+ # streamer.frame_ready_event.clear() # Reset for the next frame
# streamer.reader_busy.value = True
# target_idx = streamer.ready_buffer_idx.value
# streamer.reader_active_idx.value = target_idx
+ # if not streamer.active:
+ if getattr(streamer, "_is_stopped", False) or not streamer.active:
+ active_streams.pop(name, None)
+ streamer.stop_threads(["process_thread"])
+ async with request.app.state.stream_lock:
+ app.state.active_streams.pop(name, None)
+ break
# Frame Synchronization: ensure we don't send duplicate or out-of-order frames
current_id = streamer.shared_details.get("last_id", -1)
if current_id > last_sent_id:
ready_idx = streamer.ready_buffer_idx.value
- streamer.reader_active_idx.value = ready_idx
+ # streamer.reader_active_idx.value = ready_idx
frame_len = streamer.shm_frame_lengths[ready_idx]
if frame_len > 0:
@@ -205,35 +263,56 @@ async def frame_generator(streamer, request: Request):
# shm_name = shm_names[ready_idx]
# print(f"DEBUG: Displaying SHM {shm_name}")
# frame_bytes = streamer.latest_processed_frame
- try:
- frame_bytes = bytes(reader_shms[ready_idx].buf[:frame_len])
- finally:
- streamer.reader_active_idx.value = -1
- last_sent_id = current_id
- streamer.last_heartbeat = time.time()
+ # try:
+ frame_bytes = bytes(reader_shms[ready_idx].buf[:frame_len])
+ # finally:
+ # streamer.reader_active_idx.value = -1
+ # last_sent_id = current_id
+ # streamer.last_heartbeat = time.time()
# streamer.reader_busy.value = False
+ streamer.frame_ready_event.clear()
# Multipart JPEG delivery with explicit Content-Length for stability
- yield (
- b"--frame\r\n"
- b"Content-Type: image/jpeg\r\n"
- b"Content-Length: "
- + str(len(frame_bytes)).encode()
- + b"\r\n\r\n"
- # b"Content-Length: "
- # + str(len(frame_bytes)).encode()
- # + b"\r\n\r\n"
- + frame_bytes
- + b"\r\n"
- )
- # last_sent_id = current_id
+ if len(frame_bytes) > 0:
+ yield (
+ b"--frame\r\n"
+ b"Content-Type: image/jpeg\r\n"
+ b"Content-Length: "
+ + str(len(frame_bytes)).encode()
+ + b"\r\n\r\n"
+ # b"Content-Length: "
+ # + str(len(frame_bytes)).encode()
+ # + b"\r\n\r\n"
+ + frame_bytes
+ + b"\r\n"
+ )
+ last_sent_id = current_id
+ streamer.last_heartbeat = time.perf_counter()
# Yield control to the event loop to prevent blocking
+ # streamer.frame_ready_event.clear() # Reset for the next frame
await asyncio.sleep(0.001)
except Exception as e:
- main_app_logger.error(f"[EXCEPTION] Generator Error: {e}")
- # finally:
- # streamer.reader_busy.value = False # Safety unlock
+ # Check for standard client disconnections (Broken pipe / Connection reset)
+ if isinstance(e, OSError) and e.errno in (32, 104):
+ # main_app_logger.info(f"Stream client disconnected gracefully from {name}")
+ pass
+ else:
+ main_app_logger.error(f" [EXCEPTION] Generator Error: {e}")
+ finally:
+ # streamer.reader_busy.value = False # Safety unlock
+ for shm in reader_shms:
+ try:
+ shm.close()
+ shm = None
+ except Exception:
+ pass
+ reader_shms.clear()
+ # pass
+
+ gc.collect()
+ if torch.cuda.is_available():
+ torch.cuda.empty_cache()
return StreamingResponse(
frame_generator(streamer, request),
@@ -279,7 +358,11 @@ async def dashboard_stats(request: Request):
# Video Backlog: Frames waiting for Disk I/O
video_backlog = (
- streamer.write_queue.qsize() if streamer.config.ENABLE_QUERYING else 0
+ streamer.write_queue.qsize()
+ if streamer.config.ENABLE_QUERYING
+ and hasattr(streamer, "write_queue")
+ and hasattr(streamer.write_queue, "qsize")
+ else 0
)
# IO Backlog: Frames queued for disk storage (if enabled)
@@ -297,7 +380,7 @@ async def dashboard_stats(request: Request):
"fps": round(streamer.stat_fps, 1),
"inputfps": round(streamer.input_fps, 1),
"targetfps": round(streamer.target_fps, 1),
- "is_streaming": streamer.active,
+ "is_streaming": getattr(streamer, "active", False),
"clipper_backlog": clipper_backlog,
"ai_backlog": ai_backlog,
"video_backlog": video_backlog,
@@ -319,6 +402,11 @@ async def get_status(request: Request):
}
+import gc
+
+import torch
+
+
@app.post("/stop_stream/{name}")
async def stop_stream(name: str, request: Request):
"""
@@ -326,14 +414,24 @@ async def stop_stream(name: str, request: Request):
The blocking cleanup logic is offloaded to a separate thread to prevent API hang.
"""
# Immediately remove from the global state so polling/UI syncs instantly
- streamer = request.app.state.active_streams.pop(name, None)
+ async with request.app.state.stream_lock:
+ streamer = request.app.state.active_streams.pop(name, None)
if streamer:
streamer.active = False
# BACKGROUND CLEANUP: Fire-and-forget the heavy hardware teardown
loop = asyncio.get_event_loop()
- loop.run_in_executor(None, streamer.stop)
+ await loop.run_in_executor(None, streamer.stop)
+ # streamer.stop_threads(["process_thread"])
+
+ # Force CUDA driver to flush all residual stream contexts
+ if torch.cuda.is_available():
+ torch.cuda.synchronize()
+ torch.cuda.empty_cache()
+ if hasattr(torch.cuda, "ipc_collect"):
+ torch.cuda.ipc_collect()
+ gc.collect()
if streamer.config.DEBUG_FLAG:
print(f"--- CLEANUP | Stream '{name}' stopped and removed. ---")
@@ -362,6 +460,11 @@ async def stop_all_streams(request: Request):
# Offload to executor to keep the API responsive
loop = asyncio.get_event_loop()
await loop.run_in_executor(None, streamer.stop)
+ streamer.stop_threads(["process_thread"])
+
+ gc.collect()
+ if torch.cuda.is_available():
+ torch.cuda.empty_cache()
return {
"status": "success",
@@ -448,7 +551,7 @@ async def get_model_classes():
return {"classes": classes}
# Fallback structure matching your old format if no model is loaded
- default_classes = ["class0"]
+ default_classes = ["drone"]
return {"classes": default_classes}
diff --git a/fastapi/nginx.conf b/fastapi/nginx.conf
index debe316..5b5c2e0 100644
--- a/fastapi/nginx.conf
+++ b/fastapi/nginx.conf
@@ -18,7 +18,7 @@ http {
proxy_set_header Connection "keep-alive";
# Absolute zero buffering
- proxy_buffering off; # DISBALE Nginx buffering
+ proxy_buffering off; # DISABLE Nginx buffering
proxy_request_buffering off;
proxy_cache off; # Ensure no caching of the video
tcp_nodelay on;
diff --git a/fastapi/requirements.txt b/fastapi/requirements.txt
index ae524f7..4489546 100644
--- a/fastapi/requirements.txt
+++ b/fastapi/requirements.txt
@@ -1,173 +1,225 @@
-annotated-doc==0.0.4 \
- --hash=sha256:571ac1dc6991c450b25a9c2d84a3705e2ae7a53467b5d111c24fa8baabbed320 \
- --hash=sha256:fbcda96e87e9c92ad167c2e53839e57503ecfda18804ea28102353485033faa4
-annotated-types==0.7.0 \
- --hash=sha256:1f02e8b43a8fbbc3f3e0d4f0f4bfc8131bcb4eebe8849b8e5c773f3a1c582a53 \
- --hash=sha256:aff07c09a53a08bc8cfccb9c85b05f1aa9a2a6f23728d790723543408344ce89
-anyio==4.13.0 \
- --hash=sha256:08b310f9e24a9594186fd75b4f73f4a4152069e3853f1ed8bfbf58369f4ad708 \
- --hash=sha256:334b70e641fd2221c1505b3890c69882fe4a2df910cba14d97019b90b24439dc
-av==17.0.1 \
- --hash=sha256:09b1f1601cc4a4d9e616d197b345c363ba6abfe567cb3d6b18e45516126692b6 \
- --hash=sha256:1229e879f4b6431bc00f69d7f8891fe9a683b0a6e0e009e6c98eb7e449f0383d \
- --hash=sha256:1d33871742d1e71562db3c8e752cacc5a62766d7efc3ae408bff1c3e26ebb46e \
- --hash=sha256:3a3f33bbfed2bcc65be37941bfeb6cc20bbe9cb7afc4ef1ac8d330972df098f9 \
- --hash=sha256:3d0a7d45d9599bf9df9f8249827113d4f36df1cd6b5356227b997f0552dbc98e \
- --hash=sha256:3d3a36204cb1f1e7691e6446afa8d6b7097b09946dae732c71c5d05ce09e506e \
- --hash=sha256:3ed6bcd7021fe55832f95b8ef78dd01a4cb21faf3cd71f1e1bf4f20bf100b278 \
- --hash=sha256:42d6745d30a410ec9b22aef79a52a7ab5a001eb8f5adfd952946606a30983318 \
- --hash=sha256:4744837f4116964280bcc72285e3cdd51361e98a696205aadd924203440ef511 \
- --hash=sha256:50e58a473d65ea29b645e45c9fd8518a6783737135683ecc40571a91592bdfe4 \
- --hash=sha256:50f9dd53a8ebef77606dca3b21710f660f9a6478484e79b9abda7c787b4f2403 \
- --hash=sha256:8270634c409f8efc9a24216e5dd90313d873b26ea4b5f172b14de52cbd15121c \
- --hash=sha256:985c21095bfb9c4bb7ba362fbef7bf0194bd72b1d7d3c46e30d1f47c5d38b4df \
- --hash=sha256:987f4f46ceae4da6c614dcbd2b8149be9dbf680c3bb7a6841c58af9cff4d9230 \
- --hash=sha256:9acd0b6a6e02af2b37f63d97a03ee2c47936d58e82425c3cd075a95245937c59 \
- --hash=sha256:9af524e8632a54032e361d6b88895bd3e7c6212ca560de60f5ccc525323c764c \
- --hash=sha256:a87a42c36e29f75e7dff7281944f2a6876a2c8875e225ccbf6c1ae62748b4caa \
- --hash=sha256:b87b98afe971cde123953073bc9c95ab0b7efd2ecc082dd2dbd11f9d9abf190e \
- --hash=sha256:c58c71bffd9383908c85695ac61d3184c668accb04a5bd1b262e0fb8d09f60a5 \
- --hash=sha256:d97f54e55b18a74912f479c1978aadd1341d38d892dee95bb5c2f2dccfa72f32 \
- --hash=sha256:e6eee84afa48d0e9321047cd3e4facd44b401493f6bdc753e2e1d1e7c9e6d13e \
- --hash=sha256:f585358fe0127990aea7887e940de4cdd745a2770605c31e54b2418fd0fdd8bd \
- --hash=sha256:f63b30067e6d88a3cce0d73d01ecfc0e6f091ad2bcf689db5dc305b0b4e8348c \
- --hash=sha256:fbcbd4aa43bca6a8691816283112d1659a27f407bbeb66d1397023691339f5d4
-certifi==2026.2.25 \
- --hash=sha256:027692e4402ad994f1c42e52a4997a9763c646b73e4096e4d5d6db8af1d6f0fa \
- --hash=sha256:e887ab5cee78ea814d3472169153c2d12cd43b14bd03329a39a9c6e2e80bfba7
-charset-normalizer==3.4.7 \
- --hash=sha256:007d05ec7321d12a40227aae9e2bc6dca73f3cb21058999a1df9e193555a9dcc \
- --hash=sha256:03853ed82eeebbce3c2abfdbc98c96dc205f32a79627688ac9a27370ea61a49c \
- --hash=sha256:07d9e39b01743c3717745f4c530a6349eadbfa043c7577eef86c502c15df2c67 \
- --hash=sha256:08e721811161356f97b4059a9ba7bafb23ea5ee2255402c42881c214e173c6b4 \
- --hash=sha256:0c96c3b819b5c3e9e165495db84d41914d6894d55181d2d108cc1a69bfc9cce0 \
- --hash=sha256:0ea948db76d31190bf08bd371623927ee1339d5f2a0b4b1b4a4439a65298703c \
- --hash=sha256:0f7eb884681e3938906ed0434f20c63046eacd0111c4ba96f27b76084cd679f5 \
- --hash=sha256:12a6fff75f6bc66711b73a2f0addfc4c8c15a20e805146a02d147a318962c444 \
- --hash=sha256:12d8baf840cc7889b37c7c770f478adea7adce3dcb3944d02ec87508e2dcf153 \
- --hash=sha256:14265bfe1f09498b9d8ec91e9ec9fa52775edf90fcbde092b25f4a33d444fea9 \
- --hash=sha256:16d971e29578a5e97d7117866d15889a4a07befe0e87e703ed63cd90cb348c01 \
- --hash=sha256:177a0ba5f0211d488e295aaf82707237e331c24788d8d76c96c5a41594723217 \
- --hash=sha256:1a87ca9d5df6fe460483d9a5bbf2b18f620cbed41b432e2bddb686228282d10b \
- --hash=sha256:1c2a768fdd44ee4a9339a9b0b130049139b8ce3c01d2ce09f67f5a68048d477c \
- --hash=sha256:1c2aed2e5e41f24ea8ef1590b8e848a79b56f3a5564a65ceec43c9d692dc7d8a \
- --hash=sha256:1dc8b0ea451d6e69735094606991f32867807881400f808a106ee1d963c46a83 \
- --hash=sha256:1efde3cae86c8c273f1eb3b287be7d8499420cf2fe7585c41d370d3e790054a5 \
- --hash=sha256:202389074300232baeb53ae2569a60901f7efadd4245cf3a3bf0617d60b439d7 \
- --hash=sha256:203104ed3e428044fd943bc4bf45fa73c0730391f9621e37fe39ecf477b128cb \
- --hash=sha256:2257141f39fe65a3fdf38aeccae4b953e5f3b3324f4ff0daf9f15b8518666a2c \
- --hash=sha256:298930cec56029e05497a76988377cbd7457ba864beeea92ad7e844fe74cd1f1 \
- --hash=sha256:2cd4a60d0e2fb04537162c62bbbb4182f53541fe0ede35cdf270a1c1e723cc42 \
- --hash=sha256:2d6eb928e13016cea4f1f21d1e10c1cebd5a421bc57ddf5b1142ae3f86824fab \
- --hash=sha256:2fe249cb4651fd12605b7288b24751d8bfd46d35f12a20b1ba33dea122e690df \
- --hash=sha256:30b8d1d8c52a48c2c5690e152c169b673487a2a58de1ec7393196753063fcd5e \
- --hash=sha256:320ade88cfb846b8cd6b4ddf5ee9e80ee0c1f52401f2456b84ae1ae6a1a5f207 \
- --hash=sha256:3534e7dcbdcf757da6b85a0bbf5b6868786d5982dd959b065e65481644817a18 \
- --hash=sha256:36836d6ff945a00b88ba1e4572d721e60b5b8c98c155d465f56ad19d68f23734 \
- --hash=sha256:38c0109396c4cfc574d502df99742a45c72c08eff0a36158b6f04000043dbf38 \
- --hash=sha256:3946fa46a0cf3e4c8cb1cc52f56bb536310d34f25f01ca9b6c16afa767dab110 \
- --hash=sha256:3bec022aec2c514d9cf199522a802bd007cd588ab17ab2525f20f9c34d067c18 \
- --hash=sha256:3c9a494bc5ec77d43cea229c4f6db1e4d8fe7e1bbffa8b6f0f0032430ff8ab44 \
- --hash=sha256:3dce51d0f5e7951f8bb4900c257dad282f49190fdbebecd4ba99bcc41fef404d \
- --hash=sha256:3dedcc22d73ec993f42055eff4fcfed9318d1eeb9a6606c55892a26964964e48 \
- --hash=sha256:4042d5c8f957e15221d423ba781e85d553722fc4113f523f2feb7b188cc34c5e \
- --hash=sha256:481551899c856c704d58119b5025793fa6730adda3571971af568f66d2424bb5 \
- --hash=sha256:4dc1e73c36828f982bfe79fadf5919923f8a6f4df2860804db9a98c48824ce8d \
- --hash=sha256:4e5163c14bffd570ef2affbfdd77bba66383890797df43dc8b4cc7d6f500bf53 \
- --hash=sha256:511ef87c8aec0783e08ac18565a16d435372bc1ac25a91e6ac7f5ef2b0bff790 \
- --hash=sha256:532bc9bf33a68613fd7d65e4b1c71a6a38d7d42604ecf239c77392e9b4e8998c \
- --hash=sha256:54523e136b8948060c0fa0bc7b1b50c32c186f2fceee897a495406bb6e311d2b \
- --hash=sha256:5649fd1c7bade02f320a462fdefd0b4bd3ce036065836d4f42e0de958038e116 \
- --hash=sha256:56be790f86bfb2c98fb742ce566dfb4816e5a83384616ab59c49e0604d49c51d \
- --hash=sha256:5b77459df20e08151cd6f8b9ef8ef1f961ef73d85c21a555c7eed5b79410ec10 \
- --hash=sha256:5ed6ab538499c8644b8a3e18debabcd7ce684f3fa91cf867521a7a0279cab2d6 \
- --hash=sha256:6178f72c5508bfc5fd446a5905e698c6212932f25bcdd4b47a757a50605a90e2 \
- --hash=sha256:6370e8686f662e6a3941ee48ed4742317cafbe5707e36406e9df792cdb535776 \
- --hash=sha256:64f02c6841d7d83f832cd97ccf8eb8a906d06eb95d5276069175c696b024b60a \
- --hash=sha256:65bcd23054beab4d166035cabbc868a09c1a49d1efe458fe8e4361215df40265 \
- --hash=sha256:66671f93accb62ed07da56613636f3641f1a12c13046ce91ffc923721f23c008 \
- --hash=sha256:6696b7688f54f5af4462118f0bfa7c1621eeb87154f77fa04b9295ce7a8f2943 \
- --hash=sha256:6785f414ae0f3c733c437e0f3929197934f526d19dfaa75e18fdb4f94c6fb374 \
- --hash=sha256:67f6279d125ca0046a7fd386d01b311c6363844deac3e5b069b514ba3e63c246 \
- --hash=sha256:6c114670c45346afedc0d947faf3c7f701051d2518b943679c8ff88befe14f8e \
- --hash=sha256:6e0d51f618228538a3e8f46bd246f87a6cd030565e015803691603f55e12afb5 \
- --hash=sha256:6ed74185b2db44f41ef35fd1617c5888e59792da9bbc9190d6c7300617182616 \
- --hash=sha256:708838739abf24b2ceb208d0e22403dd018faeef86ddac04319a62ae884c4f15 \
- --hash=sha256:715479b9a2802ecac752a3b0efa2b0b60285cf962ee38414211abdfccc233b41 \
- --hash=sha256:733784b6d6def852c814bce5f318d25da2ee65dd4839a0718641c696e09a2960 \
- --hash=sha256:750e02e074872a3fad7f233b47734166440af3cdea0add3e95163110816d6752 \
- --hash=sha256:752a45dc4a6934060b3b0dab47e04edc3326575f82be64bc4fc293914566503e \
- --hash=sha256:7579e913a5339fb8fa133f6bbcfd8e6749696206cf05acdbdca71a1b436d8e72 \
- --hash=sha256:7641bb8895e77f921102f72833904dcd9901df5d6d72a2ab8f31d04b7e51e4e7 \
- --hash=sha256:7804338df6fcc08105c7745f1502ba68d900f45fd770d5bdd5288ddccb8a42d8 \
- --hash=sha256:80d04837f55fc81da168b98de4f4b797ef007fc8a79ab71c6ec9bc4dd662b15b \
- --hash=sha256:813c0e0132266c08eb87469a642cb30aaff57c5f426255419572aaeceeaa7bf4 \
- --hash=sha256:82b271f5137d07749f7bf32f70b17ab6eaabedd297e75dce75081a24f76eb545 \
- --hash=sha256:84c018e49c3bf790f9c2771c45e9313a08c2c2a6342b162cd650258b57817706 \
- --hash=sha256:8751d2787c9131302398b11e6c8068053dcb55d5a8964e114b6e196cf16cb366 \
- --hash=sha256:8778f0c7a52e56f75d12dae53ae320fae900a8b9b4164b981b9c5ce059cd1fcb \
- --hash=sha256:87fad7d9ba98c86bcb41b2dc8dbb326619be2562af1f8ff50776a39e55721c5a \
- --hash=sha256:8d828b6667a32a728a1ad1d93957cdf37489c57b97ae6c4de2860fa749b8fc1e \
- --hash=sha256:8e385e4267ab76874ae30db04c627faaaf0b509e1ccc11a95b3fc3e83f855c00 \
- --hash=sha256:92a0a01ead5e668468e952e4238cccd7c537364eb7d851ab144ab6627dbbe12f \
- --hash=sha256:94e1885b270625a9a828c9793b4d52a64445299baa1fea5a173bf1d3dd9a1a5a \
- --hash=sha256:a180c5e59792af262bf263b21a3c49353f25945d8d9f70628e73de370d55e1e1 \
- --hash=sha256:a277ab8928b9f299723bc1a2dabb1265911b1a76341f90a510368ca44ad9ab66 \
- --hash=sha256:a5fe03b42827c13cdccd08e6c0247b6a6d4b5e3cdc53fd1749f5896adcdc2356 \
- --hash=sha256:a6c5863edfbe888d9eff9c8b8087354e27618d9da76425c119293f11712a6319 \
- --hash=sha256:a89c23ef8d2c6b27fd200a42aa4ac72786e7c60d40efdc76e6011260b6e949c4 \
- --hash=sha256:adb2597b428735679446b46c8badf467b4ca5f5056aae4d51a19f9570301b1ad \
- --hash=sha256:ae196f021b5e7c78e918242d217db021ed2a6ace2bc6ae94c0fc596221c7f58d \
- --hash=sha256:ae89db9e5f98a11a4bf50407d4363e7b09b31e55bc117b4f7d80aab97ba009e5 \
- --hash=sha256:aed52fea0513bac0ccde438c188c8a471c4e0f457c2dd20cdbf6ea7a450046c7 \
- --hash=sha256:aef65cd602a6d0e0ff6f9930fcb1c8fec60dd2cfcb6facaf4bdb0e5873042db0 \
- --hash=sha256:af21eb4409a119e365397b2adbaca4c9ccab56543a65d5dbd9f920d6ac29f686 \
- --hash=sha256:b14b2d9dac08e28bb8046a1a0434b1750eb221c8f5b87a68f4fa11a6f97b5e34 \
- --hash=sha256:bb6d88045545b26da47aa879dd4a89a71d1dce0f0e549b1abcb31dfe4a8eac49 \
- --hash=sha256:bb8cc7534f51d9a017b93e3e85b260924f909601c3df002bcdb58ddb4dc41a5c \
- --hash=sha256:bc17a677b21b3502a21f66a8cc64f5bfad4df8a0b8434d661666f8ce90ac3af1 \
- --hash=sha256:bd6c2a1c7573c64738d716488d2cdd3c00e340e4835707d8fdb8dc1a66ef164e \
- --hash=sha256:bd9b23791fe793e4968dba0c447e12f78e425c59fc0e3b97f6450f4781f3ee60 \
- --hash=sha256:c03a41a8784091e67a39648f70c5f97b5b6a37f216896d44d2cdcb82615339a0 \
- --hash=sha256:c0f081d69a6e58272819b70288d3221a6ee64b98df852631c80f293514d3b274 \
- --hash=sha256:c35abb8bfff0185efac5878da64c45dafd2b37fb0383add1be155a763c1f083d \
- --hash=sha256:c36c333c39be2dbca264d7803333c896ab8fa7d4d6f0ab7edb7dfd7aea6e98c0 \
- --hash=sha256:c45e9440fb78f8ddabcf714b68f936737a121355bf59f3907f4e17721b9d1aae \
- --hash=sha256:c593052c465475e64bbfe5dbd81680f64a67fdc752c56d7a0ae205dc8aeefe0f \
- --hash=sha256:cdd68a1fb318e290a2077696b7eb7a21a49163c455979c639bf5a5dcdc46617d \
- --hash=sha256:ce3412fbe1e31eb81ea42f4169ed94861c56e643189e1e75f0041f3fe7020abe \
- --hash=sha256:cf1493cd8607bec4d8a7b9b004e699fcf8f9103a9284cc94962cb73d20f9d4a3 \
- --hash=sha256:cf29836da5119f3c8a8a70667b0ef5fdca3bb12f80fd06487cfa575b3909b393 \
- --hash=sha256:d4a48e5b3c2a489fae013b7589308a40146ee081f6f509e047e0e096084ceca1 \
- --hash=sha256:d560742f3c0d62afaccf9f41fe485ed69bd7661a241f86a3ef0f0fb8b1a397af \
- --hash=sha256:d6038d37043bced98a66e68d3aa2b6a35505dc01328cd65217cefe82f25def44 \
- --hash=sha256:d61f00a0869d77422d9b2aba989e2d24afa6ffd552af442e0e58de4f35ea6d00 \
- --hash=sha256:d635aab80466bc95771bb78d5370e74d36d1fe31467b6b29b8b57b2a3cd7d22c \
- --hash=sha256:dca4bbc466a95ba9c0234ef56d7dd9509f63da22274589ebd4ed7f1f4d4c54e3 \
- --hash=sha256:dd915403e231e6b1809fe9b6d9fc55cf8fb5e02765ac625d9cd623342a7905d7 \
- --hash=sha256:e044c39e41b92c845bc815e5ae4230804e8e7bc29e399b0437d64222d92809dd \
- --hash=sha256:e060d01aec0a910bdccb8be71faf34e7799ce36950f8294c8bf612cba65a2c9e \
- --hash=sha256:e1421b502d83040e6d7fb2fb18dff63957f720da3d77b2fbd3187ceb63755d7b \
- --hash=sha256:e17b8d5d6a8c47c85e68ca8379def1303fd360c3e22093a807cd34a71cd082b8 \
- --hash=sha256:e5f4d355f0a2b1a31bc3edec6795b46324349c9cb25eed068049e4f472fb4259 \
- --hash=sha256:e712b419df8ba5e42b226c510472b37bd57b38e897d3eca5e8cfd410a29fa859 \
- --hash=sha256:e74327fb75de8986940def6e8dee4f127cc9752bee7355bb323cc5b2659b6d46 \
- --hash=sha256:e80c8378d8f3d83cd3164da1ad2df9e37a666cdde7b1cb2298ed0b558064be30 \
- --hash=sha256:e8ac484bf18ce6975760921bb6148041faa8fef0547200386ea0b52b5d27bf7b \
- --hash=sha256:eca9705049ad3c7345d574e3510665cb2cf844c2f2dcfe675332677f081cbd46 \
- --hash=sha256:ed065083d0898c9d5b4bbec7b026fd755ff7454e6e8b73a67f8c744b13986e24 \
- --hash=sha256:edac0f1ab77644605be2cbba52e6b7f630731fc42b34cb0f634be1a6eface56a \
- --hash=sha256:effc3f449787117233702311a1b7d8f59cba9ced946ba727bdc329ec69028e24 \
- --hash=sha256:f22dec1690b584cea26fade98b2435c132c1b5f68e39f5a0b7627cd7ae31f1dc \
- --hash=sha256:f495a1652cf3fbab2eb0639776dad966c2fb874d79d87ca07f9d5f059b8bd215 \
- --hash=sha256:f496c9c3cc02230093d8330875c4c3cdfc3b73612a5fd921c65d39cbcef08063 \
- --hash=sha256:f59099f9b66f0d7145115e6f80dd8b1d847176df89b234a5a6b3f00437aa0832 \
- --hash=sha256:f59ad4c0e8f6bba240a9bb85504faa1ab438237199d4cce5f622761507b8f6a6 \
- --hash=sha256:fbccdc05410c9ee21bbf16a35f4c1d16123dcdeb8a1d38f33654fa21d0234f79 \
- --hash=sha256:fea24543955a6a729c45a73fe90e08c743f0b3334bbf3201e6c4bc1b0c7fa464
-click==8.3.2 \
- --hash=sha256:14162b8b3b3550a7d479eafa77dfd3c38d9dc8951f6f69c78913a8f9a7540fd5 \
- --hash=sha256:1924d2c27c5653561cd2cae4548d1406039cb79b858b747cfea24924bbc1616d
+annotated-doc==0.0.5 \
+ --hash=sha256:117bac03a25ede5df5440e855b32d556049ca169ead221505badf432fed4b101 \
+ --hash=sha256:c7e58ce09192557605d8bbd92836d7e1d520ac9580096042c0bfd197efacf1bb
+annotated-types==0.8.0 \
+ --hash=sha256:13b2beaad985e05e2d6407ee4c4f35590b11f8d693a258a561055cac8f64cab7 \
+ --hash=sha256:f072f4d804ea359e4eaf198b1af7a8b0943881a87f31bb764f8bf219bb9419e0
+anyio==4.15.1 \
+ --hash=sha256:6152fdbbf9a77fdec97731721bebf7c4c44f7c29b424b0065826173efc7ed101 \
+ --hash=sha256:9f28306018cbd6d329e64a36d58256edff76dd996fe423bc957326e578b82a94
+av==17.1.0 \
+ --hash=sha256:1284addf3c0dd939887a9722dc30df2241a97471ad52c3c507e31583ae22ff02 \
+ --hash=sha256:1370b11a697eb3f2555906f8ab3519b0cfe48425d7830a3996ad42e6bffafda5 \
+ --hash=sha256:19264c9bb4bee404accc7ce9ec461f2044b7f577a70234d29aafde31ed17de46 \
+ --hash=sha256:19c84fd72af5ef81a20f18fbc6f9aedff9e1455e53a7062c1d4c95926d73da4e \
+ --hash=sha256:22dff0ae582d10ef08c75c2150a4fd27cfc26653b54930c7c27b9f7b3aa20723 \
+ --hash=sha256:3453b06075c7bb973fdb6de52563f7692ff05cbc64c0bb45f4fd6e8709131f2f \
+ --hash=sha256:3dcd41e53f53f9a3260751d9c3c11d34e93d70d61e506c81f13dbc1e3606e07b \
+ --hash=sha256:43ebbe977f19a7f2d2bd1a4e119675a0b15e05852cf7309846b6ab922ba7ffe9 \
+ --hash=sha256:5327807c1219293803ef0c5d1578ff3ae1cf638c09e5998962026e1a554ec240 \
+ --hash=sha256:58f7593726437cda5bd19793027e027768450b5c4a594777bf487798a33db702 \
+ --hash=sha256:5df5c1172ef1cf65a1529d612f7da7798ce2cf82c1ff7212466b538a6cc7214c \
+ --hash=sha256:6a20658ec7d96a70e14b1196eff00b7cdd8831ac3b99868e16b8ba8b24090847 \
+ --hash=sha256:6c9b71fe5c0c5a8d303b1588d4d8ce9397d6b023f467cfef95000ba1f75507fa \
+ --hash=sha256:7f1e71ff621b66253333926f948e00faae11d855b2442133c65128bca64cdeb3 \
+ --hash=sha256:90c49bc9608377d01e82e747377505419a229464873341db18202d5dddecce5a \
+ --hash=sha256:9514cfda85180554c430695282faf4be3ffdf95775d8519733821244eecb58e0 \
+ --hash=sha256:ad7b4aa011093324b7118245f50ac6db244cfe9900d4072508a5245a2b0d3f41 \
+ --hash=sha256:b41647e42884bf543b8e8d0a1dabd4d1b006c99183eb1a2d7afc5b01f73eeff4 \
+ --hash=sha256:bbab058bd965309f39962e53caac8126987c68c0be094fc4f9427e5615b0218f \
+ --hash=sha256:bff8896454b38fcb785a70e5ae0485d7021cb776303a5849393128a30b8f850b \
+ --hash=sha256:cc5a5247622cb77e24c342364eb68f88c1442ddfaab60c1f1f483359d3cc7879 \
+ --hash=sha256:e1c90f85cd7431ede95b11e8e711571a896ebea433f298849c2c0f1594c8d86e \
+ --hash=sha256:ec630be6321b04e317862f6082e84812bbd801e55a3c2298312e3fc8a0a4af4f \
+ --hash=sha256:ee98534242a74da847af78624779ac5a3177dc7c69f956a4da9e6f0fdb37d7f6 \
+ --hash=sha256:efe9b1397300b67b644ad220c89df4892a76f2debe70f16bae1749fa20526e63 \
+ --hash=sha256:f997e3351bdf51127c07a74e21741a2996e9230cbeb2d81c14acde761b116c9c \
+ --hash=sha256:f9a65d1f48b818323fb411e80358f89d77dec340b01d27c6b2dfbb9cbf4b779f \
+ --hash=sha256:fa64e1f1500d01c4a98e7a41dc1a9a35fb4dfe71f5de0389264ec1192200c76a \
+ --hash=sha256:ff457ed419348e5b8e8c811d341389b052c5e4d5839da3794d019b125b9fe830 \
+ --hash=sha256:ffbd78d73d2c9bf31e9a007c992faec3991428b2941a3b085b84fb82e8c32d19
+certifi==2026.7.22 \
+ --hash=sha256:62f22742b58a1a33014a2b6b706588a8d7e2a88ae7bd1a6ebe8c992928483775 \
+ --hash=sha256:741e2c3b351ddf169a738da9f2c048608ff7f2c5cc02f1ebc6b118bb090d5d55
+charset-normalizer==3.5.1 \
+ --hash=sha256:00668ebb0609751758682eb0b5857e7c35b9f00e84dfdef062e103244ec94d45 \
+ --hash=sha256:012a22b88a77ca2e59b98ac5889b0deb604147666032f45e6d6e217634d2550d \
+ --hash=sha256:01e93745f7f219b703b60ba7afead36cfc4242782be5af484673fc500df12da5 \
+ --hash=sha256:04368edf83514385ffc3e1cfd4546e595f4f1272dd23ba437a93a9cc3741d47b \
+ --hash=sha256:0722590aabf9dc6a6c0343d523c05458fa2b5047dbe6302fd526bb570600753f \
+ --hash=sha256:07ffd07412fc5d5e84cd8952acf9ff7e4ed7a708e69d1bada19d8ba91711353f \
+ --hash=sha256:09a7bba9f739468c8e78c36a75c33768e53cb1959fc638f510454c14683f00d5 \
+ --hash=sha256:0b2b1b3fa5670c127b246df1d0c059defd41f689a868a3b9d79df9b1cac42d22 \
+ --hash=sha256:0c6dfb5ca6723eeed15aa8e564a014d69fcb8812f94eef11fe3631e0508199f5 \
+ --hash=sha256:0d929fc574b4d6fd9e7c0f5c2ede8716a41911923aa7fa5fce38e0818aa4a1ac \
+ --hash=sha256:13e3afe97712e8887cd516e960c63f0b93122971e5b5e4b2622fe7701771e838 \
+ --hash=sha256:15f024313246a4ed976c60f440bb8d257815513a681d212ff74fd46f7d715a90 \
+ --hash=sha256:195ce897c6153c0700078142cf8efe3e6454ca4cf4357499e4078dfd83396626 \
+ --hash=sha256:19a3dd5aa73cef1c99687c4fc57db016a9c17104ae1185da88ba566a5d3bebe4 \
+ --hash=sha256:1d1c7a53a6c2103925cdd6d7229f8c567379f211c869793df679f2e9f738c369 \
+ --hash=sha256:1f5883d77fd409a261abb5dc8ccbe335720d798b1de4abb3b1d47ccbbc76b53b \
+ --hash=sha256:21b82d8082f6f5e7f456ef0bd16323d08de1266efbfeb476e64b2a91d1471a4e \
+ --hash=sha256:252d099029bcbea642f2a06c4ed5046bdf8b5a8150b64afa5e027e88b106e5ee \
+ --hash=sha256:256dd4d85d9e4dc595e2bc983c980e73f62ddeb3165c58b4c3dfe78c5c8548c1 \
+ --hash=sha256:26422d45fd13551cf564c58932f7d72b4f58b93b0fcf18c35ba6be12b46bb102 \
+ --hash=sha256:2679de311c7946dde5d3b6f44941844133ff5c7cb86099c0061ab1e8901c20a8 \
+ --hash=sha256:29880d17a8eb0b5cfdfd8944b468322928059aa35f1f5fa8ff22b149ec0b42f8 \
+ --hash=sha256:2bced4061f000f7187254a02ad3433ae17eaf991747ceea2f478422590a5bba9 \
+ --hash=sha256:2e9cf9253119d8e5d111f05d71626786fd3d6193817316eab1ca088cdb8593cf \
+ --hash=sha256:2f06b7eae9dbe77fe1d644ca244dad508de8d302870a43f3c559b521270938a0 \
+ --hash=sha256:2f293479cce755c75f1697e87c409b7ae4c555c7dfecb6e988ad13abba943031 \
+ --hash=sha256:329fc3ccb63ad22d867d84c2adea759a64079a37ba4a343433b02c7a2816871e \
+ --hash=sha256:343fb4f2821043bd87095f7b08a1a181febc8e36ac64212143bbfd0a0e1bc235 \
+ --hash=sha256:3588e376b3ea2eea84976f67273d679f229e24c66dce7b82ae45aef04ff6e072 \
+ --hash=sha256:35aea775dc2bd5f54cd84a1cd2696cc3207c479cb9cf0bd346f0d343e4300ddb \
+ --hash=sha256:35fe081843b35aad20ffeccec3eeffbe637b15d14f3fb22cc1b59cd8ec17e93c \
+ --hash=sha256:36047af20e17097c3bb9476c2b7655f2f7aa51322c0ba58c07695bedf755a950 \
+ --hash=sha256:3617ac3cfd8b9888f145ad89dd6e692285834b0201c6074a5eeaad3fd4d668c2 \
+ --hash=sha256:366ec70f5547c640d3ce1985722490f23faf4eb5216a7eeba78277490e78dacb \
+ --hash=sha256:394fea06235c8543390050ed5f529187074b029fb027213f6c46ac11ab5d950e \
+ --hash=sha256:3d27167433c0d5f18dc850f07d0b3816221984fecdc405d6c157a6f0b8f8e9e6 \
+ --hash=sha256:3e5e1224c0a6a90e05843e07adfec669edebec17801c67072f51e59561d63c0b \
+ --hash=sha256:41876ee62a3dddf48ff1121ad8f0798032aa03f2fd35f21f34a4cab14f18d8d2 \
+ --hash=sha256:433c5a81eade63b47e522303bad236f59dba55ea6951746f5558355eeed8c75d \
+ --hash=sha256:4582c27e8c889d64811987b5967fbd3ae0c823fe1fd933b543d55ac20bb475fa \
+ --hash=sha256:485a0d363cafefcd2538a73c7c838daa2035f09b2c9f9b5e3133f80c6aeb84c2 \
+ --hash=sha256:494b70049a4d69aec6e8137c13af4cf8db8c9f9820a1392ac293b0dd2987a818 \
+ --hash=sha256:496846868fea80e479324862fa877f02411f2fd0f83b79ccee2607aa68b2a032 \
+ --hash=sha256:4abdc5f9ad448c1ecbfae2974b820535d6bc6e7eef63babbab3d81cf46968c71 \
+ --hash=sha256:4b599739b93b2cbeded49645ae3c8d1405c29ddfbceac1545c87a3f9580a9e96 \
+ --hash=sha256:4bea7f8ebe90bbd7f0e4a2de42ca6924ba23e3e76418c408ff82f1d46fabd687 \
+ --hash=sha256:4c4fb141a727957c93edfe5c32a26ceb6b5f6461d67146e2d39f51e16170bea8 \
+ --hash=sha256:4c9548dc78002099910abaebc0a72ac58b7d30931869e0351c09b507dff4ece3 \
+ --hash=sha256:4d26f14f041e83dd8edfd61f4cd4fa7285d31798b5bf1f28e70c367ba6c41d61 \
+ --hash=sha256:4f298bdadb8f0b9e5672877f647d1be9373ef5320c9e2f049795e26cad28b6a9 \
+ --hash=sha256:52ec005752a56ae79547a05c0139ca2501a0c866390b6115008456b9f0e7cde1 \
+ --hash=sha256:55261ac0d2941c42f196dd576f543d87a8ee03cd6f5e30dfb4d807b2e3b9121a \
+ --hash=sha256:56490c595a28b1bb27dfc583e816152a9767721ef58b2c03b13f954d2f707420 \
+ --hash=sha256:58d3e12c88e0950bca850ae1f7c256055c097639c2edb9eb123af9807d8b15e4 \
+ --hash=sha256:58d4aa13a59c969dbfdf9e6a9560e242cbfd9e8a8f50c2747714df1a423adf65 \
+ --hash=sha256:59171c6e45bf07d0d5cab3b0bf81d945035530f6873398b3b531c31184d46663 \
+ --hash=sha256:5b6d1386bf0096d26d3a863dc0a487a5b4eb9aa93cf5ba69683d29dde6b9d60f \
+ --hash=sha256:5c0ea61a470e070686aa30892fed79e297d2c8d0ab46b8bcdf027d38c51da591 \
+ --hash=sha256:5c84bec0ab5ae0c64bfe73a7d2adcb5ce73b467523fc27fd6a28ab2aa6cbe35a \
+ --hash=sha256:5ca0555312ae2fe82715cada7fac375530c2f3349e1eaa1bcb33d0283ac79a18 \
+ --hash=sha256:5d8531a6569d025f68e2321e7638fb7978f23db58e5f69f56913837aae03816e \
+ --hash=sha256:5e2d0e146dcb57034f8b97dc58d2d512cb90aba253960ce449f695fec6a82c6f \
+ --hash=sha256:5fc45d653ea8c9a20479167e11d4a0f8cb2fa3470737ab6f9c827532313187b7 \
+ --hash=sha256:6117b84ea48435e5356dc737f5121485c30920ba43375fa7b434fd753df0eac3 \
+ --hash=sha256:6199d5606e2bbf2b096cf64d03f8b6790c91081d5ac866b8e7bb6422738cc60c \
+ --hash=sha256:62b55f6722735a6c472f88361cde6640608773d9443cebdbb51abf436a1fcdd3 \
+ --hash=sha256:687c9ca3035544b113bea2055e180af96fb63c0c476e22a9180f51925186e7b7 \
+ --hash=sha256:6b7430cf5728e68f6c462254009a6ef4086e1bea43cf2f57aa9c55fb4f50ff96 \
+ --hash=sha256:6ba32c4d2abf1d2fe7cf27d280f4cca5664233b0f885549c7761719eb977f486 \
+ --hash=sha256:6c9cdde8becb25a7fde49924511aa2644d6f8081cc8df8e9452724303348d8e3 \
+ --hash=sha256:6df0ec430f9a831772c23ca5a224cba36517a58a84bb32c32bb59a9fa67c47f6 \
+ --hash=sha256:6e2912d4babbc65196ac13c2f53468dc57fb8b9c25ef913e8c59ddf7c6dc0e1b \
+ --hash=sha256:6e5e4d73d588ca5ed09df1b7dcd1b203d1df3c542e3f50d126c947d432b10731 \
+ --hash=sha256:70055ff39b97c99e7ae40ea3e393fb62aa2e44dbd9b29f8d14f42fb0025c3959 \
+ --hash=sha256:706bfd38730a5ac7a365793269a00f4e988178cec121391f4248d84ad8c972e9 \
+ --hash=sha256:7235dc28fc6dd9d832ac7c7bce95367dedb85929f17368a0c2bee1e080b9acbf \
+ --hash=sha256:774d157f112367ff4abd29019f38f023c24e00e56edc7829c20e358a5a913ad8 \
+ --hash=sha256:77efcff2b23071c349402ac1066667a3d011f62398d81408c9b88ad991747c9e \
+ --hash=sha256:789b8982559ae28dad2356519f841655756cdcd96616410590ae0b17454ee64f \
+ --hash=sha256:7ac76cf9afd34929d76eb7fcb63be476a4853d8a96f0dcf2d0db68a0cbdf9885 \
+ --hash=sha256:7c0c10730342b0c9b35dd1d619beb8214e520bd96a1f870f452680b238aab3e0 \
+ --hash=sha256:823f82903d189af463d7df250ef1f7f696f3cee08cc8d91deb565e8d425f6506 \
+ --hash=sha256:838648accb3a7fd9803fd45c87bce8509648eb0c11bc34e216141300977244f2 \
+ --hash=sha256:854066be00447fa8de2ccbbe893e2ffc4b123ef16d897af794c1e18bd4a714b0 \
+ --hash=sha256:85d5855daafc240cc045c026d7a15fd198a09b0fc8ff6f5ecbb5297b509cb11e \
+ --hash=sha256:85de3134b5379856e323ba37c19c9256d39425f7b76a63af52b09fb4664c2e8f \
+ --hash=sha256:87e4f41d375c0b9be2fb5251aee4b8a689169e134535aed81bf085c3b647451e \
+ --hash=sha256:88ca277405c2d3b71c4e1c2ee0e7966e807bcba86a69d11e19ba199d18ae4491 \
+ --hash=sha256:88e85ab89cb822c1e635f51d6d32e488f94e002e70e2f492bdb8b945543f345a \
+ --hash=sha256:8ac8c94b6539074e0f40899301273ac8402b9b3e01c7b7ba269ff30340aaaf20 \
+ --hash=sha256:8fe532b3c966d1fb794e0698e4589d0444017ae77fc0b31edea13c0e35bcc449 \
+ --hash=sha256:9085f87b0e38a2b92b8923059b4e8789fe40d9279712d15dcc670048d77079af \
+ --hash=sha256:90b7481fb62fbe172c558bc6fd1c4c98d82004a54a7551f20e11ac9bf0b8708c \
+ --hash=sha256:92caef967d287a407085d61176fce4012b1dd62daed4eb6d5ceb26d3d2538712 \
+ --hash=sha256:9362dd90aa7dab48c0054a21187791ccf05473f7dba5d92b8033ae62164675e7 \
+ --hash=sha256:94d78ecec2605a8d0398b0f365d5f12a63248438516f5dac536a5eff7337df4a \
+ --hash=sha256:94fbf1c0c6cc0d3d5e50f9a9313a8cdca90dd696d34b381cd1704f8c9e939f20 \
+ --hash=sha256:950f23cb393f85543777b0433f082cddd25b51ab398eac7971146495679efe5f \
+ --hash=sha256:96eefc178f8636b9c760c5829345307fd81cfae9ab1e80997dbddeb0f54ee9a3 \
+ --hash=sha256:96fef3e886d6a9874b14f27fc193fbdc69d5d8035783d86aa4e1cea594e695f9 \
+ --hash=sha256:977cdbd483a9cff38179bea4fd754289a6f2195c7abd414aba85410b3e66cc5e \
+ --hash=sha256:978eab16f55b4ab2c2a745be9a0a840bf8f09a7f227d9c76eb30214d078865a5 \
+ --hash=sha256:994e883d17c559cdfd38c84003c8b27d25424a1077272a17e7cd27bfe0bf57b2 \
+ --hash=sha256:9ac4444d8d4fd4c4bd08bf451ed3167aa9e7ec6cdb41b648794f1d1103652e36 \
+ --hash=sha256:9b5db6052055d34d41230fb78d7c439c23dc536a9896f6cb039e8dd92cfc1263 \
+ --hash=sha256:9d9a0dc7cbe9bec24c3f767c9122c41fe5a1bc43f47cd099d00d393e09769de4 \
+ --hash=sha256:9dbdd9205662134957cf0c324f639bdc5031c0ca056e2369e238db75187c0f11 \
+ --hash=sha256:9eea3ab2597a5e65fe65296e2d6a84570845a6b55532d90333d740d48bbc850a \
+ --hash=sha256:a2028475ba855475b8b4d3cfeb4994269c967aea8b9892dfba907f4263a863a3 \
+ --hash=sha256:a3a370082ce34d0612f421e15fe011c53bb1feff21a26d06ad4fb244dab5a375 \
+ --hash=sha256:a545775cfe815855ea32d7c27731d79da358ef2055b4a25830231b1622dd18aa \
+ --hash=sha256:a5cbd90ecf0fc62e64726917ad083b73001f0563657a87ec3c0b504e277dc90d \
+ --hash=sha256:a6d095662e73e74f0a49988e0593373e243e3a52e27bfeea0a859e88acf4a0f5 \
+ --hash=sha256:a6dac12ff6b846103483683f60c5f8fee205121adc58ffd87e90a90a3af69e99 \
+ --hash=sha256:a951ad59cad9145664a730d3036b40b844e74d2d3683da40111463cd3a83845d \
+ --hash=sha256:aa1099b956fb795e686d073568f6dc002a0bb89765ea6d5b055dd7d9bf1b116c \
+ --hash=sha256:aa2bb0b37202dca27175591f761108b5d34096ade1191ffe4808bdf6b1571488 \
+ --hash=sha256:aae2ee51122d3ae968a3837d97dc24a0aeebb0dea23694422cd172bd30017cd6 \
+ --hash=sha256:ab743e9bc90c1f73552ec33e10e3331315acd2c397b36065b591b0181de533cc \
+ --hash=sha256:ac00177c4831ffa650f8609e4bdddd5fe09c03b1c0c47acece7e6ea20421598b \
+ --hash=sha256:ac13b004224fb341e1e25a1ed5e19d32f57cdb2a403e01f003b46f051a550f6f \
+ --hash=sha256:acaf604462bf330b0d07e7a07c1d6e4adac79e5fb13e9c5140590542cafacc00 \
+ --hash=sha256:ae31a1a1db2ee6cc2942fccaf695c934bc7f3db9f2133a3fef1f367cf1a4ab10 \
+ --hash=sha256:ae4a097991662cd4fff0ddc74e0fe7874f82e00042fa0ea00855645ed0c79598 \
+ --hash=sha256:aea996a6aba25260827c9ea511d1addfde2da9eb686ac961838509086188b7e6 \
+ --hash=sha256:b39b69b347e5e47a3b5b8cfc005c68c1ba347474e3960236c4944a8ecd174962 \
+ --hash=sha256:b54e7e13267d49ffbfe68e25b3cbd774dab38fa37238f71265e91b36146eb21c \
+ --hash=sha256:b9af956078716df40d985fb0dfeb2c2120c5ca92ba4ff4b388acfd01cdc14d08 \
+ --hash=sha256:ba2f37ee79e6338845261a3c5b1784e5d1acdff2c0785b284f1b633033d136ab \
+ --hash=sha256:ba501e667c17d8411f98e67a022d9604ef179aff0e459b7e292c796837c13573 \
+ --hash=sha256:baf3775a2635e5a11fbd5e4e64ee69c7e86875d224a5c72aca4c141064589a90 \
+ --hash=sha256:bb57753e36e4855b8ca375069482250a6246372331a3e4f3407eaebb007443f5 \
+ --hash=sha256:bd6c173f04743d483881bffa1478d5a4624475b8cd1d2194956a75548e191c18 \
+ --hash=sha256:be47f99644b208bff7766314013f9acf57b056b04191d570d68ad14022cf5b1d \
+ --hash=sha256:c010f5581d9c612804cc59fcf7b524b707fbcb72828551237ab545bb5c7034af \
+ --hash=sha256:c1dcc36dcb96abc02236e182d17e0f71430152a6c2c7447421da2d2dc144edea \
+ --hash=sha256:c428c6c31eb5f4277d7f8eccaf767fbd548ddd5ce3c8b4f4cbbfab3d96b5904c \
+ --hash=sha256:c658c50ac0c98cd755a2dd50b7977d3bca7df401dcc47fbdfa87db53ef7d4e8b \
+ --hash=sha256:c71fb0d56c920c269cd3e2e3fe7c610e3f1fdb21a6ce60efa6430ff63676cea6 \
+ --hash=sha256:c7b742bf31c88566b4bb6335a7f393bb322e580b6bb98df7bd0c25e6e3519ce8 \
+ --hash=sha256:cc0329df4caaceb950d2f580b5ac716a377f7059624a0bafaeaf8a218c6ed774 \
+ --hash=sha256:cc5d36d96478aa9c60654bd932525bf32964c62a7281eafdf16d85003a8d6004 \
+ --hash=sha256:ce854f5f478050ade5a238731c4ca985a7d3b3cb53ff600a9b5c3b689b5f0a7a \
+ --hash=sha256:ced3fdd71aaa83ce593746c2edb42b7a59cb4c19c8b5c407781c72e493aae55a \
+ --hash=sha256:cee5dd7c6fb5dd52a0fe2a740f9bc6e3593f5f8b1788bde49de02086f30182b2 \
+ --hash=sha256:cfa1c0cc3a8f9f53f1243a5a99ac36fd003880199383b37672e86ddda9cb07e2 \
+ --hash=sha256:d1ee1e296209fdce05b81b663250eefa02213a2da7b41bf26f7829b8ba3545aa \
+ --hash=sha256:d59b75732e9b6f27388e10c14b0259cc5f2e48c78627d185e6a177b58ad3cffe \
+ --hash=sha256:d63600d620ad0064c3a748b950ac5ea38a80190e5498532efefa4b7b3f1da1f3 \
+ --hash=sha256:dd732602a7009217f658d5863d12d79d373a4de0eebc111094bcdd3bb8e0a6cc \
+ --hash=sha256:e06efa066f7dbadbc84ebc126a97c452a6451dfcf589d89d788484949e1cf795 \
+ --hash=sha256:e199fb99720074809a7720f1c0b4d919eea8b87e88713e0f8f602f7bef543d9d \
+ --hash=sha256:e4b018dc5a0eee4676e38fe84a47a427816c590b93b55d9025274ec4d6ffc2dc \
+ --hash=sha256:e6621fb2a4988d6e53eedc455e5903e2679f3967b8acb3d639f1b63c14a2e893 \
+ --hash=sha256:e71c909f353863b2b89c83de2ebed71ea6d0df8a6ef65a128193c5e650766bef \
+ --hash=sha256:e90251c0c7bdd54a100a0dce3c07b7e637278c93af29dbf78ebb89a58c4bac7d \
+ --hash=sha256:e9fbdce1e47394b09bc9f26ab117dfc8d6491977a11d86f592bb42c779db2fda \
+ --hash=sha256:eb12fb2ba69ffa05f8695f61c69e591dc4b4a12ac3757ac8af8adb259bf56d17 \
+ --hash=sha256:eda059b6bc8bc0812d626fd91a7ce01bf583df0a61296eff390fd94141a34e30 \
+ --hash=sha256:f03ac127268b43ef4fe9e6ab6794a6794b49485a0cc0c1db79876d2f33f75bc7 \
+ --hash=sha256:f298e218441525d3794428b4c8b8fb8662c6d3ea79925d4807ee6b9a96a3bca5 \
+ --hash=sha256:f5542f9b941279d82d41eb0aa9f98eba36fe4df5c7086c651df7944935b37182 \
+ --hash=sha256:f6f7deae3feb4edfa2efaf7c574fe88cbf055038a6abdb40188e4fff66d5699f \
+ --hash=sha256:f9b1e28d0e8dbfa858abdba91d6b547beaf2df1a59bec6da6faae7b96a4991a9 \
+ --hash=sha256:f9f8405c2c758532c74fed975dbee57be1f31a6e865c031870c79a6ed3212ada \
+ --hash=sha256:fa48b1b63d639f9483e0633e092f5851e2348c352f1f9bb6c8182f87884ef876 \
+ --hash=sha256:fb78f6e7fcd8ad785d28cd577168bc1aaee827b25bb8755638f694794ea98f0a \
+ --hash=sha256:fbc597639158fd7c14d55e808718848319540f51b0e6746e3eefa59723a4a348 \
+ --hash=sha256:fce8cbd4997efeb450bd298b54f755dcdff18d496f7a5ddbb4867c6d7c88fdc3 \
+ --hash=sha256:fd0350afdc3aabd5576f60ea109228bd5538139713c7b094c5cd27c73a98bc6f \
+ --hash=sha256:fd0a274c0e5f9a21565cd9d3dd749b61f96b7aa1e20a93aa1ba4029518f2e5c0 \
+ --hash=sha256:fdb8a068947befafba9952162645dc2fecaeb400e64584829ed5e9b2fbe21a7f
+click==8.5.0 \
+ --hash=sha256:255bc9599cf7748b4b1a446ccc735421bd08a2ae529a8b88597d3de5664ee360 \
+ --hash=sha256:ba0d2089de75ea0310e2dde03160e6ca10009947fb95a182f9b54021bb272e34
+cloudpickle==3.1.2 \
+ --hash=sha256:7fda9eb655c9c230dab534f1983763de5835249750e85fbcef43aaa30a9a2414 \
+ --hash=sha256:9acb47f6afd73f60dc1df93bb801b472f05ff42fa6c84167d25cb206be1fbf4a
colorama==0.4.6 \
--hash=sha256:08695f5cb7ed6e0531a20572697297273c47b8cae5a63ffc6d6ed5c201be6e44 \
--hash=sha256:4f1d9991f5acc0ca119f9d443620b77f9d6b33703e51011c16baf57afb285fc6
@@ -232,8 +284,6 @@ contourpy==1.3.2 \
--hash=sha256:f26b383144cf2d2c29f01a1e8170f50dacf0eac02d64139dcd709a8ac4eb3cfe \
--hash=sha256:f939a054192ddc596e031e50bb13b657ce318cf13d264f095ce9db7dc6ae81c0 \
--hash=sha256:fd93cc7f3139b6dd7aab2f26a90dde0aa9fc264dbf70f6740d498a70b860b82c
-cucim==23.10.0 \
- --hash=sha256:2c061ad28c3c1fa67bb62260f0a556f354c8ec2d6e3811eece375ed896d62945
cuda-bindings==12.9.4 \
--hash=sha256:1f53a7f453d4b2643d8663d036bafe29b5ba89eb904c133180f295df6dc151e5 \
--hash=sha256:20f2699d61d724de3eb3f3369d57e2b245f93085cab44fd37c3bea036cea1a6f \
@@ -259,8 +309,8 @@ cuda-bindings==12.9.4 \
--hash=sha256:d80bffc357df9988dca279734bc9674c3934a654cab10cadeed27ce17d8635ee \
--hash=sha256:f69107389e6b9948969bfd0a20c4f571fd1aefcfb1d2e1b72cc8ba5ecb7918ab \
--hash=sha256:fda147a344e8eaeca0c6ff113d2851ffca8f7dfc0a6c932374ee5c47caa649c8
-cuda-pathfinder==1.5.3 \
- --hash=sha256:dff021123aedbb4117cc7ec81717bbfe198fb4e8b5f1ee57e0e084fec5c8577d
+cuda-pathfinder==1.8.1 \
+ --hash=sha256:ae0137ff9e56ea97499bcbf54f5f2778ec25f3266715ac86da192a795af982a8
cuda-toolkit[cudart]==12.8.2.0 \
--hash=sha256:79040e750f28b959415283ccc44cce541c07003553d675b8ff45760e0189aa71
cupy-cuda12x==13.6.0 \
@@ -288,9 +338,9 @@ defusedxml==0.7.1 \
exceptiongroup==1.3.1 \
--hash=sha256:8b412432c6055b0b7d14c310000ae93352ed6754f70fa8f7c34141f91c4e3219 \
--hash=sha256:a7a39a3bd276781e98394987d3a5701d0c4edffb633bb7a5144577f82c773598
-fastapi==0.136.0 \
- --hash=sha256:8793d44ec7378e2be07f8a013cf7f7aa47d6327d0dfe9804862688ec4541a6b4 \
- --hash=sha256:cf08e067cc66e106e102d9ba659463abfac245200752f8a5b7b1e813de4ff73e
+fastapi==0.141.1 \
+ --hash=sha256:bfb91aa2d334c61cb35ba9a116fc123b3d3df31640b801cf57a7a78ec3f603b3 \
+ --hash=sha256:e8822fc40db1e1858054d7a949a888695bc9bdce70139178e33bd2871a453ca1
fastrlock==0.8.3 \
--hash=sha256:001fd86bcac78c79658bac496e8a17472d64d558cd2227fdc768aa77f877fe40 \
--hash=sha256:04bb5eef8f460d13b8c0084ea5a9d3aab2c0573991c880c0a34a56bb14951d30 \
@@ -361,207 +411,244 @@ fastrlock==0.8.3 \
--hash=sha256:f2b84b2fe858e64946e54e0e918b8a0e77fc7b09ca960ae1e50a130e8fbc9af8 \
--hash=sha256:f68c551cf8a34b6460a3a0eba44bd7897ebfc820854e19970c52a76bf064a59f \
--hash=sha256:fcb50e195ec981c92d0211a201704aecbd9e4f9451aea3a6f71ac5b1ec2c98cf
-filelock==3.28.0 \
- --hash=sha256:4ed1010aae813c4ee8d9c660e4792475ee60c4a0ba76073ceaf862bd317e3ca6 \
- --hash=sha256:de9af6712788e7171df1b28b15eba2446c69721433fa427a9bee07b17820a9db
+filelock==3.32.6 \
+ --hash=sha256:3f16ecd0117feae0dfc147e8c62eb5daeccd8bd800378c3ddf416de9b4feb6b1 \
+ --hash=sha256:a3f55a18af3652a94d8f47d6055df434f254ca1d02ef2524850c6d249ca2512c
flatbuffers==25.12.19 \
--hash=sha256:7634f50c427838bb021c2d66a3d1168e9d199b0607e6329399f04846d42e20b4
-fonttools==4.62.1 \
- --hash=sha256:0aa72c43a601cfa9273bb1ae0518f1acadc01ee181a6fc60cd758d7fdadffc04 \
- --hash=sha256:0b3ae47e8636156a9accff64c02c0924cbebad62854c4a6dbdc110cd5b4b341a \
- --hash=sha256:12859ff0b47dd20f110804c3e0d0970f7b832f561630cd879969011541a464a9 \
- --hash=sha256:149f7d84afca659d1a97e39a4778794a2f83bf344c5ee5134e09995086cc2392 \
- --hash=sha256:1596aeaddf7f78e21e68293c011316a25267b3effdaccaf4d59bc9159d681b82 \
- --hash=sha256:19177c8d96c7c36359266e571c5173bcee9157b59cfc8cb0153c5673dc5a3a7d \
- --hash=sha256:1c5c25671ce8805e0d080e2ffdeca7f1e86778c5cbfbeae86d7f866d8830517b \
- --hash=sha256:1eecc128c86c552fb963fe846ca4e011b1be053728f798185a1687502f6d398e \
- --hash=sha256:268abb1cb221e66c014acc234e872b7870d8b5d4657a83a8f4205094c32d2416 \
- --hash=sha256:2d850f66830a27b0d498ee05adb13a3781637b1826982cd7e2b3789ef0cc71ae \
- --hash=sha256:2e7abd2b1e11736f58c1de27819e1955a53267c21732e78243fa2fa2e5c1e069 \
- --hash=sha256:403d28ce06ebfc547fbcb0cb8b7f7cc2f7a2d3e1a67ba9a34b14632df9e080f9 \
- --hash=sha256:40975849bac44fb0b9253d77420c6d8b523ac4dcdcefeff6e4d706838a5b80f7 \
- --hash=sha256:486f32c8047ccd05652aba17e4a8819a3a9d78570eb8a0e3b4503142947880ed \
- --hash=sha256:49a445d2f544ce4a69338694cad575ba97b9a75fff02720da0882d1a73f12800 \
- --hash=sha256:59b372b4f0e113d3746b88985f1c796e7bf830dd54b28374cd85c2b8acd7583e \
- --hash=sha256:5a648bde915fba9da05ae98856987ca91ba832949a9e2888b48c47ef8b96c5a9 \
- --hash=sha256:5f37df1cac61d906e7b836abe356bc2f34c99d4477467755c216b72aa3dc748b \
- --hash=sha256:6706d1cb1d5e6251a97ad3c1b9347505c5615c112e66047abbef0f8545fa30d1 \
- --hash=sha256:68959f5fc58ed4599b44aad161c2837477d7f35f5f79402d97439974faebfebe \
- --hash=sha256:6acb4109f8bee00fec985c8c7afb02299e35e9c94b57287f3ea542f28bd0b0a7 \
- --hash=sha256:7487782e2113861f4ddcc07c3436450659e3caa5e470b27dc2177cade2d8e7fd \
- --hash=sha256:7aa21ff53e28a9c2157acbc44e5b401149d3c9178107130e82d74ceb500e5056 \
- --hash=sha256:7bca7a1c1faf235ffe25d4f2e555246b4750220b38de8261d94ebc5ce8a23c23 \
- --hash=sha256:8d337fdd49a79b0d51c4da87bc38169d21c3abbf0c1aa9367eff5c6656fb6dae \
- --hash=sha256:8f8fca95d3bb3208f59626a4b0ea6e526ee51f5a8ad5d91821c165903e8d9260 \
- --hash=sha256:90365821debbd7db678809c7491ca4acd1e0779b9624cdc6ddaf1f31992bf974 \
- --hash=sha256:92bb00a947e666169c99b43753c4305fc95a890a60ef3aeb2a6963e07902cc87 \
- --hash=sha256:93c316e0f5301b2adbe6a5f658634307c096fd5aae60a5b3412e4f3e1728ab24 \
- --hash=sha256:942b03094d7edbb99bdf1ae7e9090898cad7bf9030b3d21f33d7072dbcb51a53 \
- --hash=sha256:9c125ffa00c3d9003cdaaf7f2c79e6e535628093e14b5de1dccb08859b680936 \
- --hash=sha256:9dde91633f77fa576879a0c76b1d89de373cae751a98ddf0109d54e173b40f14 \
- --hash=sha256:9e7863e10b3de72376280b515d35b14f5eeed639d1aa7824f4cf06779ec65e42 \
- --hash=sha256:a24decd24d60744ee8b4679d38e88b8303d86772053afc29b19d23bb8207803c \
- --hash=sha256:a5d8825e1140f04e6c99bb7d37a9e31c172f3bc208afbe02175339e699c710e1 \
- --hash=sha256:aa69d10ed420d8121118e628ad47d86e4caa79ba37f968597b958f6cceab7eca \
- --hash=sha256:ad5cca75776cd453b1b035b530e943334957ae152a36a88a320e779d61fc980c \
- --hash=sha256:b4e0fcf265ad26e487c56cb12a42dffe7162de708762db951e1b3f755319507d \
- --hash=sha256:b820fcb92d4655513d8402d5b219f94481c4443d825b4372c75a2072aa4b357a \
- --hash=sha256:bd13b7999d59c5eb1c2b442eb2d0c427cb517a0b7a1f5798fc5c9e003f5ff782 \
- --hash=sha256:bdfe592802ef939a0e33106ea4a318eeb17822c7ee168c290273cbd5fabd746c \
- --hash=sha256:c05557a78f8fa514da0f869556eeda40887a8abc77c76ee3f74cf241778afd5a \
- --hash=sha256:c22b1014017111c401469e3acc5433e6acf6ebcc6aa9efb538a533c800971c79 \
- --hash=sha256:c9b9e288b4da2f64fd6180644221749de651703e8d0c16bd4b719533a3a7d6e3 \
- --hash=sha256:d241cdc4a67b5431c6d7f115fdf63335222414995e3a1df1a41e1182acd4bcc7 \
- --hash=sha256:e54c75fd6041f1122476776880f7c3c3295ffa31962dc6ebe2543c00dca58b5d \
- --hash=sha256:e8514f4924375f77084e81467e63238b095abda5107620f49421c368a6017ed2 \
- --hash=sha256:ee91628c08e76f77b533d65feb3fbe6d9dad699f95be51cf0d022db94089cdc4 \
- --hash=sha256:ef46db46c9447103b8f3ff91e8ba009d5fe181b1920a83757a5762551e32bb68 \
- --hash=sha256:fa1d16210b6b10a826d71bed68dd9ec24a9e218d5a5e2797f37c573e7ec215ca
-fsspec==2026.3.0 \
- --hash=sha256:1ee6a0e28677557f8c2f994e3eea77db6392b4de9cd1f5d7a9e87a0ae9d01b41 \
- --hash=sha256:d2ceafaad1b3457968ed14efa28798162f1638dbb5d2a6868a2db002a5ee39a4
+fonttools==4.64.0 \
+ --hash=sha256:043f6c572bf236f2a76e762c25f841daea11e8fc03e78088d7be66e0c5b4e4c0 \
+ --hash=sha256:06b6409b868494556a831ae33b2d9a090476c37516b38d70f45a9720b460d423 \
+ --hash=sha256:08f172961e11f4eb4f80f2f20049e09b0ea8e044fa6d456fed8346eb8588f360 \
+ --hash=sha256:09657817b75575822bcd6098ef0ebf0386f34430839ee53109e70fd40a7f6539 \
+ --hash=sha256:1c3661324f3f0fa4539a32288a3e0711a5f3ccf020036e760bb558ae9811a16f \
+ --hash=sha256:1e4e84b47839d35be24dbf476845a34f2ccf99707b66df125c1c414d3e86d25d \
+ --hash=sha256:236e59bc7e2a63557a4d7b013f9cb9e28d9aebc45bc09f85e545e6bf091db626 \
+ --hash=sha256:2524a26f8fdb9051b0d778d052f5d238285ca9f91a7dc004514c7d6cf38d35f4 \
+ --hash=sha256:2730946ca8f12c356bd98eb9b2b095c8e761ed05bed5afb0d5b380cebe4f6370 \
+ --hash=sha256:2c42237b7e8c6813643e57d3efed3be094d4c06339dc2166b626e2cc5c12ee93 \
+ --hash=sha256:3200180abc69639483cf54a17cca2e13c31ede5f665979ea0a9c829d093f372f \
+ --hash=sha256:398b14f89ca950b288bd290875f07e4e10685644fa4ac668546fb107b1ada4d4 \
+ --hash=sha256:45e3ecc3888f1637094fd75cd8fc727f3a4b06d1ddf89181126c071e244fd2a5 \
+ --hash=sha256:4691a122b8c1d0d82d6e7510ce59d5c42146518240274b53e912e255573924f7 \
+ --hash=sha256:498f02ea92c9ca18c0f9c581ea93184a9d56c25b0af14189b0767adaf34235d8 \
+ --hash=sha256:4a05783ff54ce4c7a28f18e5772efdf63c219374bd9ffc55452182e1cef8be60 \
+ --hash=sha256:507c553cdb5abe2e951b5368423849fe29911a828c2135319c3e500e3bf25b32 \
+ --hash=sha256:50e52b6f479ddb1fe32423c2ec860811f36584cf6eabf279fb9a4f98b859a8b4 \
+ --hash=sha256:53eee22af5b5a305c1ee2652955ed46b148e881456fcec1e7f0eb27f642f6bb4 \
+ --hash=sha256:5af87d1a6d247d7467ee082ae977a5443b2c45f8cd4d59375b6daa38d523c2de \
+ --hash=sha256:5b90ad6637237b636d15c9ae8b7c4a7a1c194f33def378677e468c13fd4542f8 \
+ --hash=sha256:5bfdaada437e7730c17d366bd7bb8c4a16639963ddbfc1b2f302a68a17a290e7 \
+ --hash=sha256:66a83f93579fb3493e458c4449d1d566a7b2a1c7b19915cd0fa3c9b8b5a8540b \
+ --hash=sha256:6786bed88581e19bc4f28ea7a64ad531e8f54acf50327fddca942688824a60bd \
+ --hash=sha256:6946c033a144086d5b98c976b72f476b70c93fbbedf914eee0e886f073a4e9fa \
+ --hash=sha256:6eae4376adb104c2acfa76fd9ea0cb12b572ca1d70eceac709871f638ff76e93 \
+ --hash=sha256:6f1ce9ef9a1b13098efdc2e43a2ed96d9851bbde7b31c652a87552c4efe9b422 \
+ --hash=sha256:70fd99e5a09fb77f14b29d70879a4fce9529b2d2948b14c96708e0a61e001b98 \
+ --hash=sha256:730eed859508cb7b0775ebe6bb39f18901f168eb989d8ee23a4fe082700e1e3f \
+ --hash=sha256:769fb64412ca237547ca73f111a64252d9e32c9d938bed51ed537bc9146a8f54 \
+ --hash=sha256:7d7995b906666037d7114c20a5566a372902747452af7d5bd4cd6bca8f1a2550 \
+ --hash=sha256:801fd04899d72eab34f02ab78d0451525621b3bd589da9d2d480dfffe951b643 \
+ --hash=sha256:8252f20108e557532f91d7d6dd9af87c16ed6fa930f65516aa480fa2cfed3363 \
+ --hash=sha256:83cc48d1411d2ff388dab99973dca81172cc9ceae9c9799da9548d494cfb38cb \
+ --hash=sha256:89356c0793b474af7e49ec90d39fb2363e2341516a90460e38231df5ebe8acd5 \
+ --hash=sha256:8dd18fdff0ac9759b8d67a714730abee07b2312e3656c20ba5affb0107094762 \
+ --hash=sha256:917fd520bb60809d83c14d43cfe48d5ad2516abaf2c073d65a431800dade2d29 \
+ --hash=sha256:9443eefff58aad558608f352092e1be6d278980e8c3b4e8621fcbfda97818500 \
+ --hash=sha256:9ecb2b206b5b2386f6968721a0770226b66bdd54adc4279bfff3ddf62873eed8 \
+ --hash=sha256:a0afa8bac675445dc0e2ba2891ecbedd9be89cb437afa94c823e0290cc2c4bc5 \
+ --hash=sha256:a3238a693e806a3158375c6403b8f6f71d86eb9c149b60c97f26dfd560c98ac8 \
+ --hash=sha256:a515f664cad988f2295056833a59f62220bc3e46afdaffe389a29060f6712355 \
+ --hash=sha256:a8c631303bb1fd7be3067c47536a30ff1fcb4846d6008c112bc52a03f7cd6965 \
+ --hash=sha256:b2763e452b025ee8e990f0462e76052de9bb094ebc21d296f62c6dfe958886b4 \
+ --hash=sha256:b4a7af455ffed980925bc0ebf5b8d6239e6c3e797d9d755b6db192fb3080d614 \
+ --hash=sha256:be084d19a3ac0c8b2aba696680642d703118d3b1f18cf83f5b7dbaf0ffc62ab6 \
+ --hash=sha256:c3c1fb656063a2f762db5378ea8d38ad5f7836b4f3fb8c4652270ded43df2935 \
+ --hash=sha256:c60be0aed97a32c6ba8cee21f0d0477136e495451bd97910f589ac892db120d4 \
+ --hash=sha256:cf67f96dc0bfe9607f5f2b734cedfbe2f6f995231adee4ccefa12872044d452d \
+ --hash=sha256:d16102cbcd4615b09c64e6022733faccc93200785f1ab0d4493afb8b0261edde \
+ --hash=sha256:d30c966bea2deffa19c738c81776f7182da5ccabd97e666bae4f3d6ba87341d9 \
+ --hash=sha256:d652592c71683941b768306fa1c7c6ce1bb9b072505043feafe86305d71030b7 \
+ --hash=sha256:da4c9bdeaf6b06c12d13d0addfc8ef15aa9695d26574a6dc10751258bef72f30 \
+ --hash=sha256:dac25768be4c03a990c359f408cb7e8958ed0e93061e495b3642ce7909761205 \
+ --hash=sha256:dc96150f99e05a317cb1f042b92c4cf8bc93cdb1f9f85717322e202ecdf2e505 \
+ --hash=sha256:de8acaa5f4160f537a3cf41b031171d51004b9f4aebfa6c194f18dffa9533d03 \
+ --hash=sha256:e412767d1c9765cf1b82f7b00f1686c6ca5809ebb77af363b3f9f2325a465c01 \
+ --hash=sha256:e4812f71c39d77ec5041348dafa400532adf7bf8f1fffa9aa6495fce5876d7b8 \
+ --hash=sha256:e63b63b8b5fdb8e29318dff2b15c5f852be46e972775b466f75b848f6eed4502 \
+ --hash=sha256:e662f874ab2c7da9861584db44a13573e0936df087215f63013138f6e5eba083 \
+ --hash=sha256:e7b34209eef39462563c05ea9dcf51c272a2ded56f5753da925e66bca3baa484 \
+ --hash=sha256:ecb2e59a7bc692fee64dda6010deb66222335693b30046f15cccf81233aa715f \
+ --hash=sha256:f521d79d6acda4923b264805541696f452079db0952a5bb96f9ff742f50629ec \
+ --hash=sha256:f8669ce37851b597d3435b91fefa51139e58d506ca449ca0e5bb68c63b8b6d2b \
+ --hash=sha256:fa75c7970bc6bca340cc6e20f20f069201bfcb50094c31a536fd99724d1d01ca \
+ --hash=sha256:ff7aff4637fbf71394df139c63ccfe08a47aa4252d2f91224ddb3335c716c925
+fsspec==2026.7.0 \
+ --hash=sha256:b57ddbafedfaef7018c1ecab32aa200a9d7ca26b77965f64e48b70061249d279 \
+ --hash=sha256:c803c40f4cf860b49dea58ee3e1c33cb9c790520e233537e1340049f89b82a88
h11==0.16.0 \
--hash=sha256:4e35b956cf45792e4caa5885e69fba00bdbc6ffafbfa020300e549b208ee5ff1 \
--hash=sha256:63cf8bbe7522de3bf65932fda1d9c2772064ffb3dae62d55932da54b31cb6c86
humanfriendly==10.0 \
--hash=sha256:1697e1a8a8f550fd43c2865cd84542fc175a61dcb779b6fee18cf6b6ccba1477 \
--hash=sha256:6b0b831ce8f15f7300721aa49829fc4e83921a9a301cc7f606be6686a2288ddc
-idna==3.18 \
- --hash=sha256:7f952cbe720b688055e3f87de14f5c3e5fdaa8bc3928985c4077ca689de849a2 \
- --hash=sha256:ffb385a7e039654cef1ab9ef32c6fafe283c0c0467bba1d9029738ce4a14a848
+idna==3.19 \
+ --hash=sha256:5e0811a4383b21dc5838069f801c4fb62113b7447663d2530d2bd6e77b49bf15 \
+ --hash=sha256:815e7be7a7806d54abb586dc943addc79e8b2ee16915059658cbeff4b1b43bf4
iniconfig==2.3.0 \
--hash=sha256:c76315c77db068650d49c5b56314774a7804df16fee4402c1f19d6d15d8c4730 \
--hash=sha256:f631c04d2c48c52b84d0d0549c99ff3859c98df65b3101406327ecc7d53fbf12
jinja2==3.1.6 \
--hash=sha256:0137fb05990d35f1275a587e9aee6d56da821fc83491a0fb838183be43f66d6d \
--hash=sha256:85ece4451f492d0c13c5dd7c13a64681a86afae63a5f347908daf103ce6d2f67
-joblib==1.5.3 \
- --hash=sha256:5fc3c5039fc5ca8c0276333a188bbd59d6b7ab37fe6632daa76bc7f9ec18e713 \
- --hash=sha256:8561a3269e6801106863fd0d6d84bb737be9e7631e33aaed3fb9ce5953688da3
-kiwisolver==1.5.0 \
- --hash=sha256:012b1eb16e28718fa782b5e61dc6f2da1f0792ca73bd05d54de6cb9561665fc9 \
- --hash=sha256:01808c6d15f4c3e8559595d6d1fe6411c68e4a3822b4b9972b44473b24f4e679 \
- --hash=sha256:0255a027391d52944eae1dbb5d4cc5903f57092f3674e8e544cdd2622826b3f0 \
- --hash=sha256:0b85aad90cea8ac6797a53b5d5f2e967334fa4d1149f031c4537569972596cb8 \
- --hash=sha256:0bf3acf1419fa93064a4c2189ac0b58e3be7872bf6ee6177b0d4c63dc4cea276 \
- --hash=sha256:0c50b89ffd3e1a911c69a1dd3de7173c0cd10b130f56222e57898683841e4f96 \
- --hash=sha256:0cbe94b69b819209a62cb27bdfa5dc2a8977d8de2f89dfd97ba4f53ed3af754e \
- --hash=sha256:0df54df7e686afa55e6f21fb86195224a6d9beb71d637e8d7920c95cf0f89aac \
- --hash=sha256:0e3aafb33aed7479377e5e9a82e9d4bf87063741fc99fc7ae48b0f16e32bdd6f \
- --hash=sha256:12e91c215a96e39f57989c8912ae761286ac5a9584d04030ceb3368a357f017a \
- --hash=sha256:1465387ac63576c3e125e5337a6892b9e99e0627d52317f3ca79e6930d889d15 \
- --hash=sha256:16b85d37c2cbb3253226d26e64663f755d88a03439a9c47df6246b35defbdfb7 \
- --hash=sha256:1b0feb50971481a2cc44d94e88bdb02cdd497618252ae226b8eb1201b957e368 \
- --hash=sha256:1d49a49ac4cbfb7c1375301cd1ec90169dfeae55ff84710d782260ce77a75a02 \
- --hash=sha256:1d9daea4ea6b9be74fe2f01f7fbade8d6ffab263e781274cffca0dba9be9eec9 \
- --hash=sha256:1dd9b0b119a350976a6d781e7278ec7aca0b201e1a9e2d23d9804afecb6ca681 \
- --hash=sha256:1f1489f769582498610e015a8ef2d36f28f505ab3096d0e16b4858a9ec214f57 \
- --hash=sha256:2517e24d7315eb51c10664cdb865195df38ab74456c677df67bb47f12d088a27 \
- --hash=sha256:295d9ffe712caa9f8a3081de8d32fc60191b4b51c76f02f951fd8407253528f4 \
- --hash=sha256:2a075bd7bd19c70cf67c8badfa36cf7c5d8de3c9ddb8420c51e10d9c50e94920 \
- --hash=sha256:32cc0a5365239a6ea0c6ed461e8838d053b57e397443c0ca894dcc8e388d4374 \
- --hash=sha256:332b4f0145c30b5f5ad9374881133e5aa64320428a57c2c2b61e9d891a51c2f3 \
- --hash=sha256:377815a8616074cabbf3f53354e1d040c35815a134e01d7614b7692e4bf8acfa \
- --hash=sha256:38f4a703656f493b0ad185211ccfca7f0386120f022066b018eb5296d8613e23 \
- --hash=sha256:3ac2360e93cb41be81121755c6462cff3beaa9967188c866e5fce5cf13170859 \
- --hash=sha256:3c4923e404d6bcd91b6779c009542e5647fef32e4a5d75e115e3bbac6f2335eb \
- --hash=sha256:3cdcb35dc9d807259c981a85531048ede628eabcffb3239adf3d17463518992d \
- --hash=sha256:41024ed50e44ab1a60d3fe0a9d15a4ccc9f5f2b1d814ff283c8d01134d5b81bc \
- --hash=sha256:413b820229730d358efd838ecbab79902fe97094565fdc80ddb6b0a18c18a581 \
- --hash=sha256:4432b835675f0ea7414aab3d37d119f7226d24869b7a829caeab49ebda407b0c \
- --hash=sha256:4db576bb8c3ef9365f8b40fe0f671644de6736ae2c27a2c62d7d8a1b4329f099 \
- --hash=sha256:4e7f886f47ab881692f278ae901039a234e4025a68e6dfab514263a0b1c4ae05 \
- --hash=sha256:4e9750bc21b886308024f8a54ccb9a2cc38ac9fa813bf4348434e3d54f337ff9 \
- --hash=sha256:5060731cc3ed12ca3a8b57acd4aeca5bbc2f49216dd0bec1650a1acd89486bcd \
- --hash=sha256:50847dca5d197fcbd389c805aa1a1cf32f25d2e7273dc47ab181a517666b68cc \
- --hash=sha256:5092eb5b1172947f57d6ea7d89b2f29650414e4293c47707eb499ec07a0ac796 \
- --hash=sha256:5124d1ea754509b09e53738ec185584cc609aae4a3b510aaf4ed6aa047ef9303 \
- --hash=sha256:51e8c4084897de9f05898c2c2a39af6318044ae969d46ff7a34ed3f96274adca \
- --hash=sha256:530a3fd64c87cffa844d4b6b9768774763d9caa299e9b75d8eca6a4423b31314 \
- --hash=sha256:56fa888f10d0f367155e76ce849fa1166fc9730d13bd2d65a2aa13b6f5424489 \
- --hash=sha256:58f812017cd2985c21fbffb4864d59174d4903dd66fa23815e74bbc7a0e2dd57 \
- --hash=sha256:59cd8683f575d96df5bb48f6add94afc055012c29e28124fcae2b63661b9efb1 \
- --hash=sha256:5ae8e62c147495b01a0f4765c878e9bfdf843412446a247e28df59936e99e797 \
- --hash=sha256:5b233ea3e165e43e35dba1d2b8ecc21cf070b45b65ae17dd2747d2713d942021 \
- --hash=sha256:6176c1811d9d5a04fa391c490cc44f451e240697a16977f11c6f722efb9041db \
- --hash=sha256:62f59da443c4f4849f73a51a193b1d9d258dcad0c41bc4d1b8fb2bcc04bfeb22 \
- --hash=sha256:6783e069732715ad0c3ce96dbf21dbc2235ab0593f2baf6338101f70371f4028 \
- --hash=sha256:6ab8ba9152203feec73758dad83af9a0bbe05001eb4639e547207c40cfb52083 \
- --hash=sha256:70d593af6a6ca332d1df73d519fddb5148edb15cd90d5f0155e3746a6d4fcc65 \
- --hash=sha256:72ec46b7eba5b395e0a7b63025490d3214c11013f4aacb4f5e8d6c3041829588 \
- --hash=sha256:7a32f72973f0f950c1920475d5c5ea3d971b81b6f0ec53b8d0a956cc965f22e0 \
- --hash=sha256:7a4aa69609f40fce3cbc3f87b2061f042eee32f94b8f11db707b66a26461591a \
- --hash=sha256:7c60d3c9b06fb23bd9c6139281ccbdc384297579ae037f08ae90c69f6845c0b1 \
- --hash=sha256:800ee55980c18545af444d93fdd60c56b580db5cc54867d8cbf8a1dc0829938c \
- --hash=sha256:80aa065ffd378ff784822a6d7c3212f2d5f5e9c3589614b5c228b311fd3063ac \
- --hash=sha256:86e0287879f75621ae85197b0877ed2f8b7aa57b511c7331dce2eb6f4de7d476 \
- --hash=sha256:893ff3a711d1b515ba9da14ee090519bad4610ed1962fbe298a434e8c5f8db53 \
- --hash=sha256:89fc958c702ee9a745e4700378f5d23fddbc46ff89e8fdbf5395c24d5c1452a3 \
- --hash=sha256:8c63c91f95173f9c2a67c7c526b2cea976828a0e7fced9cdcead2802dc10f8a4 \
- --hash=sha256:8df31fe574b8b3993cc61764f40941111b25c2d9fea13d3ce24a49907cd2d615 \
- --hash=sha256:8f9baf6f0a6e7571c45c8863010b45e837c3ee1c2c77fcd6ef423be91b21fedb \
- --hash=sha256:9027d773c4ff81487181a925945743413f6069634d0b122d0b37684ccf4f1e18 \
- --hash=sha256:9190426b7aa26c5229501fa297b8d0653cfd3f5a36f7990c264e157cbf886b3b \
- --hash=sha256:940dda65d5e764406b9fb92761cbf462e4e63f712ab60ed98f70552e496f3bf1 \
- --hash=sha256:94eff26096eb5395136634622515b234ecb6c9979824c1f5004c6e3c3c85ccd2 \
- --hash=sha256:9eed0f7edbb274413b6ee781cca50541c8c0facd3d6fd289779e494340a2b85c \
- --hash=sha256:ad4ae4ffd1ee9cd11357b4c66b612da9888f4f4daf2f36995eda64bd45370cac \
- --hash=sha256:b0f172dc8ffaccb8522d7c5d899de00133f2f1ca7b0a49b7da98e901de87bf2d \
- --hash=sha256:b2af221f268f5af85e776a73d62b0845fc8baf8ef0abfae79d29c77d0e776aaf \
- --hash=sha256:b7d335370ae48a780c6e6a6bbfa97342f563744c39c35562f3f367665f5c1de2 \
- --hash=sha256:b83af57bdddef03c01a9138034c6ff03181a3028d9a1003b301eb1a55e161a3f \
- --hash=sha256:bb5136fb5352d3f422df33f0c879a1b0c204004324150cc3b5e3c4f310c9049f \
- --hash=sha256:bc4d8e252f532ab46a1de9349e2d27b91fce46736a9eedaa37beaca66f574ed4 \
- --hash=sha256:bdd3e53429ff02aa319ba59dfe4ceeec345bf46cf180ec2cf6fd5b942e7975e9 \
- --hash=sha256:be12f931839a3bdfe28b584db0e640a65a8bcbc24560ae3fdb025a449b3d754e \
- --hash=sha256:be4a51a55833dc29ab5d7503e7bcb3b3af3402d266018137127450005cdfe737 \
- --hash=sha256:beb7f344487cdcb9e1efe4b7a29681b74d34c08f0043a327a74da852a6749e7b \
- --hash=sha256:bf4679a3d71012a7c2bf360e5cd878fbd5e4fcac0896b56393dec239d81529ed \
- --hash=sha256:c0e1403fd7c26d77c1f03e096dc58a5c726503fa0db0456678b8668f76f521e3 \
- --hash=sha256:c31c13da98624f957b0fb1b5bae5383b2333c2c3f6793d9825dd5ce79b525cb7 \
- --hash=sha256:c438f6ca858697c9ab67eb28246c92508af972e114cac34e57a6d4ba17a3ac08 \
- --hash=sha256:c8277104ded0a51e699c8c3aff63ce2c56d4ed5519a5f73e0fd7057f959a2b9e \
- --hash=sha256:c95cab08d1965db3d84a121f1c7ce7479bdd4072c9b3dafd8fecce48a2e6b902 \
- --hash=sha256:cc0b66c1eec9021353a4b4483afb12dfd50e3669ffbb9152d6842eb34c7e29fd \
- --hash=sha256:cdee07c4d7f6d72008d3f73b9bf027f4e11550224c7c50d8df1ae4a37c1402a6 \
- --hash=sha256:ce9bf03dad3b46408c08649c6fbd6ca28a9fce0eb32fdfffa6775a13103b5310 \
- --hash=sha256:cff8e5383db4989311f99e814feeb90c4723eb4edca425b9d5d9c3fefcdd9537 \
- --hash=sha256:d168fda2dbff7b9b5f38e693182d792a938c31db4dac3a80a4888de603c99554 \
- --hash=sha256:d1ffeb80b5676463d7a7d56acbe8e37a20ce725570e09549fe738e02ca6b7e1e \
- --hash=sha256:d36ca54cb4c6c4686f7cbb7b817f66f5911c12ddb519450bbe86707155028f87 \
- --hash=sha256:d4193f3d9dc3f6f79aaed0e5637f45d98850ebf01f7ca20e69457f3e8946b66a \
- --hash=sha256:d5cd5189fc2b6a538b75ae45433140c4823463918f7b1617c31e68b085c0022c \
- --hash=sha256:d618fd27420381a4f6044faa71f46d8bfd911bd077c555f7138ed88729bfbe79 \
- --hash=sha256:d76e2d8c75051d58177e762164d2e9ab92886534e3a12e795f103524f221dd8e \
- --hash=sha256:daae526907e262de627d8f70058a0f64acc9e2641c164c99c8f594b34a799a16 \
- --hash=sha256:db485b3847d182b908b483b2ed133c66d88d49cacf98fd278fadafe11b4478d1 \
- --hash=sha256:dd952e03bfbb096cfe2dd35cd9e00f269969b67536cb4370994afc20ff2d0875 \
- --hash=sha256:dda366d548e89a90d88a86c692377d18d8bd64b39c1fb2b92cb31370e2896bbd \
- --hash=sha256:e315e5ec90d88e140f57696ff85b484ff68bb311e36f2c414aa4286293e6dee0 \
- --hash=sha256:e4415a8db000bf49a6dd1c478bf70062eaacff0f462b92b0ba68791a905861f9 \
- --hash=sha256:e7a116ae737f0000343218c4edf5bd45893bfeaff0993c0b215d7124c9f77646 \
- --hash=sha256:e7c4c09a490dc4d4a7f8cbee56c606a320f9dc28cf92a7157a39d1ce7676a657 \
- --hash=sha256:ebae99ed6764f2b5771c522477b311be313e8841d2e0376db2b10922daebbba4 \
- --hash=sha256:ec4c85dc4b687c7f7f15f553ff26a98bfe8c58f5f7f0ac8905f0ba4c7be60232 \
- --hash=sha256:ed3a984b31da7481b103f68776f7128a89ef26ed40f4dc41a2223cda7fb24819 \
- --hash=sha256:f18c2d9782259a6dc132fdc7a63c168cbc74b35284b6d75c673958982a378384 \
- --hash=sha256:f1f9f4121ec58628c96baa3de1a55a4e3a333c5102c8e94b64e23bf7b2083309 \
- --hash=sha256:f42c23db5d1521218a3276bb08666dcb662896a0be7347cba864eca45ff64ede \
- --hash=sha256:f443b4825c50a51ee68585522ab4a1d1257fac65896f282b4c6763337ac9f5d2 \
- --hash=sha256:f6764a4ccab3078db14a632420930f6186058750df066b8ea2a7106df91d3203 \
- --hash=sha256:f7c7553b13f69c1b29a5bde08ddc6d9d0c8bfb84f9ed01c30db25944aeb852a7 \
- --hash=sha256:fa6248cd194edff41d7ea9425ced8ca3a6f838bfb295f6f1d6e6bb694a8518df \
- --hash=sha256:fa8eb9ecdb7efb0b226acec134e0d709e87a909fa4971a54c0c4f6e88635484c \
- --hash=sha256:fc20894c3d21194d8041a28b65622d5b86db786da6e3cfe73f0c762951a61167 \
- --hash=sha256:fc4d3f1fb9ca0ae9f97b095963bc6326f1dbfd3779d6679a1e016b9baaa153d3 \
- --hash=sha256:fd40bb9cd0891c4c3cb1ddf83f8bbfa15731a248fdc8162669405451e2724b09 \
- --hash=sha256:ff710414307fefa903e0d9bdf300972f892c23477829f49504e59834f4195398
-lazy-loader==0.5 \
- --hash=sha256:717f9179a0dbed357012ddad50a5ad3d5e4d9a0b8712680d4e687f5e6e6ed9b3 \
- --hash=sha256:ab0ea149e9c554d4ffeeb21105ac60bed7f3b4fd69b1d2360a4add51b170b005
-markdown-it-py==4.0.0 \
- --hash=sha256:87327c59b172c5011896038353a81343b6754500a08cd7a4973bb48c6d578147 \
- --hash=sha256:cb0a2b4aa34f932c007117b194e945bd74e0ec24133ceb5bac59009cda1cb9f3
+joblib==1.6.0 \
+ --hash=sha256:2ccc96785b12046c08fd6d55839c12857831b54a3c1673ffadd2f04bfc4eda03 \
+ --hash=sha256:3dbbf9f6e4b592a2357b854608e980fe6390d131d7a82f011a377ef2ebef7aba
+kiwisolver==1.5.1 \
+ --hash=sha256:007a5553dfc4f4e8d184f588a0200e2cd4b63a59cc8796df3c39909e679dc7a0 \
+ --hash=sha256:0324cd2567259b7a095f6cf18a52b0ffc6f3de9e69528ff1bc0e7a37bd43ff1a \
+ --hash=sha256:0627b9bceb9c3cdcf12b8a18655eedfed2692b038df27423383c120d0b7dc2d6 \
+ --hash=sha256:06a6917674de9e0fe3f66f5430787f59a9f2ddb64af9b714eaec547e29ef5c19 \
+ --hash=sha256:072bdb15a3c19a5b5dbc8f8fb1f4e1884bf4f3507eeb4cc6334401274d37a5c0 \
+ --hash=sha256:0a4faea5c6db201c6a21391d2ac926ea97acf7dacdbc3c417189e1adb1a00837 \
+ --hash=sha256:0ba9527afc80ae3d7814ed98b6572d02bf85eaf48065678342c5f0c6dab7a8c7 \
+ --hash=sha256:0d8924877ce22e17326a99a418c3c82037da078df3c6a260b13eca677444e6e7 \
+ --hash=sha256:0ebdef3eae5336568147c39a55be6a2036ffde53faa9ca2d978989ae7c2da12c \
+ --hash=sha256:1209042a623ddfda5497e4066c7b77651dde8e1d3a9dd97599dc7e97f3b9b78c \
+ --hash=sha256:16895f553ee6620a827d2da56b871f835fb70b9216cca5d188e885caf6e3bd23 \
+ --hash=sha256:17851e5dad4484be0cbccbde3b15331deae036de9aebd45eed964487802b172f \
+ --hash=sha256:1798e83840c3f627246104c4d8a9639c60fa068adf9ce92b61791781fa8a68c1 \
+ --hash=sha256:18170a77ddfecf40ec60d0928268dc95880c881864e015a8f34094ed18b9b9ad \
+ --hash=sha256:186884a58486651e3c217b6acea0a53eaa9498fdd472057c46f2f0fb5c25aad5 \
+ --hash=sha256:18a0cfb124546a4c2e6087c5f3029c7f44b37c85b142e0ced71f73a7599ac208 \
+ --hash=sha256:1983f0974a750a6f6556f368ba11105d1d8369c735b944747c9f12ae5aea7aae \
+ --hash=sha256:1a7587dc335f2c0f5bd577fd0540bd16c66006bdb60f759a1059f025e6c4f071 \
+ --hash=sha256:1acc7e5b7ef05e9da8bb70cd6c7c4513090213d2e1ad9720f599f0bf6c52aec5 \
+ --hash=sha256:1d852545c4d0e35a72728d072cbaa59e2fa7dd84bdf01e068d670dd0ceb58eb6 \
+ --hash=sha256:1ed0f5e49d0ceff8b72190824d9e59c062fbbc02c231b853112c78474b3f5ec2 \
+ --hash=sha256:1fff05e239575b1481b6ed1a782f6fad616efbf1f0b1f44e6e85c4dfe426e483 \
+ --hash=sha256:21e46b23a2da695c364124817bc01d970effd5483147f8d66a6a7167e3f6b851 \
+ --hash=sha256:22d5e5aaad6be121f2515765e3b1c444352cb8eb4c86510801db8f2e50757316 \
+ --hash=sha256:2551cf9917af48ee7c4b29cc82320489508cf96fd26a51f6fc124de661cd44c7 \
+ --hash=sha256:255605693a483db7bd5c79f60437f7bf658f7f520d61aa42722e32257c941951 \
+ --hash=sha256:26e8268480be5061d509e29669d59103c067a26377a56491630ece11762e3858 \
+ --hash=sha256:27add358abe374ebaa3b8763ef380bc99051b5a4b18d94878366a9e4f59efef0 \
+ --hash=sha256:2ae70bc59790d2af72a3f76f24b272403e135070340281108b447cb77ea70819 \
+ --hash=sha256:2e10ae1bba1899188b33557c10d73affcc12033edd18adddb57d209039976a4c \
+ --hash=sha256:3221f78211074f561c44ca42eac0619828171bec15a2c4cf6f7747d07df76e8e \
+ --hash=sha256:34633ecf50d16187ab8e5528b7a2530f2feb4e23f300db4672538b51cfc5cd38 \
+ --hash=sha256:34ec467940442c9943016fb2d4c81d1ba84351eeca2f1a78f8bc87f1ba0d414c \
+ --hash=sha256:37f801b5d7cc0e5a548921308e059fd2b057bb42972b591cfa3049f95423c4ed \
+ --hash=sha256:38f6e0deb4d0a4615efe0c4efc5990b06ae450ab50a0b321c0b078b6d238c083 \
+ --hash=sha256:3c24cd69455e1b00ddf770c13b6e2c33e07d6dc3f2d34add0bf9277c5c6bbd46 \
+ --hash=sha256:3cc210010fd2f438a3ed430b45f1b501fd13a8618bf984dc2c5ce5b69b78752e \
+ --hash=sha256:3fa5855898f6d3d01b72ccd48a2d65cbdee301251603fefe34e2025bddba219c \
+ --hash=sha256:416ba7ff9f233b7036689bb5a3783537e838ad483f63558d2a800f75afe738b1 \
+ --hash=sha256:431dc224a1a92a5c8f582d96e505196a3b5997a7271076678da2dfde67b77e9a \
+ --hash=sha256:43844c1a7ad6d723d5b5b4c4fc7f5bd399c40e288120d16257c7c9e8765c6e85 \
+ --hash=sha256:44b8faef94f1857e77fa0238f3390ff1ac51d2ea20a487e2e452a59fd2b5f5ca \
+ --hash=sha256:470d420f98d368d6f010633a20659b544c5fdfa5329e6b70219f2ef08fd4a7ef \
+ --hash=sha256:482676e5bd48d70ac99d9fc78863469845421e01184fa83f1f9366dc49f7e974 \
+ --hash=sha256:4d4ca09bf13cff792b1884f64b98ee6c2467930d632233be25c56b442d99f10e \
+ --hash=sha256:5025e36fb4fb275cef0a4e30dbb11cb4ae61d1c83deb90189cb5d7e4cafd6b55 \
+ --hash=sha256:509735237ae0d849e8a843551d423d2500d2e0a9ac1611a145658b29c0fb9f85 \
+ --hash=sha256:534f02c1abb31ed6dbd3515545285c330b2f12d00fdb1fdb71658b9ca5a13a6a \
+ --hash=sha256:5978c3340f16a35c30f8ab2fa7bcf559973c55f1a5ef6970e1f621acf3c4db13 \
+ --hash=sha256:5b973887ff782cfd6b67c9904ad8ca542e0bc5e4961503408b423b5a688b4d38 \
+ --hash=sha256:5c490db2168a508088f59140dd392556a54b8bd1048fc6383c8baff13c359673 \
+ --hash=sha256:5d142e352eb13facc7dd047489aebdff6ba78576c239f1ea04931979caaf0567 \
+ --hash=sha256:5daa1f19e097050b9c4d9a78fcc9263cb96c9dfae08037ddc1b7c4ad1889f2a2 \
+ --hash=sha256:61e9a64c7635095a6bfe483e2ff055d437c59bd45f3617a228b37277f0185d62 \
+ --hash=sha256:63fb7294b768f444eb4b068965f2662f28c2fd4161e23bd60fcf3ff27b74c046 \
+ --hash=sha256:685929988b208a911f1285e2f8ed54210b0d681a3dc0f03e00d599d291986e7e \
+ --hash=sha256:6a797a1cefc8b9c93170db580337e1fe3d011ad18b1299943231279406342048 \
+ --hash=sha256:6b92f60017dda7d877fdc546438b5e28f31c523264f49cf5a48c1d0ce1a0dfbc \
+ --hash=sha256:70ed9a45c7484d2b30cdacf60d220f494a1763b9fec1ad03285c6553fa0889f2 \
+ --hash=sha256:719a35fa1156db3640555f95ebb94f60a444e64d1c69626b0edef5df78eba225 \
+ --hash=sha256:74ad5c3dad54a4641b4c28cd15ded70899d04459c6c7aeacafea716be97cce6d \
+ --hash=sha256:74ea337e0ec3f6f342a36a4f1b5cd94dd9affddcd28ba9aae2905af932ee8c6b \
+ --hash=sha256:75d9b1cf8258462dbdc1eeda718c96ea7f079324c09067f6daabfcf37712b7fe \
+ --hash=sha256:77a4c8187a5948d7f8795adb765a3c7b553d07d86d88e43038fc32fc1fb9a3f3 \
+ --hash=sha256:7824b5e8bdbf0bccb4ccd37bbb115849a1dc45437fb4de8351385ed07c437ee0 \
+ --hash=sha256:7d38b0c279c3032e8c9cc013b405c6df8e1668dbf15465779aa7f15f61201812 \
+ --hash=sha256:7e9c01d3dd7ceba4d1d436cc021d40d592466e40b9bc7f5d83dc4e98a5c9cd8c \
+ --hash=sha256:7fd82debf43c6acd0a94359d232f6bb516ee13f269a7993736a9ac9f988bb5d9 \
+ --hash=sha256:824c3d763a05ea9e9003610145186b0e9848c7584a5575c79bac5a8e7cd80bad \
+ --hash=sha256:828f75af2b0080c8a972e75f649ab46af008e92c6104a57a759157200b835b75 \
+ --hash=sha256:83f78128fa28705fa85d01c59771c72fe81c11bd0e6155edbb9f818983a7d761 \
+ --hash=sha256:876bbfd276473d3daffe30e8c975df4ed9429967b41a6cb362dbb5155b6f13ad \
+ --hash=sha256:886fc26012f0e8b5f69d1cfe6d711f6b11f194621539bf8e6bb1c25c5dc82724 \
+ --hash=sha256:8a34616dc2521cc8dc1d7d081734da63539f021ac0450ce950908340c6e7aa2f \
+ --hash=sha256:8a708a47ade1fe19e8371d5da076bac0dd4b0a5a7985ad6c637f7f7e361b6baa \
+ --hash=sha256:8af9b142ad719ae3a911ebf616bc4b78b32bbab84d6a40d3ad2f129670509957 \
+ --hash=sha256:8bf4df63592c2a66b4f8edc5df2544998c288aa02f96ce0acd880cd1de8c8127 \
+ --hash=sha256:8de6f2a4ce7e7bd27d23dd94abf0ccafe0e0e5cc9c764b0577191f2c25f08f26 \
+ --hash=sha256:8f8fddb8e323bd6eee4e54e69a39243beab22689070f4c66b472c4cc88bb89d8 \
+ --hash=sha256:8fca690b00c4c48f6c2a547b0160ed511357093a4e4c9b47e0fadf3128066d89 \
+ --hash=sha256:9506e892bcc3b409831d363c6f53e5985e1c8d1f6f6b0256d00358684ff85378 \
+ --hash=sha256:958254518717542d02d0688d0d20cbf771da5e415e6f49543f92481c850a4540 \
+ --hash=sha256:95a02752aa032eef4aed01cda6d9b687c669bd0396bf4519eef8bba22a286720 \
+ --hash=sha256:96c30002424670b5e1e46495c2b8cbffef39cf77c1d79e76462029d50339785b \
+ --hash=sha256:98b208a7cc42c803445ef551d6753cc42a5ea13e9cab1ee66cd8b9cb70195330 \
+ --hash=sha256:9b3092d8992a1d69b7a59c3e39f35e1b9be327a17f68a7c35fc17329e337d6f2 \
+ --hash=sha256:9e51c119992ea8820706871c30a4642ec76de20ae82f9b50b9a45517d8e9f810 \
+ --hash=sha256:a5716a33bfabb2c6ce27b6cf03253467b3804f83e215f4d202685cf93c6c9874 \
+ --hash=sha256:a5a00665d1a0e26763a7338d7e911d4598fbc1d50dd0d6b7919b7dc6c5d6569f \
+ --hash=sha256:a5ca5aebae78a0bc13c1943af4af615d4966c5b650b05d5aa83b50e427196fee \
+ --hash=sha256:a7b85b2cc6ea45e5f7e8c9a30bc9fabd47cda09106cbb4b967335c3e6c43b69d \
+ --hash=sha256:a83ee7107df13abe42a54a6654670eef9bb39425cf2e27f65e0007465e1286ab \
+ --hash=sha256:aa7d00b1700966d2917e54d278aba86897890ca9276dd8b76cf6446b6c181b92 \
+ --hash=sha256:ab620eb663952455271ac37f9aaad86b73c969c02f11f53cea405b38e96a4300 \
+ --hash=sha256:ad8b9671348d7c8716715652ae11f85ed0eb99e265a2df2ca490577d69860b2c \
+ --hash=sha256:aefe930d113798330e9462f7874542977869c0613cba3262e2de3a8d5dee8f3a \
+ --hash=sha256:b03af77d77e50edba2030fd5f7c352ff209314b09030a3cba7c14edf9a09a444 \
+ --hash=sha256:b390aec180a7c054919c04898835e1c77bced23ea8383eb2c570213bf25d1a86 \
+ --hash=sha256:b3d78f7bb2b9d9a30345be1474b9aaa8685430b54afb51ba3639b5c6c11e9ed6 \
+ --hash=sha256:b5664603a253efd3a75716d793d1d3a6a82723b61dc6db767b2460bbbeec4c0f \
+ --hash=sha256:b69602970994a2ed8bbfa78c2f0394a7435226c6040489702d9f0a0ad0c07052 \
+ --hash=sha256:b6ae6a0328f0bc035741820fdeecdcd67bf4694eee03972e843663107122f450 \
+ --hash=sha256:bad20d4c69c851c982a1e3606f4c293edfd5a87885786c50082412240c4b1ffd \
+ --hash=sha256:bb7c99f0673c03017a3ee01e54a5c2617a05468b11eabe513b0080e063ed95b1 \
+ --hash=sha256:bebb89489b279b2f5661bbbb2abcc87bcd4a46607bb4a5c966f04f1db6b8df9a \
+ --hash=sha256:bfd1de989b3330420e29de39352f5c049905c9e3ee67233a50d550e3d652c148 \
+ --hash=sha256:c2306e8bb53601979fcb3fa09cc65e031876d9ae01eff2fcbcd7a84ef94d5bc1 \
+ --hash=sha256:c3a4e41e3096bf1f0f1b76e2ffd6d828d6547f574f702d59bdbef7acfa59db9c \
+ --hash=sha256:c6834b92dd2428e2dd85ef3d85f723d3c12f20aaf43a2ddd4f944ca25d833408 \
+ --hash=sha256:c90d3022d8a94778939cda8638c6c8da8fa757b8958dad7ec868ce29c87681b8 \
+ --hash=sha256:ca307d6c259e5c98d3cb9ade55342b47a6839762caf2536f3d7b46ee660cc82e \
+ --hash=sha256:ca7f6fe0f37ca978a1e5eb7a3a68e6413f417e78e838324947ffd420202b198b \
+ --hash=sha256:cb6fae641357ed2f6e533c0d3c6504a4a5703621a50c89459e46051d56b61140 \
+ --hash=sha256:cdaeeb6c350106df6bf9d873395973e5f066a9713200b72cd64f55d0a3eafab6 \
+ --hash=sha256:cea20da04494e662b83c872683bf4ff2345206043d036315ed0e924b652e7294 \
+ --hash=sha256:cea90547bfd93807e0013a004dc76552be44fad3bc1cc2b38610a9e889ed098f \
+ --hash=sha256:d09037ca068d784ebc4aec290ef952ca27ac15dd9c0b5801a88c6e1096b83e6b \
+ --hash=sha256:d27c2123977cb9269c30a49ba45f03a4323017ef693e19db4ec9dbe1299a3002 \
+ --hash=sha256:d50de98e8d807dc31822fff96f50293163a62418eb65487a21b42713d72ed0b7 \
+ --hash=sha256:d66a64dd5dec136040ec2ae94aa026a912ee60fdd45bc28d3db30037fd809e88 \
+ --hash=sha256:d79308fa689fac89cbcfbd4dbfc80b5f95c54c5a7fd4d194be221f9d33d026e6 \
+ --hash=sha256:da3275833be0edbaf4830fae08bae3dc7219f40ce0c37eaa6c25825957e06612 \
+ --hash=sha256:dc1a26b8e53395a01c2c611e58602fa47461f136fba7cd5542e6db6d64be1839 \
+ --hash=sha256:dc23390afe9f4ef9ac3bcc72a03a56eebbde03f4c571a32cb38f859cff9a6524 \
+ --hash=sha256:e05c2f7925f1d88778e53cb44f14e0223204a3bdd09a41664750363acfb1f2ef \
+ --hash=sha256:e12dfea7f5fc2a34a9080efbf79c4c44eb380ec5b9c6fea09407e08f0d1e941d \
+ --hash=sha256:e4e4523d6f336708d732516e6cfca7796cf3d96c9474eb5aecf6165f2f1fefc3 \
+ --hash=sha256:e4e49f7e1a4e7191bdf9dc67a974db714501b1fc52c24324103d06a86abd5c08 \
+ --hash=sha256:e68e151428b5384f766cd25739bf77c7e4a3dc93b5ded7a12118d9fbfdf78ab6 \
+ --hash=sha256:e8e4d953faaded9ec7ede36824e9814082d22d4c7b1eafbfa079ecba8cd0d076 \
+ --hash=sha256:ee9df1f0d77b9c6e94f4ac0fec533fbddd5ea3a327807f18d7b069ae019ded80 \
+ --hash=sha256:f0a887b6565bbfe80efde2b7f6e8890d7d9bbdb11bdb17028a3690c32fe0621f \
+ --hash=sha256:f0f4a42db92d6ec7677ab9d12830a2a8ec145a9c6d15db2b593466bc875c78d7 \
+ --hash=sha256:f1303ef2eec81262a4b708c3e858afe58d7c75ad91c1c05266eda7673369859a \
+ --hash=sha256:f1d56ec54d257d05e0b50f5780d967540cd07beeaf9e5f645b26d50cce79f4d8 \
+ --hash=sha256:f4167e87b397f273dc2356fcf1eaf50a6bac51e6105f45103ef7129c8efb0255 \
+ --hash=sha256:f76fc85bd054c806960f917ec0f329e24e436f1712267d90588e4c39890caa63 \
+ --hash=sha256:f942903fde7363d1d879057ec5de01310efda2597161784d752fa9953a01a71a \
+ --hash=sha256:f9b1c4900736e489a812c529100de4b8fb617d4db075e931e213c57424b83d9b \
+ --hash=sha256:fc271a6f0a2126958f4090e5507b9da5848927dae331f8f763bd4aa642b3d2cd \
+ --hash=sha256:febcce10f2bcdbb80b4ea919238a6a4ac13dbc4c7cadbe8d5d75c3682f8b5404
+markdown-it-py==4.2.0 \
+ --hash=sha256:04a21681d6fbb623de53f6f364d352309d4094dd4194040a10fd51833e418d49 \
+ --hash=sha256:9f7ebbcd14fe59494226453aed97c1070d83f8d24b6fc3a3bcf9a38092641c4a
markupsafe==3.0.3 \
--hash=sha256:0303439a41979d9e74d18ff5e2dd8c43ed6c6001fd40e5bf2e43f7bd9bbc523f \
--hash=sha256:068f375c472b3e7acbe2d5318dea141359e6900156b5b2ba06a30b169086b91a \
@@ -652,62 +739,62 @@ markupsafe==3.0.3 \
--hash=sha256:f71a396b3bf33ecaa1626c255855702aca4d3d9fea5e051b41ac59a9c1c41edc \
--hash=sha256:f9e130248f4462aaa8e2552d547f36ddadbeaa573879158d721bbd33dfe4743a \
--hash=sha256:fed51ac40f757d41b7c48425901843666a6677e3e8eb0abcff09e4ba6e664f50
-matplotlib==3.10.8 \
- --hash=sha256:00270d217d6b20d14b584c521f810d60c5c78406dc289859776550df837dcda7 \
- --hash=sha256:0a33deb84c15ede243aead39f77e990469fff93ad1521163305095b77b72ce4a \
- --hash=sha256:113bb52413ea508ce954a02c10ffd0d565f9c3bc7f2eddc27dfe1731e71c7b5f \
- --hash=sha256:12d90df9183093fcd479f4172ac26b322b1248b15729cb57f42f71f24c7e37a3 \
- --hash=sha256:15d30132718972c2c074cd14638c7f4592bd98719e2308bccea40e0538bc0cb5 \
- --hash=sha256:18821ace09c763ec93aef5eeff087ee493a24051936d7b9ebcad9662f66501f9 \
- --hash=sha256:1ae029229a57cd1e8fe542485f27e7ca7b23aa9e8944ddb4985d0bc444f1eca2 \
- --hash=sha256:2299372c19d56bcd35cf05a2738308758d32b9eaed2371898d8f5bd33f084aa3 \
- --hash=sha256:238b7ce5717600615c895050239ec955d91f321c209dd110db988500558e70d6 \
- --hash=sha256:24d50994d8c5816ddc35411e50a86ab05f575e2530c02752e02538122613371f \
- --hash=sha256:25d380fe8b1dc32cf8f0b1b448470a77afb195438bafdf1d858bfb876f3edf7b \
- --hash=sha256:2c1998e92cd5999e295a731bcb2911c75f597d937341f3030cc24ef2733d78a8 \
- --hash=sha256:2cf5bd12cecf46908f286d7838b2abc6c91cda506c0445b8223a7c19a00df008 \
- --hash=sha256:32f8dce744be5569bebe789e46727946041199030db8aeb2954d26013a0eb26b \
- --hash=sha256:37b3c1cc42aa184b3f738cfa18c1c1d72fd496d85467a6cf7b807936d39aa656 \
- --hash=sha256:3a48a78d2786784cc2413e57397981fb45c79e968d99656706018d6e62e57958 \
- --hash=sha256:3ab4aabc72de4ff77b3ec33a6d78a68227bf1123465887f9905ba79184a1cc04 \
- --hash=sha256:3c624e43ed56313651bc18a47f838b60d7b8032ed348911c54906b130b20071b \
- --hash=sha256:3f2e409836d7f5ac2f1c013110a4d50b9f7edc26328c108915f9075d7d7a91b6 \
- --hash=sha256:3f5c3e4da343bba819f0234186b9004faba952cc420fbc522dc4e103c1985908 \
- --hash=sha256:41703cc95688f2516b480f7f339d8851a6035f18e100ee6a32bc0b8536a12a9c \
- --hash=sha256:495672de149445ec1b772ff2c9ede9b769e3cb4f0d0aa7fa730d7f59e2d4e1c1 \
- --hash=sha256:4cf267add95b1c88300d96ca837833d4112756045364f5c734a2276038dae27d \
- --hash=sha256:56271f3dac49a88d7fca5060f004d9d22b865f743a12a23b1e937a0be4818ee1 \
- --hash=sha256:595ba4d8fe983b88f0eec8c26a241e16d6376fe1979086232f481f8f3f67494c \
- --hash=sha256:5f62550b9a30afde8c1c3ae450e5eb547d579dd69b25c2fc7a1c67f934c1717a \
- --hash=sha256:646d95230efb9ca614a7a594d4fcacde0ac61d25e37dd51710b36477594963ce \
- --hash=sha256:64fcc24778ca0404ce0cb7b6b77ae1f4c7231cdd60e6778f999ee05cbd581b9a \
- --hash=sha256:6be43b667360fef5c754dda5d25a32e6307a03c204f3c0fc5468b78fa87b4160 \
- --hash=sha256:6da7c2ce169267d0d066adcf63758f0604aa6c3eebf67458930f9d9b79ad1db1 \
- --hash=sha256:83d282364ea9f3e52363da262ce32a09dfe241e4080dcedda3c0db059d3c1f11 \
- --hash=sha256:9153c3292705be9f9c64498a8872118540c3f4123d1a1c840172edf262c8be4a \
- --hash=sha256:99eefd13c0dc3b3c1b4d561c1169e65fe47aab7b8158754d7c084088e2329466 \
- --hash=sha256:a0a7f52498f72f13d4a25ea70f35f4cb60642b466cbb0a9be951b5bc3f45a486 \
- --hash=sha256:a2b336e2d91a3d7006864e0990c83b216fcdca64b5a6484912902cef87313d78 \
- --hash=sha256:a48f2b74020919552ea25d222d5cc6af9ca3f4eb43a93e14d068457f545c2a17 \
- --hash=sha256:ad3d9833a64cf48cc4300f2b406c3d0f4f4724a91c0bd5640678a6ba7c102077 \
- --hash=sha256:b44d07310e404ba95f8c25aa5536f154c0a8ec473303535949e52eb71d0a1565 \
- --hash=sha256:b53285e65d4fa4c86399979e956235deb900be5baa7fc1218ea67fbfaeaadd6f \
- --hash=sha256:b5a2b97dbdc7d4f353ebf343744f1d1f1cca8aa8bfddb4262fcf4306c3761d50 \
- --hash=sha256:b9a5ca4ac220a0cdd1ba6bcba3608547117d30468fefce49bb26f55c1a3d5c58 \
- --hash=sha256:bab485bcf8b1c7d2060b4fcb6fc368a9e6f4cd754c9c2fea281f4be21df394a2 \
- --hash=sha256:c108a1d6fa78a50646029cb6d49808ff0fc1330fda87fa6f6250c6b5369b6645 \
- --hash=sha256:d56a1efd5bfd61486c8bc968fa18734464556f0fb8e51690f4ac25d85cbbbbc2 \
- --hash=sha256:d9050fee89a89ed57b4fb2c1bfac9a3d0c57a0d55aed95949eedbc42070fea39 \
- --hash=sha256:dd80ecb295460a5d9d260df63c43f4afbdd832d725a531f008dad1664f458adf \
- --hash=sha256:e8ea3e2d4066083e264e75c829078f9e149fa119d27e19acd503de65e0b13149 \
- --hash=sha256:eb3823f11823deade26ce3b9f40dcb4a213da7a670013929f31d5f5ed1055b22 \
- --hash=sha256:ee40c27c795bda6a5292e9cff9890189d32f7e3a0bf04e0e3c9430c4a00c37df \
- --hash=sha256:efb30e3baaea72ce5928e32bab719ab4770099079d66726a62b11b1ef7273be4 \
- --hash=sha256:f254d118d14a7f99d616271d6c3c27922c092dac11112670b157798b89bf4933 \
- --hash=sha256:f89c151aab2e2e23cb3fe0acad1e8b82841fd265379c4cecd0f3fcb34c15e0f6 \
- --hash=sha256:f97aeb209c3d2511443f8797e3e5a569aebb040d4f8bc79aa3ee78a8fb9e3dd8 \
- --hash=sha256:f9b587c9c7274c1613a30afabf65a272114cd6cdbe67b3406f818c79d7ab2e2a \
- --hash=sha256:fb061f596dad3a0f52b60dc6a5dec4a0c300dec41e058a7efe09256188d170b7
+matplotlib==3.10.9 \
+ --hash=sha256:09218df8a93712bd6ea133e83a153c755448cf7868316c531cffcc43f69d1cc9 \
+ --hash=sha256:10cc5ce06d10231c36f40e875f3c7e8050362a4ee8f0ee5d29a6b3277d57bb42 \
+ --hash=sha256:172db52c9e683f5d12eaf57f0f54834190e12581fe1cc2a19595a8f5acb4e77d \
+ --hash=sha256:1872fb212a05b729e649754a72d5da61d03e0554d76e80303b6f83d1d2c0552b \
+ --hash=sha256:1aa972116abb4c9d201bf245620b433726cb6856f3bef6a78f776a00f5c92d37 \
+ --hash=sha256:1e7698ac9868428e84d2c967424803b2472ff7167d9d6590d4204ed775343c3b \
+ --hash=sha256:2dc9477819ffd78ad12a20df1d9d6a6bd4fec6aaa9072681465fddca052f1456 \
+ --hash=sha256:3225f4e1edcb8c86c884ddf79ebe20ecd0a67d30188f279897554ccd8fded4dc \
+ --hash=sha256:336b9acc64d309063126edcdaca00db9373af3c476bb94388fe9c5a53ad13e6f \
+ --hash=sha256:345f6f68ecc8da0ca56fad2ea08fde1a115eda530079eca185d50a7bc3e146c6 \
+ --hash=sha256:34cf8167e023ad956c15f36302911d5406bd99a9862c1a8499ea6f7c0e015dc2 \
+ --hash=sha256:3fc0364dfbe1d07f6d15c5ebd0c5bf89e126916e5a8667dd4a7a6e84c36653d4 \
+ --hash=sha256:41cb28c2bd769aa3e98322c6ab09854cbcc52ab69d2759d681bba3e327b2b320 \
+ --hash=sha256:42fb814efabe95c06c1994d8ab5a8385f43a249e23badd3ba931d4308e5bca20 \
+ --hash=sha256:4e42042d54db34fda4e95a7bd3e5789c2a995d2dad3eb8850232ee534092fbbf \
+ --hash=sha256:4edcfbd8565339aa62f1cd4012f7180926fdbe71850f7b0d3c379c175cd6b66c \
+ --hash=sha256:51bf0ddbdc598e060d46c16b5590708f81a1624cefbaaf62f6a81bf9285b8c80 \
+ --hash=sha256:56fc0bd271b00025c6edfdc7c2dcd247372c8e1544971d62e1dc7c17367e8bf9 \
+ --hash=sha256:59476c6d29d612b8e9bb6ce8c5b631be6ba8f9e3a2421f22a02b192c7dd28716 \
+ --hash=sha256:6640f75af2c6148293caa0a2b39dd806a492dd66c8a8b04035813e33d0fd2585 \
+ --hash=sha256:68cfdcede415f7c8f5577b03303dd94526cdb6d11036cecdc205e08733b2d2bb \
+ --hash=sha256:6b63d9c7c769b88ab81e10dc86e4e0607cf56817b9f9e6cf24b2a5f1693b8e38 \
+ --hash=sha256:6be157fe17fc37cb95ac1d7374cf717ce9259616edec911a78d9d26dae8522d4 \
+ --hash=sha256:6c63ebcd8b4b169eb2f5c200552ae6b8be8999a005b6b507ed76fb8d7d674fe2 \
+ --hash=sha256:77210dce9cb8153dffc967efaae990543392563d5a376d4dd8539bebcb0ed217 \
+ --hash=sha256:7a8d66a55def891c33147ba3ba9bfcabf0b526a43764c818acbb4525e5ed0838 \
+ --hash=sha256:82368699727bfb7b0182e1aa13082e3c08e092fa1a25d3e1fd92405bff96f6d4 \
+ --hash=sha256:82834c3c292d24d3a8aae77cd2d20019de69d692a34a970e4fdb8d33e2ea3dda \
+ --hash=sha256:8e436d155fa8a3399dc62683f8f5d0e2e50d25d0144a73edd73f82eec8f4abfb \
+ --hash=sha256:8f3bcac1ca5ed000a6f4337d47ba67dfddf37ed6a46c15fd7f014997f7bf865f \
+ --hash=sha256:97e35e8d39ccc85859095e01a53847432ba9a53ddf7986f7a54a11b73d0e143f \
+ --hash=sha256:985f2238880e2e69093f588f5fe2e46771747febf0649f3cf7f7b7480875317f \
+ --hash=sha256:a49f1eadc84ca85fd72fa4e89e70e61bf86452df6f971af04b12c60761a0772c \
+ --hash=sha256:a5a6104ed666402ba5106d7f36e0e0cdca4e8d7fa4d39708ca88019e2835a2eb \
+ --hash=sha256:aba1615dabe83188e19d4f75a253c6a08423e04c1425e64039f800050a69de6b \
+ --hash=sha256:ae20801130378b82d647ff5047c07316295b68dc054ca6b3c13519d0ea624285 \
+ --hash=sha256:ae2f11957b27ce53497dd4d7b235c4d4f1faf383dfb39d0c5beb833bff883294 \
+ --hash=sha256:b049278ddce116aaa1c1377ebf58adea909132dfce0281cf7e3a1ea9fc2e2c65 \
+ --hash=sha256:b1b745c489cd1a77a0dc1120a05dc87af9798faebc913601feb8c73d89bf2d1e \
+ --hash=sha256:b2b9516251cb89ff618d757daec0e2ed1bf21248013844a853d87ef85ab3081d \
+ --hash=sha256:b580440f1ff81a0e34122051a3dfabb7e4b7f9e380629929bde0eff9af72165f \
+ --hash=sha256:ba7b3b8ef09eab7df0e86e9ae086faa433efbfbdb46afcb3aa16aabf779469a8 \
+ --hash=sha256:c27df8b3848f32a83d1767566595e43cfaa4460380974da06f4279a7ec143c39 \
+ --hash=sha256:d091f9d758b34aaaaa6331d13574bf01891d903b3dec59bfff458ef7551de5d6 \
+ --hash=sha256:d730e984eddf56974c3e72b6129c7ca462ac38dc624338f4b0b23eb23ecba00f \
+ --hash=sha256:d75d11c949914165976c621b2324f9ef162af7ebf4b057ddf95dd1dba7e5edcf \
+ --hash=sha256:d843374407c4017a6403b59c6c81606773d136f3259d5b6da3131bc814542cc2 \
+ --hash=sha256:da4e09638420548f31c354032a6250e473c68e5a4e96899b4844cf39ddea23fe \
+ --hash=sha256:de2445a0c6690d21b7eb6ce071cebad6d40a2e9bdf10d039074a96ba19797b99 \
+ --hash=sha256:dfca0129678bd56379db26c52b5d77ed7de314c047492fbdc763aa7501710cfb \
+ --hash=sha256:e9fae004b941b23ff2edcf1567a857ed77bafc8086ffa258190462328434faf8 \
+ --hash=sha256:f0c3c28d9fbcc1fe7a03be236d73430cf6409c41fb2383a7ac52fe932b072cb1 \
+ --hash=sha256:f4399f64b3e94cd500195490972ae1ee81170df1636fa15364d157d5bdd7b921 \
+ --hash=sha256:f76e640a5268850bfda54b5131b1b1941cc685e42c5fa98ed9f2d64038308cba \
+ --hash=sha256:fd66508e8c6877d98e586654b608a0456db8d7e8a546eb1e2600efd957302358
mdurl==0.1.2 \
--hash=sha256:84008a41e51615a49fc9966191ff91509e3c40b939176e643fd50a5c2196b8f8 \
--hash=sha256:bb413d29f5eea38f31dd4754dd7377d4465116fb207585f97bf925588687c1ba
@@ -757,29 +844,29 @@ mpmath==1.3.0 \
networkx==3.1 \
--hash=sha256:4f33f68cb2afcf86f28a45f43efc27a9386b535d567d2127f8f61d51dec58d36 \
--hash=sha256:de346335408f84de0eada6ff9fafafff9bcda11f0a0dfaa931133debb146ab61
-ninja==1.13.0 \
- --hash=sha256:11be2d22027bde06f14c343f01d31446747dbb51e72d00decca2eb99be911e2f \
- --hash=sha256:1c97223cdda0417f414bf864cfb73b72d8777e57ebb279c5f6de368de0062988 \
- --hash=sha256:3c0b40b1f0bba764644385319028650087b4c1b18cdfa6f45cb39a3669b81aa9 \
- --hash=sha256:3d00c692fb717fd511abeb44b8c5d00340c36938c12d6538ba989fe764e79630 \
- --hash=sha256:3d7d7779d12cb20c6d054c61b702139fd23a7a964ec8f2c823f1ab1b084150db \
- --hash=sha256:4a40ce995ded54d9dc24f8ea37ff3bf62ad192b547f6c7126e7e25045e76f978 \
- --hash=sha256:4be9c1b082d244b1ad7ef41eb8ab088aae8c109a9f3f0b3e56a252d3e00f42c1 \
- --hash=sha256:5f8e1e8a1a30835eeb51db05cf5a67151ad37542f5a4af2a438e9490915e5b72 \
- --hash=sha256:60056592cf495e9a6a4bea3cd178903056ecb0943e4de45a2ea825edb6dc8d3e \
- --hash=sha256:6739d3352073341ad284246f81339a384eec091d9851a886dfa5b00a6d48b3e2 \
- --hash=sha256:8cfbb80b4a53456ae8a39f90ae3d7a2129f45ea164f43fadfa15dc38c4aef1c9 \
- --hash=sha256:aa45b4037b313c2f698bc13306239b8b93b4680eb47e287773156ac9e9304714 \
- --hash=sha256:b4f2a072db3c0f944c32793e91532d8948d20d9ab83da9c0c7c15b5768072200 \
- --hash=sha256:be7f478ff9f96a128b599a964fc60a6a87b9fa332ee1bd44fa243ac88d50291c \
- --hash=sha256:d741a5e6754e0bda767e3274a0f0deeef4807f1fec6c0d7921a0244018926ae5 \
- --hash=sha256:e8bad11f8a00b64137e9b315b137d8bb6cbf3086fbdc43bf1f90fd33324d2e96 \
- --hash=sha256:fa2a8bfc62e31b08f83127d1613d10821775a0eb334197154c4d6067b7068ff1 \
- --hash=sha256:fb46acf6b93b8dd0322adc3a4945452a4e774b75b91293bafcc7b7f8e6517dfa \
- --hash=sha256:fb8ee8719f8af47fed145cced4a85f0755dd55d45b2bddaf7431fa89803c5f3e
-nncf==3.1.0 \
- --hash=sha256:cabfdc32f42a04c6c106906d1b248ace0ed0748c46b193529ec7b4e945baa374 \
- --hash=sha256:f3bcd71c537554bb90100cd91fb999904e1e222cac61eee6df7af7cf58ccd646
+ninja==1.13.2 \
+ --hash=sha256:09de9ab04f7352f51570c73fd4913acb1e6c24be0a72cd8b80243d4d3ed04925 \
+ --hash=sha256:0e083700470c02ca154a855ae6d692d03564f5064cae52e113896f9ccc078418 \
+ --hash=sha256:1293f4078278b70d0ee4b6cc8f3a9e030656c9b2f59909970343c4fe76070118 \
+ --hash=sha256:1684c60d031c54c1d049541b64243c0c567dca5463dbd77682a8901780af293d \
+ --hash=sha256:1db9852e528efa7702f5123969f86678663e46d57ff28ab13f5fd84d64a85fb1 \
+ --hash=sha256:227cbc3ae3e5e429692388103cae8c09451df086cd2d342dae0795af0d162547 \
+ --hash=sha256:525bfa3fc88aa30a4467df270fd5be6f9fcae8061d54d4df74ea1dc5abd5a975 \
+ --hash=sha256:59d71c3e15b6b6f3d903eb0c27285544e0747ca59925ada7037bb1af781ad4b3 \
+ --hash=sha256:65a24341b5ac09fcadcc37082660be40a94174e51a937fabf6e2cae26225fa2c \
+ --hash=sha256:6a87bf42b123abe2f37737300185f0a303a891899da85d73a3613ee80547e578 \
+ --hash=sha256:792cadbb9decfd1f776d4d0a6930feb46d08302eb57c176bcf26b09de5748e9f \
+ --hash=sha256:81d95081c0ad7c95f67bf682220361ed2a32659d7861b4766b03433c58f22516 \
+ --hash=sha256:915bd482c4be41c75120fd67a22e0bb3f0fbb3bbc5f95b89787deadd59e27ef2 \
+ --hash=sha256:919572cbc3f233261ecd41fe1f3efc9d44aa02464a4588867e06a8b4f6f416ea \
+ --hash=sha256:aa3d2ae5706a2c4d1e93edc951d1c6cbb45107413c404f8fde1741239efbc9a0 \
+ --hash=sha256:b2f687437fac460b27b7eadc99039b1163016fb4ba7276e2782a192d9f24ee0e \
+ --hash=sha256:d775a5e43e9088f507a6250d57fcf5678eb31268c545feb5064ffeee33735622 \
+ --hash=sha256:f90f84affc441e219f15fe52532806c1c9dbd22fb66c3addddce88a3deaabab7 \
+ --hash=sha256:fd82e26c0706ad4ab88e5fdd26f3fab0a987a90f810160f6c322e752c6af298b
+nncf==3.3.0 \
+ --hash=sha256:3dbcbc1ad4f399deed139041fae1fedd4b8aa5749e30af522fcc5a6c2af73e42 \
+ --hash=sha256:855f1e7099f01ff3a9868cf6d56850e1149691ab0db5ac76ab812ae8b0fba338
numpy==1.26.4 \
--hash=sha256:03a8c78d01d9781b28a6989f6fa1bb2c4f2d51201cf99d3dd875df6fbd96b23b \
--hash=sha256:08beddf13648eb95f8d867350f6a018a4be2e5ad54c8d8caed89ebca558b2818 \
@@ -860,6 +947,9 @@ nvidia-cusparselt-cu12==0.7.1 \
--hash=sha256:8878dce784d0fac90131b6817b607e803c36e629ba34dc5b433471382196b6a5 \
--hash=sha256:f1bb701d6b930d5a7cea44c19ceb973311500847f81b634d802b7b539dc55623 \
--hash=sha256:f67fbb5831940ec829c9117b7f33807db9f9678dc2a617fbe781cac17b4e1075
+nvidia-ml-py==13.610.43 \
+ --hash=sha256:65437eb73d68d0c62c931ca4d45038472faff03bd0b8729abba4b899f70d60f2 \
+ --hash=sha256:f13c72698edef492f985cc225f14faafe68ae065a2e407f45bdf6f4b9b43fde8
nvidia-nccl-cu12==2.27.5 \
--hash=sha256:31432ad4d1fb1004eb0c56203dc9bc2178a1ba69d1d9e02d64a6938ab5e40e7a \
--hash=sha256:ad730cf15cb5d25fe849c6e6ca9eb5b76db16a80f13f425ac68d8e2e55624457
@@ -874,35 +964,31 @@ nvidia-nvtx-cu12==12.8.90 \
--hash=sha256:5b17e2001cc0d751a5bc2c6ec6d26ad95913324a4adb86788c944f8ce9ba441f \
--hash=sha256:619c8304aedc69f02ea82dd244541a83c3d9d40993381b3b590f1adaed3db41e \
--hash=sha256:d7ad891da111ebafbf7e015d34879f7112832fc239ff0d7d776b6cb685274615
-onnx==1.21.0 \
- --hash=sha256:10c3185a232089335581fabb98fba4e86d3e8246b8140f2e406082438100ebda \
- --hash=sha256:19d9971a3e52a12968ae6c70fd0f86c349536de0b0c33922ecdbe52d1972fe60 \
- --hash=sha256:1a9baf882562c4cebf79589bebb7cd71a20e30b51158cac3e3bbaf27da6163bd \
- --hash=sha256:257d1d1deb6a652913698f1e3f33ef1ca0aa69174892fe38946d4572d89dd94f \
- --hash=sha256:2aca19949260875c14866fc77ea0bc37e4e809b24976108762843d328c92d3ce \
- --hash=sha256:3abd09872523c7e0362d767e4e63bd7c6bac52a5e2c3edbf061061fe540e2027 \
- --hash=sha256:458d91948ad9a7729a347550553b49ab6939f9af2cddf334e2116e45467dc61f \
- --hash=sha256:4d8b67d0aaec5864c87633188b91cc520877477ec0254eda122bef8be43cd764 \
- --hash=sha256:5489f25fe461e7f32128218251a466cabbeeaf1eaa791c79daebf1a80d5a2cc9 \
- --hash=sha256:5f78c411743db317a76e5d009f84f7e3d5380411a1567a868e82461a1e5c775d \
- --hash=sha256:7b58a4cfec8d9311b73dc083e4c1fa362069267881144c05139b3eba5dc3a840 \
- --hash=sha256:7cd7cb8f6459311bdb557cbf6c0ccc6d8ace11c304d1bba0a30b4a4688e245f8 \
- --hash=sha256:7ee9d8fd6a4874a5fa8b44bbcabea104ce752b20469b88bc50c7dcf9030779ad \
- --hash=sha256:82aa6ab51144df07c58c4850cb78d4f1ae969d8c0bf657b28041796d49ba6974 \
- --hash=sha256:9003d5206c01fa2ff4b46311566865d8e493e1a6998d4009ec6de39843f1b59b \
- --hash=sha256:9ea4e824964082811938a9250451d89c4ec474fe42dd36c038bfa5df31993d1e \
- --hash=sha256:a9261bd580fb8548c9c37b3c6750387eb8f21ea43c63880d37b2c622e1684285 \
- --hash=sha256:ab6a488dabbb172eebc9f3b3e7ac68763f32b0c571626d4a5004608f866cc83d \
- --hash=sha256:bba12181566acf49b35875838eba49536a327b2944664b17125577d230c637ad \
- --hash=sha256:c9b56ad04039fac6b028c07e54afa1ec7f75dd340f65311f2c292e41ed7aa4d9 \
- --hash=sha256:ca14bc4842fccc3187eb538f07eabeb25a779b39388b006db4356c07403a7bbb \
- --hash=sha256:db17fc0fec46180b6acbd1d5d8650a04e5527c02b09381da0b5b888d02a204c8 \
- --hash=sha256:e0c21cc5c7a41d1a509828e2b14fe9c30e807c6df611ec0fd64a47b8d4b16abd \
- --hash=sha256:e1931bfcc222a4c9da6475f2ffffb84b97ab3876041ec639171c11ce802bee6a \
- --hash=sha256:efba467efb316baf2a9452d892c2f982b9b758c778d23e38c7f44fa211b30bb9 \
- --hash=sha256:f2c7c234c568402e10db74e33d787e4144e394ae2bcbbf11000fbfe2e017ad68 \
- --hash=sha256:f53b3c15a3b539c16b99655c43c365622046d68c49b680c48eba4da2a4fb6f27 \
- --hash=sha256:fc2635400fe39ff37ebc4e75342cc54450eadadf39c540ff132c319bf4960095
+onnx==1.22.0 \
+ --hash=sha256:19e45e4af88e3fe3261458d4b8cc461957ae2782a358a3560503569bf3b23b72 \
+ --hash=sha256:1d0a2bdb15eb2b3cb65c438f3423d9620d14fdce32f92380e6bb1b2e09568ef5 \
+ --hash=sha256:239958534464612fbcb6ed23d5228aaa925b39b8773f58726809ffdccb4edd1c \
+ --hash=sha256:2632406b8f523ef2e2873c363f90b20a3d88c0fbcfac757d3addffccf8f452c2 \
+ --hash=sha256:2d8f229a553fa440fe623ed7b36fca5e7762da3af871c3f8f8ce451df73e2914 \
+ --hash=sha256:33ce94119bbb7f05d9caea4ea7549f5185a54369f6bbc9f70171bd5ee6935bbc \
+ --hash=sha256:596fbf0490947533c1c1045ba860851dc9fb77471023dac9a71ba5b42ceab103 \
+ --hash=sha256:5c1c0408a9d4b4df33851672e5fc7590b96301ee123396d608f9ab6f045ab06b \
+ --hash=sha256:6d0ffffd63a4ecc21ddaeddd5bf02099cb701aa4243f2de00122726869065ca4 \
+ --hash=sha256:72ccebab3bac07215c204ce8848d42e78eaaa666badbf72d25cd359b9f269e3a \
+ --hash=sha256:82e9f27fc1223cb06d68a56bed6f9d3caf3d0dad1b61bce45006d529b15bd94c \
+ --hash=sha256:8561a2c00041c07e08db0c228593b5b4694100398685f348532af7dbb84189da \
+ --hash=sha256:87a3077958f66f9a26dec10077ac28326d9cec2cbe1f0b040947243449754573 \
+ --hash=sha256:8907b9b9389893bc0dc6314cc00ee1e3a69844e48d689eacc6a0340411a7da58 \
+ --hash=sha256:8a5eccce2d5fc6c5046928a9aa7cdd9750ea4a586f8de341d3d40d820c35fdec \
+ --hash=sha256:8e268cdc0547e3949799ffd4a44451dc2b9080b57d0824a2db680b6ec65506f0 \
+ --hash=sha256:955e02e1f6d385b53d52f9cd7b9cdf5caf417c300bcfe3c64c6d542be763845b \
+ --hash=sha256:a1a89a7cb9ba13d78f009bdec448ec82a98972589734f157022a2bff7a5973a6 \
+ --hash=sha256:a3a39fc4643867aecb33417fdddb11e308ee79d2d4a584b9d50cc7aec2091b13 \
+ --hash=sha256:ae5a563f281cd9d2845622cecf6c092a57e4ee1b138f66fdbbdd4200567a5e16 \
+ --hash=sha256:c21a0e59fd967a95b358e4a6e756d1f1eec2d304a83480f329f66e30d2bf0223 \
+ --hash=sha256:cc8b66b312f8f03a53e268afb67180a2d97dd12cc79e2b61361c6c0073448016 \
+ --hash=sha256:ef40c0aaf0b643857ea9306fc7eddce17eaf9fb0407e4801f1fc5758443a38e0 \
+ --hash=sha256:f3c120dcdb70ad738f3c061b32798f408ea299eb69f84dd69ab4a6bf3c2ec01f
onnxruntime-gpu==1.23.2 \
--hash=sha256:054282614c2fc9a4a27d74242afbae706a410f1f63cc35bc72f99709029a5ba4 \
--hash=sha256:18de50c6c8eea50acc405ea13d299aec593e46478d7a22cd32cdbbdf7c42899d \
@@ -913,9 +999,9 @@ onnxruntime-gpu==1.23.2 \
--hash=sha256:d76d1ac7a479ecc3ac54482eea4ba3b10d68e888a0f8b5f420f0bdf82c5eec59 \
--hash=sha256:deba091e15357355aa836fd64c6c4ac97dd0c4609c38b08a69675073ea46b321 \
--hash=sha256:fe925a84b00e291e0ad3fac29bfd8f8e06112abc760cdc82cb711b4f3935bd95
-onnxslim==0.1.91 \
- --hash=sha256:1fdb23ca56ca3f9d12ff7b9ae1c779184468eaf911baea0930ac254b49c93ac9 \
- --hash=sha256:3222fcd9a16836a4e7917e1dc55ddf4db492dadca3f469a1fa9226f92c4a2c6e
+onnxslim==0.1.96 \
+ --hash=sha256:386a1c64e61a66fb79cdd62de996323b9fb950d22de9b117cd31f211cac96048 \
+ --hash=sha256:a43f800d39434e6b4d542abe17c25afb7fcc33aabf53d20464cb0dae0f66cbb6
opencv-python==4.11.0.86 \
--hash=sha256:03d60ccae62304860d232272e4a4fda93c39d595780cb40b161b310244b736a4 \
--hash=sha256:085ad9b77c18853ea66283e98affefe2de8cc4c1f43eda4c100cf9b2721142ec \
@@ -950,117 +1036,113 @@ openvino-dev==2024.6.0 \
openvino-telemetry==2025.2.0 \
--hash=sha256:8bf8127218e51e99547bf38b8fb85a8b31c9bf96e6f3a82eb0b3b6a34155977c \
--hash=sha256:bcb667e83a44f202ecf4cfa49281715c6d7e21499daec04ff853b7f964833599
-packaging==26.1 \
- --hash=sha256:5d9c0669c6285e491e0ced2eee587eaf67b670d94a19e94e3984a481aba6802f \
- --hash=sha256:f042152b681c4bfac5cae2742a55e103d27ab2ec0f3d88037136b6bfe7c9c5de
-pillow==12.2.0 \
- --hash=sha256:00a2865911330191c0b818c59103b58a5e697cae67042366970a6b6f1b20b7f9 \
- --hash=sha256:01afa7cf67f74f09523699b4e88c73fb55c13346d212a59a2db1f86b0a63e8c5 \
- --hash=sha256:03e7e372d5240cc23e9f07deca4d775c0817bffc641b01e9c3af208dbd300987 \
- --hash=sha256:03f6fab9219220f041c74aeaa2939ff0062bd5c364ba9ce037197f4c6d498cd9 \
- --hash=sha256:042db20a421b9bafecc4b84a8b6e444686bd9d836c7fd24542db3e7df7baad9b \
- --hash=sha256:0538bd5e05efec03ae613fd89c4ce0368ecd2ba239cc25b9f9be7ed426b0af1f \
- --hash=sha256:0a34329707af4f73cf1782a36cd2289c0368880654a2c11f027bcee9052d35dd \
- --hash=sha256:0c838a5125cee37e68edec915651521191cef1e6aa336b855f495766e77a366e \
- --hash=sha256:144748b3af2d1b358d41286056d0003f47cb339b8c43a9ea42f5fea4d8c66b6e \
- --hash=sha256:1610dd6c61621ae1cf811bef44d77e149ce3f7b95afe66a4512f8c59f25d9ebe \
- --hash=sha256:1e1757442ed87f4912397c6d35a0db6a7b52592156014706f17658ff58bbf795 \
- --hash=sha256:22db17c68434de69d8ecfc2fe821569195c0c373b25cccb9cbdacf2c6e53c601 \
- --hash=sha256:25373b66e0dd5905ed63fa3cae13c82fbddf3079f2c8bf15c6fb6a35586324c1 \
- --hash=sha256:2bb4a8d594eacdfc59d9e5ad972aa8afdd48d584ffd5f13a937a664c3e7db0ed \
- --hash=sha256:2c727a6d53cb0018aadd8018c2b938376af27914a68a492f59dfcaca650d5eea \
- --hash=sha256:2d192a155bbcec180f8564f693e6fd9bccff5a7af9b32e2e4bf8c9c69dbad6b5 \
- --hash=sha256:2e589959f10d9824d39b350472b92f0ce3b443c0a3442ebf41c40cb8361c5b97 \
- --hash=sha256:2e5a76d03a6c6dcef67edabda7a52494afa4035021a79c8558e14af25313d453 \
- --hash=sha256:325ca0528c6788d2a6c3d40e3568639398137346c3d6e66bb61db96b96511c98 \
- --hash=sha256:34c0d99ecccea270c04882cb3b86e7b57296079c9a4aff88cb3b33563d95afaa \
- --hash=sha256:390ede346628ccc626e5730107cde16c42d3836b89662a115a921f28440e6a3b \
- --hash=sha256:394167b21da716608eac917c60aa9b969421b5dcbbe02ae7f013e7b85811c69d \
- --hash=sha256:3997232e10d2920a68d25191392e3a4487d8183039e1c74c2297f00ed1c50705 \
- --hash=sha256:3adc9215e8be0448ed6e814966ecf3d9952f0ea40eb14e89a102b87f450660d8 \
- --hash=sha256:3e080565d8d7c671db5802eedfb438e5565ffa40115216eabb8cd52d0ecce024 \
- --hash=sha256:4a6c9fa44005fa37a91ebfc95d081e8079757d2e904b27103f4f5fa6f0bf78c0 \
- --hash=sha256:4bfd07bc812fbd20395212969e41931001fd59eb55a60658b0e5710872e95286 \
- --hash=sha256:4e6c62e9d237e9b65fac06857d511e90d8461a32adcc1b9065ea0c0fa3a28150 \
- --hash=sha256:50d8520da2a6ce0af445fa6d648c4273c3eeefbc32d7ce049f22e8b5c3daecc2 \
- --hash=sha256:51c4167c34b0d8ba05b547a3bb23578d0ba17b80a5593f93bd8ecb123dd336a3 \
- --hash=sha256:56a3f9c60a13133a98ecff6197af34d7824de9b7b38c3654861a725c970c197b \
- --hash=sha256:56b25336f502b6ed02e889f4ece894a72612fe885889a6e8c4c80239ff6e5f5f \
- --hash=sha256:57850958fe9c751670e49b2cecf6294acc99e562531f4bd317fa5ddee2068463 \
- --hash=sha256:58f62cc0f00fd29e64b29f4fd923ffdb3859c9f9e6105bfc37ba1d08994e8940 \
- --hash=sha256:5c0a9f29ca8e79f09de89293f82fc9b0270bb4af1d58bc98f540cc4aedf03166 \
- --hash=sha256:5cdfebd752ec52bf5bb4e35d9c64b40826bc5b40a13df7c3cda20a2c03a0f5ed \
- --hash=sha256:5d04bfa02cc2d23b497d1e90a0f927070043f6cbf303e738300532379a4b4e0f \
- --hash=sha256:5d2fd0fa6b5d9d1de415060363433f28da8b1526c1c129020435e186794b3795 \
- --hash=sha256:62f5409336adb0663b7caa0da5c7d9e7bdbaae9ce761d34669420c2a801b2780 \
- --hash=sha256:632ff19b2778e43162304d50da0181ce24ac5bb8180122cbe1bf4673428328c7 \
- --hash=sha256:6562ace0d3fb5f20ed7290f1f929cae41b25ae29528f2af1722966a0a02e2aa1 \
- --hash=sha256:673aa32138f3e7531ccdbca7b3901dba9b70940a19ccecc6a37c77d5fdeb05b5 \
- --hash=sha256:6a6e67ea2e6feda684ed370f9a1c52e7a243631c025ba42149a2cc5934dec295 \
- --hash=sha256:6a9adfc6d24b10f89588096364cc726174118c62130c817c2837c60cf08a392b \
- --hash=sha256:6bb77b2dcb06b20f9f4b4a8454caa581cd4dd0643a08bacf821216a16d9c8354 \
- --hash=sha256:6e6b2a0c538fc200b38ff9eb6628228b77908c319a005815f2dde585a0664b60 \
- --hash=sha256:71cde9a1e1551df7d34a25462fc60325e8a11a82cc2e2f54578e5e9a1e153d65 \
- --hash=sha256:7371b48c4fa448d20d2714c9a1f775a81155050d383333e0a6c15b1123dda005 \
- --hash=sha256:766cef22385fa1091258ad7e6216792b156dc16d8d3fa607e7545b2b72061f1c \
- --hash=sha256:7b14cc0106cd9aecda615dd6903840a058b4700fcb817687d0ee4fc8b6e389be \
- --hash=sha256:7f84204dee22a783350679a0333981df803dac21a0190d706a50475e361c93f5 \
- --hash=sha256:8023abc91fba39036dbce14a7d6535632f99c0b857807cbbbf21ecc9f4717f06 \
- --hash=sha256:80b2da48193b2f33ed0c32c38140f9d3186583ce7d516526d462645fd98660ae \
- --hash=sha256:8297651f5b5679c19968abefd6bb84d95fe30ef712eb1b2d9b2d31ca61267f4c \
- --hash=sha256:88d387ff40b3ff7c274947ed3125dedf5262ec6919d83946753b5f3d7c67ea4c \
- --hash=sha256:88ddbc66737e277852913bd1e07c150cc7bb124539f94c4e2df5344494e0a612 \
- --hash=sha256:8bd7903a5f2a4545f6fd5935c90058b89d30045568985a71c79f5fd6edf9b91e \
- --hash=sha256:8be29e59487a79f173507c30ddf57e733a357f67881430449bb32614075a40ab \
- --hash=sha256:8c984051042858021a54926eb597d6ee3012393ce9c181814115df4c60b9a808 \
- --hash=sha256:8cbeb542b2ebc6fcdacabf8aca8c1a97c9b3ad3927d46b8723f9d4f033288a0f \
- --hash=sha256:8e9c4f5b3c546fa3458a29ab22646c1c6c787ea8f5ef51300e5a60300736905e \
- --hash=sha256:90e6f81de50ad6b534cab6e5aef77ff6e37722b2f5d908686f4a5c9eba17a909 \
- --hash=sha256:975385f4776fafde056abb318f612ef6285b10a1f12b8570f3647ad0d74b48ec \
- --hash=sha256:9a8a34cc89c67a65ea7437ce257cea81a9dad65b29805f3ecee8c8fe8ff25ffe \
- --hash=sha256:9aba9a17b623ef750a4d11b742cbafffeb48a869821252b30ee21b5e91392c50 \
- --hash=sha256:9f08483a632889536b8139663db60f6724bfcb443c96f1b18855860d7d5c0fd4 \
- --hash=sha256:a4e8f36e677d3336f35089648c8955c51c6d386a13cf6ee9c189c5f5bd713a9f \
- --hash=sha256:a52edc8bfff4429aaabdf4d9ee0daadbbf8562364f940937b941f87a4290f5ff \
- --hash=sha256:a830b1a40919539d07806aa58e1b114df53ddd43213d9c8b75847eee6c0182b5 \
- --hash=sha256:aa88ccfe4e32d362816319ed727a004423aab09c5cea43c01a4b435643fa34eb \
- --hash=sha256:af73337013e0b3b46f175e79492d96845b16126ddf79c438d7ea7ff27783a414 \
- --hash=sha256:b1c1fbd8a5a1af3412a0810d060a78b5136ec0836c8a4ef9aa11807f2a22f4e1 \
- --hash=sha256:b85f66ae9eb53e860a873b858b789217ba505e5e405a24b85c0464822fe88032 \
- --hash=sha256:b86024e52a1b269467a802258c25521e6d742349d760728092e1bc2d135b4d76 \
- --hash=sha256:bd9c0c7a0c681a347b3194c500cb1e6ca9cab053ea4d82a5cf45b6b754560136 \
- --hash=sha256:bfa9c230d2fe991bed5318a5f119bd6780cda2915cca595393649fc118ab895e \
- --hash=sha256:d362d1878f00c142b7e1a16e6e5e780f02be8195123f164edf7eddd911eefe7c \
- --hash=sha256:d5d38f1411c0ed9f97bcb49b7bd59b6b7c314e0e27420e34d99d844b9ce3b6f3 \
- --hash=sha256:dac8d77255a37e81a2efcbd1fc05f1c15ee82200e6c240d7e127e25e365c39ea \
- --hash=sha256:dd025009355c926a84a612fecf58bb315a3f6814b17ead51a8e48d3823d9087f \
- --hash=sha256:deede7c263feb25dba4e82ea23058a235dcc2fe1f6021025dc71f2b618e26104 \
- --hash=sha256:e74473c875d78b8e9d5da2a70f7099549f9eb37ded4e2f6a463e60125bccd176 \
- --hash=sha256:ee3120ae9dff32f121610bb08e4313be87e03efeadfc6c0d18f89127e24d0c24 \
- --hash=sha256:eedf4b74eda2b5a4b2b2fb4c006d6295df3bf29e459e198c90ea48e130dc75c3 \
- --hash=sha256:efd8c21c98c5cc60653bcb311bef2ce0401642b7ce9d09e03a7da87c878289d4 \
- --hash=sha256:f1c943e96e85df3d3478f7b691f229887e143f81fedab9b20205349ab04d73ed \
- --hash=sha256:f278f034eb75b4e8a13a54a876cc4a5ab39173d2cdd93a638e1b467fc545ac43 \
- --hash=sha256:f3f40b3c5a968281fd507d519e444c35f0ff171237f4fdde090dd60699458421 \
- --hash=sha256:f490f9368b6fc026f021db16d7ec2fbf7d89e2edb42e8ec09d2c60505f5729c7 \
- --hash=sha256:fb043ee2f06b41473269765c2feae53fc2e2fbf96e5e22ca94fb5ad677856f06 \
- --hash=sha256:fc3d34d4a8fbec3e88a79b92e5465e0f9b842b628675850d860b8bd300b159f5
+packaging==26.3 \
+ --hash=sha256:94edc256424af38762eb31306eed28beb9f0efc50a8837492c9d6fd6004aed79 \
+ --hash=sha256:d7193f7c8e4e93f444fde0262bf90af30e16fa0ad0ad44cb553c87339b23cd1c
+pillow==12.3.0 \
+ --hash=sha256:00808c5e14ef63ac5161091d242999076604ff74b883423a11e5d7bbb38bf756 \
+ --hash=sha256:04f01d28a6aaff387bf842a13be313df23ba0597a44f1a976c9feb3c6ff4711a \
+ --hash=sha256:06ff022112bc9cbf83b60f8e028d94ad87b60621706487e65f673de61610ab59 \
+ --hash=sha256:0740a512dc522224c77d9aa5a8d70d8b7d73fb91f2c21125d8d025d3b8990e45 \
+ --hash=sha256:0847a763afefb695bc912d7c131e7e0632d4edc1d8698f58ddabec8e46b8b6d3 \
+ --hash=sha256:0dd2064cbc55aaec028ef5fbb60fa47bb6c3e7918e07ff17935284b227a9d2df \
+ --hash=sha256:0feb2e9d6ad6c9e3c06effe9d00f3f1e618a6643273576b016f591e9315a7139 \
+ --hash=sha256:10e41f0fbf1eec8cfd234b8fe17a4caac7c9d0db4c204d3c173a8f9f6ef3232b \
+ --hash=sha256:1182d52bc2d5e5d7d0949503aa7e36d12f42205dc287e4883f407b1988820d39 \
+ --hash=sha256:164b31cd1a0490ab6efae01aa5df49da7061be0af1b30e035b6e9a1bfe34ee6e \
+ --hash=sha256:1657923d2d45afb66526e5b933e5b3052e6bdea196c90d3abb2424e18c77dae8 \
+ --hash=sha256:186941b6aef820ad110fb01fb06eb925374dc3a21b17e37ec9a53b250c6fe2d1 \
+ --hash=sha256:1cca606cd25738df4ed873d5ad46bbdb3d83b5cbca291f6b4ff13a4df6b0bbe8 \
+ --hash=sha256:21900ce7ba264168cd50defae43cd75d25c833ad4ad6e73ffc5596d12e25ac89 \
+ --hash=sha256:236ff70b9312fb68943c703aa842ca6a758abfa45ac187a5e7c1452e96ef72b5 \
+ --hash=sha256:23aceaa007d6172b02c277f0cd359c79492bbb14f7072b4ede9fbcaf20648130 \
+ --hash=sha256:23d27a3e0307ec2244cc51e7287b919aa68d097504ebe19df4e76a98a3eea5bd \
+ --hash=sha256:24870b09b224f7ae3c39ed07d10e819d06f8720bc551847b1d623832b5b0e28d \
+ --hash=sha256:251bf95b67017e27b13d82f5b326234ca62d70f9cf4c2b9032de2358a3b12c7b \
+ --hash=sha256:25b9b82bb22e6e2b3cd07b39c68b7b862001226cb3dff7130d1cb914121b39ed \
+ --hash=sha256:28ce87c5ab450a9dd970b52e5aca5fe63ed432d18a2eaddd1979a00a1ba24ace \
+ --hash=sha256:300557495eb45ebb8aec96c2da9c4be642fbf7cd937278b4013ba894ea8eb0eb \
+ --hash=sha256:30f2aa603c41533cc25c05acd0da21636e84a315768feb631c937177db558931 \
+ --hash=sha256:331b624368d4f1d069149002f25f44bc61c8919ce8ddb3c45bdad8f6e2d89510 \
+ --hash=sha256:37d6d0a00072fd2948eb22bce7e1475f34569d90c87c59f7a2ec59541b77f7a6 \
+ --hash=sha256:37dc8f7bbb66efe481bb60defacef820c950c24713fb44962ed6aa2a50966de1 \
+ --hash=sha256:3b8182a766685eaa002637e28b4ec8d6b18819a0c71f579bf0dbaa5830297cce \
+ --hash=sha256:3edce1d53195db527e0191f84b71d02022de0540bf43a16ed734ed7537b07385 \
+ --hash=sha256:446c34dcc4324b084a53b705127dc15717b22c5e140ae0a3c38349d4efec071e \
+ --hash=sha256:4998562bf62a445225f22e07c896bb04b35b1b1f2eb6d760584c9c51d7a5f78c \
+ --hash=sha256:4b0a7fe987b14c31ebda6083f74f22b561fd3739bc0ac51e019622e3d72668c7 \
+ --hash=sha256:4e8c2a84d977f50b9daed6eeaf3baef67d00d5d74d932288f02cb94518ee3ace \
+ --hash=sha256:4f883547d4b7f0495ebe7056b0cc2aea76094e7a4abc8e933540f3271df27d9c \
+ --hash=sha256:514435a37670e3e5e08f3945b68718b6ed329bb84367777e16f9f4dfe1e61a0f \
+ --hash=sha256:53aa02d20d10c3d814d536aa4e5ac9b84ca0ff5a88377963b085ad6822f93e64 \
+ --hash=sha256:5594fc43d548a7ed94949d139aa1341b270f1863f11cfd37f5a6c8b778a6b67f \
+ --hash=sha256:571b9fcb07b97ef3a492028fb3d2dc0993ca23a06138b0315286566d29ef718a \
+ --hash=sha256:57b3d78c95ba9059768b10e28b813002261d3f3dfc55cc48b0c988f625175827 \
+ --hash=sha256:5afb51d599ea772b8365ae807ae557f18bccfe46ab261fd1c2a9ed700fc6eb17 \
+ --hash=sha256:6b02afb9b97f65fbca5f31db6a2a3ba21aa93030225f150fa3f249717e938fb4 \
+ --hash=sha256:6c0016e7b354317c4e9e525b937ac8596c38d2d232b419529b9cd7a1cd46e39a \
+ --hash=sha256:71d6097b330eea8fd15097780c8e89cb1a8ce7838669f48c5bacd6f663dd4701 \
+ --hash=sha256:756c768d0c9c2955feb7a56c37ea24aea2e369f8d36a88da270b6a9f19e62b5e \
+ --hash=sha256:78cb2c6865a35ab8ff8b75fd122f6033b92a62c82801110e48ddd6c936a45d91 \
+ --hash=sha256:7a743ff716f746fc19a9557f60dab1600d4613255f8a7aeb3cdde4db7eb15a66 \
+ --hash=sha256:85f998ea1848bc6757289e739cfbdda3a04adfd58b02fc018ce54d754a5ce468 \
+ --hash=sha256:8728f216dcdb6e6d555cf971cb34076139ad74b31fc2c14da4fafc741c5f6217 \
+ --hash=sha256:877c3f311ff35410f690861c4409e7ccbf0cd2f878e50628a28e5a0bb689e658 \
+ --hash=sha256:8cd2f7bdda092d99c9fc2fb7391354f306d01443d22785d0cbfafa2e2c8bb418 \
+ --hash=sha256:8e95e1385e4998ae9694eeaa4730ba5457ff61185b3a55e2e7bea0880aef452a \
+ --hash=sha256:962864dc93511324d51ddbb5b9f8731bf71675b93ca612a07441896f4688fb8c \
+ --hash=sha256:9cf95fe4d0f84c82d282745d9bb08ad9f926efa00be4697e767b814ce40d4330 \
+ --hash=sha256:9e881fca225083806662a5c43d627d215f258ff43c890f831966c7d7ba9c7402 \
+ --hash=sha256:a2b55dd6b2a4c4b7d87ffa56bdb33fdc5fdb9a462173861a7bc097f17d91cb09 \
+ --hash=sha256:a45650e8ce7fafffd731db8550230db6b0d306d181a90b67d3e6bca2f1990930 \
+ --hash=sha256:a876864214e136f0eb367788dbd7df045f4806801518e2cfe9e13229cfe06d8f \
+ --hash=sha256:ae26d61dfa7a47befdc7572b521024e8745f3d809bd95ca9505a7bba9ef849ec \
+ --hash=sha256:af8d94b0db561cf68b88a267c5c44b49e134f525d0dc2cb7ed413a66bc23559a \
+ --hash=sha256:b343699e8308bdc51978310e1c959c584e7869cc8c40780058c87da7781a1e94 \
+ --hash=sha256:b3c777e849237620b022f7f297dd67705f9f5cf1685f09f02e46f93e92725468 \
+ --hash=sha256:b629de27fda84b42cde7edef0d85f13b958b47f6e9bbcbba9b673c562a89bd8b \
+ --hash=sha256:ba09209fbe443b4acccebe845d8a138b89a8f4fbaeedd44953490b5315d5e965 \
+ --hash=sha256:ba54cfebe86920a559a7c4d6b9050791c20513650a1952ebe3368c7dc70306f8 \
+ --hash=sha256:bcb46e2f9feff8d06323983bd83ed00c201fdcab3d74973e7072a889b3979fcd \
+ --hash=sha256:bcc33feacfaefce60c12fd500a277533bdc02b10a19f7f6d348763d8140bbba7 \
+ --hash=sha256:bf16ba1b4d0b6b7c8e534936632270cf70eb00dbe09005bc345b2677b726855c \
+ --hash=sha256:cf1845d02ad822a369a49f2bb9345b1614744267682e7a03527dc3bf6eea1777 \
+ --hash=sha256:d69141514cc30b774ceea5e3ed3a6635c8d8a96edf664689b890f4089111fb35 \
+ --hash=sha256:d9c7f76c0673154f044e9d78c8655fb4213f6ca31a836df48b40fe5d187717b9 \
+ --hash=sha256:dbce0b29841537a2fa4a214c2bbf14de3587c9680caa9b4e217568472490b28f \
+ --hash=sha256:dc624f6bc473dacdf7ef7eb8678d0d08edf15cd94fad6ae5c7d6cc67a4e4902f \
+ --hash=sha256:e158cb00350dc278f3b91551101aa7d12415a66ebf2c91d8d5ac14e56ddd3ad0 \
+ --hash=sha256:e491916b378fba47242221bb9ead245211b70d504f495d105d17b14a24b4907c \
+ --hash=sha256:e795b7eb908249c4e43c7c99fac7c2c75dab0c43566e37db472a355f63693d71 \
+ --hash=sha256:e7e480451b9fa137494bccd3a7d69adbe8ac65a87d97be61e11f1b1050a5bac3 \
+ --hash=sha256:e91206ee562682b51b98ef4b26a6ef48fd84e15fd4c4bc5ec768eb641d206838 \
+ --hash=sha256:e9871b1ffbfa9656b60aeee92ed5136a5742696006fa322b29ea3d8da0ecc9cf \
+ --hash=sha256:e9aeb04d6aef139de265b29683e119b638208f88cf73cdd1658aa07221165321 \
+ --hash=sha256:ebaea975e03d3141d9d3a507df75c9b3ec90fa9d2ffd07567b3a978d9d790b26 \
+ --hash=sha256:f0606c8bf2cdefea14a43530f7657cbbb7ecf1c4222512492ef4a4434a9501ec \
+ --hash=sha256:f13c32a3abd6079a66d9526e18dad9b6d280384d49d7c54040cd57b6424041d9 \
+ --hash=sha256:f7401aebd7f581d7f83a439d87d474999317ee099218e5ad25d125290990ba65 \
+ --hash=sha256:fa4ecea169a355be7a3ade2c783e2ed12f0e40d2c5621cda8b3297faf7fbb9f5 \
+ --hash=sha256:fbd139c8447d25dd750ab79ee274cc5e1fe80fc56340ab10b18a195e1b6eca3e \
+ --hash=sha256:fdafc9cce40277e0f7a0feabce0ee50dd2fa1800f3b38015e51296b5e814048d \
+ --hash=sha256:fe3cca2e4e8a592be0f269a1ca4835c25199d9f3ce815c8491048f785b0a0198 \
+ --hash=sha256:ffd0c5368496f41b0944be820fcb7a838aa6e623d250b01acf2643939c3f99d7
pluggy==1.6.0 \
--hash=sha256:7dcc130b76258d33b90f61b658791dede3486c3e6bfb003ee5c9bfb396dd22f3 \
--hash=sha256:e920276dd6813095e9377c0bc5566d94c932c33b27a3e3945d8389c374dd4746
-polars==1.39.3 \
- --hash=sha256:2e016c7f3e8d14fa777ef86fe0477cec6c67023a20ba4c94d6e8431eefe4a63c \
- --hash=sha256:c2b955ccc0a08a2bc9259785decf3d5c007b489b523bf2390cf21cec2bb82a56
-polars-runtime-32==1.39.3 \
- --hash=sha256:06b47f535eb1f97a9a1e5b0053ef50db3a4276e241178e37bbb1a38b1fa53b14 \
- --hash=sha256:363d49e3a3e638fc943e2b9887940300a7d06789930855a178a4727949259dc2 \
- --hash=sha256:425c0b220b573fa097b4042edff73114cc6d23432a21dfd2dc41adf329d7d2e9 \
- --hash=sha256:7c206bdcc7bc62ea038d6adea8e44b02f0e675e0191a54c810703b4895208ea4 \
- --hash=sha256:8bc9e13dc1d2e828331f2fe8ccbc9757554dc4933a8d3e85e906b988178f95ed \
- --hash=sha256:c728e4f469cafab501947585f36311b8fb222d3e934c6209e83791e0df20b29d \
- --hash=sha256:d66ca522517554a883446957539c40dc7b75eb0c2220357fb28bc8940d305339 \
- --hash=sha256:ef5884711e3c617d7dc93519a7d038e242f5741cfe5fe9afd32d58845d86c562 \
- --hash=sha256:f49f51461de63f13e5dd4eb080421c8f23f856945f3f8bd5b2b1f59da52c2860
+polars==1.44.2 \
+ --hash=sha256:1bb331f17a40d9d931101533dcd33637b66edc61eb377b07020dac16a0f0377b \
+ --hash=sha256:86c8e26b6c2de8c8d344bb910b74dfc47b118ac3fe0f19b44909467990a0b281
+polars-runtime-32==1.44.2 \
+ --hash=sha256:10c0c695a418407617b5159db7d9a21074a733e4c6d61275b6762f25cb31ca99 \
+ --hash=sha256:1fd536720668ba203a16a20b08cd6b23057e407a0279cf36b2f35f879d6e3208 \
+ --hash=sha256:8598e7a20efba70bb74978c7df7af7c606ff4d79b9b48fdd808250b189bc9a13 \
+ --hash=sha256:a1bafb441e99199a62c63bf1bbdc0ea09ee9776dbac2bf31452b5000fb1df2f7 \
+ --hash=sha256:b84842f7d621aaca7a52e165e19a24f89db45f8aa13744941430218419a14a67 \
+ --hash=sha256:bbf9b45040291dc1c6c588c837019c33557bde25ec536562a9cca9e1f6dfcc45 \
+ --hash=sha256:c4a09fb14aad711526346efc0cb2015c2fd0555ce4118b6524e5debbaea65ff5 \
+ --hash=sha256:d51040d3ab40157f6db3c62be59cab5b80fb3c8d158924769c4982a1c8eef730 \
+ --hash=sha256:e0fd43720c8222ae39919c8ff891636d53b352706087120e62f83544dd3ff782
protobuf==5.29.6 \
--hash=sha256:36ade6ff88212e91aef4e687a971a11d7d24d6948a66751abc1b3238648f5d05 \
--hash=sha256:62e8a3114992c7c647bce37dcc93647575fc52d50e48de30c6fcb28a6a291eb1 \
@@ -1095,154 +1177,142 @@ psutil==7.2.2 \
--hash=sha256:ed0cace939114f62738d808fdcecd4c869222507e266e574799e9c0faa17d486 \
--hash=sha256:eed63d3b4d62449571547b60578c5b2c4bcccc5387148db46e0c2313dad0ee00 \
--hash=sha256:fd04ef36b4a6d599bbdb225dd1d3f51e00105f6d48a28f006da7f9822f2606d8
-pydantic==2.13.2 \
- --hash=sha256:a525087f4c03d7e7456a3de89b64cd693d2229933bb1068b9af6befd5563694e \
- --hash=sha256:b418196607e61081c3226dcd4f0672f2a194828abb9109e9cfb84026564df2d1
-pydantic-core==2.46.2 \
- --hash=sha256:0551f2d2ddb68af5a00e26497f8025c538f73ef3cb698f8e5a487042cd2792a8 \
- --hash=sha256:0d12d786e30c04a9d307c5d7080bf720d9bac7f1668191d8e37633a9562749e2 \
- --hash=sha256:0d5e6d6343b0b5dcacb3503b5de90022968da8ed0ab9ab39d3eda71c20cbf84e \
- --hash=sha256:130a6c837d819ef33e8c2bf702ed2c3429237ea69807f1140943d6f4bdaf52fa \
- --hash=sha256:13ffef637dc8370c249e5b26bd18e9a80a4fca3d809618c44e18ec834a7ca7a8 \
- --hash=sha256:154dbfdfb11b8cbd8ff4d00d0b81e3d19f4cb4bedd5aa9f091060ba071474c6a \
- --hash=sha256:15e42885b283f87846ee79e161002c5c496ef747a73f6e47054f45a13d9035bc \
- --hash=sha256:160ef93541f4f84e3e5068e6c1f64d8fd6f57586e5853d609b467d3333f8146a \
- --hash=sha256:19631e7350b7a574fb6b6db222f4b17e8bd31803074b3307d07df62379d2b2e4 \
- --hash=sha256:1a9124b63f4f40a12a0666df57450b4c24b98407ff74349221b869ec085a5d8e \
- --hash=sha256:1b0ab6d756ca2704a938e6c31b53f290c2f9c10d3914235410302a149de1a83e \
- --hash=sha256:1b877d597afb82b4898e35354bba55de6f7f048421ae0edadbb9886ec137b532 \
- --hash=sha256:1d00b99590c5bd1fabbc5d28b170923e32c1b1071b1f1de1851a4d14d89eb192 \
- --hash=sha256:20fb194788a0a50993e87013e693494ba183a2af5b44e99cf060bbae10912b11 \
- --hash=sha256:233eebac0999b6b9ba76eb56f3ec8fce13164aa16b6d2225a36a79e0f95b5973 \
- --hash=sha256:236f22b4a206b5b61db955396b7cf9e2e1ff77f372efe9570128ccfcd6a525eb \
- --hash=sha256:251a57788823230ca8cbc99e6245d1a2ed6e180ec4864f251c94182c580c7f2e \
- --hash=sha256:2643ac7eae296200dbd48762a1c852cf2cad5f5e3eba34e652053cebf03becf8 \
- --hash=sha256:28708faed0b47f9d68906551a3471421ab0b15c31519e08fdb70ae6cad04d10b \
- --hash=sha256:2c2f6e32548ac8d559b47944effcf8ae4d81c161f6b6c885edc53bc08b8f192d \
- --hash=sha256:2ca790779aa1cba1329b8dc42ccebada441d9ac1d932de980183d544682c646d \
- --hash=sha256:2d1128da41c9cb474e0a4701f9c363ec645c9d1a02229904c76bf4e0a194fde2 \
- --hash=sha256:3098446ba8cf774f61cb8d4008c1dba14a30426a15169cd95ac3392a461193b1 \
- --hash=sha256:30cacc5fb696e64b8ef6fd31d9549d394dd7d52760db072eecb98e37e3af1677 \
- --hash=sha256:315d32d1a71494d6b4e1e14a9fa7a4329597b4c4340088ad7e1a9dafbeed92a9 \
- --hash=sha256:32fbc7447be8e3be99bf7869f7066308f16be55b61f9882c2cefc7931f5c7664 \
- --hash=sha256:33741359798f9dc3d4244a66031575d8a86c004f7853eb9961a49e4b6fab2d0b \
- --hash=sha256:36b1f99dc451f1a3981f236151465bcf995bbe712d0727c9f7b236fe228a8133 \
- --hash=sha256:37a68e6f2ac95578ce3c0564802404b27b24988649616e556c07e77111ed3f1d \
- --hash=sha256:37bb079f9ee3f1a519392b73fda2a96379b31f2013c6b467fe693e7f2987f596 \
- --hash=sha256:387cbe2b2bcace397da91f9b1165a9e75da254bb306b876a43b824cc10f49ce0 \
- --hash=sha256:3a075a29ebef752784a91532a1a85be6b234ccffec0a9d7978a92696387c3da6 \
- --hash=sha256:3b0a2dee92dfaabcfb93629188c3e9cf74fdfc0f22e7c369cb444a98814a1e50 \
- --hash=sha256:404da669e5e02bf7fb2cc56715a609f63af88aea531287494467109f97865fe3 \
- --hash=sha256:41d701bb34f81f0b11c724cc544b9a10b26a28f4d0d1197f2037c91225708706 \
- --hash=sha256:48649cf2d8c358d79586e9fb2f8235902fcaa2d969ec1c5301f2d1873b2f8321 \
- --hash=sha256:48b1059e4f2a6ec3e41983148eb1eec5ef9fa3a80bbc4ac0893ac76b115fe039 \
- --hash=sha256:48b36e3235140510dc7861f0cd58b714b1cdd3d48f75e10ce52e69866b746f10 \
- --hash=sha256:4e6df5c3301e65fb42bc5338bf9a1027a02b0a31dc7f54c33775229af474daf0 \
- --hash=sha256:4f27bc4801358dc070d6697b41237fce9923d8e69a1ce1e95606ac36c1552dc1 \
- --hash=sha256:4f59b45f3ef8650c0c736a57f59031d47ed9df4c0a64e83796849d7d14863a2d \
- --hash=sha256:547381cca999be88b4715a0ed7afa11f07fc7e53cb1883687b190d25a92c56cf \
- --hash=sha256:56291ec1a11c3499890c99a8fd9053b47e60fe837a77ec72c0671b1b8b3dce24 \
- --hash=sha256:57c584af6c375ea3f826d8131a94cb212b3d9926eaff67117e3711bbff3a83a5 \
- --hash=sha256:5a3c2bc1cc8164bedbc160b7bb1e8cc1e8b9c27f69ae4f9ae2b976cdae02b2dd \
- --hash=sha256:5a8e486d238850ddf2b25739317b6551d5bef9925ab004b18c552ff6e645f8a2 \
- --hash=sha256:5e2b4adb0fa46a842c492423e61063d6639cf9aea56380a02630ddcdd4894067 \
- --hash=sha256:631bec5f951a30a4b332b4a57d0cdd5a2c8187eb71301f966425f2e54a697855 \
- --hash=sha256:67db6814beaa5fefe91101ec7eb9efda613795767be96f7cf58b1ca8c9ca9972 \
- --hash=sha256:6b865eb702c3af71cf7331919a787563ce2413f7a54ef49ec6709a01b4f22ce6 \
- --hash=sha256:73a9d2809bd8d4a7cda4d336dc996a565eb4feaaa39932f9d85a65fa18382f28 \
- --hash=sha256:78cb0d2453b50bf2035f85fd0d9cfabdb98c47f9c53ddb7c23873cd83da9560b \
- --hash=sha256:7b1c9bdca33968c0dcd875f8185b3b6275df753fe000178684b0c1738959f3cd \
- --hash=sha256:7b42c6471288dedc979ac8400d9c9770f03967dd187db1f8d3405d4d182cc714 \
- --hash=sha256:7c5a5b3dbb9e8918e223be6580da5ffcf861c0505bbc196ebed7176ce05b7b4e \
- --hash=sha256:7ccfb105fcfe91a22bbb5563ad3dc124bc1aa75bfd2e53a780ab05f78cdf6108 \
- --hash=sha256:7dcb9d40930dfad7ab6b20bcc6ca9d2b030b0f347a0cd9909b54bd53ead521b1 \
- --hash=sha256:7f700a6d6f64112ae9193709b84303bbab84424ad4b47d0253301aabce9dfc70 \
- --hash=sha256:807eeda5551f6884d3b4421578be37be50ddb7a58832348e99617a6714a73748 \
- --hash=sha256:83aef30f106edcc21a6a4cc44b82d3169a1dbe255508db788e778f3c804d3583 \
- --hash=sha256:83ee76bf2c9910513dbc19e7d82367131fa7508dedd6186a462393071cc11059 \
- --hash=sha256:8641c8d535c2d95b45c2e19b646ecd23ebba35d461e0ae48a3498277006250ab \
- --hash=sha256:8a6572f3238851fde28b3194ef98cec9dbe66f1614caf4646239ea87f324121a \
- --hash=sha256:8cbd9d67357f3a925f2af1d44db3e8ef1ce1a293ea0add98081b072d4a12e3b4 \
- --hash=sha256:8f09a713d17bcd55da8ab02ebd9110c5246a49c44182af213b5212800af8bc83 \
- --hash=sha256:8f557ce9106850c79252792962d78b987e11fcdc10e5c2252443b9a485d3bfe5 \
- --hash=sha256:91155b110788b5501abc7ea954f1d08606219e4e28e3c73a94124307c06efb80 \
- --hash=sha256:9262d11d0cd11ee3303a95156939402bed6cedfe5ed0e331b95a283a4da6eb8b \
- --hash=sha256:99ebade8c9ada4df975372d8dd25883daa0e379a05f1cd0c99aa0c04368d01a6 \
- --hash=sha256:9a7c43a0584742dface3ca0daf6f719d46c1ac2f87cf080050f9ae052c75e1b2 \
- --hash=sha256:9cc0eee720dd2f14f3b7c349469402b99ad81a174ab49d3533974529e9d93992 \
- --hash=sha256:9f0e686960ffe9e65066395af856ac2d52c159043144433602c50c221d81c1ba \
- --hash=sha256:a070c7769fec277409ad0b3d55b2f0a3703a6f00cf5031fe93090f155bf56382 \
- --hash=sha256:a0891a9be0def16fb320af21a198ece052eed72bf44d73d8ff43f702bd26fd6b \
- --hash=sha256:ac204542736aa295fa25f713b7fad6fc50b46ab7764d16087575c85f085174f3 \
- --hash=sha256:ac8a65e798f2462552c00d2e013d532c94d646729dda98458beaf51f9ec7b120 \
- --hash=sha256:b089a81c58e6ea0485562bbbbbca4f65c0549521606d5ef27fba217aac9b665a \
- --hash=sha256:b308da17b92481e0587244631c5529e5d91d04cb2b08194825627b1eca28e21e \
- --hash=sha256:b317a2b97019c0b95ce99f4f901ae383f40132da6706cdf1731066a73394c25c \
- --hash=sha256:b478652b580cd4cf7f2dd40dc9fde594ed1c84e5df4bafefffb8387ddb74049f \
- --hash=sha256:b50f9c5f826ddca1246f055148df939f5f3f2d0d96db73de28e2233f22210d4c \
- --hash=sha256:b737c0b280f41143266445de2689c0e49c79307e51c44ce3a77fef2bedad4994 \
- --hash=sha256:b839d5c802e31348b949b6473f8190cddbf7d47475856d8ac995a373ee16ec59 \
- --hash=sha256:b902f0fc7c2cf503865a05718b68147c6cd5d0a3867af38c527be574a9fa6e9d \
- --hash=sha256:bc1e8ce33d5a337f2ba862e0719b8201cd54aaed967406c748e009191d47efdd \
- --hash=sha256:bd195af20e53aaac6cf5d7862e34dfdf86351720c858581ccb6563e02ae59421 \
- --hash=sha256:c05f53362568c75476b5c96659377a5dfd982cfbe5a5c07de5106d08a04efc4f \
- --hash=sha256:c1ce5b2366f85cfdbf7f0907755043707f86d09a5b1b1acebbb7bf1600d75c64 \
- --hash=sha256:c2012f64d2cd7cca50f49f22445aa5a88691ac2b4498ee0a9a977f8ca4f7289f \
- --hash=sha256:c2e25417cec5cd9bddb151e33cb08c50160f317479ecc02b22a95ec18f8fe004 \
- --hash=sha256:c326a2b4b85e959d9a1fc3a11f32f84611b6ec07c053e1828a860edf8d068208 \
- --hash=sha256:c3ad79ed32004d9de91cacd4b5faaff44d56051392fe1d5526feda596f01af25 \
- --hash=sha256:c6b1064f3f9cf9072e1d59dd2936f9f3b668bec1c37039708c9222db703c0d5b \
- --hash=sha256:caeed15dcb1233a5a94bc6ff37ef5393cf5b33a45e4bdfb2d6042f3d24e1cb27 \
- --hash=sha256:d07d6c63106d3a9c9a333e2636f9c82c703b1a9e3b079299e58747964e4fdb72 \
- --hash=sha256:d157c48d28eebe5d46906de06a6a2f2c9e00b67d3e42de1f1b9c2d42b810f77c \
- --hash=sha256:d26e9eea3715008a09a74585fe9becd0c67fbb145dc4df9756d597d7230a652c \
- --hash=sha256:d333a50bdd814a917d8d6a7ee35ba2395d53ddaa882613bc24e54a9d8b129095 \
- --hash=sha256:d61db38eb4ee5192f0c261b7f2d38e420b554df8912245e3546aee5c45e2fd78 \
- --hash=sha256:d8060f42db3cd204871db0afd51fef54a13fa544c4dd48cdcae2e174ef40c8ba \
- --hash=sha256:d9ffa75a7ef4b97d6e5e205fabd4304ef01fec09e6f1bdde04b9ad1b07d20289 \
- --hash=sha256:dc4620a47c6fe6a39f89392c00833a82fc050ce90169798f78a25a8d4df03b6e \
- --hash=sha256:dd51dd16182b4bfdcefd27b39b856aa4a57b77f15b231a2d10c45391b0a02028 \
- --hash=sha256:de12004a7da7f1eb67ece37439a5a23a915636085dd042176fda362e006e6940 \
- --hash=sha256:de87422197cf7f83db91d89c86a21660d749b3cd76cd8a45d115b8e675670f02 \
- --hash=sha256:df73724fce8ad53c670358c905b37930bd7b9d92e57db640a65c53b2706eee00 \
- --hash=sha256:dfff584138be087457cc474791d082fdfe32b0d427613d5494a679fe9f4eaef5 \
- --hash=sha256:e094a8f85db41aa7f6a45c5dac2950afc9862e66832934231962252b5d284eed \
- --hash=sha256:e4e2c72a529fa03ff228be1d2b76944013f428220b764e03cc50ada67e17a42c \
- --hash=sha256:e698fe2d8f75c4e9368ee3f4e0d3322d1180be2ec4592d3f73b2572765b1c705 \
- --hash=sha256:e69aa5e10b7e8b1bb4a6888650fd12fcbf11d396ca11d4a44de1450875702830 \
- --hash=sha256:e80011f808b03d1d87a8f1e76ae3da19a18eb706c823e17981dcf1fae43744fc \
- --hash=sha256:e9fcabd1857492b5bf16f90258babde50f618f55d046b1309972da2396321ff9 \
- --hash=sha256:ea1ad8c89da31512fe2d249cf0638fb666925bda341901541bc5f3311c6fcc9e \
- --hash=sha256:f0c1cbb7d6112932cc188c6be007a5e2867005a069e47f42fe67bf5f122b0908 \
- --hash=sha256:f1a6197eadff5bd0bb932f12bb038d403cb75db5b0b391e70e816a647745ddaf \
- --hash=sha256:fa8ab79cea8a1bfe52a21a9b37859c15478d009f242f47737201ecea885b9dd9 \
- --hash=sha256:fb3ec2c7f54c07b30d89983ce78dc32c37dd06a972448b8716d609493802d628 \
- --hash=sha256:fcaa1c3c846a7f6686b38fe493d1b2e8007380e293bfef6a9354563c026cbf36 \
- --hash=sha256:fd05e1edb6a90ad446fa268ab09e59202766b837597b714b2492db11ee87fab9
+pydantic==2.13.5 \
+ --hash=sha256:346a034f080da3755d8e9cb5e00e8b07de1d39e4f6e2c87d8ab7cafa0b269a73 \
+ --hash=sha256:51a9c5f7b2f8e636f04c6cada605d9b6a3bf1348fdf945a3d8869b19bba0ee08
+pydantic-core==2.46.5 \
+ --hash=sha256:013d6f3483d81e02e7c328831808f336c8596ee33b4bd4026b9ffb1e960b8942 \
+ --hash=sha256:03b9666e41e35d8909852ba191a0607520f81b74eaf12ccf8737005dbb313821 \
+ --hash=sha256:045ab3b6d308439e32b81cc173bba5b9018bc6ed896afd0c65b3b009b1699af5 \
+ --hash=sha256:0bddb4020d8f04175865ccd17eff3040874fc11fb593f424edb452653b4b947c \
+ --hash=sha256:0cdbada856a1c69a7624a64d3d9aefe79300bd6ef827b43a4f265010b9b55184 \
+ --hash=sha256:0fc5be0abd4a407e200d844b404e33639a554e7bd0d448e7b9ae181be4789ac2 \
+ --hash=sha256:10416c15b8839ecc4ef4d0885da76da6fd0f67333a0eb8aff6d93c4b8f2910fc \
+ --hash=sha256:15f4a94963c95accac15b7b657bb177d3ad82bb90b0d0526d9a9b85079925db5 \
+ --hash=sha256:18a09e1e1011b462f2e32774f25859ef1223d5c2b0546a633cf56654710721e0 \
+ --hash=sha256:193375f3548919d3f0b60936ca113ada3e38f264f91b9b8e0508efaad57be931 \
+ --hash=sha256:1a353f84de772f423b5ffb11d7ae352fbbef0f446f3c0b0af0f8236d7233606e \
+ --hash=sha256:1e449def1945a462c464331254e5a44fca7c3b4f9aedf59ec2f50f8066dd8e25 \
+ --hash=sha256:1e5aad1220a1192c42341c8fd4a8686657e73ab2a920c970bdc4de334fe3193d \
+ --hash=sha256:200aa3dc9f8d54f0754f43247c0bad0999fdcfbfd2488384dd44f37279271fe6 \
+ --hash=sha256:2471fd51c61c610e1dcf7de44d7299283661654d11264ab4802b303368d69c47 \
+ --hash=sha256:24922243639cbdac66c75fcb6fd6495a9cb52b213d62f9a0d16f0310b1ff8038 \
+ --hash=sha256:28a6a556cd3b6066bea827857f9d9cce027c96f776e512f544a581f9e42161f8 \
+ --hash=sha256:2bc9419666990c06d7397831f2126a1ecc3594aaa3ff7de5bf2d066802f4e07b \
+ --hash=sha256:2cbd9a5eff05e51c447c34dfa4632145b26b09120cf04bd0c871e44c1a5e1c9a \
+ --hash=sha256:2d330aaba8621b1edcec8ae2c4050f63b84ccf6d98723a8f212e9684713abf0e \
+ --hash=sha256:2d5d76654becf5efd62c9e51c3756c67b49498b0c9a40884934c40807adbd074 \
+ --hash=sha256:337639ba62a11acde6ef3aeb08c8ea755f8ef1fe5e513356c0f36a2b0d7568b0 \
+ --hash=sha256:347ec774390c87326a2e4929d58d3f7e8763a104d5d35f4cd595a4c952366433 \
+ --hash=sha256:356c8368cbc321050b169595683a2e1d63413b1e0e2868b330af9fc14c616d3f \
+ --hash=sha256:37ae34309d7bd8c0d61ab839668058f2a7962ea1fc51d105d2db228fe0618034 \
+ --hash=sha256:37ea7b83c935e5b0d68c9449b82651accf78a10828b2c02b2f2d9e9496446c21 \
+ --hash=sha256:3a3e26b6a8274211bddee2d0e4d0d42778f17a34510f49d2ec44b58abfc41736 \
+ --hash=sha256:3aa166e99c4f2985407fb8714aebede877ecb5455cf321b606adca926d30d5a0 \
+ --hash=sha256:3d2652072b2d774947ba5cf78a9e59644ac62ee572daf6dd2e1dfe905e15b2b7 \
+ --hash=sha256:40375c2d05acec10323e45dfe2077ac44bc74659008614af5069034e2cfc781c \
+ --hash=sha256:413a717a410d0c817ef5b786a059415550b3794e1d0c2abffd9efb93a3d9f7b4 \
+ --hash=sha256:46c25dda9d092a06c08db76ffe0a197107904d0dfac653f7d5306bbcd6d6119c \
+ --hash=sha256:49776eab08766a08dfff7012f8b422dcd7e25e43b316eedf0477c24fcfa84b7c \
+ --hash=sha256:4d44cf99ddebf875f9b68cc267aa684c99b7b44fe63ee1cac4ec163807290069 \
+ --hash=sha256:4dedce55295becb61921e386b99d4f2706045306e7fa52249a33004c837379fb \
+ --hash=sha256:4f8507560a9284e1370bb048ed4282012fbef4e8d109875b95e884d228552061 \
+ --hash=sha256:4fdc8b93a41521988916eeaa271173fcca7fa0803d62f87675aac8dcec1c8e29 \
+ --hash=sha256:5086029a57366b8cf81b130a43908738095c270c21a8d7f0e8bdfdb89718e2f3 \
+ --hash=sha256:52e24eacdb536cade636aa90fb851835222becff8484b7001fdc78cb0290f2aa \
+ --hash=sha256:53feb344243bb9510a9dec7bf3cf1b64d88a98af5dc7872a5160465f8b198c8e \
+ --hash=sha256:545f26c504b27c3758439a5e6d9349931f0a04f855668d5fe323c89e82300a38 \
+ --hash=sha256:54d510bac3ee52247af28ed4bb18a1e799f040ac60fd2bf5ccd4c92f1fbe786f \
+ --hash=sha256:5cb482e9e84c851f4e623fe4acc1ced89168cf1fe18f7089db4548c8f5bbb65b \
+ --hash=sha256:5e81740c09e310f5aa5cbd3e434a01c154d4bef93241c7877b39f211d2b78ba8 \
+ --hash=sha256:5ee239d575f80b08eca11f6e20f90c4c695de7825c67eefe6091fbf20dda648e \
+ --hash=sha256:5f194189415698233dd1114a093a9b56e61e2c57e11b469be3b0506f46f0771c \
+ --hash=sha256:5f93c5fe914d75fbec9a49209b00da5f08e9e467d69da2b1510c81940cfd10be \
+ --hash=sha256:657b40d6240c0a7b6a64b30f22d1e3aa631c7e846c621b0c0f6d1d75e2e15ea6 \
+ --hash=sha256:6d30e1a4f138b8951063e9a394752a9179b51da288ffa507b1e659222f4c1793 \
+ --hash=sha256:6f7b393a8b3da82f5c1fc0751e6d01ac6c55b93c18226a60bdfba4a724efafd1 \
+ --hash=sha256:701b2e04b560eeb4bddf7a25ab8ca476176e34fdbd9a0e18196f0d12d4685f0b \
+ --hash=sha256:771cf63ae0b1b50dd22e5f3e3549fab5f3f4ff1635d352a9e1a97fe01c7b2e64 \
+ --hash=sha256:79bdfa52f843137045b2d081cc05c120ba6665d29b7559c2c47690906f39279f \
+ --hash=sha256:7ac031912d54f3d83ef3b3eb98dfabc1608802e2202263d25957eeed40b94761 \
+ --hash=sha256:7b0fc826b16c55e561e5d2a0c5c77b051ba1d92808118c4e4b5390f5e0cf191d \
+ --hash=sha256:7c6be839a5a8312626b32029a415644a0846b420bc8b52b95b28cd92da162168 \
+ --hash=sha256:816ff0a6550ffc06c098ccd2e0698600f9aa7da192a79eaa6f9af504a35db869 \
+ --hash=sha256:82a36973cf8a2ef5406f4fe2edbf8ed0c99629535d959e0b100c76a32535a111 \
+ --hash=sha256:837b396ca3d7b74091ca623f6cbd8351bd42d670a79c2683e79fb089f06a2de5 \
+ --hash=sha256:850a08d167dde16db8702c274f320c7be9d7da6f6dff2b58b18f9e815bd94f5b \
+ --hash=sha256:8816f3d218beb4b787de5c9759c259b8fa61f9dec42dc7811f320a33771778b7 \
+ --hash=sha256:892a881d5f68c2b9ea304b7a6c2c60d9343df578a311b0f86b94bc8f1ffe8129 \
+ --hash=sha256:895395f8918627b04efb1ad2a4cf605387143300ba03304cd1dfa6d03f5e095e \
+ --hash=sha256:8b10e3e8fd7ddc2bd915848a2768e44c15b22936f1cc54c462ad1164deb02655 \
+ --hash=sha256:8e24d8f05fa2d28513d94e877e9c75ad66175376209b3977f916e240e623193c \
+ --hash=sha256:8feeac04b5794e513e710af2f9c87d49f31a6dc47967bb264a1fed61a8989bec \
+ --hash=sha256:9432f3598db432cb51c5b37fdbf29a60fcccc79e30d37a05022776a6bc4ab689 \
+ --hash=sha256:976e1128455aa595ea04c79ccfedff1aaeab96ee013fcc916bed120c4f0ad94f \
+ --hash=sha256:978e7b97d4824b5be09c69fb70507cbde3b0323fc147332ca40a94d9a6a0ebbf \
+ --hash=sha256:97bf8de4d541598c94a59344eeb988a94c08ff76b5723c41f6567ec18c7892ea \
+ --hash=sha256:97cf3eb53a8cccacf9d46686a0926186c9bfb5574f2ed66d3639d5fe117cd3a9 \
+ --hash=sha256:9b68938dd5b0c783d88ff8e2dcc69451b5eb936fe212d516b21b9d5567f6d464 \
+ --hash=sha256:9c4b71f10dd532fb7a5cbc8f58707779e64f03a258c2bf8bfbaecfcd9970b519 \
+ --hash=sha256:9f47b8a949e60f027f0aa0a6f6c7b7e9c55cbf4380d10b344e282fa4e7ab1e1b \
+ --hash=sha256:a1dee1b804ff4d11c663636cf15d2ea47e9f79cd56c033fb1cbf08924842a48f \
+ --hash=sha256:a2468d93d181667a7abd66e1b64bb9f76f361b0fef8faddf687456453576f5ee \
+ --hash=sha256:a2a5e1d0ff29adddc9f6d6821a66302e4493f8ca898b715b6b1182c2c201ea0a \
+ --hash=sha256:a39ac25a9a2fa4072efdb429833c4a4c8009a51ff9eea3eeae131713cd27991e \
+ --hash=sha256:a445486499897b88a7d6c310c88ed64dd37b1b59bfd7ae9107490bbb362f47d6 \
+ --hash=sha256:a91c17edf6eea2402cb5457b4c89e99bc5ed1004aa34c4adf1d4258c1a5c22c2 \
+ --hash=sha256:ab4b66edffb32d9e951efb3814bd104b8367a7501b81b955cacb5726d897389f \
+ --hash=sha256:aca6c767f552b21b10f774aeac128e828eafb796adfa1b666a18bf6321453c3a \
+ --hash=sha256:acf8a67ba51f4ca9ddbd0e6b3000a65ac51ab734661778b3e7ba64d99a710f2f \
+ --hash=sha256:b10ec717381bdbfafef34607824db4c91de69ff085e4fca3b2af91b4fa17e68a \
+ --hash=sha256:b49924c73a235e969511bf2aabdff3beebf9820931f646c80274d5d780010c47 \
+ --hash=sha256:b6acfb46a814762367fb7ba0828b0a17d441b92ce249a0e007474c9072662dda \
+ --hash=sha256:b7ca9034437b6022f941f4857459562ee00a560b97e7cce8a0ec5a74fc6766e0 \
+ --hash=sha256:b98134087d9de723658d17a42c7d0da8d6e2ef08015dee7dc93889047315f5e4 \
+ --hash=sha256:b9fe6fb92520e3fd61f2e49000b6911b188824f089b75973ea06d6267f0b476d \
+ --hash=sha256:bce57638e08ac148e5778cce7feb968307a727d66f8e2274a543d0cf0c9ad6a3 \
+ --hash=sha256:c14ad3bdc85ee7f318742c457ca3968a92126d144b15721c759033bfb06296c2 \
+ --hash=sha256:c1c43ad4339643d70ebb8124e1305a7dab423001eff58bb41a0f731adbc98355 \
+ --hash=sha256:c3471e5c4a949c26ec00a77f01df59096aa9495877de76fd60a980f8ee6be461 \
+ --hash=sha256:c583b927a8838dab890706a6fa7573fbb8b70e24000ef9f7238e2d6f6435a5ed \
+ --hash=sha256:c76fe65e607be28c7fd4d56fc3c42b1583aa058ce3408b7ad0fd540171d31f9f \
+ --hash=sha256:c7ea57fc63aa7da93a1bd2d644e6577befae10c52c4e36377635eea1056a74f5 \
+ --hash=sha256:cd5214352ae68f3b5e9af7768bdc5253695ee069675db3480518420b3be881f2 \
+ --hash=sha256:cdbb78909f52b981d3b2d56b97328d71eb0b974c36bd77c920123a7ebb192829 \
+ --hash=sha256:cdc8b74ecc48c0cb1e9607a05ec4e9e88db60a19ffcc9a1d5f9088ede40c8dc0 \
+ --hash=sha256:d0a24b40877af2de4950252be9d21eaf7fb07660f3c2cae1f56c6b599ada5266 \
+ --hash=sha256:d22a945598fb91236b4dd793a6e42e4f3dd7740bb5aace5ebd7d4c08d13bb575 \
+ --hash=sha256:d2f9fc07a8042a8f95925b35c4f04f469707c981fc33245b6ca187cf5d2dd290 \
+ --hash=sha256:d625a186a65201c23a9e3b8ed9c47e90a026e03256608cc91851c6709096844f \
+ --hash=sha256:d925f3d9afd05a8c0fb3a1031463a8d59ebe5e2afad297e29c78be19e13b4e62 \
+ --hash=sha256:e64e88d5585bea9ce95861079de72006c7fa6d3df4e3a3b65ba31eb979c15c9f \
+ --hash=sha256:e652ab17569c94bff5475520f907b7148b8c24036a8ebbe5cf7cf7493d28579a \
+ --hash=sha256:e7b891faeedeafba41b2983e5001a81b6a915b69544c7e7570d1989ce1c36ac7 \
+ --hash=sha256:e80675d75ae2cd14372cb65cad5400d9347a3d3f6c13000183f22dfd027283ed \
+ --hash=sha256:e9c134bb666dd54b778b9fc0d2b50cbb7f979b9e3716f26a88c9ab3b6fc1dd0f \
+ --hash=sha256:eb7d8d0e5886a89a55d2eef490e272fa965a9d57c6b29a5b5088a7997ec2cad1 \
+ --hash=sha256:ecb42011e12ee19cafbc312887cbf3546959fe02fbad44f272d4be5baa997615 \
+ --hash=sha256:ef3fbbf161dc9351a2fe0422e51b129f9e97e42385bd0320b309c15f7d287dd8 \
+ --hash=sha256:efd62a42486f1bda5d24cb4f63d15a3c7768375fe83d36f9417b4ad7a2fb20b3 \
+ --hash=sha256:f077d0b97ab11fa7dcc633fca53515f290bca8a8a633e966d5b6d1879d9ed01a \
+ --hash=sha256:f332f0e72a5a0400141f830744e141bf9f97917878dbe968669e8a7fefea78ff \
+ --hash=sha256:f7b0ec93a2893de856652154d73b7ba622f26fa97726487dcac373de5f4c6084 \
+ --hash=sha256:fa10ef4112775900e7a0661068635eb67b2ab824fbde764de6e0e21982a93db0 \
+ --hash=sha256:fc5d783bd4a2387e97b8a2d5ec781cfb92b3d893bf82370548e99db5915935d3 \
+ --hash=sha256:fc8515076c11f3cfdf4fb142dcca0fe384b1230a3b5415458ac84f3e0903ec13 \
+ --hash=sha256:ff218293c9c806138dca139765e3b067621be52bcd93cdc14c7711be7ddc90a9
pydot==3.0.4 \
--hash=sha256:3ce88b2558f3808b0376f22bfa6c263909e1c3981e2a7b629b65b451eee4a25d \
--hash=sha256:bfa9c3fc0c44ba1d132adce131802d7df00429d1a79cc0346b0a5cd374dbe9c6
-pygments==2.20.0 \
- --hash=sha256:6757cd03768053ff99f3039c1a36d6c0aa0b263438fcab17520b30a303a82b5f \
- --hash=sha256:81a9e26dd42fd28a23a2d169d86d7ac03b46e2f8b59ed4698fb4785f946d0176
-pynvvideocodec==2.1.0 \
- --hash=sha256:09cdd222c761855cde9b05ddc13d21893473cebae50077eca20b4eac0fed8acb \
- --hash=sha256:133b47336cd9b3a161707b02d2ceb9e5af387cb9833335d60c16eddf7322b3ea \
- --hash=sha256:34ddbb22230f4963f5b25b41af0a1579c3731fb95c79993502dd014a4dec7fd6 \
- --hash=sha256:3e1ecef741e1943978911998aa1a9d683be8b005957646db6c9b0d53f943f9af \
- --hash=sha256:6d653240ce835e9ab531a4d61350a4af54f7f7859e5392e1323c030ffcfc8ff0 \
- --hash=sha256:7c35887e294613825c9011dc1b6962d0c0374049b1ce8939cee97db5d82b28e8 \
- --hash=sha256:af7e44d13cfb524626c72d947f12a3b192cfa1deb4b598a7778bd4aa1e6fd4ac \
- --hash=sha256:b8d9b47b5c910953d6c24fa60fd7bf6ac57f3d4a20831fcaf35e039349a91d42 \
- --hash=sha256:c317fbad725b1b81936b863edcd98920bc4be423b9965719657030359d7afe29 \
- --hash=sha256:e778af3319a759c1728065e9e4682f591c44dedbbc90c1ca71d025dce65d6a58 \
- --hash=sha256:ee5d2dc56ac5ca8d223fa6fe8f57c67faf5ee1c9360216d9194cfbcd3f9d3948
+pygments==2.21.0 \
+ --hash=sha256:2363c69b61c4a97c838da3b130dcd6468f4848992b21a82f2a63ec34377137d9 \
+ --hash=sha256:610ca751c9bc2492b38eb9a38a7fbc93edbbb2d7182edaf34e66ae493dee5c8c
pyparsing==3.3.2 \
--hash=sha256:850ba148bd908d7e2411587e247a1e4f0327839c40e2e5e6d05a007ecc69911d \
--hash=sha256:c777f4d763f140633dcb6d8a3eda953bf7a214dc4eff598413c070bcdc117cbc
-pytest==9.0.3 \
- --hash=sha256:2c5efc453d45394fdd706ade797c0a81091eccd1d6e4bccfcd476e2b8e0ab5d9 \
- --hash=sha256:b86ada508af81d19edeb213c681b1d48246c1a91d304c6c81a427674c17eb91c
+pytest==9.1.1 \
+ --hash=sha256:1088fbde8f2b49d95a549a195707afa7a76a3ce9bcadc26b6d71f0ffda5fe313 \
+ --hash=sha256:37a86b45efb9a47a61a36449063e8e18d0cab3161329fc099eb21783169c4f0c
python-dateutil==2.9.0.post0 \
--hash=sha256:37dd54208da7e1cd875388217d5e00ebd4179249f90fb72437e91a35459a0ad3 \
--hash=sha256:a8b2bc7bffae282281c8140a97d3aa9c14da0b136dfe83f850eea9a5f7470427
@@ -1320,36 +1390,30 @@ pyyaml==6.0.3 \
--hash=sha256:f7057c9a337546edc7973c0d3ba84ddcdf0daa14533c2065749c9075001090e6 \
--hash=sha256:fa160448684b4e94d80416c0fa4aac48967a969efe22931448d853ada8baf926 \
--hash=sha256:fc09d0aa354569bc501d4e787133afc08552722d3ab34836a80547331bb5d4a0
-requests==2.33.1 \
- --hash=sha256:18817f8c57c6263968bc123d237e3b8b08ac046f5456bd1e307ee8f4250d3517 \
- --hash=sha256:4e6d1ef462f3626a1f0a0a9c42dd93c63bad33f9f1c1937509b8c5c8718ab56a
+requests==2.34.2 \
+ --hash=sha256:2a0d60c172f83ac6ab31e4554906c0f3b3588d37b5cb939b1c061f4907e278e0 \
+ --hash=sha256:f288924cae4e29463698d6d60bc6a4da69c89185ad1e0bcc4104f584e960b9ed
rich==15.0.0 \
--hash=sha256:33bd4ef74232fb73fe9279a257718407f169c09b78a87ad3d296f548e27de0bb \
--hash=sha256:edd07a4824c6b40189fb7ac9bc4c52536e9780fbbfbddf6f1e2502c31b068c36
-safetensors==0.7.0 \
- --hash=sha256:0071bffba4150c2f46cae1432d31995d77acfd9f8db598b5d1a2ce67e8440ad2 \
- --hash=sha256:07663963b67e8bd9f0b8ad15bb9163606cd27cc5a1b96235a50d8369803b96b0 \
- --hash=sha256:12f49080303fa6bb424b362149a12949dfbbf1e06811a88f2307276b0c131afd \
- --hash=sha256:1d060c70284127fa805085d8f10fbd0962792aed71879d00864acda69dbab981 \
- --hash=sha256:42cb091236206bb2016d245c377ed383aa7f78691748f3bb6ee1bfa51ae2ce6a \
- --hash=sha256:473b32699f4200e69801bf5abf93f1a4ecd432a70984df164fc22ccf39c4a6f3 \
- --hash=sha256:54bef08bf00a2bff599982f6b08e8770e09cc012d7bba00783fc7ea38f1fb37d \
- --hash=sha256:5d72abdb8a4d56d4020713724ba81dac065fedb7f3667151c4a637f1d3fb26c0 \
- --hash=sha256:672132907fcad9f2aedcb705b2d7b3b93354a2aec1b2f706c4db852abe338f85 \
- --hash=sha256:6999421eb8ba9df4450a16d9184fcb7bef26240b9f98e95401f17af6c2210b71 \
- --hash=sha256:7b95a3fa7b3abb9b5b0e07668e808364d0d40f6bbbf9ae0faa8b5b210c97b140 \
- --hash=sha256:8469155f4cb518bafb4acf4865e8bb9d6804110d2d9bdcaa78564b9fd841e104 \
- --hash=sha256:94fd4858284736bb67a897a41608b5b0c2496c9bdb3bf2af1fa3409127f20d57 \
- --hash=sha256:b0f6d66c1c538d5a94a73aa9ddca8ccc4227e6c9ff555322ea40bdd142391dd4 \
- --hash=sha256:c74af94bf3ac15ac4d0f2a7c7b4663a15f8c2ab15ed0fc7531ca61d0835eccba \
- --hash=sha256:c82f4d474cf725255d9e6acf17252991c3c8aac038d6ef363a4bf8be2f6db517 \
- --hash=sha256:cdab83a366799fa730f90a4ebb563e494f28e9e92c4819e556152ad55e43591b \
- --hash=sha256:cfdead2f57330d76aa7234051dadfa7d4eedc0e5a27fd08e6f96714a92b00f09 \
- --hash=sha256:d1239932053f56f3456f32eb9625590cc7582e905021f94636202a864d470755 \
- --hash=sha256:dac7252938f0696ddea46f5e855dd3138444e82236e3be475f54929f0c510d48 \
- --hash=sha256:dc92bc2db7b45bda4510e4f51c59b00fe80b2d6be88928346e4294ce1c2abe7c \
- --hash=sha256:e07d91d0c92a31200f25351f4acb2bc6aff7f48094e13ebb1d0fb995b54b6542 \
- --hash=sha256:f4729811a6640d019a4b7ba8638ee2fd21fa5ca8c7e7bdf0fed62068fcaac737
+safetensors==0.8.0 \
+ --hash=sha256:040070828e36dc8e122178bbbd5830ff9e97920affb84cbe0f46442497bed358 \
+ --hash=sha256:096ec1a98435df7beb08853bb5aa9081a84f23d0adc67ed1a0a10550f608373f \
+ --hash=sha256:2ddf52eac562eda224f99acfa7889d02968c1fd59a5b011ae7d8137c37e9c02d \
+ --hash=sha256:3ae091f16662658bdc019a4ff6cb4c085bb7d725eb5978b183ffd265863b6d2d \
+ --hash=sha256:4124502b78f03534117c848f87a39b8f31e577b15eff423bf8bfb95f2a8c30d0 \
+ --hash=sha256:4a95ae2b05d7726d751da4ebf626a2ca782b706e101bd894c95bc2450b1cffcc \
+ --hash=sha256:7a46e5ff292c356d6991e60942ba7f79817682d3a2cef0702136448cb9c4d235 \
+ --hash=sha256:7bc0a787ba8a35be368ee3574edfa2b1ad389eebd0a72e482ae275490e3f6c98 \
+ --hash=sha256:87eec7ffed2b809f05a398a8becb7d013f19f7837cd15d9748580d6cf30dbaf4 \
+ --hash=sha256:8e080062fcde23be189565e1c3305d16751a218ecf9412c8601e64204eb6f846 \
+ --hash=sha256:8e9f537aa183a38ace122d27303dcd986b26bd2a7591f9181d7f0c396f4677ca \
+ --hash=sha256:c554f85858e05226d3c2828e32395e677434685d6d94594a41643361c5e837f0 \
+ --hash=sha256:c80201d22cbf405b80647a60ada77bba06c8fba2da2743ba1e89cdcc39a81f25 \
+ --hash=sha256:f7838e5135a406ad3e02efdcb8cf2e5397d368b0154537c4fec682dbc544d452 \
+ --hash=sha256:fabaf3e0f18a6618d9b36560682562157f77c2b71fcffc7b432be2baed9d753d \
+ --hash=sha256:fcdd41ec4628fee5799f807c73c353629130fbd942aa23d83c623dd6c9d52d78 \
+ --hash=sha256:fd6f3f93c9a0a7cc2788ee63fb763353d4bd2e89b0751bc78fcf7dda00bea774
scikit-learn==1.7.2 \
--hash=sha256:0486c8f827c2e7b64837c731c8feff72c0bd2b998067a8a9cbc10643c31f0fe1 \
--hash=sha256:0b7dacaa05e5d76759fb071558a8b5130f4845166d88654a0f9bdf3eb57851b7 \
@@ -1432,9 +1496,9 @@ scipy==1.15.3 \
six==1.17.0 \
--hash=sha256:4721f391ed90541fddacab5acf947aa0d3dc7d27b2e1e8eda2be8970586c3274 \
--hash=sha256:ff70335d468e7eb6ec65b95b99d3a2836546063f63acc5171de367e834932a81
-starlette==1.3.0 \
- --hash=sha256:bb58cbb7a699da4ee4be9ed4cdfe4bc5b0390aa6dac1d1ac714ebebe8dc3c8df \
- --hash=sha256:ff4ca1bc23de6a45cdfbbeb9b3caaea524c9221cdd8a6684ad7a4f651a83890b
+starlette==1.6.0 \
+ --hash=sha256:a86dd39d14bb45f85a3d18525215a9ef0cfd1f192ac793220e72598c90335f0c \
+ --hash=sha256:d4e3ac5e546444960c710297a3c9fc3f7ebae1b7e963f3d36173b49da535be9b
sympy==1.14.0 \
--hash=sha256:d3d3fe8df1e5a0b42f0e7bdf50541697dbe7d23746e894990c030e2b05e72517 \
--hash=sha256:e091cc3e99d2141a0ba2847328f5479b05d94a6635cb96148ccb3f34671bd8f5
@@ -1593,32 +1657,32 @@ triton==3.6.0 \
--hash=sha256:d002e07d7180fd65e622134fbd980c9a3d4211fb85224b56a0a0efbd422ab72f \
--hash=sha256:e8e323d608e3a9bfcc2d9efcc90ceefb764a82b99dea12a86d643c72539ad5d3 \
--hash=sha256:ef5523241e7d1abca00f1d240949eebdd7c673b005edbbce0aca95b8191f1d43
-typing-extensions==4.15.0 \
- --hash=sha256:0cea48d173cc12fa28ecabc3b837ea3cf6f38c6d1136f85cbaaf598984861466 \
- --hash=sha256:f0fa19c6845758ab08074a0cfa8b7aecb71c999ca73d62883bc25cc018c4e548
-typing-inspection==0.4.2 \
- --hash=sha256:4ed1cacbdc298c220f1bd249ed5287caa16f34d44ef4e9c3d0cbad5b521545e7 \
- --hash=sha256:ba561c48a67c5958007083d386c3295464928b01faa735ab8547c5692e87f464
-ultralytics==8.4.38 \
- --hash=sha256:88e4c26205a8472773725db60346a46bbd5ac35e9dab19a7f6f954fa202aedcd \
- --hash=sha256:f4d0c1efb04b75e3fab1481e4d6eecc352ad3736223e4980671f424daa95a99a
-ultralytics-thop==2.0.18 \
- --hash=sha256:21103bcd39cc9928477dc3d9374561749b66a1781b35f46256c8d8c4ac01d9cf \
- --hash=sha256:2bb44851ad224b116c3995b02dd5e474a5ccf00acf237fe0edb9e1506ede04ec
+typing-extensions==4.16.0 \
+ --hash=sha256:481caa481374e813c1b176ada14e97f1f67a4539ce9cfeb3f350d78d6370c2e8 \
+ --hash=sha256:dc983d19a509c94dba722ee6abd33940f7c05a89e243c47e907eb4db6f1a43e5
+typing-inspection==0.4.4 \
+ --hash=sha256:547274fa6b0a561ccf549cc9524b999a578e737d015d8709d021f9d0d13bea47 \
+ --hash=sha256:65b8397ba37ccbce054456aaccddfc91e6e3083c92824df348d96ca832f3f147
+ultralytics==8.4.146 \
+ --hash=sha256:5c3618f4c5f7b89ad6afdf789bf4d90645054f4eb78c211aa7eed1e186fa65fc \
+ --hash=sha256:73e826b090d35af5a54130be1875d648146d689cf281313dd1a15bc85c557051
+ultralytics-thop==2.1.6 \
+ --hash=sha256:0ec2df8ebd3db35795e1f80cdc8bce6734446dbe989bca1b0c89396353f0f08c \
+ --hash=sha256:23f7b8ad124fa3432c1a7de9279102c4fdda699216032a7dff49f87ec3d1a3af
urllib3==2.7.0 \
--hash=sha256:231e0ec3b63ceb14667c67be60f2f2c40a518cb38b03af60abc813da26505f4c \
--hash=sha256:9fb4c81ebbb1ce9531cce37674bbc6f1360472bc18ca9a553ede278ef7276897
-uvicorn==0.44.0 \
- --hash=sha256:6c942071b68f07e178264b9152f1f16dfac5da85880c4ce06366a96d70d4f31e \
- --hash=sha256:ce937c99a2cc70279556967274414c087888e8cec9f9c94644dfca11bd3ced89
+uvicorn==0.52.4 \
+ --hash=sha256:73acfee47a0b133c5de13d219492d62d8a31e935f4fe6e41a232451a15379f86 \
+ --hash=sha256:f86e41a149d7d05a9969337e3946a9c171c06a5d42680896daaba624aeac8da1
vdms==0.0.23 \
--hash=sha256:2dec153f8cb33f27cdb9ab33125198195fd29a6777cf890fc735774ed9327874 \
--hash=sha256:7cd93242df644947c009559f31b4e19eb1dab6b62ab783ce5a2a54a4ad392f57
-wheel==0.46.3 \
- --hash=sha256:4b399d56c9d9338230118d705d9737a2a468ccca63d5e813e2a4fc7815d8bc4d \
- --hash=sha256:e3e79874b07d776c40bd6033f8ddf76a7dad46a7b8aa1b2787a83083519a1803
+wheel==0.48.0 \
+ --hash=sha256:3217dcc807155e45db462d7ef2431f5ddda0d7273b700d05a67b271ceb1287ab \
+ --hash=sha256:94800765601e9171bf5d58d066e640662842bcedcbab982b2c90787a2c987322
# The following packages are considered to be unsafe in a requirements file:
-pip==26.1.2 \
- --hash=sha256:382ff9f685ee3bc25864f820aa50505825f10f5458ffff07e30a6d96e5715cab \
- --hash=sha256:f49cd134c61cf2fd75e0ce2676db03e4054504a5a4986d00f8299ae632dc4605
+pip==26.2.1 \
+ --hash=sha256:71138adf1f4ca900cdb7d289c21b7494329f2332b6d85f0e1c42108c0384ed3e \
+ --hash=sha256:f6ad667e89a1fe78046c8f13232b247200f5258d7828f3f7883d660878e0813f
diff --git a/fastapi/templates/index.html b/fastapi/templates/index.html
index 2da2d9a..cc13603 100644
--- a/fastapi/templates/index.html
+++ b/fastapi/templates/index.html
@@ -146,7 +146,7 @@ Camera: {{ id }}
" const response = await fetch(self.location.origin + '/dashboard_stats'); " +
" if (response.ok) { " +
" const stats = await response.json(); " +
- " const activeIds = Object.keys(stats); " +
+ " const activeIds = Object.keys(stats).filter(id => stats[id] && stats[id].is_streaming === true); " +
" if (activeIds.length > 0) { " +
" /* ACTIVE: Poll fast (1s) */ " +
" currentDelay = 1000; " +
@@ -163,7 +163,7 @@ Camera: {{ id }}
" timeoutId = setTimeout(fetchData, currentDelay); " +
"} " +
"self.onmessage = function(e) { " +
- " if (e.data.action === 'start' || e.data.action === 'force') { " +
+ " if (e.data.action === 'start' || e.data.action === 'force' || e.data.action === 'resume') { " +
" clearTimeout(timeoutId); " + // Cancel any pending slow poll
" fetchData(); " + // Run immediately
" } " +
@@ -185,7 +185,7 @@ Camera: {{ id }}
document.querySelectorAll('.camera-card').forEach(card => {
const id = card.id.replace('card-', '');
const streamStats = stats[id];
- if (!activeIds.includes(id) || (streamStats && streamStats.is_streaming === false)) {
+ if (!activeIds.includes(id) || !streamStats || (streamStats && streamStats.is_streaming === false)) {
console.log("Removing dead stream:", id);
card.remove();
}
diff --git a/fastapi/tests/base_test.py b/fastapi/tests/base_test.py
new file mode 100644
index 0000000..067a9be
--- /dev/null
+++ b/fastapi/tests/base_test.py
@@ -0,0 +1,3444 @@
+# ==============================================================================
+# IMPORTS
+
+import gc
+import inspect
+import io
+import os
+import queue
+import sys
+import threading
+import time
+import traceback
+import types
+import zipfile
+from pathlib import Path
+
+import cv2
+import matplotlib.pyplot as plt
+import numpy as np
+import torch
+
+# Retrieve repo packages
+REPO_DIR = str(Path(__file__).parent.parent)
+sys.path.insert(1, REPO_DIR)
+from include.default_configs import TARGET_FPS
+from include.handlers import (
+ get_bb_overlay,
+)
+from include.utils import (
+ AsyncVideoWriter,
+ default_attr_keys,
+ global_frame_prefetch_worker_v1,
+ install_and_load_pip_package,
+ release_native_linux_heap,
+ scale_bbox,
+ scale_bbox_xywh,
+)
+
+objgraph = install_and_load_pip_package("objgraph", attribute_name=None)
+target_width, target_height = 7680, 4320 # 8K
+# torch.set_grad_enabled(False)
+
+# ==============================================================================
+# LOGGING
+import logging
+
+logging.basicConfig(
+ level=logging.INFO,
+ # format="%(asctime)s - %(name)s - %(levelname)s - %(message)s",
+ format="%(asctime)s [%(levelname)s] %(name)s (%(filename)s:%(lineno)d) - %(message)s",
+ handlers=[logging.StreamHandler(sys.stdout)],
+)
+
+logging.getLogger("libav").setLevel(logging.CRITICAL)
+logging.getLogger("libav.hevc").setLevel(logging.CRITICAL)
+main_app_logger = logging.getLogger(__name__)
+
+# ==============================================================================
+# FUNCTIONS
+
+
+# Download tracking (video) data and associated ground truth
+def download_eval_data(video_name_list, target_fps=TARGET_FPS):
+ DATA_DIR = Path(__file__).parent / "eval_data"
+ DATA_DIR.mkdir(parents=True, exist_ok=True)
+
+ if "gdown" not in sys.modules:
+ gdown = install_and_load_pip_package("gdown", attribute_name=None)
+
+ # DOWNLOAD GT ZIP
+ VIDEO_GROUND_TRUTH_URL = (
+ "https://drive.google.com/open?id=16PE3tBhT0lUGZLA8-zIRYvNUvxfhFZJq"
+ )
+ VIDEO_GROUND_TRUTH_ZIP_PATH = str(DATA_DIR / "Anti-UAV-Tracking-V0GT.zip")
+ if not Path(VIDEO_GROUND_TRUTH_ZIP_PATH).exists():
+ main_app_logger.info("Downloading ground truth zip file...")
+ gdown.download(
+ url=VIDEO_GROUND_TRUTH_URL, output=VIDEO_GROUND_TRUTH_ZIP_PATH, quiet=False
+ )
+
+ # GET GT BBOXES
+ # GT bbox: [x_min, y_min, w, h]
+ gt_boxes_dict = {}
+ with zipfile.ZipFile(VIDEO_GROUND_TRUTH_ZIP_PATH, "r") as gt_ref:
+ for gt_info in gt_ref.infolist():
+ matched_vid = next(
+ (
+ v
+ for v in video_name_list
+ if gt_info.filename.endswith(f"{v}_gt.txt")
+ ),
+ None,
+ )
+ if matched_vid:
+ gt_bytes = gt_ref.read(gt_info.filename)
+ gt_stream = io.StringIO(gt_bytes.decode("utf-8"))
+ delimiter = "," if b"," in gt_bytes else None
+ gt_boxes_dict[matched_vid] = np.loadtxt(
+ gt_stream, delimiter=delimiter, dtype=np.int32
+ ).tolist()
+
+ # DOWNLOAD FRAME ZIP
+ VIDEO_SEQUENCE_URL = (
+ "https://drive.google.com/open?id=1dlSPDggg6TRFMcC1jlYIJxxzUQS1mIh9"
+ )
+ VIDEO_SEQUENCE_ZIP_PATH = str(DATA_DIR / "Anti-UAV-Tracking-V0.zip")
+ if not Path(VIDEO_SEQUENCE_ZIP_PATH).exists():
+ main_app_logger.info("Downloading video sequence zip file...")
+ gdown.download(
+ url=VIDEO_SEQUENCE_URL, output=VIDEO_SEQUENCE_ZIP_PATH, quiet=False
+ )
+
+ # WRITE VIDEO AND GET VIDEO DETAILS
+ ALL_VIDEO_DETAILS = {}
+ # ALL_VIDEO_WRITERS = {}
+ # frame_counters = {v: 0 for v in video_name_list}
+ # Open Zip once, but process each video COMPLETELY one by one
+ with zipfile.ZipFile(VIDEO_SEQUENCE_ZIP_PATH, "r") as zip_ref:
+ # Pre-index all archive paths for ultra-fast matching
+ all_files = zip_ref.infolist()
+
+ for video_name in video_name_list:
+ main_app_logger.info(f"Processing sequence: {video_name}...")
+
+ # Filter and sort frames belonging strictly to the current video
+ video_img_infos = [
+ info
+ for info in all_files
+ if info.filename.startswith(f"Anti-UAV-Tracking-V0/{video_name}/")
+ and info.filename.lower().endswith(".jpg")
+ ]
+ video_img_infos.sort(key=lambda x: x.filename)
+
+ if not video_img_infos:
+ continue
+
+ video_writer = None
+ frame_idx = 0
+
+ # Setup dictionary tracking structure
+ sample_outpath = DATA_DIR / video_img_infos[0].filename.replace(
+ "Anti-UAV-Tracking-V0", "Anti-UAV-Tracking-V0-8K"
+ )
+ sample_outpath.parent.mkdir(parents=True, exist_ok=True)
+
+ # BYPASS COMPUTE IF VIDEO EXISTS
+ target_mp4_path = sample_outpath.parent / f"{video_name}.mp4"
+ if target_mp4_path.exists():
+ main_app_logger.info(
+ f"\t[SKIP] 8K target video {target_mp4_path.name} already compiled."
+ )
+
+ first_frame_info = video_img_infos[0]
+ img_bytes = zip_ref.read(first_frame_info.filename)
+ np_array = np.frombuffer(img_bytes, np.uint8)
+ img = cv2.imdecode(np_array, cv2.IMREAD_COLOR)
+ orig_height, orig_width, _ = img.shape if img is not None else (0, 0, 0)
+
+ current_gt_boxes = gt_boxes_dict.get(video_name, [])
+ num_invalid = sum(
+ 1 for bbox in current_gt_boxes if bbox == [0, 0, 0, 0]
+ )
+ num_valid = len(current_gt_boxes) - num_invalid
+
+ scaled_gt_boxes = []
+ for bbox in current_gt_boxes:
+ if bbox == [0, 0, 0, 0]:
+ scaled_gt_boxes.append([0, 0, 0, 0])
+ else:
+ scaled_gt_boxes.append(
+ scale_bbox_xywh(
+ bbox,
+ orig_width,
+ orig_height,
+ targetW=target_width,
+ targetH=target_height,
+ )
+ )
+
+ # Reconstruct the metadata payload so downstream tracking logic doesn't break
+ ALL_VIDEO_DETAILS[video_name] = {
+ "frame_dir": str(sample_outpath.parent),
+ "orig_height": orig_height, # Safe placeholder or pull from cache if variable
+ "orig_width": orig_width,
+ "target_height": target_height,
+ "target_width": target_width,
+ "sequence": [Path(info.filename).stem for info in video_img_infos],
+ "gt_bbox": scaled_gt_boxes,
+ "num_invalid_gt_bboxes": num_invalid,
+ "num_valid_gt_bboxes": num_valid,
+ }
+ num_frames = len(ALL_VIDEO_DETAILS[video_name]["sequence"])
+ num_boxes = len(ALL_VIDEO_DETAILS[video_name]["gt_bbox"])
+ num_invalid_boxes = ALL_VIDEO_DETAILS[video_name][
+ "num_invalid_gt_bboxes"
+ ]
+ main_app_logger.info(
+ f"\t{video_name}: {num_frames} frames, {num_boxes} gt boxes ({num_invalid_boxes} out-of-view)"
+ )
+ continue # Instantly advance to the next video name in the zip reference
+
+ # Initialize detail log
+ ALL_VIDEO_DETAILS[video_name] = {
+ "frame_dir": str(sample_outpath.parent),
+ "orig_height": 0,
+ "orig_width": 0,
+ "target_height": target_height,
+ "target_width": target_width,
+ "sequence": [],
+ "gt_bbox": [],
+ "num_invalid_gt_bboxes": 0,
+ "num_valid_gt_bboxes": 0,
+ }
+
+ for file_info in video_img_infos:
+ frame_id = Path(file_info.filename).stem
+
+ # In-memory decompression
+ img_bytes = zip_ref.read(file_info.filename)
+ np_array = np.frombuffer(img_bytes, np.uint8)
+ img = cv2.imdecode(np_array, cv2.IMREAD_COLOR)
+
+ if img is not None:
+ orig_height, orig_width, _ = img.shape
+
+ # Update dimensions once per video segment
+ if frame_idx == 0:
+ ALL_VIDEO_DETAILS[video_name]["orig_height"] = orig_height
+ ALL_VIDEO_DETAILS[video_name]["orig_width"] = orig_width
+
+ # Performance upscale to 8K
+ img_8K = cv2.resize(
+ img,
+ (target_width, target_height),
+ interpolation=cv2.INTER_LINEAR, # INTER_CUBIC,
+ )
+
+ # Initialize Video Writer sequentially if missing
+ if (
+ video_writer is None
+ and not (sample_outpath.parent / f"{video_name}.mp4").exists()
+ ):
+ video_writer = AsyncVideoWriter(
+ str(sample_outpath.parent / f"{video_name}.mp4"),
+ cv2.VideoWriter_fourcc(*"avc1"),
+ target_fps,
+ (target_width, target_height),
+ )
+
+ # Synchronous frame pipeline updates
+ if video_writer is not None:
+ video_writer.write_frame(img_8K)
+ ALL_VIDEO_DETAILS[video_name]["sequence"].append(frame_id)
+
+ # Bounding Box Sync
+ if video_name in gt_boxes_dict and frame_idx < len(
+ gt_boxes_dict[video_name]
+ ):
+ gt_bbox = gt_boxes_dict[video_name][frame_idx]
+
+ if gt_bbox == [0, 0, 0, 0]:
+ x, y, w, h = [0, 0, 0, 0]
+ ALL_VIDEO_DETAILS[video_name]["num_invalid_gt_bboxes"] += 1
+ else:
+ x, y, w, h = scale_bbox_xywh(
+ gt_bbox,
+ orig_width,
+ orig_height,
+ targetW=target_width,
+ targetH=target_height,
+ )
+ ALL_VIDEO_DETAILS[video_name]["num_valid_gt_bboxes"] += 1
+
+ ALL_VIDEO_DETAILS[video_name]["gt_bbox"].append([x, y, w, h])
+
+ frame_idx += 1
+
+ # Close the writer IMMEDIATELY
+ if video_writer is not None:
+ video_writer.release()
+
+ num_frames = len(ALL_VIDEO_DETAILS[video_name]["sequence"])
+ num_boxes = len(ALL_VIDEO_DETAILS[video_name]["gt_bbox"])
+ num_invalid_boxes = ALL_VIDEO_DETAILS[video_name]["num_invalid_gt_bboxes"]
+ main_app_logger.info(
+ f"\t{video_name}: {num_frames} frames, {num_boxes} gt boxes ({num_invalid_boxes} out-of-view)"
+ )
+
+ return ALL_VIDEO_DETAILS
+
+
+# Print active CUDA tensors
+def track_gpu_tensors():
+ for obj in gc.get_objects():
+ try:
+ if torch.is_tensor(obj) and obj.is_cuda:
+ tensor_candidate = obj
+ elif (
+ hasattr(obj, "data") and torch.is_tensor(obj.data) and obj.data.is_cuda
+ ):
+ tensor_candidate = obj.data
+ else:
+ tensor_candidate = None
+
+ if tensor_candidate is not None: # and tensor_candidate.is_cuda:
+ nelement = tensor_candidate.nelement()
+ cand_shape = list(tensor_candidate.shape)
+ cand_dtype = tensor_candidate.dtype
+ tensor_bytes = tensor_candidate.element_size() * nelement
+ # main_app_logger.info(f"Leaked Tensor -> Type: {type(obj)} | Size: {obj.size()} | Device: {obj.device}")
+ main_app_logger.info(
+ f" > Tensor | Device: {tensor_candidate.device} | Shape: {str(cand_shape):<18} | Dtype: {str(cand_dtype):<12} | Size: {tensor_bytes / 1024**2:6.2f} MB"
+ )
+ except Exception:
+ pass
+ del obj
+
+
+# Save FPS comparison chart
+def fps_comparison_chart(chart_path, results, fps_key="Pipeline FPS (Video frames)"):
+ try:
+ # names = [r["Test Name"] for r in results]
+ # fps_values = [float(r[fps_key]) for r in results]
+ names = []
+ fps_values = []
+
+ for r in results:
+ if isinstance(r, dict) and "Test Name" in r:
+ # Force casting to clean, unpinned Python native types
+ names.append(str(r["Test Name"]))
+ fps_values.append(float(r.get(fps_key, 0.0)))
+
+ plt.figure(figsize=(10, 6))
+ plt.grid(axis="y", linestyle="--", alpha=0.7, zorder=0)
+
+ colors = ["#2ca02c" if "gpu" in n.lower() else "#1f77b4" for n in names]
+ bars = plt.bar(names, fps_values, color=colors, zorder=3)
+ plt.ylabel("Frames Per Second (FPS)")
+ plt.title(f"Performance Comparison: {chart_path.stem}")
+ plt.xticks(rotation=45)
+
+ for bar in bars:
+ yval = bar.get_height()
+ plt.text(
+ bar.get_x() + bar.get_width() / 2,
+ yval + 1,
+ f"{yval:.1f}",
+ ha="center",
+ va="bottom",
+ )
+
+ if fps_values:
+ plt.ylim(0, max(fps_values) * 1.2)
+
+ plt.tight_layout()
+ plt.savefig(str(chart_path))
+ main_app_logger.info(f" Comparison chart saved to: {chart_path}")
+ except Exception:
+ main_app_logger.info("Skipping chart generation: error occurred.")
+
+
+# ==============================================================================
+# CLASSES
+class BaseTest:
+ # PIPELINE FUNCTIONS --------------------------------------------
+ @torch.inference_mode()
+ def pipeline_fn(
+ self,
+ device_frame,
+ overall_frame_num,
+ stat_start_time,
+ current_clip_id,
+ gt_boxes=None,
+ read_frame_only=False,
+ ):
+ current_clip_key = f"{self.name}_{current_clip_id:03d}.mp4"
+ current_clip_path = f"{self.config.SHARED_OUTPUT}/{current_clip_key}"
+
+ metadata = {}
+ motion_detected = False
+ metrics = {"sf_time": 0, "roi_time": 0, "det_time": 0, "bbs": None}
+
+ # Given input frame, get metadata and metrics
+ try:
+ if read_frame_only:
+ # Get frame, metadata and metrics
+ _, det_frame = self.processor.format_bbs_and_frame_4_detection(
+ [], device_frame
+ )
+
+ else:
+ # Run model on frame, get metadata and metrics
+ # with torch.inference_mode():
+ metrics, metadata, det_frame, motion_detected = self.processor.run(
+ device_frame,
+ overall_frame_num,
+ frame_in_clip_count=self.frame_in_clip_count,
+ gt_boxes=None,
+ )
+
+ self.total_objects_detected += len(metadata.keys())
+
+ # Save results in video
+ self.frame2video(
+ det_frame,
+ overall_frame_num,
+ metadata,
+ getattr(self, "label_source", None),
+ stat_start_time,
+ gt_boxes=gt_boxes,
+ )
+
+ except Exception as e_detection:
+ traceback.print_exc()
+ main_app_logger.info(f"[DETECTION ERROR] Exception: {e_detection}")
+ # traceback.print_exc()
+ # if self.active:
+ # traceback.print_exc()
+ # main_app_logger.info(f"[DETECTION ERROR] Exception: {e_detection}")
+ # self.active = False
+ finally:
+ # del inf_data, bbs_full_res, device_frame, det_frame
+ if "det_frame" in locals():
+ del det_frame
+ if "device_frame" in locals():
+ del device_frame
+ # if "bbs_full_res" in locals():
+ # del bbs_full_res
+ # if "inf_data" in locals():
+ # del inf_data
+
+ # del inf_data, bbs_full_res, device_frame, det_frame
+
+ return metadata, metrics, motion_detected # Skip full detection pass
+
+ def initialize_run_realtime_inference_v1(
+ self, read_frame_only, num_prefetch_workers
+ ):
+ # self.duration_target = 30
+ # self.status = "RUNNING"
+ self.processor = self._setup_processor(read_frame_only)
+
+ if hasattr(self, "processor") and hasattr(self.processor, "label_source"):
+ self.label_source = self.processor.label_source
+ # Setup empty list to track background futures natively
+ # self._active_inference_futures = []
+
+ if self.is_cuda and not hasattr(self, "sf_start") and self.config.TEST_MODE:
+ self.sf_start, self.sf_end = (
+ torch.cuda.Event(enable_timing=True),
+ torch.cuda.Event(enable_timing=True),
+ )
+ self.roi_start, self.roi_end = (
+ torch.cuda.Event(enable_timing=True),
+ torch.cuda.Event(enable_timing=True),
+ )
+ self.det_start, self.det_end = (
+ torch.cuda.Event(enable_timing=True),
+ torch.cuda.Event(enable_timing=True),
+ )
+
+ if self.config.TEST_MODE:
+ self.component_stats = {
+ "sf": [],
+ "roi": [],
+ "det": [],
+ "dma_upload": [], # Track PCIe Transfer Latencies
+ "queue_blocked": [], # Track GIL Serialization stalls
+ "batch_sizes": [], # Track Smart Filtering density
+ "thread_backlog": [], # Track thread work pool backlog
+ }
+ self.crops_per_frame_list = []
+
+ # Bind active execution permanently to sub-stream BEFORE entering loop (Bypasses TLS Overhead)
+ # if self.device_input == "cuda" and torch.cuda.is_available():
+ # torch.cuda.set_stream(self.inference_stream)
+ # if hasattr(self, "target_fps") and hasattr(self, "duration_target"):
+ # self.max_target_frames = int(self.duration_target * float(self.target_fps))
+ # elif hasattr(self, "numFrames"):
+ # self.max_target_frames = self.numFrames
+ # else:
+ # self.max_target_frames = float("inf")
+
+ if (
+ hasattr(self, "numFrames")
+ and self.numFrames > 0
+ and not getattr(self, "is_rtsp", False)
+ ):
+ # For local test files, compute total target frames from input video length
+ self.max_target_frames = int(
+ self.numFrames / getattr(self.reader, "step_size", 1.0)
+ )
+ elif hasattr(self, "target_fps") and hasattr(self, "duration_target"):
+ self.max_target_frames = int(self.duration_target * float(self.target_fps))
+ else:
+ self.max_target_frames = float("inf")
+
+ # self.dynamic_limit = max(2, int(0.5 * self.target_fps))
+
+ # Initialize a thread-safe signaling handle at class setup (run_realtime_inference)
+ self.queue_data_ready_event = threading.Event()
+
+ try:
+ for i in range(num_prefetch_workers):
+ # prefetch_thread = mp.Process(
+ prefetch_thread = threading.Thread(
+ target=lambda: global_frame_prefetch_worker_v1(self),
+ daemon=True,
+ # name=f"{self.name}_prefetch_{i}",
+ )
+ prefetch_thread.start()
+ self.prefetch_threads.append(prefetch_thread)
+ except Exception as e:
+ main_app_logger.critical(
+ f"[{self.name}] Failed to start prefetch workers: {e}", exc_info=True
+ )
+ traceback.print_exc()
+ raise
+
+ @torch.inference_mode()
+ def run_realtime_inference_v1(
+ self,
+ sf_enabled=True,
+ profiler=None,
+ gt_enabled=False,
+ read_frame_only=False,
+ ):
+ self.status = "RUNNING"
+
+ metrics = {}
+ all_metadata = {}
+ num_objs = 0
+ total_pipeline_time_ms = 0.0
+ total_read_time = 0.0
+ total_queue_saturation = 0.0
+ total_written_frames = 0
+ total_sf_pipeline_time_ms = 0.0
+ real_world_latency_ms = 0.0
+ total_run_pipelinefn = 0
+ total_frame_loop_latency = 0.0
+ total_active_processing_overhead = 0.0
+ queue_saturation_history = []
+ consecutive_slow_frames = 0
+ coverage_percentages = []
+
+ if gt_enabled and hasattr(self, "VIDEO_GT_DETAILS") and self.VIDEO_GT_DETAILS:
+ # VIDEO_GT_DETAILS = download_eval_data([self.name])
+ gt_boxes = self.VIDEO_GT_DETAILS[self.name]["gt_bbox"]
+ gt_sequence = self.VIDEO_GT_DETAILS[self.name]["sequence"]
+
+ self.initialize_run_realtime_inference(
+ read_frame_only, self.active_workers_count
+ )
+
+ self.stat_start_time = time.perf_counter()
+ pipeline_start_time = self.stat_start_time
+ last_loop_cycle_timestamp = self.stat_start_time
+
+ missing_frame_cnt = 0
+ max_retries = int(self.target_fps)
+
+ # Release the master startup event at the last possible moment.
+ self.main_startup_event.set()
+ while (
+ self.active # or not self.prefetch_queue.empty()
+ ): # and self.frame_count_target < self.max_target_frames:
+ if self.frame_count_target >= self.max_target_frames:
+ self.active = False
+ self.reader.running_flag.value = False
+ break
+
+ stat_start_time = time.perf_counter()
+ try:
+ # FRAME RETRIEVAL ---------------------------------------------
+ try:
+ frame_start_time = time.perf_counter()
+ safe_frame, frame_details, should_continue = (
+ self._get_frame_from_queue()
+ )
+
+ if not should_continue:
+ self.active = False # Signal loop termination
+ break
+ if safe_frame is None:
+ # Allow retries while prefetch workers are still active
+ if getattr(self, "prefetch_active", True) and not getattr(
+ self.reader, "stopped", False
+ ):
+ time.sleep(0.002)
+ continue
+ missing_frame_cnt += 1
+ if missing_frame_cnt >= max_retries:
+ self.active = False
+ main_app_logger.info(
+ "Too many frames missing and workers done. Exiting..."
+ )
+ break
+ continue # Skip to the next iteration if the frame is invalid
+
+ # try:
+ # # ret, slot_idx = self.prefetch_queue.get(block=True, timeout=1.0)
+ # # ret, slot_idx = self.prefetch_queue.get(block=False)
+ # ret, slot_idx = self.prefetch_queue.get(block=True, timeout=1.0)
+ # self.prefetch_queue.task_done()
+
+ # except queue.Empty:
+ # with self.worker_tracking_lock:
+ # if (
+ # self.active_workers_count == 0
+ # and self.prefetch_queue.empty()
+ # ):
+ # self.active = False
+ # break
+ # # If the queue is empty AND the prefetch workers are done, we can exit.
+ # if not self.active:
+ # break
+ # continue
+
+ # # If self.stop() drops the queues, break out of the thread loop natively.
+ # except (OSError, ValueError, AssertionError):
+ # main_app_logger.info(
+ # "[PROCESS THREAD] Ingestion queue disconnected via stop() signal. Breaking loop."
+ # )
+ # self.active = False
+ # break
+
+ # if ret is False or slot_idx == "END_OF_STREAM":
+ # main_app_logger.info(
+ # "[PROCESS THREAD] End of video stream detected. Breaking loop naturally."
+ # )
+ # self.active = False
+ # break
+
+ # if not self.active:
+ # break
+
+ # # if slot_idx == -1 or self.shm_buffer_pool[slot_idx] is None:
+ # # continue
+ # if (
+ # slot_idx == -1
+ # or not hasattr(self, "shm_buffer_pool")
+ # or self.shm_buffer_pool is None
+ # ):
+ # continue
+ # if (
+ # slot_idx >= len(self.shm_buffer_pool)
+ # or self.shm_buffer_pool[slot_idx] is None
+ # ):
+ # continue
+
+ # # Zero-Copy Reference Extraction straight out of the memory slot array
+ # (
+ # raw_shm_frame,
+ # current_event,
+ # frame_num,
+ # abs_frame_num,
+ # true_read_latency_secs,
+ # ) = self.shm_buffer_pool[slot_idx]
+
+ # if frame_num == 0:
+ # main_app_logger.info(
+ # f"[VERIFY - CONSUMER] Main loop is officially processing Frame {frame_num}!",
+ # )
+ # # self.shm_buffer_pool[slot_idx] = None
+
+ # reader_time = (
+ # true_read_latency_secs # time.perf_counter() - frame_start_time
+ # )
+ # total_read_time += reader_time
+
+ # if "cuda" in str(self.device_input):
+ # safe_frame = self.gpu_input[slot_idx]
+ # # safe_frame = cpu_tensor.to(self.device_input, non_blocking=True)
+ # safe_frame.copy_(raw_shm_frame)
+ # # self.gpu_input[slot_idx].zero_()
+ # else:
+ # safe_frame = raw_shm_frame
+
+ # # Record the PyTorch CUDA event on the current stream
+ # if current_event is not None and isinstance(
+ # current_event, torch.cuda.Event
+ # ):
+ # # This tells the GPU that the consumer is done reading this slot
+ # current_event.record(torch.cuda.current_stream())
+
+ # Originally 0-based but make 1-based
+ # frame_num += 1
+ # abs_frame_num += 1
+ total_read_time += frame_details["reader_time"]
+ missing_frame_cnt = 0
+
+ # Calculate the exact time gap between the last frame completion
+ # and the start of the next read. This accurately captures downstream GIL blocks.
+ cycle_gap = frame_start_time - last_loop_cycle_timestamp
+ # Subtract the actual read duration to isolate the thread stall time
+ true_serialization_stall = max(
+ 0.0, (cycle_gap - frame_details["reader_time"]) * 1000.0
+ )
+ self.component_stats["queue_blocked"].append(
+ true_serialization_stall
+ )
+
+ except queue.Empty:
+ if getattr(self.reader, "reconnect_failed", False):
+ self.active = False
+ break
+ if not self.active:
+ break # Exit on error if shutdown has been initiated
+ time.sleep(0.001)
+ continue
+
+ # FRAME PROCESSING ---------------------------------------------
+ # stat_start_time = time.perf_counter()
+ # stat_start_time = self.stat_start_time
+ self.abs_frame_num = frame_details["abs_frame_num"]
+ calculated_clip_id = (
+ frame_details["frame_num"] - 1
+ ) // self.max_frames_per_clip
+ self.frame_count += 1
+ self.frame_count_target += 1
+ self.frame_in_clip_count += 1
+
+ # Real-Time Metric A: Track Queue Saturation Density Ratio
+ # current_q_size = self.reader.frame_queue.qsize()
+ saturation_ratio = (
+ len(self.reader.frame_queue) / self.reader.frame_queue.maxlen
+ )
+ total_queue_saturation += saturation_ratio
+ queue_saturation_history.append(saturation_ratio)
+
+ # RUN PIPELINE_FN ---------------------------------------------
+ run_pipelinefn_start = time.perf_counter()
+ # self.frame_count_target += 1
+ # self.frame_in_clip_count += 1
+
+ if gt_enabled:
+ target_boxes_array = self.get_frame_gt_boxes(
+ frame_details["abs_frame_num"], gt_sequence, gt_boxes
+ )
+
+ metadata_or_bbs, metrics, motion_detected = self.pipeline_fn(
+ safe_frame,
+ frame_details["frame_num"],
+ stat_start_time,
+ calculated_clip_id,
+ gt_boxes=target_boxes_array if gt_enabled else None,
+ read_frame_only=read_frame_only,
+ )
+
+ num_objs += len(metadata_or_bbs.keys())
+ # if metadata_or_bbs is not None and isinstance(metadata_or_bbs, dict):
+ # all_metadata.update(metadata_or_bbs)
+
+ total_written_frames += 1
+
+ # FRAME POST PROCESSING ---------------------------------------------
+ # Track Execution Boundaries
+ frame_end_time = time.perf_counter()
+ total_run_pipelinefn += frame_end_time - run_pipelinefn_start
+ last_loop_cycle_timestamp = frame_end_time
+
+ # Calculate frame metrics
+
+ # Real-Time Metric B: Frame Processing Latency Check
+ frame_loop_latency = (frame_end_time - frame_start_time) * 1000 # ms
+ total_frame_loop_latency += frame_loop_latency
+
+ # Isolate active processing time (excluding the queue blocking read)
+ active_processing_overhead = frame_loop_latency - (
+ frame_details["reader_time"] * 1000
+ )
+ total_active_processing_overhead += active_processing_overhead
+
+ # RUNTIME TERMINAL WARNING ALERTS
+ if saturation_ratio >= 1.0 or active_processing_overhead > (
+ 1000 / self.target_fps
+ ):
+ consecutive_slow_frames += 1
+ if (
+ consecutive_slow_frames % self.target_fps == 0
+ ): # Throttle terminal spam
+ reader_time = frame_details["reader_time"]
+ main_app_logger.info(
+ f"\033[93m⚠️ [PERF WARNING] Main loop starving! Waiting on stream ingestion... "
+ f"Read Wait: {reader_time:.1f}ms | Other Wait: {active_processing_overhead:.1f}ms | Queue Fullness: {saturation_ratio * 100:.0f}%\033[0m"
+ )
+ else:
+ consecutive_slow_frames = max(0, consecutive_slow_frames - 1)
+ # if self.device_input == "cuda" and torch.cuda.is_available():
+ # torch.cuda.default_stream().synchronize()
+
+ if gt_enabled and motion_detected:
+ self.update_eval_stats(metadata_or_bbs, target_boxes_array)
+
+ if metrics != {}:
+ num_crops = len(metrics["bbs"]) if metrics["bbs"] is not None else 0
+ self.crops_per_frame_list.append(num_crops)
+
+ # Context-aware extraction of the newly proposed metrics
+ density = metrics.get(
+ "batch_density", 0 if self.config.sf_enabled else 1
+ )
+ self.component_stats["batch_sizes"].append(density)
+
+ self.component_stats["sf"].append(metrics["sf_time"])
+ # self.component_stats["roi"].append(metrics["roi_time"])
+ self.component_stats["det"].append(metrics["det_time"])
+
+ # Calculate coverage OUTSIDE the timed block to prevent interference
+ if self.config.sf_enabled and metrics.get("bbs") is not None:
+ cov = self.calculate_unique_coverage(metrics["bbs"])
+ if hasattr(cov, "item"):
+ coverage_percentages.append(float(cov.item()))
+ else:
+ coverage_percentages.append(float(cov))
+
+ # Explicitly delete frame variables to free their references
+ frame_8k = None
+ del frame_8k, safe_frame
+ if "metadata_or_bbs" in locals():
+ del metadata_or_bbs
+
+ if total_written_frames % (10 * self.target_fps) == 0:
+ # main_app_logger.info(
+ # f"Captured {total_written_frames}/{self.max_target_frames} frames..."
+ # )
+ # Querying the cross-process property once every 5 seconds reduces proxy overhead to 0%
+ # if self.device_input == "cuda" and hasattr(
+ # self.reader, "total_h2d_time"
+ # ):
+ # avg_h2d = (
+ # self.reader.total_h2d_time
+ # / max(1, total_written_frames + 1)
+ # ) * 1000.0
+ # main_app_logger.info(
+ # f"Captured {total_written_frames}/{self.max_target_frames} frames... DMA Upload: {avg_h2d:.2f}ms"
+ # )
+ # else:
+ main_app_logger.info(
+ f"Captured {total_written_frames}/{self.max_target_frames} frames..."
+ )
+
+ if self.frame_count % 100 == 0:
+ gc.collect()
+
+ # Force PyTorch's internal allocator to release cached segments back to the OS
+ # if self.frame_count_target % (5 * self.target_fps) == 0:
+ # main_app_logger.info(
+ # f"Captured {self.frame_count_target}/{self.max_target_frames} frames..."
+ # )
+ # torch.cuda.empty_cache()
+ # torch.cuda.ipc_collect()
+
+ # Force a microscopic micro-yield if executing on CPU.
+ # This grants immediate execution priority back to the background cleaning thread,
+ # allowing the garbage collector to evict data from RAM instantly!
+ if self.device_input == "cpu":
+ time.sleep(
+ 0
+ ) # .005) # 1ms yield breaks core processor starvation lock
+
+ # END -> frame processing
+ except torch.cuda.OutOfMemoryError:
+ main_app_logger.info("!" * 70)
+ main_app_logger.info(
+ "[CRITICAL TEST CRASH] GPU MEMORY CEILING HIT INSIDE RUNNER LOOP!"
+ )
+ main_app_logger.info(
+ "Freezing allocation history registers and writing diagnostic log..."
+ )
+ main_app_logger.info("!" * 70)
+
+ try:
+ snapshot_filename = (
+ f"/tmp/test_vram_leak_profile_pid{os.getpid()}.pickle"
+ )
+ torch.cuda.memory._dump_snapshot(snapshot_filename)
+ main_app_logger.info(
+ f"[PROFILER SUCCESSFUL] Snapshot profile written to: {snapshot_filename}"
+ )
+ main_app_logger.info(
+ "--> Drag and drop this file directly into: https://pytorch.org"
+ )
+ except Exception as dump_err:
+ main_app_logger.info(
+ f"Failed to record profile data snapshot: {dump_err}"
+ )
+
+ # Force safe system unlinking of background workers to clean up OS handles
+ self.active = False
+ if hasattr(self, "reader") and self.reader is not None:
+ self.reader.stop()
+ raise
+
+ except Exception as e:
+ if not self.active or not getattr(self, "prefetch_active", True):
+ main_app_logger.info(
+ "[PROCESS THREAD] System shutdown detected during exception sweep. Exiting thread payload context."
+ )
+ break
+
+ main_app_logger.info(
+ f"[CRITICAL PIPELINE ERROR] Crash on frame: {repr(e)}"
+ )
+ traceback.print_exc()
+ raise e # Let it break so you can see the exact line number!
+
+ # END -> while self.active and self.frame_count_target < self.max_target_frames:
+
+ # PIPELINE POST PROCESSING ---------------------------------------------
+ pipeline_end_time = time.perf_counter()
+
+ real_world_latency_ms = float(
+ (pipeline_end_time - pipeline_start_time) * 1000.0
+ )
+
+ total_sf_pipeline_time_ms = float(
+ sum(self.component_stats.get("sf", [0.0]))
+ + sum(self.component_stats.get("roi", [0.0]))
+ + sum(self.component_stats.get("det", [0.0]))
+ )
+ total_pipeline_time_ms = float(
+ (pipeline_end_time - pipeline_start_time) * 1000.0
+ )
+
+ self.async_writer.release()
+
+ # Continuously sample the thread work pool backlog state mechanics
+ if hasattr(self, "executor") and self.executor is not None:
+ # Safely probe internal concurrent.futures work queue bounds
+ self.component_stats["thread_backlog"].append(
+ self.executor._work_queue.qsize()
+ )
+ else:
+ self.component_stats["thread_backlog"].append(0)
+
+ # CALCULATE PERFORMANCE METRICS ---------------------------------------------
+ main_app_logger.info(
+ f"Execution Finished. Total Output Frames Written: {self.frame_count_target}"
+ )
+
+ # Render your metrics dashboard table cleanly
+ if self.frame_count_target > 0:
+ avg_frame_loop_latency = total_frame_loop_latency / self.frame_count_target
+ self._finalize_benchmarks(
+ num_objs,
+ self.frame_count_target, # Reads the updated integer directly!
+ total_pipeline_time_ms,
+ real_world_latency_ms,
+ total_sf_pipeline_time_ms,
+ avg_frame_loop_latency,
+ coverage_percentages,
+ total_read_time,
+ total_queue_saturation,
+ queue_saturation_history,
+ computed_eval_metrics=self.evaluator.compute_final_metrics()
+ if hasattr(self, "evaluator")
+ else None
+ if gt_enabled
+ else None,
+ )
+ else:
+ main_app_logger.info(
+ f"[SKIPPED SUMMARY] Only {self.frame_count_target} frames processed. "
+ )
+
+ # Force early hardware driver sweep before unbinding threads
+ if self.device_input == "cuda" and torch.cuda.is_available():
+ torch.cuda.synchronize()
+ torch.cuda.empty_cache()
+ torch.cuda.ipc_collect()
+
+ gc.collect()
+
+ if profiler is not None:
+ profiler.stop()
+
+ self.stop()
+
+ def initialize_run_realtime_inference(self, read_frame_only, num_prefetch_workers):
+ self.processor = self._setup_processor(read_frame_only)
+
+ if hasattr(self, "processor") and hasattr(self.processor, "label_source"):
+ self.label_source = self.processor.label_source
+
+ if (
+ hasattr(self, "numFrames")
+ and self.numFrames > 0
+ and not getattr(self, "is_rtsp", False)
+ ):
+ # For local test files, compute total target frames from input video length
+ self.max_target_frames = int(
+ self.numFrames / getattr(self.reader, "step_size", 1.0)
+ )
+ elif hasattr(self, "target_fps") and hasattr(self, "duration_target"):
+ self.max_target_frames = int(self.duration_target * float(self.target_fps))
+ else:
+ self.max_target_frames = float("inf")
+ # =========================================================================
+ # 🚀 PRE-ALLOCATED METRICS SCRATCHPAD (Zero Dynamic Allocations in Loop)
+ # =========================================================================
+ cap = max(
+ 1000,
+ self.max_target_frames if self.max_target_frames != float("inf") else 10000,
+ )
+ self._metric_sf_times = np.zeros(cap, dtype=np.float32)
+ self._metric_det_times = np.zeros(cap, dtype=np.float32)
+ self._metric_reader_times = np.zeros(cap, dtype=np.float32)
+ self._metric_queue_blocked = np.zeros(cap, dtype=np.float32)
+ self._metric_queue_saturations = np.zeros(cap, dtype=np.float32)
+ self._metric_crops_count = np.zeros(cap, dtype=np.int32)
+ self._metric_batch_densities = np.zeros(cap, dtype=np.int32)
+ self._metric_coverage_tensors = []
+ # ============================
+
+ if self.is_cuda and not hasattr(self, "sf_start") and self.config.TEST_MODE:
+ self.sf_start, self.sf_end = (
+ torch.cuda.Event(enable_timing=True),
+ torch.cuda.Event(enable_timing=True),
+ )
+ self.roi_start, self.roi_end = (
+ torch.cuda.Event(enable_timing=True),
+ torch.cuda.Event(enable_timing=True),
+ )
+ self.det_start, self.det_end = (
+ torch.cuda.Event(enable_timing=True),
+ torch.cuda.Event(enable_timing=True),
+ )
+
+ if self.config.TEST_MODE:
+ self.component_stats = {
+ "sf": [],
+ "roi": [],
+ "det": [],
+ "dma_upload": [], # Track PCIe Transfer Latencies
+ "queue_blocked": [], # Track GIL Serialization stalls
+ "batch_sizes": [], # Track Smart Filtering density
+ "thread_backlog": [], # Track thread work pool backlog
+ }
+ self.crops_per_frame_list = []
+
+ # Initialize a thread-safe signaling handle at class setup (run_realtime_inference)
+ self.queue_data_ready_event = threading.Event()
+
+ try:
+ for i in range(num_prefetch_workers):
+ # prefetch_thread = mp.Process(
+ prefetch_thread = threading.Thread(
+ target=lambda: global_frame_prefetch_worker_v1(self),
+ daemon=True,
+ # name=f"{self.name}_prefetch_{i}",
+ )
+ prefetch_thread.start()
+ self.prefetch_threads.append(prefetch_thread)
+ except Exception as e:
+ main_app_logger.critical(
+ f"[{self.name}] Failed to start prefetch workers: {e}", exc_info=True
+ )
+ traceback.print_exc()
+ raise
+
+ @torch.inference_mode()
+ def run_realtime_inference(
+ self,
+ sf_enabled=True,
+ profiler=None,
+ gt_enabled=False,
+ read_frame_only=False,
+ ):
+ self.status = "RUNNING"
+ self.initialize_run_realtime_inference(
+ read_frame_only, self.active_workers_count
+ )
+
+ gt_boxes = None
+ gt_sequence = None
+ if gt_enabled and hasattr(self, "VIDEO_GT_DETAILS") and self.VIDEO_GT_DETAILS:
+ gt_boxes = self.VIDEO_GT_DETAILS[self.name]["gt_bbox"]
+ gt_sequence = self.VIDEO_GT_DETAILS[self.name]["sequence"]
+
+ self.stat_start_time = time.perf_counter()
+ pipeline_start_time = self.stat_start_time
+ last_loop_cycle_timestamp = self.stat_start_time
+
+ missing_frame_cnt = 0
+ max_retries = int(self.target_fps)
+ total_read_time = 0.0
+ total_written_frames = 0
+ total_queue_saturation = 0.0
+ total_run_pipelinefn = 0.0
+ total_frame_loop_latency = 0.0
+ total_active_processing_overhead = 0.0
+ consecutive_slow_frames = 0
+ num_objs = 0
+ queue_saturation_history = []
+ coverage_percentages = []
+
+ # Release the master startup event
+ self.main_startup_event.set()
+
+ while self.active:
+ if self.frame_count_target >= self.max_target_frames:
+ self.active = False
+ if hasattr(self.reader, "running_flag"):
+ self.reader.running_flag.value = False
+ break
+
+ stat_start_time = time.perf_counter()
+ try:
+ # 1. FRAME RETRIEVAL ------------------------------------------
+ frame_start_time = time.perf_counter()
+ safe_frame, frame_details, should_continue = (
+ self._get_frame_from_queue()
+ )
+
+ if not should_continue:
+ self.active = False
+ break
+
+ if safe_frame is None:
+ if getattr(self, "prefetch_active", True) and not getattr(
+ self.reader, "stopped", False
+ ):
+ time.sleep(0.001)
+ continue
+ missing_frame_cnt += 1
+ if missing_frame_cnt >= max_retries:
+ self.active = False
+ main_app_logger.info(
+ "Too many frames missing and workers done. Exiting..."
+ )
+ break
+ continue
+
+ missing_frame_cnt = 0
+ total_read_time += frame_details["reader_time"]
+
+ # Downstream Serialization Pressure Tracking
+ cycle_gap = frame_start_time - last_loop_cycle_timestamp
+ true_serialization_stall = max(
+ 0.0, (cycle_gap - frame_details["reader_time"]) * 1000.0
+ )
+ self.component_stats["queue_blocked"].append(true_serialization_stall)
+
+ # Frame counters and clipping indices
+ self.abs_frame_num = frame_details["abs_frame_num"]
+ calculated_clip_id = (
+ frame_details["frame_num"] - 1
+ ) // self.max_frames_per_clip
+ self.frame_count += 1
+ self.frame_count_target += 1
+ self.frame_in_clip_count += 1
+
+ # Queue Saturation Density Ratio Tracking
+ if (
+ hasattr(self.reader, "frame_queue")
+ and self.reader.frame_queue.maxlen
+ ):
+ saturation_ratio = (
+ len(self.reader.frame_queue) / self.reader.frame_queue.maxlen
+ )
+ else:
+ saturation_ratio = 0.0
+ total_queue_saturation += saturation_ratio
+ queue_saturation_history.append(saturation_ratio)
+
+ # 2. RUN PIPELINE_FN ------------------------------------------
+ run_pipelinefn_start = time.perf_counter()
+
+ target_boxes_array = None
+ if gt_enabled:
+ target_boxes_array = self.get_frame_gt_boxes(
+ frame_details["abs_frame_num"], gt_sequence, gt_boxes
+ )
+
+ metadata_or_bbs, metrics, motion_detected = self.pipeline_fn(
+ safe_frame,
+ frame_details["frame_num"],
+ stat_start_time,
+ calculated_clip_id,
+ gt_boxes=target_boxes_array if gt_enabled else None,
+ read_frame_only=read_frame_only,
+ )
+
+ num_objs += len(metadata_or_bbs.keys())
+ total_written_frames += 1
+
+ # 3. FRAME POST-PROCESSING & COMPONENT METRICS ----------------
+ frame_end_time = time.perf_counter()
+ total_run_pipelinefn += frame_end_time - run_pipelinefn_start
+ last_loop_cycle_timestamp = frame_end_time
+
+ frame_loop_latency = (frame_end_time - frame_start_time) * 1000.0
+ total_frame_loop_latency += frame_loop_latency
+
+ active_processing_overhead = frame_loop_latency - (
+ frame_details["reader_time"] * 1000.0
+ )
+ total_active_processing_overhead += active_processing_overhead
+
+ if gt_enabled and motion_detected:
+ self.update_eval_stats(metadata_or_bbs, target_boxes_array)
+
+ if metrics != {}:
+ num_crops = (
+ len(metrics["bbs"]) if metrics.get("bbs") is not None else 0
+ )
+ self.crops_per_frame_list.append(num_crops)
+
+ density = metrics.get(
+ "batch_density", 0 if self.config.sf_enabled else 1
+ )
+ self.component_stats["batch_sizes"].append(density)
+ self.component_stats["sf"].append(metrics.get("sf_time", 0.0))
+ self.component_stats["det"].append(metrics.get("det_time", 0.0))
+
+ # Asynchronous GPU coverage collection (NO per-frame GPU stall)
+ if self.config.sf_enabled and metrics.get("bbs") is not None:
+ # cov = self.calculate_unique_coverage(metrics["bbs"])
+ # if hasattr(cov, "item"):
+ # coverage_percentages.append(float(cov.item()))
+ # else:
+ # coverage_percentages.append(float(cov))
+ raw_bbs = metrics["bbs"]
+ coverage_percentages.append(
+ raw_bbs.clone() if torch.is_tensor(raw_bbs) else raw_bbs
+ )
+
+ # Clean temporary loop variables
+ del safe_frame
+ if "metadata_or_bbs" in locals():
+ del metadata_or_bbs
+
+ if total_written_frames % (10 * int(self.target_fps)) == 0:
+ main_app_logger.info(
+ f"Captured {total_written_frames}/{self.max_target_frames} frames..."
+ )
+
+ # if self.frame_count % 100 == 0:
+ # gc.collect()
+
+ except Exception as e:
+ if not self.active or not getattr(self, "prefetch_active", True):
+ break
+ main_app_logger.info(
+ f"[CRITICAL PIPELINE ERROR] Crash on frame: {repr(e)}"
+ )
+ traceback.print_exc()
+ raise e
+
+ # =====================================================================
+ # POST-LOOP BENCHMARK CALCULATIONS (Single Batch Reduction at Teardown)
+ # =====================================================================
+ pipeline_end_time = time.perf_counter()
+ total_pipeline_ms = float((pipeline_end_time - pipeline_start_time) * 1000.0)
+
+ if hasattr(self, "async_writer") and self.async_writer is not None:
+ self.async_writer.release()
+
+ main_app_logger.info(
+ f"Execution Finished. Total Output Frames Written: {total_written_frames}"
+ )
+
+ computed_eval_metrics = None
+ if gt_enabled and hasattr(self, "evaluator"):
+ computed_eval_metrics = self.evaluator.compute_final_metrics()
+
+ if hasattr(self.__class__, "benchmarks"):
+ print(f"SELF.__CLASS__.BENCHMARKS: {self.__class__.benchmarks}")
+ elif hasattr(self, "benchmarks"):
+ print(f"SELF.BENCHMARKS: {self.benchmarks}")
+
+ self._finalize_benchmarks(
+ num_objs=num_objs,
+ total_written_frames=total_written_frames,
+ total_pipeline_ms=total_pipeline_ms,
+ real_world_latency_ms=total_pipeline_ms,
+ total_sf_pipeline_ms=float(
+ sum(self.component_stats.get("sf", [0.0]))
+ + sum(self.component_stats.get("det", [0.0]))
+ ),
+ frame_loop_latency_ms=total_frame_loop_latency
+ / max(1, total_written_frames),
+ coverage_percentages=coverage_percentages,
+ total_read_time=total_read_time,
+ total_queue_saturation=total_queue_saturation,
+ queue_saturation_history=queue_saturation_history,
+ computed_eval_metrics=computed_eval_metrics,
+ )
+
+ if self.device_input == "cuda" and torch.cuda.is_available():
+ torch.cuda.synchronize()
+ torch.cuda.empty_cache()
+ if hasattr(torch.cuda, "ipc_collect"):
+ torch.cuda.ipc_collect()
+
+ # gc.collect()
+ if profiler is not None:
+ profiler.stop()
+
+ self.stop()
+
+ # CLEANUP --------------------------------------------
+
+ def clean_up_tensors_and_arrays(self):
+ main_app_logger.info("[CLEANUP] Safely stripping active runtime arrays ...")
+
+ # 1. Force disable tracking states to stop graph allocation recursion
+ # torch.set_grad_enabled(False)
+
+ # if hasattr(torch, "jit") and hasattr(torch.jit, "_builtins"):
+ # if isinstance(torch.jit._builtins, dict):
+ # torch.jit._sbuiltins.clear()
+ # else:
+ # torch.jit._builtins = {}
+
+ # 2. Collect references safely using strict type checking
+ # This completely avoids pulling unmanaged proxy objects from the heap
+ all_live_objects = gc.get_objects()
+
+ target_tensors = []
+ target_arrays = []
+
+ for obj in all_live_objects:
+ try:
+ obj_type = type(obj)
+ if isinstance(obj, types.FrameType):
+ frame_info = inspect.getframeinfo(obj)
+ if (
+ "openvino" in frame_info.filename
+ or "openvino.py" in frame_info.filename
+ ):
+ # Clear the frame's local variable dictionary to break circular references
+ obj.f_locals.clear()
+
+ # Check for concrete types to prevent triggering proxy __getattr__ hooks
+ if obj_type is torch.Tensor:
+ target_tensors.append(obj)
+ elif obj_type is np.ndarray:
+ # Explicitly guard size checks to keep it completely stable
+ if obj.base is None and obj.ndim > 0:
+ target_arrays.append(obj)
+ except Exception:
+ pass
+
+ # 3. Truncate discovered references in-place without triggering deletions
+ reclaimed_tensors = 0
+ for tensor in target_tensors:
+ try:
+ # Truncate raw storage footprint safely
+ tensor.data = torch.empty(0, device=self.device_input)
+ reclaimed_tensors += 1
+ except Exception:
+ pass
+
+ reclaimed_arrays = 0
+ for arr in target_arrays:
+ try:
+ # Shrink writeable numpy arrays down to 0 bytes safely
+ if arr.flags.writeable:
+ arr.resize((0,), refcheck=False)
+ reclaimed_arrays += 1
+ except (ValueError, SystemError):
+ pass
+
+ main_app_logger.info(
+ f"[CLEANUP] Reclaimed {reclaimed_tensors} tensors and {reclaimed_arrays} arrays safely."
+ )
+
+ # Clean local registers immediately
+ all_live_objects = None
+ target_tensors = None
+ target_arrays = None
+
+ # Clear out low-level traceback registries if active
+ # if hasattr(sys, "exc_info"):
+ if hasattr(sys, "exc_clear"):
+ sys.exc_clear() # if hasattr(sys, "exc_clear") else None
+
+ # Walk up the execution frame tree and clear out f_locals references
+ try:
+ frame_cursor = inspect.currentframe()
+ while frame_cursor:
+ # Empty out active scopes to force unlinking of trapped closures
+ frame_cursor.f_locals.clear()
+ frame_cursor = frame_cursor.f_back
+ except Exception:
+ pass
+ finally:
+ # Prevent the frame cursor property itself from pinning the block
+ del frame_cursor
+ gc.collect()
+
+ def execute_teardown(self):
+ if hasattr(self, "baseline_before_start"):
+ setattr(self, "baseline_before_start", None)
+
+ if hasattr(self, "evaluator") and self.evaluator is not None:
+ try:
+ # If your evaluator class tracks raw coordinate arrays in a list/dict,
+ # force-clear them to allow Python to free up the unmanaged memory:
+ if hasattr(self.evaluator, "history"):
+ self.evaluator.history.clear()
+ if hasattr(self.evaluator, "all_predictions"):
+ self.evaluator.all_predictions.clear()
+ except Exception:
+ pass
+ self.evaluator = None
+
+ if hasattr(self, "component_stats") and self.component_stats:
+ self.component_stats.clear()
+
+ release_native_linux_heap()
+
+ # main_app_logger.info(
+ # "[TEST HARNESS] Test Teardown complete. Releasing session cleanly.\n",
+ # ,
+ # )
+
+ # HELPER FUNCTIONS --------------------------------------------
+ def _finalize_benchmarks_v1(
+ self,
+ num_objs,
+ total_written_frames,
+ total_pipeline_ms,
+ real_world_latency_ms,
+ total_sf_pipeline_ms,
+ frame_loop_latency_ms,
+ coverage_percentages,
+ total_read_time,
+ total_queue_saturation,
+ queue_saturation_history,
+ computed_eval_metrics=None,
+ ):
+ """Aggregates metrics and adds them to the results list."""
+ # num_objs = len(all_metadata.keys())
+ sf_enabled = self.config.sf_enabled
+ stat_frame_count = self.stat_frame_count
+ stat_fps = self.stat_fps
+ total_frames = self.abs_frame_num # self.reader.total_input_frames
+
+ h2d_label = (
+ "Pinned H2D Transfer (PCIe DMA)"
+ if self.is_cuda
+ else "Data Preparation Overhead"
+ )
+ # d2h_label = (
+ # "D2H Transfer (PCIe Download)"
+ # if self.is_cuda
+ # else "Array Extraction Overhead"
+ # )
+
+ total_pipeline_s = total_pipeline_ms / 1000.0 # Full pipeline
+
+ total_real_pipeline_s = (
+ real_world_latency_ms / 1000.0
+ ) # Pipeline until last frame sent to writer
+ real_est_fps = (
+ self.frame_count_target / total_real_pipeline_s
+ if total_real_pipeline_s > 0
+ else 0
+ )
+
+ # if total_written_frames > 0:
+ duration_s = (
+ total_frames / self.reader.input_fps if self.reader.input_fps > 0 else 0
+ )
+ out_duration_s = (
+ total_written_frames / self.reader.target_fps
+ if self.reader.target_fps > 0
+ else 0
+ )
+ # main_app_logger.info(
+ # f"Expected duration: {self.duration_target:.2f} stream_duration_s: {duration_s:.2f} output_duration_s: {out_duration_s:.2f}"
+ # )
+
+ h2d_display_label = f"Reader {h2d_label}:"
+ avg_reader_ms = (total_read_time / total_written_frames) * 1000
+ # avg_resize_ms = (total_resize_time / total_written_frames)*1000
+ # avg_write_ms = (total_disk_write_overhead / total_written_frames)*1000
+ target_frame_fps = total_written_frames / total_pipeline_s
+ all_frame_fps = total_frames / total_pipeline_s
+ avg_copy_ms = (self.reader.total_shm_copy_time / total_written_frames) * 1000
+ # avg_blocked_ms = (
+ # self.reader.total_queue_wait_time / total_written_frames
+ # ) * 1000
+
+ # Avg Blocked MS tracks your downstream GIL blocks and serialization stalls
+ avg_blocked_ms = (
+ sum(self.component_stats.get("queue_blocked", [0.0]))
+ / max(1, len(self.component_stats.get("queue_blocked", [])))
+ if self.component_stats.get("queue_blocked")
+ else 0.0
+ )
+
+ avg_saturation = (total_queue_saturation / total_written_frames) * 100
+ peak_saturation = (
+ max(queue_saturation_history) * 100 if queue_saturation_history else 0
+ )
+ target_match_ratio = (
+ total_written_frames / (total_pipeline_s * self.reader.target_fps)
+ ) * 100
+
+ frame_drop = max(
+ 0.0,
+ (1.0 - (total_written_frames / max(1, total_frames))) * 100.0,
+ )
+ serial_pressure = (avg_blocked_ms / max(0.1, frame_loop_latency_ms)) * 100.0
+ evac_vel = target_frame_fps / max(0.1, all_frame_fps)
+
+ # Isolate the ratio of sequential loop stalls against full execution bounds
+ backpressure_index = (avg_blocked_ms / max(0.1, frame_loop_latency_ms)) * 100.0
+
+ if self.is_cuda:
+ avg_h2d_ms = (self.reader.total_h2d_time / total_written_frames) * 1000
+ avg_prep_ms = 0.0
+ else:
+ avg_prep_ms = (self.reader.total_h2d_time / total_written_frames) * 1000
+ avg_h2d_ms = 0.0
+
+ # Calculate structural frame size payload in Gigabytes (H * W * Channels)
+ frame_size_gb = (self.reader.frame_height * self.reader.frame_width * 3) / (
+ 1024**3
+ )
+
+ # 1. Compute Peak H2D PCIe Throughput
+ avg_h2d_s = (
+ (
+ sum(self.component_stats["dma_upload"])
+ / len(self.component_stats["dma_upload"])
+ )
+ / 1000.0
+ if self.component_stats["dma_upload"]
+ else 0
+ )
+ if self.is_cuda and avg_h2d_s > 0.00001:
+ self.pcie_throughput_gbps = frame_size_gb / avg_h2d_s
+ else:
+ self.pcie_throughput_gbps = (
+ 0.0 # Safe fallback baseline for non-PCIe pipelines (CPU)
+ )
+
+ # Calculate bus saturation relative to standard Gen4 x16 baseline ceilings (31.5 GB/s)
+ # Scale to an index percentage safely
+ pcie_bus_saturation = (self.pcie_throughput_gbps / 31.5) * 100.0
+
+ # 2. Extract PyTorch VRAM Fragmentation Delta Metrics
+ if self.is_cuda and torch.cuda.is_available():
+ peak_alloc = torch.cuda.max_memory_allocated(0) / 1024**2
+ peak_reserved = torch.cuda.max_memory_reserved(0) / 1024**2
+ self.vram_efficiency = (
+ (peak_alloc / peak_reserved * 100.0) if peak_reserved > 0 else 100.0
+ )
+ else:
+ self.vram_efficiency = (
+ 0.0 # Clean zero baseline representation for CPU runs
+ )
+
+ # Calculate averages for component breakdowns
+ # avg_reader_ms = (total_read_time / total_frames) * 1000
+ # avg_reader_ms = (total_read_time / self.frame_count_target) * 1000
+ avg_sf = (
+ sum(self.component_stats["sf"]) / len(self.component_stats["sf"])
+ if self.component_stats["sf"]
+ else 0
+ )
+ avg_roi = (
+ sum(self.component_stats["roi"]) / len(self.component_stats["roi"])
+ if self.component_stats["roi"]
+ else 0
+ )
+ avg_det = (
+ sum(self.component_stats["det"]) / len(self.component_stats["det"])
+ if self.component_stats["det"]
+ else 0
+ )
+
+ # Total Latency Sum (SF + ROI + DET)
+ det_latency_ms = avg_sf + avg_roi + avg_det
+ det_latency_s = det_latency_ms / 1000.0
+ det_est_fps = (total_written_frames / det_latency_s) if det_latency_s > 0 else 0
+
+ # Track model workloads against preprocessing overheads (sf_time + roi_time)
+ preprocessing_overhead = avg_sf + avg_roi
+ model_cost_density = avg_det / max(0.1, preprocessing_overhead)
+
+ loop_overhead = max(
+ 0.0, 100.0 - ((det_latency_ms / max(1.0, frame_loop_latency_ms)) * 100.0)
+ )
+ # avg_cov = (
+ # sum(coverage_percentages) / len(coverage_percentages)
+ # if coverage_percentages
+ # else (100.0 if not sf_enabled else 0)
+ # )
+ # peak_cov = max(coverage_percentages) if coverage_percentages else 0.0
+
+ if coverage_percentages:
+ if torch.is_tensor(coverage_percentages[0]):
+ cov_stack = torch.stack(coverage_percentages)
+ avg_cov = cov_stack.mean().item()
+ peak_cov = cov_stack.max().item()
+ else:
+ avg_cov = sum(coverage_percentages) / len(coverage_percentages)
+ peak_cov = max(coverage_percentages) if coverage_percentages else 0.0
+ else:
+ avg_cov = 100.0 if not sf_enabled else 0.0
+ peak_cov = 0.0
+ # If coverage_percentages contains un-synchronized GPU tensors, reduce them all at once!
+ # if coverage_percentages and isinstance(coverage_percentages[0], torch.Tensor):
+ # # Stack all frame elements into a single contiguous GPU tensor block
+ # coverage_tensor_stack = torch.stack(coverage_percentages)
+
+ # # Calculate reductions fast on the device without stepping back onto the CPU
+ # avg_cov = torch.mean(coverage_tensor_stack).item()
+ # peak_cov = torch.max(coverage_tensor_stack).item()
+
+ # # Safe fallback: Convert the tracking list back into plain floats for standard reporting
+ # coverage_percentages = coverage_tensor_stack.cpu().tolist()
+ # else:
+ # # Standard fallback if the script is running on a CPU target pass
+ # avg_cov = sum(coverage_percentages) / max(1, len(coverage_percentages))
+ # peak_cov = max(coverage_percentages) if coverage_percentages else 0.0
+
+ avg_crops = (
+ sum(self.crops_per_frame_list) / len(self.crops_per_frame_list)
+ if self.crops_per_frame_list
+ else 0
+ )
+
+ # Calculate how often we hit a high-motion cap (e.g., 20 crops)
+ capped_frames = sum(1 for c in self.crops_per_frame_list if c >= 20)
+ cap_rate = (
+ (capped_frames / len(self.crops_per_frame_list)) * 100
+ if self.crops_per_frame_list
+ else 0
+ )
+
+ # avg_queue = (
+ # sum(self.component_stats["queue_blocked"])
+ # / len(self.component_stats["queue_blocked"])
+ # if self.component_stats["queue_blocked"]
+ # else 0
+ # )
+ # avg_batch = (
+ # sum(self.component_stats["batch_sizes"])
+ # / len(self.component_stats["batch_sizes"])
+ # if self.component_stats["batch_sizes"]
+ # else 0
+ # )
+
+ backlog_list = self.component_stats.get("thread_backlog", [])
+ avg_backlog = (
+ sum(backlog_list) / len(backlog_list) if len(backlog_list) > 0 else 0.0
+ )
+
+ main_app_logger.info(
+ "=" * 60,
+ )
+ main_app_logger.info(
+ f" FULLY-OPTIMIZED ASYNC STAGE ({self.config.DEVICE}) PIPELINE BREAKDOWN "
+ )
+ main_app_logger.info(
+ "=" * 60,
+ )
+ main_app_logger.info(
+ f"Total Output Frames Written: {total_written_frames}",
+ )
+ main_app_logger.info(
+ f"Total Pipeline Execution Time: {total_pipeline_s:.4f} seconds",
+ )
+ main_app_logger.info(
+ f"Overall Processing Speed: {target_frame_fps:.2f} FPS",
+ )
+ main_app_logger.info(
+ "=" * 60,
+ )
+ main_app_logger.info(
+ "MAIN CONSUMER LOOP TIMELINE (SEQUENTIAL OVERHEAD):",
+ )
+ main_app_logger.info(
+ f" 1. Shared Memory Copy to Host: {avg_copy_ms:6.2f} ms"
+ )
+ main_app_logger.info(
+ f" 2. GIL / Downstream Queue Serialization Stalls: {avg_blocked_ms:6.2f} ms"
+ )
+ main_app_logger.info(
+ f" 3. Pure Video Frame File-Ingestion Read: {avg_reader_ms:6.2f} ms"
+ )
+ main_app_logger.info(
+ "=" * 60,
+ )
+ main_app_logger.info(
+ "BACKGROUND INGESTION HEALTH (ASYNC METRICS):",
+ )
+ main_app_logger.info(
+ f" A. Inbound Stream Decode Speed: {all_frame_fps:6.2f} FPS"
+ )
+ main_app_logger.info(
+ f" B. {h2d_display_label:<42} {avg_h2d_ms:6.2f} ms",
+ )
+ main_app_logger.info(
+ f" C. Consumer Queue Saturation Density: {avg_saturation:6.1f}%"
+ )
+ main_app_logger.info(
+ f" D. Peak PCIe Upload Throughput: {self.pcie_throughput_gbps:6.2f} GB/s"
+ )
+ main_app_logger.info(
+ f" E. Active Thread Pool Work Backlog: {avg_backlog:6.1f} tasks"
+ )
+ main_app_logger.info(
+ f" F. VRAM Hardware Memory Efficiency (Cache): {self.vram_efficiency:6.1f}%"
+ )
+ main_app_logger.info(
+ "=" * 60,
+ )
+
+ # EXPANDED PERFORMANCE INSIGHTS AND DIAGNOSTICS
+ main_app_logger.info("PIPELINE HEALTH & BEHAVIOR INSIGHTS:")
+ if avg_saturation > 80.0:
+ pipeline_status = "CHOKED"
+ main_app_logger.info(
+ f" • Status: \033[91m🔴 {pipeline_status}\033[0m (Downstream logic cannot keep up with inbound stream speed)"
+ )
+ elif avg_saturation > 40.0:
+ pipeline_status = "BALANCED"
+ main_app_logger.info(
+ f" • Status: \033[93m🟡 {pipeline_status}\033[0m (Queue buffer is actively pacing consumer workloads)"
+ )
+ else:
+ pipeline_status = "IDLE/LIGHT"
+ main_app_logger.info(
+ f" • Status: \033[92m🟢 {pipeline_status}\033[0m (Consumer loop finishes ahead of background ingestion clock)"
+ )
+
+ main_app_logger.info(
+ f" • Peak Queue Fullness Reached: {peak_saturation:.1f}%"
+ )
+ # Calculate stream drop indicator
+ main_app_logger.info(
+ f" • Targeted Pacing Delivery Accuracy: {target_match_ratio:.1f}%"
+ )
+ main_app_logger.info(
+ f" • Downstream Serialization Pressure Index: {serial_pressure}"
+ )
+ main_app_logger.info(
+ f" • Core Queue Evacuation Velocity Rank: {evac_vel}",
+ )
+ main_app_logger.info(
+ f" • Hardware Compute Backpressure Index: {backpressure_index:6.2f}%"
+ )
+ main_app_logger.info(
+ f" • Preprocessing to Inference Cost Density: {model_cost_density:6.2f}x"
+ )
+ main_app_logger.info(
+ f" • Physical PCIe Gen4 Bus Saturation Level: {pcie_bus_saturation:6.2f}%"
+ )
+ main_app_logger.info(
+ "=" * 60,
+ )
+
+ result_dict = {
+ "Test Name": self._testMethodName,
+ "Detection Type": self.config.DETECTION_TYPE,
+ "Device": self.config.DEVICE,
+ "Smart Filtering": "Enabled" if sf_enabled else "Disabled",
+ "Video": self.source, # self.name?
+ "Video FPS": f"{self.reader.input_fps:.2f}",
+ "Video Original Duration (s)": f"{duration_s:.4f}",
+ "Video Frames": total_frames,
+ # TARGET OVERVIEW
+ "Target FPS": f"{self.reader.target_fps:.2f}",
+ "Output Duration (s)": f"{out_duration_s:.4f}",
+ "Output Frames": total_written_frames,
+ }
+
+ if computed_eval_metrics:
+ result_dict.update(
+ {
+ # EVAL OVERVIEW
+ "Max_Recall_IoU_10": computed_eval_metrics["Max_Recall_IoU_10"],
+ "mAP_10": computed_eval_metrics["mAP_10"],
+ "mAP_50": computed_eval_metrics["mAP_50"],
+ "mAP_75": computed_eval_metrics["mAP_75"],
+ "mAP_10_95": computed_eval_metrics["mAP_10_95"],
+ "Precision_10": computed_eval_metrics["Precision_10"],
+ "Recall_10": computed_eval_metrics["Recall_10"],
+ "F1_Score_10": computed_eval_metrics["F1_Score_10"],
+ "Precision_50": computed_eval_metrics["Precision_50"],
+ "Recall_50": computed_eval_metrics["Recall_50"],
+ "F1_Score_50": computed_eval_metrics["F1_Score_50"],
+ }
+ )
+
+ result_dict.update(
+ {
+ # PIPELINE OVERVIEW
+ "Pipeline Latency (s)": f"{total_pipeline_s:.2f}",
+ "Pipeline FPS (Video frames)": f"{all_frame_fps:.2f}",
+ "Pipeline FPS (Target frames)": f"{target_frame_fps:.2f}",
+ "Real Pipeline Latency (s)": f"{total_real_pipeline_s:.2f}",
+ "Real Pipeline FPS (Target frames)": f"{real_est_fps:.2f}",
+ # MAIN CONSUMER LOOP TIMELINE (SEQUENTIAL OVERHEAD)
+ "Avg Frame Reading (ms)": f"{avg_reader_ms:.2f}",
+ # "Avg Resize (ms)": f"{avg_resize_ms:.2f}",
+ # "Avg Writer Async Handoff (ms)": f"{avg_write_ms:.2f}",
+ # BACKGROUND INGESTION HEALTH (ASYNC METRICS)
+ "Avg Host Copy (ms)": f"{avg_copy_ms:.2f}",
+ "Avg DMA Upload (ms)": f"{avg_h2d_ms:.2f}",
+ "Avg Data Prep Overhead (ms)": f"{avg_prep_ms:.2f}",
+ "Avg Queue Blocked (ms)": f"{avg_blocked_ms:.2f}",
+ "Avg Queue Saturation (%)": f"{avg_saturation:.2f}",
+ # Framework Overhead Leak Tracking (Measures loop time spent outside raw AI inference model blocks)
+ "Loop Overhead %": f"{loop_overhead:.2f}%",
+ # Inbound Stream Frame Drop Rate (Identifies if background threads are dropping packets due to I/O)
+ "Inbound Frame Drop %": f"{frame_drop:.2f}%",
+ # Serialization Pressure Index (Tracks what percentage of your frame time is lost to thread stalls)
+ "Serialization Pressure %": f"{serial_pressure:.2f}%",
+ # Queue Evacuation Velocity Coefficient (Values > 1.0 mean your consumer loop drains frames faster than ingestion)
+ "Evacuation Velocity": f"{evac_vel:.2f}x",
+ "Compute Backpressure Index": f"{backpressure_index:.2f}%",
+ "Model Cost Density Ratio": f"{model_cost_density:.2f}x",
+ "PCIe Bus Saturation %": f"{pcie_bus_saturation:.2f}%",
+ # PIPELINE HEALTH & BEHAVIOR INSIGHTS
+ "Pipeline Status": pipeline_status,
+ "Peak Queue Fullness (%)": f"{peak_saturation:.2f}",
+ "Targeted Pacing Delivery (%)": f"{target_match_ratio:.2f}",
+ # DETECTION PIPELINE (Should B NULL)
+ "SF/Detection Latency (s)": f"{det_latency_s:.2f}",
+ "SF/Detection FPS": f"{det_est_fps:.2f}",
+ "Avg SF (ms)": f"{avg_sf:.2f}",
+ "Avg ROI (ms)": f"{avg_roi:.2f}",
+ "Avg Obj. Detection (ms)": f"{avg_det:.2f}",
+ "Total Breakdown Sum (ms)": f"{det_latency_ms:.2f}",
+ "Avg Area Coverage %": f"{avg_cov:.2f}%",
+ "Peak Area Coverage %": f"{peak_cov:.2f}%",
+ "Avg Crops/Frame": f"{avg_crops:.1f}",
+ # "Frames w/o ROIs": self.no_roi_frame_cnt,
+ "Crop Cap Rate (>20)": f"{cap_rate:.1f}%",
+ "Objects Detected": num_objs,
+ # TIming to display frame (read to after send to render queue)
+ "Display Latency (s)": f"{self.elapsed_display_time:.2f}",
+ "Display Frames": stat_frame_count,
+ "Display FPS": f"{stat_fps:.2f}",
+ "PCIe Bandwidth Throughput (GB/s)": f"{self.pcie_throughput_gbps:.2f}",
+ "Avg Async Thread Pool Backlog": f"{sum(self.component_stats.get('thread_backlog', [0])) / len(self.component_stats.get('thread_backlog', [1])):.1f}",
+ "VRAM Allocation Efficiency (%)": f"{self.vram_efficiency:.1f}%",
+ }
+ )
+ self.__class__.benchmarks.append(result_dict)
+
+ main_app_logger.info(
+ f"[{self._testMethodName}] Pipeline Latency (s): {total_pipeline_s:.2f} sec"
+ )
+ main_app_logger.info(
+ f"[{self._testMethodName}] Pipeline FPS (Target frames): {target_frame_fps:.2f} ({total_written_frames} frames)"
+ )
+ main_app_logger.info(
+ f"[{self._testMethodName}] Pipeline FPS (Video frames): {all_frame_fps:.2f} ({total_frames} frames)"
+ )
+ main_app_logger.info(
+ f"[{self._testMethodName}] SF/Detection FPS (Target frames): {det_est_fps:.2f} ({total_written_frames} frames)"
+ )
+ main_app_logger.info(
+ f"[{self._testMethodName}] Display FPS: {stat_fps:.2f} ({stat_frame_count} frames)"
+ )
+
+ def _finalize_benchmarks(
+ self,
+ num_objs,
+ total_written_frames,
+ total_pipeline_ms,
+ real_world_latency_ms,
+ total_sf_pipeline_ms,
+ frame_loop_latency_ms,
+ coverage_percentages,
+ total_read_time,
+ total_queue_saturation,
+ queue_saturation_history,
+ computed_eval_metrics=None,
+ ):
+ """Aggregates metrics and adds them to the results list."""
+ sf_enabled = self.config.sf_enabled
+ stat_frame_count = self.stat_frame_count
+ stat_fps = self.stat_fps
+ total_frames = self.abs_frame_num
+
+ h2d_label = (
+ "Pinned H2D Transfer (PCIe DMA)"
+ if self.is_cuda
+ else "Data Preparation Overhead"
+ )
+
+ total_pipeline_s = total_pipeline_ms / 1000.0
+ total_real_pipeline_s = real_world_latency_ms / 1000.0
+ real_est_fps = (
+ self.frame_count_target / total_real_pipeline_s
+ if total_real_pipeline_s > 0
+ else 0.0
+ )
+
+ duration_s = (
+ total_frames / self.reader.input_fps if self.reader.input_fps > 0 else 0.0
+ )
+ out_duration_s = (
+ total_written_frames / self.reader.target_fps
+ if self.reader.target_fps > 0
+ else 0.0
+ )
+
+ h2d_display_label = f"Reader {h2d_label}:"
+ avg_reader_ms = (
+ (total_read_time / total_written_frames) * 1000.0
+ if total_written_frames > 0
+ else 0.0
+ )
+ target_frame_fps = (
+ total_written_frames / total_pipeline_s if total_pipeline_s > 0 else 0.0
+ )
+ all_frame_fps = total_frames / total_pipeline_s if total_pipeline_s > 0 else 0.0
+ avg_copy_ms = (
+ (self.reader.total_shm_copy_time / total_written_frames) * 1000.0
+ if total_written_frames > 0
+ else 0.0
+ )
+
+ avg_blocked_ms = (
+ sum(self.component_stats.get("queue_blocked", [0.0]))
+ / max(1, len(self.component_stats.get("queue_blocked", [])))
+ if self.component_stats.get("queue_blocked")
+ else 0.0
+ )
+
+ avg_saturation = (
+ (total_queue_saturation / total_written_frames) * 100.0
+ if total_written_frames > 0
+ else 0.0
+ )
+ peak_saturation = (
+ max(queue_saturation_history) * 100.0 if queue_saturation_history else 0.0
+ )
+ target_match_ratio = (
+ (total_written_frames / (total_pipeline_s * self.reader.target_fps)) * 100.0
+ if (total_pipeline_s > 0 and self.reader.target_fps > 0)
+ else 0.0
+ )
+
+ frame_drop = max(
+ 0.0,
+ (1.0 - (total_written_frames / max(1, total_frames))) * 100.0,
+ )
+ serial_pressure = (avg_blocked_ms / max(0.1, frame_loop_latency_ms)) * 100.0
+ evac_vel = target_frame_fps / max(0.1, all_frame_fps)
+ backpressure_index = (avg_blocked_ms / max(0.1, frame_loop_latency_ms)) * 100.0
+
+ if self.is_cuda:
+ avg_h2d_ms = (
+ (self.reader.total_h2d_time / total_written_frames) * 1000.0
+ if total_written_frames > 0
+ else 0.0
+ )
+ avg_prep_ms = 0.0
+ else:
+ avg_prep_ms = (
+ (self.reader.total_h2d_time / total_written_frames) * 1000.0
+ if total_written_frames > 0
+ else 0.0
+ )
+ avg_h2d_ms = 0.0
+
+ frame_size_gb = (self.reader.frame_height * self.reader.frame_width * 3) / (
+ 1024**3
+ )
+
+ avg_h2d_s = (
+ (
+ sum(self.component_stats["dma_upload"])
+ / len(self.component_stats["dma_upload"])
+ )
+ / 1000.0
+ if self.component_stats.get("dma_upload")
+ else 0.0
+ )
+ if self.is_cuda and avg_h2d_s > 0.00001:
+ self.pcie_throughput_gbps = frame_size_gb / avg_h2d_s
+ else:
+ self.pcie_throughput_gbps = 0.0
+
+ pcie_bus_saturation = (self.pcie_throughput_gbps / 31.5) * 100.0
+
+ if self.is_cuda and torch.cuda.is_available():
+ peak_alloc = torch.cuda.max_memory_allocated(0) / 1024**2
+ peak_reserved = torch.cuda.max_memory_reserved(0) / 1024**2
+ self.vram_efficiency = (
+ (peak_alloc / peak_reserved * 100.0) if peak_reserved > 0 else 100.0
+ )
+ else:
+ self.vram_efficiency = 0.0
+
+ avg_sf = (
+ sum(self.component_stats["sf"]) / len(self.component_stats["sf"])
+ if self.component_stats.get("sf")
+ else 0.0
+ )
+ avg_roi = (
+ sum(self.component_stats["roi"]) / len(self.component_stats["roi"])
+ if self.component_stats.get("roi")
+ else 0.0
+ )
+ avg_det = (
+ sum(self.component_stats["det"]) / len(self.component_stats["det"])
+ if self.component_stats.get("det")
+ else 0.0
+ )
+
+ det_latency_ms = avg_sf + avg_roi + avg_det
+ det_latency_s = det_latency_ms / 1000.0
+ det_est_fps = (
+ (total_written_frames / det_latency_s) if det_latency_s > 0 else 0.0
+ )
+
+ preprocessing_overhead = avg_sf + avg_roi
+ model_cost_density = avg_det / max(0.1, preprocessing_overhead)
+
+ loop_overhead = max(
+ 0.0, 100.0 - ((det_latency_ms / max(1.0, frame_loop_latency_ms)) * 100.0)
+ )
+
+ # Batch GPU Reduction for Coverage Tensors
+ if coverage_percentages:
+ cov_results = []
+ for bbs in coverage_percentages:
+ if bbs is not None and len(bbs) > 0:
+ cov_results.append(self.calculate_unique_coverage(bbs))
+ else:
+ cov_results.append(0.0)
+
+ avg_cov = sum(cov_results) / max(1, len(cov_results))
+ peak_cov = max(cov_results)
+ else:
+ avg_cov = 100.0 if not sf_enabled else 0.0
+ peak_cov = 0.0
+
+ avg_crops = (
+ sum(self.crops_per_frame_list) / len(self.crops_per_frame_list)
+ if self.crops_per_frame_list
+ else 0.0
+ )
+
+ capped_frames = sum(1 for c in self.crops_per_frame_list if c >= 20)
+ cap_rate = (
+ (capped_frames / len(self.crops_per_frame_list)) * 100.0
+ if self.crops_per_frame_list
+ else 0.0
+ )
+
+ backlog_list = self.component_stats.get("thread_backlog", [])
+ avg_backlog = (
+ sum(backlog_list) / len(backlog_list) if len(backlog_list) > 0 else 0.0
+ )
+
+ # Formatted Performance Breakdown Logging
+ main_app_logger.info("=" * 60)
+ main_app_logger.info(
+ f" FULLY-OPTIMIZED ASYNC STAGE ({self.config.DEVICE}) PIPELINE BREAKDOWN "
+ )
+ main_app_logger.info("=" * 60)
+ main_app_logger.info(f"Total Output Frames Written: {total_written_frames}")
+ main_app_logger.info(
+ f"Total Pipeline Execution Time: {total_pipeline_s:.4f} seconds"
+ )
+ main_app_logger.info(
+ f"Overall Processing Speed: {target_frame_fps:.2f} FPS"
+ )
+ main_app_logger.info("=" * 60)
+ main_app_logger.info("MAIN CONSUMER LOOP TIMELINE (SEQUENTIAL OVERHEAD):")
+ main_app_logger.info(
+ f" 1. Shared Memory Copy to Host: {avg_copy_ms:6.2f} ms"
+ )
+ main_app_logger.info(
+ f" 2. GIL / Downstream Queue Serialization Stalls: {avg_blocked_ms:6.2f} ms"
+ )
+ main_app_logger.info(
+ f" 3. Pure Video Frame File-Ingestion Read: {avg_reader_ms:6.2f} ms"
+ )
+ main_app_logger.info("=" * 60)
+ main_app_logger.info("BACKGROUND INGESTION HEALTH (ASYNC METRICS):")
+ main_app_logger.info(
+ f" A. Inbound Stream Decode Speed: {all_frame_fps:6.2f} FPS"
+ )
+ main_app_logger.info(f" B. {h2d_display_label:<42} {avg_h2d_ms:6.2f} ms")
+ main_app_logger.info(
+ f" C. Consumer Queue Saturation Density: {avg_saturation:6.1f}%"
+ )
+ main_app_logger.info(
+ f" D. Peak PCIe Upload Throughput: {self.pcie_throughput_gbps:6.2f} GB/s"
+ )
+ main_app_logger.info(
+ f" E. Active Thread Pool Work Backlog: {avg_backlog:6.1f} tasks"
+ )
+ main_app_logger.info(
+ f" F. VRAM Hardware Memory Efficiency (Cache): {self.vram_efficiency:6.1f}%"
+ )
+ main_app_logger.info("=" * 60)
+
+ main_app_logger.info("PIPELINE HEALTH & BEHAVIOR INSIGHTS:")
+ if avg_saturation > 80.0:
+ pipeline_status = "CHOKED"
+ main_app_logger.info(
+ f" • Status: \033[91m🔴 {pipeline_status}\033[0m (Downstream logic cannot keep up with inbound stream speed)"
+ )
+ elif avg_saturation > 40.0:
+ pipeline_status = "BALANCED"
+ main_app_logger.info(
+ f" • Status: \033[93m🟡 {pipeline_status}\033[0m (Queue buffer is actively pacing consumer workloads)"
+ )
+ else:
+ pipeline_status = "IDLE/LIGHT"
+ main_app_logger.info(
+ f" • Status: \033[92m🟢 {pipeline_status}\033[0m (Consumer loop finishes ahead of background ingestion clock)"
+ )
+
+ main_app_logger.info(
+ f" • Peak Queue Fullness Reached: {peak_saturation:.1f}%"
+ )
+ main_app_logger.info(
+ f" • Targeted Pacing Delivery Accuracy: {target_match_ratio:.1f}%"
+ )
+ main_app_logger.info(
+ f" • Downstream Serialization Pressure Index: {serial_pressure:.4f}"
+ )
+ main_app_logger.info(
+ f" • Core Queue Evacuation Velocity Rank: {evac_vel:.4f}"
+ )
+ main_app_logger.info(
+ f" • Hardware Compute Backpressure Index: {backpressure_index:6.2f}%"
+ )
+ main_app_logger.info(
+ f" • Preprocessing to Inference Cost Density: {model_cost_density:6.2f}x"
+ )
+ main_app_logger.info(
+ f" • Physical PCIe Gen4 Bus Saturation Level: {pcie_bus_saturation:6.2f}%"
+ )
+ main_app_logger.info("=" * 60)
+
+ result_dict = {
+ "Test Name": self._testMethodName,
+ "Detection Type": self.config.DETECTION_TYPE,
+ "Device": self.config.DEVICE,
+ "Smart Filtering": "Enabled" if sf_enabled else "Disabled",
+ "Video": self.source,
+ "Video FPS": f"{self.reader.input_fps:.2f}",
+ "Video Original Duration (s)": f"{duration_s:.4f}",
+ "Video Frames": total_frames,
+ "Target FPS": f"{self.reader.target_fps:.2f}",
+ "Output Duration (s)": f"{out_duration_s:.4f}",
+ "Output Frames": total_written_frames,
+ }
+
+ if computed_eval_metrics:
+ result_dict.update(
+ {
+ "Max_Recall_IoU_10": computed_eval_metrics.get(
+ "Max_Recall_IoU_10", 0.0
+ ),
+ "mAP_10": computed_eval_metrics.get("mAP_10", 0.0),
+ "mAP_50": computed_eval_metrics.get("mAP_50", 0.0),
+ "mAP_75": computed_eval_metrics.get("mAP_75", 0.0),
+ "mAP_10_95": computed_eval_metrics.get("mAP_10_95", 0.0),
+ "Precision_10": computed_eval_metrics.get("Precision_10", 0.0),
+ "Recall_10": computed_eval_metrics.get("Recall_10", 0.0),
+ "F1_Score_10": computed_eval_metrics.get("F1_Score_10", 0.0),
+ "Precision_50": computed_eval_metrics.get("Precision_50", 0.0),
+ "Recall_50": computed_eval_metrics.get("Recall_50", 0.0),
+ "F1_Score_50": computed_eval_metrics.get("F1_Score_50", 0.0),
+ }
+ )
+
+ result_dict.update(
+ {
+ "Pipeline Latency (s)": f"{total_pipeline_s:.2f}",
+ "Pipeline FPS (Video frames)": f"{all_frame_fps:.2f}",
+ "Pipeline FPS (Target frames)": f"{target_frame_fps:.2f}",
+ "Real Pipeline Latency (s)": f"{total_real_pipeline_s:.2f}",
+ "Real Pipeline FPS (Target frames)": f"{real_est_fps:.2f}",
+ "Avg Frame Reading (ms)": f"{avg_reader_ms:.2f}",
+ "Avg Host Copy (ms)": f"{avg_copy_ms:.2f}",
+ "Avg DMA Upload (ms)": f"{avg_h2d_ms:.2f}",
+ "Avg Data Prep Overhead (ms)": f"{avg_prep_ms:.2f}",
+ "Avg Queue Blocked (ms)": f"{avg_blocked_ms:.2f}",
+ "Avg Queue Saturation (%)": f"{avg_saturation:.2f}",
+ "Loop Overhead %": f"{loop_overhead:.2f}%",
+ "Inbound Frame Drop %": f"{frame_drop:.2f}%",
+ "Serialization Pressure %": f"{serial_pressure:.2f}%",
+ "Evacuation Velocity": f"{evac_vel:.2f}x",
+ "Compute Backpressure Index": f"{backpressure_index:.2f}%",
+ "Model Cost Density Ratio": f"{model_cost_density:.2f}x",
+ "PCIe Bus Saturation %": f"{pcie_bus_saturation:.2f}%",
+ "Pipeline Status": pipeline_status,
+ "Peak Queue Fullness (%)": f"{peak_saturation:.2f}",
+ "Targeted Pacing Delivery (%)": f"{target_match_ratio:.2f}",
+ "SF/Detection Latency (s)": f"{det_latency_s:.2f}",
+ "SF/Detection FPS": f"{det_est_fps:.2f}",
+ "Avg SF (ms)": f"{avg_sf:.2f}",
+ "Avg ROI (ms)": f"{avg_roi:.2f}",
+ "Avg Obj. Detection (ms)": f"{avg_det:.2f}",
+ "Total Breakdown Sum (ms)": f"{det_latency_ms:.2f}",
+ "Avg Area Coverage %": f"{avg_cov:.2f}%",
+ "Peak Area Coverage %": f"{peak_cov:.2f}%",
+ "Avg Crops/Frame": f"{avg_crops:.1f}",
+ "Crop Cap Rate (>20)": f"{cap_rate:.1f}%",
+ "Objects Detected": num_objs,
+ "Display Latency (s)": f"{self.elapsed_display_time:.2f}",
+ "Display Frames": stat_frame_count,
+ "Display FPS": f"{stat_fps:.2f}",
+ "PCIe Bandwidth Throughput (GB/s)": f"{self.pcie_throughput_gbps:.2f}",
+ "Avg Async Thread Pool Backlog": f"{sum(self.component_stats.get('thread_backlog', [0])) / max(1, len(self.component_stats.get('thread_backlog', [1]))):.1f}",
+ "VRAM Allocation Efficiency (%)": f"{self.vram_efficiency:.1f}%",
+ }
+ )
+ self.__class__.benchmarks.append(result_dict)
+
+ main_app_logger.info(
+ f"[{self._testMethodName}] Pipeline Latency (s): {total_pipeline_s:.2f} sec"
+ )
+ main_app_logger.info(
+ f"[{self._testMethodName}] Pipeline FPS (Target frames): {target_frame_fps:.2f} ({total_written_frames} frames)"
+ )
+ main_app_logger.info(
+ f"[{self._testMethodName}] Pipeline FPS (Video frames): {all_frame_fps:.2f} ({total_frames} frames)"
+ )
+ main_app_logger.info(
+ f"[{self._testMethodName}] SF/Detection FPS (Target frames): {det_est_fps:.2f} ({total_written_frames} frames)"
+ )
+ main_app_logger.info(
+ f"[{self._testMethodName}] Display FPS: {stat_fps:.2f} ({stat_frame_count} frames)"
+ )
+
+ def motion2metadata(self, merged, frame_count_target):
+ metadata = {}
+ if merged is not None and merged.size > 0:
+ # merged = merged / self.scales_tensor.cpu().numpy().reshape(-1, 4)
+ merged = merged.div(self.scales_tensor.view(-1, 4))
+
+ # Calculate area for each box: (xmax - xmin) * (ymax - ymin)
+ widths = merged[:, 2] - merged[:, 0]
+ heights = merged[:, 3] - merged[:, 1]
+ areas = widths * heights
+ max_area = np.max(areas) if np.max(areas) > 0 else 1.0
+
+ # Format motion results like detection results for evaluation
+ for i, bb in enumerate(merged):
+ disp_x, disp_y, disp_x2, disp_y2 = bb
+ disp_w = disp_x2 - disp_x
+ disp_h = disp_y2 - disp_y
+
+ if disp_w > 2 and disp_h > 2:
+ obj_id = len(metadata)
+ # num_objs += 1
+ framenum_str = f"{frame_count_target:04d}_{obj_id:04d}"
+ score = areas[i] / max_area
+
+ metadata[framenum_str] = {
+ "frameId": int(frame_count_target),
+ "bbId": framenum_str,
+ "bbox": {
+ "x": int(disp_x),
+ "y": int(disp_y),
+ "height": int(disp_h),
+ "width": int(disp_w),
+ "object": "",
+ "object_det": {
+ "confidence": score,
+ "frameH": int(self.resize_h),
+ "frameW": int(self.resize_w),
+ },
+ },
+ }
+
+ return metadata
+
+ def calculate_unique_coverage_v1(self, merged_boxes, target_w=640, target_h=640):
+ """
+ Pure Loopless Vectorized Coverage Engine.
+ Eliminates sequential Python loop slicing overhead via parallel hardware broadcast registers.
+ """
+ if merged_boxes is None or len(merged_boxes) == 0:
+ return 0.0
+
+ target_device = self.device_input if hasattr(self, "device_input") else "cpu"
+
+ if isinstance(merged_boxes, np.ndarray):
+ if merged_boxes.size == 0:
+ return 0.0
+ boxes_tensor = torch.as_tensor(
+ merged_boxes, device=target_device, dtype=torch.float32
+ )
+ else:
+ if merged_boxes.shape[0] == 0:
+ return 0.0
+ boxes_tensor = merged_boxes.to(device=target_device, dtype=torch.float32)
+
+ # Vectorized scaling projection matrix
+ scale = torch.tensor(
+ [
+ target_w / self.frame_width,
+ target_h / self.frame_height,
+ target_w / self.frame_width,
+ target_h / self.frame_height,
+ ],
+ device=target_device,
+ dtype=torch.float32,
+ )
+
+ coords = (boxes_tensor * scale).long()
+ coords[:, [0, 2]] = coords[:, [0, 2]].clamp(0, target_w)
+ coords[:, [1, 3]] = coords[:, [1, 3]].clamp(0, target_h)
+
+ # Build parallel coordinate meshes on target hardware
+ # grid_y, grid_x = torch.meshgrid(
+ # torch.arange(target_h, device=target_device),
+ # torch.arange(target_w, device=target_device),
+ # indexing="ij"
+ # )
+
+ # Extract coordinate tracking vectors
+ x1 = coords[:, 0].unsqueeze(1).unsqueeze(2)
+ y1 = coords[:, 1].unsqueeze(1).unsqueeze(2)
+ x2 = coords[:, 2].unsqueeze(1).unsqueeze(2)
+ y2 = coords[:, 3].unsqueeze(1).unsqueeze(2)
+
+ # Direct 3D array tensor broadcast reduction sum pass
+ # inside_mask = (grid_x >= x1_v) & (grid_x < x2_v) & (grid_y >= y1_v) & (grid_y < y2_v)
+ inside_boxes = (
+ (self._cached_grid_x >= x1)
+ & (self._cached_grid_x < x2)
+ & (self._cached_grid_y >= y1)
+ & (self._cached_grid_y < y2)
+ )
+ unique_mask = torch.any(inside_boxes, dim=0)
+
+ coverage_percentage = (
+ torch.sum(unique_mask).float().div(target_w * target_h).mul(100.0)
+ )
+ return coverage_percentage.item()
+
+ def calculate_unique_coverage(self, merged_boxes, target_w=640, target_h=640):
+ if merged_boxes is None or len(merged_boxes) == 0:
+ return None
+
+ target_device = self.device_input if hasattr(self, "device_input") else "cpu"
+
+ if isinstance(merged_boxes, np.ndarray):
+ if merged_boxes.size == 0:
+ return None
+ boxes_tensor = torch.as_tensor(
+ merged_boxes, device=target_device, dtype=torch.float32
+ )
+ else:
+ if merged_boxes.shape[0] == 0:
+ return None
+ boxes_tensor = merged_boxes.to(device=target_device, dtype=torch.float32)
+
+ # Scale coordinates to target space
+ scale_x = target_w / self.frame_width
+ scale_y = target_h / self.frame_height
+
+ x1 = (boxes_tensor[:, 0] * scale_x).clamp(0, target_w).long().view(-1, 1, 1)
+ y1 = (boxes_tensor[:, 1] * scale_y).clamp(0, target_h).long().view(-1, 1, 1)
+ x2 = (boxes_tensor[:, 2] * scale_x).clamp(0, target_w).long().view(-1, 1, 1)
+ y2 = (boxes_tensor[:, 3] * scale_y).clamp(0, target_h).long().view(-1, 1, 1)
+
+ # 3D broadcast evaluation
+ inside_boxes = (
+ (self._cached_grid_x >= x1)
+ & (self._cached_grid_x < x2)
+ & (self._cached_grid_y >= y1)
+ & (self._cached_grid_y < y2)
+ )
+ unique_mask = torch.any(inside_boxes, dim=0)
+
+ # Return the 0-dim GPU tensor directly without calling .item()!
+ return torch.sum(unique_mask).float().div(target_w * target_h).mul(100.0)
+
+ def print_active_cpu_tensor_memory(self):
+ """
+ CPU EQUIVALENT TO print_active_gpu_tensor_memory:
+ Scans the global Python garbage collection registries to locate, measure,
+ and map every single live PyTorch tensor actively residing on host RAM.
+ """
+
+ main_app_logger.info(
+ "=" * 60,
+ )
+ main_app_logger.info(
+ "[RAM INVESTIGATOR] Scanning Active CPU Tensors with Sources:",
+ )
+ main_app_logger.info(
+ f"{'ALLOCATED SIZE':<15} | {'TENSOR SHAPE':<25} | {'DATA TYPE':<15}"
+ )
+ main_app_logger.info(
+ "-" * 80,
+ )
+
+ total_cpu_tensor_bytes = 0
+ live_tensor_count = 0
+
+ # Force a shallow sweep to catch immediately orphaned pointers first
+ gc.collect()
+
+ # Iterate over the entire active Python heap
+ for obj in gc.get_objects():
+ try:
+ # Target only genuine PyTorch tensors residing on the host CPU
+ if torch.is_tensor(obj) and not obj.is_cuda and obj.nelement() > 0:
+ live_tensor_count += 1
+
+ # Calculate true memory footprint based on element size and count
+ element_size = obj.element_size()
+ num_elements = obj.nelement()
+ tensor_bytes = num_elements * element_size
+ total_cpu_tensor_bytes += tensor_bytes
+
+ # Filter out tiny scalar metrics to keep logs focused on leaks (e.g., > 10 KB)
+ if tensor_bytes > 10240:
+ size_kb = tensor_bytes / 1024
+ shape_str = str(list(obj.shape))
+ dtype_str = str(obj.dtype).replace("torch.", "")
+
+ main_app_logger.info(
+ f"{size_kb:>10.1f} KiB | {shape_str:<25} | {dtype_str:<15}"
+ )
+ except Exception:
+ # Guard against objects being destroyed concurrently during the scan loop
+ pass
+
+ total_cpu_tensor_mb = total_cpu_tensor_bytes / (1024 * 1024)
+ main_app_logger.info(
+ "-" * 80,
+ )
+ main_app_logger.info(
+ f"Total Live CPU Tensor Count: {live_tensor_count}",
+ )
+ main_app_logger.info(
+ f"Total Live CPU Tensor Memory: {total_cpu_tensor_mb:.2f} MB",
+ )
+ main_app_logger.info(
+ "=" * 60,
+ )
+
+ # def print_active_gpu_tensor_memory(self):
+ # main_app_logger.info(
+ # "=" * 60,
+ # )
+ # main_app_logger.info(
+ # "\033[95m[VRAM INVESTIGATOR] Scanning Active GPU Tensors with Sources:\033[0m"
+ # )
+
+ # # Capture the exact C++ memory registry structure
+ # try:
+ # raw_snapshot = torch.cuda.memory._snapshot()
+ # segments = raw_snapshot.get("segments", [])
+ # except Exception:
+ # segments = []
+ # main_app_logger.info(
+ # "[WARN] Failed parsing native memory context snapshot.",
+ # )
+
+ # # Map raw block storage memory addresses straight to Python frames
+ # addr_to_source = {}
+ # for seg in segments:
+ # for block in seg.get("blocks", []):
+ # if block.get("state") == "active_allocated":
+ # addr = block.get("address")
+ # history = block.get("history", [])
+ # if history:
+ # # Inspect the deepest frame in the allocation stack
+ # frame = history[-1]
+ # filename = frame.get("filename", "Unknown")
+ # lineno = frame.get("line", 0)
+ # func_name = frame.get("name", "unknown_func")
+ # addr_to_source[addr] = f"{filename}:{lineno} ({func_name})"
+
+ # # Scan the heap via GC and resolve actual backing storage layers
+ # leaked_tensors = []
+ # total_detected_bytes = 0
+
+ # for obj in gc.get_objects():
+ # try:
+ # if torch.is_tensor(obj) and obj.is_cuda:
+ # t_bytes = obj.element_size() * obj.nelement()
+ # total_detected_bytes += t_bytes
+ # if t_bytes > 0:
+ # leaked_tensors.append(obj)
+ # except Exception:
+ # pass
+
+ # obj = None # Break tracking register references
+
+ # for i, tensor in enumerate(leaked_tensors):
+ # t_bytes = tensor.element_size() * tensor.nelement()
+
+ # # --- THE FIX: Extract the address of the underlying storage block ---
+ # try:
+ # if hasattr(tensor, "untyped_storage"):
+ # storage_addr = tensor.untyped_storage().data_ptr()
+ # elif hasattr(tensor, "storage") and tensor.storage():
+ # storage_addr = tensor.storage().data_ptr()
+ # else:
+ # storage_addr = tensor.data_ptr()
+ # except Exception:
+ # storage_addr = tensor.data_ptr()
+
+ # # Extract absolute code trace locations from our snapshot dictionary
+ # source_loc = addr_to_source.get(
+ # storage_addr, "Unknown Native C++ Allocation / Model Context"
+ # )
+
+ # main_app_logger.info(
+ # f" > Tensor {i:3d} | Shape: {str(list(tensor.shape)):<18} | "
+ # f"Size: {t_bytes / 1024**2:6.2f} MB | Source: \033[93m{source_loc}\033[0m"
+ # )
+
+ # del leaked_tensors
+ # main_app_logger.info(
+ # f"[VRAM INVESTIGATOR] Total Live Tensor Memory: {total_detected_bytes / 1024**2:.2f} MB"
+ # )
+ # gc.collect()
+ # if (
+ # torch.cuda.is_available()
+ # ): # self.device_input == "cuda" and torch.cuda.is_available():
+ # torch.cuda.synchronize()
+ # torch.cuda.empty_cache() # Flush the pool BEFORE the guard snapshots it
+
+ # def print_active_shared_memory(self):
+ # main_app_logger.info(
+ # "=" * 60,
+ # )
+ # main_app_logger.info(
+ # "\033[96m[SHM INVESTIGATOR] Scanning Active OS Shared Memory Filesystem Tables:\033[0m"
+ # )
+ # try:
+ # shm_dir = Path("/dev/shm")
+ # if shm_dir.exists():
+ # # Extract and inventory all live POSIX memory segments allocated right now
+ # shm_files = [
+ # f
+ # for f in shm_dir.iterdir()
+ # if f.is_file()
+ # and not f.name.startswith("sem.")
+ # and not f.name.startswith("psm")
+ # ]
+ # main_app_logger.info(
+ # f" > Discovered Live OS-Mapped Memory Nodes: {len(shm_files)}"
+ # )
+
+ # for f_path in shm_files:
+ # try:
+ # f_stat = f_path.stat()
+ # size_mb = f_stat.st_size / (1024 * 1024)
+
+ # # Highlight the files using visual color anchors for scannability
+ # main_app_logger.info(
+ # f" ⚠️ \033[93m[ALIVE SHM NODE]\033[0m File: {f_path.name:<25} | Size: {size_mb:7.2f} MB"
+ # )
+ # except Exception:
+ # pass
+ # else:
+ # main_app_logger.info(
+ # " [ERROR] /dev/shm runtime directory is inaccessible on this host context."
+ # )
+ # except Exception as e:
+ # main_app_logger.info(
+ # f" [WARN] Kernel inspection execution pass failed: {e}"
+ # )
+
+ # # gc.collect()
+ # # if torch.cuda.is_available():
+ # # torch.cuda.synchronize()
+ # # torch.cuda.empty_cache() # Flush the pool BEFORE the guard snapshots it
+
+ # def calculate_leaked_memory(
+ # self, device, video_name, start_allocated, start_reserved
+ # ):
+ # _testMethodName = f"{video_name}_{device}"
+ # main_app_logger.info("=" * 60)
+ # max_allowed_leak = 1024 * 1024 # 1MB buffer allowance
+ # msg = "[LEAKAGE INVESTIGATOR] Scanning memory allocations:\n"
+ # if device == "gpu" and torch.cuda.is_available():
+ # # torch.cuda.synchronize()
+
+ # end_allocated = torch.cuda.memory_allocated(0)
+ # end_reserved = torch.cuda.memory_reserved(0)
+
+ # leak_allocated = end_allocated - start_allocated
+ # leak_reserved = end_reserved - start_reserved
+ # # max_allowed_leak = 1024 * 1024 # 1MB buffer allowance
+
+ # if leak_allocated > max_allowed_leak:
+ # msg += (
+ # f"\n🔴 GPU Memory Leak Detected for {_testMethodName}!\n"
+ # f"Check for dangling references or missing 'del' statements\n\n"
+ # )
+ # # else:
+ # # msg = "\n"
+
+ # msg += (
+ # f"\tPre-Setup Allocation: {start_allocated / 1024**2:.2f} MB\n"
+ # f"\tPost-Teardown Allocation: {end_allocated / 1024**2:.2f} MB\n"
+ # f"\tNet Leaked VRAM: {leak_allocated / 1024**2:.2f} MB\n"
+ # f"\tNet Leaked Reserved Blocks: {leak_reserved / 1024**2:.2f} MB"
+ # )
+
+ # # main_app_logger.info(msg, )
+ # else:
+ # process = psutil.Process(os.getpid())
+ # end_rss = process.memory_info().rss
+
+ # # start_allocated and start_reserved must be populated with baseline RSS in each_test_setup
+ # leak_rss = end_rss - start_allocated
+
+ # if leak_rss > max_allowed_leak:
+ # msg += (
+ # f"\n🔴 CPU Memory Leak Detected for {_testMethodName}!\n"
+ # f"Check for dangling references or missing 'del' statements\n\n"
+ # )
+ # # else:
+ # # msg = "\n"
+
+ # msg += (
+ # f" Pre-Setup Host RAM Allocation: {start_allocated / 1024**2:.2f} MB\n"
+ # )
+ # msg += f" Post-Teardown Host RAM Allocation: {end_rss / 1024**2:.2f} MB\n"
+ # msg += f" Net Leaked Host System Memory: {leak_rss / 1024**2:.2f} MB"
+ # main_app_logger.info(msg)
+ # main_app_logger.info("=" * 60)
+
+ # def diagnostic_profiler(self, device, video_name, start_allocated, start_reserved):
+ # _testMethodName = f"{video_name}_{device}"
+ # self.calculate_leaked_memory(
+ # device, video_name, start_allocated, start_reserved
+ # )
+
+ # # main_app_logger.info("=" * 60, )
+ # # main_app_logger.info(
+ # # f"\n\033[95m[DIAGNOSTICS] Starting Automated Leak Analysis for {_testMethodName}...\033[0m",
+ # # ,
+ # # )
+
+ # # 1. Run objgraph to inspect the Python object reference trees before clearing containers
+ # # try:
+ # # main_app_logger.info(
+ # # "\033[94m[DIAGNOSTICS] Python Object Registry Standings:\033[0m",
+ # # )
+ # # objgraph.show_most_common_types(limit=10)
+
+ # # Check if the metrics arrays are pinning references inside memory
+ # # for tracker_attr in ["all_preds", "all_targets"]:
+ # # if hasattr(self, tracker_attr):
+ # # tgt_list = getattr(self, tracker_attr)
+ # # if len(tgt_list) > 0:
+ # # graph_path = f"/tmp/backrefs_{tracker_attr}_{device}.png"
+ # # main_app_logger.info(
+ # # f"\033[93m[WARN] '{tracker_attr}' contains {len(tgt_list)} entries. Generating reference graph to: {graph_path}\033[0m"
+ # # )
+ # # objgraph.show_backrefs(
+ # # [tgt_list], max_depth=3, filename=graph_path
+ # # )
+ # # except ImportError:
+ # # main_app_logger.info(
+ # # "\033[91m[DIAGNOSTICS] 'objgraph' package missing. Skipping reference chain mapping. (pip install objgraph)\033[0m"
+ # # )
+
+ # # 2. Dump PyTorch Memory Snapshot before clearing VRAM caches
+ # if device == "gpu" and torch.cuda.is_available():
+ # try:
+ # # snapshot_path = f"/tmp/vram_leak_profile_{self._testMethodName}.pickle"
+ # snapshot_path = self.output_path.replace(".mp4", "_vram_profile.html")
+ # torch.cuda.memory._dump_snapshot(snapshot_path)
+ # main_app_logger.info(
+ # f"\033[92m[DIAGNOSTICS] VRAM Snapshot Trace generated successfully: {snapshot_path}\033[0m"
+ # )
+ # main_app_logger.info(
+ # "\033[92m--> Upload this file to https://pytorch.org to inspect leak allocation stacks.\033[0m"
+ # )
+
+ # with open(snapshot_path, "rb") as f:
+ # snapshot = pickle.load(f)
+
+ # # Print an HTML visualization path map of the allocations
+ # html_timeline = memory_viz.trace_plot(snapshot)
+ # html_path = snapshot_path.replace(".pickle", ".html")
+ # with open(html_path, "w", encoding="utf-8") as f:
+ # f.write(html_timeline)
+ # except Exception as e:
+ # main_app_logger.info(
+ # f"\033[91m[DIAGNOSTICS] Failed to generate PyTorch memory snapshot: {e}\033[0m"
+ # )
+
+ # def assess_memory(self, device, video_name, start_allocated, start_reserved):
+ # gc.collect()
+
+ # surviving_arrays = [
+ # obj
+ # for obj in gc.get_objects()
+ # # if isinstance(obj, np.ndarray) and obj.size >= 1 #(1920 * 1080)
+ # if type(obj) is np.ndarray and obj.ndim > 0
+ # ]
+ # analyze_tracemalloc_snapshot()
+
+ # main_app_logger.info(
+ # f"[DIAGNOSTICS] Found {len(surviving_arrays)} uncollected large arrays alive in RAM filesystem."
+ # )
+
+ # for i, arr in enumerate(surviving_arrays):
+ # referrers = gc.get_referrers(arr)
+ # main_app_logger.info(
+ # f" > Array {i} | Shape: {arr.shape} | Pinned by {len(referrers)} references:"
+ # )
+ # for ref in referrers:
+ # if isinstance(ref, dict):
+ # main_app_logger.info(
+ # f" - Dict Keys holding this array: {list(ref.keys())[:4]}"
+ # )
+ # else:
+ # main_app_logger.info(
+ # f" - Variable holding object layout: {type(ref)}",
+ # )
+ # # ───────────────────────────────────────
+
+ # # analyze_tracemalloc_snapshot()
+
+ # if device == "gpu":
+ # self.print_active_gpu_tensor_memory()
+ # else:
+ # # --- CPU RAM PATH METRICS ---
+ # # main_app_logger.info("=" * 60, )
+ # # main_app_logger.info("\n[RAM INVESTIGATOR] Scanning Host CPU Memory Standings:", )
+ # # process = psutil.Process(os.getpid())
+ # # current_rss = process.memory_info().rss / (1024 * 1024) # Host RAM in MB
+ # # main_app_logger.info(f" > Current Process Resident Set Size (RSS): {current_rss:.2f} MB", )
+ # self.print_active_cpu_tensor_memory()
+
+ # self.print_active_shared_memory()
+
+ # # AUTOMATED DIAGNOSTIC PROFILING PHASE (TRIGGERED ON TEST TEARDOWN)
+ # self.diagnostic_profiler(device, video_name, start_allocated, start_reserved)
+
+ def _print_gpu_mem(self):
+ if self.device_input == "cuda" and torch.cuda.is_available():
+ # Memory currently used by tensors
+ allocated = torch.cuda.memory_allocated(0) / 1024**2
+ # Total memory reserved by PyTorch (the "Pool")
+ reserved = torch.cuda.memory_reserved(0) / 1024**2
+ main_app_logger.info(f"\tAllocated: {allocated:0.2f} MB")
+ main_app_logger.info(f"\tReserved: {reserved:0.2f} MB")
+ else:
+ main_app_logger.info("\tCUDA not available.")
+
+ def get_frame_gt_boxes(self, abs_frame_num, gt_sequence, gt_boxes):
+ if abs_frame_num - 1 >= len(gt_sequence):
+ return np.empty((0, 4), dtype=np.float32)
+
+ gt_seq = int(gt_sequence[abs_frame_num - 1])
+ if gt_seq == abs_frame_num:
+ gt_b_xywh = gt_boxes[abs_frame_num - 1]
+ gt_b = [
+ gt_b_xywh[0],
+ gt_b_xywh[1],
+ min(gt_b_xywh[0] + gt_b_xywh[2], target_width - 1),
+ min(gt_b_xywh[1] + gt_b_xywh[3], target_height - 1),
+ ]
+ # Force a 2D matrix shape layout of (1, 4) even for a single box
+ target_boxes_array = np.array([gt_b], dtype=np.float32).reshape(-1, 4)
+ # target_labels_array = np.array([0], dtype=np.int64)
+ else:
+ # gt_b = [0,0,0,0]
+ target_boxes_array = np.empty((0, 4), dtype=np.float32)
+
+ return target_boxes_array
+
+ def update_eval_stats(self, metadata_or_bbs, target_boxes_array):
+ """
+ metadata_or_bbs: In 640 (resize) space
+ """
+
+ # gt_seq = int(gt_sequence[abs_frame_num - 1])
+ # if gt_seq == abs_frame_num:
+ # gt_b_xywh = gt_boxes[abs_frame_num - 1]
+ # gt_b = [
+ # gt_b_xywh[0],
+ # gt_b_xywh[1],
+ # min(gt_b_xywh[0] + gt_b_xywh[2], target_width - 1),
+ # min(gt_b_xywh[1] + gt_b_xywh[3], target_height - 1),
+ # ]
+ # # Force a 2D matrix shape layout of (1, 4) even for a single box
+ # target_boxes_array = np.array([gt_b], dtype=np.float32).reshape(
+ # -1, 4
+ # )
+ # # target_labels_array = np.array([0], dtype=np.int64)
+ # else:
+ # # gt_b = [0,0,0,0]
+ # target_boxes_array = np.empty((0, 4), dtype=np.float32)
+ # target_labels_array = np.empty((0,), dtype=np.int64)
+
+ bbs = []
+ scores = []
+ labels = []
+ if isinstance(metadata_or_bbs, dict):
+ for _, v in metadata_or_bbs.items():
+ labels.append(0)
+ scores.append(float(v["bbox"]["object_det"]["confidence"]))
+ bbox = [
+ v["bbox"]["x"],
+ v["bbox"]["y"],
+ v["bbox"]["width"],
+ v["bbox"]["height"],
+ ]
+ # x, y, w, h = scale_bbox_xywh(
+ # bbox,
+ # self.resize_w,
+ # self.resize_h,
+ # targetW=target_width,
+ # targetH=target_height,
+ # )
+ # x2 = min(x + w, target_width - 1)
+ # y2 = min(y + h, target_height - 1)
+
+ x, y, x2, y2 = scale_bbox(
+ bbox,
+ self.resize_w,
+ self.resize_h,
+ targetW=target_width,
+ targetH=target_height,
+ in_format="xywh",
+ out_format="xyxy",
+ )
+
+ bbs.append([x, y, x2, y2])
+ else:
+ for bbox in metadata_or_bbs:
+ labels.append(0)
+ scores.append(0.9)
+ x, y, x2, y2 = scale_bbox(
+ bbox,
+ self.resize_w,
+ self.resize_h,
+ targetW=target_width,
+ targetH=target_height,
+ in_format="xyxy",
+ out_format="xyxy",
+ )
+ bbs.append([x, y, x2, y2])
+
+ # Convert to array, then explicitly reshape to enforce the 2D constraint
+ pred_boxes_array = np.array(bbs, dtype=np.float32).reshape(-1, 4)
+ pred_scores_array = np.array(scores, dtype=np.float32)
+ # pred_labels_array = np.array(labels, dtype=np.int64)
+ # preds = {
+ # "boxes": pred_boxes_array, # Guaranteed to be (N, 4) even if N=0
+ # "scores": pred_scores_array,
+ # "labels": pred_labels_array,
+ # }
+ # ]
+ # self.all_preds.append(preds)
+ # self.all_targets.append(targets)
+ # metric_engine.update(preds, targets)
+ self.evaluator.update_frame(
+ pred_boxes_array, pred_scores_array, target_boxes_array
+ )
+
+ del pred_boxes_array, pred_scores_array
+
+ # COMPARE SNAPSHOTS --------------------------------------------
+ def capture_state_snapshot(self):
+ """Captures a metadata-only registry mapping of all active keys and values."""
+ snapshot = {}
+ for attr, val in list(self.__dict__.items()):
+ if val is None:
+ type_str = "NoneType"
+ details = None
+ else:
+ type_str = val.__class__.__name__
+
+ # Extract detailed allocation size/state contexts depending on type
+ if type_str == "Tensor":
+ details = f"shape={list(val.shape)}, device={val.device}, dtype={val.dtype}"
+ elif type_str == "ndarray":
+ details = f"shape={list(val.shape)}, dtype={val.dtype}"
+ elif type_str in ("list", "dict", "set", "tuple"):
+ try:
+ details = f"len={len(val)}"
+ except Exception:
+ details = "uncountable"
+ elif type_str in ("Thread", "Process", "DummyProcess"):
+ try:
+ details = f"alive={val.is_alive()}"
+ except Exception:
+ details = "unknown"
+ elif type_str in ("Lock", "_RLock"):
+ try:
+ details = f"locked={val.locked()}"
+ except Exception:
+ details = "unknown"
+ elif type_str == "Queue":
+ try:
+ details = f"approx_qsize={val.qsize()}"
+ except Exception:
+ details = "uncountable"
+ else:
+ # Capture values for strings, ints, bools, etc.
+ try:
+ details = (
+ str(val)[:50]
+ if type_str in ("str", "int", "bool", "float")
+ else hex(id(val))
+ )
+ except Exception:
+ details = "unresolved_pointer"
+
+ snapshot[attr] = {"type": type_str, "details": details}
+ return snapshot
+
+ def print_lifecycle_delta(
+ self,
+ before_snapshot,
+ after_snapshot,
+ keys_to_skip=default_attr_keys,
+ return_keys=False,
+ ):
+ """Compares snapshots and outputs a precise list of elements that must be evicted."""
+ main_app_logger.info("=" * 80)
+ main_app_logger.info(
+ " RESOURCE LIFECYCLE LIFESPAN ANALYSIS (BEFORE START vs AFTER STOP)"
+ )
+ main_app_logger.info("=" * 80)
+
+ # 1. New keys initialized during execution
+ new_keys = [
+ key
+ for key in sorted(
+ list(set(after_snapshot.keys()) - set(before_snapshot.keys()))
+ )
+ if key not in keys_to_skip
+ ]
+
+ main_app_logger.info(
+ f"[+] NEW ARTIFACTS GENERATED DURING THE RUN ({len(new_keys)} keys found):"
+ )
+ if new_keys:
+ for key in new_keys:
+ main_app_logger.info(
+ f" └─ {key} -> Type: {after_snapshot[key]['type']} ({after_snapshot[key]['details']})"
+ )
+ main_app_logger.info(
+ " ⚠️ ACTION REQUIRED: Must be deleted or unlinked via delattr()."
+ )
+ else:
+ main_app_logger.info(
+ " None! (No new top-level attributes were registered)."
+ )
+
+ # 2. Key values modified, grown, or mutated during execution
+ mutated_keys = []
+ static_keys = []
+ common_keys = set(before_snapshot.keys()) & set(after_snapshot.keys())
+ for key in sorted(list(common_keys)):
+ if (key not in keys_to_skip) and (
+ before_snapshot[key] != after_snapshot[key]
+ ):
+ if after_snapshot[key]["details"] not in [None, "len=0"]:
+ mutated_keys.append(key)
+ elif (
+ (key not in keys_to_skip)
+ and (before_snapshot[key] == after_snapshot[key])
+ and getattr(self, key).__class__.__name__ != "method"
+ ):
+ static_keys.append(key)
+
+ main_app_logger.info(
+ f"[Δ] RETAINED KEYS MUTATED OR GROWN DURING THE RUN ({len(mutated_keys)} keys found):"
+ )
+ if mutated_keys:
+ for key in mutated_keys:
+ b_meta = before_snapshot[key]
+ a_meta = after_snapshot[key]
+ main_app_logger.info(f" └─ {key}")
+ main_app_logger.info(
+ f" ├── Before Start: {b_meta['type']} ({b_meta['details']})"
+ )
+ main_app_logger.info(
+ f" └── After Stop: {a_meta['type']} ({a_meta['details']})"
+ )
+ main_app_logger.info(
+ " ⚠️ ACTION REQUIRED: Revert back to baseline state, call .clear(), or set to None."
+ )
+ else:
+ main_app_logger.info(
+ " None! (All original tracking keys remained perfectly static)."
+ )
+
+ # main_app_logger.info(f"\n[Δ] RETAINED STATIC KEYS DURING THE RUN ({len(static_keys)} keys found):")
+ # if static_keys:
+ # for key in static_keys:
+ # b_meta = before_snapshot[key]
+ # main_app_logger.info(f" └─ {key}")
+ # main_app_logger.info(f" ├── Before Start: {b_meta['type']} ({b_meta['details']})")
+ # main_app_logger.info(f" ⚠️ ACTION REQUIRED: Possibly remove because it has not changed.")
+ main_app_logger.info("=" * 80)
+
+ if return_keys:
+ # return new_keys, static_keys
+ return new_keys, mutated_keys, static_keys
+
+ def stop_blueprint_executor(self, new_keys, mutated_keys):
+ """Dynamically creates a blueprint of all changes and liquidates them safely."""
+ main_app_logger.info("=" * 80)
+ main_app_logger.info(" DYNAMIC TEARDOWN COMPILER & RESOURCE EVICTION ENGINE")
+ main_app_logger.info(
+ "=" * 80,
+ )
+
+ # keys_to_skip = [
+ # "_is_stopped",
+ # "active",
+ # "status",
+ # "baseline_before_start",
+ # ]
+
+ # 1. Identify dynamically added attributes (Must be destroyed / unlinked)
+ new_keys = set(
+ new_keys
+ ) # set(state_before_stop.keys()) - set(before_snapshot.keys())
+
+ # 2. Identify baseline attributes that grew or mutated (Must be reset)
+ mutated_keys = set(mutated_keys)
+ # mutated_keys = set()
+ # for key in (set(before_snapshot.keys()) & set(state_before_stop.keys())):
+ # if before_snapshot[key] != state_before_stop[key] and key not in keys_to_skip:
+ # mutated_keys.append(key) if isinstance(mutated_keys, list) else mutated_keys.add(key)
+
+ # --- PHASE 1: TARGETED HARDWARE DEALLOCATIONS (ORDER SENSITIVE) ---
+ # First, handle background threads and OS processes before severing structural arrays
+ all_targeted_keys = (
+ new_keys | mutated_keys
+ ) # [key for key in list(new_keys | mutated_keys)] # if key not in keys_to_skip]
+
+ # A. Prioritize process and thread termination hooks
+ for key in list(all_targeted_keys):
+ val = getattr(self, key, None)
+ if val is None:
+ continue
+ type_name = val.__class__.__name__
+
+ if type_name in ("Thread", "Process", "DummyProcess"):
+ main_app_logger.info(
+ f" [RECONCILING] Terminating unmanaged runtime worker: {key}"
+ )
+ # try:
+ # if hasattr(val, "is_alive") and val.is_alive():
+ # if hasattr(val, "terminate"): val.terminate()
+ # val.join(timeout=0.5)
+ # except Exception: pass
+ self.stop_thread(val)
+
+ elif type_name == "ThreadPoolExecutor":
+ main_app_logger.info(
+ f" [RECONCILING] Shutting down concurrent thread pool: {key}"
+ )
+ try:
+ val.shutdown(wait=True, cancel_futures=True)
+ except Exception:
+ pass
+
+ # B. Decouple IPC Multi-processing Queues next (Non-blocking)
+ for key in list(all_targeted_keys):
+ val = getattr(self, key, None)
+ if val is None:
+ continue
+ if val.__class__.__name__ == "Queue":
+ main_app_logger.info(
+ f" [RECONCILING] Draining and closing IPC channel: {key}"
+ )
+ try:
+ while not val.empty():
+ val.get_nowait()
+ except Exception:
+ pass
+ try:
+ val.close()
+ val.cancel_join_thread()
+ except Exception:
+ pass
+
+ # C. Unmap Shared Memory nodes from /dev/shm
+ # for key in list(all_targeted_keys):
+ # val = getattr(self, key, None)
+ # if val is None: continue
+ # type_name = val.__class__.__name__
+ # if "shm" in key.lower() or type_name in ("SharedMemory", "SharedMemoryManager"):
+ # main_app_logger.info(f" [RECONCILING] Unlinking OS Shared Memory layout node: {key}")
+ # if type_name == "list":
+ # for item in val:
+ # if item.__class__.__name__ == "tuple":
+ # for i in item:
+ # if i.__class__.__name__ == "Event":
+ # try:
+ # if hasattr(i, "_handle"):
+ # i._handle.close()
+ # except Exception:
+ # pass
+ # i = None
+ # else:
+ # i = None
+
+ # try:
+ # if hasattr(item, "close"):
+ # item.close()
+ # item.unlink()
+ # else:
+ # item = None
+ # except Exception:
+ # # pass
+ # traceback.print_exc()
+ # elif item is not None:
+ # try:
+ # item.close()
+ # item.unlink()
+ # except Exception:
+ # # pass
+ # traceback.print_exc()
+ # else:
+ # try:
+ # val.close()
+ # val.unlink()
+ # except Exception:
+ # # pass
+ # traceback.print_exc()
+
+ # --- PHASE 2: FLUSHING AND UNLINKING OBJECT REFERENCES ---
+ # Now that the background hardware loops are dead, scrub the attributes
+ for key in sorted(list(all_targeted_keys)):
+ val = getattr(self, key, None)
+ if val is None:
+ continue
+ type_name = val.__class__.__name__
+
+ # If it's a new attribute, wipe its internal space and erase it completely
+ if key in new_keys:
+ if type_name == "Tensor":
+ try:
+ val.data = torch.empty(0, device=val.device)
+ except Exception:
+ pass
+ elif isinstance(val, dict):
+ val.clear()
+ elif isinstance(val, list):
+ val.clear()
+ elif type_name == "GpuMat":
+ try:
+ val.release()
+ except Exception:
+ pass
+
+ # Permanently strip the attribute from the object namespace
+ try:
+ delattr(self, key)
+ except AttributeError:
+ pass
+
+ # If it was a mutated pre-existing baseline key, reset it to its origin baseline
+ elif key in mutated_keys:
+ if isinstance(val, (list, dict, set)):
+ try:
+ val.clear()
+ except Exception:
+ pass
+ # else:
+ # # Revert primitive counters or strings back to their exact original baseline state
+ # orig_type = before_snapshot[key]["type"]
+ elif isinstance(val, None): # orig_type == "NoneType":
+ setattr(self, key, None)
+ elif isinstance(val, int): # orig_type == "int":
+ setattr(self, key, 0)
+ elif isinstance(val, bool): # orig_type == "bool":
+ setattr(self, key, False)
+ elif isinstance(val, str): # orig_type == "str":
+ setattr(self, key, "")
+
+ # --- PHASE 3: FINAL HOST HEAP FLUSH ---
+ # import gc
+
+ gc.collect()
+ if torch.cuda.is_available():
+ torch.cuda.synchronize()
+ torch.cuda.empty_cache()
+
+ main_app_logger.info("=" * 80)
+ main_app_logger.info(
+ " [SUCCESS] Dynamic Blueprint Execution Complete. Clean state achieved."
+ )
+ main_app_logger.info("=" * 80)
+
+ # DEBUG FUNCTIONS --------------------------------------------
+ def debug_save_mask(self, frame_source, frame_num, rois=None, gt_boxes=None):
+ debug_dir = self.result_dir / "debug_stages" / self._testMethodName / "mask"
+ debug_dir.mkdir(parents=True, exist_ok=True)
+
+ save_path = (
+ debug_dir / f"frame_{frame_num:04d}_mask.jpg"
+ ) # f"mask_{frame_num:04d}.jpg"
+
+ # cv2.imwrite(
+ # # str(stage_debug_dir / f"frame_{f_num:04d}_stage6_threshold.jpg"),
+ # str(stage_debug_dir / f"frame_{self.frame_count_target:04d}_stagerois_merged.jpg"),
+ # display_frame,
+ # )
+
+ if frame_num > self.config.DEBUG_FRAME_LIMIT:
+ return
+
+ # # Download or copy the data
+ # if hasattr(frame_source, "download"):
+ # img_cpu = frame_source.download()
+ # elif torch.is_tensor(frame_source):
+ # # .contiguous() fixes the horizontal "shredding"/static look
+ # temp = frame_source.squeeze(0) if frame_source.ndim == 4 else frame_source
+ # img_cpu = temp.permute(1, 2, 0).contiguous().cpu().numpy()
+ # else:
+ # # For numpy arrays (like your pinned memory), ensure memory is linear
+ # img_cpu = np.ascontiguousarray(frame_source)
+
+ # # Fix Visibility (Normalization)
+ # # If float, scale to 0-255. If uint8, leave as is to avoid "neon" colors.
+ # if img_cpu.dtype != np.uint8:
+ # if img_cpu.max() <= 1.0:
+ # img_cpu = (img_cpu * 255).clip(0, 255).astype(np.uint8)
+ # else:
+ # img_cpu = img_cpu.astype(np.uint8)
+
+ # Download/Copy the frame
+ if torch.is_tensor(frame_source):
+ # .contiguous() is CRITICAL here to fix the "shredded" look
+ temp = frame_source.squeeze(0) if frame_source.ndim == 4 else frame_source
+ # img_cpu = temp.permute(1, 2, 0).contiguous().cpu().numpy()
+ img_cpu = temp.contiguous().cpu().numpy()
+ img_cpu = cv2.cvtColor(img_cpu, cv2.COLOR_RGB2BGR)
+ elif hasattr(frame_source, "download"):
+ img_cpu = frame_source.download()
+ else:
+ img_cpu = np.ascontiguousarray(frame_source)
+
+ # Fix Shape: restore spatial grid if flattened
+ if img_cpu.ndim == 3 and img_cpu.shape[0] == 1:
+ img_cpu = img_cpu.reshape((self.resize_h, self.resize_w, 3))
+
+ # Fix Visibility: ONLY multiply if it's actually floating point
+ # If uint8 is multiplied by 255, it wraps around and creates "neon" colors
+ if img_cpu.dtype != np.uint8:
+ if img_cpu.max() <= 1.0:
+ img_cpu = (img_cpu * 255).clip(0, 255).astype(np.uint8)
+ else:
+ img_cpu = img_cpu.astype(np.uint8)
+
+ # Color Space: Standardize to BGR for imwrite
+ if len(img_cpu.shape) != 3:
+ img_cpu = cv2.cvtColor(img_cpu, cv2.COLOR_GRAY2BGR)
+
+ h_img, w_img = img_cpu.shape[:2]
+ scale_x = float(w_img) / self.frame_width
+ scale_y = float(h_img) / self.frame_height
+ if gt_boxes is not None:
+ # Factor for converting 8K (original) bb dimensions to display dimensions
+ # disp_w, disp_h = inf_data["mask"].shape[:2] if hasattr(inf_data["mask"], "shape") else inf_data["mask"].size()
+ # scale_display_ox = disp_w / self.frame_width
+ # scale_display_oy = disp_h / self.frame_height
+
+ img_cpu = get_bb_overlay(
+ img_cpu,
+ gt_boxes,
+ (scale_x, scale_y),
+ (w_img, h_img),
+ color=(0, 255, 0), # green
+ )
+
+ # Draw 8K Boxes (Scaled down)
+ if rois is not None and len(rois) > 0:
+ boxes = rois.cpu().tolist() if torch.is_tensor(rois) else rois
+ for box in boxes:
+ x1, y1, x2, y2 = [
+ # int(box[0] * scale_x),
+ # int(box[1] * scale_y),
+ # int(box[2] * scale_x),
+ # int(box[3] * scale_y),
+ max(0, int(box[0] * scale_x)),
+ max(0, int(box[1] * scale_y)),
+ min(w_img - 1, int(box[2] * scale_x)),
+ min(h_img - 1, int(box[3] * scale_y)),
+ ]
+ cv2.rectangle(img_cpu, (x1, y1), (x2, y2), (0, 0, 255), 2)
+
+ # Save to disk
+ cv2.imwrite(str(save_path), img_cpu)
+
+ def debug_save_img_roi(self, frame_source, bbs_full_res, frame_num, gt_boxes=None):
+ debug_dir = self.result_dir / "debug_stages" / self._testMethodName / "img_roi"
+ debug_dir.mkdir(parents=True, exist_ok=True)
+
+ # out_filename = str(debug_dir / f"analysis_{frame_num:04d}.jpg")
+ save_path = debug_dir / f"frame_{frame_num:04d}_analysis.jpg"
+
+ if frame_num > self.config.DEBUG_FRAME_LIMIT:
+ return
+
+ # Download/Copy the frame
+ if torch.is_tensor(frame_source):
+ # .contiguous() is CRITICAL here to fix the "shredded" look
+ temp = frame_source.squeeze(0) if frame_source.ndim == 4 else frame_source
+ img_cpu = temp.permute(1, 2, 0).contiguous().cpu().numpy()
+ elif hasattr(frame_source, "download"):
+ img_cpu = frame_source.download()
+ else:
+ img_cpu = np.ascontiguousarray(frame_source)
+
+ # Fix Shape: restore spatial grid if flattened
+ if img_cpu.ndim == 3 and img_cpu.shape[0] == 1:
+ img_cpu = img_cpu.reshape((self.resize_h, self.resize_w, 3))
+
+ # Fix Visibility: ONLY multiply if it's actually floating point
+ # If uint8 is multiplied by 255, it wraps around and creates "neon" colors
+ if img_cpu.dtype != np.uint8:
+ if img_cpu.max() <= 1.0:
+ img_cpu = (img_cpu * 255).clip(0, 255).astype(np.uint8)
+ else:
+ img_cpu = img_cpu.astype(np.uint8)
+
+ # Color Space: Standardize to BGR for imwrite
+ if len(img_cpu.shape) != 3:
+ img_cpu = cv2.cvtColor(img_cpu, cv2.COLOR_GRAY2BGR)
+
+ # Draw 8K Boxes (Scaled down)
+ h_img, w_img = img_cpu.shape[:2]
+ scale_x = w_img / self.frame_width
+ scale_y = h_img / self.frame_height
+
+ if gt_boxes is not None:
+ # Factor for converting 8K (original) bb dimensions to display dimensions
+ # disp_w, disp_h = inf_data["mask"].shape[:2] if hasattr(inf_data["mask"], "shape") else inf_data["mask"].size()
+ # scale_display_ox = disp_w / self.frame_width
+ # scale_display_oy = disp_h / self.frame_height
+
+ img_cpu = get_bb_overlay(
+ img_cpu,
+ gt_boxes,
+ (scale_x, scale_y),
+ (w_img, h_img),
+ color=(0, 255, 0), # green
+ )
+
+ if bbs_full_res is not None:
+ boxes = (
+ bbs_full_res.cpu().tolist()
+ if torch.is_tensor(bbs_full_res)
+ else bbs_full_res
+ )
+ # Annotate Box to image shape
+ for box in boxes:
+ x1, y1, x2, y2 = [
+ max(0, int(box[0] * scale_x)),
+ max(0, int(box[1] * scale_y)),
+ min(w_img - 1, int(box[2] * scale_x)),
+ min(h_img - 1, int(box[3] * scale_y)),
+ ]
+ img_cpu = cv2.rectangle(img_cpu, (x1, y1), (x2, y2), (0, 0, 255), 2)
+
+ cv2.imwrite(str(save_path), img_cpu)
+
+ def debug_save_crops(self, cropped_batch, frame_num):
+ """Saves the first 5 crops of a batch to the results directory."""
+ debug_dir = self.result_dir / "debug_stages" / self._testMethodName / "crops"
+ # debug_dir = self.result_dir / "debug_stages" / self._testMethodName
+ debug_dir.mkdir(parents=True, exist_ok=True)
+
+ # Only save for the first self.config.DEBUG_FRAME_LIMIT frames to avoid disk bloat
+ if frame_num > self.config.DEBUG_FRAME_LIMIT:
+ return
+
+ for i, crop in enumerate(cropped_batch[: self.config.DEBUG_FRAME_LIMIT]):
+ # Convert GPU Tensor [C, H, W] -> NumPy [H, W, C]
+ if torch.is_tensor(crop):
+ # Reverse normalization (* 255) and permute to BGR
+ # img = (crop.squeeze(0).permute(1, 2, 0) * 255).byte().cpu().numpy()
+ temp = crop.squeeze(0) if crop.ndim == 4 else crop
+ # Flip the color channels from RGB to BGR before any extraction utilities run
+ # Flips only the first axis (channels) from [0, 1, 2] to [2, 1, 0]
+ temp = torch.flip(temp, dims=[0])
+ img = temp.permute(1, 2, 0).contiguous().cpu().numpy()
+ # img = cv2.cvtColor(img, cv2.COLOR_RGB2BGR)
+ else:
+ img = crop
+
+ if img.dtype != np.uint8:
+ if img.max() <= 1.0:
+ img = (img * 255).clip(0, 255).astype(np.uint8)
+ else:
+ img = img.astype(np.uint8)
+
+ cv2.imwrite(str(debug_dir / f"frame_{frame_num}_crop_{i}.jpg"), img)
+
+ def debug_save_img(self, frame_source, frame_num):
+ debug_dir = self.result_dir / "debug_stages" / self._testMethodName / "img"
+ debug_dir.mkdir(parents=True, exist_ok=True)
+
+ if frame_num > self.config.DEBUG_FRAME_LIMIT:
+ return
+
+ # Download/Copy the frame
+ if torch.is_tensor(frame_source):
+ # .contiguous() is CRITICAL here to fix the "shredded" look
+ temp = frame_source.squeeze(0) if frame_source.ndim == 4 else frame_source
+ img_cpu = temp.permute(1, 2, 0).contiguous().cpu().numpy()
+ elif hasattr(frame_source, "download"):
+ img_cpu = frame_source.download()
+ else:
+ img_cpu = np.ascontiguousarray(frame_source)
+
+ # Fix Shape: restore spatial grid if flattened
+ if img_cpu.ndim == 3 and img_cpu.shape[0] == 1:
+ img_cpu = img_cpu.reshape((self.resize_h, self.resize_w, 3))
+
+ # Fix Visibility: ONLY multiply if it's actually floating point
+ # If uint8 is multiplied by 255, it wraps around and creates "neon" colors
+ if img_cpu.dtype != np.uint8:
+ if img_cpu.max() <= 1.0:
+ img_cpu = (img_cpu * 255).clip(0, 255).astype(np.uint8)
+ else:
+ img_cpu = img_cpu.astype(np.uint8)
+
+ # Color Space: Standardize to BGR for imwrite
+ if len(img_cpu.shape) == 3:
+ pass
+ else:
+ img_cpu = cv2.cvtColor(img_cpu, cv2.COLOR_GRAY2BGR)
+
+ cv2.imwrite(str(debug_dir / f"analysis_{frame_num:04d}.jpg"), img_cpu)
diff --git a/fastapi/tests/metrics.py b/fastapi/tests/metrics.py
new file mode 100644
index 0000000..5963115
--- /dev/null
+++ b/fastapi/tests/metrics.py
@@ -0,0 +1,258 @@
+import torch
+
+
+class DeviceAgnosticOnTheFlyEvaluator:
+ def __init__(self, device="cpu"):
+ # device will be "cpu" or "cuda" / "gpu" depending on what pytest passes
+ self.device = "cuda" if "gpu" in str(device).lower() else "cpu"
+
+ iou_range = [0.10, 0.95]
+ self.num_steps = int(round(((iou_range[1] - iou_range[0]) / 0.05) + 1, 0))
+ self.iou_thresholds = torch.linspace(
+ iou_range[0], iou_range[1], self.num_steps, device=self.device
+ )
+
+ self.global_scores = []
+ self.global_matches = []
+ self.total_gt_count = 0
+ self.total_gt_hit_lenient = 0 # GT boxes hit by ANY prediction with IoU >= 0.10
+
+ @torch.inference_mode()
+ def update_frame(self, pred_boxes, pred_scores, gt_boxes):
+ """
+ Accepts bounding boxes and scores as arrays or tensors.
+ Automatically migrates inputs to the targeted CPU or GPU sandbox backend.
+ """
+ # Ensure all incoming payloads are safely handled as PyTorch tensors on the target device
+ pred_boxes = torch.as_tensor(
+ pred_boxes, dtype=torch.float32, device=self.device
+ )
+ pred_scores = torch.as_tensor(
+ pred_scores, dtype=torch.float32, device=self.device
+ )
+ gt_boxes = torch.as_tensor(gt_boxes, dtype=torch.float32, device=self.device)
+
+ # Filter invalid ground truth placeholders (-1)
+ if gt_boxes.numel() > 0:
+ valid_mask = ~torch.all(gt_boxes == -1, dim=1)
+ gt_boxes = gt_boxes[valid_mask]
+
+ num_gt_in_frame = gt_boxes.shape[0]
+ self.total_gt_count += num_gt_in_frame
+ if pred_boxes.shape[0] == 0:
+ return
+
+ # Sort frame predictions by confidence score descending
+ pred_scores, sort_idx = torch.sort(pred_scores, descending=True)
+ pred_boxes = pred_boxes[sort_idx]
+
+ # Allocate matching registers for this frame
+ frame_matches = torch.zeros(
+ (pred_boxes.shape[0], self.num_steps), dtype=torch.int32, device=self.device
+ )
+
+ if num_gt_in_frame > 0:
+ # Vectorized Broadcast IoU Calculation
+ preds_exp = pred_boxes.unsqueeze(1) # (N, 1, 4)
+ gts_exp = gt_boxes.unsqueeze(0) # (1, M, 4)
+
+ x_min = torch.maximum(preds_exp[..., 0], gts_exp[..., 0])
+ y_min = torch.maximum(preds_exp[..., 1], gts_exp[..., 1])
+ x_max = torch.minimum(preds_exp[..., 2], gts_exp[..., 2])
+ y_max = torch.minimum(preds_exp[..., 3], gts_exp[..., 3])
+
+ inter_area = torch.clamp(x_max - x_min, min=0) * torch.clamp(
+ y_max - y_min, min=0
+ )
+ pred_areas = (preds_exp[..., 2] - preds_exp[..., 0]) * (
+ preds_exp[..., 3] - preds_exp[..., 1]
+ )
+ gt_areas = (gts_exp[..., 2] - gts_exp[..., 0]) * (
+ gts_exp[..., 3] - gts_exp[..., 1]
+ )
+ union_area = pred_areas + gt_areas - inter_area
+
+ iou_matrix = torch.where(union_area > 0, inter_area / union_area, 0.0)
+
+ if iou_matrix.numel() > 0:
+ # Get the highest IoU any prediction achieved for each GT box
+ max_ious_per_gt, _ = torch.max(iou_matrix, dim=0)
+ # If a GT box was overlapped by at least 10%, consider it "found"
+ self.total_gt_hit_lenient += torch.sum(max_ious_per_gt >= 0.10).item()
+
+ # Match targets per threshold layer
+ for th_idx, th in enumerate(self.iou_thresholds):
+ matched_gts = torch.zeros(
+ num_gt_in_frame, dtype=torch.bool, device=self.device
+ )
+ for p_idx in range(pred_boxes.shape[0]):
+ row_ious = iou_matrix[p_idx]
+ # Mask out already matched ground truths
+ row_ious = torch.where(~matched_gts, row_ious, -1.0)
+
+ best_iou, best_gt_idx = torch.max(row_ious, dim=0)
+
+ if best_iou >= th and best_gt_idx != -1:
+ if row_ious[best_gt_idx] >= th: # Extra validation gate
+ frame_matches[p_idx, th_idx] = 1
+ matched_gts[best_gt_idx] = True
+
+ # Track only tiny matrix shards instead of bulky bounding box contexts
+ # self.global_scores.append(pred_scores)
+ # self.global_matches.append(frame_matches)
+
+ # Force detach data fields completely out of the PyTorch hardware execution graph
+ # and cast down to standard CPU lists before storing them over long durations
+ if hasattr(pred_scores, "detach"):
+ scores_list = pred_scores.detach().cpu().tolist()
+ else:
+ scores_list = list(pred_scores)
+
+ if hasattr(frame_matches, "detach"):
+ matches_list = frame_matches.detach().cpu().tolist()
+ else:
+ matches_list = list(frame_matches)
+
+ # EXTEND to keep the list flat and completely avoid ragged dimensions
+ self.global_scores.extend(scores_list)
+ self.global_matches.extend(matches_list)
+
+ del pred_scores, pred_boxes, gt_boxes
+
+ @torch.inference_mode()
+ def compute_final_metrics(self):
+ """Processes final precision/recall curves natively on the configured hardware device."""
+ if len(self.global_scores) == 0:
+ return {
+ "mAP_10": 0.0,
+ "mAP_50": 0.0,
+ "mAP_75": 0.0,
+ "mAP_10_95": 0.0,
+ "Precision_10": 0.0,
+ "Recall_10": 0.0,
+ "F1_Score_10": 0.0,
+ "Precision_50": 0.0,
+ "Recall_50": 0.0,
+ "F1_Score_50": 0.0,
+ "Max_Recall_IoU_10": 0.0,
+ }
+
+ # Flatten list of partial tensors into global timeline matrices
+ # scores = torch.cat(self.global_scores, dim=0)
+ # matches = torch.cat(self.global_matches, dim=0)
+
+ # Flatten list of pre-detached primitive layouts into clean standalone timeline tensors
+ scores = torch.tensor(
+ self.global_scores, dtype=torch.float32, device=self.device
+ ).view(-1)
+ matches = torch.tensor(
+ self.global_matches, dtype=torch.int32, device=self.device
+ ).view(-1, self.num_steps)
+
+ # Global Sort across the whole timeline by confidence scores
+ _, global_sort_idx = torch.sort(scores, descending=True)
+ matches = matches[global_sort_idx]
+
+ aps = []
+
+ # Track dictionary mapping for multi-threshold statistical processing
+ # Key: Threshold Value (float) -> Values: (Precision, Recall)
+ target_metrics = {
+ 0.10: {"precision": 0.0, "recall": 0.0},
+ 0.50: {"precision": 0.0, "recall": 0.0},
+ }
+
+ for th_idx in range(self.num_steps):
+ current_th = float(f"{self.iou_thresholds[th_idx].item():.2f}")
+
+ tp_cum = torch.cumsum(matches[:, th_idx], dim=0)
+ fp_cum = torch.cumsum(1 - matches[:, th_idx], dim=0)
+
+ precisions = tp_cum / (tp_cum + fp_cum)
+ if self.total_gt_count > 0:
+ recalls = tp_cum / self.total_gt_count
+ else:
+ recalls = torch.zeros_like(tp_cum)
+
+ # Vectorized 101-point COCO AP Interpolation Layout
+ mpre = torch.cat(
+ [
+ torch.tensor([0.0], device=self.device),
+ precisions,
+ torch.tensor([0.0], device=self.device),
+ ]
+ )
+ mrec = torch.cat(
+ [
+ torch.tensor([0.0], device=self.device),
+ recalls,
+ torch.tensor([1.0], device=self.device),
+ ]
+ )
+
+ # Enforce strict decreasing precision monotonicity
+ mpre = torch.flip(
+ torch.cummax(torch.flip(mpre, dims=[0]), dim=0).values, dims=[0]
+ )
+
+ recall_thresholds = torch.linspace(0.0, 1.0, 101, device=self.device)
+ inds = torch.searchsorted(mrec, recall_thresholds, side="left")
+ inds = torch.clamp(inds, max=mpre.shape[0] - 1)
+
+ ap = torch.sum(mpre[inds]) / 101.0
+ aps.append(ap.item())
+
+ # if th_idx == 0:
+ # final_precision_10 = (
+ # precisions[-1].item() if precisions.numel() > 0 else 0.0
+ # )
+ # final_recall_10 = recalls[-1].item() if recalls.numel() > 0 else 0.0
+
+ # Dynamically extract and assign metrics if threshold matches target
+ if current_th in target_metrics:
+ target_metrics[current_th]["precision"] = (
+ precisions[-1].item() if precisions.numel() > 0 else 0.0
+ )
+ target_metrics[current_th]["recall"] = (
+ recalls[-1].item() if recalls.numel() > 0 else 0.0
+ )
+
+ # f1_10 = (
+ # (2 * final_precision_10 * final_recall_10)
+ # / (final_precision_10 + final_recall_10)
+ # if (final_precision_10 + final_recall_10) > 0
+ # else 0.0
+ # )
+
+ # Calculate F1 Score harmonic balances
+ p10, r10 = target_metrics[0.10]["precision"], target_metrics[0.10]["recall"]
+ f1_10 = (2 * p10 * r10) / (p10 + r10) if (p10 + r10) > 0 else 0.0
+
+ p50, r50 = target_metrics[0.50]["precision"], target_metrics[0.50]["recall"]
+ f1_50 = (2 * p50 * r50) / (p50 + r50) if (p50 + r50) > 0 else 0.0
+
+ max_recall_10 = (
+ (self.total_gt_hit_lenient / self.total_gt_count)
+ if self.total_gt_count > 0
+ else 0.0
+ )
+
+ # Robust close-approximation lookups bypassing basic Python indexing limitations
+ map10_idx = torch.argmin(torch.abs(self.iou_thresholds - 0.10)).item()
+ map50_idx = torch.argmin(torch.abs(self.iou_thresholds - 0.50)).item()
+ map75_idx = torch.argmin(torch.abs(self.iou_thresholds - 0.75)).item()
+
+ return {
+ "mAP_10": aps[map10_idx],
+ "mAP_50": aps[map50_idx], # mAP at 0.50 IoU
+ "mAP_75": aps[map75_idx], # mAP at 0.75 IoU
+ "mAP_10_95": sum(aps)
+ / float(self.num_steps), # Averaged mAP over all 18 threshold layers
+ "Precision_10": p10,
+ "Recall_10": r10,
+ "F1_Score_10": f1_10,
+ "Precision_50": p50,
+ "Recall_50": r50,
+ "F1_Score_50": f1_50,
+ "Max_Recall_IoU_10": max_recall_10,
+ }
diff --git a/fastapi/tests/test_detections.py b/fastapi/tests/test_detections.py
index 6cac5b9..5f0f36e 100644
--- a/fastapi/tests/test_detections.py
+++ b/fastapi/tests/test_detections.py
@@ -1,23 +1,90 @@
+# ==============================================================================
+# SUPPRESS WARNINGS
+# import warnings
+
+import pytest
+
+# warnings.filterwarnings(
+# "ignore", category=FutureWarning, message=".*reduce_op` is deprecated.*"
+# )
+
+pytestmark = [
+ pytest.mark.filterwarnings(
+ "ignore:.*anyio:_pytest.warning_types.PytestAssertRewriteWarning"
+ ),
+ pytest.mark.filterwarnings(
+ "ignore:Context managers for TensorRT types are deprecated:DeprecationWarning"
+ ),
+ pytest.mark.filterwarnings(
+ "ignore:Exception ignored in.*SharedMemory.__del__:UserWarning"
+ ),
+ # Global message fallback pattern captures local execution frames
+ pytest.mark.filterwarnings("ignore:.*reduce_op.*:FutureWarning"),
+]
+
+# ==============================================================================
+# LOGGING
+import logging
+import os
+import sys
+
+logging.basicConfig(
+ level=logging.INFO,
+ # format="%(asctime)s - %(name)s - %(levelname)s - %(message)s",
+ format="%(asctime)s [%(levelname)s] %(name)s (%(filename)s:%(lineno)d) - %(message)s",
+ handlers=[logging.StreamHandler(sys.stdout)],
+)
+
+# Suppress low-delay reference block warnings from OpenCV/PyAV/FFmpeg
+# os.environ["OPENCV_FFMPEG_LOGLEVEL"] = "-8"
+# os.environ["OPENCV_LOG_LEVEL"] = "OFF"
+logging.getLogger("libav").setLevel(logging.CRITICAL)
+logging.getLogger("libav.hevc").setLevel(logging.CRITICAL)
+logging.getLogger("matplotlib").setLevel(logging.WARNING)
+logging.getLogger("ultralytics").setLevel(logging.WARNING)
+# logger = trt.Logger(trt.Logger.WARNING)
+# trt.init_libnvinfer_plugins(logger, "")
+main_app_logger = logging.getLogger(__name__)
+
+# ==============================================================================
+# IMPORTS
+
+# os.environ["OMP_NUM_THREADS"] = "2"
+# os.environ["MKL_NUM_THREADS"] = "2"
+# os.environ["PYTORCH_ALLOC_CONF"] = "expandable_segments:True"
+# try:
+# torch.set_num_interop_threads(2) # 1)
+# torch.set_num_threads(4) # 2)
+# except RuntimeError:
+# # Safe graceful fallback if a process-level fork duplicated context maps
+# pass
import argparse
+import asyncio
import csv
+import ctypes
+import faulthandler
import gc
-import logging
import multiprocessing as mp
-import os
import sys
import threading
import time
+import traceback
+import tracemalloc
from pathlib import Path
import cv2
-import matplotlib.pyplot as plt
-import numpy as np
-import pytest
-import tensorrt as trt
+import psutil
import torch
-import torch.nn.functional as F
-sys.path.insert(1, str(Path(__file__).parent.parent))
+# import torch
+
+# Retrieve repo packages
+REPO_DIR = str(Path(__file__).parent.parent)
+sys.path.insert(1, REPO_DIR)
+from base_test import (
+ BaseTest,
+ fps_comparison_chart,
+)
from include.default_configs import (
CUSTOM_MODEL_FLAG_DEFAULT,
DEBUG_DEFAULT,
@@ -25,109 +92,363 @@
THRESHOLD_VALUE,
)
from include.handlers import (
- CPUStreamHandler,
- DeviceBaseHandler,
- GPUStreamHandler,
- log_to_logger,
+ get_test_handler,
)
-from include.handlers import test_rendering_worker as rendering_worker
from include.utils import (
PipelineConfig,
- tensor2opencv,
+ ResourceTrackerFilter,
+ install_and_load_pip_package,
+ str2bool,
)
+# objgraph = install_and_load_pip_package("objgraph", attribute_name=None)
+# torch.set_grad_enabled(False)
+
+# ==============================================================================
+# MONKEY PATCH
+# Cache the hardware availability state globally.
+# This prevents internal framework loops from triggering driver/NVML checks per frame.
+# _cuda_available = torch.cuda.is_available()
+
+
+# def _patched_is_available():
+# return _cuda_available
+
+
+# torch.cuda.is_available = _patched_is_available
+
+# ==============================================================================
+# SETUP
+# Hooks directly into the OS kernel signals to force Python to print a full
+# stack traceback right before it dies, allowing you to see which line of
+# Python code caused the hard crash
+faulthandler.enable()
+
try:
+ # Retain standard spawn mode to prevent CUDA context driver deadlocks
torch.multiprocessing.set_start_method("spawn", force=True)
except RuntimeError:
pass
-logging.getLogger("matplotlib").setLevel(logging.WARNING)
-logger = trt.Logger(trt.Logger.WARNING)
-trt.init_libnvinfer_plugins(logger, "")
-
-DEBUG_FLAG_DEFAULT = True if DEBUG_DEFAULT == "1" else False
-os.environ["OMP_NUM_THREADS"] = "1"
-os.environ["PYTORCH_ALLOC_CONF"] = "expandable_segments:True"
force_export = False
+# target_width, target_height = 7680, 4320
+STATE_CAPTURE = False
+# Force Python's multiprocessing layer to quiet tracking cleanup race conditions
+os.environ["PYTHONWARNINGS"] = "ignore"
+sys.warnoptions.append("ignore")
-# def render_display_worker(queue, output_path, fps, size):
-# """Background process that draws labels and writes to video."""
-# fourcc = cv2.VideoWriter_fourcc(*"avc1")
-# writer = cv2.VideoWriter(output_path, fourcc, fps, size)
-
-# if not writer.isOpened():
-# print(f" [ERROR] VideoWriter failed to open: {output_path}")
-# return
-# while True:
-# item = queue.get()
-# if item is None:
-# break
+# =========================================================================
+# ISOLATED BACKGROUND WORKER
+# =========================================================================
-# display_frame, metadata_or_bbs, class_list = item
-# if (display_frame.shape[1], display_frame.shape[0]) != size:
-# display_frame = cv2.resize(display_frame, size)
-# if isinstance(metadata_or_bbs, dict):
-# for _, obj in metadata_or_bbs.items():
-# bbox = obj["bbox"]
-# x, y, w, h = bbox["x"], bbox["y"], bbox["width"], bbox["height"]
-# class_name = bbox["object"]
-# class_id = class_list.index(class_name) if class_name in class_list else 0
-# confidence = bbox["object_det"]["confidence"]
+def isolated_detection_worker(init_args, test_args, res_queue):
+ device = test_args["device"]
+ detection_type = test_args["detection_type"]
+ sf_enabled = test_args["sf_enabled"]
+ gt_enabled = test_args["gt_enabled"]
-# bb_color = get_detection_color(class_id, is_bgr=True)
-# label = f"{class_name} {confidence:.2f}"
+ if device == "gpu" and torch.cuda.is_available():
+ torch.cuda.synchronize()
+ torch.cuda.ipc_collect()
+ torch.cuda.empty_cache()
+ gc.collect()
-# cv2.rectangle(display_frame, (x, y), (x + w, y + h), bb_color, 2)
-# draw_label(display_frame, label, (x, y), color=bb_color, padding=5)
-# elif metadata_or_bbs is not None:
-# for box in metadata_or_bbs:
-# if torch.is_tensor(box):
-# box = box.tolist()
-# x1, y1, x2, y2 = map(int, box)
-# cv2.rectangle(display_frame, (x1, y1), (x2, y2), (0, 0, 225), 2)
+ # Establish an accurate post-initialization hardware baseline
+ if str2bool(os.getenv("ENABLE_PROFILING", "False")):
+ tracemalloc.start()
-# writer.write(display_frame)
+ start_allocated, start_reserved = 0, 0
+ if device == "gpu" and torch.cuda.is_available():
+ torch.cuda.memory._record_memory_history(
+ enabled=True,
+ trace_alloc_max_entries=250000,
+ trace_alloc_record_context=True,
+ )
-# writer.release()
+ # This captures your baseline AFTER initialize_variables() is done
+ start_allocated = torch.cuda.memory_allocated(0)
+ start_reserved = torch.cuda.memory_reserved(0)
+ try:
+ cv2.cuda.setBufferPoolUsage(False)
+ except AttributeError:
+ pass
+ else:
+ gc.collect()
+ process = psutil.Process(os.getpid())
+ start_allocated = (
+ process.memory_info().rss
+ ) # Reuse start_allocated container for host memory baseline
+ start_reserved = 0
-def fps_comparison_chart(chart_path, results, fps_key="Pipeline FPS (Video frames)"):
try:
- names = [r["Test Name"] for r in results]
- fps_values = [float(r[fps_key]) for r in results]
-
- plt.figure(figsize=(10, 6))
- plt.grid(axis="y", linestyle="--", alpha=0.7, zorder=0)
-
- colors = ["#2ca02c" if "gpu" in n.lower() else "#1f77b4" for n in names]
- bars = plt.bar(names, fps_values, color=colors, zorder=3)
- plt.ylabel("Frames Per Second (FPS)")
- plt.title(f"Performance Comparison: {chart_path.stem}")
- plt.xticks(rotation=45)
-
- for bar in bars:
- yval = bar.get_height()
- plt.text(
- bar.get_x() + bar.get_width() / 2,
- yval + 1,
- f"{yval:.1f}",
- ha="center",
- va="bottom",
+ instance, _ = get_test_handler(TestDetections(), device)
+
+ instance.source = init_args["source"]
+ instance.name = init_args["name"]
+ instance.result_dir = Path(init_args["result_dir"])
+ instance.active_streams = init_args["active_streams"]
+ instance.__class__.benchmarks = init_args["benchmarks"] # Sandbox the metrics
+
+ vid_dir = instance.result_dir / device
+ vid_dir.mkdir(parents=True, exist_ok=True)
+ os.environ["TEST_SUITE_RENDER_DIR"] = str(vid_dir)
+
+ if instance.source.startswith("rtsp"):
+ short_name = "rtsp"
+ else:
+ short_name = Path(instance.source).stem
+
+ # config definition
+ config = PipelineConfig(
+ SHARED_OUTPUT=str(instance.result_dir), # defined in context
+ CUSTOM_MODEL_FLAG=os.getenv("CUSTOM_MODEL_FLAG", CUSTOM_MODEL_FLAG_DEFAULT),
+ DEVICE=device.upper(),
+ OMIT_DETECTIONS_FLAG=True,
+ TEST_MODE=True,
+ DEBUG=os.getenv("DEBUG", DEBUG_DEFAULT),
+ DEBUG_FRAME_LIMIT=int(os.getenv("DEBUG_FRAME_LIMIT", 100)),
+ ENABLE_QUERYING=False,
+ MODEL_NAME=os.getenv("MODEL_NAME", MODEL_NAME_DEFAULT),
+ SMART_FILTERING_ENABLED=sf_enabled,
+ THRESHOLD_VALUE=int(os.getenv("THRESHOLD_VALUE", THRESHOLD_VALUE)),
+ DETECTION_TYPE=detection_type,
+ )
+
+ # INITIALIZE CLASS (mimic DeviceBaseHandler.__init__) ------------------------------
+ instance.is_rtsp = str(instance.source).startswith("rtsp:/")
+ instance.active = True
+ instance.config = config
+
+ # kwarg definition
+ instance._testMethodName = (
+ f"sf_{detection_type}_{device}"
+ if sf_enabled
+ else f"yolo_{detection_type}_{device}"
+ )
+ instance.video_output_name = f"{instance._testMethodName}_{short_name}.mp4"
+
+ instance.loop = asyncio.get_event_loop()
+ instance.frame_ready_event = asyncio.Event()
+ instance._is_stopped = False
+ instance._stop_lock = threading.Lock() # Local lock for this instance
+ instance.main_startup_event = mp.Event()
+
+ instance.device = instance.config.DEVICE
+ instance.device_input = instance.config.device_input
+ instance.disp_w, instance.disp_h = instance.config.DISPLAY_FRAME_SIZE
+ instance.resize_h, instance.resize_w = [
+ instance.config.MODEL_H,
+ instance.config.MODEL_W,
+ ]
+
+ instance.setup_reader(
+ instance.config.TARGET_FPS,
+ instance.config.CLIP_DURATION,
+ startup_event=instance.main_startup_event,
+ )
+ instance.initialize_variables()
+ instance.setup_threads()
+ instance.last_heartbeat = time.perf_counter()
+
+ if STATE_CAPTURE:
+ main_app_logger.info(
+ "[DIAGNOSTIC] Registering system state baseline layout..."
+ )
+ instance.baseline_before_start = instance.capture_state_snapshot()
+
+ def orig_fn(profiler):
+ return instance.run_realtime_inference(
+ sf_enabled=instance.config.sf_enabled,
+ profiler=profiler,
+ gt_enabled=gt_enabled,
)
- if fps_values:
- plt.ylim(0, max(fps_values) * 1.2)
+ def profiler_fn():
+ profiler = None
+ try:
+ if str2bool(os.getenv("ENABLE_PROFILING", "False")):
+ Profiler = install_and_load_pip_package(
+ "pyinstrument", attribute_name="Profiler"
+ )
+
+ profiler = Profiler(interval=0.005) # 5ms sampling interval
+
+ # Telling the statistical sampler to skip recording exception blocks completely
+ # stops stack_sampler.py from ballooning RAM over long production runs.
+ if hasattr(profiler, "_sampler") and profiler._sampler:
+ profiler._sampler.trace_exceptions = False
+
+ profiler.start()
+
+ orig_fn(profiler)
+
+ if str2bool(os.getenv("ENABLE_PROFILING", "False")):
+ # 2. Redirect standard error to the filter trap right before report compilation
+ original_stderr = sys.stderr
+ sys.stderr = ResourceTrackerFilter(original_stderr)
- plt.tight_layout()
- plt.savefig(str(chart_path))
- print(f" Comparison chart saved to: {chart_path}")
- except Exception:
- print("Skipping chart generation: error occurred.")
+ try:
+ # Force standard stdout to flush out any lingering teardown messages
+ # BEFORE pyinstrument dumps its massive ASCII tree block.
+ sys.stdout.flush()
+ # main_app_logger.info(profiler.output_text(color=True))
+ # profiler.main_app_logger.info(color=True)
+ # prof_output = profiler.output_text(color=True)
+ # main_app_logger.info(
+ # f"\n=== LATENCY BREAKDOWN FOR {self.name} ({device}) ===\n{prof_output}\n",
+ #
+ # )
+ # Save a clean, interactive tree map for visual analysis
+ output_html_path = instance.output_path.replace(
+ ".mp4", "_profile.html"
+ )
+ # output_html_path = f"/tmp/profile_{video_name}_{device}.html"
+ profiler.write_html(output_html_path)
+ main_app_logger.info(
+ f"[PROFILER] Performance tree map exported to {output_html_path}",
+ )
+
+ finally:
+ sys.stderr = original_stderr
+ if "profiler" in locals():
+ try:
+ # Force the Python interpreter to detach pyinstrument's sampling hooks
+ sys.setprofile(None)
+
+ # Completely decouple internal statistical sessions to drop C-heap frames
+ if hasattr(profiler, "_last_session"):
+ profiler._last_session = None
+ if hasattr(profiler, "last_session"):
+ profiler.last_session = None
+
+ # Forcibly clear out internal memoryview strings caching tree metrics
+ if (
+ hasattr(profiler, "session")
+ and profiler.session is not None
+ ):
+ if hasattr(profiler.session, "frame_groups"):
+ profiler.session.frame_groups = None
+ if hasattr(profiler.session, "samples"):
+ profiler.session.samples = []
+
+ # Purge compiled tree metrics structures
+ profiler.session = None
+ del profiler
+ except Exception:
+ pass
+
+ # Trigger an immediate native Linux heap compression pass
+ # This grabs the newly abandoned pyinstrument C-heap blocks and
+ # flushes them to the OS before the fixture assessment snapshot fires!
+ gc.collect()
+ try:
+ libc = ctypes.CDLL("libc.so.6")
+ libc.malloc_trim(0)
+ except Exception:
+ pass
+
+ except Exception:
+ traceback.print_exc()
+
+ if str2bool(os.getenv("ENABLE_PROFILING", "False")) and hasattr(
+ instance, "process_thread"
+ ):
+ # Re-initialize the thread context using our safe profile wrapper proxy
+ instance.process_thread = threading.Thread(target=profiler_fn, daemon=True)
+
+ instance.VIDEO_GT_DETAILS = None
+ instance.duration_target = 30
+
+ instance.start()
+
+ while instance.active or not instance._is_stopped:
+ time.sleep(0.25)
+ if getattr(instance, "status", None) == "DONE":
+ break
+
+ if (
+ hasattr(instance, "process_thread")
+ and instance.process_thread is not None
+ ):
+ if not instance.process_thread.is_alive():
+ main_app_logger.info(
+ "[TEST HARNESS] Background worker exited. Breaking loop.",
+ )
+ break
+
+ instance.stop_threads(["process_thread"])
+
+ # Force early hardware driver sweep before unbinding threads
+ if instance.device_input == "cuda" and torch.cuda.is_available():
+ torch.cuda.synchronize()
+ torch.cuda.empty_cache()
+ torch.cuda.ipc_collect()
+
+ gc.collect()
+
+ try:
+ if STATE_CAPTURE:
+ # --- CAPTURE STATE POINT B (Right before cleanup) ---
+ main_app_logger.info(
+ "[DIAGNOSTIC] Gathering active execution workspace layouts..."
+ )
+ state_after_stop = instance.capture_state_snapshot()
+
+ # 3. Print the granular delta analysis out to your terminal screen
+ new_keys_generated, mutated_keys, static_keys = (
+ instance.print_lifecycle_delta(
+ instance.baseline_before_start,
+ state_after_stop,
+ return_keys=True,
+ )
+ )
+ # self.config.DEBUG_FLAG and
+ if len(new_keys_generated) > 0 or len(mutated_keys) > 0:
+ # Inform the blueprint engine to eliminate every difference found between Point A and B
+ instance.stop_blueprint_executor(new_keys_generated, mutated_keys)
+ except Exception:
+ traceback.print_exc()
+
+ assert instance.status == "DONE"
+
+ # TEARDOWN
+ instance.execute_teardown()
+
+ if str2bool(os.getenv("ENABLE_PROFILING", "False")):
+ gc.collect()
+ if (
+ device == "gpu" and torch.cuda.is_available()
+ ): # self.device_input == "cuda" and torch.cuda.is_available():
+ torch.cuda.synchronize()
+ torch.cuda.empty_cache()
+ instance.assess_memory(
+ device, instance.name, start_allocated, start_reserved
+ )
+
+ if len(instance.__class__.benchmarks) > 0:
+ final_metrics = instance.__class__.benchmarks[-1]
+ res_queue.put({"status": "success", "metrics": final_metrics})
+ else:
+ res_queue.put({"status": "error", "error": "No benchmarks generated."})
+
+ except Exception as e:
+ res_queue.put(
+ {"status": "error", "error": str(e), "traceback": traceback.format_exc()}
+ )
+ res_queue.put(
+ {"status": "error", "error": str(e), "traceback": traceback.format_exc()}
+ )
+
+
+# =========================================================================
+# PYTEST TEST HARNESS
+# =========================================================================
@pytest.fixture(scope="class")
def setup_context(request):
"""Replaces setUpClass: Runs once per test class."""
@@ -136,36 +457,46 @@ def setup_context(request):
main_path = test_dir.parent
video_dir = main_path / "inputs"
- VIDEO_FILENAME = os.getenv("VIDEO_FILENAME", "anduril_swarm_8K.mp4")
- if video_dir.exists():
- request.cls.video_path = video_dir / VIDEO_FILENAME
+ # model_name = os.getenv("MODEL_NAME", MODEL_NAME_DEFAULT)
+ # Handler.__init__ (main items)
+ request.cls.source = os.getenv("VIDEO_FILENAME", "anduril_swarm_8K.mp4")
+ is_rtsp = "rtsp://" in request.cls.source
+ if not is_rtsp:
+ VIDEO_FILENAME = request.cls.source
+ if video_dir.exists():
+ vid_source = video_dir / VIDEO_FILENAME
+ else:
+ video_dir = Path("/watch_dir")
+ vid_source = video_dir / VIDEO_FILENAME
+
+ assert vid_source.exists()
+ request.cls.source = str(vid_source)
+ request.cls.name = vid_source.stem
+ request.cls.is_rtsp = False
else:
- video_dir = Path("/watch_dir")
- request.cls.video_path = video_dir / VIDEO_FILENAME
+ request.cls.name = "rtsp"
+ request.cls.is_rtsp = True
model_name = os.getenv("MODEL_NAME", MODEL_NAME_DEFAULT)
request.cls.result_dir = (
- test_dir
- / f"{current_test_filename}_results/{model_name}"
- / request.cls.video_path.stem
+ test_dir / f"{current_test_filename}_results/{model_name}" / request.cls.name
)
request.cls.result_dir.mkdir(parents=True, exist_ok=True)
+
+ # Benchmark statistics
request.cls.benchmarks = []
request.cls.csv_filename = (
- f"pipeline_benchmarks_{model_name}_{request.cls.video_path.stem}.csv"
+ f"detections_benchmarks_{model_name}_{request.cls.name}.csv"
)
+ request.cls.csv_path = request.cls.result_dir / request.cls.csv_filename
- request.cls.name = request.cls.video_path.stem
- request.cls.source = str(request.cls.video_path)
- request.cls.active = True
+ # request.cls.active = True
request.cls.active_streams = {}
- request.cls._shared_model = None
- request.cls._shared_model_path = None
- request.cls._shared_model_device = None
- request.cls._shared_model_sf_enabled = None
+ # RUN ALL PARAMETERIZED TESTS ----------------------------------------
yield
+ # FINAL CSV EXPORT --------------------------------------------------
if request.cls.benchmarks:
# Filter and exclude rows that were interrupted or failed initialization due to an early pytest skip
request.cls.benchmarks = [
@@ -179,8 +510,8 @@ def setup_context(request):
(r for r in results if r["Test Name"] == match_name), None
)
if cpu_row:
- gpu_fps = float(row["Pipeline FPS (Video frames)"])
- cpu_fps = float(cpu_row["Pipeline FPS (Video frames)"])
+ gpu_fps = float(row["Pipeline FPS (Target frames)"])
+ cpu_fps = float(cpu_row["Pipeline FPS (Target frames)"])
speedup = (gpu_fps / cpu_fps) if cpu_fps > 0 else 0
row["Pipeline Speedup vs CPU"] = f"{speedup:.2f}x"
else:
@@ -188,1881 +519,151 @@ def setup_context(request):
else:
row["Pipeline Speedup vs CPU"] = "Baseline (CPU)"
- keys = results[0].keys()
- with open(
- str(request.cls.result_dir / request.cls.csv_filename), "w", newline=""
- ) as f:
- dict_writer = csv.DictWriter(f, fieldnames=keys)
- dict_writer.writeheader()
- dict_writer.writerows(results)
-
- print(f"\n[FINAL] Benchmarks saved to {request.cls.csv_filename}")
+ if results:
+ # # Define video targets explicitly during collection
+ # video_names = [f"video{i:02d}" for i in range(1, 21)]
+ # ALL_VIDEO_GT_DETAILS = download_eval_data(video_names)
- for r in results:
- print(
- f" > {r['Test Name']}: {r['Pipeline FPS (Video frames)']} FPS | {r['Pipeline FPS (Target frames)']} FPS | {r['sf+roi+det FPS (Target frames)']} FPS | {r['Display FPS']} FPS | Speedup: {r.get('Pipeline Speedup vs CPU', 'N/A')}"
- )
+ keys = results[0].keys()
- chart_path = (
- request.cls.result_dir
- / f"{request.cls.csv_filename.replace('.csv', '')}_sfFPS.png"
- )
- fps_comparison_chart(
- chart_path, results, fps_key="sf+roi+det FPS (Video frames)"
- )
+ with open(str(request.cls.csv_path), "w", newline="") as f:
+ dict_writer = csv.DictWriter(f, fieldnames=keys)
+ dict_writer.writeheader()
+ dict_writer.writerows(results)
+ main_app_logger.info(f"[FINAL] Benchmarks saved to {request.cls.csv_path}")
- chart_path = (
- request.cls.result_dir
- / f"{request.cls.csv_filename.replace('.csv', '')}_pipelineFPS.png"
- )
- fps_comparison_chart(chart_path, results, fps_key="Pipeline FPS (Video frames)")
-
- chart_path = (
- request.cls.result_dir
- / f"{request.cls.csv_filename.replace('.csv', '')}_sfFPS_target.png"
- )
- fps_comparison_chart(
- chart_path, results, fps_key="sf+roi+det FPS (Target frames)"
- )
-
- chart_path = (
- request.cls.result_dir
- / f"{request.cls.csv_filename.replace('.csv', '')}_pipelineFPS_target.png"
- )
- fps_comparison_chart(
- chart_path, results, fps_key="Pipeline FPS (Target frames)"
- )
-
- chart_path = (
- request.cls.result_dir
- / f"{request.cls.csv_filename.replace('.csv', '')}_displayFPS_target.png"
- )
- fps_comparison_chart(chart_path, results, fps_key="Display FPS")
-
-
-@pytest.fixture(autouse=True)
-def each_test_setup(request):
- test_class_self = request.instance
- if torch.cuda.is_available():
- torch.cuda.synchronize()
-
- device = request.node.callspec.params.get("device")
- detection_type = request.node.callspec.params.get("detection_type")
- sf_enabled = request.node.callspec.params.get("sf_enabled")
-
- test_class_self._testMethodName = (
- f"sf_{detection_type}_{device}"
- if sf_enabled
- else f"yolo_{detection_type}_{device}"
- )
-
- # 1. Re-initialize a fresh configuration object
- test_class_self.config = PipelineConfig(
- CUSTOM_MODEL_FLAG=os.getenv("CUSTOM_MODEL_FLAG", CUSTOM_MODEL_FLAG_DEFAULT),
- DEVICE=device.upper(),
- OMIT_DETECTIONS_FLAG=True,
- TEST_MODE=True,
- DEBUG=os.getenv("DEBUG", DEBUG_DEFAULT),
- DEBUG_FRAME_LIMIT=int(os.getenv("DEBUG_FRAME_LIMIT", 100)),
- ENABLE_QUERYING=False,
- MODEL_NAME=os.getenv("MODEL_NAME", MODEL_NAME_DEFAULT),
- SMART_FILTERING_ENABLED=sf_enabled,
- THRESHOLD_VALUE=int(os.getenv("THRESHOLD_VALUE", THRESHOLD_VALUE)),
- DETECTION_TYPE=detection_type,
- )
-
- # video_output_name = f"{test_class_self._testMethodName}_detections_output.mp4"
- vid_dir = test_class_self.result_dir / "results"
- vid_dir.mkdir(parents=True, exist_ok=True)
- # test_video_output_path = os.path.join(str(vid_dir), video_output_name)
- os.environ["TEST_SUITE_RENDER_DIR"] = str(vid_dir)
- test_class_self.config.SHARED_OUTPUT = str(test_class_self.result_dir)
-
- # Reset core state machine properties before invoking any backend handlers
- test_class_self.active = True
- test_class_self._is_stopped = False
-
- # FIXING ATTRIBUTEERRORS: Explicitly provision primitives before method re-binding
- import threading
- import time
-
- test_class_self._stop_lock = threading.Lock()
- # test_class_self.stat_start_time = time.perf_counter() # timing to display detection
-
- test_class_self.device = test_class_self.config.DEVICE
- test_class_self.device_input = test_class_self.config.device_input
- test_class_self.resize_h, test_class_self.resize_w = [
- test_class_self.config.MODEL_H,
- test_class_self.config.MODEL_W,
- ]
-
- # Resolve concrete handler class type
- HandlerClass = GPUStreamHandler if device == "gpu" else CPUStreamHandler
- HandlerClass.pipeline_fn = test_class_self.__class__.pipeline_fn
-
- # 2. Dynamically re-bind backend methods to this execution instance
- import inspect
- import types
-
- handler_classes = [HandlerClass, DeviceBaseHandler]
- all_method_names = set()
-
- for cls in handler_classes:
- for name, attr in inspect.getmembers(cls, predicate=inspect.isfunction):
- all_method_names.add(name)
+ main_app_logger.info("=" * 80)
+ main_app_logger.info(
+ f"{'Test Name':<25} | {'Pipeline FPS (Target)':<21} | {'Avg Frame Reading (ms)':<22} | {'Pipeline Speedup vs CPU':<15}",
+ )
+ main_app_logger.info("-" * 80)
- for method_name in all_method_names:
- source_obj = (
- HandlerClass if hasattr(HandlerClass, method_name) else DeviceBaseHandler
- )
- if hasattr(source_obj, method_name):
- raw_func = getattr(source_obj, method_name)
- if (
- not hasattr(test_class_self.__class__, method_name)
- or method_name == "pipeline_fn"
- ):
- setattr(
- test_class_self,
- method_name,
- types.MethodType(raw_func, test_class_self),
+ for r in results:
+ main_app_logger.info(
+ f"{r['Test Name']:<25} | {r['Pipeline FPS (Target frames)']:<21} | {r['Avg Frame Reading (ms)']:<22} | {r.get('Pipeline Speedup vs CPU', 'N/A'):<10}",
)
+ main_app_logger.info("=" * 125)
- # FIXING FILEEXISTSERRORS: Proactively clear lingering POSIX layout blocks in /dev/shm
- from multiprocessing import shared_memory
-
- for i in range(4): # Matches your self.ai_ring_depth footprint
- stale_shm_name = f"shm_ai_640_{test_class_self.name}_{i}_{os.getpid()}"
- try:
- # Force link onto lingering handle and unlink it instantly from the OS map
- lingering_shm = shared_memory.SharedMemory(name=stale_shm_name)
- lingering_shm.close()
- lingering_shm.unlink()
- except FileNotFoundError:
- pass
-
- # 3. FORCE NATIVE PIPELINE PROVISIONING
- test_class_self.setup_reader(
- test_class_self.config.TARGET_FPS, test_class_self.config.CLIP_DURATION
- )
- test_class_self.initialize_variables()
- test_class_self.setup_model(None)
- test_class_self.prepare_pipeline()
-
- # Reset runtime loop state properties
- test_class_self.render_queue = None
- test_class_self.frame_count_target = 0
- test_class_self.next_process_idx = 0.0
- test_class_self.frame_in_clip_count = 0
- test_class_self.frame_count = 0
- test_class_self.elapsed_display_time = 0.0
-
- if hasattr(test_class_self, "setup_threads"):
- test_class_self.setup_threads()
-
- shared_event_buffer = {"sf": [], "roi": [], "det": []}
- test_class_self.gpu_event_buffer = shared_event_buffer
+ chart_path = (
+ request.cls.result_dir
+ / f"{request.cls.csv_filename.replace('.csv', '')}_pipelineFPS.png"
+ )
+ fps_comparison_chart(
+ chart_path, results, fps_key="Pipeline FPS (Video frames)"
+ )
- if torch.cuda.is_available():
- torch.cuda.empty_cache()
- torch.cuda.ipc_collect()
+ chart_path = (
+ request.cls.result_dir
+ / f"{request.cls.csv_filename.replace('.csv', '')}_pipelineFPS_target.png"
+ )
+ fps_comparison_chart(
+ chart_path, results, fps_key="Pipeline FPS (Target frames)"
+ )
- yield
+ chart_path = (
+ request.cls.result_dir
+ / f"{request.cls.csv_filename.replace('.csv', '')}_sfdetectFPS_target.png"
+ )
+ fps_comparison_chart(chart_path, results, fps_key="SF/Detection FPS")
- # Teardown logic
- if (
- hasattr(test_class_self, "render_queue")
- and test_class_self.render_queue is not None
- ):
- try:
- test_class_self.render_queue.put(None)
- except Exception:
- pass
+ chart_path = (
+ request.cls.result_dir
+ / f"{request.cls.csv_filename.replace('.csv', '')}_displayFPS_target.png"
+ )
+ fps_comparison_chart(chart_path, results, fps_key="Display FPS")
- render_proc_handle = getattr(test_class_self, "render_proc", None)
- if render_proc_handle is not None and render_proc_handle._started.is_set():
- test_class_self.render_proc.join(timeout=10.0)
+ else:
+ main_app_logger.info(
+ "[WARN] Benchmarking matrix empty. Skipping final CSV and chart exports."
+ )
- # Signal the state machine to drop out of loops safely
- test_class_self.active = False
- if hasattr(test_class_self, "reader") and test_class_self.reader is not None:
- try:
- test_class_self.reader.stop()
- except Exception:
- pass
-
- # Execute custom release blocks
- test_class_self.stop()
-
- # CRITICAL FIX: Block and wait for the background producer thread to die
- # BEFORE we delete the reader attribute from the namespace map.
- producer_thread_handle = getattr(test_class_self, "process_thread", None)
- if producer_thread_handle is not None and producer_thread_handle.is_alive():
- producer_thread_handle.join(timeout=5.0)
-
- # Now it is structurally safe to purge attributes without causing cross-thread collisions
- keys_to_purge = [
- "reader",
- "process_thread",
- "inference_stream",
- "bgs_stream",
- "model",
- "raw_input",
- "ai_gpu_staging",
- "ai_pinned_tensors",
- "executor",
- "io_executor",
- ]
- for key in keys_to_purge:
- if hasattr(test_class_self, key):
- delattr(test_class_self, key)
-
- if (
- hasattr(test_class_self, "gpu_event_buffer")
- and test_class_self.gpu_event_buffer
- ):
- test_class_self.gpu_event_buffer.clear()
- if hasattr(test_class_self, "component_stats") and test_class_self.component_stats:
- test_class_self.component_stats.clear()
-
- with torch.inference_mode():
- if torch.cuda.is_available():
- torch.cuda.synchronize()
- torch.cuda.empty_cache()
- torch.cuda.ipc_collect()
- gc.collect()
- time.sleep(0.2)
+@pytest.mark.usefixtures("setup_context")
+class TestDetections(BaseTest):
+ """
+ Pytest runner that spawns the isolated worker, waits for completion,
+ and merges the returned metrics into the main class for CSV/Chart generation.
+ """
+ benchmarks = [] # Class-level attribute required by _finalize_benchmarks
-@pytest.mark.usefixtures("setup_context")
-class TestSmartFilteringDetections:
- # class TestSmartFilteringDetections(DeviceBaseHandler):
- # """
- # Unified testing harness.
- # By inheriting from DeviceBaseHandler, 'self' functions as both
- # the pytest telemetry harness and the live stream execution runner.
- # """
- # # Overriding __init__ to prevent standard initialization collisions with pytest
- # def __init__(self, *args, **kwargs):
- # pass
- # SETUP --------------------------------------------
-
- @pytest.mark.parametrize("device", ["cpu", "gpu"])
+ @pytest.mark.parametrize("device", ["gpu", "cpu"])
@pytest.mark.parametrize("sf_enabled", [True, False])
- @pytest.mark.parametrize("detection_type", ["motion", "object"])
- def test_detections(self, detection_type, device, sf_enabled):
- """Unified test runner for all configurations."""
+ @pytest.mark.parametrize("detection_type", ["object", "motion"])
+ def test_detections(self, device, sf_enabled, detection_type, gt_enabled=False):
if detection_type == "motion" and not sf_enabled:
pytest.skip(
- "Pure YOLO mode is structurally invalid for detection_type 'motion'."
+ "Pure YOLO mode is structurally invalid for detection_type 'motion'.\n"
)
- # Run the actual model loader
- # self.get_model_by_device(device, sf_enabled=self.config.sf_enabled)
-
- # Execute
- self.run_pipeline() # pipeline_fn)
-
- def setup_threads(self):
- # Shared 10MB memory for display
- self.setup_shared_memory()
-
- # Executor for Async YOLO tasks and FFmpeg re-encoding
- # self.executor = ThreadPoolExecutor(max_workers=self.config.MAX_WORKERS)
- # self.clip_executor = ThreadPoolExecutor(max_workers=self.config.MAX_WORKERS)
-
- print(
- f"sf_enabled: {self.config.sf_enabled}\tTEST_MODE: {self.config.TEST_MODE}",
- flush=True,
- )
-
- # Producer: Handles acquisition and AI metadata logs
- # self.process_thread = threading.Thread(
- # target=self.run_pipeline,
- # daemon=True,
- # )
-
- self.signal_queue = mp.Queue(maxsize=1)
- self.render_queue = mp.Queue(maxsize=5)
-
- # if self.config.TEST_MODE:
- test_dir = os.getenv(
- "TEST_SUITE_RENDER_DIR", str(Path(self.config.SHARED_OUTPUT))
- )
- os.makedirs(test_dir, exist_ok=True)
-
- video_output_name = f"{self._testMethodName}_detections_output.mp4"
- out_path = os.path.join(test_dir, video_output_name)
-
- log_to_logger(
- f"[TEST MODE] Detection results saved to: {out_path}", level="info"
- )
- self.render_proc = threading.Thread(
- target=rendering_worker,
- args=(
- self.render_queue,
- (self.disp_w, self.disp_h),
- out_path,
- self.target_fps,
- ),
- daemon=True,
- )
-
- # Dummy target alignment to prevent execution signature exceptions
- self.display_proc = threading.Thread(target=lambda: None, daemon=True)
- # else:
- # self.render_proc = mp.Process(
- # target=rendering_worker,
- # args=(
- # self.render_queue,
- # self.shared_details,
- # self.ready_buffer_idx,
- # self.reader_active_idx,
- # self.shm_frame_lengths,
- # self.signal_queue,
- # (self.disp_w, self.disp_h),
- # self.config.DISPLAY_FRAME_QUALITY,
- # ),
- # )
-
- # def display_signal_sync():
- # while self.active:
- # # Wait for signal
- # # if self.mp_frame_ready_event.wait(timeout=1.0):
- # # self.mp_frame_ready_event.clear()
- # try:
- # _ = self.signal_queue.get(timeout=1.0)
- # # print(f"[DEBUG]: Signal received in FastAPI process for {self.name}", flush=True)
- # # Wake FastAPI async loop in main thread
- # self.loop.call_soon_threadsafe(self.frame_ready_event.set)
- # except queue.Empty:
- # continue
-
- # self.display_proc = threading.Thread(
- # target=display_signal_sync, daemon=True
- # )
-
- # if self.config.ENABLE_QUERYING:
- # # NEW: Dedicated I/O pool for Disk/GPU transfers (Higher worker count for 8K)
- # self.io_executor = ThreadPoolExecutor(max_workers=8)
-
- # # Dedicated FFmpeg pool so re-encoding doesn't slow down live AI
- # self.ffmpeg_executor = ThreadPoolExecutor(max_workers=2)
-
- # if not self.config.TEST_MODE:
- # # Sends metadata to VDMS
- # self.metadata_thread = threading.Thread(
- # target=send_metadata,
- # args=(
- # VDMSPool(self.config.DBHOST, self.config.DBPORT, size=10),
- # self.config.DEBUG_FLAG,
- # self.config.INGESTION,
- # self.config.TEST_MODE,
- # self.config.UDF_HOST,
- # self.config.UDF_PORT,
- # self.config.DBHOST,
- # self.config.DBPORT,
- # ),
- # daemon=True,
- # )
-
- # # Consumer: Handles GPU-to-CPU download and Disk I/O (Writing resized frames to RAM disk)
- # self.writer_thread = threading.Thread(
- # target=self.video_writer_core_loop,
- # args=(self.stop_writer,),
- # daemon=True,
- # )
-
- # HELPERS --------------------------------------------
- def _print_gpu_mem(self):
- if torch.cuda.is_available():
- # Memory currently used by tensors
- allocated = torch.cuda.memory_allocated(0) / 1024**2
- # Total memory reserved by PyTorch (the "Pool")
- reserved = torch.cuda.memory_reserved(0) / 1024**2
- print(f"\tAllocated: {allocated:0.2f} MB")
- print(f"\tReserved: {reserved:0.2f} MB")
- else:
- print("\tCUDA not available.")
-
- # def get_model_by_device(self, device, sf_enabled=False):
- # """Singleton loader: only loads if device changes or model is missing."""
- # if (
- # sf_enabled
- # and (self.frame_width * self.frame_height)
- # <= self.config.SMART_FILTERING_PIXEL_CONSTRAINT
- # ):
- # sf_enabled = False
-
- # if (
- # TestSmartFilteringDetections._shared_model is not None
- # and TestSmartFilteringDetections._shared_model_device == device
- # and TestSmartFilteringDetections._shared_model_sf_enabled == sf_enabled
- # ):
- # self.model = TestSmartFilteringDetections._shared_model
- # self.model_path = TestSmartFilteringDetections._shared_model_path
- # return
-
- # run_platform_name = "engine" if "cuda" in self.device_input else "openvino"
-
- # if self.config.CUSTOM_MODEL_FLAG:
- # dir_path = "/home/resources/models/ultralytics/custom_models"
- # else:
- # dir_path = f"/home/resources/models/ultralytics/{self.config.MODEL_NAME}/{self.config.MODEL_PRECISION}"
-
- # (
- # TestSmartFilteringDetections._shared_model,
- # TestSmartFilteringDetections._shared_model_path,
- # self.label_source,
- # ) = get_model(
- # # model_run_key, model_run_config, export=False
- # Path(dir_path),
- # self.config.MODEL_NAME,
- # run_platform_name,
- # self.device_input,
- # batch=self.config.MODEL_MAX_BATCH_SIZE,
- # force_export=force_export,
- # sf_enabled=sf_enabled,
- # model_h=self.resize_h,
- # model_w=self.resize_w,
- # )
- # TestSmartFilteringDetections._shared_model_device = device
- # TestSmartFilteringDetections._shared_model_sf_enabled = sf_enabled
- # self.model = TestSmartFilteringDetections._shared_model
- # # self.model.half()
- # self.model_path = TestSmartFilteringDetections._shared_model_path
-
- # W, H = self.resize_w, self.resize_h
- # if not sf_enabled:
- # W, H = self.frame_width, self.frame_height
- # self.model_warmup(H, W)
- # # self.pipeline_handler.model = self.model
- # # self.pipeline_handler.model_warmup(H, W)
-
- # def calculate_unique_coverage(self, merged_boxes, target_w=640, target_h=640):
- # """
- # FULLY VECTORIZED: Calculate pixel coverage without Python loops.
- # Works for any number of boxes (1 to 1000+) with near-zero overhead.
- # """
- # if merged_boxes is None or merged_boxes.shape[0] == 0:
- # return 0.0
-
- # # 1. Scaling (Vectorized)
- # # merged_boxes is [N, 4] -> [x1, y1, x2, y2] in 8K space
- # scale = torch.tensor(
- # [
- # target_w / self.frame_width,
- # target_h / self.frame_height,
- # target_w / self.frame_width,
- # target_h / self.frame_height,
- # ],
- # device=self.device_input,
- # )
-
- # # Scale and clamp all boxes at once on the GPU
- # coords = (merged_boxes * scale).long()
- # coords[:, [0, 2]] = coords[:, [0, 2]].clamp(0, target_w)
- # coords[:, [1, 3]] = coords[:, [1, 3]].clamp(0, target_h)
-
- # # 2. Vectorized Mask Filling
- # # We create a 1D representation of the 640x640 mask for fast indexing
- # mask = torch.zeros(
- # target_h * target_w, device=self.device_input, dtype=torch.uint8
- # )
-
- # # For each box, we generate a range of indices and fill them.
- # # Note: For small N (drones), a loop is okay, but for 'Swarm' noise,
- # # we use this broadcasted approach:
- # for i in range(coords.shape[0]):
- # x1, y1, x2, y2 = coords[i]
- # # Generate row indices for this box
- # rows = torch.arange(y1, y2, device=self.device_input).view(-1, 1)
- # # Calculate mask indices: (y * width) + x
- # # This fills a horizontal slice of the mask in one GPU operation
- # indices = (rows * target_w) + torch.arange(x1, x2, device=self.device_input)
- # mask[indices] = 1
-
- # # 3. Final Sum (GPU Reduction)
- # return (torch.sum(mask).item() / (target_w * target_h)) * 100
-
- def calculate_unique_coverage(self, merged_boxes, target_w=640, target_h=640):
- """
- TRUE LOOPLESS VECTORIZATION: Calculates combined bounding box pixel coverage
- in a single GPU pass, completely avoiding CPU-GPU synchronization stalls.
- """
- if merged_boxes is None or merged_boxes.shape[0] == 0:
- return 0.0
-
- scale = torch.tensor(
- [
- target_w / self.frame_width,
- target_h / self.frame_height,
- target_w / self.frame_width,
- target_h / self.frame_height,
- ],
- device=self.device_input,
- )
-
- coords = (merged_boxes * scale).long()
- coords[:, [0, 2]] = coords[:, [0, 2]].clamp(0, target_w)
- coords[:, [1, 3]] = coords[:, [1, 3]].clamp(0, target_h)
-
- x1 = coords[:, 0].view(-1, 1, 1)
- y1 = coords[:, 1].view(-1, 1, 1)
- x2 = coords[:, 2].view(-1, 1, 1)
- y2 = coords[:, 3].view(-1, 1, 1)
-
- grid_y, grid_x = torch.meshgrid(
- torch.arange(target_h, device=self.device_input),
- torch.arange(target_w, device=self.device_input),
- indexing="ij",
- )
-
- inside_boxes = (grid_x >= x1) & (grid_x < x2) & (grid_y >= y1) & (grid_y < y2)
- unique_mask = torch.any(inside_boxes, dim=0)
-
- return (torch.sum(unique_mask).item() / (target_w * target_h)) * 100
-
- def _finalize_benchmarks(
- self,
- # n_frames,
- num_objs,
- total_pipeline_ms,
- real_world_latency_ms,
- coverage_percentages,
- sf_enabled,
- stat_frame_count,
- stat_fps,
- ):
- """Aggregates metrics and adds them to the results list."""
- latency_s = total_pipeline_ms / 1000.0 # Just pipeline (sf + roi_ det)
- est_fps = self.frame_count / latency_s if latency_s > 0 else 0
- est_fps_processed = self.frame_count_target / latency_s if latency_s > 0 else 0
- duration_s = self.frame_count / self.input_fps if self.input_fps > 0 else 0
-
- real_latency_s = real_world_latency_ms / 1000.0
- real_est_fps = self.frame_count / real_latency_s if real_latency_s > 0 else 0
- real_est_fps_processed = (
- self.frame_count_target / real_latency_s if real_latency_s > 0 else 0
- )
-
- # Calculate averages for component breakdowns
- avg_sf = (
- sum(self.component_stats["sf"]) / len(self.component_stats["sf"])
- if self.component_stats["sf"]
- else 0
- )
- avg_roi = (
- sum(self.component_stats["roi"]) / len(self.component_stats["roi"])
- if self.component_stats["roi"]
- else 0
- )
- avg_det = (
- sum(self.component_stats["det"]) / len(self.component_stats["det"])
- if self.component_stats["det"]
- else 0
- )
- avg_cov = (
- sum(coverage_percentages) / len(coverage_percentages)
- if coverage_percentages
- else (100.0 if not sf_enabled else 0)
- )
-
- avg_crops = (
- sum(self.crops_per_frame_list) / len(self.crops_per_frame_list)
- if self.crops_per_frame_list
- else 0
- )
+ # Pull the values dynamically assigned by the setup_context fixture
+ init_args = {
+ "source": self.__class__.source,
+ "name": self.__class__.name,
+ "result_dir": str(self.__class__.result_dir),
+ "active_streams": {},
+ "benchmarks": self.__class__.benchmarks,
+ }
- # Total Latency Sum (SF + ROI + DET)
- total_sum = avg_sf + avg_roi + avg_det
+ test_args = {
+ "device": device,
+ "detection_type": detection_type,
+ "sf_enabled": sf_enabled,
+ "gt_enabled": gt_enabled,
+ }
- # Calculate how often we hit a high-motion cap (e.g., 20 crops)
- capped_frames = sum(1 for c in self.crops_per_frame_list if c >= 20)
- cap_rate = (
- (capped_frames / len(self.crops_per_frame_list)) * 100
- if self.crops_per_frame_list
- else 0
+ main_app_logger.info(
+ f"\n{'=' * 60}\n"
+ f"[TEST HARNESS] Spawning isolated high-speed process for {self.__class__.name} (SF: {sf_enabled})...\n"
+ f"{'=' * 60}"
)
- self.__class__.benchmarks.append(
- {
- "Test Name": self._testMethodName,
- "Detection Type": self.config.DETECTION_TYPE,
- "Device": self.device,
- "Smart Filtering": "Enabled" if sf_enabled else "Disabled",
- "Video": self.video_path.name, # self.name?
- "Video FPS": f"{self.input_fps:.2f}",
- "Video Duration (s)": f"{duration_s:.4f}",
- "Video Frames": self.frame_count,
- "Target Frames": self.frame_count_target,
- "Pipeline Latency (s)": f"{real_latency_s:.2f}",
- "Display Latency (s)": f"{self.elapsed_display_time:.2f}",
- "sf+roi+det Latency (s)": f"{latency_s:.2f}",
- # Includes time by HW decoder (reader) and preparing output video
- "Pipeline FPS (Video frames)": f"{real_est_fps:.2f}",
- "Pipeline FPS (Target frames)": f"{real_est_fps_processed:.2f}",
- # TIming to display frame (read to after send to render queue)
- "Display Frames": stat_frame_count,
- "Display FPS": f"{stat_fps:.2f}",
- # Only sf + roi + det
- "sf+roi+det FPS (Video frames)": f"{est_fps:.2f}",
- "sf+roi+det FPS (Target frames)": f"{est_fps_processed:.2f}",
- "Avg SF (ms)": f"{avg_sf:.2f}",
- "Avg ROI (ms)": f"{avg_roi:.2f}",
- "Avg Obj. Detection (ms)": f"{avg_det:.2f}",
- "Total Breakdown Sum (ms)": f"{total_sum:.2f}",
- "Avg Area Coverage %": f"{avg_cov:.2f}%",
- "Avg Crops/Frame": f"{avg_crops:.1f}",
- "Crop Cap Rate (>20)": f"{cap_rate:.1f}%",
- "Objects Detected": num_objs,
- }
- )
+ # Spawn a pristine Python process (identical to test_pipeline.py's stream_worker)
+ ctx = mp.get_context("spawn")
+ res_queue = ctx.Queue()
- print(f"\n[{self._testMethodName}] Latency: {latency_s:.2f} sec")
- print(
- f"\n[{self._testMethodName}] Pipeline FPS (Target frames): {real_est_fps_processed:.2f} ({self.frame_count_target} frames)"
- )
- print(
- f"\n[{self._testMethodName}] sf+roi+det FPS (Target frames): {est_fps_processed:.2f} ({self.frame_count_target} frames)"
+ worker_p = ctx.Process(
+ target=isolated_detection_worker, args=(init_args, test_args, res_queue)
)
- print(
- f"\n[{self._testMethodName}] Display FPS (Target frames): {stat_fps:.2f} ({stat_frame_count} frames)"
- )
-
- # def tensor2opencv_gpu(self, frame_tensor):
- # """
- # GPU-native equivalent of tensor2opencv.
- # Fixes the '3x3 ghosting' and 'ValueError' at 30+ FPS.
- # """
- # # 1. Handle Batch Dimension: (1, 3, 640, 640) -> (3, 640, 640)
- # temp = frame_tensor.squeeze(0) if frame_tensor.ndim == 4 else frame_tensor
-
- # # 2. Fix Layout & Shape: (3, 640, 640) -> (640, 640, 3)
- # # contiguous() physically rearranges pixels in VRAM to interleave colors.
- # # reshape() ensures we never see the (1, 409600, 3) shape again.
- # gpu_hwc = temp.permute(1, 2, 0).reshape(640, 640, 3).contiguous()
-
- # # 3. Visibility Fix: Scale floats (0.0-1.0) to bytes (0-255) on GPU
- # if gpu_hwc.dtype != torch.uint8:
- # gpu_hwc = (gpu_hwc * 255).clamp(0, 255).byte()
-
- # # 4. Color Space Fix: RGB -> BGR (Matches OpenCV)
- # gpu_bgr = gpu_hwc.flip(-1).contiguous()
-
- # # 5. Bridge to CuPy (Zero-Copy)
- # cp_frame = cp.from_dlpack(torch.utils.dlpack.to_dlpack(gpu_bgr))
-
- # # 6. Select Double-Buffer from Pool
- # target_buf_cpu = self.pinned_pool_np[self.pool_idx]
- # target_buf_gpu = self.cp_pool[self.pool_idx]
- # self.pool_idx = (self.pool_idx + 1) % 2
-
- # if not hasattr(self, "transfer_stream"):
- # self.transfer_stream = cp.cuda.Stream(non_blocking=True)
-
- # with self.transfer_stream:
- # # cp.copyto is more robust for pinned memory than target[:]
- # cp.copyto(target_buf_gpu, cp_frame)
- # # Download into the pinned CPU RAM
- # target_buf_gpu.get(out=target_buf_cpu)
-
- # # 7. Mandatory Sync: Wait for the DMA transfer to hit RAM
- # self.transfer_stream.synchronize()
-
- # # Return a snapshot copy so the worker has private memory for 30 FPS
- # return target_buf_cpu.copy()
-
- # DEBUG FUNCTIONS --------------------------------------------
- def debug_save_mask(self, frame_source, frame_num, rois=None):
- debug_dir = self.result_dir / "debug_mask" / self._testMethodName
- debug_dir.mkdir(parents=True, exist_ok=True)
-
- if frame_num > self.config.DEBUG_FRAME_LIMIT:
- return
-
- # Download or copy the data
- if hasattr(frame_source, "download"):
- img_cpu = frame_source.download()
- elif torch.is_tensor(frame_source):
- # .contiguous() fixes the horizontal "shredding"/static look
- temp = frame_source.squeeze(0) if frame_source.ndim == 4 else frame_source
- # img_cpu = temp.permute(1, 2, 0).contiguous().cpu().numpy()
- if temp.ndim == 1:
- temp = temp.view(self.resize_h, self.resize_w)
-
- if temp.ndim == 3:
- img_cpu = temp.permute(1, 2, 0).contiguous().cpu().numpy()
- else:
- img_cpu = temp.contiguous().cpu().numpy()
- else:
- # For numpy arrays (like your pinned memory), ensure memory is linear
- img_cpu = np.ascontiguousarray(frame_source)
-
- # Fix Visibility (Normalization)
- # If float, scale to 0-255. If uint8, leave as is to avoid "neon" colors.
- if img_cpu.dtype != np.uint8:
- if img_cpu.max() <= 1.0:
- img_cpu = (img_cpu * 255).clip(0, 255).astype(np.uint8)
- else:
- img_cpu = img_cpu.astype(np.uint8)
-
- # Handle Color Space
- # OpenCV imwrite expects BGR. If 3-channel (RGB), swap. If 1-channel, save as-is.
- if len(img_cpu.shape) == 3:
- # img_cpu = cv2.cvtColor(img_cpu, cv2.COLOR_RGB2BGR)
- pass
- else:
- img_cpu = cv2.cvtColor(img_cpu, cv2.COLOR_GRAY2BGR)
-
- # Draw 8K Boxes (Scaled down)
- if rois is not None:
- h_img, w_img = img_cpu.shape[:2]
- scale_x = float(w_img) / self.frame_width
- scale_y = float(h_img) / self.frame_height
- boxes = rois.cpu().tolist() if torch.is_tensor(rois) else rois
- for box in boxes:
- x1, y1, x2, y2 = [
- int(box[0] * scale_x),
- int(box[1] * scale_y),
- int(box[2] * scale_x),
- int(box[3] * scale_y),
- ]
- cv2.rectangle(img_cpu, (x1, y1), (x2, y2), (0, 0, 255), 2)
-
- # Save to disk
- save_path = debug_dir / f"mask_{frame_num:04d}.jpg"
- cv2.imwrite(str(save_path), img_cpu)
-
- def debug_save_img_roi(self, frame_source, bbs_full_res, frame_num):
- debug_dir = self.result_dir / "debug_analysis" / self._testMethodName
- debug_dir.mkdir(parents=True, exist_ok=True)
-
- if frame_num > self.config.DEBUG_FRAME_LIMIT:
- return
-
- # Download/Copy the frame
- if torch.is_tensor(frame_source):
- # .contiguous() is CRITICAL here to fix the "shredded" look
- temp = frame_source.squeeze(0) if frame_source.ndim == 4 else frame_source
- img_cpu = temp.permute(1, 2, 0).contiguous().cpu().numpy()
- elif hasattr(frame_source, "download"):
- img_cpu = frame_source.download()
- else:
- img_cpu = np.ascontiguousarray(frame_source)
-
- # Fix Shape: restore spatial grid if flattened
- if img_cpu.ndim == 3 and img_cpu.shape[0] == 1:
- img_cpu = img_cpu.reshape((self.resize_h, self.resize_w, 3))
-
- # Fix Visibility: ONLY multiply if it's actually floating point
- # If uint8 is multiplied by 255, it wraps around and creates "neon" colors
- if img_cpu.dtype != np.uint8:
- if img_cpu.max() <= 1.0:
- img_cpu = (img_cpu * 255).clip(0, 255).astype(np.uint8)
- else:
- img_cpu = img_cpu.astype(np.uint8)
-
- # Color Space: Standardize to BGR for imwrite
- if len(img_cpu.shape) != 3:
- img_cpu = cv2.cvtColor(img_cpu, cv2.COLOR_GRAY2BGR)
-
- # Draw 8K Boxes (Scaled down)
- h_img, w_img = img_cpu.shape[:2]
- scale_x = w_img / self.frame_width
- scale_y = h_img / self.frame_height
-
- if bbs_full_res is not None:
- boxes = (
- bbs_full_res.cpu().tolist()
- if torch.is_tensor(bbs_full_res)
- else bbs_full_res
- )
- for box in boxes:
- x1, y1, x2, y2 = [
- int(box[0] * scale_x),
- int(box[1] * scale_y),
- int(box[2] * scale_x),
- int(box[3] * scale_y),
- ]
- cv2.rectangle(img_cpu, (x1, y1), (x2, y2), (0, 0, 255), 2)
-
- cv2.imwrite(str(debug_dir / f"analysis_{frame_num:04d}.jpg"), img_cpu)
-
- def debug_save_crops(self, cropped_batch, frame_num):
- """Saves the first 5 crops of a batch to the results directory."""
- debug_dir = self.result_dir / "debug_crops" / self._testMethodName
- debug_dir.mkdir(parents=True, exist_ok=True)
-
- # Only save for the first self.config.DEBUG_FRAME_LIMIT frames to avoid disk bloat
- if frame_num > self.config.DEBUG_FRAME_LIMIT:
- return
-
- for i, crop in enumerate(cropped_batch[: self.config.DEBUG_FRAME_LIMIT]):
- # Convert GPU Tensor [C, H, W] -> NumPy [H, W, C]
- if torch.is_tensor(crop):
- # Reverse normalization (* 255) and permute to BGR
- img = (crop.squeeze(0).permute(1, 2, 0) * 255).byte().cpu().numpy()
- else:
- img = crop
-
- cv2.imwrite(str(debug_dir / f"frame_{frame_num}_crop_{i}.jpg"), img)
-
- def debug_save_img(self, frame_source, frame_num):
- debug_dir = self.result_dir / "debug_test" / self._testMethodName
- debug_dir.mkdir(parents=True, exist_ok=True)
-
- if frame_num > self.config.DEBUG_FRAME_LIMIT:
- return
-
- # Download/Copy the frame
- if torch.is_tensor(frame_source):
- # .contiguous() is CRITICAL here to fix the "shredded" look
- temp = frame_source.squeeze(0) if frame_source.ndim == 4 else frame_source
- img_cpu = temp.permute(1, 2, 0).contiguous().cpu().numpy()
- elif hasattr(frame_source, "download"):
- img_cpu = frame_source.download()
- else:
- img_cpu = np.ascontiguousarray(frame_source)
-
- # Fix Shape: restore spatial grid if flattened
- if img_cpu.ndim == 3 and img_cpu.shape[0] == 1:
- img_cpu = img_cpu.reshape((self.resize_h, self.resize_w, 3))
-
- # Fix Visibility: ONLY multiply if it's actually floating point
- # If uint8 is multiplied by 255, it wraps around and creates "neon" colors
- if img_cpu.dtype != np.uint8:
- if img_cpu.max() <= 1.0:
- img_cpu = (img_cpu * 255).clip(0, 255).astype(np.uint8)
- else:
- img_cpu = img_cpu.astype(np.uint8)
-
- # Color Space: Standardize to BGR for imwrite
- if len(img_cpu.shape) == 3:
- pass
- else:
- img_cpu = cv2.cvtColor(img_cpu, cv2.COLOR_GRAY2BGR)
-
- cv2.imwrite(str(debug_dir / f"analysis_{frame_num:04d}.jpg"), img_cpu)
-
- # # FRAME PROCESSORS --------------------------------------------
- # def run_pipeline(self): #, pipeline_fn):
- # # n_frames = 0
- # num_objs = 0
- # self.frame_count_target = 0
- # self.next_process_idx = 0.0
- # self.frame_in_clip_count = 0
- # total_pipeline_time_ms = 0.0 # Track pure latency
- # coverage_percentages = []
- # self.component_stats = {"sf": [], "roi": [], "det": []}
- # self.gpu_event_buffer = {"sf": [], "roi": [], "det": []}
- # self.crops_per_frame_list = []
-
- # # PRE-SYNC: Ensure GPU is idle before timing starts
- # # if self.device_input == "cuda":
- # # # Pre-allocate CUDA events for isolated GPU timing
- # # # start_event = torch.cuda.Event(enable_timing=True)
- # # # end_event = torch.cuda.Event(enable_timing=True)
-
- # # if torch.cuda.is_available():
- # # torch.cuda.synchronize()
-
- # self.start()
- # time.sleep(0.1)
-
- # # start_time = time.perf_counter()
- # total_session_start = time.perf_counter()
-
- # while self.active: # and n_frames < 10*self.fps:
- # # ret, frame = self.cap.read()
- # device_frame, frame_num = self.reader.read()
- # if device_frame is None:
- # if self.reader.stopped:
- # self.active = False
- # break
- # continue
-
- # # n_frames += 1
- # self.frame_count += 1
- # is_target_frame = float(frame_num) >= self.next_process_idx
- # nob = 0
-
- # # if self.device_input == "cuda":
- # # # start_eve nt.record()
- # # nob, metrics = self.pipeline_fn(device_frame, frame_num, is_target_frame)
- # # # end_event.record()
- # # # torch.cuda.synchronize()
- # # # total_pipeline_time_ms += start_event.elapsed_time(end_event)
- # # else:
- # # start_t = time.perf_counter()
- # if self.device_input == "cuda":
- # # Record the reader's current layout availability milestone
- # curr_event = torch.cuda.Event()
- # curr_event.record()
-
- # # Instruct the isolated inference stream to wait for this specific frame context
- # self.inference_stream.wait_event(curr_event)
- # with torch.cuda.stream(self.inference_stream):
- # nob, metrics = self.pipeline_fn(device_frame, frame_num, is_target_frame)
- # # self.inference_stream.synchronize()
-
- # else:
- # nob, metrics = self.pipeline_fn(device_frame, frame_num, is_target_frame)
- # # total_pipeline_time_ms += (time.perf_counter() - start_t) * 1000
-
- # num_objs += nob
-
- # if is_target_frame and metrics != {}:
- # num_crops = len(metrics["bbs"]) if metrics["bbs"] is not None else 0
- # self.crops_per_frame_list.append(num_crops)
-
- # # self.component_stats["roi"].append(metrics["roi_time"])
-
- # # if self.device_input != "cuda":
- # if metrics.get("sf_time"):
- # self.component_stats["sf"].append(metrics["sf_time"])
- # if metrics.get("det_time"):
- # self.component_stats["det"].append(metrics["det_time"])
- # if metrics.get("roi_time"):
- # self.component_stats["roi"].append(metrics["roi_time"])
-
- # # Calculate coverage OUTSIDE the timed block to prevent interference
- # if self.config.sf_enabled and metrics.get("bbs") is not None:
- # cov = self.calculate_unique_coverage(metrics["bbs"])
- # coverage_percentages.append(cov)
-
- # # POST-SYNC: Ensure all GPU tasks finished before timing ends
- # if self.device_input == "cuda":
- # torch.cuda.synchronize()
-
- # # ACTUAL speed including all overhead
- # real_world_latency_ms = (time.perf_counter() - total_session_start) * 1000
-
- # # if self.device_input == "cuda":
- # # # Move GPU event timings into component_stats lists
- # # for key in ["sf", "det", "roi"]:
- # # for start, end in self.gpu_event_buffer[key]:
- # # self.component_stats[key].append(start.elapsed_time(end))
-
- # # Calculate Pure processing latency
- # # This excludes: Reader, Tensor2OpenCV, Queueing, and Video Writing
- # total_pipeline_time_ms = (
- # sum(self.component_stats["sf"])
- # + sum(self.component_stats["roi"])
- # + sum(self.component_stats["det"])
- # )
-
- # # assert num_objs > 0
- # if num_objs == 0:
- # print(f" [WARNING] No objects detected for {self._testMethodName}")
-
- # self._finalize_benchmarks(
- # self.frame_count,
- # num_objs,
- # total_pipeline_time_ms,
- # real_world_latency_ms,
- # coverage_percentages,
- # self.config.sf_enabled,
- # self.stat_frame_count,
- # self.stat_fps,
- # )
-
- def run_pipeline(self):
- num_objs = 0
- self.frame_count_target = 0
- self.next_process_idx = 0.0
- self.frame_in_clip_count = 0
- total_pipeline_time_ms = 0.0
- coverage_percentages = []
- self.component_stats = {"sf": [], "roi": [], "det": []}
- self.crops_per_frame_list = []
-
- self.step_size = (
- float(self.input_fps) / float(self.target_fps)
- if hasattr(self, "target_fps")
- else 1.0
- )
-
- self.start()
- time.sleep(0.1)
-
- total_session_start = time.perf_counter()
-
- while self.active:
- # 1. CAPTURE COMPLETE CYCLE OVERHEAD
- # t_cycle_start = time.perf_counter()
-
- device_frame, frame_num = self.reader.read()
- if device_frame is None:
- if self.reader is None or (
- hasattr(self.reader, "stopped") and self.reader.stopped
- ):
- # --- CRITICAL PIPELINE DRAIN START ---
- print(
- "[INFO] Reader reached EOF. Draining asynchronous workers and VRAM queues..."
- )
-
- # 1. Thread Pool Flush: Force the CPU thread to block here until
- # every single background pipeline task in the executor finishes.
- if hasattr(self, "executor") and self.executor:
- self.executor.shutdown(wait=True)
- # 2. Render Queue Flush: If you utilize an asynchronous video frame
- # saving worker thread, wait for its queue tasks to bottom out.
- if hasattr(self, "render_queue") and self.render_queue:
- while not self.render_queue.empty():
- time.sleep(0.01)
+ worker_p.start()
+ worker_p.join() # Block Pytest until the isolated pipeline finishes
- # 3. GPU Hardware Flush: Force the GPU to completely finish
- # all remaining background subtraction, crops, and YOLO operations.
- if self.device_input == "cuda":
- torch.cuda.synchronize()
+ # Extract and merge results
+ if not res_queue.empty():
+ result = res_queue.get()
- self.active = False
- break
- continue
-
- self.stat_start_time = time.perf_counter() # timing to display detection
- self.frame_count += 1
- # is_target_frame = float(frame_num) >= self.next_process_idx
- # CRITICAL CADENCE DRIFT FIX: If input FPS equals target FPS, or if Smart Filtering
- # is disabled (YOLO baseline), force is_target_frame to ALWAYS be True.
- # This completely bypasses floating-point accumulation drift errors.
- if (
- not self.config.sf_enabled
- or abs(float(self.input_fps) - float(self.target_fps)) < 0.01
- ):
- is_target_frame = True
- else:
- is_target_frame = float(frame_num) >= self.next_process_idx
-
- # 2. DISPATCH WORKLOADS ASYNCHRONOUSLY
- metrics = {}
- if is_target_frame:
- self.next_process_idx += self.step_size
-
- if self.device_input == "cuda":
- # Instantiate a lightweight hardware fence event object
- curr_event = torch.cuda.Event(enable_timing=False)
- # curr_event.record()
- # Record the exact milestone on the default stream right after fetching the frame data
- curr_event.record(torch.cuda.default_stream())
- self.inference_stream.wait_event(curr_event)
- with torch.cuda.stream(self.inference_stream):
- # 1. Isolate the incoming image buffer canvas
- isolated_device_frame = (
- device_frame.clone()
- if torch.is_tensor(device_frame)
- else device_frame.copy()
- )
-
- # 2. RUN BACKGROUND SUBTRACTION IMMEDIATELY ON THE PRODUCER TIMELINE
- # This guarantees that the mask matches this exact frame_num before threads overlap!
- nob, metrics = self.pipeline_fn(
- isolated_device_frame,
- frame_num,
- is_target_frame,
- self.stat_start_time,
- )
- else:
- # 1. Isolate the incoming image buffer canvas
- isolated_device_frame = (
- device_frame.clone()
- if torch.is_tensor(device_frame)
- else device_frame.copy()
- )
-
- # 2. RUN BACKGROUND SUBTRACTION IMMEDIATELY ON THE PRODUCER TIMELINE
- # This guarantees that the mask matches this exact frame_num b
- nob, metrics = self.pipeline_fn(
- isolated_device_frame,
- frame_num,
- is_target_frame,
- self.stat_start_time,
- )
-
- num_objs += nob
- # else:
- # # Process background execution context for skipped frames
- # nob, metrics = self.pipeline_fn(
- # device_frame, frame_num, is_target_frame, self.stat_start_time
- # )
-
- # 3. ENFORCE UNIFIED HARDWARE TIMING BARRIER
- # if self.device_input == "cuda":
- # torch.cuda.synchronize()
-
- # t_cycle_end = time.perf_counter()
- # cycle_total_ms = (t_cycle_end - t_cycle_start) * 1000.0
-
- # 4. ALLOCATE ALL TRACKING TIMINGS ACCURATELY
- # if is_target_frame:
- if metrics != {}:
- num_crops = len(metrics["bbs"]) if metrics["bbs"] is not None else 0
- self.crops_per_frame_list.append(num_crops)
-
- if self.config.sf_enabled:
- self.component_stats["sf"].append(metrics["sf_time"])
- self.component_stats["roi"].append(metrics["roi_time"])
- else:
- # Full-frame YOLO baseline accounts for total data movement cycle
- # self.component_stats["det"].append(cycle_total_ms)
- self.component_stats["sf"].append(0.0)
- self.component_stats["roi"].append(0.0)
-
- self.component_stats["det"].append(metrics["det_time"])
-
- if self.config.sf_enabled and metrics.get("bbs") is not None:
- cov = self.calculate_unique_coverage(metrics["bbs"])
- coverage_percentages.append(cov)
- # else:
- # # CRITICAL METRICS FIX: If Smart Filtering runs on a skipped frame,
- # # its mask generation overhead MUST be captured and tracked!
- # if self.config.sf_enabled and metrics != {}:
- # self.component_stats["sf"].append(metrics["sf_time"])
-
- if self.device_input == "cuda":
- torch.cuda.synchronize()
-
- real_world_latency_ms = (time.perf_counter() - total_session_start) * 1000.0
-
- total_pipeline_time_ms = (
- sum(self.component_stats["sf"])
- + sum(self.component_stats["roi"])
- + sum(self.component_stats["det"])
- )
-
- if num_objs == 0:
- print(f" [WARNING] No objects detected for {self._testMethodName}")
-
- self._finalize_benchmarks(
- # self.frame_count,
- num_objs,
- total_pipeline_time_ms,
- real_world_latency_ms,
- coverage_percentages,
- self.config.sf_enabled,
- self.stat_frame_count,
- self.stat_fps,
- )
-
- # # TESTS --------------------------------------------
- # def pipeline_fn(self, device_frame, overall_frame_num, is_target_frame):
- # num_objs = 0
- # metrics = {"sf_time": 0, "roi_time": 0, "det_time": 0, "bbs": None}
-
- # # Initialize timing event handle pairs
- # sf_start, sf_end = torch.cuda.Event(enable_timing=True), torch.cuda.Event(enable_timing=True)
- # roi_start, roi_end = torch.cuda.Event(enable_timing=True), torch.cuda.Event(enable_timing=True)
- # det_start, det_end = torch.cuda.Event(enable_timing=True), torch.cuda.Event(enable_timing=True)
-
- # # --- 1. MOTION MASK GENERATION GATE ---
- # if self.config.sf_enabled:
- # if self.device_input == "cuda":
- # sf_start.record(self.inference_stream)
- # inf_data = self.rbtd_full_gpu(device_frame)
- # sf_end.record(self.inference_stream)
- # else:
- # t_start = time.perf_counter()
- # inf_data = self.rbtd_full_cpu(device_frame)
- # metrics["sf_time"] = (time.perf_counter() - t_start) * 1000.0
- # else:
- # inf_data = {}
-
- # # --- PIPELINE AT TARGET RATE ---
- # if not is_target_frame:
- # # If skipping, pull outstanding execution records immediately
- # if self.device_input == "cuda" and self.config.sf_enabled:
- # self.inference_stream.synchronize()
- # metrics["sf_time"] = sf_start.elapsed_time(sf_end)
- # return num_objs, metrics
-
- # self.next_process_idx += self.step_size
- # self.frame_count_target += 1
- # self.frame_in_clip_count += 1
- # inf_data["frameNum"] = self.frame_count_target
-
- # # --- 2. FULL-RESOLUTION ROI EXTRACTION MAPS ---
- # bbs_full_res = None
- # if self.config.sf_enabled:
- # if self.device_input == "cuda":
- # roi_start.record(self.inference_stream)
- # bbs_full_res = self.get_gpu_rois(
- # inf_data["full_frame"],
- # self.frame_count_target,
- # inf_data["mask"],
- # )
- # roi_end.record(self.inference_stream)
- # metrics["bbs"] = bbs_full_res
- # else:
- # t_start = time.perf_counter()
- # bbs_full_res = self.get_cpu_rois(
- # inf_data["full_frame"],
- # self.frame_count_target,
- # inf_data["mask"],
- # )
- # metrics["roi_time"] = (time.perf_counter() - t_start) * 1000.0
- # metrics["bbs"] = bbs_full_res
-
- # if self.config.DEBUG_FLAG:
- # if self.device_input == "cuda":
- # torch.cuda.stream(self.inference_stream)
- # torch.cuda.synchronize()
- # display_source = inf_data["mask"]
- # self.debug_save_mask(
- # display_source, self.frame_count_target, rois=bbs_full_res
- # )
-
- # clean_bbs = []
- # if self.config.sf_enabled and bbs_full_res is not None:
- # if torch.is_tensor(bbs_full_res):
- # clean_bbs = bbs_full_res.detach().cpu().numpy()
- # else:
- # clean_bbs = np.array(bbs_full_res)
-
- # # --- 3. MODEL INFERENCE TIMING BLOCK ---
- # if self.device_input == "cuda":
- # det_start.record(self.inference_stream)
- # else:
- # t_start = time.perf_counter()
-
- # if self.config.DETECTION_TYPE != "motion":
- # det_frame = inf_data["full_frame"] if "full_frame" in inf_data else device_frame
- # merged = clean_bbs if self.config.sf_enabled else None
- # metadata, _ = self.get_detections(
- # det_frame,
- # self.frame_in_clip_count,
- # merged=merged,
- # thickness=self.config.THICKNESS,
- # device_input=self.config.device_input,
- # )
- # num_objs = len(metadata.keys())
- # else:
- # num_objs = len(clean_bbs)
- # metadata = clean_bbs
-
- # if self.device_input == "cuda":
- # det_end.record(self.inference_stream)
-
- # # CRITICAL PERFORMANCE SYNCHRONIZATION POINT:
- # # We execute a single stream synchronization barrier here at the end of the entire loop.
- # # This allows the GPU kernels to run overlapped and fully concurrently!
- # self.inference_stream.synchronize()
-
- # # Unpack hardware timings efficiently
- # if self.config.sf_enabled:
- # metrics["sf_time"] = sf_start.elapsed_time(sf_end)
- # metrics["roi_time"] = roi_start.elapsed_time(roi_end)
- # metrics["det_time"] = det_start.elapsed_time(det_end)
- # metrics["bbs"] = bbs_full_res
- # else:
- # metrics["det_time"] = (time.perf_counter() - t_start) * 1000.0
-
- # # --- 4. RENDER WORKER PREPARATION ---
- # display_source = inf_data["full_frame"] if (inf_data and "full_frame" in inf_data) else device_frame
-
- # if self.device_input == "cuda":
- # gpu_resized = F.interpolate(
- # display_source.unsqueeze(0).float(),
- # size=(self.disp_h, self.disp_w),
- # mode="bilinear",
- # align_corners=False,
- # ).squeeze(0).contiguous()
- # disp_frame = np.copy(tensor2opencv(gpu_resized, self.config.device_input, is_bgr=True))
- # else:
- # cpu_resized = cv2.resize(device_frame, (self.disp_w, self.disp_h))
- # disp_frame = np.copy(tensor2opencv(cpu_resized, self.config.device_input, is_bgr=True))
-
- # data_to_draw = clean_bbs if self.config.DETECTION_TYPE == "motion" else metadata
-
- # try:
- # if hasattr(self, "render_queue") and getattr(self, "render_queue", None) is not None and not self.render_queue.full():
- # self.render_queue.put(
- # (
- # disp_frame,
- # inf_data["frameNum"] if "frameNum" in inf_data else self.frame_count_target,
- # data_to_draw,
- # self.label_source,
- # )
- # )
- # if self.config.DEBUG_FLAG:
- # self.debug_save_img(disp_frame, self.frame_count_target)
- # self.debug_save_img_roi(disp_frame, bbs_full_res, self.frame_count_target)
- # except queue.Full:
- # pass
-
- # self.update_frame()
- # return num_objs, metrics
-
- def pipeline_fn(
- self, device_frame, overall_frame_num, is_target_frame, stat_start_time
- ):
- num_objs = 0
- metrics = {"sf_time": 0, "roi_time": 0, "det_time": 0, "bbs": None}
-
- # Pre-allocate non-blocking event records to extract clean GPU timings
- sf_start, sf_end = (
- torch.cuda.Event(enable_timing=True),
- torch.cuda.Event(enable_timing=True),
- )
- roi_start, roi_end = (
- torch.cuda.Event(enable_timing=True),
- torch.cuda.Event(enable_timing=True),
- )
- det_start, det_end = (
- torch.cuda.Event(enable_timing=True),
- torch.cuda.Event(enable_timing=True),
- )
-
- # --- 1. MOTION MASK GENERATION GATE ---
- if self.config.sf_enabled:
- if self.device_input == "cuda":
- sf_start.record(self.inference_stream)
- inf_data = self.rbtd_full_gpu(device_frame)
- sf_end.record(self.inference_stream)
- else:
- t_start = time.perf_counter()
- inf_data = self.rbtd_full_cpu(device_frame)
- metrics["sf_time"] = (time.perf_counter() - t_start) * 1000.0
- else:
- inf_data = {}
-
- # --- PIPELINE AT TARGET RATE ---
- if not is_target_frame:
- # Safe, non-blocking timing extraction for skipped frames
- if self.device_input == "cuda" and self.config.sf_enabled:
- torch.cuda.synchronize()
- metrics["sf_time"] = sf_start.elapsed_time(sf_end)
- return num_objs, metrics
-
- self.frame_count_target += 1
- self.frame_in_clip_count += 1
- inf_data["frameNum"] = self.frame_count_target
-
- # --- 2. FULL-RESOLUTION ROI EXTRACTION MAPS ---
- bbs_full_res = None
- if self.config.sf_enabled:
- if self.device_input == "cuda":
- roi_start.record(self.inference_stream)
- bbs_full_res = self.get_gpu_rois(
- inf_data["full_frame"],
- self.frame_count_target,
- inf_data["mask"],
+ if result["status"] == "error":
+ pytest.fail(
+ f"Pipeline crashed in worker process:\n{result.get('error')}\n{result.get('traceback')}"
)
- roi_end.record(self.inference_stream)
- metrics["bbs"] = bbs_full_res
else:
- t_start = time.perf_counter()
- bbs_full_res = self.get_cpu_rois(
- inf_data["full_frame"],
- self.frame_count_target,
- inf_data["mask"],
- )
- metrics["roi_time"] = (time.perf_counter() - t_start) * 1000.0
- metrics["bbs"] = bbs_full_res
-
- if self.config.DEBUG_FLAG:
- # if self.device_input == "cuda":
- # # Isolate data capturing using an asynchronous memory clone operation
- # # This safely copies data without dropping your multi-stream execution pipeline concurrency
- # display_source = inf_data["mask"].clone().to("cpu", non_blocking=True)
- # else:
- display_source = inf_data["mask"]
- # self.inference_stream.synchronize()
- # display_source = inf_data["mask"]
- self.debug_save_mask(
- display_source, self.frame_count_target, rois=bbs_full_res
- )
+ metrics = result["metrics"]
- clean_bbs = []
- if self.config.sf_enabled and bbs_full_res is not None:
- if torch.is_tensor(bbs_full_res):
- clean_bbs = bbs_full_res.detach().cpu().numpy()
- else:
- clean_bbs = np.array(bbs_full_res)
+ self.__class__.benchmarks.append(metrics)
- # --- 3. MODEL INFERENCE TIMING BLOCK ---
- if self.device_input == "cuda":
- det_start.record(self.inference_stream)
- else:
- t_start = time.perf_counter()
-
- if self.config.DETECTION_TYPE != "motion":
- # det_frame = (
- # inf_data["full_frame"] if "full_frame" in inf_data else device_frame
- # )
- # Isolate your image buffer array view to prevent upstream reader pointer races
- if "full_frame" in inf_data:
- det_frame = inf_data["full_frame"]
- else:
- det_frame = (
- device_frame.clone()
- if torch.is_tensor(device_frame)
- else device_frame.copy()
+ main_app_logger.info(
+ f"[TEST HARNESS] Worker returned successfully. Display FPS: {metrics.get('Display FPS')}"
)
- merged = clean_bbs if self.config.sf_enabled else None
- metadata, _ = self.get_detections(
- det_frame,
- self.frame_in_clip_count,
- merged=merged,
- thickness=self.config.THICKNESS,
- device_input=self.config.device_input,
- )
- num_objs = len(metadata.keys())
- else:
- num_objs = len(clean_bbs)
- metadata = clean_bbs
-
- if self.device_input == "cuda":
- det_end.record(self.inference_stream)
-
- torch.cuda.synchronize()
-
- # Extract hardware timings smoothly without stalling mid-run
- if self.config.sf_enabled:
- metrics["sf_time"] = sf_start.elapsed_time(sf_end)
- metrics["roi_time"] = roi_start.elapsed_time(roi_end)
- metrics["det_time"] = det_start.elapsed_time(det_end)
- metrics["bbs"] = bbs_full_res
- else:
- metrics["det_time"] = (time.perf_counter() - t_start) * 1000.0
-
- # --- 4. RENDER WORKER PREPARATION ---
- display_source = (
- inf_data["full_frame"]
- if (inf_data and "full_frame" in inf_data)
- else device_frame
- )
-
- if self.device_input == "cuda":
- gpu_resized = (
- F.interpolate(
- display_source.unsqueeze(0).float(),
- size=(self.disp_h, self.disp_w),
- mode="bilinear",
- align_corners=False,
+ # Basic functionality assertions
+ assert metrics is not None, "Metrics dictionary should not be None."
+ assert int(metrics.get("Output Frames", 0)) > 0, (
+ "No frames were written to output."
)
- .squeeze(0)
- .contiguous()
- )
- disp_frame = np.copy(
- tensor2opencv(gpu_resized, self.config.device_input, is_bgr=True)
- )
else:
- cpu_resized = cv2.resize(device_frame, (self.disp_w, self.disp_h))
- disp_frame = np.copy(
- tensor2opencv(cpu_resized, self.config.device_input, is_bgr=True)
- )
-
- data_to_draw = clean_bbs if self.config.DETECTION_TYPE == "motion" else metadata
-
- # try:
- # if (
- # hasattr(self, "render_queue")
- # and getattr(self, "render_queue", None) is not None
- # and not self.render_queue.full()
- # ):
- # self.render_queue.put(
- # (
- # disp_frame,
- # inf_data["frameNum"]
- # if "frameNum" in inf_data
- # else self.frame_count_target,
- # data_to_draw,
- # self.label_source,
- # )
- # )
- # if self.config.DEBUG_FLAG:
- # self.debug_save_img(disp_frame, self.frame_count_target)
- # self.debug_save_img_roi(
- # disp_frame, bbs_full_res, self.frame_count_target
- # )
- # except queue.Full:
- # pass
-
- if (
- hasattr(self, "render_queue")
- and getattr(self, "render_queue", None) is not None
- ):
- self.render_queue.put(
- (
- disp_frame,
- inf_data["frameNum"]
- if "frameNum" in inf_data
- else self.frame_count_target,
- data_to_draw,
- self.label_source,
- )
- )
- if self.config.DEBUG_FLAG:
- self.debug_save_img(disp_frame, self.frame_count_target)
- self.debug_save_img_roi(disp_frame, bbs_full_res, self.frame_count_target)
-
- self.update_frame(stat_start_time)
- return num_objs, metrics
-
- # def sf_cpu_pipeline_fn(self, device_frame, overall_frame_num, is_target_frame):
- # num_objs = 0
- # metrics = {"sf_time": 0, "roi_time": 0, "det_time": 0, "bbs": None}
-
- # # Smart Filtering (Resize / Background Subtraction / Threshold / Dilate)
- # t_start = time.perf_counter()
- # inf_data = self.rbtd_full_cpu(device_frame)
- # metrics["sf_time"] = (time.perf_counter() - t_start) * 1000
-
- # # Only keep frames for TARGET_FPS
- # # Videos are written for target frames only
- # if is_target_frame:
- # self.next_process_idx += self.step_size
- # self.frame_count_target += 1 # 1-indexed
-
- # if not inf_data or inf_data["mask"] is None:
- # return num_objs, metrics
-
- # metadata = {}
- # bbs_to_send = []
- # data_to_draw = []
- # if inf_data:
- # inf_data["frameNum"] = self.frame_count_target
-
- # # Get ROIs
- # t_start = time.perf_counter()
- # bbs_full_res = self.get_cpu_rois(
- # inf_data["full_frame"],
- # self.frame_count_target,
- # inf_data["mask"],
- # )
- # metrics["roi_time"] = (time.perf_counter() - t_start) * 1000
- # metrics["bbs"] = bbs_full_res
-
- # if self.config.DEBUG_FLAG:
- # display_source = inf_data["mask"]
- # self.debug_save_mask(
- # display_source, self.frame_count_target, rois=bbs_full_res
- # )
-
- # if bbs_full_res is not None and len(bbs_full_res) == 0:
- # return num_objs, metrics
-
- # # if self.config.DEBUG_FLAG or self.config.DETECTION_TYPE == "motion":
- # # display_source = (
- # # inf_data["full_frame"]
- # # if (inf_data and "full_frame" in inf_data)
- # # else device_frame
- # # )
- # # cpu_resized = cv2.resize(
- # # display_source,
- # # (self.resize_w, self.resize_h),
- # # interpolation=cv2.INTER_NEAREST,
- # # )
-
- # t_start = time.perf_counter()
- # if self.config.DETECTION_TYPE == "motion":
- # # Motion Mode: Prepare boxes for drawing
- # if (
- # bbs_full_res is not None
- # and bbs_full_res.ndim == 2
- # and bbs_full_res.size(0) > 0
- # ):
- # scaled_resized_bbs = bbs_full_res / self.scales_tensor
- # bbs_to_send = scaled_resized_bbs.cpu().tolist()
- # else:
- # bbs_to_send = []
-
- # data_to_draw = bbs_to_send
- # num_objs = len(bbs_full_res)
- # else:
- # # Object Mode: Run YOLO and prepare metadata
- # det_frame = inf_data["full_frame"] if inf_data else device_frame
- # metadata, _ = self.get_detections(
- # det_frame,
- # self.frame_count_target, # Frame used in metadata
- # merged=bbs_full_res,
- # thickness=self.config.THICKNESS,
- # device_input=self.config.device_input,
- # )
- # data_to_draw = metadata
- # num_objs = len(metadata.keys())
-
- # metrics["det_time"] = (time.perf_counter() - t_start) * 1000
-
- # # --- OFFLOAD TO QUEUE (Run for EVERY target frame) ---
- # if not self.config.TEST_MODE:
- # display_source = (
- # inf_data["full_frame"]
- # if (inf_data and "full_frame" in inf_data)
- # else device_frame
- # )
- # cpu_resized = cv2.resize(display_source, (self.resize_w, self.resize_h))
- # display_frame = tensor2opencv(
- # cpu_resized, self.config.device_input, is_bgr=True
- # )
- # self.render_queue.put((display_frame, data_to_draw, self.label_source))
-
- # if self.config.DEBUG_FLAG:
- # cpu_resized = cv2.resize(
- # display_source, (self.resize_w, self.resize_h)
- # )
- # self.debug_save_img(cpu_resized, self.frame_count_target)
- # self.debug_save_img_roi(
- # cpu_resized, bbs_full_res, self.frame_count_target
- # )
-
- # return num_objs, metrics
-
- # def sf_gpu_pipeline_fn(self, device_frame, overall_frame_num, is_target_frame):
- # num_objs = 0
- # metrics = {"sf_time": 0, "roi_time": 0, "det_time": 0, "bbs": None}
-
- # # Smart Filtering (Resize / Background Subtraction / Threshold / Dilate)
- # sf_start, sf_end = (
- # torch.cuda.Event(enable_timing=True),
- # torch.cuda.Event(enable_timing=True),
- # )
- # sf_start.record()
- # inf_data = self.rbtd_full_gpu(device_frame)
- # sf_end.record()
- # self.gpu_event_buffer["sf"].append((sf_start, sf_end))
-
- # # Only keep frames for TARGET_FPS
- # # Videos are written for target frames only
- # if is_target_frame:
- # self.next_process_idx += self.step_size
- # self.frame_count_target += 1 # 1-indexed
-
- # if not inf_data or inf_data["mask"] is None:
- # return num_objs, metrics
-
- # metadata = {}
- # bbs_to_send = []
- # data_to_draw = []
- # if inf_data:
- # inf_data["frameNum"] = self.frame_count_target
-
- # # Get ROIs
- # # time.perf_counter accurate since called self.bgs_stream.waitForCompletion()
- # # t_start = time.perf_counter()
- # roi_start = torch.cuda.Event(enable_timing=True)
- # roi_end = torch.cuda.Event(enable_timing=True)
- # roi_start.record()
- # bbs_full_res = self.get_gpu_rois(
- # inf_data["full_frame"],
- # self.frame_count_target,
- # inf_data["mask"],
- # )
- # # metrics["roi_time"] = (time.perf_counter() - t_start) * 1000
- # # self.gpu_event_buffer["roi"].append((roi_start, roi_end))
- # roi_end.record()
- # self.gpu_event_buffer["roi"].append((roi_start, roi_end))
- # metrics["bbs"] = bbs_full_res
-
- # if self.config.DEBUG_FLAG:
- # self.bgs_stream.waitForCompletion()
- # display_source = inf_data["mask"]
- # self.debug_save_mask(
- # display_source, self.frame_count_target, rois=bbs_full_res
- # )
-
- # if bbs_full_res is not None and len(bbs_full_res) == 0:
- # return num_objs, metrics
-
- # det_start, det_end = (
- # torch.cuda.Event(enable_timing=True),
- # torch.cuda.Event(enable_timing=True),
- # )
- # det_start.record()
- # if self.config.DETECTION_TYPE == "motion":
- # # Motion Mode: Prepare boxes for drawing
- # if (
- # bbs_full_res is not None
- # and bbs_full_res.ndim == 2
- # and bbs_full_res.size(0) > 0
- # ):
- # scaled_resized_bbs = bbs_full_res / self.scales_tensor
- # bbs_to_send = scaled_resized_bbs.detach().cpu().tolist()
- # else:
- # bbs_to_send = []
- # data_to_draw = bbs_to_send
- # num_objs = len(bbs_to_send)
- # else:
- # # Object Mode: Run YOLO and prepare metadata
- # det_frame = (
- # inf_data["full_frame"] if inf_data else device_frame
- # ) # RGB
- # metadata, _ = self.get_detections(
- # det_frame,
- # self.frame_count_target,
- # merged=bbs_full_res,
- # thickness=self.config.THICKNESS,
- # device_input=self.config.device_input,
- # )
- # data_to_draw = metadata
- # num_objs = len(metadata.keys())
-
- # det_end.record()
- # self.gpu_event_buffer["det"].append((det_start, det_end))
-
- # # --- OFFLOAD TO QUEUE (Run for EVERY target frame) ---
- # if not self.config.TEST_MODE:
- # display_source = (
- # inf_data["full_frame"]
- # if (inf_data and "full_frame" in inf_data)
- # else device_frame
- # ) # RGB
- # gpu_resized = F.interpolate(
- # display_source.unsqueeze(0).half(),
- # size=(self.resize_h, self.resize_w),
- # mode="bilinear",
- # align_corners=False,
- # ).squeeze(0)
- # display_frame = tensor2opencv(
- # gpu_resized, self.config.device_input, is_bgr=True
- # )
- # # display_frame = self.tensor2opencv_gpu(gpu_resized)
- # self.render_queue.put((display_frame, data_to_draw, self.label_source))
-
- # if self.config.DEBUG_FLAG:
- # gpu_resized = F.interpolate(
- # display_source.unsqueeze(0).float(),
- # size=(self.resize_h, self.resize_w),
- # mode="bilinear",
- # align_corners=False,
- # ) # RGB
- # self.debug_save_img(gpu_resized, self.frame_count_target)
- # self.debug_save_img_roi(
- # gpu_resized, bbs_full_res, self.frame_count_target
- # )
-
- # # Async metrics stay at 0 for fairness; collected at the end of the video
- # metrics["sf_time"] = 0
- # metrics["roi_time"] = 0
- # metrics["det_time"] = 0
- # return num_objs, metrics
-
- # def yolo_cpu_pipeline_fn(self, device_frame, overall_frame_num, is_target_frame):
- # num_objs = 0
- # metadata = {}
- # metrics = {"sf_time": 0, "roi_time": 0, "det_time": 0, "bbs": None}
-
- # # Only keep frames for TARGET_FPS
- # if is_target_frame:
- # self.next_process_idx += self.step_size
- # self.frame_count_target += 1 # 1-indexed
-
- # # Get detection at original resolution (No SF)
- # t_start = time.perf_counter()
- # metadata, _ = self.get_detections(
- # device_frame,
- # self.frame_count_target,
- # thickness=self.config.THICKNESS,
- # device_input=self.config.device_input,
- # )
- # num_objs = len(metadata.keys())
- # metrics["det_time"] = (time.perf_counter() - t_start) * 1000
-
- # if not self.config.TEST_MODE:
- # cpu_resized = cv2.resize(device_frame, (self.resize_w, self.resize_h))
- # display_frame = tensor2opencv(
- # cpu_resized, self.config.device_input, is_bgr=True
- # )
- # self.render_queue.put((display_frame, metadata, self.label_source))
-
- # return num_objs, metrics
-
- # def yolo_gpu_pipeline_fn(self, frame_raw, overall_frame_num, is_target_frame):
- # num_objs = 0
- # metadata = {}
- # metrics = {"sf_time": 0, "roi_time": 0, "det_time": 0, "bbs": None}
-
- # # Only keep frames for TARGET_FPS
- # if is_target_frame:
- # self.next_process_idx += self.step_size
- # self.frame_count_target += 1 # 1-indexed
-
- # # Get detection at original resolution (No SF)
- # det_start, det_end = (
- # torch.cuda.Event(enable_timing=True),
- # torch.cuda.Event(enable_timing=True),
- # )
- # det_start.record()
- # metadata, _ = self.get_detections(
- # frame_raw,
- # self.frame_count_target,
- # thickness=self.config.THICKNESS,
- # device_input=self.config.device_input,
- # )
- # num_objs = len(metadata.keys())
- # det_end.record()
- # self.gpu_event_buffer["det"].append((det_start, det_end))
-
- # # --- OFFLOAD TO QUEUE (Run for EVERY target frame) ---
- # if not self.config.TEST_MODE:
- # gpu_resized = F.interpolate(
- # frame_raw.unsqueeze(0).half(),
- # size=(self.resize_h, self.resize_w),
- # mode="bilinear",
- # align_corners=False,
- # ).squeeze(0)
- # display_frame = tensor2opencv(
- # gpu_resized, self.config.device_input, is_bgr=True
- # )
- # # display_frame = self.tensor2opencv_gpu(gpu_resized)
- # self.render_queue.put((display_frame, metadata, self.label_source))
-
- # # Async metrics stay at 0 for fairness; collected at the end of the video
- # metrics["sf_time"] = 0
- # metrics["det_time"] = 0
- # return num_objs, metrics
-
-
-# INHERIT METHODS FROM HANDLERS -----------------------------------------------------------------
-# TestSmartFilteringDetections.setup_reader = DeviceBaseHandler.setup_reader
-# TestSmartFilteringDetections.get_frameWH = DeviceBaseHandler.get_frameWH
-# TestSmartFilteringDetections.initialize_variables = (
-# DeviceBaseHandler.initialize_variables
-# )
-# TestSmartFilteringDetections.filter_contained_boxes = (
-# DeviceBaseHandler.filter_contained_boxes
-# )
-# TestSmartFilteringDetections.get_detections = DeviceBaseHandler.get_detections
-# TestSmartFilteringDetections.prepare_pipeline = DeviceBaseHandler.prepare_pipeline
-# TestSmartFilteringDetections.get_gpu_rois_by_area = (
-# DeviceBaseHandler.get_gpu_rois_by_area
-# )
-# TestSmartFilteringDetections.get_gpu_rois = DeviceBaseHandler.get_gpu_rois
-# TestSmartFilteringDetections.get_cpu_rois = DeviceBaseHandler.get_cpu_rois
-# TestSmartFilteringDetections.run_model = DeviceBaseHandler.run_model
-# TestSmartFilteringDetections.model_warmup = DeviceBaseHandler.model_warmup
-
-# # TestSmartFilteringDetections.cleanup_gpu = GPUStreamHandler.cleanup_gpu
-# # TestSmartFilteringDetections.rbtd_full_gpu = GPUStreamHandler.rbtd_full_gpu
-# # TestSmartFilteringDetections.prepare_gpu_pipeline = (
-# # GPUStreamHandler.prepare_gpu_pipeline
-# # )
-# # TestSmartFilteringDetections.allocate_gpu = GPUStreamHandler.allocate_gpu
-# TestSmartFilteringDetections.gpu_warmup = GPUStreamHandler.gpu_warmup
-# TestSmartFilteringDetections.apply_background_subtraction_gpu = (
-# GPUStreamHandler.apply_background_subtraction_gpu
-# )
-
-# TestSmartFilteringDetections.cleanup_cpu = CPUStreamHandler.cleanup_cpu
-# # TestSmartFilteringDetections.rbtd_full_cpu = CPUStreamHandler.rbtd_full_cpu
-# # TestSmartFilteringDetections.prepare_cpu_pipeline = (
-# # CPUStreamHandler.prepare_cpu_pipeline
-# # )
-# # TestSmartFilteringDetections.allocate_cpu = CPUStreamHandler.allocate_cpu
-# TestSmartFilteringDetections.apply_background_subtraction_cpu = (
-# CPUStreamHandler.apply_background_subtraction_cpu
-# )
+ pytest.fail("Worker process died unexpectedly without returning metrics.")
+# =========================================================================
+# MAIN
+# =========================================================================
def get_pytest_filter_expression(args):
- print("\n" + "=" * 50)
- print("TARGET SELECTION PREVIEW")
- print("=" * 50)
+ main_app_logger.info("=" * 50)
+ main_app_logger.info("TARGET SELECTION PREVIEW")
+ main_app_logger.info("=" * 50)
- filter_expression = "test_detections"
+ filter_expression = Path(__file__).stem
# applied_subs = []
if args.sf_enabled is not None:
@@ -2076,14 +677,14 @@ def get_pytest_filter_expression(args):
# Target hardware context selection filter applies globally across all test cases
if args.device.lower() != "all":
- print(f" 💻 Hardware Context Constraint: {args.device.upper()}")
+ main_app_logger.info(f" 💻 Hardware Context Constraint: {args.device.upper()}")
filter_expression = f"({filter_expression}) and {args.device.lower()}"
else:
- print(" 💻 Hardware Context Constraint: ALL AVAILABLE")
+ main_app_logger.info(" 💻 Hardware Context Constraint: ALL AVAILABLE")
- print("=" * 50)
- print(f"COMPILED PYTEST KEYWORD EXPRESSION:\n 👉 {filter_expression}")
- print("=" * 50 + "\n")
+ main_app_logger.info("=" * 50)
+ main_app_logger.info(f"COMPILED PYTEST KEYWORD EXPRESSION: {filter_expression}")
+ main_app_logger.info("=" * 50)
return filter_expression
@@ -2116,6 +717,13 @@ def get_pytest_filter_expression(args):
)
# Filter tests
+ # parser.add_argument(
+ # "--device",
+ # type=str,
+ # default="all",
+ # choices=["cpu", "gpu", "all"],
+ # help="Filter by device (cpu or gpu)",
+ # )
parser.add_argument(
"--type",
type=str,
@@ -2123,13 +731,6 @@ def get_pytest_filter_expression(args):
dest="detection_type",
help="Filter by detection type (object or motion)",
)
- parser.add_argument(
- "--device",
- type=str,
- default="all",
- choices=["cpu", "gpu", "all"],
- help="Filter by device (cpu or gpu)",
- )
parser.add_argument(
"--sf",
action="store_true",
@@ -2137,6 +738,13 @@ def get_pytest_filter_expression(args):
dest="sf_enabled",
help="Filter by Smart Filtering",
)
+ parser.add_argument(
+ "--no-sf",
+ action="store_false",
+ default=None,
+ dest="sf_enabled",
+ help="Filter by Smart Filtering",
+ )
# DEBUGGING
parser.add_argument(
@@ -2152,7 +760,15 @@ def get_pytest_filter_expression(args):
help="Number of frames used for debugging [Default: 100]",
)
+ # PROFILING
+ parser.add_argument(
+ "--profile",
+ action="store_true",
+ help="Enable profiling",
+ )
+
args = parser.parse_args()
+ args.device = "gpu"
# UPDATE ENVIRONMENTAL VARIABLES
os.environ["VIDEO_FILENAME"] = args.source
@@ -2160,6 +776,7 @@ def get_pytest_filter_expression(args):
os.environ["MODEL_NAME"] = args.model_name
os.environ["DEBUG"] = "1" if args.debug else "0"
os.environ["DEBUG_FRAME_LIMIT"] = str(args.debug_frame_limit)
+ os.environ["ENABLE_PROFILING"] = "True" if args.profile else "False"
# detection_type, device, sf_enabled
filter_expression = get_pytest_filter_expression(args)
@@ -2171,9 +788,16 @@ def get_pytest_filter_expression(args):
"-s",
"-v",
# "--log-cli-level=DEBUG",
+ "-W",
+ "ignore:Exception ignored in.*SharedMemory.__del__:UserWarning",
+ # Target the exact module rewrite warning path inside the configuration framework
+ "-W",
+ "ignore:Module already imported so cannot be rewritten; anyio:_pytest.warning_types.PytestAssertRewriteWarning",
+ # Bypass Pytest's internal proxy
+ "--capture=no",
__file__,
]
- print(f"Launching tests for {args.source}")
+ # main_app_logger.info(f"Launching tests for {args.source}")
sys.exit(pytest.main(pytest_args))
diff --git a/fastapi/tests/test_eval.py b/fastapi/tests/test_eval.py
new file mode 100644
index 0000000..9ce33e6
--- /dev/null
+++ b/fastapi/tests/test_eval.py
@@ -0,0 +1,807 @@
+# ==============================================================================
+# SUPPRESS WARNINGS
+# import warnings
+
+import pytest
+
+# warnings.filterwarnings(
+# "ignore", category=FutureWarning, message=".*reduce_op` is deprecated.*"
+# )
+
+pytestmark = [
+ pytest.mark.filterwarnings(
+ "ignore:.*anyio:_pytest.warning_types.PytestAssertRewriteWarning"
+ ),
+ pytest.mark.filterwarnings(
+ "ignore:Context managers for TensorRT types are deprecated:DeprecationWarning"
+ ),
+ pytest.mark.filterwarnings(
+ "ignore:Exception ignored in.*SharedMemory.__del__:UserWarning"
+ ),
+ # Global message fallback pattern captures local execution frames
+ pytest.mark.filterwarnings("ignore:.*reduce_op.*:FutureWarning"),
+]
+
+# ==============================================================================
+# LOGGING
+import logging
+import os
+import sys
+
+logging.basicConfig(
+ level=logging.INFO,
+ # format="%(asctime)s - %(name)s - %(levelname)s - %(message)s",
+ format="%(asctime)s [%(levelname)s] %(name)s (%(filename)s:%(lineno)d) - %(message)s",
+ handlers=[logging.StreamHandler(sys.stdout)],
+)
+
+# Suppress low-delay reference block warnings from OpenCV/PyAV/FFmpeg
+# os.environ["OPENCV_FFMPEG_LOGLEVEL"] = "-8"
+# os.environ["OPENCV_LOG_LEVEL"] = "OFF"
+logging.getLogger("libav").setLevel(logging.CRITICAL)
+logging.getLogger("libav.hevc").setLevel(logging.CRITICAL)
+logging.getLogger("matplotlib").setLevel(logging.WARNING)
+logging.getLogger("ultralytics").setLevel(logging.WARNING)
+# logger = trt.Logger(trt.Logger.WARNING)
+# trt.init_libnvinfer_plugins(logger, "")
+main_app_logger = logging.getLogger(__name__)
+
+# ==============================================================================
+# IMPORTS
+
+import argparse
+import asyncio
+import csv
+import ctypes
+import faulthandler
+import gc
+import multiprocessing as mp
+import sys
+import threading
+import time
+import traceback
+import tracemalloc
+from pathlib import Path
+
+import cv2
+import psutil
+import torch
+
+# import torch
+
+# Retrieve repo packages
+REPO_DIR = str(Path(__file__).parent.parent)
+sys.path.insert(1, REPO_DIR)
+from base_test import (
+ BaseTest,
+ download_eval_data,
+ fps_comparison_chart,
+)
+from include.default_configs import (
+ CUSTOM_MODEL_FLAG_DEFAULT,
+ DEBUG_DEFAULT,
+ MODEL_NAME_DEFAULT,
+ THRESHOLD_VALUE,
+)
+from include.handlers import (
+ get_test_handler,
+)
+from include.utils import (
+ PipelineConfig,
+ ResourceTrackerFilter,
+ install_and_load_pip_package,
+ str2bool,
+)
+from metrics import DeviceAgnosticOnTheFlyEvaluator
+
+gdown = install_and_load_pip_package("gdown", attribute_name=None)
+
+
+# ==============================================================================
+# SETUP
+# Hooks directly into the OS kernel signals to force Python to print a full
+# stack traceback right before it dies, allowing you to see which line of
+# Python code caused the hard crash
+faulthandler.enable()
+
+try:
+ # Retain standard spawn mode to prevent CUDA context driver deadlocks
+ torch.multiprocessing.set_start_method("spawn", force=True)
+except RuntimeError:
+ pass
+
+force_export = False
+# target_width, target_height = 7680, 4320
+STATE_CAPTURE = False
+
+# Force Python's multiprocessing layer to quiet tracking cleanup race conditions
+os.environ["PYTHONWARNINGS"] = "ignore"
+sys.warnoptions.append("ignore")
+
+
+# =========================================================================
+# ISOLATED BACKGROUND WORKER
+# =========================================================================
+
+
+def isolated_detection_worker(init_args, test_args, res_queue):
+ device = test_args["device"]
+ detection_type = test_args["detection_type"]
+ sf_enabled = test_args["sf_enabled"]
+ video_name = test_args["video_name"]
+ gt_enabled = test_args["gt_enabled"]
+
+ if device == "gpu" and torch.cuda.is_available():
+ torch.cuda.synchronize()
+ torch.cuda.ipc_collect()
+ torch.cuda.empty_cache()
+ gc.collect()
+
+ # Establish an accurate post-initialization hardware baseline
+ if str2bool(os.getenv("ENABLE_PROFILING", "False")):
+ tracemalloc.start()
+
+ start_allocated, start_reserved = 0, 0
+ if device == "gpu" and torch.cuda.is_available():
+ torch.cuda.memory._record_memory_history(
+ enabled=True,
+ trace_alloc_max_entries=250000,
+ trace_alloc_record_context=True,
+ )
+
+ # This captures your baseline AFTER initialize_variables() is done
+ start_allocated = torch.cuda.memory_allocated(0)
+ start_reserved = torch.cuda.memory_reserved(0)
+
+ try:
+ cv2.cuda.setBufferPoolUsage(False)
+ except AttributeError:
+ pass
+ else:
+ gc.collect()
+ process = psutil.Process(os.getpid())
+ start_allocated = (
+ process.memory_info().rss
+ ) # Reuse start_allocated container for host memory baseline
+ start_reserved = 0
+
+ try:
+ instance, _ = get_test_handler(TestEvalSmartFilteringDetections(), device)
+
+ test_dir = Path(__file__).parent
+ video_dir = test_dir / "eval_data/Anti-UAV-Tracking-V0-8K"
+ vid_source = video_dir / f"{video_name}/{video_name}.mp4"
+
+ if not vid_source.exists():
+ _ = download_eval_data([video_name])
+
+ instance.source = str(vid_source)
+ instance.name = vid_source.stem
+ instance.result_dir = Path(init_args["result_dir"])
+ instance.active_streams = init_args["active_streams"]
+ instance.__class__.benchmarks = init_args["benchmarks"] # Sandbox the metrics
+
+ vid_dir = instance.result_dir / device
+ vid_dir.mkdir(parents=True, exist_ok=True)
+ os.environ["TEST_SUITE_RENDER_DIR"] = str(vid_dir)
+
+ # if instance.source.startswith("rtsp"):
+ # short_name = "rtsp"
+ # else:
+ # short_name = Path(instance.source).stem
+
+ # config definition
+ config = PipelineConfig(
+ SHARED_OUTPUT=str(instance.result_dir), # defined in context
+ CUSTOM_MODEL_FLAG=os.getenv("CUSTOM_MODEL_FLAG", CUSTOM_MODEL_FLAG_DEFAULT),
+ DEVICE=device.upper(),
+ OMIT_DETECTIONS_FLAG=True,
+ TEST_MODE=True,
+ DEBUG=os.getenv("DEBUG", DEBUG_DEFAULT),
+ DEBUG_FRAME_LIMIT=int(os.getenv("DEBUG_FRAME_LIMIT", 100)),
+ ENABLE_QUERYING=False,
+ MODEL_NAME=os.getenv("MODEL_NAME", MODEL_NAME_DEFAULT),
+ SMART_FILTERING_ENABLED=sf_enabled,
+ THRESHOLD_VALUE=int(os.getenv("THRESHOLD_VALUE", THRESHOLD_VALUE)),
+ DETECTION_TYPE=detection_type,
+ )
+
+ # INITIALIZE CLASS (mimic DeviceBaseHandler.__init__) ------------------------------
+ instance.evaluator = DeviceAgnosticOnTheFlyEvaluator(device=device)
+
+ instance.is_rtsp = str(instance.source).startswith("rtsp:/")
+ instance.active = True
+ instance.config = config
+
+ # kwarg definition
+ instance._testMethodName = (
+ f"{video_name}_sf_{detection_type}_{device}"
+ if sf_enabled
+ else f"{video_name}_yolo_{detection_type}_{device}"
+ )
+ instance.video_output_name = f"{instance._testMethodName}.mp4"
+ instance.gt_enabled = gt_enabled
+
+ instance.loop = asyncio.get_event_loop()
+ instance.frame_ready_event = asyncio.Event()
+ instance._is_stopped = False
+ instance._stop_lock = threading.Lock() # Local lock for this instance
+ instance.main_startup_event = mp.Event()
+
+ instance.device = instance.config.DEVICE
+ instance.device_input = instance.config.device_input
+ instance.disp_w, instance.disp_h = instance.config.DISPLAY_FRAME_SIZE
+ instance.resize_h, instance.resize_w = [
+ instance.config.MODEL_H,
+ instance.config.MODEL_W,
+ ]
+
+ instance.setup_reader(
+ instance.config.TARGET_FPS,
+ instance.config.CLIP_DURATION,
+ startup_event=instance.main_startup_event,
+ )
+ instance.initialize_variables()
+ instance.setup_threads()
+ instance.last_heartbeat = time.perf_counter()
+
+ if STATE_CAPTURE:
+ main_app_logger.info(
+ "[DIAGNOSTIC] Registering system state baseline layout..."
+ )
+ instance.baseline_before_start = instance.capture_state_snapshot()
+
+ def orig_fn(profiler):
+ return instance.run_realtime_inference(
+ sf_enabled=instance.config.sf_enabled,
+ profiler=profiler,
+ gt_enabled=gt_enabled,
+ )
+
+ def profiler_fn():
+ profiler = None
+ try:
+ if str2bool(os.getenv("ENABLE_PROFILING", "False")):
+ Profiler = install_and_load_pip_package(
+ "pyinstrument", attribute_name="Profiler"
+ )
+
+ profiler = Profiler(interval=0.005) # 5ms sampling interval
+
+ # Telling the statistical sampler to skip recording exception blocks completely
+ # stops stack_sampler.py from ballooning RAM over long production runs.
+ if hasattr(profiler, "_sampler") and profiler._sampler:
+ profiler._sampler.trace_exceptions = False
+
+ profiler.start()
+
+ orig_fn(profiler)
+
+ if str2bool(os.getenv("ENABLE_PROFILING", "False")):
+ # 2. Redirect standard error to the filter trap right before report compilation
+ original_stderr = sys.stderr
+ sys.stderr = ResourceTrackerFilter(original_stderr)
+
+ try:
+ # Force standard stdout to flush out any lingering teardown messages
+ # BEFORE pyinstrument dumps its massive ASCII tree block.
+ sys.stdout.flush()
+
+ # main_app_logger.info(profiler.output_text(color=True))
+ # profiler.main_app_logger.info(color=True)
+ # prof_output = profiler.output_text(color=True)
+ # main_app_logger.info(
+ # f"\n=== LATENCY BREAKDOWN FOR {self.name} ({device}) ===\n{prof_output}\n",
+ #
+ # )
+
+ # Save a clean, interactive tree map for visual analysis
+ output_html_path = instance.output_path.replace(
+ ".mp4", "_profile.html"
+ )
+ # output_html_path = f"/tmp/profile_{video_name}_{device}.html"
+ profiler.write_html(output_html_path)
+ main_app_logger.info(
+ f"[PROFILER] Performance tree map exported to {output_html_path}",
+ )
+
+ finally:
+ sys.stderr = original_stderr
+ if "profiler" in locals():
+ try:
+ # Force the Python interpreter to detach pyinstrument's sampling hooks
+ sys.setprofile(None)
+
+ # Completely decouple internal statistical sessions to drop C-heap frames
+ if hasattr(profiler, "_last_session"):
+ profiler._last_session = None
+ if hasattr(profiler, "last_session"):
+ profiler.last_session = None
+
+ # Forcibly clear out internal memoryview strings caching tree metrics
+ if (
+ hasattr(profiler, "session")
+ and profiler.session is not None
+ ):
+ if hasattr(profiler.session, "frame_groups"):
+ profiler.session.frame_groups = None
+ if hasattr(profiler.session, "samples"):
+ profiler.session.samples = []
+
+ # Purge compiled tree metrics structures
+ profiler.session = None
+ del profiler
+ except Exception:
+ pass
+
+ # Trigger an immediate native Linux heap compression pass
+ # This grabs the newly abandoned pyinstrument C-heap blocks and
+ # flushes them to the OS before the fixture assessment snapshot fires!
+ gc.collect()
+ try:
+ libc = ctypes.CDLL("libc.so.6")
+ libc.malloc_trim(0)
+ except Exception:
+ pass
+
+ except Exception:
+ traceback.print_exc()
+
+ if str2bool(os.getenv("ENABLE_PROFILING", "False")) and hasattr(
+ instance, "process_thread"
+ ):
+ # Re-initialize the thread context using our safe profile wrapper proxy
+ instance.process_thread = threading.Thread(target=profiler_fn, daemon=True)
+
+ if gt_enabled:
+ instance.VIDEO_GT_DETAILS = download_eval_data([instance.name])
+ else:
+ instance.VIDEO_GT_DETAILS = None
+
+ instance.start()
+
+ while instance.active or not instance._is_stopped:
+ time.sleep(0.25)
+ if getattr(instance, "status", None) == "DONE":
+ break
+
+ if (
+ hasattr(instance, "process_thread")
+ and instance.process_thread is not None
+ ):
+ if not instance.process_thread.is_alive():
+ main_app_logger.info(
+ "[TEST HARNESS] Background worker exited. Breaking loop.",
+ )
+ break
+
+ instance.stop_threads(["process_thread"])
+
+ # Force early hardware driver sweep before unbinding threads
+ if instance.device_input == "cuda" and torch.cuda.is_available():
+ torch.cuda.synchronize()
+ torch.cuda.empty_cache()
+ torch.cuda.ipc_collect()
+
+ gc.collect()
+
+ try:
+ if STATE_CAPTURE:
+ # --- CAPTURE STATE POINT B (Right before cleanup) ---
+ main_app_logger.info(
+ "[DIAGNOSTIC] Gathering active execution workspace layouts..."
+ )
+ state_after_stop = instance.capture_state_snapshot()
+
+ # 3. Print the granular delta analysis out to your terminal screen
+ new_keys_generated, mutated_keys, static_keys = (
+ instance.print_lifecycle_delta(
+ instance.baseline_before_start,
+ state_after_stop,
+ return_keys=True,
+ )
+ )
+ # self.config.DEBUG_FLAG and
+ if len(new_keys_generated) > 0 or len(mutated_keys) > 0:
+ # Inform the blueprint engine to eliminate every difference found between Point A and B
+ instance.stop_blueprint_executor(new_keys_generated, mutated_keys)
+ except Exception:
+ traceback.print_exc()
+
+ assert instance.status == "DONE"
+
+ # TEARDOWN
+ instance.execute_teardown()
+
+ if str2bool(os.getenv("ENABLE_PROFILING", "False")):
+ gc.collect()
+ if (
+ device == "gpu" and torch.cuda.is_available()
+ ): # self.device_input == "cuda" and torch.cuda.is_available():
+ torch.cuda.synchronize()
+ torch.cuda.empty_cache()
+ instance.assess_memory(
+ device, instance.name, start_allocated, start_reserved
+ )
+
+ if len(instance.__class__.benchmarks) > 0:
+ final_metrics = instance.__class__.benchmarks[-1]
+ res_queue.put({"status": "success", "metrics": final_metrics})
+ else:
+ res_queue.put({"status": "error", "error": "No benchmarks generated."})
+
+ except Exception as e:
+ res_queue.put(
+ {"status": "error", "error": str(e), "traceback": traceback.format_exc()}
+ )
+ res_queue.put(
+ {"status": "error", "error": str(e), "traceback": traceback.format_exc()}
+ )
+
+
+# =========================================================================
+# PYTEST TEST HARNESS
+# =========================================================================
+def pytest_generate_tests(metafunc):
+ """
+ Pytest Hook: Generates a complete cross-product matrix
+ of (video_name x device) dynamically during collection phase.
+ """
+ if (
+ "video_name" in metafunc.fixturenames
+ and "device" in metafunc.fixturenames
+ and "detection_type" in metafunc.fixturenames
+ and "sf_enabled" in metafunc.fixturenames
+ ):
+ # Define video targets explicitly during collection
+ video_names = [f"video{i:02d}" for i in range(9, 21)]
+ # video_names = ["video12"]
+
+ # Read device target filters passed from your CLI parser args block
+ # Falls back to both types if running globally via '--device all'
+ device_input = os.getenv("TEST_SUITE_DEVICE_FILTER", "all")
+ devices = ["cpu", "gpu"] if device_input == "all" else [device_input]
+
+ type_input = os.getenv("TEST_SUITE_DETECTION_FILTER", "all")
+ detection_types = ["motion", "object"] if type_input == "all" else [type_input]
+
+ sf_input = os.getenv("TEST_SUITE_SF_FILTER", "None")
+ sf_flags = [True, False] if sf_input == "None" else [str2bool(sf_input)]
+
+ # Tell Pytest to create an individual test instance for each discovered name
+ metafunc.parametrize("device", devices)
+ metafunc.parametrize("sf_enabled", sf_flags)
+ metafunc.parametrize("detection_type", detection_types)
+ metafunc.parametrize("video_name", video_names)
+
+
+@pytest.fixture(scope="class")
+def setup_context(request):
+ """Replaces setUpClass: Runs once per test class."""
+ current_test_filename = Path(__file__).stem
+ test_dir = Path(__file__).parent
+
+ # model_name = os.getenv("MODEL_NAME", MODEL_NAME_DEFAULT)
+ # Handler.__init__ (main items)
+ request.cls.is_rtsp = False
+
+ # model_name = os.getenv("MODEL_NAME", MODEL_NAME_DEFAULT)
+ request.cls.result_dir = test_dir / f"{current_test_filename}_drone_results"
+ request.cls.result_dir.mkdir(parents=True, exist_ok=True)
+
+ # Benchmark statistics
+ request.cls.benchmarks = []
+ request.cls.csv_filename = "drone_eval.csv"
+ request.cls.csv_path = request.cls.result_dir / request.cls.csv_filename
+
+ # request.cls.active = True
+ request.cls.active_streams = {}
+
+ # RUN ALL PARAMETERIZED TESTS ----------------------------------------
+ yield
+
+ # FINAL CSV EXPORT --------------------------------------------------
+ if request.cls.benchmarks:
+ # Filter and exclude rows that were interrupted or failed initialization due to an early pytest skip
+ request.cls.benchmarks = [
+ r for r in request.cls.benchmarks if r and "Test Name" in r
+ ]
+ results = request.cls.benchmarks
+ for row in results:
+ if "gpu" in row["Test Name"].lower():
+ match_name = row["Test Name"].replace("gpu", "cpu")
+ cpu_row = next(
+ (r for r in results if r["Test Name"] == match_name), None
+ )
+ if cpu_row:
+ gpu_fps = float(row["Pipeline FPS (Target frames)"])
+ cpu_fps = float(cpu_row["Pipeline FPS (Target frames)"])
+ speedup = (gpu_fps / cpu_fps) if cpu_fps > 0 else 0
+ row["Pipeline Speedup vs CPU"] = f"{speedup:.2f}x"
+ else:
+ row["Pipeline Speedup vs CPU"] = "N/A"
+ else:
+ row["Pipeline Speedup vs CPU"] = "Baseline (CPU)"
+
+ if results:
+ # # Define video targets explicitly during collection
+ # video_names = [f"video{i:02d}" for i in range(1, 21)]
+ # ALL_VIDEO_GT_DETAILS = download_eval_data(video_names)
+
+ keys = results[0].keys()
+
+ with open(str(request.cls.csv_path), "w", newline="") as f:
+ dict_writer = csv.DictWriter(f, fieldnames=keys)
+ dict_writer.writeheader()
+ dict_writer.writerows(results)
+ main_app_logger.info(f"[FINAL] Benchmarks saved to {request.cls.csv_path}")
+
+ main_app_logger.info("=" * 125)
+ main_app_logger.info(
+ f"{'Test Name':<25} | {'mAP_10':<5} | {'mAP_10_95':<10} | {'Pipeline FPS (Target)':<21} | {'Avg Frame Reading (ms)':<22} | {'Pipeline Speedup vs CPU':<15}",
+ )
+ main_app_logger.info("-" * 125)
+
+ for r in results:
+ main_app_logger.info(
+ f"{r['Test Name']:<25} | {r['mAP_10']:0.4f} | {r['mAP_10_95']:0.4f} | {r['Pipeline FPS (Target frames)']:<21} | {r['Avg Frame Reading (ms)']:<22} | {r.get('Pipeline Speedup vs CPU', 'N/A'):<10}"
+ )
+ main_app_logger.info("=" * 125)
+
+ chart_path = (
+ request.cls.result_dir
+ / f"{request.cls.csv_filename.replace('.csv', '')}_pipelineFPS.png"
+ )
+ fps_comparison_chart(
+ chart_path, results, fps_key="Pipeline FPS (Video frames)"
+ )
+
+ chart_path = (
+ request.cls.result_dir
+ / f"{request.cls.csv_filename.replace('.csv', '')}_pipelineFPS_target.png"
+ )
+ fps_comparison_chart(
+ chart_path, results, fps_key="Pipeline FPS (Target frames)"
+ )
+
+ chart_path = (
+ request.cls.result_dir
+ / f"{request.cls.csv_filename.replace('.csv', '')}_sfdetectFPS_target.png"
+ )
+ fps_comparison_chart(chart_path, results, fps_key="SF/Detection FPS")
+
+ chart_path = (
+ request.cls.result_dir
+ / f"{request.cls.csv_filename.replace('.csv', '')}_displayFPS_target.png"
+ )
+ fps_comparison_chart(chart_path, results, fps_key="Display FPS")
+
+ else:
+ main_app_logger.info(
+ "[WARN] Benchmarking matrix empty. Skipping final CSV and chart exports."
+ )
+
+
+@pytest.mark.usefixtures("setup_context")
+class TestEvalSmartFilteringDetections(BaseTest):
+ """
+ Pytest runner that spawns the isolated worker, waits for completion,
+ and merges the returned metrics into the main class for CSV/Chart generation.
+ """
+
+ benchmarks = [] # Class-level attribute required by _finalize_benchmarks
+
+ def test_eval(
+ self, device, detection_type, sf_enabled, video_name, gt_enabled=True
+ ):
+ if detection_type == "motion" and not sf_enabled:
+ pytest.skip(
+ "Pure YOLO mode is structurally invalid for detection_type 'motion'.\n"
+ )
+
+ # Pull the values dynamically assigned by the setup_context fixture
+ init_args = {
+ # "source": self.__class__.source,
+ # "name": self.__class__.name,
+ "result_dir": str(self.__class__.result_dir),
+ "active_streams": {},
+ "benchmarks": self.__class__.benchmarks,
+ }
+
+ test_args = {
+ "device": device,
+ "detection_type": detection_type,
+ "sf_enabled": sf_enabled,
+ "video_name": video_name,
+ "gt_enabled": gt_enabled,
+ }
+
+ main_app_logger.info(
+ f"\n{'=' * 60}\n"
+ f"[TEST HARNESS] Spawning isolated high-speed process for {video_name} (SF: {sf_enabled})...\n"
+ f"{'=' * 60}"
+ )
+
+ # Spawn a pristine Python process (identical to test_pipeline.py's stream_worker)
+ ctx = mp.get_context("spawn")
+ res_queue = ctx.Queue()
+
+ worker_p = ctx.Process(
+ target=isolated_detection_worker, args=(init_args, test_args, res_queue)
+ )
+
+ worker_p.start()
+ worker_p.join() # Block Pytest until the isolated pipeline finishes
+
+ # Extract and merge results
+ if not res_queue.empty():
+ result = res_queue.get()
+
+ if result["status"] == "error":
+ pytest.fail(
+ f"Pipeline crashed in worker process:\n{result.get('error')}\n{result.get('traceback')}"
+ )
+ else:
+ metrics = result["metrics"]
+
+ self.__class__.benchmarks.append(metrics)
+
+ main_app_logger.info(
+ f"[TEST HARNESS] Worker returned successfully. Display FPS: {metrics.get('Display FPS')}"
+ )
+
+ # Basic functionality assertions
+ assert metrics is not None, "Metrics dictionary should not be None."
+ assert int(metrics.get("Output Frames", 0)) > 0, (
+ "No frames were written to output."
+ )
+ else:
+ pytest.fail("Worker process died unexpectedly without returning metrics.")
+
+
+# =========================================================================
+# MAIN
+# =========================================================================
+def get_pytest_filter_expression(args):
+ main_app_logger.info("=" * 50)
+ main_app_logger.info("TARGET SELECTION PREVIEW")
+ main_app_logger.info("=" * 50)
+
+ filter_expression = Path(__file__).stem
+ # applied_subs = []
+
+ if args.sf_enabled is not None:
+ # Target exact parameter tokens generated by pytest parametrization
+ sf_str = "-True-" if args.sf_enabled else "-False-"
+ filter_expression += f" and {sf_str}"
+ # applied_subs.append(f"sf_enabled={args.sf_enabled}")
+
+ if args.detection_type:
+ filter_expression += f" and {args.detection_type}"
+
+ # Target hardware context selection filter applies globally across all test cases
+ if args.device.lower() != "all":
+ main_app_logger.info(f" 💻 Hardware Context Constraint: {args.device.upper()}")
+ filter_expression = f"({filter_expression}) and {args.device.lower()}"
+ else:
+ main_app_logger.info(" 💻 Hardware Context Constraint: ALL AVAILABLE")
+
+ main_app_logger.info("=" * 50)
+ main_app_logger.info(f"COMPILED PYTEST KEYWORD EXPRESSION: {filter_expression}")
+ main_app_logger.info("=" * 50)
+
+ return filter_expression
+
+
+if __name__ == "__main__":
+ # TEST ARGUMENTS
+ parser = argparse.ArgumentParser(description="Run Video Detection Pipeline Tests")
+ parser.add_argument(
+ "-s",
+ "--source",
+ type=str,
+ default="anduril_swarm_8K.mp4",
+ help="Video filename (located in /inputs)",
+ )
+
+ # MODEL TO USE
+ parser.add_argument(
+ "--no-custom",
+ action="store_false",
+ dest="custom_model_flag",
+ help="Enable if using Ultralytics YOLO model",
+ )
+ parser.add_argument(
+ "-m",
+ "--model",
+ type=str,
+ default="drone_detection",
+ dest="model_name",
+ help="Name of model. Required if `--no-custom` is enabled. [Default: drone_detection]",
+ )
+
+ # Filter tests
+ # parser.add_argument(
+ # "--device",
+ # type=str,
+ # default="all",
+ # choices=["cpu", "gpu", "all"],
+ # help="Filter by device (cpu or gpu)",
+ # )
+ parser.add_argument(
+ "--type",
+ type=str,
+ choices=["object", "motion"],
+ dest="detection_type",
+ help="Filter by detection type (object or motion)",
+ )
+ parser.add_argument(
+ "--sf",
+ action="store_true",
+ default=None,
+ dest="sf_enabled",
+ help="Filter by Smart Filtering",
+ )
+ parser.add_argument(
+ "--no-sf",
+ action="store_false",
+ default=None,
+ dest="sf_enabled",
+ help="Filter by Smart Filtering",
+ )
+
+ # DEBUGGING
+ parser.add_argument(
+ "--debug",
+ action="store_true",
+ help="Enable debug message and save intermediate images for Smart Filtering tests",
+ )
+ parser.add_argument(
+ "-n",
+ type=int,
+ default=100,
+ dest="debug_frame_limit",
+ help="Number of frames used for debugging [Default: 100]",
+ )
+
+ # PROFILING
+ parser.add_argument(
+ "--profile",
+ action="store_true",
+ help="Enable profiling",
+ )
+
+ args = parser.parse_args()
+ args.device = "gpu"
+
+ # UPDATE ENVIRONMENTAL VARIABLES
+ os.environ["VIDEO_FILENAME"] = args.source
+ os.environ["CUSTOM_MODEL_FLAG"] = "True" if args.custom_model_flag else "False"
+ os.environ["MODEL_NAME"] = args.model_name
+ os.environ["DEBUG"] = "1" if args.debug else "0"
+ os.environ["DEBUG_FRAME_LIMIT"] = str(args.debug_frame_limit)
+ os.environ["ENABLE_PROFILING"] = "True" if args.profile else "False"
+
+ # detection_type, device, sf_enabled
+ filter_expression = get_pytest_filter_expression(args)
+
+ # PYTEST COMMAND
+ pytest_args = [
+ "-k",
+ filter_expression,
+ "-s",
+ "-v",
+ # "--log-cli-level=DEBUG",
+ "-W",
+ "ignore:Exception ignored in.*SharedMemory.__del__:UserWarning",
+ # Target the exact module rewrite warning path inside the configuration framework
+ "-W",
+ "ignore:Module already imported so cannot be rewritten; anyio:_pytest.warning_types.PytestAssertRewriteWarning",
+ # Bypass Pytest's internal proxy
+ "--capture=no",
+ __file__,
+ ]
+
+ # main_app_logger.info(f"Launching tests for {args.source}")
+
+ sys.exit(pytest.main(pytest_args))
diff --git a/fastapi/tests/test_model.py b/fastapi/tests/test_model.py
index c2710fd..2f34da2 100644
--- a/fastapi/tests/test_model.py
+++ b/fastapi/tests/test_model.py
@@ -1,220 +1,414 @@
+# ==============================================================================
+# SUPPRESS WARNINGS
+# import warnings
+
+import pytest
+
+# warnings.filterwarnings(
+# "ignore", category=FutureWarning, message=".*reduce_op` is deprecated.*"
+# )
+
+pytestmark = [
+ pytest.mark.filterwarnings(
+ "ignore:.*anyio:_pytest.warning_types.PytestAssertRewriteWarning"
+ ),
+ pytest.mark.filterwarnings(
+ "ignore:Context managers for TensorRT types are deprecated:DeprecationWarning"
+ ),
+ pytest.mark.filterwarnings(
+ "ignore:Exception ignored in.*SharedMemory.__del__:UserWarning"
+ ),
+ # Global message fallback pattern captures local execution frames
+ pytest.mark.filterwarnings("ignore:.*reduce_op.*:FutureWarning"),
+]
+
+# ==============================================================================
+# LOGGING
+import logging
+import os
+import sys
+
+logging.basicConfig(
+ level=logging.INFO,
+ # format="%(asctime)s - %(name)s - %(levelname)s - %(message)s",
+ format="%(asctime)s [%(levelname)s] %(name)s (%(filename)s:%(lineno)d) - %(message)s",
+ handlers=[logging.StreamHandler(sys.stdout)],
+)
+
+# Suppress low-delay reference block warnings from OpenCV/PyAV/FFmpeg
+# os.environ["OPENCV_FFMPEG_LOGLEVEL"] = "-8"
+# os.environ["OPENCV_LOG_LEVEL"] = "OFF"
+logging.getLogger("libav").setLevel(logging.CRITICAL)
+logging.getLogger("libav.hevc").setLevel(logging.CRITICAL)
+logging.getLogger("matplotlib").setLevel(logging.WARNING)
+logging.getLogger("ultralytics").setLevel(logging.WARNING)
+# logger = trt.Logger(trt.Logger.WARNING)
+# trt.init_libnvinfer_plugins(logger, "")
+main_app_logger = logging.getLogger(__name__)
+
+# ==============================================================================
+# IMPORTS
+
+# os.environ["OMP_NUM_THREADS"] = "2"
+# os.environ["MKL_NUM_THREADS"] = "2"
+# os.environ["PYTORCH_ALLOC_CONF"] = "expandable_segments:True"
+# try:
+# torch.set_num_interop_threads(2) # 1)
+# torch.set_num_threads(4) # 2)
+# except RuntimeError:
+# # Safe graceful fallback if a process-level fork duplicated context maps
+# pass
import argparse
+import asyncio
import csv
+import faulthandler
import gc
-import logging
-import os
+import multiprocessing as mp
import sys
+import threading
import time
+import traceback
+import tracemalloc
from pathlib import Path
-import pytest
-import tensorrt as trt
+import cv2
+import psutil
import torch
-sys.path.insert(1, str(Path(__file__).parent.parent))
+# import torch
+
+# Retrieve repo packages
+REPO_DIR = str(Path(__file__).parent.parent)
+sys.path.insert(1, REPO_DIR)
+from base_test import (
+ BaseTest,
+)
from include.default_configs import (
CUSTOM_MODEL_FLAG_DEFAULT,
DEBUG_DEFAULT,
MODEL_NAME_DEFAULT,
THRESHOLD_VALUE,
)
-from include.handlers import DeviceBaseHandler
-from include.models import get_model
-from include.utils import PipelineConfig
-
-try:
- torch.multiprocessing.set_start_method("spawn", force=True)
-except RuntimeError:
- pass
-
-logging.getLogger("matplotlib").setLevel(logging.WARNING)
-logger = trt.Logger(trt.Logger.WARNING)
-trt.init_libnvinfer_plugins(logger, "")
+from include.detectors import GeneralObjectDetector
+from include.handlers import (
+ get_test_handler,
+)
+from include.utils import (
+ PipelineConfig,
+ str2bool,
+)
+# objgraph = install_and_load_pip_package("objgraph", attribute_name=None)
+# torch.set_grad_enabled(False)
-DEBUG_FLAG_DEFAULT = True if DEBUG_DEFAULT == "1" else False
-os.environ["OMP_NUM_THREADS"] = "1"
+# ==============================================================================
+# MONKEY PATCH
+# Cache the hardware availability state globally.
+# This prevents internal framework loops from triggering driver/NVML checks per frame.
+# _cuda_available = torch.cuda.is_available()
-force_export = False
+# def _patched_is_available():
+# return _cuda_available
-@pytest.fixture(scope="class")
-def setup_context(request):
- """Replaces setUpClass: Runs once per test class."""
- # Initialize shared paths/results
- current_test_filename = Path(__file__).stem
- test_dir = Path(__file__).parent
- main_path = test_dir.parent
- video_dir = main_path / "inputs"
-
- VIDEO_FILENAME = os.getenv("VIDEO_FILENAME", "anduril_swarm_8K.mp4")
- if video_dir.exists():
- request.cls.video_path = video_dir / VIDEO_FILENAME
- else:
- video_dir = Path("/watch_dir")
- request.cls.video_path = video_dir / VIDEO_FILENAME
- # Add any shared state to the class
- model_name = os.getenv("MODEL_NAME", MODEL_NAME_DEFAULT)
- request.cls.result_dir = (
- test_dir / f"{current_test_filename}_results/{request.cls.video_path.stem}"
- )
- request.cls.result_dir.mkdir(parents=True, exist_ok=True)
+# torch.cuda.is_available = _patched_is_available
- # Benchmark statistics
- request.cls.benchmarks = []
- request.cls.csv_filename = (
- f"model_benchmarks_{model_name}_{request.cls.video_path.stem}.csv"
- )
- request.cls.csv_path = request.cls.result_dir / request.cls.csv_filename
+# ==============================================================================
+# SETUP
+# Hooks directly into the OS kernel signals to force Python to print a full
+# stack traceback right before it dies, allowing you to see which line of
+# Python code caused the hard crash
+faulthandler.enable()
- # Initialize class vars
- request.cls.name = request.cls.video_path.stem
- request.cls.source = str(request.cls.video_path)
- request.cls.is_rtsp = str(request.cls.source).startswith("rtsp:/")
- request.cls.active = True
- request.cls.active_streams = {}
- request.cls._shared_model = None
- request.cls._shared_model_path = None
- request.cls._shared_model_device = None
- request.cls._shared_model_sf_enabled = None
+try:
+ # Retain standard spawn mode to prevent CUDA context driver deadlocks
+ torch.multiprocessing.set_start_method("spawn", force=True)
+except RuntimeError:
+ pass
- # RUN ALL PARAMETERIZED TESTS ----------------------------------------
- yield
+force_export = False
+# target_width, target_height = 7680, 4320
+STATE_CAPTURE = False
- # FINAL CSV EXPORT --------------------------------------------------
- if request.cls.benchmarks:
- results = request.cls.benchmarks
- keys = results[0].keys()
+# Force Python's multiprocessing layer to quiet tracking cleanup race conditions
+os.environ["PYTHONWARNINGS"] = "ignore"
+sys.warnoptions.append("ignore")
- with open(str(request.cls.csv_path), "w", newline="") as f:
- dict_writer = csv.DictWriter(f, fieldnames=keys)
- dict_writer.writeheader()
- dict_writer.writerows(results)
- print(f"\n[FINAL] Benchmarks saved to {request.cls.csv_filename}", flush=True)
+# =========================================================================
+# ISOLATED BACKGROUND WORKER
+# =========================================================================
-@pytest.fixture(autouse=True)
-def each_test_setup(request):
- test_class_self = request.instance
+def isolated_detection_worker(init_args, test_args, res_queue):
+ device = test_args["device"]
+ detection_type = test_args["detection_type"]
+ sf_enabled = test_args["sf_enabled"]
+ # gt_enabled = test_args["gt_enabled"]
- # setUp LOGIC --------------------------------------------------
- if torch.cuda.is_available():
+ if device == "gpu" and torch.cuda.is_available():
torch.cuda.synchronize()
+ torch.cuda.ipc_collect()
+ torch.cuda.empty_cache()
+ gc.collect()
- device = request.node.callspec.params.get("device")
- os.environ["DEVICE"] = device
- detection_type = "object" # request.node.callspec.params.get("detection_type")
- sf_enabled = False # request.node.callspec.params.get("sf_enabled")
- model_name = os.getenv("MODEL_NAME", MODEL_NAME_DEFAULT)
- test_class_self._testMethodName = f"{model_name}_{detection_type}_{device}"
-
- render_dir = test_class_self.result_dir / f"{test_class_self._testMethodName}"
- render_dir.mkdir(exist_ok=True)
- test_class_self.config = PipelineConfig(
- # GENERAL
- SHARED_OUTPUT=render_dir, # os.getenv("SHARED_OUTPUT",SHARED_OUTPUT_DEFAULT),
- CUSTOM_MODEL_FLAG=os.getenv(
- "CUSTOM_MODEL_FLAG", CUSTOM_MODEL_FLAG_DEFAULT
- ), # True,
- DEVICE=device.upper(),
- OMIT_DETECTIONS_FLAG=True,
- TEST_MODE=False,
- DEBUG=os.getenv("DEBUG", DEBUG_DEFAULT),
- DEBUG_FRAME_LIMIT=os.getenv("DEBUG_FRAME_LIMIT", 100),
- # VIDEO WRITER
- # CLIP_DURATION=None,
- # VDMS
- ENABLE_QUERYING=False,
- DBHOST="0.0.0.0",
- DBPORT=55555,
- # MODEL
- MODEL_NAME=model_name,
- # MODEL_H=360,
- # PIPELINE
- SMART_FILTERING_ENABLED=sf_enabled,
- THRESHOLD_VALUE=int(os.getenv("THRESHOLD_VALUE", THRESHOLD_VALUE)),
- # VISUALIZATION
- DETECTION_TYPE=detection_type,
- )
-
- test_class_self.device = test_class_self.config.DEVICE
- test_class_self.device_input = test_class_self.config.device_input
- test_class_self.resize_h, test_class_self.resize_w = [
- test_class_self.config.MODEL_H,
- test_class_self.config.MODEL_W,
- ]
+ # Establish an accurate post-initialization hardware baseline
+ if str2bool(os.getenv("ENABLE_PROFILING", "False")):
+ tracemalloc.start()
- test_class_self.setup_reader(
- test_class_self.config.TARGET_FPS, test_class_self.config.CLIP_DURATION
- )
+ start_allocated, start_reserved = 0, 0
+ if device == "gpu" and torch.cuda.is_available():
+ torch.cuda.memory._record_memory_history(
+ enabled=True,
+ trace_alloc_max_entries=250000,
+ trace_alloc_record_context=True,
+ )
- # RUN PARAMETERIZED TEST ----------------------------------------
- yield
+ # This captures your baseline AFTER initialize_variables() is done
+ start_allocated = torch.cuda.memory_allocated(0)
+ start_reserved = torch.cuda.memory_reserved(0)
- # tearDown LOGIC --------------------------------------------------
- print(
- f"\n--- [TearDown] Memory Before Cleanup ({test_class_self._testMethodName}) ---"
- )
-
- # Nullify the model reference to trigger automatic cleanup
- if hasattr(test_class_self, "model") and test_class_self.model is not None:
- # Check if it's an Ultralytics model with a predictor
- predictor = getattr(test_class_self.model, "predictor", None)
- if predictor is not None:
try:
- predictor.results = []
+ cv2.cuda.setBufferPoolUsage(False)
except AttributeError:
pass
- del test_class_self.model
- test_class_self.model = None
-
- # Clear the singleton references to force a reload
- TestSmartFilteringDetections._shared_model = None
- TestSmartFilteringDetections._shared_model_path = None
- TestSmartFilteringDetections._shared_model_device = None
-
- # Force Python to run destructors NOW while streams are still alive
- gc.collect()
- if torch.cuda.is_available():
- torch.cuda.synchronize() # Final sync to clear event queue
- torch.cuda.empty_cache()
- torch.cuda.ipc_collect() # Critical for shared memory cleanup
- time.sleep(0.2)
-
+ else:
+ gc.collect()
+ process = psutil.Process(os.getpid())
+ start_allocated = (
+ process.memory_info().rss
+ ) # Reuse start_allocated container for host memory baseline
+ start_reserved = 0
+
+ try:
+ instance, _ = get_test_handler(TestModel(), device)
+
+ instance.source = init_args["source"]
+ instance.name = init_args["name"]
+ instance.result_dir = Path(init_args["result_dir"])
+ instance.active_streams = init_args["active_streams"]
+ instance.__class__.benchmarks = init_args["benchmarks"] # Sandbox the metrics
+
+ # vid_dir = instance.result_dir / device
+ # vid_dir.mkdir(parents=True, exist_ok=True)
+ # os.environ["TEST_SUITE_RENDER_DIR"] = str(vid_dir)
+
+ # if instance.source.startswith("rtsp"):
+ # short_name = "rtsp"
+ # else:
+ # short_name = Path(instance.source).stem
+
+ # config definition
+ model_name = os.getenv("MODEL_NAME", MODEL_NAME_DEFAULT)
+ config = PipelineConfig(
+ SHARED_OUTPUT=str(instance.result_dir), # defined in context
+ CUSTOM_MODEL_FLAG=os.getenv("CUSTOM_MODEL_FLAG", CUSTOM_MODEL_FLAG_DEFAULT),
+ DEVICE=device.upper(),
+ OMIT_DETECTIONS_FLAG=True,
+ TEST_MODE=True,
+ DEBUG=os.getenv("DEBUG", DEBUG_DEFAULT),
+ DEBUG_FRAME_LIMIT=int(os.getenv("DEBUG_FRAME_LIMIT", 100)),
+ ENABLE_QUERYING=False,
+ MODEL_NAME=model_name,
+ SMART_FILTERING_ENABLED=sf_enabled,
+ THRESHOLD_VALUE=int(os.getenv("THRESHOLD_VALUE", THRESHOLD_VALUE)),
+ DETECTION_TYPE=detection_type,
+ )
-@pytest.mark.usefixtures("setup_context")
-class TestSmartFilteringDetections:
- # SETUP --------------------------------------------
-
- @pytest.mark.parametrize(
- "device",
- [
- ("cpu"),
- ("gpu"),
- ],
- )
- def test_pipeline(self, device):
- """Unified test runner for all configurations."""
+ # INITIALIZE CLASS (mimic DeviceBaseHandler.__init__) ------------------------------
+ instance.is_rtsp = str(instance.source).startswith("rtsp:/")
+ instance.active = True
+ instance.config = config
+
+ # kwarg definition
+ instance._testMethodName = f"{model_name}_{detection_type}_{device}"
+ # instance.video_output_name = (
+ # f"{instance._testMethodName}_{short_name}.mp4"
+ # )
+
+ instance.loop = asyncio.get_event_loop()
+ instance.frame_ready_event = asyncio.Event()
+ instance._is_stopped = False
+ instance._stop_lock = threading.Lock() # Local lock for this instance
+ instance.main_startup_event = mp.Event()
+
+ instance.device = instance.config.DEVICE
+ instance.device_input = instance.config.device_input
+ instance.disp_w, instance.disp_h = instance.config.DISPLAY_FRAME_SIZE
+ instance.resize_h, instance.resize_w = [
+ instance.config.MODEL_H,
+ instance.config.MODEL_W,
+ ]
+
+ instance.setup_reader(
+ instance.config.TARGET_FPS,
+ instance.config.CLIP_DURATION,
+ startup_event=instance.main_startup_event,
+ )
+ instance.initialize_variables()
+ instance.setup_threads()
+ instance.last_heartbeat = time.perf_counter()
+
+ if STATE_CAPTURE:
+ main_app_logger.info(
+ "[DIAGNOSTIC] Registering system state baseline layout..."
+ )
+ instance.baseline_before_start = instance.capture_state_snapshot()
+
+ # def profiler_fn():
+ # profiler = None
+ # try:
+ # if str2bool(os.getenv("ENABLE_PROFILING", "False")):
+ # Profiler = install_and_load_pip_package(
+ # "pyinstrument", attribute_name="Profiler"
+ # )
+
+ # profiler = Profiler(interval=0.005) # 5ms sampling interval
+
+ # # Telling the statistical sampler to skip recording exception blocks completely
+ # # stops stack_sampler.py from ballooning RAM over long production runs.
+ # if hasattr(profiler, "_sampler") and profiler._sampler:
+ # profiler._sampler.trace_exceptions = False
+
+ # profiler.start()
+
+ # # orig_fn(profiler)
+ # instance.run_realtime_inference(
+ # sf_enabled=instance.config.sf_enabled,
+ # profiler=profiler,
+ # gt_enabled=gt_enabled,
+ # )
+
+ # if str2bool(os.getenv("ENABLE_PROFILING", "False")):
+ # # 2. Redirect standard error to the filter trap right before report compilation
+ # original_stderr = sys.stderr
+ # sys.stderr = ResourceTrackerFilter(original_stderr)
+
+ # try:
+ # # Force standard stdout to flush out any lingering teardown messages
+ # # BEFORE pyinstrument dumps its massive ASCII tree block.
+ # sys.stdout.flush()
+
+ # # main_app_logger.info(profiler.output_text(color=True))
+ # # profiler.main_app_logger.info(color=True)
+ # # prof_output = profiler.output_text(color=True)
+ # # main_app_logger.info(
+ # # f"\n=== LATENCY BREAKDOWN FOR {self.name} ({device}) ===\n{prof_output}\n",
+ # #
+ # # )
+
+ # # Save a clean, interactive tree map for visual analysis
+ # output_html_path = instance.output_path.replace(
+ # ".mp4", "_profile.html"
+ # )
+ # # output_html_path = f"/tmp/profile_{video_name}_{device}.html"
+ # profiler.write_html(output_html_path)
+ # main_app_logger.info(
+ # f"[PROFILER] Performance tree map exported to {output_html_path}",
+ # )
+
+ # finally:
+ # sys.stderr = original_stderr
+ # if "profiler" in locals():
+ # try:
+ # # Force the Python interpreter to detach pyinstrument's sampling hooks
+ # sys.setprofile(None)
+
+ # # Completely decouple internal statistical sessions to drop C-heap frames
+ # if hasattr(profiler, "_last_session"):
+ # profiler._last_session = None
+ # if hasattr(profiler, "last_session"):
+ # profiler.last_session = None
+
+ # # Forcibly clear out internal memoryview strings caching tree metrics
+ # if (
+ # hasattr(profiler, "session")
+ # and profiler.session is not None
+ # ):
+ # if hasattr(profiler.session, "frame_groups"):
+ # profiler.session.frame_groups = None
+ # if hasattr(profiler.session, "samples"):
+ # profiler.session.samples = []
+
+ # # Purge compiled tree metrics structures
+ # profiler.session = None
+ # del profiler
+ # except Exception:
+ # pass
+
+ # # Trigger an immediate native Linux heap compression pass
+ # # This grabs the newly abandoned pyinstrument C-heap blocks and
+ # # flushes them to the OS before the fixture assessment snapshot fires!
+ # gc.collect()
+ # try:
+ # libc = ctypes.CDLL("libc.so.6")
+ # libc.malloc_trim(0)
+ # except Exception:
+ # pass
+
+ # except Exception:
+ # traceback.print_exc()
+
+ # if str2bool(os.getenv("ENABLE_PROFILING", "False")) and hasattr(
+ # instance, "process_thread"
+ # ):
+ # # Re-initialize the thread context using our safe profile wrapper proxy
+ # instance.process_thread = threading.Thread(target=profiler_fn, daemon=True)
+
+ # instance.VIDEO_GT_DETAILS = None
+ # instance.duration_target = 30
+
+ # instance.start()
+
+ # while instance.active or not instance._is_stopped:
+ # time.sleep(0.25)
+ # if getattr(instance, "status", None) == "DONE":
+ # break
+
+ # if hasattr(instance, "process_thread") and instance.process_thread is not None:
+ # if not instance.process_thread.is_alive():
+ # main_app_logger.info(
+ # "[TEST HARNESS] Background worker exited. Breaking loop.",
+ # )
+ # break
+
+ # instance.stop_threads(["process_thread"])
# Run the actual model loader
- self.get_model_by_device(device, sf_enabled=self.config.sf_enabled)
+ instance.processor = GeneralObjectDetector(
+ instance.config,
+ device=instance.device_input,
+ timer_enabled=False,
+ resize_hw=(instance.resize_h, instance.resize_w),
+ frame_hw=(instance.frame_height, instance.frame_width),
+ target_fps=instance.target_fps,
+ result_dir=instance.result_dir,
+ run_name=instance._testMethodName,
+ debug_frame_limit=-1,
+ )
total_session_start = time.perf_counter()
- results = self.model.predict(
- source=self.source,
- conf=self.config.DETECTION_THRESHOLD,
- # iou=self.config.IOU_THRESHOLD,
+ print("Running model...", flush=True)
+ results = instance.processor.model.predict(
+ source=instance.source,
+ imgsz=(instance.frame_height, instance.frame_width),
+ batch=1,
+ device=instance.config.device_input,
+ stream=any(x in instance.source for x in [".mp4", "rtsp"]),
+ conf=instance.config.DETECTION_THRESHOLD,
+ iou=instance.config.IOU_THRESHOLD,
show=False,
- imgsz=(self.frame_height, self.frame_width),
save=True,
- project=str(self.config.SHARED_OUTPUT),
- name=f"pred_{self._testMethodName}_output_video",
+ project=str(instance.config.SHARED_OUTPUT),
+ name="predict_results",
exist_ok=True, # overwrite if folder exists
- stream=True,
- data={"names": {i: name for i, name in enumerate(self.label_source)}},
+ data={
+ "names": {
+ i: name for i, name in enumerate(instance.processor.label_source)
+ }
+ },
)
frame_cnt = 0
@@ -231,67 +425,219 @@ def test_pipeline(self, device):
# Capture the true real-world duration across all operational processing layers
real_world_latency_ms = (time.perf_counter() - total_session_start) * 1000
- self._finalize_benchmarks(
+ print("Summarizing run...", flush=True)
+ instance._finalize_benchmarks(
real_world_latency_ms,
frame_cnt,
total_model_preprocess,
total_model_inference,
total_model_postprocess,
)
+ instance.stop()
+
+ # Force early hardware driver sweep before unbinding threads
+ if instance.device_input == "cuda" and torch.cuda.is_available():
+ torch.cuda.synchronize()
+ torch.cuda.empty_cache()
+ torch.cuda.ipc_collect()
+
+ gc.collect()
+
+ try:
+ if STATE_CAPTURE:
+ # --- CAPTURE STATE POINT B (Right before cleanup) ---
+ main_app_logger.info(
+ "[DIAGNOSTIC] Gathering active execution workspace layouts..."
+ )
+ state_after_stop = instance.capture_state_snapshot()
+
+ # 3. Print the granular delta analysis out to your terminal screen
+ new_keys_generated, mutated_keys, static_keys = (
+ instance.print_lifecycle_delta(
+ instance.baseline_before_start,
+ state_after_stop,
+ return_keys=True,
+ )
+ )
+ # self.config.DEBUG_FLAG and
+ if len(new_keys_generated) > 0 or len(mutated_keys) > 0:
+ # Inform the blueprint engine to eliminate every difference found between Point A and B
+ instance.stop_blueprint_executor(new_keys_generated, mutated_keys)
+ except Exception:
+ traceback.print_exc()
+
+ assert instance.status == "DONE"
+
+ # TEARDOWN
+ instance.execute_teardown()
+
+ if str2bool(os.getenv("ENABLE_PROFILING", "False")):
+ gc.collect()
+ if (
+ device == "gpu" and torch.cuda.is_available()
+ ): # self.device_input == "cuda" and torch.cuda.is_available():
+ torch.cuda.synchronize()
+ torch.cuda.empty_cache()
+ instance.assess_memory(
+ device, instance.name, start_allocated, start_reserved
+ )
+
+ if len(instance.__class__.benchmarks) > 0:
+ final_metrics = instance.__class__.benchmarks[-1]
+ res_queue.put({"status": "success", "metrics": final_metrics})
+ else:
+ res_queue.put({"status": "error", "error": "No benchmarks generated."})
- # HELPERS --------------------------------------------
- def get_model_by_device(self, device, sf_enabled=False):
- """Singleton loader: only loads if device changes or model is missing."""
- if (
- sf_enabled
- and (self.frame_width * self.frame_height)
- <= self.config.SMART_FILTERING_PIXEL_CONSTRAINT
- ):
- sf_enabled = False
-
- if (
- TestSmartFilteringDetections._shared_model is not None
- and TestSmartFilteringDetections._shared_model_device == device
- and TestSmartFilteringDetections._shared_model_sf_enabled == sf_enabled
- ):
- self.model = TestSmartFilteringDetections._shared_model
- self.model_path = TestSmartFilteringDetections._shared_model_path
- return
-
- run_platform_name = "engine" if "cuda" in self.device_input else "openvino"
-
- if self.config.CUSTOM_MODEL_FLAG:
- dir_path = "/home/resources/models/ultralytics/custom_models"
+ except Exception as e:
+ res_queue.put(
+ {"status": "error", "error": str(e), "traceback": traceback.format_exc()}
+ )
+ res_queue.put(
+ {"status": "error", "error": str(e), "traceback": traceback.format_exc()}
+ )
+
+
+# =========================================================================
+# PYTEST TEST HARNESS
+# =========================================================================
+@pytest.fixture(scope="class")
+def setup_context(request):
+ """Replaces setUpClass: Runs once per test class."""
+ # Initialize shared paths/results
+ current_test_filename = Path(__file__).stem
+ test_dir = Path(__file__).parent
+ main_path = test_dir.parent
+ video_dir = main_path / "inputs"
+
+ request.cls.source = os.getenv("VIDEO_FILENAME", "anduril_swarm_8K.mp4")
+ is_rtsp = "rtsp://" in request.cls.source
+ if not is_rtsp:
+ VIDEO_FILENAME = request.cls.source
+ if video_dir.exists():
+ vid_source = video_dir / VIDEO_FILENAME
else:
- dir_path = f"/home/resources/models/ultralytics/{self.config.MODEL_NAME}/{self.config.MODEL_PRECISION}"
-
- (
- TestSmartFilteringDetections._shared_model,
- TestSmartFilteringDetections._shared_model_path,
- self.label_source,
- ) = get_model(
- Path(dir_path),
- self.config.MODEL_NAME,
- run_platform_name,
- self.device_input,
- batch=self.config.MODEL_MAX_BATCH_SIZE,
- force_export=force_export,
- sf_enabled=sf_enabled,
- model_h=self.resize_h,
- model_w=self.resize_w,
+ video_dir = Path("/watch_dir")
+ vid_source = video_dir / VIDEO_FILENAME
+
+ assert vid_source.exists()
+ request.cls.source = str(vid_source)
+ request.cls.name = vid_source.stem
+ request.cls.is_rtsp = False
+ else:
+ request.cls.name = "rtsp"
+ request.cls.is_rtsp = True
+
+ model_name = os.getenv("MODEL_NAME", MODEL_NAME_DEFAULT)
+ request.cls.result_dir = test_dir / f"{current_test_filename}_results/{model_name}"
+ request.cls.result_dir.mkdir(parents=True, exist_ok=True)
+
+ # Benchmark statistics
+ request.cls.benchmarks = []
+ request.cls.csv_filename = f"model_benchmarks_{request.cls.name}.csv"
+ request.cls.csv_path = request.cls.result_dir / request.cls.csv_filename
+
+ # Initialize class vars
+ # request.cls.active = True
+ request.cls.active_streams = {}
+
+ # RUN ALL PARAMETERIZED TESTS ----------------------------------------
+ yield
+
+ # FINAL CSV EXPORT --------------------------------------------------
+ if request.cls.benchmarks:
+ # Filter and exclude rows that were interrupted or failed initialization due to an early pytest skip
+ request.cls.benchmarks = [
+ r for r in request.cls.benchmarks if r and "Test Name" in r
+ ]
+ results = request.cls.benchmarks
+ if results:
+ keys = results[0].keys()
+
+ with open(str(request.cls.csv_path), "w", newline="") as f:
+ dict_writer = csv.DictWriter(f, fieldnames=keys)
+ dict_writer.writeheader()
+ dict_writer.writerows(results)
+
+ print(
+ f"\n[FINAL] Benchmarks saved to {request.cls.csv_filename}", flush=True
+ )
+
+
+@pytest.mark.usefixtures("setup_context")
+class TestModel(BaseTest):
+ """
+ Pytest runner that spawns the isolated worker, waits for completion,
+ and merges the returned metrics into the main class for CSV/Chart generation.
+ """
+
+ benchmarks = [] # Class-level attribute required by _finalize_benchmarks
+
+ @pytest.mark.parametrize("device", ["gpu", "cpu"])
+ def test_model(self, device):
+ # Pull the values dynamically assigned by the setup_context fixture
+ init_args = {
+ "source": self.__class__.source,
+ "name": self.__class__.name,
+ "result_dir": str(self.__class__.result_dir),
+ "active_streams": {},
+ "benchmarks": self.__class__.benchmarks,
+ }
+
+ os.environ["DEVICE"] = device
+ detection_type = "object" # request.node.callspec.params.get("detection_type")
+ sf_enabled = False # request.node.callspec.params.get("sf_enabled")
+ gt_enabled = False
+
+ test_args = {
+ "device": device,
+ "detection_type": detection_type,
+ "sf_enabled": sf_enabled,
+ "gt_enabled": gt_enabled,
+ }
+
+ main_app_logger.info(
+ f"\n{'=' * 60}\n"
+ f"[TEST HARNESS] Spawning isolated high-speed process for {self.__class__.name} (SF: {sf_enabled})...\n"
+ f"{'=' * 60}"
)
- TestSmartFilteringDetections._shared_model_device = device
- TestSmartFilteringDetections._shared_model_sf_enabled = sf_enabled
- self.model = TestSmartFilteringDetections._shared_model
- # self.model.half()
- self.model_path = TestSmartFilteringDetections._shared_model_path
+ # Spawn a pristine Python process (identical to test_pipeline.py's stream_worker)
+ ctx = mp.get_context("spawn")
+ res_queue = ctx.Queue()
+
+ worker_p = ctx.Process(
+ target=isolated_detection_worker, args=(init_args, test_args, res_queue)
+ )
+
+ worker_p.start()
+ worker_p.join() # Block Pytest until the isolated pipeline finishes
+
+ # Extract and merge results
+ if not res_queue.empty():
+ result = res_queue.get()
- W, H = self.resize_w, self.resize_h
- if not sf_enabled:
- W, H = self.frame_width, self.frame_height
- self.model_warmup(H, W)
+ if result["status"] == "error":
+ pytest.fail(
+ f"Pipeline crashed in worker process:\n{result.get('error')}\n{result.get('traceback')}"
+ )
+ else:
+ metrics = result["metrics"]
+ self.__class__.benchmarks.append(metrics)
+
+ main_app_logger.info(
+ f"[TEST HARNESS] Worker returned successfully. Model Est. FPS: {metrics.get('Model Est. FPS')}"
+ )
+
+ # Basic functionality assertions
+ assert metrics is not None, "Metrics dictionary should not be None."
+ assert int(metrics.get("Frames Processed", 0)) > 0, (
+ "No frames were written to output."
+ )
+ else:
+ pytest.fail("Worker process died unexpectedly without returning metrics.")
+
+ # HELPERS --------------------------------------------
def _finalize_benchmarks(
self,
real_world_latency_ms,
@@ -317,7 +663,7 @@ def _finalize_benchmarks(
"Detection Type": self.config.DETECTION_TYPE,
"Device": self.device,
"Smart Filtering": "Enabled" if self.config.sf_enabled else "Disabled",
- "Video": self.video_path.name,
+ "Video": self.name,
"Video Duration (s)": f"{duration_s:.4f}",
"Video FPS": f"{self.input_fps:.2f}",
"Pipeline Latency (s)": f"{real_latency_s:.2f}",
@@ -337,14 +683,27 @@ def _finalize_benchmarks(
f"\n[{self._testMethodName}] Model Est. FPS: {model_fps:.2f} ({n_frames} frames)"
)
+ def setup_threads(self):
+ """Overrides handlers.py to bind threads dynamically to the test instance."""
+ pass
-# INHERIT METHODS FROM HANDLERS -----------------------------------------------------------------
-TestSmartFilteringDetections.setup_reader = DeviceBaseHandler.setup_reader
-TestSmartFilteringDetections.get_frameWH = DeviceBaseHandler.get_frameWH
-TestSmartFilteringDetections.run_model = DeviceBaseHandler.run_model
-TestSmartFilteringDetections.model_warmup = DeviceBaseHandler.model_warmup
+ def start(self):
+ """
+ Starts the decoupled ingestion and inference threads in the correct order.
+ """
+ # PRE-SYNC: Ensure GPU is idle before timing starts
+ if self.device_input == "cuda" and torch.cuda.is_available():
+ torch.cuda.synchronize()
+ # Start the hardware-decoupled reader first
+ # self.reader.start()
+ return self
+
+
+# =========================================================================
+# MAIN
+# =========================================================================
if __name__ == "__main__":
# TEST ARGUMENTS
parser = argparse.ArgumentParser(description="Run Video Detection Pipeline Tests")
@@ -374,41 +733,14 @@ def _finalize_benchmarks(
# Filter tests
# parser.add_argument(
- # "--type",
+ # "--device",
# type=str,
- # choices=["object", "motion"],
- # dest="detection_type",
- # help="Filter by detection type (object or motion)",
- # )
- parser.add_argument(
- "--device",
- type=str,
- choices=["cpu", "gpu"],
- help="Filter by device (cpu or gpu)",
- )
- # parser.add_argument(
- # "--sf",
- # action="store_true",
- # default=None,
- # dest="sf_enabled",
- # help="Filter by Smart Filtering",
- # )
-
- # # DEBUGGING
- # parser.add_argument(
- # "--debug",
- # action="store_true",
- # help="Enable debug message and save intermediate images for Smart Filtering tests",
- # )
- # parser.add_argument(
- # "-n",
- # type=int,
- # default=100,
- # dest="debug_frame_limit",
- # help="Number of frames used for debugging [Default: 100]",
+ # choices=["cpu", "gpu"],
+ # help="Filter by device (cpu or gpu)",
# )
args = parser.parse_args()
+ args.device = "gpu"
# UPDATE ENVIRONMENTAL VARIABLES
os.environ["VIDEO_FILENAME"] = args.source
@@ -431,6 +763,8 @@ def _finalize_benchmarks(
"-s",
"-v",
"--log-cli-level=DEBUG",
+ "-W",
+ "ignore::_pytest.warning_types.PytestAssertRewriteWarning",
__file__,
] # -s -v --log-cli-level=DEBUG
if run_args:
diff --git a/fastapi/tests/test_pipeline.py b/fastapi/tests/test_pipeline.py
index fb924dd..d5a5192 100644
--- a/fastapi/tests/test_pipeline.py
+++ b/fastapi/tests/test_pipeline.py
@@ -1,15 +1,79 @@
+# ==============================================================================
+# SUPPRESS WARNINGS
+# import warnings
+
+import pytest
+
+# warnings.filterwarnings(
+# "ignore", category=FutureWarning, message=".*reduce_op` is deprecated.*"
+# )
+
+pytestmark = [
+ pytest.mark.filterwarnings(
+ "ignore:.*anyio:_pytest.warning_types.PytestAssertRewriteWarning"
+ ),
+ pytest.mark.filterwarnings(
+ "ignore:Context managers for TensorRT types are deprecated:DeprecationWarning"
+ ),
+ pytest.mark.filterwarnings(
+ "ignore:Exception ignored in.*SharedMemory.__del__:UserWarning"
+ ),
+ # Global message fallback pattern captures local execution frames
+ pytest.mark.filterwarnings("ignore:.*reduce_op.*:FutureWarning"),
+]
+
+# ==============================================================================
+# LOGGING
+import logging
+import os
+import sys
+
+logging.basicConfig(
+ level=logging.INFO,
+ # format="%(asctime)s - %(name)s - %(levelname)s - %(message)s",
+ format="%(asctime)s [%(levelname)s] %(name)s (%(filename)s:%(lineno)d) - %(message)s",
+ handlers=[logging.StreamHandler(sys.stdout)],
+)
+# Suppress low-delay reference block warnings from OpenCV/PyAV/FFmpeg
+# os.environ["OPENCV_FFMPEG_LOGLEVEL"] = "-8"
+# os.environ["OPENCV_LOG_LEVEL"] = "OFF"
+logging.getLogger("libav").setLevel(logging.CRITICAL)
+logging.getLogger("libav.hevc").setLevel(logging.CRITICAL)
+logging.getLogger("matplotlib").setLevel(logging.WARNING)
+logging.getLogger("ultralytics").setLevel(logging.WARNING)
+# logger = trt.Logger(trt.Logger.WARNING)
+# trt.init_libnvinfer_plugins(logger, "")
+main_app_logger = logging.getLogger(__name__)
+
+# ==============================================================================
+# IMPORTS
+
+# os.environ["OMP_NUM_THREADS"] = "2"
+# os.environ["MKL_NUM_THREADS"] = "2"
+# os.environ["PYTORCH_ALLOC_CONF"] = "expandable_segments:True"
+# try:
+# torch.set_num_interop_threads(2) # 1)
+# torch.set_num_threads(4) # 2)
+# except RuntimeError:
+# # Safe graceful fallback if a process-level fork duplicated context maps
+# pass
import argparse
import csv
+import ctypes
+import faulthandler
import gc
+import inspect
import logging
import multiprocessing
-import os
import sys
+import threading
import time
import traceback
+import tracemalloc
from datetime import datetime
from pathlib import Path
+import cv2
import psutil
import pytest
import torch
@@ -21,52 +85,67 @@
MODEL_NAME_DEFAULT,
THRESHOLD_VALUE,
)
-
-# Import the exact core handlers to replace duplicate script code
-# from include.handlers import BASE_PIPELINE_CONFIG
+from include.handlers import (
+ CPUStreamHandler,
+ GPUStreamHandler,
+ log_to_logger,
+)
from include.utils import (
PipelineConfig,
+ ResourceTrackerFilter,
+ install_and_load_pip_package,
+ str2bool,
)
+# torch.set_grad_enabled(False)
+
+# ==============================================================================
+# MONKEY PATCH
+# Cache the hardware availability state globally.
+# This prevents internal framework loops from triggering driver/NVML checks per frame.
+# _cuda_available = torch.cuda.is_available()
+
+
+# def _patched_is_available():
+# return _cuda_available
+
+
+# torch.cuda.is_available = _patched_is_available
+
+# ==============================================================================
+# SETUP
+# Hooks directly into the OS kernel signals to force Python to print a full
+# stack traceback right before it dies, allowing you to see which line of
+# Python code caused the hard crash
+faulthandler.enable()
+
try:
+ # Retain standard spawn mode to prevent CUDA context driver deadlocks
torch.multiprocessing.set_start_method("spawn", force=True)
except RuntimeError:
pass
+force_export = False
+target_width, target_height = 7680, 4320
+STATE_CAPTURE = False
-logging.basicConfig(
- level=logging.INFO,
- format="%(asctime)s - %(name)s - %(levelname)s - %(message)s",
- handlers=[logging.StreamHandler(sys.stdout)],
-)
-# Suppress low-delay reference block warnings from OpenCV
-os.environ["OPENCV_LOG_LEVEL"] = "OFF"
-
-main_app_logger = logging.getLogger(__name__)
-
-
-def log_to_logger(message, level="info"):
- try:
- if level.lower() == "debug":
- main_app_logger.debug(message)
- elif level.lower() == "warning":
- main_app_logger.warning(message)
- else:
- main_app_logger.info(message)
- except Exception:
- pass
+# Force Python's multiprocessing layer to quiet tracking cleanup race conditions
+os.environ["PYTHONWARNINGS"] = "ignore"
+sys.warnoptions.append("ignore")
+# ==============================================================================
+# TEST SETUP / FUNCTIONS
@pytest.fixture(scope="class")
def setup_context(request):
- """Orchestrates test directories and configuration schemas."""
+ """Replaces setUpClass: Runs once per test class."""
current_test_filename = Path(__file__).stem
test_dir = Path(__file__).parent
main_path = test_dir.parent
video_dir = main_path / "inputs"
# Resolve source from CLI/Environment parameters
- request.cls.source = os.getenv("STREAM_SOURCE", "anduril_swarm_8K.mp4")
+ request.cls.source = os.getenv("VIDEO_FILENAME", "anduril_swarm_8K.mp4")
is_rtsp = "rtsp://" in request.cls.source
if not is_rtsp:
VIDEO_FILENAME = request.cls.source
@@ -79,23 +158,27 @@ def setup_context(request):
assert vid_source.exists()
request.cls.source = str(vid_source)
request.cls.name = vid_source.stem
+ request.cls.is_rtsp = False
else:
request.cls.name = "rtsp"
- request.cls.test_duration_mins = float(os.getenv("TEST_DURATION_MINS", 2.0))
+ request.cls.is_rtsp = True
+ request.cls.test_duration_mins = float(os.getenv("TEST_DURATION_MINS", 1.0))
- request.cls.result_dir = test_dir / f"{current_test_filename}_results"
+ model_name = os.getenv("MODEL_NAME", MODEL_NAME_DEFAULT)
+ request.cls.result_dir = test_dir / f"{current_test_filename}_results" / model_name
request.cls.result_dir.mkdir(parents=True, exist_ok=True)
request.cls.benchmarks = []
- if is_rtsp:
- request.cls.csv_filename = "reader_perf_results_rtsp.csv"
- else:
- vid_shortname = Path(request.cls.source).stem
- request.cls.csv_filename = f"reader_perf_results_{vid_shortname}.csv"
+ request.cls.csv_filename = f"pipeline_benchmarks_{request.cls.name}.csv"
request.cls.csv_path = request.cls.result_dir / request.cls.csv_filename
+ # request.cls.active = True
+ request.cls.active_streams = {}
+
+ # RUN ALL PARAMETERIZED TESTS ----------------------------------------
yield
+ # FINAL CSV EXPORT --------------------------------------------------
if request.cls.benchmarks:
ordered_headers = [
"timestamp",
@@ -116,6 +199,7 @@ def setup_context(request):
"total_frames_ingested",
"total_target_frames_processed",
"total_objects_detected",
+ "avg_detections_per_frame",
"frames_dropped_or_skipped",
"dropped_frame_sequences",
"average_read_latency_ms",
@@ -131,7 +215,8 @@ def setup_context(request):
]
# keys = {k for r in request.cls.benchmarks for k in r.keys()}
keys = []
- for r in request.cls.benchmarks:
+ results = request.cls.benchmarks
+ for r in results:
for k in r.keys():
keys.append(k)
sorted_keys = []
@@ -139,28 +224,46 @@ def setup_context(request):
if c in keys:
sorted_keys.append(c)
with open(str(request.cls.csv_path), "w", newline="") as f:
- writer = csv.DictWriter(f, fieldnames=list(sorted_keys))
- writer.writeheader()
- writer.writerows(request.cls.benchmarks)
- print(f"\n[FINAL] Telemetry saved to: {request.cls.csv_path}", flush=True)
+ dict_writer = csv.DictWriter(f, fieldnames=list(sorted_keys))
+ dict_writer.writeheader()
+ dict_writer.writerows(results)
+ main_app_logger.info(f"[FINAL] Telemetry saved to {request.cls.csv_path}")
@pytest.fixture(autouse=True)
def each_test_setup(request):
- if torch.cuda.is_available():
- torch.cuda.synchronize()
-
device = request.node.callspec.params.get("device")
os.environ["DEVICE"] = device
- yield
+ # Enforces a brief structural pause between execution sequences.
+ # This ensures old CUDA streams and file pointers are fully reclaimed by the OS
+ # before the next stage begins allocation.
+ with torch.inference_mode():
+ if device == "gpu" and torch.cuda.is_available():
+ torch.cuda.synchronize()
+ torch.cuda.ipc_collect()
+ torch.cuda.empty_cache()
+ gc.collect()
+ # Introduce a 250ms sub-slice delay to give Linux background
+ # resource tracking daemons time to complete unlinking procedures smoothly.
+ # time.sleep(0.25)
+
+ # RUN PARAMETERIZED TEST ----------------------------------------
+ try:
+ yield
+
+ except Exception as e:
+ traceback.print_exc()
+ main_app_logger.info(f"[TEST] Error: {e}")
+
+ # Evict the device cache pool entirely
gc.collect()
- if torch.cuda.is_available():
- torch.cuda.synchronize()
- torch.cuda.empty_cache()
- torch.cuda.ipc_collect()
- time.sleep(0.2)
+ with torch.inference_mode():
+ if device == "gpu" and torch.cuda.is_available():
+ torch.cuda.synchronize()
+ torch.cuda.empty_cache()
+ torch.cuda.memory._record_memory_history(enabled=False)
def stream_worker(
@@ -185,6 +288,8 @@ def stream_worker(
os.environ["OPENCV_LOG_LEVEL"] = "OFF"
os.environ["OPENCV_VIDEOIO_DEBUG"] = "0"
os.environ["FFMPEG_LOG_LEVEL"] = "quiet"
+
+ # Initialize test
test_duration_secs = test_duration_mins * 60
metrics = {
"timestamp": datetime.now().strftime("%Y-%m-%d %H:%M:%S"),
@@ -219,12 +324,12 @@ def stream_worker(
"total_frames_ingested": 0,
"total_target_frames_processed": 0,
"total_objects_detected": 0,
+ "avg_detections_per_frame": 0,
"detection_type": detection_type,
"smart_filter_active": sf_enabled,
"status": "INIT",
}
- loop_start = time.perf_counter()
process = psutil.Process(os.getpid())
cpu_samples, ram_samples, vram_samples = [], [], []
prefetch_backlog_samples = []
@@ -235,7 +340,7 @@ def stream_worker(
CUSTOM_MODEL_FLAG=os.getenv(
"CUSTOM_MODEL_FLAG", CUSTOM_MODEL_FLAG_DEFAULT
), # True,
- DEVICE="GPU" if device_type.lower() == "gpu" else "CPU",
+ DEVICE=device_type.upper(),
OMIT_DETECTIONS_FLAG=True,
TEST_MODE=True,
DEBUG=os.getenv("DEBUG", DEBUG_DEFAULT),
@@ -259,9 +364,9 @@ def stream_worker(
if out_dir:
if "Scenario_4_" in test_name:
- result_dir = out_dir / "results"
- result_dir.mkdir(parents=True, exist_ok=True)
- os.environ["TEST_SUITE_RENDER_DIR"] = str(result_dir)
+ # result_dir = out_dir / "results"
+ # result_dir.mkdir(parents=True, exist_ok=True)
+ os.environ["TEST_SUITE_RENDER_DIR"] = str(out_dir)
config.SHARED_OUTPUT = str(out_dir)
import vdms
@@ -271,39 +376,206 @@ def mock_connect(self, host, port):
vdms.vdms.connect = mock_connect
- from include.handlers import CPUStreamHandler, GPUStreamHandler
-
if device_type.lower() == "gpu":
HandlerClass = GPUStreamHandler
else:
HandlerClass = CPUStreamHandler
+ # Establish an accurate post-initialization hardware baseline
+ if str2bool(os.getenv("ENABLE_PROFILING", "False")):
+ tracemalloc.start()
+
+ start_allocated, start_reserved = 0, 0
+ if device_type.lower() and torch.cuda.is_available():
+ torch.cuda.memory._record_memory_history(
+ enabled=True,
+ trace_alloc_max_entries=250000,
+ trace_alloc_record_context=True,
+ )
+
+ # This captures your baseline AFTER initialize_variables() is done
+ start_allocated = torch.cuda.memory_allocated(0)
+ start_reserved = torch.cuda.memory_reserved(0)
+
+ try:
+ cv2.cuda.setBufferPoolUsage(False)
+ except AttributeError:
+ pass
+ else:
+ gc.collect()
+ process = psutil.Process(os.getpid())
+ start_allocated = (
+ process.memory_info().rss
+ ) # Reuse start_allocated container for host memory baseline
+ start_reserved = 0
+ log_to_logger("[STREAM_WORKER] Starting ...", level="info")
+
last_sample = time.perf_counter()
+ loop_start = last_sample
handler = None # Explicit initializing tracking state pointer 🚀
try:
- # Let handlers.py completely manage the reader, threads, ring buffers, and FFMPEG process
+ # Get Handler and start
handler = HandlerClass(
source=source, name=source_name, active_streams={}, config=config
- )
- handler.start()
-
- while (time.perf_counter() - loop_start) < test_duration_secs:
- if not handler.active:
- break
-
- curr_time = time.perf_counter()
- if (curr_time - last_sample) >= 0.5:
- cpu_samples.append(psutil.cpu_percent(interval=0))
- ram_samples.append(process.memory_info().rss / (1024 * 1024))
- if device_type.lower() == "gpu" and torch.cuda.is_available():
- vram_free, vram_total = torch.cuda.mem_get_info()
- vram_samples.append((vram_total - vram_free) / (1024 * 1024))
- if getattr(handler, "prefetch_queue", None) is not None:
- current_backlog = handler.prefetch_queue.qsize()
- prefetch_backlog_samples.append(current_backlog)
- last_sample = curr_time
- time.sleep(0.01)
+ ) # .start()
+
+ def profiler_fn():
+ profiler = None
+ try:
+ if str2bool(os.getenv("ENABLE_PROFILING", "False")):
+ Profiler = install_and_load_pip_package(
+ "pyinstrument", attribute_name="Profiler"
+ )
+
+ profiler = Profiler(interval=0.005) # 5ms sampling interval
+
+ # Telling the statistical sampler to skip recording exception blocks completely
+ # stops stack_sampler.py from ballooning RAM over long production runs.
+ if hasattr(profiler, "_sampler") and profiler._sampler:
+ profiler._sampler.trace_exceptions = False
+
+ profiler.start()
+
+ # orig_fn(profiler)
+ handler.run_realtime_inference(
+ sf_enabled=handler.config.sf_enabled,
+ profiler=profiler,
+ )
+
+ if str2bool(os.getenv("ENABLE_PROFILING", "False")):
+ # 2. Redirect standard error to the filter trap right before report compilation
+ original_stderr = sys.stderr
+ sys.stderr = ResourceTrackerFilter(original_stderr)
+
+ try:
+ # Force standard stdout to flush out any lingering teardown messages
+ # BEFORE pyinstrument dumps its massive ASCII tree block.
+ sys.stdout.flush()
+
+ # main_app_logger.info(profiler.output_text(color=True))
+ # profiler.main_app_logger.info(color=True)
+ # prof_output = profiler.output_text(color=True)
+ # main_app_logger.info(
+ # f"\n=== LATENCY BREAKDOWN FOR {self.name} ({device}) ===\n{prof_output}\n",
+ #
+ # )
+
+ # Save a clean, interactive tree map for visual analysis
+ output_html_path = handler.output_path.replace(
+ ".mp4", "_profile.html"
+ )
+ # output_html_path = f"/tmp/profile_{video_name}_{device}.html"
+ profiler.write_html(output_html_path)
+ main_app_logger.info(
+ f"[PROFILER] Performance tree map exported to {output_html_path}",
+ )
+
+ finally:
+ sys.stderr = original_stderr
+ if "profiler" in locals():
+ try:
+ # Force the Python interpreter to detach pyinstrument's sampling hooks
+ sys.setprofile(None)
+
+ # Completely decouple internal statistical sessions to drop C-heap frames
+ if hasattr(profiler, "_last_session"):
+ profiler._last_session = None
+ if hasattr(profiler, "last_session"):
+ profiler.last_session = None
+
+ # Forcibly clear out internal memoryview strings caching tree metrics
+ if (
+ hasattr(profiler, "session")
+ and profiler.session is not None
+ ):
+ if hasattr(profiler.session, "frame_groups"):
+ profiler.session.frame_groups = None
+ if hasattr(profiler.session, "samples"):
+ profiler.session.samples = []
+
+ # Purge compiled tree metrics structures
+ profiler.session = None
+ del profiler
+ except Exception:
+ pass
+
+ # Trigger an immediate native Linux heap compression pass
+ # This grabs the newly abandoned pyinstrument C-heap blocks and
+ # flushes them to the OS before the fixture assessment snapshot fires!
+ gc.collect()
+ try:
+ libc = ctypes.CDLL("libc.so.6")
+ libc.malloc_trim(0)
+ except Exception:
+ pass
+
+ except Exception:
+ traceback.print_exc()
+
+ if str2bool(os.getenv("ENABLE_PROFILING", "False")) and hasattr(
+ handler, "process_thread"
+ ):
+ # Re-initialize the thread context using our safe profile wrapper proxy
+ handler.process_thread = threading.Thread(target=profiler_fn, daemon=True)
+
+ try:
+ handler.start()
+
+ exited = False
+ loop_start = time.perf_counter()
+ while (time.perf_counter() - loop_start) < test_duration_secs:
+ time.sleep(0.25)
+ if handler._is_stopped: # not handler.active:
+ exited = True
+ break
+ if getattr(handler, "status", None) == "DONE":
+ exited = True
+ break
+
+ if (
+ hasattr(handler, "process_thread")
+ and handler.process_thread is not None
+ ):
+ if not handler.process_thread.is_alive():
+ main_app_logger.info(
+ "[TEST HARNESS] Background worker exited. Breaking loop.",
+ )
+ break
+
+ # if (
+ # hasattr(handler, "process_thread")
+ # and handler.process_thread is not None
+ # ):
+ # if not handler.process_thread.is_alive():
+ # main_app_logger.info(
+ # "[TEST HARNESS] Background worker exited. Breaking loop.",
+ # )
+ # break
+
+ curr_time = time.perf_counter()
+ if (curr_time - last_sample) >= 0.5:
+ cpu_samples.append(psutil.cpu_percent(interval=0))
+ ram_samples.append(process.memory_info().rss / (1024 * 1024))
+ if device_type.lower() == "gpu" and torch.cuda.is_available():
+ vram_free, vram_total = torch.cuda.mem_get_info()
+ vram_samples.append((vram_total - vram_free) / (1024 * 1024))
+ if getattr(handler, "prefetch_queue", None) is not None:
+ current_backlog = handler.prefetch_queue.qsize()
+ prefetch_backlog_samples.append(current_backlog)
+ last_sample = curr_time
+ time.sleep(0.001) # 0.01
+
+ handler.stop_threads(["process_thread"])
+ # Cleanup active thread worker contexts safely if they exist
+ if handler is not None and getattr(handler, "active", False) and not exited:
+ try:
+ handler.stop()
+ except Exception:
+ pass
+ except Exception as loop_e:
+ # traceback.print_exc()
+ main_app_logger.info(f"[LOOP ERROR]: {loop_e}")
# Safely capture performance values out of production thread states
actual_duration = time.perf_counter() - loop_start
@@ -324,7 +596,7 @@ def mock_connect(self, host, port):
) # Total target slices saved/evaluated
metrics["total_objects_detected"] = handler.total_objects_detected
- metrics["total_frames_read"] = getattr(handler, "frame_count", 0)
+ metrics["total_frames_read"] = handler.abs_frame_num
metrics["smart_filter_active"] = sf_enabled
metrics["device"] = device_type
@@ -348,12 +620,18 @@ def mock_connect(self, host, port):
metrics["fallback_engine_triggered"] = (
1 if getattr(r, "use_cpu_decode_fallback", False) else 0
)
- if not str(handler.source).startswith("rtsp://"):
- metrics["video_duration"] = (
- round(r.numFrames / metrics["hardware_video_fps"], 2)
- if metrics["hardware_video_fps"] > 0
- else 0.0
- )
+ # if not str(handler.source).startswith("rtsp://"):
+ # metrics["video_duration"] = (
+ # round(r.total_input_frames / metrics["hardware_video_fps"], 2)
+ # if metrics["hardware_video_fps"] > 0
+ # else 0.0
+ # )
+
+ metrics["video_duration"] = (
+ round(handler.abs_frame_num / metrics["hardware_video_fps"], 2)
+ if metrics["hardware_video_fps"] > 0
+ else 0.0
+ )
h_telemetry = getattr(handler, "telemetry", {})
io_latencies = h_telemetry.get("ram_disk_io_write_ms", [])
@@ -378,6 +656,12 @@ def mock_connect(self, host, port):
else 0.0
)
+ metrics["avg_detections_per_frame"] = (
+ metrics["total_objects_detected"] / metrics["total_target_frames_processed"]
+ )
+
+ main_app_logger.info(metrics)
+
except Exception as err:
is_expected_fail = (
"Scenario_1" in test_name
@@ -388,6 +672,7 @@ def mock_connect(self, host, port):
"could not open/connect",
"failed to initialize stream reader endpoint",
"stream reader initialization failure",
+ "opencv videocapture failed to open",
"timed out",
]
)
@@ -405,6 +690,7 @@ def mock_connect(self, host, port):
level="warning",
)
metrics["status"] = f"CRASHED: {type(err).__name__}"
+
finally:
# Calculate system baseline averages across the sample tracking matrices safely
if cpu_samples:
@@ -425,12 +711,12 @@ def mock_connect(self, host, port):
sum(prefetch_backlog_samples) / len(prefetch_backlog_samples), 2
)
- # Cleanup active thread worker contexts safely if they exist
- if handler is not None and getattr(handler, "active", False):
- try:
- handler.stop()
- except Exception:
- pass
+ # # Cleanup active thread worker contexts safely if they exist
+ # if handler is not None and getattr(handler, "active", False) and not exited:
+ # try:
+ # handler.stop()
+ # except Exception:
+ # pass
# Provide a 200ms cool-down window for OpenVINO/PyTorch C++ worker threads
# to finish their internal teardown before Python destroys the process space.
@@ -447,10 +733,18 @@ def mock_connect(self, host, port):
result_queue.put(metrics)
+ if str2bool(os.getenv("ENABLE_PROFILING", "False")):
+ handler.assess_memory(
+ handler.config.DEVICE.lower(),
+ handler.name,
+ start_allocated,
+ start_reserved,
+ )
# Absolute kill switch safely reclaims orphaned third-party threads
# at the kernel level without crashing the parent pytest framework
time.sleep(0.1)
- os._exit(0)
+ # os._exit(0)
+ log_to_logger("[STREAM_WORKER] Processing complete. Exiting...", level="info")
@pytest.mark.usefixtures("setup_context")
@@ -463,12 +757,11 @@ def test_scenario_1_invalid_rtsp(self, device):
run_clipper = False
time_limit_m = round(self.test_duration_mins, 1)
- print(
+ main_app_logger.info(
f"\n========================================\n"
f"RUNNING TEST: {test_name} | Device: {device.upper()}\n"
f"Source Destination: {bad_uri}\n"
f"========================================",
- flush=True,
)
# Execute production workflow in completely isolated spawned sandbox
@@ -489,15 +782,20 @@ def test_scenario_1_invalid_rtsp(self, device):
),
)
worker_p.start()
- worker_p.join()
+ main_app_logger.info("[SCENARIO 1] Started processing ...")
+ worker_p.join() # timeout=2.0)
+ main_app_logger.info(
+ f"[SCENARIO 1] Stopped cleanly with exit code: {worker_p.exitcode}"
+ )
if worker_p.is_alive():
worker_p.terminate()
+ worker_p.join() # Ensure kernel resource tracking fully unlinks
worker_p.close() # Reclaims underlying OS file descriptors immediately
test_metrics = res_queue.get()
self.__class__.benchmarks.append(test_metrics)
- print(f"Test Status Result: {test_metrics.get('status')}\n", flush=True)
+ main_app_logger.info(f"Test Status Result: {test_metrics.get('status')}")
# Explicitly tear down queue thread pools to prevent resource tracking leaks
try:
@@ -514,12 +812,11 @@ def test_scenario_2_longevity_throughput(self, device):
test_name = "Scenario_2_Longevity_Throughput_Evaluation"
run_clipper = False
- print(
+ main_app_logger.info(
f"\n========================================\n"
f"RUNNING TEST: {test_name} | Device: {device.upper()}\n"
f"Source Destination: {self.source}\n"
f"========================================",
- flush=True,
)
# Execute production workflow in completely isolated spawned sandbox
@@ -540,7 +837,11 @@ def test_scenario_2_longevity_throughput(self, device):
),
)
worker_p.start()
+ main_app_logger.info("[SCENARIO 2] Started processing ...")
worker_p.join()
+ main_app_logger.info(
+ f"[SCENARIO 2] Stopped cleanly with exit code: {worker_p.exitcode}"
+ )
if worker_p.is_alive():
worker_p.terminate()
@@ -548,7 +849,7 @@ def test_scenario_2_longevity_throughput(self, device):
test_metrics = res_queue.get()
self.__class__.benchmarks.append(test_metrics)
- print(f"Test Status Result: {test_metrics.get('status')}\n", flush=True)
+ main_app_logger.info(f"Test Status Result: {test_metrics.get('status')}")
# Explicitly tear down queue thread pools to prevent resource tracking leaks
try:
@@ -566,16 +867,15 @@ def test_scenario_3_video_clipper(self, device):
# test_name = "Scenario_3_Clip_Generation_Evaluation"
run_clipper = True
- print(
+ main_app_logger.info(
f"\n========================================\n"
f"RUNNING TEST: {test_name} | Device: {device.upper()}\n"
f"Source Destination: {self.source}\n"
f"========================================",
- flush=True,
)
render_dir = self.result_dir / f"{self.name}/scenario3_{device}"
render_dir.mkdir(parents=True, exist_ok=True)
- test_duration_mins = 1.0
+ # test_duration_mins = 1.0
# Execute production workflow in completely isolated spawned sandbox
ctx = multiprocessing.get_context("spawn")
@@ -589,13 +889,17 @@ def test_scenario_3_video_clipper(self, device):
self.name,
render_dir,
device,
- test_duration_mins,
+ self.test_duration_mins,
res_queue,
run_clipper,
),
)
worker_p.start()
+ main_app_logger.info("[SCENARIO 3] Started processing ...")
worker_p.join()
+ main_app_logger.info(
+ f"[SCENARIO 3] Stopped cleanly with exit code: {worker_p.exitcode}"
+ )
if worker_p.is_alive():
worker_p.terminate()
@@ -603,7 +907,7 @@ def test_scenario_3_video_clipper(self, device):
test_metrics = res_queue.get()
self.__class__.benchmarks.append(test_metrics)
- print(f"Test Status Result: {test_metrics.get('status')}\n", flush=True)
+ main_app_logger.info(f"Test Status Result: {test_metrics.get('status')}")
# Explicitly tear down queue thread pools to prevent resource tracking leaks
try:
@@ -624,19 +928,18 @@ def test_scenario_4_detection_and_clipper(self, device, sf_enabled, detection_ty
run_clipper = True
disable_detection = False
- print(
+ main_app_logger.info(
f"\n========================================\n"
f"RUNNING TEST: {test_name} | Device: {device.upper()} | SF: {sf_enabled}\n"
f"Source Name: {self.name} | Destination: {self.source}\n"
f"========================================",
- flush=True,
)
render_dir = (
self.result_dir
/ f"{self.name}/scenario4_{device}/{detection_type}_{mode_str}"
)
render_dir.mkdir(parents=True, exist_ok=True)
- test_duration_mins = 1.0
+ # test_duration_mins = 1.0
# Execute production workflow in completely isolated spawned sandbox
ctx = multiprocessing.get_context("spawn")
@@ -650,7 +953,7 @@ def test_scenario_4_detection_and_clipper(self, device, sf_enabled, detection_ty
self.name,
render_dir,
device,
- test_duration_mins,
+ self.test_duration_mins,
res_queue,
run_clipper,
disable_detection,
@@ -659,7 +962,11 @@ def test_scenario_4_detection_and_clipper(self, device, sf_enabled, detection_ty
),
)
worker_p.start()
+ main_app_logger.info("[SCENARIO 4] Started processing ...")
worker_p.join()
+ main_app_logger.info(
+ f"[SCENARIO 4] Stopped cleanly with exit code: {worker_p.exitcode}"
+ )
if worker_p.is_alive():
worker_p.terminate()
@@ -668,11 +975,10 @@ def test_scenario_4_detection_and_clipper(self, device, sf_enabled, detection_ty
test_metrics = res_queue.get()
self.__class__.benchmarks.append(test_metrics)
if disable_detection:
- print(f"Test Status Result: {test_metrics.get('status')}\n", flush=True)
+ main_app_logger.info(f"Test Status Result: {test_metrics.get('status')}")
else:
- print(
- f"Test Status Result: {test_metrics.get('status')} w/ {test_metrics.get('total_objects_detected')} detections\n",
- flush=True,
+ main_app_logger.info(
+ f"Test Status Result: {test_metrics.get('status')} w/ {test_metrics.get('total_objects_detected')} detections",
)
# Explicitly tear down queue thread pools to prevent resource tracking leaks
@@ -686,8 +992,6 @@ def test_scenario_4_detection_and_clipper(self, device, sf_enabled, detection_ty
def get_available_scenarios():
- import inspect
-
available_scenarios = set()
for attr_name, _ in inspect.getmembers(
TestHybridStreamHandlers, predicate=inspect.isfunction
@@ -710,13 +1014,15 @@ def get_pytest_filter_expression(args, sorted_scenarios):
target_scenarios = args.scenario if args.scenario else sorted_scenarios
scenario_clauses = []
- print("\n" + "=" * 50)
- print("TARGET SELECTION PREVIEW")
- print("=" * 50)
+ main_app_logger.info("=" * 50)
+ main_app_logger.info("TARGET SELECTION PREVIEW")
+ main_app_logger.info("=" * 50)
for num in target_scenarios:
if num in basic_ids:
- print(f" 🔹 Scenario {num}: Standard routing (ignoring sub-filters)")
+ main_app_logger.info(
+ f" 🔹 Scenario {num}: Standard routing (ignoring sub-filters)"
+ )
scenario_clauses.append(f"scenario_{num}")
else:
clause = f"scenario_{num}"
@@ -732,25 +1038,37 @@ def get_pytest_filter_expression(args, sorted_scenarios):
clause += f" and {args.detection_type}"
applied_subs.append(f"type={args.detection_type}")
- sub_msg = (
- f" with sub-filters: {', '.join(applied_subs)}" if applied_subs else ""
- )
- print(f" ⚙️ Scenario {num}: Active compilation{sub_msg}")
+ if applied_subs:
+ applied_subs_str = ", ".join(applied_subs)
+ sub_msg = f" with sub-filters: {applied_subs_str}"
+ main_app_logger.info(
+ f" ⚙️ Scenario {num}: Active compilation{sub_msg}"
+ )
+ else:
+ sub_msg = (
+ f" with sub-filters: {', '.join(applied_subs)}"
+ if applied_subs
+ else ""
+ )
+ main_app_logger.info(
+ f" 🔹 Scenario {num}: Standard routing (ignoring sub-filters)"
+ )
scenario_clauses.append(f"({clause})")
# Safely join scenarios together with 'or' so they can execute side-by-side
- filter_expression = f"test_scenario_ and ({' or '.join(scenario_clauses)})"
+ scenario_clauses_str = " or ".join(scenario_clauses)
+ filter_expression = f"test_scenario_ and ({scenario_clauses_str})"
# Target hardware context selection filter applies globally across all test cases
if args.device.lower() != "all":
- print(f" 💻 Hardware Context Constraint: {args.device.upper()}")
+ main_app_logger.info(f" 💻 Hardware Context Constraint: {args.device.upper()}")
filter_expression = f"({filter_expression}) and {args.device.lower()}"
else:
- print(" 💻 Hardware Context Constraint: ALL AVAILABLE")
+ main_app_logger.info(" 💻 Hardware Context Constraint: ALL AVAILABLE")
- print("=" * 50)
- print(f"COMPILED PYTEST KEYWORD EXPRESSION:\n 👉 {filter_expression}")
- print("=" * 50 + "\n")
+ main_app_logger.info("=" * 50)
+ main_app_logger.info(f"COMPILED PYTEST KEYWORD EXPRESSION: {filter_expression}")
+ main_app_logger.info("=" * 50)
return filter_expression
@@ -774,7 +1092,7 @@ def get_pytest_filter_expression(args, sorted_scenarios):
"-d",
"--duration",
type=float,
- default=2.0,
+ default=1.0,
help="Test duration in minutes.",
)
parser.add_argument(
@@ -810,13 +1128,13 @@ def get_pytest_filter_expression(args, sorted_scenarios):
dest="detection_type",
help="Filter by detection type (object or motion)",
)
- parser.add_argument(
- "--device",
- type=str,
- default="all",
- choices=["cpu", "gpu", "all"],
- help="Target hardware context selection filter.",
- )
+ # parser.add_argument(
+ # "--device",
+ # type=str,
+ # default="all",
+ # choices=["cpu", "gpu", "all"],
+ # help="Target hardware context selection filter.",
+ # )
parser.add_argument(
"--sf",
action="store_true",
@@ -838,13 +1156,23 @@ def get_pytest_filter_expression(args, sorted_scenarios):
# dest="debug_frame_limit",
# help="Number of frames used for debugging [Default: 100]",
# )
+
+ # PROFILING
+ parser.add_argument(
+ "--profile",
+ action="store_true",
+ help="Enable profiling",
+ )
+
args = parser.parse_args()
+ args.device = "gpu"
- os.environ["STREAM_SOURCE"] = args.source
+ os.environ["VIDEO_FILENAME"] = args.source
os.environ["TEST_DURATION_MINS"] = str(args.duration)
os.environ["CUSTOM_MODEL_FLAG"] = "True" if args.custom_model_flag else "False"
os.environ["MODEL_NAME"] = args.model_name
os.environ["DEBUG"] = "1" if args.debug else "0"
+ os.environ["ENABLE_PROFILING"] = "True" if args.profile else "False"
# filter_expression = "test_scenario_4_detection_and_clipper"
# filter_expression = "test_scenario_3_video_clipper"
@@ -858,11 +1186,12 @@ def get_pytest_filter_expression(args, sorted_scenarios):
"-s",
"-v",
"--log-cli-level=DEBUG",
+ "-W",
+ "ignore::_pytest.warning_types.PytestAssertRewriteWarning",
__file__,
]
- print(
+ main_app_logger.info(
f"Launching decoupled testing suite configurations for destination targets: {args.source}",
- flush=True,
)
sys.exit(pytest.main(pytest_args))
diff --git a/fastapi/tests/test_readers.py b/fastapi/tests/test_readers.py
new file mode 100644
index 0000000..6358752
--- /dev/null
+++ b/fastapi/tests/test_readers.py
@@ -0,0 +1,775 @@
+# ==============================================================================
+# SUPPRESS WARNINGS
+# import warnings
+
+import pytest
+
+# warnings.filterwarnings(
+# "ignore", category=FutureWarning, message=".*reduce_op` is deprecated.*"
+# )
+
+pytestmark = [
+ pytest.mark.filterwarnings(
+ "ignore:.*anyio:_pytest.warning_types.PytestAssertRewriteWarning"
+ ),
+ pytest.mark.filterwarnings(
+ "ignore:Context managers for TensorRT types are deprecated:DeprecationWarning"
+ ),
+ pytest.mark.filterwarnings(
+ "ignore:Exception ignored in.*SharedMemory.__del__:UserWarning"
+ ),
+ # Global message fallback pattern captures local execution frames
+ pytest.mark.filterwarnings("ignore:.*reduce_op.*:FutureWarning"),
+]
+
+# ==============================================================================
+# LOGGING
+import logging
+import os
+import sys
+
+logging.basicConfig(
+ level=logging.INFO,
+ # format="%(asctime)s - %(name)s - %(levelname)s - %(message)s",
+ format="%(asctime)s [%(levelname)s] %(name)s (%(filename)s:%(lineno)d) - %(message)s",
+ handlers=[logging.StreamHandler(sys.stdout)],
+)
+
+# Suppress low-delay reference block warnings from OpenCV/PyAV/FFmpeg
+# os.environ["OPENCV_FFMPEG_LOGLEVEL"] = "-8"
+# os.environ["OPENCV_LOG_LEVEL"] = "OFF"
+logging.getLogger("libav").setLevel(logging.CRITICAL)
+logging.getLogger("libav.hevc").setLevel(logging.CRITICAL)
+logging.getLogger("matplotlib").setLevel(logging.WARNING)
+logging.getLogger("ultralytics").setLevel(logging.WARNING)
+# logger = trt.Logger(trt.Logger.WARNING)
+# trt.init_libnvinfer_plugins(logger, "")
+main_app_logger = logging.getLogger(__name__)
+
+# ==============================================================================
+# IMPORTS
+
+# os.environ["OMP_NUM_THREADS"] = "2"
+# os.environ["MKL_NUM_THREADS"] = "2"
+# os.environ["PYTORCH_ALLOC_CONF"] = "expandable_segments:True"
+# try:
+# torch.set_num_interop_threads(2) # 1)
+# torch.set_num_threads(4) # 2)
+# except RuntimeError:
+# # Safe graceful fallback if a process-level fork duplicated context maps
+# pass
+import argparse
+import asyncio
+import csv
+import ctypes
+import faulthandler
+import gc
+import multiprocessing as mp
+import sys
+import threading
+import time
+import traceback
+import tracemalloc
+from pathlib import Path
+
+import cv2
+import psutil
+import torch
+
+# import torch
+
+# Retrieve repo packages
+REPO_DIR = str(Path(__file__).parent.parent)
+sys.path.insert(1, REPO_DIR)
+from base_test import (
+ BaseTest,
+ fps_comparison_chart,
+)
+from include.default_configs import (
+ CUSTOM_MODEL_FLAG_DEFAULT,
+ DEBUG_DEFAULT,
+ MODEL_NAME_DEFAULT,
+ THRESHOLD_VALUE,
+)
+from include.handlers import (
+ get_test_handler,
+)
+from include.utils import (
+ PipelineConfig,
+ ResourceTrackerFilter,
+ install_and_load_pip_package,
+ str2bool,
+)
+
+# objgraph = install_and_load_pip_package("objgraph", attribute_name=None)
+# torch.set_grad_enabled(False)
+
+# ==============================================================================
+# MONKEY PATCH
+# Cache the hardware availability state globally.
+# This prevents internal framework loops from triggering driver/NVML checks per frame.
+# _cuda_available = torch.cuda.is_available()
+
+
+# def _patched_is_available():
+# return _cuda_available
+
+
+# torch.cuda.is_available = _patched_is_available
+
+# ==============================================================================
+# SETUP
+# Hooks directly into the OS kernel signals to force Python to print a full
+# stack traceback right before it dies, allowing you to see which line of
+# Python code caused the hard crash
+faulthandler.enable()
+
+try:
+ # Retain standard spawn mode to prevent CUDA context driver deadlocks
+ torch.multiprocessing.set_start_method("spawn", force=True)
+except RuntimeError:
+ pass
+
+force_export = False
+# target_width, target_height = 7680, 4320
+STATE_CAPTURE = False
+
+# Force Python's multiprocessing layer to quiet tracking cleanup race conditions
+os.environ["PYTHONWARNINGS"] = "ignore"
+sys.warnoptions.append("ignore")
+
+
+# =========================================================================
+# ISOLATED BACKGROUND WORKER
+# =========================================================================
+
+
+def isolated_detection_worker(init_args, test_args, res_queue):
+ device = test_args["device"]
+ detection_type = test_args["detection_type"]
+ sf_enabled = test_args["sf_enabled"]
+ gt_enabled = test_args["gt_enabled"]
+
+ if device == "gpu" and torch.cuda.is_available():
+ torch.cuda.synchronize()
+ torch.cuda.ipc_collect()
+ torch.cuda.empty_cache()
+ gc.collect()
+
+ # Establish an accurate post-initialization hardware baseline
+ if str2bool(os.getenv("ENABLE_PROFILING", "False")):
+ tracemalloc.start()
+
+ start_allocated, start_reserved = 0, 0
+ if device == "gpu" and torch.cuda.is_available():
+ torch.cuda.memory._record_memory_history(
+ enabled=True,
+ trace_alloc_max_entries=250000,
+ trace_alloc_record_context=True,
+ )
+
+ # This captures your baseline AFTER initialize_variables() is done
+ start_allocated = torch.cuda.memory_allocated(0)
+ start_reserved = torch.cuda.memory_reserved(0)
+
+ try:
+ cv2.cuda.setBufferPoolUsage(False)
+ except AttributeError:
+ pass
+ else:
+ gc.collect()
+ process = psutil.Process(os.getpid())
+ start_allocated = (
+ process.memory_info().rss
+ ) # Reuse start_allocated container for host memory baseline
+ start_reserved = 0
+
+ try:
+ instance, _ = get_test_handler(TestReader(), device)
+
+ instance.source = init_args["source"]
+ instance.name = init_args["name"]
+ instance.result_dir = Path(init_args["result_dir"])
+ instance.active_streams = init_args["active_streams"]
+ instance.read_frame_only = init_args["read_frame_only"]
+ instance.__class__.benchmarks = init_args["benchmarks"] # Sandbox the metrics
+
+ vid_dir = instance.result_dir / device
+ vid_dir.mkdir(parents=True, exist_ok=True)
+ os.environ["TEST_SUITE_RENDER_DIR"] = str(vid_dir)
+
+ # config definition
+ config = PipelineConfig(
+ SHARED_OUTPUT=str(instance.result_dir),
+ CUSTOM_MODEL_FLAG=os.getenv("CUSTOM_MODEL_FLAG", CUSTOM_MODEL_FLAG_DEFAULT),
+ DEVICE=device.upper(),
+ OMIT_DETECTIONS_FLAG=True,
+ TEST_MODE=True,
+ DEBUG=os.getenv("DEBUG", DEBUG_DEFAULT),
+ DEBUG_FRAME_LIMIT=int(os.getenv("DEBUG_FRAME_LIMIT", 100)),
+ ENABLE_QUERYING=False,
+ MODEL_NAME=os.getenv("MODEL_NAME", MODEL_NAME_DEFAULT),
+ SMART_FILTERING_ENABLED=sf_enabled,
+ THRESHOLD_VALUE=int(os.getenv("THRESHOLD_VALUE", THRESHOLD_VALUE)),
+ DETECTION_TYPE=detection_type,
+ )
+
+ # INITIALIZE CLASS (mimic DeviceBaseHandler.__init__) ------------------------------
+ instance.is_rtsp = str(instance.source).startswith("rtsp:/")
+ instance.active = True
+ instance.config = config
+
+ # kwarg definition
+ instance._testMethodName = f"{device.upper()}Reader"
+ # instance.video_output_name = (
+ # f"{instance._testMethodName}_{short_name}.mp4"
+ # )
+
+ instance.loop = asyncio.get_event_loop()
+ instance.frame_ready_event = asyncio.Event()
+ instance._is_stopped = False
+ instance._stop_lock = threading.Lock() # Local lock for this instance
+ instance.main_startup_event = mp.Event()
+
+ instance.device = instance.config.DEVICE
+ instance.device_input = instance.config.device_input
+ instance.disp_w, instance.disp_h = instance.config.DISPLAY_FRAME_SIZE
+ instance.resize_h, instance.resize_w = [
+ instance.config.MODEL_H,
+ instance.config.MODEL_W,
+ ]
+
+ instance.setup_reader(
+ instance.config.TARGET_FPS,
+ instance.config.CLIP_DURATION,
+ startup_event=instance.main_startup_event,
+ )
+ instance.initialize_variables()
+ instance.setup_threads()
+ instance.last_heartbeat = time.perf_counter()
+
+ if STATE_CAPTURE:
+ main_app_logger.info(
+ "[DIAGNOSTIC] Registering system state baseline layout..."
+ )
+ instance.baseline_before_start = instance.capture_state_snapshot()
+
+ def orig_fn(profiler):
+ return instance.run_realtime_inference(
+ sf_enabled=instance.config.sf_enabled,
+ profiler=profiler,
+ read_frame_only=instance.read_frame_only
+ if hasattr(instance, "read_frame_only")
+ else False,
+ gt_enabled=gt_enabled,
+ )
+
+ def profiler_fn():
+ profiler = None
+ try:
+ if str2bool(os.getenv("ENABLE_PROFILING", "False")):
+ Profiler = install_and_load_pip_package(
+ "pyinstrument", attribute_name="Profiler"
+ )
+
+ profiler = Profiler(interval=0.005) # 5ms sampling interval
+
+ # Telling the statistical sampler to skip recording exception blocks completely
+ # stops stack_sampler.py from ballooning RAM over long production runs.
+ if hasattr(profiler, "_sampler") and profiler._sampler:
+ profiler._sampler.trace_exceptions = False
+
+ profiler.start()
+
+ orig_fn(profiler)
+
+ if str2bool(os.getenv("ENABLE_PROFILING", "False")):
+ # 2. Redirect standard error to the filter trap right before report compilation
+ original_stderr = sys.stderr
+ sys.stderr = ResourceTrackerFilter(original_stderr)
+
+ try:
+ # Force standard stdout to flush out any lingering teardown messages
+ # BEFORE pyinstrument dumps its massive ASCII tree block.
+ sys.stdout.flush()
+
+ # main_app_logger.info(profiler.output_text(color=True))
+ # profiler.main_app_logger.info(color=True)
+ # prof_output = profiler.output_text(color=True)
+ # main_app_logger.info(
+ # f"\n=== LATENCY BREAKDOWN FOR {self.name} ({device}) ===\n{prof_output}\n",
+ #
+ # )
+
+ # Save a clean, interactive tree map for visual analysis
+ output_html_path = instance.output_path.replace(
+ ".mp4", "_profile.html"
+ )
+ # output_html_path = f"/tmp/profile_{video_name}_{device}.html"
+ profiler.write_html(output_html_path)
+ main_app_logger.info(
+ f"[PROFILER] Performance tree map exported to {output_html_path}",
+ )
+
+ finally:
+ sys.stderr = original_stderr
+ if "profiler" in locals():
+ try:
+ # Force the Python interpreter to detach pyinstrument's sampling hooks
+ sys.setprofile(None)
+
+ # Completely decouple internal statistical sessions to drop C-heap frames
+ if hasattr(profiler, "_last_session"):
+ profiler._last_session = None
+ if hasattr(profiler, "last_session"):
+ profiler.last_session = None
+
+ # Forcibly clear out internal memoryview strings caching tree metrics
+ if (
+ hasattr(profiler, "session")
+ and profiler.session is not None
+ ):
+ if hasattr(profiler.session, "frame_groups"):
+ profiler.session.frame_groups = None
+ if hasattr(profiler.session, "samples"):
+ profiler.session.samples = []
+
+ # Purge compiled tree metrics structures
+ profiler.session = None
+ del profiler
+ except Exception:
+ pass
+
+ # Trigger an immediate native Linux heap compression pass
+ # This grabs the newly abandoned pyinstrument C-heap blocks and
+ # flushes them to the OS before the fixture assessment snapshot fires!
+ gc.collect()
+ try:
+ libc = ctypes.CDLL("libc.so.6")
+ libc.malloc_trim(0)
+ except Exception:
+ pass
+
+ except Exception:
+ traceback.print_exc()
+
+ if str2bool(os.getenv("ENABLE_PROFILING", "False")) and hasattr(
+ instance, "process_thread"
+ ):
+ # Re-initialize the thread context using our safe profile wrapper proxy
+ instance.process_thread = threading.Thread(target=profiler_fn, daemon=True)
+
+ instance.VIDEO_GT_DETAILS = None
+ instance.duration_target = 30
+
+ instance.start()
+
+ while instance.active or not instance._is_stopped:
+ time.sleep(0.25)
+ if getattr(instance, "status", None) == "DONE":
+ break
+
+ if (
+ hasattr(instance, "process_thread")
+ and instance.process_thread is not None
+ ):
+ if not instance.process_thread.is_alive():
+ main_app_logger.info(
+ "[TEST HARNESS] Background worker exited. Breaking loop.",
+ )
+ break
+
+ instance.stop_threads(["process_thread"])
+
+ # Force early hardware driver sweep before unbinding threads
+ if instance.device_input == "cuda" and torch.cuda.is_available():
+ torch.cuda.synchronize()
+ torch.cuda.empty_cache()
+ torch.cuda.ipc_collect()
+
+ gc.collect()
+
+ try:
+ if STATE_CAPTURE:
+ # --- CAPTURE STATE POINT B (Right before cleanup) ---
+ main_app_logger.info(
+ "[DIAGNOSTIC] Gathering active execution workspace layouts..."
+ )
+ state_after_stop = instance.capture_state_snapshot()
+
+ # 3. Print the granular delta analysis out to your terminal screen
+ new_keys_generated, mutated_keys, static_keys = (
+ instance.print_lifecycle_delta(
+ instance.baseline_before_start,
+ state_after_stop,
+ return_keys=True,
+ )
+ )
+ # self.config.DEBUG_FLAG and
+ if len(new_keys_generated) > 0 or len(mutated_keys) > 0:
+ # Inform the blueprint engine to eliminate every difference found between Point A and B
+ instance.stop_blueprint_executor(new_keys_generated, mutated_keys)
+ except Exception:
+ traceback.print_exc()
+
+ assert instance.status == "DONE"
+
+ # TEARDOWN
+ instance.execute_teardown()
+
+ if str2bool(os.getenv("ENABLE_PROFILING", "False")):
+ gc.collect()
+ if (
+ device == "gpu" and torch.cuda.is_available()
+ ): # self.device_input == "cuda" and torch.cuda.is_available():
+ torch.cuda.synchronize()
+ torch.cuda.empty_cache()
+ instance.assess_memory(
+ device, instance.name, start_allocated, start_reserved
+ )
+
+ if len(instance.__class__.benchmarks) > 0:
+ final_metrics = instance.__class__.benchmarks[-1]
+ res_queue.put({"status": "success", "metrics": final_metrics})
+ else:
+ res_queue.put({"status": "error", "error": "No benchmarks generated."})
+
+ except Exception as e:
+ res_queue.put(
+ {"status": "error", "error": str(e), "traceback": traceback.format_exc()}
+ )
+ res_queue.put(
+ {"status": "error", "error": str(e), "traceback": traceback.format_exc()}
+ )
+
+
+# =========================================================================
+# PYTEST TEST HARNESS
+# =========================================================================
+@pytest.fixture(scope="class")
+def setup_context(request):
+ """Replaces setUpClass: Runs once per test class."""
+ current_test_filename = Path(__file__).stem
+ test_dir = Path(__file__).parent
+ main_path = test_dir.parent
+ video_dir = main_path / "inputs"
+
+ request.cls.read_frame_only = True
+
+ # model_name = os.getenv("MODEL_NAME", MODEL_NAME_DEFAULT)
+ # Handler.__init__ (main items)
+ request.cls.source = os.getenv("VIDEO_FILENAME", "anduril_swarm_8K.mp4")
+ is_rtsp = "rtsp://" in request.cls.source
+ if not is_rtsp:
+ VIDEO_FILENAME = request.cls.source
+ if video_dir.exists():
+ vid_source = video_dir / VIDEO_FILENAME
+ else:
+ video_dir = Path("/watch_dir")
+ vid_source = video_dir / VIDEO_FILENAME
+
+ assert vid_source.exists()
+ request.cls.source = str(vid_source)
+ request.cls.name = vid_source.stem
+ request.cls.is_rtsp = False
+ else:
+ request.cls.name = "rtsp"
+ request.cls.is_rtsp = True
+ # request.cls.test_duration_mins = float(os.getenv("TEST_DURATION_MINS", 0.25))
+
+ request.cls.result_dir = (
+ test_dir
+ / f"{current_test_filename}_results" # /{model_name}"
+ / request.cls.name
+ )
+ request.cls.result_dir.mkdir(parents=True, exist_ok=True)
+
+ # Benchmark statistics
+ request.cls.benchmarks = []
+ request.cls.csv_filename = f"reader_benchmarks_{request.cls.name}.csv"
+ request.cls.csv_path = request.cls.result_dir / request.cls.csv_filename
+
+ # request.cls.active = True
+ request.cls.active_streams = {}
+
+ # RUN ALL PARAMETERIZED TESTS ----------------------------------------
+ yield
+
+ # FINAL CSV EXPORT --------------------------------------------------
+ if request.cls.benchmarks:
+ # Filter and exclude rows that were interrupted or failed initialization due to an early pytest skip
+ request.cls.benchmarks = [
+ r for r in request.cls.benchmarks if r and "Test Name" in r
+ ]
+ results = request.cls.benchmarks
+ for row in results:
+ if "GPU" in row["Test Name"]:
+ match_name = row["Test Name"].replace("GPU", "CPU")
+ cpu_row = next(
+ (r for r in results if r["Test Name"] == match_name), None
+ )
+ if cpu_row:
+ gpu_fps = float(row["Pipeline FPS (Target frames)"])
+ cpu_fps = float(cpu_row["Pipeline FPS (Target frames)"])
+ speedup = (gpu_fps / cpu_fps) if cpu_fps > 0 else 0
+ row["Pipeline Speedup vs CPU"] = f"{speedup:.2f}x"
+ else:
+ row["Pipeline Speedup vs CPU"] = "N/A"
+ else:
+ row["Pipeline Speedup vs CPU"] = "Baseline (CPU)"
+
+ keys = results[0].keys()
+ with open(str(request.cls.csv_path), "w", newline="") as f:
+ dict_writer = csv.DictWriter(f, fieldnames=keys)
+ dict_writer.writeheader()
+ dict_writer.writerows(results)
+
+ main_app_logger.info(f"[FINAL] Benchmarks saved to {request.cls.csv_path}")
+
+ main_app_logger.info("=" * 80)
+ main_app_logger.info(
+ f"{'Test Name':<25} | {'Pipeline FPS (Target)':<21} | {'Avg Frame Reading (ms)':<22} | {'Pipeline Speedup vs CPU':<15}",
+ )
+ main_app_logger.info("-" * 80)
+
+ for r in results:
+ main_app_logger.info(
+ f"{r['Test Name']:<25} | {r['Pipeline FPS (Target frames)']:<21} | {r['Avg Frame Reading (ms)']:<22} | {r.get('Pipeline Speedup vs CPU', 'N/A'):<10}",
+ )
+ main_app_logger.info("=" * 125)
+
+ chart_path = (
+ request.cls.result_dir
+ / f"{request.cls.csv_filename.replace('.csv', '')}_pipelineFPS.png"
+ )
+ fps_comparison_chart(chart_path, results, fps_key="Pipeline FPS (Video frames)")
+
+ chart_path = (
+ request.cls.result_dir
+ / f"{request.cls.csv_filename.replace('.csv', '')}_pipelineFPS_target.png"
+ )
+ fps_comparison_chart(
+ chart_path, results, fps_key="Pipeline FPS (Target frames)"
+ )
+
+
+@pytest.mark.usefixtures("setup_context")
+class TestReader(BaseTest):
+ """
+ Pytest runner that spawns the isolated worker, waits for completion,
+ and merges the returned metrics into the main class for CSV/Chart generation.
+ """
+
+ benchmarks = [] # Class-level attribute required by _finalize_benchmarks
+
+ @pytest.mark.parametrize("device", ["gpu", "cpu"])
+ def test_device_reader(self, device):
+ # Pull the values dynamically assigned by the setup_context fixture
+ init_args = {
+ "source": self.__class__.source,
+ "name": self.__class__.name,
+ "result_dir": str(self.__class__.result_dir),
+ "active_streams": {},
+ "benchmarks": self.__class__.benchmarks,
+ "read_frame_only": self.__class__.read_frame_only,
+ }
+
+ detection_type = "object" # request.node.callspec.params.get("detection_type")
+ sf_enabled = False # request.node.callspec.params.get("sf_enabled")
+ gt_enabled = False
+
+ test_args = {
+ "device": device,
+ "detection_type": detection_type,
+ "sf_enabled": sf_enabled,
+ "gt_enabled": gt_enabled,
+ }
+
+ main_app_logger.info(
+ f"\n{'=' * 60}\n"
+ f"[TEST HARNESS] Spawning isolated high-speed process for {self.__class__.name} (SF: {sf_enabled})...\n"
+ f"{'=' * 60}"
+ )
+
+ # Spawn a pristine Python process (identical to test_pipeline.py's stream_worker)
+ ctx = mp.get_context("spawn")
+ res_queue = ctx.Queue()
+
+ worker_p = ctx.Process(
+ target=isolated_detection_worker, args=(init_args, test_args, res_queue)
+ )
+
+ worker_p.start()
+ worker_p.join() # Block Pytest until the isolated pipeline finishes
+
+ # Extract and merge results
+ if not res_queue.empty():
+ result = res_queue.get()
+
+ if result["status"] == "error":
+ pytest.fail(
+ f"Pipeline crashed in worker process:\n{result.get('error')}\n{result.get('traceback')}"
+ )
+ else:
+ metrics = result["metrics"]
+
+ self.__class__.benchmarks.append(metrics)
+
+ main_app_logger.info(
+ f"[TEST HARNESS] Worker returned successfully. Display FPS: {metrics.get('Display FPS')}"
+ )
+
+ # Basic functionality assertions
+ assert metrics is not None, "Metrics dictionary should not be None."
+ assert int(metrics.get("Output Frames", 0)) > 0, (
+ "No frames were written to output."
+ )
+ else:
+ pytest.fail("Worker process died unexpectedly without returning metrics.")
+
+
+# =========================================================================
+# MAIN
+# =========================================================================
+def get_pytest_filter_expression(args):
+ main_app_logger.info("\n" + "=" * 50)
+ main_app_logger.info("TARGET SELECTION PREVIEW")
+ main_app_logger.info("=" * 50)
+
+ filter_expression = Path(__file__).stem
+ # applied_subs = []
+
+ if args.sf_enabled is not None:
+ # Target exact parameter tokens generated by pytest parametrization
+ sf_str = "-True-" if args.sf_enabled else "-False-"
+ filter_expression += f" and {sf_str}"
+ # applied_subs.append(f"sf_enabled={args.sf_enabled}")
+
+ if args.detection_type:
+ filter_expression += f" and {args.detection_type}"
+
+ # Target hardware context selection filter applies globally across all test cases
+ if args.device.lower() != "all":
+ main_app_logger.info(f" 💻 Hardware Context Constraint: {args.device.upper()}")
+ filter_expression = f"({filter_expression}) and {args.device.lower()}"
+ else:
+ main_app_logger.info(" 💻 Hardware Context Constraint: ALL AVAILABLE")
+
+ main_app_logger.info("=" * 50)
+ main_app_logger.info(
+ f"COMPILED PYTEST KEYWORD EXPRESSION:\n 👉 {filter_expression}"
+ )
+ main_app_logger.info("=" * 50 + "\n")
+
+ return filter_expression
+
+
+if __name__ == "__main__":
+ # Force all PyTorch extension handles to compile and load BEFORE threads spawn
+ # if torch.cuda.is_available():
+ # _ = torch.zeros(1).cuda()
+ # import torch._ops
+ # import torch.utils
+
+ # TEST ARGUMENTS
+ parser = argparse.ArgumentParser(description="Run Video Detection Pipeline Tests")
+ parser.add_argument(
+ "-s",
+ "--source",
+ type=str,
+ default="anduril_swarm_8K.mp4",
+ help="Video filename (located in /inputs)",
+ )
+
+ # MODEL TO USE
+ parser.add_argument(
+ "--no-custom",
+ action="store_false",
+ dest="custom_model_flag",
+ help="Enable if using Ultralytics YOLO model",
+ )
+ parser.add_argument(
+ "-m",
+ "--model",
+ type=str,
+ default="drone_detection",
+ dest="model_name",
+ help="Name of model. Required if `--no-custom` is enabled. [Default: drone_detection]",
+ )
+
+ # Filter tests
+ parser.add_argument(
+ "--type",
+ type=str,
+ choices=["object", "motion"],
+ dest="detection_type",
+ help="Filter by detection type (object or motion)",
+ )
+ # parser.add_argument(
+ # "--device",
+ # type=str,
+ # default="all",
+ # choices=["cpu", "gpu", "all"],
+ # help="Filter by device (cpu or gpu)",
+ # )
+ parser.add_argument(
+ "--sf",
+ action="store_true",
+ default=None,
+ dest="sf_enabled",
+ help="Filter by Smart Filtering",
+ )
+
+ # DEBUGGING
+ parser.add_argument(
+ "--debug",
+ action="store_true",
+ help="Enable debug message and save intermediate images for Smart Filtering tests",
+ )
+ parser.add_argument(
+ "-n",
+ type=int,
+ default=100,
+ dest="debug_frame_limit",
+ help="Number of frames used for debugging [Default: 100]",
+ )
+
+ # PROFILING
+ parser.add_argument(
+ "--profile",
+ action="store_true",
+ help="Enable profiling",
+ )
+
+ args = parser.parse_args()
+ args.device = "gpu"
+
+ # UPDATE ENVIRONMENTAL VARIABLES
+ os.environ["VIDEO_FILENAME"] = args.source
+ os.environ["CUSTOM_MODEL_FLAG"] = "True" if args.custom_model_flag else "False"
+ os.environ["MODEL_NAME"] = args.model_name
+ os.environ["DEBUG"] = "1" if args.debug else "0"
+ os.environ["DEBUG_FRAME_LIMIT"] = str(args.debug_frame_limit)
+ os.environ["ENABLE_PROFILING"] = "True" if args.profile else "False"
+
+ # detection_type, device, sf_enabled
+ filter_expression = get_pytest_filter_expression(args)
+
+ # PYTEST COMMAND
+ pytest_args = [
+ "-k",
+ filter_expression,
+ "-s",
+ "-v",
+ # "--log-cli-level=DEBUG",
+ "-W",
+ "ignore:Exception ignored in.*SharedMemory.__del__:UserWarning",
+ # Target the exact module rewrite warning path inside the configuration framework
+ "-W",
+ "ignore:Module already imported so cannot be rewritten; anyio:_pytest.warning_types.PytestAssertRewriteWarning",
+ __file__,
+ ]
+
+ # main_app_logger.info(f"Launching tests for {args.source}")
+
+ sys.exit(pytest.main(pytest_args))
diff --git a/finetune/Dockerfile b/finetune/Dockerfile
index e093a7b..d676eeb 100644
--- a/finetune/Dockerfile
+++ b/finetune/Dockerfile
@@ -3,6 +3,8 @@ FROM openvisualcloud/xeon-ubuntu2204-media-nginx:23.1@sha256:d19eb597dc210134063
ARG DEBIAN_FRONTEND=noninteractive
ENV VIRTUAL_ENV=/opt/venv
+ENV PATH="$VIRTUAL_ENV/bin:/usr/local/cuda/bin:${PATH}"
+ENV LD_LIBRARY_PATH="$VIRTUAL_ENV/lib:/usr/local/cuda/lib64:${LD_LIBRARY_PATH}"
# Prevent Python from writing .pyc files and enable unbuffered logging
ENV PYTHONDONTWRITEBYTECODE=1 \
@@ -11,138 +13,182 @@ ENV PYTHONDONTWRITEBYTECODE=1 \
# hadolint ignore=DL3008
RUN apt-get update && \
apt-get install -y --only-upgrade --no-install-recommends libc-bin libc6 && \
- apt-get install -y -q --no-install-recommends python3-setuptools \
- python3-dev python3-pip python3-venv \
- curl libgl1-mesa-glx \
- # Update and install necessary build tools and dependencies for OpenCV
- build-essential \
- cmake \
- git \
- wget \
- unzip \
- yasm \
- pkg-config \
- libgl1 \
- libglib2.0-0 \
- libgtk2.0-dev \
- libtbb-dev \
- libjpeg-dev \
- libpng-dev \
- libtiff-dev \
- libavcodec-dev \
- libavformat-dev \
- libswscale-dev \
- libxvidcore-dev \
- libxine2-dev \
- libv4l-dev \
- libdc1394-dev \
- libatlas-base-dev \
- gfortran && \
- rm -rf /var/lib/apt/lists/* && \
- apt-get clean
-
- # hadolint ignore=DL3013
+ apt-get install -y -q --no-install-recommends python3-setuptools \
+ python3-dev python3-pip python3-venv \
+ curl libgl1-mesa-glx \
+ # Update and install necessary build tools and dependencies for OpenCV
+ build-essential \
+ cmake \
+ git \
+ wget \
+ unzip \
+ yasm \
+ pkg-config \
+ libgl1 \
+ libglib2.0-0 \
+ libgtk2.0-dev \
+ libtbb-dev \
+ libjpeg-dev \
+ libpng-dev \
+ libtiff-dev \
+ libavcodec-dev \
+ libavformat-dev \
+ libswscale-dev \
+ libxvidcore-dev \
+ libxine2-dev \
+ libv4l-dev \
+ libdc1394-dev \
+ libatlas-base-dev \
+ gfortran && \
+ # Install necessary CUDA packages
+ curl -fsSL https://developer.download.nvidia.com/compute/cuda/repos/ubuntu2204/x86_64/cuda-keyring_1.1-1_all.deb -o /tmp/cuda-keyring.deb && \
+ dpkg -i /tmp/cuda-keyring.deb && rm /tmp/cuda-keyring.deb
+
+ARG CUDA_MAJOR_VERSION=12
+ARG CUDA_MINOR_VERSION=8
+
+# hadolint ignore=DL3008
+RUN apt-get update && \
+ apt-get install -y -q --no-install-recommends \
+ cuda-toolkit-${CUDA_MAJOR_VERSION}-${CUDA_MINOR_VERSION} \
+ libcudnn9-dev-cuda-${CUDA_MAJOR_VERSION} && \
+ rm -rf /var/lib/apt/lists/* && apt-get clean
+
RUN python3 -m venv ${VIRTUAL_ENV} && \
${VIRTUAL_ENV}/bin/pip install --no-cache-dir \
- "pip==26.0.1" \
+ "pip==26.2.1" \
"torch==2.10.0" \
"torchvision==0.25.0" \
- "numpy==1.26.0"
-
-ENV DEVICE="GPU"
-
-# hadolint ignore=DL3008
-RUN if [ "${DEVICE}" = "GPU" ]; then \
- curl -fsSL https://developer.download.nvidia.com/compute/cuda/repos/ubuntu2204/x86_64/cuda-keyring_1.1-1_all.deb -o /tmp/cuda-keyring.deb; \
- dpkg -i /tmp/cuda-keyring.deb; \
- rm /tmp/cuda-keyring.deb; \
- apt-get update; \
- apt-get install -y -q --no-install-recommends cuda-toolkit-12-4 libcudnn9-cuda-12 libnvinfer10 libnvonnxparsers10 libnvinfer-plugin10; \
- rm -rf /var/lib/apt/lists/*; \
- fi;
+ "numpy==1.26.4"
+# Ensure paths capture the version dynamically
+ENV PATH="/usr/local/cuda/bin:${PATH}"
+ENV LD_LIBRARY_PATH="/usr/local/cuda/lib64:${LD_LIBRARY_PATH}"
-ENV PATH="$VIRTUAL_ENV/bin:/usr/local/cuda/bin:${PATH}"
-ENV LD_LIBRARY_PATH="$VIRTUAL_ENV/lib:/usr/local/cuda/lib64:${LD_LIBRARY_PATH}"
-
-
-# ARG PYTHON_VERSION=3.10
-ENV PYTHON_VERSION=3.10
# OPENCV W/ CUDA SUPPORT
+ENV PYTHON_VERSION=3.10
ENV OPENCV_VERSION="4.11.0"
-# ENV PYTHON_VERSION="3.$(${VIRTUAL_ENV}/bin/python -V | cut -f 1 | cut -d '.' -f 2)"
ENV DEPENDENCY_DIR=/tmp/build_opencv
-# ENV PYTHONPATH=${VIRTUAL_ENV}/lib/python${PYTHON_VERSION}/site-packages:${DEPENDENCY_DIR}/opencv/modules
ENV NUMPY_PATH="${VIRTUAL_ENV}/lib/python${PYTHON_VERSION}/site-packages/numpy/core/include"
ENV SYS_PATH="/usr/include/python${PYTHON_VERSION}"
-WORKDIR ${DEPENDENCY_DIR}
# Clone OpenCV and OpenCV Contrib repositories
-RUN git clone --branch ${OPENCV_VERSION} --depth 1 https://github.com/opencv/opencv.git ${DEPENDENCY_DIR}/opencv&& \
+WORKDIR ${DEPENDENCY_DIR}
+RUN git clone --branch ${OPENCV_VERSION} --depth 1 https://github.com/opencv/opencv.git ${DEPENDENCY_DIR}/opencv && \
git clone --branch ${OPENCV_VERSION} --depth 1 https://github.com/opencv/opencv_contrib.git ${DEPENDENCY_DIR}/opencv_contrib
# Create build directory and run CMake
WORKDIR ${DEPENDENCY_DIR}/opencv/build
+RUN ln -s /usr/include/x86_64-linux-gnu/cudnn*.h /usr/local/cuda/include/ && \
+ ln -s /usr/lib/x86_64-linux-gnu/libcudnn*.so* /usr/local/cuda/lib64/
+# hadolint ignore=SC2046
RUN cmake -D CMAKE_BUILD_TYPE=RELEASE \
- -D BUILD_EXAMPLES=OFF \
- -D BUILD_JAVA=OFF \
- -D BUILD_opencv_python2=OFF \
- -D BUILD_opencv_python3=ON \
- -D BUILD_PERF_TESTS=OFF \
- -D BUILD_TESTS=OFF \
- -D CMAKE_INSTALL_PREFIX="${VIRTUAL_ENV}" \
- -D CUDA_ARCH_BIN=70,75,80,86,89,90 \
- -D CUDA_FAST_MATH=ON \
- -D ENABLE_FAST_MATH=ON \
- -D INSTALL_PYTHON_EXAMPLES=OFF \
- -D INSTALL_C_EXAMPLES=OFF \
- -D OPENCV_EXTRA_MODULES_PATH="${DEPENDENCY_DIR}/opencv_contrib/modules" \
- -D PYTHON_DEFAULT_EXECUTABLE="${VIRTUAL_ENV}/bin/python" \
- -D PYTHON3_EXECUTABLE="${VIRTUAL_ENV}/bin/python" \
- -D PYTHON3_NUMPY_INCLUDE_DIRS="${NUMPY_PATH}" \
- -D PYTHON3_PACKAGES_PATH="${VIRTUAL_ENV}/lib/python${PYTHON_VERSION}/site-packages" \
- -D PYTHON3_LIBRARY="${VIRTUAL_ENV}/lib/libpython${PYTHON_VERSION}.so" \
- -D PYTHON3_INCLUDE_DIR="${SYS_PATH}" \
- -D WITH_CUBLAS=ON \
- -D WITH_CUDA=ON \
- -D WITH_CUDNN=ON \
- -D WITH_LIBV4L=ON \
- -D WITH_NVCUVENC=ON \
- -D WITH_NVCUVID=ON \
- .. && \
- make "-j$(nproc)" && \
- make install && \
- ldconfig
-
-# # Verify where it actually installed
-# RUN find ${VIRTUAL_ENV} -name "cv2*.so"
-# # Test the import immediately
-# RUN ${VIRTUAL_ENV}/bin/python -c "import cv2; print(cv2.__version__)"
+ -D CMAKE_INSTALL_PREFIX="${VIRTUAL_ENV}" \
+ -D OPENCV_EXTRA_MODULES_PATH="${DEPENDENCY_DIR}/opencv_contrib/modules" \
+ -D WITH_CUDA=ON \
+ -D WITH_CUDNN=ON \
+ -D WITH_CUBLAS=ON \
+ -D WITH_FFMPEG=ON \
+ -D WITH_TBB=ON \
+ -D WITH_V4L=ON \
+ -D OPENCV_DNN_CUDA=ON \
+ -D CUDA_ARCH_BIN=80,86,89,90 \
+ -D CUDA_FAST_MATH=ON \
+ -D CUDA_TOOLKIT_ROOT_DIR=/usr/local/cuda \
+ -D CUDNN_INCLUDE_DIR=/usr/include/x86_64-linux-gnu \
+ -D CUDNN_LIBRARY=/usr/lib/x86_64-linux-gnu/libcudnn.so \
+ -D BUILD_opencv_cudaimgproc=ON \
+ -D BUILD_opencv_cudaarithm=ON \
+ -D BUILD_opencv_cudafilters=ON \
+ -D BUILD_opencv_cudacodec=ON \
+ -D WITH_NVCUVID=OFF \
+ -D WITH_NVCUVENC=OFF \
+ -D WITH_VAAPI=ON \
+ -D WITH_FFMPEG=ON \
+ -D BUILD_opencv_python3=ON \
+ -D PYTHON3_EXECUTABLE="${VIRTUAL_ENV}/bin/python3" \
+ -D PYTHON3_NUMPY_INCLUDE_DIRS="${VIRTUAL_ENV}/lib/python3.10/site-packages/numpy/core/include" \
+ -D OPENCV_SKIP_PYTHON_LOADER=ON \
+ -D BUILD_EXAMPLES=OFF -D BUILD_TESTS=OFF -D BUILD_PERF_TESTS=OFF .. \
+ && make -j$(nproc) && make install && ldconfig
+
+SHELL ["/bin/bash", "-o", "pipefail", "-c"]
+RUN mkdir -p /tmp/cv2_deps && \
+ PY_CV2_SO=$(find "${VIRTUAL_ENV}" -name "cv2*.so" | head -n 1) && \
+ ldd "${PY_CV2_SO}" | grep "=> /" | awk '{print $3}' | xargs -I '{}' cp -v '{}' /tmp/cv2_deps/ && \
+ # FIX: Robust fallback copying to grab NPP/CUDA libs regardless of whether they land in /usr/local or system path channels
+ (cp -v /usr/local/cuda/lib64/libnpp*.so* /tmp/cv2_deps/ || cp -v /usr/lib/x86_64-linux-gnu/libnpp*.so* /tmp/cv2_deps/) && \
+ (cp -v /usr/local/cuda/lib64/libcudart.so* /tmp/cv2_deps/ || cp -v /usr/lib/x86_64-linux-gnu/libcudart.so* /tmp/cv2_deps/) && \
+ (cp -P /usr/local/cuda/lib64/libnvrtc.so* /tmp/cv2_deps/ || cp -P /usr/lib/x86_64-linux-gnu/libnvrtc.so* /tmp/cv2_deps/) && \
+ (cp -v /usr/local/cuda/lib64/libcublas*.so* /tmp/cv2_deps/ || cp -v /usr/lib/x86_64-linux-gnu/libcublas*.so* /tmp/cv2_deps/) && \
+ cp -v /usr/lib/x86_64-linux-gnu/libnvidia-encode.so* /tmp/cv2_deps/ || true
# Clean up
RUN rm -rf ${DEPENDENCY_DIR}
-# Set the working directory in the container
-WORKDIR /home
-COPY requirements.* /home/
-# RUN pip3 install --no-cache-dir -r requirements.in
-RUN pip3 install --no-cache-dir --require-hashes -r requirements.txt
+FROM openvisualcloud/xeon-ubuntu2204-media-nginx:23.1@sha256:d19eb597dc210134063803630ae2ea1ec84dfd4189138f59551e2f5ed047284a
+ARG DEBIAN_FRONTEND=noninteractive
+ARG CUDA_MAJOR_VERSION=12
+ARG CUDA_MINOR_VERSION=8
+ENV VIRTUAL_ENV=/opt/venv
+ENV PATH="$VIRTUAL_ENV/bin:/usr/local/cuda/bin:${PATH}"
-RUN \
- # Dynamically find the site-packages folder without hardcoding '3.10'
- PY_SITEPACKAGES=$(find "${VIRTUAL_ENV}/lib/" -maxdepth 2 -name "site-packages") && \
- echo "Found site-packages at: ${PY_SITEPACKAGES}" && \
- \
- # Link TensorRT libs
- ln -sf "${PY_SITEPACKAGES}/tensorrt_libs/lib*" /usr/lib/ && \
- \
- # Link NVIDIA libs
- find "${PY_SITEPACKAGES}/nvidia/" -name "lib*.so*" -exec ln -sf {} /usr/lib/ \; && \
- \
- ldconfig;
+# Install Runtime Libraries, FFmpeg/VA-API headers, and cuDNN
+# Includes libva and ffmpeg libraries required for HEVC decoding
+# hadolint ignore=DL3008
+RUN apt-get update && \
+ apt-get install -y --no-install-recommends \
+ ca-certificates \
+ curl \
+ python3 \
+ libgl1 \
+ libglib2.0-0 \
+ libtbb12 \
+ # Critical for Hardware Decoding (HEVC)
+ libva2 \
+ libva-drm2 \
+ libva-x11-2 \
+ libavcodec58 \
+ libavformat58 \
+ libswscale5 \
+ libv4l-0 && \
+ # Install necessary CUDA packages
+ curl -fsSL https://developer.download.nvidia.com/compute/cuda/repos/ubuntu2204/x86_64/cuda-keyring_1.1-1_all.deb -o /tmp/cuda-keyring.deb && \
+ dpkg -i /tmp/cuda-keyring.deb && rm /tmp/cuda-keyring.deb && \
+ apt-get update && \
+ apt-get install -y -q --no-install-recommends \
+ libcudnn9-cuda-${CUDA_MAJOR_VERSION} \
+ libavcodec-dev \
+ libavformat-dev \
+ libavutil-dev \
+ cuda-nvcc-${CUDA_MAJOR_VERSION}-${CUDA_MINOR_VERSION} && \
+ rm -rf /var/lib/apt/lists/*
+
+# Copy the entire pre-compiled virtual environment from the build stage
+COPY --from=build ${VIRTUAL_ENV} ${VIRTUAL_ENV}
+
+# Copy additional required CUDA math/parallel libs
+RUN mkdir -p /usr/local/cuda/lib64
+COPY --from=build /tmp/cv2_deps/* /usr/local/cuda/lib64/
+
+# Tells the host engine to expose all GPUs and utility binaries (like nvidia-smi)
+ENV NVIDIA_VISIBLE_DEVICES=all
+# FIX: Cleaned duplicate line later in file to ensure capabilities are consistently initialized
+ENV NVIDIA_DRIVER_CAPABILITIES=compute,utility,video
+ENV LD_LIBRARY_PATH="/usr/local/cuda/lib64:${VIRTUAL_ENV}/lib:${LD_LIBRARY_PATH}"
+RUN ldconfig
+
+ARG DEVICE="GPU"
+ENV DEVICE="${DEVICE}"
ARG DEBUG="0"
ENV DEBUG="${DEBUG}"
+
+# Set the working directory in the container
+WORKDIR /home
+COPY requirements.txt /home/
+RUN pip3 install --no-cache-dir --require-hashes -r /home/requirements.txt && \
+ pip3 uninstall -y opencv-python opencv-contrib-python opencv-python-headless
diff --git a/finetune/docker-compose.yml b/finetune/docker-compose.yml
index d4e4971..9043573 100644
--- a/finetune/docker-compose.yml
+++ b/finetune/docker-compose.yml
@@ -14,7 +14,7 @@ services:
network_mode: "host"
# shm_size: '2gb' # Give it plenty of space for video frames
ipc: "host"
- command: ["/bin/bash", "-c", "python /home/finetune.py --no-train"]
+ command: ["/bin/bash", "-c", "python /home/finetune.py"]
environment:
YOLO_CONFIG_DIR: "/tmp"
DBHOST: "vdms-service"
@@ -50,20 +50,3 @@ services:
capabilities: [gpu]
count: all
-
-# secrets:
-# self_key:
-# file: ../certificate/self.key
-# self_crt:
-# file: ../certificate/self.crt
-
-
-# networks:
-# appnet:
-# driver: overlay
-# attachable: true
-
-
-# volumes:
-# app-content:
-# vdms-content:
diff --git a/finetune/requirements.txt b/finetune/requirements.txt
index 03548a3..d93579e 100644
--- a/finetune/requirements.txt
+++ b/finetune/requirements.txt
@@ -1,139 +1,182 @@
-certifi==2026.2.25 \
- --hash=sha256:027692e4402ad994f1c42e52a4997a9763c646b73e4096e4d5d6db8af1d6f0fa \
- --hash=sha256:e887ab5cee78ea814d3472169153c2d12cd43b14bd03329a39a9c6e2e80bfba7
-charset-normalizer==3.4.6 \
- --hash=sha256:06a7e86163334edfc5d20fe104db92fcd666e5a5df0977cb5680a506fe26cc8e \
- --hash=sha256:0c173ce3a681f309f31b87125fecec7a5d1347261ea11ebbb856fa6006b23c8c \
- --hash=sha256:0e28d62a8fc7a1fa411c43bd65e346f3bce9716dc51b897fbe930c5987b402d5 \
- --hash=sha256:0e901eb1049fdb80f5bd11ed5ea1e498ec423102f7a9b9e4645d5b8204ff2815 \
- --hash=sha256:11afb56037cbc4b1555a34dd69151e8e069bee82e613a73bef6e714ce733585f \
- --hash=sha256:150b8ce8e830eb7ccb029ec9ca36022f756986aaaa7956aad6d9ec90089338c0 \
- --hash=sha256:172985e4ff804a7ad08eebec0a1640ece87ba5041d565fff23c8f99c1f389484 \
- --hash=sha256:197c1a244a274bb016dd8b79204850144ef77fe81c5b797dc389327adb552407 \
- --hash=sha256:1ae6b62897110aa7c79ea2f5dd38d1abca6db663687c0b1ad9aed6f6bae3d9d6 \
- --hash=sha256:1cf0a70018692f85172348fe06d3a4b63f94ecb055e13a00c644d368eb82e5b8 \
- --hash=sha256:1ed80ff870ca6de33f4d953fda4d55654b9a2b340ff39ab32fa3adbcd718f264 \
- --hash=sha256:22c6f0c2fbc31e76c3b8a86fba1a56eda6166e238c29cdd3d14befdb4a4e4815 \
- --hash=sha256:231d4da14bcd9301310faf492051bee27df11f2bc7549bc0bb41fef11b82daa2 \
- --hash=sha256:259695e2ccc253feb2a016303543d691825e920917e31f894ca1a687982b1de4 \
- --hash=sha256:2a24157fa36980478dd1770b585c0f30d19e18f4fb0c47c13aa568f871718579 \
- --hash=sha256:2b1a63e8224e401cafe7739f77efd3f9e7f5f2026bda4aead8e59afab537784f \
- --hash=sha256:2bd9d128ef93637a5d7a6af25363cf5dec3fa21cf80e68055aad627f280e8afa \
- --hash=sha256:2e1d8ca8611099001949d1cdfaefc510cf0f212484fe7c565f735b68c78c3c95 \
- --hash=sha256:2ef7fedc7a6ecbe99969cd09632516738a97eeb8bd7258bf8a0f23114c057dab \
- --hash=sha256:2f7fdd9b6e6c529d6a2501a2d36b240109e78a8ceaef5687cfcfa2bbe671d297 \
- --hash=sha256:30f445ae60aad5e1f8bdbb3108e39f6fbc09f4ea16c815c66578878325f8f15a \
- --hash=sha256:31215157227939b4fb3d740cd23fe27be0439afef67b785a1eb78a3ae69cba9e \
- --hash=sha256:34315ff4fc374b285ad7f4a0bf7dcbfe769e1b104230d40f49f700d4ab6bbd84 \
- --hash=sha256:3516bbb8d42169de9e61b8520cbeeeb716f12f4ecfe3fd30a9919aa16c806ca8 \
- --hash=sha256:3778fd7d7cd04ae8f54651f4a7a0bd6e39a0cf20f801720a4c21d80e9b7ad6b0 \
- --hash=sha256:39f5068d35621da2881271e5c3205125cc456f54e9030d3f723288c873a71bf9 \
- --hash=sha256:404a1e552cf5b675a87f0651f8b79f5f1e6fd100ee88dc612f89aa16abd4486f \
- --hash=sha256:419a9d91bd238052642a51938af8ac05da5b3343becde08d5cdeab9046df9ee1 \
- --hash=sha256:423fb7e748a08f854a08a222b983f4df1912b1daedce51a72bd24fe8f26a1843 \
- --hash=sha256:4482481cb0572180b6fd976a4d5c72a30263e98564da68b86ec91f0fe35e8565 \
- --hash=sha256:461598cd852bfa5a61b09cae2b1c02e2efcd166ee5516e243d540ac24bfa68a7 \
- --hash=sha256:47955475ac79cc504ef2704b192364e51d0d473ad452caedd0002605f780101c \
- --hash=sha256:48696db7f18afb80a068821504296eb0787d9ce239b91ca15059d1d3eaacf13b \
- --hash=sha256:4be9f4830ba8741527693848403e2c457c16e499100963ec711b1c6f2049b7c7 \
- --hash=sha256:4d1d02209e06550bdaef34af58e041ad71b88e624f5d825519da3a3308e22687 \
- --hash=sha256:4f41da960b196ea355357285ad1316a00099f22d0929fe168343b99b254729c9 \
- --hash=sha256:517ad0e93394ac532745129ceabdf2696b609ec9f87863d337140317ebce1c14 \
- --hash=sha256:51fb3c322c81d20567019778cb5a4a6f2dc1c200b886bc0d636238e364848c89 \
- --hash=sha256:5273b9f0b5835ff0350c0828faea623c68bfa65b792720c453e22b25cc72930f \
- --hash=sha256:530d548084c4a9f7a16ed4a294d459b4f229db50df689bfe92027452452943a0 \
- --hash=sha256:530e8cebeea0d76bdcf93357aa5e41336f48c3dc709ac52da2bb167c5b8271d9 \
- --hash=sha256:54fae94be3d75f3e573c9a1b5402dc593de19377013c9a0e4285e3d402dd3a2a \
- --hash=sha256:572d7c822caf521f0525ba1bce1a622a0b85cf47ffbdae6c9c19e3b5ac3c4389 \
- --hash=sha256:58c948d0d086229efc484fe2f30c2d382c86720f55cd9bc33591774348ad44e0 \
- --hash=sha256:5d11595abf8dd942a77883a39d81433739b287b6aa71620f15164f8096221b30 \
- --hash=sha256:5f8ddd609f9e1af8c7bd6e2aca279c931aefecd148a14402d4e368f3171769fd \
- --hash=sha256:5feb91325bbceade6afab43eb3b508c63ee53579fe896c77137ded51c6b6958e \
- --hash=sha256:60c74963d8350241a79cb8feea80e54d518f72c26db618862a8f53e5023deaf9 \
- --hash=sha256:613f19aa6e082cf96e17e3ffd89383343d0d589abda756b7764cf78361fd41dc \
- --hash=sha256:659a1e1b500fac8f2779dd9e1570464e012f43e580371470b45277a27baa7532 \
- --hash=sha256:695f5c2823691a25f17bc5d5ffe79fa90972cc34b002ac6c843bb8a1720e950d \
- --hash=sha256:69dd852c2f0ad631b8b60cfbe25a28c0058a894de5abb566619c205ce0550eae \
- --hash=sha256:6cceb5473417d28edd20c6c984ab6fee6c6267d38d906823ebfe20b03d607dc2 \
- --hash=sha256:71be7e0e01753a89cf024abf7ecb6bca2c81738ead80d43004d9b5e3f1244e64 \
- --hash=sha256:74119174722c4349af9708993118581686f343adc1c8c9c007d59be90d077f3f \
- --hash=sha256:74a2e659c7ecbc73562e2a15e05039f1e22c75b7c7618b4b574a3ea9118d1557 \
- --hash=sha256:7504e9b7dc05f99a9bbb4525c67a2c155073b44d720470a148b34166a69c054e \
- --hash=sha256:79090741d842f564b1b2827c0b82d846405b744d31e84f18d7a7b41c20e473ff \
- --hash=sha256:7a6967aaf043bceabab5412ed6bd6bd26603dae84d5cb75bf8d9a74a4959d398 \
- --hash=sha256:7bda6eebafd42133efdca535b04ccb338ab29467b3f7bf79569883676fc628db \
- --hash=sha256:7edbed096e4a4798710ed6bc75dcaa2a21b68b6c356553ac4823c3658d53743a \
- --hash=sha256:7f9019c9cb613f084481bd6a100b12e1547cf2efe362d873c2e31e4035a6fa43 \
- --hash=sha256:802168e03fba8bbc5ce0d866d589e4b1ca751d06edee69f7f3a19c5a9fe6b597 \
- --hash=sha256:80d0a5615143c0b3225e5e3ef22c8d5d51f3f72ce0ea6fb84c943546c7b25b6c \
- --hash=sha256:82060f995ab5003a2d6e0f4ad29065b7672b6593c8c63559beefe5b443242c3e \
- --hash=sha256:836ab36280f21fc1a03c99cd05c6b7af70d2697e374c7af0b61ed271401a72a2 \
- --hash=sha256:8761ac29b6c81574724322a554605608a9960769ea83d2c73e396f3df896ad54 \
- --hash=sha256:87725cfb1a4f1f8c2fc9890ae2f42094120f4b44db9360be5d99a4c6b0e03a9e \
- --hash=sha256:899d28f422116b08be5118ef350c292b36fc15ec2daeb9ea987c89281c7bb5c4 \
- --hash=sha256:8bc5f0687d796c05b1e28ab0d38a50e6309906ee09375dd3aff6a9c09dd6e8f4 \
- --hash=sha256:8bea55c4eef25b0b19a0337dc4e3f9a15b00d569c77211fa8cde38684f234fb7 \
- --hash=sha256:8e5a94886bedca0f9b78fecd6afb6629142fd2605aa70a125d49f4edc6037ee6 \
- --hash=sha256:90ca27cd8da8118b18a52d5f547859cc1f8354a00cd1e8e5120df3e30d6279e5 \
- --hash=sha256:92734d4d8d187a354a556626c221cd1a892a4e0802ccb2af432a1d85ec012194 \
- --hash=sha256:947cf925bc916d90adba35a64c82aace04fa39b46b52d4630ece166655905a69 \
- --hash=sha256:95b52c68d64c1878818687a473a10547b3292e82b6f6fe483808fb1468e2f52f \
- --hash=sha256:97d0235baafca5f2b09cf332cc275f021e694e8362c6bb9c96fc9a0eb74fc316 \
- --hash=sha256:9ca4c0b502ab399ef89248a2c84c54954f77a070f28e546a85e91da627d1301e \
- --hash=sha256:9cc4fc6c196d6a8b76629a70ddfcd4635a6898756e2d9cac5565cf0654605d73 \
- --hash=sha256:9cc6e6d9e571d2f863fa77700701dae73ed5f78881efc8b3f9a4398772ff53e8 \
- --hash=sha256:a056d1ad2633548ca18ffa2f85c202cfb48b68615129143915b8dc72a806a923 \
- --hash=sha256:a26611d9987b230566f24a0a125f17fe0de6a6aff9f25c9f564aaa2721a5fb88 \
- --hash=sha256:a4474d924a47185a06411e0064b803c68be044be2d60e50e8bddcc2649957c1f \
- --hash=sha256:a4ea868bc28109052790eb2b52a9ab33f3aa7adc02f96673526ff47419490e21 \
- --hash=sha256:a9e68c9d88823b274cf1e72f28cb5dc89c990edf430b0bfd3e2fb0785bfeabf4 \
- --hash=sha256:aa9cccf4a44b9b62d8ba8b4dd06c649ba683e4bf04eea606d2e94cfc2d6ff4d6 \
- --hash=sha256:ab30e5e3e706e3063bc6de96b118688cb10396b70bb9864a430f67df98c61ecc \
- --hash=sha256:ac2393c73378fea4e52aa56285a3d64be50f1a12395afef9cce47772f60334c2 \
- --hash=sha256:ad8faf8df23f0378c6d527d8b0b15ea4a2e23c89376877c598c4870d1b2c7866 \
- --hash=sha256:b35b200d6a71b9839a46b9b7fff66b6638bb52fc9658aa58796b0326595d3021 \
- --hash=sha256:b3694e3f87f8ac7ce279d4355645b3c878d24d1424581b46282f24b92f5a4ae2 \
- --hash=sha256:b4ff1d35e8c5bd078be89349b6f3a845128e685e751b6ea1169cf2160b344c4d \
- --hash=sha256:bbc8c8650c6e51041ad1be191742b8b421d05bbd3410f43fa2a00c8db87678e8 \
- --hash=sha256:bc72863f4d9aba2e8fd9085e63548a324ba706d2ea2c83b260da08a59b9482de \
- --hash=sha256:bf625105bb9eef28a56a943fec8c8a98aeb80e7d7db99bd3c388137e6eb2d237 \
- --hash=sha256:c2274ca724536f173122f36c98ce188fd24ce3dad886ec2b7af859518ce008a4 \
- --hash=sha256:c45a03a4c69820a399f1dda9e1d8fbf3562eda46e7720458180302021b08f778 \
- --hash=sha256:c8ae56368f8cc97c7e40a7ee18e1cedaf8e780cd8bc5ed5ac8b81f238614facb \
- --hash=sha256:c907cdc8109f6c619e6254212e794d6548373cc40e1ec75e6e3823d9135d29cc \
- --hash=sha256:ca0276464d148c72defa8bb4390cce01b4a0e425f3b50d1435aa6d7a18107602 \
- --hash=sha256:cd5e2801c89992ed8c0a3f0293ae83c159a60d9a5d685005383ef4caca77f2c4 \
- --hash=sha256:d08ec48f0a1c48d75d0356cea971921848fb620fdeba805b28f937e90691209f \
- --hash=sha256:d1a2ee9c1499fc8f86f4521f27a973c914b211ffa87322f4ee33bb35392da2c5 \
- --hash=sha256:d5f5d1e9def3405f60e3ca8232d56f35c98fb7bf581efcc60051ebf53cb8b611 \
- --hash=sha256:d60377dce4511655582e300dc1e5a5f24ba0cb229005a1d5c8d0cb72bb758ab8 \
- --hash=sha256:d73beaac5e90173ac3deb9928a74763a6d230f494e4bfb422c217a0ad8e629bf \
- --hash=sha256:d7de2637729c67d67cf87614b566626057e95c303bc0a55ffe391f5205e7003d \
- --hash=sha256:dad6e0f2e481fffdcf776d10ebee25e0ef89f16d691f1e5dee4b586375fdc64b \
- --hash=sha256:dda86aba335c902b6149a02a55b38e96287157e609200811837678214ba2b1db \
- --hash=sha256:df01808ee470038c3f8dc4f48620df7225c49c2d6639e38f96e6d6ac6e6f7b0e \
- --hash=sha256:e1f6e2f00a6b8edb562826e4632e26d063ac10307e80f7461f7de3ad8ef3f077 \
- --hash=sha256:e25369dc110d58ddf29b949377a93e0716d72a24f62bad72b2b39f155949c1fd \
- --hash=sha256:e3c701e954abf6fc03a49f7c579cc80c2c6cc52525340ca3186c41d3f33482ef \
- --hash=sha256:e5bcc1a1ae744e0bb59641171ae53743760130600da8db48cbb6e4918e186e4e \
- --hash=sha256:e68c14b04827dd76dcbd1aeea9e604e3e4b78322d8faf2f8132c7138efa340a8 \
- --hash=sha256:e8aeb10fcbe92767f0fa69ad5a72deca50d0dca07fbde97848997d778a50c9fe \
- --hash=sha256:e985a16ff513596f217cee86c21371b8cd011c0f6f056d0920aa2d926c544058 \
- --hash=sha256:ecbbd45615a6885fe3240eb9db73b9e62518b611850fdf8ab08bd56de7ad2b17 \
- --hash=sha256:ee4ec14bc1680d6b0afab9aea2ef27e26d2024f18b24a2d7155a52b60da7e833 \
- --hash=sha256:ef5960d965e67165d75b7c7ffc60a83ec5abfc5c11b764ec13ea54fbef8b4421 \
- --hash=sha256:f0cdaecd4c953bfae0b6bb64910aaaca5a424ad9c72d85cb88417bb9814f7550 \
- --hash=sha256:f1ce721c8a7dfec21fcbdfe04e8f68174183cf4e8188e0645e92aa23985c57ff \
- --hash=sha256:f50498891691e0864dc3da965f340fada0771f6142a378083dc4608f4ea513e2 \
- --hash=sha256:f5ea69428fa1b49573eef0cc44a1d43bebd45ad0c611eb7d7eac760c7ae771bc \
- --hash=sha256:f61aa92e4aad0be58eb6eb4e0c21acf32cf8065f4b2cae5665da756c4ceef982 \
- --hash=sha256:f6e4333fb15c83f7d1482a76d45a0818897b3d33f00efd215528ff7c51b8e35d \
- --hash=sha256:f820f24b09e3e779fe84c3c456cb4108a7aa639b0d1f02c28046e11bfcd088ed \
- --hash=sha256:f98059e4fcd3e3e4e2d632b7cf81c2faae96c43c60b569e9c621468082f1d104 \
- --hash=sha256:fcce033e4021347d80ed9c66dcf1e7b1546319834b74445f561d2e2221de5659
-click==8.3.1 \
- --hash=sha256:12ff4785d337a1bb490bb7e9c2b1ee5da3112e94a8622f26a6c77f5d2fc6842a \
- --hash=sha256:981153a64e25f12d547d3426c367a4857371575ee7ad18df2a6183ab0545b2a6
+certifi==2026.7.22 \
+ --hash=sha256:62f22742b58a1a33014a2b6b706588a8d7e2a88ae7bd1a6ebe8c992928483775 \
+ --hash=sha256:741e2c3b351ddf169a738da9f2c048608ff7f2c5cc02f1ebc6b118bb090d5d55
+charset-normalizer==3.5.1 \
+ --hash=sha256:00668ebb0609751758682eb0b5857e7c35b9f00e84dfdef062e103244ec94d45 \
+ --hash=sha256:012a22b88a77ca2e59b98ac5889b0deb604147666032f45e6d6e217634d2550d \
+ --hash=sha256:01e93745f7f219b703b60ba7afead36cfc4242782be5af484673fc500df12da5 \
+ --hash=sha256:04368edf83514385ffc3e1cfd4546e595f4f1272dd23ba437a93a9cc3741d47b \
+ --hash=sha256:0722590aabf9dc6a6c0343d523c05458fa2b5047dbe6302fd526bb570600753f \
+ --hash=sha256:07ffd07412fc5d5e84cd8952acf9ff7e4ed7a708e69d1bada19d8ba91711353f \
+ --hash=sha256:09a7bba9f739468c8e78c36a75c33768e53cb1959fc638f510454c14683f00d5 \
+ --hash=sha256:0b2b1b3fa5670c127b246df1d0c059defd41f689a868a3b9d79df9b1cac42d22 \
+ --hash=sha256:0c6dfb5ca6723eeed15aa8e564a014d69fcb8812f94eef11fe3631e0508199f5 \
+ --hash=sha256:0d929fc574b4d6fd9e7c0f5c2ede8716a41911923aa7fa5fce38e0818aa4a1ac \
+ --hash=sha256:13e3afe97712e8887cd516e960c63f0b93122971e5b5e4b2622fe7701771e838 \
+ --hash=sha256:15f024313246a4ed976c60f440bb8d257815513a681d212ff74fd46f7d715a90 \
+ --hash=sha256:195ce897c6153c0700078142cf8efe3e6454ca4cf4357499e4078dfd83396626 \
+ --hash=sha256:19a3dd5aa73cef1c99687c4fc57db016a9c17104ae1185da88ba566a5d3bebe4 \
+ --hash=sha256:1d1c7a53a6c2103925cdd6d7229f8c567379f211c869793df679f2e9f738c369 \
+ --hash=sha256:1f5883d77fd409a261abb5dc8ccbe335720d798b1de4abb3b1d47ccbbc76b53b \
+ --hash=sha256:21b82d8082f6f5e7f456ef0bd16323d08de1266efbfeb476e64b2a91d1471a4e \
+ --hash=sha256:252d099029bcbea642f2a06c4ed5046bdf8b5a8150b64afa5e027e88b106e5ee \
+ --hash=sha256:256dd4d85d9e4dc595e2bc983c980e73f62ddeb3165c58b4c3dfe78c5c8548c1 \
+ --hash=sha256:26422d45fd13551cf564c58932f7d72b4f58b93b0fcf18c35ba6be12b46bb102 \
+ --hash=sha256:2679de311c7946dde5d3b6f44941844133ff5c7cb86099c0061ab1e8901c20a8 \
+ --hash=sha256:29880d17a8eb0b5cfdfd8944b468322928059aa35f1f5fa8ff22b149ec0b42f8 \
+ --hash=sha256:2bced4061f000f7187254a02ad3433ae17eaf991747ceea2f478422590a5bba9 \
+ --hash=sha256:2e9cf9253119d8e5d111f05d71626786fd3d6193817316eab1ca088cdb8593cf \
+ --hash=sha256:2f06b7eae9dbe77fe1d644ca244dad508de8d302870a43f3c559b521270938a0 \
+ --hash=sha256:2f293479cce755c75f1697e87c409b7ae4c555c7dfecb6e988ad13abba943031 \
+ --hash=sha256:329fc3ccb63ad22d867d84c2adea759a64079a37ba4a343433b02c7a2816871e \
+ --hash=sha256:343fb4f2821043bd87095f7b08a1a181febc8e36ac64212143bbfd0a0e1bc235 \
+ --hash=sha256:3588e376b3ea2eea84976f67273d679f229e24c66dce7b82ae45aef04ff6e072 \
+ --hash=sha256:35aea775dc2bd5f54cd84a1cd2696cc3207c479cb9cf0bd346f0d343e4300ddb \
+ --hash=sha256:35fe081843b35aad20ffeccec3eeffbe637b15d14f3fb22cc1b59cd8ec17e93c \
+ --hash=sha256:36047af20e17097c3bb9476c2b7655f2f7aa51322c0ba58c07695bedf755a950 \
+ --hash=sha256:3617ac3cfd8b9888f145ad89dd6e692285834b0201c6074a5eeaad3fd4d668c2 \
+ --hash=sha256:366ec70f5547c640d3ce1985722490f23faf4eb5216a7eeba78277490e78dacb \
+ --hash=sha256:394fea06235c8543390050ed5f529187074b029fb027213f6c46ac11ab5d950e \
+ --hash=sha256:3d27167433c0d5f18dc850f07d0b3816221984fecdc405d6c157a6f0b8f8e9e6 \
+ --hash=sha256:3e5e1224c0a6a90e05843e07adfec669edebec17801c67072f51e59561d63c0b \
+ --hash=sha256:41876ee62a3dddf48ff1121ad8f0798032aa03f2fd35f21f34a4cab14f18d8d2 \
+ --hash=sha256:433c5a81eade63b47e522303bad236f59dba55ea6951746f5558355eeed8c75d \
+ --hash=sha256:4582c27e8c889d64811987b5967fbd3ae0c823fe1fd933b543d55ac20bb475fa \
+ --hash=sha256:485a0d363cafefcd2538a73c7c838daa2035f09b2c9f9b5e3133f80c6aeb84c2 \
+ --hash=sha256:494b70049a4d69aec6e8137c13af4cf8db8c9f9820a1392ac293b0dd2987a818 \
+ --hash=sha256:496846868fea80e479324862fa877f02411f2fd0f83b79ccee2607aa68b2a032 \
+ --hash=sha256:4abdc5f9ad448c1ecbfae2974b820535d6bc6e7eef63babbab3d81cf46968c71 \
+ --hash=sha256:4b599739b93b2cbeded49645ae3c8d1405c29ddfbceac1545c87a3f9580a9e96 \
+ --hash=sha256:4bea7f8ebe90bbd7f0e4a2de42ca6924ba23e3e76418c408ff82f1d46fabd687 \
+ --hash=sha256:4c4fb141a727957c93edfe5c32a26ceb6b5f6461d67146e2d39f51e16170bea8 \
+ --hash=sha256:4c9548dc78002099910abaebc0a72ac58b7d30931869e0351c09b507dff4ece3 \
+ --hash=sha256:4d26f14f041e83dd8edfd61f4cd4fa7285d31798b5bf1f28e70c367ba6c41d61 \
+ --hash=sha256:4f298bdadb8f0b9e5672877f647d1be9373ef5320c9e2f049795e26cad28b6a9 \
+ --hash=sha256:52ec005752a56ae79547a05c0139ca2501a0c866390b6115008456b9f0e7cde1 \
+ --hash=sha256:55261ac0d2941c42f196dd576f543d87a8ee03cd6f5e30dfb4d807b2e3b9121a \
+ --hash=sha256:56490c595a28b1bb27dfc583e816152a9767721ef58b2c03b13f954d2f707420 \
+ --hash=sha256:58d3e12c88e0950bca850ae1f7c256055c097639c2edb9eb123af9807d8b15e4 \
+ --hash=sha256:58d4aa13a59c969dbfdf9e6a9560e242cbfd9e8a8f50c2747714df1a423adf65 \
+ --hash=sha256:59171c6e45bf07d0d5cab3b0bf81d945035530f6873398b3b531c31184d46663 \
+ --hash=sha256:5b6d1386bf0096d26d3a863dc0a487a5b4eb9aa93cf5ba69683d29dde6b9d60f \
+ --hash=sha256:5c0ea61a470e070686aa30892fed79e297d2c8d0ab46b8bcdf027d38c51da591 \
+ --hash=sha256:5c84bec0ab5ae0c64bfe73a7d2adcb5ce73b467523fc27fd6a28ab2aa6cbe35a \
+ --hash=sha256:5ca0555312ae2fe82715cada7fac375530c2f3349e1eaa1bcb33d0283ac79a18 \
+ --hash=sha256:5d8531a6569d025f68e2321e7638fb7978f23db58e5f69f56913837aae03816e \
+ --hash=sha256:5e2d0e146dcb57034f8b97dc58d2d512cb90aba253960ce449f695fec6a82c6f \
+ --hash=sha256:5fc45d653ea8c9a20479167e11d4a0f8cb2fa3470737ab6f9c827532313187b7 \
+ --hash=sha256:6117b84ea48435e5356dc737f5121485c30920ba43375fa7b434fd753df0eac3 \
+ --hash=sha256:6199d5606e2bbf2b096cf64d03f8b6790c91081d5ac866b8e7bb6422738cc60c \
+ --hash=sha256:62b55f6722735a6c472f88361cde6640608773d9443cebdbb51abf436a1fcdd3 \
+ --hash=sha256:687c9ca3035544b113bea2055e180af96fb63c0c476e22a9180f51925186e7b7 \
+ --hash=sha256:6b7430cf5728e68f6c462254009a6ef4086e1bea43cf2f57aa9c55fb4f50ff96 \
+ --hash=sha256:6ba32c4d2abf1d2fe7cf27d280f4cca5664233b0f885549c7761719eb977f486 \
+ --hash=sha256:6c9cdde8becb25a7fde49924511aa2644d6f8081cc8df8e9452724303348d8e3 \
+ --hash=sha256:6df0ec430f9a831772c23ca5a224cba36517a58a84bb32c32bb59a9fa67c47f6 \
+ --hash=sha256:6e2912d4babbc65196ac13c2f53468dc57fb8b9c25ef913e8c59ddf7c6dc0e1b \
+ --hash=sha256:6e5e4d73d588ca5ed09df1b7dcd1b203d1df3c542e3f50d126c947d432b10731 \
+ --hash=sha256:70055ff39b97c99e7ae40ea3e393fb62aa2e44dbd9b29f8d14f42fb0025c3959 \
+ --hash=sha256:706bfd38730a5ac7a365793269a00f4e988178cec121391f4248d84ad8c972e9 \
+ --hash=sha256:7235dc28fc6dd9d832ac7c7bce95367dedb85929f17368a0c2bee1e080b9acbf \
+ --hash=sha256:774d157f112367ff4abd29019f38f023c24e00e56edc7829c20e358a5a913ad8 \
+ --hash=sha256:77efcff2b23071c349402ac1066667a3d011f62398d81408c9b88ad991747c9e \
+ --hash=sha256:789b8982559ae28dad2356519f841655756cdcd96616410590ae0b17454ee64f \
+ --hash=sha256:7ac76cf9afd34929d76eb7fcb63be476a4853d8a96f0dcf2d0db68a0cbdf9885 \
+ --hash=sha256:7c0c10730342b0c9b35dd1d619beb8214e520bd96a1f870f452680b238aab3e0 \
+ --hash=sha256:823f82903d189af463d7df250ef1f7f696f3cee08cc8d91deb565e8d425f6506 \
+ --hash=sha256:838648accb3a7fd9803fd45c87bce8509648eb0c11bc34e216141300977244f2 \
+ --hash=sha256:854066be00447fa8de2ccbbe893e2ffc4b123ef16d897af794c1e18bd4a714b0 \
+ --hash=sha256:85d5855daafc240cc045c026d7a15fd198a09b0fc8ff6f5ecbb5297b509cb11e \
+ --hash=sha256:85de3134b5379856e323ba37c19c9256d39425f7b76a63af52b09fb4664c2e8f \
+ --hash=sha256:87e4f41d375c0b9be2fb5251aee4b8a689169e134535aed81bf085c3b647451e \
+ --hash=sha256:88ca277405c2d3b71c4e1c2ee0e7966e807bcba86a69d11e19ba199d18ae4491 \
+ --hash=sha256:88e85ab89cb822c1e635f51d6d32e488f94e002e70e2f492bdb8b945543f345a \
+ --hash=sha256:8ac8c94b6539074e0f40899301273ac8402b9b3e01c7b7ba269ff30340aaaf20 \
+ --hash=sha256:8fe532b3c966d1fb794e0698e4589d0444017ae77fc0b31edea13c0e35bcc449 \
+ --hash=sha256:9085f87b0e38a2b92b8923059b4e8789fe40d9279712d15dcc670048d77079af \
+ --hash=sha256:90b7481fb62fbe172c558bc6fd1c4c98d82004a54a7551f20e11ac9bf0b8708c \
+ --hash=sha256:92caef967d287a407085d61176fce4012b1dd62daed4eb6d5ceb26d3d2538712 \
+ --hash=sha256:9362dd90aa7dab48c0054a21187791ccf05473f7dba5d92b8033ae62164675e7 \
+ --hash=sha256:94d78ecec2605a8d0398b0f365d5f12a63248438516f5dac536a5eff7337df4a \
+ --hash=sha256:94fbf1c0c6cc0d3d5e50f9a9313a8cdca90dd696d34b381cd1704f8c9e939f20 \
+ --hash=sha256:950f23cb393f85543777b0433f082cddd25b51ab398eac7971146495679efe5f \
+ --hash=sha256:96eefc178f8636b9c760c5829345307fd81cfae9ab1e80997dbddeb0f54ee9a3 \
+ --hash=sha256:96fef3e886d6a9874b14f27fc193fbdc69d5d8035783d86aa4e1cea594e695f9 \
+ --hash=sha256:977cdbd483a9cff38179bea4fd754289a6f2195c7abd414aba85410b3e66cc5e \
+ --hash=sha256:978eab16f55b4ab2c2a745be9a0a840bf8f09a7f227d9c76eb30214d078865a5 \
+ --hash=sha256:994e883d17c559cdfd38c84003c8b27d25424a1077272a17e7cd27bfe0bf57b2 \
+ --hash=sha256:9ac4444d8d4fd4c4bd08bf451ed3167aa9e7ec6cdb41b648794f1d1103652e36 \
+ --hash=sha256:9b5db6052055d34d41230fb78d7c439c23dc536a9896f6cb039e8dd92cfc1263 \
+ --hash=sha256:9d9a0dc7cbe9bec24c3f767c9122c41fe5a1bc43f47cd099d00d393e09769de4 \
+ --hash=sha256:9dbdd9205662134957cf0c324f639bdc5031c0ca056e2369e238db75187c0f11 \
+ --hash=sha256:9eea3ab2597a5e65fe65296e2d6a84570845a6b55532d90333d740d48bbc850a \
+ --hash=sha256:a2028475ba855475b8b4d3cfeb4994269c967aea8b9892dfba907f4263a863a3 \
+ --hash=sha256:a3a370082ce34d0612f421e15fe011c53bb1feff21a26d06ad4fb244dab5a375 \
+ --hash=sha256:a545775cfe815855ea32d7c27731d79da358ef2055b4a25830231b1622dd18aa \
+ --hash=sha256:a5cbd90ecf0fc62e64726917ad083b73001f0563657a87ec3c0b504e277dc90d \
+ --hash=sha256:a6d095662e73e74f0a49988e0593373e243e3a52e27bfeea0a859e88acf4a0f5 \
+ --hash=sha256:a6dac12ff6b846103483683f60c5f8fee205121adc58ffd87e90a90a3af69e99 \
+ --hash=sha256:a951ad59cad9145664a730d3036b40b844e74d2d3683da40111463cd3a83845d \
+ --hash=sha256:aa1099b956fb795e686d073568f6dc002a0bb89765ea6d5b055dd7d9bf1b116c \
+ --hash=sha256:aa2bb0b37202dca27175591f761108b5d34096ade1191ffe4808bdf6b1571488 \
+ --hash=sha256:aae2ee51122d3ae968a3837d97dc24a0aeebb0dea23694422cd172bd30017cd6 \
+ --hash=sha256:ab743e9bc90c1f73552ec33e10e3331315acd2c397b36065b591b0181de533cc \
+ --hash=sha256:ac00177c4831ffa650f8609e4bdddd5fe09c03b1c0c47acece7e6ea20421598b \
+ --hash=sha256:ac13b004224fb341e1e25a1ed5e19d32f57cdb2a403e01f003b46f051a550f6f \
+ --hash=sha256:acaf604462bf330b0d07e7a07c1d6e4adac79e5fb13e9c5140590542cafacc00 \
+ --hash=sha256:ae31a1a1db2ee6cc2942fccaf695c934bc7f3db9f2133a3fef1f367cf1a4ab10 \
+ --hash=sha256:ae4a097991662cd4fff0ddc74e0fe7874f82e00042fa0ea00855645ed0c79598 \
+ --hash=sha256:aea996a6aba25260827c9ea511d1addfde2da9eb686ac961838509086188b7e6 \
+ --hash=sha256:b39b69b347e5e47a3b5b8cfc005c68c1ba347474e3960236c4944a8ecd174962 \
+ --hash=sha256:b54e7e13267d49ffbfe68e25b3cbd774dab38fa37238f71265e91b36146eb21c \
+ --hash=sha256:b9af956078716df40d985fb0dfeb2c2120c5ca92ba4ff4b388acfd01cdc14d08 \
+ --hash=sha256:ba2f37ee79e6338845261a3c5b1784e5d1acdff2c0785b284f1b633033d136ab \
+ --hash=sha256:ba501e667c17d8411f98e67a022d9604ef179aff0e459b7e292c796837c13573 \
+ --hash=sha256:baf3775a2635e5a11fbd5e4e64ee69c7e86875d224a5c72aca4c141064589a90 \
+ --hash=sha256:bb57753e36e4855b8ca375069482250a6246372331a3e4f3407eaebb007443f5 \
+ --hash=sha256:bd6c173f04743d483881bffa1478d5a4624475b8cd1d2194956a75548e191c18 \
+ --hash=sha256:be47f99644b208bff7766314013f9acf57b056b04191d570d68ad14022cf5b1d \
+ --hash=sha256:c010f5581d9c612804cc59fcf7b524b707fbcb72828551237ab545bb5c7034af \
+ --hash=sha256:c1dcc36dcb96abc02236e182d17e0f71430152a6c2c7447421da2d2dc144edea \
+ --hash=sha256:c428c6c31eb5f4277d7f8eccaf767fbd548ddd5ce3c8b4f4cbbfab3d96b5904c \
+ --hash=sha256:c658c50ac0c98cd755a2dd50b7977d3bca7df401dcc47fbdfa87db53ef7d4e8b \
+ --hash=sha256:c71fb0d56c920c269cd3e2e3fe7c610e3f1fdb21a6ce60efa6430ff63676cea6 \
+ --hash=sha256:c7b742bf31c88566b4bb6335a7f393bb322e580b6bb98df7bd0c25e6e3519ce8 \
+ --hash=sha256:cc0329df4caaceb950d2f580b5ac716a377f7059624a0bafaeaf8a218c6ed774 \
+ --hash=sha256:cc5d36d96478aa9c60654bd932525bf32964c62a7281eafdf16d85003a8d6004 \
+ --hash=sha256:ce854f5f478050ade5a238731c4ca985a7d3b3cb53ff600a9b5c3b689b5f0a7a \
+ --hash=sha256:ced3fdd71aaa83ce593746c2edb42b7a59cb4c19c8b5c407781c72e493aae55a \
+ --hash=sha256:cee5dd7c6fb5dd52a0fe2a740f9bc6e3593f5f8b1788bde49de02086f30182b2 \
+ --hash=sha256:cfa1c0cc3a8f9f53f1243a5a99ac36fd003880199383b37672e86ddda9cb07e2 \
+ --hash=sha256:d1ee1e296209fdce05b81b663250eefa02213a2da7b41bf26f7829b8ba3545aa \
+ --hash=sha256:d59b75732e9b6f27388e10c14b0259cc5f2e48c78627d185e6a177b58ad3cffe \
+ --hash=sha256:d63600d620ad0064c3a748b950ac5ea38a80190e5498532efefa4b7b3f1da1f3 \
+ --hash=sha256:dd732602a7009217f658d5863d12d79d373a4de0eebc111094bcdd3bb8e0a6cc \
+ --hash=sha256:e06efa066f7dbadbc84ebc126a97c452a6451dfcf589d89d788484949e1cf795 \
+ --hash=sha256:e199fb99720074809a7720f1c0b4d919eea8b87e88713e0f8f602f7bef543d9d \
+ --hash=sha256:e4b018dc5a0eee4676e38fe84a47a427816c590b93b55d9025274ec4d6ffc2dc \
+ --hash=sha256:e6621fb2a4988d6e53eedc455e5903e2679f3967b8acb3d639f1b63c14a2e893 \
+ --hash=sha256:e71c909f353863b2b89c83de2ebed71ea6d0df8a6ef65a128193c5e650766bef \
+ --hash=sha256:e90251c0c7bdd54a100a0dce3c07b7e637278c93af29dbf78ebb89a58c4bac7d \
+ --hash=sha256:e9fbdce1e47394b09bc9f26ab117dfc8d6491977a11d86f592bb42c779db2fda \
+ --hash=sha256:eb12fb2ba69ffa05f8695f61c69e591dc4b4a12ac3757ac8af8adb259bf56d17 \
+ --hash=sha256:eda059b6bc8bc0812d626fd91a7ce01bf583df0a61296eff390fd94141a34e30 \
+ --hash=sha256:f03ac127268b43ef4fe9e6ab6794a6794b49485a0cc0c1db79876d2f33f75bc7 \
+ --hash=sha256:f298e218441525d3794428b4c8b8fb8662c6d3ea79925d4807ee6b9a96a3bca5 \
+ --hash=sha256:f5542f9b941279d82d41eb0aa9f98eba36fe4df5c7086c651df7944935b37182 \
+ --hash=sha256:f6f7deae3feb4edfa2efaf7c574fe88cbf055038a6abdb40188e4fff66d5699f \
+ --hash=sha256:f9b1e28d0e8dbfa858abdba91d6b547beaf2df1a59bec6da6faae7b96a4991a9 \
+ --hash=sha256:f9f8405c2c758532c74fed975dbee57be1f31a6e865c031870c79a6ed3212ada \
+ --hash=sha256:fa48b1b63d639f9483e0633e092f5851e2348c352f1f9bb6c8182f87884ef876 \
+ --hash=sha256:fb78f6e7fcd8ad785d28cd577168bc1aaee827b25bb8755638f694794ea98f0a \
+ --hash=sha256:fbc597639158fd7c14d55e808718848319540f51b0e6746e3eefa59723a4a348 \
+ --hash=sha256:fce8cbd4997efeb450bd298b54f755dcdff18d496f7a5ddbb4867c6d7c88fdc3 \
+ --hash=sha256:fd0350afdc3aabd5576f60ea109228bd5538139713c7b094c5cd27c73a98bc6f \
+ --hash=sha256:fd0a274c0e5f9a21565cd9d3dd749b61f96b7aa1e20a93aa1ba4029518f2e5c0 \
+ --hash=sha256:fdb8a068947befafba9952162645dc2fecaeb400e64584829ed5e9b2fbe21a7f
+cloudpickle==3.1.2 \
+ --hash=sha256:7fda9eb655c9c230dab534f1983763de5835249750e85fbcef43aaa30a9a2414 \
+ --hash=sha256:9acb47f6afd73f60dc1df93bb801b472f05ff42fa6c84167d25cb206be1fbf4a
colorama==0.4.6 \
--hash=sha256:08695f5cb7ed6e0531a20572697297273c47b8cae5a63ffc6d6ed5c201be6e44 \
--hash=sha256:4f1d9991f5acc0ca119f9d443620b77f9d6b33703e51011c16baf57afb285fc6
@@ -198,8 +241,6 @@ contourpy==1.3.2 \
--hash=sha256:f26b383144cf2d2c29f01a1e8170f50dacf0eac02d64139dcd709a8ac4eb3cfe \
--hash=sha256:f939a054192ddc596e031e50bb13b657ce318cf13d264f095ce9db7dc6ae81c0 \
--hash=sha256:fd93cc7f3139b6dd7aab2f26a90dde0aa9fc264dbf70f6740d498a70b860b82c
-cucim==23.10.0 \
- --hash=sha256:2c061ad28c3c1fa67bb62260f0a556f354c8ec2d6e3811eece375ed896d62945
cuda-bindings==12.9.4 \
--hash=sha256:1f53a7f453d4b2643d8663d036bafe29b5ba89eb904c133180f295df6dc151e5 \
--hash=sha256:20f2699d61d724de3eb3f3369d57e2b245f93085cab44fd37c3bea036cea1a6f \
@@ -225,10 +266,10 @@ cuda-bindings==12.9.4 \
--hash=sha256:d80bffc357df9988dca279734bc9674c3934a654cab10cadeed27ce17d8635ee \
--hash=sha256:f69107389e6b9948969bfd0a20c4f571fd1aefcfb1d2e1b72cc8ba5ecb7918ab \
--hash=sha256:fda147a344e8eaeca0c6ff113d2851ffca8f7dfc0a6c932374ee5c47caa649c8
-cuda-pathfinder==1.5.2 \
- --hash=sha256:0c5f160a7756c5b072723cbbd6d861e38917ef956c68150b02f0b6e9271c71fa
-cuda-toolkit[cudart]==12.8.1 \
- --hash=sha256:adc7906af4ecbf9a352f9dca5734eceb21daec281ccfcf5675e1d2f724fc2cba
+cuda-pathfinder==1.8.1 \
+ --hash=sha256:ae0137ff9e56ea97499bcbf54f5f2778ec25f3266715ac86da192a795af982a8
+cuda-toolkit[cudart]==12.8.2.0 \
+ --hash=sha256:79040e750f28b959415283ccc44cce541c07003553d675b8ff45760e0189aa71
cupy-cuda12x==13.6.0 \
--hash=sha256:297b4268f839de67ef7865c2202d3f5a0fb8d20bd43360bc51b6e60cb4406447 \
--hash=sha256:4d2dfd9bb4705d446f542739a3616b4c9eea98d674fce247402cc9bcec89a1e4 \
@@ -318,195 +359,232 @@ fastrlock==0.8.3 \
--hash=sha256:f2b84b2fe858e64946e54e0e918b8a0e77fc7b09ca960ae1e50a130e8fbc9af8 \
--hash=sha256:f68c551cf8a34b6460a3a0eba44bd7897ebfc820854e19970c52a76bf064a59f \
--hash=sha256:fcb50e195ec981c92d0211a201704aecbd9e4f9451aea3a6f71ac5b1ec2c98cf
-filelock==3.25.2 \
- --hash=sha256:b64ece2b38f4ca29dd3e810287aa8c48182bbecd1ae6e9ae126c9b35f1382694 \
- --hash=sha256:ca8afb0da15f229774c9ad1b455ed96e85a81373065fb10446672f64444ddf70
+filelock==3.32.6 \
+ --hash=sha256:3f16ecd0117feae0dfc147e8c62eb5daeccd8bd800378c3ddf416de9b4feb6b1 \
+ --hash=sha256:a3f55a18af3652a94d8f47d6055df434f254ca1d02ef2524850c6d249ca2512c
flatbuffers==25.12.19 \
--hash=sha256:7634f50c427838bb021c2d66a3d1168e9d199b0607e6329399f04846d42e20b4
-fonttools==4.62.1 \
- --hash=sha256:0aa72c43a601cfa9273bb1ae0518f1acadc01ee181a6fc60cd758d7fdadffc04 \
- --hash=sha256:0b3ae47e8636156a9accff64c02c0924cbebad62854c4a6dbdc110cd5b4b341a \
- --hash=sha256:12859ff0b47dd20f110804c3e0d0970f7b832f561630cd879969011541a464a9 \
- --hash=sha256:149f7d84afca659d1a97e39a4778794a2f83bf344c5ee5134e09995086cc2392 \
- --hash=sha256:1596aeaddf7f78e21e68293c011316a25267b3effdaccaf4d59bc9159d681b82 \
- --hash=sha256:19177c8d96c7c36359266e571c5173bcee9157b59cfc8cb0153c5673dc5a3a7d \
- --hash=sha256:1c5c25671ce8805e0d080e2ffdeca7f1e86778c5cbfbeae86d7f866d8830517b \
- --hash=sha256:1eecc128c86c552fb963fe846ca4e011b1be053728f798185a1687502f6d398e \
- --hash=sha256:268abb1cb221e66c014acc234e872b7870d8b5d4657a83a8f4205094c32d2416 \
- --hash=sha256:2d850f66830a27b0d498ee05adb13a3781637b1826982cd7e2b3789ef0cc71ae \
- --hash=sha256:2e7abd2b1e11736f58c1de27819e1955a53267c21732e78243fa2fa2e5c1e069 \
- --hash=sha256:403d28ce06ebfc547fbcb0cb8b7f7cc2f7a2d3e1a67ba9a34b14632df9e080f9 \
- --hash=sha256:40975849bac44fb0b9253d77420c6d8b523ac4dcdcefeff6e4d706838a5b80f7 \
- --hash=sha256:486f32c8047ccd05652aba17e4a8819a3a9d78570eb8a0e3b4503142947880ed \
- --hash=sha256:49a445d2f544ce4a69338694cad575ba97b9a75fff02720da0882d1a73f12800 \
- --hash=sha256:59b372b4f0e113d3746b88985f1c796e7bf830dd54b28374cd85c2b8acd7583e \
- --hash=sha256:5a648bde915fba9da05ae98856987ca91ba832949a9e2888b48c47ef8b96c5a9 \
- --hash=sha256:5f37df1cac61d906e7b836abe356bc2f34c99d4477467755c216b72aa3dc748b \
- --hash=sha256:6706d1cb1d5e6251a97ad3c1b9347505c5615c112e66047abbef0f8545fa30d1 \
- --hash=sha256:68959f5fc58ed4599b44aad161c2837477d7f35f5f79402d97439974faebfebe \
- --hash=sha256:6acb4109f8bee00fec985c8c7afb02299e35e9c94b57287f3ea542f28bd0b0a7 \
- --hash=sha256:7487782e2113861f4ddcc07c3436450659e3caa5e470b27dc2177cade2d8e7fd \
- --hash=sha256:7aa21ff53e28a9c2157acbc44e5b401149d3c9178107130e82d74ceb500e5056 \
- --hash=sha256:7bca7a1c1faf235ffe25d4f2e555246b4750220b38de8261d94ebc5ce8a23c23 \
- --hash=sha256:8d337fdd49a79b0d51c4da87bc38169d21c3abbf0c1aa9367eff5c6656fb6dae \
- --hash=sha256:8f8fca95d3bb3208f59626a4b0ea6e526ee51f5a8ad5d91821c165903e8d9260 \
- --hash=sha256:90365821debbd7db678809c7491ca4acd1e0779b9624cdc6ddaf1f31992bf974 \
- --hash=sha256:92bb00a947e666169c99b43753c4305fc95a890a60ef3aeb2a6963e07902cc87 \
- --hash=sha256:93c316e0f5301b2adbe6a5f658634307c096fd5aae60a5b3412e4f3e1728ab24 \
- --hash=sha256:942b03094d7edbb99bdf1ae7e9090898cad7bf9030b3d21f33d7072dbcb51a53 \
- --hash=sha256:9c125ffa00c3d9003cdaaf7f2c79e6e535628093e14b5de1dccb08859b680936 \
- --hash=sha256:9dde91633f77fa576879a0c76b1d89de373cae751a98ddf0109d54e173b40f14 \
- --hash=sha256:9e7863e10b3de72376280b515d35b14f5eeed639d1aa7824f4cf06779ec65e42 \
- --hash=sha256:a24decd24d60744ee8b4679d38e88b8303d86772053afc29b19d23bb8207803c \
- --hash=sha256:a5d8825e1140f04e6c99bb7d37a9e31c172f3bc208afbe02175339e699c710e1 \
- --hash=sha256:aa69d10ed420d8121118e628ad47d86e4caa79ba37f968597b958f6cceab7eca \
- --hash=sha256:ad5cca75776cd453b1b035b530e943334957ae152a36a88a320e779d61fc980c \
- --hash=sha256:b4e0fcf265ad26e487c56cb12a42dffe7162de708762db951e1b3f755319507d \
- --hash=sha256:b820fcb92d4655513d8402d5b219f94481c4443d825b4372c75a2072aa4b357a \
- --hash=sha256:bd13b7999d59c5eb1c2b442eb2d0c427cb517a0b7a1f5798fc5c9e003f5ff782 \
- --hash=sha256:bdfe592802ef939a0e33106ea4a318eeb17822c7ee168c290273cbd5fabd746c \
- --hash=sha256:c05557a78f8fa514da0f869556eeda40887a8abc77c76ee3f74cf241778afd5a \
- --hash=sha256:c22b1014017111c401469e3acc5433e6acf6ebcc6aa9efb538a533c800971c79 \
- --hash=sha256:c9b9e288b4da2f64fd6180644221749de651703e8d0c16bd4b719533a3a7d6e3 \
- --hash=sha256:d241cdc4a67b5431c6d7f115fdf63335222414995e3a1df1a41e1182acd4bcc7 \
- --hash=sha256:e54c75fd6041f1122476776880f7c3c3295ffa31962dc6ebe2543c00dca58b5d \
- --hash=sha256:e8514f4924375f77084e81467e63238b095abda5107620f49421c368a6017ed2 \
- --hash=sha256:ee91628c08e76f77b533d65feb3fbe6d9dad699f95be51cf0d022db94089cdc4 \
- --hash=sha256:ef46db46c9447103b8f3ff91e8ba009d5fe181b1920a83757a5762551e32bb68 \
- --hash=sha256:fa1d16210b6b10a826d71bed68dd9ec24a9e218d5a5e2797f37c573e7ec215ca
-fsspec==2026.2.0 \
- --hash=sha256:6544e34b16869f5aacd5b90bdf1a71acb37792ea3ddf6125ee69a22a53fb8bff \
- --hash=sha256:98de475b5cb3bd66bedd5c4679e87b4fdfe1a3bf4d707b151b3c07e58c9a2437
+fonttools==4.64.0 \
+ --hash=sha256:043f6c572bf236f2a76e762c25f841daea11e8fc03e78088d7be66e0c5b4e4c0 \
+ --hash=sha256:06b6409b868494556a831ae33b2d9a090476c37516b38d70f45a9720b460d423 \
+ --hash=sha256:08f172961e11f4eb4f80f2f20049e09b0ea8e044fa6d456fed8346eb8588f360 \
+ --hash=sha256:09657817b75575822bcd6098ef0ebf0386f34430839ee53109e70fd40a7f6539 \
+ --hash=sha256:1c3661324f3f0fa4539a32288a3e0711a5f3ccf020036e760bb558ae9811a16f \
+ --hash=sha256:1e4e84b47839d35be24dbf476845a34f2ccf99707b66df125c1c414d3e86d25d \
+ --hash=sha256:236e59bc7e2a63557a4d7b013f9cb9e28d9aebc45bc09f85e545e6bf091db626 \
+ --hash=sha256:2524a26f8fdb9051b0d778d052f5d238285ca9f91a7dc004514c7d6cf38d35f4 \
+ --hash=sha256:2730946ca8f12c356bd98eb9b2b095c8e761ed05bed5afb0d5b380cebe4f6370 \
+ --hash=sha256:2c42237b7e8c6813643e57d3efed3be094d4c06339dc2166b626e2cc5c12ee93 \
+ --hash=sha256:3200180abc69639483cf54a17cca2e13c31ede5f665979ea0a9c829d093f372f \
+ --hash=sha256:398b14f89ca950b288bd290875f07e4e10685644fa4ac668546fb107b1ada4d4 \
+ --hash=sha256:45e3ecc3888f1637094fd75cd8fc727f3a4b06d1ddf89181126c071e244fd2a5 \
+ --hash=sha256:4691a122b8c1d0d82d6e7510ce59d5c42146518240274b53e912e255573924f7 \
+ --hash=sha256:498f02ea92c9ca18c0f9c581ea93184a9d56c25b0af14189b0767adaf34235d8 \
+ --hash=sha256:4a05783ff54ce4c7a28f18e5772efdf63c219374bd9ffc55452182e1cef8be60 \
+ --hash=sha256:507c553cdb5abe2e951b5368423849fe29911a828c2135319c3e500e3bf25b32 \
+ --hash=sha256:50e52b6f479ddb1fe32423c2ec860811f36584cf6eabf279fb9a4f98b859a8b4 \
+ --hash=sha256:53eee22af5b5a305c1ee2652955ed46b148e881456fcec1e7f0eb27f642f6bb4 \
+ --hash=sha256:5af87d1a6d247d7467ee082ae977a5443b2c45f8cd4d59375b6daa38d523c2de \
+ --hash=sha256:5b90ad6637237b636d15c9ae8b7c4a7a1c194f33def378677e468c13fd4542f8 \
+ --hash=sha256:5bfdaada437e7730c17d366bd7bb8c4a16639963ddbfc1b2f302a68a17a290e7 \
+ --hash=sha256:66a83f93579fb3493e458c4449d1d566a7b2a1c7b19915cd0fa3c9b8b5a8540b \
+ --hash=sha256:6786bed88581e19bc4f28ea7a64ad531e8f54acf50327fddca942688824a60bd \
+ --hash=sha256:6946c033a144086d5b98c976b72f476b70c93fbbedf914eee0e886f073a4e9fa \
+ --hash=sha256:6eae4376adb104c2acfa76fd9ea0cb12b572ca1d70eceac709871f638ff76e93 \
+ --hash=sha256:6f1ce9ef9a1b13098efdc2e43a2ed96d9851bbde7b31c652a87552c4efe9b422 \
+ --hash=sha256:70fd99e5a09fb77f14b29d70879a4fce9529b2d2948b14c96708e0a61e001b98 \
+ --hash=sha256:730eed859508cb7b0775ebe6bb39f18901f168eb989d8ee23a4fe082700e1e3f \
+ --hash=sha256:769fb64412ca237547ca73f111a64252d9e32c9d938bed51ed537bc9146a8f54 \
+ --hash=sha256:7d7995b906666037d7114c20a5566a372902747452af7d5bd4cd6bca8f1a2550 \
+ --hash=sha256:801fd04899d72eab34f02ab78d0451525621b3bd589da9d2d480dfffe951b643 \
+ --hash=sha256:8252f20108e557532f91d7d6dd9af87c16ed6fa930f65516aa480fa2cfed3363 \
+ --hash=sha256:83cc48d1411d2ff388dab99973dca81172cc9ceae9c9799da9548d494cfb38cb \
+ --hash=sha256:89356c0793b474af7e49ec90d39fb2363e2341516a90460e38231df5ebe8acd5 \
+ --hash=sha256:8dd18fdff0ac9759b8d67a714730abee07b2312e3656c20ba5affb0107094762 \
+ --hash=sha256:917fd520bb60809d83c14d43cfe48d5ad2516abaf2c073d65a431800dade2d29 \
+ --hash=sha256:9443eefff58aad558608f352092e1be6d278980e8c3b4e8621fcbfda97818500 \
+ --hash=sha256:9ecb2b206b5b2386f6968721a0770226b66bdd54adc4279bfff3ddf62873eed8 \
+ --hash=sha256:a0afa8bac675445dc0e2ba2891ecbedd9be89cb437afa94c823e0290cc2c4bc5 \
+ --hash=sha256:a3238a693e806a3158375c6403b8f6f71d86eb9c149b60c97f26dfd560c98ac8 \
+ --hash=sha256:a515f664cad988f2295056833a59f62220bc3e46afdaffe389a29060f6712355 \
+ --hash=sha256:a8c631303bb1fd7be3067c47536a30ff1fcb4846d6008c112bc52a03f7cd6965 \
+ --hash=sha256:b2763e452b025ee8e990f0462e76052de9bb094ebc21d296f62c6dfe958886b4 \
+ --hash=sha256:b4a7af455ffed980925bc0ebf5b8d6239e6c3e797d9d755b6db192fb3080d614 \
+ --hash=sha256:be084d19a3ac0c8b2aba696680642d703118d3b1f18cf83f5b7dbaf0ffc62ab6 \
+ --hash=sha256:c3c1fb656063a2f762db5378ea8d38ad5f7836b4f3fb8c4652270ded43df2935 \
+ --hash=sha256:c60be0aed97a32c6ba8cee21f0d0477136e495451bd97910f589ac892db120d4 \
+ --hash=sha256:cf67f96dc0bfe9607f5f2b734cedfbe2f6f995231adee4ccefa12872044d452d \
+ --hash=sha256:d16102cbcd4615b09c64e6022733faccc93200785f1ab0d4493afb8b0261edde \
+ --hash=sha256:d30c966bea2deffa19c738c81776f7182da5ccabd97e666bae4f3d6ba87341d9 \
+ --hash=sha256:d652592c71683941b768306fa1c7c6ce1bb9b072505043feafe86305d71030b7 \
+ --hash=sha256:da4c9bdeaf6b06c12d13d0addfc8ef15aa9695d26574a6dc10751258bef72f30 \
+ --hash=sha256:dac25768be4c03a990c359f408cb7e8958ed0e93061e495b3642ce7909761205 \
+ --hash=sha256:dc96150f99e05a317cb1f042b92c4cf8bc93cdb1f9f85717322e202ecdf2e505 \
+ --hash=sha256:de8acaa5f4160f537a3cf41b031171d51004b9f4aebfa6c194f18dffa9533d03 \
+ --hash=sha256:e412767d1c9765cf1b82f7b00f1686c6ca5809ebb77af363b3f9f2325a465c01 \
+ --hash=sha256:e4812f71c39d77ec5041348dafa400532adf7bf8f1fffa9aa6495fce5876d7b8 \
+ --hash=sha256:e63b63b8b5fdb8e29318dff2b15c5f852be46e972775b466f75b848f6eed4502 \
+ --hash=sha256:e662f874ab2c7da9861584db44a13573e0936df087215f63013138f6e5eba083 \
+ --hash=sha256:e7b34209eef39462563c05ea9dcf51c272a2ded56f5753da925e66bca3baa484 \
+ --hash=sha256:ecb2e59a7bc692fee64dda6010deb66222335693b30046f15cccf81233aa715f \
+ --hash=sha256:f521d79d6acda4923b264805541696f452079db0952a5bb96f9ff742f50629ec \
+ --hash=sha256:f8669ce37851b597d3435b91fefa51139e58d506ca449ca0e5bb68c63b8b6d2b \
+ --hash=sha256:fa75c7970bc6bca340cc6e20f20f069201bfcb50094c31a536fd99724d1d01ca \
+ --hash=sha256:ff7aff4637fbf71394df139c63ccfe08a47aa4252d2f91224ddb3335c716c925
+fsspec==2026.7.0 \
+ --hash=sha256:b57ddbafedfaef7018c1ecab32aa200a9d7ca26b77965f64e48b70061249d279 \
+ --hash=sha256:c803c40f4cf860b49dea58ee3e1c33cb9c790520e233537e1340049f89b82a88
humanfriendly==10.0 \
--hash=sha256:1697e1a8a8f550fd43c2865cd84542fc175a61dcb779b6fee18cf6b6ccba1477 \
--hash=sha256:6b0b831ce8f15f7300721aa49829fc4e83921a9a301cc7f606be6686a2288ddc
-idna==3.18 \
- --hash=sha256:7f952cbe720b688055e3f87de14f5c3e5fdaa8bc3928985c4077ca689de849a2 \
- --hash=sha256:ffb385a7e039654cef1ab9ef32c6fafe283c0c0467bba1d9029738ce4a14a848
+idna==3.19 \
+ --hash=sha256:5e0811a4383b21dc5838069f801c4fb62113b7447663d2530d2bd6e77b49bf15 \
+ --hash=sha256:815e7be7a7806d54abb586dc943addc79e8b2ee16915059658cbeff4b1b43bf4
jinja2==3.1.6 \
--hash=sha256:0137fb05990d35f1275a587e9aee6d56da821fc83491a0fb838183be43f66d6d \
--hash=sha256:85ece4451f492d0c13c5dd7c13a64681a86afae63a5f347908daf103ce6d2f67
-kiwisolver==1.5.0 \
- --hash=sha256:012b1eb16e28718fa782b5e61dc6f2da1f0792ca73bd05d54de6cb9561665fc9 \
- --hash=sha256:01808c6d15f4c3e8559595d6d1fe6411c68e4a3822b4b9972b44473b24f4e679 \
- --hash=sha256:0255a027391d52944eae1dbb5d4cc5903f57092f3674e8e544cdd2622826b3f0 \
- --hash=sha256:0b85aad90cea8ac6797a53b5d5f2e967334fa4d1149f031c4537569972596cb8 \
- --hash=sha256:0bf3acf1419fa93064a4c2189ac0b58e3be7872bf6ee6177b0d4c63dc4cea276 \
- --hash=sha256:0c50b89ffd3e1a911c69a1dd3de7173c0cd10b130f56222e57898683841e4f96 \
- --hash=sha256:0cbe94b69b819209a62cb27bdfa5dc2a8977d8de2f89dfd97ba4f53ed3af754e \
- --hash=sha256:0df54df7e686afa55e6f21fb86195224a6d9beb71d637e8d7920c95cf0f89aac \
- --hash=sha256:0e3aafb33aed7479377e5e9a82e9d4bf87063741fc99fc7ae48b0f16e32bdd6f \
- --hash=sha256:12e91c215a96e39f57989c8912ae761286ac5a9584d04030ceb3368a357f017a \
- --hash=sha256:1465387ac63576c3e125e5337a6892b9e99e0627d52317f3ca79e6930d889d15 \
- --hash=sha256:16b85d37c2cbb3253226d26e64663f755d88a03439a9c47df6246b35defbdfb7 \
- --hash=sha256:1b0feb50971481a2cc44d94e88bdb02cdd497618252ae226b8eb1201b957e368 \
- --hash=sha256:1d49a49ac4cbfb7c1375301cd1ec90169dfeae55ff84710d782260ce77a75a02 \
- --hash=sha256:1d9daea4ea6b9be74fe2f01f7fbade8d6ffab263e781274cffca0dba9be9eec9 \
- --hash=sha256:1dd9b0b119a350976a6d781e7278ec7aca0b201e1a9e2d23d9804afecb6ca681 \
- --hash=sha256:1f1489f769582498610e015a8ef2d36f28f505ab3096d0e16b4858a9ec214f57 \
- --hash=sha256:2517e24d7315eb51c10664cdb865195df38ab74456c677df67bb47f12d088a27 \
- --hash=sha256:295d9ffe712caa9f8a3081de8d32fc60191b4b51c76f02f951fd8407253528f4 \
- --hash=sha256:2a075bd7bd19c70cf67c8badfa36cf7c5d8de3c9ddb8420c51e10d9c50e94920 \
- --hash=sha256:32cc0a5365239a6ea0c6ed461e8838d053b57e397443c0ca894dcc8e388d4374 \
- --hash=sha256:332b4f0145c30b5f5ad9374881133e5aa64320428a57c2c2b61e9d891a51c2f3 \
- --hash=sha256:377815a8616074cabbf3f53354e1d040c35815a134e01d7614b7692e4bf8acfa \
- --hash=sha256:38f4a703656f493b0ad185211ccfca7f0386120f022066b018eb5296d8613e23 \
- --hash=sha256:3ac2360e93cb41be81121755c6462cff3beaa9967188c866e5fce5cf13170859 \
- --hash=sha256:3c4923e404d6bcd91b6779c009542e5647fef32e4a5d75e115e3bbac6f2335eb \
- --hash=sha256:3cdcb35dc9d807259c981a85531048ede628eabcffb3239adf3d17463518992d \
- --hash=sha256:41024ed50e44ab1a60d3fe0a9d15a4ccc9f5f2b1d814ff283c8d01134d5b81bc \
- --hash=sha256:413b820229730d358efd838ecbab79902fe97094565fdc80ddb6b0a18c18a581 \
- --hash=sha256:4432b835675f0ea7414aab3d37d119f7226d24869b7a829caeab49ebda407b0c \
- --hash=sha256:4db576bb8c3ef9365f8b40fe0f671644de6736ae2c27a2c62d7d8a1b4329f099 \
- --hash=sha256:4e7f886f47ab881692f278ae901039a234e4025a68e6dfab514263a0b1c4ae05 \
- --hash=sha256:4e9750bc21b886308024f8a54ccb9a2cc38ac9fa813bf4348434e3d54f337ff9 \
- --hash=sha256:5060731cc3ed12ca3a8b57acd4aeca5bbc2f49216dd0bec1650a1acd89486bcd \
- --hash=sha256:50847dca5d197fcbd389c805aa1a1cf32f25d2e7273dc47ab181a517666b68cc \
- --hash=sha256:5092eb5b1172947f57d6ea7d89b2f29650414e4293c47707eb499ec07a0ac796 \
- --hash=sha256:5124d1ea754509b09e53738ec185584cc609aae4a3b510aaf4ed6aa047ef9303 \
- --hash=sha256:51e8c4084897de9f05898c2c2a39af6318044ae969d46ff7a34ed3f96274adca \
- --hash=sha256:530a3fd64c87cffa844d4b6b9768774763d9caa299e9b75d8eca6a4423b31314 \
- --hash=sha256:56fa888f10d0f367155e76ce849fa1166fc9730d13bd2d65a2aa13b6f5424489 \
- --hash=sha256:58f812017cd2985c21fbffb4864d59174d4903dd66fa23815e74bbc7a0e2dd57 \
- --hash=sha256:59cd8683f575d96df5bb48f6add94afc055012c29e28124fcae2b63661b9efb1 \
- --hash=sha256:5ae8e62c147495b01a0f4765c878e9bfdf843412446a247e28df59936e99e797 \
- --hash=sha256:5b233ea3e165e43e35dba1d2b8ecc21cf070b45b65ae17dd2747d2713d942021 \
- --hash=sha256:6176c1811d9d5a04fa391c490cc44f451e240697a16977f11c6f722efb9041db \
- --hash=sha256:62f59da443c4f4849f73a51a193b1d9d258dcad0c41bc4d1b8fb2bcc04bfeb22 \
- --hash=sha256:6783e069732715ad0c3ce96dbf21dbc2235ab0593f2baf6338101f70371f4028 \
- --hash=sha256:6ab8ba9152203feec73758dad83af9a0bbe05001eb4639e547207c40cfb52083 \
- --hash=sha256:70d593af6a6ca332d1df73d519fddb5148edb15cd90d5f0155e3746a6d4fcc65 \
- --hash=sha256:72ec46b7eba5b395e0a7b63025490d3214c11013f4aacb4f5e8d6c3041829588 \
- --hash=sha256:7a32f72973f0f950c1920475d5c5ea3d971b81b6f0ec53b8d0a956cc965f22e0 \
- --hash=sha256:7a4aa69609f40fce3cbc3f87b2061f042eee32f94b8f11db707b66a26461591a \
- --hash=sha256:7c60d3c9b06fb23bd9c6139281ccbdc384297579ae037f08ae90c69f6845c0b1 \
- --hash=sha256:800ee55980c18545af444d93fdd60c56b580db5cc54867d8cbf8a1dc0829938c \
- --hash=sha256:80aa065ffd378ff784822a6d7c3212f2d5f5e9c3589614b5c228b311fd3063ac \
- --hash=sha256:86e0287879f75621ae85197b0877ed2f8b7aa57b511c7331dce2eb6f4de7d476 \
- --hash=sha256:893ff3a711d1b515ba9da14ee090519bad4610ed1962fbe298a434e8c5f8db53 \
- --hash=sha256:89fc958c702ee9a745e4700378f5d23fddbc46ff89e8fdbf5395c24d5c1452a3 \
- --hash=sha256:8c63c91f95173f9c2a67c7c526b2cea976828a0e7fced9cdcead2802dc10f8a4 \
- --hash=sha256:8df31fe574b8b3993cc61764f40941111b25c2d9fea13d3ce24a49907cd2d615 \
- --hash=sha256:8f9baf6f0a6e7571c45c8863010b45e837c3ee1c2c77fcd6ef423be91b21fedb \
- --hash=sha256:9027d773c4ff81487181a925945743413f6069634d0b122d0b37684ccf4f1e18 \
- --hash=sha256:9190426b7aa26c5229501fa297b8d0653cfd3f5a36f7990c264e157cbf886b3b \
- --hash=sha256:940dda65d5e764406b9fb92761cbf462e4e63f712ab60ed98f70552e496f3bf1 \
- --hash=sha256:94eff26096eb5395136634622515b234ecb6c9979824c1f5004c6e3c3c85ccd2 \
- --hash=sha256:9eed0f7edbb274413b6ee781cca50541c8c0facd3d6fd289779e494340a2b85c \
- --hash=sha256:ad4ae4ffd1ee9cd11357b4c66b612da9888f4f4daf2f36995eda64bd45370cac \
- --hash=sha256:b0f172dc8ffaccb8522d7c5d899de00133f2f1ca7b0a49b7da98e901de87bf2d \
- --hash=sha256:b2af221f268f5af85e776a73d62b0845fc8baf8ef0abfae79d29c77d0e776aaf \
- --hash=sha256:b7d335370ae48a780c6e6a6bbfa97342f563744c39c35562f3f367665f5c1de2 \
- --hash=sha256:b83af57bdddef03c01a9138034c6ff03181a3028d9a1003b301eb1a55e161a3f \
- --hash=sha256:bb5136fb5352d3f422df33f0c879a1b0c204004324150cc3b5e3c4f310c9049f \
- --hash=sha256:bc4d8e252f532ab46a1de9349e2d27b91fce46736a9eedaa37beaca66f574ed4 \
- --hash=sha256:bdd3e53429ff02aa319ba59dfe4ceeec345bf46cf180ec2cf6fd5b942e7975e9 \
- --hash=sha256:be12f931839a3bdfe28b584db0e640a65a8bcbc24560ae3fdb025a449b3d754e \
- --hash=sha256:be4a51a55833dc29ab5d7503e7bcb3b3af3402d266018137127450005cdfe737 \
- --hash=sha256:beb7f344487cdcb9e1efe4b7a29681b74d34c08f0043a327a74da852a6749e7b \
- --hash=sha256:bf4679a3d71012a7c2bf360e5cd878fbd5e4fcac0896b56393dec239d81529ed \
- --hash=sha256:c0e1403fd7c26d77c1f03e096dc58a5c726503fa0db0456678b8668f76f521e3 \
- --hash=sha256:c31c13da98624f957b0fb1b5bae5383b2333c2c3f6793d9825dd5ce79b525cb7 \
- --hash=sha256:c438f6ca858697c9ab67eb28246c92508af972e114cac34e57a6d4ba17a3ac08 \
- --hash=sha256:c8277104ded0a51e699c8c3aff63ce2c56d4ed5519a5f73e0fd7057f959a2b9e \
- --hash=sha256:c95cab08d1965db3d84a121f1c7ce7479bdd4072c9b3dafd8fecce48a2e6b902 \
- --hash=sha256:cc0b66c1eec9021353a4b4483afb12dfd50e3669ffbb9152d6842eb34c7e29fd \
- --hash=sha256:cdee07c4d7f6d72008d3f73b9bf027f4e11550224c7c50d8df1ae4a37c1402a6 \
- --hash=sha256:ce9bf03dad3b46408c08649c6fbd6ca28a9fce0eb32fdfffa6775a13103b5310 \
- --hash=sha256:cff8e5383db4989311f99e814feeb90c4723eb4edca425b9d5d9c3fefcdd9537 \
- --hash=sha256:d168fda2dbff7b9b5f38e693182d792a938c31db4dac3a80a4888de603c99554 \
- --hash=sha256:d1ffeb80b5676463d7a7d56acbe8e37a20ce725570e09549fe738e02ca6b7e1e \
- --hash=sha256:d36ca54cb4c6c4686f7cbb7b817f66f5911c12ddb519450bbe86707155028f87 \
- --hash=sha256:d4193f3d9dc3f6f79aaed0e5637f45d98850ebf01f7ca20e69457f3e8946b66a \
- --hash=sha256:d5cd5189fc2b6a538b75ae45433140c4823463918f7b1617c31e68b085c0022c \
- --hash=sha256:d618fd27420381a4f6044faa71f46d8bfd911bd077c555f7138ed88729bfbe79 \
- --hash=sha256:d76e2d8c75051d58177e762164d2e9ab92886534e3a12e795f103524f221dd8e \
- --hash=sha256:daae526907e262de627d8f70058a0f64acc9e2641c164c99c8f594b34a799a16 \
- --hash=sha256:db485b3847d182b908b483b2ed133c66d88d49cacf98fd278fadafe11b4478d1 \
- --hash=sha256:dd952e03bfbb096cfe2dd35cd9e00f269969b67536cb4370994afc20ff2d0875 \
- --hash=sha256:dda366d548e89a90d88a86c692377d18d8bd64b39c1fb2b92cb31370e2896bbd \
- --hash=sha256:e315e5ec90d88e140f57696ff85b484ff68bb311e36f2c414aa4286293e6dee0 \
- --hash=sha256:e4415a8db000bf49a6dd1c478bf70062eaacff0f462b92b0ba68791a905861f9 \
- --hash=sha256:e7a116ae737f0000343218c4edf5bd45893bfeaff0993c0b215d7124c9f77646 \
- --hash=sha256:e7c4c09a490dc4d4a7f8cbee56c606a320f9dc28cf92a7157a39d1ce7676a657 \
- --hash=sha256:ebae99ed6764f2b5771c522477b311be313e8841d2e0376db2b10922daebbba4 \
- --hash=sha256:ec4c85dc4b687c7f7f15f553ff26a98bfe8c58f5f7f0ac8905f0ba4c7be60232 \
- --hash=sha256:ed3a984b31da7481b103f68776f7128a89ef26ed40f4dc41a2223cda7fb24819 \
- --hash=sha256:f18c2d9782259a6dc132fdc7a63c168cbc74b35284b6d75c673958982a378384 \
- --hash=sha256:f1f9f4121ec58628c96baa3de1a55a4e3a333c5102c8e94b64e23bf7b2083309 \
- --hash=sha256:f42c23db5d1521218a3276bb08666dcb662896a0be7347cba864eca45ff64ede \
- --hash=sha256:f443b4825c50a51ee68585522ab4a1d1257fac65896f282b4c6763337ac9f5d2 \
- --hash=sha256:f6764a4ccab3078db14a632420930f6186058750df066b8ea2a7106df91d3203 \
- --hash=sha256:f7c7553b13f69c1b29a5bde08ddc6d9d0c8bfb84f9ed01c30db25944aeb852a7 \
- --hash=sha256:fa6248cd194edff41d7ea9425ced8ca3a6f838bfb295f6f1d6e6bb694a8518df \
- --hash=sha256:fa8eb9ecdb7efb0b226acec134e0d709e87a909fa4971a54c0c4f6e88635484c \
- --hash=sha256:fc20894c3d21194d8041a28b65622d5b86db786da6e3cfe73f0c762951a61167 \
- --hash=sha256:fc4d3f1fb9ca0ae9f97b095963bc6326f1dbfd3779d6679a1e016b9baaa153d3 \
- --hash=sha256:fd40bb9cd0891c4c3cb1ddf83f8bbfa15731a248fdc8162669405451e2724b09 \
- --hash=sha256:ff710414307fefa903e0d9bdf300972f892c23477829f49504e59834f4195398
-lazy-loader==0.5 \
- --hash=sha256:717f9179a0dbed357012ddad50a5ad3d5e4d9a0b8712680d4e687f5e6e6ed9b3 \
- --hash=sha256:ab0ea149e9c554d4ffeeb21105ac60bed7f3b4fd69b1d2360a4add51b170b005
+kiwisolver==1.5.1 \
+ --hash=sha256:007a5553dfc4f4e8d184f588a0200e2cd4b63a59cc8796df3c39909e679dc7a0 \
+ --hash=sha256:0324cd2567259b7a095f6cf18a52b0ffc6f3de9e69528ff1bc0e7a37bd43ff1a \
+ --hash=sha256:0627b9bceb9c3cdcf12b8a18655eedfed2692b038df27423383c120d0b7dc2d6 \
+ --hash=sha256:06a6917674de9e0fe3f66f5430787f59a9f2ddb64af9b714eaec547e29ef5c19 \
+ --hash=sha256:072bdb15a3c19a5b5dbc8f8fb1f4e1884bf4f3507eeb4cc6334401274d37a5c0 \
+ --hash=sha256:0a4faea5c6db201c6a21391d2ac926ea97acf7dacdbc3c417189e1adb1a00837 \
+ --hash=sha256:0ba9527afc80ae3d7814ed98b6572d02bf85eaf48065678342c5f0c6dab7a8c7 \
+ --hash=sha256:0d8924877ce22e17326a99a418c3c82037da078df3c6a260b13eca677444e6e7 \
+ --hash=sha256:0ebdef3eae5336568147c39a55be6a2036ffde53faa9ca2d978989ae7c2da12c \
+ --hash=sha256:1209042a623ddfda5497e4066c7b77651dde8e1d3a9dd97599dc7e97f3b9b78c \
+ --hash=sha256:16895f553ee6620a827d2da56b871f835fb70b9216cca5d188e885caf6e3bd23 \
+ --hash=sha256:17851e5dad4484be0cbccbde3b15331deae036de9aebd45eed964487802b172f \
+ --hash=sha256:1798e83840c3f627246104c4d8a9639c60fa068adf9ce92b61791781fa8a68c1 \
+ --hash=sha256:18170a77ddfecf40ec60d0928268dc95880c881864e015a8f34094ed18b9b9ad \
+ --hash=sha256:186884a58486651e3c217b6acea0a53eaa9498fdd472057c46f2f0fb5c25aad5 \
+ --hash=sha256:18a0cfb124546a4c2e6087c5f3029c7f44b37c85b142e0ced71f73a7599ac208 \
+ --hash=sha256:1983f0974a750a6f6556f368ba11105d1d8369c735b944747c9f12ae5aea7aae \
+ --hash=sha256:1a7587dc335f2c0f5bd577fd0540bd16c66006bdb60f759a1059f025e6c4f071 \
+ --hash=sha256:1acc7e5b7ef05e9da8bb70cd6c7c4513090213d2e1ad9720f599f0bf6c52aec5 \
+ --hash=sha256:1d852545c4d0e35a72728d072cbaa59e2fa7dd84bdf01e068d670dd0ceb58eb6 \
+ --hash=sha256:1ed0f5e49d0ceff8b72190824d9e59c062fbbc02c231b853112c78474b3f5ec2 \
+ --hash=sha256:1fff05e239575b1481b6ed1a782f6fad616efbf1f0b1f44e6e85c4dfe426e483 \
+ --hash=sha256:21e46b23a2da695c364124817bc01d970effd5483147f8d66a6a7167e3f6b851 \
+ --hash=sha256:22d5e5aaad6be121f2515765e3b1c444352cb8eb4c86510801db8f2e50757316 \
+ --hash=sha256:2551cf9917af48ee7c4b29cc82320489508cf96fd26a51f6fc124de661cd44c7 \
+ --hash=sha256:255605693a483db7bd5c79f60437f7bf658f7f520d61aa42722e32257c941951 \
+ --hash=sha256:26e8268480be5061d509e29669d59103c067a26377a56491630ece11762e3858 \
+ --hash=sha256:27add358abe374ebaa3b8763ef380bc99051b5a4b18d94878366a9e4f59efef0 \
+ --hash=sha256:2ae70bc59790d2af72a3f76f24b272403e135070340281108b447cb77ea70819 \
+ --hash=sha256:2e10ae1bba1899188b33557c10d73affcc12033edd18adddb57d209039976a4c \
+ --hash=sha256:3221f78211074f561c44ca42eac0619828171bec15a2c4cf6f7747d07df76e8e \
+ --hash=sha256:34633ecf50d16187ab8e5528b7a2530f2feb4e23f300db4672538b51cfc5cd38 \
+ --hash=sha256:34ec467940442c9943016fb2d4c81d1ba84351eeca2f1a78f8bc87f1ba0d414c \
+ --hash=sha256:37f801b5d7cc0e5a548921308e059fd2b057bb42972b591cfa3049f95423c4ed \
+ --hash=sha256:38f6e0deb4d0a4615efe0c4efc5990b06ae450ab50a0b321c0b078b6d238c083 \
+ --hash=sha256:3c24cd69455e1b00ddf770c13b6e2c33e07d6dc3f2d34add0bf9277c5c6bbd46 \
+ --hash=sha256:3cc210010fd2f438a3ed430b45f1b501fd13a8618bf984dc2c5ce5b69b78752e \
+ --hash=sha256:3fa5855898f6d3d01b72ccd48a2d65cbdee301251603fefe34e2025bddba219c \
+ --hash=sha256:416ba7ff9f233b7036689bb5a3783537e838ad483f63558d2a800f75afe738b1 \
+ --hash=sha256:431dc224a1a92a5c8f582d96e505196a3b5997a7271076678da2dfde67b77e9a \
+ --hash=sha256:43844c1a7ad6d723d5b5b4c4fc7f5bd399c40e288120d16257c7c9e8765c6e85 \
+ --hash=sha256:44b8faef94f1857e77fa0238f3390ff1ac51d2ea20a487e2e452a59fd2b5f5ca \
+ --hash=sha256:470d420f98d368d6f010633a20659b544c5fdfa5329e6b70219f2ef08fd4a7ef \
+ --hash=sha256:482676e5bd48d70ac99d9fc78863469845421e01184fa83f1f9366dc49f7e974 \
+ --hash=sha256:4d4ca09bf13cff792b1884f64b98ee6c2467930d632233be25c56b442d99f10e \
+ --hash=sha256:5025e36fb4fb275cef0a4e30dbb11cb4ae61d1c83deb90189cb5d7e4cafd6b55 \
+ --hash=sha256:509735237ae0d849e8a843551d423d2500d2e0a9ac1611a145658b29c0fb9f85 \
+ --hash=sha256:534f02c1abb31ed6dbd3515545285c330b2f12d00fdb1fdb71658b9ca5a13a6a \
+ --hash=sha256:5978c3340f16a35c30f8ab2fa7bcf559973c55f1a5ef6970e1f621acf3c4db13 \
+ --hash=sha256:5b973887ff782cfd6b67c9904ad8ca542e0bc5e4961503408b423b5a688b4d38 \
+ --hash=sha256:5c490db2168a508088f59140dd392556a54b8bd1048fc6383c8baff13c359673 \
+ --hash=sha256:5d142e352eb13facc7dd047489aebdff6ba78576c239f1ea04931979caaf0567 \
+ --hash=sha256:5daa1f19e097050b9c4d9a78fcc9263cb96c9dfae08037ddc1b7c4ad1889f2a2 \
+ --hash=sha256:61e9a64c7635095a6bfe483e2ff055d437c59bd45f3617a228b37277f0185d62 \
+ --hash=sha256:63fb7294b768f444eb4b068965f2662f28c2fd4161e23bd60fcf3ff27b74c046 \
+ --hash=sha256:685929988b208a911f1285e2f8ed54210b0d681a3dc0f03e00d599d291986e7e \
+ --hash=sha256:6a797a1cefc8b9c93170db580337e1fe3d011ad18b1299943231279406342048 \
+ --hash=sha256:6b92f60017dda7d877fdc546438b5e28f31c523264f49cf5a48c1d0ce1a0dfbc \
+ --hash=sha256:70ed9a45c7484d2b30cdacf60d220f494a1763b9fec1ad03285c6553fa0889f2 \
+ --hash=sha256:719a35fa1156db3640555f95ebb94f60a444e64d1c69626b0edef5df78eba225 \
+ --hash=sha256:74ad5c3dad54a4641b4c28cd15ded70899d04459c6c7aeacafea716be97cce6d \
+ --hash=sha256:74ea337e0ec3f6f342a36a4f1b5cd94dd9affddcd28ba9aae2905af932ee8c6b \
+ --hash=sha256:75d9b1cf8258462dbdc1eeda718c96ea7f079324c09067f6daabfcf37712b7fe \
+ --hash=sha256:77a4c8187a5948d7f8795adb765a3c7b553d07d86d88e43038fc32fc1fb9a3f3 \
+ --hash=sha256:7824b5e8bdbf0bccb4ccd37bbb115849a1dc45437fb4de8351385ed07c437ee0 \
+ --hash=sha256:7d38b0c279c3032e8c9cc013b405c6df8e1668dbf15465779aa7f15f61201812 \
+ --hash=sha256:7e9c01d3dd7ceba4d1d436cc021d40d592466e40b9bc7f5d83dc4e98a5c9cd8c \
+ --hash=sha256:7fd82debf43c6acd0a94359d232f6bb516ee13f269a7993736a9ac9f988bb5d9 \
+ --hash=sha256:824c3d763a05ea9e9003610145186b0e9848c7584a5575c79bac5a8e7cd80bad \
+ --hash=sha256:828f75af2b0080c8a972e75f649ab46af008e92c6104a57a759157200b835b75 \
+ --hash=sha256:83f78128fa28705fa85d01c59771c72fe81c11bd0e6155edbb9f818983a7d761 \
+ --hash=sha256:876bbfd276473d3daffe30e8c975df4ed9429967b41a6cb362dbb5155b6f13ad \
+ --hash=sha256:886fc26012f0e8b5f69d1cfe6d711f6b11f194621539bf8e6bb1c25c5dc82724 \
+ --hash=sha256:8a34616dc2521cc8dc1d7d081734da63539f021ac0450ce950908340c6e7aa2f \
+ --hash=sha256:8a708a47ade1fe19e8371d5da076bac0dd4b0a5a7985ad6c637f7f7e361b6baa \
+ --hash=sha256:8af9b142ad719ae3a911ebf616bc4b78b32bbab84d6a40d3ad2f129670509957 \
+ --hash=sha256:8bf4df63592c2a66b4f8edc5df2544998c288aa02f96ce0acd880cd1de8c8127 \
+ --hash=sha256:8de6f2a4ce7e7bd27d23dd94abf0ccafe0e0e5cc9c764b0577191f2c25f08f26 \
+ --hash=sha256:8f8fddb8e323bd6eee4e54e69a39243beab22689070f4c66b472c4cc88bb89d8 \
+ --hash=sha256:8fca690b00c4c48f6c2a547b0160ed511357093a4e4c9b47e0fadf3128066d89 \
+ --hash=sha256:9506e892bcc3b409831d363c6f53e5985e1c8d1f6f6b0256d00358684ff85378 \
+ --hash=sha256:958254518717542d02d0688d0d20cbf771da5e415e6f49543f92481c850a4540 \
+ --hash=sha256:95a02752aa032eef4aed01cda6d9b687c669bd0396bf4519eef8bba22a286720 \
+ --hash=sha256:96c30002424670b5e1e46495c2b8cbffef39cf77c1d79e76462029d50339785b \
+ --hash=sha256:98b208a7cc42c803445ef551d6753cc42a5ea13e9cab1ee66cd8b9cb70195330 \
+ --hash=sha256:9b3092d8992a1d69b7a59c3e39f35e1b9be327a17f68a7c35fc17329e337d6f2 \
+ --hash=sha256:9e51c119992ea8820706871c30a4642ec76de20ae82f9b50b9a45517d8e9f810 \
+ --hash=sha256:a5716a33bfabb2c6ce27b6cf03253467b3804f83e215f4d202685cf93c6c9874 \
+ --hash=sha256:a5a00665d1a0e26763a7338d7e911d4598fbc1d50dd0d6b7919b7dc6c5d6569f \
+ --hash=sha256:a5ca5aebae78a0bc13c1943af4af615d4966c5b650b05d5aa83b50e427196fee \
+ --hash=sha256:a7b85b2cc6ea45e5f7e8c9a30bc9fabd47cda09106cbb4b967335c3e6c43b69d \
+ --hash=sha256:a83ee7107df13abe42a54a6654670eef9bb39425cf2e27f65e0007465e1286ab \
+ --hash=sha256:aa7d00b1700966d2917e54d278aba86897890ca9276dd8b76cf6446b6c181b92 \
+ --hash=sha256:ab620eb663952455271ac37f9aaad86b73c969c02f11f53cea405b38e96a4300 \
+ --hash=sha256:ad8b9671348d7c8716715652ae11f85ed0eb99e265a2df2ca490577d69860b2c \
+ --hash=sha256:aefe930d113798330e9462f7874542977869c0613cba3262e2de3a8d5dee8f3a \
+ --hash=sha256:b03af77d77e50edba2030fd5f7c352ff209314b09030a3cba7c14edf9a09a444 \
+ --hash=sha256:b390aec180a7c054919c04898835e1c77bced23ea8383eb2c570213bf25d1a86 \
+ --hash=sha256:b3d78f7bb2b9d9a30345be1474b9aaa8685430b54afb51ba3639b5c6c11e9ed6 \
+ --hash=sha256:b5664603a253efd3a75716d793d1d3a6a82723b61dc6db767b2460bbbeec4c0f \
+ --hash=sha256:b69602970994a2ed8bbfa78c2f0394a7435226c6040489702d9f0a0ad0c07052 \
+ --hash=sha256:b6ae6a0328f0bc035741820fdeecdcd67bf4694eee03972e843663107122f450 \
+ --hash=sha256:bad20d4c69c851c982a1e3606f4c293edfd5a87885786c50082412240c4b1ffd \
+ --hash=sha256:bb7c99f0673c03017a3ee01e54a5c2617a05468b11eabe513b0080e063ed95b1 \
+ --hash=sha256:bebb89489b279b2f5661bbbb2abcc87bcd4a46607bb4a5c966f04f1db6b8df9a \
+ --hash=sha256:bfd1de989b3330420e29de39352f5c049905c9e3ee67233a50d550e3d652c148 \
+ --hash=sha256:c2306e8bb53601979fcb3fa09cc65e031876d9ae01eff2fcbcd7a84ef94d5bc1 \
+ --hash=sha256:c3a4e41e3096bf1f0f1b76e2ffd6d828d6547f574f702d59bdbef7acfa59db9c \
+ --hash=sha256:c6834b92dd2428e2dd85ef3d85f723d3c12f20aaf43a2ddd4f944ca25d833408 \
+ --hash=sha256:c90d3022d8a94778939cda8638c6c8da8fa757b8958dad7ec868ce29c87681b8 \
+ --hash=sha256:ca307d6c259e5c98d3cb9ade55342b47a6839762caf2536f3d7b46ee660cc82e \
+ --hash=sha256:ca7f6fe0f37ca978a1e5eb7a3a68e6413f417e78e838324947ffd420202b198b \
+ --hash=sha256:cb6fae641357ed2f6e533c0d3c6504a4a5703621a50c89459e46051d56b61140 \
+ --hash=sha256:cdaeeb6c350106df6bf9d873395973e5f066a9713200b72cd64f55d0a3eafab6 \
+ --hash=sha256:cea20da04494e662b83c872683bf4ff2345206043d036315ed0e924b652e7294 \
+ --hash=sha256:cea90547bfd93807e0013a004dc76552be44fad3bc1cc2b38610a9e889ed098f \
+ --hash=sha256:d09037ca068d784ebc4aec290ef952ca27ac15dd9c0b5801a88c6e1096b83e6b \
+ --hash=sha256:d27c2123977cb9269c30a49ba45f03a4323017ef693e19db4ec9dbe1299a3002 \
+ --hash=sha256:d50de98e8d807dc31822fff96f50293163a62418eb65487a21b42713d72ed0b7 \
+ --hash=sha256:d66a64dd5dec136040ec2ae94aa026a912ee60fdd45bc28d3db30037fd809e88 \
+ --hash=sha256:d79308fa689fac89cbcfbd4dbfc80b5f95c54c5a7fd4d194be221f9d33d026e6 \
+ --hash=sha256:da3275833be0edbaf4830fae08bae3dc7219f40ce0c37eaa6c25825957e06612 \
+ --hash=sha256:dc1a26b8e53395a01c2c611e58602fa47461f136fba7cd5542e6db6d64be1839 \
+ --hash=sha256:dc23390afe9f4ef9ac3bcc72a03a56eebbde03f4c571a32cb38f859cff9a6524 \
+ --hash=sha256:e05c2f7925f1d88778e53cb44f14e0223204a3bdd09a41664750363acfb1f2ef \
+ --hash=sha256:e12dfea7f5fc2a34a9080efbf79c4c44eb380ec5b9c6fea09407e08f0d1e941d \
+ --hash=sha256:e4e4523d6f336708d732516e6cfca7796cf3d96c9474eb5aecf6165f2f1fefc3 \
+ --hash=sha256:e4e49f7e1a4e7191bdf9dc67a974db714501b1fc52c24324103d06a86abd5c08 \
+ --hash=sha256:e68e151428b5384f766cd25739bf77c7e4a3dc93b5ded7a12118d9fbfdf78ab6 \
+ --hash=sha256:e8e4d953faaded9ec7ede36824e9814082d22d4c7b1eafbfa079ecba8cd0d076 \
+ --hash=sha256:ee9df1f0d77b9c6e94f4ac0fec533fbddd5ea3a327807f18d7b069ae019ded80 \
+ --hash=sha256:f0a887b6565bbfe80efde2b7f6e8890d7d9bbdb11bdb17028a3690c32fe0621f \
+ --hash=sha256:f0f4a42db92d6ec7677ab9d12830a2a8ec145a9c6d15db2b593466bc875c78d7 \
+ --hash=sha256:f1303ef2eec81262a4b708c3e858afe58d7c75ad91c1c05266eda7673369859a \
+ --hash=sha256:f1d56ec54d257d05e0b50f5780d967540cd07beeaf9e5f645b26d50cce79f4d8 \
+ --hash=sha256:f4167e87b397f273dc2356fcf1eaf50a6bac51e6105f45103ef7129c8efb0255 \
+ --hash=sha256:f76fc85bd054c806960f917ec0f329e24e436f1712267d90588e4c39890caa63 \
+ --hash=sha256:f942903fde7363d1d879057ec5de01310efda2597161784d752fa9953a01a71a \
+ --hash=sha256:f9b1c4900736e489a812c529100de4b8fb617d4db075e931e213c57424b83d9b \
+ --hash=sha256:fc271a6f0a2126958f4090e5507b9da5848927dae331f8f763bd4aa642b3d2cd \
+ --hash=sha256:febcce10f2bcdbb80b4ea919238a6a4ac13dbc4c7cadbe8d5d75c3682f8b5404
markupsafe==3.0.3 \
--hash=sha256:0303439a41979d9e74d18ff5e2dd8c43ed6c6001fd40e5bf2e43f7bd9bbc523f \
--hash=sha256:068f375c472b3e7acbe2d5318dea141359e6900156b5b2ba06a30b169086b91a \
@@ -597,62 +675,62 @@ markupsafe==3.0.3 \
--hash=sha256:f71a396b3bf33ecaa1626c255855702aca4d3d9fea5e051b41ac59a9c1c41edc \
--hash=sha256:f9e130248f4462aaa8e2552d547f36ddadbeaa573879158d721bbd33dfe4743a \
--hash=sha256:fed51ac40f757d41b7c48425901843666a6677e3e8eb0abcff09e4ba6e664f50
-matplotlib==3.10.8 \
- --hash=sha256:00270d217d6b20d14b584c521f810d60c5c78406dc289859776550df837dcda7 \
- --hash=sha256:0a33deb84c15ede243aead39f77e990469fff93ad1521163305095b77b72ce4a \
- --hash=sha256:113bb52413ea508ce954a02c10ffd0d565f9c3bc7f2eddc27dfe1731e71c7b5f \
- --hash=sha256:12d90df9183093fcd479f4172ac26b322b1248b15729cb57f42f71f24c7e37a3 \
- --hash=sha256:15d30132718972c2c074cd14638c7f4592bd98719e2308bccea40e0538bc0cb5 \
- --hash=sha256:18821ace09c763ec93aef5eeff087ee493a24051936d7b9ebcad9662f66501f9 \
- --hash=sha256:1ae029229a57cd1e8fe542485f27e7ca7b23aa9e8944ddb4985d0bc444f1eca2 \
- --hash=sha256:2299372c19d56bcd35cf05a2738308758d32b9eaed2371898d8f5bd33f084aa3 \
- --hash=sha256:238b7ce5717600615c895050239ec955d91f321c209dd110db988500558e70d6 \
- --hash=sha256:24d50994d8c5816ddc35411e50a86ab05f575e2530c02752e02538122613371f \
- --hash=sha256:25d380fe8b1dc32cf8f0b1b448470a77afb195438bafdf1d858bfb876f3edf7b \
- --hash=sha256:2c1998e92cd5999e295a731bcb2911c75f597d937341f3030cc24ef2733d78a8 \
- --hash=sha256:2cf5bd12cecf46908f286d7838b2abc6c91cda506c0445b8223a7c19a00df008 \
- --hash=sha256:32f8dce744be5569bebe789e46727946041199030db8aeb2954d26013a0eb26b \
- --hash=sha256:37b3c1cc42aa184b3f738cfa18c1c1d72fd496d85467a6cf7b807936d39aa656 \
- --hash=sha256:3a48a78d2786784cc2413e57397981fb45c79e968d99656706018d6e62e57958 \
- --hash=sha256:3ab4aabc72de4ff77b3ec33a6d78a68227bf1123465887f9905ba79184a1cc04 \
- --hash=sha256:3c624e43ed56313651bc18a47f838b60d7b8032ed348911c54906b130b20071b \
- --hash=sha256:3f2e409836d7f5ac2f1c013110a4d50b9f7edc26328c108915f9075d7d7a91b6 \
- --hash=sha256:3f5c3e4da343bba819f0234186b9004faba952cc420fbc522dc4e103c1985908 \
- --hash=sha256:41703cc95688f2516b480f7f339d8851a6035f18e100ee6a32bc0b8536a12a9c \
- --hash=sha256:495672de149445ec1b772ff2c9ede9b769e3cb4f0d0aa7fa730d7f59e2d4e1c1 \
- --hash=sha256:4cf267add95b1c88300d96ca837833d4112756045364f5c734a2276038dae27d \
- --hash=sha256:56271f3dac49a88d7fca5060f004d9d22b865f743a12a23b1e937a0be4818ee1 \
- --hash=sha256:595ba4d8fe983b88f0eec8c26a241e16d6376fe1979086232f481f8f3f67494c \
- --hash=sha256:5f62550b9a30afde8c1c3ae450e5eb547d579dd69b25c2fc7a1c67f934c1717a \
- --hash=sha256:646d95230efb9ca614a7a594d4fcacde0ac61d25e37dd51710b36477594963ce \
- --hash=sha256:64fcc24778ca0404ce0cb7b6b77ae1f4c7231cdd60e6778f999ee05cbd581b9a \
- --hash=sha256:6be43b667360fef5c754dda5d25a32e6307a03c204f3c0fc5468b78fa87b4160 \
- --hash=sha256:6da7c2ce169267d0d066adcf63758f0604aa6c3eebf67458930f9d9b79ad1db1 \
- --hash=sha256:83d282364ea9f3e52363da262ce32a09dfe241e4080dcedda3c0db059d3c1f11 \
- --hash=sha256:9153c3292705be9f9c64498a8872118540c3f4123d1a1c840172edf262c8be4a \
- --hash=sha256:99eefd13c0dc3b3c1b4d561c1169e65fe47aab7b8158754d7c084088e2329466 \
- --hash=sha256:a0a7f52498f72f13d4a25ea70f35f4cb60642b466cbb0a9be951b5bc3f45a486 \
- --hash=sha256:a2b336e2d91a3d7006864e0990c83b216fcdca64b5a6484912902cef87313d78 \
- --hash=sha256:a48f2b74020919552ea25d222d5cc6af9ca3f4eb43a93e14d068457f545c2a17 \
- --hash=sha256:ad3d9833a64cf48cc4300f2b406c3d0f4f4724a91c0bd5640678a6ba7c102077 \
- --hash=sha256:b44d07310e404ba95f8c25aa5536f154c0a8ec473303535949e52eb71d0a1565 \
- --hash=sha256:b53285e65d4fa4c86399979e956235deb900be5baa7fc1218ea67fbfaeaadd6f \
- --hash=sha256:b5a2b97dbdc7d4f353ebf343744f1d1f1cca8aa8bfddb4262fcf4306c3761d50 \
- --hash=sha256:b9a5ca4ac220a0cdd1ba6bcba3608547117d30468fefce49bb26f55c1a3d5c58 \
- --hash=sha256:bab485bcf8b1c7d2060b4fcb6fc368a9e6f4cd754c9c2fea281f4be21df394a2 \
- --hash=sha256:c108a1d6fa78a50646029cb6d49808ff0fc1330fda87fa6f6250c6b5369b6645 \
- --hash=sha256:d56a1efd5bfd61486c8bc968fa18734464556f0fb8e51690f4ac25d85cbbbbc2 \
- --hash=sha256:d9050fee89a89ed57b4fb2c1bfac9a3d0c57a0d55aed95949eedbc42070fea39 \
- --hash=sha256:dd80ecb295460a5d9d260df63c43f4afbdd832d725a531f008dad1664f458adf \
- --hash=sha256:e8ea3e2d4066083e264e75c829078f9e149fa119d27e19acd503de65e0b13149 \
- --hash=sha256:eb3823f11823deade26ce3b9f40dcb4a213da7a670013929f31d5f5ed1055b22 \
- --hash=sha256:ee40c27c795bda6a5292e9cff9890189d32f7e3a0bf04e0e3c9430c4a00c37df \
- --hash=sha256:efb30e3baaea72ce5928e32bab719ab4770099079d66726a62b11b1ef7273be4 \
- --hash=sha256:f254d118d14a7f99d616271d6c3c27922c092dac11112670b157798b89bf4933 \
- --hash=sha256:f89c151aab2e2e23cb3fe0acad1e8b82841fd265379c4cecd0f3fcb34c15e0f6 \
- --hash=sha256:f97aeb209c3d2511443f8797e3e5a569aebb040d4f8bc79aa3ee78a8fb9e3dd8 \
- --hash=sha256:f9b587c9c7274c1613a30afabf65a272114cd6cdbe67b3406f818c79d7ab2e2a \
- --hash=sha256:fb061f596dad3a0f52b60dc6a5dec4a0c300dec41e058a7efe09256188d170b7
+matplotlib==3.10.9 \
+ --hash=sha256:09218df8a93712bd6ea133e83a153c755448cf7868316c531cffcc43f69d1cc9 \
+ --hash=sha256:10cc5ce06d10231c36f40e875f3c7e8050362a4ee8f0ee5d29a6b3277d57bb42 \
+ --hash=sha256:172db52c9e683f5d12eaf57f0f54834190e12581fe1cc2a19595a8f5acb4e77d \
+ --hash=sha256:1872fb212a05b729e649754a72d5da61d03e0554d76e80303b6f83d1d2c0552b \
+ --hash=sha256:1aa972116abb4c9d201bf245620b433726cb6856f3bef6a78f776a00f5c92d37 \
+ --hash=sha256:1e7698ac9868428e84d2c967424803b2472ff7167d9d6590d4204ed775343c3b \
+ --hash=sha256:2dc9477819ffd78ad12a20df1d9d6a6bd4fec6aaa9072681465fddca052f1456 \
+ --hash=sha256:3225f4e1edcb8c86c884ddf79ebe20ecd0a67d30188f279897554ccd8fded4dc \
+ --hash=sha256:336b9acc64d309063126edcdaca00db9373af3c476bb94388fe9c5a53ad13e6f \
+ --hash=sha256:345f6f68ecc8da0ca56fad2ea08fde1a115eda530079eca185d50a7bc3e146c6 \
+ --hash=sha256:34cf8167e023ad956c15f36302911d5406bd99a9862c1a8499ea6f7c0e015dc2 \
+ --hash=sha256:3fc0364dfbe1d07f6d15c5ebd0c5bf89e126916e5a8667dd4a7a6e84c36653d4 \
+ --hash=sha256:41cb28c2bd769aa3e98322c6ab09854cbcc52ab69d2759d681bba3e327b2b320 \
+ --hash=sha256:42fb814efabe95c06c1994d8ab5a8385f43a249e23badd3ba931d4308e5bca20 \
+ --hash=sha256:4e42042d54db34fda4e95a7bd3e5789c2a995d2dad3eb8850232ee534092fbbf \
+ --hash=sha256:4edcfbd8565339aa62f1cd4012f7180926fdbe71850f7b0d3c379c175cd6b66c \
+ --hash=sha256:51bf0ddbdc598e060d46c16b5590708f81a1624cefbaaf62f6a81bf9285b8c80 \
+ --hash=sha256:56fc0bd271b00025c6edfdc7c2dcd247372c8e1544971d62e1dc7c17367e8bf9 \
+ --hash=sha256:59476c6d29d612b8e9bb6ce8c5b631be6ba8f9e3a2421f22a02b192c7dd28716 \
+ --hash=sha256:6640f75af2c6148293caa0a2b39dd806a492dd66c8a8b04035813e33d0fd2585 \
+ --hash=sha256:68cfdcede415f7c8f5577b03303dd94526cdb6d11036cecdc205e08733b2d2bb \
+ --hash=sha256:6b63d9c7c769b88ab81e10dc86e4e0607cf56817b9f9e6cf24b2a5f1693b8e38 \
+ --hash=sha256:6be157fe17fc37cb95ac1d7374cf717ce9259616edec911a78d9d26dae8522d4 \
+ --hash=sha256:6c63ebcd8b4b169eb2f5c200552ae6b8be8999a005b6b507ed76fb8d7d674fe2 \
+ --hash=sha256:77210dce9cb8153dffc967efaae990543392563d5a376d4dd8539bebcb0ed217 \
+ --hash=sha256:7a8d66a55def891c33147ba3ba9bfcabf0b526a43764c818acbb4525e5ed0838 \
+ --hash=sha256:82368699727bfb7b0182e1aa13082e3c08e092fa1a25d3e1fd92405bff96f6d4 \
+ --hash=sha256:82834c3c292d24d3a8aae77cd2d20019de69d692a34a970e4fdb8d33e2ea3dda \
+ --hash=sha256:8e436d155fa8a3399dc62683f8f5d0e2e50d25d0144a73edd73f82eec8f4abfb \
+ --hash=sha256:8f3bcac1ca5ed000a6f4337d47ba67dfddf37ed6a46c15fd7f014997f7bf865f \
+ --hash=sha256:97e35e8d39ccc85859095e01a53847432ba9a53ddf7986f7a54a11b73d0e143f \
+ --hash=sha256:985f2238880e2e69093f588f5fe2e46771747febf0649f3cf7f7b7480875317f \
+ --hash=sha256:a49f1eadc84ca85fd72fa4e89e70e61bf86452df6f971af04b12c60761a0772c \
+ --hash=sha256:a5a6104ed666402ba5106d7f36e0e0cdca4e8d7fa4d39708ca88019e2835a2eb \
+ --hash=sha256:aba1615dabe83188e19d4f75a253c6a08423e04c1425e64039f800050a69de6b \
+ --hash=sha256:ae20801130378b82d647ff5047c07316295b68dc054ca6b3c13519d0ea624285 \
+ --hash=sha256:ae2f11957b27ce53497dd4d7b235c4d4f1faf383dfb39d0c5beb833bff883294 \
+ --hash=sha256:b049278ddce116aaa1c1377ebf58adea909132dfce0281cf7e3a1ea9fc2e2c65 \
+ --hash=sha256:b1b745c489cd1a77a0dc1120a05dc87af9798faebc913601feb8c73d89bf2d1e \
+ --hash=sha256:b2b9516251cb89ff618d757daec0e2ed1bf21248013844a853d87ef85ab3081d \
+ --hash=sha256:b580440f1ff81a0e34122051a3dfabb7e4b7f9e380629929bde0eff9af72165f \
+ --hash=sha256:ba7b3b8ef09eab7df0e86e9ae086faa433efbfbdb46afcb3aa16aabf779469a8 \
+ --hash=sha256:c27df8b3848f32a83d1767566595e43cfaa4460380974da06f4279a7ec143c39 \
+ --hash=sha256:d091f9d758b34aaaaa6331d13574bf01891d903b3dec59bfff458ef7551de5d6 \
+ --hash=sha256:d730e984eddf56974c3e72b6129c7ca462ac38dc624338f4b0b23eb23ecba00f \
+ --hash=sha256:d75d11c949914165976c621b2324f9ef162af7ebf4b057ddf95dd1dba7e5edcf \
+ --hash=sha256:d843374407c4017a6403b59c6c81606773d136f3259d5b6da3131bc814542cc2 \
+ --hash=sha256:da4e09638420548f31c354032a6250e473c68e5a4e96899b4844cf39ddea23fe \
+ --hash=sha256:de2445a0c6690d21b7eb6ce071cebad6d40a2e9bdf10d039074a96ba19797b99 \
+ --hash=sha256:dfca0129678bd56379db26c52b5d77ed7de314c047492fbdc763aa7501710cfb \
+ --hash=sha256:e9fae004b941b23ff2edcf1567a857ed77bafc8086ffa258190462328434faf8 \
+ --hash=sha256:f0c3c28d9fbcc1fe7a03be236d73430cf6409c41fb2383a7ac52fe932b072cb1 \
+ --hash=sha256:f4399f64b3e94cd500195490972ae1ee81170df1636fa15364d157d5bdd7b921 \
+ --hash=sha256:f76e640a5268850bfda54b5131b1b1941cc685e42c5fa98ed9f2d64038308cba \
+ --hash=sha256:fd66508e8c6877d98e586654b608a0456db8d7e8a546eb1e2600efd957302358
ml-dtypes==0.5.4 \
--hash=sha256:0d2ffd05a2575b1519dc928c0b93c06339eb67173ff53acb00724502cda231cf \
--hash=sha256:11942cbf2cf92157db91e5022633c0d9474d4dfd813a909383bd23ce828a4b7d \
@@ -696,9 +774,9 @@ ml-dtypes==0.5.4 \
mpmath==1.3.0 \
--hash=sha256:7a28eb2a9774d00c7bc92411c19a89209d5da7c4c9a9e227be8330a23a25b91f \
--hash=sha256:a0b2b9fe80bbcd81a6647ff13108738cfb482d481d826cc0e02f5b35e5c88d2c
-networkx==3.1 \
- --hash=sha256:4f33f68cb2afcf86f28a45f43efc27a9386b535d567d2127f8f61d51dec58d36 \
- --hash=sha256:de346335408f84de0eada6ff9fafafff9bcda11f0a0dfaa931133debb146ab61
+networkx==3.4.2 \
+ --hash=sha256:307c3669428c5362aab27c8a1260aa8f47c4e91d3891f48be0141738d8d053e1 \
+ --hash=sha256:df5d4365b724cf81b8c6a7312509d0c22386097011ad1abe274afd5e9d3bbc5f
numpy==1.26.4 \
--hash=sha256:03a8c78d01d9781b28a6989f6fa1bb2c4f2d51201cf99d3dd875df6fbd96b23b \
--hash=sha256:08beddf13648eb95f8d867350f6a018a4be2e5ad54c8d8caed89ebca558b2818 \
@@ -779,6 +857,9 @@ nvidia-cusparselt-cu12==0.7.1 \
--hash=sha256:8878dce784d0fac90131b6817b607e803c36e629ba34dc5b433471382196b6a5 \
--hash=sha256:f1bb701d6b930d5a7cea44c19ceb973311500847f81b634d802b7b539dc55623 \
--hash=sha256:f67fbb5831940ec829c9117b7f33807db9f9678dc2a617fbe781cac17b4e1075
+nvidia-ml-py==13.610.43 \
+ --hash=sha256:65437eb73d68d0c62c931ca4d45038472faff03bd0b8729abba4b899f70d60f2 \
+ --hash=sha256:f13c72698edef492f985cc225f14faafe68ae065a2e407f45bdf6f4b9b43fde8
nvidia-nccl-cu12==2.27.5 \
--hash=sha256:31432ad4d1fb1004eb0c56203dc9bc2178a1ba69d1d9e02d64a6938ab5e40e7a \
--hash=sha256:ad730cf15cb5d25fe849c6e6ca9eb5b76db16a80f13f425ac68d8e2e55624457
@@ -793,35 +874,31 @@ nvidia-nvtx-cu12==12.8.90 \
--hash=sha256:5b17e2001cc0d751a5bc2c6ec6d26ad95913324a4adb86788c944f8ce9ba441f \
--hash=sha256:619c8304aedc69f02ea82dd244541a83c3d9d40993381b3b590f1adaed3db41e \
--hash=sha256:d7ad891da111ebafbf7e015d34879f7112832fc239ff0d7d776b6cb685274615
-onnx==1.21.0 \
- --hash=sha256:10c3185a232089335581fabb98fba4e86d3e8246b8140f2e406082438100ebda \
- --hash=sha256:19d9971a3e52a12968ae6c70fd0f86c349536de0b0c33922ecdbe52d1972fe60 \
- --hash=sha256:1a9baf882562c4cebf79589bebb7cd71a20e30b51158cac3e3bbaf27da6163bd \
- --hash=sha256:257d1d1deb6a652913698f1e3f33ef1ca0aa69174892fe38946d4572d89dd94f \
- --hash=sha256:2aca19949260875c14866fc77ea0bc37e4e809b24976108762843d328c92d3ce \
- --hash=sha256:3abd09872523c7e0362d767e4e63bd7c6bac52a5e2c3edbf061061fe540e2027 \
- --hash=sha256:458d91948ad9a7729a347550553b49ab6939f9af2cddf334e2116e45467dc61f \
- --hash=sha256:4d8b67d0aaec5864c87633188b91cc520877477ec0254eda122bef8be43cd764 \
- --hash=sha256:5489f25fe461e7f32128218251a466cabbeeaf1eaa791c79daebf1a80d5a2cc9 \
- --hash=sha256:5f78c411743db317a76e5d009f84f7e3d5380411a1567a868e82461a1e5c775d \
- --hash=sha256:7b58a4cfec8d9311b73dc083e4c1fa362069267881144c05139b3eba5dc3a840 \
- --hash=sha256:7cd7cb8f6459311bdb557cbf6c0ccc6d8ace11c304d1bba0a30b4a4688e245f8 \
- --hash=sha256:7ee9d8fd6a4874a5fa8b44bbcabea104ce752b20469b88bc50c7dcf9030779ad \
- --hash=sha256:82aa6ab51144df07c58c4850cb78d4f1ae969d8c0bf657b28041796d49ba6974 \
- --hash=sha256:9003d5206c01fa2ff4b46311566865d8e493e1a6998d4009ec6de39843f1b59b \
- --hash=sha256:9ea4e824964082811938a9250451d89c4ec474fe42dd36c038bfa5df31993d1e \
- --hash=sha256:a9261bd580fb8548c9c37b3c6750387eb8f21ea43c63880d37b2c622e1684285 \
- --hash=sha256:ab6a488dabbb172eebc9f3b3e7ac68763f32b0c571626d4a5004608f866cc83d \
- --hash=sha256:bba12181566acf49b35875838eba49536a327b2944664b17125577d230c637ad \
- --hash=sha256:c9b56ad04039fac6b028c07e54afa1ec7f75dd340f65311f2c292e41ed7aa4d9 \
- --hash=sha256:ca14bc4842fccc3187eb538f07eabeb25a779b39388b006db4356c07403a7bbb \
- --hash=sha256:db17fc0fec46180b6acbd1d5d8650a04e5527c02b09381da0b5b888d02a204c8 \
- --hash=sha256:e0c21cc5c7a41d1a509828e2b14fe9c30e807c6df611ec0fd64a47b8d4b16abd \
- --hash=sha256:e1931bfcc222a4c9da6475f2ffffb84b97ab3876041ec639171c11ce802bee6a \
- --hash=sha256:efba467efb316baf2a9452d892c2f982b9b758c778d23e38c7f44fa211b30bb9 \
- --hash=sha256:f2c7c234c568402e10db74e33d787e4144e394ae2bcbbf11000fbfe2e017ad68 \
- --hash=sha256:f53b3c15a3b539c16b99655c43c365622046d68c49b680c48eba4da2a4fb6f27 \
- --hash=sha256:fc2635400fe39ff37ebc4e75342cc54450eadadf39c540ff132c319bf4960095
+onnx==1.22.0 \
+ --hash=sha256:19e45e4af88e3fe3261458d4b8cc461957ae2782a358a3560503569bf3b23b72 \
+ --hash=sha256:1d0a2bdb15eb2b3cb65c438f3423d9620d14fdce32f92380e6bb1b2e09568ef5 \
+ --hash=sha256:239958534464612fbcb6ed23d5228aaa925b39b8773f58726809ffdccb4edd1c \
+ --hash=sha256:2632406b8f523ef2e2873c363f90b20a3d88c0fbcfac757d3addffccf8f452c2 \
+ --hash=sha256:2d8f229a553fa440fe623ed7b36fca5e7762da3af871c3f8f8ce451df73e2914 \
+ --hash=sha256:33ce94119bbb7f05d9caea4ea7549f5185a54369f6bbc9f70171bd5ee6935bbc \
+ --hash=sha256:596fbf0490947533c1c1045ba860851dc9fb77471023dac9a71ba5b42ceab103 \
+ --hash=sha256:5c1c0408a9d4b4df33851672e5fc7590b96301ee123396d608f9ab6f045ab06b \
+ --hash=sha256:6d0ffffd63a4ecc21ddaeddd5bf02099cb701aa4243f2de00122726869065ca4 \
+ --hash=sha256:72ccebab3bac07215c204ce8848d42e78eaaa666badbf72d25cd359b9f269e3a \
+ --hash=sha256:82e9f27fc1223cb06d68a56bed6f9d3caf3d0dad1b61bce45006d529b15bd94c \
+ --hash=sha256:8561a2c00041c07e08db0c228593b5b4694100398685f348532af7dbb84189da \
+ --hash=sha256:87a3077958f66f9a26dec10077ac28326d9cec2cbe1f0b040947243449754573 \
+ --hash=sha256:8907b9b9389893bc0dc6314cc00ee1e3a69844e48d689eacc6a0340411a7da58 \
+ --hash=sha256:8a5eccce2d5fc6c5046928a9aa7cdd9750ea4a586f8de341d3d40d820c35fdec \
+ --hash=sha256:8e268cdc0547e3949799ffd4a44451dc2b9080b57d0824a2db680b6ec65506f0 \
+ --hash=sha256:955e02e1f6d385b53d52f9cd7b9cdf5caf417c300bcfe3c64c6d542be763845b \
+ --hash=sha256:a1a89a7cb9ba13d78f009bdec448ec82a98972589734f157022a2bff7a5973a6 \
+ --hash=sha256:a3a39fc4643867aecb33417fdddb11e308ee79d2d4a584b9d50cc7aec2091b13 \
+ --hash=sha256:ae5a563f281cd9d2845622cecf6c092a57e4ee1b138f66fdbbdd4200567a5e16 \
+ --hash=sha256:c21a0e59fd967a95b358e4a6e756d1f1eec2d304a83480f329f66e30d2bf0223 \
+ --hash=sha256:cc8b66b312f8f03a53e268afb67180a2d97dd12cc79e2b61361c6c0073448016 \
+ --hash=sha256:ef40c0aaf0b643857ea9306fc7eddce17eaf9fb0407e4801f1fc5758443a38e0 \
+ --hash=sha256:f3c120dcdb70ad738f3c061b32798f408ea299eb69f84dd69ab4a6bf3c2ec01f
onnxruntime-gpu==1.23.2 \
--hash=sha256:054282614c2fc9a4a27d74242afbae706a410f1f63cc35bc72f99709029a5ba4 \
--hash=sha256:18de50c6c8eea50acc405ea13d299aec593e46478d7a22cd32cdbbdf7c42899d \
@@ -832,9 +909,9 @@ onnxruntime-gpu==1.23.2 \
--hash=sha256:d76d1ac7a479ecc3ac54482eea4ba3b10d68e888a0f8b5f420f0bdf82c5eec59 \
--hash=sha256:deba091e15357355aa836fd64c6c4ac97dd0c4609c38b08a69675073ea46b321 \
--hash=sha256:fe925a84b00e291e0ad3fac29bfd8f8e06112abc760cdc82cb711b4f3935bd95
-onnxslim==0.1.90 \
- --hash=sha256:bd8695ddb446dd7fb1680c687f4f7a923b1598c9ed05b0b69ff6a8eb8a30050d \
- --hash=sha256:cdea94ee294f4403c7dbe63f9176916b4396a1cd79cd05d67078f212cf9ade28
+onnxslim==0.1.96 \
+ --hash=sha256:386a1c64e61a66fb79cdd62de996323b9fb950d22de9b117cd31f211cac96048 \
+ --hash=sha256:a43f800d39434e6b4d542abe17c25afb7fcc33aabf53d20464cb0dae0f66cbb6
opencv-python==4.11.0.86 \
--hash=sha256:03d60ccae62304860d232272e4a4fda93c39d595780cb40b161b310244b736a4 \
--hash=sha256:085ad9b77c18853ea66283e98affefe2de8cc4c1f43eda4c100cf9b2721142ec \
@@ -843,114 +920,110 @@ opencv-python==4.11.0.86 \
--hash=sha256:6b02611523803495003bd87362db3e1d2a0454a6a63025dc6658a9830570aa0d \
--hash=sha256:810549cb2a4aedaa84ad9a1c92fbfdfc14090e2749cedf2c1589ad8359aa169b \
--hash=sha256:9d05ef13d23fe97f575153558653e2d6e87103995d54e6a35db3f282fe1f9c66
-packaging==26.0 \
- --hash=sha256:00243ae351a257117b6a241061796684b084ed1c516a08c48a3f7e147a9d80b4 \
- --hash=sha256:b36f1fef9334a5588b4166f8bcd26a14e521f2b55e6b9de3aaa80d3ff7a37529
-pillow==12.2.0 \
- --hash=sha256:00a2865911330191c0b818c59103b58a5e697cae67042366970a6b6f1b20b7f9 \
- --hash=sha256:01afa7cf67f74f09523699b4e88c73fb55c13346d212a59a2db1f86b0a63e8c5 \
- --hash=sha256:03e7e372d5240cc23e9f07deca4d775c0817bffc641b01e9c3af208dbd300987 \
- --hash=sha256:03f6fab9219220f041c74aeaa2939ff0062bd5c364ba9ce037197f4c6d498cd9 \
- --hash=sha256:042db20a421b9bafecc4b84a8b6e444686bd9d836c7fd24542db3e7df7baad9b \
- --hash=sha256:0538bd5e05efec03ae613fd89c4ce0368ecd2ba239cc25b9f9be7ed426b0af1f \
- --hash=sha256:0a34329707af4f73cf1782a36cd2289c0368880654a2c11f027bcee9052d35dd \
- --hash=sha256:0c838a5125cee37e68edec915651521191cef1e6aa336b855f495766e77a366e \
- --hash=sha256:144748b3af2d1b358d41286056d0003f47cb339b8c43a9ea42f5fea4d8c66b6e \
- --hash=sha256:1610dd6c61621ae1cf811bef44d77e149ce3f7b95afe66a4512f8c59f25d9ebe \
- --hash=sha256:1e1757442ed87f4912397c6d35a0db6a7b52592156014706f17658ff58bbf795 \
- --hash=sha256:22db17c68434de69d8ecfc2fe821569195c0c373b25cccb9cbdacf2c6e53c601 \
- --hash=sha256:25373b66e0dd5905ed63fa3cae13c82fbddf3079f2c8bf15c6fb6a35586324c1 \
- --hash=sha256:2bb4a8d594eacdfc59d9e5ad972aa8afdd48d584ffd5f13a937a664c3e7db0ed \
- --hash=sha256:2c727a6d53cb0018aadd8018c2b938376af27914a68a492f59dfcaca650d5eea \
- --hash=sha256:2d192a155bbcec180f8564f693e6fd9bccff5a7af9b32e2e4bf8c9c69dbad6b5 \
- --hash=sha256:2e589959f10d9824d39b350472b92f0ce3b443c0a3442ebf41c40cb8361c5b97 \
- --hash=sha256:2e5a76d03a6c6dcef67edabda7a52494afa4035021a79c8558e14af25313d453 \
- --hash=sha256:325ca0528c6788d2a6c3d40e3568639398137346c3d6e66bb61db96b96511c98 \
- --hash=sha256:34c0d99ecccea270c04882cb3b86e7b57296079c9a4aff88cb3b33563d95afaa \
- --hash=sha256:390ede346628ccc626e5730107cde16c42d3836b89662a115a921f28440e6a3b \
- --hash=sha256:394167b21da716608eac917c60aa9b969421b5dcbbe02ae7f013e7b85811c69d \
- --hash=sha256:3997232e10d2920a68d25191392e3a4487d8183039e1c74c2297f00ed1c50705 \
- --hash=sha256:3adc9215e8be0448ed6e814966ecf3d9952f0ea40eb14e89a102b87f450660d8 \
- --hash=sha256:3e080565d8d7c671db5802eedfb438e5565ffa40115216eabb8cd52d0ecce024 \
- --hash=sha256:4a6c9fa44005fa37a91ebfc95d081e8079757d2e904b27103f4f5fa6f0bf78c0 \
- --hash=sha256:4bfd07bc812fbd20395212969e41931001fd59eb55a60658b0e5710872e95286 \
- --hash=sha256:4e6c62e9d237e9b65fac06857d511e90d8461a32adcc1b9065ea0c0fa3a28150 \
- --hash=sha256:50d8520da2a6ce0af445fa6d648c4273c3eeefbc32d7ce049f22e8b5c3daecc2 \
- --hash=sha256:51c4167c34b0d8ba05b547a3bb23578d0ba17b80a5593f93bd8ecb123dd336a3 \
- --hash=sha256:56a3f9c60a13133a98ecff6197af34d7824de9b7b38c3654861a725c970c197b \
- --hash=sha256:56b25336f502b6ed02e889f4ece894a72612fe885889a6e8c4c80239ff6e5f5f \
- --hash=sha256:57850958fe9c751670e49b2cecf6294acc99e562531f4bd317fa5ddee2068463 \
- --hash=sha256:58f62cc0f00fd29e64b29f4fd923ffdb3859c9f9e6105bfc37ba1d08994e8940 \
- --hash=sha256:5c0a9f29ca8e79f09de89293f82fc9b0270bb4af1d58bc98f540cc4aedf03166 \
- --hash=sha256:5cdfebd752ec52bf5bb4e35d9c64b40826bc5b40a13df7c3cda20a2c03a0f5ed \
- --hash=sha256:5d04bfa02cc2d23b497d1e90a0f927070043f6cbf303e738300532379a4b4e0f \
- --hash=sha256:5d2fd0fa6b5d9d1de415060363433f28da8b1526c1c129020435e186794b3795 \
- --hash=sha256:62f5409336adb0663b7caa0da5c7d9e7bdbaae9ce761d34669420c2a801b2780 \
- --hash=sha256:632ff19b2778e43162304d50da0181ce24ac5bb8180122cbe1bf4673428328c7 \
- --hash=sha256:6562ace0d3fb5f20ed7290f1f929cae41b25ae29528f2af1722966a0a02e2aa1 \
- --hash=sha256:673aa32138f3e7531ccdbca7b3901dba9b70940a19ccecc6a37c77d5fdeb05b5 \
- --hash=sha256:6a6e67ea2e6feda684ed370f9a1c52e7a243631c025ba42149a2cc5934dec295 \
- --hash=sha256:6a9adfc6d24b10f89588096364cc726174118c62130c817c2837c60cf08a392b \
- --hash=sha256:6bb77b2dcb06b20f9f4b4a8454caa581cd4dd0643a08bacf821216a16d9c8354 \
- --hash=sha256:6e6b2a0c538fc200b38ff9eb6628228b77908c319a005815f2dde585a0664b60 \
- --hash=sha256:71cde9a1e1551df7d34a25462fc60325e8a11a82cc2e2f54578e5e9a1e153d65 \
- --hash=sha256:7371b48c4fa448d20d2714c9a1f775a81155050d383333e0a6c15b1123dda005 \
- --hash=sha256:766cef22385fa1091258ad7e6216792b156dc16d8d3fa607e7545b2b72061f1c \
- --hash=sha256:7b14cc0106cd9aecda615dd6903840a058b4700fcb817687d0ee4fc8b6e389be \
- --hash=sha256:7f84204dee22a783350679a0333981df803dac21a0190d706a50475e361c93f5 \
- --hash=sha256:8023abc91fba39036dbce14a7d6535632f99c0b857807cbbbf21ecc9f4717f06 \
- --hash=sha256:80b2da48193b2f33ed0c32c38140f9d3186583ce7d516526d462645fd98660ae \
- --hash=sha256:8297651f5b5679c19968abefd6bb84d95fe30ef712eb1b2d9b2d31ca61267f4c \
- --hash=sha256:88d387ff40b3ff7c274947ed3125dedf5262ec6919d83946753b5f3d7c67ea4c \
- --hash=sha256:88ddbc66737e277852913bd1e07c150cc7bb124539f94c4e2df5344494e0a612 \
- --hash=sha256:8bd7903a5f2a4545f6fd5935c90058b89d30045568985a71c79f5fd6edf9b91e \
- --hash=sha256:8be29e59487a79f173507c30ddf57e733a357f67881430449bb32614075a40ab \
- --hash=sha256:8c984051042858021a54926eb597d6ee3012393ce9c181814115df4c60b9a808 \
- --hash=sha256:8cbeb542b2ebc6fcdacabf8aca8c1a97c9b3ad3927d46b8723f9d4f033288a0f \
- --hash=sha256:8e9c4f5b3c546fa3458a29ab22646c1c6c787ea8f5ef51300e5a60300736905e \
- --hash=sha256:90e6f81de50ad6b534cab6e5aef77ff6e37722b2f5d908686f4a5c9eba17a909 \
- --hash=sha256:975385f4776fafde056abb318f612ef6285b10a1f12b8570f3647ad0d74b48ec \
- --hash=sha256:9a8a34cc89c67a65ea7437ce257cea81a9dad65b29805f3ecee8c8fe8ff25ffe \
- --hash=sha256:9aba9a17b623ef750a4d11b742cbafffeb48a869821252b30ee21b5e91392c50 \
- --hash=sha256:9f08483a632889536b8139663db60f6724bfcb443c96f1b18855860d7d5c0fd4 \
- --hash=sha256:a4e8f36e677d3336f35089648c8955c51c6d386a13cf6ee9c189c5f5bd713a9f \
- --hash=sha256:a52edc8bfff4429aaabdf4d9ee0daadbbf8562364f940937b941f87a4290f5ff \
- --hash=sha256:a830b1a40919539d07806aa58e1b114df53ddd43213d9c8b75847eee6c0182b5 \
- --hash=sha256:aa88ccfe4e32d362816319ed727a004423aab09c5cea43c01a4b435643fa34eb \
- --hash=sha256:af73337013e0b3b46f175e79492d96845b16126ddf79c438d7ea7ff27783a414 \
- --hash=sha256:b1c1fbd8a5a1af3412a0810d060a78b5136ec0836c8a4ef9aa11807f2a22f4e1 \
- --hash=sha256:b85f66ae9eb53e860a873b858b789217ba505e5e405a24b85c0464822fe88032 \
- --hash=sha256:b86024e52a1b269467a802258c25521e6d742349d760728092e1bc2d135b4d76 \
- --hash=sha256:bd9c0c7a0c681a347b3194c500cb1e6ca9cab053ea4d82a5cf45b6b754560136 \
- --hash=sha256:bfa9c230d2fe991bed5318a5f119bd6780cda2915cca595393649fc118ab895e \
- --hash=sha256:d362d1878f00c142b7e1a16e6e5e780f02be8195123f164edf7eddd911eefe7c \
- --hash=sha256:d5d38f1411c0ed9f97bcb49b7bd59b6b7c314e0e27420e34d99d844b9ce3b6f3 \
- --hash=sha256:dac8d77255a37e81a2efcbd1fc05f1c15ee82200e6c240d7e127e25e365c39ea \
- --hash=sha256:dd025009355c926a84a612fecf58bb315a3f6814b17ead51a8e48d3823d9087f \
- --hash=sha256:deede7c263feb25dba4e82ea23058a235dcc2fe1f6021025dc71f2b618e26104 \
- --hash=sha256:e74473c875d78b8e9d5da2a70f7099549f9eb37ded4e2f6a463e60125bccd176 \
- --hash=sha256:ee3120ae9dff32f121610bb08e4313be87e03efeadfc6c0d18f89127e24d0c24 \
- --hash=sha256:eedf4b74eda2b5a4b2b2fb4c006d6295df3bf29e459e198c90ea48e130dc75c3 \
- --hash=sha256:efd8c21c98c5cc60653bcb311bef2ce0401642b7ce9d09e03a7da87c878289d4 \
- --hash=sha256:f1c943e96e85df3d3478f7b691f229887e143f81fedab9b20205349ab04d73ed \
- --hash=sha256:f278f034eb75b4e8a13a54a876cc4a5ab39173d2cdd93a638e1b467fc545ac43 \
- --hash=sha256:f3f40b3c5a968281fd507d519e444c35f0ff171237f4fdde090dd60699458421 \
- --hash=sha256:f490f9368b6fc026f021db16d7ec2fbf7d89e2edb42e8ec09d2c60505f5729c7 \
- --hash=sha256:fb043ee2f06b41473269765c2feae53fc2e2fbf96e5e22ca94fb5ad677856f06 \
- --hash=sha256:fc3d34d4a8fbec3e88a79b92e5465e0f9b842b628675850d860b8bd300b159f5
-polars==1.39.3 \
- --hash=sha256:2e016c7f3e8d14fa777ef86fe0477cec6c67023a20ba4c94d6e8431eefe4a63c \
- --hash=sha256:c2b955ccc0a08a2bc9259785decf3d5c007b489b523bf2390cf21cec2bb82a56
-polars-runtime-32==1.39.3 \
- --hash=sha256:06b47f535eb1f97a9a1e5b0053ef50db3a4276e241178e37bbb1a38b1fa53b14 \
- --hash=sha256:363d49e3a3e638fc943e2b9887940300a7d06789930855a178a4727949259dc2 \
- --hash=sha256:425c0b220b573fa097b4042edff73114cc6d23432a21dfd2dc41adf329d7d2e9 \
- --hash=sha256:7c206bdcc7bc62ea038d6adea8e44b02f0e675e0191a54c810703b4895208ea4 \
- --hash=sha256:8bc9e13dc1d2e828331f2fe8ccbc9757554dc4933a8d3e85e906b988178f95ed \
- --hash=sha256:c728e4f469cafab501947585f36311b8fb222d3e934c6209e83791e0df20b29d \
- --hash=sha256:d66ca522517554a883446957539c40dc7b75eb0c2220357fb28bc8940d305339 \
- --hash=sha256:ef5884711e3c617d7dc93519a7d038e242f5741cfe5fe9afd32d58845d86c562 \
- --hash=sha256:f49f51461de63f13e5dd4eb080421c8f23f856945f3f8bd5b2b1f59da52c2860
+packaging==26.3 \
+ --hash=sha256:94edc256424af38762eb31306eed28beb9f0efc50a8837492c9d6fd6004aed79 \
+ --hash=sha256:d7193f7c8e4e93f444fde0262bf90af30e16fa0ad0ad44cb553c87339b23cd1c
+pillow==12.3.0 \
+ --hash=sha256:00808c5e14ef63ac5161091d242999076604ff74b883423a11e5d7bbb38bf756 \
+ --hash=sha256:04f01d28a6aaff387bf842a13be313df23ba0597a44f1a976c9feb3c6ff4711a \
+ --hash=sha256:06ff022112bc9cbf83b60f8e028d94ad87b60621706487e65f673de61610ab59 \
+ --hash=sha256:0740a512dc522224c77d9aa5a8d70d8b7d73fb91f2c21125d8d025d3b8990e45 \
+ --hash=sha256:0847a763afefb695bc912d7c131e7e0632d4edc1d8698f58ddabec8e46b8b6d3 \
+ --hash=sha256:0dd2064cbc55aaec028ef5fbb60fa47bb6c3e7918e07ff17935284b227a9d2df \
+ --hash=sha256:0feb2e9d6ad6c9e3c06effe9d00f3f1e618a6643273576b016f591e9315a7139 \
+ --hash=sha256:10e41f0fbf1eec8cfd234b8fe17a4caac7c9d0db4c204d3c173a8f9f6ef3232b \
+ --hash=sha256:1182d52bc2d5e5d7d0949503aa7e36d12f42205dc287e4883f407b1988820d39 \
+ --hash=sha256:164b31cd1a0490ab6efae01aa5df49da7061be0af1b30e035b6e9a1bfe34ee6e \
+ --hash=sha256:1657923d2d45afb66526e5b933e5b3052e6bdea196c90d3abb2424e18c77dae8 \
+ --hash=sha256:186941b6aef820ad110fb01fb06eb925374dc3a21b17e37ec9a53b250c6fe2d1 \
+ --hash=sha256:1cca606cd25738df4ed873d5ad46bbdb3d83b5cbca291f6b4ff13a4df6b0bbe8 \
+ --hash=sha256:21900ce7ba264168cd50defae43cd75d25c833ad4ad6e73ffc5596d12e25ac89 \
+ --hash=sha256:236ff70b9312fb68943c703aa842ca6a758abfa45ac187a5e7c1452e96ef72b5 \
+ --hash=sha256:23aceaa007d6172b02c277f0cd359c79492bbb14f7072b4ede9fbcaf20648130 \
+ --hash=sha256:23d27a3e0307ec2244cc51e7287b919aa68d097504ebe19df4e76a98a3eea5bd \
+ --hash=sha256:24870b09b224f7ae3c39ed07d10e819d06f8720bc551847b1d623832b5b0e28d \
+ --hash=sha256:251bf95b67017e27b13d82f5b326234ca62d70f9cf4c2b9032de2358a3b12c7b \
+ --hash=sha256:25b9b82bb22e6e2b3cd07b39c68b7b862001226cb3dff7130d1cb914121b39ed \
+ --hash=sha256:28ce87c5ab450a9dd970b52e5aca5fe63ed432d18a2eaddd1979a00a1ba24ace \
+ --hash=sha256:300557495eb45ebb8aec96c2da9c4be642fbf7cd937278b4013ba894ea8eb0eb \
+ --hash=sha256:30f2aa603c41533cc25c05acd0da21636e84a315768feb631c937177db558931 \
+ --hash=sha256:331b624368d4f1d069149002f25f44bc61c8919ce8ddb3c45bdad8f6e2d89510 \
+ --hash=sha256:37d6d0a00072fd2948eb22bce7e1475f34569d90c87c59f7a2ec59541b77f7a6 \
+ --hash=sha256:37dc8f7bbb66efe481bb60defacef820c950c24713fb44962ed6aa2a50966de1 \
+ --hash=sha256:3b8182a766685eaa002637e28b4ec8d6b18819a0c71f579bf0dbaa5830297cce \
+ --hash=sha256:3edce1d53195db527e0191f84b71d02022de0540bf43a16ed734ed7537b07385 \
+ --hash=sha256:446c34dcc4324b084a53b705127dc15717b22c5e140ae0a3c38349d4efec071e \
+ --hash=sha256:4998562bf62a445225f22e07c896bb04b35b1b1f2eb6d760584c9c51d7a5f78c \
+ --hash=sha256:4b0a7fe987b14c31ebda6083f74f22b561fd3739bc0ac51e019622e3d72668c7 \
+ --hash=sha256:4e8c2a84d977f50b9daed6eeaf3baef67d00d5d74d932288f02cb94518ee3ace \
+ --hash=sha256:4f883547d4b7f0495ebe7056b0cc2aea76094e7a4abc8e933540f3271df27d9c \
+ --hash=sha256:514435a37670e3e5e08f3945b68718b6ed329bb84367777e16f9f4dfe1e61a0f \
+ --hash=sha256:53aa02d20d10c3d814d536aa4e5ac9b84ca0ff5a88377963b085ad6822f93e64 \
+ --hash=sha256:5594fc43d548a7ed94949d139aa1341b270f1863f11cfd37f5a6c8b778a6b67f \
+ --hash=sha256:571b9fcb07b97ef3a492028fb3d2dc0993ca23a06138b0315286566d29ef718a \
+ --hash=sha256:57b3d78c95ba9059768b10e28b813002261d3f3dfc55cc48b0c988f625175827 \
+ --hash=sha256:5afb51d599ea772b8365ae807ae557f18bccfe46ab261fd1c2a9ed700fc6eb17 \
+ --hash=sha256:6b02afb9b97f65fbca5f31db6a2a3ba21aa93030225f150fa3f249717e938fb4 \
+ --hash=sha256:6c0016e7b354317c4e9e525b937ac8596c38d2d232b419529b9cd7a1cd46e39a \
+ --hash=sha256:71d6097b330eea8fd15097780c8e89cb1a8ce7838669f48c5bacd6f663dd4701 \
+ --hash=sha256:756c768d0c9c2955feb7a56c37ea24aea2e369f8d36a88da270b6a9f19e62b5e \
+ --hash=sha256:78cb2c6865a35ab8ff8b75fd122f6033b92a62c82801110e48ddd6c936a45d91 \
+ --hash=sha256:7a743ff716f746fc19a9557f60dab1600d4613255f8a7aeb3cdde4db7eb15a66 \
+ --hash=sha256:85f998ea1848bc6757289e739cfbdda3a04adfd58b02fc018ce54d754a5ce468 \
+ --hash=sha256:8728f216dcdb6e6d555cf971cb34076139ad74b31fc2c14da4fafc741c5f6217 \
+ --hash=sha256:877c3f311ff35410f690861c4409e7ccbf0cd2f878e50628a28e5a0bb689e658 \
+ --hash=sha256:8cd2f7bdda092d99c9fc2fb7391354f306d01443d22785d0cbfafa2e2c8bb418 \
+ --hash=sha256:8e95e1385e4998ae9694eeaa4730ba5457ff61185b3a55e2e7bea0880aef452a \
+ --hash=sha256:962864dc93511324d51ddbb5b9f8731bf71675b93ca612a07441896f4688fb8c \
+ --hash=sha256:9cf95fe4d0f84c82d282745d9bb08ad9f926efa00be4697e767b814ce40d4330 \
+ --hash=sha256:9e881fca225083806662a5c43d627d215f258ff43c890f831966c7d7ba9c7402 \
+ --hash=sha256:a2b55dd6b2a4c4b7d87ffa56bdb33fdc5fdb9a462173861a7bc097f17d91cb09 \
+ --hash=sha256:a45650e8ce7fafffd731db8550230db6b0d306d181a90b67d3e6bca2f1990930 \
+ --hash=sha256:a876864214e136f0eb367788dbd7df045f4806801518e2cfe9e13229cfe06d8f \
+ --hash=sha256:ae26d61dfa7a47befdc7572b521024e8745f3d809bd95ca9505a7bba9ef849ec \
+ --hash=sha256:af8d94b0db561cf68b88a267c5c44b49e134f525d0dc2cb7ed413a66bc23559a \
+ --hash=sha256:b343699e8308bdc51978310e1c959c584e7869cc8c40780058c87da7781a1e94 \
+ --hash=sha256:b3c777e849237620b022f7f297dd67705f9f5cf1685f09f02e46f93e92725468 \
+ --hash=sha256:b629de27fda84b42cde7edef0d85f13b958b47f6e9bbcbba9b673c562a89bd8b \
+ --hash=sha256:ba09209fbe443b4acccebe845d8a138b89a8f4fbaeedd44953490b5315d5e965 \
+ --hash=sha256:ba54cfebe86920a559a7c4d6b9050791c20513650a1952ebe3368c7dc70306f8 \
+ --hash=sha256:bcb46e2f9feff8d06323983bd83ed00c201fdcab3d74973e7072a889b3979fcd \
+ --hash=sha256:bcc33feacfaefce60c12fd500a277533bdc02b10a19f7f6d348763d8140bbba7 \
+ --hash=sha256:bf16ba1b4d0b6b7c8e534936632270cf70eb00dbe09005bc345b2677b726855c \
+ --hash=sha256:cf1845d02ad822a369a49f2bb9345b1614744267682e7a03527dc3bf6eea1777 \
+ --hash=sha256:d69141514cc30b774ceea5e3ed3a6635c8d8a96edf664689b890f4089111fb35 \
+ --hash=sha256:d9c7f76c0673154f044e9d78c8655fb4213f6ca31a836df48b40fe5d187717b9 \
+ --hash=sha256:dbce0b29841537a2fa4a214c2bbf14de3587c9680caa9b4e217568472490b28f \
+ --hash=sha256:dc624f6bc473dacdf7ef7eb8678d0d08edf15cd94fad6ae5c7d6cc67a4e4902f \
+ --hash=sha256:e158cb00350dc278f3b91551101aa7d12415a66ebf2c91d8d5ac14e56ddd3ad0 \
+ --hash=sha256:e491916b378fba47242221bb9ead245211b70d504f495d105d17b14a24b4907c \
+ --hash=sha256:e795b7eb908249c4e43c7c99fac7c2c75dab0c43566e37db472a355f63693d71 \
+ --hash=sha256:e7e480451b9fa137494bccd3a7d69adbe8ac65a87d97be61e11f1b1050a5bac3 \
+ --hash=sha256:e91206ee562682b51b98ef4b26a6ef48fd84e15fd4c4bc5ec768eb641d206838 \
+ --hash=sha256:e9871b1ffbfa9656b60aeee92ed5136a5742696006fa322b29ea3d8da0ecc9cf \
+ --hash=sha256:e9aeb04d6aef139de265b29683e119b638208f88cf73cdd1658aa07221165321 \
+ --hash=sha256:ebaea975e03d3141d9d3a507df75c9b3ec90fa9d2ffd07567b3a978d9d790b26 \
+ --hash=sha256:f0606c8bf2cdefea14a43530f7657cbbb7ecf1c4222512492ef4a4434a9501ec \
+ --hash=sha256:f13c32a3abd6079a66d9526e18dad9b6d280384d49d7c54040cd57b6424041d9 \
+ --hash=sha256:f7401aebd7f581d7f83a439d87d474999317ee099218e5ad25d125290990ba65 \
+ --hash=sha256:fa4ecea169a355be7a3ade2c783e2ed12f0e40d2c5621cda8b3297faf7fbb9f5 \
+ --hash=sha256:fbd139c8447d25dd750ab79ee274cc5e1fe80fc56340ab10b18a195e1b6eca3e \
+ --hash=sha256:fdafc9cce40277e0f7a0feabce0ee50dd2fa1800f3b38015e51296b5e814048d \
+ --hash=sha256:fe3cca2e4e8a592be0f269a1ca4835c25199d9f3ce815c8491048f785b0a0198 \
+ --hash=sha256:ffd0c5368496f41b0944be820fcb7a838aa6e623d250b01acf2643939c3f99d7
+polars==1.44.2 \
+ --hash=sha256:1bb331f17a40d9d931101533dcd33637b66edc61eb377b07020dac16a0f0377b \
+ --hash=sha256:86c8e26b6c2de8c8d344bb910b74dfc47b118ac3fe0f19b44909467990a0b281
+polars-runtime-32==1.44.2 \
+ --hash=sha256:10c0c695a418407617b5159db7d9a21074a733e4c6d61275b6762f25cb31ca99 \
+ --hash=sha256:1fd536720668ba203a16a20b08cd6b23057e407a0279cf36b2f35f879d6e3208 \
+ --hash=sha256:8598e7a20efba70bb74978c7df7af7c606ff4d79b9b48fdd808250b189bc9a13 \
+ --hash=sha256:a1bafb441e99199a62c63bf1bbdc0ea09ee9776dbac2bf31452b5000fb1df2f7 \
+ --hash=sha256:b84842f7d621aaca7a52e165e19a24f89db45f8aa13744941430218419a14a67 \
+ --hash=sha256:bbf9b45040291dc1c6c588c837019c33557bde25ec536562a9cca9e1f6dfcc45 \
+ --hash=sha256:c4a09fb14aad711526346efc0cb2015c2fd0555ce4118b6524e5debbaea65ff5 \
+ --hash=sha256:d51040d3ab40157f6db3c62be59cab5b80fb3c8d158924769c4982a1c8eef730 \
+ --hash=sha256:e0fd43720c8222ae39919c8ff891636d53b352706087120e62f83544dd3ff782
protobuf==5.29.6 \
--hash=sha256:36ade6ff88212e91aef4e687a971a11d7d24d6948a66751abc1b3238648f5d05 \
--hash=sha256:62e8a3114992c7c647bce37dcc93647575fc52d50e48de30c6fcb28a6a291eb1 \
@@ -1065,56 +1138,9 @@ pyyaml==6.0.3 \
--hash=sha256:f7057c9a337546edc7973c0d3ba84ddcdf0daa14533c2065749c9075001090e6 \
--hash=sha256:fa160448684b4e94d80416c0fa4aac48967a969efe22931448d853ada8baf926 \
--hash=sha256:fc09d0aa354569bc501d4e787133afc08552722d3ab34836a80547331bb5d4a0
-requests==2.33.1 \
- --hash=sha256:18817f8c57c6263968bc123d237e3b8b08ac046f5456bd1e307ee8f4250d3517 \
- --hash=sha256:4e6d1ef462f3626a1f0a0a9c42dd93c63bad33f9f1c1937509b8c5c8718ab56a
-scipy==1.15.3 \
- --hash=sha256:05dc6abcd105e1a29f95eada46d4a3f251743cfd7d3ae8ddb4088047f24ea477 \
- --hash=sha256:06efcba926324df1696931a57a176c80848ccd67ce6ad020c810736bfd58eb1c \
- --hash=sha256:0a769105537aa07a69468a0eefcd121be52006db61cdd8cac8a0e68980bbb723 \
- --hash=sha256:0bdd905264c0c9cfa74a4772cdb2070171790381a5c4d312c973382fc6eaf730 \
- --hash=sha256:0ff17c0bb1cb32952c09217d8d1eed9b53d1463e5f1dd6052c7857f83127d539 \
- --hash=sha256:14ed70039d182f411ffc74789a16df3835e05dc469b898233a245cdfd7f162cb \
- --hash=sha256:185cd3d6d05ca4b44a8f1595af87f9c372bb6acf9c808e99aa3e9aa03bd98cf6 \
- --hash=sha256:18aaacb735ab38b38db42cb01f6b92a2d0d4b6aabefeb07f02849e47f8fb3594 \
- --hash=sha256:1c832e1bd78dea67d5c16f786681b28dd695a8cb1fb90af2e27580d3d0967e92 \
- --hash=sha256:263961f658ce2165bbd7b99fa5135195c3a12d9bef045345016b8b50c315cb82 \
- --hash=sha256:271e3713e645149ea5ea3e97b57fdab61ce61333f97cfae392c28ba786f9bb49 \
- --hash=sha256:2c620736bcc334782e24d173c0fdbb7590a0a436d2fdf39310a8902505008759 \
- --hash=sha256:34716e281f181a02341ddeaad584205bd2fd3c242063bd3423d61ac259ca7eba \
- --hash=sha256:39cb9c62e471b1bb3750066ecc3a3f3052b37751c7c3dfd0fd7e48900ed52982 \
- --hash=sha256:3ac07623267feb3ae308487c260ac684b32ea35fd81e12845039952f558047b8 \
- --hash=sha256:3b0334816afb8b91dab859281b1b9786934392aa3d527cd847e41bb6f45bee65 \
- --hash=sha256:40e54d5c7e7ebf1aa596c374c49fa3135f04648a0caabcb66c52884b943f02b4 \
- --hash=sha256:50f9e62461c95d933d5c5ef4a1f2ebf9a2b4e83b0db374cb3f1de104d935922e \
- --hash=sha256:52092bc0472cfd17df49ff17e70624345efece4e1a12b23783a1ac59a1b728ed \
- --hash=sha256:5380741e53df2c566f4d234b100a484b420af85deb39ea35a1cc1be84ff53a5c \
- --hash=sha256:5e721fed53187e71d0ccf382b6bf977644c533e506c4d33c3fb24de89f5c3ed5 \
- --hash=sha256:6487aa99c2a3d509a5227d9a5e889ff05830a06b2ce08ec30df6d79db5fcd5c5 \
- --hash=sha256:6ac6310fdbfb7aa6612408bd2f07295bcbd3fda00d2d702178434751fe48e019 \
- --hash=sha256:6cfd56fc1a8e53f6e89ba3a7a7251f7396412d655bca2aa5611c8ec9a6784a1e \
- --hash=sha256:6db907c7368e3092e24919b5e31c76998b0ce1684d51a90943cb0ed1b4ffd6c1 \
- --hash=sha256:721d6b4ef5dc82ca8968c25b111e307083d7ca9091bc38163fb89243e85e3889 \
- --hash=sha256:76ad1fb5f8752eabf0fa02e4cc0336b4e8f021e2d5f061ed37d6d264db35e3ca \
- --hash=sha256:79167bba085c31f38603e11a267d862957cbb3ce018d8b38f79ac043bc92d825 \
- --hash=sha256:795c46999bae845966368a3c013e0e00947932d68e235702b5c3f6ea799aa8c9 \
- --hash=sha256:7e11270a000969409d37ed399585ee530b9ef6aa99d50c019de4cb01e8e54e62 \
- --hash=sha256:8c9ed3ba2c8a2ce098163a9bdb26f891746d02136995df25227a20e71c396ebb \
- --hash=sha256:993439ce220d25e3696d1b23b233dd010169b62f6456488567e830654ee37a6b \
- --hash=sha256:9d61e97b186a57350f6d6fd72640f9e99d5a4a2b8fbf4b9ee9a841eab327dc13 \
- --hash=sha256:9db984639887e3dffb3928d118145ffe40eff2fa40cb241a306ec57c219ebbbb \
- --hash=sha256:9e2abc762b0811e09a0d3258abee2d98e0c703eee49464ce0069590846f31d40 \
- --hash=sha256:a345928c86d535060c9c2b25e71e87c39ab2f22fc96e9636bd74d1dbf9de448c \
- --hash=sha256:ad3432cb0f9ed87477a8d97f03b763fd1d57709f1bbde3c9369b1dff5503b253 \
- --hash=sha256:ae48a786a28412d744c62fd7816a4118ef97e5be0bee968ce8f0a2fba7acf3bb \
- --hash=sha256:aef683a9ae6eb00728a542b796f52a5477b78252edede72b8327a886ab63293f \
- --hash=sha256:b90ab29d0c37ec9bf55424c064312930ca5f4bde15ee8619ee44e69319aab163 \
- --hash=sha256:c05045d8b9bfd807ee1b9f38761993297b10b245f012b11b13b91ba8945f7e45 \
- --hash=sha256:c9deabd6d547aee2c9a81dee6cc96c6d7e9a9b1953f74850c179f91fdc729cb7 \
- --hash=sha256:dde4fc32993071ac0c7dd2d82569e544f0bdaff66269cb475e0f369adad13f11 \
- --hash=sha256:eae3cf522bc7df64b42cad3925c876e1b0b6c35c1337c93e12c0f366f55b0eaf \
- --hash=sha256:ed7284b21a7a0c8f1b6e5977ac05396c0d008b89e05498c8b7e8f4a1423bba0e \
- --hash=sha256:f77f853d584e72e874d87357ad70f44b437331507d1c311457bed8ed2b956126
+requests==2.34.2 \
+ --hash=sha256:2a0d60c172f83ac6ab31e4554906c0f3b3588d37b5cb939b1c061f4907e278e0 \
+ --hash=sha256:f288924cae4e29463698d6d60bc6a4da69c89185ad1e0bcc4104f584e960b9ed
six==1.17.0 \
--hash=sha256:4721f391ed90541fddacab5acf947aa0d3dc7d27b2e1e8eda2be8970586c3274 \
--hash=sha256:ff70335d468e7eb6ec65b95b99d3a2836546063f63acc5171de367e834932a81
@@ -1222,20 +1248,20 @@ triton==3.6.0 \
--hash=sha256:d002e07d7180fd65e622134fbd980c9a3d4211fb85224b56a0a0efbd422ab72f \
--hash=sha256:e8e323d608e3a9bfcc2d9efcc90ceefb764a82b99dea12a86d643c72539ad5d3 \
--hash=sha256:ef5523241e7d1abca00f1d240949eebdd7c673b005edbbce0aca95b8191f1d43
-typing-extensions==4.15.0 \
- --hash=sha256:0cea48d173cc12fa28ecabc3b837ea3cf6f38c6d1136f85cbaaf598984861466 \
- --hash=sha256:f0fa19c6845758ab08074a0cfa8b7aecb71c999ca73d62883bc25cc018c4e548
-ultralytics==8.4.34 \
- --hash=sha256:e932da39976879ddf4f398993bd44587c1f5aafaac138efab38d34030680c113 \
- --hash=sha256:fe94bf945c2c69b4df245ee9327dc3c7ce6d06ec9c647843331b0d497a48db78
-ultralytics-thop==2.0.18 \
- --hash=sha256:21103bcd39cc9928477dc3d9374561749b66a1781b35f46256c8d8c4ac01d9cf \
- --hash=sha256:2bb44851ad224b116c3995b02dd5e474a5ccf00acf237fe0edb9e1506ede04ec
+typing-extensions==4.16.0 \
+ --hash=sha256:481caa481374e813c1b176ada14e97f1f67a4539ce9cfeb3f350d78d6370c2e8 \
+ --hash=sha256:dc983d19a509c94dba722ee6abd33940f7c05a89e243c47e907eb4db6f1a43e5
+ultralytics==8.4.146 \
+ --hash=sha256:5c3618f4c5f7b89ad6afdf789bf4d90645054f4eb78c211aa7eed1e186fa65fc \
+ --hash=sha256:73e826b090d35af5a54130be1875d648146d689cf281313dd1a15bc85c557051
+ultralytics-thop==2.1.6 \
+ --hash=sha256:0ec2df8ebd3db35795e1f80cdc8bce6734446dbe989bca1b0c89396353f0f08c \
+ --hash=sha256:23f7b8ad124fa3432c1a7de9279102c4fdda699216032a7dff49f87ec3d1a3af
urllib3==2.7.0 \
--hash=sha256:231e0ec3b63ceb14667c67be60f2f2c40a518cb38b03af60abc813da26505f4c \
--hash=sha256:9fb4c81ebbb1ce9531cce37674bbc6f1360472bc18ca9a553ede278ef7276897
# The following packages are considered to be unsafe in a requirements file:
-pip==26.1.2 \
- --hash=sha256:382ff9f685ee3bc25864f820aa50505825f10f5458ffff07e30a6d96e5715cab \
- --hash=sha256:f49cd134c61cf2fd75e0ce2676db03e4054504a5a4986d00f8299ae632dc4605
+pip==26.2.1 \
+ --hash=sha256:71138adf1f4ca900cdb7d289c21b7494329f2332b6d85f0e1c42108c0384ed3e \
+ --hash=sha256:f6ad667e89a1fe78046c8f13232b247200f5258d7828f3f7883d660878e0813f
diff --git a/frontend/requirements.txt b/frontend/requirements.txt
index 0919578..cfce0a9 100644
--- a/frontend/requirements.txt
+++ b/frontend/requirements.txt
@@ -1,123 +1,182 @@
-certifi==2026.1.4 \
- --hash=sha256:9943707519e4add1115f44c2bc244f782c0249876bf51b6599fee1ffbedd685c \
- --hash=sha256:ac726dd470482006e014ad384921ed6438c457018f4b3d204aea4281258b2120
-charset-normalizer==3.4.4 \
- --hash=sha256:027f6de494925c0ab2a55eab46ae5129951638a49a34d87f4c3eda90f696b4ad \
- --hash=sha256:077fbb858e903c73f6c9db43374fd213b0b6a778106bc7032446a8e8b5b38b93 \
- --hash=sha256:0a98e6759f854bd25a58a73fa88833fba3b7c491169f86ce1180c948ab3fd394 \
- --hash=sha256:0d3d8f15c07f86e9ff82319b3d9ef6f4bf907608f53fe9d92b28ea9ae3d1fd89 \
- --hash=sha256:0f04b14ffe5fdc8c4933862d8306109a2c51e0704acfa35d51598eb45a1e89fc \
- --hash=sha256:11d694519d7f29d6cd09f6ac70028dba10f92f6cdd059096db198c283794ac86 \
- --hash=sha256:194f08cbb32dc406d6e1aea671a68be0823673db2832b38405deba2fb0d88f63 \
- --hash=sha256:1bee1e43c28aa63cb16e5c14e582580546b08e535299b8b6158a7c9c768a1f3d \
- --hash=sha256:21d142cc6c0ec30d2efee5068ca36c128a30b0f2c53c1c07bd78cb6bc1d3be5f \
- --hash=sha256:2437418e20515acec67d86e12bf70056a33abdacb5cb1655042f6538d6b085a8 \
- --hash=sha256:244bfb999c71b35de57821b8ea746b24e863398194a4014e4c76adc2bbdfeff0 \
- --hash=sha256:2677acec1a2f8ef614c6888b5b4ae4060cc184174a938ed4e8ef690e15d3e505 \
- --hash=sha256:277e970e750505ed74c832b4bf75dac7476262ee2a013f5574dd49075879e161 \
- --hash=sha256:2aaba3b0819274cc41757a1da876f810a3e4d7b6eb25699253a4effef9e8e4af \
- --hash=sha256:2b7d8f6c26245217bd2ad053761201e9f9680f8ce52f0fcd8d0755aeae5b2152 \
- --hash=sha256:2c9d3c380143a1fedbff95a312aa798578371eb29da42106a29019368a475318 \
- --hash=sha256:3162d5d8ce1bb98dd51af660f2121c55d0fa541b46dff7bb9b9f86ea1d87de72 \
- --hash=sha256:31fd66405eaf47bb62e8cd575dc621c56c668f27d46a61d975a249930dd5e2a4 \
- --hash=sha256:362d61fd13843997c1c446760ef36f240cf81d3ebf74ac62652aebaf7838561e \
- --hash=sha256:376bec83a63b8021bb5c8ea75e21c4ccb86e7e45ca4eb81146091b56599b80c3 \
- --hash=sha256:44c2a8734b333e0578090c4cd6b16f275e07aa6614ca8715e6c038e865e70576 \
- --hash=sha256:47cc91b2f4dd2833fddaedd2893006b0106129d4b94fdb6af1f4ce5a9965577c \
- --hash=sha256:4902828217069c3c5c71094537a8e623f5d097858ac6ca8252f7b4d10b7560f1 \
- --hash=sha256:4bd5d4137d500351a30687c2d3971758aac9a19208fc110ccb9d7188fbe709e8 \
- --hash=sha256:4fe7859a4e3e8457458e2ff592f15ccb02f3da787fcd31e0183879c3ad4692a1 \
- --hash=sha256:542d2cee80be6f80247095cc36c418f7bddd14f4a6de45af91dfad36d817bba2 \
- --hash=sha256:554af85e960429cf30784dd47447d5125aaa3b99a6f0683589dbd27e2f45da44 \
- --hash=sha256:5833d2c39d8896e4e19b689ffc198f08ea58116bee26dea51e362ecc7cd3ed26 \
- --hash=sha256:5947809c8a2417be3267efc979c47d76a079758166f7d43ef5ae8e9f92751f88 \
- --hash=sha256:5ae497466c7901d54b639cf42d5b8c1b6a4fead55215500d2f486d34db48d016 \
- --hash=sha256:5bd2293095d766545ec1a8f612559f6b40abc0eb18bb2f5d1171872d34036ede \
- --hash=sha256:5bfbb1b9acf3334612667b61bd3002196fe2a1eb4dd74d247e0f2a4d50ec9bbf \
- --hash=sha256:5cb4d72eea50c8868f5288b7f7f33ed276118325c1dfd3957089f6b519e1382a \
- --hash=sha256:5dbe56a36425d26d6cfb40ce79c314a2e4dd6211d51d6d2191c00bed34f354cc \
- --hash=sha256:5f819d5fe9234f9f82d75bdfa9aef3a3d72c4d24a6e57aeaebba32a704553aa0 \
- --hash=sha256:64b55f9dce520635f018f907ff1b0df1fdc31f2795a922fb49dd14fbcdf48c84 \
- --hash=sha256:6515f3182dbe4ea06ced2d9e8666d97b46ef4c75e326b79bb624110f122551db \
- --hash=sha256:65e2befcd84bc6f37095f5961e68a6f077bf44946771354a28ad434c2cce0ae1 \
- --hash=sha256:6aee717dcfead04c6eb1ce3bd29ac1e22663cdea57f943c87d1eab9a025438d7 \
- --hash=sha256:6b39f987ae8ccdf0d2642338faf2abb1862340facc796048b604ef14919e55ed \
- --hash=sha256:6e1fcf0720908f200cd21aa4e6750a48ff6ce4afe7ff5a79a90d5ed8a08296f8 \
- --hash=sha256:74018750915ee7ad843a774364e13a3db91682f26142baddf775342c3f5b1133 \
- --hash=sha256:74664978bb272435107de04e36db5a9735e78232b85b77d45cfb38f758efd33e \
- --hash=sha256:74bb723680f9f7a6234dcf67aea57e708ec1fbdf5699fb91dfd6f511b0a320ef \
- --hash=sha256:752944c7ffbfdd10c074dc58ec2d5a8a4cd9493b314d367c14d24c17684ddd14 \
- --hash=sha256:778d2e08eda00f4256d7f672ca9fef386071c9202f5e4607920b86d7803387f2 \
- --hash=sha256:780236ac706e66881f3b7f2f32dfe90507a09e67d1d454c762cf642e6e1586e0 \
- --hash=sha256:798d75d81754988d2565bff1b97ba5a44411867c0cf32b77a7e8f8d84796b10d \
- --hash=sha256:799a7a5e4fb2d5898c60b640fd4981d6a25f1c11790935a44ce38c54e985f828 \
- --hash=sha256:7a32c560861a02ff789ad905a2fe94e3f840803362c84fecf1851cb4cf3dc37f \
- --hash=sha256:7c308f7e26e4363d79df40ca5b2be1c6ba9f02bdbccfed5abddb7859a6ce72cf \
- --hash=sha256:7fa17817dc5625de8a027cb8b26d9fefa3ea28c8253929b8d6649e705d2835b6 \
- --hash=sha256:81d5eb2a312700f4ecaa977a8235b634ce853200e828fbadf3a9c50bab278328 \
- --hash=sha256:82004af6c302b5d3ab2cfc4cc5f29db16123b1a8417f2e25f9066f91d4411090 \
- --hash=sha256:837c2ce8c5a65a2035be9b3569c684358dfbf109fd3b6969630a87535495ceaa \
- --hash=sha256:840c25fb618a231545cbab0564a799f101b63b9901f2569faecd6b222ac72381 \
- --hash=sha256:8a6562c3700cce886c5be75ade4a5db4214fda19fede41d9792d100288d8f94c \
- --hash=sha256:8af65f14dc14a79b924524b1e7fffe304517b2bff5a58bf64f30b98bbc5079eb \
- --hash=sha256:8ef3c867360f88ac904fd3f5e1f902f13307af9052646963ee08ff4f131adafc \
- --hash=sha256:94537985111c35f28720e43603b8e7b43a6ecfb2ce1d3058bbe955b73404e21a \
- --hash=sha256:99ae2cffebb06e6c22bdc25801d7b30f503cc87dbd283479e7b606f70aff57ec \
- --hash=sha256:9a26f18905b8dd5d685d6d07b0cdf98a79f3c7a918906af7cc143ea2e164c8bc \
- --hash=sha256:9b35f4c90079ff2e2edc5b26c0c77925e5d2d255c42c74fdb70fb49b172726ac \
- --hash=sha256:9cd98cdc06614a2f768d2b7286d66805f94c48cde050acdbbb7db2600ab3197e \
- --hash=sha256:9d1bb833febdff5c8927f922386db610b49db6e0d4f4ee29601d71e7c2694313 \
- --hash=sha256:9f7fcd74d410a36883701fafa2482a6af2ff5ba96b9a620e9e0721e28ead5569 \
- --hash=sha256:a59cb51917aa591b1c4e6a43c132f0cdc3c76dbad6155df4e28ee626cc77a0a3 \
- --hash=sha256:a61900df84c667873b292c3de315a786dd8dac506704dea57bc957bd31e22c7d \
- --hash=sha256:a79cfe37875f822425b89a82333404539ae63dbdddf97f84dcbc3d339aae9525 \
- --hash=sha256:a8a8b89589086a25749f471e6a900d3f662d1d3b6e2e59dcecf787b1cc3a1894 \
- --hash=sha256:a8bf8d0f749c5757af2142fe7903a9df1d2e8aa3841559b2bad34b08d0e2bcf3 \
- --hash=sha256:a9768c477b9d7bd54bc0c86dbaebdec6f03306675526c9927c0e8a04e8f94af9 \
- --hash=sha256:ac1c4a689edcc530fc9d9aa11f5774b9e2f33f9a0c6a57864e90908f5208d30a \
- --hash=sha256:af2d8c67d8e573d6de5bc30cdb27e9b95e49115cd9baad5ddbd1a6207aaa82a9 \
- --hash=sha256:b435cba5f4f750aa6c0a0d92c541fb79f69a387c91e61f1795227e4ed9cece14 \
- --hash=sha256:b5b290ccc2a263e8d185130284f8501e3e36c5e02750fc6b6bdeb2e9e96f1e25 \
- --hash=sha256:b5d84d37db046c5ca74ee7bb47dd6cbc13f80665fdde3e8040bdd3fb015ecb50 \
- --hash=sha256:b7cf1017d601aa35e6bb650b6ad28652c9cd78ee6caff19f3c28d03e1c80acbf \
- --hash=sha256:bc7637e2f80d8530ee4a78e878bce464f70087ce73cf7c1caf142416923b98f1 \
- --hash=sha256:c0463276121fdee9c49b98908b3a89c39be45d86d1dbaa22957e38f6321d4ce3 \
- --hash=sha256:c4ef880e27901b6cc782f1b95f82da9313c0eb95c3af699103088fa0ac3ce9ac \
- --hash=sha256:c8ae8a0f02f57a6e61203a31428fa1d677cbe50c93622b4149d5c0f319c1d19e \
- --hash=sha256:ca5862d5b3928c4940729dacc329aa9102900382fea192fc5e52eb69d6093815 \
- --hash=sha256:cb01158d8b88ee68f15949894ccc6712278243d95f344770fa7593fa2d94410c \
- --hash=sha256:cb6254dc36b47a990e59e1068afacdcd02958bdcce30bb50cc1700a8b9d624a6 \
- --hash=sha256:cc00f04ed596e9dc0da42ed17ac5e596c6ccba999ba6bd92b0e0aef2f170f2d6 \
- --hash=sha256:cd09d08005f958f370f539f186d10aec3377d55b9eeb0d796025d4886119d76e \
- --hash=sha256:cd4b7ca9984e5e7985c12bc60a6f173f3c958eae74f3ef6624bb6b26e2abbae4 \
- --hash=sha256:ce8a0633f41a967713a59c4139d29110c07e826d131a316b50ce11b1d79b4f84 \
- --hash=sha256:cead0978fc57397645f12578bfd2d5ea9138ea0fac82b2f63f7f7c6877986a69 \
- --hash=sha256:d055ec1e26e441f6187acf818b73564e6e6282709e9bcb5b63f5b23068356a15 \
- --hash=sha256:d1f13550535ad8cff21b8d757a3257963e951d96e20ec82ab44bc64aeb62a191 \
- --hash=sha256:d9c7f57c3d666a53421049053eaacdd14bbd0a528e2186fcb2e672effd053bb0 \
- --hash=sha256:d9e45d7faa48ee908174d8fe84854479ef838fc6a705c9315372eacbc2f02897 \
- --hash=sha256:da3326d9e65ef63a817ecbcc0df6e94463713b754fe293eaa03da99befb9a5bd \
- --hash=sha256:de00632ca48df9daf77a2c65a484531649261ec9f25489917f09e455cb09ddb2 \
- --hash=sha256:e1f185f86a6f3403aa2420e815904c67b2f9ebc443f045edd0de921108345794 \
- --hash=sha256:e824f1492727fa856dd6eda4f7cee25f8518a12f3c4a56a74e8095695089cf6d \
- --hash=sha256:e912091979546adf63357d7e2ccff9b44f026c075aeaf25a52d0e95ad2281074 \
- --hash=sha256:eaabd426fe94daf8fd157c32e571c85cb12e66692f15516a83a03264b08d06c3 \
- --hash=sha256:ebf3e58c7ec8a8bed6d66a75d7fb37b55e5015b03ceae72a8e7c74495551e224 \
- --hash=sha256:ecaae4149d99b1c9e7b88bb03e3221956f68fd6d50be2ef061b2381b61d20838 \
- --hash=sha256:eecbc200c7fd5ddb9a7f16c7decb07b566c29fa2161a16cf67b8d068bd21690a \
- --hash=sha256:f155a433c2ec037d4e8df17d18922c3a0d9b3232a396690f17175d2946f0218d \
- --hash=sha256:f1e34719c6ed0b92f418c7c780480b26b5d9c50349e9a9af7d76bf757530350d \
- --hash=sha256:f34be2938726fc13801220747472850852fe6b1ea75869a048d6f896838c896f \
- --hash=sha256:f820802628d2694cb7e56db99213f930856014862f3fd943d290ea8438d07ca8 \
- --hash=sha256:f8bf04158c6b607d747e93949aa60618b61312fe647a6369f88ce2ff16043490 \
- --hash=sha256:f8e160feb2aed042cd657a72acc0b481212ed28b1b9a95c0cee1621b524e1966 \
- --hash=sha256:f9d332f8c2a2fcbffe1378594431458ddbef721c1769d78e2cbc06280d8155f9 \
- --hash=sha256:fa09f53c465e532f4d3db095e0c55b615f010ad81803d383195b6b5ca6cbf5f3 \
- --hash=sha256:faa3a41b2b66b6e50f84ae4a68c64fcd0c44355741c6374813a800cd6695db9e \
- --hash=sha256:fd44c878ea55ba351104cb93cc85e74916eb8fa440ca7903e57575e97394f608
-idna==3.18 \
- --hash=sha256:7f952cbe720b688055e3f87de14f5c3e5fdaa8bc3928985c4077ca689de849a2 \
- --hash=sha256:ffb385a7e039654cef1ab9ef32c6fafe283c0c0467bba1d9029738ce4a14a848
+certifi==2026.7.22 \
+ --hash=sha256:62f22742b58a1a33014a2b6b706588a8d7e2a88ae7bd1a6ebe8c992928483775 \
+ --hash=sha256:741e2c3b351ddf169a738da9f2c048608ff7f2c5cc02f1ebc6b118bb090d5d55
+charset-normalizer==3.5.1 \
+ --hash=sha256:00668ebb0609751758682eb0b5857e7c35b9f00e84dfdef062e103244ec94d45 \
+ --hash=sha256:012a22b88a77ca2e59b98ac5889b0deb604147666032f45e6d6e217634d2550d \
+ --hash=sha256:01e93745f7f219b703b60ba7afead36cfc4242782be5af484673fc500df12da5 \
+ --hash=sha256:04368edf83514385ffc3e1cfd4546e595f4f1272dd23ba437a93a9cc3741d47b \
+ --hash=sha256:0722590aabf9dc6a6c0343d523c05458fa2b5047dbe6302fd526bb570600753f \
+ --hash=sha256:07ffd07412fc5d5e84cd8952acf9ff7e4ed7a708e69d1bada19d8ba91711353f \
+ --hash=sha256:09a7bba9f739468c8e78c36a75c33768e53cb1959fc638f510454c14683f00d5 \
+ --hash=sha256:0b2b1b3fa5670c127b246df1d0c059defd41f689a868a3b9d79df9b1cac42d22 \
+ --hash=sha256:0c6dfb5ca6723eeed15aa8e564a014d69fcb8812f94eef11fe3631e0508199f5 \
+ --hash=sha256:0d929fc574b4d6fd9e7c0f5c2ede8716a41911923aa7fa5fce38e0818aa4a1ac \
+ --hash=sha256:13e3afe97712e8887cd516e960c63f0b93122971e5b5e4b2622fe7701771e838 \
+ --hash=sha256:15f024313246a4ed976c60f440bb8d257815513a681d212ff74fd46f7d715a90 \
+ --hash=sha256:195ce897c6153c0700078142cf8efe3e6454ca4cf4357499e4078dfd83396626 \
+ --hash=sha256:19a3dd5aa73cef1c99687c4fc57db016a9c17104ae1185da88ba566a5d3bebe4 \
+ --hash=sha256:1d1c7a53a6c2103925cdd6d7229f8c567379f211c869793df679f2e9f738c369 \
+ --hash=sha256:1f5883d77fd409a261abb5dc8ccbe335720d798b1de4abb3b1d47ccbbc76b53b \
+ --hash=sha256:21b82d8082f6f5e7f456ef0bd16323d08de1266efbfeb476e64b2a91d1471a4e \
+ --hash=sha256:252d099029bcbea642f2a06c4ed5046bdf8b5a8150b64afa5e027e88b106e5ee \
+ --hash=sha256:256dd4d85d9e4dc595e2bc983c980e73f62ddeb3165c58b4c3dfe78c5c8548c1 \
+ --hash=sha256:26422d45fd13551cf564c58932f7d72b4f58b93b0fcf18c35ba6be12b46bb102 \
+ --hash=sha256:2679de311c7946dde5d3b6f44941844133ff5c7cb86099c0061ab1e8901c20a8 \
+ --hash=sha256:29880d17a8eb0b5cfdfd8944b468322928059aa35f1f5fa8ff22b149ec0b42f8 \
+ --hash=sha256:2bced4061f000f7187254a02ad3433ae17eaf991747ceea2f478422590a5bba9 \
+ --hash=sha256:2e9cf9253119d8e5d111f05d71626786fd3d6193817316eab1ca088cdb8593cf \
+ --hash=sha256:2f06b7eae9dbe77fe1d644ca244dad508de8d302870a43f3c559b521270938a0 \
+ --hash=sha256:2f293479cce755c75f1697e87c409b7ae4c555c7dfecb6e988ad13abba943031 \
+ --hash=sha256:329fc3ccb63ad22d867d84c2adea759a64079a37ba4a343433b02c7a2816871e \
+ --hash=sha256:343fb4f2821043bd87095f7b08a1a181febc8e36ac64212143bbfd0a0e1bc235 \
+ --hash=sha256:3588e376b3ea2eea84976f67273d679f229e24c66dce7b82ae45aef04ff6e072 \
+ --hash=sha256:35aea775dc2bd5f54cd84a1cd2696cc3207c479cb9cf0bd346f0d343e4300ddb \
+ --hash=sha256:35fe081843b35aad20ffeccec3eeffbe637b15d14f3fb22cc1b59cd8ec17e93c \
+ --hash=sha256:36047af20e17097c3bb9476c2b7655f2f7aa51322c0ba58c07695bedf755a950 \
+ --hash=sha256:3617ac3cfd8b9888f145ad89dd6e692285834b0201c6074a5eeaad3fd4d668c2 \
+ --hash=sha256:366ec70f5547c640d3ce1985722490f23faf4eb5216a7eeba78277490e78dacb \
+ --hash=sha256:394fea06235c8543390050ed5f529187074b029fb027213f6c46ac11ab5d950e \
+ --hash=sha256:3d27167433c0d5f18dc850f07d0b3816221984fecdc405d6c157a6f0b8f8e9e6 \
+ --hash=sha256:3e5e1224c0a6a90e05843e07adfec669edebec17801c67072f51e59561d63c0b \
+ --hash=sha256:41876ee62a3dddf48ff1121ad8f0798032aa03f2fd35f21f34a4cab14f18d8d2 \
+ --hash=sha256:433c5a81eade63b47e522303bad236f59dba55ea6951746f5558355eeed8c75d \
+ --hash=sha256:4582c27e8c889d64811987b5967fbd3ae0c823fe1fd933b543d55ac20bb475fa \
+ --hash=sha256:485a0d363cafefcd2538a73c7c838daa2035f09b2c9f9b5e3133f80c6aeb84c2 \
+ --hash=sha256:494b70049a4d69aec6e8137c13af4cf8db8c9f9820a1392ac293b0dd2987a818 \
+ --hash=sha256:496846868fea80e479324862fa877f02411f2fd0f83b79ccee2607aa68b2a032 \
+ --hash=sha256:4abdc5f9ad448c1ecbfae2974b820535d6bc6e7eef63babbab3d81cf46968c71 \
+ --hash=sha256:4b599739b93b2cbeded49645ae3c8d1405c29ddfbceac1545c87a3f9580a9e96 \
+ --hash=sha256:4bea7f8ebe90bbd7f0e4a2de42ca6924ba23e3e76418c408ff82f1d46fabd687 \
+ --hash=sha256:4c4fb141a727957c93edfe5c32a26ceb6b5f6461d67146e2d39f51e16170bea8 \
+ --hash=sha256:4c9548dc78002099910abaebc0a72ac58b7d30931869e0351c09b507dff4ece3 \
+ --hash=sha256:4d26f14f041e83dd8edfd61f4cd4fa7285d31798b5bf1f28e70c367ba6c41d61 \
+ --hash=sha256:4f298bdadb8f0b9e5672877f647d1be9373ef5320c9e2f049795e26cad28b6a9 \
+ --hash=sha256:52ec005752a56ae79547a05c0139ca2501a0c866390b6115008456b9f0e7cde1 \
+ --hash=sha256:55261ac0d2941c42f196dd576f543d87a8ee03cd6f5e30dfb4d807b2e3b9121a \
+ --hash=sha256:56490c595a28b1bb27dfc583e816152a9767721ef58b2c03b13f954d2f707420 \
+ --hash=sha256:58d3e12c88e0950bca850ae1f7c256055c097639c2edb9eb123af9807d8b15e4 \
+ --hash=sha256:58d4aa13a59c969dbfdf9e6a9560e242cbfd9e8a8f50c2747714df1a423adf65 \
+ --hash=sha256:59171c6e45bf07d0d5cab3b0bf81d945035530f6873398b3b531c31184d46663 \
+ --hash=sha256:5b6d1386bf0096d26d3a863dc0a487a5b4eb9aa93cf5ba69683d29dde6b9d60f \
+ --hash=sha256:5c0ea61a470e070686aa30892fed79e297d2c8d0ab46b8bcdf027d38c51da591 \
+ --hash=sha256:5c84bec0ab5ae0c64bfe73a7d2adcb5ce73b467523fc27fd6a28ab2aa6cbe35a \
+ --hash=sha256:5ca0555312ae2fe82715cada7fac375530c2f3349e1eaa1bcb33d0283ac79a18 \
+ --hash=sha256:5d8531a6569d025f68e2321e7638fb7978f23db58e5f69f56913837aae03816e \
+ --hash=sha256:5e2d0e146dcb57034f8b97dc58d2d512cb90aba253960ce449f695fec6a82c6f \
+ --hash=sha256:5fc45d653ea8c9a20479167e11d4a0f8cb2fa3470737ab6f9c827532313187b7 \
+ --hash=sha256:6117b84ea48435e5356dc737f5121485c30920ba43375fa7b434fd753df0eac3 \
+ --hash=sha256:6199d5606e2bbf2b096cf64d03f8b6790c91081d5ac866b8e7bb6422738cc60c \
+ --hash=sha256:62b55f6722735a6c472f88361cde6640608773d9443cebdbb51abf436a1fcdd3 \
+ --hash=sha256:687c9ca3035544b113bea2055e180af96fb63c0c476e22a9180f51925186e7b7 \
+ --hash=sha256:6b7430cf5728e68f6c462254009a6ef4086e1bea43cf2f57aa9c55fb4f50ff96 \
+ --hash=sha256:6ba32c4d2abf1d2fe7cf27d280f4cca5664233b0f885549c7761719eb977f486 \
+ --hash=sha256:6c9cdde8becb25a7fde49924511aa2644d6f8081cc8df8e9452724303348d8e3 \
+ --hash=sha256:6df0ec430f9a831772c23ca5a224cba36517a58a84bb32c32bb59a9fa67c47f6 \
+ --hash=sha256:6e2912d4babbc65196ac13c2f53468dc57fb8b9c25ef913e8c59ddf7c6dc0e1b \
+ --hash=sha256:6e5e4d73d588ca5ed09df1b7dcd1b203d1df3c542e3f50d126c947d432b10731 \
+ --hash=sha256:70055ff39b97c99e7ae40ea3e393fb62aa2e44dbd9b29f8d14f42fb0025c3959 \
+ --hash=sha256:706bfd38730a5ac7a365793269a00f4e988178cec121391f4248d84ad8c972e9 \
+ --hash=sha256:7235dc28fc6dd9d832ac7c7bce95367dedb85929f17368a0c2bee1e080b9acbf \
+ --hash=sha256:774d157f112367ff4abd29019f38f023c24e00e56edc7829c20e358a5a913ad8 \
+ --hash=sha256:77efcff2b23071c349402ac1066667a3d011f62398d81408c9b88ad991747c9e \
+ --hash=sha256:789b8982559ae28dad2356519f841655756cdcd96616410590ae0b17454ee64f \
+ --hash=sha256:7ac76cf9afd34929d76eb7fcb63be476a4853d8a96f0dcf2d0db68a0cbdf9885 \
+ --hash=sha256:7c0c10730342b0c9b35dd1d619beb8214e520bd96a1f870f452680b238aab3e0 \
+ --hash=sha256:823f82903d189af463d7df250ef1f7f696f3cee08cc8d91deb565e8d425f6506 \
+ --hash=sha256:838648accb3a7fd9803fd45c87bce8509648eb0c11bc34e216141300977244f2 \
+ --hash=sha256:854066be00447fa8de2ccbbe893e2ffc4b123ef16d897af794c1e18bd4a714b0 \
+ --hash=sha256:85d5855daafc240cc045c026d7a15fd198a09b0fc8ff6f5ecbb5297b509cb11e \
+ --hash=sha256:85de3134b5379856e323ba37c19c9256d39425f7b76a63af52b09fb4664c2e8f \
+ --hash=sha256:87e4f41d375c0b9be2fb5251aee4b8a689169e134535aed81bf085c3b647451e \
+ --hash=sha256:88ca277405c2d3b71c4e1c2ee0e7966e807bcba86a69d11e19ba199d18ae4491 \
+ --hash=sha256:88e85ab89cb822c1e635f51d6d32e488f94e002e70e2f492bdb8b945543f345a \
+ --hash=sha256:8ac8c94b6539074e0f40899301273ac8402b9b3e01c7b7ba269ff30340aaaf20 \
+ --hash=sha256:8fe532b3c966d1fb794e0698e4589d0444017ae77fc0b31edea13c0e35bcc449 \
+ --hash=sha256:9085f87b0e38a2b92b8923059b4e8789fe40d9279712d15dcc670048d77079af \
+ --hash=sha256:90b7481fb62fbe172c558bc6fd1c4c98d82004a54a7551f20e11ac9bf0b8708c \
+ --hash=sha256:92caef967d287a407085d61176fce4012b1dd62daed4eb6d5ceb26d3d2538712 \
+ --hash=sha256:9362dd90aa7dab48c0054a21187791ccf05473f7dba5d92b8033ae62164675e7 \
+ --hash=sha256:94d78ecec2605a8d0398b0f365d5f12a63248438516f5dac536a5eff7337df4a \
+ --hash=sha256:94fbf1c0c6cc0d3d5e50f9a9313a8cdca90dd696d34b381cd1704f8c9e939f20 \
+ --hash=sha256:950f23cb393f85543777b0433f082cddd25b51ab398eac7971146495679efe5f \
+ --hash=sha256:96eefc178f8636b9c760c5829345307fd81cfae9ab1e80997dbddeb0f54ee9a3 \
+ --hash=sha256:96fef3e886d6a9874b14f27fc193fbdc69d5d8035783d86aa4e1cea594e695f9 \
+ --hash=sha256:977cdbd483a9cff38179bea4fd754289a6f2195c7abd414aba85410b3e66cc5e \
+ --hash=sha256:978eab16f55b4ab2c2a745be9a0a840bf8f09a7f227d9c76eb30214d078865a5 \
+ --hash=sha256:994e883d17c559cdfd38c84003c8b27d25424a1077272a17e7cd27bfe0bf57b2 \
+ --hash=sha256:9ac4444d8d4fd4c4bd08bf451ed3167aa9e7ec6cdb41b648794f1d1103652e36 \
+ --hash=sha256:9b5db6052055d34d41230fb78d7c439c23dc536a9896f6cb039e8dd92cfc1263 \
+ --hash=sha256:9d9a0dc7cbe9bec24c3f767c9122c41fe5a1bc43f47cd099d00d393e09769de4 \
+ --hash=sha256:9dbdd9205662134957cf0c324f639bdc5031c0ca056e2369e238db75187c0f11 \
+ --hash=sha256:9eea3ab2597a5e65fe65296e2d6a84570845a6b55532d90333d740d48bbc850a \
+ --hash=sha256:a2028475ba855475b8b4d3cfeb4994269c967aea8b9892dfba907f4263a863a3 \
+ --hash=sha256:a3a370082ce34d0612f421e15fe011c53bb1feff21a26d06ad4fb244dab5a375 \
+ --hash=sha256:a545775cfe815855ea32d7c27731d79da358ef2055b4a25830231b1622dd18aa \
+ --hash=sha256:a5cbd90ecf0fc62e64726917ad083b73001f0563657a87ec3c0b504e277dc90d \
+ --hash=sha256:a6d095662e73e74f0a49988e0593373e243e3a52e27bfeea0a859e88acf4a0f5 \
+ --hash=sha256:a6dac12ff6b846103483683f60c5f8fee205121adc58ffd87e90a90a3af69e99 \
+ --hash=sha256:a951ad59cad9145664a730d3036b40b844e74d2d3683da40111463cd3a83845d \
+ --hash=sha256:aa1099b956fb795e686d073568f6dc002a0bb89765ea6d5b055dd7d9bf1b116c \
+ --hash=sha256:aa2bb0b37202dca27175591f761108b5d34096ade1191ffe4808bdf6b1571488 \
+ --hash=sha256:aae2ee51122d3ae968a3837d97dc24a0aeebb0dea23694422cd172bd30017cd6 \
+ --hash=sha256:ab743e9bc90c1f73552ec33e10e3331315acd2c397b36065b591b0181de533cc \
+ --hash=sha256:ac00177c4831ffa650f8609e4bdddd5fe09c03b1c0c47acece7e6ea20421598b \
+ --hash=sha256:ac13b004224fb341e1e25a1ed5e19d32f57cdb2a403e01f003b46f051a550f6f \
+ --hash=sha256:acaf604462bf330b0d07e7a07c1d6e4adac79e5fb13e9c5140590542cafacc00 \
+ --hash=sha256:ae31a1a1db2ee6cc2942fccaf695c934bc7f3db9f2133a3fef1f367cf1a4ab10 \
+ --hash=sha256:ae4a097991662cd4fff0ddc74e0fe7874f82e00042fa0ea00855645ed0c79598 \
+ --hash=sha256:aea996a6aba25260827c9ea511d1addfde2da9eb686ac961838509086188b7e6 \
+ --hash=sha256:b39b69b347e5e47a3b5b8cfc005c68c1ba347474e3960236c4944a8ecd174962 \
+ --hash=sha256:b54e7e13267d49ffbfe68e25b3cbd774dab38fa37238f71265e91b36146eb21c \
+ --hash=sha256:b9af956078716df40d985fb0dfeb2c2120c5ca92ba4ff4b388acfd01cdc14d08 \
+ --hash=sha256:ba2f37ee79e6338845261a3c5b1784e5d1acdff2c0785b284f1b633033d136ab \
+ --hash=sha256:ba501e667c17d8411f98e67a022d9604ef179aff0e459b7e292c796837c13573 \
+ --hash=sha256:baf3775a2635e5a11fbd5e4e64ee69c7e86875d224a5c72aca4c141064589a90 \
+ --hash=sha256:bb57753e36e4855b8ca375069482250a6246372331a3e4f3407eaebb007443f5 \
+ --hash=sha256:bd6c173f04743d483881bffa1478d5a4624475b8cd1d2194956a75548e191c18 \
+ --hash=sha256:be47f99644b208bff7766314013f9acf57b056b04191d570d68ad14022cf5b1d \
+ --hash=sha256:c010f5581d9c612804cc59fcf7b524b707fbcb72828551237ab545bb5c7034af \
+ --hash=sha256:c1dcc36dcb96abc02236e182d17e0f71430152a6c2c7447421da2d2dc144edea \
+ --hash=sha256:c428c6c31eb5f4277d7f8eccaf767fbd548ddd5ce3c8b4f4cbbfab3d96b5904c \
+ --hash=sha256:c658c50ac0c98cd755a2dd50b7977d3bca7df401dcc47fbdfa87db53ef7d4e8b \
+ --hash=sha256:c71fb0d56c920c269cd3e2e3fe7c610e3f1fdb21a6ce60efa6430ff63676cea6 \
+ --hash=sha256:c7b742bf31c88566b4bb6335a7f393bb322e580b6bb98df7bd0c25e6e3519ce8 \
+ --hash=sha256:cc0329df4caaceb950d2f580b5ac716a377f7059624a0bafaeaf8a218c6ed774 \
+ --hash=sha256:cc5d36d96478aa9c60654bd932525bf32964c62a7281eafdf16d85003a8d6004 \
+ --hash=sha256:ce854f5f478050ade5a238731c4ca985a7d3b3cb53ff600a9b5c3b689b5f0a7a \
+ --hash=sha256:ced3fdd71aaa83ce593746c2edb42b7a59cb4c19c8b5c407781c72e493aae55a \
+ --hash=sha256:cee5dd7c6fb5dd52a0fe2a740f9bc6e3593f5f8b1788bde49de02086f30182b2 \
+ --hash=sha256:cfa1c0cc3a8f9f53f1243a5a99ac36fd003880199383b37672e86ddda9cb07e2 \
+ --hash=sha256:d1ee1e296209fdce05b81b663250eefa02213a2da7b41bf26f7829b8ba3545aa \
+ --hash=sha256:d59b75732e9b6f27388e10c14b0259cc5f2e48c78627d185e6a177b58ad3cffe \
+ --hash=sha256:d63600d620ad0064c3a748b950ac5ea38a80190e5498532efefa4b7b3f1da1f3 \
+ --hash=sha256:dd732602a7009217f658d5863d12d79d373a4de0eebc111094bcdd3bb8e0a6cc \
+ --hash=sha256:e06efa066f7dbadbc84ebc126a97c452a6451dfcf589d89d788484949e1cf795 \
+ --hash=sha256:e199fb99720074809a7720f1c0b4d919eea8b87e88713e0f8f602f7bef543d9d \
+ --hash=sha256:e4b018dc5a0eee4676e38fe84a47a427816c590b93b55d9025274ec4d6ffc2dc \
+ --hash=sha256:e6621fb2a4988d6e53eedc455e5903e2679f3967b8acb3d639f1b63c14a2e893 \
+ --hash=sha256:e71c909f353863b2b89c83de2ebed71ea6d0df8a6ef65a128193c5e650766bef \
+ --hash=sha256:e90251c0c7bdd54a100a0dce3c07b7e637278c93af29dbf78ebb89a58c4bac7d \
+ --hash=sha256:e9fbdce1e47394b09bc9f26ab117dfc8d6491977a11d86f592bb42c779db2fda \
+ --hash=sha256:eb12fb2ba69ffa05f8695f61c69e591dc4b4a12ac3757ac8af8adb259bf56d17 \
+ --hash=sha256:eda059b6bc8bc0812d626fd91a7ce01bf583df0a61296eff390fd94141a34e30 \
+ --hash=sha256:f03ac127268b43ef4fe9e6ab6794a6794b49485a0cc0c1db79876d2f33f75bc7 \
+ --hash=sha256:f298e218441525d3794428b4c8b8fb8662c6d3ea79925d4807ee6b9a96a3bca5 \
+ --hash=sha256:f5542f9b941279d82d41eb0aa9f98eba36fe4df5c7086c651df7944935b37182 \
+ --hash=sha256:f6f7deae3feb4edfa2efaf7c574fe88cbf055038a6abdb40188e4fff66d5699f \
+ --hash=sha256:f9b1e28d0e8dbfa858abdba91d6b547beaf2df1a59bec6da6faae7b96a4991a9 \
+ --hash=sha256:f9f8405c2c758532c74fed975dbee57be1f31a6e865c031870c79a6ed3212ada \
+ --hash=sha256:fa48b1b63d639f9483e0633e092f5851e2348c352f1f9bb6c8182f87884ef876 \
+ --hash=sha256:fb78f6e7fcd8ad785d28cd577168bc1aaee827b25bb8755638f694794ea98f0a \
+ --hash=sha256:fbc597639158fd7c14d55e808718848319540f51b0e6746e3eefa59723a4a348 \
+ --hash=sha256:fce8cbd4997efeb450bd298b54f755dcdff18d496f7a5ddbb4867c6d7c88fdc3 \
+ --hash=sha256:fd0350afdc3aabd5576f60ea109228bd5538139713c7b094c5cd27c73a98bc6f \
+ --hash=sha256:fd0a274c0e5f9a21565cd9d3dd749b61f96b7aa1e20a93aa1ba4029518f2e5c0 \
+ --hash=sha256:fdb8a068947befafba9952162645dc2fecaeb400e64584829ed5e9b2fbe21a7f
+idna==3.19 \
+ --hash=sha256:5e0811a4383b21dc5838069f801c4fb62113b7447663d2530d2bd6e77b49bf15 \
+ --hash=sha256:815e7be7a7806d54abb586dc943addc79e8b2ee16915059658cbeff4b1b43bf4
ply==3.11 \
--hash=sha256:00c7c1aaa88358b9c765b6d3000c6eec0ba42abca5351b095321aef446081da3 \
--hash=sha256:096f9b8350b65ebd2fd1346b12452efe5b9607f7482813ffca50c22722a807ce
@@ -133,53 +192,42 @@ protobuf==5.29.6 \
--hash=sha256:cb4c86de9cd8a7f3a256b9744220d87b847371c6b2f10bde87768918ef33ba49 \
--hash=sha256:da9ee6a5424b6b30fd5e45c5ea663aef540ca95f9ad99d1e887e819cdf9b8723 \
--hash=sha256:e3387f44798ac1106af0233c04fb8abf543772ff241169946f698b3a9a3d3ab9
-psutil==5.9.0 \
- --hash=sha256:072664401ae6e7c1bfb878c65d7282d4b4391f1bc9a56d5e03b5a490403271b5 \
- --hash=sha256:1070a9b287846a21a5d572d6dddd369517510b68710fca56b0e9e02fd24bed9a \
- --hash=sha256:1d7b433519b9a38192dfda962dd8f44446668c009833e1429a52424624f408b4 \
- --hash=sha256:3151a58f0fbd8942ba94f7c31c7e6b310d2989f4da74fcbf28b934374e9bf841 \
- --hash=sha256:32acf55cb9a8cbfb29167cd005951df81b567099295291bcfd1027365b36591d \
- --hash=sha256:3611e87eea393f779a35b192b46a164b1d01167c9d323dda9b1e527ea69d697d \
- --hash=sha256:3d00a664e31921009a84367266b35ba0aac04a2a6cad09c550a89041034d19a0 \
- --hash=sha256:4e2fb92e3aeae3ec3b7b66c528981fd327fb93fd906a77215200404444ec1845 \
- --hash=sha256:539e429da49c5d27d5a58e3563886057f8fc3868a5547b4f1876d9c0f007bccf \
- --hash=sha256:55ce319452e3d139e25d6c3f85a1acf12d1607ddedea5e35fb47a552c051161b \
- --hash=sha256:58c7d923dc209225600aec73aa2c4ae8ea33b1ab31bc11ef8a5933b027476f07 \
- --hash=sha256:7336292a13a80eb93c21f36bde4328aa748a04b68c13d01dfddd67fc13fd0618 \
- --hash=sha256:742c34fff804f34f62659279ed5c5b723bb0195e9d7bd9907591de9f8f6558e2 \
- --hash=sha256:7641300de73e4909e5d148e90cc3142fb890079e1525a840cf0dfd39195239fd \
- --hash=sha256:76cebf84aac1d6da5b63df11fe0d377b46b7b500d892284068bacccf12f20666 \
- --hash=sha256:7779be4025c540d1d65a2de3f30caeacc49ae7a2152108adeaf42c7534a115ce \
- --hash=sha256:7d190ee2eaef7831163f254dc58f6d2e2a22e27382b936aab51c835fc080c3d3 \
- --hash=sha256:8293942e4ce0c5689821f65ce6522ce4786d02af57f13c0195b40e1edb1db61d \
- --hash=sha256:869842dbd66bb80c3217158e629d6fceaecc3a3166d3d1faee515b05dd26ca25 \
- --hash=sha256:90a58b9fcae2dbfe4ba852b57bd4a1dded6b990a33d6428c7614b7d48eccb492 \
- --hash=sha256:9b51917c1af3fa35a3f2dabd7ba96a2a4f19df3dec911da73875e1edaf22a40b \
- --hash=sha256:b2237f35c4bbae932ee98902a08050a27821f8f6dfa880a47195e5993af4702d \
- --hash=sha256:c3400cae15bdb449d518545cbd5b649117de54e3596ded84aacabfbb3297ead2 \
- --hash=sha256:c51f1af02334e4b516ec221ee26b8fdf105032418ca5a5ab9737e8c87dafe203 \
- --hash=sha256:cb8d10461c1ceee0c25a64f2dd54872b70b89c26419e147a05a10b753ad36ec2 \
- --hash=sha256:d62a2796e08dd024b8179bd441cb714e0f81226c352c802fca0fd3f89eeacd94 \
- --hash=sha256:df2c8bd48fb83a8408c8390b143c6a6fa10cb1a674ca664954de193fdcab36a9 \
- --hash=sha256:e5c783d0b1ad6ca8a5d3e7b680468c9c926b804be83a3a8e95141b05c39c9f64 \
- --hash=sha256:e9805fed4f2a81de98ae5fe38b75a74c6e6ad2df8a5c479594c7629a1fe35f56 \
- --hash=sha256:ea42d747c5f71b5ccaa6897b216a7dadb9f52c72a0fe2b872ef7d3e1eacf3ba3 \
- --hash=sha256:ef216cc9feb60634bda2f341a9559ac594e2eeaadd0ba187a4c2eb5b5d40b91c \
- --hash=sha256:ff0d41f8b3e9ebb6b6110057e40019a432e96aae2008951121ba4e56040b84f3
-requests==2.33.1 \
- --hash=sha256:18817f8c57c6263968bc123d237e3b8b08ac046f5456bd1e307ee8f4250d3517 \
- --hash=sha256:4e6d1ef462f3626a1f0a0a9c42dd93c63bad33f9f1c1937509b8c5c8718ab56a
-tornado==6.5.7 \
- --hash=sha256:148b2eb15c2c765a50796172c1e499649b35f30d2e3c3d3e15913cfa56bfb163 \
- --hash=sha256:66c513a76cda70d53907bc27cf1447557699c2e95aa48ba27a442ff61c3ddfc2 \
- --hash=sha256:7778b30bef919231265e91c69963ce0f49a1e9c07ac900bbe75b19ce2575ba92 \
- --hash=sha256:8a46347a18f23fb92b396beebe0fb78f61dda0cc302445202c16203d8a18848b \
- --hash=sha256:8d759e71906ee783f8867b93bf26a265743da4c1e2f4a018464c1ba019862972 \
- --hash=sha256:9da38de27f1da3b78a966f0dae12b5a1ea9afe72ca805d84ff06508272ddf100 \
- --hash=sha256:de942f843533a039ef9fa3d9c88c7cd8a7c94553fb5ad0154270989b3d99a2c4 \
- --hash=sha256:e726f0c75da7726eec023aa62751ff8878bd2737e34fbdd33b1ae5897d2200f5 \
- --hash=sha256:f8de3bf12d3efdd0cbe7c8887868198f8a91415e3f29fcf258d9b8eb7b1d9ae4 \
- --hash=sha256:ff934fce95643af5f11efdae618eaa73d469dc588641e5c8d19295a0c65c4796
+psutil==7.2.2 \
+ --hash=sha256:0746f5f8d406af344fd547f1c8daa5f5c33dbc293bb8d6a16d80b4bb88f59372 \
+ --hash=sha256:076a2d2f923fd4821644f5ba89f059523da90dc9014e85f8e45a5774ca5bc6f9 \
+ --hash=sha256:11fe5a4f613759764e79c65cf11ebdf26e33d6dd34336f8a337aa2996d71c841 \
+ --hash=sha256:1a571f2330c966c62aeda00dd24620425d4b0cc86881c89861fbc04549e5dc63 \
+ --hash=sha256:1a7b04c10f32cc88ab39cbf606e117fd74721c831c98a27dc04578deb0c16979 \
+ --hash=sha256:1fa4ecf83bcdf6e6c8f4449aff98eefb5d0604bf88cb883d7da3d8d2d909546a \
+ --hash=sha256:2edccc433cbfa046b980b0df0171cd25bcaeb3a68fe9022db0979e7aa74a826b \
+ --hash=sha256:7b6d09433a10592ce39b13d7be5a54fbac1d1228ed29abc880fb23df7cb694c9 \
+ --hash=sha256:8c233660f575a5a89e6d4cb65d9f938126312bca76d8fe087b947b3a1aaac9ee \
+ --hash=sha256:917e891983ca3c1887b4ef36447b1e0873e70c933afc831c6b6da078ba474312 \
+ --hash=sha256:ab486563df44c17f5173621c7b198955bd6b613fb87c71c161f827d3fb149a9b \
+ --hash=sha256:ae0aefdd8796a7737eccea863f80f81e468a1e4cf14d926bd9b6f5f2d5f90ca9 \
+ --hash=sha256:b0726cecd84f9474419d67252add4ac0cd9811b04d61123054b9fb6f57df6e9e \
+ --hash=sha256:b58fabe35e80b264a4e3bb23e6b96f9e45a3df7fb7eed419ac0e5947c61e47cc \
+ --hash=sha256:c7663d4e37f13e884d13994247449e9f8f574bc4655d509c3b95e9ec9e2b9dc1 \
+ --hash=sha256:e452c464a02e7dc7822a05d25db4cde564444a67e58539a00f929c51eddda0cf \
+ --hash=sha256:e78c8603dcd9a04c7364f1a3e670cea95d51ee865e4efb3556a3a63adef958ea \
+ --hash=sha256:eb7e81434c8d223ec4a219b5fc1c47d0417b12be7ea866e24fb5ad6e84b3d988 \
+ --hash=sha256:ed0cace939114f62738d808fdcecd4c869222507e266e574799e9c0faa17d486 \
+ --hash=sha256:eed63d3b4d62449571547b60578c5b2c4bcccc5387148db46e0c2313dad0ee00 \
+ --hash=sha256:fd04ef36b4a6d599bbdb225dd1d3f51e00105f6d48a28f006da7f9822f2606d8
+requests==2.34.2 \
+ --hash=sha256:2a0d60c172f83ac6ab31e4554906c0f3b3588d37b5cb939b1c061f4907e278e0 \
+ --hash=sha256:f288924cae4e29463698d6d60bc6a4da69c89185ad1e0bcc4104f584e960b9ed
+tornado==6.5.8 \
+ --hash=sha256:11881db6b7c168494be2c2d12e65931451bdf7ee718535418ae1d8855dd5a0ee \
+ --hash=sha256:547d63f450d570c14fe0e8db2cfb14c9bbd1c2503b4a6612586267955aa47b58 \
+ --hash=sha256:5d242290bdf7ab3151bc1065fdd75c0dcc21cbc7b49f22a4c56329c2d6566d22 \
+ --hash=sha256:67832909c4779c64942380cb5f044a5c6163d00831472d80e25e115de9917836 \
+ --hash=sha256:68a7468c7e289f8514d7d664101753903217eff1bb6822c6b5994a0b5f5bcb26 \
+ --hash=sha256:7b94ff0e128fe0542f3bd331fb44d06260fc4ac16881545159f34ef08aad4195 \
+ --hash=sha256:7e2360a0ffbe145eca8af0b19cb7203d79b1a98dd4cccdd6b368f6f49c2e3808 \
+ --hash=sha256:9452e1b208a8bd771e2cb1f2ff564985b9b214bdebbe622793e1799e0a6bd23f \
+ --hash=sha256:9715b5eb79735b2bcd454ce216a9275b7c0470e64ea1bf5742f78b2f72b26eeb \
+ --hash=sha256:cc6aa787d7cfab7c3d35189dc7a56fbd2399a569624c730c6b55b3d6531d0403
urllib3==2.7.0 \
--hash=sha256:231e0ec3b63ceb14667c67be60f2f2c40a518cb38b03af60abc813da26505f4c \
--hash=sha256:9fb4c81ebbb1ce9531cce37674bbc6f1360472bc18ca9a553ede278ef7276897
@@ -188,6 +236,6 @@ vdms==0.0.23 \
--hash=sha256:7cd93242df644947c009559f31b4e19eb1dab6b62ab783ce5a2a54a4ad392f57
# The following packages are considered to be unsafe in a requirements file:
-pip==26.1.2 \
- --hash=sha256:382ff9f685ee3bc25864f820aa50505825f10f5458ffff07e30a6d96e5715cab \
- --hash=sha256:f49cd134c61cf2fd75e0ce2676db03e4054504a5a4986d00f8299ae632dc4605
+pip==26.2.1 \
+ --hash=sha256:71138adf1f4ca900cdb7d289c21b7494329f2332b6d85f0e1c42108c0384ed3e \
+ --hash=sha256:f6ad667e89a1fe78046c8f13232b247200f5258d7828f3f7883d660878e0813f
diff --git a/start_app.sh b/start_app.sh
index a8b6a0a..956987c 100755
--- a/start_app.sh
+++ b/start_app.sh
@@ -6,10 +6,10 @@
INGESTION="object" #,face"
EXP_TYPE=compose
DEBUG="0"
-DEVICE="CPU"
+DEVICE="GPU"
DOCKER_TAR="0"
RESIZE_FLAG="False"
-OMIT_DETECTIONS_FLAG="False"
+OMIT_DETECTIONS_FLAG="True"
MODEL_NAME=""
DIR=$(dirname $(readlink -f "$0"))
@@ -53,13 +53,8 @@ script_usage()
Options:
-h optional Print this help message
-d or --debug optional Flag to enable debug messages
- -e or --device optional Device for inference (CPU, GPU) [Default: CPU]
- -i or --ingestion optional Ingestion type (object, face) [Default: "object,face"]
-l or --tars optional Flag to load docker images instead of building from Dockerfiles
-m or --model optional Custom YOLO model name (.pt). If not provided model YOLO11n is used.
- -o or --omit-det optional By default, object detections are printed. To omit printing detections to screen, enable flag.
- -t or --type optional Deployment method (compose) [Default: compose]
- -z or --resize optional Flag to resize video to model input size
EOF
}
@@ -69,12 +64,7 @@ while true; do
-h) script_usage; exit 0 ;;
-d | --debug) shift; DEBUG="1" ;;
-l | --tars) shift; DOCKER_TAR="1" ;;
- -e | --device) shift; DEVICE=$1; shift ;;
- -i | --ingestion) shift; INGESTION=$1; shift ;;
-m | --model) shift; MODEL_NAME=$1; shift ;;
- -t | --type) shift; EXP_TYPE="$1"; shift ;;
- -z | --resize) shift; RESIZE_FLAG="True" ;;
- -o | --omit-det) shift; OMIT_DETECTIONS_FLAG="True" ;;
--) shift; break ;;
*) script_usage; exit 0 ;;
esac
diff --git a/stop_app.sh b/stop_app.sh
index deb3e94..fadbe84 100755
--- a/stop_app.sh
+++ b/stop_app.sh
@@ -34,7 +34,6 @@ script_usage()
Options:
-h optional Print this help message
- -t or --type optional Deployment method (compose) [Default: compose]
-p or --prune optional Flag to prune docker builder
EOF
@@ -43,7 +42,6 @@ EOF
while true; do
case "$1" in
-h) script_usage; exit 0 ;;
- -t | --type) shift; EXP_TYPE="$1"; shift ;;
-p | --prune) shift; DOCKER_PRUNE="1" ;;
--) shift; break ;;
*) script_usage; exit 0 ;;
diff --git a/udf/requirements.txt b/udf/requirements.txt
index d223a26..c9688ac 100644
--- a/udf/requirements.txt
+++ b/udf/requirements.txt
@@ -1,12 +1,12 @@
blinker==1.9.0 \
--hash=sha256:b4ce2265a7abece45e7cc896e98dbebe6cead56bcf805a3d23136d145f5445bf \
--hash=sha256:ba0efaa9080b619ff2f3459d1d500c57bddea4a6b424b60a91141db6fd2f08bc
-build==1.4.0 \
- --hash=sha256:6a07c1b8eb6f2b311b96fcbdbce5dab5fe637ffda0fd83c9cac622e927501596 \
- --hash=sha256:f1b91b925aa322be454f8330c6fb48b465da993d1e7e7e6fa35027ec49f3c936
-click==8.3.1 \
- --hash=sha256:12ff4785d337a1bb490bb7e9c2b1ee5da3112e94a8622f26a6c77f5d2fc6842a \
- --hash=sha256:981153a64e25f12d547d3426c367a4857371575ee7ad18df2a6183ab0545b2a6
+build==1.6.0 \
+ --hash=sha256:bd2c8afc603e7a2e0ce70e2ea85f0a6d02043bafbd307f5bada0f98669eca5af \
+ --hash=sha256:f7aaf1ebbb79178a02ba248bb524f2176b256017e17e8e4bd4289c7b38cc2bad
+click==8.5.0 \
+ --hash=sha256:255bc9599cf7748b4b1a446ccc735421bd08a2ae529a8b88597d3de5664ee360 \
+ --hash=sha256:ba0d2089de75ea0310e2dde03160e6ca10009947fb95a182f9b54021bb272e34
flask==3.1.3 \
--hash=sha256:0ef0e52b8a9cd932855379197dd8f94047b359ca0a78695144304cb45f87c9eb \
--hash=sha256:f4bcbefc124291925f1a26446da31a5178f9483862233b23c0c96a20701f670c
@@ -109,74 +109,54 @@ markupsafe==3.0.3 \
--hash=sha256:f71a396b3bf33ecaa1626c255855702aca4d3d9fea5e051b41ac59a9c1c41edc \
--hash=sha256:f9e130248f4462aaa8e2552d547f36ddadbeaa573879158d721bbd33dfe4743a \
--hash=sha256:fed51ac40f757d41b7c48425901843666a6677e3e8eb0abcff09e4ba6e664f50
-numpy==2.2.6 \
- --hash=sha256:038613e9fb8c72b0a41f025a7e4c3f0b7a1b5d768ece4796b674c8f3fe13efff \
- --hash=sha256:0678000bb9ac1475cd454c6b8c799206af8107e310843532b04d49649c717a47 \
- --hash=sha256:0811bb762109d9708cca4d0b13c4f67146e3c3b7cf8d34018c722adb2d957c84 \
- --hash=sha256:0b605b275d7bd0c640cad4e5d30fa701a8d59302e127e5f79138ad62762c3e3d \
- --hash=sha256:0bca768cd85ae743b2affdc762d617eddf3bcf8724435498a1e80132d04879e6 \
- --hash=sha256:1bc23a79bfabc5d056d106f9befb8d50c31ced2fbc70eedb8155aec74a45798f \
- --hash=sha256:287cc3162b6f01463ccd86be154f284d0893d2b3ed7292439ea97eafa8170e0b \
- --hash=sha256:37c0ca431f82cd5fa716eca9506aefcabc247fb27ba69c5062a6d3ade8cf8f49 \
- --hash=sha256:37e990a01ae6ec7fe7fa1c26c55ecb672dd98b19c3d0e1d1f326fa13cb38d163 \
- --hash=sha256:389d771b1623ec92636b0786bc4ae56abafad4a4c513d36a55dce14bd9ce8571 \
- --hash=sha256:3d70692235e759f260c3d837193090014aebdf026dfd167834bcba43e30c2a42 \
- --hash=sha256:41c5a21f4a04fa86436124d388f6ed60a9343a6f767fced1a8a71c3fbca038ff \
- --hash=sha256:481b49095335f8eed42e39e8041327c05b0f6f4780488f61286ed3c01368d491 \
- --hash=sha256:4eeaae00d789f66c7a25ac5f34b71a7035bb474e679f410e5e1a94deb24cf2d4 \
- --hash=sha256:55a4d33fa519660d69614a9fad433be87e5252f4b03850642f88993f7b2ca566 \
- --hash=sha256:5a6429d4be8ca66d889b7cf70f536a397dc45ba6faeb5f8c5427935d9592e9cf \
- --hash=sha256:5bd4fc3ac8926b3819797a7c0e2631eb889b4118a9898c84f585a54d475b7e40 \
- --hash=sha256:5beb72339d9d4fa36522fc63802f469b13cdbe4fdab4a288f0c441b74272ebfd \
- --hash=sha256:6031dd6dfecc0cf9f668681a37648373bddd6421fff6c66ec1624eed0180ee06 \
- --hash=sha256:71594f7c51a18e728451bb50cc60a3ce4e6538822731b2933209a1f3614e9282 \
- --hash=sha256:74d4531beb257d2c3f4b261bfb0fc09e0f9ebb8842d82a7b4209415896adc680 \
- --hash=sha256:7befc596a7dc9da8a337f79802ee8adb30a552a94f792b9c9d18c840055907db \
- --hash=sha256:894b3a42502226a1cac872f840030665f33326fc3dac8e57c607905773cdcde3 \
- --hash=sha256:8e41fd67c52b86603a91c1a505ebaef50b3314de0213461c7a6e99c9a3beff90 \
- --hash=sha256:8e9ace4a37db23421249ed236fdcdd457d671e25146786dfc96835cd951aa7c1 \
- --hash=sha256:8fc377d995680230e83241d8a96def29f204b5782f371c532579b4f20607a289 \
- --hash=sha256:9551a499bf125c1d4f9e250377c1ee2eddd02e01eac6644c080162c0c51778ab \
- --hash=sha256:b0544343a702fa80c95ad5d3d608ea3599dd54d4632df855e4c8d24eb6ecfa1c \
- --hash=sha256:b093dd74e50a8cba3e873868d9e93a85b78e0daf2e98c6797566ad8044e8363d \
- --hash=sha256:b412caa66f72040e6d268491a59f2c43bf03eb6c96dd8f0307829feb7fa2b6fb \
- --hash=sha256:b4f13750ce79751586ae2eb824ba7e1e8dba64784086c98cdbbcc6a42112ce0d \
- --hash=sha256:b64d8d4d17135e00c8e346e0a738deb17e754230d7e0810ac5012750bbd85a5a \
- --hash=sha256:ba10f8411898fc418a521833e014a77d3ca01c15b0c6cdcce6a0d2897e6dbbdf \
- --hash=sha256:bd48227a919f1bafbdda0583705e547892342c26fb127219d60a5c36882609d1 \
- --hash=sha256:c1f9540be57940698ed329904db803cf7a402f3fc200bfe599334c9bd84a40b2 \
- --hash=sha256:c820a93b0255bc360f53eca31a0e676fd1101f673dda8da93454a12e23fc5f7a \
- --hash=sha256:ce47521a4754c8f4593837384bd3424880629f718d87c5d44f8ed763edd63543 \
- --hash=sha256:d042d24c90c41b54fd506da306759e06e568864df8ec17ccc17e9e884634fd00 \
- --hash=sha256:de749064336d37e340f640b05f24e9e3dd678c57318c7289d222a8a2f543e90c \
- --hash=sha256:e1dda9c7e08dc141e0247a5b8f49cf05984955246a327d4c48bda16821947b2f \
- --hash=sha256:e29554e2bef54a90aa5cc07da6ce955accb83f21ab5de01a62c8478897b264fd \
- --hash=sha256:e3143e4451880bed956e706a3220b4e5cf6172ef05fcc397f6f36a550b1dd868 \
- --hash=sha256:e8213002e427c69c45a52bbd94163084025f533a55a59d6f9c5b820774ef3303 \
- --hash=sha256:efd28d4e9cd7d7a8d39074a4d44c63eda73401580c5c76acda2ce969e0a38e83 \
- --hash=sha256:f0fd6321b839904e15c46e0d257fdd101dd7f530fe03fd6359c1ea63738703f3 \
- --hash=sha256:f1372f041402e37e5e633e586f62aa53de2eac8d98cbfb822806ce4bbefcb74d \
- --hash=sha256:f2618db89be1b4e05f7a1a847a9c1c0abd63e63a1607d892dd54668dd92faf87 \
- --hash=sha256:f447e6acb680fd307f40d3da4852208af94afdfab89cf850986c3ca00562f4fa \
- --hash=sha256:f92729c95468a2f4f15e9bb94c432a9229d0d50de67304399627a943201baa2f \
- --hash=sha256:f9f1adb22318e121c5c69a09142811a201ef17ab257a1e66ca3025065b7f53ae \
- --hash=sha256:fc0c5673685c508a142ca65209b4e79ed6740a4ed6b2267dbba90f34b0b3cfda \
- --hash=sha256:fc7b73d02efb0e18c000e9ad8b83480dfcd5dfd11065997ed4c6747470ae8915 \
- --hash=sha256:fd83c01228a688733f1ded5201c678f0c53ecc1006ffbc404db9f7a899ac6249 \
- --hash=sha256:fe27749d33bb772c80dcd84ae7e8df2adc920ae8297400dabec45f0dedb3f6de \
- --hash=sha256:fee4236c876c4e8369388054d02d0e9bb84821feb1a64dd59e137e6511a551f8
-opencv-python-headless==4.13.0.90 \
- --hash=sha256:0e0c8c9f620802fddc4fa7f471a1d263c7b0dca16cd9e7e2f996bb8bd2128c0c \
- --hash=sha256:12a28674f215542c9bf93338de1b5bffd76996d32da9acb9e739fdb9c8bbd738 \
- --hash=sha256:32255203040dc98803be96362e13f9e4bce20146898222d2e5c242f80de50da5 \
- --hash=sha256:96060fc57a1abb1144b0b8129e2ff3bfcdd0ccd8e8bd05bd85256ff4ed587d3b \
- --hash=sha256:dbc1f4625e5af3a80ebdbd84380227c0f445228588f2521b11af47710caca1ba \
- --hash=sha256:e13790342591557050157713af17a7435ac1b50c65282715093c9297fa045d8f \
- --hash=sha256:eba38bc255d0b7d1969c5bcc90a060ca2b61a3403b613872c750bfa5dfe9e03b \
- --hash=sha256:f46b17ea0aa7e4124ca6ad71143f89233ae9557f61d2326bcdb34329a1ddf9bd
-packaging==26.0 \
- --hash=sha256:00243ae351a257117b6a241061796684b084ed1c516a08c48a3f7e147a9d80b4 \
- --hash=sha256:b36f1fef9334a5588b4166f8bcd26a14e521f2b55e6b9de3aaa80d3ff7a37529
+numpy==1.26.4 \
+ --hash=sha256:03a8c78d01d9781b28a6989f6fa1bb2c4f2d51201cf99d3dd875df6fbd96b23b \
+ --hash=sha256:08beddf13648eb95f8d867350f6a018a4be2e5ad54c8d8caed89ebca558b2818 \
+ --hash=sha256:1af303d6b2210eb850fcf03064d364652b7120803a0b872f5211f5234b399f20 \
+ --hash=sha256:1dda2e7b4ec9dd512f84935c5f126c8bd8b9f2fc001e9f54af255e8c5f16b0e0 \
+ --hash=sha256:2a02aba9ed12e4ac4eb3ea9421c420301a0c6460d9830d74a9df87efa4912010 \
+ --hash=sha256:2e4ee3380d6de9c9ec04745830fd9e2eccb3e6cf790d39d7b98ffd19b0dd754a \
+ --hash=sha256:3373d5d70a5fe74a2c1bb6d2cfd9609ecf686d47a2d7b1d37a8f3b6bf6003aea \
+ --hash=sha256:47711010ad8555514b434df65f7d7b076bb8261df1ca9bb78f53d3b2db02e95c \
+ --hash=sha256:4c66707fabe114439db9068ee468c26bbdf909cac0fb58686a42a24de1760c71 \
+ --hash=sha256:50193e430acfc1346175fcbdaa28ffec49947a06918b7b92130744e81e640110 \
+ --hash=sha256:52b8b60467cd7dd1e9ed082188b4e6bb35aa5cdd01777621a1658910745b90be \
+ --hash=sha256:60dedbb91afcbfdc9bc0b1f3f402804070deed7392c23eb7a7f07fa857868e8a \
+ --hash=sha256:62b8e4b1e28009ef2846b4c7852046736bab361f7aeadeb6a5b89ebec3c7055a \
+ --hash=sha256:666dbfb6ec68962c033a450943ded891bed2d54e6755e35e5835d63f4f6931d5 \
+ --hash=sha256:675d61ffbfa78604709862923189bad94014bef562cc35cf61d3a07bba02a7ed \
+ --hash=sha256:679b0076f67ecc0138fd2ede3a8fd196dddc2ad3254069bcb9faf9a79b1cebcd \
+ --hash=sha256:7349ab0fa0c429c82442a27a9673fc802ffdb7c7775fad780226cb234965e53c \
+ --hash=sha256:7ab55401287bfec946ced39700c053796e7cc0e3acbef09993a9ad2adba6ca6e \
+ --hash=sha256:7e50d0a0cc3189f9cb0aeb3a6a6af18c16f59f004b866cd2be1c14b36134a4a0 \
+ --hash=sha256:95a7476c59002f2f6c590b9b7b998306fba6a5aa646b1e22ddfeaf8f78c3a29c \
+ --hash=sha256:96ff0b2ad353d8f990b63294c8986f1ec3cb19d749234014f4e7eb0112ceba5a \
+ --hash=sha256:9fad7dcb1aac3c7f0584a5a8133e3a43eeb2fe127f47e3632d43d677c66c102b \
+ --hash=sha256:9ff0f4f29c51e2803569d7a51c2304de5554655a60c5d776e35b4a41413830d0 \
+ --hash=sha256:a354325ee03388678242a4d7ebcd08b5c727033fcff3b2f536aea978e15ee9e6 \
+ --hash=sha256:a4abb4f9001ad2858e7ac189089c42178fcce737e4169dc61321660f1a96c7d2 \
+ --hash=sha256:ab47dbe5cc8210f55aa58e4805fe224dac469cde56b9f731a4c098b91917159a \
+ --hash=sha256:afedb719a9dcfc7eaf2287b839d8198e06dcd4cb5d276a3df279231138e83d30 \
+ --hash=sha256:b3ce300f3644fb06443ee2222c2201dd3a89ea6040541412b8fa189341847218 \
+ --hash=sha256:b97fe8060236edf3662adfc2c633f56a08ae30560c56310562cb4f95500022d5 \
+ --hash=sha256:bfe25acf8b437eb2a8b2d49d443800a5f18508cd811fea3181723922a8a82b07 \
+ --hash=sha256:cd25bcecc4974d09257ffcd1f098ee778f7834c3ad767fe5db785be9a4aa9cb2 \
+ --hash=sha256:d209d8969599b27ad20994c8e41936ee0964e6da07478d6c35016bc386b66ad4 \
+ --hash=sha256:d5241e0a80d808d70546c697135da2c613f30e28251ff8307eb72ba696945764 \
+ --hash=sha256:edd8b5fe47dab091176d21bb6de568acdd906d1887a4584a15a9a96a1dca06ef \
+ --hash=sha256:f870204a840a60da0b12273ef34f7051e98c3b5961b61b0c2c1be6dfd64fbcd3 \
+ --hash=sha256:ffa75af20b44f8dba823498024771d5ac50620e6915abac414251bd971b4529f
+opencv-python-headless==4.11.0.86 \
+ --hash=sha256:0e0a27c19dd1f40ddff94976cfe43066fbbe9dfbb2ec1907d66c19caef42a57b \
+ --hash=sha256:48128188ade4a7e517237c8e1e11a9cdf5c282761473383e77beb875bb1e61ca \
+ --hash=sha256:6c304df9caa7a6a5710b91709dd4786bf20a74d57672b3c31f7033cc638174ca \
+ --hash=sha256:6efabcaa9df731f29e5ea9051776715b1bdd1845d7c9530065c7951d2a2899eb \
+ --hash=sha256:996eb282ca4b43ec6a3972414de0e2331f5d9cda2b41091a49739c19fb843798 \
+ --hash=sha256:a66c1b286a9de872c343ee7c3553b084244299714ebb50fbdcd76f07ebbe6c81 \
+ --hash=sha256:f447d8acbb0b6f2808da71fddd29c1cdd448d2bc98f72d9bb78a7a898fc9621b
+packaging==26.3 \
+ --hash=sha256:94edc256424af38762eb31306eed28beb9f0efc50a8837492c9d6fd6004aed79 \
+ --hash=sha256:d7193f7c8e4e93f444fde0262bf90af30e16fa0ad0ad44cb553c87339b23cd1c
protobuf==5.29.6 \
--hash=sha256:36ade6ff88212e91aef4e687a971a11d7d24d6948a66751abc1b3238648f5d05 \
--hash=sha256:62e8a3114992c7c647bce37dcc93647575fc52d50e48de30c6fcb28a6a291eb1 \
@@ -192,20 +172,65 @@ protobuf==5.29.6 \
pyproject-hooks==1.2.0 \
--hash=sha256:1e859bd5c40fae9448642dd871adf459e5e2084186e8d2c2a79a824c970da1f8 \
--hash=sha256:9e5c6bfa8dcc30091c74b0cf803c81fdd29d94f01992a7707bc97babb1141913
-tomli==1.1.0 \
- --hash=sha256:33d7984738f8bb699c9b0a816eb646a8178a69eaa792d258486776a5d21b8ca5 \
- --hash=sha256:f4a182048010e89cbec0ae4686b21f550a7f2903f665e34a6de58ec15424f919
+tomli==2.4.1 \
+ --hash=sha256:01f520d4f53ef97964a240a035ec2a869fe1a37dde002b57ebc4417a27ccd853 \
+ --hash=sha256:0d85819802132122da43cb86656f8d1f8c6587d54ae7dcaf30e90533028b49fe \
+ --hash=sha256:136443dbd7e1dee43c68ac2694fde36b2849865fa258d39bf822c10e8068eac5 \
+ --hash=sha256:1d8591993e228b0c930c4bb0db464bdad97b3289fb981255d6c9a41aedc84b2d \
+ --hash=sha256:2190f2e9dd7508d2a90ded5ed369255980a1bcdd58e52f7fe24b8162bf9fedbd \
+ --hash=sha256:2c1c351919aca02858f740c6d33adea0c5deea37f9ecca1cc1ef9e884a619d26 \
+ --hash=sha256:36d2bd2ad5fb9eaddba5226aa02c8ec3fa4f192631e347b3ed28186d43be6b54 \
+ --hash=sha256:3d48a93ee1c9b79c04bb38772ee1b64dcf18ff43085896ea460ca8dec96f35f6 \
+ --hash=sha256:47149d5bd38761ac8be13a84864bf0b7b70bc051806bc3669ab1cbc56216b23c \
+ --hash=sha256:4ab97e64ccda8756376892c53a72bd1f964e519c77236368527f758fbc36a53a \
+ --hash=sha256:4b605484e43cdc43f0954ddae319fb75f04cc10dd80d830540060ee7cd0243cd \
+ --hash=sha256:504aa796fe0569bb43171066009ead363de03675276d2d121ac1a4572397870f \
+ --hash=sha256:51529d40e3ca50046d7606fa99ce3956a617f9b36380da3b7f0dd3dd28e68cb5 \
+ --hash=sha256:52c8ef851d9a240f11a88c003eacb03c31fc1c9c4ec64a99a0f922b93874fda9 \
+ --hash=sha256:559db847dc486944896521f68d8190be1c9e719fced785720d2216fe7022b662 \
+ --hash=sha256:5a881ab208c0baf688221f8cecc5401bd291d67e38a1ac884d6736cbcd8247e9 \
+ --hash=sha256:5cb41aa38891e073ee49d55fbc7839cfdb2bc0e600add13874d048c94aadddd1 \
+ --hash=sha256:5e262d41726bc187e69af7825504c933b6794dc3fbd5945e41a79bb14c31f585 \
+ --hash=sha256:5ee18d9ebdb417e384b58fe414e8d6af9f4e7a0ae761519fb50f721de398dd4e \
+ --hash=sha256:7008df2e7655c495dd12d2a4ad038ff878d4ca4b81fccaf82b714e07eae4402c \
+ --hash=sha256:734e20b57ba95624ecf1841e72b53f6e186355e216e5412de414e3c51e5e3c41 \
+ --hash=sha256:7c7e1a961a0b2f2472c1ac5b69affa0ae1132c39adcb67aba98568702b9cc23f \
+ --hash=sha256:7f86fd587c4ed9dd76f318225e7d9b29cfc5a9d43de44e5754db8d1128487085 \
+ --hash=sha256:7f94b27a62cfad8496c8d2513e1a222dd446f095fca8987fceef261225538a15 \
+ --hash=sha256:88dceee75c2c63af144e456745e10101eb67361050196b0b6af5d717254dddf7 \
+ --hash=sha256:8a650c2dbafa08d42e51ba0b62740dae4ecb9338eefa093aa5c78ceb546fcd5c \
+ --hash=sha256:8d65a2fbf9d2f8352685bc1364177ee3923d6baf5e7f43ea4959d7d8bc326a36 \
+ --hash=sha256:96481a5786729fd470164b47cdb3e0e58062a496f455ee41b4403be77cb5a076 \
+ --hash=sha256:a120733b01c45e9a0c34aeef92bf0cf1d56cfe81ed9d47d562f9ed591a9828ac \
+ --hash=sha256:b1d22e6e9387bf4739fbe23bfa80e93f6b0373a7f1b96c6227c32bef95a4d7a8 \
+ --hash=sha256:b8c198f8c1805dc42708689ed6864951fd2494f924149d3e4bce7710f8eb5232 \
+ --hash=sha256:c2541745709bad0264b7d4705ad453b76ccd191e64aa6f0fc66b69a293a45ece \
+ --hash=sha256:c742f741d58a28940ce01d58f0ab2ea3ced8b12402f162f4d534dfe18ba1cd6a \
+ --hash=sha256:c7f2c7f2b9ca6bdeef8f0fa897f8e05085923eb091721675170254cbc5b02897 \
+ --hash=sha256:d312ef37c91508b0ab2cee7da26ec0b3ed2f03ce12bd87a588d771ae15dcf82d \
+ --hash=sha256:d4d8fe59808a54658fcc0160ecfb1b30f9089906c50b23bcb4c69eddc19ec2b4 \
+ --hash=sha256:da25dc3563bff5965356133435b757a795a17b17d01dbc0f42fb32447ddfd917 \
+ --hash=sha256:eab21f45c7f66c13f2a9e0e1535309cee140182a9cdae1e041d02e47291e8396 \
+ --hash=sha256:eb0dc4e38e6a1fd579e5d50369aa2e10acfc9cace504579b2faabb478e76941a \
+ --hash=sha256:ec9bfaf3ad2df51ace80688143a6a4ebc09a248f6ff781a9945e51937008fcbc \
+ --hash=sha256:ede3e6487c5ef5d28634ba3f31f989030ad6af71edfb0055cbbd14189ff240ba \
+ --hash=sha256:f3c6818a1a86dd6dca7ddcaaf76947d5ba31aecc28cb1b67009a5877c9a64f3f \
+ --hash=sha256:f758f1b9299d059cc3f6546ae2af89670cb1c4d48ea29c3cacc4fe7de3058257 \
+ --hash=sha256:f8f0fc26ec2cc2b965b7a3b87cd19c5c6b8c5e5f436b984e85f486d652285c30 \
+ --hash=sha256:fd0409a3653af6c147209d267a0e4243f0ae46b011aa978b1080359fddc9b6cf \
+ --hash=sha256:ff18e6a727ee0ab0388507b89d1bc6a22b138d1e2fa56d1ad494586d61d2eae9 \
+ --hash=sha256:ff2983983d34813c1aeb0fa89091e76c3a22889ee83ab27c5eeb45100560c049
vdms==0.0.23 \
--hash=sha256:2dec153f8cb33f27cdb9ab33125198195fd29a6777cf890fc735774ed9327874 \
--hash=sha256:7cd93242df644947c009559f31b4e19eb1dab6b62ab783ce5a2a54a4ad392f57
werkzeug==3.1.8 \
--hash=sha256:63a77fb8892bf28ebc3178683445222aa500e48ebad5ec77b0ad80f8726b1f50 \
--hash=sha256:9bad61a4268dac112f1c5cd4630a56ede601b6ed420300677a869083d70a4c44
-wheel==0.46.3 \
- --hash=sha256:4b399d56c9d9338230118d705d9737a2a468ccca63d5e813e2a4fc7815d8bc4d \
- --hash=sha256:e3e79874b07d776c40bd6033f8ddf76a7dad46a7b8aa1b2787a83083519a1803
+wheel==0.48.0 \
+ --hash=sha256:3217dcc807155e45db462d7ef2431f5ddda0d7273b700d05a67b271ceb1287ab \
+ --hash=sha256:94800765601e9171bf5d58d066e640662842bcedcbab982b2c90787a2c987322
# The following packages are considered to be unsafe in a requirements file:
-pip==26.1.2 \
- --hash=sha256:382ff9f685ee3bc25864f820aa50505825f10f5458ffff07e30a6d96e5715cab \
- --hash=sha256:f49cd134c61cf2fd75e0ce2676db03e4054504a5a4986d00f8299ae632dc4605
+pip==26.2.1 \
+ --hash=sha256:71138adf1f4ca900cdb7d289c21b7494329f2332b6d85f0e1c42108c0384ed3e \
+ --hash=sha256:f6ad667e89a1fe78046c8f13232b247200f5258d7828f3f7883d660878e0813f
diff --git a/video/Dockerfile b/video/Dockerfile
index ad5732d..2a51caf 100644
--- a/video/Dockerfile
+++ b/video/Dockerfile
@@ -28,7 +28,7 @@ WORKDIR /home
COPY requirements.* /home/
-ARG DEVICE="CPU"
+ARG DEVICE="GPU"
ENV DEVICE="${DEVICE}"
# RUN pip3 install --no-cache-dir -r /home/requirements.in
diff --git a/video/requirements.txt b/video/requirements.txt
index 9e4a641..5dbccbc 100644
--- a/video/requirements.txt
+++ b/video/requirements.txt
@@ -1,148 +1,191 @@
-build==1.4.0 \
- --hash=sha256:6a07c1b8eb6f2b311b96fcbdbce5dab5fe637ffda0fd83c9cac622e927501596 \
- --hash=sha256:f1b91b925aa322be454f8330c6fb48b465da993d1e7e7e6fa35027ec49f3c936
-certifi==2026.2.25 \
- --hash=sha256:027692e4402ad994f1c42e52a4997a9763c646b73e4096e4d5d6db8af1d6f0fa \
- --hash=sha256:e887ab5cee78ea814d3472169153c2d12cd43b14bd03329a39a9c6e2e80bfba7
-charset-normalizer==3.4.6 \
- --hash=sha256:06a7e86163334edfc5d20fe104db92fcd666e5a5df0977cb5680a506fe26cc8e \
- --hash=sha256:0c173ce3a681f309f31b87125fecec7a5d1347261ea11ebbb856fa6006b23c8c \
- --hash=sha256:0e28d62a8fc7a1fa411c43bd65e346f3bce9716dc51b897fbe930c5987b402d5 \
- --hash=sha256:0e901eb1049fdb80f5bd11ed5ea1e498ec423102f7a9b9e4645d5b8204ff2815 \
- --hash=sha256:11afb56037cbc4b1555a34dd69151e8e069bee82e613a73bef6e714ce733585f \
- --hash=sha256:150b8ce8e830eb7ccb029ec9ca36022f756986aaaa7956aad6d9ec90089338c0 \
- --hash=sha256:172985e4ff804a7ad08eebec0a1640ece87ba5041d565fff23c8f99c1f389484 \
- --hash=sha256:197c1a244a274bb016dd8b79204850144ef77fe81c5b797dc389327adb552407 \
- --hash=sha256:1ae6b62897110aa7c79ea2f5dd38d1abca6db663687c0b1ad9aed6f6bae3d9d6 \
- --hash=sha256:1cf0a70018692f85172348fe06d3a4b63f94ecb055e13a00c644d368eb82e5b8 \
- --hash=sha256:1ed80ff870ca6de33f4d953fda4d55654b9a2b340ff39ab32fa3adbcd718f264 \
- --hash=sha256:22c6f0c2fbc31e76c3b8a86fba1a56eda6166e238c29cdd3d14befdb4a4e4815 \
- --hash=sha256:231d4da14bcd9301310faf492051bee27df11f2bc7549bc0bb41fef11b82daa2 \
- --hash=sha256:259695e2ccc253feb2a016303543d691825e920917e31f894ca1a687982b1de4 \
- --hash=sha256:2a24157fa36980478dd1770b585c0f30d19e18f4fb0c47c13aa568f871718579 \
- --hash=sha256:2b1a63e8224e401cafe7739f77efd3f9e7f5f2026bda4aead8e59afab537784f \
- --hash=sha256:2bd9d128ef93637a5d7a6af25363cf5dec3fa21cf80e68055aad627f280e8afa \
- --hash=sha256:2e1d8ca8611099001949d1cdfaefc510cf0f212484fe7c565f735b68c78c3c95 \
- --hash=sha256:2ef7fedc7a6ecbe99969cd09632516738a97eeb8bd7258bf8a0f23114c057dab \
- --hash=sha256:2f7fdd9b6e6c529d6a2501a2d36b240109e78a8ceaef5687cfcfa2bbe671d297 \
- --hash=sha256:30f445ae60aad5e1f8bdbb3108e39f6fbc09f4ea16c815c66578878325f8f15a \
- --hash=sha256:31215157227939b4fb3d740cd23fe27be0439afef67b785a1eb78a3ae69cba9e \
- --hash=sha256:34315ff4fc374b285ad7f4a0bf7dcbfe769e1b104230d40f49f700d4ab6bbd84 \
- --hash=sha256:3516bbb8d42169de9e61b8520cbeeeb716f12f4ecfe3fd30a9919aa16c806ca8 \
- --hash=sha256:3778fd7d7cd04ae8f54651f4a7a0bd6e39a0cf20f801720a4c21d80e9b7ad6b0 \
- --hash=sha256:39f5068d35621da2881271e5c3205125cc456f54e9030d3f723288c873a71bf9 \
- --hash=sha256:404a1e552cf5b675a87f0651f8b79f5f1e6fd100ee88dc612f89aa16abd4486f \
- --hash=sha256:419a9d91bd238052642a51938af8ac05da5b3343becde08d5cdeab9046df9ee1 \
- --hash=sha256:423fb7e748a08f854a08a222b983f4df1912b1daedce51a72bd24fe8f26a1843 \
- --hash=sha256:4482481cb0572180b6fd976a4d5c72a30263e98564da68b86ec91f0fe35e8565 \
- --hash=sha256:461598cd852bfa5a61b09cae2b1c02e2efcd166ee5516e243d540ac24bfa68a7 \
- --hash=sha256:47955475ac79cc504ef2704b192364e51d0d473ad452caedd0002605f780101c \
- --hash=sha256:48696db7f18afb80a068821504296eb0787d9ce239b91ca15059d1d3eaacf13b \
- --hash=sha256:4be9f4830ba8741527693848403e2c457c16e499100963ec711b1c6f2049b7c7 \
- --hash=sha256:4d1d02209e06550bdaef34af58e041ad71b88e624f5d825519da3a3308e22687 \
- --hash=sha256:4f41da960b196ea355357285ad1316a00099f22d0929fe168343b99b254729c9 \
- --hash=sha256:517ad0e93394ac532745129ceabdf2696b609ec9f87863d337140317ebce1c14 \
- --hash=sha256:51fb3c322c81d20567019778cb5a4a6f2dc1c200b886bc0d636238e364848c89 \
- --hash=sha256:5273b9f0b5835ff0350c0828faea623c68bfa65b792720c453e22b25cc72930f \
- --hash=sha256:530d548084c4a9f7a16ed4a294d459b4f229db50df689bfe92027452452943a0 \
- --hash=sha256:530e8cebeea0d76bdcf93357aa5e41336f48c3dc709ac52da2bb167c5b8271d9 \
- --hash=sha256:54fae94be3d75f3e573c9a1b5402dc593de19377013c9a0e4285e3d402dd3a2a \
- --hash=sha256:572d7c822caf521f0525ba1bce1a622a0b85cf47ffbdae6c9c19e3b5ac3c4389 \
- --hash=sha256:58c948d0d086229efc484fe2f30c2d382c86720f55cd9bc33591774348ad44e0 \
- --hash=sha256:5d11595abf8dd942a77883a39d81433739b287b6aa71620f15164f8096221b30 \
- --hash=sha256:5f8ddd609f9e1af8c7bd6e2aca279c931aefecd148a14402d4e368f3171769fd \
- --hash=sha256:5feb91325bbceade6afab43eb3b508c63ee53579fe896c77137ded51c6b6958e \
- --hash=sha256:60c74963d8350241a79cb8feea80e54d518f72c26db618862a8f53e5023deaf9 \
- --hash=sha256:613f19aa6e082cf96e17e3ffd89383343d0d589abda756b7764cf78361fd41dc \
- --hash=sha256:659a1e1b500fac8f2779dd9e1570464e012f43e580371470b45277a27baa7532 \
- --hash=sha256:695f5c2823691a25f17bc5d5ffe79fa90972cc34b002ac6c843bb8a1720e950d \
- --hash=sha256:69dd852c2f0ad631b8b60cfbe25a28c0058a894de5abb566619c205ce0550eae \
- --hash=sha256:6cceb5473417d28edd20c6c984ab6fee6c6267d38d906823ebfe20b03d607dc2 \
- --hash=sha256:71be7e0e01753a89cf024abf7ecb6bca2c81738ead80d43004d9b5e3f1244e64 \
- --hash=sha256:74119174722c4349af9708993118581686f343adc1c8c9c007d59be90d077f3f \
- --hash=sha256:74a2e659c7ecbc73562e2a15e05039f1e22c75b7c7618b4b574a3ea9118d1557 \
- --hash=sha256:7504e9b7dc05f99a9bbb4525c67a2c155073b44d720470a148b34166a69c054e \
- --hash=sha256:79090741d842f564b1b2827c0b82d846405b744d31e84f18d7a7b41c20e473ff \
- --hash=sha256:7a6967aaf043bceabab5412ed6bd6bd26603dae84d5cb75bf8d9a74a4959d398 \
- --hash=sha256:7bda6eebafd42133efdca535b04ccb338ab29467b3f7bf79569883676fc628db \
- --hash=sha256:7edbed096e4a4798710ed6bc75dcaa2a21b68b6c356553ac4823c3658d53743a \
- --hash=sha256:7f9019c9cb613f084481bd6a100b12e1547cf2efe362d873c2e31e4035a6fa43 \
- --hash=sha256:802168e03fba8bbc5ce0d866d589e4b1ca751d06edee69f7f3a19c5a9fe6b597 \
- --hash=sha256:80d0a5615143c0b3225e5e3ef22c8d5d51f3f72ce0ea6fb84c943546c7b25b6c \
- --hash=sha256:82060f995ab5003a2d6e0f4ad29065b7672b6593c8c63559beefe5b443242c3e \
- --hash=sha256:836ab36280f21fc1a03c99cd05c6b7af70d2697e374c7af0b61ed271401a72a2 \
- --hash=sha256:8761ac29b6c81574724322a554605608a9960769ea83d2c73e396f3df896ad54 \
- --hash=sha256:87725cfb1a4f1f8c2fc9890ae2f42094120f4b44db9360be5d99a4c6b0e03a9e \
- --hash=sha256:899d28f422116b08be5118ef350c292b36fc15ec2daeb9ea987c89281c7bb5c4 \
- --hash=sha256:8bc5f0687d796c05b1e28ab0d38a50e6309906ee09375dd3aff6a9c09dd6e8f4 \
- --hash=sha256:8bea55c4eef25b0b19a0337dc4e3f9a15b00d569c77211fa8cde38684f234fb7 \
- --hash=sha256:8e5a94886bedca0f9b78fecd6afb6629142fd2605aa70a125d49f4edc6037ee6 \
- --hash=sha256:90ca27cd8da8118b18a52d5f547859cc1f8354a00cd1e8e5120df3e30d6279e5 \
- --hash=sha256:92734d4d8d187a354a556626c221cd1a892a4e0802ccb2af432a1d85ec012194 \
- --hash=sha256:947cf925bc916d90adba35a64c82aace04fa39b46b52d4630ece166655905a69 \
- --hash=sha256:95b52c68d64c1878818687a473a10547b3292e82b6f6fe483808fb1468e2f52f \
- --hash=sha256:97d0235baafca5f2b09cf332cc275f021e694e8362c6bb9c96fc9a0eb74fc316 \
- --hash=sha256:9ca4c0b502ab399ef89248a2c84c54954f77a070f28e546a85e91da627d1301e \
- --hash=sha256:9cc4fc6c196d6a8b76629a70ddfcd4635a6898756e2d9cac5565cf0654605d73 \
- --hash=sha256:9cc6e6d9e571d2f863fa77700701dae73ed5f78881efc8b3f9a4398772ff53e8 \
- --hash=sha256:a056d1ad2633548ca18ffa2f85c202cfb48b68615129143915b8dc72a806a923 \
- --hash=sha256:a26611d9987b230566f24a0a125f17fe0de6a6aff9f25c9f564aaa2721a5fb88 \
- --hash=sha256:a4474d924a47185a06411e0064b803c68be044be2d60e50e8bddcc2649957c1f \
- --hash=sha256:a4ea868bc28109052790eb2b52a9ab33f3aa7adc02f96673526ff47419490e21 \
- --hash=sha256:a9e68c9d88823b274cf1e72f28cb5dc89c990edf430b0bfd3e2fb0785bfeabf4 \
- --hash=sha256:aa9cccf4a44b9b62d8ba8b4dd06c649ba683e4bf04eea606d2e94cfc2d6ff4d6 \
- --hash=sha256:ab30e5e3e706e3063bc6de96b118688cb10396b70bb9864a430f67df98c61ecc \
- --hash=sha256:ac2393c73378fea4e52aa56285a3d64be50f1a12395afef9cce47772f60334c2 \
- --hash=sha256:ad8faf8df23f0378c6d527d8b0b15ea4a2e23c89376877c598c4870d1b2c7866 \
- --hash=sha256:b35b200d6a71b9839a46b9b7fff66b6638bb52fc9658aa58796b0326595d3021 \
- --hash=sha256:b3694e3f87f8ac7ce279d4355645b3c878d24d1424581b46282f24b92f5a4ae2 \
- --hash=sha256:b4ff1d35e8c5bd078be89349b6f3a845128e685e751b6ea1169cf2160b344c4d \
- --hash=sha256:bbc8c8650c6e51041ad1be191742b8b421d05bbd3410f43fa2a00c8db87678e8 \
- --hash=sha256:bc72863f4d9aba2e8fd9085e63548a324ba706d2ea2c83b260da08a59b9482de \
- --hash=sha256:bf625105bb9eef28a56a943fec8c8a98aeb80e7d7db99bd3c388137e6eb2d237 \
- --hash=sha256:c2274ca724536f173122f36c98ce188fd24ce3dad886ec2b7af859518ce008a4 \
- --hash=sha256:c45a03a4c69820a399f1dda9e1d8fbf3562eda46e7720458180302021b08f778 \
- --hash=sha256:c8ae56368f8cc97c7e40a7ee18e1cedaf8e780cd8bc5ed5ac8b81f238614facb \
- --hash=sha256:c907cdc8109f6c619e6254212e794d6548373cc40e1ec75e6e3823d9135d29cc \
- --hash=sha256:ca0276464d148c72defa8bb4390cce01b4a0e425f3b50d1435aa6d7a18107602 \
- --hash=sha256:cd5e2801c89992ed8c0a3f0293ae83c159a60d9a5d685005383ef4caca77f2c4 \
- --hash=sha256:d08ec48f0a1c48d75d0356cea971921848fb620fdeba805b28f937e90691209f \
- --hash=sha256:d1a2ee9c1499fc8f86f4521f27a973c914b211ffa87322f4ee33bb35392da2c5 \
- --hash=sha256:d5f5d1e9def3405f60e3ca8232d56f35c98fb7bf581efcc60051ebf53cb8b611 \
- --hash=sha256:d60377dce4511655582e300dc1e5a5f24ba0cb229005a1d5c8d0cb72bb758ab8 \
- --hash=sha256:d73beaac5e90173ac3deb9928a74763a6d230f494e4bfb422c217a0ad8e629bf \
- --hash=sha256:d7de2637729c67d67cf87614b566626057e95c303bc0a55ffe391f5205e7003d \
- --hash=sha256:dad6e0f2e481fffdcf776d10ebee25e0ef89f16d691f1e5dee4b586375fdc64b \
- --hash=sha256:dda86aba335c902b6149a02a55b38e96287157e609200811837678214ba2b1db \
- --hash=sha256:df01808ee470038c3f8dc4f48620df7225c49c2d6639e38f96e6d6ac6e6f7b0e \
- --hash=sha256:e1f6e2f00a6b8edb562826e4632e26d063ac10307e80f7461f7de3ad8ef3f077 \
- --hash=sha256:e25369dc110d58ddf29b949377a93e0716d72a24f62bad72b2b39f155949c1fd \
- --hash=sha256:e3c701e954abf6fc03a49f7c579cc80c2c6cc52525340ca3186c41d3f33482ef \
- --hash=sha256:e5bcc1a1ae744e0bb59641171ae53743760130600da8db48cbb6e4918e186e4e \
- --hash=sha256:e68c14b04827dd76dcbd1aeea9e604e3e4b78322d8faf2f8132c7138efa340a8 \
- --hash=sha256:e8aeb10fcbe92767f0fa69ad5a72deca50d0dca07fbde97848997d778a50c9fe \
- --hash=sha256:e985a16ff513596f217cee86c21371b8cd011c0f6f056d0920aa2d926c544058 \
- --hash=sha256:ecbbd45615a6885fe3240eb9db73b9e62518b611850fdf8ab08bd56de7ad2b17 \
- --hash=sha256:ee4ec14bc1680d6b0afab9aea2ef27e26d2024f18b24a2d7155a52b60da7e833 \
- --hash=sha256:ef5960d965e67165d75b7c7ffc60a83ec5abfc5c11b764ec13ea54fbef8b4421 \
- --hash=sha256:f0cdaecd4c953bfae0b6bb64910aaaca5a424ad9c72d85cb88417bb9814f7550 \
- --hash=sha256:f1ce721c8a7dfec21fcbdfe04e8f68174183cf4e8188e0645e92aa23985c57ff \
- --hash=sha256:f50498891691e0864dc3da965f340fada0771f6142a378083dc4608f4ea513e2 \
- --hash=sha256:f5ea69428fa1b49573eef0cc44a1d43bebd45ad0c611eb7d7eac760c7ae771bc \
- --hash=sha256:f61aa92e4aad0be58eb6eb4e0c21acf32cf8065f4b2cae5665da756c4ceef982 \
- --hash=sha256:f6e4333fb15c83f7d1482a76d45a0818897b3d33f00efd215528ff7c51b8e35d \
- --hash=sha256:f820f24b09e3e779fe84c3c456cb4108a7aa639b0d1f02c28046e11bfcd088ed \
- --hash=sha256:f98059e4fcd3e3e4e2d632b7cf81c2faae96c43c60b569e9c621468082f1d104 \
- --hash=sha256:fcce033e4021347d80ed9c66dcf1e7b1546319834b74445f561d2e2221de5659
-idna==3.18 \
- --hash=sha256:7f952cbe720b688055e3f87de14f5c3e5fdaa8bc3928985c4077ca689de849a2 \
- --hash=sha256:ffb385a7e039654cef1ab9ef32c6fafe283c0c0467bba1d9029738ce4a14a848
+build==1.6.0 \
+ --hash=sha256:bd2c8afc603e7a2e0ce70e2ea85f0a6d02043bafbd307f5bada0f98669eca5af \
+ --hash=sha256:f7aaf1ebbb79178a02ba248bb524f2176b256017e17e8e4bd4289c7b38cc2bad
+certifi==2026.7.22 \
+ --hash=sha256:62f22742b58a1a33014a2b6b706588a8d7e2a88ae7bd1a6ebe8c992928483775 \
+ --hash=sha256:741e2c3b351ddf169a738da9f2c048608ff7f2c5cc02f1ebc6b118bb090d5d55
+charset-normalizer==3.5.1 \
+ --hash=sha256:00668ebb0609751758682eb0b5857e7c35b9f00e84dfdef062e103244ec94d45 \
+ --hash=sha256:012a22b88a77ca2e59b98ac5889b0deb604147666032f45e6d6e217634d2550d \
+ --hash=sha256:01e93745f7f219b703b60ba7afead36cfc4242782be5af484673fc500df12da5 \
+ --hash=sha256:04368edf83514385ffc3e1cfd4546e595f4f1272dd23ba437a93a9cc3741d47b \
+ --hash=sha256:0722590aabf9dc6a6c0343d523c05458fa2b5047dbe6302fd526bb570600753f \
+ --hash=sha256:07ffd07412fc5d5e84cd8952acf9ff7e4ed7a708e69d1bada19d8ba91711353f \
+ --hash=sha256:09a7bba9f739468c8e78c36a75c33768e53cb1959fc638f510454c14683f00d5 \
+ --hash=sha256:0b2b1b3fa5670c127b246df1d0c059defd41f689a868a3b9d79df9b1cac42d22 \
+ --hash=sha256:0c6dfb5ca6723eeed15aa8e564a014d69fcb8812f94eef11fe3631e0508199f5 \
+ --hash=sha256:0d929fc574b4d6fd9e7c0f5c2ede8716a41911923aa7fa5fce38e0818aa4a1ac \
+ --hash=sha256:13e3afe97712e8887cd516e960c63f0b93122971e5b5e4b2622fe7701771e838 \
+ --hash=sha256:15f024313246a4ed976c60f440bb8d257815513a681d212ff74fd46f7d715a90 \
+ --hash=sha256:195ce897c6153c0700078142cf8efe3e6454ca4cf4357499e4078dfd83396626 \
+ --hash=sha256:19a3dd5aa73cef1c99687c4fc57db016a9c17104ae1185da88ba566a5d3bebe4 \
+ --hash=sha256:1d1c7a53a6c2103925cdd6d7229f8c567379f211c869793df679f2e9f738c369 \
+ --hash=sha256:1f5883d77fd409a261abb5dc8ccbe335720d798b1de4abb3b1d47ccbbc76b53b \
+ --hash=sha256:21b82d8082f6f5e7f456ef0bd16323d08de1266efbfeb476e64b2a91d1471a4e \
+ --hash=sha256:252d099029bcbea642f2a06c4ed5046bdf8b5a8150b64afa5e027e88b106e5ee \
+ --hash=sha256:256dd4d85d9e4dc595e2bc983c980e73f62ddeb3165c58b4c3dfe78c5c8548c1 \
+ --hash=sha256:26422d45fd13551cf564c58932f7d72b4f58b93b0fcf18c35ba6be12b46bb102 \
+ --hash=sha256:2679de311c7946dde5d3b6f44941844133ff5c7cb86099c0061ab1e8901c20a8 \
+ --hash=sha256:29880d17a8eb0b5cfdfd8944b468322928059aa35f1f5fa8ff22b149ec0b42f8 \
+ --hash=sha256:2bced4061f000f7187254a02ad3433ae17eaf991747ceea2f478422590a5bba9 \
+ --hash=sha256:2e9cf9253119d8e5d111f05d71626786fd3d6193817316eab1ca088cdb8593cf \
+ --hash=sha256:2f06b7eae9dbe77fe1d644ca244dad508de8d302870a43f3c559b521270938a0 \
+ --hash=sha256:2f293479cce755c75f1697e87c409b7ae4c555c7dfecb6e988ad13abba943031 \
+ --hash=sha256:329fc3ccb63ad22d867d84c2adea759a64079a37ba4a343433b02c7a2816871e \
+ --hash=sha256:343fb4f2821043bd87095f7b08a1a181febc8e36ac64212143bbfd0a0e1bc235 \
+ --hash=sha256:3588e376b3ea2eea84976f67273d679f229e24c66dce7b82ae45aef04ff6e072 \
+ --hash=sha256:35aea775dc2bd5f54cd84a1cd2696cc3207c479cb9cf0bd346f0d343e4300ddb \
+ --hash=sha256:35fe081843b35aad20ffeccec3eeffbe637b15d14f3fb22cc1b59cd8ec17e93c \
+ --hash=sha256:36047af20e17097c3bb9476c2b7655f2f7aa51322c0ba58c07695bedf755a950 \
+ --hash=sha256:3617ac3cfd8b9888f145ad89dd6e692285834b0201c6074a5eeaad3fd4d668c2 \
+ --hash=sha256:366ec70f5547c640d3ce1985722490f23faf4eb5216a7eeba78277490e78dacb \
+ --hash=sha256:394fea06235c8543390050ed5f529187074b029fb027213f6c46ac11ab5d950e \
+ --hash=sha256:3d27167433c0d5f18dc850f07d0b3816221984fecdc405d6c157a6f0b8f8e9e6 \
+ --hash=sha256:3e5e1224c0a6a90e05843e07adfec669edebec17801c67072f51e59561d63c0b \
+ --hash=sha256:41876ee62a3dddf48ff1121ad8f0798032aa03f2fd35f21f34a4cab14f18d8d2 \
+ --hash=sha256:433c5a81eade63b47e522303bad236f59dba55ea6951746f5558355eeed8c75d \
+ --hash=sha256:4582c27e8c889d64811987b5967fbd3ae0c823fe1fd933b543d55ac20bb475fa \
+ --hash=sha256:485a0d363cafefcd2538a73c7c838daa2035f09b2c9f9b5e3133f80c6aeb84c2 \
+ --hash=sha256:494b70049a4d69aec6e8137c13af4cf8db8c9f9820a1392ac293b0dd2987a818 \
+ --hash=sha256:496846868fea80e479324862fa877f02411f2fd0f83b79ccee2607aa68b2a032 \
+ --hash=sha256:4abdc5f9ad448c1ecbfae2974b820535d6bc6e7eef63babbab3d81cf46968c71 \
+ --hash=sha256:4b599739b93b2cbeded49645ae3c8d1405c29ddfbceac1545c87a3f9580a9e96 \
+ --hash=sha256:4bea7f8ebe90bbd7f0e4a2de42ca6924ba23e3e76418c408ff82f1d46fabd687 \
+ --hash=sha256:4c4fb141a727957c93edfe5c32a26ceb6b5f6461d67146e2d39f51e16170bea8 \
+ --hash=sha256:4c9548dc78002099910abaebc0a72ac58b7d30931869e0351c09b507dff4ece3 \
+ --hash=sha256:4d26f14f041e83dd8edfd61f4cd4fa7285d31798b5bf1f28e70c367ba6c41d61 \
+ --hash=sha256:4f298bdadb8f0b9e5672877f647d1be9373ef5320c9e2f049795e26cad28b6a9 \
+ --hash=sha256:52ec005752a56ae79547a05c0139ca2501a0c866390b6115008456b9f0e7cde1 \
+ --hash=sha256:55261ac0d2941c42f196dd576f543d87a8ee03cd6f5e30dfb4d807b2e3b9121a \
+ --hash=sha256:56490c595a28b1bb27dfc583e816152a9767721ef58b2c03b13f954d2f707420 \
+ --hash=sha256:58d3e12c88e0950bca850ae1f7c256055c097639c2edb9eb123af9807d8b15e4 \
+ --hash=sha256:58d4aa13a59c969dbfdf9e6a9560e242cbfd9e8a8f50c2747714df1a423adf65 \
+ --hash=sha256:59171c6e45bf07d0d5cab3b0bf81d945035530f6873398b3b531c31184d46663 \
+ --hash=sha256:5b6d1386bf0096d26d3a863dc0a487a5b4eb9aa93cf5ba69683d29dde6b9d60f \
+ --hash=sha256:5c0ea61a470e070686aa30892fed79e297d2c8d0ab46b8bcdf027d38c51da591 \
+ --hash=sha256:5c84bec0ab5ae0c64bfe73a7d2adcb5ce73b467523fc27fd6a28ab2aa6cbe35a \
+ --hash=sha256:5ca0555312ae2fe82715cada7fac375530c2f3349e1eaa1bcb33d0283ac79a18 \
+ --hash=sha256:5d8531a6569d025f68e2321e7638fb7978f23db58e5f69f56913837aae03816e \
+ --hash=sha256:5e2d0e146dcb57034f8b97dc58d2d512cb90aba253960ce449f695fec6a82c6f \
+ --hash=sha256:5fc45d653ea8c9a20479167e11d4a0f8cb2fa3470737ab6f9c827532313187b7 \
+ --hash=sha256:6117b84ea48435e5356dc737f5121485c30920ba43375fa7b434fd753df0eac3 \
+ --hash=sha256:6199d5606e2bbf2b096cf64d03f8b6790c91081d5ac866b8e7bb6422738cc60c \
+ --hash=sha256:62b55f6722735a6c472f88361cde6640608773d9443cebdbb51abf436a1fcdd3 \
+ --hash=sha256:687c9ca3035544b113bea2055e180af96fb63c0c476e22a9180f51925186e7b7 \
+ --hash=sha256:6b7430cf5728e68f6c462254009a6ef4086e1bea43cf2f57aa9c55fb4f50ff96 \
+ --hash=sha256:6ba32c4d2abf1d2fe7cf27d280f4cca5664233b0f885549c7761719eb977f486 \
+ --hash=sha256:6c9cdde8becb25a7fde49924511aa2644d6f8081cc8df8e9452724303348d8e3 \
+ --hash=sha256:6df0ec430f9a831772c23ca5a224cba36517a58a84bb32c32bb59a9fa67c47f6 \
+ --hash=sha256:6e2912d4babbc65196ac13c2f53468dc57fb8b9c25ef913e8c59ddf7c6dc0e1b \
+ --hash=sha256:6e5e4d73d588ca5ed09df1b7dcd1b203d1df3c542e3f50d126c947d432b10731 \
+ --hash=sha256:70055ff39b97c99e7ae40ea3e393fb62aa2e44dbd9b29f8d14f42fb0025c3959 \
+ --hash=sha256:706bfd38730a5ac7a365793269a00f4e988178cec121391f4248d84ad8c972e9 \
+ --hash=sha256:7235dc28fc6dd9d832ac7c7bce95367dedb85929f17368a0c2bee1e080b9acbf \
+ --hash=sha256:774d157f112367ff4abd29019f38f023c24e00e56edc7829c20e358a5a913ad8 \
+ --hash=sha256:77efcff2b23071c349402ac1066667a3d011f62398d81408c9b88ad991747c9e \
+ --hash=sha256:789b8982559ae28dad2356519f841655756cdcd96616410590ae0b17454ee64f \
+ --hash=sha256:7ac76cf9afd34929d76eb7fcb63be476a4853d8a96f0dcf2d0db68a0cbdf9885 \
+ --hash=sha256:7c0c10730342b0c9b35dd1d619beb8214e520bd96a1f870f452680b238aab3e0 \
+ --hash=sha256:823f82903d189af463d7df250ef1f7f696f3cee08cc8d91deb565e8d425f6506 \
+ --hash=sha256:838648accb3a7fd9803fd45c87bce8509648eb0c11bc34e216141300977244f2 \
+ --hash=sha256:854066be00447fa8de2ccbbe893e2ffc4b123ef16d897af794c1e18bd4a714b0 \
+ --hash=sha256:85d5855daafc240cc045c026d7a15fd198a09b0fc8ff6f5ecbb5297b509cb11e \
+ --hash=sha256:85de3134b5379856e323ba37c19c9256d39425f7b76a63af52b09fb4664c2e8f \
+ --hash=sha256:87e4f41d375c0b9be2fb5251aee4b8a689169e134535aed81bf085c3b647451e \
+ --hash=sha256:88ca277405c2d3b71c4e1c2ee0e7966e807bcba86a69d11e19ba199d18ae4491 \
+ --hash=sha256:88e85ab89cb822c1e635f51d6d32e488f94e002e70e2f492bdb8b945543f345a \
+ --hash=sha256:8ac8c94b6539074e0f40899301273ac8402b9b3e01c7b7ba269ff30340aaaf20 \
+ --hash=sha256:8fe532b3c966d1fb794e0698e4589d0444017ae77fc0b31edea13c0e35bcc449 \
+ --hash=sha256:9085f87b0e38a2b92b8923059b4e8789fe40d9279712d15dcc670048d77079af \
+ --hash=sha256:90b7481fb62fbe172c558bc6fd1c4c98d82004a54a7551f20e11ac9bf0b8708c \
+ --hash=sha256:92caef967d287a407085d61176fce4012b1dd62daed4eb6d5ceb26d3d2538712 \
+ --hash=sha256:9362dd90aa7dab48c0054a21187791ccf05473f7dba5d92b8033ae62164675e7 \
+ --hash=sha256:94d78ecec2605a8d0398b0f365d5f12a63248438516f5dac536a5eff7337df4a \
+ --hash=sha256:94fbf1c0c6cc0d3d5e50f9a9313a8cdca90dd696d34b381cd1704f8c9e939f20 \
+ --hash=sha256:950f23cb393f85543777b0433f082cddd25b51ab398eac7971146495679efe5f \
+ --hash=sha256:96eefc178f8636b9c760c5829345307fd81cfae9ab1e80997dbddeb0f54ee9a3 \
+ --hash=sha256:96fef3e886d6a9874b14f27fc193fbdc69d5d8035783d86aa4e1cea594e695f9 \
+ --hash=sha256:977cdbd483a9cff38179bea4fd754289a6f2195c7abd414aba85410b3e66cc5e \
+ --hash=sha256:978eab16f55b4ab2c2a745be9a0a840bf8f09a7f227d9c76eb30214d078865a5 \
+ --hash=sha256:994e883d17c559cdfd38c84003c8b27d25424a1077272a17e7cd27bfe0bf57b2 \
+ --hash=sha256:9ac4444d8d4fd4c4bd08bf451ed3167aa9e7ec6cdb41b648794f1d1103652e36 \
+ --hash=sha256:9b5db6052055d34d41230fb78d7c439c23dc536a9896f6cb039e8dd92cfc1263 \
+ --hash=sha256:9d9a0dc7cbe9bec24c3f767c9122c41fe5a1bc43f47cd099d00d393e09769de4 \
+ --hash=sha256:9dbdd9205662134957cf0c324f639bdc5031c0ca056e2369e238db75187c0f11 \
+ --hash=sha256:9eea3ab2597a5e65fe65296e2d6a84570845a6b55532d90333d740d48bbc850a \
+ --hash=sha256:a2028475ba855475b8b4d3cfeb4994269c967aea8b9892dfba907f4263a863a3 \
+ --hash=sha256:a3a370082ce34d0612f421e15fe011c53bb1feff21a26d06ad4fb244dab5a375 \
+ --hash=sha256:a545775cfe815855ea32d7c27731d79da358ef2055b4a25830231b1622dd18aa \
+ --hash=sha256:a5cbd90ecf0fc62e64726917ad083b73001f0563657a87ec3c0b504e277dc90d \
+ --hash=sha256:a6d095662e73e74f0a49988e0593373e243e3a52e27bfeea0a859e88acf4a0f5 \
+ --hash=sha256:a6dac12ff6b846103483683f60c5f8fee205121adc58ffd87e90a90a3af69e99 \
+ --hash=sha256:a951ad59cad9145664a730d3036b40b844e74d2d3683da40111463cd3a83845d \
+ --hash=sha256:aa1099b956fb795e686d073568f6dc002a0bb89765ea6d5b055dd7d9bf1b116c \
+ --hash=sha256:aa2bb0b37202dca27175591f761108b5d34096ade1191ffe4808bdf6b1571488 \
+ --hash=sha256:aae2ee51122d3ae968a3837d97dc24a0aeebb0dea23694422cd172bd30017cd6 \
+ --hash=sha256:ab743e9bc90c1f73552ec33e10e3331315acd2c397b36065b591b0181de533cc \
+ --hash=sha256:ac00177c4831ffa650f8609e4bdddd5fe09c03b1c0c47acece7e6ea20421598b \
+ --hash=sha256:ac13b004224fb341e1e25a1ed5e19d32f57cdb2a403e01f003b46f051a550f6f \
+ --hash=sha256:acaf604462bf330b0d07e7a07c1d6e4adac79e5fb13e9c5140590542cafacc00 \
+ --hash=sha256:ae31a1a1db2ee6cc2942fccaf695c934bc7f3db9f2133a3fef1f367cf1a4ab10 \
+ --hash=sha256:ae4a097991662cd4fff0ddc74e0fe7874f82e00042fa0ea00855645ed0c79598 \
+ --hash=sha256:aea996a6aba25260827c9ea511d1addfde2da9eb686ac961838509086188b7e6 \
+ --hash=sha256:b39b69b347e5e47a3b5b8cfc005c68c1ba347474e3960236c4944a8ecd174962 \
+ --hash=sha256:b54e7e13267d49ffbfe68e25b3cbd774dab38fa37238f71265e91b36146eb21c \
+ --hash=sha256:b9af956078716df40d985fb0dfeb2c2120c5ca92ba4ff4b388acfd01cdc14d08 \
+ --hash=sha256:ba2f37ee79e6338845261a3c5b1784e5d1acdff2c0785b284f1b633033d136ab \
+ --hash=sha256:ba501e667c17d8411f98e67a022d9604ef179aff0e459b7e292c796837c13573 \
+ --hash=sha256:baf3775a2635e5a11fbd5e4e64ee69c7e86875d224a5c72aca4c141064589a90 \
+ --hash=sha256:bb57753e36e4855b8ca375069482250a6246372331a3e4f3407eaebb007443f5 \
+ --hash=sha256:bd6c173f04743d483881bffa1478d5a4624475b8cd1d2194956a75548e191c18 \
+ --hash=sha256:be47f99644b208bff7766314013f9acf57b056b04191d570d68ad14022cf5b1d \
+ --hash=sha256:c010f5581d9c612804cc59fcf7b524b707fbcb72828551237ab545bb5c7034af \
+ --hash=sha256:c1dcc36dcb96abc02236e182d17e0f71430152a6c2c7447421da2d2dc144edea \
+ --hash=sha256:c428c6c31eb5f4277d7f8eccaf767fbd548ddd5ce3c8b4f4cbbfab3d96b5904c \
+ --hash=sha256:c658c50ac0c98cd755a2dd50b7977d3bca7df401dcc47fbdfa87db53ef7d4e8b \
+ --hash=sha256:c71fb0d56c920c269cd3e2e3fe7c610e3f1fdb21a6ce60efa6430ff63676cea6 \
+ --hash=sha256:c7b742bf31c88566b4bb6335a7f393bb322e580b6bb98df7bd0c25e6e3519ce8 \
+ --hash=sha256:cc0329df4caaceb950d2f580b5ac716a377f7059624a0bafaeaf8a218c6ed774 \
+ --hash=sha256:cc5d36d96478aa9c60654bd932525bf32964c62a7281eafdf16d85003a8d6004 \
+ --hash=sha256:ce854f5f478050ade5a238731c4ca985a7d3b3cb53ff600a9b5c3b689b5f0a7a \
+ --hash=sha256:ced3fdd71aaa83ce593746c2edb42b7a59cb4c19c8b5c407781c72e493aae55a \
+ --hash=sha256:cee5dd7c6fb5dd52a0fe2a740f9bc6e3593f5f8b1788bde49de02086f30182b2 \
+ --hash=sha256:cfa1c0cc3a8f9f53f1243a5a99ac36fd003880199383b37672e86ddda9cb07e2 \
+ --hash=sha256:d1ee1e296209fdce05b81b663250eefa02213a2da7b41bf26f7829b8ba3545aa \
+ --hash=sha256:d59b75732e9b6f27388e10c14b0259cc5f2e48c78627d185e6a177b58ad3cffe \
+ --hash=sha256:d63600d620ad0064c3a748b950ac5ea38a80190e5498532efefa4b7b3f1da1f3 \
+ --hash=sha256:dd732602a7009217f658d5863d12d79d373a4de0eebc111094bcdd3bb8e0a6cc \
+ --hash=sha256:e06efa066f7dbadbc84ebc126a97c452a6451dfcf589d89d788484949e1cf795 \
+ --hash=sha256:e199fb99720074809a7720f1c0b4d919eea8b87e88713e0f8f602f7bef543d9d \
+ --hash=sha256:e4b018dc5a0eee4676e38fe84a47a427816c590b93b55d9025274ec4d6ffc2dc \
+ --hash=sha256:e6621fb2a4988d6e53eedc455e5903e2679f3967b8acb3d639f1b63c14a2e893 \
+ --hash=sha256:e71c909f353863b2b89c83de2ebed71ea6d0df8a6ef65a128193c5e650766bef \
+ --hash=sha256:e90251c0c7bdd54a100a0dce3c07b7e637278c93af29dbf78ebb89a58c4bac7d \
+ --hash=sha256:e9fbdce1e47394b09bc9f26ab117dfc8d6491977a11d86f592bb42c779db2fda \
+ --hash=sha256:eb12fb2ba69ffa05f8695f61c69e591dc4b4a12ac3757ac8af8adb259bf56d17 \
+ --hash=sha256:eda059b6bc8bc0812d626fd91a7ce01bf583df0a61296eff390fd94141a34e30 \
+ --hash=sha256:f03ac127268b43ef4fe9e6ab6794a6794b49485a0cc0c1db79876d2f33f75bc7 \
+ --hash=sha256:f298e218441525d3794428b4c8b8fb8662c6d3ea79925d4807ee6b9a96a3bca5 \
+ --hash=sha256:f5542f9b941279d82d41eb0aa9f98eba36fe4df5c7086c651df7944935b37182 \
+ --hash=sha256:f6f7deae3feb4edfa2efaf7c574fe88cbf055038a6abdb40188e4fff66d5699f \
+ --hash=sha256:f9b1e28d0e8dbfa858abdba91d6b547beaf2df1a59bec6da6faae7b96a4991a9 \
+ --hash=sha256:f9f8405c2c758532c74fed975dbee57be1f31a6e865c031870c79a6ed3212ada \
+ --hash=sha256:fa48b1b63d639f9483e0633e092f5851e2348c352f1f9bb6c8182f87884ef876 \
+ --hash=sha256:fb78f6e7fcd8ad785d28cd577168bc1aaee827b25bb8755638f694794ea98f0a \
+ --hash=sha256:fbc597639158fd7c14d55e808718848319540f51b0e6746e3eefa59723a4a348 \
+ --hash=sha256:fce8cbd4997efeb450bd298b54f755dcdff18d496f7a5ddbb4867c6d7c88fdc3 \
+ --hash=sha256:fd0350afdc3aabd5576f60ea109228bd5538139713c7b094c5cd27c73a98bc6f \
+ --hash=sha256:fd0a274c0e5f9a21565cd9d3dd749b61f96b7aa1e20a93aa1ba4029518f2e5c0 \
+ --hash=sha256:fdb8a068947befafba9952162645dc2fecaeb400e64584829ed5e9b2fbe21a7f
+idna==3.19 \
+ --hash=sha256:5e0811a4383b21dc5838069f801c4fb62113b7447663d2530d2bd6e77b49bf15 \
+ --hash=sha256:815e7be7a7806d54abb586dc943addc79e8b2ee16915059658cbeff4b1b43bf4
inotify==0.2.12 \
--hash=sha256:9aee407f92c7d51a2ce50f3b78291a9094e334e34bd68e82bf60020795fa2c94 \
--hash=sha256:e4f1c8ec7ba5ec2a1a7fce48c0c917234af9d756495ebae7ffa00e41a305ab90
-packaging==26.0 \
- --hash=sha256:00243ae351a257117b6a241061796684b084ed1c516a08c48a3f7e147a9d80b4 \
- --hash=sha256:b36f1fef9334a5588b4166f8bcd26a14e521f2b55e6b9de3aaa80d3ff7a37529
+packaging==26.3 \
+ --hash=sha256:94edc256424af38762eb31306eed28beb9f0efc50a8837492c9d6fd6004aed79 \
+ --hash=sha256:d7193f7c8e4e93f444fde0262bf90af30e16fa0ad0ad44cb553c87339b23cd1c
pyproject-hooks==1.2.0 \
--hash=sha256:1e859bd5c40fae9448642dd871adf459e5e2084186e8d2c2a79a824c970da1f8 \
--hash=sha256:9e5c6bfa8dcc30091c74b0cf803c81fdd29d94f01992a7707bc97babb1141913
@@ -220,73 +263,73 @@ pyyaml==6.0.3 \
--hash=sha256:f7057c9a337546edc7973c0d3ba84ddcdf0daa14533c2065749c9075001090e6 \
--hash=sha256:fa160448684b4e94d80416c0fa4aac48967a969efe22931448d853ada8baf926 \
--hash=sha256:fc09d0aa354569bc501d4e787133afc08552722d3ab34836a80547331bb5d4a0
-requests==2.33.1 \
- --hash=sha256:18817f8c57c6263968bc123d237e3b8b08ac046f5456bd1e307ee8f4250d3517 \
- --hash=sha256:4e6d1ef462f3626a1f0a0a9c42dd93c63bad33f9f1c1937509b8c5c8718ab56a
-tomli==2.4.0 \
- --hash=sha256:0408e3de5ec77cc7f81960c362543cbbd91ef883e3138e81b729fc3eea5b9729 \
- --hash=sha256:0dc56fef0e2c1c470aeac5b6ca8cc7b640bb93e92d9803ddaf9ea03e198f5b0b \
- --hash=sha256:0e0fe8a0b8312acf3a88077a0802565cb09ee34107813bba1c7cd591fa6cfc8d \
- --hash=sha256:0f2e3955efea4d1cfbcb87bc321e00dc08d2bcb737fd1d5e398af111d86db5df \
- --hash=sha256:133e93646ec4300d651839d382d63edff11d8978be23da4cc106f5a18b7d0576 \
- --hash=sha256:1b168f2731796b045128c45982d3a4874057626da0e2ef1fdd722848b741361d \
- --hash=sha256:1c8a885b370751837c029ef9bc014f27d80840e48bac415f3412e6593bbc18c1 \
- --hash=sha256:1f776e7d669ebceb01dee46484485f43a4048746235e683bcdffacdf1fb4785a \
- --hash=sha256:1fb2945cbe303b1419e2706e711b7113da57b7db31ee378d08712d678a34e51e \
- --hash=sha256:20cedb4ee43278bc4f2fee6cb50daec836959aadaf948db5172e776dd3d993fc \
- --hash=sha256:20ffd184fb1df76a66e34bd1b36b4a4641bd2b82954befa32fe8163e79f1a702 \
- --hash=sha256:26ab906a1eb794cd4e103691daa23d95c6919cc2fa9160000ac02370cc9dd3f6 \
- --hash=sha256:2add28aacc7425117ff6364fe9e06a183bb0251b03f986df0e78e974047571fd \
- --hash=sha256:2b1e3b80e1d5e52e40e9b924ec43d81570f0e7d09d11081b797bc4692765a3d4 \
- --hash=sha256:31d556d079d72db7c584c0627ff3a24c5d3fb4f730221d3444f3efb1b2514776 \
- --hash=sha256:36b9d05b51e65b254ea6c2585b59d2c4cb91c8a3d91d0ed0f17591a29aaea54a \
- --hash=sha256:39b0b5d1b6dd03684b3fb276407ebed7090bbec989fa55838c98560c01113b66 \
- --hash=sha256:3cf226acb51d8f1c394c1b310e0e0e61fecdd7adcb78d01e294ac297dd2e7f87 \
- --hash=sha256:3d895d56bd3f82ddd6faaff993c275efc2ff38e52322ea264122d72729dca2b2 \
- --hash=sha256:413540dce94673591859c4c6f794dfeaa845e98bf35d72ed59636f869ef9f86f \
- --hash=sha256:43e685b9b2341681907759cf3a04e14d7104b3580f808cfde1dfdb60ada85475 \
- --hash=sha256:4cbcb367d44a1f0c2be408758b43e1ffb5308abe0ea222897d6bfc8e8281ef2f \
- --hash=sha256:551e321c6ba03b55676970b47cb1b73f14a0a4dce6a3e1a9458fd6d921d72e95 \
- --hash=sha256:5572e41282d5268eb09a697c89a7bee84fae66511f87533a6f88bd2f7b652da9 \
- --hash=sha256:5aa48d7c2356055feef06a43611fc401a07337d5b006be13a30f6c58f869e3c3 \
- --hash=sha256:5b5807f3999fb66776dbce568cc9a828544244a8eb84b84b9bafc080c99597b9 \
- --hash=sha256:5e3f639a7a8f10069d0e15408c0b96a2a828cfdec6fca05296ebcdcc28ca7c76 \
- --hash=sha256:685306e2cc7da35be4ee914fd34ab801a6acacb061b6a7abca922aaf9ad368da \
- --hash=sha256:75c2f8bbddf170e8effc98f5e9084a8751f8174ea6ccf4fca5398436e0320bc8 \
- --hash=sha256:7b438885858efd5be02a9a133caf5812b8776ee0c969fea02c45e8e3f296ba51 \
- --hash=sha256:7d49c66a7d5e56ac959cb6fc583aff0651094ec071ba9ad43df785abc2320d86 \
- --hash=sha256:7d6d9a4aee98fac3eab4952ad1d73aee87359452d1c086b5ceb43ed02ddb16b8 \
- --hash=sha256:84d081fbc252d1b6a982e1870660e7330fb8f90f676f6e78b052ad4e64714bf0 \
- --hash=sha256:8768715ffc41f0008abe25d808c20c3d990f42b6e2e58305d5da280ae7d1fa3b \
- --hash=sha256:920b1de295e72887bafa3ad9f7a792f811847d57ea6b1215154030cf131f16b1 \
- --hash=sha256:9a08144fa4cba33db5255f9b74f0b89888622109bd2776148f2597447f92a94e \
- --hash=sha256:a26d7ff68dfdb9f87a016ecfd1e1c2bacbe3108f4e0f8bcd2228ef9a766c787d \
- --hash=sha256:aa89c3f6c277dd275d8e243ad24f3b5e701491a860d5121f2cdd399fbb31fc9c \
- --hash=sha256:b5ef256a3fd497d4973c11bf142e9ed78b150d36f5773f1ca6088c230ffc5867 \
- --hash=sha256:b6c78bdf37764092d369722d9946cb65b8767bfa4110f902a1b2542d8d173c8a \
- --hash=sha256:bbb1b10aa643d973366dc2cb1ad94f99c1726a02343d43cbc011edbfac579e7c \
- --hash=sha256:c084ad935abe686bd9c898e62a02a19abfc9760b5a79bc29644463eaf2840cb0 \
- --hash=sha256:c73add4bb52a206fd0c0723432db123c0c75c280cbd67174dd9d2db228ebb1b4 \
- --hash=sha256:cae9c19ed12d4e8f3ebf46d1a75090e4c0dc16271c5bce1c833ac168f08fb614 \
- --hash=sha256:d20b797a5c1ad80c516e41bc1fb0443ddb5006e9aaa7bda2d71978346aeb9132 \
- --hash=sha256:d3d1654e11d724760cdb37a3d7691f0be9db5fbdaef59c9f532aabf87006dbaa \
- --hash=sha256:d878f2a6707cc9d53a1be1414bbb419e629c3d6e67f69230217bb663e76b5087
-tornado==6.5.7 \
- --hash=sha256:148b2eb15c2c765a50796172c1e499649b35f30d2e3c3d3e15913cfa56bfb163 \
- --hash=sha256:66c513a76cda70d53907bc27cf1447557699c2e95aa48ba27a442ff61c3ddfc2 \
- --hash=sha256:7778b30bef919231265e91c69963ce0f49a1e9c07ac900bbe75b19ce2575ba92 \
- --hash=sha256:8a46347a18f23fb92b396beebe0fb78f61dda0cc302445202c16203d8a18848b \
- --hash=sha256:8d759e71906ee783f8867b93bf26a265743da4c1e2f4a018464c1ba019862972 \
- --hash=sha256:9da38de27f1da3b78a966f0dae12b5a1ea9afe72ca805d84ff06508272ddf100 \
- --hash=sha256:de942f843533a039ef9fa3d9c88c7cd8a7c94553fb5ad0154270989b3d99a2c4 \
- --hash=sha256:e726f0c75da7726eec023aa62751ff8878bd2737e34fbdd33b1ae5897d2200f5 \
- --hash=sha256:f8de3bf12d3efdd0cbe7c8887868198f8a91415e3f29fcf258d9b8eb7b1d9ae4 \
- --hash=sha256:ff934fce95643af5f11efdae618eaa73d469dc588641e5c8d19295a0c65c4796
+requests==2.34.2 \
+ --hash=sha256:2a0d60c172f83ac6ab31e4554906c0f3b3588d37b5cb939b1c061f4907e278e0 \
+ --hash=sha256:f288924cae4e29463698d6d60bc6a4da69c89185ad1e0bcc4104f584e960b9ed
+tomli==2.4.1 \
+ --hash=sha256:01f520d4f53ef97964a240a035ec2a869fe1a37dde002b57ebc4417a27ccd853 \
+ --hash=sha256:0d85819802132122da43cb86656f8d1f8c6587d54ae7dcaf30e90533028b49fe \
+ --hash=sha256:136443dbd7e1dee43c68ac2694fde36b2849865fa258d39bf822c10e8068eac5 \
+ --hash=sha256:1d8591993e228b0c930c4bb0db464bdad97b3289fb981255d6c9a41aedc84b2d \
+ --hash=sha256:2190f2e9dd7508d2a90ded5ed369255980a1bcdd58e52f7fe24b8162bf9fedbd \
+ --hash=sha256:2c1c351919aca02858f740c6d33adea0c5deea37f9ecca1cc1ef9e884a619d26 \
+ --hash=sha256:36d2bd2ad5fb9eaddba5226aa02c8ec3fa4f192631e347b3ed28186d43be6b54 \
+ --hash=sha256:3d48a93ee1c9b79c04bb38772ee1b64dcf18ff43085896ea460ca8dec96f35f6 \
+ --hash=sha256:47149d5bd38761ac8be13a84864bf0b7b70bc051806bc3669ab1cbc56216b23c \
+ --hash=sha256:4ab97e64ccda8756376892c53a72bd1f964e519c77236368527f758fbc36a53a \
+ --hash=sha256:4b605484e43cdc43f0954ddae319fb75f04cc10dd80d830540060ee7cd0243cd \
+ --hash=sha256:504aa796fe0569bb43171066009ead363de03675276d2d121ac1a4572397870f \
+ --hash=sha256:51529d40e3ca50046d7606fa99ce3956a617f9b36380da3b7f0dd3dd28e68cb5 \
+ --hash=sha256:52c8ef851d9a240f11a88c003eacb03c31fc1c9c4ec64a99a0f922b93874fda9 \
+ --hash=sha256:559db847dc486944896521f68d8190be1c9e719fced785720d2216fe7022b662 \
+ --hash=sha256:5a881ab208c0baf688221f8cecc5401bd291d67e38a1ac884d6736cbcd8247e9 \
+ --hash=sha256:5cb41aa38891e073ee49d55fbc7839cfdb2bc0e600add13874d048c94aadddd1 \
+ --hash=sha256:5e262d41726bc187e69af7825504c933b6794dc3fbd5945e41a79bb14c31f585 \
+ --hash=sha256:5ee18d9ebdb417e384b58fe414e8d6af9f4e7a0ae761519fb50f721de398dd4e \
+ --hash=sha256:7008df2e7655c495dd12d2a4ad038ff878d4ca4b81fccaf82b714e07eae4402c \
+ --hash=sha256:734e20b57ba95624ecf1841e72b53f6e186355e216e5412de414e3c51e5e3c41 \
+ --hash=sha256:7c7e1a961a0b2f2472c1ac5b69affa0ae1132c39adcb67aba98568702b9cc23f \
+ --hash=sha256:7f86fd587c4ed9dd76f318225e7d9b29cfc5a9d43de44e5754db8d1128487085 \
+ --hash=sha256:7f94b27a62cfad8496c8d2513e1a222dd446f095fca8987fceef261225538a15 \
+ --hash=sha256:88dceee75c2c63af144e456745e10101eb67361050196b0b6af5d717254dddf7 \
+ --hash=sha256:8a650c2dbafa08d42e51ba0b62740dae4ecb9338eefa093aa5c78ceb546fcd5c \
+ --hash=sha256:8d65a2fbf9d2f8352685bc1364177ee3923d6baf5e7f43ea4959d7d8bc326a36 \
+ --hash=sha256:96481a5786729fd470164b47cdb3e0e58062a496f455ee41b4403be77cb5a076 \
+ --hash=sha256:a120733b01c45e9a0c34aeef92bf0cf1d56cfe81ed9d47d562f9ed591a9828ac \
+ --hash=sha256:b1d22e6e9387bf4739fbe23bfa80e93f6b0373a7f1b96c6227c32bef95a4d7a8 \
+ --hash=sha256:b8c198f8c1805dc42708689ed6864951fd2494f924149d3e4bce7710f8eb5232 \
+ --hash=sha256:c2541745709bad0264b7d4705ad453b76ccd191e64aa6f0fc66b69a293a45ece \
+ --hash=sha256:c742f741d58a28940ce01d58f0ab2ea3ced8b12402f162f4d534dfe18ba1cd6a \
+ --hash=sha256:c7f2c7f2b9ca6bdeef8f0fa897f8e05085923eb091721675170254cbc5b02897 \
+ --hash=sha256:d312ef37c91508b0ab2cee7da26ec0b3ed2f03ce12bd87a588d771ae15dcf82d \
+ --hash=sha256:d4d8fe59808a54658fcc0160ecfb1b30f9089906c50b23bcb4c69eddc19ec2b4 \
+ --hash=sha256:da25dc3563bff5965356133435b757a795a17b17d01dbc0f42fb32447ddfd917 \
+ --hash=sha256:eab21f45c7f66c13f2a9e0e1535309cee140182a9cdae1e041d02e47291e8396 \
+ --hash=sha256:eb0dc4e38e6a1fd579e5d50369aa2e10acfc9cace504579b2faabb478e76941a \
+ --hash=sha256:ec9bfaf3ad2df51ace80688143a6a4ebc09a248f6ff781a9945e51937008fcbc \
+ --hash=sha256:ede3e6487c5ef5d28634ba3f31f989030ad6af71edfb0055cbbd14189ff240ba \
+ --hash=sha256:f3c6818a1a86dd6dca7ddcaaf76947d5ba31aecc28cb1b67009a5877c9a64f3f \
+ --hash=sha256:f758f1b9299d059cc3f6546ae2af89670cb1c4d48ea29c3cacc4fe7de3058257 \
+ --hash=sha256:f8f0fc26ec2cc2b965b7a3b87cd19c5c6b8c5e5f436b984e85f486d652285c30 \
+ --hash=sha256:fd0409a3653af6c147209d267a0e4243f0ae46b011aa978b1080359fddc9b6cf \
+ --hash=sha256:ff18e6a727ee0ab0388507b89d1bc6a22b138d1e2fa56d1ad494586d61d2eae9 \
+ --hash=sha256:ff2983983d34813c1aeb0fa89091e76c3a22889ee83ab27c5eeb45100560c049
+tornado==6.5.8 \
+ --hash=sha256:11881db6b7c168494be2c2d12e65931451bdf7ee718535418ae1d8855dd5a0ee \
+ --hash=sha256:547d63f450d570c14fe0e8db2cfb14c9bbd1c2503b4a6612586267955aa47b58 \
+ --hash=sha256:5d242290bdf7ab3151bc1065fdd75c0dcc21cbc7b49f22a4c56329c2d6566d22 \
+ --hash=sha256:67832909c4779c64942380cb5f044a5c6163d00831472d80e25e115de9917836 \
+ --hash=sha256:68a7468c7e289f8514d7d664101753903217eff1bb6822c6b5994a0b5f5bcb26 \
+ --hash=sha256:7b94ff0e128fe0542f3bd331fb44d06260fc4ac16881545159f34ef08aad4195 \
+ --hash=sha256:7e2360a0ffbe145eca8af0b19cb7203d79b1a98dd4cccdd6b368f6f49c2e3808 \
+ --hash=sha256:9452e1b208a8bd771e2cb1f2ff564985b9b214bdebbe622793e1799e0a6bd23f \
+ --hash=sha256:9715b5eb79735b2bcd454ce216a9275b7c0470e64ea1bf5742f78b2f72b26eeb \
+ --hash=sha256:cc6aa787d7cfab7c3d35189dc7a56fbd2399a569624c730c6b55b3d6531d0403
urllib3==2.7.0 \
--hash=sha256:231e0ec3b63ceb14667c67be60f2f2c40a518cb38b03af60abc813da26505f4c \
--hash=sha256:9fb4c81ebbb1ce9531cce37674bbc6f1360472bc18ca9a553ede278ef7276897
# The following packages are considered to be unsafe in a requirements file:
-pip==26.1.2 \
- --hash=sha256:382ff9f685ee3bc25864f820aa50505825f10f5458ffff07e30a6d96e5715cab \
- --hash=sha256:f49cd134c61cf2fd75e0ce2676db03e4054504a5a4986d00f8299ae632dc4605
+pip==26.2.1 \
+ --hash=sha256:71138adf1f4ca900cdb7d289c21b7494329f2332b6d85f0e1c42108c0384ed3e \
+ --hash=sha256:f6ad667e89a1fe78046c8f13232b247200f5258d7828f3f7883d660878e0813f