Compare commits
203 Commits
| Author | SHA1 | Date |
|---|---|---|
|
|
7d325d23e3 | |
|
|
fb22686a5e | |
|
|
6b143e4e41 | |
|
|
bc20dd19fb | |
|
|
e82349fabf | |
|
|
01df50b440 | |
|
|
a95d85ce07 | |
|
|
813c4295bf | |
|
|
7fc91180fe | |
|
|
f7f53d0a60 | |
|
|
f2efc328f7 | |
|
|
475d551908 | |
|
|
1a2b9b24db | |
|
|
3c19d176b3 | |
|
|
794971ba0f | |
|
|
541b0226cd | |
|
|
5212af7b70 | |
|
|
e138f6c544 | |
|
|
5f0654e5cc | |
|
|
caff4d1ec2 | |
|
|
ceddf331c6 | |
|
|
9c06bfeddb | |
|
|
fe77de5b9f | |
|
|
bcbe8de1e4 | |
|
|
156ee33740 | |
|
|
7d915a7357 | |
|
|
9837c17878 | |
|
|
b20d6eac46 | |
|
|
25e879ec05 | |
|
|
89d49c2e93 | |
|
|
140c05f273 | |
|
|
206adc083c | |
|
|
5fccf8a966 | |
|
|
8be58fa0aa | |
|
|
f1f8213c01 | |
|
|
937ed4af37 | |
|
|
60d748e57d | |
|
|
f7b63f149a | |
|
|
9aaf7fdbd6 | |
|
|
a4b9c4e097 | |
|
|
68e63f9e39 | |
|
|
bc7b9fc69e | |
|
|
1efa5b8eaa | |
|
|
20b7c085b7 | |
|
|
c3496134bc | |
|
|
39eb6571ce | |
|
|
7096ee911c | |
|
|
3d669f1ab4 | |
|
|
8ecd9a6680 | |
|
|
16814acff3 | |
|
|
94cfb7f290 | |
|
|
3e80c0ce3c | |
|
|
d5cadf526a | |
|
|
6306b7ca71 | |
|
|
14c3c86e22 | |
|
|
39c015c0ff | |
|
|
9897b0790f | |
|
|
287868e171 | |
|
|
5344cb99dd | |
|
|
8dedc3474d | |
|
|
1e4489f61b | |
|
|
75023c5f2f | |
|
|
23a2227ae7 | |
|
|
072f78471c | |
|
|
74db9e29ff | |
|
|
dde422703c | |
|
|
814a226eba | |
|
|
6a69197177 | |
|
|
5b4c8b6d0d | |
|
|
c3413a8f10 | |
|
|
beede8a638 | |
|
|
e13090f84b | |
|
|
8fd47a9ec6 | |
|
|
ced30b48f7 | |
|
|
bd0f44fcfd | |
|
|
78aec073c4 | |
|
|
eea04b3656 | |
|
|
afcf13a6f5 | |
|
|
0e1056df19 | |
|
|
f173905c8b | |
|
|
15dbbb5cb1 | |
|
|
0d4c3a4fcf | |
|
|
f196e15f26 | |
|
|
99049d84e1 | |
|
|
8692148c67 | |
|
|
058d8fd990 | |
|
|
8f576b02a8 | |
|
|
a32323d5bd | |
|
|
d590eb6658 | |
|
|
04858a2727 | |
|
|
3f816c0b6c | |
|
|
3ecd5d0744 | |
|
|
098f5315b5 | |
|
|
beb047095f | |
|
|
4624dde770 | |
|
|
97f5c0ca53 | |
|
|
e1b7a16101 | |
|
|
15f56dea30 | |
|
|
934da124f5 | |
|
|
ad2c75021f | |
|
|
0a95bae8a8 | |
|
|
5a90113b15 | |
|
|
58aacd3f74 | |
|
|
b5f0752f41 | |
|
|
939097bdce | |
|
|
46fdd5ceba | |
|
|
09b21992c5 | |
|
|
10b538373b | |
|
|
f34a940c0a | |
|
|
4b867f282b | |
|
|
57bb5e7e8b | |
|
|
2aa43bceab | |
|
|
7239af5048 | |
|
|
27ba0aa92f | |
|
|
25d98c85f3 | |
|
|
2169192492 | |
|
|
a3a860cf3f | |
|
|
9ff41b9706 | |
|
|
44b62cd164 | |
|
|
a179d9120f | |
|
|
14f6f245c8 | |
|
|
3e610a0558 | |
|
|
af6365acfb | |
|
|
49ecac0376 | |
|
|
9251893bd9 | |
|
|
ccd7098f2f | |
|
|
07d182be56 | |
|
|
cca3c14911 | |
|
|
44546a13f2 | |
|
|
4b60bbc9cc | |
|
|
6918d44190 | |
|
|
a48532f6f5 | |
|
|
11a586c133 | |
|
|
31e84f7909 | |
|
|
7e765eb054 | |
|
|
393ff52954 | |
|
|
c9962c9262 | |
|
|
04b6d45ea2 | |
|
|
2854932965 | |
|
|
bcc785fbb5 | |
|
|
c0df72b84a | |
|
|
15cf80abac | |
|
|
b9c24ddafd | |
|
|
c2490e6015 | |
|
|
a914a00ab8 | |
|
|
d3a0a84dd5 | |
|
|
1e551ec9c0 | |
|
|
1fa735d0d9 | |
|
|
62a24d7960 | |
|
|
3d07099cf9 | |
|
|
483e3e9335 | |
|
|
e0e1abd865 | |
|
|
117a9baab8 | |
|
|
497f336d13 | |
|
|
ef8ba557a1 | |
|
|
caf060521d | |
|
|
b8ebc14489 | |
|
|
8a4063086f | |
|
|
a549f44792 | |
|
|
cb9d3dccb8 | |
|
|
ace3ebd03e | |
|
|
9faa4f6133 | |
|
|
97f4951f08 | |
|
|
3b485f719a | |
|
|
c92458b70e | |
|
|
eef8c96d95 | |
|
|
b82d6f95be | |
|
|
b18a30ea1a | |
|
|
35006d7342 | |
|
|
3410d92daa | |
|
|
5a8e211a04 | |
|
|
1e8a48559b | |
|
|
e03111e67c | |
|
|
0c67942d32 | |
|
|
7d2259669b | |
|
|
fb02e5c959 | |
|
|
918b6139ef | |
|
|
81218c5f1b | |
|
|
2c3a2ef6f9 | |
|
|
2fdb970430 | |
|
|
fb2dec9775 | |
|
|
befdb7c661 | |
|
|
cb259065a5 | |
|
|
4b856f1c04 | |
|
|
6461d3fbac | |
|
|
2c13b269e6 | |
|
|
f1176270eb | |
|
|
5b883fed5b | |
|
|
e7376d558e | |
|
|
17d7717515 | |
|
|
7e6df44d86 | |
|
|
01ec36b31d | |
|
|
f4b0767a88 | |
|
|
7e28315595 | |
|
|
b68b41ae7a | |
|
|
6fb4e80837 | |
|
|
7dce60697f | |
|
|
26ae68a2b6 | |
|
|
878e6414da | |
|
|
2beeda7851 | |
|
|
6ec8fa9a23 | |
|
|
1f199de502 | |
|
|
d8a25481c4 |
|
|
@ -2,129 +2,81 @@
|
||||||
|
|
||||||
## Our Pledge
|
## Our Pledge
|
||||||
|
|
||||||
We as members, contributors, and leaders pledge to make participation in our
|
We as members, contributors, and leaders pledge to make participation in our community a harassment-free experience for everyone, regardless of age, body size, visible or invisible disability, ethnicity, sex characteristics, gender identity and expression, level of experience, education, socioeconomic status, nationality, personal appearance, race, caste, color, religion, or sexual identity and orientation.
|
||||||
community a harassment-free experience for everyone, regardless of age, body
|
|
||||||
size, visible or invisible disability, ethnicity, sex characteristics, gender
|
|
||||||
identity and expression, level of experience, education, socioeconomic status,
|
|
||||||
nationality, personal appearance, race, caste, color, religion, or sexual
|
|
||||||
identity and orientation.
|
|
||||||
|
|
||||||
We pledge to act and interact in ways that contribute to an open, welcoming,
|
We pledge to act and interact in ways that contribute to an open, welcoming, diverse, inclusive, and healthy community.
|
||||||
diverse, inclusive, and healthy community.
|
|
||||||
|
|
||||||
## Our Standards
|
## Our Standards
|
||||||
|
|
||||||
Examples of behavior that contributes to a positive environment for our
|
Examples of behavior that contributes to a positive environment for our community include:
|
||||||
community include:
|
|
||||||
|
|
||||||
- Demonstrating empathy and kindness toward other people
|
- Demonstrating empathy and kindness toward other people
|
||||||
- Being respectful of differing opinions, viewpoints, and experiences
|
- Being respectful of differing opinions, viewpoints, and experiences
|
||||||
- Giving and gracefully accepting constructive feedback
|
- Giving and gracefully accepting constructive feedback
|
||||||
- Accepting responsibility and apologizing to those affected by our mistakes,
|
- Accepting responsibility and apologizing to those affected by our mistakes, and learning from the experience
|
||||||
and learning from the experience
|
- Focusing on what is best not just for us as individuals, but for the overall community
|
||||||
- Focusing on what is best not just for us as individuals, but for the overall
|
|
||||||
community
|
|
||||||
|
|
||||||
Examples of unacceptable behavior include:
|
Examples of unacceptable behavior include:
|
||||||
|
|
||||||
- The use of sexualized language or imagery, and sexual attention or advances of
|
- The use of sexualized language or imagery, and sexual attention or advances of any kind
|
||||||
any kind
|
|
||||||
- Trolling, insulting or derogatory comments, and personal or political attacks
|
- Trolling, insulting or derogatory comments, and personal or political attacks
|
||||||
- Public or private harassment
|
- Public or private harassment
|
||||||
- Publishing others' private information, such as a physical or email address,
|
- Publishing others' private information, such as a physical or email address, without their explicit permission
|
||||||
without their explicit permission
|
- Other conduct which could reasonably be considered inappropriate in a professional setting
|
||||||
- Other conduct which could reasonably be considered inappropriate in a
|
|
||||||
professional setting
|
|
||||||
|
|
||||||
## Enforcement Responsibilities
|
## Enforcement Responsibilities
|
||||||
|
|
||||||
Community leaders are responsible for clarifying and enforcing our standards of
|
Community leaders are responsible for clarifying and enforcing our standards of acceptable behavior and will take appropriate and fair corrective action in response to any behavior that they deem inappropriate, threatening, offensive, or harmful.
|
||||||
acceptable behavior and will take appropriate and fair corrective action in
|
|
||||||
response to any behavior that they deem inappropriate, threatening, offensive,
|
|
||||||
or harmful.
|
|
||||||
|
|
||||||
Community leaders have the right and responsibility to remove, edit, or reject
|
Community leaders have the right and responsibility to remove, edit, or reject comments, commits, code, wiki edits, issues, and other contributions that are not aligned to this Code of Conduct, and will communicate reasons for moderation decisions when appropriate.
|
||||||
comments, commits, code, wiki edits, issues, and other contributions that are
|
|
||||||
not aligned to this Code of Conduct, and will communicate reasons for moderation
|
|
||||||
decisions when appropriate.
|
|
||||||
|
|
||||||
## Scope
|
## Scope
|
||||||
|
|
||||||
This Code of Conduct applies within all community spaces, and also applies when
|
This Code of Conduct applies within all community spaces, and also applies when an individual is officially representing the community in public spaces. Examples of representing our community include using an official e-mail address, posting via an official social media account, or acting as an appointed representative at an online or offline event.
|
||||||
an individual is officially representing the community in public spaces.
|
|
||||||
Examples of representing our community include using an official e-mail address,
|
|
||||||
posting via an official social media account, or acting as an appointed
|
|
||||||
representative at an online or offline event.
|
|
||||||
|
|
||||||
## Enforcement
|
## Enforcement
|
||||||
|
|
||||||
Instances of abusive, harassing, or otherwise unacceptable behavior may be
|
Instances of abusive, harassing, or otherwise unacceptable behavior may be reported to the community leaders responsible for enforcement at community-reports@roboflow.com.
|
||||||
reported to the community leaders responsible for enforcement at
|
|
||||||
community-reports@roboflow.com.
|
|
||||||
|
|
||||||
All complaints will be reviewed and investigated promptly and fairly.
|
All complaints will be reviewed and investigated promptly and fairly.
|
||||||
|
|
||||||
All community leaders are obligated to respect the privacy and security of the
|
All community leaders are obligated to respect the privacy and security of the reporter of any incident.
|
||||||
reporter of any incident.
|
|
||||||
|
|
||||||
## Enforcement Guidelines
|
## Enforcement Guidelines
|
||||||
|
|
||||||
Community leaders will follow these Community Impact Guidelines in determining
|
Community leaders will follow these Community Impact Guidelines in determining the consequences for any action they deem in violation of this Code of Conduct:
|
||||||
the consequences for any action they deem in violation of this Code of Conduct:
|
|
||||||
|
|
||||||
### 1. Correction
|
### 1. Correction
|
||||||
|
|
||||||
**Community Impact**: Use of inappropriate language or other behavior deemed
|
**Community Impact**: Use of inappropriate language or other behavior deemed unprofessional or unwelcome in the community.
|
||||||
unprofessional or unwelcome in the community.
|
|
||||||
|
|
||||||
**Consequence**: A private, written warning from community leaders, providing
|
**Consequence**: A private, written warning from community leaders, providing clarity around the nature of the violation and an explanation of why the behavior was inappropriate. A public apology may be requested.
|
||||||
clarity around the nature of the violation and an explanation of why the
|
|
||||||
behavior was inappropriate. A public apology may be requested.
|
|
||||||
|
|
||||||
### 2. Warning
|
### 2. Warning
|
||||||
|
|
||||||
**Community Impact**: A violation through a single incident or series of
|
**Community Impact**: A violation through a single incident or series of actions.
|
||||||
actions.
|
|
||||||
|
|
||||||
**Consequence**: A warning with consequences for continued behavior. No
|
**Consequence**: A warning with consequences for continued behavior. No interaction with the people involved, including unsolicited interaction with those enforcing the Code of Conduct, for a specified period of time. This includes avoiding interactions in community spaces as well as external channels like social media. Violating these terms may lead to a temporary or permanent ban.
|
||||||
interaction with the people involved, including unsolicited interaction with
|
|
||||||
those enforcing the Code of Conduct, for a specified period of time. This
|
|
||||||
includes avoiding interactions in community spaces as well as external channels
|
|
||||||
like social media. Violating these terms may lead to a temporary or permanent
|
|
||||||
ban.
|
|
||||||
|
|
||||||
### 3. Temporary Ban
|
### 3. Temporary Ban
|
||||||
|
|
||||||
**Community Impact**: A serious violation of community standards, including
|
**Community Impact**: A serious violation of community standards, including sustained inappropriate behavior.
|
||||||
sustained inappropriate behavior.
|
|
||||||
|
|
||||||
**Consequence**: A temporary ban from any sort of interaction or public
|
**Consequence**: A temporary ban from any sort of interaction or public communication with the community for a specified period of time. No public or private interaction with the people involved, including unsolicited interaction with those enforcing the Code of Conduct, is allowed during this period. Violating these terms may lead to a permanent ban.
|
||||||
communication with the community for a specified period of time. No public or
|
|
||||||
private interaction with the people involved, including unsolicited interaction
|
|
||||||
with those enforcing the Code of Conduct, is allowed during this period.
|
|
||||||
Violating these terms may lead to a permanent ban.
|
|
||||||
|
|
||||||
### 4. Permanent Ban
|
### 4. Permanent Ban
|
||||||
|
|
||||||
**Community Impact**: Demonstrating a pattern of violation of community
|
**Community Impact**: Demonstrating a pattern of violation of community standards, including sustained inappropriate behavior, harassment of an individual, or aggression toward or disparagement of classes of individuals.
|
||||||
standards, including sustained inappropriate behavior, harassment of an
|
|
||||||
individual, or aggression toward or disparagement of classes of individuals.
|
|
||||||
|
|
||||||
**Consequence**: A permanent ban from any sort of public interaction within the
|
**Consequence**: A permanent ban from any sort of public interaction within the community.
|
||||||
community.
|
|
||||||
|
|
||||||
## Attribution
|
## Attribution
|
||||||
|
|
||||||
This Code of Conduct is adapted from the [Contributor Covenant][homepage],
|
This Code of Conduct is adapted from the [Contributor Covenant][homepage], version 2.1, available at [https://www.contributor-covenant.org/version/2/1/code_of_conduct.html][v2.1].
|
||||||
version 2.1, available at
|
|
||||||
[https://www.contributor-covenant.org/version/2/1/code_of_conduct.html][v2.1].
|
|
||||||
|
|
||||||
Community Impact Guidelines were inspired by
|
Community Impact Guidelines were inspired by [Mozilla's code of conduct enforcement ladder][mozilla coc].
|
||||||
[Mozilla's code of conduct enforcement ladder][mozilla coc].
|
|
||||||
|
|
||||||
For answers to common questions about this code of conduct, see the FAQ at
|
For answers to common questions about this code of conduct, see the FAQ at [https://www.contributor-covenant.org/faq][faq]. Translations are available at [https://www.contributor-covenant.org/translations][translations].
|
||||||
[https://www.contributor-covenant.org/faq][faq]. Translations are available at
|
|
||||||
[https://www.contributor-covenant.org/translations][translations].
|
|
||||||
|
|
||||||
[faq]: https://www.contributor-covenant.org/faq
|
[faq]: https://www.contributor-covenant.org/faq
|
||||||
[homepage]: https://www.contributor-covenant.org
|
[homepage]: https://www.contributor-covenant.org
|
||||||
|
|
|
||||||
|
|
@ -11,13 +11,14 @@ Please read and adhere to our [Code of Conduct](https://supervision.roboflow.com
|
||||||
## Table of Contents
|
## Table of Contents
|
||||||
|
|
||||||
- [Contribution Guidelines](#contribution-guidelines)
|
- [Contribution Guidelines](#contribution-guidelines)
|
||||||
- [Contributing Features](#contributing-features)
|
- [Contributing Features](#contributing-features)
|
||||||
|
- [API Design Principles](#api-design-principles)
|
||||||
- [How to Contribute Changes](#how-to-contribute-changes)
|
- [How to Contribute Changes](#how-to-contribute-changes)
|
||||||
- [Installation for Contributors](#installation-for-contributors)
|
- [Installation for Contributors](#installation-for-contributors)
|
||||||
- [Code Style and Quality](#code-style-and-quality)
|
- [Code Style and Quality](#code-style-and-quality)
|
||||||
- [Pre-commit tool](#pre-commit-tool)
|
- [Pre-commit tool](#pre-commit-tool)
|
||||||
- [Docstrings](#docstrings)
|
- [Docstrings](#docstrings)
|
||||||
- [Type checking](#type-checking)
|
- [Type checking](#type-checking)
|
||||||
- [Documentation](#documentation)
|
- [Documentation](#documentation)
|
||||||
- [Cookbooks](#cookbooks)
|
- [Cookbooks](#cookbooks)
|
||||||
- [Tests](#tests)
|
- [Tests](#tests)
|
||||||
|
|
@ -41,6 +42,15 @@ For example, counting objects that cross a line anywhere on an image is a common
|
||||||
|
|
||||||
Before you contribute a new feature, consider submitting an Issue to discuss the feature so the community can weigh in and assist.
|
Before you contribute a new feature, consider submitting an Issue to discuss the feature so the community can weigh in and assist.
|
||||||
|
|
||||||
|
### API Design Principles
|
||||||
|
|
||||||
|
Supervision APIs should remain generic, composable, and predictable across model families. Before adding a new integration, annotator option, or data conversion method, check the existing `sv.Detections`, `sv.KeyPoints`, and annotator patterns and follow these principles:
|
||||||
|
|
||||||
|
1. **Model integrations normalize raw external outputs into existing Supervision containers.** Use `sv.Detections` for detection, segmentation, and other instance-level predictions that include boxes, masks, class ids, confidence scores, or extra per-instance fields. Use `sv.KeyPoints` for standalone keypoint or pose predictions when keypoints exist independently of detection boxes (e.g. pure pose estimation, landmark detection on pre-cropped images). Use `Detections.keypoints` when keypoints are always co-incident with boxes from the same model — the field stores an `(n, K, 2)` or `(n, K, 3)` array where the optional third channel is per-point confidence in `[0, 1]`.
|
||||||
|
2. **Do not add a `from_<model>` method when the model already returns a Supervision object.** `from_*` methods are for converting raw outputs from external packages such as Ultralytics, Transformers, Inference, or MediaPipe. If a model's `predict()` method already returns `sv.Detections`, keep that result type and store additional structured payloads in `detections.data` or `detections.metadata` using documented keys.
|
||||||
|
3. **Annotators render data; filtering and visibility are container state.** Filtering by confidence, class id, tracker id, geometry, or custom data should happen before annotation through the container slicing APIs, for example `detections[detections.confidence > 0.7]` or `key_points[key_points.confidence > 0.5]`. Per-point presentation state, such as a `KeyPoints.visible` mask, may live on the container and be honored consistently by annotators.
|
||||||
|
4. **Annotator constructor arguments should describe visual presentation, not model-quality gates.** Use constructor arguments for color, thickness, opacity, text, position, style, and generic visualization parameters such as sigma levels. Annotators may skip invalid geometry defensively, including missing points, zero-area boxes, non-finite coordinates, or points marked invisible on the container. They should not introduce confidence thresholds or model-specific quality gates as rendering options.
|
||||||
|
|
||||||
## How to Contribute Changes
|
## How to Contribute Changes
|
||||||
|
|
||||||
First, fork this repository to your own GitHub account. Click "fork" in the top corner of the `supervision` repository to get started:
|
First, fork this repository to your own GitHub account. Click "fork" in the top corner of the `supervision` repository to get started:
|
||||||
|
|
@ -132,63 +142,63 @@ Before starting your work on the project, set up your development environment:
|
||||||
|
|
||||||
1. **Clone your fork of the project:**
|
1. **Clone your fork of the project:**
|
||||||
|
|
||||||
**Option A: Recommended for most contributors (shallow clone of develop branch):**
|
**Option A: Recommended for most contributors (shallow clone of develop branch):**
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
git clone --depth 1 -b develop https://github.com/YOUR_USERNAME/supervision.git
|
git clone --depth 1 -b develop https://github.com/YOUR_USERNAME/supervision.git
|
||||||
cd supervision
|
cd supervision
|
||||||
```
|
```
|
||||||
|
|
||||||
Replace `YOUR_USERNAME` with your GitHub username.
|
Replace `YOUR_USERNAME` with your GitHub username.
|
||||||
|
|
||||||
> **Note**: Using `--depth 1` creates a shallow clone with minimal history and `-b develop` ensures you start with the development branch. This significantly reduces download size while providing everything needed to contribute.
|
> **Note**: Using `--depth 1` creates a shallow clone with minimal history and `-b develop` ensures you start with the development branch. This significantly reduces download size while providing everything needed to contribute.
|
||||||
|
|
||||||
**Option B: Full repository clone (if you need complete history):**
|
**Option B: Full repository clone (if you need complete history):**
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
git clone https://github.com/YOUR_USERNAME/supervision.git
|
git clone https://github.com/YOUR_USERNAME/supervision.git
|
||||||
cd supervision
|
cd supervision
|
||||||
git checkout develop
|
git checkout develop
|
||||||
```
|
```
|
||||||
|
|
||||||
2. **Set up the upstream remote:**
|
2. **Set up the upstream remote:**
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
git remote add upstream https://github.com/roboflow/supervision.git
|
git remote add upstream https://github.com/roboflow/supervision.git
|
||||||
git fetch upstream
|
git fetch upstream
|
||||||
```
|
```
|
||||||
|
|
||||||
3. **Create and activate a virtual environment:**
|
3. **Create and activate a virtual environment:**
|
||||||
|
|
||||||
**On Linux/macOS:**
|
**On Linux/macOS:**
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
python3 -m venv .venv
|
python3 -m venv .venv
|
||||||
source .venv/bin/activate
|
source .venv/bin/activate
|
||||||
```
|
```
|
||||||
|
|
||||||
**On Windows:**
|
**On Windows:**
|
||||||
|
|
||||||
```cmd
|
```cmd
|
||||||
python -m venv .venv
|
python -m venv .venv
|
||||||
.venv\Scripts\activate
|
.venv\Scripts\activate
|
||||||
```
|
```
|
||||||
|
|
||||||
4. **Install `uv`:**
|
4. **Install `uv`:**
|
||||||
|
|
||||||
Follow the instructions on the [uv installation page](https://docs.astral.sh/uv/getting-started/installation/).
|
Follow the instructions on the [uv installation page](https://docs.astral.sh/uv/getting-started/installation/).
|
||||||
|
|
||||||
5. **Install project dependencies:**
|
5. **Install project dependencies:**
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
uv pip install -r pyproject.toml --group dev --group docs --extra metrics
|
uv pip install -r pyproject.toml --group dev --group docs --extra metrics
|
||||||
```
|
```
|
||||||
|
|
||||||
6. **Verify the setup:**
|
6. **Verify the setup:**
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
uv run pytest
|
uv run pytest
|
||||||
```
|
```
|
||||||
|
|
||||||
## 🎨 Code Style and Quality
|
## 🎨 Code Style and Quality
|
||||||
|
|
||||||
|
|
@ -202,27 +212,27 @@ To run the pre-commit tool, follow these steps:
|
||||||
|
|
||||||
1. **Install pre-commit** (already included if you followed the installation steps above):
|
1. **Install pre-commit** (already included if you followed the installation steps above):
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
uv sync --group dev
|
uv sync --group dev
|
||||||
```
|
```
|
||||||
|
|
||||||
2. **Navigate to the project's root directory** (if not already there).
|
2. **Navigate to the project's root directory** (if not already there).
|
||||||
|
|
||||||
3. **Run pre-commit checks**:
|
3. **Run pre-commit checks**:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
uv run pre-commit run --all-files
|
uv run pre-commit run --all-files
|
||||||
```
|
```
|
||||||
|
|
||||||
This will execute the pre-commit hooks configured for this project. If any issues are found, the pre-commit tool will provide feedback on how to resolve them. Make the necessary changes and re-run the command until all issues are resolved.
|
This will execute the pre-commit hooks configured for this project. If any issues are found, the pre-commit tool will provide feedback on how to resolve them. Make the necessary changes and re-run the command until all issues are resolved.
|
||||||
|
|
||||||
4. **Install pre-commit as a git hook** (optional but recommended):
|
4. **Install pre-commit as a git hook** (optional but recommended):
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
uv run pre-commit install
|
uv run pre-commit install
|
||||||
```
|
```
|
||||||
|
|
||||||
This will automatically run pre-commit checks every time you make a `git commit`.
|
This will automatically run pre-commit checks every time you make a `git commit`.
|
||||||
|
|
||||||
### Docstrings
|
### Docstrings
|
||||||
|
|
||||||
|
|
@ -230,9 +240,41 @@ All new functions and classes in `supervision` should include docstrings. This i
|
||||||
|
|
||||||
`supervision` adheres to the [Google Python docstring style](https://google.github.io/styleguide/pyguide.html#383-functions-and-methods). Please refer to the style guide while writing docstrings for your contribution.
|
`supervision` adheres to the [Google Python docstring style](https://google.github.io/styleguide/pyguide.html#383-functions-and-methods). Please refer to the style guide while writing docstrings for your contribution.
|
||||||
|
|
||||||
|
Every docstring should include a usage example. When the example only uses `supervision`, NumPy, and the standard library — no optional extras, no external files or network access — strongly prefer `>>>` doctest format so it is automatically verified by the test suite. See [Doctests](#doctests) below for syntax guidance and for when fenced ```` ```python ```` blocks are appropriate instead.
|
||||||
|
|
||||||
### Type checking
|
### Type checking
|
||||||
|
|
||||||
Currently, there is no systematic type checking with mypy implemented in the project. This is a known limitation that may be addressed in future updates.
|
Type hints are required on all new code. mypy is enforced by the pre-commit hook configured in `.pre-commit-config.yaml` — your PR will fail CI if mypy reports errors.
|
||||||
|
|
||||||
|
### Readability
|
||||||
|
|
||||||
|
Avoid multi-branch conditional expressions inside function or constructor arguments. If an argument needs more than a simple `a if condition else b`, assign it to a named local variable before the call.
|
||||||
|
|
||||||
|
### Performance
|
||||||
|
|
||||||
|
- Avoid unnecessary copies of NumPy arrays.
|
||||||
|
- Prefer vectorized operations over Python loops in hot paths.
|
||||||
|
- Lazy-import heavy framework dependencies (`torch`, `transformers`, `ultralytics`) inside the function that needs them — never at module top level.
|
||||||
|
|
||||||
|
### Deprecation policy
|
||||||
|
|
||||||
|
**Minimum window**: deprecated APIs must remain for at least **3 minor releases** before removal. Example: deprecated in `0.29.0` → removed in `0.32.0`.
|
||||||
|
|
||||||
|
Use the appropriate mechanism depending on what is being deprecated:
|
||||||
|
|
||||||
|
- **Module-level alias**: `supervision.utils.internal.warn_deprecated` in the deprecated module's `__init__.py`
|
||||||
|
- **Renamed parameter**: `supervision.utils.internal.deprecated_parameter` decorator
|
||||||
|
- **Public function, method, or class**: `@deprecated` from `pydeprecate`
|
||||||
|
|
||||||
|
Always specify both the deprecation version and the planned removal version in the message or decorator arguments.
|
||||||
|
|
||||||
|
### Deprecated module aliases
|
||||||
|
|
||||||
|
`supervision.keypoint` is deprecated since `0.27.0` and will be removed in `0.30.0`. Always import from `supervision.key_points`:
|
||||||
|
|
||||||
|
```python
|
||||||
|
from supervision.key_points import KeyPoints # correct
|
||||||
|
```
|
||||||
|
|
||||||
## 📝 Documentation
|
## 📝 Documentation
|
||||||
|
|
||||||
|
|
@ -242,15 +284,15 @@ To run the documentation locally:
|
||||||
|
|
||||||
1. **Install documentation dependencies** (if not already installed):
|
1. **Install documentation dependencies** (if not already installed):
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
uv sync --group docs
|
uv sync --group docs
|
||||||
```
|
```
|
||||||
|
|
||||||
2. **Start the documentation server**:
|
2. **Start the documentation server**:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
uv run mkdocs serve
|
uv run mkdocs serve
|
||||||
```
|
```
|
||||||
|
|
||||||
3. **Access the documentation** at `http://127.0.0.1:8000` in your browser.
|
3. **Access the documentation** at `http://127.0.0.1:8000` in your browser.
|
||||||
|
|
||||||
|
|
@ -258,9 +300,7 @@ You can learn more about mkdocs on the [mkdocs website](https://www.mkdocs.org/)
|
||||||
|
|
||||||
## 🧑🍳 Cookbooks
|
## 🧑🍳 Cookbooks
|
||||||
|
|
||||||
We are always looking for new examples and cookbooks to add to the `supervision`
|
We are always looking for new examples and cookbooks to add to the `supervision` documentation. If you have a use case that you think would be helpful to others, please submit a PR with your example. Here are some guidelines for submitting a new example:
|
||||||
documentation. If you have a use case that you think would be helpful to others, please
|
|
||||||
submit a PR with your example. Here are some guidelines for submitting a new example:
|
|
||||||
|
|
||||||
- Create a new notebook in the [`docs/notebooks`](https://github.com/roboflow/supervision/tree/develop/docs/notebooks) folder.
|
- Create a new notebook in the [`docs/notebooks`](https://github.com/roboflow/supervision/tree/develop/docs/notebooks) folder.
|
||||||
- Add a link to the new notebook in [`docs/theme/cookbooks.html`](https://github.com/roboflow/supervision/blob/develop/docs/theme/cookbooks.html). Make sure to add the path to the new notebook, as well as a title, labels, author and supervision version.
|
- Add a link to the new notebook in [`docs/theme/cookbooks.html`](https://github.com/roboflow/supervision/blob/develop/docs/theme/cookbooks.html). Make sure to add the path to the new notebook, as well as a title, labels, author and supervision version.
|
||||||
|
|
@ -286,6 +326,87 @@ To run tests with coverage:
|
||||||
uv run pytest --cov=supervision
|
uv run pytest --cov=supervision
|
||||||
```
|
```
|
||||||
|
|
||||||
|
### Test Structure
|
||||||
|
|
||||||
|
Follow **Arrange-Act-Assert (AAA)**: one setup block, one action, one assertion group per test. Never put two independent actions in the same test.
|
||||||
|
|
||||||
|
**Class grouping:** Group related tests into a class. The class name carries the unit under test; method names describe the expected outcome only — not the mechanism.
|
||||||
|
|
||||||
|
```python
|
||||||
|
class TestDetectionsWithNms:
|
||||||
|
def test_keeps_highest_confidence_detection(self): ...
|
||||||
|
def test_suppresses_lower_score_when_overlap_exceeds_threshold(self): ...
|
||||||
|
def test_raises_when_confidence_missing(self): ...
|
||||||
|
```
|
||||||
|
|
||||||
|
**Parametrize aggressively:** Three or more structurally identical tests should become a single `@pytest.mark.parametrize` case. Use `pytest.param(..., id="slug")` per case — not `ids=[...]` on the decorator — so the ID stays co-located with its arguments and survives reordering.
|
||||||
|
|
||||||
|
```python
|
||||||
|
@pytest.mark.parametrize(
|
||||||
|
("overlap_metric", "expected_keep"),
|
||||||
|
[
|
||||||
|
pytest.param(OverlapMetric.IOU, [True, True], id="iou-keeps-both"),
|
||||||
|
pytest.param(OverlapMetric.IOS, [True, False], id="ios-suppresses-small"),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
def test_overlap_metric_determines_suppression(
|
||||||
|
overlap_metric: OverlapMetric, expected_keep: list[bool]
|
||||||
|
) -> None:
|
||||||
|
"""Small box inside large: IOU keeps both; IOS suppresses small."""
|
||||||
|
...
|
||||||
|
```
|
||||||
|
|
||||||
|
**Docstrings:** Every test function/method requires at minimum a one-line docstring (within the project line length configured in `pyproject.toml`). Describe the scenario, not the implementation.
|
||||||
|
|
||||||
|
### Doctests
|
||||||
|
|
||||||
|
**Guidance:** when an example uses only `supervision`, NumPy, and the standard library — no optional extras (e.g. no `--extra metrics` packages), no external files, no network, no devices — prefer `>>>` doctest format so it is automatically verified by the test suite. Fenced ```` ```python ```` blocks are appropriate when the example cannot reasonably be executed (e.g. loading a third-party model, reading a video file) or when the primary purpose is demonstrating error/exception behaviour rather than return values.
|
||||||
|
|
||||||
|
Doctests run automatically as part of the test suite via `--doctest-modules` in `pyproject.toml`. The `ELLIPSIS` and `NORMALIZE_WHITESPACE` flags are enabled globally, so `...` matches any output fragment and minor whitespace differences are ignored.
|
||||||
|
|
||||||
|
```bash
|
||||||
|
uv run pytest --doctest-modules src/
|
||||||
|
```
|
||||||
|
|
||||||
|
**Writing a doctest**
|
||||||
|
|
||||||
|
Use the `Example:` section of a Google-style docstring. Prefix each input line with `>>>` and each continuation line with `...`. Place expected output immediately after the last input line with no blank line between them.
|
||||||
|
|
||||||
|
```python
|
||||||
|
def clip_boxes(xyxy: np.ndarray, resolution_wh: tuple) -> np.ndarray:
|
||||||
|
"""Clip bounding boxes to frame boundaries.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
xyxy: Box coordinates as (N, 4) float array.
|
||||||
|
resolution_wh: Frame size as (width, height).
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Clipped boxes as (N, 4) float array.
|
||||||
|
|
||||||
|
Example:
|
||||||
|
>>> import numpy as np
|
||||||
|
>>> import supervision as sv
|
||||||
|
>>> boxes = np.array([[-10, -5, 120, 80]], dtype=np.float32)
|
||||||
|
>>> sv.clip_boxes(boxes, resolution_wh=(100, 60))
|
||||||
|
array([[ 0., 0., 100., 60.]], dtype=float32)
|
||||||
|
"""
|
||||||
|
```
|
||||||
|
|
||||||
|
### Key rules
|
||||||
|
|
||||||
|
- **Single-line expression** — write the repr as expected output: `>>> len(result)` → `1`
|
||||||
|
- **Multi-line statement** — use `...` continuation: `>>> arr = np.array([` / `... [1, 2],` / `... ])`
|
||||||
|
- **Print output** — write the printed string as expected output (no quotes).
|
||||||
|
- **`None` return** — no output line needed (suppress with assignment or `_ =`).
|
||||||
|
- **Large/variable arrays** — use `ELLIPSIS`: `array([...])` matches any content.
|
||||||
|
- **`# doctest: +SKIP`** — use only as a last resort for genuinely non-runnable lines (e.g. a GPU-only call inside an otherwise runnable example). Prefer splitting the example into two blocks instead.
|
||||||
|
|
||||||
|
Fenced ```` ```python ```` blocks remain appropriate for:
|
||||||
|
|
||||||
|
- Examples that import optional extras (`supervision[metrics]`, `torch`, `ultralytics`).
|
||||||
|
- Examples that read files, capture video, or require a running service.
|
||||||
|
- Illustrative pseudocode that is intentionally incomplete.
|
||||||
|
|
||||||
## 🔍 PR Review Guidelines
|
## 🔍 PR Review Guidelines
|
||||||
|
|
||||||
These guidelines help reviewers provide consistent, actionable feedback efficiently. Your goals: validate completeness, identify risks, provide actionable feedback, and highlight quality gaps.
|
These guidelines help reviewers provide consistent, actionable feedback efficiently. Your goals: validate completeness, identify risks, provide actionable feedback, and highlight quality gaps.
|
||||||
|
|
|
||||||
|
|
@ -2,17 +2,17 @@
|
||||||
|
|
||||||
This file provides context-aware guidance for GitHub Copilot when working in the Supervision repository.
|
This file provides context-aware guidance for GitHub Copilot when working in the Supervision repository.
|
||||||
|
|
||||||
---
|
______________________________________________________________________
|
||||||
|
|
||||||
## 📚 Repository Overview
|
## 📚 Repository Overview
|
||||||
|
|
||||||
**Supervision** is a Python library providing reusable computer vision utilities for working with object detection models (YOLO, SAM, etc.). It offers tools for detections processing, tracking, annotation, and dataset management.
|
**Supervision** is a Python library providing reusable computer vision utilities for working with object detection models (YOLO, SAM, etc.). It offers tools for detections processing, tracking, annotation, and dataset management.
|
||||||
|
|
||||||
- **Languages**: Python 3.9+
|
- **Languages**: Python 3.10+
|
||||||
- **Key Dependencies**: NumPy, OpenCV, SciPy
|
- **Key Dependencies**: NumPy, OpenCV, SciPy
|
||||||
- **License**: MIT
|
- **License**: MIT
|
||||||
|
|
||||||
---
|
______________________________________________________________________
|
||||||
|
|
||||||
## 🏗️ Project Structure
|
## 🏗️ Project Structure
|
||||||
|
|
||||||
|
|
@ -30,7 +30,7 @@ supervision/
|
||||||
└── examples/ # Usage examples
|
└── examples/ # Usage examples
|
||||||
```
|
```
|
||||||
|
|
||||||
---
|
______________________________________________________________________
|
||||||
|
|
||||||
## 🔧 Development Commands
|
## 🔧 Development Commands
|
||||||
|
|
||||||
|
|
@ -61,7 +61,7 @@ uv run pytest --cov=supervision
|
||||||
uv run mkdocs serve
|
uv run mkdocs serve
|
||||||
```
|
```
|
||||||
|
|
||||||
---
|
______________________________________________________________________
|
||||||
|
|
||||||
## 💻 Code Conventions
|
## 💻 Code Conventions
|
||||||
|
|
||||||
|
|
@ -77,8 +77,8 @@ uv run mkdocs serve
|
||||||
- **Linting**: Enforced by `ruff-check` (pre-commit)
|
- **Linting**: Enforced by `ruff-check` (pre-commit)
|
||||||
- **Type Hints**: Required on all new code
|
- **Type Hints**: Required on all new code
|
||||||
- **Docstrings**: Required using [Google Python style](https://google.github.io/styleguide/pyguide.html#383-functions-and-methods)
|
- **Docstrings**: Required using [Google Python style](https://google.github.io/styleguide/pyguide.html#383-functions-and-methods)
|
||||||
- Must include usage examples with primitive values
|
- Must include usage examples with primitive values
|
||||||
- Serve as runnable documentation
|
- Serve as runnable documentation
|
||||||
|
|
||||||
### Performance
|
### Performance
|
||||||
|
|
||||||
|
|
@ -92,7 +92,7 @@ uv run mkdocs serve
|
||||||
- Maintain backward compatibility unless explicitly breaking
|
- Maintain backward compatibility unless explicitly breaking
|
||||||
- Prefer functional utilities over complex classes
|
- Prefer functional utilities over complex classes
|
||||||
|
|
||||||
---
|
______________________________________________________________________
|
||||||
|
|
||||||
## 🧪 Testing Requirements
|
## 🧪 Testing Requirements
|
||||||
|
|
||||||
|
|
@ -103,7 +103,7 @@ All new features must include:
|
||||||
- Clear test names describing what they validate
|
- Clear test names describing what they validate
|
||||||
- Proper assertions (not just "no exception raised")
|
- Proper assertions (not just "no exception raised")
|
||||||
|
|
||||||
---
|
______________________________________________________________________
|
||||||
|
|
||||||
## 📝 Documentation Requirements
|
## 📝 Documentation Requirements
|
||||||
|
|
||||||
|
|
@ -114,7 +114,7 @@ For new public functions/classes:
|
||||||
- Entry in appropriate `docs/*.md` file
|
- Entry in appropriate `docs/*.md` file
|
||||||
- Reference in `mkdocs.yml` navigation
|
- Reference in `mkdocs.yml` navigation
|
||||||
|
|
||||||
---
|
______________________________________________________________________
|
||||||
|
|
||||||
## 🔍 Pull Request Reviews
|
## 🔍 Pull Request Reviews
|
||||||
|
|
||||||
|
|
@ -129,7 +129,7 @@ Quick checklist:
|
||||||
- Score code quality, testing, docs (n/5 scale)
|
- Score code quality, testing, docs (n/5 scale)
|
||||||
- Use inline comments + GitHub suggestion format
|
- Use inline comments + GitHub suggestion format
|
||||||
|
|
||||||
---
|
______________________________________________________________________
|
||||||
|
|
||||||
## 🌿 Branching & Commits
|
## 🌿 Branching & Commits
|
||||||
|
|
||||||
|
|
@ -137,10 +137,10 @@ Quick checklist:
|
||||||
- Use **conventional commits**: `feat:`, `fix:`, `docs:`, `refactor:`, `perf:`, `test:`, `chore:`
|
- Use **conventional commits**: `feat:`, `fix:`, `docs:`, `refactor:`, `perf:`, `test:`, `chore:`
|
||||||
- All PRs target `develop` branch
|
- All PRs target `develop` branch
|
||||||
|
|
||||||
---
|
______________________________________________________________________
|
||||||
|
|
||||||
## 🎯 Context-Aware Behavior
|
## 🎯 Context-Aware Behavior
|
||||||
|
|
||||||
**For general development tasks**: Follow [AGENTS.md](../AGENTS.md)
|
- **For general development tasks**: Follow [AGENTS.md](../AGENTS.md)
|
||||||
**For pull request reviews**: Follow [PR Review Guidelines](CONTRIBUTING.md#pr-review-guidelines)
|
- **For pull request reviews**: Follow [PR Review Guidelines](CONTRIBUTING.md#pr-review-guidelines)
|
||||||
**For detailed processes**: Consult [CONTRIBUTING.md](CONTRIBUTING.md)
|
- **For detailed processes**: Consult [CONTRIBUTING.md](CONTRIBUTING.md)
|
||||||
|
|
|
||||||
|
|
@ -5,6 +5,8 @@ updates:
|
||||||
directory: "/"
|
directory: "/"
|
||||||
schedule:
|
schedule:
|
||||||
interval: "weekly"
|
interval: "weekly"
|
||||||
|
cooldown:
|
||||||
|
default-days: 7
|
||||||
commit-message:
|
commit-message:
|
||||||
prefix: ⬆️
|
prefix: ⬆️
|
||||||
target-branch: "develop"
|
target-branch: "develop"
|
||||||
|
|
@ -17,6 +19,8 @@ updates:
|
||||||
directory: "/"
|
directory: "/"
|
||||||
schedule:
|
schedule:
|
||||||
interval: "weekly"
|
interval: "weekly"
|
||||||
|
cooldown:
|
||||||
|
default-days: 7
|
||||||
commit-message:
|
commit-message:
|
||||||
prefix: ⬆️
|
prefix: ⬆️
|
||||||
target-branch: "develop"
|
target-branch: "develop"
|
||||||
|
|
|
||||||
|
|
@ -9,9 +9,16 @@ accept = [
|
||||||
200, # OK
|
200, # OK
|
||||||
408, # Request Timeout
|
408, # Request Timeout
|
||||||
# 429 means the server received the request and is actively rate-limiting — the URL is
|
# 429 means the server received the request and is actively rate-limiting — the URL is
|
||||||
# reachable. Real dead links return 404, 410, 5xx, or fail to connect; none of those
|
# reachable. Real dead links return 404, 410, or fail to connect/resolve; none of
|
||||||
# produce a 429, so accepting it here does not hide broken links.
|
# those produce a 429, so accepting it here does not hide broken links.
|
||||||
429, # Too Many Requests (rate-limited but reachable; does not mask dead links)
|
429, # Too Many Requests (rate-limited but reachable; does not mask dead links)
|
||||||
|
# CI regularly sees momentary 502/503/504 from large, healthy hosts (github.com,
|
||||||
|
# supervision.roboflow.com), and in-run retries tend to land inside the same
|
||||||
|
# incident window. Genuinely dead links surface as 404, 410, or connection/DNS
|
||||||
|
# failures, which remain rejected.
|
||||||
|
502, # Bad Gateway (transient upstream hiccup)
|
||||||
|
503, # Service Unavailable (transient overload or maintenance)
|
||||||
|
504, # Gateway Timeout (transient upstream hiccup)
|
||||||
]
|
]
|
||||||
|
|
||||||
exclude = [
|
exclude = [
|
||||||
|
|
@ -19,6 +26,7 @@ exclude = [
|
||||||
"http://127.0.0.1:8000", # hint for local docs server
|
"http://127.0.0.1:8000", # hint for local docs server
|
||||||
"https://sam2.metademolab.com/", # returns 403 Forbidden
|
"https://sam2.metademolab.com/", # returns 403 Forbidden
|
||||||
"https://snyk.io/advisor/python/supervision/badge.svg", # badge URL
|
"https://snyk.io/advisor/python/supervision/badge.svg", # badge URL
|
||||||
|
"https://trendshift.io", # badge API times out in CI
|
||||||
"https://universe.roboflow.com/",
|
"https://universe.roboflow.com/",
|
||||||
"https://universe.roboflow.com/model-examples/segmented-animals-basic",
|
"https://universe.roboflow.com/model-examples/segmented-animals-basic",
|
||||||
# fixme: this page returns 401 Unauthorized when accessed and 404 Not Found when accessed with browser,
|
# fixme: this page returns 401 Unauthorized when accessed and 404 Not Found when accessed with browser,
|
||||||
|
|
|
||||||
|
|
@ -0,0 +1,109 @@
|
||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Validate doctest prompt formatting in source docstrings."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import re
|
||||||
|
import sys
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
DOCTEST_PROMPT_RE = re.compile(r"^\s*>>>")
|
||||||
|
FENCE_RE = re.compile(r"^\s*```(?P<language>[A-Za-z0-9_-]*)\s*$")
|
||||||
|
|
||||||
|
|
||||||
|
def _check_content(content: str, path: Path) -> list[str]:
|
||||||
|
r"""Return doctest fence violations for file content.
|
||||||
|
|
||||||
|
Examples:
|
||||||
|
```pycon
|
||||||
|
>>> from pathlib import Path
|
||||||
|
>>> _check_content('```pycon\n>>> len([1])\n1\n\n```\n', Path('src/a.py'))
|
||||||
|
[]
|
||||||
|
>>> _check_content('>>> len([1])\n1\n', Path('src/a.py'))
|
||||||
|
['src/a.py:1: doctest prompt must be inside a ```pycon fenced block']
|
||||||
|
>>> violations = _check_content(
|
||||||
|
... '```pycon\n>>> len([1])\n1\n```\n', Path('src/a.py')
|
||||||
|
... )
|
||||||
|
>>> violations == [
|
||||||
|
... 'src/a.py:4: pycon doctest block must include exactly one blank line '
|
||||||
|
... 'before the closing fence'
|
||||||
|
... ]
|
||||||
|
True
|
||||||
|
|
||||||
|
```
|
||||||
|
"""
|
||||||
|
violations: list[str] = []
|
||||||
|
active_fence_language: str | None = None
|
||||||
|
in_invalid_doctest_block = False
|
||||||
|
line_before_previous = ""
|
||||||
|
previous_line = ""
|
||||||
|
|
||||||
|
for line_number, line in enumerate(content.splitlines(), start=1):
|
||||||
|
fence_match = FENCE_RE.match(line)
|
||||||
|
if fence_match is not None:
|
||||||
|
in_invalid_doctest_block = False
|
||||||
|
if active_fence_language is None:
|
||||||
|
active_fence_language = fence_match.group("language")
|
||||||
|
else:
|
||||||
|
has_exactly_one_blank_line = (
|
||||||
|
previous_line.strip() == "" and line_before_previous.strip() != ""
|
||||||
|
)
|
||||||
|
if active_fence_language == "pycon" and not has_exactly_one_blank_line:
|
||||||
|
violations.append(
|
||||||
|
f"{path}:{line_number}: pycon doctest block must include "
|
||||||
|
"exactly one blank line before the closing fence"
|
||||||
|
)
|
||||||
|
active_fence_language = None
|
||||||
|
line_before_previous = previous_line
|
||||||
|
previous_line = line
|
||||||
|
continue
|
||||||
|
|
||||||
|
if not line.strip():
|
||||||
|
in_invalid_doctest_block = False
|
||||||
|
|
||||||
|
if (
|
||||||
|
DOCTEST_PROMPT_RE.match(line)
|
||||||
|
and active_fence_language != "pycon"
|
||||||
|
and not in_invalid_doctest_block
|
||||||
|
):
|
||||||
|
violations.append(
|
||||||
|
f"{path}:{line_number}: doctest prompt must be inside a "
|
||||||
|
"```pycon fenced block"
|
||||||
|
)
|
||||||
|
in_invalid_doctest_block = True
|
||||||
|
|
||||||
|
line_before_previous = previous_line
|
||||||
|
previous_line = line
|
||||||
|
|
||||||
|
return violations
|
||||||
|
|
||||||
|
|
||||||
|
def check_file(path: Path) -> list[str]:
|
||||||
|
"""Return doctest fence violations for a single source file."""
|
||||||
|
if not path.is_file() or path.suffix != ".py" or "src" not in path.parts:
|
||||||
|
return []
|
||||||
|
|
||||||
|
return _check_content(content=path.read_text(encoding="utf-8"), path=path)
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> int:
|
||||||
|
"""Run the doctest fence check for pre-commit supplied files."""
|
||||||
|
parser = argparse.ArgumentParser(
|
||||||
|
description="Validate doctest prompts in src/ are fenced as pycon blocks."
|
||||||
|
)
|
||||||
|
parser.add_argument("files", nargs="*", type=Path)
|
||||||
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
violations = [
|
||||||
|
violation for path in args.files for violation in check_file(path=path)
|
||||||
|
]
|
||||||
|
if violations:
|
||||||
|
print("\n".join(violations))
|
||||||
|
return 1
|
||||||
|
|
||||||
|
return 0
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
sys.exit(main())
|
||||||
|
|
@ -0,0 +1,125 @@
|
||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Verify a built Supervision wheel works without OpenCV."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import subprocess
|
||||||
|
import zipfile
|
||||||
|
from email import message_from_bytes
|
||||||
|
from email.message import Message
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
_MANIFEST_CHECKS = {
|
||||||
|
"import-supervision",
|
||||||
|
"no-cv2-module",
|
||||||
|
"fallback-backend",
|
||||||
|
"bgr-to-gray",
|
||||||
|
"draw-rectangle",
|
||||||
|
"required-pyav",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _wheel_metadata(wheel: Path) -> Message:
|
||||||
|
"""Read the core metadata embedded in a wheel archive."""
|
||||||
|
with zipfile.ZipFile(wheel) as archive:
|
||||||
|
metadata_paths = [
|
||||||
|
name for name in archive.namelist() if name.endswith("/METADATA")
|
||||||
|
]
|
||||||
|
if len(metadata_paths) != 1:
|
||||||
|
raise ValueError(
|
||||||
|
f"expected one METADATA file in {wheel}, found {metadata_paths}"
|
||||||
|
)
|
||||||
|
return message_from_bytes(archive.read(metadata_paths[0]))
|
||||||
|
|
||||||
|
|
||||||
|
def _validate_metadata(wheel: Path) -> None:
|
||||||
|
"""Reject wheels that retain an OpenCV runtime requirement or extra."""
|
||||||
|
metadata = _wheel_metadata(wheel)
|
||||||
|
requirements = metadata.get_all("Requires-Dist", [])
|
||||||
|
extras = metadata.get_all("Provides-Extra", [])
|
||||||
|
if any("opencv" in requirement.lower() for requirement in requirements):
|
||||||
|
raise ValueError(
|
||||||
|
f"OpenCV runtime requirement remains in {wheel}: {requirements}"
|
||||||
|
)
|
||||||
|
if any("opencv" in extra.lower() for extra in extras):
|
||||||
|
raise ValueError(f"OpenCV extra remains in {wheel}: {extras}")
|
||||||
|
|
||||||
|
|
||||||
|
def _validate_manifest(manifest: Path) -> None:
|
||||||
|
"""Keep the installed-wheel fallback smoke contract explicit and complete."""
|
||||||
|
checks = {
|
||||||
|
s
|
||||||
|
for line in manifest.read_text(encoding="utf-8").splitlines()
|
||||||
|
if (s := line.strip()) and not s.startswith("#")
|
||||||
|
}
|
||||||
|
if checks != _MANIFEST_CHECKS:
|
||||||
|
raise ValueError(
|
||||||
|
f"unexpected fallback manifest {checks}; expected {_MANIFEST_CHECKS}"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _run_installed_wheel_probe(python: Path) -> None:
|
||||||
|
"""Exercise the installed fallback without allowing the source tree on sys.path."""
|
||||||
|
source = """
|
||||||
|
import importlib.util
|
||||||
|
from importlib import metadata
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
import av
|
||||||
|
import numpy as np
|
||||||
|
import supervision
|
||||||
|
from supervision import _cv2
|
||||||
|
|
||||||
|
package_path = Path(supervision.__file__).resolve()
|
||||||
|
if "site-packages" not in package_path.parts:
|
||||||
|
raise AssertionError(
|
||||||
|
f"supervision did not import from site-packages: {package_path}"
|
||||||
|
)
|
||||||
|
if importlib.util.find_spec("cv2") is not None:
|
||||||
|
raise AssertionError("cv2 is installed in the clean-wheel environment")
|
||||||
|
opencv_distributions = [
|
||||||
|
distribution.metadata["Name"]
|
||||||
|
for distribution in metadata.distributions()
|
||||||
|
if "opencv" in distribution.metadata["Name"].lower()
|
||||||
|
]
|
||||||
|
if opencv_distributions:
|
||||||
|
raise AssertionError(
|
||||||
|
"OpenCV distributions remain in the clean-wheel environment: "
|
||||||
|
f"{opencv_distributions}"
|
||||||
|
)
|
||||||
|
if _cv2.BACKEND_NAME != "fallback":
|
||||||
|
raise AssertionError(f"expected fallback backend, got {_cv2.BACKEND_NAME!r}")
|
||||||
|
|
||||||
|
image = np.array([[[0, 0, 255]]], dtype=np.uint8)
|
||||||
|
assert _cv2.cvtColor(image, _cv2.COLOR_BGR2GRAY).tolist() == [[76]]
|
||||||
|
canvas = np.zeros((3, 3, 3), dtype=np.uint8)
|
||||||
|
assert _cv2.rectangle(canvas, (0, 0), (2, 2), (1, 2, 3), -1) is canvas
|
||||||
|
assert canvas.tolist() == [[[1, 2, 3]] * 3] * 3
|
||||||
|
assert av.__version__
|
||||||
|
"""
|
||||||
|
subprocess.run( # noqa: S603 - the caller passes the clean CI interpreter explicitly.
|
||||||
|
[str(python), "-c", source],
|
||||||
|
check=True,
|
||||||
|
cwd=Path.cwd().parent,
|
||||||
|
)
|
||||||
|
subprocess.run( # noqa: S603 - the caller passes the clean CI interpreter explicitly.
|
||||||
|
[str(python), "-m", "pip", "check"], check=True
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> None:
|
||||||
|
"""Validate one wheel against a previously prepared clean environment."""
|
||||||
|
parser = argparse.ArgumentParser(description=__doc__)
|
||||||
|
parser.add_argument("--wheel", type=Path, required=True)
|
||||||
|
parser.add_argument("--python", type=Path, required=True)
|
||||||
|
parser.add_argument("--manifest", type=Path, required=True)
|
||||||
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
_validate_metadata(args.wheel)
|
||||||
|
_validate_manifest(args.manifest)
|
||||||
|
_run_installed_wheel_probe(args.python)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
|
|
@ -20,10 +20,10 @@ jobs:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
steps:
|
steps:
|
||||||
- name: 📥 Checkout the repository
|
- name: 📥 Checkout the repository
|
||||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||||
|
|
||||||
- name: 🐍 Install uv and set Python version ${{ inputs.python-version }}
|
- name: 🐍 Install uv and set Python version ${{ inputs.python-version }}
|
||||||
uses: astral-sh/setup-uv@08807647e7069bb48b6ef5acd8ec9567f424441b # v8.1.0
|
uses: astral-sh/setup-uv@c771a70e6277c0a99b617c7a806ffedaca235ff9 # v9.0.0
|
||||||
with:
|
with:
|
||||||
python-version: ${{ inputs.python-version }}
|
python-version: ${{ inputs.python-version }}
|
||||||
activate-environment: true
|
activate-environment: true
|
||||||
|
|
|
||||||
|
|
@ -22,10 +22,10 @@ jobs:
|
||||||
timeout-minutes: 10
|
timeout-minutes: 10
|
||||||
steps:
|
steps:
|
||||||
- name: 📥 Checkout the repository
|
- name: 📥 Checkout the repository
|
||||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||||
|
|
||||||
- name: 🐍 Install uv and set Python
|
- name: 🐍 Install uv and set Python
|
||||||
uses: astral-sh/setup-uv@08807647e7069bb48b6ef5acd8ec9567f424441b # v8.1.0
|
uses: astral-sh/setup-uv@c771a70e6277c0a99b617c7a806ffedaca235ff9 # v9.0.0
|
||||||
with:
|
with:
|
||||||
python-version: "3.10"
|
python-version: "3.10"
|
||||||
activate-environment: true
|
activate-environment: true
|
||||||
|
|
|
||||||
|
|
@ -19,7 +19,7 @@ jobs:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
steps:
|
steps:
|
||||||
- name: Checkout repository
|
- name: Checkout repository
|
||||||
uses: actions/checkout@v6
|
uses: actions/checkout@v7.0.1
|
||||||
|
|
||||||
- name: 🔗 Link Checker
|
- name: 🔗 Link Checker
|
||||||
uses: lycheeverse/lychee-action@v2
|
uses: lycheeverse/lychee-action@v2
|
||||||
|
|
|
||||||
|
|
@ -22,31 +22,57 @@ jobs:
|
||||||
link-check: ${{ github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository }}
|
link-check: ${{ github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository }}
|
||||||
|
|
||||||
run-tests:
|
run-tests:
|
||||||
name: Import Test and Pytest Run
|
name: Pytest Run
|
||||||
# needs: build # todo: consider using this build package for testing
|
# needs: build # todo: consider using this build package for testing
|
||||||
timeout-minutes: 10
|
timeout-minutes: 10
|
||||||
strategy:
|
strategy:
|
||||||
fail-fast: false
|
fail-fast: false
|
||||||
matrix:
|
matrix:
|
||||||
os: ["ubuntu-latest", "windows-latest", "macos-latest"]
|
os: ["ubuntu-latest", "windows-latest", "macos-latest"]
|
||||||
python-version: ["3.9", "3.10", "3.11", "3.12", "3.13"]
|
python-version: ["3.10", "3.11", "3.12", "3.13"]
|
||||||
|
cv2: ["none"]
|
||||||
|
include:
|
||||||
|
- { os: "ubuntu-latest", python-version: "3.13", cv2: "opencv-python" }
|
||||||
|
- { os: "windows-latest", python-version: "3.13", cv2: "opencv-python" }
|
||||||
|
- { os: "macos-latest", python-version: "3.13", cv2: "opencv-python" }
|
||||||
|
- { os: "ubuntu-latest", python-version: "3.13", cv2: "opencv-python-headless" }
|
||||||
|
- { os: "windows-latest", python-version: "3.13", cv2: "opencv-python-headless" }
|
||||||
|
- { os: "macos-latest", python-version: "3.13", cv2: "opencv-python-headless" }
|
||||||
runs-on: ${{ matrix.os }}
|
runs-on: ${{ matrix.os }}
|
||||||
steps:
|
steps:
|
||||||
- name: 📥 Checkout the repository
|
- name: 📥 Checkout the repository
|
||||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||||
|
|
||||||
- name: 🐍 Install uv and set Python version ${{ matrix.python-version }}
|
- name: 🐍 Install uv and set Python version ${{ matrix.python-version }}
|
||||||
uses: astral-sh/setup-uv@08807647e7069bb48b6ef5acd8ec9567f424441b # v8.1.0
|
uses: astral-sh/setup-uv@c771a70e6277c0a99b617c7a806ffedaca235ff9 # v9.0.0
|
||||||
with:
|
with:
|
||||||
python-version: ${{ matrix.python-version }}
|
python-version: ${{ matrix.python-version }}
|
||||||
activate-environment: true
|
activate-environment: true
|
||||||
|
|
||||||
- name: 🚀 Install Packages
|
- name: 🚀 Install Packages
|
||||||
run: uv sync --frozen --group dev --group docs --extra metrics
|
run: uv sync --frozen --group dev --extra metrics
|
||||||
|
|
||||||
|
- name: 📷 Install selected OpenCV package
|
||||||
|
if: matrix.cv2 != 'none'
|
||||||
|
run: uv pip install ${{ matrix.cv2 }}
|
||||||
|
|
||||||
|
- name: 🧭 Confirm selected cv2 backend
|
||||||
|
env:
|
||||||
|
EXPECTED_BACKEND: ${{ matrix.cv2 == 'none' && 'fallback' || 'opencv' }}
|
||||||
|
run: |
|
||||||
|
import os
|
||||||
|
from supervision import _cv2
|
||||||
|
|
||||||
|
expected = os.environ["EXPECTED_BACKEND"]
|
||||||
|
assert _cv2.BACKEND_NAME == expected, f"expected {expected!r}, got {_cv2.BACKEND_NAME!r}"
|
||||||
|
shell: python
|
||||||
|
|
||||||
- name: 📦 Run the Import test
|
- name: 📦 Run the Import test
|
||||||
run: python -c "import supervision; from supervision import assets; from supervision import metrics; print(supervision.__version__)"
|
run: python -c "import supervision; from supervision import assets; from supervision import metrics; print(supervision.__version__)"
|
||||||
|
|
||||||
|
- name: 📋 Print installed packages
|
||||||
|
run: uv pip list
|
||||||
|
|
||||||
- name: 🧪 Run the Test
|
- name: 🧪 Run the Test
|
||||||
run: pytest src/ tests/ --cov=supervision --cov-report=xml
|
run: pytest src/ tests/ --cov=supervision --cov-report=xml
|
||||||
|
|
||||||
|
|
@ -56,7 +82,7 @@ jobs:
|
||||||
coverage report
|
coverage report
|
||||||
|
|
||||||
- name: Upload coverage to Codecov
|
- name: Upload coverage to Codecov
|
||||||
uses: codecov/codecov-action@v6
|
uses: codecov/codecov-action@v7
|
||||||
with:
|
with:
|
||||||
token: ${{ secrets.CODECOV_TOKEN }}
|
token: ${{ secrets.CODECOV_TOKEN }}
|
||||||
files: "coverage.xml"
|
files: "coverage.xml"
|
||||||
|
|
@ -68,23 +94,66 @@ jobs:
|
||||||
- name: Minimize uv cache
|
- name: Minimize uv cache
|
||||||
run: uv cache prune --ci
|
run: uv cache prune --ci
|
||||||
|
|
||||||
|
clean-wheel:
|
||||||
|
name: Clean Wheel on ${{ matrix.os }} / Python ${{ matrix.python-version }}
|
||||||
|
timeout-minutes: 10
|
||||||
|
strategy:
|
||||||
|
fail-fast: false
|
||||||
|
matrix:
|
||||||
|
os: ["ubuntu-latest", "windows-latest", "macos-latest"]
|
||||||
|
python-version: ["3.10", "3.13"]
|
||||||
|
runs-on: ${{ matrix.os }}
|
||||||
|
steps:
|
||||||
|
- name: 📥 Checkout the repository
|
||||||
|
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||||
|
|
||||||
|
- name: 🐍 Install uv and set Python version ${{ matrix.python-version }}
|
||||||
|
uses: astral-sh/setup-uv@c771a70e6277c0a99b617c7a806ffedaca235ff9 # v9.0.0
|
||||||
|
with:
|
||||||
|
python-version: ${{ matrix.python-version }}
|
||||||
|
activate-environment: true
|
||||||
|
|
||||||
|
- name: 🏗️ Build the wheel
|
||||||
|
run: |
|
||||||
|
uv sync --frozen --group build
|
||||||
|
uv build --wheel
|
||||||
|
|
||||||
|
- name: 🧼 Create an isolated wheel environment
|
||||||
|
shell: bash
|
||||||
|
run: |
|
||||||
|
uv venv --clear --seed .clean-wheel
|
||||||
|
if [ "$RUNNER_OS" = "Windows" ]; then
|
||||||
|
echo "CLEAN_PYTHON=$PWD/.clean-wheel/Scripts/python.exe" >> "$GITHUB_ENV"
|
||||||
|
else
|
||||||
|
echo "CLEAN_PYTHON=$PWD/.clean-wheel/bin/python" >> "$GITHUB_ENV"
|
||||||
|
fi
|
||||||
|
|
||||||
|
- name: 📦 Install and verify the wheel without OpenCV
|
||||||
|
shell: bash
|
||||||
|
run: |
|
||||||
|
uv pip install --python "$CLEAN_PYTHON" --strict dist/*.whl
|
||||||
|
"$CLEAN_PYTHON" .github/scripts/verify_clean_wheel.py \
|
||||||
|
--wheel dist/*.whl \
|
||||||
|
--python "$CLEAN_PYTHON" \
|
||||||
|
--manifest tests/cv2/installed_wheel_fallback_manifest.txt
|
||||||
|
|
||||||
testing-guardian:
|
testing-guardian:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
needs: run-tests
|
needs: [run-tests, clean-wheel]
|
||||||
if: always()
|
if: always()
|
||||||
steps:
|
steps:
|
||||||
- name: 📋 Display test result
|
- name: 📋 Display test result
|
||||||
run: echo "${{ needs.run-tests.result }}"
|
run: echo "tests=${{ needs.run-tests.result }}, clean-wheel=${{ needs.clean-wheel.result }}"
|
||||||
- name: ❌ Fail guardian on test failure
|
- name: ❌ Fail guardian on test failure
|
||||||
if: needs.run-tests.result == 'failure'
|
if: needs.run-tests.result == 'failure' || needs.clean-wheel.result == 'failure'
|
||||||
run: exit 1
|
run: exit 1
|
||||||
# Ensure that cancelled or skipped test runs still cause this guardian job to fail,
|
# Ensure that cancelled or skipped test runs still cause this guardian job to fail,
|
||||||
# using an explicit exit code instead of relying on timeout behavior.
|
# using an explicit exit code instead of relying on timeout behavior.
|
||||||
- name: ⚠️ cancelled or skipped...
|
- name: ⚠️ cancelled or skipped...
|
||||||
if: contains(fromJSON('["cancelled", "skipped"]'), needs.run-tests.result)
|
if: contains(fromJSON('["cancelled", "skipped"]'), needs.run-tests.result) || contains(fromJSON('["cancelled", "skipped"]'), needs.clean-wheel.result)
|
||||||
run: |
|
run: |
|
||||||
echo "run-tests job result is '${{ needs.run-tests.result }}'; failing explicitly."
|
echo "run-tests job result is '${{ needs.run-tests.result }}'; failing explicitly."
|
||||||
exit 1
|
exit 1
|
||||||
- name: ✅ tests succeeded
|
- name: ✅ tests succeeded
|
||||||
if: needs.run-tests.result == 'success'
|
if: needs.run-tests.result == 'success'
|
||||||
run: echo "All tests completed successfully in job 'run-tests'."
|
run: echo "All tests completed successfully."
|
||||||
|
|
|
||||||
|
|
@ -32,12 +32,12 @@ jobs:
|
||||||
timeout-minutes: 10
|
timeout-minutes: 10
|
||||||
steps:
|
steps:
|
||||||
- name: 📥 Checkout the repository
|
- name: 📥 Checkout the repository
|
||||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||||
with:
|
with:
|
||||||
fetch-depth: 0
|
fetch-depth: 0
|
||||||
|
|
||||||
- name: 🐍 Install uv and set Python
|
- name: 🐍 Install uv and set Python
|
||||||
uses: astral-sh/setup-uv@08807647e7069bb48b6ef5acd8ec9567f424441b # v8.1.0
|
uses: astral-sh/setup-uv@c771a70e6277c0a99b617c7a806ffedaca235ff9 # v9.0.0
|
||||||
with:
|
with:
|
||||||
python-version: "3.10"
|
python-version: "3.10"
|
||||||
activate-environment: true
|
activate-environment: true
|
||||||
|
|
@ -62,15 +62,32 @@ jobs:
|
||||||
env:
|
env:
|
||||||
MKDOCS_GIT_COMMITTERS_APIKEY: ${{ secrets.GITHUB_TOKEN }}
|
MKDOCS_GIT_COMMITTERS_APIKEY: ${{ secrets.GITHUB_TOKEN }}
|
||||||
run: |
|
run: |
|
||||||
|
if mike list | grep -Eq '^latest(\s|$)'; then mike delete latest; fi
|
||||||
mike deploy --push latest
|
mike deploy --push latest
|
||||||
|
|
||||||
|
- name: 🏷️ Determine release deployment metadata
|
||||||
|
id: release_metadata
|
||||||
|
run: |
|
||||||
|
is_rc=false
|
||||||
|
release_tag=""
|
||||||
|
if [[ "$GITHUB_EVENT_NAME" == "release" ]]; then
|
||||||
|
release_tag="${GITHUB_REF_NAME#v}"
|
||||||
|
release_tag="${release_tag%.post*}"
|
||||||
|
release_tag_lower="${release_tag,,}"
|
||||||
|
# Match RC suffixes with separators (1.0-rc1, 1.0.rc1) or compact form (1.0rc1).
|
||||||
|
if [[ "$release_tag_lower" =~ (^|[._-])rc[0-9]+$ ]] || [[ "$release_tag_lower" =~ [0-9]rc[0-9]+$ ]]; then
|
||||||
|
is_rc=true
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
echo "is_rc=$is_rc" >> "$GITHUB_OUTPUT"
|
||||||
|
echo "release_tag=$release_tag" >> "$GITHUB_OUTPUT"
|
||||||
|
|
||||||
- name: 🚀 Deploy Release Docs
|
- name: 🚀 Deploy Release Docs
|
||||||
if: github.event_name == 'release' && github.event.action == 'published'
|
if: github.event_name == 'release' && github.event.action == 'published' && steps.release_metadata.outputs.is_rc != 'true'
|
||||||
env:
|
env:
|
||||||
MKDOCS_GIT_COMMITTERS_APIKEY: ${{ secrets.GITHUB_TOKEN }}
|
MKDOCS_GIT_COMMITTERS_APIKEY: ${{ secrets.GITHUB_TOKEN }}
|
||||||
run: |
|
run: |
|
||||||
release_tag="${GITHUB_REF_NAME#v}"
|
mike deploy --push "${{ steps.release_metadata.outputs.release_tag }}"
|
||||||
mike deploy --push "$release_tag"
|
|
||||||
|
|
||||||
# IndexNow key: 0d5d9799b1cc4a39825146388c6781eb
|
# IndexNow key: 0d5d9799b1cc4a39825146388c6781eb
|
||||||
# This key must stay in sync across three files:
|
# This key must stay in sync across three files:
|
||||||
|
|
@ -84,7 +101,7 @@ jobs:
|
||||||
(github.event_name == 'push' && github.ref == 'refs/heads/develop') ||
|
(github.event_name == 'push' && github.ref == 'refs/heads/develop') ||
|
||||||
github.event_name == 'workflow_dispatch' ||
|
github.event_name == 'workflow_dispatch' ||
|
||||||
(github.event_name == 'push' && github.ref == 'refs/heads/release/latest') ||
|
(github.event_name == 'push' && github.ref == 'refs/heads/release/latest') ||
|
||||||
(github.event_name == 'release' && github.event.action == 'published')
|
(github.event_name == 'release' && github.event.action == 'published' && steps.release_metadata.outputs.is_rc != 'true')
|
||||||
run: |
|
run: |
|
||||||
cp docs/robots.txt /tmp/robots.txt
|
cp docs/robots.txt /tmp/robots.txt
|
||||||
cp docs/llms.txt /tmp/llms.txt
|
cp docs/llms.txt /tmp/llms.txt
|
||||||
|
|
|
||||||
|
|
@ -3,9 +3,12 @@ name: Publish Supervision Pre-Releases to PyPI
|
||||||
on:
|
on:
|
||||||
push:
|
push:
|
||||||
tags:
|
tags:
|
||||||
- "[0-9]+.[0-9]+[0-9]+.[0-9]+a[0-9]"
|
- "[0-9]+.[0-9]+.[0-9]+a[0-9]+"
|
||||||
- "[0-9]+.[0-9]+[0-9]+.[0-9]+b[0-9]"
|
- "[0-9]+.[0-9]+.[0-9]+b[0-9]+"
|
||||||
- "[0-9]+.[0-9]+[0-9]+.[0-9]+rc[0-9]"
|
- "[0-9]+.[0-9]+.[0-9]+rc[0-9]+"
|
||||||
|
- "[0-9]+.[0-9]+.[0-9]+.a[0-9]+"
|
||||||
|
- "[0-9]+.[0-9]+.[0-9]+.b[0-9]+"
|
||||||
|
- "[0-9]+.[0-9]+.[0-9]+.rc[0-9]+"
|
||||||
workflow_dispatch:
|
workflow_dispatch:
|
||||||
pull_request:
|
pull_request:
|
||||||
branches: [main, develop]
|
branches: [main, develop]
|
||||||
|
|
@ -43,6 +46,6 @@ jobs:
|
||||||
|
|
||||||
- name: 🚀 Publish to PyPi
|
- name: 🚀 Publish to PyPi
|
||||||
if: github.event_name != 'pull_request'
|
if: github.event_name != 'pull_request'
|
||||||
uses: pypa/gh-action-pypi-publish@cef221092ed1bacb1cc03d23a2d87d1d172e277b # v1.14.0
|
uses: pypa/gh-action-pypi-publish@ba38be9e461d3875417946c167d0b5f3d385a247 # v1.14.1
|
||||||
with:
|
with:
|
||||||
attestations: true
|
attestations: true
|
||||||
|
|
|
||||||
|
|
@ -40,7 +40,7 @@ jobs:
|
||||||
|
|
||||||
- name: 📦 Upload assets to Release
|
- name: 📦 Upload assets to Release
|
||||||
if: github.event_name == 'release'
|
if: github.event_name == 'release'
|
||||||
uses: AButler/upload-release-assets@v3.0
|
uses: AButler/upload-release-assets@v4.0
|
||||||
with:
|
with:
|
||||||
files: "dist/*"
|
files: "dist/*"
|
||||||
repo-token: ${{ secrets.GITHUB_TOKEN }}
|
repo-token: ${{ secrets.GITHUB_TOKEN }}
|
||||||
|
|
@ -48,6 +48,6 @@ jobs:
|
||||||
- name: 🚀 Publish to PyPi
|
- name: 🚀 Publish to PyPi
|
||||||
# We only want to publish to PyPi if the event is a release and it's not a pre-release.
|
# We only want to publish to PyPi if the event is a release and it's not a pre-release.
|
||||||
if: (github.event_name == 'release' && github.event.release.prerelease != true) || github.event_name == 'workflow_dispatch'
|
if: (github.event_name == 'release' && github.event.release.prerelease != true) || github.event_name == 'workflow_dispatch'
|
||||||
uses: pypa/gh-action-pypi-publish@cef221092ed1bacb1cc03d23a2d87d1d172e277b # v1.14.0
|
uses: pypa/gh-action-pypi-publish@ba38be9e461d3875417946c167d0b5f3d385a247 # v1.14.1
|
||||||
with:
|
with:
|
||||||
attestations: true
|
attestations: true
|
||||||
|
|
|
||||||
|
|
@ -38,7 +38,7 @@ jobs:
|
||||||
|
|
||||||
- name: 🚀 Publish to Test-PyPi
|
- name: 🚀 Publish to Test-PyPi
|
||||||
if: github.event_name != 'pull_request'
|
if: github.event_name != 'pull_request'
|
||||||
uses: pypa/gh-action-pypi-publish@cef221092ed1bacb1cc03d23a2d87d1d172e277b # v1.14.0
|
uses: pypa/gh-action-pypi-publish@ba38be9e461d3875417946c167d0b5f3d385a247 # v1.14.1
|
||||||
with:
|
with:
|
||||||
repository-url: https://test.pypi.org/legacy/
|
repository-url: https://test.pypi.org/legacy/
|
||||||
attestations: true
|
attestations: true
|
||||||
|
|
|
||||||
|
|
@ -156,11 +156,27 @@ Desktop.ini
|
||||||
|
|
||||||
# local data
|
# local data
|
||||||
data/
|
data/
|
||||||
|
examples/*/outputs/
|
||||||
|
!src/supervision/_cv2/data/
|
||||||
|
!src/supervision/_cv2/data/*
|
||||||
|
*.mp4
|
||||||
|
*.pt
|
||||||
|
|
||||||
|
# some artifacts
|
||||||
|
/*.py
|
||||||
|
/*.jpg
|
||||||
|
/*.png
|
||||||
|
/*.tif
|
||||||
|
/*.json
|
||||||
|
|
||||||
|
|
||||||
# Claude working scratchpad (plans, lessons, ephemeral artefacts)
|
# Claude working scratchpad (plans, lessons, ephemeral artefacts)
|
||||||
.claude/logs
|
.claude/logs/
|
||||||
.claude/state/
|
.claude/state/
|
||||||
|
.claude/worktrees/
|
||||||
|
.developments/
|
||||||
.plans/
|
.plans/
|
||||||
|
.notes/
|
||||||
.reports/
|
.reports/
|
||||||
.temp/
|
.temp/
|
||||||
.tmp/
|
.tmp/
|
||||||
|
|
@ -169,3 +185,7 @@ _resolutions/
|
||||||
_reviews/
|
_reviews/
|
||||||
tasks/
|
tasks/
|
||||||
*.local.md
|
*.local.md
|
||||||
|
|
||||||
|
output/
|
||||||
|
notebooks/
|
||||||
|
releases/
|
||||||
|
|
|
||||||
|
|
@ -1,5 +1,5 @@
|
||||||
default_language_version:
|
default_language_version:
|
||||||
python: python3
|
python: python3.10
|
||||||
|
|
||||||
ci:
|
ci:
|
||||||
autofix_prs: true
|
autofix_prs: true
|
||||||
|
|
@ -8,6 +8,14 @@ ci:
|
||||||
autoupdate_commit_msg: "chore(pre_commit): ⬆ pre_commit autoupdate"
|
autoupdate_commit_msg: "chore(pre_commit): ⬆ pre_commit autoupdate"
|
||||||
|
|
||||||
repos:
|
repos:
|
||||||
|
- repo: local
|
||||||
|
hooks:
|
||||||
|
- id: check-doctest-fences
|
||||||
|
name: check doctest fences
|
||||||
|
entry: python .github/scripts/check_doctest_fences.py
|
||||||
|
language: python
|
||||||
|
files: ^src/.*\.py$
|
||||||
|
|
||||||
- repo: https://github.com/pre-commit/pre-commit-hooks
|
- repo: https://github.com/pre-commit/pre-commit-hooks
|
||||||
rev: v6.0.0
|
rev: v6.0.0
|
||||||
hooks:
|
hooks:
|
||||||
|
|
@ -28,8 +36,8 @@ repos:
|
||||||
- id: end-of-file-fixer
|
- id: end-of-file-fixer
|
||||||
- id: mixed-line-ending
|
- id: mixed-line-ending
|
||||||
|
|
||||||
- repo: https://github.com/JoC0de/pre-commit-prettier
|
- repo: https://github.com/rbubley/mirrors-prettier
|
||||||
rev: v3.8.3 # using tag; previously pinned SHA when tags were not persistent
|
rev: v3.9.6
|
||||||
hooks:
|
hooks:
|
||||||
- id: prettier
|
- id: prettier
|
||||||
files: \.(ya?ml|toml)$
|
files: \.(ya?ml|toml)$
|
||||||
|
|
@ -37,9 +45,11 @@ repos:
|
||||||
args: ["--print-width=120"]
|
args: ["--print-width=120"]
|
||||||
|
|
||||||
- repo: https://github.com/tox-dev/pyproject-fmt
|
- repo: https://github.com/tox-dev/pyproject-fmt
|
||||||
rev: v2.21.1
|
rev: v2.26.0
|
||||||
hooks:
|
hooks:
|
||||||
- id: pyproject-fmt
|
- id: pyproject-fmt
|
||||||
|
additional_dependencies:
|
||||||
|
- "tomli>=2.0.1"
|
||||||
|
|
||||||
- repo: https://github.com/abravalheri/validate-pyproject
|
- repo: https://github.com/abravalheri/validate-pyproject
|
||||||
rev: v0.25
|
rev: v0.25
|
||||||
|
|
@ -47,7 +57,7 @@ repos:
|
||||||
- id: validate-pyproject
|
- id: validate-pyproject
|
||||||
|
|
||||||
- repo: https://github.com/astral-sh/ruff-pre-commit
|
- repo: https://github.com/astral-sh/ruff-pre-commit
|
||||||
rev: v0.15.12
|
rev: v0.16.1
|
||||||
hooks:
|
hooks:
|
||||||
- id: ruff-check
|
- id: ruff-check
|
||||||
args: ["--fix"]
|
args: ["--fix"]
|
||||||
|
|
@ -58,24 +68,37 @@ repos:
|
||||||
rev: 1.0.0
|
rev: 1.0.0
|
||||||
hooks:
|
hooks:
|
||||||
- id: mdformat
|
- id: mdformat
|
||||||
|
name: mdformat (gfm)
|
||||||
|
exclude: ^docs/
|
||||||
additional_dependencies:
|
additional_dependencies:
|
||||||
|
- "mdformat-frontmatter"
|
||||||
|
- "mdformat-gfm"
|
||||||
|
- "mdformat-ruff"
|
||||||
|
args: ["--number", "--wrap=no"]
|
||||||
|
- id: mdformat
|
||||||
|
name: mdformat (mkdocs)
|
||||||
|
files: ^docs/
|
||||||
|
additional_dependencies:
|
||||||
|
- "mdformat-frontmatter"
|
||||||
- "mdformat-mkdocs[recommended]>=2.1.0"
|
- "mdformat-mkdocs[recommended]>=2.1.0"
|
||||||
- "mdformat-ruff"
|
- "mdformat-ruff"
|
||||||
args: ["--number"]
|
args: ["--number", "--wrap=no"]
|
||||||
exclude: ^(docs/changelog\.md|docs/deprecated\.md)$
|
|
||||||
|
|
||||||
- repo: https://github.com/pre-commit/mirrors-mypy
|
- repo: https://github.com/pre-commit/mirrors-mypy
|
||||||
rev: v1.20.2
|
rev: v2.3.0
|
||||||
hooks:
|
hooks:
|
||||||
- id: mypy
|
- id: mypy
|
||||||
|
language_version: python3.11
|
||||||
additional_dependencies:
|
additional_dependencies:
|
||||||
- "numpy>=2.0"
|
- "numpy>=2.0"
|
||||||
- "types-PyYAML"
|
- "types-PyYAML"
|
||||||
- "types-requests"
|
- "types-requests"
|
||||||
|
- "types-tqdm"
|
||||||
|
|
||||||
- repo: https://github.com/codespell-project/codespell
|
- repo: https://github.com/codespell-project/codespell
|
||||||
rev: v2.4.2
|
rev: v2.4.3
|
||||||
hooks:
|
hooks:
|
||||||
- id: codespell
|
- id: codespell
|
||||||
|
exclude: ^src/supervision/_cv2/data/hershey_fonts\.json$
|
||||||
additional_dependencies:
|
additional_dependencies:
|
||||||
- "tomli>=2.0.1"
|
- "tomli>=2.0.1"
|
||||||
|
|
|
||||||
181
AGENTS.md
181
AGENTS.md
|
|
@ -1,98 +1,153 @@
|
||||||
# Agent Guidelines for `supervision`
|
# Agent Guidelines for `supervision`
|
||||||
|
|
||||||
These instructions define how AI agents (GitHub Copilot, Claude, etc.) should behave when
|
Behave like a senior contributor: precise, efficient, maintainable. When this file and [CONTRIBUTING.md](.github/CONTRIBUTING.md) conflict, **CONTRIBUTING.md wins**.
|
||||||
assigned an issue, task, or multi-step problem in this repository.
|
|
||||||
|
|
||||||
Behave like a senior contributor: precise, efficient, aligned with the project's
|
______________________________________________________________________
|
||||||
philosophy, and focused on maintainability and clarity.
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## 1. Before You Code
|
## 1. Before You Code
|
||||||
|
|
||||||
- Read the task/issue thoroughly before acting.
|
- Read the task thoroughly; group clarifications into one ask.
|
||||||
- Identify missing information; ask **one targeted clarification question** if needed.
|
- Outline a plan before making changes.
|
||||||
- Outline a step-by-step plan before making changes.
|
- Check whether the feature already exists under a different name.
|
||||||
- Check whether the feature or fix already exists under a different name.
|
- Confirm alignment with `src/supervision/` architecture.
|
||||||
- Confirm alignment with the repository's architecture (`src/supervision/`).
|
|
||||||
|
|
||||||
---
|
______________________________________________________________________
|
||||||
|
|
||||||
## 2. Repository Conventions
|
## 2. Repository Architecture
|
||||||
|
|
||||||
All work must follow the conventions of the `supervision` library
|
**Package root**: `src/supervision/` — all library code. **Tests**: `tests/` — mirrors `src/supervision/`. **Public API**: `src/supervision/__init__.py`.
|
||||||
(see [CONTRIBUTING.md](.github/CONTRIBUTING.md) for full details).
|
|
||||||
|
|
||||||
### Branching & Commits
|
```
|
||||||
|
src/supervision/
|
||||||
|
├── detection/
|
||||||
|
│ ├── core.py — Detections dataclass; all model connectors as classmethods
|
||||||
|
│ ├── compact_mask.py — compact mask representation
|
||||||
|
│ ├── vlm.py — VLM connectors (Florence-2, Gemini, Qwen, PaliGemma)
|
||||||
|
│ ├── utils/ — pure NumPy helpers: boxes, converters, iou_and_nms, masks, polygons
|
||||||
|
│ ├── line_zone.py — LineZone
|
||||||
|
│ └── tools/ — InferenceSlicer, PolygonZone, CSVSink, JSONSink, DetectionsSmoother
|
||||||
|
├── annotators/core.py — BoxAnnotator, MaskAnnotator, LabelAnnotator, … each: .annotate(scene, detections)
|
||||||
|
├── key_points/ — KeyPoints, EdgeAnnotator, VertexAnnotator (use this, NOT keypoint/ — see §4)
|
||||||
|
├── tracker/ — DEPRECATED
|
||||||
|
├── dataset/core.py — DetectionDataset / ClassificationDataset (YOLO / COCO / Pascal VOC)
|
||||||
|
├── geometry/core.py — Point, Rect, Vector, Position
|
||||||
|
├── metrics/ — mAP, confusion matrix (requires --extra metrics)
|
||||||
|
├── utils/internal.py — warn_deprecated, deprecated_parameter, internal helpers
|
||||||
|
└── config.py — string constants; always import from here, never use literals
|
||||||
|
```
|
||||||
|
|
||||||
- Branch from `develop` using prefixes: `feat/`, `fix/`, `docs/`, `refactor/`, `test/`, `chore/`.
|
### Key design patterns
|
||||||
- Use **conventional commits**: `feat:`, `fix:`, `docs:`, `refactor:`, `perf:`, `test:`, `chore:`.
|
|
||||||
- PRs must target the `develop` branch.
|
|
||||||
|
|
||||||
### Code Style
|
- **`Detections` is the lingua franca** — every connector, tracker, and annotator speaks `Detections`. New connector = `@classmethod from_<framework>(cls, result) -> Detections`.
|
||||||
|
- **Annotators are composable** — receive `scene` (BGR `np.ndarray`) + `detections`, return annotated copy.
|
||||||
|
- **`data` dict extensibility** — per-detection metadata in `detections.data` as `np.ndarray` aligned with `xyxy`. Keys are constants from `config.py`.
|
||||||
|
- **Vectorized throughout** — NumPy arrays, no Python loops in hot paths. Never write `for det in detections`.
|
||||||
|
- **Lazy-import heavy deps** — `torch`, `transformers`, `ultralytics` must be imported inside the function that needs them, never at module top level.
|
||||||
|
|
||||||
- **Formatting and linting** are enforced by **pre-commit**.
|
______________________________________________________________________
|
||||||
The hook chain typically includes: ruff-check, ruff-format, codespell, mdformat,
|
|
||||||
prettier, pyproject-fmt, and standard pre-commit-hooks (trailing whitespace, YAML, TOML, etc.).
|
|
||||||
- **Type hints**: required on all new code. Type checking with mypy is encouraged but not
|
|
||||||
currently enforced systematically by pre-commit; see [.github/CONTRIBUTING.md](.github/CONTRIBUTING.md)
|
|
||||||
for the latest type-checking expectations.
|
|
||||||
- **Docstrings**: Google Python docstring style. Required for all new functions and classes.
|
|
||||||
Docstrings should include usage examples demonstrating the function with primitive values
|
|
||||||
so they serve as runnable documentation.
|
|
||||||
|
|
||||||
### API Consistency
|
## 3. Agent-Critical Rules
|
||||||
|
|
||||||
- Follow existing naming patterns.
|
These supplement [CONTRIBUTING.md](.github/CONTRIBUTING.md) — covering gaps or agent-specific failure modes.
|
||||||
- Maintain backward compatibility unless explicitly allowed.
|
|
||||||
- Prefer functional utilities over complex classes unless justified.
|
|
||||||
|
|
||||||
### Performance
|
**Doc headings**: `###` max in docstrings and docs. `####` renders identically to bold in mkdocs — use `**bold**` instead.
|
||||||
|
|
||||||
- Avoid unnecessary copies of NumPy arrays.
|
**Type hints**: required on all new code. mypy is enforced by pre-commit (`.pre-commit-config.yaml`).
|
||||||
- Prefer vectorized operations over Python loops in hot paths.
|
|
||||||
- Use OpenCV operations efficiently.
|
|
||||||
|
|
||||||
---
|
**Function docstrings**: every new or modified function, including private helpers and tests, must have a succinct docstring explaining its purpose. Put function-level why/what/how context inside the function docstring, not in a comment before the function. Public APIs still require the full Google-style structure described below.
|
||||||
|
|
||||||
## 3. Implementing Features
|
**Readable argument lists**: do not put multi-branch conditional expressions inside function or constructor arguments. If an argument needs more than a simple `a if condition else b`, assign it to a named local variable before the call.
|
||||||
|
|
||||||
- Provide a minimal, clean implementation.
|
**Inline comments**: write code so the intent is clear from names, small helpers, and straightforward control flow. For non-trivial logic inside a function that still needs context, add concise inline comments explaining why the code exists, what invariant it protects, and how the tricky part works. Do not put comments before functions; use the function docstring instead. Do not comment obvious assignments, mechanical plumbing, lint-only changes, typing-only changes, or pure docs edits.
|
||||||
- Include type hints and Google-style docstrings with usage examples.
|
|
||||||
- All new functionality must be covered with tests, including edge cases.
|
|
||||||
- Add or update documentation (docstrings + mkdocs entries if applicable).
|
|
||||||
- Ensure compatibility with core dependencies: NumPy, OpenCV, SciPy.
|
|
||||||
|
|
||||||
---
|
**Doctest determinism** — output must be reproducible across platforms:
|
||||||
|
|
||||||
## 4. Fixing Bugs
|
- Use `# doctest: +ELLIPSIS` for floats that vary by platform.
|
||||||
|
- Seed any RNG before calling it.
|
||||||
|
- Never assert `dict` or `set` iteration order.
|
||||||
|
- No network or filesystem access outside `supervision/assets/`.
|
||||||
|
|
||||||
1. Reproduce and understand the root cause.
|
**⚠ Test structure** — agents frequently fail here; read [CONTRIBUTING.md §Tests](.github/CONTRIBUTING.md#-tests) carefully: AAA structure, class grouping, parametrize with `pytest.param(..., id="slug")`, one-line docstring per test.
|
||||||
2. Write a test that reproduces the bug (it should fail before the fix).
|
|
||||||
3. Apply a minimal, targeted fix.
|
|
||||||
4. Verify the test passes and no other components break.
|
|
||||||
|
|
||||||
---
|
For branching, commit, code style, and API design conventions see [CONTRIBUTING.md](.github/CONTRIBUTING.md).
|
||||||
|
|
||||||
## 5. Refactoring
|
______________________________________________________________________
|
||||||
|
|
||||||
- Preserve behavior and API stability.
|
## 4. Deprecated Module Aliases
|
||||||
- Improve readability or performance.
|
|
||||||
- Reduce duplication.
|
|
||||||
- Avoid large, sweeping refactors unless explicitly requested.
|
|
||||||
|
|
||||||
---
|
`supervision.keypoint` deprecated since `0.27.0`, removed in `0.31.0`. Always import from `supervision.key_points`, not `supervision.keypoint`.
|
||||||
|
|
||||||
## 6. Before You Commit
|
______________________________________________________________________
|
||||||
|
|
||||||
Always run these before committing:
|
## 5. Deprecating APIs
|
||||||
|
|
||||||
|
**Minimum window**: deprecated APIs must remain for at least **3 minor releases** before removal. Example: deprecated in `0.29.0` → removed in `0.32.0`.
|
||||||
|
|
||||||
|
- Module-level: `supervision.utils.internal.warn_deprecated` in the deprecated module's own `__init__.py`
|
||||||
|
- Parameter renamed (old→new): `supervision.utils.internal.deprecated_parameter` decorator
|
||||||
|
- Public function, method, or class: `@deprecated` from `pydeprecate`
|
||||||
|
|
||||||
|
Always name the version introduced and the removal version:
|
||||||
|
|
||||||
|
```python
|
||||||
|
warn_deprecated("'foo' deprecated in `0.29.0`, removed in `0.32.0`. Use 'bar'.")
|
||||||
|
```
|
||||||
|
|
||||||
|
______________________________________________________________________
|
||||||
|
|
||||||
|
## 6. Implementing Features
|
||||||
|
|
||||||
|
- Minimal implementation; type hints and Google docstrings with usage examples.
|
||||||
|
- Tests covering new functionality and edge cases (see [CONTRIBUTING.md §Tests](.github/CONTRIBUTING.md#-tests)).
|
||||||
|
- Update docstrings and mkdocs entries as needed.
|
||||||
|
- Update [docs/changelog.md](docs/changelog.md) for every functional change or bug fix, including user-visible behavior changes. Skip changelog entries for lint-only, type-only, formatting-only, and pure documentation-only changes.
|
||||||
|
|
||||||
|
**Extending `Detections`**: store metadata in `detections.data` as `np.ndarray` aligned with `xyxy`; define the key as a constant in `config.py` (e.g. `CLASS_NAME_DATA_FIELD`, `ORIENTED_BOX_COORDINATES`).
|
||||||
|
|
||||||
|
**New model connector** (`detection/core.py`):
|
||||||
|
|
||||||
|
```python
|
||||||
|
@classmethod
|
||||||
|
def from_myframework(cls, result) -> "Detections":
|
||||||
|
import myframework # noqa: F401 — lazy import
|
||||||
|
|
||||||
|
xyxy = ... # (N, 4)
|
||||||
|
return cls(
|
||||||
|
xyxy=xyxy,
|
||||||
|
confidence=...,
|
||||||
|
class_id=...,
|
||||||
|
data={CLASS_NAME_DATA_FIELD: np.array([...])},
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
VLM connectors go in `detection/vlm.py`, not `core.py`.
|
||||||
|
|
||||||
|
______________________________________________________________________
|
||||||
|
|
||||||
|
## 7. Bugs & Refactoring
|
||||||
|
|
||||||
|
**Bugs**: reproduce → write failing test → minimal fix → verify no regressions.
|
||||||
|
|
||||||
|
**Refactoring**: preserve behavior and API; reduce duplication; avoid sweeping changes unless requested; apply §5 deprecation when removing public API.
|
||||||
|
|
||||||
|
______________________________________________________________________
|
||||||
|
|
||||||
|
## 8. Before You Commit
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
uv run pytest --cov=supervision
|
uv run pytest --cov=supervision
|
||||||
uv run pre-commit run --all-files
|
uv run pre-commit run --all-files
|
||||||
```
|
```
|
||||||
|
|
||||||
- All pre-commit hooks must pass (formatting, linting, type checking, spell check, etc.).
|
Capture a baseline before changes to avoid introducing new failures:
|
||||||
- All tests must pass before opening a PR. Note: some existing tests in the repo may
|
|
||||||
already be failing — your changes must not introduce new failures.
|
```bash
|
||||||
- Fix any issues reported and re-run until clean.
|
STASH_BEFORE=$(git rev-parse refs/stash 2>/dev/null)
|
||||||
|
git stash push --include-untracked
|
||||||
|
uv run pytest -q 2>&1 | tee /tmp/baseline.txt
|
||||||
|
[ "$(git rev-parse refs/stash 2>/dev/null)" != "$STASH_BEFORE" ] && git stash pop
|
||||||
|
uv run pytest -q 2>&1 | tee /tmp/after.txt
|
||||||
|
diff /tmp/baseline.txt /tmp/after.txt
|
||||||
|
```
|
||||||
|
|
||||||
|
Any test passing in baseline but failing after = blocker.
|
||||||
|
|
|
||||||
18
LICENSE.md
18
LICENSE.md
|
|
@ -2,20 +2,8 @@ MIT License
|
||||||
|
|
||||||
Copyright (c) 2022 Roboflow
|
Copyright (c) 2022 Roboflow
|
||||||
|
|
||||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
Permission is hereby granted, free of charge, to any person obtaining a copy of this software and associated documentation files (the "Software"), to deal in the Software without restriction, including without limitation the rights to use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the Software, and to permit persons to whom the Software is furnished to do so, subject to the following conditions:
|
||||||
of this software and associated documentation files (the "Software"), to deal
|
|
||||||
in the Software without restriction, including without limitation the rights
|
|
||||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
||||||
copies of the Software, and to permit persons to whom the Software is
|
|
||||||
furnished to do so, subject to the following conditions:
|
|
||||||
|
|
||||||
The above copyright notice and this permission notice shall be included in all
|
The above copyright notice and this permission notice shall be included in all copies or substantial portions of the Software.
|
||||||
copies or substantial portions of the Software.
|
|
||||||
|
|
||||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
||||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
||||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
||||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
||||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
||||||
SOFTWARE.
|
|
||||||
|
|
|
||||||
216
README.md
216
README.md
|
|
@ -1,6 +1,6 @@
|
||||||
<div align="center">
|
<div align="center">
|
||||||
<p>
|
<p>
|
||||||
<a align="center" href="" target="https://supervision.roboflow.com">
|
<a align="center" href="https://supervision.roboflow.com" target="_blank">
|
||||||
<img
|
<img
|
||||||
width="100%"
|
width="100%"
|
||||||
src="https://media.roboflow.com/open-source/supervision/rf-supervision-banner.png?updatedAt=1678995927529"
|
src="https://media.roboflow.com/open-source/supervision/rf-supervision-banner.png?updatedAt=1678995927529"
|
||||||
|
|
@ -14,16 +14,9 @@
|
||||||
|
|
||||||
<br>
|
<br>
|
||||||
|
|
||||||
[](https://badge.fury.io/py/supervision)
|
[](https://badge.fury.io/py/supervision) [](https://pypistats.org/packages/supervision) [](LICENSE.md) [](https://badge.fury.io/py/supervision) [](https://codecov.io/gh/roboflow/supervision)
|
||||||
[](https://pypistats.org/packages/supervision)
|
|
||||||
[](LICENSE.md)
|
|
||||||
[](https://badge.fury.io/py/supervision)
|
|
||||||
[](https://codecov.io/gh/roboflow/supervision)
|
|
||||||
|
|
||||||
[](https://snyk.io/advisor/python/supervision)
|
[](https://snyk.io/advisor/python/supervision) [](https://colab.research.google.com/github/roboflow/supervision/blob/main/demo.ipynb) [](https://huggingface.co/spaces/Roboflow/Annotators) [](https://discord.gg/GbfgXGJ8Bk)
|
||||||
[](https://colab.research.google.com/github/roboflow/supervision/blob/main/demo.ipynb)
|
|
||||||
[](https://huggingface.co/spaces/Roboflow/Annotators)
|
|
||||||
[](https://discord.gg/GbfgXGJ8Bk)
|
|
||||||
|
|
||||||
<div align="center">
|
<div align="center">
|
||||||
<a href="https://trendshift.io/repositories/124" target="_blank"><img src="https://trendshift.io/api/badge/repositories/124" alt="roboflow%2Fsupervision | Trendshift" style="width: 250px; height: 55px;" width="250" height="55"/></a>
|
<a href="https://trendshift.io/repositories/124" target="_blank"><img src="https://trendshift.io/api/badge/repositories/124" alt="roboflow%2Fsupervision | Trendshift" style="width: 250px; height: 55px;" width="250" height="55"/></a>
|
||||||
|
|
@ -31,14 +24,29 @@
|
||||||
|
|
||||||
</div>
|
</div>
|
||||||
|
|
||||||
## 👋 hello
|
<details>
|
||||||
|
<summary><strong>📑 Table of Contents</strong></summary>
|
||||||
|
|
||||||
**We write your reusable computer vision tools.** Whether you need to load your dataset from your hard drive, draw detections on an image or video, or count how many detections are in a zone. You can count on us! 🤝
|
- [👋 Hello](#-hello)
|
||||||
|
- [💻 Install](#-install)
|
||||||
|
- [🔥 Quickstart](#-quickstart)
|
||||||
|
- [Models](#models)
|
||||||
|
- [Annotators](#annotators)
|
||||||
|
- [Datasets](#datasets)
|
||||||
|
- [🎬 Tutorials](#-tutorials)
|
||||||
|
- [💜 Built with Supervision](#-built-with-supervision)
|
||||||
|
- [📚 Documentation](#-documentation)
|
||||||
|
- [🏆 Contribution](#-contribution)
|
||||||
|
|
||||||
## 💻 install
|
</details>
|
||||||
|
|
||||||
Pip install the supervision package in a
|
## 👋 Hello
|
||||||
[**Python>=3.9**](https://www.python.org/) environment.
|
|
||||||
|
**We are your essential toolkit for computer vision.** From data loading to real-time zone counting, we provide the building blocks so you can focus on building applications around your models. 🤝
|
||||||
|
|
||||||
|
## 💻 Install
|
||||||
|
|
||||||
|
Pip install the supervision package in a [**Python>=3.10**](https://www.python.org/) environment.
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
pip install supervision
|
pip install supervision
|
||||||
|
|
@ -46,9 +54,9 @@ pip install supervision
|
||||||
|
|
||||||
Read more about conda, mamba, and installing from source in our [guide](https://roboflow.github.io/supervision/).
|
Read more about conda, mamba, and installing from source in our [guide](https://roboflow.github.io/supervision/).
|
||||||
|
|
||||||
## 🔥 quickstart
|
## 🔥 Quickstart
|
||||||
|
|
||||||
### models
|
### Models
|
||||||
|
|
||||||
Supervision was designed to be model agnostic. Just plug in any classification, detection, or segmentation model. For your convenience, we have created [connectors](https://supervision.roboflow.com/latest/detection/core/#detections) for the most popular libraries like Ultralytics, Transformers, MMDetection, or Inference. Other integrations, like `rfdetr`, already return `sv.Detections` directly.
|
Supervision was designed to be model agnostic. Just plug in any classification, detection, or segmentation model. For your convenience, we have created [connectors](https://supervision.roboflow.com/latest/detection/core/#detections) for the most popular libraries like Ultralytics, Transformers, MMDetection, or Inference. Other integrations, like `rfdetr`, already return `sv.Detections` directly.
|
||||||
|
|
||||||
|
|
@ -59,7 +67,7 @@ import supervision as sv
|
||||||
from PIL import Image
|
from PIL import Image
|
||||||
from rfdetr import RFDETRSmall
|
from rfdetr import RFDETRSmall
|
||||||
|
|
||||||
image = Image.open(...)
|
image = Image.open("path/to/image.jpg")
|
||||||
model = RFDETRSmall()
|
model = RFDETRSmall()
|
||||||
detections = model.predict(image, threshold=0.5)
|
detections = model.predict(image, threshold=0.5)
|
||||||
|
|
||||||
|
|
@ -72,25 +80,25 @@ len(detections)
|
||||||
|
|
||||||
- inference
|
- inference
|
||||||
|
|
||||||
Running with [Inference](https://github.com/roboflow/inference) requires a [Roboflow API KEY](https://docs.roboflow.com/api-reference/authentication#retrieve-an-api-key).
|
Running with [Inference](https://github.com/roboflow/inference) requires a [Roboflow API KEY](https://docs.roboflow.com/api-reference/authentication#retrieve-an-api-key).
|
||||||
|
|
||||||
```python
|
```python
|
||||||
import supervision as sv
|
import supervision as sv
|
||||||
from PIL import Image
|
from PIL import Image
|
||||||
from inference import get_model
|
from inference import get_model
|
||||||
|
|
||||||
image = Image.open(...)
|
image = Image.open("path/to/image.jpg")
|
||||||
model = get_model(model_id="rfdetr-small", api_key="ROBOFLOW_API_KEY")
|
model = get_model(model_id="rfdetr-small", api_key="ROBOFLOW_API_KEY")
|
||||||
result = model.infer(image)[0]
|
result = model.infer(image)[0]
|
||||||
detections = sv.Detections.from_inference(result)
|
detections = sv.Detections.from_inference(result)
|
||||||
|
|
||||||
len(detections)
|
len(detections)
|
||||||
# 5
|
# 5
|
||||||
```
|
```
|
||||||
|
|
||||||
</details>
|
</details>
|
||||||
|
|
||||||
### annotators
|
### Annotators
|
||||||
|
|
||||||
Supervision offers a wide range of highly customizable [annotators](https://supervision.roboflow.com/latest/detection/annotators/), allowing you to compose the perfect visualization for your use case.
|
Supervision offers a wide range of highly customizable [annotators](https://supervision.roboflow.com/latest/detection/annotators/), allowing you to compose the perfect visualization for your use case.
|
||||||
|
|
||||||
|
|
@ -98,7 +106,8 @@ Supervision offers a wide range of highly customizable [annotators](https://supe
|
||||||
import cv2
|
import cv2
|
||||||
import supervision as sv
|
import supervision as sv
|
||||||
|
|
||||||
image = cv2.imread(...)
|
image = cv2.imread("path/to/image.jpg")
|
||||||
|
# Assuming detections are obtained from a model
|
||||||
detections = sv.Detections(...)
|
detections = sv.Detections(...)
|
||||||
|
|
||||||
box_annotator = sv.BoxAnnotator()
|
box_annotator = sv.BoxAnnotator()
|
||||||
|
|
@ -107,7 +116,7 @@ annotated_frame = box_annotator.annotate(scene=image.copy(), detections=detectio
|
||||||
|
|
||||||
https://github.com/roboflow/supervision/assets/26109316/691e219c-0565-4403-9218-ab5644f39bce
|
https://github.com/roboflow/supervision/assets/26109316/691e219c-0565-4403-9218-ab5644f39bce
|
||||||
|
|
||||||
### datasets
|
### Datasets
|
||||||
|
|
||||||
Supervision provides a set of [utils](https://supervision.roboflow.com/latest/datasets/core/) that allow you to load, split, merge, and save datasets in one of the supported formats.
|
Supervision provides a set of [utils](https://supervision.roboflow.com/latest/datasets/core/) that allow you to load, split, merge, and save datasets in one of the supported formats.
|
||||||
|
|
||||||
|
|
@ -131,97 +140,97 @@ for path, image, annotation in ds:
|
||||||
pass
|
pass
|
||||||
```
|
```
|
||||||
|
|
||||||
<details close>
|
<details>
|
||||||
<summary>👉 more dataset utils</summary>
|
<summary>👉 more dataset utils</summary>
|
||||||
|
|
||||||
- load
|
- load
|
||||||
|
|
||||||
```python
|
```python
|
||||||
dataset = sv.DetectionDataset.from_yolo(
|
dataset = sv.DetectionDataset.from_yolo(
|
||||||
images_directory_path=...,
|
images_directory_path=...,
|
||||||
annotations_directory_path=...,
|
annotations_directory_path=...,
|
||||||
data_yaml_path=...,
|
data_yaml_path=...,
|
||||||
)
|
)
|
||||||
|
|
||||||
dataset = sv.DetectionDataset.from_pascal_voc(
|
dataset = sv.DetectionDataset.from_pascal_voc(
|
||||||
images_directory_path=...,
|
images_directory_path=...,
|
||||||
annotations_directory_path=...,
|
annotations_directory_path=...,
|
||||||
)
|
)
|
||||||
|
|
||||||
dataset = sv.DetectionDataset.from_coco(
|
dataset = sv.DetectionDataset.from_coco(
|
||||||
images_directory_path=...,
|
images_directory_path=...,
|
||||||
annotations_path=...,
|
annotations_path=...,
|
||||||
)
|
)
|
||||||
```
|
```
|
||||||
|
|
||||||
- split
|
- split
|
||||||
|
|
||||||
```python
|
```python
|
||||||
train_dataset, test_dataset = dataset.split(split_ratio=0.7)
|
train_dataset, test_dataset = dataset.split(split_ratio=0.7)
|
||||||
test_dataset, valid_dataset = test_dataset.split(split_ratio=0.5)
|
test_dataset, valid_dataset = test_dataset.split(split_ratio=0.5)
|
||||||
|
|
||||||
len(train_dataset), len(test_dataset), len(valid_dataset)
|
len(train_dataset), len(test_dataset), len(valid_dataset)
|
||||||
# (700, 150, 150)
|
# (700, 150, 150)
|
||||||
```
|
```
|
||||||
|
|
||||||
- merge
|
- merge
|
||||||
|
|
||||||
```python
|
```python
|
||||||
ds_1 = sv.DetectionDataset(...)
|
ds_1 = sv.DetectionDataset(...)
|
||||||
len(ds_1)
|
len(ds_1)
|
||||||
# 100
|
# 100
|
||||||
ds_1.classes
|
ds_1.classes
|
||||||
# ['dog', 'person']
|
# ['dog', 'person']
|
||||||
|
|
||||||
ds_2 = sv.DetectionDataset(...)
|
ds_2 = sv.DetectionDataset(...)
|
||||||
len(ds_2)
|
len(ds_2)
|
||||||
# 200
|
# 200
|
||||||
ds_2.classes
|
ds_2.classes
|
||||||
# ['cat']
|
# ['cat']
|
||||||
|
|
||||||
ds_merged = sv.DetectionDataset.merge([ds_1, ds_2])
|
ds_merged = sv.DetectionDataset.merge([ds_1, ds_2])
|
||||||
len(ds_merged)
|
len(ds_merged)
|
||||||
# 300
|
# 300
|
||||||
ds_merged.classes
|
ds_merged.classes
|
||||||
# ['cat', 'dog', 'person']
|
# ['cat', 'dog', 'person']
|
||||||
```
|
```
|
||||||
|
|
||||||
- save
|
- save
|
||||||
|
|
||||||
```python
|
```python
|
||||||
dataset.as_yolo(
|
dataset.as_yolo(
|
||||||
images_directory_path=...,
|
images_directory_path=...,
|
||||||
annotations_directory_path=...,
|
annotations_directory_path=...,
|
||||||
data_yaml_path=...,
|
data_yaml_path=...,
|
||||||
)
|
)
|
||||||
|
|
||||||
dataset.as_pascal_voc(
|
dataset.as_pascal_voc(
|
||||||
images_directory_path=...,
|
images_directory_path=...,
|
||||||
annotations_directory_path=...,
|
annotations_directory_path=...,
|
||||||
)
|
)
|
||||||
|
|
||||||
dataset.as_coco(
|
dataset.as_coco(
|
||||||
images_directory_path=...,
|
images_directory_path=...,
|
||||||
annotations_path=...,
|
annotations_path=...,
|
||||||
)
|
)
|
||||||
```
|
```
|
||||||
|
|
||||||
- convert
|
- convert
|
||||||
|
|
||||||
```python
|
```python
|
||||||
sv.DetectionDataset.from_yolo(
|
sv.DetectionDataset.from_yolo(
|
||||||
images_directory_path=...,
|
images_directory_path=...,
|
||||||
annotations_directory_path=...,
|
annotations_directory_path=...,
|
||||||
data_yaml_path=...,
|
data_yaml_path=...,
|
||||||
).as_pascal_voc(
|
).as_pascal_voc(
|
||||||
images_directory_path=...,
|
images_directory_path=...,
|
||||||
annotations_directory_path=...,
|
annotations_directory_path=...,
|
||||||
)
|
)
|
||||||
```
|
```
|
||||||
|
|
||||||
</details>
|
</details>
|
||||||
|
|
||||||
## 🎬 tutorials
|
## 🎬 Tutorials
|
||||||
|
|
||||||
Want to learn how to use Supervision? Explore our [how-to guides](https://supervision.roboflow.com/develop/how_to/detect_and_annotate/), [end-to-end examples](./examples), [cheatsheet](https://roboflow.github.io/cheatsheet-supervision/), and [cookbooks](https://supervision.roboflow.com/develop/cookbooks/)!
|
Want to learn how to use Supervision? Explore our [how-to guides](https://supervision.roboflow.com/develop/how_to/detect_and_annotate/), [end-to-end examples](./examples), [cheatsheet](https://roboflow.github.io/cheatsheet-supervision/), and [cookbooks](https://supervision.roboflow.com/develop/cookbooks/)!
|
||||||
|
|
||||||
|
|
@ -241,7 +250,7 @@ Want to learn how to use Supervision? Explore our [how-to guides](https://superv
|
||||||
<div><strong>Created: 11 Jan 2024</strong></div>
|
<div><strong>Created: 11 Jan 2024</strong></div>
|
||||||
<br/>Learn how to track and estimate the speed of vehicles using YOLO, ByteTrack, and Roboflow Inference. This comprehensive tutorial covers object detection, multi-object tracking, filtering detections, perspective transformation, speed estimation, visualization improvements, and more.</p>
|
<br/>Learn how to track and estimate the speed of vehicles using YOLO, ByteTrack, and Roboflow Inference. This comprehensive tutorial covers object detection, multi-object tracking, filtering detections, perspective transformation, speed estimation, visualization improvements, and more.</p>
|
||||||
|
|
||||||
## 💜 built with supervision
|
## 💜 Built with Supervision
|
||||||
|
|
||||||
Did you build something cool using supervision? [Let us know!](https://github.com/roboflow/supervision/discussions/categories/built-with-supervision)
|
Did you build something cool using supervision? [Let us know!](https://github.com/roboflow/supervision/discussions/categories/built-with-supervision)
|
||||||
|
|
||||||
|
|
@ -251,11 +260,11 @@ https://github.com/roboflow/supervision/assets/26109316/c9436828-9fbf-4c25-ae8c-
|
||||||
|
|
||||||
https://github.com/roboflow/supervision/assets/26109316/3ac6982f-4943-4108-9b7f-51787ef1a69f
|
https://github.com/roboflow/supervision/assets/26109316/3ac6982f-4943-4108-9b7f-51787ef1a69f
|
||||||
|
|
||||||
## 📚 documentation
|
## 📚 Documentation
|
||||||
|
|
||||||
Visit our [documentation](https://roboflow.github.io/supervision) page to learn how supervision can help you build computer vision applications faster and more reliably.
|
Visit our [documentation](https://roboflow.github.io/supervision) page to learn how supervision can help you build computer vision applications faster and more reliably.
|
||||||
|
|
||||||
## 🏆 contribution
|
## 🏆 Contribution
|
||||||
|
|
||||||
We love your input! Please see our [contributing guide](.github/CONTRIBUTING.md) to get started. Thank you 🙏 to all our contributors!
|
We love your input! Please see our [contributing guide](.github/CONTRIBUTING.md) to get started. Thank you 🙏 to all our contributors!
|
||||||
|
|
||||||
|
|
@ -267,8 +276,6 @@ We love your input! Please see our [contributing guide](.github/CONTRIBUTING.md)
|
||||||
|
|
||||||
<br>
|
<br>
|
||||||
|
|
||||||
<div align="center">
|
|
||||||
|
|
||||||
<div align="center">
|
<div align="center">
|
||||||
<a href="https://youtube.com/roboflow">
|
<a href="https://youtube.com/roboflow">
|
||||||
<img
|
<img
|
||||||
|
|
@ -303,6 +310,7 @@ We love your input! Please see our [contributing guide](.github/CONTRIBUTING.md)
|
||||||
src="https://media.roboflow.com/notebooks/template/icons/purple/forum.png?ik-sdk-version=javascript-1.4.3&updatedAt=1672949633584"
|
src="https://media.roboflow.com/notebooks/template/icons/purple/forum.png?ik-sdk-version=javascript-1.4.3&updatedAt=1672949633584"
|
||||||
width="3%"
|
width="3%"
|
||||||
/>
|
/>
|
||||||
|
</a>
|
||||||
<img src="https://raw.githubusercontent.com/ultralytics/assets/main/social/logo-transparent.png" width="3%"/>
|
<img src="https://raw.githubusercontent.com/ultralytics/assets/main/social/logo-transparent.png" width="3%"/>
|
||||||
<a href="https://blog.roboflow.com">
|
<a href="https://blog.roboflow.com">
|
||||||
<img
|
<img
|
||||||
|
|
@ -310,6 +318,4 @@ We love your input! Please see our [contributing guide](.github/CONTRIBUTING.md)
|
||||||
width="3%"
|
width="3%"
|
||||||
/>
|
/>
|
||||||
</a>
|
</a>
|
||||||
</a>
|
|
||||||
</div>
|
</div>
|
||||||
</div>
|
|
||||||
|
|
|
||||||
|
|
@ -5,8 +5,7 @@ description: API reference for supervision's assets module — download sample v
|
||||||
|
|
||||||
# Assets
|
# Assets
|
||||||
|
|
||||||
Supervision offers an assets download utility that allows you to download image and video files
|
Supervision offers an assets download utility that allows you to download image and video files that you can use in your demos.
|
||||||
that you can use in your demos.
|
|
||||||
|
|
||||||
<div class="md-typeset">
|
<div class="md-typeset">
|
||||||
<h2><a href="#supervision.assets.downloader.download_assets.download_assets">download_assets</a></h2>
|
<h2><a href="#supervision.assets.downloader.download_assets.download_assets">download_assets</a></h2>
|
||||||
|
|
|
||||||
1512
docs/changelog.md
1512
docs/changelog.md
File diff suppressed because it is too large
Load Diff
|
|
@ -1,14 +1,13 @@
|
||||||
---
|
---
|
||||||
comments: true
|
comments: true
|
||||||
description: API reference for supervision's DetectionDataset and ClassificationDataset — load, merge, split, and convert datasets in YOLO, COCO, and VOC formats.
|
description: API reference for supervision's DetectionDataset and ClassificationDataset — load, merge, split, and convert datasets in YOLO, COCO, VOC, CreateML, and LabelMe formats.
|
||||||
---
|
---
|
||||||
|
|
||||||
# Datasets
|
# Datasets
|
||||||
|
|
||||||
!!! warning
|
!!! warning
|
||||||
|
|
||||||
Dataset API is still fluid and may change. If you use Dataset API in your project until further notice, freeze the
|
Dataset API is still fluid and may change. If you use Dataset API in your project until further notice, freeze the `supervision` version in your `requirements.txt` or `setup.py`.
|
||||||
`supervision` version in your `requirements.txt` or `setup.py`.
|
|
||||||
|
|
||||||
<div class="md-typeset">
|
<div class="md-typeset">
|
||||||
<h2>DetectionDataset</h2>
|
<h2>DetectionDataset</h2>
|
||||||
|
|
|
||||||
|
|
@ -5,20 +5,32 @@ status: deprecated
|
||||||
|
|
||||||
# Deprecated
|
# Deprecated
|
||||||
|
|
||||||
These features are phased out due to better alternatives or potential issues in future versions. Deprecated functionalities are supported for **five subsequent releases**, providing time for users to transition to updated methods.
|
These features are phased out due to better alternatives or potential issues in future versions. Deprecated functionalities are typically supported for multiple subsequent releases, providing time for users to transition to updated methods.
|
||||||
|
|
||||||
- `overlap_ratio_wh` in [`InferenceSlicer.__init__`](https://supervision.roboflow.com/latest/detection/tools/inference_slicer/) is deprecated and will be removed in `supervision-0.27.0`. Please set it to `None` and use `overlap_wh` instead.
|
- [`sv.ByteTrack`](https://supervision.roboflow.com/latest/trackers/#supervision.tracker.byte_tracker.core.ByteTrack) is deprecated in `supervision-0.28.0` in favour of `ByteTrackTracker` from the external [`trackers`](https://pypi.org/project/trackers/) package (`pip install trackers`). The update method is renamed from `update_with_detections()` to `update()`. Removal is planned for `supervision-0.31.0`.
|
||||||
- `sv.LMM` enum is deprecated and will be removed in `supervision-0.31.0`. Use `sv.VLM` instead.
|
- `supervision.keypoint` module is deprecated in `supervision-0.27.0`; use `supervision.key_points` instead. It will be removed in `supervision-0.31.0`.
|
||||||
- [`sv.Detections.from_lmm`](https://supervision.roboflow.com/0.26.0/detection/core/#supervision.detection.core.Detections.from_lmm) property is deprecated and will be removed in `supervision-0.31.0`. Use [`sv.Detections.from_vlm`](https://supervision.roboflow.com/0.26.0/detection/core/#supervision.detection.core.Detections.from_vlm) instead.
|
- `create_tiles` in `supervision.utils.image` is deprecated in `supervision-0.27.0`. It will be removed in `supervision-0.31.0`.
|
||||||
|
- `ensure_cv2_image_for_processing` in `supervision.utils.conversion` is deprecated in `supervision-0.27.0`. It will be removed in `supervision-0.31.0`.
|
||||||
|
- Keypoint validation utilities in `supervision.validators` are deprecated in `supervision-0.27.0`. They will be removed in `supervision-0.31.0`.
|
||||||
|
- `normalized_xyxy` argument in [`sv.denormalize_boxes`](https://supervision.roboflow.com/latest/detection/utils/boxes/#supervision.detection.utils.boxes.denormalize_boxes) is deprecated in `supervision-0.27.0` and renamed to `xyxy`. Passing `normalized_xyxy=` emits a `FutureWarning`; support will be removed in `supervision-0.31.0`.
|
||||||
|
- `supervision.dataset.utils` import path for [`sv.rle_to_mask`](https://supervision.roboflow.com/latest/detection/utils/converters/#supervision.detection.utils.converters.rle_to_mask) and [`sv.mask_to_rle`](https://supervision.roboflow.com/latest/detection/utils/converters/#supervision.detection.utils.converters.mask_to_rle) is deprecated in `supervision-0.28.0`. These functions moved to `supervision.detection.utils.converters` and will be removed from `supervision.dataset.utils` in `supervision-0.31.0`.
|
||||||
|
- `sv.LMM` enum is deprecated in `supervision-0.27.0` and will be removed in `supervision-0.31.0`. Use `sv.VLM` instead.
|
||||||
|
- [`sv.Detections.from_lmm`](https://supervision.roboflow.com/latest/detection/core/#supervision.detection.core.Detections.from_lmm) classmethod is deprecated in `supervision-0.26.0` and will be removed in `supervision-0.31.0`. Use [`sv.Detections.from_vlm`](https://supervision.roboflow.com/latest/detection/core/#supervision.detection.core.Detections.from_vlm) instead.
|
||||||
|
- `KeyPoints.confidence` is deprecated in `supervision-0.29.0`. Use `KeyPoints.keypoint_confidence` instead. It will be removed in `supervision-0.32.0`.
|
||||||
|
- Public `validate_*` helper functions are deprecated in `supervision-0.29.0` and will be removed in `supervision-0.32.0`. Supervision internals now use private `_validate_*` helpers.
|
||||||
|
|
||||||
# Removed
|
# Removed
|
||||||
|
|
||||||
|
### 0.27.0
|
||||||
|
|
||||||
|
- `overlap_ratio_wh` parameter in [`sv.InferenceSlicer`](https://supervision.roboflow.com/latest/detection/tools/inference_slicer/) has been removed. Use the pixel-based `overlap_wh` parameter instead.
|
||||||
|
- `overlap_filter_strategy` parameter in [`sv.InferenceSlicer`](https://supervision.roboflow.com/latest/detection/tools/inference_slicer/) has been removed. Use `overlap_strategy` instead.
|
||||||
|
|
||||||
### 0.26.0
|
### 0.26.0
|
||||||
|
|
||||||
- The `sv.DetectionDataset.images` property has been removed in `supervision-0.26.0`. Please loop over images with `for path, image, annotation in dataset:`, as that does not require loading all images into memory. Also, constructing `sv.DetectionDataset` with parameter `images` as `Dict[str, np.ndarray]` is deprecated and has been removed in `supervision-0.26.0`. Please pass a list of paths `List[str]` instead.
|
- The `sv.DetectionDataset.images` property has been removed in `supervision-0.26.0`. Please loop over images with `for path, image, annotation in dataset:`, as that does not require loading all images into memory. Also, constructing `sv.DetectionDataset` with parameter `images` as `Dict[str, np.ndarray]` is deprecated and has been removed in `supervision-0.26.0`. Please pass a list of paths `List[str]` instead.
|
||||||
- The name `sv.BoundingBoxAnnotator` is deprecated and has been removed in `supervision-0.26.0`. It has been renamed to [`sv.BoxAnnotator`](https://supervision.roboflow.com/0.22.0/detection/annotators/#supervision.annotators.core.BoxAnnotator).
|
- The name `sv.BoundingBoxAnnotator` is deprecated and has been removed in `supervision-0.26.0`. It has been renamed to [`sv.BoxAnnotator`](https://supervision.roboflow.com/0.22.0/detection/annotators/#supervision.annotators.core.BoxAnnotator).
|
||||||
|
|
||||||
|
|
||||||
### 0.24.0
|
### 0.24.0
|
||||||
|
|
||||||
- The `frame_resolution_wh ` parameter in [`sv.PolygonZone`](detection/tools/polygon_zone.md/#supervision.detection.tools.polygon_zone.PolygonZone) has been removed.
|
- The `frame_resolution_wh ` parameter in [`sv.PolygonZone`](detection/tools/polygon_zone.md/#supervision.detection.tools.polygon_zone.PolygonZone) has been removed.
|
||||||
|
|
|
||||||
|
|
@ -7,510 +7,531 @@ description: API reference for supervision's annotator classes — draw bounding
|
||||||
|
|
||||||
Annotators accept detections and apply box or mask visualizations to the detections. Annotators have many available styles.
|
Annotators accept detections and apply box or mask visualizations to the detections. Annotators have many available styles.
|
||||||
|
|
||||||
=== "Box"
|
=== "Outlines"
|
||||||
|
|
||||||
```python
|
=== "Box"
|
||||||
import supervision as sv
|
|
||||||
|
|
||||||
image = ...
|
```python
|
||||||
detections = sv.Detections(...)
|
import supervision as sv
|
||||||
|
|
||||||
box_annotator = sv.BoxAnnotator()
|
image = ...
|
||||||
annotated_frame = box_annotator.annotate(
|
detections = sv.Detections(...)
|
||||||
scene=image.copy(),
|
|
||||||
detections=detections,
|
|
||||||
)
|
|
||||||
```
|
|
||||||
|
|
||||||
<div class="result" markdown>
|
box_annotator = sv.BoxAnnotator()
|
||||||
|
annotated_frame = box_annotator.annotate(
|
||||||
{ align=center width="800" }
|
scene=image.copy(),
|
||||||
|
detections=detections,
|
||||||
</div>
|
|
||||||
|
|
||||||
=== "RoundBox"
|
|
||||||
|
|
||||||
```python
|
|
||||||
import supervision as sv
|
|
||||||
|
|
||||||
image = ...
|
|
||||||
detections = sv.Detections(...)
|
|
||||||
|
|
||||||
round_box_annotator = sv.RoundBoxAnnotator()
|
|
||||||
annotated_frame = round_box_annotator.annotate(
|
|
||||||
scene=image.copy(),
|
|
||||||
detections=detections,
|
|
||||||
)
|
|
||||||
```
|
|
||||||
|
|
||||||
<div class="result" markdown>
|
|
||||||
|
|
||||||
{ align=center width="800" }
|
|
||||||
|
|
||||||
</div>
|
|
||||||
|
|
||||||
=== "BoxCorner"
|
|
||||||
|
|
||||||
```python
|
|
||||||
import supervision as sv
|
|
||||||
|
|
||||||
image = ...
|
|
||||||
detections = sv.Detections(...)
|
|
||||||
|
|
||||||
corner_annotator = sv.BoxCornerAnnotator()
|
|
||||||
annotated_frame = corner_annotator.annotate(
|
|
||||||
scene=image.copy(),
|
|
||||||
detections=detections,
|
|
||||||
)
|
|
||||||
```
|
|
||||||
|
|
||||||
<div class="result" markdown>
|
|
||||||
|
|
||||||
{ align=center width="800" }
|
|
||||||
|
|
||||||
</div>
|
|
||||||
|
|
||||||
=== "Color"
|
|
||||||
|
|
||||||
```python
|
|
||||||
import supervision as sv
|
|
||||||
|
|
||||||
image = ...
|
|
||||||
detections = sv.Detections(...)
|
|
||||||
|
|
||||||
color_annotator = sv.ColorAnnotator()
|
|
||||||
annotated_frame = color_annotator.annotate(
|
|
||||||
scene=image.copy(),
|
|
||||||
detections=detections,
|
|
||||||
)
|
|
||||||
```
|
|
||||||
|
|
||||||
<div class="result" markdown>
|
|
||||||
|
|
||||||
{ align=center width="800" }
|
|
||||||
|
|
||||||
</div>
|
|
||||||
|
|
||||||
=== "Circle"
|
|
||||||
|
|
||||||
```python
|
|
||||||
import supervision as sv
|
|
||||||
|
|
||||||
image = ...
|
|
||||||
detections = sv.Detections(...)
|
|
||||||
|
|
||||||
circle_annotator = sv.CircleAnnotator()
|
|
||||||
annotated_frame = circle_annotator.annotate(
|
|
||||||
scene=image.copy(),
|
|
||||||
detections=detections,
|
|
||||||
)
|
|
||||||
```
|
|
||||||
|
|
||||||
<div class="result" markdown>
|
|
||||||
|
|
||||||
{ align=center width="800" }
|
|
||||||
|
|
||||||
</div>
|
|
||||||
|
|
||||||
=== "Dot"
|
|
||||||
|
|
||||||
```python
|
|
||||||
import supervision as sv
|
|
||||||
|
|
||||||
image = ...
|
|
||||||
detections = sv.Detections(...)
|
|
||||||
|
|
||||||
dot_annotator = sv.DotAnnotator()
|
|
||||||
annotated_frame = dot_annotator.annotate(
|
|
||||||
scene=image.copy(),
|
|
||||||
detections=detections,
|
|
||||||
)
|
|
||||||
```
|
|
||||||
|
|
||||||
<div class="result" markdown>
|
|
||||||
|
|
||||||
{ align=center width="800" }
|
|
||||||
|
|
||||||
</div>
|
|
||||||
|
|
||||||
=== "Triangle"
|
|
||||||
|
|
||||||
```python
|
|
||||||
import supervision as sv
|
|
||||||
|
|
||||||
image = ...
|
|
||||||
detections = sv.Detections(...)
|
|
||||||
|
|
||||||
triangle_annotator = sv.TriangleAnnotator()
|
|
||||||
annotated_frame = triangle_annotator.annotate(
|
|
||||||
scene=image.copy(),
|
|
||||||
detections=detections,
|
|
||||||
)
|
|
||||||
```
|
|
||||||
|
|
||||||
<div class="result" markdown>
|
|
||||||
|
|
||||||
{ align=center width="800" }
|
|
||||||
|
|
||||||
</div>
|
|
||||||
|
|
||||||
=== "Ellipse"
|
|
||||||
|
|
||||||
```python
|
|
||||||
import supervision as sv
|
|
||||||
|
|
||||||
image = ...
|
|
||||||
detections = sv.Detections(...)
|
|
||||||
|
|
||||||
ellipse_annotator = sv.EllipseAnnotator()
|
|
||||||
annotated_frame = ellipse_annotator.annotate(
|
|
||||||
scene=image.copy(),
|
|
||||||
detections=detections,
|
|
||||||
)
|
|
||||||
```
|
|
||||||
|
|
||||||
<div class="result" markdown>
|
|
||||||
|
|
||||||
{ align=center width="800" }
|
|
||||||
|
|
||||||
</div>
|
|
||||||
|
|
||||||
=== "Halo"
|
|
||||||
|
|
||||||
```python
|
|
||||||
import supervision as sv
|
|
||||||
|
|
||||||
image = ...
|
|
||||||
detections = sv.Detections(...)
|
|
||||||
|
|
||||||
halo_annotator = sv.HaloAnnotator()
|
|
||||||
annotated_frame = halo_annotator.annotate(
|
|
||||||
scene=image.copy(),
|
|
||||||
detections=detections,
|
|
||||||
)
|
|
||||||
```
|
|
||||||
|
|
||||||
<div class="result" markdown>
|
|
||||||
|
|
||||||
{ align=center width="800" }
|
|
||||||
|
|
||||||
</div>
|
|
||||||
|
|
||||||
=== "PercentageBar"
|
|
||||||
|
|
||||||
```python
|
|
||||||
import supervision as sv
|
|
||||||
|
|
||||||
image = ...
|
|
||||||
detections = sv.Detections(...)
|
|
||||||
|
|
||||||
percentage_bar_annotator = sv.PercentageBarAnnotator()
|
|
||||||
annotated_frame = percentage_bar_annotator.annotate(
|
|
||||||
scene=image.copy(),
|
|
||||||
detections=detections,
|
|
||||||
)
|
|
||||||
```
|
|
||||||
|
|
||||||
<div class="result" markdown>
|
|
||||||
|
|
||||||
{ align=center width="800" }
|
|
||||||
|
|
||||||
</div>
|
|
||||||
|
|
||||||
=== "Mask"
|
|
||||||
|
|
||||||
```python
|
|
||||||
import supervision as sv
|
|
||||||
|
|
||||||
image = ...
|
|
||||||
detections = sv.Detections(...)
|
|
||||||
|
|
||||||
mask_annotator = sv.MaskAnnotator()
|
|
||||||
annotated_frame = mask_annotator.annotate(
|
|
||||||
scene=image.copy(),
|
|
||||||
detections=detections,
|
|
||||||
)
|
|
||||||
```
|
|
||||||
|
|
||||||
<div class="result" markdown>
|
|
||||||
|
|
||||||
{ align=center width="800" }
|
|
||||||
|
|
||||||
</div>
|
|
||||||
|
|
||||||
=== "Polygon"
|
|
||||||
|
|
||||||
```python
|
|
||||||
import supervision as sv
|
|
||||||
|
|
||||||
image = ...
|
|
||||||
detections = sv.Detections(...)
|
|
||||||
|
|
||||||
polygon_annotator = sv.PolygonAnnotator()
|
|
||||||
annotated_frame = polygon_annotator.annotate(
|
|
||||||
scene=image.copy(),
|
|
||||||
detections=detections,
|
|
||||||
)
|
|
||||||
```
|
|
||||||
|
|
||||||
<div class="result" markdown>
|
|
||||||
|
|
||||||
{ align=center width="800" }
|
|
||||||
|
|
||||||
</div>
|
|
||||||
|
|
||||||
=== "Label"
|
|
||||||
|
|
||||||
```python
|
|
||||||
import supervision as sv
|
|
||||||
|
|
||||||
image = ...
|
|
||||||
detections = sv.Detections(...)
|
|
||||||
|
|
||||||
labels = [
|
|
||||||
f"{class_name} {confidence:.2f}"
|
|
||||||
for class_name, confidence in zip(
|
|
||||||
detections["class_name"],
|
|
||||||
detections.confidence,
|
|
||||||
)
|
)
|
||||||
]
|
```
|
||||||
|
|
||||||
label_annotator = sv.LabelAnnotator(text_position=sv.Position.CENTER)
|
<div class="result" markdown>
|
||||||
annotated_frame = label_annotator.annotate(
|
|
||||||
scene=image.copy(), detections=detections, labels=labels
|
|
||||||
)
|
|
||||||
```
|
|
||||||
|
|
||||||
<div class="result" markdown>
|
{ align=center width="800" }
|
||||||
|
|
||||||
{ align=center width="800" }
|
</div>
|
||||||
|
|
||||||
</div>
|
=== "RoundBox"
|
||||||
|
|
||||||
=== "RichLabel"
|
```python
|
||||||
|
import supervision as sv
|
||||||
|
|
||||||
```python
|
image = ...
|
||||||
import supervision as sv
|
detections = sv.Detections(...)
|
||||||
|
|
||||||
image = ...
|
round_box_annotator = sv.RoundBoxAnnotator()
|
||||||
detections = sv.Detections(...)
|
annotated_frame = round_box_annotator.annotate(
|
||||||
|
scene=image.copy(),
|
||||||
labels = [
|
detections=detections,
|
||||||
f"{class_name} {confidence:.2f}"
|
|
||||||
for class_name, confidence in zip(
|
|
||||||
detections["class_name"],
|
|
||||||
detections.confidence,
|
|
||||||
)
|
)
|
||||||
]
|
```
|
||||||
|
|
||||||
rich_label_annotator = sv.RichLabelAnnotator(
|
<div class="result" markdown>
|
||||||
font_path="TTF_FONT_PATH",
|
|
||||||
text_position=sv.Position.CENTER,
|
|
||||||
)
|
|
||||||
annotated_frame = rich_label_annotator.annotate(
|
|
||||||
scene=image.copy(),
|
|
||||||
detections=detections,
|
|
||||||
labels=labels,
|
|
||||||
)
|
|
||||||
```
|
|
||||||
|
|
||||||
<div class="result" markdown>
|
{ align=center width="800" }
|
||||||
|
|
||||||
{ align=center width="800" }
|
</div>
|
||||||
|
|
||||||
</div>
|
=== "BoxCorner"
|
||||||
|
|
||||||
=== "Icon"
|
```python
|
||||||
|
import supervision as sv
|
||||||
|
|
||||||
```python
|
image = ...
|
||||||
import supervision as sv
|
detections = sv.Detections(...)
|
||||||
|
|
||||||
image = ...
|
corner_annotator = sv.BoxCornerAnnotator()
|
||||||
detections = sv.Detections(...)
|
annotated_frame = corner_annotator.annotate(
|
||||||
|
scene=image.copy(),
|
||||||
|
detections=detections,
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
icon_paths = ["<ICON_PATH>" for _ in detections]
|
<div class="result" markdown>
|
||||||
|
|
||||||
icon_annotator = sv.IconAnnotator()
|
{ align=center width="800" }
|
||||||
annotated_frame = icon_annotator.annotate(
|
|
||||||
scene=image.copy(),
|
|
||||||
detections=detections,
|
|
||||||
icon_path=icon_paths,
|
|
||||||
)
|
|
||||||
```
|
|
||||||
|
|
||||||
<div class="result" markdown>
|
</div>
|
||||||
|
|
||||||
{ align=center width="800" }
|
=== "Circle"
|
||||||
|
|
||||||
</div>
|
```python
|
||||||
|
import supervision as sv
|
||||||
|
|
||||||
<!-- === "Crop"
|
image = ...
|
||||||
|
detections = sv.Detections(...)
|
||||||
|
|
||||||
```python
|
circle_annotator = sv.CircleAnnotator()
|
||||||
import supervision as sv
|
annotated_frame = circle_annotator.annotate(
|
||||||
|
scene=image.copy(),
|
||||||
|
detections=detections,
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
image = ...
|
<div class="result" markdown>
|
||||||
detections = sv.Detections(...)
|
|
||||||
|
|
||||||
crop_annotator = sv.CropAnnotator()
|
{ align=center width="800" }
|
||||||
annotated_frame = crop_annotator.annotate(
|
|
||||||
scene=image.copy(),
|
|
||||||
detections=detections,
|
|
||||||
)
|
|
||||||
```
|
|
||||||
|
|
||||||
<div class="result" markdown>
|
</div>
|
||||||
|
|
||||||
{ align=center width="800" }
|
=== "Ellipse"
|
||||||
|
|
||||||
</div>
|
```python
|
||||||
|
import supervision as sv
|
||||||
|
|
||||||
-->
|
image = ...
|
||||||
|
detections = sv.Detections(...)
|
||||||
|
|
||||||
=== "Blur"
|
ellipse_annotator = sv.EllipseAnnotator()
|
||||||
|
annotated_frame = ellipse_annotator.annotate(
|
||||||
|
scene=image.copy(),
|
||||||
|
detections=detections,
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
```python
|
<div class="result" markdown>
|
||||||
import supervision as sv
|
|
||||||
|
|
||||||
image = ...
|
{ align=center width="800" }
|
||||||
detections = sv.Detections(...)
|
|
||||||
|
|
||||||
blur_annotator = sv.BlurAnnotator()
|
</div>
|
||||||
annotated_frame = (blur_annotator.annotate(scene=image.copy(), detections=detections),)
|
|
||||||
```
|
|
||||||
|
|
||||||
<div class="result" markdown>
|
=== "Polygon"
|
||||||
|
|
||||||
{ align=center width="800" }
|
```python
|
||||||
|
import supervision as sv
|
||||||
|
|
||||||
</div>
|
image = ...
|
||||||
|
detections = sv.Detections(...)
|
||||||
|
|
||||||
=== "Pixelate"
|
polygon_annotator = sv.PolygonAnnotator()
|
||||||
|
annotated_frame = polygon_annotator.annotate(
|
||||||
|
scene=image.copy(),
|
||||||
|
detections=detections,
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
```python
|
<div class="result" markdown>
|
||||||
import supervision as sv
|
|
||||||
|
|
||||||
image = ...
|
{ align=center width="800" }
|
||||||
detections = sv.Detections(...)
|
|
||||||
|
|
||||||
pixelate_annotator = sv.PixelateAnnotator()
|
</div>
|
||||||
annotated_frame = pixelate_annotator.annotate(
|
|
||||||
scene=image.copy(),
|
|
||||||
detections=detections,
|
|
||||||
)
|
|
||||||
```
|
|
||||||
|
|
||||||
<div class="result" markdown>
|
=== "Shading"
|
||||||
|
|
||||||
{ align=center width="800" }
|
=== "Color"
|
||||||
|
|
||||||
</div>
|
```python
|
||||||
|
import supervision as sv
|
||||||
|
|
||||||
=== "Trace"
|
image = ...
|
||||||
|
detections = sv.Detections(...)
|
||||||
|
|
||||||
```python
|
color_annotator = sv.ColorAnnotator()
|
||||||
import supervision as sv
|
annotated_frame = color_annotator.annotate(
|
||||||
from ultralytics import YOLO
|
scene=image.copy(),
|
||||||
|
detections=detections,
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
model = YOLO("yolov8x.pt")
|
<div class="result" markdown>
|
||||||
|
|
||||||
trace_annotator = sv.TraceAnnotator()
|
{ align=center width="800" }
|
||||||
|
|
||||||
video_info = sv.VideoInfo.from_video_path(video_path="...")
|
</div>
|
||||||
frames_generator = sv.get_video_frames_generator(source_path="...")
|
|
||||||
tracker = sv.ByteTrack()
|
|
||||||
|
|
||||||
with sv.VideoSink(target_path="...", video_info=video_info) as sink:
|
=== "Halo"
|
||||||
for frame in frames_generator:
|
|
||||||
result = model(frame)[0]
|
```python
|
||||||
detections = sv.Detections.from_ultralytics(result)
|
import supervision as sv
|
||||||
detections = tracker.update_with_detections(detections)
|
|
||||||
annotated_frame = trace_annotator.annotate(
|
image = ...
|
||||||
scene=frame.copy(),
|
detections = sv.Detections(...)
|
||||||
detections=detections,
|
|
||||||
|
halo_annotator = sv.HaloAnnotator()
|
||||||
|
annotated_frame = halo_annotator.annotate(
|
||||||
|
scene=image.copy(),
|
||||||
|
detections=detections,
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
<div class="result" markdown>
|
||||||
|
|
||||||
|
{ align=center width="800" }
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
=== "Mask"
|
||||||
|
|
||||||
|
```python
|
||||||
|
import supervision as sv
|
||||||
|
|
||||||
|
image = ...
|
||||||
|
detections = sv.Detections(...)
|
||||||
|
|
||||||
|
mask_annotator = sv.MaskAnnotator()
|
||||||
|
annotated_frame = mask_annotator.annotate(
|
||||||
|
scene=image.copy(),
|
||||||
|
detections=detections,
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
!!! note
|
||||||
|
|
||||||
|
`MaskAnnotator` expects `detections.mask` to contain instance segmentation masks aligned to the image passed to `annotate`. For dense masks, provide a boolean array of shape `(N, H, W)` where `(H, W)` matches the image height and width (it also accepts `sv.CompactMask`). If your model returns framework-specific results, convert them to `sv.Detections` first, for example with `sv.Detections.from_ultralytics(...)` or `sv.Detections.from_inference(...)`.
|
||||||
|
|
||||||
|
<div class="result" markdown>
|
||||||
|
|
||||||
|
{ align=center width="800" }
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
=== "Markers"
|
||||||
|
|
||||||
|
=== "Dot"
|
||||||
|
|
||||||
|
```python
|
||||||
|
import supervision as sv
|
||||||
|
|
||||||
|
image = ...
|
||||||
|
detections = sv.Detections(...)
|
||||||
|
|
||||||
|
dot_annotator = sv.DotAnnotator()
|
||||||
|
annotated_frame = dot_annotator.annotate(
|
||||||
|
scene=image.copy(),
|
||||||
|
detections=detections,
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
<div class="result" markdown>
|
||||||
|
|
||||||
|
{ align=center width="800" }
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
=== "Triangle"
|
||||||
|
|
||||||
|
```python
|
||||||
|
import supervision as sv
|
||||||
|
|
||||||
|
image = ...
|
||||||
|
detections = sv.Detections(...)
|
||||||
|
|
||||||
|
triangle_annotator = sv.TriangleAnnotator()
|
||||||
|
annotated_frame = triangle_annotator.annotate(
|
||||||
|
scene=image.copy(),
|
||||||
|
detections=detections,
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
<div class="result" markdown>
|
||||||
|
|
||||||
|
{ align=center width="800" }
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
=== "Labels"
|
||||||
|
|
||||||
|
=== "Label"
|
||||||
|
|
||||||
|
```python
|
||||||
|
import supervision as sv
|
||||||
|
|
||||||
|
image = ...
|
||||||
|
detections = sv.Detections(...)
|
||||||
|
|
||||||
|
labels = [
|
||||||
|
f"{class_name} {confidence:.2f}"
|
||||||
|
for class_name, confidence in zip(
|
||||||
|
detections["class_name"],
|
||||||
|
detections.confidence,
|
||||||
)
|
)
|
||||||
sink.write_frame(frame=annotated_frame)
|
]
|
||||||
```
|
|
||||||
|
|
||||||
<div class="result" markdown>
|
label_annotator = sv.LabelAnnotator(text_position=sv.Position.CENTER)
|
||||||
|
annotated_frame = label_annotator.annotate(
|
||||||
|
scene=image.copy(), detections=detections, labels=labels
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
{ align=center width="800" }
|
<div class="result" markdown>
|
||||||
|
|
||||||
</div>
|
{ align=center width="800" }
|
||||||
|
|
||||||
=== "HeatMap"
|
</div>
|
||||||
|
|
||||||
```python
|
=== "RichLabel"
|
||||||
import supervision as sv
|
|
||||||
from ultralytics import YOLO
|
|
||||||
|
|
||||||
model = YOLO("yolov8x.pt")
|
```python
|
||||||
|
import supervision as sv
|
||||||
|
|
||||||
heat_map_annotator = sv.HeatMapAnnotator()
|
image = ...
|
||||||
|
detections = sv.Detections(...)
|
||||||
|
|
||||||
video_info = sv.VideoInfo.from_video_path(video_path="...")
|
labels = [
|
||||||
frames_generator = sv.get_video_frames_generator(source_path="...")
|
f"{class_name} {confidence:.2f}"
|
||||||
|
for class_name, confidence in zip(
|
||||||
with sv.VideoSink(target_path="...", video_info=video_info) as sink:
|
detections["class_name"],
|
||||||
for frame in frames_generator:
|
detections.confidence,
|
||||||
result = model(frame)[0]
|
|
||||||
detections = sv.Detections.from_ultralytics(result)
|
|
||||||
annotated_frame = heat_map_annotator.annotate(
|
|
||||||
scene=frame.copy(),
|
|
||||||
detections=detections,
|
|
||||||
)
|
)
|
||||||
sink.write_frame(frame=annotated_frame)
|
]
|
||||||
```
|
|
||||||
|
|
||||||
<div class="result" markdown>
|
rich_label_annotator = sv.RichLabelAnnotator(
|
||||||
|
font_path="TTF_FONT_PATH",
|
||||||
|
text_position=sv.Position.CENTER,
|
||||||
|
)
|
||||||
|
annotated_frame = rich_label_annotator.annotate(
|
||||||
|
scene=image.copy(),
|
||||||
|
detections=detections,
|
||||||
|
labels=labels,
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
{ align=center width="800" }
|
<div class="result" markdown>
|
||||||
|
|
||||||
</div>
|
{ align=center width="800" }
|
||||||
|
|
||||||
=== "Background Color"
|
</div>
|
||||||
|
|
||||||
```python
|
=== "Transformative"
|
||||||
import supervision as sv
|
|
||||||
|
|
||||||
image = ...
|
=== "Blur"
|
||||||
detections = sv.Detections(...)
|
|
||||||
|
|
||||||
background_overlay_annotator = sv.BackgroundOverlayAnnotator()
|
```python
|
||||||
annotated_frame = background_overlay_annotator.annotate(
|
import supervision as sv
|
||||||
scene=image.copy(),
|
|
||||||
detections=detections,
|
|
||||||
)
|
|
||||||
```
|
|
||||||
|
|
||||||
<div class="result" markdown>
|
image = ...
|
||||||
|
detections = sv.Detections(...)
|
||||||
|
|
||||||
{ align=center width="800" }
|
blur_annotator = sv.BlurAnnotator()
|
||||||
|
annotated_frame = blur_annotator.annotate(
|
||||||
|
scene=image.copy(),
|
||||||
|
detections=detections,
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
</div>
|
<div class="result" markdown>
|
||||||
|
|
||||||
=== "Comparison"
|
{ align=center width="800" }
|
||||||
|
|
||||||
```python
|
</div>
|
||||||
import supervision as sv
|
|
||||||
|
|
||||||
image = ...
|
=== "Pixelate"
|
||||||
detections_1 = sv.Detections(...)
|
|
||||||
detections_2 = sv.Detections(...)
|
|
||||||
|
|
||||||
comparison_annotator = sv.ComparisonAnnotator()
|
```python
|
||||||
annotated_frame = comparison_annotator.annotate(
|
import supervision as sv
|
||||||
scene=image.copy(),
|
|
||||||
detections_1=detections_1,
|
|
||||||
detections_2=detections_2,
|
|
||||||
)
|
|
||||||
```
|
|
||||||
|
|
||||||
<div class="result" markdown>
|
image = ...
|
||||||
|
detections = sv.Detections(...)
|
||||||
|
|
||||||
{ align=center width="800" }
|
pixelate_annotator = sv.PixelateAnnotator()
|
||||||
|
annotated_frame = pixelate_annotator.annotate(
|
||||||
|
scene=image.copy(),
|
||||||
|
detections=detections,
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
</div>
|
<div class="result" markdown>
|
||||||
|
|
||||||
|
{ align=center width="800" }
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<!-- === "Crop"
|
||||||
|
|
||||||
|
```python
|
||||||
|
import supervision as sv
|
||||||
|
|
||||||
|
image = ...
|
||||||
|
detections = sv.Detections(...)
|
||||||
|
|
||||||
|
crop_annotator = sv.CropAnnotator()
|
||||||
|
annotated_frame = crop_annotator.annotate(
|
||||||
|
scene=image.copy(),
|
||||||
|
detections=detections,
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
<div class="result" markdown>
|
||||||
|
|
||||||
|
{ align=center width="800" }
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
-->
|
||||||
|
|
||||||
|
=== "Tracking & Aggregation"
|
||||||
|
|
||||||
|
=== "Trace"
|
||||||
|
|
||||||
|
```python
|
||||||
|
import supervision as sv
|
||||||
|
from ultralytics import YOLO
|
||||||
|
|
||||||
|
model = YOLO("yolov8x.pt")
|
||||||
|
|
||||||
|
trace_annotator = sv.TraceAnnotator()
|
||||||
|
|
||||||
|
video_info = sv.VideoInfo.from_video_path(video_path="...")
|
||||||
|
frames_generator = sv.get_video_frames_generator(source_path="...")
|
||||||
|
tracker = sv.ByteTrack()
|
||||||
|
|
||||||
|
with sv.VideoSink(target_path="...", video_info=video_info) as sink:
|
||||||
|
for frame in frames_generator:
|
||||||
|
result = model(frame)[0]
|
||||||
|
detections = sv.Detections.from_ultralytics(result)
|
||||||
|
detections = tracker.update_with_detections(detections)
|
||||||
|
annotated_frame = trace_annotator.annotate(
|
||||||
|
scene=frame.copy(),
|
||||||
|
detections=detections,
|
||||||
|
)
|
||||||
|
sink.write_frame(frame=annotated_frame)
|
||||||
|
```
|
||||||
|
|
||||||
|
<div class="result" markdown>
|
||||||
|
|
||||||
|
{ align=center width="800" }
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
=== "HeatMap"
|
||||||
|
|
||||||
|
```python
|
||||||
|
import supervision as sv
|
||||||
|
from ultralytics import YOLO
|
||||||
|
|
||||||
|
model = YOLO("yolov8x.pt")
|
||||||
|
|
||||||
|
heat_map_annotator = sv.HeatMapAnnotator()
|
||||||
|
|
||||||
|
video_info = sv.VideoInfo.from_video_path(video_path="...")
|
||||||
|
frames_generator = sv.get_video_frames_generator(source_path="...")
|
||||||
|
|
||||||
|
with sv.VideoSink(target_path="...", video_info=video_info) as sink:
|
||||||
|
for frame in frames_generator:
|
||||||
|
result = model(frame)[0]
|
||||||
|
detections = sv.Detections.from_ultralytics(result)
|
||||||
|
annotated_frame = heat_map_annotator.annotate(
|
||||||
|
scene=frame.copy(),
|
||||||
|
detections=detections,
|
||||||
|
)
|
||||||
|
sink.write_frame(frame=annotated_frame)
|
||||||
|
```
|
||||||
|
|
||||||
|
<div class="result" markdown>
|
||||||
|
|
||||||
|
{ align=center width="800" }
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
=== "Others"
|
||||||
|
|
||||||
|
=== "PercentageBar"
|
||||||
|
|
||||||
|
```python
|
||||||
|
import supervision as sv
|
||||||
|
|
||||||
|
image = ...
|
||||||
|
detections = sv.Detections(...)
|
||||||
|
|
||||||
|
percentage_bar_annotator = sv.PercentageBarAnnotator()
|
||||||
|
annotated_frame = percentage_bar_annotator.annotate(
|
||||||
|
scene=image.copy(),
|
||||||
|
detections=detections,
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
<div class="result" markdown>
|
||||||
|
|
||||||
|
{ align=center width="800" }
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
=== "Icon"
|
||||||
|
|
||||||
|
```python
|
||||||
|
import supervision as sv
|
||||||
|
|
||||||
|
image = ...
|
||||||
|
detections = sv.Detections(...)
|
||||||
|
|
||||||
|
icon_paths = ["<ICON_PATH>" for _ in detections]
|
||||||
|
|
||||||
|
icon_annotator = sv.IconAnnotator()
|
||||||
|
annotated_frame = icon_annotator.annotate(
|
||||||
|
scene=image.copy(),
|
||||||
|
detections=detections,
|
||||||
|
icon_path=icon_paths,
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
<div class="result" markdown>
|
||||||
|
|
||||||
|
{ align=center width="800" }
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
=== "Background Color"
|
||||||
|
|
||||||
|
```python
|
||||||
|
import supervision as sv
|
||||||
|
|
||||||
|
image = ...
|
||||||
|
detections = sv.Detections(...)
|
||||||
|
|
||||||
|
background_overlay_annotator = sv.BackgroundOverlayAnnotator()
|
||||||
|
annotated_frame = background_overlay_annotator.annotate(
|
||||||
|
scene=image.copy(),
|
||||||
|
detections=detections,
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
<div class="result" markdown>
|
||||||
|
|
||||||
|
{ align=center width="800" }
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
=== "Comparison"
|
||||||
|
|
||||||
|
```python
|
||||||
|
import supervision as sv
|
||||||
|
|
||||||
|
image = ...
|
||||||
|
detections_1 = sv.Detections(...)
|
||||||
|
detections_2 = sv.Detections(...)
|
||||||
|
|
||||||
|
comparison_annotator = sv.ComparisonAnnotator()
|
||||||
|
annotated_frame = comparison_annotator.annotate(
|
||||||
|
scene=image.copy(),
|
||||||
|
detections_1=detections_1,
|
||||||
|
detections_2=detections_2,
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
<div class="result" markdown>
|
||||||
|
|
||||||
|
{ align=center width="800" }
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
<div class="md-typeset">
|
<div class="md-typeset">
|
||||||
<h2>Try Supervision Annotators on your own image</h2>
|
<h2>Try Supervision Annotators on your own image</h2>
|
||||||
|
|
|
||||||
|
|
@ -4,8 +4,13 @@ comments: true
|
||||||
|
|
||||||
# Legacy Metrics
|
# Legacy Metrics
|
||||||
|
|
||||||
Starting with `0.23.0`, a new metrics module is being introduced to supervision.
|
Starting with `0.23.0`, a new metrics module is being introduced to supervision. Metrics here are part of the legacy evaluation API and will be deprecated in the future.
|
||||||
Metrics here are part of the legacy evaluation API and will be deprecated in the future.
|
|
||||||
|
Install the metrics extra before using this page's APIs:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
pip install "supervision[metrics]"
|
||||||
|
```
|
||||||
|
|
||||||
<div class="md-typeset">
|
<div class="md-typeset">
|
||||||
<h2><a href="#supervision.metrics.detection.ConfusionMatrix">ConfusionMatrix</a></h2>
|
<h2><a href="#supervision.metrics.detection.ConfusionMatrix">ConfusionMatrix</a></h2>
|
||||||
|
|
|
||||||
|
|
@ -4,4 +4,51 @@ comments: true
|
||||||
|
|
||||||
# InferenceSlicer
|
# InferenceSlicer
|
||||||
|
|
||||||
|
## GeoTIFF Datasets
|
||||||
|
|
||||||
|
Install the optional GeoTIFF dependencies before running this example:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
pip install "supervision[geotiff]"
|
||||||
|
wget -O RGB.byte.tif https://raw.githubusercontent.com/rasterio/rasterio/main/tests/data/RGB.byte.tif
|
||||||
|
```
|
||||||
|
|
||||||
|
`InferenceSlicer` can read an open `rasterio` dataset window-by-window. This keeps large GeoTIFFs out of memory while passing each tile to the callback as an `(H, W, C)` NumPy array.
|
||||||
|
|
||||||
|
```python
|
||||||
|
import numpy as np
|
||||||
|
import rasterio
|
||||||
|
import supervision as sv
|
||||||
|
|
||||||
|
|
||||||
|
def callback(tile: np.ndarray) -> sv.Detections:
|
||||||
|
h, w = tile.shape[:2]
|
||||||
|
return sv.Detections(
|
||||||
|
xyxy=np.array([[w * 0.25, h * 0.25, w * 0.75, h * 0.75]], dtype=float),
|
||||||
|
confidence=np.array([0.9]),
|
||||||
|
class_id=np.array([0]),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
slicer = sv.InferenceSlicer(
|
||||||
|
callback=callback,
|
||||||
|
slice_wh=(256, 256),
|
||||||
|
overlap_wh=(64, 64),
|
||||||
|
overlap_filter=sv.OverlapFilter.NONE,
|
||||||
|
)
|
||||||
|
|
||||||
|
with rasterio.open("RGB.byte.tif") as dataset:
|
||||||
|
detections = slicer(dataset)
|
||||||
|
|
||||||
|
print(len(detections))
|
||||||
|
```
|
||||||
|
|
||||||
|
GeoTIFF inputs must use a projected coordinate reference system. Reproject geographic rasters before passing them to `InferenceSlicer`.
|
||||||
|
|
||||||
|
<div class="md-typeset">
|
||||||
|
<h2><a href="#supervision.detection.tools.inference_slicer.WindowedRasterDataset">WindowedRasterDataset</a></h2>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
:::supervision.detection.tools.inference_slicer.WindowedRasterDataset
|
||||||
|
|
||||||
:::supervision.detection.tools.inference_slicer.InferenceSlicer
|
:::supervision.detection.tools.inference_slicer.InferenceSlicer
|
||||||
|
|
|
||||||
|
|
@ -33,3 +33,9 @@ comments: true
|
||||||
</div>
|
</div>
|
||||||
|
|
||||||
:::supervision.detection.utils.boxes.denormalize_boxes
|
:::supervision.detection.utils.boxes.denormalize_boxes
|
||||||
|
|
||||||
|
<div class="md-typeset">
|
||||||
|
<h2><a href="#supervision.detection.utils.boxes.xyxyxyxy_to_xyxy">xyxyxyxy_to_xyxy</a></h2>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
:::supervision.detection.utils.boxes.xyxyxyxy_to_xyxy
|
||||||
|
|
|
||||||
|
|
@ -76,3 +76,9 @@ status: new
|
||||||
</div>
|
</div>
|
||||||
|
|
||||||
:::supervision.detection.utils.converters.mask_to_rle
|
:::supervision.detection.utils.converters.mask_to_rle
|
||||||
|
|
||||||
|
<div class="md-typeset">
|
||||||
|
<h2><a href="#supervision.detection.utils.converters.is_compressed_rle">is_compressed_rle</a></h2>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
:::supervision.detection.utils.converters.is_compressed_rle
|
||||||
|
|
|
||||||
|
|
@ -52,12 +52,24 @@ comments: true
|
||||||
|
|
||||||
:::supervision.detection.utils.iou_and_nms.box_non_max_suppression
|
:::supervision.detection.utils.iou_and_nms.box_non_max_suppression
|
||||||
|
|
||||||
|
<div class="md-typeset">
|
||||||
|
<h2><a href="#supervision.detection.utils.iou_and_nms.box_soft_non_max_suppression">box_soft_non_max_suppression</a></h2>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
:::supervision.detection.utils.iou_and_nms.box_soft_non_max_suppression
|
||||||
|
|
||||||
<div class="md-typeset">
|
<div class="md-typeset">
|
||||||
<h2><a href="#supervision.detection.utils.iou_and_nms.mask_non_max_suppression">mask_non_max_suppression</a></h2>
|
<h2><a href="#supervision.detection.utils.iou_and_nms.mask_non_max_suppression">mask_non_max_suppression</a></h2>
|
||||||
</div>
|
</div>
|
||||||
|
|
||||||
:::supervision.detection.utils.iou_and_nms.mask_non_max_suppression
|
:::supervision.detection.utils.iou_and_nms.mask_non_max_suppression
|
||||||
|
|
||||||
|
<div class="md-typeset">
|
||||||
|
<h2><a href="#supervision.detection.utils.iou_and_nms.mask_soft_non_max_suppression">mask_soft_non_max_suppression</a></h2>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
:::supervision.detection.utils.iou_and_nms.mask_soft_non_max_suppression
|
||||||
|
|
||||||
<div class="md-typeset">
|
<div class="md-typeset">
|
||||||
<h2><a href="#supervision.detection.utils.iou_and_nms.box_non_max_merge">box_non_max_merge</a></h2>
|
<h2><a href="#supervision.detection.utils.iou_and_nms.box_non_max_merge">box_non_max_merge</a></h2>
|
||||||
</div>
|
</div>
|
||||||
|
|
@ -69,3 +81,15 @@ comments: true
|
||||||
</div>
|
</div>
|
||||||
|
|
||||||
:::supervision.detection.utils.iou_and_nms.mask_non_max_merge
|
:::supervision.detection.utils.iou_and_nms.mask_non_max_merge
|
||||||
|
|
||||||
|
<div class="md-typeset">
|
||||||
|
<h2><a href="#supervision.detection.utils.iou_and_nms.oriented_box_non_max_suppression">oriented_box_non_max_suppression</a></h2>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
:::supervision.detection.utils.iou_and_nms.oriented_box_non_max_suppression
|
||||||
|
|
||||||
|
<div class="md-typeset">
|
||||||
|
<h2><a href="#supervision.detection.utils.iou_and_nms.oriented_box_non_max_merge">oriented_box_non_max_merge</a></h2>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
:::supervision.detection.utils.iou_and_nms.oriented_box_non_max_merge
|
||||||
|
|
|
||||||
|
|
@ -5,6 +5,12 @@ status: new
|
||||||
|
|
||||||
# Masks Utils
|
# Masks Utils
|
||||||
|
|
||||||
|
<div class="md-typeset">
|
||||||
|
<h2><a href="#supervision.detection.utils.masks.mask_to_roi">mask_to_roi</a></h2>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
:::supervision.detection.utils.masks.mask_to_roi
|
||||||
|
|
||||||
<div class="md-typeset">
|
<div class="md-typeset">
|
||||||
<h2><a href="#supervision.detection.utils.masks.move_masks">move_masks</a></h2>
|
<h2><a href="#supervision.detection.utils.masks.move_masks">move_masks</a></h2>
|
||||||
</div>
|
</div>
|
||||||
|
|
@ -28,3 +34,9 @@ status: new
|
||||||
</div>
|
</div>
|
||||||
|
|
||||||
:::supervision.detection.utils.masks.filter_segments_by_distance
|
:::supervision.detection.utils.masks.filter_segments_by_distance
|
||||||
|
|
||||||
|
<div class="md-typeset">
|
||||||
|
<h2><a href="#supervision.detection.utils.masks.calculate_masks_centroids">calculate_masks_centroids</a></h2>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
:::supervision.detection.utils.masks.calculate_masks_centroids
|
||||||
|
|
|
||||||
|
|
@ -3,7 +3,25 @@ comments: true
|
||||||
status: new
|
status: new
|
||||||
---
|
---
|
||||||
|
|
||||||
# VLMs Utils
|
# VLM Utils
|
||||||
|
|
||||||
|
<div class="md-typeset">
|
||||||
|
<h2><a href="#supervision.detection.vlm.VLM">VLM</a></h2>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
:::supervision.detection.vlm.VLM
|
||||||
|
|
||||||
|
<div class="md-typeset">
|
||||||
|
<h2><a href="#supervision.detection.vlm.LMM">LMM</a></h2>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
:::supervision.detection.vlm.LMM
|
||||||
|
|
||||||
|
<div class="md-typeset">
|
||||||
|
<h2><a href="#supervision.detection.vlm.validate_vlm_parameters">validate_vlm_parameters</a></h2>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
:::supervision.detection.vlm.validate_vlm_parameters
|
||||||
|
|
||||||
<div class="md-typeset">
|
<div class="md-typeset">
|
||||||
<h2><a href="#supervision.detection.utils.vlms.edit_distance">edit_distance</a></h2>
|
<h2><a href="#supervision.detection.utils.vlms.edit_distance">edit_distance</a></h2>
|
||||||
|
|
|
||||||
29
docs/faq.md
29
docs/faq.md
|
|
@ -25,6 +25,8 @@ pip install "supervision[metrics]"
|
||||||
|
|
||||||
Sample asset utilities are part of the base package under `supervision.assets`.
|
Sample asset utilities are part of the base package under `supervision.assets`.
|
||||||
|
|
||||||
|
Supervision does not install OpenCV. Its image, drawing, and file-video APIs use the included fallback when `cv2` is unavailable, and automatically use a compatible `cv2` already present in your environment. See the [OpenCV migration guide](how_to/opencv_migration.md) when upgrading an existing environment or choosing an OpenCV wheel yourself.
|
||||||
|
|
||||||
## Which object detection models work with Supervision?
|
## Which object detection models work with Supervision?
|
||||||
|
|
||||||
Supervision is model agnostic. `sv.Detections` includes converters for Ultralytics YOLO, Roboflow Inference, Hugging Face Transformers outputs, SAM, Detectron2, MMDetection, YOLO-NAS, PaddleDet, NCNN, Azure AI Vision, and VLM parsers including Florence-2, PaliGemma, Qwen VL, Gemini, DeepSeek VL 2, and Moondream. Keypoint outputs have separate `sv.KeyPoints` converters, including MediaPipe.
|
Supervision is model agnostic. `sv.Detections` includes converters for Ultralytics YOLO, Roboflow Inference, Hugging Face Transformers outputs, SAM, Detectron2, MMDetection, YOLO-NAS, PaddleDet, NCNN, Azure AI Vision, and VLM parsers including Florence-2, PaliGemma, Qwen VL, Gemini, DeepSeek VL 2, and Moondream. Keypoint outputs have separate `sv.KeyPoints` converters, including MediaPipe.
|
||||||
|
|
@ -35,11 +37,11 @@ You can annotate images and video, filter detections, track objects, count objec
|
||||||
|
|
||||||
## How do I track objects across video frames?
|
## How do I track objects across video frames?
|
||||||
|
|
||||||
Assign persistent tracker IDs before visualization. The built-in `sv.ByteTrack` wrapper accepts `Detections` through `update_with_detections()`. After tracking, combine the output with annotators such as `sv.TraceAnnotator`, `sv.BoxAnnotator`, and `sv.LabelAnnotator`.
|
Assign persistent tracker IDs before visualization. The built-in `sv.ByteTrack` wrapper accepts `Detections` through `update_with_detections()`, but it is deprecated in favor of `ByteTrackTracker` from the external `trackers` package. After tracking, combine the output with annotators such as `sv.TraceAnnotator`, `sv.BoxAnnotator`, and `sv.LabelAnnotator`.
|
||||||
|
|
||||||
## What dataset formats does Supervision support?
|
## What dataset formats does Supervision support?
|
||||||
|
|
||||||
For detection datasets, Supervision supports YOLO, COCO JSON, and Pascal VOC. Use `DetectionDataset.from_yolo()`, `DetectionDataset.from_coco()`, or `DetectionDataset.from_pascal_voc()` to load datasets, and the matching `as_*` methods to export them.
|
For detection datasets, Supervision supports YOLO, COCO JSON, Pascal VOC, CreateML, and LabelMe. Use `DetectionDataset.from_yolo()`, `DetectionDataset.from_coco()`, `DetectionDataset.from_pascal_voc()`, `DetectionDataset.from_createml()`, or `DetectionDataset.from_labelme()` to load datasets, and the matching `as_*` methods to export them.
|
||||||
|
|
||||||
## How do I count objects in a zone?
|
## How do I count objects in a zone?
|
||||||
|
|
||||||
|
|
@ -47,12 +49,33 @@ Use `sv.PolygonZone` for arbitrary polygon regions and `sv.LineZone` for line-cr
|
||||||
|
|
||||||
## How do I benchmark a model?
|
## How do I benchmark a model?
|
||||||
|
|
||||||
Use `supervision.metrics.mean_average_precision.MeanAveragePrecision` for mAP and `sv.ConfusionMatrix` for confusion matrices. Accumulate predictions and ground-truth `Detections`, then call `compute()` to calculate metrics.
|
Install `supervision[metrics]`, then use `supervision.metrics.mean_average_precision.MeanAveragePrecision` for mAP and `sv.ConfusionMatrix` for confusion matrices. Accumulate predictions and ground-truth `Detections`, then call `compute()` to calculate metrics.
|
||||||
|
|
||||||
## Is Supervision free to use?
|
## Is Supervision free to use?
|
||||||
|
|
||||||
Yes. Supervision is free and open source under the MIT license.
|
Yes. Supervision is free and open source under the MIT license.
|
||||||
|
|
||||||
|
## How do I process frames from a webcam with supervision?
|
||||||
|
|
||||||
|
Supervision does not support live camera capture. Manage the capture device yourself with `cv2.VideoCapture`, which works regardless of which OpenCV wheel (`opencv-python` or `opencv-python-headless`) is installed, and pass individual frames to supervision annotators:
|
||||||
|
|
||||||
|
```python
|
||||||
|
import cv2 # requires: pip install opencv-python (or opencv-python-headless)
|
||||||
|
import supervision as sv
|
||||||
|
|
||||||
|
cap = cv2.VideoCapture(0)
|
||||||
|
annotator = sv.BoxAnnotator()
|
||||||
|
|
||||||
|
while True:
|
||||||
|
ret, frame = cap.read()
|
||||||
|
if not ret:
|
||||||
|
break
|
||||||
|
# run your detector, then annotate:
|
||||||
|
# annotated = annotator.annotate(frame, detections)
|
||||||
|
|
||||||
|
cap.release()
|
||||||
|
```
|
||||||
|
|
||||||
## Where is the source code?
|
## Where is the source code?
|
||||||
|
|
||||||
The source code is available at [github.com/roboflow/supervision](https://github.com/roboflow/supervision).
|
The source code is available at [github.com/roboflow/supervision](https://github.com/roboflow/supervision).
|
||||||
|
|
|
||||||
|
|
@ -42,14 +42,9 @@ We'll use the following libraries:
|
||||||
- `supervision` to evaluate the model results
|
- `supervision` to evaluate the model results
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
pip install roboflow supervision
|
pip install roboflow inference "supervision[metrics]"
|
||||||
pip install git+https://github.com/roboflow/inference.git@linas/allow-latest-rc-supervision
|
|
||||||
```
|
```
|
||||||
|
|
||||||
!!! info
|
|
||||||
|
|
||||||
We're updating `inference` at the moment. Please install it as shown above.
|
|
||||||
|
|
||||||
Here's how you can download a dataset:
|
Here's how you can download a dataset:
|
||||||
|
|
||||||
```python
|
```python
|
||||||
|
|
@ -130,8 +125,7 @@ Evaluating your model requires careful selection of the dataset. Which images sh
|
||||||
- **Validation Set**: This is the set of images used to validate the model during training. Every Nth training epoch, the model is evaluated on the validation set. Often the training is stopped once the validation loss stops improving. Therefore, even while the images aren't used to train the model, it still indirectly influences the training outcome.
|
- **Validation Set**: This is the set of images used to validate the model during training. Every Nth training epoch, the model is evaluated on the validation set. Often the training is stopped once the validation loss stops improving. Therefore, even while the images aren't used to train the model, it still indirectly influences the training outcome.
|
||||||
- **Test Set**: This is the set of images kept aside for model testing. It is exactly the set you should use for benchmarking. If the dataset was split correctly, none of these images would be shown to the model during training.
|
- **Test Set**: This is the set of images kept aside for model testing. It is exactly the set you should use for benchmarking. If the dataset was split correctly, none of these images would be shown to the model during training.
|
||||||
|
|
||||||
Therefore, an unrelated dataset or the `test` set is the best choice for benchmarking.
|
Therefore, an unrelated dataset or the `test` set is the best choice for benchmarking. Several other problems may arise:
|
||||||
Several other problems may arise:
|
|
||||||
|
|
||||||
- **Extra Classes**: An unrelated dataset may contain additional classes which you may need to [filter out](https://supervision.roboflow.com/how_to/filter_detections/#by-set-of-classes) before computing metrics.
|
- **Extra Classes**: An unrelated dataset may contain additional classes which you may need to [filter out](https://supervision.roboflow.com/how_to/filter_detections/#by-set-of-classes) before computing metrics.
|
||||||
- **Class Mismatch**: In an unrelated dataset, the class names or IDs may be different to what your model produces, you'll need to remap them, which is [shown in this guide](#running-a-model).
|
- **Class Mismatch**: In an unrelated dataset, the class names or IDs may be different to what your model produces, you'll need to remap them, which is [shown in this guide](#running-a-model).
|
||||||
|
|
@ -145,8 +139,7 @@ At this stage, you should have:
|
||||||
- A dataset of labeled images to evaluate the model.
|
- A dataset of labeled images to evaluate the model.
|
||||||
- A model prepared for benchmarking.
|
- A model prepared for benchmarking.
|
||||||
|
|
||||||
With these ready, we can now run the model and obtain predictions.
|
With these ready, we can now run the model and obtain predictions. We'll use `supervision` to create a dataset iterator, and then run the model on each image.
|
||||||
We'll use `supervision` to create a dataset iterator, and then run the model on each image.
|
|
||||||
|
|
||||||
=== "Inference"
|
=== "Inference"
|
||||||
|
|
||||||
|
|
@ -198,8 +191,7 @@ We'll use `supervision` to create a dataset iterator, and then run the model on
|
||||||
|
|
||||||
## Remapping classes
|
## Remapping classes
|
||||||
|
|
||||||
Did you notice an issue in the above logic?
|
Did you notice an issue in the above logic? Since we're using an unrelated dataset, the class names and IDs may be different from what the model was trained on.
|
||||||
Since we're using an unrelated dataset, the class names and IDs may be different from what the model was trained on.
|
|
||||||
|
|
||||||
We need to remap them to match the dataset classes. Here's how to do it:
|
We need to remap them to match the dataset classes. Here's how to do it:
|
||||||
|
|
||||||
|
|
@ -259,8 +251,7 @@ Let's also remove the predictions that are not in the dataset classes.
|
||||||
|
|
||||||
Dataset class names and IDs can be found in the `data.yaml` file, or by printing `dataset.classes`.
|
Dataset class names and IDs can be found in the `data.yaml` file, or by printing `dataset.classes`.
|
||||||
|
|
||||||
Each model will have a different class mapping, so make sure to check the model's documentation. In this case, the model was trained on the COCO dataset, with a class
|
Each model will have a different class mapping, so make sure to check the model's documentation. In this case, the model was trained on the COCO dataset, with a class configuration found [here](https://github.com/ultralytics/ultralytics/blob/main/ultralytics/cfg/datasets/coco8.yaml).
|
||||||
configuration found [here](https://github.com/ultralytics/ultralytics/blob/main/ultralytics/cfg/datasets/coco8.yaml).
|
|
||||||
|
|
||||||
```python
|
```python
|
||||||
import supervision as sv
|
import supervision as sv
|
||||||
|
|
@ -293,8 +284,7 @@ Let's also remove the predictions that are not in the dataset classes.
|
||||||
|
|
||||||
## Visualizing Predictions
|
## Visualizing Predictions
|
||||||
|
|
||||||
The first step in evaluating your model’s performance is to visualize its predictions.
|
The first step in evaluating your model’s performance is to visualize its predictions. This gives an intuitive sense of how well your model is detecting objects and where it might be failing.
|
||||||
This gives an intuitive sense of how well your model is detecting objects and where it might be failing.
|
|
||||||
|
|
||||||
```python
|
```python
|
||||||
import supervision as sv
|
import supervision as sv
|
||||||
|
|
@ -334,6 +324,20 @@ Here, predictions in purple are targets (ground truth), and predictions in teal
|
||||||
|
|
||||||
See [annotator documentation](https://supervision.roboflow.com/latest/detection/annotators/) for even more options.
|
See [annotator documentation](https://supervision.roboflow.com/latest/detection/annotators/) for even more options.
|
||||||
|
|
||||||
|
## Visual Benchmarking
|
||||||
|
|
||||||
|
To inspect where a model succeeds and fails, pass `save_directory_path` to `sv.ConfusionMatrix.benchmark(...)`. For every dataset image it writes a 2x2 result grid — `Ground Truth`, `True Positives`, `False Positives`, and `False Negatives` panels — directly into that directory, reusing the original image filenames. This makes it easy to skim through per-image outcomes alongside the aggregate confusion matrix.
|
||||||
|
|
||||||
|
```python
|
||||||
|
import supervision as sv
|
||||||
|
|
||||||
|
confusion_matrix = sv.ConfusionMatrix.benchmark(
|
||||||
|
dataset=test_set,
|
||||||
|
callback=callback,
|
||||||
|
save_directory_path="./results",
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
## Benchmarking Metrics
|
## Benchmarking Metrics
|
||||||
|
|
||||||
With multiple models, fine details matter. Visual inspection may not be enough. `supervision` provides a collection of metrics that help obtain precise numerical results of model performance.
|
With multiple models, fine details matter. Visual inspection may not be enough. `supervision` provides a collection of metrics that help obtain precise numerical results of model performance.
|
||||||
|
|
@ -467,7 +471,7 @@ Yes, if you want to evaluate their bounding boxes. Convert model outputs to `Det
|
||||||
|
|
||||||
### What is a ConfusionMatrix and how do I use it?
|
### What is a ConfusionMatrix and how do I use it?
|
||||||
|
|
||||||
`sv.ConfusionMatrix` visualizes true positives, false positives, and false negatives per class. Create one with `sv.ConfusionMatrix.from_detections(predictions=predictions, targets=targets, classes=classes, conf_threshold=0.5, iou_threshold=0.5)`, then call `metric.plot()` to render a heatmap.
|
`sv.ConfusionMatrix` visualizes true positives, false positives, and false negatives per class. Create one with `sv.ConfusionMatrix.from_detections(predictions=predictions, targets=targets, classes=classes, conf_threshold=0.5, iou_threshold=0.5)`, then call `confusion_matrix.plot()` to render a heatmap. If you want per-image validation visualizations saved to disk, pass `save_directory_path="./results"` to `sv.ConfusionMatrix.benchmark(...)`; it will write 2x2 result grids directly into that directory using the original image filenames, with `Ground Truth`, `True Positives`, `False Positives`, and `False Negatives` panels.
|
||||||
|
|
||||||
## Author
|
## Author
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -24,7 +24,7 @@ download_assets(VideoAssets.VEHICLES_2)
|
||||||
|
|
||||||
First, we need to initialize a model. Let's use a YOLOv8 model with the default COCO checkpoint. We also need to load a video on which to run inference.
|
First, we need to initialize a model. Let's use a YOLOv8 model with the default COCO checkpoint. We also need to load a video on which to run inference.
|
||||||
|
|
||||||
Create a YOLO model instance and load the source video using supervision's `VideoInfo` helper. The model will process each frame during inference, while `VideoInfo` extracts resolution and frame-rate metadata needed by the polygon zone annotator. A shared color palette ensures consistent zone coloring throughout the output video.
|
Create a YOLO model instance and download the source video. The model will process each frame during inference. A shared color palette ensures consistent zone coloring throughout the output video.
|
||||||
|
|
||||||
```python
|
```python
|
||||||
import numpy as np
|
import numpy as np
|
||||||
|
|
@ -32,13 +32,13 @@ import supervision as sv
|
||||||
import cv2
|
import cv2
|
||||||
|
|
||||||
from ultralytics import YOLO
|
from ultralytics import YOLO
|
||||||
|
from supervision.assets import VideoAssets, download_assets
|
||||||
|
|
||||||
model = YOLO("yolov8s.pt")
|
model = YOLO("yolov8s.pt")
|
||||||
|
|
||||||
VIDEO = str(VideoAssets.VEHICLES_2)
|
VIDEO = download_assets(VideoAssets.VEHICLES_2)
|
||||||
|
|
||||||
colors = sv.ColorPalette.default()
|
colors = sv.ColorPalette.DEFAULT
|
||||||
video_info = sv.VideoInfo.from_video_path(VIDEO)
|
|
||||||
```
|
```
|
||||||
|
|
||||||
## Calculate Coordinates
|
## Calculate Coordinates
|
||||||
|
|
@ -80,10 +80,7 @@ With the coordinates of the zones to draw ready, we can set up our zones:
|
||||||
Instantiate a `PolygonZone` for each polygon array, pairing it with a `PolygonZoneAnnotator` for visual overlay and a `BoxAnnotator` for drawing detection boxes. Each zone will later trigger on incoming detections to determine which objects fall inside its boundaries, enabling per-zone counting in the inference callback.
|
Instantiate a `PolygonZone` for each polygon array, pairing it with a `PolygonZoneAnnotator` for visual overlay and a `BoxAnnotator` for drawing detection boxes. Each zone will later trigger on incoming detections to determine which objects fall inside its boundaries, enabling per-zone counting in the inference callback.
|
||||||
|
|
||||||
```python
|
```python
|
||||||
zones = [
|
zones = [sv.PolygonZone(polygon=polygon) for polygon in polygons]
|
||||||
sv.PolygonZone(polygon=polygon, frame_resolution_wh=video_info.resolution_wh)
|
|
||||||
for polygon in polygons
|
|
||||||
]
|
|
||||||
zone_annotators = [
|
zone_annotators = [
|
||||||
sv.PolygonZoneAnnotator(
|
sv.PolygonZoneAnnotator(
|
||||||
zone=zone,
|
zone=zone,
|
||||||
|
|
@ -98,8 +95,6 @@ box_annotators = [
|
||||||
sv.BoxAnnotator(
|
sv.BoxAnnotator(
|
||||||
color=colors.by_idx(index),
|
color=colors.by_idx(index),
|
||||||
thickness=4,
|
thickness=4,
|
||||||
text_thickness=4,
|
|
||||||
text_scale=2,
|
|
||||||
)
|
)
|
||||||
for index in range(len(polygons))
|
for index in range(len(polygons))
|
||||||
]
|
]
|
||||||
|
|
@ -121,9 +116,7 @@ def process_frame(frame: np.ndarray, i) -> np.ndarray:
|
||||||
):
|
):
|
||||||
mask = zone.trigger(detections=detections)
|
mask = zone.trigger(detections=detections)
|
||||||
detections_filtered = detections[mask]
|
detections_filtered = detections[mask]
|
||||||
frame = box_annotator.annotate(
|
frame = box_annotator.annotate(scene=frame, detections=detections_filtered)
|
||||||
scene=frame, detections=detections_filtered, skip_label=True
|
|
||||||
)
|
|
||||||
frame = zone_annotator.annotate(scene=frame)
|
frame = zone_annotator.annotate(scene=frame)
|
||||||
|
|
||||||
return frame
|
return frame
|
||||||
|
|
|
||||||
|
|
@ -13,20 +13,25 @@ date_modified: 2026-04-22
|
||||||
|
|
||||||
# Detect and Annotate
|
# Detect and Annotate
|
||||||
|
|
||||||
Supervision provides a seamless process for annotating predictions generated by various
|
!!! tip "Sample Image"
|
||||||
object detection and segmentation models. This guide shows how to perform inference
|
|
||||||
with the [Inference](https://github.com/roboflow/inference),
|
Don't have an image? Download the one used in this tutorial:
|
||||||
[Ultralytics](https://github.com/ultralytics/ultralytics) or
|
|
||||||
[Transformers](https://github.com/huggingface/transformers) packages. Following this,
|
```bash
|
||||||
you'll learn how to import these predictions into Supervision and use them to annotate
|
wget https://media.roboflow.com/notebooks/examples/dog.jpeg
|
||||||
source image.
|
```
|
||||||
|
|
||||||
|
```
|
||||||
|
Then replace `<SOURCE_IMAGE_PATH>` with `"dog.jpeg"`.
|
||||||
|
```
|
||||||
|
|
||||||
|
Supervision provides a seamless process for annotating predictions generated by various object detection and segmentation models. This guide shows how to perform inference with the [Inference](https://github.com/roboflow/inference), [Ultralytics](https://github.com/ultralytics/ultralytics) or [Transformers](https://github.com/huggingface/transformers) packages. Following this, you'll learn how to import these predictions into Supervision and use them to annotate source image.
|
||||||
|
|
||||||

|

|
||||||
|
|
||||||
## Run Detection
|
## Run Detection
|
||||||
|
|
||||||
First, you'll need to obtain predictions from your object detection or segmentation
|
First, you'll need to obtain predictions from your object detection or segmentation model.
|
||||||
model.
|
|
||||||
|
|
||||||
To run inference, initialize your chosen model and pass the source image to its predict or infer method. Supervision supports Roboflow Inference, Ultralytics YOLO, and Hugging Face Transformers -- select the tab matching your framework. The result is a framework-specific object you will convert to a `Detections` instance in the next step.
|
To run inference, initialize your chosen model and pass the source image to its predict or infer method. Supervision supports Roboflow Inference, Ultralytics YOLO, and Hugging Face Transformers -- select the tab matching your framework. The result is a framework-specific object you will convert to a `Detections` instance in the next step.
|
||||||
|
|
||||||
|
|
@ -37,7 +42,7 @@ To run inference, initialize your chosen model and pass the source image to its
|
||||||
from inference import get_model
|
from inference import get_model
|
||||||
|
|
||||||
model = get_model(model_id="yolov8n-640")
|
model = get_model(model_id="yolov8n-640")
|
||||||
image = cv2.imread("<SOURCE_IMAGE_PATH>")
|
image = cv2.imread("dog.jpeg")
|
||||||
results = model.infer(image)[0]
|
results = model.infer(image)[0]
|
||||||
```
|
```
|
||||||
|
|
||||||
|
|
@ -48,7 +53,7 @@ To run inference, initialize your chosen model and pass the source image to its
|
||||||
from ultralytics import YOLO
|
from ultralytics import YOLO
|
||||||
|
|
||||||
model = YOLO("yolov8n.pt")
|
model = YOLO("yolov8n.pt")
|
||||||
image = cv2.imread("<SOURCE_IMAGE_PATH>")
|
image = cv2.imread("dog.jpeg")
|
||||||
results = model(image)[0]
|
results = model(image)[0]
|
||||||
```
|
```
|
||||||
|
|
||||||
|
|
@ -62,7 +67,7 @@ To run inference, initialize your chosen model and pass the source image to its
|
||||||
processor = DetrImageProcessor.from_pretrained("facebook/detr-resnet-50")
|
processor = DetrImageProcessor.from_pretrained("facebook/detr-resnet-50")
|
||||||
model = DetrForObjectDetection.from_pretrained("facebook/detr-resnet-50")
|
model = DetrForObjectDetection.from_pretrained("facebook/detr-resnet-50")
|
||||||
|
|
||||||
image = Image.open("<SOURCE_IMAGE_PATH>")
|
image = Image.open("dog.jpeg")
|
||||||
inputs = processor(images=image, return_tensors="pt")
|
inputs = processor(images=image, return_tensors="pt")
|
||||||
|
|
||||||
with torch.no_grad():
|
with torch.no_grad():
|
||||||
|
|
@ -91,7 +96,7 @@ Each supported framework has a dedicated class method on `sv.Detections` that co
|
||||||
from inference import get_model
|
from inference import get_model
|
||||||
|
|
||||||
model = get_model(model_id="yolov8n-640")
|
model = get_model(model_id="yolov8n-640")
|
||||||
image = cv2.imread("<SOURCE_IMAGE_PATH>")
|
image = cv2.imread("dog.jpeg")
|
||||||
results = model.infer(image)[0]
|
results = model.infer(image)[0]
|
||||||
detections = sv.Detections.from_inference(results)
|
detections = sv.Detections.from_inference(results)
|
||||||
```
|
```
|
||||||
|
|
@ -106,7 +111,7 @@ Each supported framework has a dedicated class method on `sv.Detections` that co
|
||||||
from ultralytics import YOLO
|
from ultralytics import YOLO
|
||||||
|
|
||||||
model = YOLO("yolov8n.pt")
|
model = YOLO("yolov8n.pt")
|
||||||
image = cv2.imread("<SOURCE_IMAGE_PATH>")
|
image = cv2.imread("dog.jpeg")
|
||||||
results = model(image)[0]
|
results = model(image)[0]
|
||||||
detections = sv.Detections.from_ultralytics(results)
|
detections = sv.Detections.from_ultralytics(results)
|
||||||
```
|
```
|
||||||
|
|
@ -124,7 +129,7 @@ Each supported framework has a dedicated class method on `sv.Detections` that co
|
||||||
processor = DetrImageProcessor.from_pretrained("facebook/detr-resnet-50")
|
processor = DetrImageProcessor.from_pretrained("facebook/detr-resnet-50")
|
||||||
model = DetrForObjectDetection.from_pretrained("facebook/detr-resnet-50")
|
model = DetrForObjectDetection.from_pretrained("facebook/detr-resnet-50")
|
||||||
|
|
||||||
image = Image.open("<SOURCE_IMAGE_PATH>")
|
image = Image.open("dog.jpeg")
|
||||||
inputs = processor(images=image, return_tensors="pt")
|
inputs = processor(images=image, return_tensors="pt")
|
||||||
|
|
||||||
with torch.no_grad():
|
with torch.no_grad():
|
||||||
|
|
@ -161,7 +166,7 @@ To draw bounding boxes and class labels on your image, create a `BoxAnnotator` a
|
||||||
from inference import get_model
|
from inference import get_model
|
||||||
|
|
||||||
model = get_model(model_id="yolov8n-640")
|
model = get_model(model_id="yolov8n-640")
|
||||||
image = cv2.imread("<SOURCE_IMAGE_PATH>")
|
image = cv2.imread("dog.jpeg")
|
||||||
results = model.infer(image)[0]
|
results = model.infer(image)[0]
|
||||||
detections = sv.Detections.from_inference(results)
|
detections = sv.Detections.from_inference(results)
|
||||||
|
|
||||||
|
|
@ -182,7 +187,7 @@ To draw bounding boxes and class labels on your image, create a `BoxAnnotator` a
|
||||||
from ultralytics import YOLO
|
from ultralytics import YOLO
|
||||||
|
|
||||||
model = YOLO("yolov8n.pt")
|
model = YOLO("yolov8n.pt")
|
||||||
image = cv2.imread("<SOURCE_IMAGE_PATH>")
|
image = cv2.imread("dog.jpeg")
|
||||||
results = model(image)[0]
|
results = model(image)[0]
|
||||||
detections = sv.Detections.from_ultralytics(results)
|
detections = sv.Detections.from_ultralytics(results)
|
||||||
|
|
||||||
|
|
@ -206,7 +211,7 @@ To draw bounding boxes and class labels on your image, create a `BoxAnnotator` a
|
||||||
processor = DetrImageProcessor.from_pretrained("facebook/detr-resnet-50")
|
processor = DetrImageProcessor.from_pretrained("facebook/detr-resnet-50")
|
||||||
model = DetrForObjectDetection.from_pretrained("facebook/detr-resnet-50")
|
model = DetrForObjectDetection.from_pretrained("facebook/detr-resnet-50")
|
||||||
|
|
||||||
image = Image.open("<SOURCE_IMAGE_PATH>")
|
image = Image.open("dog.jpeg")
|
||||||
inputs = processor(images=image, return_tensors="pt")
|
inputs = processor(images=image, return_tensors="pt")
|
||||||
|
|
||||||
with torch.no_grad():
|
with torch.no_grad():
|
||||||
|
|
@ -233,9 +238,7 @@ To draw bounding boxes and class labels on your image, create a `BoxAnnotator` a
|
||||||
|
|
||||||
## Display Custom Labels
|
## Display Custom Labels
|
||||||
|
|
||||||
By default, [`sv.LabelAnnotator`](https://supervision.roboflow.com/latest/detection/annotators/#supervision.annotators.core.LabelAnnotator)
|
By default, [`sv.LabelAnnotator`](https://supervision.roboflow.com/latest/detection/annotators/#supervision.annotators.core.LabelAnnotator) will label each detection with its `class_name` (if possible) or `class_id`. You can override this behavior by passing a list of custom `labels` to the `annotate` method.
|
||||||
will label each detection with its `class_name` (if possible) or `class_id`. You can
|
|
||||||
override this behavior by passing a list of custom `labels` to the `annotate` method.
|
|
||||||
|
|
||||||
=== "Inference"
|
=== "Inference"
|
||||||
|
|
||||||
|
|
@ -245,7 +248,7 @@ override this behavior by passing a list of custom `labels` to the `annotate` me
|
||||||
from inference import get_model
|
from inference import get_model
|
||||||
|
|
||||||
model = get_model(model_id="yolov8n-640")
|
model = get_model(model_id="yolov8n-640")
|
||||||
image = cv2.imread("<SOURCE_IMAGE_PATH>")
|
image = cv2.imread("dog.jpeg")
|
||||||
results = model.infer(image)[0]
|
results = model.infer(image)[0]
|
||||||
detections = sv.Detections.from_inference(results)
|
detections = sv.Detections.from_inference(results)
|
||||||
|
|
||||||
|
|
@ -272,7 +275,7 @@ override this behavior by passing a list of custom `labels` to the `annotate` me
|
||||||
from ultralytics import YOLO
|
from ultralytics import YOLO
|
||||||
|
|
||||||
model = YOLO("yolov8n.pt")
|
model = YOLO("yolov8n.pt")
|
||||||
image = cv2.imread("<SOURCE_IMAGE_PATH>")
|
image = cv2.imread("dog.jpeg")
|
||||||
results = model(image)[0]
|
results = model(image)[0]
|
||||||
detections = sv.Detections.from_ultralytics(results)
|
detections = sv.Detections.from_ultralytics(results)
|
||||||
|
|
||||||
|
|
@ -302,7 +305,7 @@ override this behavior by passing a list of custom `labels` to the `annotate` me
|
||||||
processor = DetrImageProcessor.from_pretrained("facebook/detr-resnet-50")
|
processor = DetrImageProcessor.from_pretrained("facebook/detr-resnet-50")
|
||||||
model = DetrForObjectDetection.from_pretrained("facebook/detr-resnet-50")
|
model = DetrForObjectDetection.from_pretrained("facebook/detr-resnet-50")
|
||||||
|
|
||||||
image = Image.open("<SOURCE_IMAGE_PATH>")
|
image = Image.open("dog.jpeg")
|
||||||
inputs = processor(images=image, return_tensors="pt")
|
inputs = processor(images=image, return_tensors="pt")
|
||||||
|
|
||||||
with torch.no_grad():
|
with torch.no_grad():
|
||||||
|
|
@ -335,11 +338,7 @@ override this behavior by passing a list of custom `labels` to the `annotate` me
|
||||||
|
|
||||||
## Annotate Image with Segmentations
|
## Annotate Image with Segmentations
|
||||||
|
|
||||||
If you are running the segmentation model
|
If you are running the segmentation model [`sv.MaskAnnotator`](https://supervision.roboflow.com/latest/detection/annotators/#supervision.annotators.core.MaskAnnotator) is a drop-in replacement for [`sv.BoxAnnotator`](https://supervision.roboflow.com/latest/detection/annotators/#supervision.annotators.core.BoxAnnotator) that will allow you to draw masks instead of boxes.
|
||||||
[`sv.MaskAnnotator`](https://supervision.roboflow.com/latest/detection/annotators/#supervision.annotators.core.MaskAnnotator)
|
|
||||||
is a drop-in replacement for
|
|
||||||
[`sv.BoxAnnotator`](https://supervision.roboflow.com/latest/detection/annotators/#supervision.annotators.core.BoxAnnotator)
|
|
||||||
that will allow you to draw masks instead of boxes.
|
|
||||||
|
|
||||||
=== "Inference"
|
=== "Inference"
|
||||||
|
|
||||||
|
|
@ -349,7 +348,7 @@ that will allow you to draw masks instead of boxes.
|
||||||
from inference import get_model
|
from inference import get_model
|
||||||
|
|
||||||
model = get_model(model_id="yolov8n-seg-640")
|
model = get_model(model_id="yolov8n-seg-640")
|
||||||
image = cv2.imread("<SOURCE_IMAGE_PATH>")
|
image = cv2.imread("dog.jpeg")
|
||||||
results = model.infer(image)[0]
|
results = model.infer(image)[0]
|
||||||
detections = sv.Detections.from_inference(results)
|
detections = sv.Detections.from_inference(results)
|
||||||
|
|
||||||
|
|
@ -364,6 +363,7 @@ that will allow you to draw masks instead of boxes.
|
||||||
scene=annotated_image,
|
scene=annotated_image,
|
||||||
detections=detections,
|
detections=detections,
|
||||||
)
|
)
|
||||||
|
sv.plot_image(annotated_image)
|
||||||
```
|
```
|
||||||
|
|
||||||
=== "Ultralytics"
|
=== "Ultralytics"
|
||||||
|
|
@ -374,7 +374,7 @@ that will allow you to draw masks instead of boxes.
|
||||||
from ultralytics import YOLO
|
from ultralytics import YOLO
|
||||||
|
|
||||||
model = YOLO("yolov8n-seg.pt")
|
model = YOLO("yolov8n-seg.pt")
|
||||||
image = cv2.imread("<SOURCE_IMAGE_PATH>")
|
image = cv2.imread("dog.jpeg")
|
||||||
results = model(image)[0]
|
results = model(image)[0]
|
||||||
detections = sv.Detections.from_ultralytics(results)
|
detections = sv.Detections.from_ultralytics(results)
|
||||||
|
|
||||||
|
|
@ -402,7 +402,7 @@ that will allow you to draw masks instead of boxes.
|
||||||
processor = DetrImageProcessor.from_pretrained("facebook/detr-resnet-50-panoptic")
|
processor = DetrImageProcessor.from_pretrained("facebook/detr-resnet-50-panoptic")
|
||||||
model = DetrForSegmentation.from_pretrained("facebook/detr-resnet-50-panoptic")
|
model = DetrForSegmentation.from_pretrained("facebook/detr-resnet-50-panoptic")
|
||||||
|
|
||||||
image = Image.open("<SOURCE_IMAGE_PATH>")
|
image = Image.open("dog.jpeg")
|
||||||
inputs = processor(images=image, return_tensors="pt")
|
inputs = processor(images=image, return_tensors="pt")
|
||||||
|
|
||||||
with torch.no_grad():
|
with torch.no_grad():
|
||||||
|
|
|
||||||
|
|
@ -10,11 +10,7 @@ date_modified: 2026-04-22
|
||||||
|
|
||||||
# Detect Small Objects
|
# Detect Small Objects
|
||||||
|
|
||||||
This guide shows how to detect small objects
|
This guide shows how to detect small objects with the [Inference](https://github.com/roboflow/inference), [Ultralytics](https://github.com/ultralytics/ultralytics) or [Transformers](https://github.com/huggingface/transformers) packages using [`InferenceSlicer`](https://supervision.roboflow.com/latest/detection/tools/inference_slicer/#supervision.detection.tools.inference_slicer.InferenceSlicer).
|
||||||
with the [Inference](https://github.com/roboflow/inference),
|
|
||||||
[Ultralytics](https://github.com/ultralytics/ultralytics) or
|
|
||||||
[Transformers](https://github.com/huggingface/transformers) packages using
|
|
||||||
[`InferenceSlicer`](https://supervision.roboflow.com/latest/detection/tools/inference_slicer/#supervision.detection.tools.inference_slicer.InferenceSlicer).
|
|
||||||
|
|
||||||
<video controls>
|
<video controls>
|
||||||
<source src="https://media.roboflow.com/supervision_detect_small_objects_example.mp4" type="video/mp4">
|
<source src="https://media.roboflow.com/supervision_detect_small_objects_example.mp4" type="video/mp4">
|
||||||
|
|
@ -22,8 +18,7 @@ with the [Inference](https://github.com/roboflow/inference),
|
||||||
|
|
||||||
## Baseline Detection
|
## Baseline Detection
|
||||||
|
|
||||||
Small object detection in high-resolution images presents challenges due to the objects'
|
Small object detection in high-resolution images presents challenges due to the objects' size relative to the image resolution.
|
||||||
size relative to the image resolution.
|
|
||||||
|
|
||||||
Running a standard detection model on the full image establishes a baseline for comparison. Load your chosen model, pass the image through it, and convert the results into a `Detections` object. This baseline reveals how many small objects the model misses at native resolution, motivating the sliced inference approach shown later.
|
Running a standard detection model on the full image establishes a baseline for comparison. Load your chosen model, pass the image through it, and convert the results into a `Detections` object. This baseline reveals how many small objects the model misses at native resolution, motivating the sliced inference approach shown later.
|
||||||
|
|
||||||
|
|
@ -116,9 +111,7 @@ Running a standard detection model on the full image establishes a baseline for
|
||||||
|
|
||||||
## Input Resolution
|
## Input Resolution
|
||||||
|
|
||||||
Modifying the input resolution of images before detection can enhance small object
|
Modifying the input resolution of images before detection can enhance small object identification at the cost of processing speed and increased memory usage. This method is less effective for ultra-high-resolution images (4K and above).
|
||||||
identification at the cost of processing speed and increased memory usage. This method
|
|
||||||
is less effective for ultra-high-resolution images (4K and above).
|
|
||||||
|
|
||||||
=== "Inference"
|
=== "Inference"
|
||||||
|
|
||||||
|
|
@ -166,9 +159,7 @@ is less effective for ultra-high-resolution images (4K and above).
|
||||||
|
|
||||||
## Inference Slicer
|
## Inference Slicer
|
||||||
|
|
||||||
[`InferenceSlicer`](https://supervision.roboflow.com/latest/detection/tools/inference_slicer/#supervision.detection.tools.inference_slicer.InferenceSlicer)
|
[`InferenceSlicer`](https://supervision.roboflow.com/latest/detection/tools/inference_slicer/#supervision.detection.tools.inference_slicer.InferenceSlicer) processes high-resolution images by dividing them into smaller segments, detecting objects within each, and aggregating the results.
|
||||||
processes high-resolution images by dividing them into smaller segments, detecting
|
|
||||||
objects within each, and aggregating the results.
|
|
||||||
|
|
||||||
<video controls>
|
<video controls>
|
||||||
<source src="https://media.roboflow.com/supervision_detect_small_objects_example_2.mp4" type="video/mp4">
|
<source src="https://media.roboflow.com/supervision_detect_small_objects_example_2.mp4" type="video/mp4">
|
||||||
|
|
|
||||||
|
|
@ -10,11 +10,7 @@ date_modified: 2026-04-22
|
||||||
|
|
||||||
# Filter Detections
|
# Filter Detections
|
||||||
|
|
||||||
The advanced filtering capabilities of the `Detections` class offer users a versatile and efficient way to narrow down
|
The advanced filtering capabilities of the `Detections` class offer users a versatile and efficient way to narrow down and refine object detections. This section outlines various filtering methods, including filtering by specific class or a set of classes, confidence, object area, bounding box area, relative area, box dimensions, and designated zones. Each method is demonstrated with concise code examples to provide users with a clear understanding of how to implement the filters in their applications.
|
||||||
and refine object detections. This section outlines various filtering methods, including filtering by specific class
|
|
||||||
or a set of classes, confidence, object area, bounding box area, relative area, box dimensions, and designated zones.
|
|
||||||
Each method is demonstrated with concise code examples to provide users with a clear understanding of how to implement
|
|
||||||
the filters in their applications.
|
|
||||||
|
|
||||||
### by specific class
|
### by specific class
|
||||||
|
|
||||||
|
|
@ -124,8 +120,7 @@ Allows you to select detections with specific confidence value, for example high
|
||||||
|
|
||||||
### by area
|
### by area
|
||||||
|
|
||||||
Allows you to select detections based on their size. We define the area as the number of pixels occupied by the
|
Allows you to select detections based on their size. We define the area as the number of pixels occupied by the detection in the image. In the example below, we have sifted out the detections that are too small.
|
||||||
detection in the image. In the example below, we have sifted out the detections that are too small.
|
|
||||||
|
|
||||||
=== "After"
|
=== "After"
|
||||||
|
|
||||||
|
|
@ -159,10 +154,7 @@ detection in the image. In the example below, we have sifted out the detections
|
||||||
|
|
||||||
### by relative area
|
### by relative area
|
||||||
|
|
||||||
Allows you to select detections based on their size in relation to the size of whole image. Sometimes the concept of
|
Allows you to select detections based on their size in relation to the size of whole image. Sometimes the concept of detection size changes depending on the image. Detection occupying 10000 square px can be large on a 1280x720 image but small on a 3840x2160 image. In such cases, we can filter out detections based on the percentage of the image area occupied by them. In the example below, we remove too large detections.
|
||||||
detection size changes depending on the image. Detection occupying 10000 square px can be large on a 1280x720 image
|
|
||||||
but small on a 3840x2160 image. In such cases, we can filter out detections based on the percentage of the image area
|
|
||||||
occupied by them. In the example below, we remove too large detections.
|
|
||||||
|
|
||||||
=== "After"
|
=== "After"
|
||||||
|
|
||||||
|
|
@ -204,9 +196,7 @@ occupied by them. In the example below, we remove too large detections.
|
||||||
|
|
||||||
### by box dimensions
|
### by box dimensions
|
||||||
|
|
||||||
Allows you to select detections based on their dimensions. The size of the bounding box, as well as its coordinates,
|
Allows you to select detections based on their dimensions. The size of the bounding box, as well as its coordinates, can be criteria for rejecting detection. Implementing such filtering requires a bit of custom code but is relatively simple and fast.
|
||||||
can be criteria for rejecting detection. Implementing such filtering requires a bit of custom code but is relatively
|
|
||||||
simple and fast.
|
|
||||||
|
|
||||||
=== "After"
|
=== "After"
|
||||||
|
|
||||||
|
|
@ -244,8 +234,7 @@ simple and fast.
|
||||||
|
|
||||||
### by `PolygonZone`
|
### by `PolygonZone`
|
||||||
|
|
||||||
Allows you to use `Detections` in combination with `PolygonZone` to weed out bounding boxes that are in and out of the
|
Allows you to use `Detections` in combination with `PolygonZone` to weed out bounding boxes that are in and out of the zone. In the example below you can see how to filter out all detections located in the lower part of the image.
|
||||||
zone. In the example below you can see how to filter out all detections located in the lower part of the image.
|
|
||||||
|
|
||||||
=== "After"
|
=== "After"
|
||||||
|
|
||||||
|
|
@ -331,7 +320,7 @@ Use NumPy-style boolean indexing: `detections[detections.class_id == 0]` for cla
|
||||||
|
|
||||||
### How do I filter by bounding box area?
|
### How do I filter by bounding box area?
|
||||||
|
|
||||||
`detections[detections.area > 1000]` filters by pixel area. If masks are present, `detections.area` uses mask area; otherwise it uses bounding box area from `xyxy`. Use `detections.box_area` when you specifically need bounding box area.
|
`detections[detections.area > 1000]` filters by pixel area. If masks are present, `detections.area` uses mask area; otherwise, if oriented-box coordinates are present, it uses oriented polygon area; all remaining detections use bounding box area from `xyxy`. Use `detections.box_area` when you specifically need axis-aligned bounding box area.
|
||||||
|
|
||||||
### Can I filter by box aspect ratio or dimensions?
|
### Can I filter by box aspect ratio or dimensions?
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -0,0 +1,63 @@
|
||||||
|
---
|
||||||
|
comments: true
|
||||||
|
description: Migrate Supervision installations after OpenCV becomes an ambient optional backend: use the included fallback by default or select one compatible OpenCV wheel for your application.
|
||||||
|
date_modified: 2026-07-17
|
||||||
|
---
|
||||||
|
|
||||||
|
# Migrate to Supervision Without an OpenCV Dependency
|
||||||
|
|
||||||
|
Supervision no longer installs OpenCV or offers an OpenCV extra. A standard installation includes the NumPy, Pillow, SciPy, and PyAV fallback needed by Supervision's image, drawing, and file-video APIs. When a compatible `cv2` is already installed, Supervision selects it once when the process imports the package.
|
||||||
|
|
||||||
|
## Keep the default fallback
|
||||||
|
|
||||||
|
Install Supervision normally when your application does not otherwise require OpenCV:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
pip install supervision
|
||||||
|
```
|
||||||
|
|
||||||
|
The fallback keeps Supervision's documented APIs operational. Some text and anti-aliased drawing pixels can differ from OpenCV, so use the same backend while validating image-level baselines.
|
||||||
|
|
||||||
|
## Prefer OpenCV behavior
|
||||||
|
|
||||||
|
Install exactly one OpenCV wheel family when your application relies on OpenCV outside Supervision or needs its native behavior:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# Servers and containers without OpenCV GUI modules
|
||||||
|
pip install opencv-python-headless supervision
|
||||||
|
|
||||||
|
# Desktop applications that need OpenCV GUI modules
|
||||||
|
pip install opencv-python supervision
|
||||||
|
```
|
||||||
|
|
||||||
|
Do not install both `opencv-python` and `opencv-python-headless`. If another dependency, such as a model runtime, already provides a compatible `cv2`, keep that installation instead of adding a second wheel family.
|
||||||
|
|
||||||
|
## Verify the selected backend
|
||||||
|
|
||||||
|
Backend selection happens at import time and lasts for the process lifetime. Run this command in a fresh Python process after changing dependencies:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
python -c "from supervision import _cv2; print(_cv2.BACKEND_NAME)"
|
||||||
|
```
|
||||||
|
|
||||||
|
It prints `fallback` without OpenCV and the OpenCV backend name when `cv2` is available. `_cv2` is private; use this command only as an installation diagnostic, not as application API.
|
||||||
|
|
||||||
|
## Capture webcams yourself
|
||||||
|
|
||||||
|
Supervision's video helpers support file paths through either backend. Live camera capture remains application-owned, so install your chosen OpenCV wheel if you use `cv2.VideoCapture(0)`:
|
||||||
|
|
||||||
|
```python
|
||||||
|
import cv2
|
||||||
|
import supervision as sv
|
||||||
|
|
||||||
|
capture = cv2.VideoCapture(0)
|
||||||
|
annotator = sv.BoxAnnotator()
|
||||||
|
```
|
||||||
|
|
||||||
|
## Roll back an upgrade
|
||||||
|
|
||||||
|
If a downstream image baseline requires the pre-migration package behavior, pin Supervision below the first release that removes the OpenCV dependency, then plan a backend-specific migration separately:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
pip install "supervision<0.30.0"
|
||||||
|
```
|
||||||
|
|
@ -1,34 +1,24 @@
|
||||||
---
|
---
|
||||||
comments: true
|
comments: true
|
||||||
description: Load, split, merge, and convert computer vision datasets between YOLO, COCO, and Pascal VOC formats using supervision's DetectionDataset.
|
description: Load, split, merge, and convert computer vision datasets between YOLO, COCO, Pascal VOC, CreateML, and LabelMe formats using supervision's DetectionDataset.
|
||||||
authors:
|
authors:
|
||||||
- name: Piotr Skalski
|
- name: Piotr Skalski
|
||||||
role: Computer Vision Engineer, Roboflow
|
role: Computer Vision Engineer, Roboflow
|
||||||
github: https://github.com/SkalskiP
|
github: https://github.com/SkalskiP
|
||||||
date_modified: 2026-04-22
|
date_modified: 2026-06-25
|
||||||
---
|
---
|
||||||
|
|
||||||
With Supervision, you can load and manipulate classification, object detection, and
|
With Supervision, you can load and manipulate classification, object detection, and segmentation datasets. This tutorial will walk you through how to load, split, merge, visualize, and augment datasets in Supervision.
|
||||||
segmentation datasets. This tutorial will walk you through how to load, split, merge,
|
|
||||||
visualize, and augment datasets in Supervision.
|
|
||||||
|
|
||||||
## Download Dataset
|
## Download Dataset
|
||||||
|
|
||||||
In this tutorial, we will use a dataset from
|
In this tutorial, we will use a dataset from [Roboflow Universe](https://universe.roboflow.com/), a public repository of thousands of computer vision datasets. If you already have your dataset in [COCO](https://roboflow.com/formats/coco-json), [YOLO](https://roboflow.com/formats/yolov8-pytorch-txt), [Pascal VOC](https://roboflow.com/formats/pascal-voc-xml), [CreateML](https://roboflow.com/formats/createml-json), or [LabelMe](https://roboflow.com/formats/labelme-json) format, you can skip this section.
|
||||||
[Roboflow Universe](https://universe.roboflow.com/), a public repository of
|
|
||||||
thousands of computer vision datasets. If you already have your dataset in
|
|
||||||
[COCO](https://roboflow.com/formats/coco-json),
|
|
||||||
[YOLO](https://roboflow.com/formats/yolov8-pytorch-txt),
|
|
||||||
or [Pascal VOC](https://roboflow.com/formats/pascal-voc-xml) format, you can skip this
|
|
||||||
section.
|
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
pip install roboflow
|
pip install roboflow
|
||||||
```
|
```
|
||||||
|
|
||||||
Next, log into your Roboflow account and download the dataset of your choice in the
|
Next, log into your Roboflow account and download the dataset of your choice. The following snippets show common COCO, YOLO, Pascal VOC, and CreateML exports; LabelMe datasets can also be loaded directly from per-image JSON files in the next section. You can customize the code with your workspace ID, project ID, and version number.
|
||||||
COCO, YOLO, or Pascal VOC format. You can customize the following code snippet with
|
|
||||||
your workspace ID, project ID, and version number.
|
|
||||||
|
|
||||||
=== "COCO"
|
=== "COCO"
|
||||||
|
|
||||||
|
|
@ -66,12 +56,21 @@ your workspace ID, project ID, and version number.
|
||||||
dataset = project.version("<PROJECT_VERSION>").download("voc")
|
dataset = project.version("<PROJECT_VERSION>").download("voc")
|
||||||
```
|
```
|
||||||
|
|
||||||
|
=== "CreateML"
|
||||||
|
|
||||||
|
```python
|
||||||
|
import roboflow
|
||||||
|
|
||||||
|
roboflow.login()
|
||||||
|
|
||||||
|
rf = roboflow.Roboflow()
|
||||||
|
project = rf.workspace("<WORKSPACE_ID>").project("<PROJECT_ID>")
|
||||||
|
dataset = project.version("<PROJECT_VERSION>").download("createml")
|
||||||
|
```
|
||||||
|
|
||||||
## Load Dataset
|
## Load Dataset
|
||||||
|
|
||||||
The Supervision library provides convenient functions to load datasets in various
|
The Supervision library provides convenient functions to load datasets in various formats. If your dataset is already split into train, test, and valid subsets, you can load each of those as separate [`sv.DetectionDataset`](https://supervision.roboflow.com/latest/datasets/core/#supervision.dataset.core.DetectionDataset) instances.
|
||||||
formats. If your dataset is already split into train, test, and valid subsets, you can
|
|
||||||
load each of those as separate [`sv.DetectionDataset`](https://supervision.roboflow.com/latest/datasets/core/#supervision.dataset.core.DetectionDataset)
|
|
||||||
instances.
|
|
||||||
|
|
||||||
=== "COCO"
|
=== "COCO"
|
||||||
|
|
||||||
|
|
@ -157,11 +156,63 @@ instances.
|
||||||
# 800, 100, 100
|
# 800, 100, 100
|
||||||
```
|
```
|
||||||
|
|
||||||
|
=== "CreateML"
|
||||||
|
|
||||||
|
We can do so using the [`sv.DetectionDataset.from_createml`](https://supervision.roboflow.com/latest/datasets/core/#supervision.dataset.core.DetectionDataset.from_createml) to load annotations in [CreateML](https://roboflow.com/formats/createml-json) format.
|
||||||
|
|
||||||
|
```python
|
||||||
|
import supervision as sv
|
||||||
|
|
||||||
|
ds_train = sv.DetectionDataset.from_createml(
|
||||||
|
images_directory_path=f"{dataset.location}/train",
|
||||||
|
annotations_path=f"{dataset.location}/train/_annotations.createml.json",
|
||||||
|
)
|
||||||
|
ds_valid = sv.DetectionDataset.from_createml(
|
||||||
|
images_directory_path=f"{dataset.location}/valid",
|
||||||
|
annotations_path=f"{dataset.location}/valid/_annotations.createml.json",
|
||||||
|
)
|
||||||
|
ds_test = sv.DetectionDataset.from_createml(
|
||||||
|
images_directory_path=f"{dataset.location}/test",
|
||||||
|
annotations_path=f"{dataset.location}/test/_annotations.createml.json",
|
||||||
|
)
|
||||||
|
|
||||||
|
ds_train.classes
|
||||||
|
# ['person', 'bicycle', 'car', ...]
|
||||||
|
|
||||||
|
len(ds_train), len(ds_valid), len(ds_test)
|
||||||
|
# 800, 100, 100
|
||||||
|
```
|
||||||
|
|
||||||
|
=== "LabelMe"
|
||||||
|
|
||||||
|
We can do so using the [`sv.DetectionDataset.from_labelme`](https://supervision.roboflow.com/latest/datasets/core/#supervision.dataset.core.DetectionDataset.from_labelme) to load annotations in [LabelMe](https://roboflow.com/formats/labelme-json) format. LabelMe `rectangle` shapes are loaded as bounding boxes and `polygon` shapes are loaded as masks with bounding boxes.
|
||||||
|
|
||||||
|
```python
|
||||||
|
import supervision as sv
|
||||||
|
|
||||||
|
ds_train = sv.DetectionDataset.from_labelme(
|
||||||
|
images_directory_path="<TRAIN_IMAGES_DIRECTORY_PATH>",
|
||||||
|
annotations_directory_path="<TRAIN_ANNOTATIONS_DIRECTORY_PATH>",
|
||||||
|
)
|
||||||
|
ds_valid = sv.DetectionDataset.from_labelme(
|
||||||
|
images_directory_path="<VALID_IMAGES_DIRECTORY_PATH>",
|
||||||
|
annotations_directory_path="<VALID_ANNOTATIONS_DIRECTORY_PATH>",
|
||||||
|
)
|
||||||
|
ds_test = sv.DetectionDataset.from_labelme(
|
||||||
|
images_directory_path="<TEST_IMAGES_DIRECTORY_PATH>",
|
||||||
|
annotations_directory_path="<TEST_ANNOTATIONS_DIRECTORY_PATH>",
|
||||||
|
)
|
||||||
|
|
||||||
|
ds_train.classes
|
||||||
|
# ['person', 'bicycle', 'car', ...]
|
||||||
|
|
||||||
|
len(ds_train), len(ds_valid), len(ds_test)
|
||||||
|
# 800, 100, 100
|
||||||
|
```
|
||||||
|
|
||||||
## Split Dataset
|
## Split Dataset
|
||||||
|
|
||||||
If your dataset is not already split into train, test, and valid subsets, you can
|
If your dataset is not already split into train, test, and valid subsets, you can easily do so using the [`sv.DetectionDataset.split`](https://supervision.roboflow.com/latest/datasets/core/#supervision.dataset.core.DetectionDataset.split) method. We can split it as follows, ensuring a random shuffle of the data.
|
||||||
easily do so using the [`sv.DetectionDataset.split`](https://supervision.roboflow.com/latest/datasets/core/#supervision.dataset.core.DetectionDataset.split)
|
|
||||||
method. We can split it as follows, ensuring a random shuffle of the data.
|
|
||||||
|
|
||||||
```python
|
```python
|
||||||
import supervision as sv
|
import supervision as sv
|
||||||
|
|
@ -180,9 +231,7 @@ len(ds_train), len(ds_valid), len(ds_test)
|
||||||
|
|
||||||
## Merge Dataset
|
## Merge Dataset
|
||||||
|
|
||||||
If you have multiple datasets that you would like to merge, you can do so using the
|
If you have multiple datasets that you would like to merge, you can do so using the [`sv.DetectionDataset.merge`](https://supervision.roboflow.com/latest/datasets/core/#supervision.dataset.core.DetectionDataset.merge) method.
|
||||||
[`sv.DetectionDataset.merge`](https://supervision.roboflow.com/latest/datasets/core/#supervision.dataset.core.DetectionDataset.merge)
|
|
||||||
method.
|
|
||||||
|
|
||||||
=== "COCO"
|
=== "COCO"
|
||||||
|
|
||||||
|
|
@ -286,12 +335,75 @@ method.
|
||||||
# 1000
|
# 1000
|
||||||
```
|
```
|
||||||
|
|
||||||
|
=== "CreateML"
|
||||||
|
|
||||||
|
```{ .py hl_lines="22-28" }
|
||||||
|
import supervision as sv
|
||||||
|
|
||||||
|
ds_train = sv.DetectionDataset.from_createml(
|
||||||
|
images_directory_path=f'{dataset.location}/train',
|
||||||
|
annotations_path=f'{dataset.location}/train/_annotations.createml.json',
|
||||||
|
)
|
||||||
|
ds_valid = sv.DetectionDataset.from_createml(
|
||||||
|
images_directory_path=f'{dataset.location}/valid',
|
||||||
|
annotations_path=f'{dataset.location}/valid/_annotations.createml.json',
|
||||||
|
)
|
||||||
|
ds_test = sv.DetectionDataset.from_createml(
|
||||||
|
images_directory_path=f'{dataset.location}/test',
|
||||||
|
annotations_path=f'{dataset.location}/test/_annotations.createml.json',
|
||||||
|
)
|
||||||
|
|
||||||
|
ds_train.classes
|
||||||
|
# ['person', 'bicycle', 'car', ...]
|
||||||
|
|
||||||
|
len(ds_train), len(ds_valid), len(ds_test)
|
||||||
|
# 800, 100, 100
|
||||||
|
|
||||||
|
ds = sv.DetectionDataset.merge([ds_train, ds_valid, ds_test])
|
||||||
|
|
||||||
|
ds.classes
|
||||||
|
# ['person', 'bicycle', 'car', ...]
|
||||||
|
|
||||||
|
len(ds)
|
||||||
|
# 1000
|
||||||
|
```
|
||||||
|
|
||||||
|
=== "LabelMe"
|
||||||
|
|
||||||
|
```{ .py hl_lines="22-28" }
|
||||||
|
import supervision as sv
|
||||||
|
|
||||||
|
ds_train = sv.DetectionDataset.from_labelme(
|
||||||
|
images_directory_path="<TRAIN_IMAGES_DIRECTORY_PATH>",
|
||||||
|
annotations_directory_path="<TRAIN_ANNOTATIONS_DIRECTORY_PATH>",
|
||||||
|
)
|
||||||
|
ds_valid = sv.DetectionDataset.from_labelme(
|
||||||
|
images_directory_path="<VALID_IMAGES_DIRECTORY_PATH>",
|
||||||
|
annotations_directory_path="<VALID_ANNOTATIONS_DIRECTORY_PATH>",
|
||||||
|
)
|
||||||
|
ds_test = sv.DetectionDataset.from_labelme(
|
||||||
|
images_directory_path="<TEST_IMAGES_DIRECTORY_PATH>",
|
||||||
|
annotations_directory_path="<TEST_ANNOTATIONS_DIRECTORY_PATH>",
|
||||||
|
)
|
||||||
|
|
||||||
|
ds_train.classes
|
||||||
|
# ['person', 'bicycle', 'car', ...]
|
||||||
|
|
||||||
|
len(ds_train), len(ds_valid), len(ds_test)
|
||||||
|
# 800, 100, 100
|
||||||
|
|
||||||
|
ds = sv.DetectionDataset.merge([ds_train, ds_valid, ds_test])
|
||||||
|
|
||||||
|
ds.classes
|
||||||
|
# ['person', 'bicycle', 'car', ...]
|
||||||
|
|
||||||
|
len(ds)
|
||||||
|
# 1000
|
||||||
|
```
|
||||||
|
|
||||||
## Iterate over Dataset
|
## Iterate over Dataset
|
||||||
|
|
||||||
There are two ways to loop over a `sv.DetectionDataset`: using a direct
|
There are two ways to loop over a `sv.DetectionDataset`: using a direct [for loop](https://supervision.roboflow.com/latest/datasets/core/#supervision.dataset.core.DetectionDataset.__iter__) called on the `sv.DetectionDataset` instance or loading `sv.DetectionDataset` entries [by index](https://supervision.roboflow.com/latest/datasets/core/#supervision.dataset.core.DetectionDataset.__getitem__).
|
||||||
[for loop](https://supervision.roboflow.com/latest/datasets/core/#supervision.dataset.core.DetectionDataset.__iter__)
|
|
||||||
called on the `sv.DetectionDataset` instance or loading `sv.DetectionDataset` entries
|
|
||||||
[by index](https://supervision.roboflow.com/latest/datasets/core/#supervision.dataset.core.DetectionDataset.__getitem__).
|
|
||||||
|
|
||||||
```python
|
```python
|
||||||
import supervision as sv
|
import supervision as sv
|
||||||
|
|
@ -310,13 +422,7 @@ for idx in range(len(ds)):
|
||||||
|
|
||||||
## Visualize Dataset
|
## Visualize Dataset
|
||||||
|
|
||||||
The Supervision library provides tools for easily visualizing your detection dataset.
|
The Supervision library provides tools for easily visualizing your detection dataset. You can create a grid of annotated images to quickly inspect your data and labels. First, initialize the [`sv.BoxAnnotator`](https://supervision.roboflow.com/latest/detection/annotators/#supervision.annotators.core.BoxAnnotator) and [`sv.LabelAnnotator`](https://supervision.roboflow.com/latest/detection/annotators/#supervision.annotators.core.LabelAnnotator). Then, iterate through a subset of the dataset (e.g., the first 25 images), drawing bounding boxes and class labels on each image. Finally, combine the annotated images into a grid for display.
|
||||||
You can create a grid of annotated images to quickly inspect your data and labels.
|
|
||||||
First, initialize the [`sv.BoxAnnotator`](https://supervision.roboflow.com/latest/detection/annotators/#supervision.annotators.core.BoxAnnotator)
|
|
||||||
and [`sv.LabelAnnotator`](https://supervision.roboflow.com/latest/detection/annotators/#supervision.annotators.core.LabelAnnotator).
|
|
||||||
Then, iterate through a subset of the dataset (e.g., the first 25 images), drawing
|
|
||||||
bounding boxes and class labels on each image. Finally, combine the annotated images
|
|
||||||
into a grid for display.
|
|
||||||
|
|
||||||
```python
|
```python
|
||||||
import supervision as sv
|
import supervision as sv
|
||||||
|
|
@ -393,24 +499,45 @@ sv.plot_images_grid(
|
||||||
)
|
)
|
||||||
```
|
```
|
||||||
|
|
||||||
|
=== "CreateML"
|
||||||
|
|
||||||
|
We can do so using the [`sv.DetectionDataset.as_createml`](https://supervision.roboflow.com/latest/datasets/core/#supervision.dataset.core.DetectionDataset.as_createml) method to save annotations in [CreateML](https://roboflow.com/formats/createml-json) format.
|
||||||
|
|
||||||
|
```python
|
||||||
|
import supervision as sv
|
||||||
|
|
||||||
|
ds = sv.DetectionDataset(...)
|
||||||
|
|
||||||
|
ds.as_createml(
|
||||||
|
images_directory_path="<IMAGE_DIRECTORY_PATH>",
|
||||||
|
annotations_path="<ANNOTATIONS_PATH>",
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
=== "LabelMe"
|
||||||
|
|
||||||
|
We can do so using the [`sv.DetectionDataset.as_labelme`](https://supervision.roboflow.com/latest/datasets/core/#supervision.dataset.core.DetectionDataset.as_labelme) method to save annotations in [LabelMe](https://roboflow.com/formats/labelme-json) format. Detections with masks are exported as `polygon` shapes; box-only detections are exported as `rectangle` shapes.
|
||||||
|
|
||||||
|
```python
|
||||||
|
import supervision as sv
|
||||||
|
|
||||||
|
ds = sv.DetectionDataset(...)
|
||||||
|
|
||||||
|
ds.as_labelme(
|
||||||
|
images_directory_path="<IMAGE_DIRECTORY_PATH>",
|
||||||
|
annotations_directory_path="<ANNOTATIONS_DIRECTORY_PATH>",
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
## Augment Dataset
|
## Augment Dataset
|
||||||
|
|
||||||
In this section, we'll explore using Supervision in combination with Albumentations to
|
In this section, we'll explore using Supervision in combination with Albumentations to augment our dataset. Data augmentation is a common technique in computer vision to increase the size and diversity of training datasets, leading to improved model performance and generalization.
|
||||||
augment our dataset. Data augmentation is a common technique in computer vision to
|
|
||||||
increase the size and diversity of training datasets, leading to improved model
|
|
||||||
performance and generalization.
|
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
pip install albumentations
|
pip install albumentations
|
||||||
```
|
```
|
||||||
|
|
||||||
Albumentations provides a flexible and powerful API for image augmentation. The core of
|
Albumentations provides a flexible and powerful API for image augmentation. The core of the library is the [`Compose`](https://albumentations.ai/docs/api-reference/albumentations/core/composition/#Compose) class, which allows you to chain multiple image transformations together. Each transformation is defined using a dedicated class, such as [`HorizontalFlip`](https://albumentations.ai/docs/api-reference/albumentations/augmentations/geometric/flip/#HorizontalFlip), [`RandomBrightnessContrast`](https://albumentations.ai/docs/api-reference/albumentations/augmentations/pixel/transforms/#RandomBrightnessContrast), or [`Perspective`](https://albumentations.ai/docs/api-reference/albumentations/augmentations/geometric/transforms/#Perspective).
|
||||||
the library is the [`Compose`](https://albumentations.ai/docs/api-reference/albumentations/core/composition/#Compose)
|
|
||||||
class, which allows you to chain multiple image transformations together. Each
|
|
||||||
transformation is defined using a dedicated class, such as
|
|
||||||
[`HorizontalFlip`](https://albumentations.ai/docs/api-reference/albumentations/augmentations/geometric/flip/#HorizontalFlip),
|
|
||||||
[`RandomBrightnessContrast`](https://albumentations.ai/docs/api-reference/albumentations/augmentations/pixel/transforms/#RandomBrightnessContrast),
|
|
||||||
or [`Perspective`](https://albumentations.ai/docs/api-reference/albumentations/augmentations/geometric/transforms/#Perspective).
|
|
||||||
|
|
||||||
```python
|
```python
|
||||||
import albumentations as A
|
import albumentations as A
|
||||||
|
|
@ -428,8 +555,7 @@ augmentation = A.Compose(
|
||||||
)
|
)
|
||||||
```
|
```
|
||||||
|
|
||||||
The key is to set `format='pascal_voc'`, which corresponds to the
|
The key is to set `format='pascal_voc'`, which corresponds to the `[x_min, y_min, x_max, y_max]` bounding box format used in Supervision.
|
||||||
`[x_min, y_min, x_max, y_max]` bounding box format used in Supervision.
|
|
||||||
|
|
||||||
```python
|
```python
|
||||||
import numpy as np
|
import numpy as np
|
||||||
|
|
@ -460,7 +586,7 @@ augmented_annotations = replace(
|
||||||
|
|
||||||
### What dataset formats does supervision support?
|
### What dataset formats does supervision support?
|
||||||
|
|
||||||
For detection datasets, supervision supports YOLO, COCO JSON, and Pascal VOC. Use `DetectionDataset.from_yolo()`, `from_coco()`, or `from_pascal_voc()` to load, and `as_yolo()`, `as_coco()`, or `as_pascal_voc()` to save. Classification datasets use `ClassificationDataset.from_folder_structure()` and `as_folder_structure()`.
|
For detection datasets, supervision supports YOLO, COCO JSON, Pascal VOC, CreateML, and LabelMe. Use `DetectionDataset.from_yolo()`, `from_coco()`, `from_pascal_voc()`, `from_createml()`, or `from_labelme()` to load, and `as_yolo()`, `as_coco()`, `as_pascal_voc()`, `as_createml()`, or `as_labelme()` to save. Classification datasets use `ClassificationDataset.from_folder_structure()` and `as_folder_structure()`.
|
||||||
|
|
||||||
### Can I split a dataset into train/val/test sets?
|
### Can I split a dataset into train/val/test sets?
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -10,19 +10,11 @@ date_modified: 2026-04-22
|
||||||
|
|
||||||
# Save Detections
|
# Save Detections
|
||||||
|
|
||||||
Supervision enables an easy way to save detections in .CSV and .JSON files for offline
|
Supervision enables an easy way to save detections in .CSV and .JSON files for offline processing. This guide demonstrates how to perform video inference using the [Inference](https://github.com/roboflow/inference), [Ultralytics](https://github.com/ultralytics/ultralytics) or [Transformers](https://github.com/huggingface/transformers) packages and save their results with [`sv.CSVSink`](https://supervision.roboflow.com/latest/detection/tools/save_detections/#supervision.detection.tools.csv_sink.CSVSink) and [`sv.JSONSink`](https://supervision.roboflow.com/latest/detection/tools/save_detections/#supervision.detection.tools.json_sink.JSONSink).
|
||||||
processing. This guide demonstrates how to perform video inference using the
|
|
||||||
[Inference](https://github.com/roboflow/inference),
|
|
||||||
[Ultralytics](https://github.com/ultralytics/ultralytics) or
|
|
||||||
[Transformers](https://github.com/huggingface/transformers) packages and save their results with
|
|
||||||
[`sv.CSVSink`](https://supervision.roboflow.com/latest/detection/tools/save_detections/#supervision.detection.tools.csv_sink.CSVSink) and
|
|
||||||
[`sv.JSONSink`](https://supervision.roboflow.com/latest/detection/tools/save_detections/#supervision.detection.tools.json_sink.JSONSink).
|
|
||||||
|
|
||||||
## Run Detection
|
## Run Detection
|
||||||
|
|
||||||
First, you'll need to obtain predictions from your object detection or segmentation
|
First, you'll need to obtain predictions from your object detection or segmentation model. You can learn more on this topic in our [How to Detect and Annotate](https://supervision.roboflow.com/latest/how_to/detect_and_annotate/) guide.
|
||||||
model. You can learn more on this topic in our
|
|
||||||
[How to Detect and Annotate](https://supervision.roboflow.com/latest/how_to/detect_and_annotate/) guide.
|
|
||||||
|
|
||||||
To generate predictions for saving, initialize your model and iterate over video frames using `sv.get_video_frames_generator`. Each frame is passed to the model, and the raw output is converted into a `sv.Detections` object. This detection loop forms the foundation for both CSV and JSON export workflows shown below.
|
To generate predictions for saving, initialize your model and iterate over video frames using `sv.get_video_frames_generator`. Each frame is passed to the model, and the raw output is converted into a `sv.Detections` object. This detection loop forms the foundation for both CSV and JSON export workflows shown below.
|
||||||
|
|
||||||
|
|
@ -82,11 +74,7 @@ To generate predictions for saving, initialize your model and iterate over video
|
||||||
|
|
||||||
## Save Detections as CSV
|
## Save Detections as CSV
|
||||||
|
|
||||||
To save detections to a `.CSV` file, open our
|
To save detections to a `.CSV` file, open our [`sv.CSVSink`](https://supervision.roboflow.com/latest/detection/tools/save_detections/#supervision.detection.tools.csv_sink.CSVSink) and then pass the [`sv.Detections`](https://supervision.roboflow.com/latest/detection/core/#supervision.detection.core.Detections) object resulting from the inference to it. Its fields are parsed and saved on disk.
|
||||||
[`sv.CSVSink`](https://supervision.roboflow.com/latest/detection/tools/save_detections/#supervision.detection.tools.csv_sink.CSVSink)
|
|
||||||
and then pass the
|
|
||||||
[`sv.Detections`](https://supervision.roboflow.com/latest/detection/core/#supervision.detection.core.Detections)
|
|
||||||
object resulting from the inference to it. Its fields are parsed and saved on disk.
|
|
||||||
|
|
||||||
=== "Inference"
|
=== "Inference"
|
||||||
|
|
||||||
|
|
@ -158,12 +146,7 @@ object resulting from the inference to it. Its fields are parsed and saved on di
|
||||||
|
|
||||||
## Custom Fields
|
## Custom Fields
|
||||||
|
|
||||||
Besides regular fields in
|
Besides regular fields in [`sv.Detections`](https://supervision.roboflow.com/latest/detection/core/#supervision.detection.core.Detections), [`sv.CSVSink`](https://supervision.roboflow.com/latest/detection/tools/save_detections/#supervision.detection.tools.csv_sink.CSVSink) also allows you to add custom information to each row, which can be passed via the `custom_data` dictionary. Let's utilize this feature to save information about the frame index from which the detections originate.
|
||||||
[`sv.Detections`](https://supervision.roboflow.com/latest/detection/core/#supervision.detection.core.Detections),
|
|
||||||
[`sv.CSVSink`](https://supervision.roboflow.com/latest/detection/tools/save_detections/#supervision.detection.tools.csv_sink.CSVSink)
|
|
||||||
also allows you to add custom information to each row, which can be passed via the
|
|
||||||
`custom_data` dictionary. Let's utilize this feature to save information about the
|
|
||||||
frame index from which the detections originate.
|
|
||||||
|
|
||||||
=== "Inference"
|
=== "Inference"
|
||||||
|
|
||||||
|
|
@ -235,11 +218,7 @@ frame index from which the detections originate.
|
||||||
|
|
||||||
## Save Detections as JSON
|
## Save Detections as JSON
|
||||||
|
|
||||||
If you prefer to save the result in a `.JSON` file instead of a `.CSV` file, all you
|
If you prefer to save the result in a `.JSON` file instead of a `.CSV` file, all you need to do is replace [`sv.CSVSink`](https://supervision.roboflow.com/latest/detection/tools/save_detections/#supervision.detection.tools.csv_sink.CSVSink) with [`sv.JSONSink`](https://supervision.roboflow.com/latest/detection/tools/save_detections/#supervision.detection.tools.json_sink.JSONSink).
|
||||||
need to do is replace
|
|
||||||
[`sv.CSVSink`](https://supervision.roboflow.com/latest/detection/tools/save_detections/#supervision.detection.tools.csv_sink.CSVSink)
|
|
||||||
with
|
|
||||||
[`sv.JSONSink`](https://supervision.roboflow.com/latest/detection/tools/save_detections/#supervision.detection.tools.json_sink.JSONSink).
|
|
||||||
|
|
||||||
=== "Inference"
|
=== "Inference"
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -13,20 +13,11 @@ date_modified: 2026-04-22
|
||||||
|
|
||||||
# Track Objects
|
# Track Objects
|
||||||
|
|
||||||
Leverage Supervision's advanced capabilities for enhancing your video analysis by
|
Leverage Supervision's advanced capabilities for enhancing your video analysis by seamlessly [tracking](https://supervision.roboflow.com/latest/trackers/) objects recognized by a multitude of object detection, segmentation and keypoint models. This comprehensive guide will take you through the steps to perform inference using the YOLOv8 model via either the [Inference](https://github.com/roboflow/inference) or [Ultralytics](https://github.com/ultralytics/ultralytics) packages. Following this, you'll discover how to track these objects efficiently and annotate your video content for a deeper analysis.
|
||||||
seamlessly [tracking](https://supervision.roboflow.com/latest/trackers/) objects recognized by
|
|
||||||
a multitude of object detection, segmentation and keypoint models. This comprehensive guide will
|
|
||||||
take you through the steps to perform inference using the YOLOv8 model via either the
|
|
||||||
[Inference](https://github.com/roboflow/inference) or
|
|
||||||
[Ultralytics](https://github.com/ultralytics/ultralytics) packages. Following this,
|
|
||||||
you'll discover how to track these objects efficiently and annotate your video content
|
|
||||||
for a deeper analysis.
|
|
||||||
|
|
||||||
## Object Detection & Segmentation
|
## Object Detection & Segmentation
|
||||||
|
|
||||||
To make it easier for you to follow our tutorial download the video we will use as an
|
To make it easier for you to follow our tutorial download the video we will use as an example. You can do this using the [`supervision.assets`](https://supervision.roboflow.com/latest/assets/) module included in the base package.
|
||||||
example. You can do this using the
|
|
||||||
[`supervision.assets`](https://supervision.roboflow.com/latest/assets/) module included in the base package.
|
|
||||||
|
|
||||||
This section demonstrates how to detect and segment objects in video frames using YOLOv8 with either the Inference or Ultralytics package. You will download a sample video, define a per-frame callback function that runs model prediction, and process the entire video to produce an annotated output file.
|
This section demonstrates how to detect and segment objects in video frames using YOLOv8 with either the Inference or Ultralytics package. You will download a sample video, define a per-frame callback function that runs model prediction, and process the entire video to produce an annotated output file.
|
||||||
|
|
||||||
|
|
@ -42,16 +33,9 @@ download_assets(VideoAssets.PEOPLE_WALKING)
|
||||||
|
|
||||||
### Run Inference
|
### Run Inference
|
||||||
|
|
||||||
First, you'll need to obtain predictions from your object detection or segmentation
|
First, you'll need to obtain predictions from your object detection or segmentation model. In this tutorial, we are using the YOLOv8 model as an example. However, Supervision is versatile and compatible with various models. Check this [link](https://supervision.roboflow.com/latest/how_to/detect_and_annotate/#load-predictions-into-supervision) for guidance on how to plug in other models.
|
||||||
model. In this tutorial, we are using the YOLOv8 model as an example. However,
|
|
||||||
Supervision is versatile and compatible with various models. Check this
|
|
||||||
[link](https://supervision.roboflow.com/latest/how_to/detect_and_annotate/#load-predictions-into-supervision)
|
|
||||||
for guidance on how to plug in other models.
|
|
||||||
|
|
||||||
We will define a `callback` function, which will process each frame of the video
|
We will define a `callback` function, which will process each frame of the video by obtaining model predictions and then annotating the frame based on these predictions. This `callback` function will be essential in the subsequent steps of the tutorial, as it will be modified to include tracking, labeling, and trace annotations.
|
||||||
by obtaining model predictions and then annotating the frame based on these predictions.
|
|
||||||
This `callback` function will be essential in the subsequent steps of the tutorial, as
|
|
||||||
it will be modified to include tracking, labeling, and trace annotations.
|
|
||||||
|
|
||||||
!!! tip
|
!!! tip
|
||||||
|
|
||||||
|
|
@ -107,11 +91,11 @@ it will be modified to include tracking, labeling, and trace annotations.
|
||||||
|
|
||||||
### Tracking
|
### Tracking
|
||||||
|
|
||||||
After running inference and obtaining predictions, the next step is to track the
|
After running inference and obtaining predictions, the next step is to track the detected objects throughout the video. Utilizing Supervision’s [`sv.ByteTrack`](https://supervision.roboflow.com/latest/trackers/#supervision.tracker.byte_tracker.core.ByteTrack) functionality, each detected object is assigned a unique tracker ID, enabling the continuous following of the object's motion path across different frames.
|
||||||
detected objects throughout the video. Utilizing Supervision’s
|
|
||||||
[`sv.ByteTrack`](https://supervision.roboflow.com/latest/trackers/#supervision.tracker.byte_tracker.core.ByteTrack)
|
!!! warning "Deprecated tracker wrapper"
|
||||||
functionality, each detected object is assigned a unique tracker ID,
|
|
||||||
enabling the continuous following of the object's motion path across different frames.
|
`sv.ByteTrack` is deprecated in favor of `ByteTrackTracker` from the external `trackers` package. The external tracker uses `update()` instead of `update_with_detections()`.
|
||||||
|
|
||||||
=== "Ultralytics"
|
=== "Ultralytics"
|
||||||
|
|
||||||
|
|
@ -163,11 +147,7 @@ enabling the continuous following of the object's motion path across different f
|
||||||
|
|
||||||
### Annotate Video with Tracking IDs
|
### Annotate Video with Tracking IDs
|
||||||
|
|
||||||
Annotating the video with tracking IDs helps in distinguishing and following each object
|
Annotating the video with tracking IDs helps in distinguishing and following each object distinctly. With the [`sv.LabelAnnotator`](https://supervision.roboflow.com/latest/detection/annotators/#supervision.annotators.core.LabelAnnotator) in Supervision, we can overlay the tracker IDs and class labels on the detected objects, offering a clear visual representation of each object's class and unique identifier.
|
||||||
distinctly. With the
|
|
||||||
[`sv.LabelAnnotator`](https://supervision.roboflow.com/latest/detection/annotators/#supervision.annotators.core.LabelAnnotator)
|
|
||||||
in Supervision, we can overlay the tracker IDs and class labels on the detected objects,
|
|
||||||
offering a clear visual representation of each object's class and unique identifier.
|
|
||||||
|
|
||||||
=== "Ultralytics"
|
=== "Ultralytics"
|
||||||
|
|
||||||
|
|
@ -245,11 +225,7 @@ offering a clear visual representation of each object's class and unique identif
|
||||||
|
|
||||||
### Annotate Video with Traces
|
### Annotate Video with Traces
|
||||||
|
|
||||||
Adding traces to the video involves overlaying the historical paths of the detected
|
Adding traces to the video involves overlaying the historical paths of the detected objects. This feature, powered by the [`sv.TraceAnnotator`](https://supervision.roboflow.com/latest/detection/annotators/#supervision.annotators.core.TraceAnnotator), allows for visualizing the trajectories of objects, helping in understanding the movement patterns and interactions between objects in the video.
|
||||||
objects. This feature, powered by the
|
|
||||||
[`sv.TraceAnnotator`](https://supervision.roboflow.com/latest/detection/annotators/#supervision.annotators.core.TraceAnnotator),
|
|
||||||
allows for visualizing the trajectories of objects, helping in understanding the
|
|
||||||
movement patterns and interactions between objects in the video.
|
|
||||||
|
|
||||||
=== "Ultralytics"
|
=== "Ultralytics"
|
||||||
|
|
||||||
|
|
@ -335,8 +311,7 @@ movement patterns and interactions between objects in the video.
|
||||||
|
|
||||||
Models aren't limited to object detection and segmentation. Keypoint detection allows for detailed analysis of body joints and connections, especially valuable for applications like human pose estimation. This section introduces keypoint tracking. We'll walk through the steps of annotating keypoints, converting them into bounding box detections compatible with `ByteTrack`, and applying detection smoothing for enhanced stability.
|
Models aren't limited to object detection and segmentation. Keypoint detection allows for detailed analysis of body joints and connections, especially valuable for applications like human pose estimation. This section introduces keypoint tracking. We'll walk through the steps of annotating keypoints, converting them into bounding box detections compatible with `ByteTrack`, and applying detection smoothing for enhanced stability.
|
||||||
|
|
||||||
To make it easier for you to follow our tutorial, let's download the video we will use as an
|
To make it easier for you to follow our tutorial, let's download the video we will use as an example. You can do this using the [`supervision.assets`](https://supervision.roboflow.com/latest/assets/) module included in the base package.
|
||||||
example. You can do this using the [`supervision.assets`](https://supervision.roboflow.com/latest/assets/) module included in the base package.
|
|
||||||
|
|
||||||
```python
|
```python
|
||||||
from supervision.assets import download_assets, VideoAssets
|
from supervision.assets import download_assets, VideoAssets
|
||||||
|
|
@ -350,8 +325,7 @@ download_assets(VideoAssets.SKIING)
|
||||||
|
|
||||||
### Keypoint Detection
|
### Keypoint Detection
|
||||||
|
|
||||||
First, you'll need to obtain predictions from your keypoint detection model. In this tutorial, we are using the YOLOv8 model as an example. However,
|
First, you'll need to obtain predictions from your keypoint detection model. In this tutorial, we are using the YOLOv8 model as an example. However, Supervision is versatile and compatible with various models. Check this [link](https://supervision.roboflow.com/latest/keypoint/core/) for guidance on how to plug in other models.
|
||||||
Supervision is versatile and compatible with various models. Check this [link](https://supervision.roboflow.com/latest/keypoint/core/) for guidance on how to plug in other models.
|
|
||||||
|
|
||||||
We will define a `callback` function, which will process each frame of the video by obtaining model predictions and then annotating the frame based on these predictions.
|
We will define a `callback` function, which will process each frame of the video by obtaining model predictions and then annotating the frame based on these predictions.
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -0,0 +1,193 @@
|
||||||
|
---
|
||||||
|
comments: true
|
||||||
|
description: Use CompactMask for memory-efficient instance segmentation in supervision — ingest COCO RLE payloads, skip mask materialisation, and merge mixed dense and compact detections without allocating a full pixel stack.
|
||||||
|
authors:
|
||||||
|
- name: Borda
|
||||||
|
role: Open Source Engineer, Roboflow
|
||||||
|
github: https://github.com/borda
|
||||||
|
date_modified: 2026-07-01
|
||||||
|
---
|
||||||
|
|
||||||
|
# Use Compact Masks for Memory-Efficient Segmentation
|
||||||
|
|
||||||
|
[CompactMask][supervision.detection.compact_mask.CompactMask] stores each instance mask as a run-length encoding of its bounding-box **crop** rather than a full `(H, W)` boolean frame. For high-resolution images with many sparse masks this can reduce memory from tens of gigabytes to tens of megabytes, and eliminates full-frame decode work in annotators that only need the cropped region.
|
||||||
|
|
||||||
|
!!! Note
|
||||||
|
|
||||||
|
`sv.mask_to_xyxy` keeps supervision's inclusive max-coordinate convention for compatibility with `CompactMask` and current box-based adapters. Use `sv.mask_to_roi` when you need exclusive slice bounds for NumPy indexing or crop extraction.
|
||||||
|
|
||||||
|
This guide covers the four main integration points:
|
||||||
|
|
||||||
|
1. [Ingesting COCO RLE payloads directly as CompactMask](#ingest-coco-rle-payloads)
|
||||||
|
2. [Parsing Roboflow Inference results without a dense stack](#parse-inference-results)
|
||||||
|
3. [Skipping mask materialisation for box/label annotators](#skip-unnecessary-materialisation)
|
||||||
|
4. [Merging mixed dense and compact detections](#merge-mixed-detections)
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Ingest COCO RLE Payloads
|
||||||
|
|
||||||
|
If your model or API returns masks in the COCO RLE format (`{"size": [H, W], "counts": "..."}`) you can convert them directly to `CompactMask` without allocating an `(N, H, W)` boolean array:
|
||||||
|
|
||||||
|
```python
|
||||||
|
import numpy as np
|
||||||
|
import supervision as sv
|
||||||
|
from supervision.detection.compact_mask import CompactMask
|
||||||
|
|
||||||
|
# Example: two COCO RLE masks for a 720×1280 frame.
|
||||||
|
# Replace the counts strings with actual compressed RLE payloads from your
|
||||||
|
# model or API — e.g., from pycocotools mask.encode() or an Inference response.
|
||||||
|
rles = [
|
||||||
|
{"size": [720, 1280], "counts": "YOUR_RLE_COUNTS_STRING_HERE"},
|
||||||
|
{"size": [720, 1280], "counts": "YOUR_RLE_COUNTS_STRING_HERE"},
|
||||||
|
]
|
||||||
|
xyxy = np.array(
|
||||||
|
[
|
||||||
|
[100.0, 50.0, 400.0, 300.0],
|
||||||
|
[500.0, 200.0, 900.0, 600.0],
|
||||||
|
]
|
||||||
|
)
|
||||||
|
|
||||||
|
compact = CompactMask.from_coco_rle(rles, xyxy, image_shape=(720, 1280))
|
||||||
|
|
||||||
|
detections = sv.Detections(
|
||||||
|
xyxy=xyxy,
|
||||||
|
mask=compact,
|
||||||
|
class_id=np.array([0, 1]),
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
`from_coco_rle` uses run-length arithmetic scoped to each bounding box so no dense pixel array is ever created. Uncompressed integer count lists are also accepted in place of compressed strings.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Parse Inference Results
|
||||||
|
|
||||||
|
`Detections.from_inference` accepts a `compact_masks=True` flag that routes the Roboflow RLE payload through `CompactMask.from_coco_rle` instead of decoding to a dense stack:
|
||||||
|
|
||||||
|
```python
|
||||||
|
import supervision as sv
|
||||||
|
|
||||||
|
# result: a Roboflow Inference v2 response dict with instance masks.
|
||||||
|
detections = sv.Detections.from_inference(result, compact_masks=True)
|
||||||
|
|
||||||
|
from supervision.detection.compact_mask import CompactMask
|
||||||
|
|
||||||
|
assert isinstance(detections.mask, CompactMask)
|
||||||
|
```
|
||||||
|
|
||||||
|
!!! Warning
|
||||||
|
|
||||||
|
`compact_masks=True` crops each mask to its detector bounding box. Pixels outside the box are silently dropped. For masks that extend meaningfully beyond the reported bounding box, use the default `compact_masks=False` (dense decode) to preserve all pixels.
|
||||||
|
|
||||||
|
To convert an existing dense-mask `Detections` to compact at any point:
|
||||||
|
|
||||||
|
```python
|
||||||
|
detections_compact = detections.to_compact_masks()
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Skip Unnecessary Materialisation
|
||||||
|
|
||||||
|
Annotators that do not draw masks (box, label, circle, ellipse, trace, keypoint) expose `requires_mask = False`. Integrations can branch on this flag to avoid decoding compact or RLE masks before annotation:
|
||||||
|
|
||||||
|
```python
|
||||||
|
import supervision as sv
|
||||||
|
|
||||||
|
annotators = [
|
||||||
|
sv.BoxAnnotator(),
|
||||||
|
sv.LabelAnnotator(),
|
||||||
|
sv.MaskAnnotator(), # requires_mask = True
|
||||||
|
]
|
||||||
|
|
||||||
|
for ann in annotators:
|
||||||
|
if ann.requires_mask:
|
||||||
|
# Annotator reads mask pixels — CompactMask decodes lazily per crop.
|
||||||
|
scene = ann.annotate(scene, detections)
|
||||||
|
else:
|
||||||
|
# Annotator ignores masks — strip mask field to eliminate any decode cost.
|
||||||
|
det_no_mask = sv.Detections(
|
||||||
|
xyxy=detections.xyxy,
|
||||||
|
confidence=detections.confidence,
|
||||||
|
class_id=detections.class_id,
|
||||||
|
)
|
||||||
|
scene = ann.annotate(scene, det_no_mask)
|
||||||
|
```
|
||||||
|
|
||||||
|
Annotators that set `requires_mask = True`: [MaskAnnotator][supervision.annotators.core.MaskAnnotator], [PolygonAnnotator][supervision.annotators.core.PolygonAnnotator], [HaloAnnotator][supervision.annotators.core.HaloAnnotator].
|
||||||
|
|
||||||
|
All others default to `requires_mask = False`.
|
||||||
|
|
||||||
|
!!! Note
|
||||||
|
|
||||||
|
`PolygonAnnotator` and `MaskAnnotator` both operate directly on `CompactMask` without materialising the full `(N, H, W)` frame — passing compact detections to them is already efficient.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Merge Mixed Detections
|
||||||
|
|
||||||
|
When merging `Detections` objects that mix dense `ndarray` masks and `CompactMask` instances, `Detections.merge` converts dense inputs to `CompactMask` automatically. No full `(N, H, W)` stack is allocated:
|
||||||
|
|
||||||
|
```python
|
||||||
|
import numpy as np
|
||||||
|
import supervision as sv
|
||||||
|
from supervision.detection.compact_mask import CompactMask
|
||||||
|
|
||||||
|
H, W = 720, 1280
|
||||||
|
|
||||||
|
# Compact detections from an RLE-based source.
|
||||||
|
# Replace the counts string with a real compressed RLE payload from your model or API.
|
||||||
|
rles = [{"size": [H, W], "counts": "YOUR_RLE_COUNTS_STRING_HERE"}]
|
||||||
|
xyxy_a = np.array([[100.0, 50.0, 400.0, 300.0]])
|
||||||
|
cm = CompactMask.from_coco_rle(rles, xyxy_a, image_shape=(H, W))
|
||||||
|
det_a = sv.Detections(xyxy=xyxy_a, mask=cm, class_id=np.array([0]))
|
||||||
|
|
||||||
|
# Dense detections from a different source.
|
||||||
|
masks_b = np.zeros((1, H, W), dtype=bool)
|
||||||
|
masks_b[0, 200:400, 500:800] = True
|
||||||
|
xyxy_b = np.array([[500.0, 200.0, 799.0, 399.0]])
|
||||||
|
det_b = sv.Detections(xyxy=xyxy_b, mask=masks_b, class_id=np.array([1]))
|
||||||
|
|
||||||
|
# Output is CompactMask regardless of input order.
|
||||||
|
merged = sv.Detections.merge([det_a, det_b])
|
||||||
|
assert isinstance(merged.mask, CompactMask)
|
||||||
|
assert len(merged) == 2
|
||||||
|
```
|
||||||
|
|
||||||
|
Merge rules:
|
||||||
|
|
||||||
|
| Inputs | Output mask type |
|
||||||
|
| ------------------------------------- | ------------------------------- |
|
||||||
|
| All `CompactMask` | `CompactMask` |
|
||||||
|
| Mixed `CompactMask` + dense `ndarray` | `CompactMask` |
|
||||||
|
| All dense `ndarray` | `ndarray` (backward compatible) |
|
||||||
|
|
||||||
|
All `CompactMask` inputs must share the same `image_shape`; mismatches raise `ValueError`.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Performance Notes
|
||||||
|
|
||||||
|
These estimates apply to the **parsing and annotation stage**, not end-to-end pipeline FPS. Model inference typically dominates total runtime.
|
||||||
|
|
||||||
|
| Optimisation | Realistic gain | Applies when |
|
||||||
|
| ---------------------------- | -------------------------- | ------------------------------------------------------------- |
|
||||||
|
| `from_coco_rle` ingestion | 25–60% faster parse | Full-frame COCO RLE payload; current dense decode path |
|
||||||
|
| `MaskAnnotator` ROI blending | 10–35% faster annotation | Many small, sparse masks on high-res frames |
|
||||||
|
| `PolygonAnnotator` crop path | 15–45% faster polygon draw | Many compact masks; full-frame materialise was the bottleneck |
|
||||||
|
| Mixed-mask merge | 5–20% faster merge | Mix of compact and dense sources (e.g. multi-camera stitch) |
|
||||||
|
|
||||||
|
Upper-end gains assume: ≥1080p frames, tens to hundreds of instances, masks covering less than ~20% of total pixels.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## API Reference
|
||||||
|
|
||||||
|
- [CompactMask][supervision.detection.compact_mask.CompactMask]
|
||||||
|
- [CompactMask.from_coco_rle][supervision.detection.compact_mask.CompactMask.from_coco_rle]
|
||||||
|
- [CompactMask.from_dense][supervision.detection.compact_mask.CompactMask.from_dense]
|
||||||
|
- [Detections.from_inference][supervision.detection.core.Detections.from_inference]
|
||||||
|
- [Detections.to_compact_masks][supervision.detection.core.Detections.to_compact_masks]
|
||||||
|
- [Detections.merge][supervision.detection.core.Detections.merge]
|
||||||
|
- [BaseAnnotator.requires_mask][supervision.annotators.base.BaseAnnotator]
|
||||||
|
|
@ -45,17 +45,13 @@ We write your reusable computer vision tools. Whether you need to load your data
|
||||||
|
|
||||||
## 💻 Install
|
## 💻 Install
|
||||||
|
|
||||||
You can install `supervision` in a
|
You can install `supervision` in a [**Python>=3.10**](https://www.python.org/) environment.
|
||||||
[**Python>=3.9**](https://www.python.org/) environment.
|
|
||||||
|
|
||||||
!!! example "Installation"
|
!!! example "Installation"
|
||||||
|
|
||||||
=== "pip (recommended)"
|
=== "pip (recommended)"
|
||||||
|
|
||||||
[](https://badge.fury.io/py/supervision)
|
[](https://badge.fury.io/py/supervision) [](https://pypistats.org/packages/supervision) [](../LICENSE.md) [](https://badge.fury.io/py/supervision)
|
||||||
[](https://pypistats.org/packages/supervision)
|
|
||||||
[](../LICENSE.md)
|
|
||||||
[](https://badge.fury.io/py/supervision)
|
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
pip install supervision
|
pip install supervision
|
||||||
|
|
@ -63,10 +59,7 @@ You can install `supervision` in a
|
||||||
|
|
||||||
=== "poetry"
|
=== "poetry"
|
||||||
|
|
||||||
[](https://badge.fury.io/py/supervision)
|
[](https://badge.fury.io/py/supervision) [](https://pypistats.org/packages/supervision) [](../LICENSE.md) [](https://badge.fury.io/py/supervision)
|
||||||
[](https://pypistats.org/packages/supervision)
|
|
||||||
[](../LICENSE.md)
|
|
||||||
[](https://badge.fury.io/py/supervision)
|
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
poetry add supervision
|
poetry add supervision
|
||||||
|
|
@ -74,10 +67,7 @@ You can install `supervision` in a
|
||||||
|
|
||||||
=== "uv"
|
=== "uv"
|
||||||
|
|
||||||
[](https://badge.fury.io/py/supervision)
|
[](https://badge.fury.io/py/supervision) [](https://pypistats.org/packages/supervision) [](../LICENSE.md) [](https://badge.fury.io/py/supervision)
|
||||||
[](https://pypistats.org/packages/supervision)
|
|
||||||
[](../LICENSE.md)
|
|
||||||
[](https://badge.fury.io/py/supervision)
|
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
uv pip install supervision
|
uv pip install supervision
|
||||||
|
|
@ -91,10 +81,7 @@ You can install `supervision` in a
|
||||||
|
|
||||||
=== "rye"
|
=== "rye"
|
||||||
|
|
||||||
[](https://badge.fury.io/py/supervision)
|
[](https://badge.fury.io/py/supervision) [](https://pypistats.org/packages/supervision) [](../LICENSE.md) [](https://badge.fury.io/py/supervision)
|
||||||
[](https://pypistats.org/packages/supervision)
|
|
||||||
[](../LICENSE.md)
|
|
||||||
[](https://badge.fury.io/py/supervision)
|
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
rye add supervision
|
rye add supervision
|
||||||
|
|
|
||||||
|
|
@ -77,6 +77,63 @@ comments: true
|
||||||
|
|
||||||
</div>
|
</div>
|
||||||
|
|
||||||
|
=== "VertexEllipseAreaAnnotator"
|
||||||
|
|
||||||
|
```python
|
||||||
|
import supervision as sv
|
||||||
|
|
||||||
|
image = ...
|
||||||
|
key_points = sv.KeyPoints(...)
|
||||||
|
|
||||||
|
area_annotator = sv.VertexEllipseAreaAnnotator(
|
||||||
|
color=sv.Color.GREEN,
|
||||||
|
sigma=2.0,
|
||||||
|
)
|
||||||
|
annotated_frame = area_annotator.annotate(
|
||||||
|
scene=image.copy(),
|
||||||
|
key_points=key_points,
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
`sv.VertexEllipseAnnotator` is a compatibility alias for `sv.VertexEllipseAreaAnnotator`.
|
||||||
|
|
||||||
|
=== "VertexEllipseOutlineAnnotator"
|
||||||
|
|
||||||
|
```python
|
||||||
|
import supervision as sv
|
||||||
|
|
||||||
|
image = ...
|
||||||
|
key_points = sv.KeyPoints(...)
|
||||||
|
|
||||||
|
outline_annotator = sv.VertexEllipseOutlineAnnotator(
|
||||||
|
color=sv.Color.GREEN,
|
||||||
|
sigma=2.0,
|
||||||
|
thickness=2,
|
||||||
|
)
|
||||||
|
annotated_frame = outline_annotator.annotate(
|
||||||
|
scene=image.copy(),
|
||||||
|
key_points=key_points,
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
=== "VertexEllipseHaloAnnotator"
|
||||||
|
|
||||||
|
```python
|
||||||
|
import supervision as sv
|
||||||
|
|
||||||
|
image = ...
|
||||||
|
key_points = sv.KeyPoints(...)
|
||||||
|
|
||||||
|
halo_annotator = sv.VertexEllipseHaloAnnotator(
|
||||||
|
color=sv.Color.GREEN,
|
||||||
|
sigma=2.0,
|
||||||
|
)
|
||||||
|
annotated_frame = halo_annotator.annotate(
|
||||||
|
scene=image.copy(),
|
||||||
|
key_points=key_points,
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
<div class="md-typeset">
|
<div class="md-typeset">
|
||||||
<h2><a href="#supervision.key_points.annotators.VertexAnnotator">VertexAnnotator</a></h2>
|
<h2><a href="#supervision.key_points.annotators.VertexAnnotator">VertexAnnotator</a></h2>
|
||||||
</div>
|
</div>
|
||||||
|
|
@ -94,3 +151,21 @@ comments: true
|
||||||
</div>
|
</div>
|
||||||
|
|
||||||
:::supervision.key_points.annotators.VertexLabelAnnotator
|
:::supervision.key_points.annotators.VertexLabelAnnotator
|
||||||
|
|
||||||
|
<div class="md-typeset">
|
||||||
|
<h2><a href="#supervision.key_points.annotators.VertexEllipseAreaAnnotator">VertexEllipseAreaAnnotator</a></h2>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
:::supervision.key_points.annotators.VertexEllipseAreaAnnotator
|
||||||
|
|
||||||
|
<div class="md-typeset">
|
||||||
|
<h2><a href="#supervision.key_points.annotators.VertexEllipseOutlineAnnotator">VertexEllipseOutlineAnnotator</a></h2>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
:::supervision.key_points.annotators.VertexEllipseOutlineAnnotator
|
||||||
|
|
||||||
|
<div class="md-typeset">
|
||||||
|
<h2><a href="#supervision.key_points.annotators.VertexEllipseHaloAnnotator">VertexEllipseHaloAnnotator</a></h2>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
:::supervision.key_points.annotators.VertexEllipseHaloAnnotator
|
||||||
|
|
|
||||||
|
|
@ -76,7 +76,7 @@ The built-in `sv.ByteTrack` wrapper assigns persistent IDs across video frames t
|
||||||
|
|
||||||
### Datasets
|
### Datasets
|
||||||
|
|
||||||
`sv.DetectionDataset` loads, merges, splits, and converts object detection datasets. Supported formats include YOLO, COCO JSON, and Pascal VOC. `sv.ClassificationDataset` supports folder-structured classification datasets.
|
`sv.DetectionDataset` loads, merges, splits, and converts object detection datasets. Supported formats include YOLO, COCO JSON, Pascal VOC, and LabelMe. `sv.ClassificationDataset` supports folder-structured classification datasets.
|
||||||
|
|
||||||
### Metrics
|
### Metrics
|
||||||
|
|
||||||
|
|
@ -162,7 +162,7 @@ No. Supervision is model agnostic. It is designed to normalize model outputs int
|
||||||
|
|
||||||
### What dataset formats are supported?
|
### What dataset formats are supported?
|
||||||
|
|
||||||
For object detection datasets, Supervision supports YOLO, COCO JSON, and Pascal VOC import and export. For classification datasets, it supports folder-structure import and export.
|
For object detection datasets, Supervision supports YOLO, COCO JSON, Pascal VOC, and LabelMe import and export. For classification datasets, it supports folder-structure import and export.
|
||||||
|
|
||||||
### How do I detect small objects?
|
### How do I detect small objects?
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -53,7 +53,7 @@ Zone-based counting. `PolygonZone.trigger(detections)` returns a boolean mask fo
|
||||||
|
|
||||||
### sv.DetectionDataset and sv.ClassificationDataset
|
### sv.DetectionDataset and sv.ClassificationDataset
|
||||||
|
|
||||||
For detection datasets, load, merge, split, and convert between YOLO, COCO JSON, and Pascal VOC formats. Classification datasets use folder-structure import and export via `ClassificationDataset.from_folder_structure()` and `as_folder_structure()`.
|
For detection datasets, load, merge, split, and convert between YOLO, COCO JSON, Pascal VOC, and LabelMe formats. Classification datasets use folder-structure import and export via `ClassificationDataset.from_folder_structure()` and `as_folder_structure()`.
|
||||||
|
|
||||||
### sv.InferenceSlicer
|
### sv.InferenceSlicer
|
||||||
|
|
||||||
|
|
@ -129,11 +129,11 @@ Supervision is an open-source Python library by Roboflow for computer vision wor
|
||||||
|
|
||||||
### How do I install supervision?
|
### How do I install supervision?
|
||||||
|
|
||||||
Install with `pip install supervision`. For optional metric dependencies use `pip install supervision[metrics]`. Sample asset utilities are included in the base package under `supervision.assets`. The current package metadata requires Python 3.9+.
|
Install with `pip install supervision`. For optional metric dependencies use `pip install supervision[metrics]`. Sample asset utilities are included in the base package under `supervision.assets`. The current package metadata requires Python 3.10+.
|
||||||
|
|
||||||
### What can I do with supervision?
|
### What can I do with supervision?
|
||||||
|
|
||||||
Annotate images and video with bounding boxes, masks, and labels; track objects across frames with persistent IDs; count detections inside polygon zones or line crossings; filter and query detection results; load, split, and convert detection datasets between YOLO, COCO, and Pascal VOC formats; manage classification datasets with folder structures; and benchmark model performance with mAP and confusion matrices.
|
Annotate images and video with bounding boxes, masks, and labels; track objects across frames with persistent IDs; count detections inside polygon zones or line crossings; filter and query detection results; load, split, and convert detection datasets between YOLO, COCO, Pascal VOC, and LabelMe formats; manage classification datasets with folder structures; and benchmark model performance with mAP and confusion matrices.
|
||||||
|
|
||||||
### Is supervision free to use?
|
### Is supervision free to use?
|
||||||
|
|
||||||
|
|
@ -153,7 +153,7 @@ Use a tracker to assign persistent IDs. The built-in `sv.ByteTrack` wrapper acce
|
||||||
|
|
||||||
### What dataset formats does supervision support?
|
### What dataset formats does supervision support?
|
||||||
|
|
||||||
For detection datasets, supervision supports YOLO, COCO JSON, and Pascal VOC. Use `DetectionDataset.from_yolo()`, `from_coco()`, or `from_pascal_voc()` to load, and `as_yolo()`, `as_coco()`, or `as_pascal_voc()` to save. For classification datasets, use `ClassificationDataset.from_folder_structure()` and `as_folder_structure()`.
|
For detection datasets, supervision supports YOLO, COCO JSON, Pascal VOC, and LabelMe. Use `DetectionDataset.from_yolo()`, `from_coco()`, `from_pascal_voc()`, or `from_labelme()` to load, and `as_yolo()`, `as_coco()`, `as_pascal_voc()`, or `as_labelme()` to save. For classification datasets, use `ClassificationDataset.from_folder_structure()` and `as_folder_structure()`.
|
||||||
|
|
||||||
### How do I count objects in a zone?
|
### How do I count objects in a zone?
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -47,7 +47,7 @@ Object tracker wrapper that assigns persistent IDs across video frames. The buil
|
||||||
Zone-based counting. `PolygonZone.trigger(detections)` returns a boolean mask for detections currently inside an arbitrary polygon. `LineZone.trigger(detections)` returns `(crossed_in, crossed_out)` arrays for line crossings and requires `detections.tracker_id` so objects can be matched across frames. Both are commonly paired with zone annotators for visualization.
|
Zone-based counting. `PolygonZone.trigger(detections)` returns a boolean mask for detections currently inside an arbitrary polygon. `LineZone.trigger(detections)` returns `(crossed_in, crossed_out)` arrays for line crossings and requires `detections.tracker_id` so objects can be matched across frames. Both are commonly paired with zone annotators for visualization.
|
||||||
|
|
||||||
### sv.DetectionDataset and sv.ClassificationDataset
|
### sv.DetectionDataset and sv.ClassificationDataset
|
||||||
For detection datasets, load, merge, split, and convert between YOLO, COCO JSON, and Pascal VOC formats. Classification datasets use folder-structure import and export via `ClassificationDataset.from_folder_structure()` and `as_folder_structure()`.
|
For detection datasets, load, merge, split, and convert between YOLO, COCO JSON, Pascal VOC, CreateML, and LabelMe formats. Classification datasets use folder-structure import and export via `ClassificationDataset.from_folder_structure()` and `as_folder_structure()`.
|
||||||
|
|
||||||
### sv.InferenceSlicer
|
### sv.InferenceSlicer
|
||||||
SAHI-style inference slicing: split high-resolution images into overlapping tiles, run detection on each tile, merge results with non-maximum suppression or non-maximum merge. Configure tile overlap in pixels with `overlap_wh`.
|
SAHI-style inference slicing: split high-resolution images into overlapping tiles, run detection on each tile, merge results with non-maximum suppression or non-maximum merge. Configure tile overlap in pixels with `overlap_wh`.
|
||||||
|
|
@ -120,11 +120,11 @@ Supervision is an open-source Python library by Roboflow for computer vision wor
|
||||||
|
|
||||||
### How do I install supervision?
|
### How do I install supervision?
|
||||||
|
|
||||||
Install with `pip install supervision`. For optional metric dependencies use `pip install supervision[metrics]`. Sample asset utilities are included in the base package under `supervision.assets`. The current package metadata requires Python 3.9+.
|
Install with `pip install supervision`. For optional metric dependencies use `pip install supervision[metrics]`. Sample asset utilities are included in the base package under `supervision.assets`. The current package metadata requires Python 3.10+.
|
||||||
|
|
||||||
### What can I do with supervision?
|
### What can I do with supervision?
|
||||||
|
|
||||||
Annotate images and video with bounding boxes, masks, and labels; track objects across frames with persistent IDs; count detections inside polygon zones or line crossings; filter and query detection results; load, split, and convert detection datasets between YOLO, COCO, and Pascal VOC formats; manage classification datasets with folder structures; and benchmark model performance with mAP and confusion matrices.
|
Annotate images and video with bounding boxes, masks, and labels; track objects across frames with persistent IDs; count detections inside polygon zones or line crossings; filter and query detection results; load, split, and convert detection datasets between YOLO, COCO, Pascal VOC, and LabelMe formats; manage classification datasets with folder structures; and benchmark model performance with mAP and confusion matrices.
|
||||||
|
|
||||||
### Is supervision free to use?
|
### Is supervision free to use?
|
||||||
|
|
||||||
|
|
@ -144,7 +144,7 @@ Use a tracker to assign persistent IDs. The built-in `sv.ByteTrack` wrapper acce
|
||||||
|
|
||||||
### What dataset formats does supervision support?
|
### What dataset formats does supervision support?
|
||||||
|
|
||||||
For detection datasets, supervision supports YOLO, COCO JSON, and Pascal VOC. Use `DetectionDataset.from_yolo()`, `from_coco()`, or `from_pascal_voc()` to load, and `as_yolo()`, `as_coco()`, or `as_pascal_voc()` to save. For classification datasets, use `ClassificationDataset.from_folder_structure()` and `as_folder_structure()`.
|
For detection datasets, supervision supports YOLO, COCO JSON, Pascal VOC, CreateML, and LabelMe. Use `DetectionDataset.from_yolo()`, `from_coco()`, `from_pascal_voc()`, `from_createml()`, or `from_labelme()` to load, and `as_yolo()`, `as_coco()`, `as_pascal_voc()`, `as_createml()`, or `as_labelme()` to save. For classification datasets, use `ClassificationDataset.from_folder_structure()` and `as_folder_structure()`.
|
||||||
|
|
||||||
### How do I count objects in a zone?
|
### How do I count objects in a zone?
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -6,6 +6,12 @@ comments: true
|
||||||
|
|
||||||
This page contains supplementary values, types and enums that metrics use.
|
This page contains supplementary values, types and enums that metrics use.
|
||||||
|
|
||||||
|
Install the metrics extra before using metrics APIs:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
pip install "supervision[metrics]"
|
||||||
|
```
|
||||||
|
|
||||||
<div class="md-typeset">
|
<div class="md-typeset">
|
||||||
<h2><a href="#supervision.metrics.core.MetricTarget">MetricTarget</a></h2>
|
<h2><a href="#supervision.metrics.core.MetricTarget">MetricTarget</a></h2>
|
||||||
</div>
|
</div>
|
||||||
|
|
|
||||||
|
|
@ -4,6 +4,12 @@ comments: true
|
||||||
|
|
||||||
# F1 Score
|
# F1 Score
|
||||||
|
|
||||||
|
Install the metrics extra before using this API:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
pip install "supervision[metrics]"
|
||||||
|
```
|
||||||
|
|
||||||
<div class="md-typeset">
|
<div class="md-typeset">
|
||||||
<h2><a href="#supervision.metrics.f1_score.F1Score">F1Score</a></h2>
|
<h2><a href="#supervision.metrics.f1_score.F1Score">F1Score</a></h2>
|
||||||
</div>
|
</div>
|
||||||
|
|
|
||||||
|
|
@ -1,10 +1,16 @@
|
||||||
---
|
---
|
||||||
comments: true
|
comments: true
|
||||||
description: API reference for MeanAveragePrecision — compute mAP for object detection benchmarking with bounding boxes.
|
description: API reference for MeanAveragePrecision — compute mAP for object detection benchmarking with boxes, masks, and oriented boxes.
|
||||||
---
|
---
|
||||||
|
|
||||||
# Mean Average Precision
|
# Mean Average Precision
|
||||||
|
|
||||||
|
Install the metrics extra before using this API:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
pip install "supervision[metrics]"
|
||||||
|
```
|
||||||
|
|
||||||
<div class="md-typeset">
|
<div class="md-typeset">
|
||||||
<h2><a href="#supervision.metrics.mean_average_precision.MeanAveragePrecision">MeanAveragePrecision</a></h2>
|
<h2><a href="#supervision.metrics.mean_average_precision.MeanAveragePrecision">MeanAveragePrecision</a></h2>
|
||||||
</div>
|
</div>
|
||||||
|
|
|
||||||
|
|
@ -4,6 +4,12 @@ comments: true
|
||||||
|
|
||||||
# Mean Average Recall
|
# Mean Average Recall
|
||||||
|
|
||||||
|
Install the metrics extra before using this API:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
pip install "supervision[metrics]"
|
||||||
|
```
|
||||||
|
|
||||||
<div class="md-typeset">
|
<div class="md-typeset">
|
||||||
<h2><a href="#supervision.metrics.mean_average_recall.MeanAverageRecall">MeanAverageRecall</a></h2>
|
<h2><a href="#supervision.metrics.mean_average_recall.MeanAverageRecall">MeanAverageRecall</a></h2>
|
||||||
</div>
|
</div>
|
||||||
|
|
|
||||||
|
|
@ -4,6 +4,12 @@ comments: true
|
||||||
|
|
||||||
# Precision
|
# Precision
|
||||||
|
|
||||||
|
Install the metrics extra before using this API:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
pip install "supervision[metrics]"
|
||||||
|
```
|
||||||
|
|
||||||
<div class="md-typeset">
|
<div class="md-typeset">
|
||||||
<h2><a href="#supervision.metrics.precision.Precision">Precision</a></h2>
|
<h2><a href="#supervision.metrics.precision.Precision">Precision</a></h2>
|
||||||
</div>
|
</div>
|
||||||
|
|
|
||||||
|
|
@ -4,6 +4,12 @@ comments: true
|
||||||
|
|
||||||
# Recall
|
# Recall
|
||||||
|
|
||||||
|
Install the metrics extra before using this API:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
pip install "supervision[metrics]"
|
||||||
|
```
|
||||||
|
|
||||||
<div class="md-typeset">
|
<div class="md-typeset">
|
||||||
<h2><a href="#supervision.metrics.recall.Recall">Recall</a></h2>
|
<h2><a href="#supervision.metrics.recall.Recall">Recall</a></h2>
|
||||||
</div>
|
</div>
|
||||||
|
|
|
||||||
|
|
@ -0,0 +1,176 @@
|
||||||
|
{
|
||||||
|
"cells": [
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {
|
||||||
|
"id": "jxcxFKy2hRnA"
|
||||||
|
},
|
||||||
|
"source": [
|
||||||
|
"# Blurring Faces\n",
|
||||||
|
"\n",
|
||||||
|
"---\n",
|
||||||
|
"\n",
|
||||||
|
"[](https://colab.research.google.com/github/roboflow/supervision/blob/develop/docs/notebooks/blurring_faces.ipynb)"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {
|
||||||
|
"id": "FAxD-vAgkadG"
|
||||||
|
},
|
||||||
|
"source": [
|
||||||
|
"Click the `Open in Colab` button to run the cookbook on Google Colab.\n",
|
||||||
|
"\n",
|
||||||
|
"## Introduction\n",
|
||||||
|
"\n",
|
||||||
|
"In this cookbook we'll use a frame from a video of someone in a supermarket. We'll download this video via the `supervision` assets module. We'll then run inference on this frame using the hosted Roboflow API to fetch detections of faces utilizing an open source face detection model on Roboflow Universe. Finally, we'll use supervision to blur the detected faces."
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {
|
||||||
|
"id": "o_F-fJGskuz9"
|
||||||
|
},
|
||||||
|
"source": [
|
||||||
|
"## Install packages\n",
|
||||||
|
"\n",
|
||||||
|
"Let's quickly install the `supervision` package with the assets module, as well as the roboflow `inference_sdk` with pip. We'll also install `tqdm` to show a progress bar, but this is optional in production code."
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {
|
||||||
|
"id": "c2X8k1opItiO"
|
||||||
|
},
|
||||||
|
"outputs": [],
|
||||||
|
"source": "!pip3 install -q supervision inference tqdm \"Pillow<12\""
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {
|
||||||
|
"id": "9xy1jXMrm5iU"
|
||||||
|
},
|
||||||
|
"source": [
|
||||||
|
"## Download Video and Extract Frame\n",
|
||||||
|
"\n",
|
||||||
|
"In order to blur a face in a frame, we'll need a frame with a face in it. Let's download a video, and grab a frame in the middle of the video. I played around a little, and found that the 800th frame is great frame for us to test, since the customer is facing the camera. In this code, we're also using tqdm to display a progress bar of our script."
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"metadata": {},
|
||||||
|
"cell_type": "code",
|
||||||
|
"outputs": [],
|
||||||
|
"execution_count": null,
|
||||||
|
"source": [
|
||||||
|
"import supervision as sv\n",
|
||||||
|
"\n",
|
||||||
|
"video = sv.download_assets(sv.VideoAssets.GROCERY_STORE)\n",
|
||||||
|
"\n",
|
||||||
|
"# Seek directly to frame 800 using the start parameter (O(1) seek)\n",
|
||||||
|
"frame = next(sv.get_video_frames_generator(video, start=800))\n",
|
||||||
|
"\n",
|
||||||
|
"sv.plot_image(frame)"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {
|
||||||
|
"id": "tgejF7Zlosyd"
|
||||||
|
},
|
||||||
|
"source": [
|
||||||
|
"## Detecting Faces\n",
|
||||||
|
"\n",
|
||||||
|
"Now that we've got our image we'll need a good face detecting model. For this task, there are already an impressive amount of open source models available on [Roboflow Universe](https://universe.roboflow.com/). After a little digging, this [face detection model](https://universe.roboflow.com/mohamed-traore-2ekkp/face-detection-mik1i) has over 1300 images. Some models, including this one, require a Roboflow API key. You can [create a free account here](https://app.roboflow.com/login). From there, you can find the key under Settings > Workspaces > Roboflow API. Let's give it a try."
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {
|
||||||
|
"colab": {
|
||||||
|
"base_uri": "https://localhost:8080/"
|
||||||
|
},
|
||||||
|
"id": "3MDjVxEzKWGn",
|
||||||
|
"outputId": "8fee3446-cc81-4e3c-c4a0-b531d286f1a4"
|
||||||
|
},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"import os\n",
|
||||||
|
"from inference_sdk import InferenceHTTPClient\n",
|
||||||
|
"\n",
|
||||||
|
"try:\n",
|
||||||
|
" from google.colab import userdata\n",
|
||||||
|
" ROBOFLOW_API_KEY = userdata.get(\"ROBOFLOW_API_KEY\") or \"\"\n",
|
||||||
|
"except ImportError:\n",
|
||||||
|
" ROBOFLOW_API_KEY = os.environ.get(\"ROBOFLOW_API_KEY\", \"\")\n",
|
||||||
|
"\n",
|
||||||
|
"assert ROBOFLOW_API_KEY, \"Set ROBOFLOW_API_KEY in Colab secrets or as env var\"\n",
|
||||||
|
"\n",
|
||||||
|
"client = InferenceHTTPClient(\n",
|
||||||
|
" api_url=\"https://detect.roboflow.com\",\n",
|
||||||
|
" api_key=ROBOFLOW_API_KEY\n",
|
||||||
|
")\n",
|
||||||
|
"\n",
|
||||||
|
"results = client.infer(frame, model_id=\"face-detection-mik1i/18\")\n",
|
||||||
|
"\n",
|
||||||
|
"print(f\"Detected {len(results['predictions'])} face(s)\")\n",
|
||||||
|
"print(results)"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {
|
||||||
|
"id": "dkPMOHa_univ"
|
||||||
|
},
|
||||||
|
"source": [
|
||||||
|
"## Blurring the Face\n",
|
||||||
|
"\n",
|
||||||
|
"Now that we're detecting faces, bluring them is easy with supervision. Let's pass our results into a `Detections` object and annotate the frame with a `BlurAnnotator`."
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"metadata": {},
|
||||||
|
"cell_type": "code",
|
||||||
|
"outputs": [],
|
||||||
|
"execution_count": null,
|
||||||
|
"source": [
|
||||||
|
"blur = sv.BlurAnnotator(kernel_size=100)\n",
|
||||||
|
"\n",
|
||||||
|
"detections = sv.Detections.from_inference(results)\n",
|
||||||
|
"\n",
|
||||||
|
"annotated_frame = blur.annotate(scene=frame.copy(), detections=detections)\n",
|
||||||
|
"\n",
|
||||||
|
"sv.plot_image(annotated_frame)"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {
|
||||||
|
"id": "YVNe8oe4vY4N"
|
||||||
|
},
|
||||||
|
"source": [
|
||||||
|
"## Conclusion\n",
|
||||||
|
"\n",
|
||||||
|
"With supervision, inference, and Roboflow Universe we were able to blur faces in minutes with an open source model. There are many other impressive use cases out there, so feel free to share in your own cookbooks. Happy building!"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"metadata": {
|
||||||
|
"accelerator": "GPU",
|
||||||
|
"colab": {
|
||||||
|
"cell_execution_strategy": "setup",
|
||||||
|
"gpuType": "T4",
|
||||||
|
"provenance": []
|
||||||
|
},
|
||||||
|
"kernelspec": {
|
||||||
|
"display_name": "Python 3",
|
||||||
|
"name": "python3"
|
||||||
|
},
|
||||||
|
"language_info": {
|
||||||
|
"name": "python"
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"nbformat": 4,
|
||||||
|
"nbformat_minor": 0
|
||||||
|
}
|
||||||
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
|
|
@ -34,6 +34,10 @@
|
||||||
<p class="card repo-card" data-name="Object Tracking" data-labels="TRACKING, ANNOTATOR" data-version="v0.18.0"
|
<p class="card repo-card" data-name="Object Tracking" data-labels="TRACKING, ANNOTATOR" data-version="v0.18.0"
|
||||||
data-author="nickherrig"></p>
|
data-author="nickherrig"></p>
|
||||||
</a>
|
</a>
|
||||||
|
<a href="../notebooks/blurring-faces/">
|
||||||
|
<p class="card repo-card" data-name="Blurring Faces" data-labels="ANNOTATOR,API,UNIVERSE"
|
||||||
|
data-version="v0.18.0" data-author="nickherrig"></p>
|
||||||
|
</a>
|
||||||
<a href="../notebooks/occupancy_analytics/">
|
<a href="../notebooks/occupancy_analytics/">
|
||||||
<p class="card repo-card" data-name="Analyzing Zone Occupancy" data-labels="ANNOTATOR,DETECTION,ZONES"
|
<p class="card repo-card" data-name="Analyzing Zone Occupancy" data-labels="ANNOTATOR,DETECTION,ZONES"
|
||||||
data-version="v0.26.0" data-author="stellasphere"></p>
|
data-version="v0.26.0" data-author="stellasphere"></p>
|
||||||
|
|
@ -58,6 +62,14 @@
|
||||||
<p class="card repo-card" data-name="Understand Visitors with YOLO-World"
|
<p class="card repo-card" data-name="Understand Visitors with YOLO-World"
|
||||||
data-labels="ANNOTATORS,DETECTION,INFERENCE" data-version="v0.19.0" data-author="AdonaiVera"></p>
|
data-labels="ANNOTATORS,DETECTION,INFERENCE" data-version="v0.19.0" data-author="AdonaiVera"></p>
|
||||||
</a>
|
</a>
|
||||||
|
<a href="../notebooks/compact-mask-sam3/">
|
||||||
|
<p class="card repo-card" data-name="Memory-Efficient Instance Segmentation"
|
||||||
|
data-labels="COMPACT MASK,SAM3,SEGMENTATION" data-version="v0.28.0" data-author="Borda"></p>
|
||||||
|
</a>
|
||||||
|
<a href="../notebooks/oriented-bounding-boxes/">
|
||||||
|
<p class="card repo-card" data-name="Oriented Bounding Boxes for Densely Packed Objects"
|
||||||
|
data-labels="OBB,DETECTIONS,NMS,DATASET" data-version="v0.29.0" data-author="kounelisagis"></p>
|
||||||
|
</a>
|
||||||
</div>
|
</div>
|
||||||
</div>
|
</div>
|
||||||
</section>
|
</section>
|
||||||
|
|
|
||||||
|
|
@ -120,7 +120,7 @@
|
||||||
"name": "How do I install supervision?",
|
"name": "How do I install supervision?",
|
||||||
"acceptedAnswer": {
|
"acceptedAnswer": {
|
||||||
"@type": "Answer",
|
"@type": "Answer",
|
||||||
"text": "Install supervision with pip: pip install supervision. For optional metric dependencies use pip install supervision[metrics]. Sample asset utilities are included in the base package under supervision.assets. The current package metadata requires Python 3.9+."
|
"text": "Install supervision with pip: pip install supervision. For optional metric dependencies use pip install supervision[metrics]. Sample asset utilities are included in the base package under supervision.assets. The current package metadata requires Python 3.10+."
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
|
|
|
||||||
|
|
@ -1,8 +1,12 @@
|
||||||
---
|
---
|
||||||
comments: true
|
comments: true
|
||||||
description: API reference for supervision's object trackers — ByteTrack and SORT implementations that assign persistent IDs across video frames.
|
description: API reference for supervision's deprecated ByteTrack tracker wrapper.
|
||||||
---
|
---
|
||||||
|
|
||||||
# ByteTrack
|
# ByteTrack
|
||||||
|
|
||||||
|
!!! warning "Deprecated"
|
||||||
|
|
||||||
|
`sv.ByteTrack` is deprecated in `supervision-0.28.0` and will be removed in `supervision-0.31.0`. Install `trackers` and use `ByteTrackTracker` instead.
|
||||||
|
|
||||||
:::supervision.tracker.byte_tracker.core.ByteTrack
|
:::supervision.tracker.byte_tracker.core.ByteTrack
|
||||||
|
|
|
||||||
|
|
@ -0,0 +1,36 @@
|
||||||
|
---
|
||||||
|
comments: true
|
||||||
|
status: new
|
||||||
|
---
|
||||||
|
|
||||||
|
# Conversion Utils
|
||||||
|
|
||||||
|
<div class="md-typeset">
|
||||||
|
<h2><a href="#supervision.utils.conversion.cv2_to_pillow">cv2_to_pillow</a></h2>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
:::supervision.utils.conversion.cv2_to_pillow
|
||||||
|
|
||||||
|
<div class="md-typeset">
|
||||||
|
<h2><a href="#supervision.utils.conversion.pillow_to_cv2">pillow_to_cv2</a></h2>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
:::supervision.utils.conversion.pillow_to_cv2
|
||||||
|
|
||||||
|
<div class="md-typeset">
|
||||||
|
<h2><a href="#supervision.utils.conversion.ensure_cv2_image_for_annotation">ensure_cv2_image_for_annotation</a></h2>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
:::supervision.utils.conversion.ensure_cv2_image_for_annotation
|
||||||
|
|
||||||
|
<div class="md-typeset">
|
||||||
|
<h2><a href="#supervision.utils.conversion.ensure_pil_image_for_annotation">ensure_pil_image_for_annotation</a></h2>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
:::supervision.utils.conversion.ensure_pil_image_for_annotation
|
||||||
|
|
||||||
|
<div class="md-typeset">
|
||||||
|
<h2><a href="#supervision.utils.conversion.images_to_cv2">images_to_cv2</a></h2>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
:::supervision.utils.conversion.images_to_cv2
|
||||||
|
|
@ -13,3 +13,21 @@ comments: true
|
||||||
</div>
|
</div>
|
||||||
|
|
||||||
:::supervision.geometry.core.Position
|
:::supervision.geometry.core.Position
|
||||||
|
|
||||||
|
<div class="md-typeset">
|
||||||
|
<h2><a href="#supervision.geometry.core.Point">Point</a></h2>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
:::supervision.geometry.core.Point
|
||||||
|
|
||||||
|
<div class="md-typeset">
|
||||||
|
<h2><a href="#supervision.geometry.core.Rect">Rect</a></h2>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
:::supervision.geometry.core.Rect
|
||||||
|
|
||||||
|
<div class="md-typeset">
|
||||||
|
<h2><a href="#supervision.geometry.core.Vector">Vector</a></h2>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
:::supervision.geometry.core.Vector
|
||||||
|
|
|
||||||
|
|
@ -11,6 +11,12 @@ status: new
|
||||||
|
|
||||||
:::supervision.utils.image.crop_image
|
:::supervision.utils.image.crop_image
|
||||||
|
|
||||||
|
<div class="md-typeset">
|
||||||
|
<h2><a href="#supervision.utils.image.load_image_from_url">load_image_from_url</a></h2>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
:::supervision.utils.image.load_image_from_url
|
||||||
|
|
||||||
<div class="md-typeset">
|
<div class="md-typeset">
|
||||||
<h2><a href="#supervision.utils.image.scale_image">scale_image</a></h2>
|
<h2><a href="#supervision.utils.image.scale_image">scale_image</a></h2>
|
||||||
</div>
|
</div>
|
||||||
|
|
|
||||||
|
|
@ -0,0 +1,12 @@
|
||||||
|
---
|
||||||
|
comments: true
|
||||||
|
status: new
|
||||||
|
---
|
||||||
|
|
||||||
|
# Image Window
|
||||||
|
|
||||||
|
<div class="md-typeset">
|
||||||
|
<h2><a href="#supervision.utils.image_window.ImageWindow">ImageWindow</a></h2>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
:::supervision.utils.image_window.ImageWindow
|
||||||
|
|
@ -2,7 +2,7 @@
|
||||||
|
|
||||||
This example benchmarks `CompactMask`, a new mask representation introduced in `supervision` that replaces dense `(N, H, W)` boolean arrays with a crop-scoped Run-Length Encoding (RLE). The benchmark demonstrates full API compatibility, massive memory savings, and order-of-magnitude annotation speedups — with no change to your existing `Detections` code.
|
This example benchmarks `CompactMask`, a new mask representation introduced in `supervision` that replaces dense `(N, H, W)` boolean arrays with a crop-scoped Run-Length Encoding (RLE). The benchmark demonstrates full API compatibility, massive memory savings, and order-of-magnitude annotation speedups — with no change to your existing `Detections` code.
|
||||||
|
|
||||||
---
|
______________________________________________________________________
|
||||||
|
|
||||||
## The Problem
|
## The Problem
|
||||||
|
|
||||||
|
|
@ -16,7 +16,7 @@ For a 4K image with 1 000 detected objects:
|
||||||
|
|
||||||
At this scale, typical pipelines crash with `MemoryError` before a single frame is annotated. Aerial imagery, satellite tiles, and high-density crowd scenes all hit this wall.
|
At this scale, typical pipelines crash with `MemoryError` before a single frame is annotated. Aerial imagery, satellite tiles, and high-density crowd scenes all hit this wall.
|
||||||
|
|
||||||
---
|
______________________________________________________________________
|
||||||
|
|
||||||
## The Solution — Crop-RLE Storage
|
## The Solution — Crop-RLE Storage
|
||||||
|
|
||||||
|
|
@ -95,7 +95,7 @@ Crop RLE's `.crop()` method powers the `MaskAnnotator` optimisation — it never
|
||||||
|
|
||||||
At N=1 000 with 1 % overlap, bbox pre-filter reduces 499 500 candidate pairs to ~5 000 overlapping pairs — a ~2 000x reduction in pixel-level work.
|
At N=1 000 with 1 % overlap, bbox pre-filter reduces 499 500 candidate pairs to ~5 000 overlapping pairs — a ~2 000x reduction in pixel-level work.
|
||||||
|
|
||||||
---
|
______________________________________________________________________
|
||||||
|
|
||||||
## Why Crop-RLE Was Chosen over Local Crop
|
## Why Crop-RLE Was Chosen over Local Crop
|
||||||
|
|
||||||
|
|
@ -107,7 +107,7 @@ Both formats compress extremely well; the deciding factors for Crop-RLE are:
|
||||||
|
|
||||||
The main trade-off: crop-only decode is O(A) rather than O(1). For the common solid-fill segmentation mask this is negligible (\<0.1 ms per mask).
|
The main trade-off: crop-only decode is O(A) rather than O(1). For the common solid-fill segmentation mask this is negligible (\<0.1 ms per mask).
|
||||||
|
|
||||||
---
|
______________________________________________________________________
|
||||||
|
|
||||||
## Operation-by-Operation Speedup Analysis
|
## Operation-by-Operation Speedup Analysis
|
||||||
|
|
||||||
|
|
@ -115,7 +115,7 @@ This section walks through every `Detections` operation that touches masks and s
|
||||||
|
|
||||||
At 50% fill on an FHD image each mask's bounding box covers a large portion of the frame, producing many RLE runs per row.
|
At 50% fill on an FHD image each mask's bounding box covers a large portion of the frame, producing many RLE runs per row.
|
||||||
|
|
||||||
---
|
______________________________________________________________________
|
||||||
|
|
||||||
### Memory
|
### Memory
|
||||||
|
|
||||||
|
|
@ -146,7 +146,7 @@ Scaled to N=200: 200 x 4.7 KB = ~933 KB of RLE data, plus `_crop_shapes` (1.6 KB
|
||||||
|
|
||||||
At 5% fill with 8-vertex polygons, the ratio reaches 10 000x–20 000x because crops are tiny and RLEs are extremely short. The benchmark's 4K-200-5%-v8 scenario measures 21 786x (theory) / ~6 000x (malloc). The SAT-200-5%-v8 scenario reaches 62 968x theoretical.
|
At 5% fill with 8-vertex polygons, the ratio reaches 10 000x–20 000x because crops are tiny and RLEs are extremely short. The benchmark's 4K-200-5%-v8 scenario measures 21 786x (theory) / ~6 000x (malloc). The SAT-200-5%-v8 scenario reaches 62 968x theoretical.
|
||||||
|
|
||||||
---
|
______________________________________________________________________
|
||||||
|
|
||||||
### `.area`
|
### `.area`
|
||||||
|
|
||||||
|
|
@ -179,7 +179,7 @@ At FHD-200-50%-v600, dense `.area` takes 84.66 ms; compact takes 0.48 ms — a *
|
||||||
| No (H, W) allocation per mask | latency |
|
| No (H, W) allocation per mask | latency |
|
||||||
| **Combined** | **~1 000x** |
|
| **Combined** | **~1 000x** |
|
||||||
|
|
||||||
---
|
______________________________________________________________________
|
||||||
|
|
||||||
### `filter` / `__getitem__` (boolean index)
|
### `filter` / `__getitem__` (boolean index)
|
||||||
|
|
||||||
|
|
@ -212,7 +212,7 @@ At FHD-200-50%-v600, dense `filter` takes 14.56 ms; compact takes 0.03 ms — a
|
||||||
| Allocation | new `(K, H, W)` array | new `CompactMask` shell (~trivial) |
|
| Allocation | new `(K, H, W)` array | new `CompactMask` shell (~trivial) |
|
||||||
| **Speedup** | | **hundreds to tens of thousands x** |
|
| **Speedup** | | **hundreds to tens of thousands x** |
|
||||||
|
|
||||||
---
|
______________________________________________________________________
|
||||||
|
|
||||||
### `annotate` (`MaskAnnotator`)
|
### `annotate` (`MaskAnnotator`)
|
||||||
|
|
||||||
|
|
@ -246,7 +246,7 @@ colored_mask[y1 : y1 + crop_h, x1 : x1 + crop_w][crop_m] = color.as_bgr()
|
||||||
| x N masks | compounds |
|
| x N masks | compounds |
|
||||||
| **Combined** | **~26 – 400x** |
|
| **Combined** | **~26 – 400x** |
|
||||||
|
|
||||||
---
|
______________________________________________________________________
|
||||||
|
|
||||||
### IoU (`mask_iou_batch` / `compact_mask_iou_batch`)
|
### IoU (`mask_iou_batch` / `compact_mask_iou_batch`)
|
||||||
|
|
||||||
|
|
@ -315,7 +315,7 @@ At FHD-200-50%-v600, dense IoU takes 23 915 ms; compact takes 51.58 ms — a **4
|
||||||
|
|
||||||
At 20% fill the gaps close — more pairs overlap, larger crops — speedup drops toward the lower end of the range.
|
At 20% fill the gaps close — more pairs overlap, larger crops — speedup drops toward the lower end of the range.
|
||||||
|
|
||||||
---
|
______________________________________________________________________
|
||||||
|
|
||||||
### NMS (`mask_non_max_suppression`)
|
### NMS (`mask_non_max_suppression`)
|
||||||
|
|
||||||
|
|
@ -339,7 +339,7 @@ All three IoU optimisations apply to the compact path:
|
||||||
|
|
||||||
At FHD-200-50%-v600, dense NMS takes 5 231 ms; compact takes 48.15 ms — a **481x speedup**. Dense IoU/NMS is skipped for scenarios above 1 GB (4K-200 and SAT-200 tiers); compact NMS still runs on those.
|
At FHD-200-50%-v600, dense NMS takes 5 231 ms; compact takes 48.15 ms — a **481x speedup**. Dense IoU/NMS is skipped for scenarios above 1 GB (4K-200 and SAT-200 tiers); compact NMS still runs on those.
|
||||||
|
|
||||||
---
|
______________________________________________________________________
|
||||||
|
|
||||||
### `merge` (`Detections.merge`)
|
### `merge` (`Detections.merge`)
|
||||||
|
|
||||||
|
|
@ -387,7 +387,7 @@ if len(self.xyxy) > 0:
|
||||||
|
|
||||||
This O(1) check avoids the O(N x H x W) dense materialisation that previously dominated compact merge time.
|
This O(1) check avoids the O(N x H x W) dense materialisation that previously dominated compact merge time.
|
||||||
|
|
||||||
---
|
______________________________________________________________________
|
||||||
|
|
||||||
### `offset` / `with_offset` (`InferenceSlicer` tile stitching)
|
### `offset` / `with_offset` (`InferenceSlicer` tile stitching)
|
||||||
|
|
||||||
|
|
@ -425,7 +425,7 @@ At FHD-200-50%-v600, dense offset takes 42.30 ms; compact takes 0.02 ms — a **
|
||||||
|
|
||||||
In the `InferenceSlicer` pipeline the canvas is always expanded by the tile offset, so no crop ever overflows — the fast path is always taken. Clipping only activates for objects that genuinely straddle the image boundary.
|
In the `InferenceSlicer` pipeline the canvas is always expanded by the tile offset, so no crop ever overflows — the fast path is always taken. Clipping only activates for objects that genuinely straddle the image boundary.
|
||||||
|
|
||||||
---
|
______________________________________________________________________
|
||||||
|
|
||||||
### `centroids` (`calculate_masks_centroids`)
|
### `centroids` (`calculate_masks_centroids`)
|
||||||
|
|
||||||
|
|
@ -460,7 +460,7 @@ At FHD-200-50%-v600, dense centroids takes 1 133.68 ms; compact takes 60.39 ms
|
||||||
| No global `np.indices((H, W))` allocation | saves large float64 |
|
| No global `np.indices((H, W))` allocation | saves large float64 |
|
||||||
| **Combined (N=200)** | **~19 – 1 000x** |
|
| **Combined (N=200)** | **~19 – 1 000x** |
|
||||||
|
|
||||||
---
|
______________________________________________________________________
|
||||||
|
|
||||||
### Summary
|
### Summary
|
||||||
|
|
||||||
|
|
@ -480,7 +480,7 @@ Measured speedups at the **FHD-200-50%-v600** operating point (dense fill, compl
|
||||||
|
|
||||||
All speedups are larger at sparser fill fractions and larger resolutions. At SAT-200-20%-v128, `.area` reaches 1 204x and `merge` reaches 89 046x. At the sparsest scenarios (5% fill, 8-vertex polygons), memory ratios exceed 60 000x.
|
All speedups are larger at sparser fill fractions and larger resolutions. At SAT-200-20%-v128, `.area` reaches 1 204x and `merge` reaches 89 046x. At the sparsest scenarios (5% fill, 8-vertex polygons), memory ratios exceed 60 000x.
|
||||||
|
|
||||||
---
|
______________________________________________________________________
|
||||||
|
|
||||||
## Drop-In Compatibility
|
## Drop-In Compatibility
|
||||||
|
|
||||||
|
|
@ -517,7 +517,7 @@ Supported indexing patterns:
|
||||||
| `mask[slice]` | New `CompactMask` |
|
| `mask[slice]` | New `CompactMask` |
|
||||||
| `np.asarray(mask)` | Dense `(N, H, W)` bool array |
|
| `np.asarray(mask)` | Dense `(N, H, W)` bool array |
|
||||||
|
|
||||||
---
|
______________________________________________________________________
|
||||||
|
|
||||||
## Benchmark
|
## Benchmark
|
||||||
|
|
||||||
|
|
@ -527,6 +527,69 @@ Run on any machine — no GPU or real model required:
|
||||||
uv run python examples/compact_mask/benchmark.py
|
uv run python examples/compact_mask/benchmark.py
|
||||||
```
|
```
|
||||||
|
|
||||||
|
For a focused benchmark of the Roboflow inference-result parser API, run:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
uv run python examples/compact_mask/bench_inference_api.py
|
||||||
|
```
|
||||||
|
|
||||||
|
This script downloads all supervision image assets plus the middle frame from every supervision video asset by default, runs one real segmentation inference per source image, requests native RLE masks from Inference, freezes that result, and then compares parser performance:
|
||||||
|
|
||||||
|
```python
|
||||||
|
sv.Detections.from_inference(result)
|
||||||
|
sv.Detections.from_inference(result, compact_masks=True)
|
||||||
|
```
|
||||||
|
|
||||||
|
Timing repetitions, warmups, confidence, IoU, response mask format, and the default model live as constants in `bench_inference_api.py`.
|
||||||
|
|
||||||
|
Inference runs and segmentation-derived box fields are outside the timed benchmark loop. By default the script uses `rfdetr-seg-large` with `response_mask_format="rle"`; set `BENCH_INFERENCE_MODEL_ID` to override the model. Set `ROBOFLOW_API_KEY` when your model requires authentication. Sources where the model returns no native RLE segmentation masks are skipped because there is no RLE parser work to benchmark. `rfdetr-large` is a valid local Inference model id, but it is object detection only; use an `rfdetr-seg-*` model for instance segmentation.
|
||||||
|
|
||||||
|
Run one specific supervision image or video asset with `--asset`:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
uv run python examples/compact_mask/bench_inference_api.py --asset people-walking
|
||||||
|
uv run python examples/compact_mask/bench_inference_api.py --asset soccer
|
||||||
|
uv run python examples/compact_mask/bench_inference_api.py --asset vehicles
|
||||||
|
uv run python examples/compact_mask/bench_inference_api.py --asset people-walking-video
|
||||||
|
```
|
||||||
|
|
||||||
|
The output reports image size, segmented objects, median parser time, peak traced allocations, mask storage, and parser speedup (`dense parser time / compact parser time`).
|
||||||
|
|
||||||
|
**Speedup column:** The `speedup` value reflects allocation savings — how much time is saved by skipping the dense `(N, H, W)` bool-stack allocation — not a faster RLE decode. Compact RLE arithmetic is typically slower than the dense NumPy path. The net result:
|
||||||
|
|
||||||
|
- **Compact is faster** only when the dense `(N, H, W)` bool-stack allocation dominates — large images with many sparse masks where avoiding that allocation outweighs the RLE arithmetic cost.
|
||||||
|
- **Compact is slower** for small images or dense/overlapping masks, where Python RLE arithmetic dominates and the allocation cost is negligible.
|
||||||
|
- **The primary guaranteed benefit is memory**: compact masks use roughly 99% less memory than dense stacks for typical segmentation output, regardless of which parse direction is faster.
|
||||||
|
|
||||||
|
The default run includes a `synthetic-dense-64` row (64×64 image, 4 fully-filled masks) to demonstrate the adversarial regime where compact is slower than dense. For each real source with segmentation masks, the script also writes a validation overlay to `examples/compact_mask/outputs/*_segmentations.jpg`.
|
||||||
|
|
||||||
|
### Sample results — inference API
|
||||||
|
|
||||||
|
Measured on macOS Apple M4 Max, 50 reps after 3 warmups, using `rfdetr-seg-large` via Roboflow Inference.
|
||||||
|
|
||||||
|
| src | res | seg | dense ms | CM ms | speedup | peak MB (dense/compact) | mask MB (dense/compact) | ok |
|
||||||
|
| -------------------------- | --------- | --- | -------- | ----- | ------- | ----------------------- | ----------------------- | --- |
|
||||||
|
| synthetic-dense-64 | 64×64 | 4 | 0.03 | 0.11 | 0.31× | 0.04 / 0.05 | 0.02 / 0.00 | ✓ |
|
||||||
|
| people-walking.jpg | 1920×1080 | 53 | 85.56 | 12.55 | 6.82× | 219.86 / 0.11 | 109.90 / 0.02 | ✓ |
|
||||||
|
| soccer.jpg | 398×224 | 21 | 1.36 | 1.07 | 1.27× | 3.77 / 0.05 | 1.87 / 0.00 | ✓ |
|
||||||
|
| vehicles.mp4#269 | 3840×2160 | 7 | 46.03 | 2.60 | 18× | 116.13 / 0.07 | 58.06 / 0.00 | ✓ |
|
||||||
|
| milk-bottling-plant.mp4#94 | 1920×1080 | 9 | 15.61 | 11.57 | 1.35× | 37.34 / 0.53 | 18.66 / 0.03 | ✓ |
|
||||||
|
| vehicles-2.mp4#637 | 1920×1080 | 47 | 76.87 | 13.59 | 5.66× | 194.97 / 0.13 | 97.46 / 0.03 | ✓ |
|
||||||
|
| grocery-store.mp4#501 | 3840×2160 | 4 | 27.20 | 4.36 | 6.24× | 66.36 / 0.22 | 33.18 / 0.01 | ✓ |
|
||||||
|
| subway.mp4#649 | 2160×3840 | 42 | 325.71 | 32.21 | 10× | 696.78 / 0.80 | 348.36 / 0.09 | ✓ |
|
||||||
|
| market-square.mp4#237 | 2160×3840 | 96 | 732.98 | 27.24 | 27× | 1592.61 / 0.22 | 796.26 / 0.05 | ✓ |
|
||||||
|
| people-walking.mp4#170 | 1920×1080 | 60 | 100.99 | 12.69 | 7.96× | 248.89 / 0.12 | 124.42 / 0.02 | ✓ |
|
||||||
|
| beach-1.mp4#223 | 3840×2160 | 33 | 223.50 | 13.39 | 17× | 547.47 / 0.12 | 273.72 / 0.02 | ✓ |
|
||||||
|
| basketball-1.mp4#238 | 1920×1080 | 2 | 3.61 | 2.05 | 1.76× | 8.30 / 0.15 | 4.15 / 0.01 | ✓ |
|
||||||
|
| skiing.mp4#176 | 1920×1080 | 11 | 16.47 | 3.07 | 5.37× | 45.63 / 0.08 | 22.81 / 0.01 | ✓ |
|
||||||
|
|
||||||
|
- **seg** — number of instance segmentations returned by the model
|
||||||
|
- **dense ms / CM ms** — median parse time for `from_inference()` vs `from_inference(compact_masks=True)`
|
||||||
|
- **speedup** — dense / compact parse time; values below 1× (e.g., synthetic-dense-64) indicate the adversarial regime where RLE arithmetic cost exceeds allocation savings
|
||||||
|
- **peak MB** — peak traced allocations during parsing (dense / compact)
|
||||||
|
- **mask MB** — mask storage only (dense / compact); compact is typically 100–5 000× smaller
|
||||||
|
- **ok** — `compact.to_dense()` pixel-exactly matches dense masks
|
||||||
|
|
||||||
Six image tiers x three fill fractions (5 / 20 / 50 %) x three vertex counts (8 / 128 / 600):
|
Six image tiers x three fill fractions (5 / 20 / 50 %) x three vertex counts (8 / 128 / 600):
|
||||||
|
|
||||||
| Tier | Resolution | Objects | Dense array | Notes |
|
| Tier | Resolution | Objects | Dense array | Notes |
|
||||||
|
|
@ -563,7 +626,7 @@ Dense timing is skipped automatically when the dense IoU/NMS array would exceed
|
||||||
|
|
||||||
All non-skipped scenarios pass: pixel-perfect annotation, exact area, lossless `to_dense()` roundtrip.
|
All non-skipped scenarios pass: pixel-perfect annotation, exact area, lossless `to_dense()` roundtrip.
|
||||||
|
|
||||||
---
|
______________________________________________________________________
|
||||||
|
|
||||||
## Use-Cases
|
## Use-Cases
|
||||||
|
|
||||||
|
|
@ -573,7 +636,7 @@ All non-skipped scenarios pass: pixel-perfect annotation, exact area, lossless `
|
||||||
- **Long-running tracking** — accumulated `Detections` across many frames stay in kilobytes rather than gigabytes.
|
- **Long-running tracking** — accumulated `Detections` across many frames stay in kilobytes rather than gigabytes.
|
||||||
- **`InferenceSlicer`** — `with_offset()` adjusts crop origins directly when stitching tile results; no dense materialisation needed.
|
- **`InferenceSlicer`** — `with_offset()` adjusts crop origins directly when stitching tile results; no dense materialisation needed.
|
||||||
|
|
||||||
---
|
______________________________________________________________________
|
||||||
|
|
||||||
## Limitations
|
## Limitations
|
||||||
|
|
||||||
|
|
@ -581,11 +644,12 @@ All non-skipped scenarios pass: pixel-perfect annotation, exact area, lossless `
|
||||||
- RLE format is **column-major (F-order), crop-scoped** — pixel-scan order matches COCO / pycocotools, but crop scope differs from full-image scope. Use `.to_dense()` to materialize a full-image dense mask, then encode that mask to COCO RLE before passing it to pycocotools.
|
- RLE format is **column-major (F-order), crop-scoped** — pixel-scan order matches COCO / pycocotools, but crop scope differs from full-image scope. Use `.to_dense()` to materialize a full-image dense mask, then encode that mask to COCO RLE before passing it to pycocotools.
|
||||||
- `from_dense()` requires the input `(N, H, W)` array to fit in memory. For truly OOM-scale data, build `CompactMask` per-detection directly from model output crops rather than from a pre-allocated dense stack.
|
- `from_dense()` requires the input `(N, H, W)` array to fit in memory. For truly OOM-scale data, build `CompactMask` per-detection directly from model output crops rather than from a pre-allocated dense stack.
|
||||||
|
|
||||||
---
|
______________________________________________________________________
|
||||||
|
|
||||||
## Files
|
## Files
|
||||||
|
|
||||||
| File | Description |
|
| File | Description |
|
||||||
| -------------- | ------------------------------------------------ |
|
| ------------------------ | --------------------------------------------------- |
|
||||||
| `benchmark.py` | Full benchmark across FHD / 4K / satellite tiers |
|
| `benchmark.py` | Full benchmark across FHD / 4K / satellite tiers |
|
||||||
| `README.md` | This file |
|
| `bench_inference_api.py` | Focused dense vs compact `from_inference` benchmark |
|
||||||
|
| `README.md` | This file |
|
||||||
|
|
|
||||||
|
|
@ -0,0 +1,505 @@
|
||||||
|
"""Benchmark dense vs compact Roboflow RLE ingestion.
|
||||||
|
|
||||||
|
Run with:
|
||||||
|
uv run python examples/compact_mask/bench_inference_api.py
|
||||||
|
|
||||||
|
The benchmark downloads supervision assets, runs one segmentation inference per
|
||||||
|
source image, then times dense vs compact parsing of that fixed inference result.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import gc
|
||||||
|
import os
|
||||||
|
import statistics
|
||||||
|
import time
|
||||||
|
import tracemalloc
|
||||||
|
from collections.abc import Callable
|
||||||
|
from dataclasses import dataclass
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
import cv2
|
||||||
|
import numpy as np
|
||||||
|
from rich import box
|
||||||
|
from rich.console import Console
|
||||||
|
from rich.table import Table
|
||||||
|
|
||||||
|
import supervision as sv
|
||||||
|
from supervision.assets import ImageAssets, VideoAssets, download_assets
|
||||||
|
from supervision.config import CLASS_NAME_DATA_FIELD
|
||||||
|
from supervision.detection.compact_mask import CompactMask
|
||||||
|
|
||||||
|
console = Console(width=120, force_terminal=True)
|
||||||
|
|
||||||
|
# Default segmentation model; use an rfdetr-seg-* id so masks are returned.
|
||||||
|
MODEL_ID = "rfdetr-seg-large"
|
||||||
|
# Environment variable that can override MODEL_ID without adding CLI noise.
|
||||||
|
MODEL_ID_ENV = "BENCH_INFERENCE_MODEL_ID"
|
||||||
|
# Optional Roboflow API key for models that require authentication.
|
||||||
|
API_KEY_ENV = "ROBOFLOW_API_KEY"
|
||||||
|
# Model confidence threshold used only for the one inference call per source.
|
||||||
|
CONFIDENCE = 0.2
|
||||||
|
# Model IoU threshold used only for the one inference call per source.
|
||||||
|
IOU = 0.5
|
||||||
|
# Request native RLE masks so the benchmark measures RLE parser ingestion.
|
||||||
|
RESPONSE_MASK_FORMAT = "rle"
|
||||||
|
# Parser timing repetitions; inference itself is not repeated.
|
||||||
|
REPETITIONS = 50
|
||||||
|
# Untimed parser warmup calls before measurements.
|
||||||
|
WARMUP = 3
|
||||||
|
# Visual segmentation overlays for manual validation.
|
||||||
|
ARTIFACT_DIR = Path("examples/compact_mask/outputs")
|
||||||
|
|
||||||
|
ASSETS = {Path(asset.filename).stem: asset for asset in ImageAssets}
|
||||||
|
for video_asset in VideoAssets:
|
||||||
|
key = Path(video_asset.filename).stem
|
||||||
|
ASSETS[key if key not in ASSETS else f"{key}-video"] = video_asset
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class ApiBenchmarkResult:
|
||||||
|
"""Result for one dense-vs-compact parser benchmark run."""
|
||||||
|
|
||||||
|
source: str
|
||||||
|
resolution: str
|
||||||
|
segmented_objects: int
|
||||||
|
dense_s: float
|
||||||
|
compact_s: float
|
||||||
|
dense_peak_bytes: int
|
||||||
|
compact_peak_bytes: int
|
||||||
|
dense_mask_bytes: int
|
||||||
|
compact_mask_bytes: int
|
||||||
|
pixel_perfect: bool
|
||||||
|
|
||||||
|
|
||||||
|
def load_image_from_asset(path: Path | None, asset: str) -> tuple[np.ndarray, str]:
|
||||||
|
"""Return ``(image, label)`` for an image or video middle frame."""
|
||||||
|
if path is not None:
|
||||||
|
image = cv2.imread(str(path))
|
||||||
|
if image is None:
|
||||||
|
raise FileNotFoundError(f"Could not read image: {path}")
|
||||||
|
return image, str(path)
|
||||||
|
|
||||||
|
asset_obj = ASSETS[asset]
|
||||||
|
asset_path = Path(download_assets(asset_obj))
|
||||||
|
if isinstance(asset_obj, ImageAssets):
|
||||||
|
image = cv2.imread(str(asset_path))
|
||||||
|
if image is None:
|
||||||
|
raise FileNotFoundError(f"Could not read image: {asset_path}")
|
||||||
|
return image, str(asset_path)
|
||||||
|
|
||||||
|
video = cv2.VideoCapture(str(asset_path))
|
||||||
|
if not video.isOpened():
|
||||||
|
raise FileNotFoundError(f"Could not read video: {asset_path}")
|
||||||
|
frame_count = int(video.get(cv2.CAP_PROP_FRAME_COUNT))
|
||||||
|
frame_index = max(0, frame_count // 2)
|
||||||
|
if frame_index:
|
||||||
|
video.set(cv2.CAP_PROP_POS_FRAMES, frame_index)
|
||||||
|
ok, frame = video.read()
|
||||||
|
video.release()
|
||||||
|
if not ok or frame is None:
|
||||||
|
raise FileNotFoundError(f"Could not read middle frame: {asset_path}")
|
||||||
|
return frame, f"{asset_path}#{frame_index}"
|
||||||
|
|
||||||
|
|
||||||
|
def freeze_result(inference_result: Any) -> dict[str, Any]:
|
||||||
|
"""Convert one Inference result to a reusable dictionary."""
|
||||||
|
if isinstance(inference_result, dict):
|
||||||
|
return inference_result
|
||||||
|
if hasattr(inference_result, "model_dump"):
|
||||||
|
return inference_result.model_dump(exclude_none=True, by_alias=True)
|
||||||
|
if hasattr(inference_result, "dict"):
|
||||||
|
return inference_result.dict(exclude_none=True, by_alias=True)
|
||||||
|
raise TypeError(
|
||||||
|
f"Expected dict-like Inference result, got {type(inference_result).__name__}"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def count_rle_predictions(result: dict[str, Any]) -> int:
|
||||||
|
"""Return the number of predictions carrying Roboflow RLE masks."""
|
||||||
|
return sum(
|
||||||
|
isinstance(prediction.get("rle") or prediction.get("rle_mask"), dict)
|
||||||
|
for prediction in result.get("predictions", [])
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def synthetic_dense_small_result() -> tuple[np.ndarray, str, dict[str, Any]]:
|
||||||
|
"""Return a small dense-mask adversarial payload where compact parsing is slower.
|
||||||
|
|
||||||
|
Uses a 64x64 image with 4 fully-filled masks. At this scale the dense
|
||||||
|
``(N, H, W)`` allocation cost is negligible; Python RLE arithmetic dominates,
|
||||||
|
making compact ingestion slower than the dense NumPy path. Included as a
|
||||||
|
clearly labeled adversarial row in the default benchmark run to show that
|
||||||
|
the ``speedup`` column reflects allocation savings, not decode speed.
|
||||||
|
"""
|
||||||
|
height, width = 64, 64
|
||||||
|
image = np.zeros((height, width, 3), dtype=np.uint8)
|
||||||
|
predictions = [
|
||||||
|
{
|
||||||
|
"x": width / 2,
|
||||||
|
"y": height / 2,
|
||||||
|
"width": width,
|
||||||
|
"height": height,
|
||||||
|
"confidence": 0.9,
|
||||||
|
"class_id": index,
|
||||||
|
"class": f"dense-{index}",
|
||||||
|
"rle": {"size": [height, width], "counts": [0, height * width]},
|
||||||
|
}
|
||||||
|
for index in range(4)
|
||||||
|
]
|
||||||
|
return (
|
||||||
|
image,
|
||||||
|
"synthetic-dense-64",
|
||||||
|
{
|
||||||
|
"predictions": predictions,
|
||||||
|
"image": {"width": width, "height": height},
|
||||||
|
},
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def derive_boxes_from_rle_masks(result: dict[str, Any]) -> dict[str, Any]:
|
||||||
|
"""Set prediction boxes from native RLE segmentation masks."""
|
||||||
|
predictions = []
|
||||||
|
for prediction in result.get("predictions", []):
|
||||||
|
rle = prediction.get("rle") or prediction.get("rle_mask")
|
||||||
|
if not isinstance(rle, dict):
|
||||||
|
predictions.append(prediction)
|
||||||
|
continue
|
||||||
|
|
||||||
|
height, width = rle["size"]
|
||||||
|
mask = sv.rle_to_mask(rle["counts"], resolution_wh=(int(width), int(height)))
|
||||||
|
if not mask.any():
|
||||||
|
predictions.append(prediction)
|
||||||
|
continue
|
||||||
|
|
||||||
|
x1, y1, x2, y2 = sv.mask_to_xyxy(mask[np.newaxis, ...])[0]
|
||||||
|
predictions.append(
|
||||||
|
{
|
||||||
|
**prediction,
|
||||||
|
"x": float((x1 + x2) / 2),
|
||||||
|
"y": float((y1 + y2) / 2),
|
||||||
|
"width": float(x2 - x1),
|
||||||
|
"height": float(y2 - y1),
|
||||||
|
}
|
||||||
|
)
|
||||||
|
return {**result, "predictions": predictions}
|
||||||
|
|
||||||
|
|
||||||
|
def artifact_path(source: str) -> Path:
|
||||||
|
"""Return the segmentation validation artifact path for a source."""
|
||||||
|
source_path, separator, frame = source.partition("#")
|
||||||
|
stem = Path(source_path).stem
|
||||||
|
suffix = f"_frame_{frame}" if separator else ""
|
||||||
|
return ARTIFACT_DIR / f"{stem}{suffix}_segmentations.jpg"
|
||||||
|
|
||||||
|
|
||||||
|
def detection_labels(detections: sv.Detections) -> list[str]:
|
||||||
|
"""Return compact class/confidence labels for validation artifacts."""
|
||||||
|
raw_class_names = detections.get_data(CLASS_NAME_DATA_FIELD)
|
||||||
|
class_names = (
|
||||||
|
raw_class_names.astype(str).tolist()
|
||||||
|
if isinstance(raw_class_names, np.ndarray)
|
||||||
|
else [""] * len(detections)
|
||||||
|
)
|
||||||
|
|
||||||
|
labels = []
|
||||||
|
for index in range(len(detections)):
|
||||||
|
class_name = class_names[index] if index < len(class_names) else ""
|
||||||
|
confidence = (
|
||||||
|
""
|
||||||
|
if detections.confidence is None
|
||||||
|
else f" {detections.confidence[index]:.2f}"
|
||||||
|
)
|
||||||
|
labels.append(f"{class_name}{confidence}".strip() or str(index))
|
||||||
|
return labels
|
||||||
|
|
||||||
|
|
||||||
|
def save_segmentation_artifact(
|
||||||
|
image: np.ndarray,
|
||||||
|
result: dict[str, Any],
|
||||||
|
source: str,
|
||||||
|
) -> Path | None:
|
||||||
|
"""Draw parsed segmentation masks and save a validation artifact."""
|
||||||
|
detections = sv.Detections.from_inference(result)
|
||||||
|
if detections.mask is None:
|
||||||
|
return None
|
||||||
|
|
||||||
|
annotated = image.copy()
|
||||||
|
annotated = sv.MaskAnnotator(
|
||||||
|
color_lookup=sv.ColorLookup.INDEX,
|
||||||
|
opacity=0.45,
|
||||||
|
).annotate(scene=annotated, detections=detections)
|
||||||
|
annotated = sv.LabelAnnotator(
|
||||||
|
color_lookup=sv.ColorLookup.INDEX,
|
||||||
|
text_scale=0.35,
|
||||||
|
text_padding=4,
|
||||||
|
).annotate(
|
||||||
|
scene=annotated,
|
||||||
|
detections=detections,
|
||||||
|
labels=detection_labels(detections),
|
||||||
|
)
|
||||||
|
|
||||||
|
path = artifact_path(source)
|
||||||
|
path.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
if not cv2.imwrite(str(path), annotated):
|
||||||
|
raise OSError(f"Could not write segmentation artifact: {path}")
|
||||||
|
return path
|
||||||
|
|
||||||
|
|
||||||
|
def load_inference_model(model_id: str, api_key: str | None) -> Any:
|
||||||
|
"""Load the requested Inference model."""
|
||||||
|
try:
|
||||||
|
from inference import get_model
|
||||||
|
except ImportError as exc:
|
||||||
|
raise ImportError(
|
||||||
|
"Install the `inference` package to run this benchmark."
|
||||||
|
) from exc
|
||||||
|
|
||||||
|
model_kwargs = {"api_key": api_key} if api_key is not None else {}
|
||||||
|
return get_model(model_id=model_id, **model_kwargs)
|
||||||
|
|
||||||
|
|
||||||
|
def run_inference_once(
|
||||||
|
image: np.ndarray,
|
||||||
|
model: Any,
|
||||||
|
model_id: str,
|
||||||
|
confidence: float,
|
||||||
|
iou: float,
|
||||||
|
) -> dict[str, Any] | None:
|
||||||
|
"""Run one real segmentation inference and return a frozen result."""
|
||||||
|
# Inference still serializes instance segmentations with x/y/width/height.
|
||||||
|
# Derive those fields from the RLE masks so the benchmark uses segmentations,
|
||||||
|
# not the model-reported detector boxes, as the source of truth.
|
||||||
|
result = derive_boxes_from_rle_masks(
|
||||||
|
freeze_result(
|
||||||
|
model.infer(
|
||||||
|
image,
|
||||||
|
confidence=confidence,
|
||||||
|
iou=iou,
|
||||||
|
response_mask_format=RESPONSE_MASK_FORMAT,
|
||||||
|
)[0]
|
||||||
|
)
|
||||||
|
)
|
||||||
|
rle_count = count_rle_predictions(result)
|
||||||
|
if rle_count == 0:
|
||||||
|
console.print(
|
||||||
|
f"[yellow]skipped[/yellow] {model_id}: no native RLE segmentation "
|
||||||
|
f"predictions for response_mask_format={RESPONSE_MASK_FORMAT!r}"
|
||||||
|
)
|
||||||
|
return None
|
||||||
|
return result
|
||||||
|
|
||||||
|
|
||||||
|
def median_seconds(fn: Callable[[], object], reps: int, warmup: int) -> float:
|
||||||
|
"""Return median runtime for ``fn``."""
|
||||||
|
for _ in range(warmup):
|
||||||
|
fn()
|
||||||
|
gc.collect()
|
||||||
|
|
||||||
|
timings = []
|
||||||
|
for _ in range(reps):
|
||||||
|
start = time.perf_counter()
|
||||||
|
fn()
|
||||||
|
timings.append(time.perf_counter() - start)
|
||||||
|
return statistics.median(timings)
|
||||||
|
|
||||||
|
|
||||||
|
def peak_bytes(fn: Callable[[], object]) -> int:
|
||||||
|
"""Return peak traced allocations for one call."""
|
||||||
|
gc.collect()
|
||||||
|
tracemalloc.start()
|
||||||
|
fn()
|
||||||
|
_, peak = tracemalloc.get_traced_memory()
|
||||||
|
tracemalloc.stop()
|
||||||
|
return int(peak)
|
||||||
|
|
||||||
|
|
||||||
|
def dense_mask_bytes(detections: sv.Detections) -> int:
|
||||||
|
"""Return dense mask storage bytes."""
|
||||||
|
return 0 if detections.mask is None else int(np.asarray(detections.mask).nbytes)
|
||||||
|
|
||||||
|
|
||||||
|
def compact_mask_bytes(detections: sv.Detections) -> int:
|
||||||
|
"""Return compact mask storage bytes."""
|
||||||
|
if not isinstance(detections.mask, CompactMask):
|
||||||
|
return 0
|
||||||
|
return sum(rle.nbytes for rle in detections.mask._rles)
|
||||||
|
|
||||||
|
|
||||||
|
def _fmt_ratio(ratio: float) -> str:
|
||||||
|
"""Format a speedup/compression ratio with colour coding."""
|
||||||
|
fmt = f"{ratio:.0f}x" if ratio >= 10 else f"{ratio:.2f}x"
|
||||||
|
if ratio >= 10:
|
||||||
|
return f"[green]{fmt}[/green]"
|
||||||
|
elif ratio >= 1:
|
||||||
|
return f"[yellow]{fmt}[/yellow]"
|
||||||
|
else:
|
||||||
|
return f"[red]{fmt}[/red]"
|
||||||
|
|
||||||
|
|
||||||
|
def _fmt_mb(num_bytes: int) -> str:
|
||||||
|
"""Format bytes as compact megabytes."""
|
||||||
|
return f"{num_bytes / 1e6:.2f}"
|
||||||
|
|
||||||
|
|
||||||
|
def run_benchmark(
|
||||||
|
source: str,
|
||||||
|
image: np.ndarray,
|
||||||
|
result: dict[str, Any],
|
||||||
|
reps: int,
|
||||||
|
warmup: int,
|
||||||
|
) -> ApiBenchmarkResult:
|
||||||
|
"""Run one dense-vs-compact parser benchmark."""
|
||||||
|
|
||||||
|
# Benchmark the public Roboflow/Inference adapter; RLE masks enter through
|
||||||
|
# the result payload and should stay compact when compact_masks=True.
|
||||||
|
def dense() -> sv.Detections:
|
||||||
|
return sv.Detections.from_inference(result)
|
||||||
|
|
||||||
|
def compact() -> sv.Detections:
|
||||||
|
return sv.Detections.from_inference(result, compact_masks=True)
|
||||||
|
|
||||||
|
dense_once = dense()
|
||||||
|
compact_once = compact()
|
||||||
|
if not isinstance(dense_once.mask, np.ndarray):
|
||||||
|
raise TypeError(f"Expected dense ndarray mask, got {type(dense_once.mask)}")
|
||||||
|
if not isinstance(compact_once.mask, CompactMask):
|
||||||
|
raise TypeError(f"Expected CompactMask, got {type(compact_once.mask)}")
|
||||||
|
np.testing.assert_array_equal(compact_once.mask.to_dense(), dense_once.mask)
|
||||||
|
|
||||||
|
dense_s = median_seconds(dense, reps, warmup)
|
||||||
|
compact_s = median_seconds(compact, reps, warmup)
|
||||||
|
dense_peak = peak_bytes(dense)
|
||||||
|
compact_peak = peak_bytes(compact)
|
||||||
|
|
||||||
|
return ApiBenchmarkResult(
|
||||||
|
source=source,
|
||||||
|
resolution=f"{image.shape[1]}x{image.shape[0]}",
|
||||||
|
segmented_objects=len(dense_once),
|
||||||
|
dense_s=dense_s,
|
||||||
|
compact_s=compact_s,
|
||||||
|
dense_peak_bytes=dense_peak,
|
||||||
|
compact_peak_bytes=compact_peak,
|
||||||
|
dense_mask_bytes=dense_mask_bytes(dense_once),
|
||||||
|
compact_mask_bytes=compact_mask_bytes(compact_once),
|
||||||
|
pixel_perfect=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def print_summary(results: list[ApiBenchmarkResult], reps: int, warmup: int) -> None:
|
||||||
|
"""Print a Rich summary table matching the compact mask benchmark style."""
|
||||||
|
table = Table(
|
||||||
|
title="CompactMask from_inference",
|
||||||
|
box=box.ROUNDED,
|
||||||
|
show_lines=False,
|
||||||
|
header_style="bold cyan",
|
||||||
|
)
|
||||||
|
table.add_column("src", style="bold", no_wrap=True)
|
||||||
|
table.add_column("res", no_wrap=True)
|
||||||
|
table.add_column("seg", justify="right")
|
||||||
|
table.add_column("dense ms", justify="right")
|
||||||
|
table.add_column("CM ms", justify="right", style="green")
|
||||||
|
table.add_column("speedup", justify="right")
|
||||||
|
table.add_column("peak MB", justify="right", style="cyan")
|
||||||
|
table.add_column("mask MB", justify="right")
|
||||||
|
table.add_column("ok", justify="center")
|
||||||
|
|
||||||
|
for result in results:
|
||||||
|
speedup = result.dense_s / max(result.compact_s, 1e-9)
|
||||||
|
table.add_row(
|
||||||
|
result.source,
|
||||||
|
result.resolution,
|
||||||
|
str(result.segmented_objects),
|
||||||
|
f"{result.dense_s * 1e3:.2f}",
|
||||||
|
f"{result.compact_s * 1e3:.2f}",
|
||||||
|
_fmt_ratio(speedup),
|
||||||
|
f"{_fmt_mb(result.dense_peak_bytes)}/{_fmt_mb(result.compact_peak_bytes)}",
|
||||||
|
f"{_fmt_mb(result.dense_mask_bytes)}/{_fmt_mb(result.compact_mask_bytes)}",
|
||||||
|
"[green]✓[/green]" if result.pixel_perfect else "[red]✗[/red]",
|
||||||
|
)
|
||||||
|
|
||||||
|
console.print(table)
|
||||||
|
console.print(
|
||||||
|
"[dim]"
|
||||||
|
+ " · ".join(
|
||||||
|
[
|
||||||
|
f"timings are median of {reps} reps after {warmup} warmups",
|
||||||
|
"peak MB and mask MB are dense/compact",
|
||||||
|
"speedup = dense / compact parse time; gains are allocation-driven"
|
||||||
|
" (avoiding the dense (N,H,W) bool-stack), not faster RLE decode",
|
||||||
|
"compact RLE arithmetic is typically slower than the dense NumPy path"
|
||||||
|
" — synthetic-dense-64 shows this adversarial regime (speedup < 1x)",
|
||||||
|
"OK means compact.to_dense() exactly matches dense masks",
|
||||||
|
]
|
||||||
|
)
|
||||||
|
+ "[/dim]"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> None:
|
||||||
|
"""Run the benchmark."""
|
||||||
|
parser = argparse.ArgumentParser()
|
||||||
|
parser.add_argument("--asset", choices=ASSETS.keys(), default=None)
|
||||||
|
parser.add_argument("--image", type=Path, default=None)
|
||||||
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
assets = [args.asset] if args.asset is not None else list(ASSETS)
|
||||||
|
if args.image is not None:
|
||||||
|
assets = ["custom"]
|
||||||
|
|
||||||
|
results = []
|
||||||
|
if args.asset is None and args.image is None:
|
||||||
|
image, source, inference_result = synthetic_dense_small_result()
|
||||||
|
console.rule(f"[bold]{source}[/bold] | {image.shape[1]}x{image.shape[0]}")
|
||||||
|
results.append(
|
||||||
|
run_benchmark(
|
||||||
|
source=source,
|
||||||
|
image=image,
|
||||||
|
result=inference_result,
|
||||||
|
reps=REPETITIONS,
|
||||||
|
warmup=WARMUP,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
model_id = os.getenv(MODEL_ID_ENV, MODEL_ID)
|
||||||
|
model = load_inference_model(model_id=model_id, api_key=os.getenv(API_KEY_ENV))
|
||||||
|
for asset in assets:
|
||||||
|
image, source = load_image_from_asset(args.image, asset)
|
||||||
|
console.rule(f"[bold]{source}[/bold] | {image.shape[1]}x{image.shape[0]}")
|
||||||
|
inference_result = run_inference_once(
|
||||||
|
image=image,
|
||||||
|
model=model,
|
||||||
|
model_id=model_id,
|
||||||
|
confidence=CONFIDENCE,
|
||||||
|
iou=IOU,
|
||||||
|
)
|
||||||
|
if inference_result is None:
|
||||||
|
continue
|
||||||
|
console.print(
|
||||||
|
f"[dim]captured {count_rle_predictions(inference_result)} RLE masks "
|
||||||
|
f"from {model_id}[/dim]"
|
||||||
|
)
|
||||||
|
artifact = save_segmentation_artifact(
|
||||||
|
image=image,
|
||||||
|
result=inference_result,
|
||||||
|
source=source,
|
||||||
|
)
|
||||||
|
if artifact is not None:
|
||||||
|
console.print(f"[dim]saved segmentation artifact: {artifact}[/dim]")
|
||||||
|
results.append(
|
||||||
|
run_benchmark(
|
||||||
|
source=source,
|
||||||
|
image=image,
|
||||||
|
result=inference_result,
|
||||||
|
reps=REPETITIONS,
|
||||||
|
warmup=WARMUP,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
if not results:
|
||||||
|
raise ValueError(f"Model {model_id!r} returned no segmentation masks.")
|
||||||
|
print_summary(results, reps=REPETITIONS, warmup=WARMUP)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
|
|
@ -2,7 +2,9 @@
|
||||||
|
|
||||||
Demonstrates that ``CompactMask`` is a drop-in replacement for dense
|
Demonstrates that ``CompactMask`` is a drop-in replacement for dense
|
||||||
``(N, H, W)`` bool arrays in ``supervision.Detections``, while using
|
``(N, H, W)`` bool arrays in ``supervision.Detections``, while using
|
||||||
significantly less memory and enabling faster annotation.
|
significantly less memory and enabling faster annotation. The annotation
|
||||||
|
timing reports frame size, detection count, mask area ratio, and
|
||||||
|
``MaskAnnotator`` speedup from ROI-only blending.
|
||||||
|
|
||||||
Run with:
|
Run with:
|
||||||
uv run python examples/compact_mask/benchmark.py
|
uv run python examples/compact_mask/benchmark.py
|
||||||
|
|
@ -12,19 +14,17 @@ Mask complexity is controlled by ``num_vertices``: random polygons with more
|
||||||
vertices produce jaggier boundaries and more RLE runs per row.
|
vertices produce jaggier boundaries and more RLE runs per row.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import dataclasses
|
import dataclasses
|
||||||
import gc
|
import gc
|
||||||
import json
|
import json
|
||||||
import math
|
import math
|
||||||
import time
|
import time
|
||||||
import tracemalloc
|
import tracemalloc
|
||||||
|
from collections.abc import Callable
|
||||||
from concurrent.futures import ThreadPoolExecutor
|
from concurrent.futures import ThreadPoolExecutor
|
||||||
from dataclasses import dataclass, field
|
from dataclasses import dataclass, field
|
||||||
from datetime import datetime, timezone
|
from datetime import datetime, timezone
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Callable
|
|
||||||
|
|
||||||
import cv2
|
import cv2
|
||||||
import numpy as np
|
import numpy as np
|
||||||
|
|
@ -74,7 +74,7 @@ class ScenarioResult:
|
||||||
name: str
|
name: str
|
||||||
resolution: str # e.g. "1920x1080"
|
resolution: str # e.g. "1920x1080"
|
||||||
num_objects: int
|
num_objects: int
|
||||||
fill_name: str # e.g. "5%"
|
fill_name: str # mask area ratio, e.g. "5%"
|
||||||
num_vertices: int # polygon vertex count — complexity proxy
|
num_vertices: int # polygon vertex count — complexity proxy
|
||||||
# memory (theoretical: raw numpy nbytes)
|
# memory (theoretical: raw numpy nbytes)
|
||||||
dense_bytes: int
|
dense_bytes: int
|
||||||
|
|
@ -956,7 +956,7 @@ def print_summary(results: list[ScenarioResult]) -> None:
|
||||||
table.add_column("Scenario", style="bold", min_width=22)
|
table.add_column("Scenario", style="bold", min_width=22)
|
||||||
table.add_column("Objects", justify="right", min_width=7)
|
table.add_column("Objects", justify="right", min_width=7)
|
||||||
table.add_column("Resolution", min_width=12, no_wrap=True)
|
table.add_column("Resolution", min_width=12, no_wrap=True)
|
||||||
table.add_column("Fill", justify="right", min_width=5, no_wrap=True)
|
table.add_column("Mask\narea", justify="right", min_width=5, no_wrap=True)
|
||||||
table.add_column("Vertices", justify="right", min_width=8, no_wrap=True)
|
table.add_column("Vertices", justify="right", min_width=8, no_wrap=True)
|
||||||
table.add_column("Dense\ntheory", justify="right", min_width=10)
|
table.add_column("Dense\ntheory", justify="right", min_width=10)
|
||||||
table.add_column("Compact\ntheory", justify="right", style="green", min_width=9)
|
table.add_column("Compact\ntheory", justify="right", style="green", min_width=9)
|
||||||
|
|
@ -1037,7 +1037,8 @@ def print_summary(results: list[ScenarioResult]) -> None:
|
||||||
"Decode ms/mask — to_dense() / N (compact→dense overhead per mask)",
|
"Decode ms/mask — to_dense() / N (compact→dense overhead per mask)",
|
||||||
"Area x — .area speedup (RLE sum, no materialisation)",
|
"Area x — .area speedup (RLE sum, no materialisation)",
|
||||||
"Filter x — boolean-index speedup",
|
"Filter x — boolean-index speedup",
|
||||||
"Annot x — MaskAnnotator speedup (crop-paint vs full-frame alloc)",
|
"Annot x — MaskAnnotator speedup "
|
||||||
|
"(ROI-only blend vs full-frame overlay)",
|
||||||
f"IoU x — pairwise self-IoU speedup "
|
f"IoU x — pairwise self-IoU speedup "
|
||||||
f"(dense skipped >{IOU_DENSE_SKIP_GB:.0f} GB)",
|
f"(dense skipped >{IOU_DENSE_SKIP_GB:.0f} GB)",
|
||||||
"NMS x — mask_non_max_suppression speedup",
|
"NMS x — mask_non_max_suppression speedup",
|
||||||
|
|
|
||||||
|
|
@ -1,14 +1,10 @@
|
||||||
# count people in zone
|
# count people in zone
|
||||||
|
|
||||||
[](https://colab.research.google.com/github/roboflow-ai/notebooks/blob/main/notebooks/how-to-detect-and-count-objects-in-polygon-zone.ipynb)
|
[](https://colab.research.google.com/github/roboflow-ai/notebooks/blob/main/notebooks/how-to-detect-and-count-objects-in-polygon-zone.ipynb) [](https://www.youtube.com/watch?v=l_kf9CfZ_8M)
|
||||||
[](https://www.youtube.com/watch?v=l_kf9CfZ_8M)
|
|
||||||
|
|
||||||
## 👋 hello
|
## 👋 hello
|
||||||
|
|
||||||
This demo is a video analysis tool that counts and highlights objects in specific zones
|
This demo is a video analysis tool that counts and highlights objects in specific zones of a video. Each zone and the objects within it are marked in different colors, making it easy to see and count the objects in each area. The tool can save this enhanced video or display it live on the screen.
|
||||||
of a video. Each zone and the objects within it are marked in different colors, making
|
|
||||||
it easy to see and count the objects in each area. The tool can save this enhanced
|
|
||||||
video or display it live on the screen.
|
|
||||||
|
|
||||||
https://github.com/roboflow/supervision/assets/26109316/f84db7b5-79e2-4142-a1da-64daa43ce667
|
https://github.com/roboflow/supervision/assets/26109316/f84db7b5-79e2-4142-a1da-64daa43ce667
|
||||||
|
|
||||||
|
|
@ -16,76 +12,61 @@ https://github.com/roboflow/supervision/assets/26109316/f84db7b5-79e2-4142-a1da-
|
||||||
|
|
||||||
- clone repository and navigate to example directory
|
- clone repository and navigate to example directory
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
git clone --depth 1 -b develop https://github.com/roboflow/supervision.git
|
git clone --depth 1 -b develop https://github.com/roboflow/supervision.git
|
||||||
cd supervision/examples/count_people_in_zone
|
cd supervision/examples/count_people_in_zone
|
||||||
```
|
```
|
||||||
|
|
||||||
- setup python environment and activate it [optional]
|
- setup python environment and activate it [optional]
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
uv venv
|
uv venv
|
||||||
source .venv/bin/activate
|
source .venv/bin/activate
|
||||||
```
|
```
|
||||||
|
|
||||||
- install required dependencies
|
- install required dependencies
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
uv pip install -r requirements.txt
|
uv pip install -r requirements.txt
|
||||||
```
|
```
|
||||||
|
|
||||||
- download `traffic_analysis.pt` and `traffic_analysis.mov` files
|
- download `traffic_analysis.pt` and `traffic_analysis.mov` files
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
./setup.sh
|
./setup.sh
|
||||||
```
|
```
|
||||||
|
|
||||||
## 🛠️ script arguments
|
## 🛠️ script arguments
|
||||||
|
|
||||||
- ultralytics
|
- ultralytics
|
||||||
|
|
||||||
- `--source_weights_path` (optional): The path to the YOLO model's weights file.
|
- `--source_weights_path` (optional): The path to the YOLO model's weights file. Defaults to `"yolov8x.pt"` if not specified.
|
||||||
Defaults to `"yolov8x.pt"` if not specified.
|
|
||||||
|
|
||||||
- `--zone_configuration_path`: Specifies the path to the JSON file containing zone
|
- `--zone_configuration_path`: Specifies the path to the JSON file containing zone configurations. This file defines the polygonal areas in the video where objects will be counted.
|
||||||
configurations. This file defines the polygonal areas in the video where objects will
|
|
||||||
be counted.
|
|
||||||
|
|
||||||
- `--source_video_path`: The path to the source video file that will be analyzed.
|
- `--source_video_path`: The path to the source video file that will be analyzed.
|
||||||
|
|
||||||
- `--target_video_path` (optional): The path to save the output video with annotations.
|
- `--target_video_path` (optional): The path to save the output video with annotations. If not provided, the processed video will be displayed in real-time.
|
||||||
If not provided, the processed video will be displayed in real-time.
|
|
||||||
|
|
||||||
- `--confidence_threshold` (optional): Sets the confidence threshold for the YOLO model
|
- `--confidence_threshold` (optional): Sets the confidence threshold for the YOLO model to filter detections. Default is `0.3`.
|
||||||
to filter detections. Default is `0.3`.
|
|
||||||
|
|
||||||
- `--iou_threshold` (optional): Specifies the IOU (Intersection Over Union) threshold
|
- `--iou_threshold` (optional): Specifies the IOU (Intersection Over Union) threshold for the model. Default is `0.7`.
|
||||||
for the model. Default is `0.7`.
|
|
||||||
|
|
||||||
- inference
|
- inference
|
||||||
|
|
||||||
- `--roboflow_api_key` (optional): The API key for Roboflow services. If not provided
|
- `--roboflow_api_key` (optional): The API key for Roboflow services. If not provided directly, the script tries to fetch it from the `ROBOFLOW_API_KEY` environment variable. Follow [this guide](https://docs.roboflow.com/api-reference/authentication#retrieve-an-api-key) to acquire your `API KEY`.
|
||||||
directly, the script tries to fetch it from the `ROBOFLOW_API_KEY` environment
|
|
||||||
variable. Follow [this guide](https://docs.roboflow.com/api-reference/authentication#retrieve-an-api-key)
|
|
||||||
to acquire your `API KEY`.
|
|
||||||
|
|
||||||
- `--model_id` (optional): Designates the Roboflow model ID to be used. The default
|
- `--model_id` (optional): Designates the Roboflow model ID to be used. The default value is `"yolov8x-1280"`.
|
||||||
value is `"yolov8x-1280"`.
|
|
||||||
|
|
||||||
- `--zone_configuration_path`: Specifies the path to the JSON file containing zone
|
- `--zone_configuration_path`: Specifies the path to the JSON file containing zone configurations. This file defines the polygonal areas in the video where objects will be counted.
|
||||||
configurations. This file defines the polygonal areas in the video where objects will
|
|
||||||
be counted.
|
|
||||||
|
|
||||||
- `--source_video_path`: The path to the source video file that will be analyzed.
|
- `--source_video_path`: The path to the source video file that will be analyzed.
|
||||||
|
|
||||||
- `--target_video_path` (optional): The path to save the output video with annotations.
|
- `--target_video_path` (optional): The path to save the output video with annotations. If not provided, the processed video will be displayed in real-time.
|
||||||
If not provided, the processed video will be displayed in real-time.
|
|
||||||
|
|
||||||
- `--confidence_threshold` (optional): Sets the confidence threshold for the YOLO model
|
- `--confidence_threshold` (optional): Sets the confidence threshold for the YOLO model to filter detections. Default is `0.3`.
|
||||||
to filter detections. Default is `0.3`.
|
|
||||||
|
|
||||||
- `--iou_threshold` (optional): Specifies the IOU (Intersection Over Union) threshold
|
- `--iou_threshold` (optional): Specifies the IOU (Intersection Over Union) threshold for the model. Default is `0.7`.
|
||||||
for the model. Default is `0.7`.
|
|
||||||
|
|
||||||
## 📌 zone configuration
|
## 📌 zone configuration
|
||||||
|
|
||||||
|
|
@ -98,35 +79,29 @@ https://github.com/roboflow/supervision/assets/26109316/f84db7b5-79e2-4142-a1da-
|
||||||
|
|
||||||
- ultralytics
|
- ultralytics
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
python ultralytics_example.py \
|
python ultralytics_example.py \
|
||||||
--zone_configuration_path data/multi-zone-config.json \
|
--zone_configuration_path data/multi-zone-config.json \
|
||||||
--source_video_path data/market-square.mp4 \
|
--source_video_path data/market-square.mp4 \
|
||||||
--confidence_threshold 0.3 \
|
--confidence_threshold 0.3 \
|
||||||
--iou_threshold 0.5
|
--iou_threshold 0.5
|
||||||
```
|
```
|
||||||
|
|
||||||
- inference
|
- inference
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
python inference_example.py \
|
python inference_example.py \
|
||||||
--roboflow_api_key "ROBOFLOW_API_KEY" \
|
--roboflow_api_key "ROBOFLOW_API_KEY" \
|
||||||
--zone_configuration_path data/multi-zone-config.json \
|
--zone_configuration_path data/multi-zone-config.json \
|
||||||
--source_video_path data/market-square.mp4 \
|
--source_video_path data/market-square.mp4 \
|
||||||
--confidence_threshold 0.3 \
|
--confidence_threshold 0.3 \
|
||||||
--iou_threshold 0.5
|
--iou_threshold 0.5
|
||||||
```
|
```
|
||||||
|
|
||||||
## © license
|
## © license
|
||||||
|
|
||||||
This demo integrates two main components, each with its own licensing:
|
This demo integrates two main components, each with its own licensing:
|
||||||
|
|
||||||
- ultralytics: The object detection model used in this demo, YOLOv8, is distributed
|
- ultralytics: The object detection model used in this demo, YOLOv8, is distributed under the [AGPL-3.0 license](https://github.com/ultralytics/ultralytics/blob/main/LICENSE). You can find more details about this license here.
|
||||||
under the [AGPL-3.0 license](https://github.com/ultralytics/ultralytics/blob/main/LICENSE).
|
|
||||||
You can find more details about this license here.
|
|
||||||
|
|
||||||
- supervision: The analytics code that powers the zone-based analysis in this demo is
|
- supervision: The analytics code that powers the zone-based analysis in this demo is based on the Supervision library, which is licensed under the [MIT license](https://github.com/roboflow/supervision/blob/develop/LICENSE.md). This makes the Supervision part of the code fully open source and freely usable in your projects.
|
||||||
based on the Supervision library, which is licensed under the
|
|
||||||
[MIT license](https://github.com/roboflow/supervision/blob/develop/LICENSE.md). This
|
|
||||||
makes the Supervision part of the code fully open source and freely usable in your
|
|
||||||
projects.
|
|
||||||
|
|
|
||||||
|
|
@ -1,7 +1,6 @@
|
||||||
import json
|
import json
|
||||||
import os
|
import os
|
||||||
|
|
||||||
import cv2
|
|
||||||
import numpy as np
|
import numpy as np
|
||||||
from inference.core.models.roboflow import RoboflowInferenceModel
|
from inference.core.models.roboflow import RoboflowInferenceModel
|
||||||
from inference.models.utils import get_roboflow_model
|
from inference.models.utils import get_roboflow_model
|
||||||
|
|
@ -117,7 +116,7 @@ def annotate(
|
||||||
"""
|
"""
|
||||||
annotated_frame = frame.copy()
|
annotated_frame = frame.copy()
|
||||||
for zone, zone_annotator, box_annotator in zip(
|
for zone, zone_annotator, box_annotator in zip(
|
||||||
zones, zone_annotators, box_annotators
|
zones, zone_annotators, box_annotators, strict=True
|
||||||
):
|
):
|
||||||
detections_in_zone = detections[zone.trigger(detections=detections)]
|
detections_in_zone = detections[zone.trigger(detections=detections)]
|
||||||
annotated_frame = zone_annotator.annotate(scene=annotated_frame)
|
annotated_frame = zone_annotator.annotate(scene=annotated_frame)
|
||||||
|
|
@ -135,7 +134,7 @@ def main(
|
||||||
target_video_path: str | None = None,
|
target_video_path: str | None = None,
|
||||||
confidence_threshold: float = 0.3,
|
confidence_threshold: float = 0.3,
|
||||||
iou_threshold: float = 0.7,
|
iou_threshold: float = 0.7,
|
||||||
):
|
) -> None:
|
||||||
"""
|
"""
|
||||||
Counting people in zones with Inference and Supervision.
|
Counting people in zones with Inference and Supervision.
|
||||||
|
|
||||||
|
|
@ -179,6 +178,7 @@ def main(
|
||||||
)
|
)
|
||||||
sink.write_frame(annotated_frame)
|
sink.write_frame(annotated_frame)
|
||||||
else:
|
else:
|
||||||
|
window = sv.ImageWindow("Processed Video")
|
||||||
for frame in tqdm(frames_generator, total=video_info.total_frames):
|
for frame in tqdm(frames_generator, total=video_info.total_frames):
|
||||||
detections = detect(frame, model, confidence_threshold, iou_threshold)
|
detections = detect(frame, model, confidence_threshold, iou_threshold)
|
||||||
annotated_frame = annotate(
|
annotated_frame = annotate(
|
||||||
|
|
@ -188,11 +188,12 @@ def main(
|
||||||
box_annotators=box_annotators,
|
box_annotators=box_annotators,
|
||||||
detections=detections,
|
detections=detections,
|
||||||
)
|
)
|
||||||
cv2.imshow("Processed Video", annotated_frame)
|
window.show(annotated_frame)
|
||||||
if cv2.waitKey(1) & 0xFF == ord("q"):
|
key = window.wait_key(1)
|
||||||
|
if not window.is_open or key == "q":
|
||||||
break
|
break
|
||||||
|
|
||||||
cv2.destroyAllWindows()
|
window.close()
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
|
|
|
||||||
|
|
@ -1,6 +1,5 @@
|
||||||
import json
|
import json
|
||||||
|
|
||||||
import cv2
|
|
||||||
import numpy as np
|
import numpy as np
|
||||||
from tqdm import tqdm
|
from tqdm import tqdm
|
||||||
from ultralytics import YOLO
|
from ultralytics import YOLO
|
||||||
|
|
@ -116,7 +115,7 @@ def annotate(
|
||||||
"""
|
"""
|
||||||
annotated_frame = frame.copy()
|
annotated_frame = frame.copy()
|
||||||
for zone, zone_annotator, box_annotator in zip(
|
for zone, zone_annotator, box_annotator in zip(
|
||||||
zones, zone_annotators, box_annotators
|
zones, zone_annotators, box_annotators, strict=True
|
||||||
):
|
):
|
||||||
detections_in_zone = detections[zone.trigger(detections=detections)]
|
detections_in_zone = detections[zone.trigger(detections=detections)]
|
||||||
annotated_frame = zone_annotator.annotate(scene=annotated_frame)
|
annotated_frame = zone_annotator.annotate(scene=annotated_frame)
|
||||||
|
|
@ -133,7 +132,7 @@ def main(
|
||||||
target_video_path: str | None = None,
|
target_video_path: str | None = None,
|
||||||
confidence_threshold: float = 0.3,
|
confidence_threshold: float = 0.3,
|
||||||
iou_threshold: float = 0.7,
|
iou_threshold: float = 0.7,
|
||||||
):
|
) -> None:
|
||||||
"""
|
"""
|
||||||
Counting people in zones with YOLO and Supervision.
|
Counting people in zones with YOLO and Supervision.
|
||||||
|
|
||||||
|
|
@ -167,6 +166,7 @@ def main(
|
||||||
)
|
)
|
||||||
sink.write_frame(annotated_frame)
|
sink.write_frame(annotated_frame)
|
||||||
else:
|
else:
|
||||||
|
window = sv.ImageWindow("Processed Video")
|
||||||
for frame in tqdm(frames_generator, total=video_info.total_frames):
|
for frame in tqdm(frames_generator, total=video_info.total_frames):
|
||||||
detections = detect(frame, model, confidence_threshold, iou_threshold)
|
detections = detect(frame, model, confidence_threshold, iou_threshold)
|
||||||
annotated_frame = annotate(
|
annotated_frame = annotate(
|
||||||
|
|
@ -176,11 +176,12 @@ def main(
|
||||||
box_annotators=box_annotators,
|
box_annotators=box_annotators,
|
||||||
detections=detections,
|
detections=detections,
|
||||||
)
|
)
|
||||||
cv2.imshow("Processed Video", annotated_frame)
|
window.show(annotated_frame)
|
||||||
if cv2.waitKey(1) & 0xFF == ord("q"):
|
key = window.wait_key(1)
|
||||||
|
if not window.is_open or key == "q":
|
||||||
break
|
break
|
||||||
|
|
||||||
cv2.destroyAllWindows()
|
window.close()
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
|
|
|
||||||
|
|
@ -2,51 +2,42 @@
|
||||||
|
|
||||||
## 👋 hello
|
## 👋 hello
|
||||||
|
|
||||||
This script performs heatmap and tracking analysis using YOLOv8, an object-detection method and
|
This script performs heatmap and tracking analysis using YOLOv8, an object-detection method and ByteTrack, a simple yet effective online multi-object tracking method. It uses the supervision package for multiple tasks such as drawing heatmap annotations, tracking objects, etc.
|
||||||
ByteTrack, a simple yet effective online multi-object tracking method. It uses the
|
|
||||||
supervision package for multiple tasks such as drawing heatmap annotations, tracking objects, etc.
|
|
||||||
|
|
||||||
## 💻 install
|
## 💻 install
|
||||||
|
|
||||||
- clone repository and navigate to example directory
|
- clone repository and navigate to example directory
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
git clone --depth 1 -b develop https://github.com/roboflow/supervision.git
|
git clone --depth 1 -b develop https://github.com/roboflow/supervision.git
|
||||||
cd supervision/examples/heatmap_and_track
|
cd supervision/examples/heatmap_and_track
|
||||||
```
|
```
|
||||||
|
|
||||||
- setup python environment and activate it [optional]
|
- setup python environment and activate it [optional]
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
uv venv
|
uv venv
|
||||||
source .venv/bin/activate
|
source .venv/bin/activate
|
||||||
```
|
```
|
||||||
|
|
||||||
- install required dependencies
|
- install required dependencies
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
uv pip install -r requirements.txt
|
uv pip install -r requirements.txt
|
||||||
```
|
```
|
||||||
|
|
||||||
## 🛠️ script arguments
|
## 🛠️ script arguments
|
||||||
|
|
||||||
- `--source_weights_path`: Required. Specifies the path to the weights file for the
|
- `--source_weights_path`: Required. Specifies the path to the weights file for the YOLO model. This file contains the trained model data necessary for object detection.
|
||||||
YOLO model. This file contains the trained model data necessary for object detection.
|
- `--source_video_path` (optional): The path to the source video file that will be analyzed. This is the input video on which crowd analysis will be performed. If not specified default is `people-walking.mp4` from supervision assets
|
||||||
- `--source_video_path` (optional): The path to the source video file that will be
|
|
||||||
analyzed. This is the input video on which crowd analysis will be performed.
|
|
||||||
If not specified default is `people-walking.mp4` from supervision assets
|
|
||||||
- `--target_video_path` (optional): The path to save the output.mp4 video with annotations.
|
- `--target_video_path` (optional): The path to save the output.mp4 video with annotations.
|
||||||
- `--confidence_threshold` (optional): Sets the confidence threshold for the YOLO model
|
- `--confidence_threshold` (optional): Sets the confidence threshold for the YOLO model to filter detections. Default is `0.3`. This determines how confident the model should be to recognize an object in the video.
|
||||||
to filter detections. Default is `0.3`. This determines how confident the model should
|
- `--iou_threshold` (optional): Specifies the IOU (Intersection Over Union) threshold for the model. Default is 0.7. This value is used to manage object detection accuracy, particularly in distinguishing between different objects.
|
||||||
be to recognize an object in the video.
|
|
||||||
- `--iou_threshold` (optional): Specifies the IOU (Intersection Over Union) threshold
|
|
||||||
for the model. Default is 0.7. This value is used to manage object detection accuracy,
|
|
||||||
particularly in distinguishing between different objects.
|
|
||||||
- `--heatmap_alpha` (optional): Opacity of the overlay mask, between 0 and 1.
|
- `--heatmap_alpha` (optional): Opacity of the overlay mask, between 0 and 1.
|
||||||
- `--radius` (optional): Radius of the heat circle.
|
- `--radius` (optional): Radius of the heat circle.
|
||||||
- `--track_threshold` (optional): Detection confidence threshold for track activation.
|
- `--track_activation_threshold` (optional): Detection confidence threshold for track activation.
|
||||||
- `--track_seconds` (optional): Number of seconds to buffer when a track is lost.
|
- `--track_seconds` (optional): Number of seconds to buffer when a track is lost.
|
||||||
- `--match_threshold` (optional): Threshold for matching tracks with detections.
|
- `--minimum_matching_threshold` (optional): Threshold for matching tracks with detections.
|
||||||
|
|
||||||
## ⚙️ run
|
## ⚙️ run
|
||||||
|
|
||||||
|
|
@ -63,12 +54,6 @@ python script.py \
|
||||||
|
|
||||||
This demo integrates two main components, each with its own licensing:
|
This demo integrates two main components, each with its own licensing:
|
||||||
|
|
||||||
- ultralytics: The object detection model used in this demo, YOLOv8, is distributed
|
- ultralytics: The object detection model used in this demo, YOLOv8, is distributed under the [AGPL-3.0 license](https://github.com/ultralytics/ultralytics/blob/main/LICENSE). You can find more details about this license here.
|
||||||
under the [AGPL-3.0 license](https://github.com/ultralytics/ultralytics/blob/main/LICENSE).
|
|
||||||
You can find more details about this license here.
|
|
||||||
|
|
||||||
- supervision: The analytics code that powers the zone-based analysis in this demo is
|
- supervision: The analytics code that powers the zone-based analysis in this demo is based on the Supervision library, which is licensed under the [MIT license](https://github.com/roboflow/supervision/blob/develop/LICENSE.md). This makes the Supervision part of the code fully open source and freely usable in your projects.
|
||||||
based on the Supervision library, which is licensed under the
|
|
||||||
[MIT license](https://github.com/roboflow/supervision/blob/develop/LICENSE.md). This
|
|
||||||
makes the Supervision part of the code fully open source and freely usable in your
|
|
||||||
projects.
|
|
||||||
|
|
|
||||||
|
|
@ -1,123 +1,121 @@
|
||||||
from typing import Optional
|
import cv2
|
||||||
|
from ultralytics import YOLO
|
||||||
import cv2
|
|
||||||
from ultralytics import YOLO
|
import supervision as sv
|
||||||
|
from supervision.assets import VideoAssets, download_assets
|
||||||
import supervision as sv
|
|
||||||
from supervision.assets import VideoAssets, download_assets
|
|
||||||
|
def download_video() -> str:
|
||||||
|
download_assets(VideoAssets.PEOPLE_WALKING)
|
||||||
def download_video() -> str:
|
return VideoAssets.PEOPLE_WALKING.value
|
||||||
download_assets(VideoAssets.PEOPLE_WALKING)
|
|
||||||
return VideoAssets.PEOPLE_WALKING.value
|
|
||||||
|
def main(
|
||||||
|
source_weights_path: str,
|
||||||
def main(
|
source_video_path: str | None = None,
|
||||||
source_weights_path: str,
|
target_video_path: str = "output.mp4",
|
||||||
source_video_path: Optional[str] = None,
|
confidence_threshold: float = 0.35,
|
||||||
target_video_path: str = "output.mp4",
|
iou_threshold: float = 0.5,
|
||||||
confidence_threshold: float = 0.35,
|
heatmap_alpha: float = 0.5,
|
||||||
iou_threshold: float = 0.5,
|
radius: int = 25,
|
||||||
heatmap_alpha: float = 0.5,
|
track_activation_threshold: float = 0.35,
|
||||||
radius: int = 25,
|
track_seconds: int = 5,
|
||||||
track_activation_threshold: float = 0.35,
|
minimum_matching_threshold: float = 0.99,
|
||||||
track_seconds: int = 5,
|
) -> None:
|
||||||
minimum_matching_threshold: float = 0.99,
|
"""
|
||||||
) -> None:
|
Heatmap and Tracking with Supervision.
|
||||||
"""
|
|
||||||
Heatmap and Tracking with Supervision.
|
Args:
|
||||||
|
source_weights_path: Path to the source weights file
|
||||||
Args:
|
source_video_path: Path to the source video file
|
||||||
source_weights_path: Path to the source weights file
|
target_video_path: Path to the target video file
|
||||||
source_video_path: Path to the source video file
|
confidence_threshold: Confidence threshold for the model
|
||||||
target_video_path: Path to the target video file
|
iou_threshold: IOU threshold for the model
|
||||||
confidence_threshold: Confidence threshold for the model
|
heatmap_alpha: Opacity of the overlay mask, between 0 and 1
|
||||||
iou_threshold: IOU threshold for the model
|
radius: Radius of the heat circle
|
||||||
heatmap_alpha: Opacity of the overlay mask, between 0 and 1
|
track_activation_threshold: Detection confidence threshold for track activation
|
||||||
radius: Radius of the heat circle
|
track_seconds: Number of seconds to buffer when a track is lost
|
||||||
track_activation_threshold: Detection confidence threshold for track activation
|
minimum_matching_threshold: Threshold for matching tracks with detections
|
||||||
track_seconds: Number of seconds to buffer when a track is lost
|
"""
|
||||||
minimum_matching_threshold: Threshold for matching tracks with detections
|
### instantiate model
|
||||||
"""
|
model = YOLO(source_weights_path)
|
||||||
### instantiate model
|
source_video_path = source_video_path or download_video()
|
||||||
model = YOLO(source_weights_path)
|
|
||||||
source_video_path = source_video_path or download_video()
|
### heatmap config
|
||||||
|
heat_map_annotator = sv.HeatMapAnnotator(
|
||||||
### heatmap config
|
position=sv.Position.BOTTOM_CENTER,
|
||||||
heat_map_annotator = sv.HeatMapAnnotator(
|
opacity=heatmap_alpha,
|
||||||
position=sv.Position.BOTTOM_CENTER,
|
radius=radius,
|
||||||
opacity=heatmap_alpha,
|
kernel_size=25,
|
||||||
radius=radius,
|
top_hue=0,
|
||||||
kernel_size=25,
|
low_hue=125,
|
||||||
top_hue=0,
|
)
|
||||||
low_hue=125,
|
|
||||||
)
|
### annotation config
|
||||||
|
label_annotator = sv.LabelAnnotator(text_position=sv.Position.CENTER)
|
||||||
### annotation config
|
|
||||||
label_annotator = sv.LabelAnnotator(text_position=sv.Position.CENTER)
|
### get the video fps
|
||||||
|
cap = cv2.VideoCapture(source_video_path)
|
||||||
### get the video fps
|
fps = int(cap.get(cv2.CAP_PROP_FPS))
|
||||||
cap = cv2.VideoCapture(source_video_path)
|
cap.release()
|
||||||
fps = int(cap.get(cv2.CAP_PROP_FPS))
|
|
||||||
cap.release()
|
### tracker config
|
||||||
|
byte_tracker = sv.ByteTrack(
|
||||||
### tracker config
|
track_activation_threshold=track_activation_threshold,
|
||||||
byte_tracker = sv.ByteTrack(
|
lost_track_buffer=track_seconds * fps,
|
||||||
track_activation_threshold=track_activation_threshold,
|
minimum_matching_threshold=minimum_matching_threshold,
|
||||||
lost_track_buffer=track_seconds * fps,
|
frame_rate=fps,
|
||||||
minimum_matching_threshold=minimum_matching_threshold,
|
)
|
||||||
frame_rate=fps,
|
|
||||||
)
|
### video config
|
||||||
|
video_info = sv.VideoInfo.from_video_path(video_path=source_video_path)
|
||||||
### video config
|
frames_generator = sv.get_video_frames_generator(
|
||||||
video_info = sv.VideoInfo.from_video_path(video_path=source_video_path)
|
source_path=source_video_path, stride=1
|
||||||
frames_generator = sv.get_video_frames_generator(
|
)
|
||||||
source_path=source_video_path, stride=1
|
|
||||||
)
|
### Detect, track, annotate, save
|
||||||
|
with sv.VideoSink(target_path=target_video_path, video_info=video_info) as sink:
|
||||||
### Detect, track, annotate, save
|
for frame in frames_generator:
|
||||||
with sv.VideoSink(target_path=target_video_path, video_info=video_info) as sink:
|
result = model(
|
||||||
for frame in frames_generator:
|
source=frame,
|
||||||
result = model(
|
classes=[0], # only person class
|
||||||
source=frame,
|
conf=confidence_threshold,
|
||||||
classes=[0], # only person class
|
iou=iou_threshold,
|
||||||
conf=confidence_threshold,
|
# show_conf = True,
|
||||||
iou=iou_threshold,
|
# save_txt = True,
|
||||||
# show_conf = True,
|
# save_conf = True,
|
||||||
# save_txt = True,
|
# save = True,
|
||||||
# save_conf = True,
|
device=None, # use None = CPU, 0 = single GPU, or [0,1] = dual GPU
|
||||||
# save = True,
|
)[0]
|
||||||
device=None, # use None = CPU, 0 = single GPU, or [0,1] = dual GPU
|
|
||||||
)[0]
|
detections = sv.Detections.from_ultralytics(result) # get detections
|
||||||
|
|
||||||
detections = sv.Detections.from_ultralytics(result) # get detections
|
detections = byte_tracker.update_with_detections(
|
||||||
|
detections
|
||||||
detections = byte_tracker.update_with_detections(
|
) # update tracker
|
||||||
detections
|
|
||||||
) # update tracker
|
### draw heatmap
|
||||||
|
annotated_frame = heat_map_annotator.annotate(
|
||||||
### draw heatmap
|
scene=frame.copy(), detections=detections
|
||||||
annotated_frame = heat_map_annotator.annotate(
|
)
|
||||||
scene=frame.copy(), detections=detections
|
|
||||||
)
|
### draw other attributes from `detections` object
|
||||||
|
labels = [
|
||||||
### draw other attributes from `detections` object
|
f"#{tracker_id}"
|
||||||
labels = [
|
for class_id, tracker_id in zip(
|
||||||
f"#{tracker_id}"
|
detections.class_id, detections.tracker_id
|
||||||
for class_id, tracker_id in zip(
|
)
|
||||||
detections.class_id, detections.tracker_id
|
]
|
||||||
)
|
|
||||||
]
|
label_annotator.annotate(
|
||||||
|
scene=annotated_frame, detections=detections, labels=labels
|
||||||
label_annotator.annotate(
|
)
|
||||||
scene=annotated_frame, detections=detections, labels=labels
|
|
||||||
)
|
sink.write_frame(frame=annotated_frame)
|
||||||
|
|
||||||
sink.write_frame(frame=annotated_frame)
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
from jsonargparse import auto_cli, set_parsing_settings
|
||||||
if __name__ == "__main__":
|
|
||||||
from jsonargparse import auto_cli, set_parsing_settings
|
set_parsing_settings(parse_optionals_as_positionals=True)
|
||||||
|
auto_cli(main, as_positional=False)
|
||||||
set_parsing_settings(parse_optionals_as_positionals=True)
|
|
||||||
auto_cli(main, as_positional=False)
|
|
||||||
|
|
|
||||||
|
|
@ -1,122 +1,96 @@
|
||||||
# speed estimation
|
# speed estimation
|
||||||
|
|
||||||
[](https://colab.research.google.com/github/roboflow-ai/notebooks/blob/main/notebooks/how-to-estimate-vehicle-speed-with-computer-vision.ipynb)
|
[](https://colab.research.google.com/github/roboflow-ai/notebooks/blob/main/notebooks/how-to-estimate-vehicle-speed-with-computer-vision.ipynb) [](https://youtu.be/uWP6UjDeZvY)
|
||||||
[](https://youtu.be/uWP6UjDeZvY)
|
|
||||||
|
|
||||||
## 👋 hello
|
## 👋 hello
|
||||||
|
|
||||||
This example performs speed estimation analysis using various object-detection models
|
This example performs speed estimation analysis using various object-detection models and ByteTrack - a simple yet effective online multi-object tracking method. It uses the supervision package for multiple tasks such as tracking, annotations, etc.
|
||||||
and ByteTrack - a simple yet effective online multi-object tracking method. It uses the
|
|
||||||
supervision package for multiple tasks such as tracking, annotations, etc.
|
|
||||||
|
|
||||||
https://github.com/roboflow/supervision/assets/26109316/d50118c1-2ae4-458d-915a-5d860fd36f71
|
https://github.com/roboflow/supervision/assets/26109316/d50118c1-2ae4-458d-915a-5d860fd36f71
|
||||||
|
|
||||||
> [!IMPORTANT]
|
> [!IMPORTANT] Adjust the [`SOURCE`](https://github.com/roboflow/supervision/blob/e32b05a636dab2ea1f39299e529c4b22b8baa8da/examples/speed_estimation/ultralytics_example.py#L10) and [`TARGET`](https://github.com/roboflow/supervision/blob/e32b05a636dab2ea1f39299e529c4b22b8baa8da/examples/speed_estimation/ultralytics_example.py#L15) configuration if you plan to run a speed estimation script on your video file. Those must be adjusted separately for each camera view. You can learn more from our YouTube [tutorial](https://youtu.be/uWP6UjDeZvY).
|
||||||
> Adjust the [`SOURCE`](https://github.com/roboflow/supervision/blob/e32b05a636dab2ea1f39299e529c4b22b8baa8da/examples/speed_estimation/ultralytics_example.py#L10)
|
|
||||||
> and [`TARGET`](https://github.com/roboflow/supervision/blob/e32b05a636dab2ea1f39299e529c4b22b8baa8da/examples/speed_estimation/ultralytics_example.py#L15)
|
|
||||||
> configuration if you plan to run a speed estimation script on your video file. Those must be adjusted separately for each camera view. You can learn more
|
|
||||||
> from our YouTube [tutorial](https://youtu.be/uWP6UjDeZvY).
|
|
||||||
|
|
||||||
## 💻 install
|
## 💻 install
|
||||||
|
|
||||||
- clone repository and navigate to example directory
|
- clone repository and navigate to example directory
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
git clone --depth 1 -b develop https://github.com/roboflow/supervision.git
|
git clone --depth 1 -b develop https://github.com/roboflow/supervision.git
|
||||||
cd supervision/examples/speed_estimation
|
cd supervision/examples/speed_estimation
|
||||||
```
|
```
|
||||||
|
|
||||||
- setup python environment and activate it [optional]
|
- setup python environment and activate it [optional]
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
uv venv
|
uv venv
|
||||||
source .venv/bin/activate
|
source .venv/bin/activate
|
||||||
```
|
```
|
||||||
|
|
||||||
- install required dependencies
|
- install required dependencies
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
uv pip install -r requirements.txt
|
uv pip install -r requirements.txt
|
||||||
```
|
```
|
||||||
|
|
||||||
- download `vehicles.mp4` file
|
- download `vehicles.mp4` file
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
python video_downloader.py
|
python video_downloader.py
|
||||||
```
|
```
|
||||||
|
|
||||||
## 🛠️ script arguments
|
## 🛠️ script arguments
|
||||||
|
|
||||||
- `--roboflow_api_key` (optional): The API key for Roboflow services. If not provided
|
- `--roboflow_api_key` (optional): The API key for Roboflow services. If not provided directly, the script tries to fetch it from the `ROBOFLOW_API_KEY` environment variable. Follow [this guide](https://docs.roboflow.com/api-reference/authentication#retrieve-an-api-key) to acquire your `API KEY`.
|
||||||
directly, the script tries to fetch it from the `ROBOFLOW_API_KEY` environment
|
|
||||||
variable. Follow [this guide](https://docs.roboflow.com/api-reference/authentication#retrieve-an-api-key)
|
|
||||||
to acquire your `API KEY`.
|
|
||||||
|
|
||||||
- `--model_id` (optional): Designates the Roboflow model ID to be used. The default
|
- `--model_id` (optional): Designates the Roboflow model ID to be used. The default value is `"yolov8x-1280"`.
|
||||||
value is `"yolov8x-1280"`.
|
|
||||||
|
|
||||||
- `--source_weights_path`: Required. Specifies the path to the YOLO model's weights
|
- `--source_weights_path`: Required. Specifies the path to the YOLO model's weights file, which is essential for the object detection process. This file contains the data that the model uses to identify objects in the video.
|
||||||
file, which is essential for the object detection process. This file contains the
|
|
||||||
data that the model uses to identify objects in the video.
|
|
||||||
|
|
||||||
- `--source_video_path`: Required. The path to the source video file that will be
|
- `--source_video_path`: Required. The path to the source video file that will be analyzed. This is the input video on which traffic flow analysis will be performed.
|
||||||
analyzed. This is the input video on which traffic flow analysis will be performed.
|
|
||||||
|
|
||||||
- `--target_video_path`: The path to save the output video with
|
- `--target_video_path`: The path to save the output video with annotations. If not specified, the processed video will be displayed in real-time without being saved.
|
||||||
annotations. If not specified, the processed video will be displayed in real-time
|
|
||||||
without being saved.
|
|
||||||
|
|
||||||
- `--confidence_threshold` (optional): Sets the confidence threshold for the YOLO
|
- `--confidence_threshold` (optional): Sets the confidence threshold for the YOLO model to filter detections. Default is `0.3`. This determines how confident the model should be to recognize an object in the video.
|
||||||
model to filter detections. Default is `0.3`. This determines how confident the
|
|
||||||
model should be to recognize an object in the video.
|
|
||||||
|
|
||||||
- `--iou_threshold` (optional): Specifies the IOU (Intersection Over Union) threshold
|
- `--iou_threshold` (optional): Specifies the IOU (Intersection Over Union) threshold for the model. Default is 0.7. This value is used to manage object detection accuracy, particularly in distinguishing between different objects.
|
||||||
for the model. Default is 0.7. This value is used to manage object detection
|
|
||||||
accuracy, particularly in distinguishing between different objects.
|
|
||||||
|
|
||||||
## ⚙️ run
|
## ⚙️ run
|
||||||
|
|
||||||
- yolo-nas
|
- yolo-nas
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
python yolo_nas_example.py \
|
python yolo_nas_example.py \
|
||||||
--source_video_path data/vehicles.mp4 \
|
--source_video_path data/vehicles.mp4 \
|
||||||
--target_video_path data/vehicles-result.mp4 \
|
--target_video_path data/vehicles-result.mp4 \
|
||||||
--confidence_threshold 0.3 \
|
--confidence_threshold 0.3 \
|
||||||
--iou_threshold 0.5
|
--iou_threshold 0.5
|
||||||
```
|
```
|
||||||
|
|
||||||
- inference
|
- inference
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
python inference_example.py \
|
python inference_example.py \
|
||||||
--roboflow_api_key "ROBOFLOW_API_KEY" \
|
--roboflow_api_key "ROBOFLOW_API_KEY" \
|
||||||
--source_video_path data/vehicles.mp4 \
|
--source_video_path data/vehicles.mp4 \
|
||||||
--target_video_path data/vehicles-result.mp4 \
|
--target_video_path data/vehicles-result.mp4 \
|
||||||
--confidence_threshold 0.3 \
|
--confidence_threshold 0.3 \
|
||||||
--iou_threshold 0.5
|
--iou_threshold 0.5
|
||||||
```
|
```
|
||||||
|
|
||||||
- ultralytics
|
- ultralytics
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
python ultralytics_example.py \
|
python ultralytics_example.py \
|
||||||
--source_video_path data/vehicles.mp4 \
|
--source_video_path data/vehicles.mp4 \
|
||||||
--target_video_path data/vehicles-result.mp4 \
|
--target_video_path data/vehicles-result.mp4 \
|
||||||
--confidence_threshold 0.3 \
|
--confidence_threshold 0.3 \
|
||||||
--iou_threshold 0.5
|
--iou_threshold 0.5
|
||||||
```
|
```
|
||||||
|
|
||||||
## © license
|
## © license
|
||||||
|
|
||||||
This demo integrates two main components, each with its own licensing:
|
This demo integrates two main components, each with its own licensing:
|
||||||
|
|
||||||
- ultralytics: The object detection model used in this demo, YOLOv8, is distributed
|
- ultralytics: The object detection model used in this demo, YOLOv8, is distributed under the [AGPL-3.0 license](https://github.com/ultralytics/ultralytics/blob/main/LICENSE). You can find more details about this license here.
|
||||||
under the [AGPL-3.0 license](https://github.com/ultralytics/ultralytics/blob/main/LICENSE).
|
|
||||||
You can find more details about this license here.
|
|
||||||
|
|
||||||
- supervision: The analytics code that powers the zone-based analysis in this demo is
|
- supervision: The analytics code that powers the zone-based analysis in this demo is based on the Supervision library, which is licensed under the [MIT license](https://github.com/roboflow/supervision/blob/develop/LICENSE.md). This makes the Supervision part of the code fully open source and freely usable in your projects.
|
||||||
based on the Supervision library, which is licensed under the
|
|
||||||
[MIT license](https://github.com/roboflow/supervision/blob/develop/LICENSE.md). This
|
|
||||||
makes the Supervision part of the code fully open source and freely usable in your
|
|
||||||
projects.
|
|
||||||
|
|
|
||||||
|
|
@ -44,7 +44,7 @@ def main(
|
||||||
roboflow_api_key: str | None = None,
|
roboflow_api_key: str | None = None,
|
||||||
confidence_threshold: float = 0.3,
|
confidence_threshold: float = 0.3,
|
||||||
iou_threshold: float = 0.7,
|
iou_threshold: float = 0.7,
|
||||||
):
|
) -> None:
|
||||||
"""
|
"""
|
||||||
Vehicle Speed Estimation using Inference and Supervision.
|
Vehicle Speed Estimation using Inference and Supervision.
|
||||||
|
|
||||||
|
|
@ -96,6 +96,7 @@ def main(
|
||||||
coordinates = defaultdict(lambda: deque(maxlen=int(video_info.fps)))
|
coordinates = defaultdict(lambda: deque(maxlen=int(video_info.fps)))
|
||||||
|
|
||||||
with sv.VideoSink(target_video_path, video_info) as sink:
|
with sv.VideoSink(target_video_path, video_info) as sink:
|
||||||
|
window = sv.ImageWindow("frame")
|
||||||
for frame in frame_generator:
|
for frame in frame_generator:
|
||||||
results = model.infer(
|
results = model.infer(
|
||||||
frame, confidence=confidence_threshold, iou=iou_threshold
|
frame, confidence=confidence_threshold, iou=iou_threshold
|
||||||
|
|
@ -109,7 +110,7 @@ def main(
|
||||||
)
|
)
|
||||||
points = view_transformer.transform_points(points=points).astype(int)
|
points = view_transformer.transform_points(points=points).astype(int)
|
||||||
|
|
||||||
for tracker_id, [_, y] in zip(detections.tracker_id, points):
|
for tracker_id, [_, y] in zip(detections.tracker_id, points, strict=True):
|
||||||
coordinates[tracker_id].append(y)
|
coordinates[tracker_id].append(y)
|
||||||
|
|
||||||
labels = []
|
labels = []
|
||||||
|
|
@ -136,10 +137,11 @@ def main(
|
||||||
)
|
)
|
||||||
|
|
||||||
sink.write_frame(annotated_frame)
|
sink.write_frame(annotated_frame)
|
||||||
cv2.imshow("frame", annotated_frame)
|
window.show(annotated_frame)
|
||||||
if cv2.waitKey(1) & 0xFF == ord("q"):
|
key = window.wait_key(1)
|
||||||
|
if not window.is_open or key == "q":
|
||||||
break
|
break
|
||||||
cv2.destroyAllWindows()
|
window.close()
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
|
|
|
||||||
|
|
@ -41,7 +41,7 @@ def main(
|
||||||
target_video_path: str,
|
target_video_path: str,
|
||||||
confidence_threshold: float = 0.3,
|
confidence_threshold: float = 0.3,
|
||||||
iou_threshold: float = 0.7,
|
iou_threshold: float = 0.7,
|
||||||
):
|
) -> None:
|
||||||
"""
|
"""
|
||||||
Vehicle Speed Estimation using Ultralytics and Supervision.
|
Vehicle Speed Estimation using Ultralytics and Supervision.
|
||||||
|
|
||||||
|
|
@ -82,6 +82,7 @@ def main(
|
||||||
coordinates = defaultdict(lambda: deque(maxlen=int(video_info.fps)))
|
coordinates = defaultdict(lambda: deque(maxlen=int(video_info.fps)))
|
||||||
|
|
||||||
with sv.VideoSink(target_video_path, video_info) as sink:
|
with sv.VideoSink(target_video_path, video_info) as sink:
|
||||||
|
window = sv.ImageWindow("frame")
|
||||||
for frame in frame_generator:
|
for frame in frame_generator:
|
||||||
result = model(frame, conf=confidence_threshold, iou=iou_threshold)[0]
|
result = model(frame, conf=confidence_threshold, iou=iou_threshold)[0]
|
||||||
detections = sv.Detections.from_ultralytics(result)
|
detections = sv.Detections.from_ultralytics(result)
|
||||||
|
|
@ -93,7 +94,7 @@ def main(
|
||||||
)
|
)
|
||||||
points = view_transformer.transform_points(points=points).astype(int)
|
points = view_transformer.transform_points(points=points).astype(int)
|
||||||
|
|
||||||
for tracker_id, [_, y] in zip(detections.tracker_id, points):
|
for tracker_id, [_, y] in zip(detections.tracker_id, points, strict=True):
|
||||||
coordinates[tracker_id].append(y)
|
coordinates[tracker_id].append(y)
|
||||||
|
|
||||||
labels = []
|
labels = []
|
||||||
|
|
@ -120,10 +121,11 @@ def main(
|
||||||
)
|
)
|
||||||
|
|
||||||
sink.write_frame(annotated_frame)
|
sink.write_frame(annotated_frame)
|
||||||
cv2.imshow("frame", annotated_frame)
|
window.show(annotated_frame)
|
||||||
if cv2.waitKey(1) & 0xFF == ord("q"):
|
key = window.wait_key(1)
|
||||||
|
if not window.is_open or key == "q":
|
||||||
break
|
break
|
||||||
cv2.destroyAllWindows()
|
window.close()
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
|
|
|
||||||
|
|
@ -42,7 +42,7 @@ def main(
|
||||||
target_video_path: str,
|
target_video_path: str,
|
||||||
confidence_threshold: float = 0.3,
|
confidence_threshold: float = 0.3,
|
||||||
iou_threshold: float = 0.7,
|
iou_threshold: float = 0.7,
|
||||||
):
|
) -> None:
|
||||||
"""
|
"""
|
||||||
Vehicle Speed Estimation using YOLO-NAS and Supervision.
|
Vehicle Speed Estimation using YOLO-NAS and Supervision.
|
||||||
|
|
||||||
|
|
@ -83,6 +83,7 @@ def main(
|
||||||
coordinates = defaultdict(lambda: deque(maxlen=int(video_info.fps)))
|
coordinates = defaultdict(lambda: deque(maxlen=int(video_info.fps)))
|
||||||
|
|
||||||
with sv.VideoSink(target_video_path, video_info) as sink:
|
with sv.VideoSink(target_video_path, video_info) as sink:
|
||||||
|
window = sv.ImageWindow("frame")
|
||||||
for frame in frame_generator:
|
for frame in frame_generator:
|
||||||
result = model.predict(frame, conf=confidence_threshold, iou=iou_threshold)[
|
result = model.predict(frame, conf=confidence_threshold, iou=iou_threshold)[
|
||||||
0
|
0
|
||||||
|
|
@ -123,10 +124,11 @@ def main(
|
||||||
)
|
)
|
||||||
|
|
||||||
sink.write_frame(annotated_frame)
|
sink.write_frame(annotated_frame)
|
||||||
cv2.imshow("frame", annotated_frame)
|
window.show(annotated_frame)
|
||||||
if cv2.waitKey(1) & 0xFF == ord("q"):
|
key = window.wait_key(1)
|
||||||
|
if not window.is_open or key == "q":
|
||||||
break
|
break
|
||||||
cv2.destroyAllWindows()
|
window.close()
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
|
|
|
||||||
|
|
@ -4,10 +4,7 @@
|
||||||
|
|
||||||
## 👋 hello
|
## 👋 hello
|
||||||
|
|
||||||
Practical demonstration on leveraging computer vision for analyzing wait times and
|
Practical demonstration on leveraging computer vision for analyzing wait times and monitoring the duration that objects or individuals spend in predefined areas of video frames. This example project, perfect for retail analytics or traffic management applications.
|
||||||
monitoring the duration that objects or individuals spend in predefined areas of video
|
|
||||||
frames. This example project, perfect for retail analytics or traffic management
|
|
||||||
applications.
|
|
||||||
|
|
||||||
https://github.com/roboflow/supervision/assets/26109316/d051cc8a-dd15-41d4-aa36-d38b86334c39
|
https://github.com/roboflow/supervision/assets/26109316/d051cc8a-dd15-41d4-aa36-d38b86334c39
|
||||||
|
|
||||||
|
|
@ -15,23 +12,25 @@ https://github.com/roboflow/supervision/assets/26109316/d051cc8a-dd15-41d4-aa36-
|
||||||
|
|
||||||
- clone repository and navigate to example directory
|
- clone repository and navigate to example directory
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
git clone --depth 1 -b develop https://github.com/roboflow/supervision.git
|
git clone --depth 1 -b develop https://github.com/roboflow/supervision.git
|
||||||
cd supervision/examples/time_in_zone
|
cd supervision/examples/time_in_zone
|
||||||
```
|
```
|
||||||
|
|
||||||
- setup python environment and activate it [optional]
|
- setup python environment and activate it [optional]
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
uv venv
|
uv venv
|
||||||
source .venv/bin/activate
|
source .venv/bin/activate
|
||||||
```
|
```
|
||||||
|
|
||||||
- install required dependencies
|
- install required dependencies
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
uv pip install -r requirements.txt
|
uv pip install -r requirements.txt
|
||||||
```
|
```
|
||||||
|
|
||||||
|
The three RTSP `*_stream_example.py` scripts display frames from an `InferencePipeline` callback running on a worker thread, so they use OpenCV HighGUI instead of `sv.ImageWindow`. Install `opencv-python` and keep only one OpenCV wheel installed to run those scripts. The file and naive-stream examples use `sv.ImageWindow`, which works regardless of which OpenCV wheel (or none) is installed.
|
||||||
|
|
||||||
## 🛠 scripts
|
## 🛠 scripts
|
||||||
|
|
||||||
|
|
@ -59,9 +58,7 @@ python scripts/download_from_youtube.py \
|
||||||
|
|
||||||
### `stream_from_file`
|
### `stream_from_file`
|
||||||
|
|
||||||
This script allows you to stream video files from a directory. It's an awesome way to
|
This script allows you to stream video files from a directory. It's an awesome way to mock a live video stream for local testing. Video will be streamed in a loop under `rtsp://localhost:8554/live0.stream` URL. This script requires docker to be installed.
|
||||||
mock a live video stream for local testing. Video will be streamed in a loop under
|
|
||||||
`rtsp://localhost:8554/live0.stream` URL. This script requires docker to be installed.
|
|
||||||
|
|
||||||
- `--video_directory`: Directory containing video files to stream.
|
- `--video_directory`: Directory containing video files to stream.
|
||||||
- `--number_of_streams`: Number of video files to stream.
|
- `--number_of_streams`: Number of video files to stream.
|
||||||
|
|
@ -80,10 +77,7 @@ python scripts/stream_from_file.py \
|
||||||
|
|
||||||
### `draw_zones`
|
### `draw_zones`
|
||||||
|
|
||||||
If you want to test zone time in zone analysis on your own video, you can use this
|
If you want to test zone time in zone analysis on your own video, you can use this script to design custom zones and save results as a JSON file. The script will open a window where you can draw polygons on the source image or video file. The polygons will be saved as a JSON file.
|
||||||
script to design custom zones and save results as a JSON file. The script will open a
|
|
||||||
window where you can draw polygons on the source image or video file. The polygons will
|
|
||||||
be saved as a JSON file.
|
|
||||||
|
|
||||||
- `--source_path`: Path to the source image or video file for drawing polygons.
|
- `--source_path`: Path to the source image or video file for drawing polygons.
|
||||||
- `--zone_configuration_path`: Path where the polygon annotations will be saved as a JSON file.
|
- `--zone_configuration_path`: Path where the polygon annotations will be saved as a JSON file.
|
||||||
|
|
@ -324,12 +318,6 @@ python ultralytics_stream_example.py \
|
||||||
|
|
||||||
This demo integrates two main components, each with its own licensing:
|
This demo integrates two main components, each with its own licensing:
|
||||||
|
|
||||||
- ultralytics: The object detection model used in this demo, YOLOv8, is distributed
|
- ultralytics: The object detection model used in this demo, YOLOv8, is distributed under the [AGPL-3.0 license](https://github.com/ultralytics/ultralytics/blob/main/LICENSE). You can find more details about this license here.
|
||||||
under the [AGPL-3.0 license](https://github.com/ultralytics/ultralytics/blob/main/LICENSE).
|
|
||||||
You can find more details about this license here.
|
|
||||||
|
|
||||||
- supervision: The analytics code that powers the zone-based analysis in this demo is
|
- supervision: The analytics code that powers the zone-based analysis in this demo is based on the Supervision library, which is licensed under the [MIT license](https://github.com/roboflow/supervision/blob/develop/LICENSE.md). This makes the Supervision part of the code fully open source and freely usable in your projects.
|
||||||
based on the Supervision library, which is licensed under the
|
|
||||||
[MIT license](https://github.com/roboflow/supervision/blob/develop/LICENSE.md). This
|
|
||||||
makes the Supervision part of the code fully open source and freely usable in your
|
|
||||||
projects.
|
|
||||||
|
|
|
||||||
|
|
@ -1,4 +1,3 @@
|
||||||
import cv2
|
|
||||||
import numpy as np
|
import numpy as np
|
||||||
from inference import get_model
|
from inference import get_model
|
||||||
from utils.general import find_in_list, load_zones_config
|
from utils.general import find_in_list, load_zones_config
|
||||||
|
|
@ -49,6 +48,7 @@ def main(
|
||||||
]
|
]
|
||||||
timers = [FPSBasedTimer(video_info.fps) for _ in zones]
|
timers = [FPSBasedTimer(video_info.fps) for _ in zones]
|
||||||
|
|
||||||
|
window = sv.ImageWindow("Processed Video")
|
||||||
for frame in frames_generator:
|
for frame in frames_generator:
|
||||||
results = model.infer(
|
results = model.infer(
|
||||||
frame, confidence=confidence_threshold, iou_threshold=iou_threshold
|
frame, confidence=confidence_threshold, iou_threshold=iou_threshold
|
||||||
|
|
@ -84,10 +84,11 @@ def main(
|
||||||
custom_color_lookup=custom_color_lookup,
|
custom_color_lookup=custom_color_lookup,
|
||||||
)
|
)
|
||||||
|
|
||||||
cv2.imshow("Processed Video", annotated_frame)
|
window.show(annotated_frame)
|
||||||
if cv2.waitKey(1) & 0xFF == ord("q"):
|
key = window.wait_key(1)
|
||||||
|
if not window.is_open or key == "q":
|
||||||
break
|
break
|
||||||
cv2.destroyAllWindows()
|
window.close()
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
|
|
|
||||||
|
|
@ -1,4 +1,3 @@
|
||||||
import cv2
|
|
||||||
import numpy as np
|
import numpy as np
|
||||||
from inference import get_model
|
from inference import get_model
|
||||||
from utils.general import find_in_list, get_stream_frames_generator, load_zones_config
|
from utils.general import find_in_list, get_stream_frames_generator, load_zones_config
|
||||||
|
|
@ -49,6 +48,7 @@ def main(
|
||||||
]
|
]
|
||||||
timers = [ClockBasedTimer() for _ in zones]
|
timers = [ClockBasedTimer() for _ in zones]
|
||||||
|
|
||||||
|
window = sv.ImageWindow("Processed Video")
|
||||||
for frame in frames_generator:
|
for frame in frames_generator:
|
||||||
fps_monitor.tick()
|
fps_monitor.tick()
|
||||||
fps = fps_monitor.fps
|
fps = fps_monitor.fps
|
||||||
|
|
@ -94,10 +94,11 @@ def main(
|
||||||
custom_color_lookup=custom_color_lookup,
|
custom_color_lookup=custom_color_lookup,
|
||||||
)
|
)
|
||||||
|
|
||||||
cv2.imshow("Processed Video", annotated_frame)
|
window.show(annotated_frame)
|
||||||
if cv2.waitKey(1) & 0xFF == ord("q"):
|
key = window.wait_key(1)
|
||||||
|
if not window.is_open or key == "q":
|
||||||
break
|
break
|
||||||
cv2.destroyAllWindows()
|
window.close()
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
|
|
|
||||||
|
|
@ -15,7 +15,7 @@ LABEL_ANNOTATOR = sv.LabelAnnotator(
|
||||||
|
|
||||||
|
|
||||||
class CustomSink:
|
class CustomSink:
|
||||||
def __init__(self, zone_configuration_path: str, classes: list[int]):
|
def __init__(self, zone_configuration_path: str, classes: list[int]) -> None:
|
||||||
self.classes = classes
|
self.classes = classes
|
||||||
self.tracker = sv.ByteTrack(minimum_matching_threshold=0.5)
|
self.tracker = sv.ByteTrack(minimum_matching_threshold=0.5)
|
||||||
self.fps_monitor = sv.FPSMonitor()
|
self.fps_monitor = sv.FPSMonitor()
|
||||||
|
|
|
||||||
|
|
@ -2,7 +2,6 @@ from __future__ import annotations
|
||||||
|
|
||||||
from enum import Enum
|
from enum import Enum
|
||||||
|
|
||||||
import cv2
|
|
||||||
import numpy as np
|
import numpy as np
|
||||||
from rfdetr import RFDETRBase, RFDETRLarge, RFDETRMedium, RFDETRNano, RFDETRSmall
|
from rfdetr import RFDETRBase, RFDETRLarge, RFDETRMedium, RFDETRNano, RFDETRSmall
|
||||||
from utils.general import find_in_list, load_zones_config
|
from utils.general import find_in_list, load_zones_config
|
||||||
|
|
@ -25,7 +24,7 @@ class ModelSize(Enum):
|
||||||
LARGE = "large"
|
LARGE = "large"
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def list(cls):
|
def list(cls) -> list[str]:
|
||||||
return list(map(lambda c: c.value, cls))
|
return list(map(lambda c: c.value, cls))
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
|
|
@ -44,7 +43,9 @@ class ModelSize(Enum):
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def load_model(checkpoint: ModelSize | str, device: str, resolution: int):
|
def load_model(
|
||||||
|
checkpoint: ModelSize | str, device: str, resolution: int
|
||||||
|
) -> RFDETRBase | RFDETRLarge | RFDETRMedium | RFDETRNano | RFDETRSmall:
|
||||||
checkpoint = ModelSize.from_value(checkpoint)
|
checkpoint = ModelSize.from_value(checkpoint)
|
||||||
|
|
||||||
if checkpoint == ModelSize.NANO:
|
if checkpoint == ModelSize.NANO:
|
||||||
|
|
@ -126,6 +127,7 @@ def main(
|
||||||
]
|
]
|
||||||
timers = [FPSBasedTimer(video_info.fps) for _ in zones]
|
timers = [FPSBasedTimer(video_info.fps) for _ in zones]
|
||||||
|
|
||||||
|
window = sv.ImageWindow("Processed Video")
|
||||||
for frame in frames_generator:
|
for frame in frames_generator:
|
||||||
detections = model.predict(frame, threshold=confidence_threshold)
|
detections = model.predict(frame, threshold=confidence_threshold)
|
||||||
detections = detections[find_in_list(detections.class_id, classes)]
|
detections = detections[find_in_list(detections.class_id, classes)]
|
||||||
|
|
@ -159,10 +161,11 @@ def main(
|
||||||
custom_color_lookup=custom_color_lookup,
|
custom_color_lookup=custom_color_lookup,
|
||||||
)
|
)
|
||||||
|
|
||||||
cv2.imshow("Processed Video", annotated_frame)
|
window.show(annotated_frame)
|
||||||
if cv2.waitKey(1) & 0xFF == ord("q"):
|
key = window.wait_key(1)
|
||||||
|
if not window.is_open or key == "q":
|
||||||
break
|
break
|
||||||
cv2.destroyAllWindows()
|
window.close()
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
|
|
|
||||||
|
|
@ -2,7 +2,6 @@ from __future__ import annotations
|
||||||
|
|
||||||
from enum import Enum
|
from enum import Enum
|
||||||
|
|
||||||
import cv2
|
|
||||||
import numpy as np
|
import numpy as np
|
||||||
from rfdetr import RFDETRBase, RFDETRLarge, RFDETRMedium, RFDETRNano, RFDETRSmall
|
from rfdetr import RFDETRBase, RFDETRLarge, RFDETRMedium, RFDETRNano, RFDETRSmall
|
||||||
from utils.general import find_in_list, get_stream_frames_generator, load_zones_config
|
from utils.general import find_in_list, get_stream_frames_generator, load_zones_config
|
||||||
|
|
@ -25,7 +24,7 @@ class ModelSize(Enum):
|
||||||
LARGE = "large"
|
LARGE = "large"
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def list(cls):
|
def list(cls) -> list[str]:
|
||||||
return list(map(lambda c: c.value, cls))
|
return list(map(lambda c: c.value, cls))
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
|
|
@ -44,7 +43,9 @@ class ModelSize(Enum):
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def load_model(checkpoint: ModelSize | str, device: str, resolution: int):
|
def load_model(
|
||||||
|
checkpoint: ModelSize | str, device: str, resolution: int
|
||||||
|
) -> RFDETRBase | RFDETRLarge | RFDETRMedium | RFDETRNano | RFDETRSmall:
|
||||||
checkpoint = ModelSize.from_value(checkpoint)
|
checkpoint = ModelSize.from_value(checkpoint)
|
||||||
|
|
||||||
if checkpoint == ModelSize.NANO:
|
if checkpoint == ModelSize.NANO:
|
||||||
|
|
@ -126,6 +127,7 @@ def main(
|
||||||
]
|
]
|
||||||
timers = [ClockBasedTimer() for _ in zones]
|
timers = [ClockBasedTimer() for _ in zones]
|
||||||
|
|
||||||
|
window = sv.ImageWindow("Processed Video")
|
||||||
for frame in frames_generator:
|
for frame in frames_generator:
|
||||||
fps_monitor.tick()
|
fps_monitor.tick()
|
||||||
fps = fps_monitor.fps
|
fps = fps_monitor.fps
|
||||||
|
|
@ -169,11 +171,12 @@ def main(
|
||||||
custom_color_lookup=custom_color_lookup,
|
custom_color_lookup=custom_color_lookup,
|
||||||
)
|
)
|
||||||
|
|
||||||
cv2.imshow("Processed Video", annotated_frame)
|
window.show(annotated_frame)
|
||||||
if cv2.waitKey(1) & 0xFF == ord("q"):
|
key = window.wait_key(1)
|
||||||
|
if not window.is_open or key == "q":
|
||||||
break
|
break
|
||||||
|
|
||||||
cv2.destroyAllWindows()
|
window.close()
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
|
|
|
||||||
|
|
@ -21,7 +21,7 @@ class ModelSize(Enum):
|
||||||
LARGE = "large"
|
LARGE = "large"
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def list(cls):
|
def list(cls) -> list[str]:
|
||||||
return [c.value for c in cls]
|
return [c.value for c in cls]
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
|
|
@ -41,7 +41,9 @@ class ModelSize(Enum):
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def load_model(checkpoint: ModelSize | str, device: str, resolution: int):
|
def load_model(
|
||||||
|
checkpoint: ModelSize | str, device: str, resolution: int
|
||||||
|
) -> RFDETRBase | RFDETRLarge | RFDETRMedium | RFDETRNano | RFDETRSmall:
|
||||||
checkpoint = ModelSize.from_value(checkpoint)
|
checkpoint = ModelSize.from_value(checkpoint)
|
||||||
if checkpoint == ModelSize.NANO:
|
if checkpoint == ModelSize.NANO:
|
||||||
return RFDETRNano(device=device, resolution=resolution)
|
return RFDETRNano(device=device, resolution=resolution)
|
||||||
|
|
@ -77,7 +79,7 @@ LABEL_ANNOTATOR = sv.LabelAnnotator(
|
||||||
|
|
||||||
|
|
||||||
class CustomSink:
|
class CustomSink:
|
||||||
def __init__(self, zone_configuration_path: str, classes: list[int]):
|
def __init__(self, zone_configuration_path: str, classes: list[int]) -> None:
|
||||||
self.classes = classes
|
self.classes = classes
|
||||||
self.tracker = sv.ByteTrack(minimum_matching_threshold=0.8)
|
self.tracker = sv.ByteTrack(minimum_matching_threshold=0.8)
|
||||||
self.fps_monitor = sv.FPSMonitor()
|
self.fps_monitor = sv.FPSMonitor()
|
||||||
|
|
|
||||||
|
|
@ -1,5 +1,3 @@
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import os
|
import os
|
||||||
import sys
|
import sys
|
||||||
from typing import Any
|
from typing import Any
|
||||||
|
|
|
||||||
|
|
@ -1,8 +1,5 @@
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import json
|
import json
|
||||||
import os
|
import os
|
||||||
from typing import Any
|
|
||||||
|
|
||||||
import cv2
|
import cv2
|
||||||
import numpy as np
|
import numpy as np
|
||||||
|
|
@ -10,11 +7,10 @@ from jsonargparse import auto_cli
|
||||||
|
|
||||||
import supervision as sv
|
import supervision as sv
|
||||||
|
|
||||||
KEY_ENTER = 13
|
KEY_ENTER = {"Return", "KP_Enter"}
|
||||||
KEY_NEWLINE = 10
|
KEY_ESCAPE = "Escape"
|
||||||
KEY_ESCAPE = 27
|
KEY_QUIT = "q"
|
||||||
KEY_QUIT = ord("q")
|
KEY_SAVE = "s"
|
||||||
KEY_SAVE = ord("s")
|
|
||||||
|
|
||||||
THICKNESS = 2
|
THICKNESS = 2
|
||||||
COLORS = sv.ColorPalette.DEFAULT
|
COLORS = sv.ColorPalette.DEFAULT
|
||||||
|
|
@ -37,15 +33,17 @@ def resolve_source(source_path: str) -> np.ndarray | None:
|
||||||
return frame
|
return frame
|
||||||
|
|
||||||
|
|
||||||
def mouse_event(event: int, x: int, y: int, flags: int, param: Any) -> None:
|
def mouse_event(x: int, y: int, event_type: str) -> None:
|
||||||
global current_mouse_position
|
global current_mouse_position
|
||||||
if event == cv2.EVENT_MOUSEMOVE:
|
if event_type == "move":
|
||||||
current_mouse_position = (x, y)
|
current_mouse_position = (x, y)
|
||||||
elif event == cv2.EVENT_LBUTTONDOWN:
|
elif event_type == "down":
|
||||||
POLYGONS[-1].append((x, y))
|
POLYGONS[-1].append((x, y))
|
||||||
|
|
||||||
|
|
||||||
def redraw(image: np.ndarray, original_image: np.ndarray) -> None:
|
def redraw(
|
||||||
|
image: np.ndarray, original_image: np.ndarray, window: sv.ImageWindow
|
||||||
|
) -> None:
|
||||||
global POLYGONS, current_mouse_position
|
global POLYGONS, current_mouse_position
|
||||||
image[:] = original_image.copy()
|
image[:] = original_image.copy()
|
||||||
for idx, polygon in enumerate(POLYGONS):
|
for idx, polygon in enumerate(POLYGONS):
|
||||||
|
|
@ -80,10 +78,12 @@ def redraw(image: np.ndarray, original_image: np.ndarray) -> None:
|
||||||
color=color,
|
color=color,
|
||||||
thickness=THICKNESS,
|
thickness=THICKNESS,
|
||||||
)
|
)
|
||||||
cv2.imshow(WINDOW_NAME, image)
|
window.show(image)
|
||||||
|
|
||||||
|
|
||||||
def close_and_finalize_polygon(image: np.ndarray, original_image: np.ndarray) -> None:
|
def close_and_finalize_polygon(
|
||||||
|
image: np.ndarray, original_image: np.ndarray, window: sv.ImageWindow
|
||||||
|
) -> None:
|
||||||
if len(POLYGONS[-1]) > 2:
|
if len(POLYGONS[-1]) > 2:
|
||||||
cv2.line(
|
cv2.line(
|
||||||
img=image,
|
img=image,
|
||||||
|
|
@ -95,7 +95,7 @@ def close_and_finalize_polygon(image: np.ndarray, original_image: np.ndarray) ->
|
||||||
POLYGONS.append([])
|
POLYGONS.append([])
|
||||||
image[:] = original_image.copy()
|
image[:] = original_image.copy()
|
||||||
redraw_polygons(image)
|
redraw_polygons(image)
|
||||||
cv2.imshow(WINDOW_NAME, image)
|
window.show(image)
|
||||||
|
|
||||||
|
|
||||||
def redraw_polygons(image: np.ndarray) -> None:
|
def redraw_polygons(image: np.ndarray) -> None:
|
||||||
|
|
@ -119,7 +119,9 @@ def redraw_polygons(image: np.ndarray) -> None:
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def save_polygons_to_json(polygons, target_path):
|
def save_polygons_to_json(
|
||||||
|
polygons: list[list[tuple[int, int]]], target_path: str | os.PathLike[str]
|
||||||
|
) -> None:
|
||||||
data_to_save = polygons if polygons[-1] else polygons[:-1]
|
data_to_save = polygons if polygons[-1] else polygons[:-1]
|
||||||
with open(target_path, "w") as f:
|
with open(target_path, "w") as f:
|
||||||
json.dump(data_to_save, f)
|
json.dump(data_to_save, f)
|
||||||
|
|
@ -140,13 +142,16 @@ def main(source_path: str, zone_configuration_path: str) -> None:
|
||||||
return
|
return
|
||||||
|
|
||||||
image = original_image.copy()
|
image = original_image.copy()
|
||||||
cv2.imshow(WINDOW_NAME, image)
|
window = sv.ImageWindow(WINDOW_NAME)
|
||||||
cv2.setMouseCallback(WINDOW_NAME, mouse_event, image)
|
window.set_mouse_callback(mouse_event)
|
||||||
|
window.show(image)
|
||||||
|
|
||||||
while True:
|
while True:
|
||||||
key = cv2.waitKey(1) & 0xFF
|
key = window.wait_key(1)
|
||||||
if key == KEY_ENTER or key == KEY_NEWLINE:
|
if not window.is_open:
|
||||||
close_and_finalize_polygon(image, original_image)
|
break
|
||||||
|
if key in KEY_ENTER:
|
||||||
|
close_and_finalize_polygon(image, original_image, window)
|
||||||
elif key == KEY_ESCAPE:
|
elif key == KEY_ESCAPE:
|
||||||
POLYGONS[-1] = []
|
POLYGONS[-1] = []
|
||||||
current_mouse_position = None
|
current_mouse_position = None
|
||||||
|
|
@ -154,11 +159,11 @@ def main(source_path: str, zone_configuration_path: str) -> None:
|
||||||
save_polygons_to_json(POLYGONS, zone_configuration_path)
|
save_polygons_to_json(POLYGONS, zone_configuration_path)
|
||||||
print(f"Polygons saved to {zone_configuration_path}")
|
print(f"Polygons saved to {zone_configuration_path}")
|
||||||
break
|
break
|
||||||
redraw(image, original_image)
|
redraw(image, original_image, window)
|
||||||
if key == KEY_QUIT:
|
if key == KEY_QUIT:
|
||||||
break
|
break
|
||||||
|
|
||||||
cv2.destroyAllWindows()
|
window.close()
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
|
|
|
||||||
|
|
@ -1,4 +1,3 @@
|
||||||
import cv2
|
|
||||||
import numpy as np
|
import numpy as np
|
||||||
from ultralytics import YOLO
|
from ultralytics import YOLO
|
||||||
from utils.general import find_in_list, load_zones_config
|
from utils.general import find_in_list, load_zones_config
|
||||||
|
|
@ -49,6 +48,7 @@ def main(
|
||||||
]
|
]
|
||||||
timers = [FPSBasedTimer(video_info.fps) for _ in zones]
|
timers = [FPSBasedTimer(video_info.fps) for _ in zones]
|
||||||
|
|
||||||
|
window = sv.ImageWindow("Processed Video")
|
||||||
for frame in frames_generator:
|
for frame in frames_generator:
|
||||||
results = model(
|
results = model(
|
||||||
frame,
|
frame,
|
||||||
|
|
@ -88,10 +88,11 @@ def main(
|
||||||
custom_color_lookup=custom_color_lookup,
|
custom_color_lookup=custom_color_lookup,
|
||||||
)
|
)
|
||||||
|
|
||||||
cv2.imshow("Processed Video", annotated_frame)
|
window.show(annotated_frame)
|
||||||
if cv2.waitKey(1) & 0xFF == ord("q"):
|
key = window.wait_key(1)
|
||||||
|
if not window.is_open or key == "q":
|
||||||
break
|
break
|
||||||
cv2.destroyAllWindows()
|
window.close()
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
|
|
|
||||||
|
|
@ -1,4 +1,3 @@
|
||||||
import cv2
|
|
||||||
import numpy as np
|
import numpy as np
|
||||||
from ultralytics import YOLO
|
from ultralytics import YOLO
|
||||||
from utils.general import find_in_list, get_stream_frames_generator, load_zones_config
|
from utils.general import find_in_list, get_stream_frames_generator, load_zones_config
|
||||||
|
|
@ -49,6 +48,7 @@ def main(
|
||||||
]
|
]
|
||||||
timers = [ClockBasedTimer() for _ in zones]
|
timers = [ClockBasedTimer() for _ in zones]
|
||||||
|
|
||||||
|
window = sv.ImageWindow("Processed Video")
|
||||||
for frame in frames_generator:
|
for frame in frames_generator:
|
||||||
fps_monitor.tick()
|
fps_monitor.tick()
|
||||||
fps = fps_monitor.fps
|
fps = fps_monitor.fps
|
||||||
|
|
@ -98,10 +98,11 @@ def main(
|
||||||
custom_color_lookup=custom_color_lookup,
|
custom_color_lookup=custom_color_lookup,
|
||||||
)
|
)
|
||||||
|
|
||||||
cv2.imshow("Processed Video", annotated_frame)
|
window.show(annotated_frame)
|
||||||
if cv2.waitKey(1) & 0xFF == ord("q"):
|
key = window.wait_key(1)
|
||||||
|
if not window.is_open or key == "q":
|
||||||
break
|
break
|
||||||
cv2.destroyAllWindows()
|
window.close()
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
|
|
|
||||||
|
|
@ -1,5 +1,3 @@
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import cv2
|
import cv2
|
||||||
import numpy as np
|
import numpy as np
|
||||||
from inference import InferencePipeline
|
from inference import InferencePipeline
|
||||||
|
|
@ -18,7 +16,7 @@ LABEL_ANNOTATOR = sv.LabelAnnotator(
|
||||||
|
|
||||||
|
|
||||||
class CustomSink:
|
class CustomSink:
|
||||||
def __init__(self, zone_configuration_path: str, classes: list[int]):
|
def __init__(self, zone_configuration_path: str, classes: list[int]) -> None:
|
||||||
self.classes = classes
|
self.classes = classes
|
||||||
self.tracker = sv.ByteTrack(minimum_matching_threshold=0.8)
|
self.tracker = sv.ByteTrack(minimum_matching_threshold=0.8)
|
||||||
self.fps_monitor = sv.FPSMonitor()
|
self.fps_monitor = sv.FPSMonitor()
|
||||||
|
|
|
||||||
|
|
@ -2,107 +2,82 @@
|
||||||
|
|
||||||
## 👋 hello
|
## 👋 hello
|
||||||
|
|
||||||
This script provides functionality for processing videos using YOLOv8 for object
|
This script provides functionality for processing videos using YOLOv8 for object detection and Supervision for tracking and annotation.
|
||||||
detection and Supervision for tracking and annotation.
|
|
||||||
|
|
||||||
## 💻 install
|
## 💻 install
|
||||||
|
|
||||||
- clone repository and navigate to example directory
|
- clone repository and navigate to example directory
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
git clone --depth 1 -b develop https://github.com/roboflow/supervision.git
|
git clone --depth 1 -b develop https://github.com/roboflow/supervision.git
|
||||||
cd supervision/examples/tracking
|
cd supervision/examples/tracking
|
||||||
```
|
```
|
||||||
|
|
||||||
- setup python environment and activate it [optional]
|
- setup python environment and activate it [optional]
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
uv venv
|
uv venv
|
||||||
source .venv/bin/activate
|
source .venv/bin/activate
|
||||||
```
|
```
|
||||||
|
|
||||||
- install required dependencies
|
- install required dependencies
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
uv pip install -r requirements.txt
|
uv pip install -r requirements.txt
|
||||||
```
|
```
|
||||||
|
|
||||||
## 🛠️ script arguments
|
## 🛠️ script arguments
|
||||||
|
|
||||||
- ultralytics
|
- ultralytics
|
||||||
|
|
||||||
- `--source_weights_path`: Required. Specifies the path to the YOLO model's weights
|
- `--source_weights_path`: Required. Specifies the path to the YOLO model's weights file, which is essential for the object detection process. This file contains the data that the model uses to identify objects in the video.
|
||||||
file, which is essential for the object detection process. This file contains the data
|
|
||||||
that the model uses to identify objects in the video.
|
|
||||||
|
|
||||||
- `--source_video_path`: Required. The path to the source video file to be processed.
|
- `--source_video_path`: Required. The path to the source video file to be processed. This is the video on which object detection and annotation will be performed.
|
||||||
This is the video on which object detection and annotation will be performed.
|
|
||||||
|
|
||||||
- `--target_video_path`: Required. The path where the processed video, with annotations
|
- `--target_video_path`: Required. The path where the processed video, with annotations added, will be saved. This is your output video file.
|
||||||
added, will be saved. This is your output video file.
|
|
||||||
|
|
||||||
- `--confidence_threshold` (optional): Sets the confidence level at which the model
|
- `--confidence_threshold` (optional): Sets the confidence level at which the model identifies objects in the video. Default is `0.3`. A higher threshold makes the model more selective, while a lower threshold makes it more inclusive in identifying objects.
|
||||||
identifies objects in the video. Default is `0.3`. A higher threshold makes the model
|
|
||||||
more selective, while a lower threshold makes it more inclusive in identifying objects.
|
|
||||||
|
|
||||||
- `--iou_threshold` (optional): Specifies the IOU (Intersection Over Union) threshold
|
- `--iou_threshold` (optional): Specifies the IOU (Intersection Over Union) threshold for the model, defaulting to `0.7`. This parameter helps in differentiating between distinct objects, especially in crowded scenes.
|
||||||
for the model, defaulting to `0.7`. This parameter helps in differentiating between
|
|
||||||
distinct objects, especially in crowded scenes.
|
|
||||||
|
|
||||||
- inference
|
- inference
|
||||||
|
|
||||||
- `--roboflow_api_key` (optional): The API key for Roboflow services. If not provided
|
- `--roboflow_api_key` (optional): The API key for Roboflow services. If not provided directly, the script tries to fetch it from the `ROBOFLOW_API_KEY` environment variable. Follow [this guide](https://docs.roboflow.com/api-reference/authentication#retrieve-an-api-key) to acquire your `API KEY`.
|
||||||
directly, the script tries to fetch it from the `ROBOFLOW_API_KEY` environment
|
|
||||||
variable. Follow [this guide](https://docs.roboflow.com/api-reference/authentication#retrieve-an-api-key)
|
|
||||||
to acquire your `API KEY`.
|
|
||||||
|
|
||||||
- `--model_id` (optional): Designates the Roboflow model ID to be used. The default
|
- `--model_id` (optional): Designates the Roboflow model ID to be used. The default value is `"yolov8x-1280"`.
|
||||||
value is `"yolov8x-1280"`.
|
|
||||||
|
|
||||||
- `--source_video_path`: Required. The path to the source video file to be processed.
|
- `--source_video_path`: Required. The path to the source video file to be processed. This is the video on which object detection and annotation will be performed.
|
||||||
This is the video on which object detection and annotation will be performed.
|
|
||||||
|
|
||||||
- `--target_video_path`: Required. The path where the processed video, with annotations
|
- `--target_video_path`: Required. The path where the processed video, with annotations added, will be saved. This is your output video file.
|
||||||
added, will be saved. This is your output video file.
|
|
||||||
|
|
||||||
- `--confidence_threshold` (optional): Sets the confidence level at which the model
|
- `--confidence_threshold` (optional): Sets the confidence level at which the model identifies objects in the video. Default is `0.3`. A higher threshold makes the model more selective, while a lower threshold makes it more inclusive in identifying objects.
|
||||||
identifies objects in the video. Default is `0.3`. A higher threshold makes the model
|
|
||||||
more selective, while a lower threshold makes it more inclusive in identifying objects.
|
|
||||||
|
|
||||||
- `--iou_threshold` (optional): Specifies the IOU (Intersection Over Union) threshold
|
- `--iou_threshold` (optional): Specifies the IOU (Intersection Over Union) threshold for the model, defaulting to `0.7`. This parameter helps in differentiating between distinct objects, especially in crowded scenes.
|
||||||
for the model, defaulting to `0.7`. This parameter helps in differentiating between
|
|
||||||
distinct objects, especially in crowded scenes.
|
|
||||||
|
|
||||||
## ⚙️ run
|
## ⚙️ run
|
||||||
|
|
||||||
- inference
|
- inference
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
python inference_example.py \
|
python inference_example.py \
|
||||||
--roboflow_api_key "ROBOFLOW_API_KEY" \
|
--roboflow_api_key "ROBOFLOW_API_KEY" \
|
||||||
--source_video_path input.mp4 \
|
--source_video_path input.mp4 \
|
||||||
--target_video_path tracking_result.mp4
|
--target_video_path tracking_result.mp4
|
||||||
```
|
```
|
||||||
|
|
||||||
- ultralytics
|
- ultralytics
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
python ultralytics_example.py \
|
python ultralytics_example.py \
|
||||||
--source_weights_path yolov8s.pt \
|
--source_weights_path yolov8s.pt \
|
||||||
--source_video_path input.mp4 \
|
--source_video_path input.mp4 \
|
||||||
--target_video_path tracking_result.mp4
|
--target_video_path tracking_result.mp4
|
||||||
```
|
```
|
||||||
|
|
||||||
## © license
|
## © license
|
||||||
|
|
||||||
This demo integrates two main components, each with its own licensing:
|
This demo integrates two main components, each with its own licensing:
|
||||||
|
|
||||||
- ultralytics: The object detection model used in this demo, YOLOv8, is distributed
|
- ultralytics: The object detection model used in this demo, YOLOv8, is distributed under the [AGPL-3.0 license](https://github.com/ultralytics/ultralytics/blob/main/LICENSE). You can find more details about this license here.
|
||||||
under the [AGPL-3.0 license](https://github.com/ultralytics/ultralytics/blob/main/LICENSE).
|
|
||||||
You can find more details about this license here.
|
|
||||||
|
|
||||||
- supervision: The analytics code that powers the zone-based analysis in this demo is
|
- supervision: The analytics code that powers the zone-based analysis in this demo is based on the Supervision library, which is licensed under the [MIT license](https://github.com/roboflow/supervision/blob/develop/LICENSE.md). This makes the Supervision part of the code fully open source and freely usable in your projects.
|
||||||
based on the Supervision library, which is licensed under the
|
|
||||||
[MIT license](https://github.com/roboflow/supervision/blob/develop/LICENSE.md). This
|
|
||||||
makes the Supervision part of the code fully open source and freely usable in your
|
|
||||||
projects.
|
|
||||||
|
|
|
||||||
|
|
@ -2,9 +2,7 @@
|
||||||
|
|
||||||
## 👋 hello
|
## 👋 hello
|
||||||
|
|
||||||
This script performs traffic flow analysis using YOLOv8, an object-detection method and
|
This script performs traffic flow analysis using YOLOv8, an object-detection method and ByteTrack, a simple yet effective online multi-object tracking method. It uses the supervision package for multiple tasks such as tracking, annotations, etc.
|
||||||
ByteTrack, a simple yet effective online multi-object tracking method. It uses the
|
|
||||||
supervision package for multiple tasks such as tracking, annotations, etc.
|
|
||||||
|
|
||||||
https://github.com/roboflow/supervision/assets/26109316/c9436828-9fbf-4c25-ae8c-60e9c81b3900
|
https://github.com/roboflow/supervision/assets/26109316/c9436828-9fbf-4c25-ae8c-60e9c81b3900
|
||||||
|
|
||||||
|
|
@ -12,112 +10,86 @@ https://github.com/roboflow/supervision/assets/26109316/c9436828-9fbf-4c25-ae8c-
|
||||||
|
|
||||||
- clone repository and navigate to example directory
|
- clone repository and navigate to example directory
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
git clone --depth 1 -b develop https://github.com/roboflow/supervision.git
|
git clone --depth 1 -b develop https://github.com/roboflow/supervision.git
|
||||||
cd supervision/examples/traffic_analysis
|
cd supervision/examples/traffic_analysis
|
||||||
```
|
```
|
||||||
|
|
||||||
- setup python environment and activate it [optional]
|
- setup python environment and activate it [optional]
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
uv venv
|
uv venv
|
||||||
source .venv/bin/activate
|
source .venv/bin/activate
|
||||||
```
|
```
|
||||||
|
|
||||||
- install required dependencies
|
- install required dependencies
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
uv pip install -r requirements.txt
|
uv pip install -r requirements.txt
|
||||||
```
|
```
|
||||||
|
|
||||||
- download `traffic_analysis.pt` and `traffic_analysis.mov` files
|
- download `traffic_analysis.pt` and `traffic_analysis.mov` files
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
./setup.sh
|
./setup.sh
|
||||||
```
|
```
|
||||||
|
|
||||||
## 🛠️ script arguments
|
## 🛠️ script arguments
|
||||||
|
|
||||||
- ultralytics
|
- ultralytics
|
||||||
|
|
||||||
- `--source_weights_path`: Required. Specifies the path to the YOLO model's weights
|
- `--source_weights_path`: Required. Specifies the path to the YOLO model's weights file, which is essential for the object detection process. This file contains the data that the model uses to identify objects in the video.
|
||||||
file, which is essential for the object detection process. This file contains the
|
|
||||||
data that the model uses to identify objects in the video.
|
|
||||||
|
|
||||||
- `--source_video_path`: Required. The path to the source video file that will be
|
- `--source_video_path`: Required. The path to the source video file that will be analyzed. This is the input video on which traffic flow analysis will be performed.
|
||||||
analyzed. This is the input video on which traffic flow analysis will be performed.
|
|
||||||
|
|
||||||
- `--target_video_path` (optional): The path to save the output video with
|
- `--target_video_path` (optional): The path to save the output video with annotations. If not specified, the processed video will be displayed in real-time without being saved.
|
||||||
annotations. If not specified, the processed video will be displayed in real-time
|
|
||||||
without being saved.
|
|
||||||
|
|
||||||
- `--confidence_threshold` (optional): Sets the confidence threshold for the YOLO
|
- `--confidence_threshold` (optional): Sets the confidence threshold for the YOLO model to filter detections. Default is `0.3`. This determines how confident the model should be to recognize an object in the video.
|
||||||
model to filter detections. Default is `0.3`. This determines how confident the
|
|
||||||
model should be to recognize an object in the video.
|
|
||||||
|
|
||||||
- `--iou_threshold` (optional): Specifies the IOU (Intersection Over Union) threshold
|
- `--iou_threshold` (optional): Specifies the IOU (Intersection Over Union) threshold for the model. Default is 0.7. This value is used to manage object detection accuracy, particularly in distinguishing between different objects.
|
||||||
for the model. Default is 0.7. This value is used to manage object detection
|
|
||||||
accuracy, particularly in distinguishing between different objects.
|
|
||||||
|
|
||||||
- inference
|
- inference
|
||||||
|
|
||||||
- `--roboflow_api_key` (optional): The API key for Roboflow services. If not provided
|
- `--roboflow_api_key` (optional): The API key for Roboflow services. If not provided directly, the script tries to fetch it from the `ROBOFLOW_API_KEY` environment variable. Follow [this guide](https://docs.roboflow.com/api-reference/authentication#retrieve-an-api-key) to acquire your `API KEY`.
|
||||||
directly, the script tries to fetch it from the `ROBOFLOW_API_KEY` environment
|
|
||||||
variable. Follow [this guide](https://docs.roboflow.com/api-reference/authentication#retrieve-an-api-key)
|
|
||||||
to acquire your `API KEY`.
|
|
||||||
|
|
||||||
- `--model_id` (optional): Designates the Roboflow model ID to be used. The default
|
- `--model_id` (optional): Designates the Roboflow model ID to be used. The default value is `"vehicle-count-in-drone-video/6"`.
|
||||||
value is `"vehicle-count-in-drone-video/6"`.
|
|
||||||
|
|
||||||
- `--source_video_path`: Required. The path to the source video file that will be
|
- `--source_video_path`: Required. The path to the source video file that will be analyzed. This is the input video on which traffic flow analysis will be performed.
|
||||||
analyzed. This is the input video on which traffic flow analysis will be performed.
|
|
||||||
|
|
||||||
- `--target_video_path` (optional): The path to save the output video with
|
- `--target_video_path` (optional): The path to save the output video with annotations. If not specified, the processed video will be displayed in real-time without being saved.
|
||||||
annotations. If not specified, the processed video will be displayed in real-time
|
|
||||||
without being saved.
|
|
||||||
|
|
||||||
- `--confidence_threshold` (optional): Sets the confidence threshold for the YOLO
|
- `--confidence_threshold` (optional): Sets the confidence threshold for the YOLO model to filter detections. Default is `0.3`. This determines how confident the model should be to recognize an object in the video.
|
||||||
model to filter detections. Default is `0.3`. This determines how confident the
|
|
||||||
model should be to recognize an object in the video.
|
|
||||||
|
|
||||||
- `--iou_threshold` (optional): Specifies the IOU (Intersection Over Union) threshold
|
- `--iou_threshold` (optional): Specifies the IOU (Intersection Over Union) threshold for the model. Default is 0.7. This value is used to manage object detection accuracy, particularly in distinguishing between different objects.
|
||||||
for the model. Default is 0.7. This value is used to manage object detection
|
|
||||||
accuracy, particularly in distinguishing between different objects.
|
|
||||||
|
|
||||||
## ⚙️ run
|
## ⚙️ run
|
||||||
|
|
||||||
- ultralytics
|
- ultralytics
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
python ultralytics_example.py \
|
python ultralytics_example.py \
|
||||||
--source_weights_path data/traffic_analysis.pt \
|
--source_weights_path data/traffic_analysis.pt \
|
||||||
--source_video_path data/traffic_analysis.mov \
|
--source_video_path data/traffic_analysis.mov \
|
||||||
--confidence_threshold 0.3 \
|
--confidence_threshold 0.3 \
|
||||||
--iou_threshold 0.5 \
|
--iou_threshold 0.5 \
|
||||||
--target_video_path data/traffic_analysis_result.mov
|
--target_video_path data/traffic_analysis_result.mov
|
||||||
```
|
```
|
||||||
|
|
||||||
- inference
|
- inference
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
python inference_example.py \
|
python inference_example.py \
|
||||||
--roboflow_api_key "ROBOFLOW_API_KEY" \
|
--roboflow_api_key "ROBOFLOW_API_KEY" \
|
||||||
--source_video_path data/traffic_analysis.mov \
|
--source_video_path data/traffic_analysis.mov \
|
||||||
--confidence_threshold 0.3 \
|
--confidence_threshold 0.3 \
|
||||||
--iou_threshold 0.5 \
|
--iou_threshold 0.5 \
|
||||||
--target_video_path data/traffic_analysis_result.mov
|
--target_video_path data/traffic_analysis_result.mov
|
||||||
```
|
```
|
||||||
|
|
||||||
## © license
|
## © license
|
||||||
|
|
||||||
This demo integrates two main components, each with its own licensing:
|
This demo integrates two main components, each with its own licensing:
|
||||||
|
|
||||||
- ultralytics: The object detection model used in this demo, YOLOv8, is distributed
|
- ultralytics: The object detection model used in this demo, YOLOv8, is distributed under the [AGPL-3.0 license](https://github.com/ultralytics/ultralytics/blob/main/LICENSE). You can find more details about this license here.
|
||||||
under the [AGPL-3.0 license](https://github.com/ultralytics/ultralytics/blob/main/LICENSE).
|
|
||||||
You can find more details about this license here.
|
|
||||||
|
|
||||||
- supervision: The analytics code that powers the zone-based analysis in this demo is
|
- supervision: The analytics code that powers the zone-based analysis in this demo is based on the Supervision library, which is licensed under the [MIT license](https://github.com/roboflow/supervision/blob/develop/LICENSE.md). This makes the Supervision part of the code fully open source and freely usable in your projects.
|
||||||
based on the Supervision library, which is licensed under the
|
|
||||||
[MIT license](https://github.com/roboflow/supervision/blob/develop/LICENSE.md). This
|
|
||||||
makes the Supervision part of the code fully open source and freely usable in your
|
|
||||||
projects.
|
|
||||||
|
|
|
||||||
|
|
@ -1,9 +1,6 @@
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import os
|
import os
|
||||||
from collections.abc import Iterable
|
from collections.abc import Iterable
|
||||||
|
|
||||||
import cv2
|
|
||||||
import numpy as np
|
import numpy as np
|
||||||
from inference.models.utils import get_roboflow_model
|
from inference.models.utils import get_roboflow_model
|
||||||
from tqdm import tqdm
|
from tqdm import tqdm
|
||||||
|
|
@ -103,7 +100,7 @@ class VideoProcessor:
|
||||||
)
|
)
|
||||||
self.detections_manager = DetectionsManager()
|
self.detections_manager = DetectionsManager()
|
||||||
|
|
||||||
def process_video(self):
|
def process_video(self) -> None:
|
||||||
frame_generator = sv.get_video_frames_generator(
|
frame_generator = sv.get_video_frames_generator(
|
||||||
source_path=self.source_video_path
|
source_path=self.source_video_path
|
||||||
)
|
)
|
||||||
|
|
@ -114,12 +111,14 @@ class VideoProcessor:
|
||||||
annotated_frame = self.process_frame(frame)
|
annotated_frame = self.process_frame(frame)
|
||||||
sink.write_frame(annotated_frame)
|
sink.write_frame(annotated_frame)
|
||||||
else:
|
else:
|
||||||
|
window = sv.ImageWindow("Processed Video")
|
||||||
for frame in tqdm(frame_generator, total=self.video_info.total_frames):
|
for frame in tqdm(frame_generator, total=self.video_info.total_frames):
|
||||||
annotated_frame = self.process_frame(frame)
|
annotated_frame = self.process_frame(frame)
|
||||||
cv2.imshow("Processed Video", annotated_frame)
|
window.show(annotated_frame)
|
||||||
if cv2.waitKey(1) & 0xFF == ord("q"):
|
key = window.wait_key(1)
|
||||||
|
if not window.is_open or key == "q":
|
||||||
break
|
break
|
||||||
cv2.destroyAllWindows()
|
window.close()
|
||||||
|
|
||||||
def annotate_frame(
|
def annotate_frame(
|
||||||
self, frame: np.ndarray, detections: sv.Detections
|
self, frame: np.ndarray, detections: sv.Detections
|
||||||
|
|
|
||||||
|
|
@ -1,8 +1,5 @@
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
from collections.abc import Iterable
|
from collections.abc import Iterable
|
||||||
|
|
||||||
import cv2
|
|
||||||
import numpy as np
|
import numpy as np
|
||||||
from tqdm import tqdm
|
from tqdm import tqdm
|
||||||
from ultralytics import YOLO
|
from ultralytics import YOLO
|
||||||
|
|
@ -100,7 +97,7 @@ class VideoProcessor:
|
||||||
)
|
)
|
||||||
self.detections_manager = DetectionsManager()
|
self.detections_manager = DetectionsManager()
|
||||||
|
|
||||||
def process_video(self):
|
def process_video(self) -> None:
|
||||||
frame_generator = sv.get_video_frames_generator(
|
frame_generator = sv.get_video_frames_generator(
|
||||||
source_path=self.source_video_path
|
source_path=self.source_video_path
|
||||||
)
|
)
|
||||||
|
|
@ -111,12 +108,14 @@ class VideoProcessor:
|
||||||
annotated_frame = self.process_frame(frame)
|
annotated_frame = self.process_frame(frame)
|
||||||
sink.write_frame(annotated_frame)
|
sink.write_frame(annotated_frame)
|
||||||
else:
|
else:
|
||||||
|
window = sv.ImageWindow("Processed Video")
|
||||||
for frame in tqdm(frame_generator, total=self.video_info.total_frames):
|
for frame in tqdm(frame_generator, total=self.video_info.total_frames):
|
||||||
annotated_frame = self.process_frame(frame)
|
annotated_frame = self.process_frame(frame)
|
||||||
cv2.imshow("Processed Video", annotated_frame)
|
window.show(annotated_frame)
|
||||||
if cv2.waitKey(1) & 0xFF == ord("q"):
|
key = window.wait_key(1)
|
||||||
|
if not window.is_open or key == "q":
|
||||||
break
|
break
|
||||||
cv2.destroyAllWindows()
|
window.close()
|
||||||
|
|
||||||
def annotate_frame(
|
def annotate_frame(
|
||||||
self, frame: np.ndarray, detections: sv.Detections
|
self, frame: np.ndarray, detections: sv.Detections
|
||||||
|
|
|
||||||
15
mkdocs.yml
15
mkdocs.yml
|
|
@ -20,7 +20,7 @@ extra:
|
||||||
link: https://discord.gg/GbfgXGJ8Bk
|
link: https://discord.gg/GbfgXGJ8Bk
|
||||||
analytics:
|
analytics:
|
||||||
provider: google
|
provider: google
|
||||||
property: G-P7ZG0Y19G5
|
property: G-SEKT4K1EWR
|
||||||
version:
|
version:
|
||||||
provider: mike
|
provider: mike
|
||||||
|
|
||||||
|
|
@ -42,6 +42,8 @@ nav:
|
||||||
- Process Datasets: how_to/process_datasets.md
|
- Process Datasets: how_to/process_datasets.md
|
||||||
- Benchmark a Model: how_to/benchmark_a_model.md
|
- Benchmark a Model: how_to/benchmark_a_model.md
|
||||||
- Count in Zone: how_to/count_in_zone.md
|
- Count in Zone: how_to/count_in_zone.md
|
||||||
|
- Use Compact Masks: how_to/use_compact_masks.md
|
||||||
|
- OpenCV Migration: how_to/opencv_migration.md
|
||||||
- Reference:
|
- Reference:
|
||||||
- Detection and Segmentation:
|
- Detection and Segmentation:
|
||||||
- Core: detection/core.md
|
- Core: detection/core.md
|
||||||
|
|
@ -52,7 +54,7 @@ nav:
|
||||||
- Boxes: detection/utils/boxes.md
|
- Boxes: detection/utils/boxes.md
|
||||||
- Masks: detection/utils/masks.md
|
- Masks: detection/utils/masks.md
|
||||||
- Polygons: detection/utils/polygons.md
|
- Polygons: detection/utils/polygons.md
|
||||||
- VLMs: detection/utils/vlms.md
|
- VLM Utils: detection/utils/vlms.md
|
||||||
- Keypoint Detection:
|
- Keypoint Detection:
|
||||||
- Core: keypoint/core.md
|
- Core: keypoint/core.md
|
||||||
- Annotators: keypoint/annotators.md
|
- Annotators: keypoint/annotators.md
|
||||||
|
|
@ -76,8 +78,10 @@ nav:
|
||||||
- Common Values: metrics/common_values.md
|
- Common Values: metrics/common_values.md
|
||||||
- Legacy Metrics: detection/metrics.md
|
- Legacy Metrics: detection/metrics.md
|
||||||
- Utils:
|
- Utils:
|
||||||
|
- Conversion: utils/conversion.md
|
||||||
- Video: utils/video.md
|
- Video: utils/video.md
|
||||||
- Image: utils/image.md
|
- Image: utils/image.md
|
||||||
|
- Image Window: utils/image_window.md
|
||||||
- Iterables: utils/iterables.md
|
- Iterables: utils/iterables.md
|
||||||
- Notebook: utils/notebook.md
|
- Notebook: utils/notebook.md
|
||||||
- File: utils/file.md
|
- File: utils/file.md
|
||||||
|
|
@ -85,6 +89,9 @@ nav:
|
||||||
- Geometry: utils/geometry.md
|
- Geometry: utils/geometry.md
|
||||||
- Assets: assets.md
|
- Assets: assets.md
|
||||||
- Cookbooks: cookbooks.md
|
- Cookbooks: cookbooks.md
|
||||||
|
- Contributing: contributing.md
|
||||||
|
- Code of Conduct: code_of_conduct.md
|
||||||
|
- License: license.md
|
||||||
- Changelog:
|
- Changelog:
|
||||||
- Changelog: changelog.md
|
- Changelog: changelog.md
|
||||||
- Deprecated: deprecated.md
|
- Deprecated: deprecated.md
|
||||||
|
|
@ -132,10 +139,10 @@ plugins:
|
||||||
default_handler: python
|
default_handler: python
|
||||||
handlers:
|
handlers:
|
||||||
python:
|
python:
|
||||||
|
paths: [supervision]
|
||||||
|
load_external_modules: true
|
||||||
options:
|
options:
|
||||||
parameter_headings: true
|
parameter_headings: true
|
||||||
paths: [supervision]
|
|
||||||
load_external_modules: true
|
|
||||||
allow_inspection: true
|
allow_inspection: true
|
||||||
show_bases: true
|
show_bases: true
|
||||||
group_by_category: true
|
group_by_category: true
|
||||||
|
|
|
||||||
|
|
@ -1,9 +0,0 @@
|
||||||
#!/usr/bin/env bash
|
|
||||||
set -euo pipefail
|
|
||||||
|
|
||||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
|
||||||
|
|
||||||
for py in "$SCRIPT_DIR"/*.py; do
|
|
||||||
echo "Converting: $(basename "$py")"
|
|
||||||
jupytext --to ipynb "$py"
|
|
||||||
done
|
|
||||||
|
|
@ -1,450 +0,0 @@
|
||||||
# ---
|
|
||||||
# jupyter:
|
|
||||||
# jupytext:
|
|
||||||
# cell_metadata_filter: -all
|
|
||||||
# formats: ipynb,py:percent
|
|
||||||
# text_representation:
|
|
||||||
# extension: .py
|
|
||||||
# format_name: percent
|
|
||||||
# format_version: '1.3'
|
|
||||||
# jupytext_version: 1.19.1
|
|
||||||
# ---
|
|
||||||
# ruff: noqa: E402
|
|
||||||
|
|
||||||
# %% [markdown]
|
|
||||||
# # supervision 0.28.0: Memory-Efficient Instance Segmentation
|
|
||||||
#
|
|
||||||
# [](https://colab.research.google.com/github/roboflow/supervision/blob/develop/notebooks/release-demo_0-28.ipynb)
|
|
||||||
#
|
|
||||||
# **supervision** is a set of reusable tools for computer vision.
|
|
||||||
# Two headlining changes in 0.28.0:
|
|
||||||
#
|
|
||||||
# 1. **`sv.Detections.from_sam3`** -- first-class support for SAM3 (Segment
|
|
||||||
# Anything Model 3) inference responses. supervision now parses both the
|
|
||||||
# PCS (prompt-controlled segmentation) and PVS (point-video segmentation)
|
|
||||||
# output formats directly into a `sv.Detections` object.
|
|
||||||
#
|
|
||||||
# 2. **`sv.CompactMask`** -- instance masks stored as RLE-encoded bounding-box
|
|
||||||
# crops instead of full-resolution bitmaps. Any segmentation model --
|
|
||||||
# RF-DETR Seg, SAM3, YOLO-Seg -- can feed into CompactMask. Memory drops
|
|
||||||
# 10-100x without changing the API anywhere in supervision.
|
|
||||||
#
|
|
||||||
# **Story**: run RF-DETR Seg on a real image, visualise the masks, then convert
|
|
||||||
# to CompactMask and watch the memory footprint collapse.
|
|
||||||
#
|
|
||||||
# **Sections:**
|
|
||||||
# 1. [Install](#1-install)
|
|
||||||
# 2. [Download sample image](#2-download-sample-image)
|
|
||||||
# 3. [RF-DETR Seg -- instance segmentation](#3-rf-detr-seg)
|
|
||||||
# 4. [CompactMask -- memory-efficient storage](#4-compactmask)
|
|
||||||
# 5. [SAM3 -- text-prompted segmentation](#5-sam3)
|
|
||||||
# 6. [Other notable changes in 0.28.0](#6-other-notable-changes)
|
|
||||||
# 7. [Next steps](#7-next-steps)
|
|
||||||
|
|
||||||
# %% [markdown]
|
|
||||||
# ## 1. Install
|
|
||||||
|
|
||||||
# %%
|
|
||||||
# !pip install -q 'supervision==0.28.0' 'rfdetr' 'inference-sdk>=0.9' numpy matplotlib
|
|
||||||
|
|
||||||
# %% [markdown]
|
|
||||||
# ## 2. Download sample image
|
|
||||||
#
|
|
||||||
# `sv.ImageAssets` is new in 0.28.0 -- a counterpart to the existing
|
|
||||||
# `sv.VideoAssets`. `download_assets` caches locally and returns the path.
|
|
||||||
|
|
||||||
# %%
|
|
||||||
# %matplotlib inline
|
|
||||||
|
|
||||||
import cv2
|
|
||||||
import matplotlib.pyplot as plt
|
|
||||||
import numpy as np
|
|
||||||
|
|
||||||
import supervision as sv
|
|
||||||
from supervision.assets import ImageAssets, download_assets
|
|
||||||
|
|
||||||
image_path = download_assets(ImageAssets.PEOPLE_WALKING)
|
|
||||||
print(f"Image: {image_path}")
|
|
||||||
|
|
||||||
image_bgr = cv2.imread(image_path)
|
|
||||||
image_rgb = cv2.cvtColor(image_bgr, cv2.COLOR_BGR2RGB)
|
|
||||||
H, W = image_bgr.shape[:2]
|
|
||||||
print(f"Resolution: {W} x {H}")
|
|
||||||
|
|
||||||
plt.figure(figsize=(12, 7))
|
|
||||||
plt.imshow(image_rgb)
|
|
||||||
plt.axis("off")
|
|
||||||
plt.title("people-walking.jpg")
|
|
||||||
plt.tight_layout()
|
|
||||||
plt.show()
|
|
||||||
|
|
||||||
# %% [markdown]
|
|
||||||
# ## 3. RF-DETR Seg
|
|
||||||
#
|
|
||||||
# **RF-DETR** is a real-time transformer-based object detection model from Roboflow.
|
|
||||||
# The `RFDETRSegSmall` variant adds an instance segmentation head -- it produces
|
|
||||||
# one binary mask per detected instance alongside the bounding box.
|
|
||||||
#
|
|
||||||
# Key facts for this demo:
|
|
||||||
#
|
|
||||||
# - Pretrained on **COCO** (80 object categories) -- detects people, bags, cars, etc.
|
|
||||||
# - Weights download automatically on first `RFDETRSegSmall()` call (~100 MB).
|
|
||||||
# - `model.predict()` returns **`sv.Detections`** directly -- no converter needed.
|
|
||||||
# Masks are a `(N, H, W)` bool array attached as `detections.mask`.
|
|
||||||
|
|
||||||
# %%
|
|
||||||
from rfdetr.detr import RFDETRSegSmall
|
|
||||||
|
|
||||||
model = RFDETRSegSmall()
|
|
||||||
model.optimize_for_inference()
|
|
||||||
|
|
||||||
# predict accepts a file path, PIL Image, or RGB numpy array
|
|
||||||
detections = model.predict(image_path, threshold=0.3)
|
|
||||||
if not isinstance(detections, sv.Detections):
|
|
||||||
raise TypeError(f"Expected sv.Detections, got {type(detections).__name__}")
|
|
||||||
|
|
||||||
n_masks = 0 if detections.mask is None else len(detections.mask)
|
|
||||||
print(f"Detections: {len(detections)} (with masks: {n_masks})")
|
|
||||||
|
|
||||||
# %% [markdown]
|
|
||||||
# ### 3.1 COCO class names
|
|
||||||
#
|
|
||||||
# COCO has 90 numeric class IDs; map them to readable names for annotation.
|
|
||||||
|
|
||||||
# %%
|
|
||||||
# Subset of COCO class names (IDs 0-based after RF-DETR's remapping).
|
|
||||||
COCO_NAMES: dict[int, str] = {
|
|
||||||
0: "person",
|
|
||||||
1: "bicycle",
|
|
||||||
2: "car",
|
|
||||||
3: "motorcycle",
|
|
||||||
4: "airplane",
|
|
||||||
5: "bus",
|
|
||||||
6: "train",
|
|
||||||
7: "truck",
|
|
||||||
8: "boat",
|
|
||||||
24: "backpack",
|
|
||||||
25: "umbrella",
|
|
||||||
26: "handbag",
|
|
||||||
28: "suitcase",
|
|
||||||
56: "chair",
|
|
||||||
57: "couch",
|
|
||||||
58: "potted plant",
|
|
||||||
59: "bed",
|
|
||||||
60: "dining table",
|
|
||||||
62: "tv",
|
|
||||||
63: "laptop",
|
|
||||||
67: "cell phone",
|
|
||||||
72: "refrigerator",
|
|
||||||
74: "clock",
|
|
||||||
76: "scissors",
|
|
||||||
}
|
|
||||||
|
|
||||||
labels = []
|
|
||||||
assert detections.class_id is not None
|
|
||||||
for cid, conf in zip(
|
|
||||||
detections.class_id,
|
|
||||||
detections.confidence
|
|
||||||
if detections.confidence is not None
|
|
||||||
else [None] * len(detections),
|
|
||||||
):
|
|
||||||
name = COCO_NAMES.get(int(cid), f"cls_{cid}")
|
|
||||||
labels.append(f"{name} {conf:.2f}" if conf is not None else name)
|
|
||||||
|
|
||||||
# %% [markdown]
|
|
||||||
# ### 3.2 Visualise RF-DETR Seg output
|
|
||||||
|
|
||||||
# %%
|
|
||||||
PALETTE = sv.ColorPalette.DEFAULT
|
|
||||||
|
|
||||||
annotated = image_bgr.copy()
|
|
||||||
annotated = sv.MaskAnnotator(color=PALETTE, opacity=0.45).annotate(
|
|
||||||
annotated, detections
|
|
||||||
)
|
|
||||||
annotated = sv.BoxAnnotator(color=PALETTE, thickness=2).annotate(annotated, detections)
|
|
||||||
annotated = sv.LabelAnnotator(color=PALETTE, text_scale=0.5, text_thickness=1).annotate(
|
|
||||||
annotated, detections, labels=labels
|
|
||||||
)
|
|
||||||
|
|
||||||
plt.figure(figsize=(12, 7))
|
|
||||||
plt.imshow(cv2.cvtColor(annotated, cv2.COLOR_BGR2RGB))
|
|
||||||
plt.axis("off")
|
|
||||||
plt.title(f"RF-DETR Seg -- {len(detections)} instance(s)")
|
|
||||||
plt.tight_layout()
|
|
||||||
plt.show()
|
|
||||||
|
|
||||||
# %% [markdown]
|
|
||||||
# ## 4. CompactMask
|
|
||||||
#
|
|
||||||
# RF-DETR Seg returns one full-resolution binary mask per detected instance.
|
|
||||||
# On a 1280 x 720 image with 12 people that is:
|
|
||||||
#
|
|
||||||
# `12 x 720 x 1280 x 1 byte = 11 MB`
|
|
||||||
#
|
|
||||||
# Most of those pixels are background. The actual person silhouette fits in a
|
|
||||||
# tight bounding box. `sv.CompactMask` stores **only the bounding-box crop**,
|
|
||||||
# RLE-encoded:
|
|
||||||
#
|
|
||||||
# - A 200 x 100 person crop: `~2.5 KB` instead of `900 KB`
|
|
||||||
# - Drop-in replacement -- all annotators, filters, and `area` keep working
|
|
||||||
|
|
||||||
# %% [markdown]
|
|
||||||
# ### 4.1 Measure dense mask footprint
|
|
||||||
|
|
||||||
# %%
|
|
||||||
from typing import Any
|
|
||||||
|
|
||||||
dense_bytes: int = 0
|
|
||||||
dense_mask: "np.ndarray[Any, np.dtype[np.bool_]] | None" = None
|
|
||||||
|
|
||||||
assert detections.mask is not None and isinstance(detections.mask, np.ndarray)
|
|
||||||
|
|
||||||
dense_mask = detections.mask
|
|
||||||
dense_bytes = dense_mask.nbytes
|
|
||||||
n_inst = len(dense_mask)
|
|
||||||
print(f"Instances: {n_inst}")
|
|
||||||
print(f"Mask shape: {dense_mask.shape} (N x H x W, bool)")
|
|
||||||
print(f"Dense footprint: {dense_bytes / 1024:.1f} KB")
|
|
||||||
print(f" = {n_inst} masks x {H} x {W} x 1 byte")
|
|
||||||
|
|
||||||
# %% [markdown]
|
|
||||||
# ### 4.2 Convert to CompactMask
|
|
||||||
|
|
||||||
# %%
|
|
||||||
compact: "sv.CompactMask | None" = None
|
|
||||||
crop_bytes: int = 0
|
|
||||||
|
|
||||||
assert dense_mask is not None
|
|
||||||
compact = sv.CompactMask.from_dense(
|
|
||||||
masks=dense_mask,
|
|
||||||
xyxy=detections.xyxy,
|
|
||||||
image_shape=(H, W),
|
|
||||||
)
|
|
||||||
|
|
||||||
# Measure compact size via uncompressed crop booleans (upper bound; RLE < this).
|
|
||||||
crop_bytes = sum(compact.crop(i).nbytes for i in range(len(compact)))
|
|
||||||
|
|
||||||
print(f"Crop size (est.): {crop_bytes / 1024:.1f} KB (uncompressed crops)")
|
|
||||||
if crop_bytes > 0 and dense_bytes > 0:
|
|
||||||
ratio = dense_bytes / crop_bytes
|
|
||||||
print(f"Reduction factor: {ratio:.1f}x (before RLE compression)")
|
|
||||||
|
|
||||||
# Swap in CompactMask -- supervision uses it transparently from here on.
|
|
||||||
detections.mask = compact
|
|
||||||
print(f"\ndetections.mask type: {type(detections.mask).__name__}")
|
|
||||||
|
|
||||||
# %% [markdown]
|
|
||||||
# ### 4.3 Filtering by mask area
|
|
||||||
#
|
|
||||||
# `compact.area` returns the true pixel count of each instance mask.
|
|
||||||
# Filter out tiny detections (partial occlusions, image-edge artefacts).
|
|
||||||
|
|
||||||
# %%
|
|
||||||
large: sv.Detections = detections
|
|
||||||
large_labels: list[str] = labels
|
|
||||||
|
|
||||||
assert isinstance(detections.mask, sv.CompactMask)
|
|
||||||
areas = detections.mask.area
|
|
||||||
print(
|
|
||||||
f"Mask areas (px): min={areas.min():.0f} "
|
|
||||||
f"mean={areas.mean():.0f} max={areas.max():.0f}"
|
|
||||||
)
|
|
||||||
|
|
||||||
# Keep instances larger than 0.1% of the image.
|
|
||||||
min_area = 0.001 * H * W
|
|
||||||
keep_idx = np.where(areas > min_area)[0]
|
|
||||||
_filtered = detections[keep_idx]
|
|
||||||
if isinstance(_filtered, sv.Detections):
|
|
||||||
large = _filtered
|
|
||||||
large_labels = [labels[i] for i in keep_idx] if labels else []
|
|
||||||
print(f"\nInstances > {min_area:.0f} px: {len(large)}")
|
|
||||||
|
|
||||||
# %% [markdown]
|
|
||||||
# ### 4.4 Annotate with CompactMask
|
|
||||||
#
|
|
||||||
# Annotators call `.to_dense()` internally -- CompactMask is invisible to them.
|
|
||||||
|
|
||||||
# %%
|
|
||||||
assert isinstance(detections.mask, sv.CompactMask) and dense_bytes > 0
|
|
||||||
|
|
||||||
annotated_compact = image_bgr.copy()
|
|
||||||
annotated_compact = sv.MaskAnnotator(color=PALETTE, opacity=0.45).annotate(
|
|
||||||
annotated_compact, large
|
|
||||||
)
|
|
||||||
annotated_compact = sv.BoxAnnotator(color=PALETTE, thickness=2).annotate(
|
|
||||||
annotated_compact, large
|
|
||||||
)
|
|
||||||
annotated_compact = sv.LabelAnnotator(
|
|
||||||
color=PALETTE, text_scale=0.5, text_thickness=1
|
|
||||||
).annotate(annotated_compact, large, labels=large_labels)
|
|
||||||
|
|
||||||
plt.figure(figsize=(12, 7))
|
|
||||||
plt.imshow(cv2.cvtColor(annotated_compact, cv2.COLOR_BGR2RGB))
|
|
||||||
plt.axis("off")
|
|
||||||
plt.title(
|
|
||||||
f"CompactMask (filtered) -- {len(large)} instance(s) "
|
|
||||||
f"| {dense_bytes / 1024:.0f} KB dense -> {crop_bytes / 1024:.0f} KB crops"
|
|
||||||
)
|
|
||||||
plt.tight_layout()
|
|
||||||
plt.show()
|
|
||||||
|
|
||||||
# %% [markdown]
|
|
||||||
# ### 4.5 Per-instance crop
|
|
||||||
#
|
|
||||||
# `compact.crop(i)` decodes only the bounding-box crop for instance `i` as a
|
|
||||||
# `(H_crop, W_crop)` bool array -- no full mask materialised.
|
|
||||||
|
|
||||||
# %%
|
|
||||||
assert isinstance(detections.mask, sv.CompactMask) and len(detections) > 0
|
|
||||||
|
|
||||||
crop = detections.mask.crop(0)
|
|
||||||
bbox = detections.mask.bbox_xyxy[0].astype(int)
|
|
||||||
|
|
||||||
fig, axes = plt.subplots(1, 2, figsize=(10, 4))
|
|
||||||
axes[0].imshow(image_rgb[bbox[1] : bbox[3], bbox[0] : bbox[2]])
|
|
||||||
axes[0].set_title("Image crop (instance 0)")
|
|
||||||
axes[0].axis("off")
|
|
||||||
|
|
||||||
axes[1].imshow(crop, cmap="gray")
|
|
||||||
axes[1].set_title(f"Mask crop ({crop.shape[1]} x {crop.shape[0]} px)")
|
|
||||||
axes[1].axis("off")
|
|
||||||
|
|
||||||
plt.tight_layout()
|
|
||||||
plt.show()
|
|
||||||
|
|
||||||
full_px = H * W
|
|
||||||
crop_kb = crop.nbytes / 1024
|
|
||||||
print(f"Full-res mask slot: {H} x {W} = {full_px / 1024:.0f} KB")
|
|
||||||
print(f"Compact crop: {crop.shape[0]} x {crop.shape[1]} = {crop_kb:.1f} KB")
|
|
||||||
|
|
||||||
# %% [markdown]
|
|
||||||
# ## 5. SAM3
|
|
||||||
#
|
|
||||||
# `sv.Detections.from_sam3()` is the other headline in 0.28.0.
|
|
||||||
# SAM3 segments objects by free-text prompts -- `"person"`, `"bag"`, any phrase.
|
|
||||||
# supervision parses both the PCS and PVS response formats into a standard
|
|
||||||
# `sv.Detections`, with `class_id` set to the prompt index.
|
|
||||||
#
|
|
||||||
# This section runs only when `ROBOFLOW_API_KEY` is available.
|
|
||||||
|
|
||||||
# %%
|
|
||||||
import base64
|
|
||||||
import os
|
|
||||||
from typing import Optional
|
|
||||||
|
|
||||||
import requests
|
|
||||||
|
|
||||||
try:
|
|
||||||
from google.colab import userdata # type: ignore[import, unused-ignore]
|
|
||||||
|
|
||||||
ROBOFLOW_API_KEY: str = userdata.get("ROBOFLOW_API_KEY") or ""
|
|
||||||
except Exception:
|
|
||||||
ROBOFLOW_API_KEY = os.environ.get("ROBOFLOW_API_KEY", "")
|
|
||||||
|
|
||||||
PROMPTS = ["person", "bag"]
|
|
||||||
sam3_detections: Optional[sv.Detections] = None
|
|
||||||
|
|
||||||
assert ROBOFLOW_API_KEY
|
|
||||||
|
|
||||||
with open(image_path, "rb") as _f:
|
|
||||||
_img_b64 = base64.b64encode(_f.read()).decode("utf-8")
|
|
||||||
|
|
||||||
_response = requests.post(
|
|
||||||
f"https://api.roboflow.com/inferenceproxy/seg-preview?api_key={ROBOFLOW_API_KEY}",
|
|
||||||
json={
|
|
||||||
"image": {"type": "base64", "value": _img_b64},
|
|
||||||
"prompts": [{"type": "text", "text": p} for p in PROMPTS],
|
|
||||||
"output_prob_thresh": 0.3,
|
|
||||||
},
|
|
||||||
headers={"Content-Type": "application/json"},
|
|
||||||
timeout=60,
|
|
||||||
)
|
|
||||||
_response.raise_for_status()
|
|
||||||
sam3_result: dict[str, Any] = _response.json()
|
|
||||||
sam3_detections = sv.Detections.from_sam3(sam3_result=sam3_result, resolution_wh=(W, H))
|
|
||||||
print(f"SAM3 detections: {len(sam3_detections)}")
|
|
||||||
if sam3_detections.class_id is not None:
|
|
||||||
for idx, prompt in enumerate(PROMPTS):
|
|
||||||
count = int((sam3_detections.class_id == idx).sum())
|
|
||||||
print(f" [{idx}] '{prompt}': {count} instance(s)")
|
|
||||||
|
|
||||||
# %%
|
|
||||||
assert sam3_detections is not None and len(sam3_detections) > 0
|
|
||||||
|
|
||||||
sam3_labels = (
|
|
||||||
[PROMPTS[c] for c in sam3_detections.class_id]
|
|
||||||
if sam3_detections.class_id is not None
|
|
||||||
else []
|
|
||||||
)
|
|
||||||
SAM3_PALETTE = sv.ColorPalette.from_hex(["#ff6b6b", "#4ecdc4"])
|
|
||||||
|
|
||||||
annotated_sam3 = image_bgr.copy()
|
|
||||||
annotated_sam3 = sv.MaskAnnotator(color=SAM3_PALETTE, opacity=0.45).annotate(
|
|
||||||
annotated_sam3, sam3_detections
|
|
||||||
)
|
|
||||||
annotated_sam3 = sv.BoxAnnotator(color=SAM3_PALETTE, thickness=2).annotate(
|
|
||||||
annotated_sam3, sam3_detections
|
|
||||||
)
|
|
||||||
annotated_sam3 = sv.LabelAnnotator(
|
|
||||||
color=SAM3_PALETTE, text_scale=0.5, text_thickness=1
|
|
||||||
).annotate(annotated_sam3, sam3_detections, labels=sam3_labels)
|
|
||||||
|
|
||||||
plt.figure(figsize=(12, 7))
|
|
||||||
plt.imshow(cv2.cvtColor(annotated_sam3, cv2.COLOR_BGR2RGB))
|
|
||||||
plt.axis("off")
|
|
||||||
plt.title(f"SAM3 -- from_sam3() -- {len(sam3_detections)} instance(s)")
|
|
||||||
plt.tight_layout()
|
|
||||||
plt.show()
|
|
||||||
|
|
||||||
# %% [markdown]
|
|
||||||
# ## 6. Other notable changes in 0.28.0
|
|
||||||
#
|
|
||||||
# ### `VideoInfo.fps` is now `float`
|
|
||||||
#
|
|
||||||
# NTSC frame rates (23.976, 29.97, 59.94) were silently truncated to `int`.
|
|
||||||
# Wrap with `int()` at call sites that require an integer.
|
|
||||||
|
|
||||||
# %%
|
|
||||||
import collections
|
|
||||||
|
|
||||||
from supervision.assets import VideoAssets
|
|
||||||
|
|
||||||
video_path = download_assets(VideoAssets.PEOPLE_WALKING)
|
|
||||||
info = sv.VideoInfo.from_video_path(video_path)
|
|
||||||
|
|
||||||
print(f"fps: {info.fps} ({type(info.fps).__name__}) -- was int before 0.28.0")
|
|
||||||
|
|
||||||
fps_int = int(info.fps)
|
|
||||||
buf: collections.deque[sv.Detections] = collections.deque(maxlen=fps_int)
|
|
||||||
trace = sv.TraceAnnotator(trace_length=fps_int)
|
|
||||||
print(f"deque maxlen: {buf.maxlen} (= int({info.fps}))")
|
|
||||||
|
|
||||||
# %% [markdown]
|
|
||||||
# ### `sv.ByteTrack` deprecated
|
|
||||||
#
|
|
||||||
# `sv.ByteTrack` still works in 0.28.0 and 0.29.0 but emits a
|
|
||||||
# `DeprecationWarning`. Migrate to `ByteTrackTracker` from the external
|
|
||||||
# [`trackers`](https://pypi.org/project/trackers/) package before 0.30.0.
|
|
||||||
#
|
|
||||||
# ```python
|
|
||||||
# # Before
|
|
||||||
# tracker = sv.ByteTrack()
|
|
||||||
# detections = tracker.update_with_detections(detections)
|
|
||||||
#
|
|
||||||
# # After (pip install trackers)
|
|
||||||
# from trackers import ByteTrackTracker
|
|
||||||
# tracker = ByteTrackTracker()
|
|
||||||
# detections = tracker.update(detections)
|
|
||||||
# ```
|
|
||||||
|
|
||||||
# %% [markdown]
|
|
||||||
# ## 7. Next steps
|
|
||||||
#
|
|
||||||
# - [`sv.CompactMask` docs](https://supervision.roboflow.com/develop/detection/compact_mask/)
|
|
||||||
# -- full API reference: `resize`, `merge`, `with_offset`
|
|
||||||
# - [`sv.Detections.from_sam3` docs](https://supervision.roboflow.com/develop/detection/core/)
|
|
||||||
# -- PCS and PVS format reference
|
|
||||||
# - [RF-DETR docs](https://github.com/roboflow/rf-detr)
|
|
||||||
# -- training, export, and deployment
|
|
||||||
# - [Full changelog](https://supervision.roboflow.com/develop/changelog/)
|
|
||||||
# -- every change in 0.28.0
|
|
||||||
|
|
@ -4,7 +4,7 @@ requires = [ "setuptools>=61" ]
|
||||||
|
|
||||||
[project]
|
[project]
|
||||||
name = "supervision"
|
name = "supervision"
|
||||||
version = "0.28.0"
|
version = "0.31.0.dev0"
|
||||||
description = "A set of easy-to-use utils that will come in handy in any Computer Vision project"
|
description = "A set of easy-to-use utils that will come in handy in any Computer Vision project"
|
||||||
readme = "README.md"
|
readme = "README.md"
|
||||||
keywords = [
|
keywords = [
|
||||||
|
|
@ -23,7 +23,7 @@ maintainers = [
|
||||||
authors = [
|
authors = [
|
||||||
{ name = "Roboflow et al.", email = "develop@roboflow.com" },
|
{ name = "Roboflow et al.", email = "develop@roboflow.com" },
|
||||||
]
|
]
|
||||||
requires-python = ">=3.9"
|
requires-python = ">=3.10"
|
||||||
classifiers = [
|
classifiers = [
|
||||||
"Development Status :: 5 - Production/Stable",
|
"Development Status :: 5 - Production/Stable",
|
||||||
"Intended Audience :: Developers",
|
"Intended Audience :: Developers",
|
||||||
|
|
@ -33,7 +33,6 @@ classifiers = [
|
||||||
"Operating System :: Microsoft :: Windows",
|
"Operating System :: Microsoft :: Windows",
|
||||||
"Operating System :: POSIX :: Linux",
|
"Operating System :: POSIX :: Linux",
|
||||||
"Programming Language :: Python :: 3 :: Only",
|
"Programming Language :: Python :: 3 :: Only",
|
||||||
"Programming Language :: Python :: 3.9",
|
|
||||||
"Programming Language :: Python :: 3.10",
|
"Programming Language :: Python :: 3.10",
|
||||||
"Programming Language :: Python :: 3.11",
|
"Programming Language :: Python :: 3.11",
|
||||||
"Programming Language :: Python :: 3.12",
|
"Programming Language :: Python :: 3.12",
|
||||||
|
|
@ -48,17 +47,20 @@ classifiers = [
|
||||||
"Typing :: Typed",
|
"Typing :: Typed",
|
||||||
]
|
]
|
||||||
dependencies = [
|
dependencies = [
|
||||||
|
"av>=14.2",
|
||||||
"defusedxml>=0.7.1",
|
"defusedxml>=0.7.1",
|
||||||
"matplotlib>=3.6",
|
"matplotlib>=3.6",
|
||||||
"numpy>=1.21.2",
|
"numpy>=1.21.2",
|
||||||
"opencv-python>=4.5.5.64",
|
|
||||||
"pillow>=9.4",
|
"pillow>=9.4",
|
||||||
"pydeprecate>=0.7,<0.8",
|
"pydeprecate>=0.9,<0.12",
|
||||||
"pyyaml>=5.3",
|
"pyyaml>=5.3",
|
||||||
"requests>=2.26",
|
"requests>=2.26",
|
||||||
"scipy>=1.10",
|
"scipy>=1.10",
|
||||||
"tqdm>=4.62.3"
|
"tqdm>=4.62.3"
|
||||||
]
|
]
|
||||||
|
optional-dependencies.geotiff = [
|
||||||
|
"rasterio>=1.3", # 1.3 introduced stable window-read API and CRS.is_projected
|
||||||
|
]
|
||||||
optional-dependencies.metrics = [
|
optional-dependencies.metrics = [
|
||||||
"pandas>=2",
|
"pandas>=2",
|
||||||
]
|
]
|
||||||
|
|
@ -74,34 +76,36 @@ dev = [
|
||||||
"nbconvert>=7.14.2",
|
"nbconvert>=7.14.2",
|
||||||
"notebook>=6.5.3,<8",
|
"notebook>=6.5.3,<8",
|
||||||
"pre-commit>=3.8",
|
"pre-commit>=3.8",
|
||||||
"pytest>=7.2.2,<9",
|
"pytest>=7.2.2,<10",
|
||||||
"pytest-cov>=4,<8",
|
"pytest-cov>=4,<8",
|
||||||
|
"scikit-learn>=1.7",
|
||||||
"tox>=4.11.4",
|
"tox>=4.11.4",
|
||||||
|
"types-tqdm",
|
||||||
]
|
]
|
||||||
docs = [
|
docs = [
|
||||||
"mike>=2",
|
"mike>=2",
|
||||||
"mkdocs-git-committers-plugin-2>=2.4.1; python_version>='3.9' and python_version<'4'",
|
"mkdocs-git-committers-plugin-2>=2.4.1; python_version>='3.10' and python_version<'4'",
|
||||||
"mkdocs-git-revision-date-localized-plugin>=1.2.4",
|
"mkdocs-git-revision-date-localized-plugin>=1.2.4",
|
||||||
"mkdocs-jupyter>=0.24.3",
|
"mkdocs-jupyter>=0.24.3",
|
||||||
"mkdocs-material[imaging]>=9.7",
|
"mkdocs-material[imaging]>=9.7",
|
||||||
"mkdocstrings>=0.25.2,<0.31",
|
"mkdocstrings>=1,<1.1",
|
||||||
"mkdocstrings-python>=1.10.9,<2", # todo: breaking changes in 2.x
|
"mkdocstrings-python>=2,<3",
|
||||||
]
|
]
|
||||||
build = [
|
build = [
|
||||||
"build>=0.10,<1.5",
|
"build>=1,<1.6",
|
||||||
"twine>=5.1.1,<7",
|
"twine>=5.1.1,<7",
|
||||||
"wheel>=0.40,<0.48",
|
"wheel>=0.40,<0.48",
|
||||||
]
|
]
|
||||||
|
|
||||||
[tool.setuptools]
|
[tool.setuptools]
|
||||||
include-package-data = false
|
|
||||||
package-data.supervision = [ "py.typed" ]
|
|
||||||
packages.find.where = [ "src" ]
|
packages.find.where = [ "src" ]
|
||||||
packages.find.include = [ "supervision*" ]
|
packages.find.include = [ "supervision*" ]
|
||||||
|
include-package-data = false
|
||||||
|
package-data.supervision = [ "py.typed" ]
|
||||||
# exclude = [ "docs*", "tests*", "examples*" ]
|
# exclude = [ "docs*", "tests*", "examples*" ]
|
||||||
|
|
||||||
[tool.ruff]
|
[tool.ruff]
|
||||||
target-version = "py39"
|
target-version = "py310"
|
||||||
line-length = 88
|
line-length = 88
|
||||||
indent-width = 4
|
indent-width = 4
|
||||||
# Exclude a variety of commonly ignored directories.
|
# Exclude a variety of commonly ignored directories.
|
||||||
|
|
@ -163,6 +167,7 @@ lint.per-file-ignores."src/**" = [
|
||||||
]
|
]
|
||||||
lint.per-file-ignores."tests/**" = [
|
lint.per-file-ignores."tests/**" = [
|
||||||
"S101", # Use of `assert` detected
|
"S101", # Use of `assert` detected
|
||||||
|
"S603", # `subprocess` call: subprocess with hardcoded args in test utilities is safe
|
||||||
]
|
]
|
||||||
lint.unfixable = []
|
lint.unfixable = []
|
||||||
# Allow unused variables when underscore-prefixed.
|
# Allow unused variables when underscore-prefixed.
|
||||||
|
|
@ -178,35 +183,33 @@ lint.pydocstyle.convention = "google"
|
||||||
lint.pylint.max-args = 20
|
lint.pylint.max-args = 20
|
||||||
|
|
||||||
[tool.codespell]
|
[tool.codespell]
|
||||||
|
ignore-words-list = "STrack,sTrack,strack"
|
||||||
skip = "*.ipynb"
|
skip = "*.ipynb"
|
||||||
count = true
|
count = true
|
||||||
quiet-level = 3
|
quiet-level = 3
|
||||||
ignore-words-list = "STrack,sTrack,strack"
|
|
||||||
|
|
||||||
[tool.mypy]
|
[tool.mypy]
|
||||||
python_version = "3.9"
|
|
||||||
ignore_missing_imports = false
|
|
||||||
explicit_package_bases = true
|
|
||||||
strict = true
|
|
||||||
mypy_path = "src"
|
mypy_path = "src"
|
||||||
|
explicit_package_bases = true
|
||||||
|
ignore_missing_imports = false
|
||||||
|
python_version = "3.10"
|
||||||
|
warn_unused_ignores = true
|
||||||
|
strict = true
|
||||||
overrides = [
|
overrides = [
|
||||||
# exclude = [
|
{ module = [ "examples.*", "tests.*" ], ignore_errors = true },
|
||||||
# "docs",
|
{ module = [ "supervision._cv2" ], warn_unused_ignores = false },
|
||||||
# "test",
|
|
||||||
# "examples",
|
|
||||||
# "setup.py",
|
|
||||||
# ]
|
|
||||||
{ module = [
|
|
||||||
"tests.*",
|
|
||||||
"examples.*",
|
|
||||||
], ignore_errors = true },
|
|
||||||
]
|
]
|
||||||
|
|
||||||
[tool.pytest]
|
[tool.pytest]
|
||||||
|
ini_options.testpaths = [ "src", "tests" ]
|
||||||
|
ini_options.norecursedirs = [ ".git", ".venv", "build", "dist", "docs", "examples", "notebooks" ]
|
||||||
ini_options.addopts = [
|
ini_options.addopts = [
|
||||||
"--doctest-modules",
|
"--doctest-modules",
|
||||||
"--color=yes",
|
"--color=yes",
|
||||||
]
|
]
|
||||||
|
ini_options.filterwarnings = [
|
||||||
|
"error::DeprecationWarning",
|
||||||
|
]
|
||||||
ini_options.doctest_optionflags = "ELLIPSIS NORMALIZE_WHITESPACE"
|
ini_options.doctest_optionflags = "ELLIPSIS NORMALIZE_WHITESPACE"
|
||||||
|
|
||||||
[tool.autoflake]
|
[tool.autoflake]
|
||||||
|
|
|
||||||
|
|
@ -1,24 +0,0 @@
|
||||||
# Release Process
|
|
||||||
|
|
||||||
This doc outlines how supervision is released into production.
|
|
||||||
|
|
||||||
It assumes you already have the code changes, as well as a draft of the release notes.
|
|
||||||
|
|
||||||
1. Make sure you have all required changes were merged into `develop`.
|
|
||||||
2. Create and merge a PR, merging `develop` into `main`, containing:
|
|
||||||
- A commit that updates the project version in `pyproject.toml`.
|
|
||||||
- All changes made during the release.
|
|
||||||
3. Tag the commit with the new supervision version.
|
|
||||||
- make sure to pull from `main` !
|
|
||||||
- Verify that the latest merge commits exists. `git log`.
|
|
||||||
- Run `git tag x.y.z`, with your version
|
|
||||||
- Check with `git log`.
|
|
||||||
- Run `git push origin --tags`
|
|
||||||
- Upon pushing the Github release, the [PyPi](https://pypi.org/project/supervision/) should update to the new version. Check this!
|
|
||||||
4. Open and merge a PR, merging `main` into `develop`.
|
|
||||||
5. Update the docs by running the [Supervision Release Documentation Workflow 📚](https://github.com/roboflow/supervision/actions/workflows/publish-release-docs.yml) workflow from GitHub.
|
|
||||||
- Select the `main` branch from the dropdown.
|
|
||||||
6. Create a release on GitHub.
|
|
||||||
- Go to releases
|
|
||||||
- Assign the release notes to the tag created in step 3.
|
|
||||||
- Publish the release.
|
|
||||||
|
|
@ -1,4 +1,5 @@
|
||||||
import importlib.metadata as importlib_metadata
|
import importlib.metadata as importlib_metadata
|
||||||
|
from typing import TYPE_CHECKING, Any
|
||||||
|
|
||||||
try:
|
try:
|
||||||
# This will read version from pyproject.toml
|
# This will read version from pyproject.toml
|
||||||
|
|
@ -52,7 +53,10 @@ from supervision.detection.line_zone import (
|
||||||
LineZoneAnnotatorMulticlass,
|
LineZoneAnnotatorMulticlass,
|
||||||
)
|
)
|
||||||
from supervision.detection.tools.csv_sink import CSVSink
|
from supervision.detection.tools.csv_sink import CSVSink
|
||||||
from supervision.detection.tools.inference_slicer import InferenceSlicer
|
from supervision.detection.tools.inference_slicer import (
|
||||||
|
InferenceSlicer,
|
||||||
|
WindowedRasterDataset,
|
||||||
|
)
|
||||||
from supervision.detection.tools.json_sink import JSONSink
|
from supervision.detection.tools.json_sink import JSONSink
|
||||||
from supervision.detection.tools.polygon_zone import PolygonZone, PolygonZoneAnnotator
|
from supervision.detection.tools.polygon_zone import PolygonZone, PolygonZoneAnnotator
|
||||||
from supervision.detection.tools.smoother import DetectionsSmoother
|
from supervision.detection.tools.smoother import DetectionsSmoother
|
||||||
|
|
@ -62,6 +66,7 @@ from supervision.detection.utils.boxes import (
|
||||||
move_boxes,
|
move_boxes,
|
||||||
pad_boxes,
|
pad_boxes,
|
||||||
scale_boxes,
|
scale_boxes,
|
||||||
|
xyxyxyxy_to_xyxy,
|
||||||
)
|
)
|
||||||
from supervision.detection.utils.converters import (
|
from supervision.detection.utils.converters import (
|
||||||
is_compressed_rle,
|
is_compressed_rle,
|
||||||
|
|
@ -86,16 +91,21 @@ from supervision.detection.utils.iou_and_nms import (
|
||||||
box_iou_batch_with_jaccard,
|
box_iou_batch_with_jaccard,
|
||||||
box_non_max_merge,
|
box_non_max_merge,
|
||||||
box_non_max_suppression,
|
box_non_max_suppression,
|
||||||
|
box_soft_non_max_suppression,
|
||||||
mask_iou_batch,
|
mask_iou_batch,
|
||||||
mask_non_max_merge,
|
mask_non_max_merge,
|
||||||
mask_non_max_suppression,
|
mask_non_max_suppression,
|
||||||
|
mask_soft_non_max_suppression,
|
||||||
oriented_box_iou_batch,
|
oriented_box_iou_batch,
|
||||||
|
oriented_box_non_max_merge,
|
||||||
|
oriented_box_non_max_suppression,
|
||||||
)
|
)
|
||||||
from supervision.detection.utils.masks import (
|
from supervision.detection.utils.masks import (
|
||||||
calculate_masks_centroids,
|
calculate_masks_centroids,
|
||||||
contains_holes,
|
contains_holes,
|
||||||
contains_multiple_segments,
|
contains_multiple_segments,
|
||||||
filter_segments_by_distance,
|
filter_segments_by_distance,
|
||||||
|
mask_to_roi,
|
||||||
move_masks,
|
move_masks,
|
||||||
)
|
)
|
||||||
from supervision.detection.utils.polygons import (
|
from supervision.detection.utils.polygons import (
|
||||||
|
|
@ -121,11 +131,14 @@ from supervision.geometry.utils import get_polygon_center
|
||||||
from supervision.key_points.annotators import (
|
from supervision.key_points.annotators import (
|
||||||
EdgeAnnotator,
|
EdgeAnnotator,
|
||||||
VertexAnnotator,
|
VertexAnnotator,
|
||||||
|
VertexEllipseAnnotator,
|
||||||
|
VertexEllipseAreaAnnotator,
|
||||||
|
VertexEllipseHaloAnnotator,
|
||||||
|
VertexEllipseOutlineAnnotator,
|
||||||
VertexLabelAnnotator,
|
VertexLabelAnnotator,
|
||||||
)
|
)
|
||||||
from supervision.key_points.core import KeyPoints
|
from supervision.key_points.core import KeyPoints
|
||||||
from supervision.metrics.detection import ConfusionMatrix, MeanAveragePrecision
|
from supervision.metrics.detection import ConfusionMatrix, MeanAveragePrecision
|
||||||
from supervision.tracker.byte_tracker.core import ByteTrack
|
|
||||||
from supervision.utils.conversion import cv2_to_pillow, pillow_to_cv2
|
from supervision.utils.conversion import cv2_to_pillow, pillow_to_cv2
|
||||||
from supervision.utils.file import list_files_with_extensions
|
from supervision.utils.file import list_files_with_extensions
|
||||||
from supervision.utils.image import (
|
from supervision.utils.image import (
|
||||||
|
|
@ -134,11 +147,13 @@ from supervision.utils.image import (
|
||||||
get_image_resolution_wh,
|
get_image_resolution_wh,
|
||||||
grayscale_image,
|
grayscale_image,
|
||||||
letterbox_image,
|
letterbox_image,
|
||||||
|
load_image_from_url,
|
||||||
overlay_image,
|
overlay_image,
|
||||||
resize_image,
|
resize_image,
|
||||||
scale_image,
|
scale_image,
|
||||||
tint_image,
|
tint_image,
|
||||||
)
|
)
|
||||||
|
from supervision.utils.image_window import ImageWindow
|
||||||
from supervision.utils.notebook import plot_image, plot_images_grid
|
from supervision.utils.notebook import plot_image, plot_images_grid
|
||||||
from supervision.utils.video import (
|
from supervision.utils.video import (
|
||||||
FPSMonitor,
|
FPSMonitor,
|
||||||
|
|
@ -148,8 +163,12 @@ from supervision.utils.video import (
|
||||||
process_video,
|
process_video,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
from supervision.tracker.byte_tracker.core import ByteTrack
|
||||||
|
|
||||||
__all__ = [
|
__all__ = [
|
||||||
"LMM",
|
"LMM",
|
||||||
|
"VLM",
|
||||||
"BackgroundOverlayAnnotator",
|
"BackgroundOverlayAnnotator",
|
||||||
"BaseDataset",
|
"BaseDataset",
|
||||||
"BlurAnnotator",
|
"BlurAnnotator",
|
||||||
|
|
@ -179,6 +198,7 @@ __all__ = [
|
||||||
"HeatMapAnnotator",
|
"HeatMapAnnotator",
|
||||||
"IconAnnotator",
|
"IconAnnotator",
|
||||||
"ImageSink",
|
"ImageSink",
|
||||||
|
"ImageWindow",
|
||||||
"InferenceSlicer",
|
"InferenceSlicer",
|
||||||
"JSONSink",
|
"JSONSink",
|
||||||
"KeyPoints",
|
"KeyPoints",
|
||||||
|
|
@ -204,15 +224,21 @@ __all__ = [
|
||||||
"TraceAnnotator",
|
"TraceAnnotator",
|
||||||
"TriangleAnnotator",
|
"TriangleAnnotator",
|
||||||
"VertexAnnotator",
|
"VertexAnnotator",
|
||||||
|
"VertexEllipseAnnotator",
|
||||||
|
"VertexEllipseAreaAnnotator",
|
||||||
|
"VertexEllipseHaloAnnotator",
|
||||||
|
"VertexEllipseOutlineAnnotator",
|
||||||
"VertexLabelAnnotator",
|
"VertexLabelAnnotator",
|
||||||
"VideoInfo",
|
"VideoInfo",
|
||||||
"VideoSink",
|
"VideoSink",
|
||||||
|
"WindowedRasterDataset",
|
||||||
"approximate_polygon",
|
"approximate_polygon",
|
||||||
"box_iou",
|
"box_iou",
|
||||||
"box_iou_batch",
|
"box_iou_batch",
|
||||||
"box_iou_batch_with_jaccard",
|
"box_iou_batch_with_jaccard",
|
||||||
"box_non_max_merge",
|
"box_non_max_merge",
|
||||||
"box_non_max_suppression",
|
"box_non_max_suppression",
|
||||||
|
"box_soft_non_max_suppression",
|
||||||
"calculate_masks_centroids",
|
"calculate_masks_centroids",
|
||||||
"calculate_optimal_line_thickness",
|
"calculate_optimal_line_thickness",
|
||||||
"calculate_optimal_text_scale",
|
"calculate_optimal_text_scale",
|
||||||
|
|
@ -221,6 +247,7 @@ __all__ = [
|
||||||
"contains_multiple_segments",
|
"contains_multiple_segments",
|
||||||
"crop_image",
|
"crop_image",
|
||||||
"cv2_to_pillow",
|
"cv2_to_pillow",
|
||||||
|
"denormalize_boxes",
|
||||||
"draw_filled_polygon",
|
"draw_filled_polygon",
|
||||||
"draw_filled_rectangle",
|
"draw_filled_rectangle",
|
||||||
"draw_image",
|
"draw_image",
|
||||||
|
|
@ -242,15 +269,20 @@ __all__ = [
|
||||||
"is_valid_hex",
|
"is_valid_hex",
|
||||||
"letterbox_image",
|
"letterbox_image",
|
||||||
"list_files_with_extensions",
|
"list_files_with_extensions",
|
||||||
|
"load_image_from_url",
|
||||||
"mask_iou_batch",
|
"mask_iou_batch",
|
||||||
"mask_non_max_merge",
|
"mask_non_max_merge",
|
||||||
"mask_non_max_suppression",
|
"mask_non_max_suppression",
|
||||||
|
"mask_soft_non_max_suppression",
|
||||||
"mask_to_polygons",
|
"mask_to_polygons",
|
||||||
"mask_to_rle",
|
"mask_to_rle",
|
||||||
|
"mask_to_roi",
|
||||||
"mask_to_xyxy",
|
"mask_to_xyxy",
|
||||||
"move_boxes",
|
"move_boxes",
|
||||||
"move_masks",
|
"move_masks",
|
||||||
"oriented_box_iou_batch",
|
"oriented_box_iou_batch",
|
||||||
|
"oriented_box_non_max_merge",
|
||||||
|
"oriented_box_non_max_suppression",
|
||||||
"overlay_image",
|
"overlay_image",
|
||||||
"pad_boxes",
|
"pad_boxes",
|
||||||
"pillow_to_cv2",
|
"pillow_to_cv2",
|
||||||
|
|
@ -271,4 +303,15 @@ __all__ = [
|
||||||
"xyxy_to_polygons",
|
"xyxy_to_polygons",
|
||||||
"xyxy_to_xcycarh",
|
"xyxy_to_xcycarh",
|
||||||
"xyxy_to_xywh",
|
"xyxy_to_xywh",
|
||||||
|
"xyxyxyxy_to_xyxy",
|
||||||
]
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def __getattr__(name: str) -> Any:
|
||||||
|
"""Lazily resolve deprecated compatibility exports."""
|
||||||
|
if name == "ByteTrack":
|
||||||
|
from supervision.tracker.byte_tracker.core import ByteTrack as byte_track
|
||||||
|
|
||||||
|
globals()[name] = byte_track
|
||||||
|
return byte_track
|
||||||
|
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
|
||||||
|
|
|
||||||
|
|
@ -0,0 +1,302 @@
|
||||||
|
"""Private OpenCV compatibility surface used by Supervision."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import warnings
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
import numpy.typing as npt
|
||||||
|
|
||||||
|
from supervision._cv2._color import _cvt_color, _merge, _split
|
||||||
|
from supervision._cv2._common import BackendUnavailableError
|
||||||
|
from supervision._cv2._components import (
|
||||||
|
_connected_components,
|
||||||
|
_connected_components_with_stats,
|
||||||
|
)
|
||||||
|
from supervision._cv2._contours import _find_contours
|
||||||
|
from supervision._cv2._drawing import (
|
||||||
|
_circle,
|
||||||
|
_draw_contours,
|
||||||
|
_ellipse,
|
||||||
|
_fill_poly,
|
||||||
|
_line,
|
||||||
|
_polylines,
|
||||||
|
_rectangle,
|
||||||
|
)
|
||||||
|
from supervision._cv2._geometry import (
|
||||||
|
_approx_poly_dp,
|
||||||
|
_contour_area,
|
||||||
|
_intersect_convex_convex,
|
||||||
|
)
|
||||||
|
from supervision._cv2._image import (
|
||||||
|
_add_weighted,
|
||||||
|
_convert_scale_abs,
|
||||||
|
_copy_make_border,
|
||||||
|
_flip,
|
||||||
|
_imdecode,
|
||||||
|
_imencode,
|
||||||
|
_imread,
|
||||||
|
_imwrite,
|
||||||
|
_mean,
|
||||||
|
_resize,
|
||||||
|
)
|
||||||
|
from supervision._cv2._text import _get_text_size, _put_text
|
||||||
|
from supervision._cv2._transform import _blur
|
||||||
|
from supervision._cv2._video import (
|
||||||
|
_video_writer_fourcc,
|
||||||
|
_VideoCapture,
|
||||||
|
_VideoWriter,
|
||||||
|
)
|
||||||
|
from supervision._cv2.constants import (
|
||||||
|
_BORDER_CONSTANT,
|
||||||
|
_CAP_PROP_FPS,
|
||||||
|
_CAP_PROP_FRAME_COUNT,
|
||||||
|
_CAP_PROP_FRAME_HEIGHT,
|
||||||
|
_CAP_PROP_FRAME_WIDTH,
|
||||||
|
_CAP_PROP_POS_FRAMES,
|
||||||
|
_CC_STAT_AREA,
|
||||||
|
_CHAIN_APPROX_SIMPLE,
|
||||||
|
_COLOR_BGR2GRAY,
|
||||||
|
_COLOR_BGR2RGB,
|
||||||
|
_COLOR_GRAY2BGR,
|
||||||
|
_COLOR_HSV2BGR,
|
||||||
|
_COLOR_RGB2BGR,
|
||||||
|
_FONT_HERSHEY_COMPLEX,
|
||||||
|
_FONT_HERSHEY_COMPLEX_SMALL,
|
||||||
|
_FONT_HERSHEY_DUPLEX,
|
||||||
|
_FONT_HERSHEY_PLAIN,
|
||||||
|
_FONT_HERSHEY_SCRIPT_COMPLEX,
|
||||||
|
_FONT_HERSHEY_SCRIPT_SIMPLEX,
|
||||||
|
_FONT_HERSHEY_SIMPLEX,
|
||||||
|
_FONT_HERSHEY_TRIPLEX,
|
||||||
|
_FONT_ITALIC,
|
||||||
|
_IMREAD_COLOR,
|
||||||
|
_IMREAD_UNCHANGED,
|
||||||
|
_INTER_LINEAR,
|
||||||
|
_INTER_NEAREST,
|
||||||
|
_LINE_4,
|
||||||
|
_LINE_8,
|
||||||
|
_LINE_AA,
|
||||||
|
_RETR_TREE,
|
||||||
|
)
|
||||||
|
|
||||||
|
try:
|
||||||
|
import cv2
|
||||||
|
except (ImportError, OSError):
|
||||||
|
_IS_CV2_AVAILABLE = False
|
||||||
|
else:
|
||||||
|
_IS_CV2_AVAILABLE = True
|
||||||
|
|
||||||
|
if _IS_CV2_AVAILABLE:
|
||||||
|
from cv2 import ( # type: ignore[attr-defined]
|
||||||
|
BORDER_CONSTANT,
|
||||||
|
CAP_PROP_FPS,
|
||||||
|
CAP_PROP_FRAME_COUNT,
|
||||||
|
CAP_PROP_FRAME_HEIGHT,
|
||||||
|
CAP_PROP_FRAME_WIDTH,
|
||||||
|
CAP_PROP_POS_FRAMES,
|
||||||
|
CC_STAT_AREA,
|
||||||
|
COLOR_BGR2GRAY,
|
||||||
|
COLOR_BGR2RGB,
|
||||||
|
COLOR_GRAY2BGR,
|
||||||
|
COLOR_HSV2BGR,
|
||||||
|
COLOR_RGB2BGR,
|
||||||
|
FONT_HERSHEY_COMPLEX,
|
||||||
|
FONT_HERSHEY_COMPLEX_SMALL,
|
||||||
|
FONT_HERSHEY_DUPLEX,
|
||||||
|
FONT_HERSHEY_PLAIN,
|
||||||
|
FONT_HERSHEY_SCRIPT_COMPLEX,
|
||||||
|
FONT_HERSHEY_SCRIPT_SIMPLEX,
|
||||||
|
FONT_HERSHEY_SIMPLEX,
|
||||||
|
FONT_HERSHEY_TRIPLEX,
|
||||||
|
FONT_ITALIC,
|
||||||
|
IMREAD_COLOR,
|
||||||
|
IMREAD_UNCHANGED,
|
||||||
|
INTER_LINEAR,
|
||||||
|
INTER_NEAREST,
|
||||||
|
LINE_4,
|
||||||
|
LINE_8,
|
||||||
|
LINE_AA,
|
||||||
|
VideoCapture,
|
||||||
|
VideoWriter,
|
||||||
|
VideoWriter_fourcc, # type: ignore[attr-defined]
|
||||||
|
addWeighted,
|
||||||
|
approxPolyDP,
|
||||||
|
blur,
|
||||||
|
circle,
|
||||||
|
connectedComponents,
|
||||||
|
connectedComponentsWithStats,
|
||||||
|
contourArea,
|
||||||
|
convertScaleAbs,
|
||||||
|
copyMakeBorder,
|
||||||
|
cvtColor,
|
||||||
|
drawContours,
|
||||||
|
ellipse,
|
||||||
|
fillPoly,
|
||||||
|
flip,
|
||||||
|
getTextSize,
|
||||||
|
imdecode,
|
||||||
|
imencode,
|
||||||
|
imread,
|
||||||
|
imwrite,
|
||||||
|
intersectConvexConvex,
|
||||||
|
line,
|
||||||
|
mean,
|
||||||
|
merge,
|
||||||
|
polylines,
|
||||||
|
putText,
|
||||||
|
rectangle,
|
||||||
|
resize,
|
||||||
|
split,
|
||||||
|
)
|
||||||
|
from cv2 import (
|
||||||
|
findContours as _find_contours_impl,
|
||||||
|
)
|
||||||
|
|
||||||
|
BACKEND_NAME = "opencv"
|
||||||
|
else:
|
||||||
|
BACKEND_NAME = "fallback"
|
||||||
|
|
||||||
|
warnings.warn(
|
||||||
|
"OpenCV (`opencv-python`) is not installed; supervision is using its "
|
||||||
|
"pure NumPy fallback backend instead. Some operations may be slower "
|
||||||
|
"or behave slightly differently. Install `opencv-python` for full "
|
||||||
|
"performance and compatibility.",
|
||||||
|
stacklevel=2,
|
||||||
|
)
|
||||||
|
|
||||||
|
BORDER_CONSTANT = _BORDER_CONSTANT
|
||||||
|
CAP_PROP_FPS = _CAP_PROP_FPS
|
||||||
|
CAP_PROP_FRAME_COUNT = _CAP_PROP_FRAME_COUNT
|
||||||
|
CAP_PROP_FRAME_HEIGHT = _CAP_PROP_FRAME_HEIGHT
|
||||||
|
CAP_PROP_FRAME_WIDTH = _CAP_PROP_FRAME_WIDTH
|
||||||
|
CAP_PROP_POS_FRAMES = _CAP_PROP_POS_FRAMES
|
||||||
|
CC_STAT_AREA = _CC_STAT_AREA
|
||||||
|
COLOR_BGR2GRAY = _COLOR_BGR2GRAY
|
||||||
|
COLOR_BGR2RGB = _COLOR_BGR2RGB
|
||||||
|
COLOR_GRAY2BGR = _COLOR_GRAY2BGR
|
||||||
|
COLOR_HSV2BGR = _COLOR_HSV2BGR
|
||||||
|
COLOR_RGB2BGR = _COLOR_RGB2BGR
|
||||||
|
FONT_HERSHEY_COMPLEX = _FONT_HERSHEY_COMPLEX
|
||||||
|
FONT_HERSHEY_COMPLEX_SMALL = _FONT_HERSHEY_COMPLEX_SMALL
|
||||||
|
FONT_HERSHEY_DUPLEX = _FONT_HERSHEY_DUPLEX
|
||||||
|
FONT_HERSHEY_PLAIN = _FONT_HERSHEY_PLAIN
|
||||||
|
FONT_HERSHEY_SCRIPT_COMPLEX = _FONT_HERSHEY_SCRIPT_COMPLEX
|
||||||
|
FONT_HERSHEY_SCRIPT_SIMPLEX = _FONT_HERSHEY_SCRIPT_SIMPLEX
|
||||||
|
FONT_HERSHEY_SIMPLEX = _FONT_HERSHEY_SIMPLEX
|
||||||
|
FONT_HERSHEY_TRIPLEX = _FONT_HERSHEY_TRIPLEX
|
||||||
|
FONT_ITALIC = _FONT_ITALIC
|
||||||
|
IMREAD_COLOR = _IMREAD_COLOR
|
||||||
|
IMREAD_UNCHANGED = _IMREAD_UNCHANGED
|
||||||
|
INTER_LINEAR = _INTER_LINEAR
|
||||||
|
INTER_NEAREST = _INTER_NEAREST
|
||||||
|
LINE_4 = _LINE_4
|
||||||
|
LINE_8 = _LINE_8
|
||||||
|
LINE_AA = _LINE_AA
|
||||||
|
|
||||||
|
# Fallback implementations when cv2 is not available. Suppress type errors because
|
||||||
|
# fallback types differ from cv2 types, but are functionally equivalent.
|
||||||
|
VideoCapture = _VideoCapture # type: ignore[assignment,misc]
|
||||||
|
VideoWriter = _VideoWriter # type: ignore[assignment,misc]
|
||||||
|
VideoWriter_fourcc = _video_writer_fourcc # type: ignore[assignment]
|
||||||
|
addWeighted = _add_weighted # type: ignore[assignment]
|
||||||
|
approxPolyDP = _approx_poly_dp # type: ignore[assignment]
|
||||||
|
blur = _blur # type: ignore[assignment]
|
||||||
|
circle = _circle # type: ignore[assignment]
|
||||||
|
connectedComponents = _connected_components # type: ignore[assignment]
|
||||||
|
connectedComponentsWithStats = _connected_components_with_stats # type: ignore[assignment]
|
||||||
|
contourArea = _contour_area # type: ignore[assignment]
|
||||||
|
convertScaleAbs = _convert_scale_abs # type: ignore[assignment]
|
||||||
|
copyMakeBorder = _copy_make_border # type: ignore[assignment]
|
||||||
|
cvtColor = _cvt_color # type: ignore[assignment]
|
||||||
|
drawContours = _draw_contours # type: ignore[assignment]
|
||||||
|
ellipse = _ellipse # type: ignore[assignment]
|
||||||
|
fillPoly = _fill_poly # type: ignore[assignment]
|
||||||
|
_find_contours_impl = _find_contours
|
||||||
|
flip = _flip # type: ignore[assignment]
|
||||||
|
getTextSize = _get_text_size # type: ignore[assignment]
|
||||||
|
imdecode = _imdecode # type: ignore[assignment]
|
||||||
|
imencode = _imencode # type: ignore[assignment]
|
||||||
|
imread = _imread # type: ignore[assignment]
|
||||||
|
imwrite = _imwrite # type: ignore[assignment]
|
||||||
|
intersectConvexConvex = _intersect_convex_convex # type: ignore[assignment]
|
||||||
|
line = _line # type: ignore[assignment]
|
||||||
|
mean = _mean # type: ignore[assignment]
|
||||||
|
merge = _merge # type: ignore[assignment]
|
||||||
|
polylines = _polylines # type: ignore[assignment]
|
||||||
|
putText = _put_text # type: ignore[assignment]
|
||||||
|
rectangle = _rectangle # type: ignore[assignment]
|
||||||
|
resize = _resize # type: ignore[assignment]
|
||||||
|
split = _split # type: ignore[assignment]
|
||||||
|
|
||||||
|
|
||||||
|
def find_contours(image: npt.NDArray[Any]) -> list[npt.NDArray[Any]]:
|
||||||
|
"""Return the contour geometry required by mask-to-polygon conversion."""
|
||||||
|
contours, _ = _find_contours_impl(image, _RETR_TREE, _CHAIN_APPROX_SIMPLE)
|
||||||
|
return list(contours)
|
||||||
|
|
||||||
|
|
||||||
|
__all__ = [
|
||||||
|
"BACKEND_NAME",
|
||||||
|
"BORDER_CONSTANT",
|
||||||
|
"CAP_PROP_FPS",
|
||||||
|
"CAP_PROP_FRAME_COUNT",
|
||||||
|
"CAP_PROP_FRAME_HEIGHT",
|
||||||
|
"CAP_PROP_FRAME_WIDTH",
|
||||||
|
"CAP_PROP_POS_FRAMES",
|
||||||
|
"CC_STAT_AREA",
|
||||||
|
"COLOR_BGR2GRAY",
|
||||||
|
"COLOR_BGR2RGB",
|
||||||
|
"COLOR_GRAY2BGR",
|
||||||
|
"COLOR_HSV2BGR",
|
||||||
|
"COLOR_RGB2BGR",
|
||||||
|
"FONT_HERSHEY_COMPLEX",
|
||||||
|
"FONT_HERSHEY_COMPLEX_SMALL",
|
||||||
|
"FONT_HERSHEY_DUPLEX",
|
||||||
|
"FONT_HERSHEY_PLAIN",
|
||||||
|
"FONT_HERSHEY_SCRIPT_COMPLEX",
|
||||||
|
"FONT_HERSHEY_SCRIPT_SIMPLEX",
|
||||||
|
"FONT_HERSHEY_SIMPLEX",
|
||||||
|
"FONT_HERSHEY_TRIPLEX",
|
||||||
|
"FONT_ITALIC",
|
||||||
|
"IMREAD_COLOR",
|
||||||
|
"IMREAD_UNCHANGED",
|
||||||
|
"INTER_LINEAR",
|
||||||
|
"INTER_NEAREST",
|
||||||
|
"LINE_4",
|
||||||
|
"LINE_8",
|
||||||
|
"LINE_AA",
|
||||||
|
"BackendUnavailableError",
|
||||||
|
"VideoCapture",
|
||||||
|
"VideoWriter",
|
||||||
|
"VideoWriter_fourcc",
|
||||||
|
"addWeighted",
|
||||||
|
"approxPolyDP",
|
||||||
|
"blur",
|
||||||
|
"circle",
|
||||||
|
"connectedComponents",
|
||||||
|
"connectedComponentsWithStats",
|
||||||
|
"contourArea",
|
||||||
|
"convertScaleAbs",
|
||||||
|
"copyMakeBorder",
|
||||||
|
"cvtColor",
|
||||||
|
"drawContours",
|
||||||
|
"ellipse",
|
||||||
|
"fillPoly",
|
||||||
|
"find_contours",
|
||||||
|
"flip",
|
||||||
|
"getTextSize",
|
||||||
|
"imdecode",
|
||||||
|
"imencode",
|
||||||
|
"imread",
|
||||||
|
"imwrite",
|
||||||
|
"intersectConvexConvex",
|
||||||
|
"line",
|
||||||
|
"mean",
|
||||||
|
"merge",
|
||||||
|
"polylines",
|
||||||
|
"putText",
|
||||||
|
"rectangle",
|
||||||
|
"resize",
|
||||||
|
"split",
|
||||||
|
]
|
||||||
|
|
@ -0,0 +1,94 @@
|
||||||
|
"""Private color and channel-operation fallbacks."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from collections.abc import Sequence
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
import numpy as np
|
||||||
|
import numpy.typing as npt
|
||||||
|
|
||||||
|
from supervision._cv2._common import _cast_array_like_opencv
|
||||||
|
from supervision._cv2.constants import (
|
||||||
|
_COLOR_BGR2GRAY,
|
||||||
|
_COLOR_BGR2RGB,
|
||||||
|
_COLOR_GRAY2BGR,
|
||||||
|
_COLOR_HSV2BGR,
|
||||||
|
_COLOR_RGB2BGR,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _cvt_color(image: npt.NDArray[Any], code: int) -> npt.NDArray[Any]:
|
||||||
|
"""Convert the BGR, RGB, grayscale, and 8-bit HSV formats used by Supervision."""
|
||||||
|
if code in (_COLOR_BGR2RGB, _COLOR_RGB2BGR):
|
||||||
|
if image.ndim != 3 or image.shape[2] != 3:
|
||||||
|
raise ValueError("BGR/RGB conversion requires a three-channel image")
|
||||||
|
return np.ascontiguousarray(image[..., ::-1])
|
||||||
|
|
||||||
|
if code == _COLOR_GRAY2BGR:
|
||||||
|
if image.ndim != 2:
|
||||||
|
raise ValueError("GRAY2BGR conversion requires a two-dimensional image")
|
||||||
|
return np.repeat(image[..., np.newaxis], 3, axis=2)
|
||||||
|
|
||||||
|
if code == _COLOR_BGR2GRAY:
|
||||||
|
if image.ndim != 3 or image.shape[2] != 3:
|
||||||
|
raise ValueError("BGR2GRAY conversion requires a three-channel image")
|
||||||
|
if image.dtype == np.uint8:
|
||||||
|
values = image.astype(np.uint32)
|
||||||
|
weighted = (
|
||||||
|
values[..., 0] * 3735
|
||||||
|
+ values[..., 1] * 19235
|
||||||
|
+ values[..., 2] * 9798
|
||||||
|
+ (1 << 14)
|
||||||
|
) >> 15
|
||||||
|
return weighted.astype(np.uint8)
|
||||||
|
float_values = (
|
||||||
|
image[..., 0].astype(np.float64) * 0.114
|
||||||
|
+ image[..., 1].astype(np.float64) * 0.587
|
||||||
|
+ image[..., 2].astype(np.float64) * 0.299
|
||||||
|
)
|
||||||
|
return _cast_array_like_opencv(float_values, image.dtype)
|
||||||
|
|
||||||
|
if code == _COLOR_HSV2BGR:
|
||||||
|
if image.ndim != 3 or image.shape[2] != 3:
|
||||||
|
raise ValueError("HSV2BGR conversion requires a three-channel image")
|
||||||
|
return _hsv_to_bgr(image)
|
||||||
|
|
||||||
|
raise ValueError(f"Unsupported color conversion code: {code}")
|
||||||
|
|
||||||
|
|
||||||
|
def _hsv_to_bgr(image: npt.NDArray[Any]) -> npt.NDArray[Any]:
|
||||||
|
"""Convert OpenCV's 8-bit HSV representation to BGR."""
|
||||||
|
values = image.astype(np.float64)
|
||||||
|
hue = values[..., 0] / 30.0
|
||||||
|
saturation = values[..., 1] / 255.0
|
||||||
|
value = values[..., 2] / 255.0
|
||||||
|
|
||||||
|
chroma = value * saturation
|
||||||
|
sector_index = np.floor(hue).astype(np.int64) % 6
|
||||||
|
sector = hue - np.floor(hue)
|
||||||
|
x = chroma * (1 - np.abs(((sector_index + sector) % 2) - 1))
|
||||||
|
match = value - chroma
|
||||||
|
zeros = np.zeros_like(chroma)
|
||||||
|
|
||||||
|
red = np.choose(sector_index, (chroma, x, zeros, zeros, x, chroma))
|
||||||
|
green = np.choose(sector_index, (x, chroma, chroma, x, zeros, zeros))
|
||||||
|
blue = np.choose(sector_index, (zeros, zeros, x, chroma, chroma, x))
|
||||||
|
bgr = np.stack((blue + match, green + match, red + match), axis=-1) * 255
|
||||||
|
return _cast_array_like_opencv(bgr, image.dtype)
|
||||||
|
|
||||||
|
|
||||||
|
def _split(image: npt.NDArray[Any]) -> tuple[npt.NDArray[Any], ...]:
|
||||||
|
"""Split an image into contiguous single-channel arrays."""
|
||||||
|
if image.ndim == 2:
|
||||||
|
return (np.ascontiguousarray(image),)
|
||||||
|
return tuple(
|
||||||
|
np.ascontiguousarray(image[..., index]) for index in range(image.shape[2])
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _merge(channels: Sequence[npt.NDArray[Any]]) -> npt.NDArray[Any]:
|
||||||
|
"""Merge single-channel arrays along their final axis."""
|
||||||
|
if not channels:
|
||||||
|
raise ValueError("At least one channel is required")
|
||||||
|
return np.ascontiguousarray(np.stack(channels, axis=-1))
|
||||||
Some files were not shown because too many files have changed in this diff Show More
Loading…
Reference in New Issue