initial commit
This commit is contained in:
parent
4729e20a94
commit
574c61b6b0
|
|
@ -15,7 +15,7 @@ source image.
|
|||
|
||||

|
||||
|
||||
## Run Inference
|
||||
## Run Detection
|
||||
|
||||
First, you'll need to obtain predictions from your object detection or segmentation
|
||||
model.
|
||||
|
|
|
|||
|
|
@ -0,0 +1,124 @@
|
|||
---
|
||||
comments: true
|
||||
status: new
|
||||
---
|
||||
|
||||
# Save Detections
|
||||
|
||||
TODO
|
||||
|
||||
## Run Detection
|
||||
|
||||
=== "Inference"
|
||||
|
||||
```python
|
||||
import cv2
|
||||
from inference import get_model
|
||||
|
||||
model = get_model(model_id="yolov8n-640")
|
||||
image = cv2.imread(<SOURCE_IMAGE_APTH>)
|
||||
results = model.infer(image)[0]
|
||||
```
|
||||
|
||||
=== "Ultralytics"
|
||||
|
||||
```python
|
||||
import cv2
|
||||
from ultralytics import YOLO
|
||||
|
||||
model = YOLO("yolov8n.pt")
|
||||
image = cv2.imread(<SOURCE_IMAGE_APTH>)
|
||||
results = model(image)[0]
|
||||
```
|
||||
|
||||
=== "Transformers"
|
||||
|
||||
```python
|
||||
import torch
|
||||
from PIL import Image
|
||||
from transformers import DetrImageProcessor, DetrForObjectDetection
|
||||
|
||||
processor = DetrImageProcessor.from_pretrained("facebook/detr-resnet-50")
|
||||
model = DetrForObjectDetection.from_pretrained("facebook/detr-resnet-50")
|
||||
|
||||
image = Image.open(<SOURCE_IMAGE_APTH>)
|
||||
inputs = processor(images=image, return_tensors="pt")
|
||||
|
||||
with torch.no_grad():
|
||||
outputs = model(**inputs)
|
||||
|
||||
width, height = image.size
|
||||
target_size = torch.tensor([[height, width]])
|
||||
results = processor.post_process_object_detection(
|
||||
outputs=outputs, target_sizes=target_size)[0]
|
||||
```
|
||||
|
||||
## Save Detections as CSV
|
||||
|
||||
TODO
|
||||
|
||||
=== "Inference"
|
||||
|
||||
```python
|
||||
import supervision as sv
|
||||
from inference import get_model
|
||||
|
||||
model = get_model(model_id="yolov8n-640")
|
||||
|
||||
with sv.CSVSink(<TARGET_CSV_PATH>) as sink:
|
||||
for frame in sv.get_video_frames_generator(<SOURCE_VIDEO_PATH>):
|
||||
|
||||
results = model.infer(image)[0]
|
||||
detections = sv.Detections.from_inference(results)
|
||||
sink.append(detections, {})
|
||||
```
|
||||
|
||||
=== "Ultralytics"
|
||||
|
||||
```python
|
||||
import supervision as sv
|
||||
from ultralytics import YOLO
|
||||
|
||||
model = YOLO("yolov8n.pt")
|
||||
|
||||
with sv.CSVSink(<TARGET_CSV_PATH>) as sink:
|
||||
for frame in sv.get_video_frames_generator(<SOURCE_VIDEO_PATH>):
|
||||
|
||||
results = model(frame)[0]
|
||||
detections = sv.Detections.from_ultralytics(results)
|
||||
sink.append(detections, {})
|
||||
```
|
||||
|
||||
=== "Transformers"
|
||||
|
||||
```python
|
||||
import torch
|
||||
from PIL import Image
|
||||
from transformers import DetrImageProcessor, DetrForObjectDetection
|
||||
|
||||
processor = DetrImageProcessor.from_pretrained("facebook/detr-resnet-50")
|
||||
model = DetrForObjectDetection.from_pretrained("facebook/detr-resnet-50")
|
||||
|
||||
image = Image.open(<SOURCE_IMAGE_APTH>)
|
||||
inputs = processor(images=image, return_tensors="pt")
|
||||
|
||||
with torch.no_grad():
|
||||
outputs = model(**inputs)
|
||||
|
||||
width, height = image.size
|
||||
target_size = torch.tensor([[height, width]])
|
||||
results = processor.post_process_object_detection(
|
||||
outputs=outputs, target_sizes=target_size)[0]
|
||||
```
|
||||
|
||||
## Custom Fields
|
||||
|
||||
TODO
|
||||
|
||||
## Save Detections as JSON
|
||||
|
||||
TODO
|
||||
|
||||
## Process Video and Save Detections
|
||||
|
||||
TODO
|
||||
|
|
@ -38,9 +38,11 @@ nav:
|
|||
- Home: index.md
|
||||
- How to:
|
||||
- Detect and Annotate: how_to/detect_and_annotate.md
|
||||
- Save Detections: how_to/save_detections.md
|
||||
- Filter Detections: how_to/filter_detections.md
|
||||
- Detect Small Objects: how_to/detect_small_objects.md
|
||||
- Track Objects: how_to/track_objects.md
|
||||
- Filter Detections: how_to/filter_detections.md
|
||||
|
||||
- API:
|
||||
- Annotators: annotators.md
|
||||
- Classifications:
|
||||
|
|
|
|||
|
|
@ -1,7 +1,7 @@
|
|||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from typing import List, Tuple
|
||||
from typing import List, Tuple, Union
|
||||
|
||||
import matplotlib.pyplot as plt
|
||||
|
||||
|
|
@ -448,3 +448,19 @@ class ColorPalette:
|
|||
raise ValueError("idx argument should not be negative")
|
||||
idx = idx % len(self.colors)
|
||||
return self.colors[idx]
|
||||
|
||||
|
||||
def unify_to_bgr(color: Union[Tuple[int, int, int], Color]) -> Tuple[int, int, int]:
|
||||
"""
|
||||
Converts a color input in multiple formats to a standardized BGR format.
|
||||
|
||||
Args:
|
||||
color (Union[Tuple[int, int, int], Color]): The color input to be converted,
|
||||
which can be either a tuple of RGB values or an instance of a Color class.
|
||||
|
||||
Returns:
|
||||
Tuple[int, int, int]: The color in BGR format as a tuple of three integers.
|
||||
"""
|
||||
if issubclass(type(color), Color):
|
||||
return color.as_bgr()
|
||||
return color
|
||||
|
|
|
|||
|
|
@ -9,7 +9,7 @@ import cv2
|
|||
import numpy as np
|
||||
|
||||
from supervision.annotators.base import ImageType
|
||||
from supervision.draw.color import Color
|
||||
from supervision.draw.color import Color, unify_to_bgr
|
||||
from supervision.draw.utils import calculate_optimal_text_scale, draw_text
|
||||
from supervision.geometry.core import Point
|
||||
from supervision.utils.conversion import (
|
||||
|
|
@ -25,17 +25,18 @@ MAX_COLUMNS_FOR_SINGLE_ROW_GRID = 3
|
|||
|
||||
|
||||
@convert_for_image_processing
|
||||
def crop_image(image: np.ndarray, xyxy: np.ndarray) -> np.ndarray:
|
||||
def crop_image(image: ImageType, xyxy: np.ndarray) -> np.ndarray:
|
||||
"""
|
||||
Crops the given image based on the given bounding box.
|
||||
|
||||
Args:
|
||||
image (np.ndarray): The image to be cropped, represented as a numpy array.
|
||||
image (ImageType): The image to be cropped. `ImageType` is a flexible type,
|
||||
accepting either `numpy.ndarray` or `PIL.Image.Image`.
|
||||
xyxy (np.ndarray): A numpy array containing the bounding box coordinates
|
||||
in the format (x1, y1, x2, y2).
|
||||
|
||||
Returns:
|
||||
(np.ndarray): The cropped image as a numpy array.
|
||||
(ImageType): The cropped image.
|
||||
|
||||
Examples:
|
||||
```python
|
||||
|
|
@ -74,10 +75,10 @@ def resize_image(image: np.ndarray, scale_factor: float) -> np.ndarray:
|
|||
raise ValueError("Scale factor must be positive.")
|
||||
|
||||
old_width, old_height = image.shape[1], image.shape[0]
|
||||
nwe_width = int(old_width * scale_factor)
|
||||
new_width = int(old_width * scale_factor)
|
||||
new_height = int(old_height * scale_factor)
|
||||
|
||||
return cv2.resize(image, (nwe_width, new_height), interpolation=cv2.INTER_LINEAR)
|
||||
return cv2.resize(image, (new_width, new_height), interpolation=cv2.INTER_LINEAR)
|
||||
|
||||
|
||||
def place_image(
|
||||
|
|
@ -285,14 +286,14 @@ def create_tiles(
|
|||
raise ValueError("Could not create image tiles from empty list of images.")
|
||||
if return_type == "auto":
|
||||
return_type = _negotiate_tiles_format(images=images)
|
||||
tile_padding_color = _color_to_bgr(color=tile_padding_color)
|
||||
tile_margin_color = _color_to_bgr(color=tile_margin_color)
|
||||
tile_padding_color = unify_to_bgr(color=tile_padding_color)
|
||||
tile_margin_color = unify_to_bgr(color=tile_margin_color)
|
||||
images = images_to_cv2(images=images)
|
||||
if single_tile_size is None:
|
||||
single_tile_size = _aggregate_images_shape(images=images, mode=tile_scaling)
|
||||
resized_images = [
|
||||
letterbox_image(
|
||||
image=i, desired_size=single_tile_size, color=tile_padding_color
|
||||
image=i, target_resolution_wh=single_tile_size, color=tile_padding_color
|
||||
)
|
||||
for i in images
|
||||
]
|
||||
|
|
@ -311,8 +312,8 @@ def create_tiles(
|
|||
titles_anchors = fill(
|
||||
sequence=titles_anchors, desired_size=len(images), content=None
|
||||
)
|
||||
titles_color = _color_to_bgr(color=titles_color)
|
||||
titles_background_color = _color_to_bgr(color=titles_background_color)
|
||||
titles_color = unify_to_bgr(color=titles_color)
|
||||
titles_background_color = unify_to_bgr(color=titles_background_color)
|
||||
tiles = _generate_tiles(
|
||||
images=resized_images,
|
||||
grid_size=grid_size,
|
||||
|
|
@ -546,8 +547,8 @@ def _generate_color_image(
|
|||
|
||||
@convert_for_image_processing
|
||||
def letterbox_image(
|
||||
image: np.ndarray,
|
||||
desired_size: Tuple[int, int],
|
||||
image: ImageType,
|
||||
target_resolution_wh: Tuple[int, int],
|
||||
color: Union[Tuple[int, int, int], Color] = (0, 0, 0),
|
||||
) -> np.ndarray:
|
||||
"""
|
||||
|
|
@ -555,27 +556,27 @@ def letterbox_image(
|
|||
ratio, adding padding of given color if needed to maintain aspect ratio.
|
||||
|
||||
Args:
|
||||
image (np.ndarray): Input image (type will be adjusted by decorator,
|
||||
image (ImageType): Input image (type will be adjusted by decorator,
|
||||
you can provide PIL.Image)
|
||||
desired_size (Tuple[int, int]): image size (width, height) representing
|
||||
target_resolution_wh (Tuple[int, int]): image size (width, height) representing
|
||||
the target dimensions.
|
||||
color (Union[Tuple[int, int, int], Color]): the color to pad with - If
|
||||
tuple provided - should be BGR.
|
||||
|
||||
Returns:
|
||||
np.ndarray: letterboxed image (type may be adjusted to PIL.Image by
|
||||
ImageType: letterboxed image (type may be adjusted to PIL.Image by
|
||||
decorator if function was called with PIL.Image)
|
||||
"""
|
||||
color = _color_to_bgr(color=color)
|
||||
color = unify_to_bgr(color=color)
|
||||
resized_img = resize_image_keeping_aspect_ratio(
|
||||
image=image,
|
||||
desired_size=desired_size,
|
||||
desired_size=target_resolution_wh,
|
||||
)
|
||||
new_height, new_width = resized_img.shape[:2]
|
||||
top_padding = (desired_size[1] - new_height) // 2
|
||||
bottom_padding = desired_size[1] - new_height - top_padding
|
||||
left_padding = (desired_size[0] - new_width) // 2
|
||||
right_padding = desired_size[0] - new_width - left_padding
|
||||
top_padding = (target_resolution_wh[1] - new_height) // 2
|
||||
bottom_padding = target_resolution_wh[1] - new_height - top_padding
|
||||
left_padding = (target_resolution_wh[0] - new_width) // 2
|
||||
right_padding = target_resolution_wh[0] - new_width - left_padding
|
||||
return cv2.copyMakeBorder(
|
||||
resized_img,
|
||||
top_padding,
|
||||
|
|
@ -625,9 +626,3 @@ def resize_image_keeping_aspect_ratio(
|
|||
new_height = desired_size[1]
|
||||
new_width = int(desired_size[1] * img_ratio)
|
||||
return cv2.resize(image, (new_width, new_height))
|
||||
|
||||
|
||||
def _color_to_bgr(color: Union[Tuple[int, int, int], Color]) -> Tuple[int, int, int]:
|
||||
if issubclass(type(color), Color):
|
||||
return color.as_bgr()
|
||||
return color
|
||||
|
|
|
|||
|
|
@ -62,7 +62,7 @@ def test_letterbox_image_for_opencv_image() -> None:
|
|||
|
||||
# when
|
||||
result = letterbox_image(
|
||||
image=image, desired_size=(1024, 1024), color=(255, 255, 255)
|
||||
image=image, target_resolution_wh=(1024, 1024), color=(255, 255, 255)
|
||||
)
|
||||
|
||||
# then
|
||||
|
|
@ -88,7 +88,7 @@ def test_letterbox_image_for_pillow_image() -> None:
|
|||
|
||||
# when
|
||||
result = letterbox_image(
|
||||
image=image, desired_size=(1024, 1024), color=(255, 255, 255)
|
||||
image=image, target_resolution_wh=(1024, 1024), color=(255, 255, 255)
|
||||
)
|
||||
|
||||
# then
|
||||
|
|
|
|||
Loading…
Reference in New Issue