Inspect any ONNX model in your browser

Computer vision demo · Object tracking

Bird's-Eye View Object Tracking with YOLO26 and Homography

Watch YOLO26 and TrackTrack follow road users as OpenCV homography projects their estimated ground-contact points onto a top-down map with motion trails.

Watch tracked objects in the camera footage alongside their estimated positions and movement trails on a calibrated bird’s-eye map.

How the bird's-eye view works

The Python pipeline calibrates a road plane, detects and tracks vehicles with YOLO26, projects approximate ground-contact points with homography, and renders their positions on a top-down map. Follow the complete implementation below, from imports through video export.

Python implementation, step by step

The complete original Python source is presented below in file order, divided into eight readable sections. Read the sections in sequence to reconstruct the full script, including its imports, configuration, helper methods, tracking, rendering, and entry point.

1. Imports, configuration and tracker state

The script imports OpenCV, NumPy, Ultralytics YOLO, pathlib and deque before defining its model and video defaults. BirdEyeTracker stores the output layout, map dimensions, confidence threshold, tracker configuration and visualization preferences in one place. It also creates independent state for each tracking ID, including recent positions, direction and motion trails. Input validation catches unsupported layouts, invalid dimensions and incompatible map settings before video processing begins.

Python
"""
Track objects and display approximate 2D BEV symbols, not 3D cuboids.

# Steps
1- GroundPlane Calibration - selection of exactly 4 points for homography.
2- BEV Background
3- Object Tracking
4- Ground Projection
5- BEV Shapes
6- Frame Composition 
7- Video Output
"""


from collections import deque
from pathlib import Path

import cv2
import numpy as np
from ultralytics import YOLO


DEFAULT_MODEL = "yolo26s-drone.pt"
DEFAULT_VIDEO = "bev-raw-video.mp4"
VEHICLE_CLASSES = {"car", "truck", "bus", "motorcycle", "bicycle", "train", "vehicle"}
DARK = (18, 18, 18)
VIDEO_BACKGROUND = (104, 31, 17)  # #111F68 in OpenCV BGR.
WHITE = (245, 245, 245)
ID_BACKGROUND = (255, 42, 17)  # Yellow in BGR.
ID_TEXT = (255, 255, 255)  # Navy blue in BGR.
SELECTION_ORANGE = (0, 165, 255)  # Orange in OpenCV BGR.
BEV_DARK_YELLOW = (0, 164, 201)  # Dark yellow (#C9A400) in OpenCV BGR.



class BirdEyeTracker:
    """Calibrate, track, render BEV symbols, and write video in one class."""

    SYMBOL_SIZES = {"car": (10, 24), "truck": (13, 36), "bus": (13, 40),
                    "train": (14, 44), "motorcycle": (6, 14), "bicycle": (5, 13),
                    "person": (7, 9)}

    def __init__(
        self, video_path, model_path, output_path="birds_eye_tracking.mp4",
        corners=None, tracker="tracktrack.yaml", bird_width=640, bird_height=360,
        show=False, bird_only=False, map_ids=False, conf=0.25, classes=None, device=None,
        map_style="road", road_background=None, trail_length=45,
        layout="dashboard", dashboard_width=1920, dashboard_height=1080, map_marker="footprint",
    ):
        """Initialize video settings, layout, calibration, and per-track state."""

        self.video_path, self.model_path, self.output_path = video_path, model_path, output_path
        self.input_corners, self.show = corners, show
        self.layout, self.bird_only = layout, bird_only
        self.camera_width = 3 * dashboard_width // 4
        self.camera_height = dashboard_height
        dashboard = layout == "dashboard" and not bird_only
        self.bird_width = dashboard_width - self.camera_width if dashboard else bird_width
        self.bird_height = dashboard_height if dashboard else bird_height
        self.output_width = self.bird_width if bird_only else dashboard_width if dashboard else 2 * self.bird_width
        self.output_height = dashboard_height if dashboard else self.bird_height
        self.bev_header_height = min(64, self.bird_height // 4)
        self.map_style, self.road_background = map_style, road_background
        self.trail_length, self.map_marker, self.map_ids = trail_length, map_marker, map_ids
        self.track_options = dict(persist=True, tracker=tracker, conf=conf, classes=classes, verbose=False)
        if device is not None:
            self.track_options["device"] = device
        scale = max(0.5, min(2.0, self.bird_height / 540))
        self.symbol_sizes = {name: (width * scale, length * scale)
                             for name, (width, length) in self.SYMBOL_SIZES.items()}
        self.default_symbol_size = (10 * scale, 22 * scale)
        self.frame_width = self.frame_height = 0
        self.corners = self.homography = self.background = None
        self.corner_pixels = None
        self.capture = self.writer = self.model = None
        self.states = {}
        self.frame_index = 0

    def _new_track_state(self):
        """Create independent smoothing, direction, trail, and expiry state for one ID."""
        return {"position": None, "angle": 0.0, "last_seen": -1,
                "positions": deque(maxlen=8), "trail": deque(maxlen=self.trail_length)}


    def _validate_options(self):
        """
        Reject invalid layouts, map modes, dimensions, and trail lengths. Validate before opening 
        video resources; MP4 output dimensions must be even and large enough for the selected layout.
        """
        if self.layout not in {"dashboard", "side"}:
            raise SystemExit("Layout must be 'dashboard' or 'side'")
        if self.bird_width < 160 or self.bird_height < 160:
            raise SystemExit("Bird's-eye width and height must be at least 160 pixels")
        if self.output_width % 2 or self.output_height % 2:
            raise SystemExit("Output video dimensions must be even")
        if self.layout == "dashboard" and not self.bird_only:
            if self.output_width < 640 or self.output_height < 480:
                raise SystemExit("Dashboard dimensions must be at least 640 x 480")
        if self.map_style not in {"road", "pastel"}:
            raise SystemExit("Map style must be 'road' or 'pastel'")
        if self.map_marker not in {"footprint", "dot"}:
            raise SystemExit("Map marker must be 'footprint' or 'dot'")
        if not isinstance(self.trail_length, int) or self.trail_length < 0:
            raise SystemExit("Trail length must be a nonnegative integer")

2. Video input and four-point calibration

This section opens the input video and reads the first frame to establish its dimensions. You can provide four road-plane corners directly or select them interactively in top-left, top-right, bottom-right and bottom-left order. The calibration logic checks the supplied coordinates and geometry before using cv2.getPerspectiveTransform to compute the camera-to-map homography. Because the mapping assumes a planar surface, the chosen corners should lie on the road rather than on elevated objects.

Python
    def _open_video(self):
        """Open the configured video file, stream, or numeric camera device."""

        source = str(self.video_path)
        self.capture = cv2.VideoCapture(int(source) if source.isdecimal() else source)
        if not self.capture.isOpened():
            raise SystemExit(f"Could not open video source: {self.video_path}")
        ok, frame = self.capture.read()
        if not ok:
            raise SystemExit("Video source contains no readable frames")
        return frame


    def _select_corners(self, frame):
        """Select exactly four road corners; ignore further clicks until undo or reset."""

        scale = min(1.0, 1920 / self.frame_width, 720 / self.frame_height)
        preview = cv2.resize(frame, (round(self.frame_width * scale), round(self.frame_height * scale)))
        picked = []
        redraw = True
        window = "Select 4 road corners | Enter: confirm  R: reset  Backspace: undo  Esc: quit"

        def click(event, x, y, _flags, _data):
            """Record up to four road corners and refresh the selection overlay."""
            nonlocal redraw
            if event == cv2.EVENT_LBUTTONDOWN and len(picked) < 4:
                picked.append((x / scale, y / scale))
                redraw = True

        cv2.namedWindow(window, cv2.WINDOW_NORMAL)
        cv2.setMouseCallback(window, click)
        try:
            while True:
                if redraw:
                    display = preview.copy()
                    for index, (x, y) in enumerate(picked):
                        point = (round(x * scale), round(y * scale))
                        cv2.circle(display, point, 6, SELECTION_ORANGE, -1)
                        cv2.putText(display, str(index + 1), (point[0] + 8, point[1] - 8),
                                    cv2.FONT_HERSHEY_SIMPLEX, 0.7, SELECTION_ORANGE, 2)
                    redraw = False
                cv2.imshow(window, display)
                key = cv2.waitKey(30) & 0xFF
                if key in (13, 10) and len(picked) == 4:
                    return np.float32(picked)
                if key == ord("r"):
                    picked.clear()
                    redraw = True
                if key in (8, 127) and picked:
                    picked.pop()
                    redraw = True
                if key == 27 or cv2.getWindowProperty(window, cv2.WND_PROP_VISIBLE) < 1:
                    raise SystemExit("Calibration cancelled")
        finally:
            cv2.destroyWindow(window)


    @staticmethod
    def _point_array(points, name):
        """Validate and return exactly four finite (x, y) pairs as a float32 array."""
        try:
            array = np.asarray(points, dtype=np.float32)
        except (TypeError, ValueError) as error:
            raise SystemExit(f"{name} must contain exactly four (x, y) pairs") from error
        if array.ndim != 2 or array.shape[1] != 2 or len(array) != 4:
            raise SystemExit(f"{name} must contain exactly four (x, y) pairs")
        if not np.isfinite(array).all():
            raise SystemExit(f"{name} must contain finite coordinates")
        return array


    def _calibrate(self, frame, corners=None):
        """Validate four road corners and map them to the inset BEV rectangle.

        Mouse selection is used by default. Corners must be inside the video
        and form a convex TL, TR, BR, BL quadrilateral with nonzero area.
        """
        self.frame_height, self.frame_width = frame.shape[:2]
        points = (self._point_array(corners, "Source points") if corners is not None
                  else self._select_corners(frame))
        if (np.any(points < 0) or np.any(points[:, 0] >= self.frame_width)
                or np.any(points[:, 1] >= self.frame_height)):
            raise SystemExit("All source points must be inside the video frame")
        contour = points.reshape(-1, 1, 2)
        if cv2.contourArea(contour) < 100 or not cv2.isContourConvex(contour):
            raise SystemExit("Four corners must form a convex TL, TR, BR, BL quadrilateral")
        target = np.float32([[20, self.bev_header_height + 20],
                             [self.bird_width - 21, self.bev_header_height + 20],
                             [self.bird_width - 21, self.bird_height - 21], [20, self.bird_height - 21]])
        homography = cv2.getPerspectiveTransform(points, target)
        if not np.isfinite(homography).all() or abs(np.linalg.det(homography)) < 1e-12:
            raise SystemExit("Degenerate homography; choose spread-out ground points")
        self.corners, self.homography = points, homography
        self.corner_pixels = np.rint(points).astype(np.int32)

3. Homography helpers and ground projection

These helpers convert image coordinates into positions on the bird's-eye canvas using the calibrated perspective matrix. A ground mask defines the road region that should appear on the map, while each detected bounding box contributes its bottom-center as an approximate contact point. The code projects these points with cv2.perspectiveTransform and rejects positions outside the calibrated polygon or output bounds. This avoids drawing markers for detections that cannot be mapped reliably to the selected road plane.

Python
    def _warp(self, image, interpolation=cv2.INTER_LINEAR):
        """
        Return an image or mask warped into the calibrated BEV dimensions.
        """
        return cv2.warpPerspective(image, self.homography, (self.bird_width, self.bird_height),
                                   flags=interpolation)


    def _ground_mask(self):
        """
        Return a Boolean BEV mask covering only the selected road polygon.
        Fill the polygon in source coordinates, then warp it without blending
        mask values. Pixels outside calibrated ground stay hidden.
        """
        mask = np.zeros((self.frame_height, self.frame_width), dtype=np.uint8)
        cv2.fillConvexPoly(mask, self.corners.astype(np.int32), 255)
        return self._warp(mask, cv2.INTER_NEAREST) > 0


    def _project_boxes(self, boxes):
        """
        Return source bottom-centers and their BEV projections as N x 2 arrays.
        The bottom-center approximates ground contact. Empty detections return
        empty arrays; these projections do not reconstruct 3D keypoints.
        """
        xyxy = boxes.xyxy.cpu().numpy()
        feet = np.column_stack(((xyxy[:, 0] + xyxy[:, 2]) / 2, xyxy[:, 3])).astype(np.float32)
        if not len(feet):
            return feet, feet.copy()
        mapped = cv2.perspectiveTransform(feet.reshape(-1, 1, 2), self.homography)
        return feet, mapped.reshape(-1, 2)


    def _contains(self, foot, mapped):
        """
        Check whether a source contact point and its BEV projection are valid.
        Reject nonfinite coordinates, points outside the road polygon, and
        projections outside the map dimensions.
        """
        if not np.isfinite(foot).all() or not np.isfinite(mapped).all():
            return False
        x, y = mapped
        return (cv2.pointPolygonTest(self.corners, tuple(map(float, foot)), False) >= 0
                and 0 <= x < self.bird_width and 0 <= y < self.bird_height)

4. Build the BEV road background

The BEV background is prepared once so it does not need to be reconstructed for every tracked frame. For a recorded video, the script samples up to 31 frames, warps them into map coordinates and takes a temporal median to reduce the influence of moving traffic. You can instead provide an empty-road image or select the plain pastel map style. The resulting canvas includes the calibrated road boundary and a BEV heading, then serves as the base for animated tracking overlays.

Python
    def _sample_background(self, video_path, first_frame):
        """
        Return a temporal median of up to 31 evenly spaced, warped video frames.
        A separate reader preserves tracking position and always closes. Camera
        devices, streams, or failed samples fall back to the first frame.
        The camera must be fixed; parked vehicles can remain in the median.
        """
        samples = []
        source = str(video_path)
        if not source.isdecimal() and Path(source).is_file():
            reader = cv2.VideoCapture(source)
            try:
                count = int(reader.get(cv2.CAP_PROP_FRAME_COUNT))
                indices = np.unique(np.linspace(0, max(0, count - 1), min(31, max(1, count))).astype(int))
                for index in indices:
                    reader.set(cv2.CAP_PROP_POS_FRAMES, int(index))
                    ok, frame = reader.read()
                    if ok:
                        samples.append(self._warp(frame))
            finally:
                reader.release()
        return (np.median(np.stack(samples), axis=0, overwrite_input=True).astype(np.uint8)
                if samples else self._warp(first_frame))


    def _build_background(self, video_path, first_frame):
        """
        Build and cache the static road map or plain light background.
        A supplied empty-road image must match source dimensions. Road imagery
        is warped, dimmed, and masked to calibrated ground for reuse per frame.
        Add a white boundary and a large centered BEV/Homography header above the road.
        """
        canvas = np.full((self.bird_height, self.bird_width, 3), VIDEO_BACKGROUND, dtype=np.uint8)
        if self.map_style == "pastel":
            canvas[:] = (245, 242, 240)
        else:
            if self.road_background is not None:
                plate = cv2.imread(str(Path(self.road_background).expanduser()))
                expected = (self.frame_height, self.frame_width)
                if plate is None or plate.shape[:2] != expected:
                    raise SystemExit("Road background must be a readable image matching the video dimensions")
                road = self._warp(plate)
            else:
                road = self._sample_background(video_path, first_frame)
            road = cv2.addWeighted(road, 0.65, np.full_like(road, DARK), 0.35, 0)
            valid = self._ground_mask()
            canvas[valid] = road[valid]
        boundary = cv2.perspectiveTransform(
            self.corners.reshape(-1, 1, 2), self.homography)
        cv2.polylines(canvas, [np.rint(boundary).astype(np.int32)], True, WHITE, 3, cv2.LINE_AA)
        canvas[:self.bev_header_height] = VIDEO_BACKGROUND
        title = "BEV/Homography"
        font, thickness = cv2.FONT_HERSHEY_SIMPLEX, 2
        (text_width, text_height), baseline = cv2.getTextSize(title, font, 1.0, thickness)
        text_scale = min(1.2, (self.bird_width - 32) / text_width,
                         (self.bev_header_height - 12) / (text_height + baseline))
        (text_width, text_height), baseline = cv2.getTextSize(title, font, text_scale, thickness)
        position = ((self.bird_width - text_width) // 2,
                    (self.bev_header_height + text_height - baseline) // 2)
        cv2.putText(canvas, title, position, font, text_scale, WHITE, thickness, cv2.LINE_AA)
        self.background = canvas

5. Track smoothing and BEV visualization

Each tracked object retains a smoothed map position to reduce jitter from changing detection boxes. Recent positions help estimate motion direction, which determines the orientation of an illustrative footprint for vehicle classes. The renderer can also display simpler dot markers, track IDs and trails that show recent movement across the road. Old track state is discarded after its expiry window so previous identities do not remain on the map indefinitely.

Python
    def _update_position(self, state, mapped, category):
        """
        Update a track position and return its rounded BEV contact estimate.
        Vehicles use a two-pixel dead zone and a 25% smoothing update; other
        classes use the projected position directly.
        """
        raw = np.float32(mapped)
        if category in VEHICLE_CLASSES:
            previous = raw if state["position"] is None else state["position"]
            delta = raw - previous
            state["position"] = previous if np.linalg.norm(delta) < 2 else previous + 0.25 * delta
        else:
            state["position"] = raw
        return tuple(np.rint(state["position"]).astype(int))


    def _footprint_corners(self, state, point, category):
        """
        Return four rotated 2D rectangle corners for an approximate BEV symbol.
        Recent motion above three pixels updates the smoothed long-axis angle.
        Class pixel sizes are illustrative, not measured footprints or 3D
        corners. Stationary tracks keep their previous angle.
        """
        state["positions"].append(point)
        if len(state["positions"]) > 1:
            delta = np.float32(point) - np.float32(state["positions"][0])
            if np.linalg.norm(delta) >= 3:
                target = float(np.degrees(np.arctan2(delta[1], delta[0])) - 90)
                difference = (target - state["angle"] + 90) % 180 - 90
                state["angle"] += 0.25 * difference
        size = self.symbol_sizes.get(category, self.default_symbol_size)
        rectangle = (tuple(map(float, point)), size, state["angle"])
        return np.rint(cv2.boxPoints(rectangle)).astype(np.int32)


    def _draw_bev_label(self, canvas, polygon, track_id):
        """Center a tiny numeric ID inside its marker, scaling to fit the rotated box.

        Use an inscribed square so the yellow badge and navy digits remain
        inside the shape at any angle. Large IDs shrink to the available space.
        """
        edges = np.roll(polygon, -1, axis=0) - polygon
        side = max(1, int(np.linalg.norm(edges, axis=1).min() / np.sqrt(2)) - 2)
        label = str(track_id)
        font, font_scale = cv2.FONT_HERSHEY_SIMPLEX, 0.25
        (width, height), baseline = cv2.getTextSize(label, font, font_scale, 1)
        badge = np.full((height + baseline + 2, width + 2, 3), ID_BACKGROUND, dtype=np.uint8)
        cv2.putText(badge, label, (1, height + 1), font, font_scale, ID_TEXT, 1, cv2.LINE_AA)
        scale = min(1.0, side / badge.shape[1], side / badge.shape[0])
        badge = cv2.resize(badge, (max(1, round(badge.shape[1] * scale)),
                                   max(1, round(badge.shape[0] * scale))), interpolation=cv2.INTER_AREA)
        center = np.rint(polygon.mean(axis=0)).astype(int)
        x, y = center - np.array([badge.shape[1] // 2, badge.shape[0] // 2])
        left, top = max(0, x), max(0, y)
        right, bottom = min(self.bird_width, x + badge.shape[1]), min(self.bird_height, y + badge.shape[0])
        if left < right and top < bottom:
            canvas[top:bottom, left:right] = badge[top - y:bottom - y, left - x:right - x]


    def _draw_bev_track(self, canvas, mapped, track_id, category, frame_index):
        """
        Update one track and draw its marker, optional trail, and optional ID.
        Initialize state only for new identities. Missed frames reset motion
        history and trails; footprint markers retain the last known angle.
        """
        state = self.states.get(track_id)
        if state is None:
            state = self._new_track_state()
            self.states[track_id] = state
        if state["last_seen"] != frame_index - 1:
            state["positions"].clear()
            state["trail"].clear()
        point = self._update_position(state, mapped, category)
        color = BEV_DARK_YELLOW
        if self.trail_length:
            state["trail"].append(point)
            if len(state["trail"]) > 1:
                cv2.polylines(canvas, [np.array(state["trail"], dtype=np.int32)], False, color, 2, cv2.LINE_AA)
        if self.map_marker == "footprint":
            polygon = self._footprint_corners(state, point, category)
            cv2.fillConvexPoly(canvas, polygon, color, cv2.LINE_AA)
            cv2.polylines(canvas, [polygon], True, WHITE, 1, cv2.LINE_AA)
        else:
            cv2.circle(canvas, point, 7, (255, 255, 255), -1, cv2.LINE_AA)
            cv2.circle(canvas, point, 5, color, -1, cv2.LINE_AA)
            polygon = cv2.boxPoints((tuple(map(float, point)), (8, 8), 0))
        state["last_seen"] = frame_index
        if self.map_ids:
            self._draw_bev_label(canvas, polygon, track_id)


    def _forget_old_tracks(self, frame_index):
        """
        Remove all state for tracks absent from the BEV for over 120 frames.
        Position, direction history, and trail are released together.
        """
        expired = [track_id for track_id, state in self.states.items()
                   if frame_index - state["last_seen"] > 120]
        for track_id in expired:
            del self.states[track_id]


    def _render_bev(self, result, frame_index):
        """
        Return a fresh map with the current tracked detections drawn on it.
        Move box data to CPU once, project contact points, and render only
        tracks on calibrated ground. Expire missing tracks, including frames
        with no detections.
        """
        canvas = self.background.copy()
        boxes = result.boxes
        if boxes is not None and boxes.id is not None:
            boxes = boxes.cpu()
            feet, mapped = self._project_boxes(boxes)
            ids = boxes.id.tolist()
            classes = boxes.cls.tolist()
            for foot, map_point, track_id, class_id in zip(feet, mapped, ids, classes):
                track_id, class_id = int(track_id), int(class_id)
                if self._contains(foot, map_point):
                    self._draw_bev_track(canvas, map_point, track_id, result.names[class_id].lower(), frame_index)
        self._forget_old_tracks(frame_index)
        return canvas


    @staticmethod

6. Compose camera and bird’s-eye layouts

This part turns the annotated camera view and BEV map into a consistently sized output frame. The default dashboard allocates most of the 1920 × 1080 frame to camera footage and a narrower panel to the top-down visualization. The alternative side-by-side layout displays both views together, while map-only mode writes just the BEV output. Frame fitting and composition keep the selected layout dimensions consistent for the video writer.

Python
    def _fit_frame(frame, width, height):
        """
        Return the full frame resized to fit, with Original centered in top padding.
        Preserve proportions without cropping; add centered navy padding only
        when source and panel aspect ratios differ. Center the title in top
        padding when there is sufficient room, keeping it off the video image.
        """
        source_height, source_width = frame.shape[:2]
        scale = min(width / source_width, height / source_height)
        display_width = max(1, min(width, round(source_width * scale)))
        display_height = max(1, min(height, round(source_height * scale)))
        view = cv2.resize(frame, (display_width, display_height),
                          interpolation=cv2.INTER_AREA if scale < 1 else cv2.INTER_LINEAR)
        if (display_width, display_height) == (width, height):
            return view
        panel = np.full((height, width, 3), VIDEO_BACKGROUND, dtype=np.uint8)
        left, top = (width - display_width) // 2, (height - display_height) // 2
        panel[top:top + display_height, left:left + display_width] = view
        if top >= 16:
            title = "Original"
            font, thickness = cv2.FONT_HERSHEY_SIMPLEX, 2
            (text_width, text_height), baseline = cv2.getTextSize(title, font, 1.0, thickness)
            text_scale = min(1.2, (width - 24) / text_width,
                             (top - 12) / (text_height + baseline))
            (text_width, text_height), baseline = cv2.getTextSize(title, font, text_scale, thickness)
            position = ((width - text_width) // 2, (top + text_height - baseline) // 2)
            cv2.putText(panel, title, position, font, text_scale, WHITE, thickness, cv2.LINE_AA)
        return panel


    def _compose(self, annotated, map_canvas):
        """
        Combine the labeled camera frame and BEV into the selected output layout.
        Return the map directly for map-only mode. Dashboard adds a thin
        vertical divider. The camera title uses existing top padding.
        """
        if self.bird_only:
            return map_canvas
        if self.layout == "side":
            camera = self._fit_frame(annotated, self.bird_width, self.bird_height)
            return np.concatenate((camera, map_canvas), axis=1)
        canvas = np.empty((self.output_height, self.output_width, 3), dtype=np.uint8)
        canvas[:, :self.camera_width] = self._fit_frame(annotated, self.camera_width, self.camera_height)
        canvas[:, self.camera_width:] = map_canvas
        cv2.line(canvas, (self.camera_width, 0), (self.camera_width, self.output_height - 1), (60, 60, 60), 1)
        return canvas

7. YOLO26 tracking and camera annotations

The script loads the configured Ultralytics checkpoint and checks that it is a detection model before running inference. YOLO tracking uses persistent identities, with TrackTrack configured by default, so detections can be associated across successive frames. The camera annotation code draws the tracking information and calibration outline on the original perspective view. This section also creates the OpenCV MP4 writer with the required output size and source frame rate, using a fallback when needed.

Python
    def _load_detector(self):
        """
        Load the configured YOLO weights and verify the detection task.
        Pose and segmentation models require different geometry handling and
        are rejected by this 2D-box tracking pipeline.
        """
        self.model = YOLO(self.model_path)
        if self.model.task != "detect":
            raise SystemExit("Model must be a YOLO detection model")


    def _create_writer(self):
        """
        Open an MP4 writer and return the resolved output path.
        Create parent directories and use source FPS, falling back to 30 when
        invalid. Output dimensions come from the selected layout.
        """
        output = Path(self.output_path).expanduser().resolve()
        output.parent.mkdir(parents=True, exist_ok=True)
        fps = self.capture.get(cv2.CAP_PROP_FPS)
        fps = fps if np.isfinite(fps) and fps > 0 else 30.0
        self.writer = cv2.VideoWriter(str(output), cv2.VideoWriter_fourcc(*"mp4v"), fps,
                                      (self.output_width, self.output_height))
        if not self.writer.isOpened():
            raise SystemExit(f"Could not write output video: {output}")
        return output


    def _detect_and_track(self, frame):
        """
        Return a YOLO tracking result for one source frame.
        Persistent tracker state keeps identities across frames; configured
        confidence, class filters, device, and tracker options are reused. Move
        results to CPU once for both camera and BEV rendering.
        """
        return self.model.track(frame, **self.track_options)[0].cpu()


    def _annotate_camera(self, result):
        """Draw yellow camera boxes and smaller ID-only badges with navy text.

        Scale label size for the final camera panel so downsampling does not
        make IDs too small. Show a thick white calibration outline and white points,
        including frames with no detections. Clamp badges inside the frame.
        """
        if self.bird_only:
            return None
        annotated = result.orig_img.copy()
        if self.corners is not None:
            corners = self.corner_pixels
            white = (255, 255, 255)
            cv2.polylines(annotated, [corners], True, white, 4, cv2.LINE_AA)
            for corner in corners:
                cv2.circle(annotated, tuple(corner), 10, white, -1, cv2.LINE_AA)
        boxes = result.boxes
        if boxes is None:
            return annotated
        boxes = boxes.cpu()
        for box in boxes.xyxy.numpy():
            x1, y1, x2, y2 = np.rint(box).astype(int)
            cv2.rectangle(annotated, (x1, y1), (x2, y2), ID_BACKGROUND, 2, cv2.LINE_AA)
        if boxes.id is None:
            return annotated
        height, width = annotated.shape[:2]
        panel_width = self.bird_width if self.layout == "side" else self.camera_width
        panel_height = self.bird_height if self.layout == "side" else self.camera_height
        resize_scale = min(panel_width / width, panel_height / height)
        font_scale = 1.2
        thickness = 1
        padding = max(2, round(5 / resize_scale))
        font = 0
        for box, track_id in zip(boxes.xyxy.numpy(), boxes.id.tolist()):
            track_id = int(track_id)
            label = f"#{track_id}"
            (text_width, text_height), baseline = cv2.getTextSize(label, font, font_scale, thickness)
            badge_width = text_width + 2 * padding
            badge_height = text_height + baseline + 2 * padding
            left = max(0, min(round(float(box[0])), width - badge_width))
            top = max(0, min(round(float(box[1])) - badge_height, height - badge_height))
            cv2.rectangle(annotated, (left, top), (left + badge_width, top + badge_height), ID_BACKGROUND, -1)
            cv2.putText(annotated, label, (left + padding, top + padding + text_height),
                        font, font_scale, ID_TEXT, thickness, cv2.LINE_AA)
        return annotated

8. Process frames, preview and export MP4

The final methods connect calibration, background generation, detection, projection, rendering and frame composition into a single processing loop. Every processed frame is written to the output MP4, and the optional preview lets you inspect results while processing or stop early. Video capture, writer resources and preview windows are released even if processing is interrupted. The script finishes with run_bird_eye and a __main__ entry point using the default dashboard settings.

Python
    def _process_frame(self, frame):
        """
        Detect objects, label the camera frame, render the BEV, and compose output.
        Camera labels show only track IDs. Skip camera annotation
        entirely for map-only output.
        """
        result = self._detect_and_track(frame)
        annotated = self._annotate_camera(result)
        map_canvas = self._render_bev(result, self.frame_index)
        return self._compose(annotated, map_canvas)


    def _show(self, canvas):
        """
        Display a screen-sized preview and return whether processing should stop.
        Keep preview proportions; stop when Q is pressed or the window closes.
        """
        scale = min(1.0, 1600 / self.output_width, 900 / self.output_height)
        preview = cv2.resize(canvas, (round(self.output_width * scale), round(self.output_height * scale)))
        window = "Homography | Bird Eye View using Ultralytics YOLO26 Object Detection + Tracking"
        cv2.imshow(window, preview)
        return cv2.waitKey(1) & 0xFF == ord("q") or cv2.getWindowProperty(window, cv2.WND_PROP_VISIBLE) < 1


    def _close(self):
        """
        Release capture, writer, and any preview windows.
        This cleanup also runs after cancellation or a processing failure.
        """
        if self.capture is not None:
            self.capture.release()
        if self.writer is not None:
            self.writer.release()
        if self.show:
            cv2.destroyAllWindows()


    def run(self):
        """
        Execute calibration, background creation, tracking, and video output.
        Process the first calibration frame too, then continue until EOF or
        preview exit. Always release resources and return the saved video path.
        """
        self._validate_options()
        try:
            frame = self._open_video()
            self._calibrate(frame, self.input_corners)
            self._build_background(self.video_path, frame)
            self._load_detector()
            output = self._create_writer()
            while True:
                canvas = self._process_frame(frame)
                self.writer.write(canvas)
                self.frame_index += 1
                if self.show and self._show(canvas):
                    break
                ok, frame = self.capture.read()
                if not ok:
                    break
        finally:
            self._close()
        print(f"Saved {self.frame_index} frames to {output}")
        return output


def run_bird_eye(video_path, model_path, **kwargs):
    """
    Construct BirdEyeTracker, execute the pipeline, and return its output path.
    Additional keyword arguments are passed directly to the tracker constructor.
    """
    return BirdEyeTracker(video_path, model_path, **kwargs).run()


if __name__ == "__main__":
    run_bird_eye(
        video_path=DEFAULT_VIDEO,
        model_path=DEFAULT_MODEL,
        layout="dashboard",
        dashboard_width=1920,
        dashboard_height=1080,
        show=True,
    )

What this visualization can and cannot measure

This is a planar 2D visualization, not 3D reconstruction or a calibrated measurement of physical vehicle dimensions. Bounding-box bottom centers approximate ground contact, while the displayed class-based footprint sizes are illustrative. Detection errors, occlusions, camera motion and imperfect road calibration can affect positions and trails. The method assumes a fixed camera and a reasonably planar road.

The source script uses Python, OpenCV, NumPy, Ultralytics YOLO and TrackTrack by default. It writes a processed MP4 using OpenCV; the player above streams that rendered result rather than running inference in the browser.

← Explore more computer vision demos