From 17e27a39e05a5ff51944fe7e4352ba1cf3572ff3 Mon Sep 17 00:00:00 2001 From: Raphael van Kempen Date: Mon, 28 Sep 2026 17:30:11 +0000 Subject: [PATCH 01/17] rename ros package to autonomy_evaluation --- .devcontainer/devcontainer.json | 2 +- .github/workflows/docker-ros.yml | 2 +- README.md | 48 +++---- .../benchmarks/__init__.py | 8 -- autonomy_benchmarks/setup.cfg | 4 - .../tests/benchmarks/__init__.py | 4 - .../README.md | 33 +++-- .../autonomy_evaluation/__init__.py | 4 + .../autonomy_evaluation.py | 126 +++++++++--------- .../evaluations/Evaluation.py | 28 ++-- .../evaluations/__init__.py | 8 ++ .../NuscenesLidarObjectDetection.py | 22 +-- .../lidar_object_detection/__init__.py | 4 +- .../utils/ObjectDetectionUtils.py | 8 +- .../autonomy_evaluation}/utils/__init__.py | 4 +- .../config/conf.rviz | 6 +- .../launch/autonomy_evaluation.launch.py | 30 ++--- .../package.xml | 4 +- .../resource/autonomy_evaluation | 0 autonomy_evaluation/setup.cfg | 4 + .../setup.py | 4 +- .../tests/__init__.py | 2 +- .../tests/evaluations}/__init__.py | 2 +- .../lidar_object_detection}/__init__.py | 2 +- .../test_NuscenesLidarObjectDetection.py | 36 ++--- .../tests/evaluations/test_Evaluation.py | 86 ++++++------ .../tests/test_autonomy_evaluation.py | 94 ++++++------- .../tests/utils/__init__.py | 2 +- .../tests/utils/test_ObjectDetectionUtils.py | 2 +- deployment/compose/docker-compose.yml | 12 +- deployment/helm/Chart.yaml | 4 +- deployment/helm/values.yaml | 12 +- docker-compose.dev.yml | 8 +- docker-compose.yml | 12 +- docs/IMPLEMENTATION.md | 20 +-- 35 files changed, 326 insertions(+), 321 deletions(-) delete mode 100644 autonomy_benchmarks/autonomy_benchmarks/benchmarks/__init__.py delete mode 100644 autonomy_benchmarks/setup.cfg delete mode 100644 autonomy_benchmarks/tests/benchmarks/__init__.py rename {autonomy_benchmarks => autonomy_evaluation}/README.md (75%) create mode 100644 autonomy_evaluation/autonomy_evaluation/__init__.py rename autonomy_benchmarks/autonomy_benchmarks/autonomy_benchmarks.py => autonomy_evaluation/autonomy_evaluation/autonomy_evaluation.py (85%) rename autonomy_benchmarks/autonomy_benchmarks/benchmarks/AutonomyBenchmark.py => autonomy_evaluation/autonomy_evaluation/evaluations/Evaluation.py (89%) create mode 100644 autonomy_evaluation/autonomy_evaluation/evaluations/__init__.py rename {autonomy_benchmarks/autonomy_benchmarks/benchmarks => autonomy_evaluation/autonomy_evaluation/evaluations}/lidar_object_detection/NuscenesLidarObjectDetection.py (98%) rename {autonomy_benchmarks/autonomy_benchmarks/benchmarks => autonomy_evaluation/autonomy_evaluation/evaluations}/lidar_object_detection/__init__.py (51%) rename {autonomy_benchmarks/autonomy_benchmarks => autonomy_evaluation/autonomy_evaluation}/utils/ObjectDetectionUtils.py (97%) rename {autonomy_benchmarks/autonomy_benchmarks => autonomy_evaluation/autonomy_evaluation}/utils/__init__.py (51%) rename {autonomy_benchmarks => autonomy_evaluation}/config/conf.rviz (99%) rename autonomy_benchmarks/launch/autonomy_benchmarks.launch.py => autonomy_evaluation/launch/autonomy_evaluation.launch.py (83%) rename {autonomy_benchmarks => autonomy_evaluation}/package.xml (86%) rename autonomy_benchmarks/resource/autonomy_benchmarks => autonomy_evaluation/resource/autonomy_evaluation (100%) create mode 100644 autonomy_evaluation/setup.cfg rename {autonomy_benchmarks => autonomy_evaluation}/setup.py (86%) rename {autonomy_benchmarks => autonomy_evaluation}/tests/__init__.py (56%) rename {autonomy_benchmarks/tests/benchmarks/lidar_object_detection => autonomy_evaluation/tests/evaluations}/__init__.py (56%) rename {autonomy_benchmarks/autonomy_benchmarks => autonomy_evaluation/tests/evaluations/lidar_object_detection}/__init__.py (59%) rename {autonomy_benchmarks/tests/benchmarks => autonomy_evaluation/tests/evaluations}/lidar_object_detection/test_NuscenesLidarObjectDetection.py (95%) rename autonomy_benchmarks/tests/benchmarks/test_AutonomyBenchmark.py => autonomy_evaluation/tests/evaluations/test_Evaluation.py (70%) rename autonomy_benchmarks/tests/test_autonomy_benchmarks.py => autonomy_evaluation/tests/test_autonomy_evaluation.py (76%) rename {autonomy_benchmarks => autonomy_evaluation}/tests/utils/__init__.py (57%) rename {autonomy_benchmarks => autonomy_evaluation}/tests/utils/test_ObjectDetectionUtils.py (99%) diff --git a/.devcontainer/devcontainer.json b/.devcontainer/devcontainer.json index f1baedb..45f71cf 100644 --- a/.devcontainer/devcontainer.json +++ b/.devcontainer/devcontainer.json @@ -5,7 +5,7 @@ // attach to container from docker compose setup "dockerComposeFile": ["../docker-compose.dev.yml"], - "service": "autonomy_benchmarks", + "service": "autonomy_evaluation", // set up git pre-commit hooks "onCreateCommand": "pip install pre-commit && apt update && apt install locales -y && locale-gen en_US.UTF-8", diff --git a/.github/workflows/docker-ros.yml b/.github/workflows/docker-ros.yml index 82f4d8d..bd99cc8 100644 --- a/.github/workflows/docker-ros.yml +++ b/.github/workflows/docker-ros.yml @@ -22,7 +22,7 @@ jobs: target: dev,run base-image: rwthika/ros2:jazzy rmw-implementation: rmw_zenoh_cpp - command: ros2 launch autonomy_benchmarks autonomy_benchmarks.launch.py + command: ros2 launch autonomy_evaluation autonomy_evaluation.launch.py enable-slim: false env: TARGET_CMAKE_ARGS: -DCMAKE_EXPORT_COMPILE_COMMANDS=1 diff --git a/README.md b/README.md index a008294..5bba453 100644 --- a/README.md +++ b/README.md @@ -1,36 +1,36 @@ -# autonomy_benchmarks +# autonomy_evaluation

- - + +
- - - - - + + + + +

-> This repository will be part of the **Autonomy.Hub Ecosystem** +> This repository is part of the **Autonomy.Benchmarks** suite of the **Autonomy.Hub Ecosystem** -As part of the Autonomy.Hub Ecosystem, **Autonomy.Benchmarks** enables the Automated Driving community to easily benchmark their automated driving building blocks across different tasks and datasets: +Within the Autonomy.Benchmarks suite, **Autonomy.Evaluation** generates the metrics-based evidence for benchmarking automated driving deployments. It evaluates the output of a system under test on the samples replayed by [Autonomy.Datasets](https://github.com/thinking-cars/autonomy_datasets) and reports the resulting metrics per scene and over all evaluated samples: -- 🔄 **Unified ROS 2 Interface**: Work with multiple datasets using the benefits of the ROS 2 ecosystem -- 📊 **Comprehensive Benchmarks**: Use the provided benchmarks with [Autonomy.Datasets](https://github.com/thinking-cars/autonomy_datasets) to benchmark building blocks across different automated driving tasks +- 🔄 **Unified ROS 2 Interface**: Evaluate systems under test on multiple datasets using the benefits of the ROS 2 ecosystem +- 📊 **Established Metrics**: Use the provided evaluations, which follow the protocols of established challenges, with [Autonomy.Datasets](https://github.com/thinking-cars/autonomy_datasets) across different automated driving tasks - ⚡ **Efficient Data Pipeline**: Works seamlessly with preprocessed Rosbag files from [Autonomy.Datasets](https://github.com/thinking-cars/autonomy_datasets) for fast execution during development - 🐳 **Dockerized Environment**: Reproducible setup with all dependencies included - 🔌 **Modular Architecture**: Easy integration with other ROS 2 packages -## Supported Benchmarks +## Supported Evaluations -This repository supports various automated driving evaluation benchmarks. +This repository supports evaluations of various automated driving tasks. Detailed metric definitions and computation notes are documented in [docs/IMPLEMENTATION.md](docs/IMPLEMENTATION.md). -> [**Contributions**](docs/IMPLEMENTATION.md#adding-more-benchmarks) adding more benchmarks are welcome +> [**Contributions**](docs/IMPLEMENTATION.md#adding-more-evaluations) adding more evaluations are welcome -| Benchmark | Challenge | Dataset | Task | +| Evaluation | Challenge | Dataset | Task | | --------- | --------- | ------- | ---- | | [**nuScenes 3D Lidar Object Detection**](docs/IMPLEMENTATION.md#3d-lidar-object-detection) | [![3D Object Detection Challenge](https://img.shields.io/badge/origin-3D_Object_Detection_Challenge-green)](https://www.nuscenes.org/object-detection) | [nuScenes](https://github.com/thinking-cars/autonomy_datasets) | 3D bounding box detection from lidar | @@ -43,7 +43,7 @@ Detailed metric definitions and computation notes are documented in [docs/IMPLEM Clone [autonomy_datasets](https://github.com/thinking-cars/autonomy_datasets) and follow its setup instructions to prepare your dataset. -Use the provided [docker-compose.yml](docker-compose.yml) to start the full pipeline — dataset publisher, system-under-test, and benchmark node: +Use the provided [docker-compose.yml](docker-compose.yml) to start the full pipeline — dataset publisher, system under test, and evaluation node: ```bash # enable GUI output from Docker container @@ -57,13 +57,13 @@ docker compose up -d docker compose down ``` -Configure the benchmark task and dataset via ROS launch arguments in [docker-compose.yml](docker-compose.yml): +Configure the evaluation and dataset via ROS launch arguments in [docker-compose.yml](docker-compose.yml): ```yaml -command: ros2 launch autonomy_benchmarks autonomy_benchmarks.launch.py benchmark:=nuscenes_lidar_object_detection prediction:=$your_prediction_topic label:=$your_label_topic request_samples:=/datasets/request_samples visualize:=true +command: ros2 launch autonomy_evaluation autonomy_evaluation.launch.py evaluation:=nuscenes_lidar_object_detection prediction:=$your_prediction_topic label:=$your_label_topic request_samples:=/datasets/request_samples visualize:=true ``` -The benchmark node requests the samples it evaluates from the dataset node via its `request_samples` service, which publishes them and responds once they have been published. The dataset therefore publishes the next sample only once the system under test has processed the current one. As soon as all samples have been published, the benchmark aggregates its metrics per sample, per scene of the dataset and over the whole benchmark. See the [node documentation](autonomy_benchmarks/README.md#autonomy_benchmarks) for the sample request settings and the results. +The evaluation node requests the samples it evaluates from the dataset node via its `request_samples` service, which publishes them and responds once they have been published. The dataset therefore publishes the next sample only once the system under test has processed the current one. As soon as all samples have been published, the node aggregates its metrics per scene of the dataset and over all evaluated samples. See the [node documentation](autonomy_evaluation/README.md#autonomy_evaluation) for the sample request settings and the results. ## 💻 Development @@ -71,11 +71,11 @@ The benchmark node requests the samples it evaluates from the dataset node via i 1. Clone the repository. ```bash - git clone https://github.com/thinking-cars/autonomy_benchmarks.git + git clone https://github.com/thinking-cars/autonomy_evaluation.git ``` 1. Initialize the [`.openads-dev-environment`](https://github.com/openads-project/openads-dev-environment) submodule containing development environment configuration. ```bash - cd autonomy_benchmarks + cd autonomy_evaluation git submodule update --init --recursive ``` 1. Open the repository in [Visual Studio Code](https://code.visualstudio.com). @@ -108,11 +108,11 @@ colcon test-result --verbose ## 📝 Documentation -Package and node interfaces are documented in the respective package READMEs listed below. Implementation details are found in the [Source Code Documentation](https://thinking-cars.github.io/autonomy_benchmarks). +Package and node interfaces are documented in the respective package READMEs listed below. Implementation details are found in the [Source Code Documentation](https://thinking-cars.github.io/autonomy_evaluation). | Package | Description | | --- | --- | -| [autonomy_benchmarks](autonomy_benchmarks/README.md) | Benchmarking suite for automated driving tasks | +| [autonomy_evaluation](autonomy_evaluation/README.md) | Metrics-based evaluation of automated driving tasks, generating the evidence for benchmarking automated driving deployments | ## ⚖️ Licensing diff --git a/autonomy_benchmarks/autonomy_benchmarks/benchmarks/__init__.py b/autonomy_benchmarks/autonomy_benchmarks/benchmarks/__init__.py deleted file mode 100644 index bdb62a9..0000000 --- a/autonomy_benchmarks/autonomy_benchmarks/benchmarks/__init__.py +++ /dev/null @@ -1,8 +0,0 @@ -# Copyright Thinking Cars GmbH -# SPDX-License-Identifier: Apache-2.0 - -"""Benchmark implementations for AutonomyHub.""" - -from autonomy_benchmarks.benchmarks.AutonomyBenchmark import AutonomyBenchmark - -__all__ = ["AutonomyBenchmark"] diff --git a/autonomy_benchmarks/setup.cfg b/autonomy_benchmarks/setup.cfg deleted file mode 100644 index 7e58c70..0000000 --- a/autonomy_benchmarks/setup.cfg +++ /dev/null @@ -1,4 +0,0 @@ -[develop] -script_dir=$base/lib/autonomy_benchmarks -[install] -install_scripts=$base/lib/autonomy_benchmarks diff --git a/autonomy_benchmarks/tests/benchmarks/__init__.py b/autonomy_benchmarks/tests/benchmarks/__init__.py deleted file mode 100644 index 5b0f7f7..0000000 --- a/autonomy_benchmarks/tests/benchmarks/__init__.py +++ /dev/null @@ -1,4 +0,0 @@ -# Copyright Thinking Cars GmbH -# SPDX-License-Identifier: Apache-2.0 - -"""Benchmark-oriented test modules for autonomy_benchmarks.""" diff --git a/autonomy_benchmarks/README.md b/autonomy_evaluation/README.md similarity index 75% rename from autonomy_benchmarks/README.md rename to autonomy_evaluation/README.md index 33ac320..27dd054 100644 --- a/autonomy_benchmarks/README.md +++ b/autonomy_evaluation/README.md @@ -1,21 +1,26 @@ -# `autonomy_benchmarks` +# `autonomy_evaluation` -Benchmarking suite for automated driving tasks +Metrics-based evaluation of automated driving tasks, generating the evidence for benchmarking automated driving deployments + +`autonomy_evaluation` is the part of the **Autonomy.Benchmarks** suite that turns the output of a system under test into +metrics. It evaluates the samples that [autonomy_datasets](https://github.com/thinking-cars/autonomy_datasets) replays +against the labels of the dataset, and reports the metrics per scene and over all evaluated samples, as the evidence +an automated driving deployment is benchmarked on. ## Nodes -### `autonomy_benchmarks` +### `autonomy_evaluation` The node requests the samples it evaluates from the dataset, using the `request_samples` service of [autonomy_datasets](https://github.com/thinking-cars/autonomy_datasets), which publishes them and responds once they have been published. By default one sample is requested at a time, so the dataset only publishes the next sample once the system under test has delivered its output for -the current one and the benchmark has evaluated it. Increase `samples_per_request` to publish +the current one and the node has evaluated it. Increase `samples_per_request` to publish samples in batches, set it to `0` to publish the whole dataset with a single request, or list the IDs of individual samples in `sample_ids` to evaluate only those. Once the dataset reports that all requested samples have been published, the per-sample metrics -are aggregated and reported on three levels: for the whole benchmark (`aggregated_metrics`), for +are aggregated and reported on three levels: over all evaluated samples (`aggregated_metrics`), for the samples of each scene the dataset published them from (`scene_results`, matched with the samples via the `published_scene_ids` of the responses), and for every single sample (`sample_results`). The dataset metrics are logged, and the results of all three levels are @@ -24,7 +29,7 @@ written to a JSON file if `results_path` is set. Samples that are not evaluated are left out. ```bash -ros2 launch autonomy_benchmarks autonomy_benchmarks.launch.py \ +ros2 launch autonomy_evaluation autonomy_evaluation.launch.py \ prediction:=/object_list/prediction \ label:=/object_list/lidar_01 \ request_samples:=/datasets/request_samples \ @@ -34,7 +39,7 @@ ros2 launch autonomy_benchmarks autonomy_benchmarks.launch.py \ To look at the samples one by one, set `manual_playback` together with `visualize`. The node then requests no samples itself, and the samples are published with the [playback panel](https://github.com/thinking-cars/autonomy_datasets/blob/main/autonomy_datasets_rviz_plugins/README.md) in RViz instead, whose _Service_ field has to name the `request_samples` service of the dataset node (`/datasets/request_samples` by default). Every sample that arrives is evaluated and shown in RViz, and `samples_per_request`, `sample_ids` and `evaluation_timeout` have no effect. As the responses of the dataset only reach the panel, the node neither learns the scenes of the samples, so `scene_results` stays empty, nor when the dataset has ended: the results of the evaluated samples are reported once the node is stopped, e.g. with Ctrl-C, and are marked as incomplete. ```bash -ros2 launch autonomy_benchmarks autonomy_benchmarks.launch.py \ +ros2 launch autonomy_evaluation autonomy_evaluation.launch.py \ prediction:=/object_list/prediction \ label:=/object_list/lidar_01 \ visualize:=true \ @@ -43,7 +48,7 @@ ros2 launch autonomy_benchmarks autonomy_benchmarks.launch.py \ ```mermaid flowchart LR - NODE("autonomy_benchmarks") + NODE("autonomy_evaluation") NODE o--o|~/request_samples| SC0:::hidden classDef hidden display: none; ``` @@ -58,23 +63,23 @@ flowchart LR | Parameter | Type | Default | Description | | --- | --- | --- | --- | -| `benchmark` | `string` | `nuscenes_lidar_object_detection` | benchmark name | +| `evaluation` | `string` | `nuscenes_lidar_object_detection` | evaluation name | | `visualize` | `bool` | `false` | publish the per-sample true positives, false positives and false negatives for RViz | | `manual_playback` | `bool` | `false` | leave requesting the samples of the dataset to the user | | `samples_per_request` | `int` | `1` | number of samples to request from the dataset at a time; 0 requests all remaining samples at once, 1 evaluates every sample before the next one is published | | `sample_ids` | `string` | - | comma-separated IDs of the dataset samples to evaluate (e.g. '0,10,20'); if empty, all samples of the dataset are evaluated | | `evaluation_timeout` | `float` | `60.0` | seconds to wait for a published sample to be evaluated before continuing without it | -| `results_path` | `string` | - | path of the JSON file the benchmark results are written to; results are only logged if empty | +| `results_path` | `string` | - | path of the JSON file the evaluation results are written to; results are only logged if empty | ## Launch Files -### [`autonomy_benchmarks.launch.py`](launch/autonomy_benchmarks.launch.py) +### [`autonomy_evaluation.launch.py`](launch/autonomy_evaluation.launch.py) | Argument | Default | Description | | --- | --- | --- | | `request_samples` | `"~/request_samples"` | service of the dataset node used to request the samples to evaluate | -| `benchmark` | `"nuscenes_lidar_object_detection"` | benchmark to run | -| `name` | `"autonomy_benchmarks"` | node name | +| `evaluation` | `"nuscenes_lidar_object_detection"` | evaluation to run | +| `name` | `"autonomy_evaluation"` | node name | | `namespace` | `""` | node namespace | | `log_level` | `"info"` | ros logging level | | `use_sim_time` | `"true"` | use sim time | @@ -83,4 +88,4 @@ flowchart LR | `samples_per_request` | `"1"` | number of samples to request from the dataset at a time (0 requests all remaining samples at once) | | `sample_ids` | `""` | comma-separated IDs of the dataset samples to evaluate (all samples if empty) | | `evaluation_timeout` | `"60.0"` | seconds to wait for a published sample to be evaluated before continuing without it | -| `results_path` | `""` | path of the JSON file the benchmark results are written to (results are only logged if empty) | +| `results_path` | `""` | path of the JSON file the evaluation results are written to (results are only logged if empty) | diff --git a/autonomy_evaluation/autonomy_evaluation/__init__.py b/autonomy_evaluation/autonomy_evaluation/__init__.py new file mode 100644 index 0000000..de1b036 --- /dev/null +++ b/autonomy_evaluation/autonomy_evaluation/__init__.py @@ -0,0 +1,4 @@ +# Copyright Thinking Cars GmbH +# SPDX-License-Identifier: Apache-2.0 + +"""Metrics-based evaluation of automated driving tasks.""" diff --git a/autonomy_benchmarks/autonomy_benchmarks/autonomy_benchmarks.py b/autonomy_evaluation/autonomy_evaluation/autonomy_evaluation.py similarity index 85% rename from autonomy_benchmarks/autonomy_benchmarks/autonomy_benchmarks.py rename to autonomy_evaluation/autonomy_evaluation/autonomy_evaluation.py index ff0b450..f82639d 100644 --- a/autonomy_benchmarks/autonomy_benchmarks/autonomy_benchmarks.py +++ b/autonomy_evaluation/autonomy_evaluation/autonomy_evaluation.py @@ -20,15 +20,15 @@ from rclpy.task import Future from rclpy.timer import Timer -# Interval in seconds at which the benchmark checks whether it can request further samples; the -# benchmark also advances whenever a request is answered or a sample has been evaluated +# Interval in seconds at which the evaluation checks whether it can request further samples; the +# evaluation also advances whenever a request is answered or a sample has been evaluated _REQUEST_TIMER_PERIOD_S = 0.5 # Interval in seconds at which waiting for the sample request service of the dataset is logged _SERVICE_WAIT_LOG_INTERVAL_S = 10.0 # Number of samples whose messages are kept while they wait for the messages of their remaining -# benchmark inputs +# evaluation inputs _SYNCHRONIZER_QUEUE_SIZE = 10 @@ -48,7 +48,7 @@ def parse_sample_ids(sample_ids: str) -> list[int]: class SampleSynchronizer: - """Matches the messages of the benchmark inputs that belong to the same dataset sample. + """Matches the messages of the evaluation inputs that belong to the same dataset sample. The dataset stamps all messages of a sample with the recording time of that sample, so the messages of a sample are matched by their exact header stamp. Messages of a sample that never @@ -100,18 +100,18 @@ def add(self, topic: str, message: Any): self.incomplete_samples.popitem(last=False) -class AutonomyBenchmarks(Node): - """ROS 2 node for benchmarking automated driving tasks.""" +class AutonomyEvaluation(Node): + """ROS 2 node evaluating automated driving tasks, generating metrics-based evidence for benchmarking them.""" def __init__(self): """Constructor""" - super().__init__("autonomy_benchmarks") + super().__init__("autonomy_evaluation") self.auto_reconfigurable_params: list[str] = [] - self.benchmark = self.declare_and_load_parameter( - name="benchmark", + self.evaluation = self.declare_and_load_parameter( + name="evaluation", param_type=rclpy.Parameter.Type.STRING, - description="benchmark name", + description="evaluation name", default="nuscenes_lidar_object_detection", ) @@ -168,7 +168,7 @@ def __init__(self): self.results_path = self.declare_and_load_parameter( name="results_path", param_type=rclpy.Parameter.Type.STRING, - description="path of the JSON file the benchmark results are written to; results are only logged if empty", + description="path of the JSON file the evaluation results are written to; results are only logged if empty", default="", ) @@ -274,26 +274,26 @@ def setup(self): self.data_subscriptions: dict[str, Subscription] = {} - # get handler for specified benchmark - benchmark_handler = None - if self.benchmark == "nuscenes_lidar_object_detection": - from autonomy_benchmarks.benchmarks.lidar_object_detection.NuscenesLidarObjectDetection import ( + # get handler for specified evaluation + evaluation_handler = None + if self.evaluation == "nuscenes_lidar_object_detection": + from autonomy_evaluation.evaluations.lidar_object_detection.NuscenesLidarObjectDetection import ( NuscenesLidarObjectDetection, ) - benchmark_handler = NuscenesLidarObjectDetection() + evaluation_handler = NuscenesLidarObjectDetection() else: - self.get_logger().fatal(f"Benchmark '{self.benchmark}' not recognized, exiting") + self.get_logger().fatal(f"Evaluation '{self.evaluation}' not recognized, exiting") raise SystemExit(1) - # create subscriptions for benchmark data inputs, whose messages are matched into the + # create subscriptions for evaluation data inputs, whose messages are matched into the # samples to evaluate by their header stamp self.message_synchronizer = SampleSynchronizer( - topics=list(benchmark_handler.required_inputs()), + topics=list(evaluation_handler.required_inputs()), callback=self.evaluate_sample, queue_size=_SYNCHRONIZER_QUEUE_SIZE, ) - for msg_topic, msg_type in benchmark_handler.required_inputs().items(): + for msg_topic, msg_type in evaluation_handler.required_inputs().items(): self.data_subscriptions[msg_topic] = self.create_subscription( msg_type, msg_topic, @@ -306,10 +306,10 @@ def setup(self): ), ) - # create publishers visualizing the benchmark's per-sample matching outcome + # create publishers visualizing the evaluation's per-sample matching outcome self.visualization_publishers: dict[str, Publisher] = {} if self.visualize: - for msg_topic, msg_type in benchmark_handler.visualization_outputs().items(): + for msg_topic, msg_type in evaluation_handler.visualization_outputs().items(): self.visualization_publishers[msg_topic] = self.create_publisher( msg_type, f"~/{msg_topic}", @@ -320,11 +320,11 @@ def setup(self): depth=10, ), ) - self.get_logger().info(f"Visualizing benchmark results on: {sorted(self.visualization_publishers)}") + self.get_logger().info(f"Visualizing evaluation results on: {sorted(self.visualization_publishers)}") # store handler and ordered topic list for use in evaluate_sample - self.benchmark_handler = benchmark_handler - self.input_topics: list = list(benchmark_handler.required_inputs().keys()) + self.evaluation_handler = evaluation_handler + self.input_topics: list = list(evaluation_handler.required_inputs().keys()) self.published_sample_ids: list[int] = [] # A sample may already be evaluated before the dataset node answers the request that @@ -336,7 +336,7 @@ def setup(self): self.pending_request: Optional[Future] = None self.evaluation_deadline: Optional[float] = None self.publishing_finished = False - self.benchmark_finished = False + self.evaluation_finished = False self.num_evaluated_samples = 0 if self.manual_playback: @@ -347,22 +347,22 @@ def setup(self): return self.sample_request_client = self.create_client(RequestSamples, "~/request_samples") - # driven by a steady clock, so that the benchmark also advances while the simulation clock + # driven by a steady clock, so that the evaluation also advances while the simulation clock # of the dataset stands still, i.e. while no sample is being published self.request_timer = self.create_timer( _REQUEST_TIMER_PERIOD_S, - self.advance_benchmark, + self.advance_evaluation, clock=Clock(clock_type=ClockType.STEADY_TIME), ) self.get_logger().info(f"Requesting samples to evaluate from '{self.sample_request_client.srv_name}'") - def advance_benchmark(self): - """Requests the next samples to evaluate, or finalizes the benchmark once all were published + def advance_evaluation(self): + """Requests the next samples to evaluate, or finalizes the evaluation once all were published Called periodically as well as whenever a request has been answered or a sample has been - evaluated, and does nothing while the benchmark is waiting for one of those. + evaluated, and does nothing while the evaluation is waiting for one of those. """ - if self.benchmark_finished or self.pending_request is not None: + if self.evaluation_finished or self.pending_request is not None: return if not self.sample_request_client.service_is_ready(): self.get_logger().warn( @@ -373,7 +373,7 @@ def advance_benchmark(self): if self.awaiting_evaluations(): return if self.publishing_finished: - self.finalize_benchmark() + self.finalize_evaluation() self.shutdown() return self.request_samples() @@ -381,13 +381,13 @@ def advance_benchmark(self): def awaiting_evaluations(self) -> bool: """Reports whether published samples are still waiting to be evaluated - A sample is evaluated once all benchmark inputs have been received for it, which happens + A sample is evaluated once all evaluation inputs have been received for it, which happens once the system under test has processed the sample the dataset published. Samples that are not evaluated within 'evaluation_timeout' seconds are given up on, so that a system - under test which skips samples does not stall the benchmark. + under test which skips samples does not stall the evaluation. Returns: - bool: whether the benchmark waits for published samples to be evaluated + bool: whether the evaluation waits for published samples to be evaluated """ outstanding_evaluations = len(self.scenes_awaiting_sample) - len(self.samples_awaiting_scene) if outstanding_evaluations <= 0: @@ -425,7 +425,7 @@ def request_samples(self): self.pending_request.add_done_callback(self.samples_published_callback) def samples_published_callback(self, future: Future): - """Records the samples the dataset node has published and continues the benchmark + """Records the samples the dataset node has published and continues the evaluation Args: future (Future): future of the request, holding the response of the dataset node @@ -436,7 +436,7 @@ def samples_published_callback(self, future: Future): except Exception as exception: self.get_logger().error(f"Requesting samples of the dataset failed: {exception}") self.publishing_finished = True - self.advance_benchmark() + self.advance_evaluation() return published_sample_ids = [int(sample_id) for sample_id in response.published_sample_ids] @@ -455,12 +455,12 @@ def samples_published_callback(self, future: Future): self.publishing_finished = ( response.end_of_dataset or not response.success or bool(self.requested_sample_ids) or self.samples_per_request <= 0 ) - self.advance_benchmark() + self.advance_evaluation() def assign_scenes(self, published_scene_ids: Sequence[str]): """Attributes the scenes of published samples to the samples that are evaluated for them - The benchmark aggregates the metrics of the samples of a scene, so every evaluated sample + The evaluation aggregates the metrics of the samples of a scene, so every evaluated sample needs the scene the dataset published it from. Samples are evaluated in the order the dataset published them, so both are matched in that order; a scene whose sample has not been evaluated yet waits for it, and vice versa. @@ -475,12 +475,12 @@ def assign_scenes(self, published_scene_ids: Sequence[str]): else: self.scenes_awaiting_sample.append(str(scene_id)) - def finalize_benchmark(self, complete: bool = True): - """Aggregates the metrics of the evaluated samples per scene and over the whole benchmark + def finalize_evaluation(self, complete: bool = True): + """Aggregates the metrics of the evaluated samples per scene and over the whole evaluation The results hold the metrics of every single sample, of the samples of each scene, and of - all evaluated samples, of which the metrics over all samples are logged. Calling this on a - benchmark that has already been finalized does nothing, so that a benchmark which ran to + all evaluated samples, of which the metrics over all samples are logged. Calling this on an + evaluation that has already been finalized does nothing, so that an evaluation which ran to its end is not finalized a second time when the node shuts down. Args: @@ -488,31 +488,31 @@ def finalize_benchmark(self, complete: bool = True): results of an evaluation that was interrupted before its last sample, e.g. with Ctrl-C, are marked as incomplete via '"complete": false' """ - if self.benchmark_finished: + if self.evaluation_finished: return - self.benchmark_finished = True + self.evaluation_finished = True if self.request_timer is not None: self.request_timer.cancel() if not self.num_evaluated_samples: - self.get_logger().warn(f"Benchmark '{self.benchmark}' evaluated no sample, no metrics are aggregated") + self.get_logger().warn(f"Evaluation '{self.evaluation}' evaluated no sample, no metrics are aggregated") return - results = self.benchmark_handler.finalize(complete=complete) + results = self.evaluation_handler.finalize(complete=complete) aggregated_metrics = json.dumps(results["aggregated_metrics"], indent=2, default=str) evaluated_samples = f"{results['num_samples']} evaluated sample(s) of {results['num_scenes']} scene(s)" if complete: - self.get_logger().info(f"Benchmark '{self.benchmark}' finished after {evaluated_samples}.") + self.get_logger().info(f"Evaluation '{self.evaluation}' finished after {evaluated_samples}.") else: self.get_logger().warn( - f"Benchmark '{self.benchmark}' was interrupted after {evaluated_samples}, " + f"Evaluation '{self.evaluation}' was interrupted after {evaluated_samples}, " "its results are marked as incomplete." ) if self.results_path: - reported_results = "benchmark results" if complete else "incomplete benchmark results" + reported_results = "evaluation results" if complete else "incomplete evaluation results" try: - results_path = self.benchmark_handler.save_results(self.results_path, results=results) + results_path = self.evaluation_handler.save_results(self.results_path, results=results) self.get_logger().info(f"Wrote {reported_results} to '{results_path}'") except OSError as exception: self.get_logger().error(f"Failed to write {reported_results} to '{self.results_path}': {exception}") @@ -520,10 +520,10 @@ def finalize_benchmark(self, complete: bool = True): self.get_logger().info(f"Aggregated dataset metrics:\n{aggregated_metrics}") def shutdown(self): - """Stops the node, as there is nothing left to evaluate once the benchmark has finished + """Stops the node, as there is nothing left to evaluate once the evaluation has finished Shutting down the ROS context ends the spinning of the node in 'main', which lets the - process exit with the benchmark results reported. + process exit with the evaluation results reported. """ self.get_logger().info("Nothing left to evaluate, shutting down") rclpy.try_shutdown() @@ -533,10 +533,10 @@ def evaluate_sample(self, *args): The positional *args* are the synchronized ROS messages in the same order as the input names (dict keys) returned by - ``benchmark_handler.required_inputs()``. Those keys match the + ``evaluation_handler.required_inputs()``. Those keys match the ``compute_sample_metrics`` parameter names, so the raw ``ObjectList`` - messages are forwarded to ``benchmark_handler.record_sample`` by keyword; - the benchmark extracts the fields it needs inside + messages are forwarded to ``evaluation_handler.record_sample`` by keyword; + the evaluation extracts the fields it needs inside ``compute_sample_metrics``. Samples are identified by the ROS header stamp of their messages, which is the stamp the @@ -558,7 +558,7 @@ def evaluate_sample(self, *args): sample_id = f"{stamp.sec}.{stamp.nanosec:09d}" self.get_logger().debug(f"Sample ID: '{sample_id}'") - result = self.benchmark_handler.record_sample(sample_id=sample_id, **messages) + result = self.evaluation_handler.record_sample(sample_id=sample_id, **messages) self.num_evaluated_samples += 1 # attribute the sample to the scene the dataset published it from, which the dataset may # only report after the sample has been evaluated; with the playback controlled manually, @@ -576,31 +576,31 @@ def evaluate_sample(self, *args): # publish the sample's matching outcome for inspection in RViz if self.visualization_publishers: - visualization = self.benchmark_handler.visualize_sample(sample_id=sample_id, **messages) + visualization = self.evaluation_handler.visualize_sample(sample_id=sample_id, **messages) for msg_topic, publisher in self.visualization_publishers.items(): publisher.publish(visualization[msg_topic]) # request the next samples, or aggregate the dataset metrics if this was the last one; with # the playback controlled manually, the user requests the next samples instead if not self.manual_playback: - self.advance_benchmark() + self.advance_evaluation() def main(): """Initializes ROS, runs the node event loop, and performs shutdown cleanup.""" rclpy.init() - node = AutonomyBenchmarks() + node = AutonomyEvaluation() try: rclpy.spin(node) except (KeyboardInterrupt, ExternalShutdownException): pass finally: # An evaluation that is stopped before its last sample, e.g. with Ctrl-C, still reports the - # samples it did evaluate, marked as incomplete results. A benchmark that ran to its end + # samples it did evaluate, marked as incomplete results. An evaluation that ran to its end # has been finalized already and is left untouched. try: - node.finalize_benchmark(complete=False) + node.finalize_evaluation(complete=False) finally: node.destroy_node() rclpy.try_shutdown() diff --git a/autonomy_benchmarks/autonomy_benchmarks/benchmarks/AutonomyBenchmark.py b/autonomy_evaluation/autonomy_evaluation/evaluations/Evaluation.py similarity index 89% rename from autonomy_benchmarks/autonomy_benchmarks/benchmarks/AutonomyBenchmark.py rename to autonomy_evaluation/autonomy_evaluation/evaluations/Evaluation.py index 2f405a7..6e8f986 100644 --- a/autonomy_benchmarks/autonomy_benchmarks/benchmarks/AutonomyBenchmark.py +++ b/autonomy_evaluation/autonomy_evaluation/evaluations/Evaluation.py @@ -1,9 +1,9 @@ # Copyright Thinking Cars GmbH # SPDX-License-Identifier: Apache-2.0 -"""Abstract base class for all AutonomyHub benchmarks. +"""Abstract base class for all evaluations of automated driving tasks. -Each benchmark defines how to compute per-sample and aggregated metrics for a +Each evaluation defines how to compute per-sample and aggregated metrics for a specific perception task (e.g. 2-D / 3-D object detection). Concrete subclasses must override the abstract methods. """ @@ -16,10 +16,10 @@ from typing import Any, Dict, List, Optional -class AutonomyBenchmark(ABC): - """Meta-class (abstract base class) for AutonomyHub benchmarks. +class Evaluation(ABC): + """Meta-class (abstract base class) for the evaluations of automated driving tasks. - A benchmark is responsible for: + An evaluation is responsible for: * extracting the model input from a dataset sample, * computing per-sample metrics from a prediction and the ground-truth label, * aggregating per-sample metrics into scene-level and dataset-level metrics, @@ -27,13 +27,13 @@ class AutonomyBenchmark(ABC): """ def __init__(self, name: str, description: str = "") -> None: - """Initialize a benchmark definition and empty result store.""" + """Initialize an evaluation definition and empty result store.""" self.name: str = name self.description: str = description self._sample_results: List[Dict[str, Any]] = [] # ------------------------------------------------------------------ - # Abstract interface – must be implemented by every concrete benchmark + # Abstract interface – must be implemented by every concrete evaluation # ------------------------------------------------------------------ @abstractmethod @@ -84,12 +84,12 @@ def compute_aggregated_metrics(self, sample_results: List[Dict[str, Any]]) -> Di # ------------------------------------------------------------------ def visualization_outputs(self) -> Dict[str, Any]: - """Define the ROS message types this benchmark publishes for visualization. + """Define the ROS message types this evaluation publishes for visualization. Returns ------- A dictionary mapping output names to their ROS message types, empty for - a benchmark that offers no visualization. The names are node-relative + an evaluation that offers no visualization. The names are node-relative topics and match the keys of :meth:`visualize_sample`. """ return {} @@ -103,7 +103,7 @@ def visualize_sample(self, prediction: Any, label: Any, sample_id: Optional[str] Returns ------- A dictionary mapping the names of :meth:`visualization_outputs` to ready - ROS messages, empty for a benchmark that offers no visualization. + ROS messages, empty for an evaluation that offers no visualization. """ return {} @@ -123,7 +123,7 @@ def record_sample( This is the main entry point used by the evaluation loop. Any keyword arguments in *auxiliary* are forwarded verbatim to - :meth:`compute_sample_metrics`, allowing benchmarks to receive + :meth:`compute_sample_metrics`, allowing evaluations to receive auxiliary per-sample data (e.g. static-map annotations) without changing the abstract interface. @@ -171,7 +171,7 @@ def finalize(self, complete: bool = True) -> Dict[str, Any]: Parameters ---------- complete: - Whether all samples of the benchmark have been evaluated. An + Whether all samples of the evaluation have been evaluated. An evaluation that was interrupted, e.g. with Ctrl-C, still reports the samples it did evaluate, marked as ``"complete": false`` so that they are not mistaken for the results over the whole dataset. @@ -179,7 +179,7 @@ def finalize(self, complete: bool = True) -> Dict[str, Any]: aggregated = self.compute_aggregated_metrics(self._sample_results) scenes = self.sample_results_by_scene() return { - "benchmark": self.name, + "evaluation": self.name, "description": self.description, "complete": complete, "num_samples": len(self._sample_results), @@ -207,7 +207,7 @@ def save_results(self, output_path: str, results: Optional[Dict[str, Any]] = Non A results payload previously obtained from :meth:`finalize`, which is computed here when omitted. complete: - Whether all samples of the benchmark have been evaluated, see + Whether all samples of the evaluation have been evaluated, see :meth:`finalize`. Only used while the results are computed here; a given payload is written with the flag it was finalized with. diff --git a/autonomy_evaluation/autonomy_evaluation/evaluations/__init__.py b/autonomy_evaluation/autonomy_evaluation/evaluations/__init__.py new file mode 100644 index 0000000..b3d9504 --- /dev/null +++ b/autonomy_evaluation/autonomy_evaluation/evaluations/__init__.py @@ -0,0 +1,8 @@ +# Copyright Thinking Cars GmbH +# SPDX-License-Identifier: Apache-2.0 + +"""Evaluations of automated driving tasks, each computing the metrics of one task.""" + +from autonomy_evaluation.evaluations.Evaluation import Evaluation + +__all__ = ["Evaluation"] diff --git a/autonomy_benchmarks/autonomy_benchmarks/benchmarks/lidar_object_detection/NuscenesLidarObjectDetection.py b/autonomy_evaluation/autonomy_evaluation/evaluations/lidar_object_detection/NuscenesLidarObjectDetection.py similarity index 98% rename from autonomy_benchmarks/autonomy_benchmarks/benchmarks/lidar_object_detection/NuscenesLidarObjectDetection.py rename to autonomy_evaluation/autonomy_evaluation/evaluations/lidar_object_detection/NuscenesLidarObjectDetection.py index 519c975..42e4f0b 100644 --- a/autonomy_benchmarks/autonomy_benchmarks/benchmarks/lidar_object_detection/NuscenesLidarObjectDetection.py +++ b/autonomy_evaluation/autonomy_evaluation/evaluations/lidar_object_detection/NuscenesLidarObjectDetection.py @@ -1,7 +1,7 @@ # Copyright Thinking Cars GmbH # SPDX-License-Identifier: Apache-2.0 -"""nuScenes - Lidar object detection benchmark. +"""nuScenes - Lidar object detection evaluation. Evaluates 3D bounding-box predictions from lidar point cloud input against nuScenes ground-truth labels using the official nuScenes evaluation protocol. @@ -50,9 +50,9 @@ from typing import Any, Dict, List, Optional, Tuple import numpy as np -from autonomy_benchmarks.benchmarks.AutonomyBenchmark import AutonomyBenchmark -from autonomy_benchmarks.utils.ObjectDetectionUtils import ObjectDetectionUtils from autonomy_datasets_msgs.msg import ObjectListMetaInfo +from autonomy_evaluation.evaluations.Evaluation import Evaluation +from autonomy_evaluation.utils.ObjectDetectionUtils import ObjectDetectionUtils from perception_msgs.msg import ObjectClassification, ObjectList from perception_msgs_utils.state_getters import ( get_height, @@ -186,8 +186,8 @@ class DetectionClass: _SUPPORTED_CLASSES: frozenset = frozenset(set(_PERCEPTION_TYPE_TO_CLASS.values()) - {_CLASS_IGNORE}) & _DETECTION_CLASSES -class NuscenesLidarObjectDetection(AutonomyBenchmark): - """Benchmark for lidar 3D object detection on the nuScenes dataset. +class NuscenesLidarObjectDetection(Evaluation): + """Evaluation of lidar 3D object detection on the nuScenes dataset. See the module docstring for full evaluation settings. Objects are keyed by their canonical nuScenes detection class name. The classes evaluated with a @@ -202,7 +202,7 @@ def __init__(self) -> None: """Configure nuScenes thresholds, class ranges, and TP metric rules.""" super().__init__( name="nuscenes_lidar_object_detection", - description=("3D bounding-box object detection benchmark from lidar point clouds using the nuScenes dataset."), + description=("3D bounding-box object detection evaluation from lidar point clouds using the nuScenes dataset."), ) # BEV distance thresholds used for AP and TP-metric computation. @@ -514,7 +514,7 @@ def compute_aggregated_metrics(self, sample_results: List[Dict[str, Any]]) -> Di Returns: A dict containing ``threshold_metrics`` (per-threshold AP / mAP with the P-R filter), ``tp_metrics`` (per-class ATE, ASE, AOE, AVE, AAE - at the 2 m threshold), and ``benchmark_score`` (flat mAP and NDS + at the 2 m threshold), and ``score`` (flat mAP and NDS summary). """ all_match_records = [entry["metrics"]["match_records"] for entry in sample_results] @@ -526,13 +526,13 @@ def compute_aggregated_metrics(self, sample_results: List[Dict[str, Any]]) -> Di threshold_metrics = self._compute_threshold_metrics(cumulative_results) # Compute TP error metrics at the 2 m threshold. tp_metrics = self._compute_tp_metrics(merged_records) - # Compute benchmark score (mAP and NDS). - benchmark_score = self._compute_benchmark_score(threshold_metrics, tp_metrics) + # Compute the score (mAP and NDS). + score = self._compute_score(threshold_metrics, tp_metrics) return { "threshold_metrics": threshold_metrics, "tp_metrics": tp_metrics, - "benchmark_score": benchmark_score, + "score": score, } # ------------------------------------------------------------------ @@ -1055,7 +1055,7 @@ class with no GT/predictions falls back to worst-case ``1.0`` errors. } return tp_metrics - def _compute_benchmark_score( + def _compute_score( self, threshold_metrics: Dict[str, Any], tp_metrics: Dict[str, Dict[str, Any]], diff --git a/autonomy_benchmarks/autonomy_benchmarks/benchmarks/lidar_object_detection/__init__.py b/autonomy_evaluation/autonomy_evaluation/evaluations/lidar_object_detection/__init__.py similarity index 51% rename from autonomy_benchmarks/autonomy_benchmarks/benchmarks/lidar_object_detection/__init__.py rename to autonomy_evaluation/autonomy_evaluation/evaluations/lidar_object_detection/__init__.py index 3da1540..de18a9e 100644 --- a/autonomy_benchmarks/autonomy_benchmarks/benchmarks/lidar_object_detection/__init__.py +++ b/autonomy_evaluation/autonomy_evaluation/evaluations/lidar_object_detection/__init__.py @@ -1,9 +1,9 @@ # Copyright Thinking Cars GmbH # SPDX-License-Identifier: Apache-2.0 -"""LiDAR object detection benchmarks.""" +"""LiDAR object detection evaluations.""" -from autonomy_benchmarks.benchmarks.lidar_object_detection.NuscenesLidarObjectDetection import ( +from autonomy_evaluation.evaluations.lidar_object_detection.NuscenesLidarObjectDetection import ( NuscenesLidarObjectDetection, ) diff --git a/autonomy_benchmarks/autonomy_benchmarks/utils/ObjectDetectionUtils.py b/autonomy_evaluation/autonomy_evaluation/utils/ObjectDetectionUtils.py similarity index 97% rename from autonomy_benchmarks/autonomy_benchmarks/utils/ObjectDetectionUtils.py rename to autonomy_evaluation/autonomy_evaluation/utils/ObjectDetectionUtils.py index e610bbf..a381835 100644 --- a/autonomy_benchmarks/autonomy_benchmarks/utils/ObjectDetectionUtils.py +++ b/autonomy_evaluation/autonomy_evaluation/utils/ObjectDetectionUtils.py @@ -1,9 +1,9 @@ # Copyright Thinking Cars GmbH # SPDX-License-Identifier: Apache-2.0 -"""Utility helpers for object-detection benchmarks. +"""Utility helpers for object-detection evaluations. -Contains reusable functions shared across all benchmark implementations: +Contains reusable functions shared across all evaluation implementations: - **AP integration** (``compute_ap_101_point``): 101-point recall interpolation. - **TP metric integration** (``compute_tp_101_point``): average TP metric over a @@ -30,9 +30,9 @@ class ObjectDetectionUtils: - """Static helpers shared across the object-detection benchmarks. + """Static helpers shared across the object-detection evaluations. - Groups two kinds of stateless utilities used by every benchmark: + Groups two kinds of stateless utilities used by every evaluation: * Geometry on plain ``dict`` object records: 2D and volumetric 3D IoU, the rotated BEV footprint polygon, BEV center distance, and BEV diff --git a/autonomy_benchmarks/autonomy_benchmarks/utils/__init__.py b/autonomy_evaluation/autonomy_evaluation/utils/__init__.py similarity index 51% rename from autonomy_benchmarks/autonomy_benchmarks/utils/__init__.py rename to autonomy_evaluation/autonomy_evaluation/utils/__init__.py index 2089998..469f404 100644 --- a/autonomy_benchmarks/autonomy_benchmarks/utils/__init__.py +++ b/autonomy_evaluation/autonomy_evaluation/utils/__init__.py @@ -1,8 +1,8 @@ # Copyright Thinking Cars GmbH # SPDX-License-Identifier: Apache-2.0 -"""Utility functions for AutonomyHub benchmarks.""" +"""Utility functions shared by the evaluations.""" -from autonomy_benchmarks.utils.ObjectDetectionUtils import ObjectDetectionUtils +from autonomy_evaluation.utils.ObjectDetectionUtils import ObjectDetectionUtils __all__ = ["ObjectDetectionUtils"] diff --git a/autonomy_benchmarks/config/conf.rviz b/autonomy_evaluation/config/conf.rviz similarity index 99% rename from autonomy_benchmarks/config/conf.rviz rename to autonomy_evaluation/config/conf.rviz index af56693..571471e 100644 --- a/autonomy_benchmarks/config/conf.rviz +++ b/autonomy_evaluation/config/conf.rviz @@ -372,7 +372,7 @@ Visualization Manager: Filter size: 10 History Policy: Keep Last Reliability Policy: Reliable - Value: /autonomy_benchmarks/true_positives + Value: /autonomy_evaluation/true_positives Value: true Velocity arrow: Use velocity color: false @@ -447,7 +447,7 @@ Visualization Manager: Filter size: 10 History Policy: Keep Last Reliability Policy: Reliable - Value: /autonomy_benchmarks/false_positives + Value: /autonomy_evaluation/false_positives Value: true Velocity arrow: Use velocity color: false @@ -522,7 +522,7 @@ Visualization Manager: Filter size: 10 History Policy: Keep Last Reliability Policy: Reliable - Value: /autonomy_benchmarks/false_negatives + Value: /autonomy_evaluation/false_negatives Value: true Velocity arrow: Use velocity color: false diff --git a/autonomy_benchmarks/launch/autonomy_benchmarks.launch.py b/autonomy_evaluation/launch/autonomy_evaluation.launch.py similarity index 83% rename from autonomy_benchmarks/launch/autonomy_benchmarks.launch.py rename to autonomy_evaluation/launch/autonomy_evaluation.launch.py index 9a567d6..49f3b37 100644 --- a/autonomy_benchmarks/launch/autonomy_benchmarks.launch.py +++ b/autonomy_evaluation/launch/autonomy_evaluation.launch.py @@ -11,8 +11,8 @@ from launch_ros.parameter_descriptions import ParameterValue from launch_ros.substitutions import FindPackageShare -# Names of the benchmark's data inputs. Each name must match a key returned by -# the benchmark's ``required_inputs()`` (and thus a ``compute_sample_metrics`` +# Names of the evaluation's data inputs. Each name must match a key returned by +# the evaluation's ``required_inputs()`` (and thus a ``compute_sample_metrics`` # parameter). This single list is the source of truth: it drives both the # per-input launch arguments and the topic remappings below, so adding a new # input (e.g. "map", "radar") is a one-line change here. @@ -26,10 +26,10 @@ def generate_launch_description(): - """Create and return the launch description for the autonomy_benchmarks node.""" + """Create and return the launch description for the autonomy_evaluation node.""" - # Service of the dataset node the benchmark requests the samples to evaluate from, remapped - # onto the topic its argument resolves to just like the benchmark's data inputs. + # Service of the dataset node the evaluation requests the samples to evaluate from, remapped + # onto the topic its argument resolves to just like the evaluation's data inputs. remappable_topics = [ DeclareLaunchArgument( "request_samples", @@ -41,12 +41,12 @@ def generate_launch_description(): args = [ *remappable_topics, DeclareLaunchArgument( - "benchmark", + "evaluation", default_value="nuscenes_lidar_object_detection", - description="benchmark name", + description="evaluation name", choices=["nuscenes_lidar_object_detection"], ), - DeclareLaunchArgument("name", default_value="autonomy_benchmarks", description="node name"), + DeclareLaunchArgument("name", default_value="autonomy_evaluation", description="node name"), DeclareLaunchArgument("namespace", default_value="", description="node namespace"), DeclareLaunchArgument( "log_level", default_value="info", description="ROS logging level (debug, info, warn, error, fatal)" @@ -83,15 +83,15 @@ def generate_launch_description(): DeclareLaunchArgument( "results_path", default_value="", - description="path of the JSON file the benchmark results are written to (results are only logged if empty)", + description="path of the JSON file the evaluation results are written to (results are only logged if empty)", ), - # One argument per benchmark input; defaults to the node-relative name so + # One argument per evaluation input; defaults to the node-relative name so # an unset input is a no-op remap. Override with e.g. prediction:=/real/topic. *[ DeclareLaunchArgument( name, default_value=f"~/{name}", - description=f"real ROS topic feeding the benchmark's '{name}' input", + description=f"real ROS topic feeding the evaluation's '{name}' input", ) for name in _INPUTS ], @@ -104,12 +104,12 @@ def generate_launch_description(): remappings += [(la.default_value[0].text, LaunchConfiguration(la.name)) for la in remappable_topics] node = Node( - package="autonomy_benchmarks", - executable="autonomy_benchmarks", + package="autonomy_evaluation", + executable="autonomy_evaluation", namespace=LaunchConfiguration("namespace"), name=LaunchConfiguration("name"), parameters=[ - {"benchmark": LaunchConfiguration("benchmark")}, + {"evaluation": LaunchConfiguration("evaluation")}, {"visualize": ParameterValue(LaunchConfiguration("visualize"), value_type=bool)}, {"manual_playback": ParameterValue(LaunchConfiguration("manual_playback"), value_type=bool)}, {"samples_per_request": ParameterValue(LaunchConfiguration("samples_per_request"), value_type=int)}, @@ -128,7 +128,7 @@ def generate_launch_description(): name="rviz2", arguments=[ "-d", - PathJoinSubstitution([FindPackageShare("autonomy_benchmarks"), "config", "conf.rviz"]), + PathJoinSubstitution([FindPackageShare("autonomy_evaluation"), "config", "conf.rviz"]), ], condition=IfCondition(LaunchConfiguration("visualize")), output="screen", diff --git a/autonomy_benchmarks/package.xml b/autonomy_evaluation/package.xml similarity index 86% rename from autonomy_benchmarks/package.xml rename to autonomy_evaluation/package.xml index 2121ce3..18add39 100644 --- a/autonomy_benchmarks/package.xml +++ b/autonomy_evaluation/package.xml @@ -2,9 +2,9 @@ - autonomy_benchmarks + autonomy_evaluation 1.0.0 - Benchmarking suite for automated driving tasks + Metrics-based evaluation of automated driving tasks, generating the evidence for benchmarking automated driving deployments Haohao Hu Raphael van Kempen diff --git a/autonomy_benchmarks/resource/autonomy_benchmarks b/autonomy_evaluation/resource/autonomy_evaluation similarity index 100% rename from autonomy_benchmarks/resource/autonomy_benchmarks rename to autonomy_evaluation/resource/autonomy_evaluation diff --git a/autonomy_evaluation/setup.cfg b/autonomy_evaluation/setup.cfg new file mode 100644 index 0000000..59b63fe --- /dev/null +++ b/autonomy_evaluation/setup.cfg @@ -0,0 +1,4 @@ +[develop] +script_dir=$base/lib/autonomy_evaluation +[install] +install_scripts=$base/lib/autonomy_evaluation diff --git a/autonomy_benchmarks/setup.py b/autonomy_evaluation/setup.py similarity index 86% rename from autonomy_benchmarks/setup.py rename to autonomy_evaluation/setup.py index c481265..4043461 100644 --- a/autonomy_benchmarks/setup.py +++ b/autonomy_evaluation/setup.py @@ -6,7 +6,7 @@ from setuptools import find_packages, setup -package_name = "autonomy_benchmarks" +package_name = "autonomy_evaluation" setup( name=package_name, @@ -26,6 +26,6 @@ license="TODO: License declaration", tests_require=["pytest"], entry_points={ - "console_scripts": ["autonomy_benchmarks = autonomy_benchmarks.autonomy_benchmarks:main"], + "console_scripts": ["autonomy_evaluation = autonomy_evaluation.autonomy_evaluation:main"], }, ) diff --git a/autonomy_benchmarks/tests/__init__.py b/autonomy_evaluation/tests/__init__.py similarity index 56% rename from autonomy_benchmarks/tests/__init__.py rename to autonomy_evaluation/tests/__init__.py index 23e5cd9..f3e5696 100644 --- a/autonomy_benchmarks/tests/__init__.py +++ b/autonomy_evaluation/tests/__init__.py @@ -1,4 +1,4 @@ # Copyright Thinking Cars GmbH # SPDX-License-Identifier: Apache-2.0 -"""Test suite for the autonomy_benchmarks package.""" +"""Test suite for the autonomy_evaluation package.""" diff --git a/autonomy_benchmarks/tests/benchmarks/lidar_object_detection/__init__.py b/autonomy_evaluation/tests/evaluations/__init__.py similarity index 56% rename from autonomy_benchmarks/tests/benchmarks/lidar_object_detection/__init__.py rename to autonomy_evaluation/tests/evaluations/__init__.py index 65f82a8..ae03f57 100644 --- a/autonomy_benchmarks/tests/benchmarks/lidar_object_detection/__init__.py +++ b/autonomy_evaluation/tests/evaluations/__init__.py @@ -1,4 +1,4 @@ # Copyright Thinking Cars GmbH # SPDX-License-Identifier: Apache-2.0 -"""Tests for the nuScenes lidar benchmark.""" +"""Tests of the evaluations of autonomy_evaluation.""" diff --git a/autonomy_benchmarks/autonomy_benchmarks/__init__.py b/autonomy_evaluation/tests/evaluations/lidar_object_detection/__init__.py similarity index 59% rename from autonomy_benchmarks/autonomy_benchmarks/__init__.py rename to autonomy_evaluation/tests/evaluations/lidar_object_detection/__init__.py index ebbf0e0..cfb62fd 100644 --- a/autonomy_benchmarks/autonomy_benchmarks/__init__.py +++ b/autonomy_evaluation/tests/evaluations/lidar_object_detection/__init__.py @@ -1,4 +1,4 @@ # Copyright Thinking Cars GmbH # SPDX-License-Identifier: Apache-2.0 -"""Autonomy benchmarks package.""" +"""Tests for the nuScenes lidar evaluation.""" diff --git a/autonomy_benchmarks/tests/benchmarks/lidar_object_detection/test_NuscenesLidarObjectDetection.py b/autonomy_evaluation/tests/evaluations/lidar_object_detection/test_NuscenesLidarObjectDetection.py similarity index 95% rename from autonomy_benchmarks/tests/benchmarks/lidar_object_detection/test_NuscenesLidarObjectDetection.py rename to autonomy_evaluation/tests/evaluations/lidar_object_detection/test_NuscenesLidarObjectDetection.py index a073600..d63cdcf 100644 --- a/autonomy_benchmarks/tests/benchmarks/lidar_object_detection/test_NuscenesLidarObjectDetection.py +++ b/autonomy_evaluation/tests/evaluations/lidar_object_detection/test_NuscenesLidarObjectDetection.py @@ -1,10 +1,10 @@ # Copyright Thinking Cars GmbH # SPDX-License-Identifier: Apache-2.0 -"""Tests for NuscenesLidarObjectDetection benchmark. +"""Tests for NuscenesLidarObjectDetection evaluation. Tests focus on public API methods: compute_sample_metrics(), -compute_aggregated_metrics(), and benchmark configuration. +compute_aggregated_metrics(), and evaluation configuration. Inputs are real ROS messages (built via the helpers below), matching how ``_extract_objects`` reads them: geometry through the ``perception_msgs_utils`` @@ -21,10 +21,10 @@ from typing import List, Optional, Tuple import pytest -from autonomy_benchmarks.benchmarks.lidar_object_detection.NuscenesLidarObjectDetection import ( +from autonomy_datasets_msgs.msg import ObjectListMetaInfo, ObjectMetaInfo +from autonomy_evaluation.evaluations.lidar_object_detection.NuscenesLidarObjectDetection import ( NuscenesLidarObjectDetection, ) -from autonomy_datasets_msgs.msg import ObjectListMetaInfo, ObjectMetaInfo from diagnostic_msgs.msg import KeyValue from perception_msgs.msg import HEXAMOTION, Object, ObjectClassification, ObjectList, ObjectState @@ -164,20 +164,20 @@ def _label( class TestNuscenesLidarObjectDetection: - """Tests for NuscenesLidarObjectDetection benchmark. + """Tests for NuscenesLidarObjectDetection evaluation. Covers sample metrics computation (Pass 1), aggregated metrics (Pass 2), and NDS score calculation. """ def setup_method(self): - """Create a fresh benchmark instance for each test.""" + """Create a fresh evaluation instance for each test.""" self.bm = NuscenesLidarObjectDetection() def _metrics(self, pred_objs, gts, reverse_meta: bool = False) -> dict: """Compute sample metrics from prediction objects and ``_gt`` pairs. - Mirrors the node's synchronized callback, which hands the benchmark the + Mirrors the node's synchronized callback, which hands the evaluation the label object list and its meta info message together. """ label, label_meta_info = _label(gts, reverse_meta=reverse_meta) @@ -367,19 +367,19 @@ def test_perfect_matching_yields_high_map(self): preds = [_pred(x=float(i), confidence=1.0 - i * 0.01) for i in range(n)] gts = [_gt(x=float(i), num_lidar_pts=5) for i in range(n)] result = self.bm.compute_aggregated_metrics([self._make_sample_result(preds, gts)]) - assert result["benchmark_score"]["map"] >= 0.8 + assert result["score"]["map"] >= 0.8 def test_no_pred_yields_zero_map(self): """Verify missing predictions yield zero mAP.""" gts = [_gt(x=0.0, num_lidar_pts=5)] result = self.bm.compute_aggregated_metrics([self._make_sample_result([], gts)]) - assert result["benchmark_score"]["map"] == 0.0 + assert result["score"]["map"] == 0.0 def test_all_fp_yields_zero_map(self): """Verify all false positives yield zero mAP.""" preds = [_pred(x=float(i), confidence=0.9) for i in range(5)] result = self.bm.compute_aggregated_metrics([self._make_sample_result(preds, [])]) - assert result["benchmark_score"]["map"] == 0.0 + assert result["score"]["map"] == 0.0 def test_unsupported_class_gt_excluded_from_map(self): """A GT class the interface cannot express (truck) must not dilute mAP. @@ -393,7 +393,7 @@ def test_unsupported_class_gt_excluded_from_map(self): gts.append(_gt(x=5.0, y=20.0, original_class="vehicle.truck", num_lidar_pts=5)) result = self.bm.compute_aggregated_metrics([self._make_sample_result(preds, gts)]) # truck is unsupported -> excluded; mAP stays high (car only). - assert result["benchmark_score"]["map"] >= 0.8 + assert result["score"]["map"] >= 0.8 assert "truck" not in self.bm.supported_classes def test_supported_but_undetected_class_still_counts_as_zero(self): @@ -408,13 +408,13 @@ def test_supported_but_undetected_class_still_counts_as_zero(self): gts.append(_gt(x=5.0, y=20.0, original_class="human.pedestrian.adult", num_lidar_pts=5)) result = self.bm.compute_aggregated_metrics([self._make_sample_result(preds, gts)]) # pedestrian is supported -> AP 0 pulls the mean well below the car-only case. - assert result["benchmark_score"]["map"] < 0.6 + assert result["score"]["map"] < 0.6 assert "pedestrian" in self.bm.supported_classes def test_capability_lists_reported(self): """The score declares which classes the interface can / cannot express.""" gts = [_gt(x=0.0, num_lidar_pts=5)] - score = self.bm.compute_aggregated_metrics([self._make_sample_result([], gts)])["benchmark_score"] + score = self.bm.compute_aggregated_metrics([self._make_sample_result([], gts)])["score"] assert score["supported_classes"] == sorted(self.bm.supported_classes) assert "car" in score["supported_classes"] assert "truck" in score["unsupported_classes"] @@ -431,7 +431,7 @@ def test_multi_frame_accumulation(self): # --- visualization of the per-sample matching outcome --- def test_visualization_outputs_are_object_lists(self): - """The benchmark declares one ObjectList output per matching outcome.""" + """The evaluation declares one ObjectList output per matching outcome.""" assert self.bm.visualization_outputs() == { "true_positives": ObjectList, "false_positives": ObjectList, @@ -488,7 +488,7 @@ def _tm(overall_map: float) -> dict: def test_nds_with_valid_aae(self): """Verify NDS uses full 6-term formula when AAE data is available.""" tp_metrics = {"car": {"ate": 0.2, "ase": 0.1, "aoe": 0.3, "ave": 0.4, "aae": 0.05}} - score = self.bm._compute_benchmark_score(self._tm(0.5), tp_metrics) + score = self.bm._compute_score(self._tm(0.5), tp_metrics) expected = ( 5.0 * 0.5 + max(1.0 - 0.2, 0.0) @@ -503,13 +503,13 @@ def test_nds_without_aae_is_lower(self): """Verify missing AAE data (falls back to 1.0) reduces NDS score.""" tp_with = {"car": {"ate": 0.0, "ase": 0.0, "aoe": 0.0, "ave": 0.0, "aae": 0.0}} tp_without = {"car": {"ate": 0.0, "ase": 0.0, "aoe": 0.0, "ave": 0.0, "aae": None}} - score_with = self.bm._compute_benchmark_score(self._tm(1.0), tp_with) - score_without = self.bm._compute_benchmark_score(self._tm(1.0), tp_without) + score_with = self.bm._compute_score(self._tm(1.0), tp_with) + score_without = self.bm._compute_score(self._tm(1.0), tp_without) assert score_with["nds"] == pytest.approx(1.0, abs=1e-4) assert score_without["nds"] == pytest.approx(0.9, abs=1e-4) def test_nds_clamps_tp_errors(self): """Verify TP errors above 1.0 are clamped in NDS computation.""" tp_metrics = {"car": {"ate": 2.0, "ase": 3.0, "aoe": 5.0, "ave": 1.5, "aae": 2.0}} - score = self.bm._compute_benchmark_score(self._tm(0.0), tp_metrics) + score = self.bm._compute_score(self._tm(0.0), tp_metrics) assert score["nds"] >= 0.0 diff --git a/autonomy_benchmarks/tests/benchmarks/test_AutonomyBenchmark.py b/autonomy_evaluation/tests/evaluations/test_Evaluation.py similarity index 70% rename from autonomy_benchmarks/tests/benchmarks/test_AutonomyBenchmark.py rename to autonomy_evaluation/tests/evaluations/test_Evaluation.py index 4bc831e..0e76a2f 100644 --- a/autonomy_benchmarks/tests/benchmarks/test_AutonomyBenchmark.py +++ b/autonomy_evaluation/tests/evaluations/test_Evaluation.py @@ -1,10 +1,10 @@ # Copyright Thinking Cars GmbH # SPDX-License-Identifier: Apache-2.0 -"""Tests for the result store of the AutonomyBenchmark base class. +"""Tests for the result store of the Evaluation base class. Metrics are reported on three levels: for every single sample, aggregated over the samples of each -scene of the dataset, and aggregated over all evaluated samples. A minimal benchmark whose metrics +scene of the dataset, and aggregated over all evaluated samples. A minimal evaluation whose metrics are trivial to predict is used, so that the tests cover the grouping and not a metric definition. """ @@ -13,18 +13,18 @@ import json from typing import Any, Dict, List -from autonomy_benchmarks.benchmarks.AutonomyBenchmark import AutonomyBenchmark +from autonomy_evaluation.evaluations.Evaluation import Evaluation -class CountingBenchmark(AutonomyBenchmark): - """Minimal benchmark counting the objects of a sample and summing them up when aggregating.""" +class CountingEvaluation(Evaluation): + """Minimal evaluation counting the objects of a sample and summing them up when aggregating.""" def __init__(self) -> None: - """Name the benchmark.""" + """Name the evaluation.""" super().__init__(name="counting", description="counts objects") def required_inputs(self) -> Dict[str, Any]: - """Declare the inputs, which this benchmark does not read from ROS messages.""" + """Declare the inputs, which this evaluation does not read from ROS messages.""" return {"prediction": object, "label": object} def compute_sample_metrics(self, prediction: Any, label: Any, sample_id: str = None) -> Dict[str, Any]: @@ -39,44 +39,44 @@ def compute_aggregated_metrics(self, sample_results: List[Dict[str, Any]]) -> Di } -def _benchmark_of(samples) -> CountingBenchmark: - """Record ``(sample_id, scene_id, num_predictions, num_labels)`` samples in a benchmark.""" - benchmark = CountingBenchmark() +def _evaluation_of(samples) -> CountingEvaluation: + """Record ``(sample_id, scene_id, num_predictions, num_labels)`` samples in an evaluation.""" + evaluation = CountingEvaluation() for sample_id, scene_id, num_predictions, num_labels in samples: - benchmark.record_sample(prediction=num_predictions, label=num_labels, sample_id=sample_id, scene_id=scene_id) - return benchmark + evaluation.record_sample(prediction=num_predictions, label=num_labels, sample_id=sample_id, scene_id=scene_id) + return evaluation class TestSampleResults: """Tests recording samples with the scene of the dataset they belong to.""" # The recorded samples themselves are no longer reported alongside the aggregated - # results, while 'sample_results' is commented out in AutonomyBenchmark.finalize() + # results, while 'sample_results' is commented out in Evaluation.finalize() # def test_records_sample_with_its_scene(self): # """A recorded sample keeps its ID, its scene and its metrics.""" - # benchmark = _benchmark_of([("0", "scene_a", 2, 3)]) + # evaluation = _evaluation_of([("0", "scene_a", 2, 3)]) # - # assert benchmark.finalize()["sample_results"] == [ + # assert evaluation.finalize()["sample_results"] == [ # {"sample_id": "0", "scene_id": "scene_a", "metrics": {"num_predictions": 2, "num_labels": 3}} # ] def test_scene_can_be_set_after_the_sample_was_recorded(self): """An evaluation loop that learns the scene late sets it on the returned entry.""" - benchmark = CountingBenchmark() + evaluation = CountingEvaluation() - entry = benchmark.record_sample(prediction=1, label=1, sample_id="0") + entry = evaluation.record_sample(prediction=1, label=1, sample_id="0") assert entry["scene_id"] is None entry["scene_id"] = "scene_a" - assert benchmark.sample_results_by_scene() == {"scene_a": [entry]} + assert evaluation.sample_results_by_scene() == {"scene_a": [entry]} class TestFinalize: - """Tests aggregating the recorded samples per scene and over the whole benchmark.""" + """Tests aggregating the recorded samples per scene and over the whole evaluation.""" - def test_aggregates_per_sample_scene_and_benchmark(self): + def test_aggregates_per_sample_scene_and_evaluation(self): """Metrics are reported for every sample, every scene and all samples together.""" - results = _benchmark_of( + results = _evaluation_of( [ ("0", "scene_a", 1, 1), ("1", "scene_a", 2, 3), @@ -94,7 +94,7 @@ def test_aggregates_per_sample_scene_and_benchmark(self): } assert results["scene_results"]["scene_b"]["aggregated_metrics"] == {"num_predictions": 4, "num_labels": 5} # The metrics of the single samples are no longer reported alongside the aggregated - # results, while 'sample_results' is commented out in AutonomyBenchmark.finalize() + # results, while 'sample_results' is commented out in Evaluation.finalize() # assert [entry["metrics"] for entry in results["sample_results"]] == [ # {"num_predictions": 1, "num_labels": 1}, # {"num_predictions": 2, "num_labels": 3}, @@ -103,7 +103,7 @@ def test_aggregates_per_sample_scene_and_benchmark(self): def test_groups_samples_of_a_scene_that_are_not_recorded_consecutively(self): """Samples are grouped by their scene, not by the order they were recorded in.""" - results = _benchmark_of( + results = _evaluation_of( [ ("0", "scene_a", 1, 0), ("1", "scene_b", 2, 0), @@ -115,9 +115,9 @@ def test_groups_samples_of_a_scene_that_are_not_recorded_consecutively(self): assert results["scene_results"]["scene_a"]["aggregated_metrics"]["num_predictions"] == 5 assert results["scene_results"]["scene_b"]["sample_ids"] == ["1"] - def test_samples_without_a_scene_are_only_aggregated_over_the_benchmark(self): - """A sample that cannot be attributed to a scene still counts for the whole benchmark.""" - results = _benchmark_of([("0", "scene_a", 1, 0), ("1", None, 2, 0)]).finalize() + def test_samples_without_a_scene_are_only_aggregated_over_the_evaluation(self): + """A sample that cannot be attributed to a scene still counts for the whole evaluation.""" + results = _evaluation_of([("0", "scene_a", 1, 0), ("1", None, 2, 0)]).finalize() assert results["num_samples"] == 2 assert results["num_scenes"] == 1 @@ -126,7 +126,7 @@ def test_samples_without_a_scene_are_only_aggregated_over_the_benchmark(self): def test_reports_no_scene_without_recorded_scenes(self): """Samples recorded without a scene aggregate to no scene results at all.""" - results = _benchmark_of([("0", None, 1, 0)]).finalize() + results = _evaluation_of([("0", None, 1, 0)]).finalize() assert results["num_scenes"] == 0 assert results["scene_results"] == {} @@ -135,25 +135,25 @@ def test_reports_no_scene_without_recorded_scenes(self): class TestSaveResults: """Tests writing the results of all three levels to a JSON file.""" - def test_writes_sample_scene_and_benchmark_metrics(self, tmp_path): - """The stored results hold the metrics of every sample, every scene and the benchmark.""" - benchmark = _benchmark_of([("0", "scene_a", 1, 1), ("1", "scene_b", 2, 2)]) + def test_writes_sample_scene_and_evaluation_metrics(self, tmp_path): + """The stored results hold the metrics of every sample, every scene and the evaluation.""" + evaluation = _evaluation_of([("0", "scene_a", 1, 1), ("1", "scene_b", 2, 2)]) - output_path = benchmark.save_results(str(tmp_path / "results" / "counting.json")) + output_path = evaluation.save_results(str(tmp_path / "results" / "counting.json")) stored = json.loads(open(output_path).read()) assert stored["aggregated_metrics"] == {"num_predictions": 3, "num_labels": 3} assert sorted(stored["scene_results"]) == ["scene_a", "scene_b"] # The single samples are no longer written alongside the aggregated - # results, while 'sample_results' is commented out in AutonomyBenchmark.finalize() + # results, while 'sample_results' is commented out in Evaluation.finalize() # assert [entry["sample_id"] for entry in stored["sample_results"]] == ["0", "1"] def test_writes_previously_computed_results(self, tmp_path): """Results that have already been computed are written as they are.""" - benchmark = _benchmark_of([("0", "scene_a", 1, 1)]) - results = benchmark.finalize() + evaluation = _evaluation_of([("0", "scene_a", 1, 1)]) + results = evaluation.finalize() - output_path = benchmark.save_results(str(tmp_path / "counting.json"), results=results) + output_path = evaluation.save_results(str(tmp_path / "counting.json"), results=results) assert json.loads(open(output_path).read())["aggregated_metrics"] == results["aggregated_metrics"] @@ -162,12 +162,12 @@ class TestIncompleteResults: """Tests marking the results of an evaluation that did not process all samples.""" def test_results_of_all_samples_are_complete(self): - """Results aggregated after the last sample of the benchmark are marked as complete.""" - assert _benchmark_of([("0", "scene_a", 1, 1)]).finalize()["complete"] is True + """Results aggregated after the last sample of the evaluation are marked as complete.""" + assert _evaluation_of([("0", "scene_a", 1, 1)]).finalize()["complete"] is True def test_interrupted_results_are_marked_incomplete(self): """Results aggregated before the last sample, e.g. after Ctrl-C, are marked as incomplete.""" - results = _benchmark_of([("0", "scene_a", 1, 1)]).finalize(complete=False) + results = _evaluation_of([("0", "scene_a", 1, 1)]).finalize(complete=False) assert results["complete"] is False # the samples that were evaluated are still reported @@ -176,9 +176,9 @@ def test_interrupted_results_are_marked_incomplete(self): def test_writes_incomplete_results_to_the_results_file(self, tmp_path): """The results file of an interrupted evaluation marks the results it holds as incomplete.""" - benchmark = _benchmark_of([("0", "scene_a", 1, 1)]) + evaluation = _evaluation_of([("0", "scene_a", 1, 1)]) - output_path = benchmark.save_results(str(tmp_path / "counting.json"), complete=False) + output_path = evaluation.save_results(str(tmp_path / "counting.json"), complete=False) stored = json.loads(open(output_path).read()) assert stored["complete"] is False @@ -186,9 +186,9 @@ def test_writes_incomplete_results_to_the_results_file(self, tmp_path): def test_written_results_keep_the_flag_they_were_finalized_with(self, tmp_path): """A given payload is written as it is, marked the way it was finalized.""" - benchmark = _benchmark_of([("0", "scene_a", 1, 1)]) - results = benchmark.finalize(complete=False) + evaluation = _evaluation_of([("0", "scene_a", 1, 1)]) + results = evaluation.finalize(complete=False) - output_path = benchmark.save_results(str(tmp_path / "counting.json"), results=results) + output_path = evaluation.save_results(str(tmp_path / "counting.json"), results=results) assert json.loads(open(output_path).read())["complete"] is False diff --git a/autonomy_benchmarks/tests/test_autonomy_benchmarks.py b/autonomy_evaluation/tests/test_autonomy_evaluation.py similarity index 76% rename from autonomy_benchmarks/tests/test_autonomy_benchmarks.py rename to autonomy_evaluation/tests/test_autonomy_evaluation.py index 651eed7..5f0374e 100644 --- a/autonomy_benchmarks/tests/test_autonomy_benchmarks.py +++ b/autonomy_evaluation/tests/test_autonomy_evaluation.py @@ -1,7 +1,7 @@ # Copyright Thinking Cars GmbH # SPDX-License-Identifier: Apache-2.0 -"""Tests for the helpers of the autonomy_benchmarks node. +"""Tests for the helpers of the autonomy_evaluation node. The node itself drives the evaluation via the ``request_samples`` service of the dataset and is covered by running it against a dataset. Tested here are the parsing of the samples to evaluate, @@ -18,7 +18,7 @@ from types import SimpleNamespace import pytest -from autonomy_benchmarks.autonomy_benchmarks import AutonomyBenchmarks, parse_sample_ids, SampleSynchronizer +from autonomy_evaluation.autonomy_evaluation import AutonomyEvaluation, parse_sample_ids, SampleSynchronizer _TOPICS = ["prediction", "label", "label_meta_info"] @@ -34,7 +34,7 @@ def _message(stamp: tuple[int, int]) -> SimpleNamespace: def _synchronizer(queue_size: int = 10) -> tuple[SampleSynchronizer, list]: - """Create a synchronizer of the benchmark inputs next to the list of the samples it matched.""" + """Create a synchronizer of the evaluation inputs next to the list of the samples it matched.""" matched_samples: list = [] synchronizer = SampleSynchronizer(_TOPICS, callback=lambda *msgs: matched_samples.append(msgs), queue_size=queue_size) return synchronizer, matched_samples @@ -76,7 +76,7 @@ def test_rejects_values_that_are_no_ids(self, sample_ids): class TestSampleSynchronizer: - """Tests matching the messages of the benchmark inputs into the samples to evaluate.""" + """Tests matching the messages of the evaluation inputs into the samples to evaluate.""" def test_matches_the_messages_of_a_sample_in_input_order(self): """A sample is reported once every input has been received, in the order of the inputs.""" @@ -152,8 +152,8 @@ def info(self, message: str, **kwargs) -> None: debug = warn = error = info -class _FakeBenchmarkHandler: - """Stand in for the benchmark whose results the node aggregates and writes.""" +class _FakeEvaluationHandler: + """Stand in for the evaluation whose results the node aggregates and writes.""" def __init__(self): """Start without finalized or written results.""" @@ -171,26 +171,26 @@ def save_results(self, output_path: str, results: dict = None) -> str: return output_path -class _RecordingBenchmarkHandler: - """Stand in for the benchmark that records the samples the node evaluates.""" +class _RecordingEvaluationHandler: + """Stand in for the evaluation that records the samples the node evaluates.""" def __init__(self): """Start without recorded samples.""" self.recorded_samples: list = [] def record_sample(self, sample_id: str, **messages) -> dict: - """Record a sample the way the benchmark stores it, still without a scene.""" + """Record a sample the way the evaluation stores it, still without a scene.""" entry = {"sample_id": sample_id, "scene_id": None, "metrics": {}} self.recorded_samples.append(entry) return entry def _evaluating_node(manual_playback: bool) -> SimpleNamespace: - """Stub the node state that evaluating a sample reads, counting the attempts to continue the benchmark.""" + """Stub the node state that evaluating a sample reads, counting the attempts to continue the evaluation.""" node = SimpleNamespace( manual_playback=manual_playback, input_topics=_TOPICS, - benchmark_handler=_RecordingBenchmarkHandler(), + evaluation_handler=_RecordingEvaluationHandler(), num_evaluated_samples=0, scenes_awaiting_sample=deque(), samples_awaiting_scene=deque(), @@ -201,7 +201,7 @@ def _evaluating_node(manual_playback: bool) -> SimpleNamespace: num_advances=0, get_logger=lambda logger=_FakeLogger(): logger, ) - node.advance_benchmark = lambda: setattr(node, "num_advances", node.num_advances + 1) + node.advance_evaluation = lambda: setattr(node, "num_advances", node.num_advances + 1) return node @@ -211,24 +211,24 @@ def _sample_messages(stamp: tuple[int, int]) -> tuple: class TestEvaluateSample: - """Tests evaluating a sample with the playback driven by the benchmark or by the user in RViz.""" + """Tests evaluating a sample with the playback driven by the evaluation or by the user in RViz.""" - def test_benchmark_requests_the_next_samples_after_evaluating_one(self): - """Without manual playback, the benchmark continues with the next samples on its own.""" + def test_evaluation_requests_the_next_samples_after_evaluating_one(self): + """Without manual playback, the evaluation continues with the next samples on its own.""" node = _evaluating_node(manual_playback=False) - AutonomyBenchmarks.evaluate_sample(node, *_sample_messages(_NEXT_SCENE[0])) + AutonomyEvaluation.evaluate_sample(node, *_sample_messages(_NEXT_SCENE[0])) assert node.num_advances == 1 # the dataset reports the scene of the sample with the response to the request - assert list(node.samples_awaiting_scene) == node.benchmark_handler.recorded_samples + assert list(node.samples_awaiting_scene) == node.evaluation_handler.recorded_samples def test_manual_playback_leaves_requesting_samples_to_the_user(self): """With manual playback, samples are evaluated as they arrive, without requesting further ones.""" node = _evaluating_node(manual_playback=True) for stamp in _NEXT_SCENE: - AutonomyBenchmarks.evaluate_sample(node, *_sample_messages(stamp)) + AutonomyEvaluation.evaluate_sample(node, *_sample_messages(stamp)) assert node.num_evaluated_samples == len(_NEXT_SCENE) assert node.num_advances == 0 @@ -236,12 +236,12 @@ def test_manual_playback_leaves_requesting_samples_to_the_user(self): assert not node.samples_awaiting_scene -def _node(num_evaluated_samples: int = 2, results_path: str = "/results/benchmark.json") -> SimpleNamespace: - """Stub the node state that finalizing a benchmark reads, without initializing ROS.""" +def _node(num_evaluated_samples: int = 2, results_path: str = "/results/evaluation.json") -> SimpleNamespace: + """Stub the node state that finalizing an evaluation reads, without initializing ROS.""" return SimpleNamespace( - benchmark="counting", - benchmark_finished=False, - benchmark_handler=_FakeBenchmarkHandler(), + evaluation="counting", + evaluation_finished=False, + evaluation_handler=_FakeEvaluationHandler(), num_evaluated_samples=num_evaluated_samples, results_path=results_path, request_timer=SimpleNamespace(cancel=lambda: None), @@ -249,59 +249,59 @@ def _node(num_evaluated_samples: int = 2, results_path: str = "/results/benchmar ) -class TestFinalizeBenchmark: - """Tests reporting the results of a benchmark that ran to its end or was interrupted.""" +class TestFinalizeEvaluation: + """Tests reporting the results of an evaluation that ran to its end or was interrupted.""" - def test_finished_benchmark_writes_complete_results(self): - """A benchmark that evaluated all its samples reports complete results.""" + def test_finished_evaluation_writes_complete_results(self): + """An evaluation that evaluated all its samples reports complete results.""" node = _node() - AutonomyBenchmarks.finalize_benchmark(node) + AutonomyEvaluation.finalize_evaluation(node) - assert node.benchmark_handler.finalized_complete is True - assert node.benchmark_handler.written_results["complete"] is True + assert node.evaluation_handler.finalized_complete is True + assert node.evaluation_handler.written_results["complete"] is True - def test_interrupted_benchmark_writes_incomplete_results(self): + def test_interrupted_evaluation_writes_incomplete_results(self): """An evaluation stopped before its last sample, e.g. with Ctrl-C, still writes its results.""" node = _node() - AutonomyBenchmarks.finalize_benchmark(node, complete=False) + AutonomyEvaluation.finalize_evaluation(node, complete=False) - assert node.benchmark_handler.finalized_complete is False - assert node.benchmark_handler.written_results["complete"] is False + assert node.evaluation_handler.finalized_complete is False + assert node.evaluation_handler.written_results["complete"] is False - def test_finished_benchmark_is_not_finalized_again_on_shutdown(self): + def test_finished_evaluation_is_not_finalized_again_on_shutdown(self): """Shutting down after the last sample must not overwrite the results with incomplete ones.""" node = _node() - AutonomyBenchmarks.finalize_benchmark(node) + AutonomyEvaluation.finalize_evaluation(node) - AutonomyBenchmarks.finalize_benchmark(node, complete=False) + AutonomyEvaluation.finalize_evaluation(node, complete=False) - assert node.benchmark_handler.finalized_complete is True - assert node.benchmark_handler.written_results["complete"] is True + assert node.evaluation_handler.finalized_complete is True + assert node.evaluation_handler.written_results["complete"] is True def test_manual_playback_writes_its_results_when_stopped(self): """With manual playback, which runs no request timer, the results are written once the node is stopped.""" node = _node() node.request_timer = None - AutonomyBenchmarks.finalize_benchmark(node, complete=False) + AutonomyEvaluation.finalize_evaluation(node, complete=False) - assert node.benchmark_handler.written_results["complete"] is False + assert node.evaluation_handler.written_results["complete"] is False - def test_interrupted_benchmark_without_samples_writes_nothing(self): + def test_interrupted_evaluation_without_samples_writes_nothing(self): """An evaluation interrupted before its first sample has no results to write.""" node = _node(num_evaluated_samples=0) - AutonomyBenchmarks.finalize_benchmark(node, complete=False) + AutonomyEvaluation.finalize_evaluation(node, complete=False) - assert node.benchmark_handler.written_results is None + assert node.evaluation_handler.written_results is None def test_results_are_only_logged_without_a_results_path(self): """Without 'results_path' the interrupted results are logged instead of written.""" node = _node(results_path="") - AutonomyBenchmarks.finalize_benchmark(node, complete=False) + AutonomyEvaluation.finalize_evaluation(node, complete=False) - assert node.benchmark_handler.finalized_complete is False - assert node.benchmark_handler.written_results is None + assert node.evaluation_handler.finalized_complete is False + assert node.evaluation_handler.written_results is None diff --git a/autonomy_benchmarks/tests/utils/__init__.py b/autonomy_evaluation/tests/utils/__init__.py similarity index 57% rename from autonomy_benchmarks/tests/utils/__init__.py rename to autonomy_evaluation/tests/utils/__init__.py index 75af748..2f60957 100644 --- a/autonomy_benchmarks/tests/utils/__init__.py +++ b/autonomy_evaluation/tests/utils/__init__.py @@ -1,4 +1,4 @@ # Copyright Thinking Cars GmbH # SPDX-License-Identifier: Apache-2.0 -"""Utility test modules for autonomy_benchmarks.""" +"""Utility test modules for autonomy_evaluation.""" diff --git a/autonomy_benchmarks/tests/utils/test_ObjectDetectionUtils.py b/autonomy_evaluation/tests/utils/test_ObjectDetectionUtils.py similarity index 99% rename from autonomy_benchmarks/tests/utils/test_ObjectDetectionUtils.py rename to autonomy_evaluation/tests/utils/test_ObjectDetectionUtils.py index be2fcdd..aeb3b64 100644 --- a/autonomy_benchmarks/tests/utils/test_ObjectDetectionUtils.py +++ b/autonomy_evaluation/tests/utils/test_ObjectDetectionUtils.py @@ -7,7 +7,7 @@ import numpy as np import pytest -from autonomy_benchmarks.utils.ObjectDetectionUtils import ObjectDetectionUtils +from autonomy_evaluation.utils.ObjectDetectionUtils import ObjectDetectionUtils def _box2d(x1, y1, x2, y2): diff --git a/deployment/compose/docker-compose.yml b/deployment/compose/docker-compose.yml index 70c08a2..e4bf1b7 100644 --- a/deployment/compose/docker-compose.yml +++ b/deployment/compose/docker-compose.yml @@ -1,16 +1,16 @@ services: - autonomy-benchmarks: - image: ghcr.io/thinking-cars/autonomy_benchmarks:v1.0.0 + autonomy-evaluation: + image: ghcr.io/thinking-cars/autonomy_evaluation:v1.0.0 environment: # --- name ------ NAMESPACE: / - NAME: autonomy_benchmarks + NAME: autonomy_evaluation # --- other ----- REQUEST_SAMPLES: ~/request_samples LOG_LEVEL: ${LOG_LEVEL:-info} USE_SIM_TIME: ${USE_SIM_TIME:-true} - BENCHMARK: nuscenes_lidar_object_detection + EVALUATION: nuscenes_lidar_object_detection VISUALIZE: false MANUAL_PLAYBACK: false SAMPLES_PER_REQUEST: 1 @@ -21,12 +21,12 @@ services: - /bin/bash - -ic - | - ros2 launch autonomy_benchmarks autonomy_benchmarks.launch.py \ + ros2 launch autonomy_evaluation autonomy_evaluation.launch.py \ namespace:=$${NAMESPACE} \ name:=$${NAME} \ log_level:=$${LOG_LEVEL} \ use_sim_time:=$${USE_SIM_TIME} \ - benchmark:=$${BENCHMARK} \ + evaluation:=$${EVALUATION} \ visualize:=$${VISUALIZE} \ manual_playback:=$${MANUAL_PLAYBACK} \ samples_per_request:=$${SAMPLES_PER_REQUEST} \ diff --git a/deployment/helm/Chart.yaml b/deployment/helm/Chart.yaml index db887d2..d7ee7ae 100644 --- a/deployment/helm/Chart.yaml +++ b/deployment/helm/Chart.yaml @@ -1,8 +1,8 @@ apiVersion: v2 -name: autonomy-benchmarks +name: autonomy-evaluation version: 1.0.0 appVersion: 1.0.0 -description: Benchmarking suite for automated driving tasks +description: Metrics-based evaluation of automated driving tasks, generating the evidence for benchmarking automated driving deployments dependencies: - repository: oci://ghcr.io/openads-project/openads-helm name: openadservice diff --git a/deployment/helm/values.yaml b/deployment/helm/values.yaml index f917b77..a5e6d8f 100644 --- a/deployment/helm/values.yaml +++ b/deployment/helm/values.yaml @@ -1,15 +1,15 @@ openadservice: - name: autonomy-benchmarks - image: ghcr.io/thinking-cars/autonomy_benchmarks:v1.0.0 + name: autonomy-evaluation + image: ghcr.io/thinking-cars/autonomy_evaluation:v1.0.0 env: # --- name ------ NAMESPACE: / - NAME: autonomy_benchmarks + NAME: autonomy_evaluation # --- other ----- REQUEST_SAMPLES: ~/request_samples LOG_LEVEL: info USE_SIM_TIME: true - BENCHMARK: nuscenes_lidar_object_detection + EVALUATION: nuscenes_lidar_object_detection VISUALIZE: false MANUAL_PLAYBACK: false SAMPLES_PER_REQUEST: 1 @@ -20,12 +20,12 @@ openadservice: - /bin/bash - -ic - | - ros2 launch autonomy_benchmarks autonomy_benchmarks.launch.py \ + ros2 launch autonomy_evaluation autonomy_evaluation.launch.py \ namespace:=${NAMESPACE} \ name:=${NAME} \ log_level:=${LOG_LEVEL} \ use_sim_time:=${USE_SIM_TIME} \ - benchmark:=${BENCHMARK} \ + evaluation:=${EVALUATION} \ visualize:=${VISUALIZE} \ manual_playback:=${MANUAL_PLAYBACK} \ samples_per_request:=${SAMPLES_PER_REQUEST} \ diff --git a/docker-compose.dev.yml b/docker-compose.dev.yml index 63075d6..2f54860 100644 --- a/docker-compose.dev.yml +++ b/docker-compose.dev.yml @@ -1,12 +1,12 @@ -name: ${USER}-autonomy-benchmarks +name: ${USER}-autonomy-evaluation services: - autonomy_benchmarks: + autonomy_evaluation: extends: file: docker-compose.yml - service: autonomy_benchmarks - image: ghcr.io/thinking-cars/autonomy_benchmarks:latest-dev + service: autonomy_evaluation + image: ghcr.io/thinking-cars/autonomy_evaluation:latest-dev command: sleep infinity volumes: - .:/docker-ros/ws/src/target diff --git a/docker-compose.yml b/docker-compose.yml index 38a5097..dc3392c 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -1,15 +1,15 @@ -name: ${USER}-autonomy-benchmarks +name: ${USER}-autonomy-evaluation services: - autonomy_benchmarks: - image: ghcr.io/thinking-cars/autonomy_benchmarks:latest + autonomy_evaluation: + image: ghcr.io/thinking-cars/autonomy_evaluation:latest command: - /bin/bash - -ic - | - ros2 launch autonomy_benchmarks autonomy_benchmarks.launch.py \ - benchmark:=nuscenes_lidar_object_detection \ + ros2 launch autonomy_evaluation autonomy_evaluation.launch.py \ + evaluation:=nuscenes_lidar_object_detection \ prediction:=/object_list/prediction \ label:=/object_list/lidar_01 \ request_samples:=/datasets/request_samples \ @@ -49,7 +49,7 @@ services: - ./config/params_nuscenes.yml:/params_nuscenes.yml - $DATASET_DIR:/datasets depends_on: - - autonomy_benchmarks + - autonomy_evaluation autoware_lidar_centerpoint: profiles: diff --git a/docs/IMPLEMENTATION.md b/docs/IMPLEMENTATION.md index 741c484..844e002 100644 --- a/docs/IMPLEMENTATION.md +++ b/docs/IMPLEMENTATION.md @@ -1,6 +1,6 @@ # Implementation Details -This repository supports the following benchmarks for object detection in automated driving systems: +This repository supports the following evaluations of object detection in automated driving systems: - [nuScenes Challenge](#nuscenes-challenge): 3D lidar object detection @@ -16,7 +16,7 @@ Supported Datasets: - [nuScenes Dataset](https://github.com/thinking-cars/autonomy_datasets/blob/main/docs/IMPLEMENTATION.md#nuscenes-dataset) -> This benchmark uses [nuScenes Dataset](https://github.com/thinking-cars/autonomy_datasets/blob/main/docs/IMPLEMENTATION.md#nuscenes-dataset) via the [autonomy_datasets](https://github.com/thinking-cars/autonomy_datasets) ROS package. +> This evaluation uses [nuScenes Dataset](https://github.com/thinking-cars/autonomy_datasets/blob/main/docs/IMPLEMENTATION.md#nuscenes-dataset) via the [autonomy_datasets](https://github.com/thinking-cars/autonomy_datasets) ROS package. [![3D Object Detection Challenge](https://img.shields.io/badge/origin-3D_Object_Detection_Challenge-green)](https://www.nuscenes.org/object-detection) ![2019](https://img.shields.io/badge/published-2019-green) @@ -99,14 +99,14 @@ Metrics are computed based on the following assumptions: -### Adding more Benchmarks +### Adding more Evaluations -To contribute a new benchmark for a dataset or evaluation protocol: +To contribute a new evaluation for a dataset or evaluation protocol: -1. Create a new benchmark class in [autonomy_benchmarks/benchmarks/](../autonomy_benchmarks/autonomy_benchmarks/benchmarks/) that inherits from `AutonomyBenchmark`. +1. Create a new evaluation class in [autonomy_evaluation/evaluations/](../autonomy_evaluation/autonomy_evaluation/evaluations/) that inherits from `Evaluation`. 2. Implement the three abstract methods: `required_inputs()`, `compute_sample_metrics()`, and `compute_aggregated_metrics()`. -3. Configure the benchmark in `__init__` (thresholds, per-class ranges, and metric rules as instance attributes); keep static lookup tables (e.g. category-to-class mappings) as module-level `_CONSTANT_NAME` constants. -4. Register the benchmark in the node's handler dispatch in [autonomy_benchmarks.py](../autonomy_benchmarks/autonomy_benchmarks/autonomy_benchmarks.py) so it can be selected via the `benchmark:=` launch argument. -5. Add comprehensive tests in [tests/benchmarks/](../autonomy_benchmarks/tests/benchmarks/) following existing test patterns. -6. Update documentation with benchmark details, metrics table, and dataset requirements. -7. Create a [Pull Request](https://github.com/thinking-cars/autonomy_benchmarks-internal/pulls) on GitHub and wait for maintainer feedback. +3. Configure the evaluation in `__init__` (thresholds, per-class ranges, and metric rules as instance attributes); keep static lookup tables (e.g. category-to-class mappings) as module-level `_CONSTANT_NAME` constants. +4. Register the evaluation in the node's handler dispatch in [autonomy_evaluation.py](../autonomy_evaluation/autonomy_evaluation/autonomy_evaluation.py) so it can be selected via the `evaluation:=` launch argument. +5. Add comprehensive tests in [tests/evaluations/](../autonomy_evaluation/tests/evaluations/) following existing test patterns. +6. Update documentation with evaluation details, metrics table, and dataset requirements. +7. Create a [Pull Request](https://github.com/thinking-cars/autonomy_evaluation/pulls) on GitHub and wait for maintainer feedback. From 6eebac2dc1fb5f269642a9676ef86a7dd46d9e91 Mon Sep 17 00:00:00 2001 From: Raphael van Kempen Date: Tue, 29 Sep 2026 12:28:15 +0000 Subject: [PATCH 02/17] make node more generic --- README.md | 7 +- autonomy_evaluation/README.md | 56 ++-- .../autonomy_evaluation.py | 297 +++++++++++++----- .../evaluations/Evaluation.py | 131 ++++++-- .../evaluations/__init__.py | 3 +- .../NuscenesLidarObjectDetection.py | 36 ++- .../evaluations/registry.py | 62 ++++ .../launch/autonomy_evaluation.launch.py | 135 ++++---- .../test_NuscenesLidarObjectDetection.py | 21 ++ .../tests/evaluations/test_Evaluation.py | 110 ++++++- .../tests/evaluations/test_registry.py | 51 +++ .../tests/test_autonomy_evaluation.py | 195 ++++++++++-- autonomy_evaluation/tests/test_launch.py | 54 ++++ deployment/compose/docker-compose.yml | 6 +- deployment/helm/values.yaml | 6 +- docs/IMPLEMENTATION.md | 40 ++- 16 files changed, 960 insertions(+), 250 deletions(-) create mode 100644 autonomy_evaluation/autonomy_evaluation/evaluations/registry.py create mode 100644 autonomy_evaluation/tests/evaluations/test_registry.py create mode 100644 autonomy_evaluation/tests/test_launch.py diff --git a/README.md b/README.md index 5bba453..a2bd9e4 100644 --- a/README.md +++ b/README.md @@ -14,9 +14,10 @@ > This repository is part of the **Autonomy.Benchmarks** suite of the **Autonomy.Hub Ecosystem** -Within the Autonomy.Benchmarks suite, **Autonomy.Evaluation** generates the metrics-based evidence for benchmarking automated driving deployments. It evaluates the output of a system under test on the samples replayed by [Autonomy.Datasets](https://github.com/thinking-cars/autonomy_datasets) and reports the resulting metrics per scene and over all evaluated samples: +Within the Autonomy.Benchmarks suite, **Autonomy.Evaluation** generates the metrics-based evidence for benchmarking automated driving deployments. It evaluates arbitrary ROS systems under test, either from their own topics alone, e.g. a closed-loop planner by its time to collision, or against ground truth, e.g. a perception algorithm against the labels of a dataset replayed by [Autonomy.Datasets](https://github.com/thinking-cars/autonomy_datasets), and reports the resulting metrics per scene and over all evaluated samples: -- 🔄 **Unified ROS 2 Interface**: Evaluate systems under test on multiple datasets using the benefits of the ROS 2 ecosystem +- 🔄 **Unified ROS 2 Interface**: Evaluate any ROS system under test, on datasets, in simulation or live, using the benefits of the ROS 2 ecosystem +- 🧩 **Pluggable Evaluations**: Select an evaluation by name, or bring your own from another package, reading any topics as inputs and, where needed, ground truth - 📊 **Established Metrics**: Use the provided evaluations, which follow the protocols of established challenges, with [Autonomy.Datasets](https://github.com/thinking-cars/autonomy_datasets) across different automated driving tasks - ⚡ **Efficient Data Pipeline**: Works seamlessly with preprocessed Rosbag files from [Autonomy.Datasets](https://github.com/thinking-cars/autonomy_datasets) for fast execution during development - 🐳 **Dockerized Environment**: Reproducible setup with all dependencies included @@ -63,7 +64,7 @@ Configure the evaluation and dataset via ROS launch arguments in [docker-compose command: ros2 launch autonomy_evaluation autonomy_evaluation.launch.py evaluation:=nuscenes_lidar_object_detection prediction:=$your_prediction_topic label:=$your_label_topic request_samples:=/datasets/request_samples visualize:=true ``` -The evaluation node requests the samples it evaluates from the dataset node via its `request_samples` service, which publishes them and responds once they have been published. The dataset therefore publishes the next sample only once the system under test has processed the current one. As soon as all samples have been published, the node aggregates its metrics per scene of the dataset and over all evaluated samples. See the [node documentation](autonomy_evaluation/README.md#autonomy_evaluation) for the sample request settings and the results. +The evaluation node requests the samples it evaluates from the dataset node via its `request_samples` service, which publishes them and responds once they have been published. The dataset therefore publishes the next sample only once the system under test has processed the current one. As soon as all samples have been published, the node aggregates its metrics per scene of the dataset and over all evaluated samples. To evaluate samples published by others instead, e.g. by a closed-loop simulation, set `sample_source:=external`. See the [node documentation](autonomy_evaluation/README.md#autonomy_evaluation) for the topics of the evaluations, the sample settings and the results. ## 💻 Development diff --git a/autonomy_evaluation/README.md b/autonomy_evaluation/README.md index 27dd054..e014c52 100644 --- a/autonomy_evaluation/README.md +++ b/autonomy_evaluation/README.md @@ -2,31 +2,23 @@ Metrics-based evaluation of automated driving tasks, generating the evidence for benchmarking automated driving deployments -`autonomy_evaluation` is the part of the **Autonomy.Benchmarks** suite that turns the output of a system under test into -metrics. It evaluates the samples that [autonomy_datasets](https://github.com/thinking-cars/autonomy_datasets) replays -against the labels of the dataset, and reports the metrics per scene and over all evaluated samples, as the evidence -an automated driving deployment is benchmarked on. +`autonomy_evaluation` is the part of the **Autonomy.Benchmarks** suite that turns the output of a system under test into metrics. It evaluates arbitrary ROS systems under test, either from their own topics alone, e.g. the time to collision of a closed-loop planner, or against ground truth, e.g. the predictions of a perception algorithm against the labels of a dataset replayed by [autonomy_datasets](https://github.com/thinking-cars/autonomy_datasets). The metrics are reported per scene and over all evaluated samples, as the evidence an automated driving deployment is benchmarked on. ## Nodes ### `autonomy_evaluation` -The node requests the samples it evaluates from the dataset, using the `request_samples` service -of [autonomy_datasets](https://github.com/thinking-cars/autonomy_datasets), which publishes them -and responds once they have been published. By default one sample is requested at a time, so the -dataset only publishes the next sample once the system under test has delivered its output for -the current one and the node has evaluated it. Increase `samples_per_request` to publish -samples in batches, set it to `0` to publish the whole dataset with a single request, or list the -IDs of individual samples in `sample_ids` to evaluate only those. - -Once the dataset reports that all requested samples have been published, the per-sample metrics -are aggregated and reported on three levels: over all evaluated samples (`aggregated_metrics`), for -the samples of each scene the dataset published them from (`scene_results`, matched with the -samples via the `published_scene_ids` of the responses), and for every single sample -(`sample_results`). The dataset metrics are logged, and the results of all three levels are -written to a JSON file if `results_path` is set. Samples that are not evaluated within -`evaluation_timeout` seconds of being published, e.g. because the system under test skipped them, -are left out. +The node runs the evaluation selected by `evaluation`, either one of this package by its name or one implemented in another package as `:` (see [Adding more Evaluations](../docs/IMPLEMENTATION.md#adding-more-evaluations)). Each evaluation declares the topics it reads: its _inputs_ from the system under test, and, if it compares them with a reference, its _ground truth_. The node subscribes to all of them, each on its node-relative name, which the launch file remaps onto the topic given by the launch argument of the same name, e.g. `prediction:=/object_list/prediction`. A topic without such an argument is subscribed in the private namespace of the node. The node logs which topic it evaluates as input and which as ground truth. + +| Evaluation | Inputs | Ground truth | +| --- | --- | --- | +| `nuscenes_lidar_object_detection` | `prediction` (`perception_msgs/ObjectList`) | `label` (`perception_msgs/ObjectList`), `label_meta_info` (`autonomy_datasets_msgs/ObjectListMetaInfo`, subscribed next to `label` on `