Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
55 changes: 55 additions & 0 deletions ami/main/management/commands/export_embeddings.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,55 @@
"""
Write one algorithm's detection feature vectors for a project to a directory.

The vectors go to ``vectors.npy`` and each row's detection is named in ``index.csv`` by
its natural key (capture path, timestamp, station, box, detector), so
``import_embeddings`` can put them back on a database where the ids differ.
"""

import pathlib

from django.core.management.base import BaseCommand, CommandError

from ami.main.models import Project
from ami.main.models_future.embedding_transfer import DEFAULT_VECTOR_KEY, export_embeddings, resolve_algorithm
from ami.ml.models import Algorithm


class Command(BaseCommand):
help = "Export detection feature vectors from one algorithm as vectors.npy plus an index of detection keys."

def add_arguments(self, parser):
parser.add_argument("--project", type=int, required=True, help="Project ID to export from.")
parser.add_argument(
"--algorithm", required=True, help="Key, name or ID of the algorithm whose vectors to export."
)
parser.add_argument("--output", type=pathlib.Path, required=True, help="Directory to write into.")
parser.add_argument(
"--vector-key",
default=DEFAULT_VECTOR_KEY,
help=f"Which of the algorithm's vectors to export when it stores several (default {DEFAULT_VECTOR_KEY}).",
)
parser.add_argument(
"--dtype", default="float32", choices=["float16", "float32"], help="Storage precision (default float32)."
)

def handle(self, *args, **options):
try:
project = Project.objects.get(pk=options["project"])
except Project.DoesNotExist as err:
raise CommandError(f"Project {options['project']} does not exist") from err
try:
algorithm = resolve_algorithm(options["algorithm"])
except Algorithm.DoesNotExist as err:
raise CommandError(str(err)) from err

try:
manifest = export_embeddings(
project, algorithm, options["output"], vector_key=options["vector_key"], dtype=options["dtype"]
)
except ValueError as err:
raise CommandError(str(err)) from err
self.stdout.write(
f"Wrote {manifest.count} vectors of {manifest.dimensions} dimensions ({manifest.dtype}) from "
f"{manifest.source_store} for algorithm {algorithm.key!r} in project #{project.pk} to {options['output']}"
)
42 changes: 42 additions & 0 deletions ami/main/management/commands/export_validated_occurrences.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,42 @@
"""
Write the occurrences people confirmed or identified in one project to a portable bundle.

The bundle names detections, taxa and users by natural keys only (capture path, timestamp,
station, bounding box, detector; GBIF key or name and rank; email), so it can be replayed
onto a database where every id is different with ``import_validated_occurrences``.
"""

import json
import pathlib

from django.core.management.base import BaseCommand, CommandError

from ami.main.models import Project
from ami.main.models_future.validated_occurrences import build_bundle


class Command(BaseCommand):
help = "Export a project's confirmed occurrences and identifications as a bundle keyed by natural keys."

def add_arguments(self, parser):
parser.add_argument("--project", type=int, required=True, help="Project ID to export from.")
parser.add_argument("--output", type=pathlib.Path, required=True, help="Path of the JSON bundle to write.")

def handle(self, *args, **options):
try:
project = Project.objects.get(pk=options["project"])
except Project.DoesNotExist as err:
raise CommandError(f"Project {options['project']} does not exist") from err

bundle = build_bundle(project)
output: pathlib.Path = options["output"]
output.parent.mkdir(parents=True, exist_ok=True)
output.write_text(json.dumps(bundle.as_dict(), indent=1, ensure_ascii=False))

confirmed = sum(1 for record in bundle.occurrences if record.confirmation)
identifications = sum(len(record.identifications) for record in bundle.occurrences)
detections = sum(len(record.detections) for record in bundle.occurrences)
self.stdout.write(
f"Wrote {len(bundle.occurrences)} occurrences ({confirmed} confirmed, {identifications} identifications, "
f"{detections} detections) from project #{project.pk} to {output}"
)
80 changes: 80 additions & 0 deletions ami/main/management/commands/import_embeddings.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,80 @@
"""
Put detection feature vectors from an ``export_embeddings`` directory onto a project.

Each row's detection is found again by its natural key (capture path and box, with an
overlap fallback) and the vector is stored under the same algorithm. By default this is a
dry run that reports how many detections were found; ``--execute`` writes the rows. A
detection that already has a vector from that algorithm is skipped unless ``--replace``.
"""

import pathlib

from django.core.management.base import BaseCommand, CommandError

from ami.main.models import Project
from ami.main.models_future.detection_matching import DEFAULT_IOU_THRESHOLD
from ami.main.models_future.embedding_transfer import import_embeddings, resolve_algorithm
from ami.ml.models import Algorithm


class Command(BaseCommand):
help = "Import detection feature vectors from an export directory onto a project (dry run by default)."

def add_arguments(self, parser):
parser.add_argument("--project", type=int, required=True, help="Project ID to import into.")
parser.add_argument(
"--input", type=pathlib.Path, required=True, help="Directory written by export_embeddings."
)
parser.add_argument("--execute", action="store_true", default=False, help="Write the rows.")
parser.add_argument(
"--algorithm",
default=None,
help="Key, name or ID of the algorithm to store the vectors under (default: the one in the manifest).",
)
parser.add_argument("--iou-threshold", type=float, default=DEFAULT_IOU_THRESHOLD)
parser.add_argument(
"--replace", action="store_true", default=False, help="Overwrite vectors the detections already have."
)

def handle(self, *args, **options):
try:
project = Project.objects.get(pk=options["project"])
except Project.DoesNotExist as err:
raise CommandError(f"Project {options['project']} does not exist") from err
algorithm = None
if options["algorithm"]:
try:
algorithm = resolve_algorithm(options["algorithm"])
except Algorithm.DoesNotExist as err:
raise CommandError(str(err)) from err

try:
report = import_embeddings(
project,
options["input"],
execute=options["execute"],
iou_threshold=options["iou_threshold"],
replace=options["replace"],
algorithm=algorithm,
)
except (OSError, ValueError, RuntimeError, Algorithm.DoesNotExist) as err:
raise CommandError(str(err)) from err

summary = report.summary()
mode = "Applied" if options["execute"] else "Dry run"
self.stdout.write(
f"{mode} on project #{project.pk}: {summary['vectors_total']} vectors {summary['sources']}, "
f"detections {summary['detections']}"
)
if options["execute"]:
self.stdout.write(
f" written: {summary['written']} skipped (already stored): {summary['skipped_existing']}"
f" replaced: {summary['replaced']}"
f" classifier features without a classification on the target: {summary['skipped_no_classification']}"
)
if summary["skipped_no_field"]:
self.stdout.write(
self.style.WARNING(f" rows this branch has no table or column for: {summary['skipped_no_field']}")
)
else:
self.stdout.write("Nothing was changed. Re-run with --execute to write the vectors.")
95 changes: 95 additions & 0 deletions ami/main/management/commands/import_validated_occurrences.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,95 @@
"""
Replay a bundle of confirmed occurrences and identifications onto a project.

By default this is a dry run: it finds each detection again and reports how many matched
exactly, how many only by overlap, and how many are missing, without changing a row. With
``--execute`` it regroups the matched detections with the same operations the review
interface uses, records each confirmation under its original reviewer and time, and
re-creates the identifications. Occurrences with a missing detection are reported as
partial and left unconfirmed. Running it twice changes nothing the second time.
"""

import json
import pathlib

from django.core.management.base import BaseCommand, CommandError

from ami.main.models import Project
from ami.main.models_future.detection_matching import DEFAULT_IOU_THRESHOLD
from ami.main.models_future.validated_occurrences import Bundle, ImportOptions, import_bundle


class Command(BaseCommand):
help = "Replay confirmed occurrences and identifications from a bundle onto a project (dry run by default)."

def add_arguments(self, parser):
parser.add_argument("--project", type=int, required=True, help="Project ID to import into.")
parser.add_argument("--input", type=pathlib.Path, required=True, help="Path of the JSON bundle to read.")
parser.add_argument(
"--execute",
action="store_true",
default=False,
help="Write the changes. Without this flag the command only reports what it would do.",
)
parser.add_argument(
"--iou-threshold",
type=float,
default=DEFAULT_IOU_THRESHOLD,
help=(
"Lowest intersection over union accepted when no detection has the exact box "
f"(default {DEFAULT_IOU_THRESHOLD})."
),
)
parser.add_argument(
"--create-missing-detections",
action="store_true",
default=False,
help="Recreate a box whose capture exists but that no detector found, as a reviewer-drawn detection.",
)
parser.add_argument(
"--report", type=pathlib.Path, default=None, help="Write the per-occurrence match report as JSON here."
)

def handle(self, *args, **options):
try:
project = Project.objects.get(pk=options["project"])
except Project.DoesNotExist as err:
raise CommandError(f"Project {options['project']} does not exist") from err
try:
bundle = Bundle.from_dict(json.loads(options["input"].read_text()))
except (OSError, ValueError, KeyError) as err:
raise CommandError(f"Could not read bundle {options['input']}: {err}") from err

import_options = ImportOptions(
execute=options["execute"],
iou_threshold=options["iou_threshold"],
create_missing_detections=options["create_missing_detections"],
)
report = import_bundle(project, bundle, import_options)
summary = report.summary()

mode = "Applied" if import_options.execute else "Dry run"
self.stdout.write(f"{mode} on project #{project.pk} ({project.name}), bundle from {bundle.project_name!r}:")
self.stdout.write(f" occurrences: {summary['occurrences_total']} {dict(summary['occurrences'])}")
self.stdout.write(f" detections: {summary['detections_total']} {dict(summary['detections'])}")
self.stdout.write(
f" confirmed: {summary['confirmed']} identifications applied: {summary['identifications_applied']}"
f" skipped: {summary['identifications_skipped']} detections created: {summary['detections_created']}"
)
if summary["determination_mismatches"]:
self.stdout.write(f" determinations differing from the bundle: {summary['determination_mismatches']}")
if summary["missing_users"]:
self.stdout.write(self.style.WARNING(f" users not found: {', '.join(summary['missing_users'])}"))
if summary["missing_taxa"]:
names = ", ".join(f"{t['name']} ({t['rank']})" for t in summary["missing_taxa"])
self.stdout.write(self.style.WARNING(f" taxa not found: {names}"))
for outcome in report.outcomes:
if outcome.outcome == "error":
self.stdout.write(self.style.ERROR(f" record {outcome.ref}: {outcome.error}"))

if options["report"]:
options["report"].parent.mkdir(parents=True, exist_ok=True)
options["report"].write_text(json.dumps(report.as_dict(), indent=1, default=str))
self.stdout.write(f" report written to {options['report']}")
if not import_options.execute:
self.stdout.write("Nothing was changed. Re-run with --execute to apply.")
Loading