#!/usr/bin/env python3
"""Build a license-aware, human-review queue for Santiago building candidates.

This module deliberately separates *candidate discovery* from OSM editing.  Candidate
geometry may come from only two inputs:

* the AI screening layer derived from the licensed OpenAerialMap (OAM) image; or
* Microsoft Global Building Footprints (CDLA-Permissive-2.0).

IGAC and Google Open Buildings are excluded from queue processing entirely so that
the shareable output has no influence from either dataset.  Existing OSM features
are used only for strict de-duplication/conflation screening.

Spatial comparisons happen in the local UTM zone.  Two footprints match when any of
these deliberately permissive screening rules is true:

* intersection-over-union is at least 0.10;
* intersection covers at least 35% of the smaller footprint; or
* their edges are no more than 2 m apart AND centroids no more than 8 m apart.

Those permissive thresholds compare AI and Microsoft only.  OSM matching requires
actual overlap and assigns each OSM source feature to at most one best candidate.
Candidates also require at least 95% coverage by the valid OAM footprint.  Every
feature remains ``upload_ready=false`` and must be traced/corrected by a human from
the OAM image.
"""

from __future__ import annotations

import argparse
import hashlib
import json
import math
from collections import Counter, defaultdict
from dataclasses import dataclass
from pathlib import Path
from typing import Any, Iterable, Sequence

from pyproj import CRS, Transformer
from shapely import make_valid, normalize, set_precision
from shapely.geometry import mapping, shape
from shapely.geometry.base import BaseGeometry
from shapely.ops import transform, unary_union
from shapely.strtree import STRtree
from shapely.wkb import dumps as wkb_dumps


DEFAULT_AI = Path("santiago_results/ai_buildings_screening.geojson")
DEFAULT_SOURCES = Path("santiago_results/source_comparison.geojson")
DEFAULT_FOOTPRINT = Path("santiago_results/oam_valid_footprint.geojson")
DEFAULT_QUEUE = Path("santiago_results/review_queue.geojson")
DEFAULT_PILOT = Path("santiago_results/pilot_batch.geojson")
DEFAULT_SUMMARY = Path("santiago_results/review_queue_summary.json")

MIN_IOU = 0.10
MIN_SMALLER_OVERLAP = 0.35
MAX_EDGE_DISTANCE_M = 2.0
MAX_CENTROID_DISTANCE_M = 8.0
OAM_EDGE_WARNING_M = 8.0
MIN_OAM_COVERAGE_FRACTION = 0.95
OSM_MIN_INTERSECTION_M2 = 1.0
OSM_MIN_IOU = 0.20
OSM_MIN_CANDIDATE_COVERAGE = 0.50
OSM_MIN_SOURCE_COVERAGE = 0.50
# Full exclusion is deliberately stricter than merely finding an overlap.  A
# lower-coverage match stays in the queue for manual conflation.
OSM_MIN_WHOLE_CANDIDATE_COVERAGE = 0.85


@dataclass(frozen=True)
class SourceFeature:
    """A valid source feature in WGS84 and local metric coordinates."""

    geometry_wgs84: BaseGeometry
    geometry_metric: BaseGeometry
    properties: dict[str, Any]
    stable_key: str


def _read_geojson(path: Path) -> dict[str, Any]:
    with path.open(encoding="utf-8") as handle:
        data = json.load(handle)
    if data.get("type") != "FeatureCollection":
        raise ValueError(f"Expected a GeoJSON FeatureCollection: {path}")
    return data


def _clean_geometry(raw: dict[str, Any] | None) -> BaseGeometry | None:
    if not raw:
        return None
    geom = make_valid(shape(raw))
    if geom.is_empty:
        return None
    if geom.geom_type not in {"Polygon", "MultiPolygon"}:
        polygonal = [part for part in getattr(geom, "geoms", []) if part.geom_type in {"Polygon", "MultiPolygon"}]
        if not polygonal:
            return None
        geom = unary_union(polygonal)
    return geom


def _utm_crs_for(features: Iterable[dict[str, Any]]) -> CRS:
    """Choose a local UTM CRS from the combined WGS84 feature centroid."""

    centers: list[tuple[float, float]] = []
    for feature in features:
        geom = _clean_geometry(feature.get("geometry"))
        if geom is not None:
            point = geom.representative_point()
            centers.append((point.x, point.y))
    if not centers:
        raise ValueError("Cannot select a metric CRS from empty inputs")
    longitude = sum(x for x, _ in centers) / len(centers)
    latitude = sum(y for _, y in centers) / len(centers)
    zone = int(math.floor((longitude + 180.0) / 6.0) + 1)
    epsg = (32600 if latitude >= 0 else 32700) + zone
    return CRS.from_epsg(epsg)


def _stable_geometry_key(prefix: str, geom: BaseGeometry) -> str:
    """Hash normalized, rounded WGS84 geometry so IDs survive input reordering."""

    canonical = normalize(set_precision(geom, grid_size=1e-7))
    payload = prefix.encode("utf-8") + b"|" + wkb_dumps(
        canonical,
        hex=False,
        output_dimension=2,
        big_endian=True,
    )
    return hashlib.sha1(payload).hexdigest()


def _source_features(
    raw_features: Sequence[dict[str, Any]],
    to_metric: Transformer,
    source_prefix: str,
) -> list[SourceFeature]:
    converted: list[SourceFeature] = []
    for feature in raw_features:
        geom_wgs84 = _clean_geometry(feature.get("geometry"))
        if geom_wgs84 is None:
            continue
        geom_metric = transform(to_metric.transform, geom_wgs84)
        if geom_metric.is_empty or geom_metric.area <= 0:
            continue
        stable_key = _stable_geometry_key(source_prefix, geom_wgs84)
        converted.append(
            SourceFeature(
                geometry_wgs84=geom_wgs84,
                geometry_metric=geom_metric,
                properties=dict(feature.get("properties") or {}),
                stable_key=stable_key,
            )
        )
    converted.sort(key=lambda item: item.stable_key)
    return converted


def _match_metrics(left: BaseGeometry, right: BaseGeometry) -> dict[str, float | bool]:
    intersection_area = left.intersection(right).area
    union_area = left.union(right).area
    smaller_area = min(left.area, right.area)
    iou = intersection_area / union_area if union_area else 0.0
    smaller_overlap = intersection_area / smaller_area if smaller_area else 0.0
    edge_distance = left.distance(right)
    centroid_distance = left.centroid.distance(right.centroid)
    matched = (
        iou >= MIN_IOU
        or smaller_overlap >= MIN_SMALLER_OVERLAP
        or (
            edge_distance <= MAX_EDGE_DISTANCE_M
            and centroid_distance <= MAX_CENTROID_DISTANCE_M
        )
    )
    # A continuous ordering score is used only to select the best match when one
    # source footprint plausibly touches multiple candidates.
    quality = (
        (100.0 * iou)
        + (60.0 * smaller_overlap)
        + max(0.0, 12.0 - centroid_distance)
        + max(0.0, 5.0 - edge_distance)
    )
    return {
        "matched": matched,
        "iou": iou,
        "smaller_overlap": smaller_overlap,
        "edge_distance_m": edge_distance,
        "centroid_distance_m": centroid_distance,
        "quality": quality,
    }


def _best_matches(
    anchors: Sequence[SourceFeature],
    others: Sequence[SourceFeature],
) -> tuple[dict[int, list[int]], dict[tuple[int, int], dict[str, float | bool]]]:
    """Assign each ``other`` to its best matching anchor, allowing one-to-many."""

    assignments: dict[int, list[int]] = defaultdict(list)
    metrics_by_pair: dict[tuple[int, int], dict[str, float | bool]] = {}
    if not anchors or not others:
        return assignments, metrics_by_pair

    anchor_geometries = [item.geometry_metric for item in anchors]
    tree = STRtree(anchor_geometries)
    for other_index, other in enumerate(others):
        query_geom = other.geometry_metric.buffer(MAX_EDGE_DISTANCE_M)
        possible = tree.query(query_geom)
        ranked: list[tuple[float, str, int, dict[str, float | bool]]] = []
        for raw_anchor_index in possible:
            anchor_index = int(raw_anchor_index)
            metrics = _match_metrics(
                anchors[anchor_index].geometry_metric,
                other.geometry_metric,
            )
            if bool(metrics["matched"]):
                ranked.append(
                    (
                        -float(metrics["quality"]),
                        anchors[anchor_index].stable_key,
                        anchor_index,
                        metrics,
                    )
                )
        if ranked:
            _, _, anchor_index, metrics = min(ranked)
            assignments[anchor_index].append(other_index)
            metrics_by_pair[(anchor_index, other_index)] = metrics
    for indexes in assignments.values():
        indexes.sort(key=lambda index: others[index].stable_key)
    return assignments, metrics_by_pair


def _osm_overlap_metrics(
    candidate: BaseGeometry,
    osm_feature: BaseGeometry,
) -> dict[str, float | bool]:
    """Apply the stricter, overlap-only rule used for OSM de-duplication."""

    intersection_area = candidate.intersection(osm_feature).area
    union_area = candidate.union(osm_feature).area
    iou = intersection_area / union_area if union_area else 0.0
    candidate_coverage = intersection_area / candidate.area if candidate.area else 0.0
    osm_coverage = intersection_area / osm_feature.area if osm_feature.area else 0.0
    matched = intersection_area >= OSM_MIN_INTERSECTION_M2 and (
        iou >= OSM_MIN_IOU
        or candidate_coverage >= OSM_MIN_CANDIDATE_COVERAGE
        or osm_coverage >= OSM_MIN_SOURCE_COVERAGE
    )
    quality = (100.0 * iou) + (70.0 * candidate_coverage) + (50.0 * osm_coverage)
    return {
        "matched": matched,
        "intersection_area_m2": intersection_area,
        "iou": iou,
        "candidate_coverage": candidate_coverage,
        "osm_coverage": osm_coverage,
        "quality": quality,
    }


def _best_osm_matches(
    candidates: Sequence[SourceFeature],
    osm_features: Sequence[SourceFeature],
) -> tuple[dict[int, list[int]], dict[tuple[int, int], dict[str, float | bool]]]:
    """Assign each OSM source feature to no more than one best candidate."""

    assignments: dict[int, list[int]] = defaultdict(list)
    metrics_by_pair: dict[tuple[int, int], dict[str, float | bool]] = {}
    if not candidates or not osm_features:
        return assignments, metrics_by_pair

    tree = STRtree([item.geometry_metric for item in candidates])
    for osm_index, osm_feature in enumerate(osm_features):
        ranked: list[tuple[float, str, int, dict[str, float | bool]]] = []
        for raw_candidate_index in tree.query(osm_feature.geometry_metric):
            candidate_index = int(raw_candidate_index)
            metrics = _osm_overlap_metrics(
                candidates[candidate_index].geometry_metric,
                osm_feature.geometry_metric,
            )
            if bool(metrics["matched"]):
                ranked.append(
                    (
                        -float(metrics["quality"]),
                        candidates[candidate_index].stable_key,
                        candidate_index,
                        metrics,
                    )
                )
        if ranked:
            _, _, candidate_index, metrics = min(ranked)
            assignments[candidate_index].append(osm_index)
            metrics_by_pair[(candidate_index, osm_index)] = metrics
    for indexes in assignments.values():
        indexes.sort(key=lambda index: osm_features[index].stable_key)
    return assignments, metrics_by_pair


def _osm_source_id(feature: SourceFeature) -> str:
    properties = feature.properties
    for osm_type, property_name in (
        ("way", "osm_way_id"),
        ("relation", "osm_relation_id"),
        ("node", "osm_node_id"),
    ):
        value = properties.get(property_name)
        if value not in (None, ""):
            return f"{osm_type}/{value}"
    return f"geometry/{feature.stable_key[:16]}"


def _deduplicate_osm(features: Sequence[SourceFeature]) -> list[SourceFeature]:
    """Keep one deterministic geometry for each OSM source object."""

    unique: dict[str, SourceFeature] = {}
    for feature in features:
        unique.setdefault(_osm_source_id(feature), feature)
    return [unique[source_id] for source_id in sorted(unique)]


def _category(source_ai: bool, source_microsoft: bool) -> str:
    if source_ai and source_microsoft:
        return "ai_microsoft_consensus"
    if source_ai:
        return "ai_only"
    return "microsoft_only"


def _priority(
    category: str,
    area_m2: float,
    compactness: float,
    near_oam_edge: bool,
    match_count_microsoft: int,
) -> tuple[int, str, list[str]]:
    base = {
        "ai_microsoft_consensus": 82,
        "ai_only": 55,
        "microsoft_only": 47,
    }[category]
    flags: list[str] = []

    if 12.0 <= area_m2 <= 500.0:
        base += 8
    elif 8.0 <= area_m2 <= 900.0:
        base += 3
    else:
        base -= 8
        flags.append("unusual_area")

    if compactness < 0.18:
        base -= 7
        flags.append("irregular_geometry")
    elif compactness >= 0.55:
        base += 3

    if near_oam_edge:
        base -= 10
        flags.append("near_oam_valid_data_edge")
    if match_count_microsoft > 1:
        base -= 6
        flags.append("possible_merged_roofs")

    score = max(0, min(100, int(round(base))))
    tier = "A" if score >= 85 else "B" if score >= 70 else "C" if score >= 55 else "D"
    return score, tier, flags


def _compactness(geometry: BaseGeometry) -> float:
    perimeter = geometry.length
    if not perimeter:
        return 0.0
    return max(0.0, min(1.0, 4.0 * math.pi * geometry.area / (perimeter * perimeter)))


def _recommended_action(
    category: str,
    fully_preexisting_osm: bool,
    osm_overlap: bool,
    insufficient_oam_coverage: bool,
) -> str:
    if fully_preexisting_osm:
        return "exclude_preexisting_osm; optionally_review_existing_shape"
    if insufficient_oam_coverage:
        return "exclude_insufficient_valid_oam_coverage"
    if osm_overlap:
        return "hold_for_manual_osm_conflation; review_remaining_roof_units_from_oam"
    if category == "ai_microsoft_consensus":
        return "high_priority_human_trace_or_correct_from_oam"
    if category == "microsoft_only":
        return "human_verify_microsoft_candidate_and_trace_or_correct_from_oam"
    return "human_inspect_ai_signal_and_trace_from_oam_if_building"


def _feature_collection(features: Sequence[dict[str, Any]], metadata: dict[str, Any]) -> dict[str, Any]:
    return {
        "type": "FeatureCollection",
        "name": metadata.get("name", "santiago_review_queue"),
        "metadata": metadata,
        "features": list(features),
    }


def _pilot_candidates(
    eligible_features: Sequence[dict[str, Any]],
    pilot_size: int,
) -> list[dict[str, Any]]:
    """Select a deterministic, stratified pilot with a consensus majority."""

    size = min(max(0, pilot_size), len(eligible_features))
    if size == 0:
        return []

    ranked = sorted(
        eligible_features,
        key=lambda feature: (
            -int(feature["properties"]["priority_score"]),
            feature["properties"]["candidate_id"],
        ),
    )
    strata = {
        "consensus": lambda category: category == "ai_microsoft_consensus",
        "ai_only": lambda category: category == "ai_only",
        "microsoft_only": lambda category: category == "microsoft_only",
    }
    # 65/17.5/17.5 percent. Rounding is resolved by the general fill below.
    target_counts = {
        "consensus": int(round(size * 0.65)),
        "ai_only": int(round(size * 0.175)),
        "microsoft_only": int(round(size * 0.175)),
    }

    chosen_ids: set[str] = set()
    chosen: list[dict[str, Any]] = []
    chosen_stratum: dict[str, str] = {}
    for stratum, predicate in strata.items():
        eligible = [
            feature
            for feature in ranked
            if predicate(feature["properties"]["category"])
        ]
        for feature in eligible[: target_counts[stratum]]:
            if len(chosen) >= size:
                break
            candidate_id = feature["properties"]["candidate_id"]
            chosen_ids.add(candidate_id)
            chosen_stratum[candidate_id] = stratum
            chosen.append(feature)

    for feature in ranked:
        if len(chosen) >= size:
            break
        candidate_id = feature["properties"]["candidate_id"]
        if candidate_id in chosen_ids:
            continue
        chosen_ids.add(candidate_id)
        chosen_stratum[candidate_id] = next(
            stratum
            for stratum, predicate in strata.items()
            if predicate(feature["properties"]["category"])
        )
        chosen.append(feature)

    output: list[dict[str, Any]] = []
    for sequence, feature in enumerate(
        sorted(
            chosen,
            key=lambda item: (
                -int(item["properties"]["priority_score"]),
                item["properties"]["candidate_id"],
            ),
        ),
        start=1,
    ):
        copied = {
            "type": "Feature",
            "geometry": feature["geometry"],
            "properties": dict(feature["properties"]),
        }
        candidate_id = copied["properties"]["candidate_id"]
        copied["properties"]["pilot_sequence"] = sequence
        copied["properties"]["pilot_stratum"] = chosen_stratum[candidate_id]
        output.append(copied)
    return output


def build_queue(
    ai_geojson: dict[str, Any],
    source_geojson: dict[str, Any],
    footprint_geojson: dict[str, Any] | None = None,
    pilot_size: int = 200,
) -> tuple[dict[str, Any], dict[str, Any], dict[str, Any]]:
    """Return full queue, stratified pilot, and summary dictionaries."""

    ai_raw = list(ai_geojson.get("features") or [])
    source_raw = list(source_geojson.get("features") or [])
    footprint_raw = list((footprint_geojson or {}).get("features") or [])
    by_source: dict[str, list[dict[str, Any]]] = defaultdict(list)
    for feature in source_raw:
        by_source[str((feature.get("properties") or {}).get("source", ""))].append(feature)

    # Deliberately do not use IGAC or Google even to choose the projection. Changing
    # either excluded source cannot change queue/pilot geometry, scores, or order.
    projection_inputs = [
        *ai_raw,
        *by_source["Microsoft"],
        *by_source["OpenStreetMap"],
        *footprint_raw,
    ]
    metric_crs = _utm_crs_for(projection_inputs)
    to_metric = Transformer.from_crs("EPSG:4326", metric_crs, always_xy=True)

    ai = _source_features(ai_raw, to_metric, "ai_oam")
    microsoft = _source_features(by_source["Microsoft"], to_metric, "microsoft")
    osm_raw_valid = _source_features(by_source["OpenStreetMap"], to_metric, "osm_existing")
    osm = _deduplicate_osm(osm_raw_valid)
    footprint = _source_features(footprint_raw, to_metric, "oam_valid_footprint")

    ai_ms, ai_ms_metrics = _best_matches(ai, microsoft)
    matched_microsoft = {index for indexes in ai_ms.values() for index in indexes}

    # Candidate geometry is now locked to AI or unmatched Microsoft only.
    anchors: list[tuple[str, SourceFeature, list[int]]] = [
        ("ai_oam", feature, ai_ms.get(index, [])) for index, feature in enumerate(ai)
    ]
    anchors.extend(
        ("microsoft", feature, [])
        for index, feature in enumerate(microsoft)
        if index not in matched_microsoft
    )
    anchors.sort(key=lambda item: (item[0], item[1].stable_key))
    anchor_features = [item[1] for item in anchors]

    anchor_osm, anchor_osm_metrics = _best_osm_matches(anchor_features, osm)

    footprint_metric = unary_union([item.geometry_metric for item in footprint]) if footprint else None
    ai_index_by_key = {item.stable_key: index for index, item in enumerate(ai)}
    queue_features: list[dict[str, Any]] = []
    for anchor_index, (origin, anchor, microsoft_indexes) in enumerate(anchors):
        source_ai = origin == "ai_oam"
        source_microsoft = origin == "microsoft" or bool(microsoft_indexes)
        osm_indexes = anchor_osm.get(anchor_index, [])
        osm_overlap = bool(osm_indexes)
        provisional_roof_units = max(1, len(microsoft_indexes))
        potential_mapped_units = min(provisional_roof_units, len(osm_indexes))
        if osm_indexes:
            assigned_osm_union = unary_union(
                [osm[index].geometry_metric for index in osm_indexes]
            )
            osm_union_candidate_coverage = (
                anchor.geometry_metric.intersection(assigned_osm_union).area
                / anchor.geometry_metric.area
            )
        else:
            osm_union_candidate_coverage = 0.0
        plausible_whole_candidate_match = (
            potential_mapped_units == provisional_roof_units
            and osm_union_candidate_coverage >= OSM_MIN_WHOLE_CANDIDATE_COVERAGE
        )
        # For a nominal one-unit component with only a small OSM overlap, do not
        # claim that its provisional unit was mapped. It may be an unrecognized
        # merged cluster and remains active for manual conflation.
        mapped_provisional_roof_units = (
            potential_mapped_units
            if provisional_roof_units > 1 or plausible_whole_candidate_match
            else 0
        )
        remaining_provisional_roof_units = (
            provisional_roof_units - mapped_provisional_roof_units
        )
        fully_preexisting_osm = osm_overlap and plausible_whole_candidate_match

        if footprint_metric is None:
            oam_coverage_fraction = 0.0
        else:
            oam_coverage_fraction = (
                anchor.geometry_metric.intersection(footprint_metric).area
                / anchor.geometry_metric.area
            )
        sufficient_oam_coverage = oam_coverage_fraction >= MIN_OAM_COVERAGE_FRACTION
        insufficient_oam_coverage = not sufficient_oam_coverage
        outside_oam = oam_coverage_fraction <= 0.0
        near_oam_edge = bool(
            footprint_metric is not None
            and anchor.geometry_metric.distance(footprint_metric.boundary) <= OAM_EDGE_WARNING_M
        )
        category = _category(source_ai, source_microsoft)
        compactness = _compactness(anchor.geometry_metric)
        priority_score, priority_tier, risk_flags = _priority(
            category=category,
            area_m2=anchor.geometry_metric.area,
            compactness=compactness,
            near_oam_edge=near_oam_edge,
            match_count_microsoft=len(microsoft_indexes),
        )
        if fully_preexisting_osm:
            risk_flags.append("fully_preexisting_osm")
        elif osm_overlap:
            risk_flags.extend(
                ["partial_osm_conflation_required", "osm_overlap_excluded_from_pilot"]
            )
        if insufficient_oam_coverage:
            risk_flags.append("insufficient_valid_oam_coverage")

        candidate_id = "sgo-" + _stable_geometry_key(origin, anchor.geometry_wgs84)[:12]
        ai_index = ai_index_by_key.get(anchor.stable_key) if source_ai else None
        ms_match_quality = [
            round(float(ai_ms_metrics[(ai_index, ms_index)]["quality"]), 3)  # type: ignore[index]
            for ms_index in microsoft_indexes
            if ai_index is not None and (ai_index, ms_index) in ai_ms_metrics
        ]
        matched_osm_source_ids = sorted(_osm_source_id(osm[index]) for index in osm_indexes)
        matched_osm_way_ids = sorted(
            int(osm[index].properties["osm_way_id"])
            for index in osm_indexes
            if osm[index].properties.get("osm_way_id") not in (None, "")
        )
        osm_match_details = [
            {
                "source_id": _osm_source_id(osm[index]),
                "iou": round(float(anchor_osm_metrics[(anchor_index, index)]["iou"]), 4),
                "candidate_coverage": round(
                    float(anchor_osm_metrics[(anchor_index, index)]["candidate_coverage"]),
                    4,
                ),
                "osm_coverage": round(
                    float(anchor_osm_metrics[(anchor_index, index)]["osm_coverage"]),
                    4,
                ),
                "intersection_area_m2": round(
                    float(anchor_osm_metrics[(anchor_index, index)]["intersection_area_m2"]),
                    2,
                ),
            }
            for index in osm_indexes
        ]

        active = remaining_provisional_roof_units > 0 and sufficient_oam_coverage
        pilot_eligible = active and not osm_overlap
        properties = {
            "candidate_id": candidate_id,
            "category": category,
            "priority_score": priority_score,
            "priority_tier": priority_tier,
            "review_status": "unreviewed",
            "active": active,
            "pilot_eligible": pilot_eligible,
            "upload_ready": False,
            "recommended_action": _recommended_action(
                category,
                fully_preexisting_osm,
                osm_overlap,
                insufficient_oam_coverage,
            ),
            "geometry_origin": origin,
            "geometry_license_rule": (
                "AI screening derived from OAM; human must retrace/correct from OAM"
                if origin == "ai_oam"
                else "Microsoft CDLA-Permissive-2.0; human must verify/retrace from OAM"
            ),
            "source_ai": source_ai,
            "source_microsoft": source_microsoft,
            "source_osm": osm_overlap,
            "osm_overlap": osm_overlap,
            "preexisting_osm": fully_preexisting_osm,
            "fully_preexisting_osm": fully_preexisting_osm,
            "outside_valid_oam_coverage": outside_oam,
            "insufficient_valid_oam_coverage": insufficient_oam_coverage,
            "oam_valid_coverage_fraction": round(oam_coverage_fraction, 6),
            "minimum_required_oam_coverage_fraction": MIN_OAM_COVERAGE_FRACTION,
            "near_oam_valid_data_edge": near_oam_edge,
            "area_m2": round(anchor.geometry_metric.area, 2),
            "compactness": round(compactness, 4),
            "microsoft_match_count": len(microsoft_indexes),
            "provisional_roof_units": provisional_roof_units,
            "mapped_provisional_roof_units": mapped_provisional_roof_units,
            "remaining_provisional_roof_units": remaining_provisional_roof_units,
            "osm_matched_source_feature_count": len(osm_indexes),
            "osm_union_candidate_coverage": round(osm_union_candidate_coverage, 6),
            "matched_osm_source_ids": matched_osm_source_ids,
            "matched_osm_way_ids": matched_osm_way_ids,
            "microsoft_match_quality": ms_match_quality,
            "osm_overlap_match_details": osm_match_details,
            "risk_flags": sorted(set(risk_flags)),
        }
        queue_features.append(
            {"type": "Feature", "geometry": mapping(anchor.geometry_wgs84), "properties": properties}
        )

    queue_features.sort(key=lambda item: item["properties"]["candidate_id"])
    active_features = [item for item in queue_features if item["properties"]["active"]]
    pilot_eligible_features = [
        item for item in queue_features if item["properties"]["pilot_eligible"]
    ]
    pilot_features = _pilot_candidates(pilot_eligible_features, pilot_size)

    category_counts = Counter(item["properties"]["category"] for item in queue_features)
    active_category_counts = Counter(item["properties"]["category"] for item in active_features)
    pilot_strata = Counter(item["properties"]["pilot_stratum"] for item in pilot_features)
    candidate_source_presence_counts = {
        key: sum(bool(item["properties"][key]) for item in queue_features)
        for key in (
            "source_ai",
            "source_microsoft",
            "source_osm",
        )
    }
    matched_osm_indexes = {index for indexes in anchor_osm.values() for index in indexes}
    matching_thresholds = {
        "ai_microsoft": {
            "minimum_iou": MIN_IOU,
            "minimum_smaller_footprint_overlap": MIN_SMALLER_OVERLAP,
            "maximum_edge_distance_m": MAX_EDGE_DISTANCE_M,
            "maximum_centroid_distance_m_for_proximity_match": MAX_CENTROID_DISTANCE_M,
        },
        "osm_deduplication": {
            "assignment": "each unique OSM source feature to at most one best candidate",
            "minimum_intersection_area_m2": OSM_MIN_INTERSECTION_M2,
            "minimum_iou": OSM_MIN_IOU,
            "minimum_candidate_coverage": OSM_MIN_CANDIDATE_COVERAGE,
            "minimum_osm_source_coverage": OSM_MIN_SOURCE_COVERAGE,
            "minimum_union_candidate_coverage_for_full_exclusion": OSM_MIN_WHOLE_CANDIDATE_COVERAGE,
            "proximity_only_matching": False,
        },
        "minimum_valid_oam_coverage_fraction": MIN_OAM_COVERAGE_FRACTION,
    }
    source_provenance = {
        "oam_imagery": {
            "role": "authoritative tracing imagery",
            "acquisition_dates": sorted(
                {
                    str((feature.get("properties") or {}).get("acquisition_date"))
                    for feature in footprint_raw
                    if (feature.get("properties") or {}).get("acquisition_date")
                }
            ),
            "gsd_m_values": sorted(
                {
                    float((feature.get("properties") or {}).get("gsd_m"))
                    for feature in footprint_raw
                    if (feature.get("properties") or {}).get("gsd_m") is not None
                }
            ),
        },
        "ai_screening": {
            "role": "candidate geometry; screening only",
            "models": sorted(
                {
                    str((feature.get("properties") or {}).get("model"))
                    for feature in ai_raw
                    if (feature.get("properties") or {}).get("model")
                }
            ),
        },
        "microsoft": {
            "role": "candidate geometry and AI corroboration",
            "license": "CDLA-Permissive-2.0",
            "snapshot_verified": "2026-08-13",
        },
        "google_open_buildings": {
            "role": "excluded from queue processing",
            "reason": "sole local detection is below the dataset's recommended confidence",
        },
        "openstreetmap": {
            "role": "strict de-duplication and conflation signal only",
            "snapshot_verified": "2026-08-19",
            "unique_input_source_feature_count": len(osm),
        },
    }
    summary = {
        "metric_crs": metric_crs.to_string(),
        "matching_thresholds": matching_thresholds,
        "source_provenance": source_provenance,
        "license_boundary": {
            "candidate_geometry_sources": ["AI derived from OAM", "Microsoft"],
            "qa_only_sources": ["OpenStreetMap"],
            "excluded_from_queue_processing": ["IGAC", "Google Open Buildings"],
            "upload_ready_candidate_count": 0,
        },
        "input_source_feature_counts": {
            "ai_valid_feature_count": len(ai),
            "microsoft_valid_feature_count": len(microsoft),
            "igac_feature_count_excluded_from_queue_processing": len(
                by_source["IGAC urban"]
            ),
            "google_feature_count_excluded_from_queue_processing": len(
                by_source["Google Open Buildings v3"]
            ),
            "osm_raw_feature_count": len(by_source["OpenStreetMap"]),
            "osm_unique_valid_source_feature_count": len(osm),
        },
        "queue_candidate_count": len(queue_features),
        "active_candidate_count": len(active_features),
        "pilot_eligible_candidate_count": len(pilot_eligible_features),
        "total_provisional_roof_units": sum(
            int(item["properties"]["provisional_roof_units"]) for item in queue_features
        ),
        "osm_mapped_provisional_roof_units": sum(
            int(item["properties"]["mapped_provisional_roof_units"])
            for item in queue_features
        ),
        "remaining_provisional_roof_units": sum(
            int(item["properties"]["remaining_provisional_roof_units"])
            for item in queue_features
        ),
        "active_remaining_provisional_roof_units": sum(
            int(item["properties"]["remaining_provisional_roof_units"])
            for item in active_features
        ),
        "provisional_roof_units_note": "A screening workload measure: one per candidate, or the number of Microsoft footprints when an AI component spans several. It is not a confirmed building count.",
        "candidate_source_presence_counts": candidate_source_presence_counts,
        "osm_overlap_candidate_count": sum(
            bool(item["properties"]["osm_overlap"]) for item in queue_features
        ),
        "fully_preexisting_osm_candidate_count": sum(
            bool(item["properties"]["fully_preexisting_osm"])
            for item in queue_features
        ),
        "partial_osm_conflation_candidate_count": sum(
            bool(item["properties"]["osm_overlap"])
            and not bool(item["properties"]["fully_preexisting_osm"])
            for item in queue_features
        ),
        "unique_osm_input_source_feature_count": len(osm),
        "unique_osm_matched_source_feature_count": len(matched_osm_indexes),
        "insufficient_oam_coverage_candidate_count": sum(
            bool(item["properties"]["insufficient_valid_oam_coverage"])
            for item in queue_features
        ),
        "candidate_category_counts": dict(sorted(category_counts.items())),
        "active_candidate_category_counts": dict(sorted(active_category_counts.items())),
        "pilot_candidate_count": len(pilot_features),
        "pilot_candidate_strata_counts": dict(sorted(pilot_strata.items())),
        "warning": "Screening output only. Human review, OSM community consultation, and license/import-policy checks are required before editing OSM.",
    }
    metadata = {
        "name": "santiago_license_aware_review_queue",
        "source_manifest": "source_manifest.json",
        "metric_crs_used_for_analysis": metric_crs.to_string(),
        "matching_thresholds": matching_thresholds,
        "sources": source_provenance,
        "upload_ready": False,
    }
    pilot_metadata = {
        "name": "santiago_stratified_200_candidate_pilot",
        "source_manifest": "source_manifest.json",
        "selection": "deterministic 65% AI+Microsoft consensus, 17.5% AI-only, 17.5% Microsoft-only; shortages filled by priority",
        "osm_overlap_policy": "all OSM-overlap candidates excluded from initial pilot",
        "minimum_valid_oam_coverage_fraction": MIN_OAM_COVERAGE_FRACTION,
        "sources": source_provenance,
        "upload_ready": False,
    }
    return (
        _feature_collection(queue_features, metadata),
        _feature_collection(pilot_features, pilot_metadata),
        summary,
    )


def write_outputs(
    queue: dict[str, Any],
    pilot: dict[str, Any],
    summary: dict[str, Any],
    queue_path: Path,
    pilot_path: Path,
    summary_path: Path,
) -> None:
    for path in (queue_path, pilot_path, summary_path):
        path.parent.mkdir(parents=True, exist_ok=True)
    with queue_path.open("w", encoding="utf-8") as handle:
        json.dump(queue, handle, indent=2, sort_keys=True)
        handle.write("\n")
    with pilot_path.open("w", encoding="utf-8") as handle:
        json.dump(pilot, handle, indent=2, sort_keys=True)
        handle.write("\n")
    with summary_path.open("w", encoding="utf-8") as handle:
        json.dump(summary, handle, indent=2, sort_keys=True)
        handle.write("\n")


def parse_args() -> argparse.Namespace:
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument("--ai", type=Path, default=DEFAULT_AI)
    parser.add_argument("--sources", type=Path, default=DEFAULT_SOURCES)
    parser.add_argument("--footprint", type=Path, default=DEFAULT_FOOTPRINT)
    parser.add_argument("--queue-out", type=Path, default=DEFAULT_QUEUE)
    parser.add_argument("--pilot-out", type=Path, default=DEFAULT_PILOT)
    parser.add_argument("--summary-out", type=Path, default=DEFAULT_SUMMARY)
    parser.add_argument("--pilot-size", type=int, default=200)
    return parser.parse_args()


def main() -> None:
    args = parse_args()
    queue, pilot, summary = build_queue(
        _read_geojson(args.ai),
        _read_geojson(args.sources),
        _read_geojson(args.footprint) if args.footprint.exists() else None,
        pilot_size=args.pilot_size,
    )
    write_outputs(
        queue,
        pilot,
        summary,
        args.queue_out,
        args.pilot_out,
        args.summary_out,
    )
    print(json.dumps(summary, indent=2, sort_keys=True))


if __name__ == "__main__":
    main()
