
    ^jG                        d dl Z d dlmZ d dlmZ d dlmZmZmZ d dl	Z
d dlZd dlmc mZ d dlmZ ddlmZ ddlmZ dd	lmZmZ dd
lmZmZmZmZmZmZ ddl m!Z!m"Z" ddl#m$Z$m%Z%m&Z&  e&       rd dl'Z' G d de!d      Z(ddde)de*fdZ+d1dZ,d2dZ-de.e*   de*de*fdZ/	 	 	 	 d3de*de*de)de*dz  d e.e*   dz  d!e0e.e.e*      e.e*   f   fd"Z1d# Z2d$e*d!ejf                  fd%Z4	 d4d&Z5	 d5de*d'ejf                  d(e0e*e*f   d!ejf                  fd)Z6d*e7e8ef   d!ejf                  fd+Z9d6d,Z:d7d-Z;d. Z<e% G d/ d0e             Z=d0gZ>y)8    N)deepcopy)product)AnyOptionalUnion)batched_nms   )TorchvisionBackend)'SemanticSegmentationPostProcessorOutput)BatchFeatureget_size_dict)IMAGENET_STANDARD_MEANIMAGENET_STANDARD_STDChannelDimension
ImageInputPILImageResamplingSizeDict)ImagesKwargsUnpack)
TensorTypeauto_docstringis_vision_availablec                   &    e Zd ZU dZeeef   ed<   y)Sam3ImageProcessorKwargsz
    mask_size (`dict[str, int]`, *optional*):
        The size `{"height": int, "width": int}` to resize the segmentation maps to.
    	mask_sizeN)__name__
__module____qualname____doc__dictstrint__annotations__     y/var/www/ramen.bs-engineer-server.com/venv/lib/python3.12/site-packages/transformers/models/sam3/image_processing_sam3.pyr   r   3   s    
 CH~r%   r   F)totalmaskstorch.Tensormask_thresholdstability_score_offsetc                 (   | ||z   kD  j                  dt        j                        j                  dt        j                        }| ||z
  kD  j                  dt        j                        j                  dt        j                        }||z  }|S )Ndtype)sumtorchint16int32)r(   r*   r+   intersectionsunionsstability_scoress         r&   _compute_stability_scorer7   <   s     
.#99	:??%++?VZZ[]ejepepZq  ~(>>?DDRu{{D[__`bjojuju_vF$v-r%   c                 l   t        j                  |       dk(  r1t        j                  g | j                  dd dd| j                  iS | j                  }|dd \  }}t        j
                  | d      \  }}|t        j                  ||j                        dddf   z  }t        j
                  |d      \  }}||| z  z   }t        j                  |d      \  }}t        j
                  | d      \  }	}|	t        j                  ||	j                        dddf   z  }
t        j
                  |
d      \  }}|
||	 z  z   }
t        j                  |
d      \  }}||k  ||k  z  }t        j                  ||||gd      }|| j                  d      z  } |j                  g |dd d }|S )	aL  
    Computes the bounding boxes around the given input masks. The bounding boxes are in the XYXY format which
    corresponds the following required indices:
        - LEFT: left hand side of the bounding box
        - TOP: top of the bounding box
        - RIGHT: right of the bounding box
        - BOTTOM: bottom of the bounding box

    Return [0,0,0,0] for an empty mask. For input shape channel_1 x channel_2 x ... x height x width, the output shape
    is channel_1 x channel_2 x ... x 4.

    Args:
        - masks (`torch.Tensor` of shape `(batch, nb_mask, height, width)`)
    r   N   devicer-   dimr;   )r1   numelzerosshaper;   maxarangeminstack	unsqueezereshape)r(   rA   heightwidth	in_height_in_height_coordsbottom_edges	top_edgesin_widthin_width_coordsright_edges
left_edgesempty_filterouts                  r&   _batched_mask_to_boxrU   G   s   " {{5Q{{EEKK,EaEEE KKE"#JMFE 99U+LIq 5<<y?O?O#PQUWXQX#YYii 0b9OL!'&YJ*??99-26LIq ))Er*KHaeHOO!LTSTW!UUOYYB7NK%((;;OIIo26MJ  *,	1IJL
++z9k<Hb
QC
,))"-
-C #++
%uSbz
%1
%CJr%   c                 p   t        j                  |t         j                  | j                        }t        j                  |t         j                  | j                        }|\  }}}}t        j                  ||||gg| j                        }	t        | j                        dk(  r|	j                  d      }	| |	z   j                         } t        j                  | |dddf   |d      }
t        j                  | |dddf   |d      }t        j                  |
|       }
t        j                  |
d      S )	zNFilter masks at the edge of a crop, but not at the edge of the original image.r/   r;   r>   r	      Nr   )atolrtolr<   )r1   	as_tensorfloatr;   tensorlenrA   rF   iscloselogical_andany)boxescrop_boxorig_boxrY   crop_box_torchorig_box_torchlefttoprK   offsetnear_crop_edgenear_image_edges               r&   _is_box_near_crop_edgerl   x   s    __XU[[VN__XU[[VNOD#q!\\D#tS125<<HF
5;;1!!!$V^""$E]]5.q*ASTUNmmE>$'+BTUVO&&~7GHN99^++r%   rc   orig_height
orig_widthc                     |\  }}}}|dk(  r|dk(  r||k(  r||k(  r| S |||z
  z
  |||z
  z
  }	}|||z
  ||	|z
  f}
t         j                  j                  j                  | |
d      S )Nr   )value)r1   nn
functionalpad)r(   rc   rm   rn   rg   rh   rightbottompad_xpad_yrs   s              r&   
_pad_masksrx      s    'D#ufqySAX%:"5&K:O.v|0L5EsECK
0C88""5#Q"77r%   target_sizecrop_n_layersoverlap_ratiopoints_per_cropcrop_n_points_downscale_factorreturnc                 p   t        | t              rt        d      | j                  dd }g }t	        |dz         D ]-  }t        |||z  z        }	|j                  t        |	             / t        |||      \  }
}t        |
| ||||      \  }}t        j                  |
      }
|
j                         }
t        j                  |      }|j                  d      j                  dddd      }t        j                  |      }t        j                   |dddddddf   t        j"                        }|
|||fS )	a  
    Generates a list of crop boxes of different sizes. Each layer has (2**i)**2 boxes for the ith layer.

    Args:
        image (`torch.Tensor`):
            Image to generate crops for.
        target_size (`int`):
            Size of the smallest crop.
        crop_n_layers (`int`, *optional*):
            If `crops_n_layers>0`, mask prediction will be run again on crops of the image. Sets the number of layers
            to run, where each layer has 2**i_layer number of image crops.
        overlap_ratio (`int`, *optional*):
            Sets the degree to which crops overlap. In the first crop layer, crops will overlap by this fraction of the
            image length. Later layers with more crops scale down this overlap.
        points_per_crop (`int`, *optional*):
            Number of points to sam3ple per crop.
        crop_n_points_downscale_factor (`int`, *optional*):
            The number of points-per-side sam3pled in layer n is scaled down by crop_n_points_downscale_factor**n.
    z.Only one image is allowed for crop generation.r9   NrX   r      r	   r.   )
isinstancelist
ValueErrorrA   ranger"   append_build_point_grid_generate_per_layer_crops_generate_crop_imagesr1   r]   r\   rE   rF   permute	ones_likeint64)imagery   rz   r{   r|   r}   original_sizepoints_gridin_points
crop_boxes
layer_idxscropped_imagespoint_grid_per_cropinput_labelss                  r&   _generate_crop_boxesr      s0   8 %IJJKK$MK=1$% 8*H!*KLM,X678 7}mUbcJ
*?E;
K+'N' j)J!!#Jkk"56O%//2::1aAFO[[0N???1aA:#>ekkRLDDr%   c           	         g g }}|\  }}t        ||      }|j                  dd||g       |j                  d       t        |       D ]  }d|dz   z  }	t        ||z  d|	z  z        }
t        t	        j
                  |
|	dz
  z  |z   |	z              }t        t	        j
                  |
|	dz
  z  |z   |	z              }t        |	      D cg c]  }t        ||
z
  |z         }}t        |	      D cg c]  }t        ||
z
  |z         }}t        ||      D ]J  \  }}||t        ||z   |      t        ||z   |      g}|j                  |       |j                  |dz          L  ||fS c c}w c c}w )aq  
    Generates 2 ** (layers idx + 1) crops for each crop_n_layers. Crops are in the XYWH format : The XYWH format
    consists of the following required indices:
        - X: X coordinate of the top left of the bounding box
        - Y: Y coordinate of the top left of the bounding box
        - W: width of the bounding box
        - H: height of the bounding box
    r   r   rX   )rD   r   r   r"   mathceilr   )rz   r{   r   r   r   	im_heightim_width
short_sidei_layern_crops_per_sideoverlap
crop_widthcrop_heightr   crop_box_x0crop_box_y0rg   rh   boxs                      r&   r   r      s     
J'IxY)J q!Xy12a' +1-mj0A8H4HIJG/?!/C$Dx$OSc#cde
$))W0@10D%E	%QUe$efg@EFV@WX1sJ0A56XXAFGWAXYAsK'1Q67YY k: 	+ID#c$"3X>C+DUW`@abCc"gk*	++ z!! YYs   E)/E.
n_per_sidec                    dd| z  z  }t        j                  |d|z
  |       }t        j                  |dddf   | df      }t        j                  |dddf   d| f      }t        j                  ||gd      j	                  dd      }|S )z;Generates a 2D grid of points evenly spaced in [0,1]x[0,1].rX   r   Nr-   r<   )r1   linspacetilerE   rG   )r   ri   points_one_sidepoints_xpoints_ypointss         r&   r   r      s    !j.!FnnVQZDOzz/$'2ZODHzz/!T'2Q
ODH[[(H-26>>r1EFMr%   c                 \   g }g }t        |       D ]  \  }	}
|
\  }}}}|dd||||f   }|j                  |       |j                  dd }t        j                  |      j                  d      j                  d      }|||	      |z  }t        |||      }|j                  |        ||fS )z
    Takes as an input bounding boxes that are used to crop the image. Based in the crops, the corresponding points are
    also passed.
    Nr9   )r   )dimsr   )	enumerater   rA   r1   r]   fliprF   _normalize_coordinates)r   r   r   r   ry   r   input_data_formatr   total_points_per_cropr   rc   rg   rh   rt   ru   
cropped_imcropped_im_sizepoints_scaler   normalized_pointss                       r&   r   r      s     N , 88#+ c5&1c&j$u*45
j)$**23/||O499t9DNNqQZ]+l:2;V$$%678 000r%   coordsr   c                 <   |\  }}| dz  t        ||      z  }||z  ||z  }}t        |dz         }t        |dz         }t        |      j                         }|r|j	                  ddd      }|d   ||z  z  |d<   |d   ||z  z  |d<   |r|j	                  dd      }|S )z
    Expects a numpy array of length 2 in the final dimension. Requires the original image size in (height, width)
    format.
    g      ?      ?r-   r   ).r   ).rX   r:   )rB   r"   r   r\   rG   )	ry   r   r   is_bounding_box
old_height	old_widthscale
new_height	new_widths	            r&   r   r     s     *J	#J	 ::E&.	E0A	JIO$IZ#%&Jf##%FAq)F^y9'<=F6NF^zJ'>?F6NA&Mr%   rlec                     | d   \  }}t        j                  ||z  t              }d}d}| d   D ]  }|||||z    ||z  }| } |j                  ||      }|j	                  dd      S )z/Compute a binary mask from an uncompressed RLE.sizer.   r   FcountsrX   )r1   emptyboolrG   	transpose)r   rH   rI   maskidxparitycounts          r&   _rle_to_maskr   *  s    KMFE;;v~T2D
CFX "(S3;u <<v&D>>!Qr%   c                    t        |j                         |t        j                  |j                  d         |      }||   }|D cg c]  }| |   	 } }||   }| D cg c]  }t        |       }}||| |fS c c}w c c}w )a  
    Perform NMS (Non Maximum Suppression) on the outputs.

    Args:
            rle_masks (`torch.Tensor`):
                binary masks in the RLE format
            iou_scores (`torch.Tensor` of shape (nb_masks, 1)):
                iou_scores predicted by the model
            mask_boxes (`torch.Tensor`):
                The bounding boxes corresponding to segmentation masks
            amg_crops_nms_thresh (`float`, *optional*, defaults to 0.7):
                NMS threshold.
    r   )rb   scoresidxsiou_threshold)r   r\   r1   r@   rA   r   )	rle_masks
iou_scores
mask_boxesamg_crops_nms_threshkeep_by_nmsr   r   r(   s           r&   !_post_process_for_mask_generationr   8  s      [[))!,-*	K K(J'23!13I3K(J*343\#4E4*i33	 44s   A8A=c                    | j                   \  }}}| j                  ddd      j                  d      } | ddddf   | ddddf   z  }|j                         }g }t	        |      D ]  }||dddf   |k(  df   dz   }t        |      dk(  rA| |df   dk(  r|j                  ||g||z  gd       n|j                  ||gd||z  gd       f|dd |dd z
  }	| |df   dk(  rg ndg}
|
|d   j                         g|	j                         z   ||z  |d   j                         z
  gz   z  }
|j                  ||g|
d        |S )z^
    Encodes masks the run-length encoding (RLE), in the format expected by pycoco tools.
    r   r   rX   Nr-   )r   r   )	rA   r   flattennonzeror   r^   r   itemtolist)
input_mask
batch_sizerH   rI   diffchange_indicesrT   r   cur_idxsbtw_idxsr   s              r&   _mask_to_rler   U  s   
 !+ 0 0J##Aq!,44Q7J aez!SbS&11D\\^N C: @!.A"6!";Q">?!Cx=A !Q$1$

VUO?OPQ

VUO6E>?RSTAB<(3B-/!!Q$'1,1#8A;##%&)::funxXZ|O`O`Ob>b=ccc

VUOv>?@ Jr%   c                    t        |t        t        f      rMt        j                  |D cg c]  }|d   	 c}      }t        j                  |D cg c]  }|d   	 c}      }n:t        |t        j
                        r|j                  d      \  }}nt        d      t        j                  ||||gd      }|j                  d      j                  | j                        }| |z  } | S c c}w c c}w )a  
    Scale batch of bounding boxes to the target sizes.

    Args:
        boxes (`torch.Tensor` of shape `(batch_size, num_boxes, 4)`):
            Bounding boxes to scale. Each box is expected to be in (x1, y1, x2, y2) format.
        target_sizes (`list[tuple[int, int]]` or `torch.Tensor` of shape `(batch_size, 2)`):
            Target sizes to scale the boxes to. Each target size is expected to be in (height, width) format.

    Returns:
        `torch.Tensor` of shape `(batch_size, num_boxes, 4)`: Scaled bounding boxes.
    r   rX   z4`target_sizes` must be a list, tuple or torch.Tensorr<   )r   r   tupler1   r]   Tensorunbind	TypeErrorrE   rF   tor;   )rb   target_sizesr   image_heightimage_widthscale_factors         r&   _scale_boxesr   t  s     ,u.||<$@aQqT$@All,#?QAaD#?@	L%,,	/$0$7$7$:!kNOO;;\;U[\]L))!,//=LL EL %A#?s   C*C/c                   `    e Zd ZeZej                  ZeZ	e
ZdddZdddZdZdZdZdZdZdZdZdee   f fdZe	 d*ded	edz  dee   d
ef fd       Z	 d*dedz  d
ef fdZ	 d*ded	edz  dedede e!df   dz  dee   d
efdZ"de#d   de!e$z  dz  d
df fdZ%	 	 	 	 	 d+ddde&de'de&dz  de#e&   dz  de(d   fdZ)	 	 	 	 d,dZ*	 	 	 	 	 d-dZ+d Z,d e-j\                  d
e-j\                  fd!Z/	 	 	 d.d"e#e0e&e&f      dz  d#e'd$ed
d%fd&Z1d/d#e'd"e#e0   dz  fd'Z2	 	 	 d0d#e'd(e'd"e#e0   dz  fd)Z3 xZ4S )1Sam3ImageProcessori  )rH   rI   i   TNkwargsc                 $    t        |   di | y )Nr$   )super__init__)selfr   	__class__s     r&   r   zSam3ImageProcessor.__init__  s    "6"r%   imagessegmentation_mapsr~   c                 &    t        |   ||fi |S )zp
        segmentation_maps (`ImageInput`, *optional*):
            The segmentation maps to preprocess.
        )r   
preprocess)r   r   r   r   r   s       r&   r   zSam3ImageProcessor.preprocess  s     w!&*;FvFFr%   r   c                 |    |&t        |t              st        di t        |d      }||d<   t        |   di |S )zT
        Update kwargs that need further processing before being validated.
        r   )
param_namer$   )r   r   r   r   _standardize_kwargs)r   r   r   r   s      r&   r   z&Sam3ImageProcessor._standardize_kwargs  sE      Ix)H T={#STI'{w*4V44r%   do_convert_rgbr   r;   ztorch.devicec                 8   | j                  ||||      }|D cg c]  }|j                  dd  }}|j                         }	 | j                  |fi |	}
|
|d}|| j                  |ddt        j
                        }|j                         }|j                  ddt        j                  |j                  d      d	        | j                  dd
|i|}|j                  d      j                  t        j                        |d<   t        ||d         S c c}w )z/
        Preprocess image-like inputs.
        )r   r   r   r;   r9   N)pixel_valuesoriginal_sizesr   F)r   expected_ndimsr   r   r   )do_normalize
do_rescaleresampler   r   rX   labelsreturn_tensors)datatensor_typer$   )_prepare_image_like_inputsrA   copy_preprocessr   FIRSTupdater   NEARESTpopsqueezer   r1   r   r   )r   r   r   r   r   r;   r   r   r   images_kwargsr   r  processed_segmentation_mapssegmentation_maps_kwargss                 r&   _preprocess_image_like_inputsz0Sam3ImageProcessor._preprocess_image_like_inputs  sE    00.L]fl 1 
 9??u%++bc*??'t''@-@(,

 (*.*I*I( $"2"8"8	 +J +' (.{{}$$++$)"' 2 : :488E	 +;$*:*: +2+6N+' 9@@CFFu{{SDN6:J3KLL= @s   Dr)   r  c                 <    t        |   |fd|i|j                  S )Nr  )r   r  r   )r   r   r  r   r   s       r&   r  zSam3ImageProcessor._preprocess  s%     w"6S.SFS```r%   r   z+np.ndarray | PIL.Image.Image | torch.Tensorrz   r{   r|   r}   c                     | j                  |      }t        ||||||      \  }}}	}
|t        j                  d      }|j	                  |      }|j	                  |      }|
j	                  |      }
|||	|
fS )a  
        Generates a list of crop boxes of different sizes. Each layer has (2**i)**2 boxes for the ith layer.

        Args:
            image (`torch.Tensor`):
                Input original image
            target_size (`int`):
                Target size of the resized image
            crop_n_layers (`int`, *optional*, defaults to 0):
                If >0, mask prediction will be run again on crops of the image. Sets the number of layers to run, where
                each layer has 2**i_layer number of image crops.
            overlap_ratio (`float`, *optional*, defaults to 512/1500):
                Sets the degree to which crops overlap. In the first crop layer, crops will overlap by this fraction of
                the image length. Later layers with more crops scale down this overlap.
            points_per_crop (`int`, *optional*, defaults to 32):
                Number of points to sam3ple from each crop.
            crop_n_points_downscale_factor (`list[int]`, *optional*, defaults to 1):
                The number of points-per-side sam3pled in layer n is scaled down by crop_n_points_downscale_factor**n.
            device (`torch.device`, *optional*, defaults to None):
                Device to use for the computation. If None, cpu will be used.
        cpu)process_imager   r1   r;   r   )r   r   ry   rz   r{   r|   r}   r;   r   r   r   s              r&   generate_crop_boxesz&Sam3ImageProcessor.generate_crop_boxes  s    > ""5)DX*E
A
O^\ >\\%(F]]6*
),,V4#v.?NLHHr%   c	                    |\  }	}
|j                  dd      }|j                  dd      }|j                  d   |j                  d   k7  rt        d      |j                  |j                  k7  r|j	                  |j                        }|j                  d   }t        j                  |t
        j                  |j                        }|dkD  r|||kD  z  }|dkD  rt        |||      }|||kD  z  }||   }||   }||kD  }t        |      }t        ||dd|
|	g       }||   }||   }||   }t        |||	|
      }t        |      }|||fS )a  
        Filters the predicted masks by selecting only the ones that meets several criteria. The first criterion being
        that the iou scores needs to be greater than `pred_iou_thresh`. The second criterion is that the stability
        score needs to be greater than `stability_score_thresh`. The method also converts the predicted masks to
        bounding boxes and pad the predicted masks if necessary.

        Args:
            masks (`torch.Tensor`):
                Input masks.
            iou_scores (`torch.Tensor`):
                List of IoU scores.
            original_size (`tuple[int,int]`):
                Size of the original image.
            cropped_box_image (`torch.Tensor`):
                The cropped image.
            pred_iou_thresh (`float`, *optional*, defaults to 0.88):
                The threshold for the iou scores.
            stability_score_thresh (`float`, *optional*, defaults to 0.95):
                The threshold for the stability score.
            mask_threshold (`float`, *optional*, defaults to 0):
                The threshold for the predicted masks.
            stability_score_offset (`float`, *optional*, defaults to 1):
                The offset for the stability score used in the `_compute_stability_score` method.
        r   rX   z4masks and iou_scores must have the sam3e batch size.rW           )r   rA   r   r;   r   r1   onesr   r7   rU   rl   rx   r   )r   r(   r   r   cropped_box_imagepred_iou_threshstability_score_threshr*   r+   original_heightoriginal_widthr   	keep_maskr6   r   converted_boxess                   r&   filter_maskszSam3ImageProcessor.filter_masks'  sw   F +8'''1-
a#;;q>Z--a00STT<<:,,,#u||4J[[^
JJzELLQ	S !Z/%ABI "C'7~Oef!%58N%NOII&i  &.u5+.A~0W
 
	 	"i ))45"3_nUU#fo--r%   c                    t        |t        j                  t        j                  f      r|j                         }g }	t        |      D ]  \  }
}t        ||
   t        j                        rt        j                  ||
         ||
<   n(t        ||
   t        j                        st        d      t        j                  ||
   |dd      }|r| j                  |      }|r||kD  }|	j                  |        |	S )aG  
        Remove padding and upscale masks to the original image size.

        Args:
            masks (`Union[torch.Tensor, List[torch.Tensor], np.ndarray, List[np.ndarray]]`):
                Batched masks from the mask_decoder in (batch_size, num_channels, height, width) format.
            original_sizes (`Union[torch.Tensor, List[Tuple[int,int]]]`):
                The original sizes of each image before it was resized to the model's expected input shape, in (height,
                width) format.
            mask_threshold (`float`, *optional*, defaults to 0.0):
                Threshold for binarization and post-processing operations.
            binarize (`bool`, *optional*, defaults to `True`):
                Whether to binarize the masks.
            max_hole_area (`float`, *optional*, defaults to 0.0):
                The maximum area of a hole to fill.
            max_sprinkle_area (`float`, *optional*, defaults to 0.0):
                The maximum area of a sprinkle to fill.
            apply_non_overlapping_constraints (`bool`, *optional*, defaults to `False`):
                Whether to apply non-overlapping constraints to the masks.

        Returns:
            (`torch.Tensor`): Batched masks in batch_size, num_channels, height, width) format, where (height, width)
            is given by original_size.
        zIInput masks should be a list of `torch.tensors` or a list of `np.ndarray`bilinearF)modealign_corners)r   r1   r   npndarrayr   r   
from_numpyr   Finterpolate"_apply_non_overlapping_constraintsr   )r   r(   r   r*   binarizemax_hole_areamax_sprinkle_area!apply_non_overlapping_constraintsr   output_masksr   r   interpolated_masks                r&   post_process_masksz%Sam3ImageProcessor.post_process_masksu  s    F nu||RZZ&@A+224N ). 9 
	3A}%(BJJ/ ++E!H5aa%,,7 kll !eAhJfk l0$($K$KL]$^!$5$F! 12
	3 r%   c                     t        ||||      S )a$  
        Post processes mask that are generated by calling the Non Maximum Suppression algorithm on the predicted masks.

        Args:
            all_masks (`torch.Tensor`):
                List of all predicted segmentation masks
            all_scores (`torch.Tensor`):
                List of all predicted iou scores
            all_boxes (`torch.Tensor`):
                List of all bounding boxes of the predicted masks
            crops_nms_thresh (`float`):
                Threshold for NMS (Non Maximum Suppression) algorithm.
        )r   )r   	all_masks
all_scores	all_boxescrops_nms_threshs        r&    post_process_for_mask_generationz3Sam3ImageProcessor.post_process_for_mask_generation  s     1J	Scddr%   
pred_masksc                     |j                  d      }|dk(  r|S |j                  }t        j                  |dd      }t        j                  ||      dddddf   }||k(  }t        j
                  ||t        j                  |d            }|S )	z
        Apply non-overlapping constraints to the object scores in pred_masks. Here we
        keep only the highest scoring object at each spatial location in pred_masks.
        r   rX   T)r=   keepdimr>   Ng      $)rB   )r   r;   r1   argmaxrC   whereclamp)r   r<  r   r;   max_obj_indsbatch_obj_indskeeps          r&   r.  z5Sam3ImageProcessor._apply_non_overlapping_constraints  s    
  __Q'
?""||JAtDj@D$PTATU~- [[z5;;zu3UV
r%   r   	thresholdreturn_segmentation_scoreszBlist[torch.Tensor] | list[SemanticSegmentationPostProcessorOutput]c                    |j                   }|t        d      |j                         }t        |      }|t        |      t        |      k7  rt        d      g }t	        |      D ]  }	t
        j                  j                  j                  ||	   j                  d      ||	   dd      }
|j                  t        |
d   |kD  j                  t
        j                        |
d   d	
              nMt	        |      D cg c]9  }t        ||df   |kD  j                  t
        j                        ||   d	
      ; }}|s|D cg c]  }|j                   }}|S c c}w c c}w )a  
        Converts the output of [`Sam3Model`] into semantic segmentation maps.

        Args:
            outputs ([`Sam3ImageSegmentationOutput`]):
                Raw outputs of the model containing semantic_seg.
            target_sizes (`list[tuple[int, int]]` of length `batch_size`, *optional*):
                List of tuples corresponding to the requested final size (height, width) of each prediction. If unset,
                predictions will not be resized.
            threshold (`float`, *optional*, defaults to 0.5):
                Threshold for binarizing the semantic segmentation masks.
            return_segmentation_scores (`bool`, *optional*, defaults to `False`):
                Whether to return segmentation scores alongside the segmentation map. When `True`, each element of
                the returned list is a [`SemanticSegmentationPostProcessorOutput`] with fields `segmentation`
                (binary class IDs, shape `(height, width)`) and `segmentation_scores` (sigmoid probabilities,
                shape `(1, height, width)`).

        Returns:
            `list[torch.Tensor]` or `list[SemanticSegmentationPostProcessorOutput]`: When
            `return_segmentation_scores=False` (default), a list of length `batch_size` where each item is a
            segmentation map of shape `(height, width)` with class IDs. When `return_segmentation_scores=True`,
            a list of [`SemanticSegmentationPostProcessorOutput`] with fields `segmentation` (class IDs, shape
            `(height, width)`) and `segmentation_scores` (shape `(1, height, width)`). In both cases,
            `(height, width)` corresponds to the target size (if `target_sizes` is specified).
        zSemantic segmentation output is not available in the model outputs. Make sure the model was run with semantic segmentation enabled.zTMake sure that you pass in as many target sizes as the batch dimension of the logitsr   r<   r&  Fr   r'  r(  )r   r   )segmentationsegmentation_scores)r  )semantic_segr   sigmoidr^   r   r1   rq   rr   r-  rF   r   r   r   longrI  )r   outputsr   rE  rF  semantic_logitssemantic_probsr   semantic_segmentationr   resized_probsr   r   s                r&   "post_process_semantic_segmentationz5Sam3ImageProcessor.post_process_semantic_segmentation  s   D ".."R  )002)
 #?#s<'88 j  %'!Z(  % 3 3 ? ?"3'11a18%c*#"'	 !@ ! &,,;-:4-@9-L,P,PQVQ[Q[,\3@3C. z*%  8)71)=	)I(M(Mejj(Y/=a/@%! % *CX$Y4T%6%6$Y!$Y$$% %Zs    >EE"c                    |j                   }|j                  }|j                  }|j                  d   }|t	        |      |k7  rt        d      |j                         }||j                         }	||	z  }|}
|t        |
|      }
g }t        ||
      D ](  \  }}||kD  }||   }||   }|j                  ||d       * |S )aD  
        Converts the raw output of [`Sam3Model`] into final bounding boxes in (top_left_x, top_left_y,
        bottom_right_x, bottom_right_y) format.

        Args:
            outputs ([`Sam3ImageSegmentationOutput`]):
                Raw outputs of the model containing pred_boxes, pred_logits, and optionally presence_logits.
            threshold (`float`, *optional*, defaults to 0.3):
                Score threshold to keep object detection predictions.
            target_sizes (`list[tuple[int, int]]`, *optional*):
                List of tuples (`tuple[int, int]`) containing the target size `(height, width)` of each image in the
                batch. If unset, predictions will not be resized.

        Returns:
            `list[dict]`: A list of dictionaries, each dictionary containing the following keys:
                - **scores** (`torch.Tensor`): The confidence scores for each predicted box on the image.
                - **boxes** (`torch.Tensor`): Image bounding boxes in (top_left_x, top_left_y, bottom_right_x,
                  bottom_right_y) format.
        r   9Make sure that you pass in as many target sizes as images)r   rb   )
pred_logits
pred_boxespresence_logitsrA   r^   r   rL  r   zipr   )r   rN  rE  r   rV  rW  rX  r   batch_scorespresence_scoresbatch_boxesresultsr   rb   rD  s                  r&   post_process_object_detectionz0Sam3ImageProcessor.post_process_object_detection#  s    ( ))''
!11 &&q)
#L(9Z(GXYY #**,&-557O'/9L ! #&{LAK {; 	?MFEI%DD\F$KENNfu=>		? r%   r*   c                    |j                   }|j                  }|j                  }|j                  }|j                  d   }	|t        |      |	k7  rt        d      |j                         }
||j                         }|
|z  }
|j                         }|}|t        ||      }g }t        t        |
||            D ]  \  }\  }}}||kD  }||   }||   }||   }|^||   }t        |      dkD  rKt        j                  j                  j                  |j                  d      |dd      j!                  d      }||kD  j#                  t        j$                        }|j'                  |||d        |S )aQ  
        Converts the raw output of [`Sam3Model`] into instance segmentation predictions with bounding boxes and masks.

        Args:
            outputs ([`Sam3ImageSegmentationOutput`]):
                Raw outputs of the model containing pred_boxes, pred_logits, pred_masks, and optionally
                presence_logits.
            threshold (`float`, *optional*, defaults to 0.3):
                Score threshold to keep instance predictions.
            mask_threshold (`float`, *optional*, defaults to 0.5):
                Threshold for binarizing the predicted masks.
            target_sizes (`list[tuple[int, int]]`, *optional*):
                List of tuples (`tuple[int, int]`) containing the target size `(height, width)` of each image in the
                batch. If unset, predictions will not be resized.

        Returns:
            `list[dict]`: A list of dictionaries, each dictionary containing the following keys:
                - **scores** (`torch.Tensor`): The confidence scores for each predicted instance on the image.
                - **boxes** (`torch.Tensor`): Image bounding boxes in (top_left_x, top_left_y, bottom_right_x,
                  bottom_right_y) format.
                - **masks** (`torch.Tensor`): Binary segmentation masks for each instance, shape (num_instances,
                  height, width).
        r   rU  r&  FrH  )r   rb   r(   )rV  rW  r<  rX  rA   r^   r   rL  r   r   rY  r1   rq   rr   r-  rF   r  r   rM  r   )r   rN  rE  r*   r   rV  rW  r<  rX  r   rZ  r[  batch_masksr\  r]  r   r   rb   r(   rD  ry   s                        r&   "post_process_instance_segmentationz5Sam3ImageProcessor.post_process_instance_segmentationV  s   < ))''
''
!11 &&q)
#L(9Z(GXYY #**,&-557O'/9L !((* ! #&{LAK+4S{T_5`+a 	O'C'&%I%DD\F$KE$KE '*3/u:>!HH//;;*('&+	 < 
 gaj  ^+//

;ENNfuuMN+	O. r%   N)r   g?    rX   N)g)\(?gffffff?r   rX   )r  Tr  r  F)Nr   F)333333?N)re  r   N)5r   r   r   r   valid_kwargsr   BILINEARr  r   
image_meanr   	image_stdr   r   	do_resizer  r  r   do_padpad_sizemask_pad_sizer   r   r   r   r   r   r   r    r   r   r   r   r!   r  r   r   r  r"   r\   r   r  r$  r5  r;  r1   r   r.  r   rS  r^  ra  __classcell__)r   s   @r&   r   r     s   +L!**H'J%IT*D-IIJLN FHM#(@!A #  04
G
G &,
G 12	
G
 

G 
G &*5d?5 
	5& 59-M-M &,-M 	-M
 ,-M c>)*T1-M 12-M 
-M^a^$a j(4/a
 
a )&(;<+//I</I 	/I
 /I t/I )-S	D(8/I (/In # L.d */3je U\\ ell . 6:+0S% 5c?+d2S% 	S%
 %)S% 
NS%j1 1[_`e[fim[m 1l  #+/P P 	P
 5kD(Pr%   r   )r(   r)   )g      4@)r   rc  rd  rX   rb  )F)gffffff?)r   r)   )?r   r
  r   	itertoolsr   typingr   r   r   numpyr)  r1   torch.nn.functionalrq   rr   r,  torchvision.ops.boxesr   image_processing_backendsr
   image_processing_outputsr   image_processing_utilsr   r   image_utilsr   r   r   r   r   r   processing_utilsr   r   utilsr   r   r   PILr   r\   r"   r7   rU   rl   r   rx   r   r   r   r   r   r   r   r    r!   r   r   r   r   r   __all__r$   r%   r&   <module>r|     s  ,    ' '     - ; O A  5 D D |5 N E cf .b,$8S	 8 8 8 %"$782E2E 2E 	2E
 4Z2E %)I$42E 4S	?DI%&2Ej"D# %,,  _c14 ]b#ll;@c?
\\8 d38n    4:>8 U+ U Up  
 r%   