Ë
    ¡[;jÏ&  ã                  ó²   — d dl mZ d dlmZ d dlmZmZ d dlmZ	m
ZmZmZ dddœd„Z	 dd„Zdddœd	„Zdddœd
„Z	 dd„Zdddœd„Z	 dd„Zddœd„Z
ddœd„Zy)é    )Úannotations)Úconv_sequences)Úis_noneÚsetupPandas)Ú_block_similarityÚeditopsÚopcodesÚ
similarityN)Ú	processorÚscore_cutoffc               ó¶   — |� || «      }  ||«      }t        | |«      \  } }t        | «      t        |«      z   }t        | |«      }|d|z  z
  }|�||k  r|S |dz   S )aÝ  
    Calculates the minimum number of insertions and deletions
    required to change one sequence into the other. This is equivalent to the
    Levenshtein distance with a substitution weight of 2.

    Parameters
    ----------
    s1 : Sequence[Hashable]
        First string to compare.
    s2 : Sequence[Hashable]
        Second string to compare.
    processor: callable, optional
        Optional callable that is used to preprocess the strings before
        comparing them. Default is None, which deactivates this behaviour.
    score_cutoff : int, optional
        Maximum distance between s1 and s2, that is
        considered as a result. If the distance is bigger than score_cutoff,
        score_cutoff + 1 is returned instead. Default is None, which deactivates
        this behaviour.

    Returns
    -------
    distance : int
        distance between s1 and s2

    Examples
    --------
    Find the Indel distance between two strings:

    >>> from rapidfuzz.distance import Indel
    >>> Indel.distance("lewenstein", "levenshtein")
    3

    Setting a maximum distance allows the implementation to select
    a more efficient implementation:

    >>> Indel.distance("lewenstein", "levenshtein", score_cutoff=1)
    2

    é   é   )r   ÚlenÚlcs_seq_similarity)Ús1Ús2r   r   ÚmaximumÚlcs_simÚdists          údG:\00. PROJECTS\API\Inventory\templateJSON\kerjaOCR\Lib\site-packages\rapidfuzz/distance/Indel_py.pyÚdistancer      sv   € ð^ ÐÙ�r‹]ˆÙ�r‹]ˆä˜B Ó#�F€BˆÜ�"‹gœ˜B›Ñ€GÜ   RÓ(€GØ�Q˜‘[Ñ €DØ Ð(¨D°LÒ,@ˆ4ÐWÀ|ÐVWÑGWÐWó    c                óv   — t        |«      t        |«      z   }t        | ||«      }|d|z  z
  }|�||k  r|S |dz   S )Nr   r   )r   Úlcs_seq_block_similarity)Úblockr   r   r   r   r   r   s          r   Ú_block_distancer   I   sO   € ô �"‹gœ˜B›Ñ€GÜ& u¨b°"Ó5€GØ�Q˜‘[Ñ €DØ Ð(¨D°LÒ,@ˆ4ÐWÀ|ÐVWÑGWÐWr   c               óª   — |� || «      }  ||«      }t        | |«      \  } }t        | «      t        |«      z   }t        | |«      }||z
  }|�||k\  r|S dS )a  
    Calculates the Indel similarity in the range [max, 0].

    This is calculated as ``(len1 + len2) - distance``.

    Parameters
    ----------
    s1 : Sequence[Hashable]
        First string to compare.
    s2 : Sequence[Hashable]
        Second string to compare.
    processor: callable, optional
        Optional callable that is used to preprocess the strings before
        comparing them. Default is None, which deactivates this behaviour.
    score_cutoff : int, optional
        Maximum distance between s1 and s2, that is
        considered as a result. If the similarity is smaller than score_cutoff,
        0 is returned instead. Default is None, which deactivates
        this behaviour.

    Returns
    -------
    similarity : int
        similarity between s1 and s2
    r   )r   r   r   )r   r   r   r   r   r   Úsims          r   r
   r
   U   sk   € ð@ ÐÙ�r‹]ˆÙ�r‹]ˆä˜B Ó#�F€BˆÜ�"‹gœ˜B›Ñ€GÜ�B˜Ó€DØ
�D‰.€CØÐ'¨3°,Ò+>ˆ3ÐFÀQÐFr   c               óô   — t        «        t        | «      st        |«      ry|� || «      }  ||«      }t        | |«      \  } }t        | «      t        |«      z   }t	        | |«      }|r||z  nd}|�||k  r|S dS )a8  
    Calculates a normalized levenshtein similarity in the range [1, 0].

    This is calculated as ``distance / (len1 + len2)``.

    Parameters
    ----------
    s1 : Sequence[Hashable]
        First string to compare.
    s2 : Sequence[Hashable]
        Second string to compare.
    processor: callable, optional
        Optional callable that is used to preprocess the strings before
        comparing them. Default is None, which deactivates this behaviour.
    score_cutoff : float, optional
        Optional argument for a score threshold as a float between 0 and 1.0.
        For norm_dist > score_cutoff 1.0 is returned instead. Default is 1.0,
        which deactivates this behaviour.

    Returns
    -------
    norm_dist : float
        normalized distance between s1 and s2 as a float between 0 and 1.0
    ç      ð?r   r   )r   r   r   r   r   )r   r   r   r   r   r   Ú	norm_dists          r   Únormalized_distancer#   €   s„   € ô> „MÜˆr„{”g˜b”kØàÐÙ�r‹]ˆÙ�r‹]ˆä˜B Ó#�F€BˆÜ�"‹gœ˜B›Ñ€GÜ�B˜Ó€DÙ")��w’¨q€IØ%Ð-°¸lÒ1Jˆ9ÐRÐQRÐRr   c                ór   — t        |«      t        |«      z   }t        | ||«      }|r||z  nd}|�||k  r|S dS )Nr   r   )r   r   )r   r   r   r   r   r   r"   s          r   Ú_block_normalized_distancer%   ®   sI   € ô �"‹gœ˜B›Ñ€GÜ˜5 " bÓ)€DÙ")��w’¨q€IØ%Ð-°¸lÒ1Jˆ9ÐRÐQRÐRr   c               ó¾   — t        «        t        | «      st        |«      ry|� || «      }  ||«      }t        | |«      \  } }t        | |«      }d|z
  }|�||k\  r|S dS )a�  
    Calculates a normalized indel similarity in the range [0, 1].

    This is calculated as ``1 - normalized_distance``

    Parameters
    ----------
    s1 : Sequence[Hashable]
        First string to compare.
    s2 : Sequence[Hashable]
        Second string to compare.
    processor: callable, optional
        Optional callable that is used to preprocess the strings before
        comparing them. Default is None, which deactivates this behaviour.
    score_cutoff : float, optional
        Optional argument for a score threshold as a float between 0 and 1.0.
        For norm_sim < score_cutoff 0 is returned instead. Default is 0,
        which deactivates this behaviour.

    Returns
    -------
    norm_sim : float
        normalized similarity between s1 and s2 as a float between 0 and 1.0

    Examples
    --------
    Find the normalized Indel similarity between two strings:

    >>> from rapidfuzz.distance import Indel
    >>> Indel.normalized_similarity("lewenstein", "levenshtein")
    0.85714285714285

    Setting a score_cutoff allows the implementation to select
    a more efficient implementation:

    >>> Indel.normalized_similarity("lewenstein", "levenshtein", score_cutoff=0.9)
    0.0

    When a different processor is used s1 and s2 do not have to be strings

    >>> Indel.normalized_similarity(["lewenstein"], ["levenshtein"], processor=lambda s: s[0])
    0.8571428571428572
    g        r!   r   )r   r   r   r#   )r   r   r   r   r"   Únorm_sims         r   Únormalized_similarityr(   º   sn   € ôd „MÜˆr„{”g˜b”kØàÐÙ�r‹]ˆÙ�r‹]ˆä˜B Ó#�F€BˆÜ# B¨Ó+€IØ�Y‰€HØ$Ð,°¸LÒ0Hˆ8ÐPÈqÐPr   c                ó<   — t        | ||«      }d|z
  }|�||k\  r|S dS )Nr!   r   )r%   )r   r   r   r   r"   r'   s         r   Ú_block_normalized_similarityr*   ú   s2   € ô +¨5°"°bÓ9€IØ�Y‰€HØ$Ð,°¸LÒ0Hˆ8ÐPÈqÐPr   ©r   c               ó   — t        | ||¬«      S )ua  
    Return Editops describing how to turn s1 into s2.

    Parameters
    ----------
    s1 : Sequence[Hashable]
        First string to compare.
    s2 : Sequence[Hashable]
        Second string to compare.
    processor: callable, optional
        Optional callable that is used to preprocess the strings before
        comparing them. Default is None, which deactivates this behaviour.

    Returns
    -------
    editops : Editops
        edit operations required to turn s1 into s2

    Notes
    -----
    The alignment is calculated using an algorithm of Heikki HyyrÃ¶, which is
    described [6]_. It has a time complexity and memory usage of ``O([N/64] * M)``.

    References
    ----------
    .. [6] HyyrÃ¶, Heikki. "A Note on Bit-Parallel Alignment Computation."
           Stringology (2004).

    Examples
    --------
    >>> from rapidfuzz.distance import Indel
    >>> for tag, src_pos, dest_pos in Indel.editops("qabxcd", "abycdf"):
    ...    print(("%7s s1[%d] s2[%d]" % (tag, src_pos, dest_pos)))
     delete s1[0] s2[0]
     delete s1[3] s2[2]
     insert s1[4] s2[2]
     insert s1[6] s2[5]
    r+   )Úlcs_seq_editops©r   r   r   s      r   r   r     s   € ôX ˜2˜r¨YÔ7Ð7r   c               ó   — t        | ||¬«      S )u  
    Return Opcodes describing how to turn s1 into s2.

    Parameters
    ----------
    s1 : Sequence[Hashable]
        First string to compare.
    s2 : Sequence[Hashable]
        Second string to compare.
    processor: callable, optional
        Optional callable that is used to preprocess the strings before
        comparing them. Default is None, which deactivates this behaviour.

    Returns
    -------
    opcodes : Opcodes
        edit operations required to turn s1 into s2

    Notes
    -----
    The alignment is calculated using an algorithm of Heikki HyyrÃ¶, which is
    described [7]_. It has a time complexity and memory usage of ``O([N/64] * M)``.

    References
    ----------
    .. [7] HyyrÃ¶, Heikki. "A Note on Bit-Parallel Alignment Computation."
           Stringology (2004).

    Examples
    --------
    >>> from rapidfuzz.distance import Indel

    >>> a = "qabxcd"
    >>> b = "abycdf"
    >>> for tag, i1, i2, j1, j2 in Indel.opcodes(a, b):
    ...    print(("%7s a[%d:%d] (%s) b[%d:%d] (%s)" %
    ...           (tag, i1, i2, a[i1:i2], j1, j2, b[j1:j2])))
     delete a[0:1] (q) b[0:0] ()
      equal a[1:3] (ab) b[0:2] (ab)
     delete a[3:4] (x) b[2:2] ()
     insert a[4:4] () b[2:3] (y)
      equal a[4:6] (cd) b[3:5] (cd)
     insert a[6:6] () b[5:6] (f)
    r+   )Úlcs_seq_opcodesr.   s      r   r	   r	   4  s   € ôd ˜2˜r¨YÔ7Ð7r   )N)Ú
__future__r   Úrapidfuzz._common_pyr   Úrapidfuzz._utilsr   r   Úrapidfuzz.distance.LCSseq_pyr   r   r   r-   r	   r0   r
   r   r   r   r#   r%   r(   r*   © r   r   Ú<module>r6      sŒ   ðõ #å /ß 1÷ó ð Øô7Xð| ó		Xð  Øô(Gð^ Øô+Sðd ó		Sð  Øô=QðH ó	Qð ô	,8ðf õ	28r   