/*
 * Copyright (C) 2025 The Android Open Source Project
 *
 * Licensed under the Apache License, Version 2.0 (the "License");
 * you may not use this file except in compliance with the License.
 * You may obtain a copy of the License at
 *
 *      http://www.apache.org/licenses/LICENSE-2.0
 *
 * Unless required by applicable law or agreed to in writing, software
 * distributed under the License is distributed on an "AS IS" BASIS,
 * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
 * See the License for the specific language governing permissions and
 * limitations under the License.
 */

package android.ext.services.notification;

import static java.lang.String.format;

import android.annotation.SuppressLint;
import android.icu.util.ULocale;
import android.os.Build;
import android.util.ArrayMap;
import android.view.textclassifier.TextClassifier;
import android.view.textclassifier.TextLanguage;

import androidx.annotation.Nullable;
import androidx.annotation.RequiresApi;

import com.android.modules.utils.build.SdkLevel;

import java.util.ArrayList;
import java.util.List;
import java.util.regex.Matcher;
import java.util.regex.Pattern;

/**
 * Class with helper methods related to detecting OTP codes in a text for V only.
 *
 * @deprecated in B+. The OTP detection functionality has been integrated into TextClassifier. This
 * class is intended for use only on V.
 */
@SuppressLint("ObsoleteSdkInt")
@Deprecated
@RequiresApi(Build.VERSION_CODES.VANILLA_ICE_CREAM)
public class LegacyOtpDetector {
    private static final int PATTERN_FLAGS =
            Pattern.DOTALL | Pattern.CASE_INSENSITIVE | Pattern.MULTILINE;

    private static ThreadLocal<Matcher> compileToRegex(String pattern) {
        return ThreadLocal.withInitial(() -> Pattern.compile(pattern, PATTERN_FLAGS).matcher(""));
    }

    private static final float TC_THRESHOLD = 0.6f;

    private static final ArrayMap<String, ThreadLocal<Matcher>> EXTRA_LANG_OTP_REGEX =
            new ArrayMap<>();

    /**
     * A regex matching a line start, open paren, arrow, colon (not proceeded by a digit), open
     * square
     * bracket, equals sign, double or single quote, ideographic char, or a space that is not
     * preceded
     * by a number. It will not consume the start char (meaning START won't be included in the
     * matched
     * string)
     */
    private static final String START =
            "(^|(?<=((^|[^0-9])\\s)|[>(\"'=\\[\\p{IsIdeographic}]|[^0-9]:))";

    /** One single OTP char. A number or alphabetical char (that isn't also ideographic) */
    private static final String OTP_CHAR = "([0-9\\p{IsAlphabetic}&&[^\\p{IsIdeographic}]])";

    /** One OTP char, followed by an optional dash */
    private static final String OTP_CHAR_WITH_DASH = format("(%s-?)", OTP_CHAR);

    /**
     * Performs a lookahead to find a digit after 0 to 7 OTP_CHARs. This ensures that our potential
     * OTP code contains at least one number
     */
    private static final String FIND_DIGIT = format("(?=%s{0,7}\\d)", OTP_CHAR_WITH_DASH);

    /**
     * Matches between 5 and 8 otp chars, with dashes in between. Here, we are assuming an OTP code
     * is
     * 5-8 characters long. The last char must not be followed by a dash
     */
    private static final String OTP_CHARS = format("(%s{4,7}%s)", OTP_CHAR_WITH_DASH, OTP_CHAR);

    /**
     * A regex matching a line end, a space that is not followed by a number, an ideographic char,
     * or
     * a period, close paren, close square bracket, single or double quote, exclamation point,
     * question mark, or comma. It will not consume the end char
     */
    private static final String END = "(?=\\s[^0-9]|$|\\p{IsIdeographic}|[.?!,)'\\]\"])";

    /** A regex matching four digit numerical codes */
    private static final String FOUR_DIGITS = "(\\d{4})";

    private static final String FIVE_TO_EIGHT_ALPHANUM_AT_LEAST_ONE_NUM =
            format("(%s%s)", FIND_DIGIT, OTP_CHARS);

    /** A regex matching two pairs of 3 digits (ex "123 456") */
    private static final String SIX_DIGITS_WITH_SPACE = "(\\d{3}\\s\\d{3})";

    /**
     * Combining the regular expressions above, we get an OTP regex: 1. search for START, THEN 2.
     * match ONE of a. alphanumeric sequence, at least one number, length 5-8, with optional dashes
     * b.
     * 4 numbers in a row c. pair of 3 digit codes separated by a space THEN 3. search for END Ex:
     * "6454", " 345 678.", "[YDT-456]"
     */
    private static final String ALL_OTP =
            format(
                    "%s(%s|%s|%s)%s",
                    START, FIVE_TO_EIGHT_ALPHANUM_AT_LEAST_ONE_NUM, FOUR_DIGITS,
                    SIX_DIGITS_WITH_SPACE, END);

    private static final ThreadLocal<Matcher> OTP_REGEX = compileToRegex(ALL_OTP);

    /**
     * A Date regular expression. Looks for dates with the month, day, and year separated by
     * dashes.
     * Handles one and two digit months and days, and four or two-digit years. It makes the
     * following
     * assumptions: Dates and months will never be higher than 39 If a four digit year is used, the
     * leading digit will be 1 or 2
     */
    private static final String DATE_WITH_DASHES = "([0-3]?\\d-[0-3]?\\d-([12]\\d)?\\d\\d)";

    /**
     * matches a ten digit phone number, when the area code is separated by a space or dash.
     * Supports
     * optional parentheses around the area code, and an optional dash or space in between the rest
     * of
     * the numbers. This format registers as an otp match due to the space between the area code
     * and
     * the rest, but shouldn't.
     */
    private static final String PHONE_WITH_SPACE = "(\\(?\\d{3}\\)?(-|\\s)?\\d{3}(-|\\s)?\\d{4})";

    /**
     * A combination of common false positives. These matches are expected to be longer than (or
     * equal
     * in length to) otp matches, and are always run, even if we have a language specific regex
     */
    private static final ThreadLocal<Matcher> FALSE_POSITIVE_LONGER_REGEX =
            compileToRegex(format("%s(%s|%s)%s", START, DATE_WITH_DASHES, PHONE_WITH_SPACE, END));

    /** A regex matching the common years of 19xx and 20xx. Used for false positive reduction */
    private static final String COMMON_YEARS = format("%s((19|20)\\d\\d)%s", START, END);

    /**
     * A regex matching three lower case letters. Used for false positive reduction, as no known
     * OTPs
     * have 3 lowercase letters in sequence.
     */
    private static final String THREE_LOWERCASE = "(\\p{Ll}{3})";

    /**
     * A combination of common false positives. Run in cases where we don't have a language specific
     * regular expression. These matches are expect to be shorter than (or equal in length to) otp
     * matches
     */
    private static final ThreadLocal<Matcher> FALSE_POSITIVE_SHORTER_REGEX =
            compileToRegex(format("%s|%s", COMMON_YEARS, THREE_LOWERCASE));

    /**
     * A list of regular expressions representing words found in an OTP context (non case sensitive)
     * Note: TAN is short for Transaction Authentication Number
     */
    private static final String[] ENGLISH_CONTEXT_WORDS =
            new String[]{
                    "pin",
                    "pass[-\\s]?(code|word)",
                    "TAN",
                    "otp",
                    "2fa",
                    "(two|2)[-\\s]?factor",
                    "log[-\\s]?in",
                    "auth(enticat(e|ion))?",
                    "code",
                    "secret",
                    "verif(y|ication)",
                    "one(\\s|-)?time",
                    "access",
                    "validat(e|ion)"
            };

    /**
     * Creates a regular expression to match any of a series of individual words, case insensitive.
     * It
     * also verifies the position of the word, relative to the OTP match
     */
    private static ThreadLocal<Matcher> createDictionaryRegex(String[] words) {
        StringBuilder regex = new StringBuilder("(");
        for (int i = 0; i < words.length; i++) {
            String boundedWord = "\\b" + words[i] + "\\b";
            regex.append(boundedWord);
            if (i != words.length - 1) {
                regex.append("|");
            }
        }
        regex.append(")");
        return compileToRegex(regex.toString());
    }

    static {
        EXTRA_LANG_OTP_REGEX.put(
                ULocale.ENGLISH.toLanguageTag(), createDictionaryRegex(ENGLISH_CONTEXT_WORDS));
    }

    /**
     * Checks if a string of text might contain an OTP, based on several regular expressions, and
     * potentially using a textClassifier to eliminate false positives
     *
     * @param sensitiveText          The text whose content should be checked
     * @param checkForFalsePositives If true, will ensure the content does not match the date
     *                               regex.
     *                               If a TextClassifier is provided, it will then try to find a
     *                               language specific regex. If it
     *                               is successful, it will use that regex to check for false
     *                               positives. If it is not, it will
     *                               use the TextClassifier (if provided), plus the year and three
     *                               lowercase regexes to remove
     *                               possible false positives.
     * @param tc                     If non null, the provided TextClassifier will be used to find
     *                               the language of the
     *                               text, and look for a language-specific regex for it. If
     *                               checkForFalsePositives is true will
     *                               also use the classifier to find flight codes and addresses.
     * @param language               If non null, then the TextClassifier (if provided), will not
     *                               perform language
     *                               id, and the system will assume the text is in the specified
     *                               language
     * @return True if we believe an OTP is in the message, false otherwise.
     */
    public static boolean containsOtp(
            String sensitiveText,
            boolean checkForFalsePositives,
            @Nullable TextClassifier tc,
            @Nullable ULocale language) {
        if (sensitiveText == null || !SdkLevel.isAtLeastV()) {
            return false;
        }
        Matcher otpMatcher = OTP_REGEX.get();
        otpMatcher.reset(sensitiveText);
        boolean otpMatch = otpMatcher.find();
        if (!checkForFalsePositives || !otpMatch) {
            return otpMatch;
        }

        if (allOtpMatchesAreFalsePositives(sensitiveText, FALSE_POSITIVE_LONGER_REGEX.get(),
                true)) {
            return false;
        }

        if (tc != null || language != null) {
            if (language == null) {
                language = getLanguageWithRegex(sensitiveText, tc);
            }
            Matcher languageSpecificMatcher =
                    language != null ? EXTRA_LANG_OTP_REGEX.get(language.toLanguageTag()).get()
                            : null;
            if (languageSpecificMatcher != null) {
                languageSpecificMatcher.reset(sensitiveText);
                // Only use the language-specific regex for false positives
                return languageSpecificMatcher.find();
            }
            // Only check for OTPs when there is a language specific matcher
            return false;
        }

        return !allOtpMatchesAreFalsePositives(
                sensitiveText, FALSE_POSITIVE_SHORTER_REGEX.get(), false);
    }

    /**
     * Checks that a given text has at least one match for one regex, that doesn't match another
     *
     * @param text                      The full text to check
     * @param falsePositiveRegex        A regex that should not match the OTP regex (for at least
     *                                  one match
     *                                  found by the OTP regex). The false positive regex matches
     *                                  may be longer or shorter than the
     *                                  OTP matches.
     * @param fpMatchesAreLongerThanOtp Whether the false positives are longer than the otp
     *                                  matches.
     *                                  If true, this method will search the whole text for false
     *                                  positives, and verify at least
     *                                  one OTP match is not contained by any of the false
     *                                  positives. If false, then this method
     *                                  will search individual OTP matches for false positives, and
     *                                  will verify at least one OTP
     *                                  match doesn't contain a false positive.
     * @return true, if all matches found by OTP_REGEX are contained in, or themselves contain a
     * match
     * to falsePositiveRegex, or there are no OTP matches, false otherwise.
     */
    private static boolean allOtpMatchesAreFalsePositives(
            String text, Matcher falsePositiveRegex, boolean fpMatchesAreLongerThanOtp) {
        List<String> falsePositives = new ArrayList<>();
        if (fpMatchesAreLongerThanOtp) {
            // if the false positives are longer than the otp, search for them in the whole text
            falsePositives = getAllMatches(text, falsePositiveRegex);
        }
        List<String> otpMatches = getAllMatches(text, OTP_REGEX.get());
        for (String otpMatch : otpMatches) {
            boolean otpMatchContainsNoFp = true;
            boolean noFpContainsOtpMatch = true;
            if (!fpMatchesAreLongerThanOtp) {
                // if the false positives are shorter than the otp, search for them in the otp match
                falsePositives = getAllMatches(otpMatch, falsePositiveRegex);
            }
            for (String falsePositive : falsePositives) {
                otpMatchContainsNoFp =
                        fpMatchesAreLongerThanOtp
                                || (otpMatchContainsNoFp && !otpMatch.contains(falsePositive));
                noFpContainsOtpMatch =
                        !fpMatchesAreLongerThanOtp
                                || (noFpContainsOtpMatch && !falsePositive.contains(otpMatch));
            }
            if (otpMatchContainsNoFp && noFpContainsOtpMatch) {
                return false;
            }
        }
        return true;
    }

    private static List<String> getAllMatches(String text, Matcher regex) {
        ArrayList<String> matches = new ArrayList<>();
        regex.reset(text);
        while (regex.find()) {
            matches.add(regex.group());
        }
        return matches;
    }

  /**
   * Tries to determine the language of the given text. Will return the language with the highest
   * confidence score that meets the minimum threshold, and has a language-specific regex, null
   * otherwise.
   *
   * @param text The text to analyze for language detection
   * @param tc   The {@link TextClassifier} to use for language detection. Can be null
   * @return The {@link ULocale} of the detected language, or null if no language meets the criteria
   */
    @Nullable
    public static ULocale getLanguageWithRegex(String text, @Nullable TextClassifier tc) {
        if (tc == null) {
            return null;
        }

        float highestConfidence = 0;
        ULocale highestConfidenceLocale = null;
        TextLanguage.Request langRequest = new TextLanguage.Request.Builder(text).build();
        TextLanguage lang = tc.detectLanguage(langRequest);
        for (int i = 0; i < lang.getLocaleHypothesisCount(); i++) {
            ULocale locale = lang.getLocale(i);
            float confidence = lang.getConfidenceScore(locale);
            if (confidence >= TC_THRESHOLD
                    && confidence >= highestConfidence
                    && EXTRA_LANG_OTP_REGEX.containsKey(locale.toLanguageTag())) {
                highestConfidence = confidence;
                highestConfidenceLocale = locale;
            }
        }
        return highestConfidenceLocale;
    }

    private LegacyOtpDetector() {
    }
}
