Spaces:
Build error
Build error
| import re | |
| import pandas as pd | |
| from urllib.parse import urlparse | |
| from tld import get_tld | |
| def abnormal_url(url): | |
| hostname = urlparse(url).hostname | |
| hostname = str(hostname) | |
| match = re.search(hostname, url) | |
| return 1 if match else 0 | |
| def httpSecure(url): | |
| htp = urlparse(url).scheme | |
| match = str(htp) | |
| return 1 if match == 'https' else 0 | |
| def digit_count(url): | |
| return sum(1 for i in url if i.isnumeric()) | |
| def letter_count(url): | |
| return sum(1 for i in url if i.isalpha()) | |
| def Shortening_Service(url): | |
| match = re.search(r'bit\.ly|goo\.gl|shorte\.st|go2l\.ink|x\.co|ow\.ly|t\.co|tinyurl|tr\.im|is\.gd|cli\.gs|' | |
| r'yfrog\.com|migre\.me|ff\.im|tiny\.cc|url4\.eu|twit\.ac|su\.pr|twurl\.nl|snipurl\.com|' | |
| r'short\.to|BudURL\.com|ping\.fm|post\.ly|Just\.as|bkite\.com|snipr\.com|fic\.kr|loopt\.us|' | |
| r'doiop\.com|short\.ie|kl\.am|wp\.me|rubyurl\.com|om\.ly|to\.ly|bit\.do|t\.co|lnkd\.in|' | |
| r'db\.tt|qr\.ae|adf\.ly|goo\.gl|bitly\.com|cur\.lv|tinyurl\.com|ow\.ly|bit\.ly|ity\.im|' | |
| r'q\.gs|is\.gd|po\.st|bc\.vc|twitthis\.com|u\.to|j\.mp|buzurl\.com|cutt\.us|u\.bb|yourls\.org|' | |
| r'x\.co|prettylinkpro\.com|scrnch\.me|filoops\.info|vzturl\.com|qr\.net|1url\.com|tweez\.me|v\.gd|' | |
| r'tr\.im|link\.zip\.net', url) | |
| return 1 if match else 0 | |
| def ip_address_detect(url): | |
| match = re.search( | |
| r'(([01]?\d\d?|2[0-4]\d|25[0-5])\.([01]?\d\d?|2[0-4]\d|25[0-5])\.([01]?\d\d?|2[0-4]\d|25[0-5])\.' | |
| r'([01]?\d\d?|2[0-4]\d|25[0-5])\/)|' | |
| r'(([01]?\d\d?|2[0-4]\d|25[0-5])\.([01]?\d\d?|2[0-4]\d|25[0-5])\.([01]?\d\d?|2[0-4]\d|25[0-5])\.' | |
| r'([01]?\d\d?|2[0-4]\d|25[0-5])\/)|' | |
| r'((0x[0-9a-fA-F]{1,2})\.(0x[0-9a-fA-F]{1,2})\.(0x[0-9a-fA-F]{1,2})\.(0x[0-9a-fA-F]{1,2})\/)|' | |
| r'(?:[a-fA-F0-9]{1,4}:){7}[a-fA-F0-9]{1,4}|' | |
| r'([0-9]+(?:\.[0-9]+){3}:[0-9]+)|' | |
| r'((?:(?:\d|[01]?\d\d|2[0-4]\d|25[0-5])\.){3}(?:25[0-5]|2[0-4]\d|[01]?\d\d|\d)(?:\/\d{1,2})?)', url) | |
| return 1 if match else 0 | |
| def num_subdomains(url): | |
| try: | |
| parsed = urlparse(url) | |
| netloc = parsed.netloc or parsed.path.split('/')[0] | |
| if ':' in netloc: | |
| netloc = netloc.split(':')[0] | |
| parts = netloc.split('.') | |
| return 0 if len(parts) <= 2 else len(parts) - 2 | |
| except: | |
| return 0 | |
| def is_suspicious_tld(url): | |
| try: | |
| res = get_tld(url, as_object=True, fail_silently=False, fix_protocol=True) | |
| tld = res.tld | |
| except: | |
| tld = 'unknown' | |
| suspicious_tlds = ['.tk', '.xyz', '.top', '.ml', '.ga', '.cf', '.gq'] | |
| return 1 if tld in suspicious_tlds else 0 | |
| def extract_features(url): | |
| """Mengekstrak 22 fitur leksikal dari URL untuk model ML""" | |
| url = url.replace('www.', '') | |
| features = {} | |
| features['url_len'] = len(str(url)) | |
| special_chars = ['@','?','-','=','.','#','%','+','$','!','*',',','//'] | |
| for char in special_chars: | |
| features[char] = url.count(char) | |
| features['abnormal_url'] = abnormal_url(url) | |
| features['https'] = httpSecure(url) | |
| features['digits'] = digit_count(url) | |
| features['letters'] = letter_count(url) | |
| features['Shortening_Service'] = Shortening_Service(url) | |
| features['ip_address_detect'] = ip_address_detect(url) | |
| features['num_subdomains'] = num_subdomains(url) | |
| features['suspicious_tld'] = is_suspicious_tld(url) | |
| return pd.DataFrame([features]) |