Ken2707's picture
Upload features.py
6a73dfd verified
Raw
History Blame Contribute Delete
3.6 kB
import re
import pandas as pd
from urllib.parse import urlparse
from tld import get_tld
def abnormal_url(url):
hostname = urlparse(url).hostname
hostname = str(hostname)
match = re.search(hostname, url)
return 1 if match else 0
def httpSecure(url):
htp = urlparse(url).scheme
match = str(htp)
return 1 if match == 'https' else 0
def digit_count(url):
return sum(1 for i in url if i.isnumeric())
def letter_count(url):
return sum(1 for i in url if i.isalpha())
def Shortening_Service(url):
match = re.search(r'bit\.ly|goo\.gl|shorte\.st|go2l\.ink|x\.co|ow\.ly|t\.co|tinyurl|tr\.im|is\.gd|cli\.gs|'
r'yfrog\.com|migre\.me|ff\.im|tiny\.cc|url4\.eu|twit\.ac|su\.pr|twurl\.nl|snipurl\.com|'
r'short\.to|BudURL\.com|ping\.fm|post\.ly|Just\.as|bkite\.com|snipr\.com|fic\.kr|loopt\.us|'
r'doiop\.com|short\.ie|kl\.am|wp\.me|rubyurl\.com|om\.ly|to\.ly|bit\.do|t\.co|lnkd\.in|'
r'db\.tt|qr\.ae|adf\.ly|goo\.gl|bitly\.com|cur\.lv|tinyurl\.com|ow\.ly|bit\.ly|ity\.im|'
r'q\.gs|is\.gd|po\.st|bc\.vc|twitthis\.com|u\.to|j\.mp|buzurl\.com|cutt\.us|u\.bb|yourls\.org|'
r'x\.co|prettylinkpro\.com|scrnch\.me|filoops\.info|vzturl\.com|qr\.net|1url\.com|tweez\.me|v\.gd|'
r'tr\.im|link\.zip\.net', url)
return 1 if match else 0
def ip_address_detect(url):
match = re.search(
r'(([01]?\d\d?|2[0-4]\d|25[0-5])\.([01]?\d\d?|2[0-4]\d|25[0-5])\.([01]?\d\d?|2[0-4]\d|25[0-5])\.'
r'([01]?\d\d?|2[0-4]\d|25[0-5])\/)|'
r'(([01]?\d\d?|2[0-4]\d|25[0-5])\.([01]?\d\d?|2[0-4]\d|25[0-5])\.([01]?\d\d?|2[0-4]\d|25[0-5])\.'
r'([01]?\d\d?|2[0-4]\d|25[0-5])\/)|'
r'((0x[0-9a-fA-F]{1,2})\.(0x[0-9a-fA-F]{1,2})\.(0x[0-9a-fA-F]{1,2})\.(0x[0-9a-fA-F]{1,2})\/)|'
r'(?:[a-fA-F0-9]{1,4}:){7}[a-fA-F0-9]{1,4}|'
r'([0-9]+(?:\.[0-9]+){3}:[0-9]+)|'
r'((?:(?:\d|[01]?\d\d|2[0-4]\d|25[0-5])\.){3}(?:25[0-5]|2[0-4]\d|[01]?\d\d|\d)(?:\/\d{1,2})?)', url)
return 1 if match else 0
def num_subdomains(url):
try:
parsed = urlparse(url)
netloc = parsed.netloc or parsed.path.split('/')[0]
if ':' in netloc:
netloc = netloc.split(':')[0]
parts = netloc.split('.')
return 0 if len(parts) <= 2 else len(parts) - 2
except:
return 0
def is_suspicious_tld(url):
try:
res = get_tld(url, as_object=True, fail_silently=False, fix_protocol=True)
tld = res.tld
except:
tld = 'unknown'
suspicious_tlds = ['.tk', '.xyz', '.top', '.ml', '.ga', '.cf', '.gq']
return 1 if tld in suspicious_tlds else 0
def extract_features(url):
"""Mengekstrak 22 fitur leksikal dari URL untuk model ML"""
url = url.replace('www.', '')
features = {}
features['url_len'] = len(str(url))
special_chars = ['@','?','-','=','.','#','%','+','$','!','*',',','//']
for char in special_chars:
features[char] = url.count(char)
features['abnormal_url'] = abnormal_url(url)
features['https'] = httpSecure(url)
features['digits'] = digit_count(url)
features['letters'] = letter_count(url)
features['Shortening_Service'] = Shortening_Service(url)
features['ip_address_detect'] = ip_address_detect(url)
features['num_subdomains'] = num_subdomains(url)
features['suspicious_tld'] = is_suspicious_tld(url)
return pd.DataFrame([features])