File size: 4,877 Bytes
ac704e3
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
"""
features.py 
Extracting and Cleaning Data from the NVD API 
"""

import logging
from datetime import datetime

# We are in a .py so we dont't use 'print' but 'logger'
logger = logging.getLogger(__name__)


def extract_features(cve: dict) -> dict:
    """
    Function to extract and clean data from the NVD API

    Args: 
        cve: The NVD API returns a dictionary containing the raw CVE. 

    Returns: 
        Non-nested dictionary ready for model training
        returns 'None' if the CVE dictionary is invalid.
    """ 
    cve_id = cve["cve"]["id"]
    published = cve["cve"]["published"]
    # '.get' to provide a fallback in case the data is missing 
    source_identifier = cve["cve"].get("sourceIdentifier")
    vulnerability_status = cve["cve"].get("vulnStatus")
    description = cve["cve"].get("descriptions", [])
    # Our model cannot read a description
    # But a detailed—and therefore lengthy—description indicates a well-documented vulnerability, which makes it more vulnerable
    description_words = " ".join([d.get("value", "") for d in description if d.get("lang") == "en"])
    description_length = len(description_words)
    reference = cve["cve"].get("references", [])

    # metrics: 
    # security, if the metrics are missing: 
    metrics = cve["cve"].get("metrics", {})
    weaknesses = cve["cve"].get("weaknesses", [])
    cvss_v31 = metrics.get("cvssMetricV31", [])  # "cvssMetricV31" is a list
    cvss_data = cvss_v31[0]["cvssData"] if cvss_v31 else {}
    configurations_data = cve["cve"].get("configurations", [])


    base_score = cvss_data.get("baseScore")
    attack_vector = cvss_data.get("attackVector")
    attack_complexity = cvss_data.get("attackComplexity")
    privileges_required = cvss_data.get("privilegesRequired")
    user_interaction = cvss_data.get("userInteraction")
    confidentiality_impact = cvss_data.get("confidentialityImpact")
    integrity_impact = cvss_data.get("integrityImpact")
    availability_impact = cvss_data.get("availabilityImpact")
    # We are adding back these two metrics, which had been excluded due to probable collinearity
    exploitability_score = cvss_v31[0].get("exploitabilityScore") if cvss_v31 else None
    impact_score = cvss_v31[0].get("impactScore") if cvss_v31 else None

    # addition of content type to weakness
    # empty dictionary if there is no dictionary
    weakness_data = weaknesses[0] if weaknesses else {}
    description_list = weakness_data.get("description", [])
    # Empty dictionary if no elements
    description_data = description_list[0] if description_list else {}
    cwe = description_data.get("value")

    # Adding scope
    scope = cvss_data.get("scope")

    # Adding configuration 
    # Directory tree up to "criteria"
    config_0 = configurations_data[0] if configurations_data else {}
    nodes = config_0.get("nodes", [])
    node_0 = nodes[0] if nodes else {}
    cpe_matches = node_0.get("cpeMatch", [])
    cpe_0 = cpe_matches[0] if cpe_matches else {}
    criteria = cpe_0.get("criteria")
    criteria_parts = criteria.split(":") if criteria else []
    # If there is an application, hardware, OS 
    component_type = criteria_parts[2] if len(criteria_parts) > 2 else None
    # The vendor
    vendor = criteria_parts[3] if len(criteria_parts) > 3 else None

    # Date
    publication_date = datetime.strptime(published, "%Y-%m-%dT%H:%M:%S.%f")  # %f for milliseconds
    age_in_days = (datetime.now() - publication_date).days

    reference_count = len(reference)
    # We're looking for websites that contain code capable of exploiting vulnerabilities, such as exploit-db.com or GitHub
    # If “exploits” are published on these sites, the likelihood of being hacked is much higher
    has_exploit_reference = any (
        "exploit" in ref.get("url", "") or "github" in ref.get("url", "")
        for ref in reference
    )

    return { 
        "cve_id": cve_id,
        "published": published,
        "source_identifier": source_identifier,
        "vulnerability_status": vulnerability_status,
        "description_length" : description_length,
        "description_words" : description_words,
        "base_score": base_score,
        "attack_vector": attack_vector,
        "attack_complexity": attack_complexity,
        "privileges_required": privileges_required,
        "user_interaction": user_interaction,
        "confidentiality_impact": confidentiality_impact,
        "integrity_impact": integrity_impact,
        "availability_impact": availability_impact,
        "age_in_days": age_in_days,
        "reference_count": reference_count,
        "has_exploit_reference": has_exploit_reference,
        "cwe": cwe,
        "scope": scope,
        "component_type": component_type,
        "vendor": vendor,
        "exploitability_score": exploitability_score,
        "impact_score": impact_score
    }