File size: 4,106 Bytes
29f25be
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
"""Summarize exported Instruments sysmon-process XML for Vime processes."""
import argparse
import json
import statistics
import xml.etree.ElementTree as ET
from pathlib import Path


def summarize(source):
    root = ET.parse(source).getroot()
    columns = [c.findtext('mnemonic') for c in root.find('.//schema').findall('col')]
    references = {e.attrib['id']: e for e in root.iter() if 'id' in e.attrib}

    def resolve(element):
        while 'ref' in element.attrib:
            element = references[element.attrib['ref']]
        return element

    def number(element):
        element = resolve(element)
        return None if element.tag == 'sentinel' or element.text is None else float(element.text)

    processes = {}
    for row in root.findall('.//row'):
        values = dict(zip(columns, row))
        label = resolve(values['process']).get('fmt', '')
        if not (label.startswith('VimeKeyboard (') or label.startswith('Vime (')):
            continue
        footprint = number(values['memory-physical-footprint'])
        if footprint is None:
            continue
        item = {'seconds': number(values['time']) / 1e9,
                'physical_mib': footprint / 1048576,
                'resident_mib': number(values['memory-resident-size']) / 1048576,
                'private_resident_mib': number(values['memory-real-private']) / 1048576,
                'shared_resident_mib': number(values['memory-real-shared']) / 1048576,
                'threads': number(values['thread-count']),
                'recently_died': bool(number(values['recently-died']))}
        processes.setdefault(label, []).append(item)
    summary = {}
    for label, samples in processes.items():
        samples.sort(key=lambda s: s['seconds'])
        peak = max(samples, key=lambda s: s['physical_mib'])
        windows = []
        for start in range(0, int(samples[-1]['seconds']) + 1, 30):
            selected = [s['physical_mib'] for s in samples if start <= s['seconds'] < start + 30]
            if selected:
                windows.append({'start_seconds': start, 'end_seconds': start + 30,
                                'samples': len(selected), 'min_mib': min(selected),
                                'median_mib': statistics.median(selected), 'max_mib': max(selected)})
        intervals = [b['seconds'] - a['seconds'] for a, b in zip(samples, samples[1:])]
        summary[label] = {'sample_count': len(samples), 'first': samples[0], 'last': samples[-1],
                          'sampled_peak': peak, 'sampled_min_mib': min(s['physical_mib'] for s in samples),
                          'resident_metrics': {key: {'first_mib': samples[0][key],
                                                     'sampled_peak_mib': max(s[key] for s in samples),
                                                     'last_mib': samples[-1][key]}
                                               for key in ('private_resident_mib', 'shared_resident_mib', 'resident_mib')},
                          'median_interval_seconds': statistics.median(intervals) if intervals else None,
                          'recently_died_samples': sum(s['recently_died'] for s in samples),
                          'windows': windows, 'samples': samples}
    return {'source': str(source), 'metric': 'sysmon physical footprint; sampled peak, not lifetime high-water mark',
            'processes': summary}


if __name__ == '__main__':
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument('--input', type=Path, required=True)
    parser.add_argument('--output', type=Path, required=True)
    args = parser.parse_args()
    if args.output.exists():
        parser.error("Output exists; choose a new report path.")
    report = summarize(args.input)
    args.output.parent.mkdir(parents=True, exist_ok=True)
    with args.output.open('x', encoding='utf-8') as stream:
        stream.write(json.dumps(report, indent=2) + '\n')
    print(json.dumps({k: {key: value for key, value in v.items() if key != 'samples'}
                      for k, v in report['processes'].items()}, indent=2))