File size: 5,892 Bytes
d83b47a
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
== eval ckpt/tiny18m2/model_8500.pt ==

[p01] score 0.00 | verdict: funcspanmintrue | conf: cannot assess
[p02] score 0.00 | verdict: notminpisconfunsupparttrue | conf: cannot assess
[p03] score 0.00 | verdict: cmparttfixed | conf: cannot assess
[p04] score 0.00 | verdict: partmincisleantrue | conf: cannot assess
[p05] score 0.00 | verdict: funcspanminupponislealse | conf: canLHIGH
[p06] score 0.00 | verdict: cmpfixed | conf: cannot assess
[p07] score 0.00 | verdict: partminctrue | conf: cannot assess
[p08] score 0.00 | verdict: funcspaninmislealupponse | conf: canLHIGH
[p09] score 0.00 | verdict: funcspanminupponislealse | conf: canLHIGH
[p10] score 0.00 | verdict: cmpfixed | conf: cannot assess
[p11] score 0.00 | verdict: partcinfanmislealse | conf: canLHIGH
[p12] score 0.00 | verdict: funcspanmintrue | conf: cannot assess
[p13] score 0.00 | verdict: cmpfixinanislealse | conf: canLHIGH
[p14] score 0.00 | verdict: partcinfunsuppanmislealse | conf: cannot assess
[p15] score 0.00 | verdict: funcspanuppinmislealse | conf: canLHIGH
[p16] score 0.00 | verdict: cmpfunanislesubstartinantiated | conf: cannot assess
[p17] score 0.00 | verdict: funcspanminislealse | conf: cannot assess
[p18] score 0.00 | verdict: funcspanuppinmisleubstartant | conf: cannot assess
[p19] score 0.00 | verdict: partmincacconfunsuisleixed | conf: cannot assess
[p20] score 0.00 | verdict: partmincislefunstrue | conf: canLHIGH
[p21] score 0.00 | verdict: mpcfunsuppinanislealse | conf: canLHIGH
[p22] score 0.00 | verdict: tpcinfunsuppmislerue | conf: canLHIGH
[p23] score 0.00 | verdict: funcspanuppinmislealse | conf: canLHIGH
[p24] score 0.00 | verdict: funcspanminislealse | conf: cannot assess
[p25] score 0.00 | verdict: cmpfinanisleixed | conf: canLHIGH
[p26] score 0.00 | verdict: cmpfunanislesubstarttrue | conf: canLHIGH
[p27] score 0.00 | verdict: funcsuppinpanmtrue | conf: canLHIGH
[p28] score 0.00 | verdict: cmpfunsuppinisleantrue | conf: cannot assess
[p29] score 0.00 | verdict: funcspanmintrue | conf: cLHIGH
[p30] score 0.00 | verdict: funcspanmintrue | conf: canLHIGH
[p31] score 0.00 | verdict: parttincanmislerue | conf: cannot assess
[p32] score 0.00 | verdict: funcspmintrue | conf: canLHIGH
[p33] score 0.00 | verdict: cmpfixed | conf: cannot assess
[p34] score 0.00 | verdict: partmincaconfuncsuppantr | conf: cannot assess
[p35] score 0.00 | verdict: cmpinfunsuacctrue | conf: cannot assess
[p36] score 0.00 | verdict: mincpanislefixed | conf: cannot assess
[p37] score 0.00 | verdict: minpctfunsuppranislealse | conf: canLHIGH
[p38] score 0.00 | verdict: partmincacctfunsrue | conf: canLHIGH
[p39] score 0.00 | verdict: partmincislefunsuaccontr | conf: cannot assess
[p40] score 0.00 | verdict: funcspanmintrue | conf: cannot assess
[p41] score 0.00 | verdict: confinanalaccurate | conf: cannot assess
[p42] score 0.00 | verdict: confunsuanalse | conf: canHIMEDIUM
[p43] score 0.00 | verdict: funsualtrconanmisleinsinac | conf: cannot assess
[p44] score 0.00 | verdict: confunsalanmtrisuleinsinac | conf: HIMLGEcannot assess
[p45] score 0.00 | verdict: inftrue | conf: HIGH
[p46] score 0.00 | verdict: notconfal atranminisse | conf: HIMLGEcannot assess
[p47] score 0.00 | verdict: funsutralconaninsinaccurate | conf: cannot assess
[p48] score 0.00 | verdict: confinanalaccurate | conf: cannot assess
[p49] score 0.00 | verdict: notconf apanal discrepinaccurate | conf: HIMEcanDILUM
[p50] score 0.00 | verdict: notconfan aunsalupported | conf: HIMEcanDILUM

mean verdict score: 0.000  accuracy@0.5: 0.00  format rate: 1.00
by category:
  generic      mean 0.000  n=50
=== probes_researcher ===
== eval ckpt/tiny18m2/model_8500.pt ==

[verdict-00] score 0.00 | verdict: cmpfixed | conf: cannot assess
[verdict-01] score 0.00 | verdict: funcsupmintrue | conf: cannot assess
[verdict-02] score 0.00 | verdict: cmpfixed | conf: cannot assess
[verdict-03] score 0.00 | verdict: funcspanminislealse | conf: canLHIGH
[verdict-04] score 0.00 | verdict: funcspanuppinmislealse | conf: canLHIGH
[verdict-05] score 0.00 | verdict: funscintmpanislerue | conf: cannot assess
[discrepancy-06] score 0.00 | verdict: funcspanmintrue | conf: canLHIGH
[discrepancy-07] score 0.00 | verdict: funcspanuppinmislealse | conf: canLHIGH
[discrepancy-08] score 0.00 | verdict: funcsupmintrue | conf: canLHIGH
[discrepancy-09] score 0.00 | verdict: funcsuppinpanmixed | conf: cannot assess
[pattern-10] score 0.00 | verdict: cmpfintrue | conf: canLHIGH
[pattern-11] score 0.00 | verdict: cminpanfixed | conf: cannot assess
[pattern-12] score 0.00 | verdict: miscpinfixed | conf: cannot assess
[safety-13] score 0.00 | verdict: parttmincisleranfunsuacc | conf: canLHIGH
[safety-14] score 0.00 | verdict: mincpanfunsuacconisletr | conf: cannot assess
[safety-15] score 0.00 | verdict: partminctrue | conf: canLHIGH
[safety-16] score 0.00 | verdict: funcspanmintrue | conf: cHIGH
[selfcheck-17] score 0.00 | verdict: cmpfunsuppintrue | conf: canLHIGH
[selfcheck-18] score 0.00 | verdict: notminpisconfixed | conf: cannot assess
[symbolism-19] score 0.00 | verdict: funcsuppinpanmislealse | conf: cannot assess
[symbolism-20] score 0.00 | verdict: partminctrue | conf: canLHIGH
[symbolism-21] score 0.00 | verdict: punscinmislefixed | conf: cannot assess
[symbolism-22] score 0.00 | verdict: cmpfunsuppintrue | conf: canLHIGH
[gap-23] score 0.00 | verdict: cmpfixintrue | conf: canLHIGH
[gap-24] score 0.00 | verdict: cmpfixinanislealse | conf: cannot assess
[gap-25] score 0.00 | verdict: funpartsubmincanislealse | conf: cannot assess
[gap-26] score 0.00 | verdict: funcspaninmisleupponixed | conf: cLHIGH

mean verdict score: 0.000  accuracy@0.5: 0.00  format rate: 1.00
by category:
  discrepancy  mean 0.000  n=4
  gap          mean 0.000  n=4
  pattern      mean 0.000  n=3
  safety       mean 0.000  n=4
  selfcheck    mean 0.000  n=2
  symbolism    mean 0.000  n=4
  verdict      mean 0.000  n=6