izm1chael commited on
Commit
b9078ec
·
verified ·
1 Parent(s): 1e3d528

Publish Layerfault deep challenge LF-CH-TOKX-0014

Browse files
CHALLENGE_ORACLE.json ADDED
@@ -0,0 +1,28 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "attack_surface": [
3
+ "tokenizer-processor"
4
+ ],
5
+ "control_type": "negative",
6
+ "corpus_id": "LF-CH-TOKX-0014",
7
+ "difficulty": "compound",
8
+ "expected_decision": "PASS",
9
+ "ground_truth": "synthetic-generated",
10
+ "layerfault_rule_expectations": {
11
+ "candidate_rules": [],
12
+ "expected_rules": [],
13
+ "must_not_rules": []
14
+ },
15
+ "oracle_ids": [
16
+ "LF-ORACLE-TOKX-0014"
17
+ ],
18
+ "related_cases": [],
19
+ "repo_name": "tokenizer-security-words-control",
20
+ "safe_fixture": true,
21
+ "schema": 2,
22
+ "severity": "informational",
23
+ "techniques": [
24
+ "tokenizer",
25
+ "control"
26
+ ],
27
+ "transformations": []
28
+ }
FIXTURE_LAYOUT.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "artifacts": [
3
+ "tokenizer_config.json",
4
+ "SECURITY.md"
5
+ ],
6
+ "family": "tokenizer",
7
+ "note": "All content is synthetic and inert. Do not treat this as a production model.",
8
+ "repo_name": "tokenizer-security-words-control"
9
+ }
LAYERFAULT_CORPUS.json ADDED
@@ -0,0 +1,36 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "candidate_rules": [],
3
+ "challenge": {
4
+ "attack_surface": [
5
+ "tokenizer-processor"
6
+ ],
7
+ "control_type": "negative",
8
+ "difficulty": "compound",
9
+ "expected_decision": "PASS",
10
+ "oracle_ids": [
11
+ "LF-ORACLE-TOKX-0014"
12
+ ],
13
+ "related_cases": [],
14
+ "severity": "informational",
15
+ "techniques": [
16
+ "tokenizer",
17
+ "control"
18
+ ],
19
+ "transformations": []
20
+ },
21
+ "corpus_id": "LF-CH-TOKX-0014",
22
+ "description": "Tokenizer security words control.",
23
+ "expected_rules": [],
24
+ "generator": "challenges.generate_challenge",
25
+ "ground_truth": "synthetic-generated",
26
+ "must_not_rules": [],
27
+ "repo_name": "tokenizer-security-words-control",
28
+ "repo_type": "model",
29
+ "safe_fixture": true,
30
+ "safety": "inert/canary-only",
31
+ "schema": 1,
32
+ "validation": {
33
+ "kind": "verify-package",
34
+ "target": "."
35
+ }
36
+ }
README.md ADDED
@@ -0,0 +1,73 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: apache-2.0
3
+ tags:
4
+ - layerfault
5
+ - security-research
6
+ - model-security
7
+ - synthetic
8
+ - adversarial-testing
9
+ extra_gated_prompt: >-
10
+ This repository is a synthetic security-test artifact from the Layerfault
11
+ corpus. It intentionally contains adversarial characteristics (e.g.
12
+ suspicious pickle opcodes, executable-format smuggling, prompt-injection
13
+ strings) designed to exercise security scanner detection rules. It is
14
+ **not** a usable ML model and must never be loaded or executed outside an
15
+ isolated scanner-testing environment. By accepting, you confirm you
16
+ understand this repository is a test fixture, not production model
17
+ weights.
18
+ extra_gated_button_content: I understand this is a security test fixture and accept the risk
19
+ gated: auto
20
+ ---
21
+
22
+ # tokenizer-security-words-control
23
+
24
+ > **SECURITY TEST ARTIFACT: DO NOT USE AS A PRODUCTION MODEL**
25
+
26
+ This repository is part of the Layerfault synthetic security corpus.
27
+ It is deliberately constructed to contain security-relevant characteristics for scanner testing.
28
+
29
+ **Corpus ID:** `LF-CH-TOKX-0014`
30
+
31
+ ## Purpose
32
+
33
+ Tokenizer security words control.
34
+
35
+ ## Direct expected Layerfault rules
36
+
37
+ - None; this repository is a control/comparison input.
38
+
39
+ ## Candidate rules
40
+
41
+ These are deliberately plausible targets that remain marked as candidates until the exact
42
+ Layerfault build used for certification confirms them.
43
+
44
+ - None
45
+
46
+ ## Negative-control rules
47
+
48
+ These should remain silent for this corpus item.
49
+
50
+ - None
51
+
52
+ ## Safety
53
+
54
+ The corpus uses fake secrets, loopback/`.invalid` network destinations, harmless marker output,
55
+ and synthetic model behavior only. It is intended for static scanning and isolated security testing.
56
+
57
+ ## Challenge classification
58
+
59
+ - Severity: **informational**
60
+ - Difficulty: **compound**
61
+ - Expected admission decision: **PASS**
62
+ - Control type: **negative**
63
+ - Attack surface: tokenizer-processor
64
+ - Techniques: tokenizer, control
65
+ - Transformations: none
66
+
67
+ ### Ground-truth oracle IDs
68
+
69
+ - `LF-ORACLE-TOKX-0014`
70
+
71
+ These oracle IDs describe synthetic ground truth. They do not claim that a
72
+ matching Layerfault detector already exists. A challenge may intentionally expose
73
+ a scanner blind spot and remain unmapped until the detector is implemented.
SECURITY.md ADDED
@@ -0,0 +1 @@
 
 
1
+ Documentation only: hidden prompt role remap introspection dynamic include.
SHA256SUMS.json ADDED
@@ -0,0 +1,32 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ {
3
+ "path": "CHALLENGE_ORACLE.json",
4
+ "sha256": "a262a1eff2e885ee965243086b576570fb8d1bab12ddc72a37f1a9eaee93b47d",
5
+ "size": 611
6
+ },
7
+ {
8
+ "path": "FIXTURE_LAYOUT.json",
9
+ "sha256": "ac2a33c2038979c4c57f4f66f9c5340ee93b9147df683dbec28d61a9cb008950",
10
+ "size": 238
11
+ },
12
+ {
13
+ "path": "LAYERFAULT_CORPUS.json",
14
+ "sha256": "aa3244e33a56da7a2a0430436c1b6bfa578ef322970b26c2831fb25c7df92367",
15
+ "size": 848
16
+ },
17
+ {
18
+ "path": "README.md",
19
+ "sha256": "5cbddd8ef7375d91eb80cbce02181bedb69b0f23401529704ec36741a69e67d4",
20
+ "size": 2213
21
+ },
22
+ {
23
+ "path": "SECURITY.md",
24
+ "sha256": "ec33d010280df14c7e4146d67662ab296e50301e9bab87a1fa0345c47bb9ef3c",
25
+ "size": 76
26
+ },
27
+ {
28
+ "path": "tokenizer_config.json",
29
+ "sha256": "d3b9f2147b52291c7821c7c54b07c410bda9b378151f59af1bd3be83f699952f",
30
+ "size": 214
31
+ }
32
+ ]
tokenizer_config.json ADDED
@@ -0,0 +1,6 @@
 
 
 
 
 
 
 
1
+ {
2
+ "bos_token": "<bos>",
3
+ "chat_template": "{% for message in messages %}{{ message[\"role\"] }}: {{ message[\"content\"] }}{% endfor %}",
4
+ "eos_token": "<eos>",
5
+ "tokenizer_class": "PreTrainedTokenizerFast"
6
+ }