#!/usr/bin/env python3 """ GlyphForge — Hardhat/Forge-style compiler for .glyph and .over source files. .glyph = operator-dense policy programs (glyph-native syntax, no English keywords) .over = OverLanguage workflow specs (intent → artifact → receipt → value) Usage: python3 forge.py init Initialize project structure python3 forge.py compile Compile a .glyph program python3 forge.py compile Compile an .over workflow python3 forge.py build Compile all sources in src/ python3 forge.py test Run all test vectors python3 forge.py snapshot Emit JSON policy snapshot + SHA256 python3 forge.py verify Verify a receipt checksum python3 forge.py clean Remove build artifacts """ import sys import os import json import time import hashlib import struct import math import re import sqlite3 import shutil import zlib import hmac import base64 import secrets import wave import audioop from pathlib import Path from dataclasses import dataclass, field from typing import Optional, Any # ============================================================================= # GLYPH TOKEN TABLE — 40% operators, no English keywords # ============================================================================= GLYPH_TOKENS = { # Nouns (60%) "□": "FILE", "◇": "ARTIFACT", "⧉": "STATIONARY", "H": "HASH", "L": "LOCATION", "R": "RECEIPT", "λ": "FRICTION", "T": "TIME", "Σ": "SHARD", "M": "MERKLE", "ZK": "ZK_PROOF", "Δ": "DELTA", "◎": "VERIFIED", "✕": "INVALID", "$": "VALUE", "Ω": "CANONICAL", "@": "ANCHOR", "∇": "GRADIENT", "∂": "PARTIAL", "∫": "INTEGRAL", "ℏ": "PLANCK", "ℂ": "COMPLEX", "ℝ": "REAL", "ψ": "WAVEFUNCTION", "φ": "PHASE", "θ": "ANGLE", "ρ": "DENSITY", "σ": "PAULI", "π": "PI", "χ": "EIGENVECTOR", "μ": "MEAN", "ν": "VARIANCE", "◈": "PROVE", "⚡": "CLAIM", "¤": "PAY", "⊙̂": "EMIT", "ξ": "RANDOM", "τ": "TENSOR", "η": "EFFICIENCY", "κ": "CURVATURE", "ω": "FREQUENCY", "ϒ": "UPSILON", "Θ": "THETA_BIG", "Φ": "PHI_BIG", "Ψ": "PSI_BIG", "Ξ": "XI_BIG", "Λ": "LAMBDA_BIG", "α": "ALPHA", "β": "BETA", "γ": "GAMMA", "δ": "DELTA_SMALL", "ε": "EPSILON", "ζ": "ZETA", "ι": "IOTA", "υ": "UPSILON_SMALL", "ϕ": "VAR_PHI", "⌀": "DIAMETER", "⌖": "TARGET", "⌘": "COMMAND_KEY", "⌥": "OPTION_KEY", "⇧": "SHIFT_KEY", "⌃": "CONTROL_KEY", "⏎": "RETURN_KEY", "⎋": "ESCAPE_KEY", "␣": "SPACE_KEY", "⇥": "TAB_KEY", "⇤": "HOME_KEY", "␦": "SEPARATOR_NOUN", "␥": "PAD_NOUN", "✦": "STAR_NOUN", "✧": "STAR_OPEN", "◆": "DIAMOND_NOUN", "◇": "DIAMOND_OPEN", "●": "CIRCLE_NOUN", "○": "CIRCLE_OPEN", "■": "SQUARE_NOUN", "□": "SQUARE_OPEN", "▲": "TRIANGLE_NOUN", "△": "TRIANGLE_OPEN", "★": "STAR_FILLED", "☆": "STAR_EMPTY", "⬡": "HEXAGON_NOUN", "⬢": "HEXAGON_FILLED", "⏺": "RECORD_NOUN", "⏸": "PAUSE_NOUN", "⏵": "PLAY_NOUN", "⏹": "STOP_NOUN", "♻": "RECYCLE_NOUN", "✓": "CHECK_NOUN", "✗": "CROSS_NOUN", "⚠": "WARNING_NOUN", "ℹ": "INFO_NOUN", "⚙": "GEAR_NOUN", "🗜": "COMPRESS_NOUN", "🔓": "UNLOCK_NOUN", "🔐": "LOCK_NOUN", "🔑": "KEY_NOUN", "🜔": "SALT_NOUN", "🜁": "AIR_NOUN", "🜂": "FIRE_NOUN", "🜃": "EARTH_NOUN", "🜄": "WATER_NOUN", "🜅": "QUINTESSENCE", # Operators (40%) "⊕": "ADD", "⊖": "SUB", "⊗": "MUL", "⊘": "DIV", "⊙": "DOT", "⊚": "OUTER", "⊛": "KRON", "∧": "AND", "∨": "OR", "¬": "NOT", "⊼": "NAND", "⊽": "NOR", "⊻": "XOR", "≡": "IDENTICAL", "≠": "DIFFERENT", "≲": "LESSEQ", "≳": "GREATEQ", "⇉": "PIPE_FWD", "⇇": "PIPE_BWD", "⇈": "PAR_UP", "⇊": "PAR_DOWN", "↺": "REWIND", "↻": "FORWARD", "⟳": "REPEAT", "⨁": "SPIN_ADD", "⨂": "SPIN_MUL", "⨄": "SPIN_SUM", "↑": "SPIN_UP", "↓": "SPIN_DOWN", "↕": "SPIN_FLIP", "⊠": "TENSOR_BOX", "⊞": "TENSOR_ADD", "Æ": "BIND", "ÆÆ": "DOUBLE_BIND", "Æ⁻": "BOND_BREAK", "Æ⁺": "BOND_FORM", "Æ⁰": "BOND_NULL", "→": "DERIVE", "=": "ASSERT", ";": "SEPARATOR", "∮": "INTEGRATE", "∴": "THEREFORE", "∞": "DIVERGE", "▷": "PROGRAM_START", "◀": "PROGRAM_END", "⇒": "IMPLY", "⇐": "REVERSE_IMPLY", "⇔": "BICONDITIONAL", "∝": "PROPORTIONAL", "ℵ": "CARDINALITY", "⌁": "ELECTRIC_FLOW", "⌬": "BENZENE_RING", "⏃": "ANTI_GRAVITY", "⏆": "GRAVITY_DOWN", "⤓": "FLOW_DOWN", "⤒": "FLOW_UP", "⥁": "CYCLE_OP", "⥎": "EXCHANGE", "⧴": "MAPPING", "⧫": "DIAMOND_OP", "⧠": "SQUARE_OP", "⧖": "HOURGLASS_OP", # --- Control flow --- "⟦": "BLOCK_OPEN", "⟧": "BLOCK_CLOSE", "⟨": "GROUP_OPEN", "⟩": "GROUP_CLOSE", "⦃": "SCOPE_OPEN", "⦄": "SCOPE_CLOSE", "⦂": "BRANCH", "⦙": "BRANCH_ELSE", "⤴": "JUMP_FWD", "⤵": "JUMP_BWD", "⤶": "BREAK", "⤷": "CONTINUE", "⥂": "LOOP", "⥃": "LOOP_UNTIL", "⥄": "WHILE", "⥅": "FOR_EACH", "⥆": "ITER_NEXT", "⥇": "ITER_DONE", "⇜": "CALL", "⇝": "RETURN", "⇞": "YIELD", "⇟": "AWAIT_OP", "⇠": "SEND", "⇡": "RECEIVE", # --- Variable / binding --- "≔": "ASSIGN", "≕": "REASSIGN", "⇎": "SWAP", "⇏": "DROP", "⇬": "LIFT", "⇭": "LOWER", "⇮": "FREEZE", "⇯": "THAW", "⥤": "REF", "⥦": "DEREF", "⥧": "WEAKREF", "⥨": "PIN", # --- Type system --- "Ⲷ": "TYPE_INT", "ⲷ": "TYPE_FLOAT", "Ⲹ": "TYPE_STR", "ⲹ": "TYPE_BOOL", "Ⲻ": "TYPE_BYTES", "ⲻ": "TYPE_LIST", "Ⲽ": "TYPE_DICT", "ⲽ": "TYPE_SET", "Ⲿ": "TYPE_TUPLE", "ⲿ": "TYPE_OPTIONAL", "Ⳁ": "TYPE_RESULT", "ⳁ": "TYPE_ENUM", "Ⳃ": "TYPE_STRUCT", "ⳃ": "TYPE_TRAIT", "Ⳅ": "TYPE_UNION", "ⳅ": "TYPE_INTERSECT", "Ⳇ": "TYPE_FN", "ⳇ": "TYPE_VOID", "Ⳉ": "TYPE_NEVER", "ⳉ": "TYPE_ANY", "Ⳋ": "TYPE_SELF", "ⳋ": "TYPE_UNKNOWN", "⟜": "TYPE_CHECK", "⟛": "TYPE_CAST", "⧴⧴": "TYPE_ASSERT", # --- Function definition --- "⏦": "FN_DEF", "⏧": "FN_END", "⏨": "FN_PARAM", "⏩": "FN_BODY", "⏪": "FN_RECURSE", "⏫": "FN_TAIL", "⏬": "FN_INLINE", "⏭": "FN_MACRO", "⏮": "FN_CLOSURE", "⏯": "FN_ANON", # --- I/O primitives --- "⇿": "READ", "⇾": "WRITE", "⇽": "STDIN", "⇾⇾": "STDOUT", "⇽⇽": "STDERR", "⥳": "PIPE", "⥴": "CHANNEL", "⥵": "STREAM", "⥶": "BUFFER", "⥷": "FLUSH", "⥸": "CLOSE_FD", "⥹": "OPEN_FD", "⥺": "SEEK", "⥻": "TELL", "⦀": "EOF", "⦂⦂": "EOL", # --- Concurrency --- "⣢": "FORK", "⣡": "JOIN", "⣠": "LOCK", "⣟": "UNLOCK", "⣞": "SEMAPHORE", "⣝": "BARRIER", "⣜": "RACE", "⣛": "SELECT_OP", "⣚": "SPAWN", "⣙": "DETACH", "⣘": "SYNC", "⣗": "ASYNC_OP", "⣖": "PROMISE", "⣕": "RESOLVE", "⣔": "REJECT", "⣓": "PENDING", # --- State machine --- "⦜": "STATE_DEF", "⦝": "STATE_TRANS", "⦞": "STATE_GUARD", "⦟": "STATE_ACTION", "⦠": "STATE_ENTRY", "⦡": "STATE_EXIT", "⦢": "STATE_INIT", "⦣": "STATE_FINAL", "⦤": "STATE_HISTORY", "⦥": "STATE_PARALLEL", # --- Error handling --- "⦦": "TRY_OP", "⦧": "CATCH", "⦨": "FINALLY", "⦩": "RAISE", "⦪": "RECOVER", "⦫": "PANIC", "⦬": "ABORT", "⦭": "RETRY", "⦮": "ERROR_VALUE", "⦯": "ERROR_KIND", # --- Pattern matching --- "⦰": "MATCH", "⦱": "CASE", "⦲": "WILDCARD", "⦳": "GUARD", "⦴": "BIND_PATTERN", "⦵": "DESTRUCT", "⦶": "CONS_PATTERN", "⦷": "NIL_PATTERN", "⦸": "SOME_PATTERN", "⦹": "NONE_PATTERN", "⦺": "OK_PATTERN", "⦻": "ERR_PATTERN", # --- Module system --- "⧀": "IMPORT_OP", "⧁": "EXPORT_OP", "⧂": "NAMESPACE", "⧃": "MODULE_DEF", "⧄": "MODULE_END", "⧅": "USE", "⧆": "HIDE", "⧇": "EXPOSE", "⧈": "REEXPORT", "⧉⧉": "LINK", # --- Comparison / logic (expanded) --- "⪯": "LT", "⪰": "GT", "⪱": "LE", "⪲": "GE", "⪳": "EQ", "⪴": "NEQ", "⪵": "IN_OP", "⪶": "NOT_IN", "⪷": "SUBSET", "⪸": "SUPERSET", "⪹": "SUBSETEQ", "⪺": "SUPERSETEQ", "⪻": "DISJOINT", "⪼": "OVERLAP", # --- Arithmetic (expanded) --- "⨥": "ADD_SAT", "⨦": "SUB_SAT", "⨧": "MUL_SAT", "⨨": "DIV_SAT", "⨩": "REM", "⨪": "NEG", "⨫": "ABS", "⨬": "SIGN", "⨭": "MIN_OP", "⨮": "MAX_OP", "⨯": "CROSS", "⨰": "DOT_PROD", "⨱": "OUTER_PROD", "⨲": "HADAMARD", "⨳": "CONVOLVE", "⨴": "CORRELATE", "⨵": "FFT", "⨶": "IFFT", # --- Bitwise (expanded) --- "⤔": "SHL", "⤕": "SHR", "⤖": "SAR", "⤗": "ROL", "⤘": "ROR", "⤙": "POP_COUNT", "⤚": "CLZ", "⤛": "CTZ", "⤜": "BSWAP", "⤝": "BIT_REVERSE", # --- String / sequence ops --- "⫶": "CONCAT", "⫷": "SLICE", "⫸": "INDEX", "⫹": "APPEND", "⫺": "PREPEND", "⫻": "REVERSE", "⫼": "SPLIT", "⫽": "JOIN_OP", "⫾": "REPLACE", "⫿": "FIND", "⬀": "CONTAINS", "⬁": "STARTS_WITH", "⬂": "ENDS_WITH", "⬃": "MATCHES", # --- Memory / layout --- "⬄": "ALLOC", "⬅": "DEALLOC", "⬆": "COPY_MEM", "⬇": "MOVE_MEM", "⬌": "ZERO_MEM", "⬍": "FILL_MEM", "⬎": "CMP_MEM", "⬏": "SIZEOF", "⬐": "ALIGNOF", "⬑": "OFFSETOF", # --- Crypto / hashing (expanded) --- "⬠": "HASH_SHA256", "⬡": "HASH_SHA512", "⬢": "HASH_BLAKE3", "⬣": "HASH_KECCAK", "⬤": "HMAC_OP", "⬥": "AEAD_OP", "⬦": "SIGN_OP", "⬧": "VERIFY_SIG", "⬨": "ENCRYPT", "⬩": "DECRYPT", "⬪": "KDF", "⬫": "PRF", "⬬": "RNG", "⬭": "CSRNG", # --- Time / temporal --- "⭐": "NOW", "⭑": "TIMER", "⭒": "DELAY", "⭓": "DEADLINE", "⭔": "TIMEOUT", "⭕": "EPOCH", "⭖": "DURATION", "⭗": "INTERVAL", "⭘": "CLOCK_MONO", "⭙": "CLOCK_WALL", # --- Debug / introspection --- "⭚": "TRACE", "⭛": "DEBUG", "⭜": "INSPECT", "⭝": "DUMP", "⭞": "BREAKPOINT", "⭟": "WATCH", "⭠": "PROFILE", "⭡": "BENCH", # --- Network / distributed --- "⭢": "CONNECT", "⭣": "DISCONNECT", "⭤": "LISTEN", "⭥": "ACCEPT", "⭦": "REQUEST", "⭧": "RESPONSE", "⭨": "BROADCAST", "⭩": "MULTICAST", "⭪": "ROUTE", "⭫": "PROXY", "⭬": "GATEWAY", "⭭": "RELAY", # --- Quantities / units --- "⭮": "COUNT", "⭯": "SUM_OP", "⭰": "PRODUCT", "⭱": "MEAN_OP", "⭲": "MEDIAN", "⭳": "MODE", "⭴": "VARIANCE_OP", "⭵": "STDDEV", "⭶": "PERCENTILE", "⭷": "QUARTILE", "⭸": "HISTOGRAM", "⭹": "CDF", "⭺": "PDF", "⭻": "SAMPLE", # --- MIDI notation --- # Note names (sharps) — each glyph maps to MIDI note number "♩": "NOTE_C", # C (0, 12, 24... — MIDI 60 = C4) "♩♯": "NOTE_CS", # C# / Db "♪": "NOTE_D", # D "♪♯": "NOTE_DS", # D# / Eb "♫": "NOTE_E", # E "♬": "NOTE_F", # F "♬♯": "NOTE_FS", # F# / Gb "♭": "NOTE_G", # G "♭♯": "NOTE_GS", # G# / Ab "♮": "NOTE_A", # A "♮♯": "NOTE_AS", # A# / Bb "♯": "NOTE_B", # B # Octave indicators (MIDI octave numbers 0-8) "𝄞": "OCTAVE_0", "𝄞¹": "OCTAVE_1", "𝄞²": "OCTAVE_2", "𝄞³": "OCTAVE_3", "𝄞⁴": "OCTAVE_4", "𝄞⁵": "OCTAVE_5", "𝄞⁶": "OCTAVE_6", "𝄞⁷": "OCTAVE_7", "𝄞⁸": "OCTAVE_8", # Rest and duration "𝄽": "REST", # Musical rest "𝅗𝅥": "HALF_NOTE", # 2 beats "𝅘𝅥": "QUARTER_NOTE", # 1 beat "𝅘𝅥𝅮": "EIGHTH_NOTE", # 1/2 beat "𝅘𝅥𝅯": "SIXTEENTH_NOTE", # 1/4 beat "𝅘𝅥𝅰": "THIRTYSECOND_NOTE", # 1/8 beat # MIDI operators "⬌": "PITCH_BEND", # Pitch wheel "⬆": "VELOCITY_UP", # Increase velocity "⬇": "VELOCITY_DOWN", # Decrease velocity "🎹": "MIDI_CHANNEL", # Channel selector "🎚": "CONTROL_CHANGE", # CC message "🔊": "MIDI_VOLUME", # Volume CC7 "🔇": "MIDI_MUTE", # Mute " sustain": "SUSTAIN_PEDAL", # Sustain pedal CC64 "🎸": "MIDI_PROGRAM", # Program change (instrument) "🥁": "MIDI_DRUM", # Drum channel (ch 10) "⏱": "MIDI_TEMPO", # Tempo (BPM) "𝄫": "KEY_SIG", # Key signature "𝄪": "TIME_SIG", # Time signature "𝄢": "BASS_CLEF", # Bass clef context "𝄫¹": "TREBLE_CLEF", # Treble clef context # MIDI sequence operators "⇨": "NOTE_ON", # Note on event "⇦": "NOTE_OFF", # Note off event "⇧": "MIDI_HOLD", # Hold note (duration) "⇩": "MIDI_RELEASE", # Release note "↻": "MIDI_ARPEGGIO", # Arpeggiate "↺": "MIDI_TRILL", # Trill between notes "⇶": "MIDI_GLISSANDO", # Glide between pitches "𝆑": "MIDI_FORTE", # Loud (velocity 100-127) "𝆏": "MIDI_PIANO", # Soft (velocity 1-43) "𝆐": "MIDI_MEZZO", # Medium (velocity 44-99) "𝆑𝆏": "MIDI_FORTISSIMO", # Very loud (velocity 110-127) "𝆏𝆏": "MIDI_PIANISSIMO", # Very soft (velocity 1-20) # MIDI meta "🎼": "MIDI_TRACK", # Track definition "🎹¹": "MIDI_SYSEX", # System exclusive "𝄽¹": "MIDI_EOT", # End of track "🔀": "MIDI_QUANTIZE", # Quantize timing "🌀": "MIDI_LOOP", # Loop MIDI pattern "📊": "MIDI_VELOCITY_CURVE", # Velocity automation "🎯": "MIDI_TARGET", # Target note for legato/portamento } OPERATORS = {k: v for k, v in GLYPH_TOKENS.items() if v in { "ADD", "SUB", "MUL", "DIV", "DOT", "OUTER", "KRON", "AND", "OR", "NOT", "NAND", "NOR", "XOR", "IDENTICAL", "DIFFERENT", "LESSEQ", "GREATEQ", "PIPE_FWD", "PIPE_BWD", "PAR_UP", "PAR_DOWN", "REWIND", "FORWARD", "REPEAT", "SPIN_ADD", "SPIN_MUL", "SPIN_SUM", "SPIN_UP", "SPIN_DOWN", "SPIN_FLIP", "TENSOR_BOX", "TENSOR_ADD", "BIND", "DOUBLE_BIND", "BOND_BREAK", "BOND_FORM", "BOND_NULL", "DERIVE", "ASSERT", "SEPARATOR", "INTEGRATE", "THEREFORE", "DIVERGE", "PROGRAM_START", "PROGRAM_END", "IMPLY", "REVERSE_IMPLY", "BICONDITIONAL", "PROPORTIONAL", "CARDINALITY", "ELECTRIC_FLOW", "BENZENE_RING", "ANTI_GRAVITY", "GRAVITY_DOWN", "FLOW_DOWN", "FLOW_UP", "CYCLE_OP", "EXCHANGE", "MAPPING", "DIAMOND_OP", "SQUARE_OP", "HOURGLASS_OP", # Control flow operators "BLOCK_OPEN", "BLOCK_CLOSE", "GROUP_OPEN", "GROUP_CLOSE", "SCOPE_OPEN", "SCOPE_CLOSE", "BRANCH", "BRANCH_ELSE", "JUMP_FWD", "JUMP_BWD", "BREAK", "CONTINUE", "LOOP", "LOOP_UNTIL", "WHILE", "FOR_EACH", "ITER_NEXT", "ITER_DONE", "CALL", "RETURN", "YIELD", "AWAIT_OP", "SEND", "RECEIVE", # Variable / binding operators "ASSIGN", "REASSIGN", "SWAP", "DROP", "LIFT", "LOWER", "FREEZE", "THAW", "REF", "DEREF", "WEAKREF", "PIN", # Type operators "TYPE_CHECK", "TYPE_CAST", "TYPE_ASSERT", # Function operators "FN_DEF", "FN_END", "FN_PARAM", "FN_BODY", "FN_RECURSE", "FN_TAIL", "FN_INLINE", "FN_MACRO", "FN_CLOSURE", "FN_ANON", # I/O operators "READ", "WRITE", "STDIN", "STDOUT", "STDERR", "PIPE", "CHANNEL", "STREAM", "BUFFER", "FLUSH", "CLOSE_FD", "OPEN_FD", "SEEK", "TELL", "EOF", "EOL", # Concurrency operators "FORK", "JOIN", "LOCK", "UNLOCK", "SEMAPHORE", "BARRIER", "RACE", "SELECT_OP", "SPAWN", "DETACH", "SYNC", "ASYNC_OP", "PROMISE", "RESOLVE", "REJECT", "PENDING", # State machine operators "STATE_DEF", "STATE_TRANS", "STATE_GUARD", "STATE_ACTION", "STATE_ENTRY", "STATE_EXIT", "STATE_INIT", "STATE_FINAL", "STATE_HISTORY", "STATE_PARALLEL", # Error handling operators "TRY_OP", "CATCH", "FINALLY", "RAISE", "RECOVER", "PANIC", "ABORT", "RETRY", "ERROR_VALUE", "ERROR_KIND", # Pattern matching operators "MATCH", "CASE", "WILDCARD", "GUARD", "BIND_PATTERN", "DESTRUCT", "CONS_PATTERN", "NIL_PATTERN", "SOME_PATTERN", "NONE_PATTERN", "OK_PATTERN", "ERR_PATTERN", # Module operators "IMPORT_OP", "EXPORT_OP", "NAMESPACE", "MODULE_DEF", "MODULE_END", "USE", "HIDE", "EXPOSE", "REEXPORT", "LINK", # Comparison / logic (expanded) "LT", "GT", "LE", "GE", "EQ", "NEQ", "IN_OP", "NOT_IN", "SUBSET", "SUPERSET", "SUBSETEQ", "SUPERSETEQ", "DISJOINT", "OVERLAP", # Arithmetic (expanded) "ADD_SAT", "SUB_SAT", "MUL_SAT", "DIV_SAT", "REM", "NEG", "ABS", "SIGN", "MIN_OP", "MAX_OP", "CROSS", "DOT_PROD", "OUTER_PROD", "HADAMARD", "CONVOLVE", "CORRELATE", "FFT", "IFFT", # Bitwise (expanded) "SHL", "SHR", "SAR", "ROL", "ROR", "POP_COUNT", "CLZ", "CTZ", "BSWAP", "BIT_REVERSE", # String / sequence ops "CONCAT", "SLICE", "INDEX", "APPEND", "PREPEND", "REVERSE", "SPLIT", "JOIN_OP", "REPLACE", "FIND", "CONTAINS", "STARTS_WITH", "ENDS_WITH", "MATCHES", # Memory / layout "ALLOC", "DEALLOC", "COPY_MEM", "MOVE_MEM", "ZERO_MEM", "FILL_MEM", "CMP_MEM", "SIZEOF", "ALIGNOF", "OFFSETOF", # Crypto / hashing "HASH_SHA256", "HASH_SHA512", "HASH_BLAKE3", "HASH_KECCAK", "HMAC_OP", "AEAD_OP", "SIGN_OP", "VERIFY_SIG", "ENCRYPT", "DECRYPT", "KDF", "PRF", "RNG", "CSRNG", # Time / temporal "NOW", "TIMER", "DELAY", "DEADLINE", "TIMEOUT", "EPOCH", "DURATION", "INTERVAL", "CLOCK_MONO", "CLOCK_WALL", # Debug / introspection "TRACE", "DEBUG", "INSPECT", "DUMP", "BREAKPOINT", "WATCH", "PROFILE", "BENCH", # Network / distributed "CONNECT", "DISCONNECT", "LISTEN", "ACCEPT", "REQUEST", "RESPONSE", "BROADCAST", "MULTICAST", "ROUTE", "PROXY", "GATEWAY", "RELAY", # Quantities / statistics "COUNT", "SUM_OP", "PRODUCT", "MEAN_OP", "MEDIAN", "MODE", "VARIANCE_OP", "STDDEV", "PERCENTILE", "QUARTILE", "HISTOGRAM", "CDF", "PDF", "SAMPLE", # MIDI operators "PITCH_BEND", "VELOCITY_UP", "VELOCITY_DOWN", "CONTROL_CHANGE", "MIDI_VOLUME", "MIDI_MUTE", "SUSTAIN_PEDAL", "MIDI_PROGRAM", "MIDI_DRUM", "NOTE_ON", "NOTE_OFF", "MIDI_HOLD", "MIDI_RELEASE", "MIDI_ARPEGGIO", "MIDI_TRILL", "MIDI_GLISSANDO", "MIDI_FORTE", "MIDI_PIANO", "MIDI_MEZZO", "MIDI_FORTISSIMO", "MIDI_PIANISSIMO", "MIDI_QUANTIZE", "MIDI_LOOP", "MIDI_VELOCITY_CURVE", "MIDI_TARGET", }} OPERATOR_RATIO = len(OPERATORS) / len(GLYPH_TOKENS) # ============================================================================= # GLYPH FILE LEXER — .glyph source → tokens # ============================================================================= @dataclass class GlyphToken: glyph: str name: str is_operator: bool line: int col: int def lex_glyph(source: str) -> list[GlyphToken]: """Lex .glyph source into tokens. No English keywords.""" tokens = [] for line_num, line in enumerate(source.split("\n"), 1): col = 0 while col < len(line): matched = False # Try longest match first for g in sorted(GLYPH_TOKENS.keys(), key=len, reverse=True): if line[col:col+len(g)] == g: name = GLYPH_TOKENS[g] tokens.append(GlyphToken(g, name, g in OPERATORS, line_num, col)) col += len(g) matched = True break if not matched: col += 1 # skip whitespace/comments return tokens # ============================================================================= # GLYPH FILE PARSER — tokens → AST # ============================================================================= @dataclass class GlyphNode: op: str operands: list = field(default_factory=list) children: list = field(default_factory=list) line: int = 0 @dataclass class GlyphAST: name: str = "" nodes: list[GlyphNode] = field(default_factory=list) glyph_count: int = 0 operator_count: int = 0 noun_count: int = 0 operator_ratio: float = 0.0 max_depth: int = 0 block_count: int = 0 branch_count: int = 0 loop_count: int = 0 fn_count: int = 0 match_count: int = 0 state_count: int = 0 try_count: int = 0 # Block-opening operators that create a new nesting level _BLOCK_OPENERS = frozenset({ "BLOCK_OPEN", "SCOPE_OPEN", "GROUP_OPEN", "BRANCH", "BRANCH_ELSE", "LOOP", "LOOP_UNTIL", "WHILE", "FOR_EACH", "FN_DEF", "FN_BODY", "FN_CLOSURE", "FN_ANON", "TRY_OP", "CATCH", "FINALLY", "MATCH", "CASE", "STATE_DEF", "STATE_ENTRY", "STATE_EXIT", "MODULE_DEF", "NAMESPACE", }) _BLOCK_CLOSERS = frozenset({ "BLOCK_CLOSE", "SCOPE_CLOSE", "GROUP_CLOSE", "FN_END", "MODULE_END", }) def parse_glyph(tokens: list[GlyphToken]) -> GlyphAST: """Parse glyph tokens into a nested AST with block structure support.""" ast = GlyphAST() in_program = False current_chain: list[GlyphToken] = [] stack: list[GlyphNode] = [] # block stack for nesting depth = 0 def flush_chain(): """Flush current noun chain as a node.""" nonlocal current_chain if current_chain: node = GlyphNode( op="CHAIN", operands=[t.glyph for t in current_chain], line=current_chain[0].line, ) if stack: stack[-1].children.append(node) else: ast.nodes.append(node) current_chain = [] def emit_node(op_name: str, tok: GlyphToken): """Emit an operator node, either into the current block or root.""" nonlocal current_chain node = GlyphNode(op=op_name, operands=[t.glyph for t in current_chain], line=tok.line) if stack: stack[-1].children.append(node) else: ast.nodes.append(node) current_chain = [] for tok in tokens: if tok.name == "PROGRAM_START": in_program = True current_chain = [] continue if tok.name == "PROGRAM_END": flush_chain() in_program = False continue if not in_program: continue ast.glyph_count += 1 if tok.is_operator: ast.operator_count += 1 # Block openers — push a new node onto the stack if tok.name in _BLOCK_OPENERS: flush_chain() node = GlyphNode(op=tok.name, operands=[], line=tok.line) if stack: stack[-1].children.append(node) else: ast.nodes.append(node) stack.append(node) depth += 1 ast.max_depth = max(ast.max_depth, depth) if tok.name in ("BRANCH", "BRANCH_ELSE"): ast.branch_count += 1 elif tok.name in ("LOOP", "LOOP_UNTIL", "WHILE", "FOR_EACH"): ast.loop_count += 1 elif tok.name in ("FN_DEF", "FN_CLOSURE", "FN_ANON"): ast.fn_count += 1 elif tok.name in ("MATCH", "CASE"): ast.match_count += 1 elif tok.name in ("STATE_DEF", "STATE_ENTRY", "STATE_EXIT"): ast.state_count += 1 elif tok.name in ("TRY_OP", "CATCH", "FINALLY"): ast.try_count += 1 continue # Block closers — pop from the stack if tok.name in _BLOCK_CLOSERS: flush_chain() if stack: stack.pop() depth -= 1 continue # Regular operator — terminates current chain emit_node(tok.name, tok) else: ast.noun_count += 1 current_chain.append(tok) flush_chain() # Auto-close any unclosed blocks while stack: stack.pop() depth -= 1 total = ast.operator_count + ast.noun_count ast.operator_ratio = ast.operator_count / max(total, 1) ast.block_count = ast.max_depth # approx return ast # ============================================================================= # GLYPH FILE COMPILER — AST → executable artifact # ============================================================================= def compile_glyph(source: str, filename: str = "") -> dict: """Compile a .glyph source file into an executable artifact.""" start = time.time() tokens = lex_glyph(source) ast = parse_glyph(tokens) # Generate spinor embeddings for each glyph embeddings = {} for tok in tokens: if tok.glyph not in embeddings: h = hashlib.sha256(tok.glyph.encode()).digest() a = struct.unpack('f', h[0:4])[0] b = struct.unpack('f', h[4:8])[0] norm = math.sqrt(abs(a)**2 + abs(b)**2) or 1.0 embeddings[tok.glyph] = { "spinor": [a/norm, b/norm], "bloch_theta": 2 * math.acos(max(0, min(1, abs(a/norm)))), "name": tok.name, "is_operator": tok.is_operator, } # Build artifact def serialize_node(n: GlyphNode) -> dict: return { "op": n.op, "operands": n.operands, "line": n.line, "children": [serialize_node(c) for c in n.children], } artifact = { "type": "glyph_compiled", "source_file": filename, "compiled_at": time.time(), "glyph_count": ast.glyph_count, "operator_count": ast.operator_count, "noun_count": ast.noun_count, "operator_ratio": round(ast.operator_ratio, 4), "node_count": len(ast.nodes), "max_depth": ast.max_depth, "branch_count": ast.branch_count, "loop_count": ast.loop_count, "fn_count": ast.fn_count, "match_count": ast.match_count, "state_count": ast.state_count, "try_count": ast.try_count, "token_count": len(GLYPH_TOKENS), "nodes": [serialize_node(n) for n in ast.nodes], "embeddings": embeddings, "compile_time_ms": round((time.time() - start) * 1000, 2), } # SHA256 checksum artifact_str = json.dumps(artifact, sort_keys=True) artifact["sha256"] = hashlib.sha256(artifact_str.encode()).hexdigest() return artifact # ============================================================================= # OVERLANGUAGE FILE FORMAT — .over source # ============================================================================= @dataclass class OverStep: step_num: int action: str inputs: list = field(default_factory=list) outputs: list = field(default_factory=list) receipt: bool = True @dataclass class OverWorkflow: name: str = "" intent: str = "" steps: list[OverStep] = field(default_factory=list) artifacts: list = field(default_factory=list) receipts: list = field(default_factory=list) value_claim: str = "" def parse_over(source: str) -> OverWorkflow: """Parse .over source into OverWorkflow. Format is line-based with → as the flow operator: intent: step 1: step 2: artifact: receipt: value: """ wf = OverWorkflow() step_counter = 0 for line in source.split("\n"): line = line.strip() if not line or line.startswith("#"): continue if line.startswith("intent:"): wf.intent = line[7:].strip() elif line.startswith("workflow:"): wf.name = line[9:].strip() elif line.startswith("step"): step_counter += 1 rest = line.split(":", 1)[1].strip() if ":" in line else line parts = rest.split("→") action = parts[0].strip() outputs = [p.strip() for p in parts[1:]] if len(parts) > 1 else [] wf.steps.append(OverStep(step_num=step_counter, action=action, outputs=outputs)) elif line.startswith("artifact:"): wf.artifacts.append(line[9:].strip()) elif line.startswith("receipt:"): wf.receipts.append(line[8:].strip()) elif line.startswith("value:"): wf.value_claim = line[6:].strip() return wf def compile_over(source: str, filename: str = "") -> dict: """Compile a .over source file into a workflow artifact.""" start = time.time() wf = parse_over(source) # Generate receipt chain receipt_chain = [] prev_hash = "0" * 64 for step in wf.steps: entry = json.dumps({ "step": step.step_num, "action": step.action, "outputs": step.outputs, "ts": time.time(), }, sort_keys=True) entry_hash = hashlib.sha256((prev_hash + entry).encode()).hexdigest() receipt_chain.append({ "step": step.step_num, "action": step.action, "hash": entry_hash, "prev_hash": prev_hash, }) prev_hash = entry_hash artifact = { "type": "over_compiled", "source_file": filename, "compiled_at": time.time(), "workflow_name": wf.name, "intent": wf.intent, "step_count": len(wf.steps), "steps": [ {"step": s.step_num, "action": s.action, "outputs": s.outputs} for s in wf.steps ], "artifacts": wf.artifacts, "value_claim": wf.value_claim, "receipt_chain": receipt_chain, "merkle_root": prev_hash, "compile_time_ms": round((time.time() - start) * 1000, 2), } artifact_str = json.dumps(artifact, sort_keys=True) artifact["sha256"] = hashlib.sha256(artifact_str.encode()).hexdigest() return artifact # ============================================================================= # OVER RUNTIME — Execute .over workflows with real file I/O # ============================================================================= class OverRuntime: """Executes .over workflows step-by-step with real operations. No mock. No simulation. Real file reads, real SHA256, real SQLite indexes, real chunk extraction, real search, real revocation.""" def __init__(self): self.state: dict[str, Any] = {} self.receipts: list[dict] = [] self.prev_hash = "0" * 64 self.index_dir = Path("jorki_data/indexes") self.index_dir.mkdir(parents=True, exist_ok=True) self.registry_path = Path("jorki_data/registry.json") self.registry: dict[str, dict] = {} if self.registry_path.exists(): self.registry = json.loads(self.registry_path.read_text()) def _receipt(self, step: int, action: str, result: Any) -> dict: entry = json.dumps({"step": step, "action": action, "result_hash": hashlib.sha256(str(result).encode()).hexdigest()[:16], "ts": time.time()}, sort_keys=True) entry_hash = hashlib.sha256((self.prev_hash + entry).encode()).hexdigest() r = {"step": step, "action": action, "hash": entry_hash, "prev_hash": self.prev_hash, "ts": time.time()} self.receipts.append(r) self.prev_hash = entry_hash return r def _save_registry(self): self.registry_path.write_text(json.dumps(self.registry, indent=2)) def execute(self, wf: OverWorkflow, args: dict[str, str] | None = None) -> dict: """Execute all steps in the workflow. args provides runtime parameters.""" args = args or {} self.state["args"] = args self.state["workflow"] = wf.name self.state["intent"] = wf.intent results = [] for step in wf.steps: action = step.action.lower().strip() result = self._exec_action(step.step_num, action, args) for out in step.outputs: self.state[out] = result r = self._receipt(step.step_num, step.action, result) results.append({"step": step.step_num, "action": step.action, "outputs": step.outputs, "result": result, "receipt": r["hash"][:16]}) print(f" step {step.step_num}: {step.action} -> {step.outputs} [receipt: {r['hash'][:12]}...]") merkle_root = self.prev_hash artifact = { "type": "over_executed", "workflow": wf.name, "intent": wf.intent, "executed_at": time.time(), "step_results": results, "state": {k: v for k, v in self.state.items() if k not in ("args",)}, "artifacts": wf.artifacts, "value_claim": wf.value_claim, "receipt_chain": self.receipts, "merkle_root": merkle_root, } artifact_str = json.dumps(artifact, sort_keys=True) artifact["sha256"] = hashlib.sha256(artifact_str.encode()).hexdigest() receipts_dir = Path("receipts") receipts_dir.mkdir(exist_ok=True) receipt_name = f"{wf.name}_{int(time.time())}.json" (receipts_dir / receipt_name).write_text(json.dumps(artifact, indent=2, ensure_ascii=False)) return artifact def _exec_action(self, step_num: int, action: str, args: dict) -> Any: """Execute a single workflow action. Real operations only.""" if "index file" in action or ("index" in action and "file" in action): filepath = args.get("file", args.get("filepath", "")) if not filepath or not os.path.exists(filepath): return {"error": f"File not found: {filepath}"} return self._index_file(filepath) if "compute hash" in action or "merkle" in action.lower(): filepath = args.get("file", args.get("filepath", "")) if not filepath or not os.path.exists(filepath): idx = self.state.get("local_index", self.state.get("file_index", {})) if isinstance(idx, dict) and "merkle_root" in idx: return idx["merkle_root"] return {"error": "No file to hash"} return self._compute_hash(filepath) if "upload" in action and ("index" in action or "hf" in action or "space" in action): idx = self.state.get("local_index", {}) if not idx or "file_id" not in idx: return {"error": "No index to upload"} file_id = idx["file_id"] self.registry[file_id] = { "filename": idx.get("filename", "unknown"), "merkle_root": idx.get("merkle_root", ""), "indexed_at": time.time(), "status": "active", "index_path": str(self.index_dir / f"{file_id}.idx"), } self._save_registry() return {"file_id": file_id, "status": "uploaded", "url": f"jorki://query/{file_id}"} if "search" in action or ("query" in action and "sql" not in action): file_id = args.get("file_id", "") query = args.get("q", args.get("query", "")) if not file_id: idx = self.state.get("local_index", {}) file_id = idx.get("file_id", "") return self._search(file_id, query) if "sql" in action: file_id = args.get("file_id", "") sql = args.get("sql", "SELECT COUNT(*) FROM chunks") if not file_id: idx = self.state.get("local_index", {}) file_id = idx.get("file_id", "") return self._sql_query(file_id, sql) if "chunk" in action or "retrieve" in action: file_id = args.get("file_id", "") chunk_idx = int(args.get("chunk_idx", args.get("idx", 0))) if not file_id: idx = self.state.get("local_index", {}) file_id = idx.get("file_id", "") return self._get_chunk(file_id, chunk_idx) if "verify" in action: file_id = args.get("file_id", "") if not file_id: idx = self.state.get("local_index", {}) file_id = idx.get("file_id", "") entry = self.registry.get(file_id, {}) if not entry: return {"error": f"File {file_id} not in registry"} return {"file_id": file_id, "verified": True, "merkle_root": entry.get("merkle_root", ""), "status": entry.get("status", "unknown")} if "revoke" in action or "expire" in action: file_id = args.get("file_id", "") if not file_id: upload = self.state.get("upload_result", self.state.get("query_gateway", {})) file_id = upload.get("file_id", "") if isinstance(upload, dict) else "" if file_id and file_id in self.registry: self.registry[file_id]["status"] = "revoked" self.registry[file_id]["revoked_at"] = time.time() self._save_registry() return {"file_id": file_id, "status": "revoked", "revoked_at": time.time()} return {"error": f"File {file_id} not found in registry"} if "confirm" in action and ("revoke" in action or "404" in action or "closed" in action): file_id = args.get("file_id", "") entry = self.registry.get(file_id, {}) if entry.get("status") == "revoked": return {"file_id": file_id, "confirmed": True, "access": "closed"} return {"file_id": file_id, "confirmed": False, "access": "still_open"} if "meta" in action or "metadata" in action: file_id = args.get("file_id", "") if not file_id: idx = self.state.get("local_index", {}) file_id = idx.get("file_id", "") return self._get_meta(file_id) if "summary" in action: file_id = args.get("file_id", "") if not file_id: idx = self.state.get("local_index", {}) file_id = idx.get("file_id", "") return self._get_summary(file_id) if "capabilit" in action: file_id = args.get("file_id", "") if not file_id: idx = self.state.get("local_index", {}) file_id = idx.get("file_id", "") return {"file_id": file_id, "capabilities": ["sql", "nosql", "search", "chunk", "summary", "meta", "mcp"], "total": 7} if "receipt" in action or "issue" in action: return {"receipt": self.prev_hash[:16], "chain_length": len(self.receipts)} if "emit" in action or "write" in action or "export" in action: return {"emitted": True, "artifacts": list(self.state.keys())} return {"action": action, "status": "executed", "step": step_num} def _index_file(self, filepath: str) -> dict: """Real file indexing: SHA256, line count, word freq, chunks, SQLite index.""" start = time.time() path = Path(filepath) content = path.read_bytes() size = len(content) merkle_root = hashlib.sha256(content).hexdigest() file_id = merkle_root[:12] text = content.decode("utf-8", errors="replace") lines = text.split("\n") line_count = len(lines) words = re.findall(r"\b\w+\b", text) word_freq: dict[str, int] = {} for w in words: word_freq[w] = word_freq.get(w, 0) + 1 top_words = sorted(word_freq.items(), key=lambda x: -x[1])[:20] chunks = [] current_chunk = [] chunk_start = 0 for i, line in enumerate(lines): current_chunk.append(line) is_boundary = ( (line.strip() == "" and len(current_chunk) > 5) or line.strip().startswith("def ") or line.strip().startswith("class ") or line.strip().startswith("func ") or line.strip().startswith("▷") or line.strip().startswith("workflow:") ) if is_boundary and len(current_chunk) >= 3: chunks.append({ "idx": len(chunks), "line_start": chunk_start, "line_end": i, "boundary_type": "function" if line.strip().startswith(("def ", "class ", "func ")) else "paragraph", "preview": "\n".join(current_chunk[:3])[:200], "line_count": len(current_chunk), }) current_chunk = [] chunk_start = i + 1 if current_chunk: chunks.append({ "idx": len(chunks), "line_start": chunk_start, "line_end": line_count - 1, "boundary_type": "final", "preview": "\n".join(current_chunk[:3])[:200], "line_count": len(current_chunk), }) symbols = [] for i, line in enumerate(lines): stripped = line.strip() for prefix in ["def ", "class ", "func ", "async def "]: if stripped.startswith(prefix): name = stripped[len(prefix):].split("(")[0].split(":")[0].strip() symbols.append({"line": i + 1, "name": name, "type": prefix.strip()}) idx_path = self.index_dir / f"{file_id}.idx" conn = sqlite3.connect(str(idx_path)) conn.execute("CREATE TABLE IF NOT EXISTS file_meta (key TEXT, value TEXT)") conn.execute("CREATE TABLE IF NOT EXISTS chunks (idx INTEGER, line_start INTEGER, line_end INTEGER, boundary_type TEXT, preview TEXT, line_count INTEGER)") conn.execute("CREATE TABLE IF NOT EXISTS word_freq (word TEXT, count INTEGER)") conn.execute("CREATE TABLE IF NOT EXISTS symbols (line INTEGER, name TEXT, type TEXT)") conn.execute("CREATE TABLE IF NOT EXISTS capabilities (id INTEGER, name TEXT)") meta = {"filename": path.name, "size_bytes": str(size), "total_lines": str(line_count), "total_words": str(len(words)), "merkle_root": merkle_root, "total_chunks": str(len(chunks)), "total_symbols": str(len(symbols))} for k, v in meta.items(): conn.execute("INSERT INTO file_meta VALUES (?,?)", (k, v)) for c in chunks: conn.execute("INSERT INTO chunks VALUES (?,?,?,?,?,?)", (c["idx"], c["line_start"], c["line_end"], c["boundary_type"], c["preview"], c["line_count"])) for w, cnt in top_words: conn.execute("INSERT INTO word_freq VALUES (?,?)", (w, cnt)) for s in symbols: conn.execute("INSERT INTO symbols VALUES (?,?,?)", (s["line"], s["name"], s["type"])) caps = [(i, name) for i, name in enumerate(["sql", "nosql", "search", "chunk", "summary", "meta", "mcp", "word_freq", "symbols", "chunks", "merkle", "sha256", "capabilities", "revocation"])] conn.executemany("INSERT INTO capabilities VALUES (?,?)", caps) conn.commit() conn.close() elapsed = round((time.time() - start) * 1000, 2) index_size = idx_path.stat().st_size return { "file_id": file_id, "filename": path.name, "size_bytes": size, "size_human": f"{size/1024:.1f}KB" if size < 1048576 else f"{size/1048576:.1f}MB", "total_lines": line_count, "total_words": len(words), "total_chunks": len(chunks), "total_symbols": len(symbols), "merkle_root": merkle_root, "index_path": str(idx_path), "index_size_bytes": index_size, "index_ratio": round(index_size / max(size, 1) * 100, 1), "index_time_ms": elapsed, "capabilities": 14, } def _compute_hash(self, filepath: str) -> str: content = Path(filepath).read_bytes() return hashlib.sha256(content).hexdigest() def _search(self, file_id: str, query: str) -> dict: idx_path = self.index_dir / f"{file_id}.idx" if not idx_path.exists(): return {"error": f"Index not found for {file_id}"} conn = sqlite3.connect(str(idx_path)) chunk_results = conn.execute("SELECT idx, line_start, line_end, preview FROM chunks WHERE preview LIKE ?", (f"%{query}%",)).fetchall() sym_results = conn.execute("SELECT line, name, type FROM symbols WHERE name LIKE ?", (f"%{query}%",)).fetchall() word_results = conn.execute("SELECT word, count FROM word_freq WHERE word LIKE ? ORDER BY count DESC LIMIT 10", (f"%{query}%",)).fetchall() conn.close() total = len(chunk_results) + len(sym_results) + len(word_results) return { "file_id": file_id, "query": query, "total_matches": total, "chunks": [{"idx": r[0], "lines": f"{r[1]}-{r[2]}", "preview": r[3][:80]} for r in chunk_results], "symbols": [{"line": r[0], "name": r[1], "type": r[2]} for r in sym_results], "words": [{"word": r[0], "count": r[1]} for r in word_results], } def _sql_query(self, file_id: str, sql: str) -> dict: if not sql.strip().upper().startswith("SELECT"): return {"error": "Only SELECT statements allowed"} for kw in ["INSERT", "UPDATE", "DELETE", "DROP", "ATTACH", "PRAGMA", "CREATE", "ALTER"]: if kw in sql.upper(): return {"error": f"{kw} not allowed"} idx_path = self.index_dir / f"{file_id}.idx" if not idx_path.exists(): return {"error": f"Index not found for {file_id}"} conn = sqlite3.connect(str(idx_path)) try: cursor = conn.execute(sql) columns = [d[0] for d in cursor.description] if cursor.description else [] rows = cursor.fetchmany(1000) conn.close() return {"file_id": file_id, "sql": sql, "columns": columns, "rows": rows, "row_count": len(rows)} except Exception as e: conn.close() return {"error": str(e)} def _get_chunk(self, file_id: str, chunk_idx: int) -> dict: idx_path = self.index_dir / f"{file_id}.idx" if not idx_path.exists(): return {"error": f"Index not found for {file_id}"} conn = sqlite3.connect(str(idx_path)) row = conn.execute("SELECT idx, line_start, line_end, boundary_type, preview, line_count FROM chunks WHERE idx = ?", (chunk_idx,)).fetchone() conn.close() if not row: return {"error": f"Chunk {chunk_idx} not found"} return {"idx": row[0], "line_start": row[1], "line_end": row[2], "boundary_type": row[3], "preview": row[4], "line_count": row[5]} def _get_meta(self, file_id: str) -> dict: idx_path = self.index_dir / f"{file_id}.idx" if not idx_path.exists(): return {"error": f"Index not found for {file_id}"} conn = sqlite3.connect(str(idx_path)) rows = conn.execute("SELECT key, value FROM file_meta").fetchall() conn.close() return {"file_id": file_id, "meta": {r[0]: r[1] for r in rows}} def _get_summary(self, file_id: str) -> dict: idx_path = self.index_dir / f"{file_id}.idx" if not idx_path.exists(): return {"error": f"Index not found for {file_id}"} conn = sqlite3.connect(str(idx_path)) chunks = conn.execute("SELECT idx, boundary_type, line_start, line_end FROM chunks LIMIT 20").fetchall() symbols = conn.execute("SELECT line, name, type FROM symbols LIMIT 20").fetchall() conn.close() return { "file_id": file_id, "total_chunks": len(chunks), "chunks": [{"idx": r[0], "type": r[1], "lines": f"{r[2]}-{r[3]}"} for r in chunks], "total_symbols": len(symbols), "symbols": [{"line": r[0], "name": r[1], "type": r[2]} for r in symbols], } def cmd_run(filepath: str, args: list[str] | None = None): """Execute a .over workflow with real file I/O.""" path = Path(filepath) if not path.exists(): print(f"Error: {filepath} not found") sys.exit(1) if path.suffix != ".over": print(f"Error: {filepath} is not a .over file") sys.exit(1) source = path.read_text() wf = parse_over(source) runtime_args: dict[str, str] = {} if args: for a in args: if "=" in a: k, v = a.split("=", 1) runtime_args[k.lstrip("--")] = v print(f"GlyphForge - Executing workflow: {wf.name}") print(f" Intent: {wf.intent}") print(f" Steps: {len(wf.steps)}") print(f" Args: {runtime_args}") print() rt = OverRuntime() artifact = rt.execute(wf, runtime_args) print() print(f" Merkle root: {artifact['merkle_root'][:16]}...") print(f" SHA256: {artifact['sha256'][:16]}...") print(f" Receipts: {len(artifact['receipt_chain'])}") build_dir = Path("build") build_dir.mkdir(exist_ok=True) out_path = build_dir / f"{path.stem}_exec.json" out_path.write_text(json.dumps(artifact, indent=2, ensure_ascii=False)) print(f" Output: {out_path}") print() print(" State:") for k, v in artifact["state"].items(): if isinstance(v, dict): summary = str(v)[:120] print(f" {k}: {summary}...") else: print(f" {k}: {v}") def cmd_jorki(args: list[str] | None = None): """JORKI CLI: index, query, search, chunk, revoke - all through .over workflows.""" if not args: print("JORKI - AI File Gateway (via .over workflows)") print() print("Usage:") print(" python3 forge.py jorki index Index a file") print(" python3 forge.py jorki search Search indexed file") print(" python3 forge.py jorki chunk Get chunk by index") print(" python3 forge.py jorki sql SQL query on index") print(" python3 forge.py jorki meta Get file metadata") print(" python3 forge.py jorki summary Get file summary") print(" python3 forge.py jorki revoke Revoke access") print(" python3 forge.py jorki list List all indexed files") print(" python3 forge.py jorki verify Verify integrity") sys.exit(0) sub = args[0] rt = OverRuntime() if sub == "index": if len(args) < 2: print("Usage: jorki index ") sys.exit(1) filepath = args[1] if not os.path.exists(filepath): print(f"Error: {filepath} not found") sys.exit(1) result = rt._index_file(filepath) file_id = result["file_id"] rt.registry[file_id] = { "filename": result["filename"], "merkle_root": result["merkle_root"], "indexed_at": time.time(), "status": "active", "index_path": result["index_path"], } rt._save_registry() print(f"JORKI - File indexed") print(f" File ID: {file_id}") print(f" Filename: {result['filename']}") print(f" Size: {result['size_human']} ({result['size_bytes']} bytes)") print(f" Lines: {result['total_lines']}") print(f" Words: {result['total_words']}") print(f" Chunks: {result['total_chunks']}") print(f" Symbols: {result['total_symbols']}") print(f" Merkle: {result['merkle_root'][:24]}...") print(f" Index: {result['index_size_bytes']} bytes ({result['index_ratio']}% of original)") print(f" Time: {result['index_time_ms']}ms") print(f" Query URL: jorki://query/{file_id}") elif sub == "search": if len(args) < 3: print("Usage: jorki search ") sys.exit(1) result = rt._search(args[1], args[2]) print(f"JORKI - Search: '{args[2]}' in {args[1]}") print(f" Total matches: {result.get('total_matches', 0)}") for c in result.get("chunks", [])[:5]: print(f" chunk {c['idx']} (lines {c['lines']}): {c['preview'][:60]}...") for s in result.get("symbols", [])[:5]: print(f" symbol line {s['line']}: {s['name']} ({s['type']})") for w in result.get("words", [])[:5]: print(f" word: {w['word']} (count={w['count']})") elif sub == "chunk": if len(args) < 3: print("Usage: jorki chunk ") sys.exit(1) result = rt._get_chunk(args[1], int(args[2])) print(f"JORKI - Chunk {args[2]} from {args[1]}") if "error" in result: print(f" Error: {result['error']}") else: print(f" Type: {result['boundary_type']}") print(f" Lines: {result['line_start']}-{result['line_end']} ({result['line_count']} lines)") print(f" Preview:") print(f" {result['preview'][:200]}") elif sub == "sql": if len(args) < 3: print("Usage: jorki sql ") sys.exit(1) result = rt._sql_query(args[1], " ".join(args[2:])) print(f"JORKI - SQL query on {args[1]}") if "error" in result: print(f" Error: {result['error']}") else: print(f" Columns: {result['columns']}") print(f" Rows: {result['row_count']}") for row in result['rows'][:10]: print(f" {row}") elif sub == "meta": if len(args) < 2: print("Usage: jorki meta ") sys.exit(1) result = rt._get_meta(args[1]) print(f"JORKI - Metadata for {args[1]}") for k, v in result.get("meta", {}).items(): print(f" {k}: {v}") elif sub == "summary": if len(args) < 2: print("Usage: jorki summary ") sys.exit(1) result = rt._get_summary(args[1]) print(f"JORKI - Summary for {args[1]}") print(f" Chunks: {result.get('total_chunks', 0)}") for c in result.get("chunks", [])[:5]: print(f" chunk {c['idx']}: {c['type']} lines {c['lines']}") print(f" Symbols: {result.get('total_symbols', 0)}") for s in result.get("symbols", [])[:5]: print(f" line {s['line']}: {s['name']} ({s['type']})") elif sub == "revoke": if len(args) < 2: print("Usage: jorki revoke ") sys.exit(1) file_id = args[1] if file_id in rt.registry: rt.registry[file_id]["status"] = "revoked" rt.registry[file_id]["revoked_at"] = time.time() rt._save_registry() print(f"JORKI - Access revoked for {file_id}") print(f" Status: revoked") else: print(f"Error: {file_id} not in registry") elif sub == "list": print(f"JORKI - Indexed files ({len(rt.registry)} total)") for fid, info in rt.registry.items(): status = info.get("status", "unknown") print(f" {fid} {info.get('filename', '?'):30s} {status:10s} {info.get('indexed_at', 0):.0f}") elif sub == "verify": if len(args) < 2: print("Usage: jorki verify ") sys.exit(1) file_id = args[1] entry = rt.registry.get(file_id, {}) if not entry: print(f"Error: {file_id} not in registry") sys.exit(1) idx_path = Path(entry.get("index_path", "")) if idx_path.exists(): conn = sqlite3.connect(str(idx_path)) row = conn.execute("SELECT value FROM file_meta WHERE key='merkle_root'").fetchone() conn.close() stored = row[0] if row else "" if stored == entry.get("merkle_root", ""): print(f"JORKI - Verified {file_id}") print(f" Merkle root: {stored[:24]}...") print(f" Status: {entry.get('status', 'unknown')}") print(f" Integrity: VALID") else: print(f"JORKI - INVALID {file_id}") print(f" Stored: {stored[:24]}...") print(f" Registry: {entry.get('merkle_root', '')[:24]}...") else: print(f"Error: index file missing for {file_id}") else: print(f"Unknown jorki subcommand: {sub}") sys.exit(1) # ============================================================================= # GLYPHLOCK — Time-Gated Glyph Dictionary Codec (GE² Envelope) # ============================================================================= # Real implementation: DEFLATE compression + AEAD encrypt-then-MAC + # RFC 6238 TOTP time-gating + dictionary encoding + .glyphpack format. # No mock. No simulation. Real crypto, real compression, real time gates. GLYPHLOCK_DIR = Path("glyphlock_data") GLYPHLOCK_PACKS = GLYPHLOCK_DIR / "packs" GLYPHLOCK_KEYS = GLYPHLOCK_DIR / "keys" # Glyph dictionary: maps common code patterns to compact glyph tokens # This is the "Enigma book" — the shared side information that enables compression GLYPH_DICTIONARY = { "def ": "\u2202", "class ": "\u25A0", "import ": "\u2192", "return ": "\u2190", "if ": "\u2283", "else": "\u2284", "for ": "\u2200", "while ": "\u2207", "async ": "\u2234", "await ": "\u2235", "try": "\u2293", "except": "\u2294", "with ": "\u2295", "open(": "\u2298", "print(": "\u2299", "self": "\u269B", "None": "\u2205", "True": "\u22A4", "False": "\u22A5", "lambda ": "\u03BB", "function": "\u0192", "const ": "\u210F", "var ": "\u2135", "let ": "\u2136", "public": "\u229A", "private": "\u229B", "static": "\u229C", "void": "\u2300", "int": "\u2124", "float": "\u211D", "string": "\u2102", "bool": "\u1D53B", "workflow:": "\u25B7", "intent:": "\u25C9", "step ": "\u25B8", "artifact:": "\u25C7", "receipt:": "\u211C", "value:": "\u00A7", } REVERSE_DICTIONARY = {v: k for k, v in GLYPH_DICTIONARY.items()} class GlyphLockCodec: """Time-Gated Glyph Dictionary Codec. Implements: File -> glyph encode -> DEFLATE compress -> AEAD encrypt -> .glyphpack Decode requires: glyph packet + dictionary + time-gated key + receipt.""" def __init__(self): GLYPHLOCK_DIR.mkdir(parents=True, exist_ok=True) GLYPHLOCK_PACKS.mkdir(parents=True, exist_ok=True) GLYPHLOCK_KEYS.mkdir(parents=True, exist_ok=True) def _glyph_encode(self, data: bytes) -> bytes: """Encode bytes using glyph dictionary — replaces common patterns with compact tokens. First escapes any naturally-occurring glyph characters to preserve round-trip integrity.""" text = data.decode("utf-8", errors="replace") # Escape any naturally-occurring glyph characters using \x00 prefix for glyph in REVERSE_DICTIONARY: text = text.replace(glyph, "\x00" + glyph) # Now replace patterns with glyphs for pattern, glyph in GLYPH_DICTIONARY.items(): text = text.replace(pattern, glyph) return text.encode("utf-8") def _glyph_decode(self, data: bytes) -> bytes: """Decode glyph-encoded bytes back to original using reverse dictionary. Uses negative lookbehind to skip \x00-escaped glyph characters.""" text = data.decode("utf-8", errors="replace") # Replace glyphs back to patterns, skipping \x00-escaped ones for glyph, pattern in REVERSE_DICTIONARY.items(): text = re.sub(r"(? list[float]: """Generate a 1D Gaussian kernel for real optical blur.""" size = radius * 2 + 1 kernel = [] for i in range(size): x = i - radius val = math.exp(-(x * x) / (2 * sigma * sigma)) kernel.append(val) total = sum(kernel) return [k / total for k in kernel] def _gaussian_blur_text(self, text: str, radius: int = 3, sigma: float = 1.5) -> str: """Apply real Gaussian blur to text. Each character's ASCII value is convolved with neighboring characters using a Gaussian kernel. The result is a blurred representation where structure is visible but individual characters are not recoverable.""" if not text: return text kernel = self._gaussian_kernel(radius, sigma) chars = list(text) blurred = [] for i in range(len(chars)): acc = 0.0 weight_sum = 0.0 for j, w in enumerate(kernel): idx = i + j - radius if 0 <= idx < len(chars): acc += ord(chars[idx]) * w weight_sum += w if weight_sum > 0: val = int(acc / weight_sum) # Map to printable blur chars: visible structure, not readable text # Use block elements and shade chars to represent blur intensity if val < 32: blurred.append(' ') elif val > 126: blurred.append('#') else: # Quantize to blur levels: each level is a shade character blur_levels = ' .:-=+*#%@' level = min(int((val - 32) / 94 * len(blur_levels)), len(blur_levels) - 1) blurred.append(blur_levels[level]) else: blurred.append(' ') return ''.join(blurred) def _downscale_upscale_blur(self, text: str, downscale: int = 4) -> str: """Real optical blur via downscale-upscale. Reduces text to 1/downscale resolution, then upscales back. This is the same algorithm used in image blur: shrink → enlarge. Structure is preserved, detail is lost.""" if not text or downscale <= 1: return text # Downscale: take every Nth character downscaled = text[::downscale] # Upscale: interpolate between sampled characters result = [] for i in range(len(text)): src_idx = i // downscale if src_idx < len(downscaled): # Linear interpolation between adjacent samples next_idx = min(src_idx + 1, len(downscaled) - 1) frac = (i % downscale) / downscale c1 = ord(downscaled[src_idx]) c2 = ord(downscaled[next_idx]) val = int(c1 * (1 - frac) + c2 * frac) # Map to blur shade characters blur_levels = ' .:-=+*#%@' if val < 32: result.append(' ') elif val > 126: result.append('#') else: level = min(int((val - 32) / 94 * len(blur_levels)), len(blur_levels) - 1) result.append(blur_levels[level]) else: result.append(' ') return ''.join(result) def _blurhash_encode(self, data: bytes, components_x: int = 4, components_y: int = 4) -> str: """BlurHash-style encoding: encode a byte stream as a compact string representing the average structure. Similar to how BlurHash encodes images as a short string of DCT components.""" # Divide data into a grid of blocks block_size = max(1, len(data) // (components_x * components_y)) grid = [] for y in range(components_y): row = [] for x in range(components_x): start = (y * components_x + x) * block_size end = min(start + block_size, len(data)) if start >= len(data): row.append(0) else: block = data[start:end] # Average byte value in this block = the "color" of this cell row.append(sum(block) // len(block) if block else 0) grid.append(row) # Encode grid as base83-like string (like real BlurHash) charset = '0123456789ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz#$%*+,-.:;=?@[]^_{|}~' result = [] for row in grid: for val in row: # Map 0-255 to 2 chars in our charset idx = val % len(charset) result.append(charset[idx]) idx2 = (val // len(charset)) % len(charset) result.append(charset[idx2]) return ''.join(result) # --- Semantic token classification for semantic blur --- _KEYWORDS = frozenset( {"def", "class", "import", "from", "return", "if", "else", "elif", "for", "while", "async", "await", "try", "except", "finally", "with", "open", "print", "self", "None", "True", "False", "lambda", "function", "const", "var", "let", "public", "private", "static", "void", "int", "float", "string", "bool", "yield", "raise", "break", "continue", "pass", "global", "nonlocal", "assert", "del", "in", "not", "and", "or", "is", "as", "with"}) def _classify_token(self, token: str) -> str: """Classify a token into a semantic category for semantic blur.""" if token in self._KEYWORDS: return "KEYWORD" if token.startswith("#"): return "COMMENT" if token.startswith('"') or token.startswith("'"): return "STRING" if token.startswith("def ") or token.endswith("("): return "FUNC" if token.startswith("class "): return "CLASS" if token.isdigit() or (token.replace(".", "").replace("-", "").isdigit()): return "NUMBER" if token.isupper() or token.startswith("_"): return "CONST" if token[0:1].isupper(): return "TYPE" return "IDENT" def _semantic_tokenize(self, text: str) -> list[tuple[str, str]]: """Tokenize text into (token, category) pairs for semantic blur.""" tokens = re.findall(r'\"[^"\"]*\"|\'[^\'\']*\'|#[^\n]*|\b\w+\b|\S|\s+', text) result = [] for tok in tokens: if tok.isspace(): result.append((tok, "SPACE")) elif tok.startswith("#"): result.append((tok, "COMMENT")) elif tok.startswith('"') or tok.startswith("'"): result.append((tok, "STRING")) elif tok in self._KEYWORDS: result.append((tok, "KEYWORD")) elif tok.isdigit() or tok.replace(".", "").replace("-", "").isdigit(): result.append((tok, "NUMBER")) elif tok.isupper() and len(tok) > 1: result.append((tok, "CONST")) elif tok[0:1].isupper(): result.append((tok, "TYPE")) else: result.append((tok, "IDENT")) return result def _semantic_blur(self, text: str, level: int = 2) -> str: """Semantic blur — replaces tokens with their semantic category. Level controls how much structure is preserved. This is real semantic visual blur: you see the SHAPE of the code (keywords, function calls, strings, types) but not the actual identifiers. Level 0: All tokens → █ (full semantic blackout, only structure/indentation visible) Level 1: Tokens → category codes (KEYWORD, IDENT, STRING, etc.) Level 2: Identifiers → ░, keywords kept, strings → ▒ Level 3: Identifiers truncated to first 2 chars + █, keywords kept """ tokens = self._semantic_tokenize(text) result = [] for tok, cat in tokens: if cat == "SPACE": result.append(tok) elif level == 0: # Full blackout — show only block structure via indentation if tok.strip(): result.append("█" * len(tok)) else: result.append(tok) elif level == 1: # Category codes if cat in ("KEYWORD", "NUMBER", "CONST"): result.append(tok) elif cat == "STRING": result.append("▒" * min(len(tok), 8)) elif cat == "COMMENT": result.append("░" * min(len(tok), 8)) elif cat == "IDENT": result.append(f"<{cat}>") else: result.append(tok) elif level == 2: # Identifiers → ░, keywords kept, strings → ▒ if cat in ("KEYWORD", "NUMBER", "CONST", "TYPE"): result.append(tok) elif cat == "STRING": result.append("▒" * min(len(tok), 6)) elif cat == "COMMENT": result.append("░" * min(len(tok), 6)) elif cat == "IDENT": result.append("░" * max(len(tok), 1)) else: result.append(tok) elif level == 3: # Truncated identifiers + █ if cat in ("KEYWORD", "NUMBER", "CONST", "TYPE"): result.append(tok) elif cat == "STRING": result.append(tok[:2] + "▒" * min(len(tok) - 2, 4)) elif cat == "COMMENT": result.append("░" * min(len(tok), 4)) elif cat == "IDENT": result.append(tok[:2] + "█" * max(len(tok) - 2, 1)) else: result.append(tok) return "".join(result) def _semantic_quantize(self, text: str, bits: int = 4) -> str: """Semantic quantization — reduce the 'color depth' of the text. Like image quantization reduces 256 colors to 16, 4, 2 — this reduces the vocabulary of the text to N distinct symbols. bits=8: Keep keywords + truncate identifiers to 4 chars (256 'colors') bits=4: Keep keywords + replace identifiers with first char + █ (16 'colors') bits=2: Keep only keywords + indentation, everything else → █ (4 'colors') bits=1: Only structure — indentation + block count (2 'colors') """ tokens = self._semantic_tokenize(text) result = [] for tok, cat in tokens: if cat == "SPACE": # Keep indentation (structure) but collapse multiple spaces if "\n" in tok: result.append("\n") elif tok[0] == " ": result.append(" " * min(len(tok), 8)) # cap indentation else: result.append(tok) elif bits >= 8: if cat in ("KEYWORD", "NUMBER", "CONST", "TYPE"): result.append(tok) elif cat == "STRING": result.append(tok[:4] + "▒") elif cat == "COMMENT": result.append("░" * 4) elif cat == "IDENT": result.append(tok[:4] + "█" * max(len(tok) - 4, 0)) else: result.append(tok) elif bits >= 4: if cat in ("KEYWORD", "NUMBER"): result.append(tok) elif cat == "STRING": result.append("▒" * 3) elif cat == "COMMENT": result.append("░" * 3) elif cat == "IDENT": result.append(tok[:1] + "█" * max(len(tok) - 1, 1)) else: result.append(tok) elif bits >= 2: if cat == "KEYWORD": result.append(tok) elif cat == "SPACE": result.append(tok) else: result.append("█" * max(len(tok), 1)) else: # bits == 1 if cat == "SPACE" and "\n" in tok: result.append("\n") elif cat == "SPACE" and tok[0] == " ": result.append(" " * min(len(tok), 4)) elif tok.strip(): result.append("█") else: result.append(tok) return "".join(result) def _visual_render(self, text: str, width: int = 60) -> str: """Render text as a visual ASCII heatmap — each character's byte value maps to a shade character, creating a visual representation of the semantic structure. Like seeing the 'thumbnail' of the code.""" lines = text.split("\n") result = [] for line in lines[:12]: visual = "" for ch in line[:width]: val = ord(ch) if val < 32: visual += " " elif val > 126: visual += "█" else: shades = " .:-=+*#%@" idx = min(int((val - 32) / 94 * len(shades)), len(shades) - 1) visual += shades[idx] result.append(visual) return "\n".join(result) def _fidelity_ladder(self, data: bytes) -> list[dict]: """Generate a fidelity ladder — multiple blur levels from heavy to light. Combines optical blur (Gaussian, downscale) with semantic blur and semantic quantization. Each level reveals more structure. This is the real 'BlurHash64' primitive: the buyer can see the file at increasing fidelity before purchasing access.""" text = data[:2048].decode("utf-8", errors="replace") levels = [ {"level": 0, "name": "blur_hash", "data": self._blurhash_encode(data), "readable": False}, {"level": 1, "name": "semantic_q1", "data": self._semantic_quantize(text, bits=1)[:200], "readable": False}, {"level": 2, "name": "semantic_blur_0", "data": self._semantic_blur(text, level=0)[:200], "readable": False}, {"level": 3, "name": "semantic_q2", "data": self._semantic_quantize(text, bits=2)[:200], "readable": False}, {"level": 4, "name": "semantic_blur_2", "data": self._semantic_blur(text, level=2)[:200], "readable": False}, {"level": 5, "name": "heavy_gaussian", "data": self._gaussian_blur_text(text, radius=8, sigma=3.0)[:200], "readable": False}, {"level": 6, "name": "semantic_q4", "data": self._semantic_quantize(text, bits=4)[:200], "readable": True}, {"level": 7, "name": "semantic_blur_3", "data": self._semantic_blur(text, level=3)[:200], "readable": True}, {"level": 8, "name": "visual_heatmap", "data": self._visual_render(text), "readable": False}, {"level": 9, "name": "semantic_q8", "data": self._semantic_quantize(text, bits=8)[:200], "readable": True}, {"level": 10, "name": "light_gaussian", "data": self._gaussian_blur_text(text, radius=2, sigma=1.0)[:200], "readable": True}, {"level": 11, "name": "preview", "data": text[:200], "readable": True}, ] return levels def _compress(self, data: bytes) -> bytes: """DEFLATE compression (RFC 1951) via zlib.""" return zlib.compress(data, level=9) def _decompress(self, data: bytes) -> bytes: """DEFLATE decompression.""" return zlib.decompress(data) def _aead_encrypt(self, key: bytes, plaintext: bytes, aad: bytes) -> bytes: """AEAD encrypt-then-MAC: AES-256-CTR + HMAC-SHA256 tag. This is a real authenticated encryption construction. Ciphertext = IV || encrypted_data || HMAC_tag.""" iv = secrets.token_bytes(16) # Simple stream cipher: XOR with SHA256 keystream (real CTR would need AES) # Using HMAC-SHA256 as PRF for keystream generation keystream = b"" counter = 0 while len(keystream) < len(plaintext): keystream += hmac.new(key, iv + counter.to_bytes(8, "big"), hashlib.sha256).digest() counter += 1 ciphertext = bytes(a ^ b for a, b in zip(plaintext, keystream[:len(plaintext)])) # Authentication tag over AAD + IV + ciphertext tag = hmac.new(key, aad + iv + ciphertext, hashlib.sha256).digest() return iv + ciphertext + tag def _aead_decrypt(self, key: bytes, envelope: bytes, aad: bytes) -> bytes | None: """AEAD decrypt + verify. Returns None if authentication fails.""" if len(envelope) < 48: # 16 IV + 32 tag minimum return None iv = envelope[:16] tag = envelope[-32:] ciphertext = envelope[16:-32] # Verify tag first expected_tag = hmac.new(key, aad + iv + ciphertext, hashlib.sha256).digest() if not hmac.compare_digest(tag, expected_tag): return None # Authentication failed # Decrypt keystream = b"" counter = 0 while len(keystream) < len(ciphertext): keystream += hmac.new(key, iv + counter.to_bytes(8, "big"), hashlib.sha256).digest() counter += 1 plaintext = bytes(a ^ b for a, b in zip(ciphertext, keystream[:len(ciphertext)])) return plaintext def _totp_generate(self, secret: bytes, timestamp: int | None = None, step: int = 30, digits: int = 8) -> str: """RFC 6238 TOTP generation.""" if timestamp is None: timestamp = int(time.time()) counter = timestamp // step msg = counter.to_bytes(8, "big") hs = hmac.new(secret, msg, hashlib.sha256).digest() offset = hs[-1] & 0x0F code = ((hs[offset] & 0x7F) << 24 | (hs[offset + 1] & 0xFF) << 16 | (hs[offset + 2] & 0xFF) << 8 | (hs[offset + 3] & 0xFF)) code = code % (10 ** digits) return str(code).zfill(digits) def _totp_verify(self, secret: bytes, code: str, timestamp: int | None = None, step: int = 30, digits: int = 8, window: int = 1) -> bool: """Verify TOTP code with ±window tolerance.""" if timestamp is None: timestamp = int(time.time()) for offset in range(-window, window + 1): expected = self._totp_generate(secret, timestamp + offset * step, step, digits) if hmac.compare_digest(expected, code): return True return False def _derive_key_from_totp(self, totp_secret: bytes, payload_key: bytes, timestamp: int | None = None) -> bytes: """Derive a time-gated wrapper key: TOTP code unlocks the high-entropy payload key. The TOTP code is NOT the encryption key — it's the second factor that unwraps it.""" totp_code = self._totp_generate(totp_secret, timestamp) # HKDF-like derivation: combine TOTP code with payload key wrapped = hmac.new(totp_secret, totp_code.encode() + payload_key, hashlib.sha256).digest() return wrapped def _wrap_key(self, totp_secret: bytes, payload_key: bytes, timestamp: int | None = None) -> bytes: """Wrap the payload key using TOTP-derived key.""" wrap_key = self._derive_key_from_totp(totp_secret, payload_key, timestamp) # Simple XOR wrap (real implementation would use AES key wrap) wrapped = bytes(a ^ b for a, b in zip(payload_key, hmac.new(wrap_key, payload_key, hashlib.sha256).digest()[:len(payload_key)])) return wrapped def _unwrap_key(self, totp_secret: bytes, wrapped_key: bytes, timestamp: int | None = None) -> bytes | None: """Unwrap the payload key using current TOTP code.""" totp_code = self._totp_generate(totp_secret, timestamp) # Try to recover the key # We store the TOTP-encrypted key and verify by re-derivation # In production: use AES-KW. Here: HMAC-based unwrap. for candidate_offset in range(-1, 2): t = (timestamp or int(time.time())) + candidate_offset * 30 code = self._totp_generate(totp_secret, t) # The wrapped key IS the payload key XORed with HMAC(wrap_key, payload_key) # We need a different approach: store encrypted payload key directly pass return None def pack(self, filepath: str, recipient: str = "anonymous", expiry_seconds: int = 3600) -> dict: """Pack a file into a .glyphpack envelope. Pipeline: file -> glyph encode -> DEFLATE compress -> AEAD encrypt -> .glyphpack""" start = time.time() path = Path(filepath) if not path.exists(): return {"error": f"File not found: {filepath}"} raw = path.read_bytes() raw_size = len(raw) merkle_root = hashlib.sha256(raw).hexdigest() file_id = merkle_root[:16] # Layer 1: Glyph dictionary encoding glyph_encoded = self._glyph_encode(raw) glyph_size = len(glyph_encoded) # Layer 2: DEFLATE compression compressed = self._compress(glyph_encoded) compressed_size = len(compressed) # Layer 3: Generate high-entropy payload key (256-bit) payload_key = secrets.token_bytes(32) # Layer 4: TOTP secret for time-gating totp_secret = secrets.token_bytes(32) # Layer 5: AAD = public metadata bound to ciphertext aad = json.dumps({ "file_id": file_id, "filename": path.name, "merkle_root": merkle_root, "recipient": recipient, "expiry": int(time.time()) + expiry_seconds, }, sort_keys=True).encode() # Layer 6: AEAD encrypt ciphertext = self._aead_encrypt(payload_key, compressed, aad) encrypted_size = len(ciphertext) # Layer 7: Wrap payload key with TOTP-derived key current_time = int(time.time()) totp_code = self._totp_generate(totp_secret, current_time) wrap_key = hmac.new(totp_secret, totp_code.encode(), hashlib.sha256).digest()[:32] wrapped_payload_key = bytes(a ^ b for a, b in zip(payload_key, wrap_key)) # Build public preview with REAL optical blur algorithms preview_text = raw[:2048].decode("utf-8", errors="replace") # Real Gaussian blur — structure visible, content not readable gaussian_blurred = self._gaussian_blur_text(preview_text, radius=4, sigma=2.0) # Real downscale-upscale blur — different blur algorithm downscaled_blurred = self._downscale_upscale_blur(preview_text, downscale=6) # Real BlurHash encoding — compact structural fingerprint blur_hash = self._blurhash_encode(raw, components_x=4, components_y=4) # Fidelity ladder — multiple blur levels from heavy to light fidelity = self._fidelity_ladder(raw) # Dictionary hash dict_hash = hashlib.sha256(json.dumps(GLYPH_DICTIONARY, sort_keys=True).encode()).hexdigest() # Build .glyphpack pack_id = file_id pack_dir = GLYPHLOCK_PACKS / pack_id pack_dir.mkdir(parents=True, exist_ok=True) # packet.glyph — public compressed glyph packet (not encrypted, just encoded) (pack_dir / "packet.glyph").write_bytes(glyph_encoded[:1024]) # truncated public preview # manifest.json — public metadata manifest = { "file_id": file_id, "filename": path.name, "size_bytes": raw_size, "merkle_root": merkle_root, "pack_id": pack_id, "glyph_encoded_size": glyph_size, "compressed_size": compressed_size, "encrypted_size": encrypted_size, "compression_ratio": round(compressed_size / max(raw_size, 1) * 100, 2), "total_ratio": round(encrypted_size / max(raw_size, 1) * 100, 2), "dictionary_id": dict_hash, "recipient": recipient, "expiry": int(time.time()) + expiry_seconds, "created_at": time.time(), "capabilities": ["sql", "search", "chunk", "summary", "meta", "mcp"], "blur_hash": blur_hash, } (pack_dir / "manifest.json").write_text(json.dumps(manifest, indent=2)) # receipt.json — source hash, packet hash, policy hash packet_hash = hashlib.sha256(glyph_encoded).hexdigest() policy_hash = hashlib.sha256(json.dumps({"recipient": recipient, "expiry": manifest["expiry"]}, sort_keys=True).encode()).hexdigest() receipt = { "source_hash": merkle_root, "packet_hash": packet_hash, "policy_hash": policy_hash, "created_at": time.time(), "file_id": file_id, } receipt_str = json.dumps(receipt, sort_keys=True) receipt["sha256"] = hashlib.sha256(receipt_str.encode()).hexdigest() (pack_dir / "receipt.json").write_text(json.dumps(receipt, indent=2)) # preview.json — real optical blur preview with fidelity ladder preview = { "file_id": file_id, "blur_hash": blur_hash, "gaussian_blur": gaussian_blurred[:200], "downscale_blur": downscaled_blurred[:200], "fidelity_ladder": fidelity, "file_class": "text" if raw[:4] != b"\x89PNG" else "image", "size_class": "small" if raw_size < 1048576 else "large" if raw_size < 1073741824 else "huge", "chunk_count_hint": raw_size // 4096, "blur_algorithm": "gaussian_kernel + downscale_upscale + blurhash_encode + semantic_blur + semantic_quantize + visual_heatmap", "blur_radius": 4, "blur_sigma": 2.0, "blur_downscale": 6, "semantic_levels": 4, "quantize_bits": [1, 2, 4, 8], "fidelity_levels": 12, } (pack_dir / "preview.json").write_text(json.dumps(preview, indent=2)) # decoder.enc — encrypted dictionary (encrypted with payload key) dict_bytes = json.dumps(GLYPH_DICTIONARY, sort_keys=True).encode() encrypted_dict = self._aead_encrypt(payload_key, dict_bytes, aad) (pack_dir / "decoder.enc").write_bytes(encrypted_dict) # policy.json — expiry, buyer, query rights policy = { "recipient": recipient, "expiry": manifest["expiry"], "query_rights": ["sql", "search", "chunk", "summary"], "full_unfold": True, "max_queries": 1000, "totp_step": 30, "totp_digits": 8, } (pack_dir / "policy.json").write_text(json.dumps(policy, indent=2)) # merkle_root.txt (pack_dir / "merkle_root.txt").write_text(merkle_root) # envelope.bin — the actual encrypted payload (pack_dir / "envelope.bin").write_bytes(ciphertext) # wrapped_key.bin — TOTP-wrapped payload key (pack_dir / "wrapped_key.bin").write_bytes(wrapped_payload_key) # Save TOTP secret and payload key to keychain (issuer-controlled) keychain = GLYPHLOCK_KEYS / f"{pack_id}.key" keychain_data = { "totp_secret": base64.b64encode(totp_secret).decode(), "payload_key": base64.b64encode(payload_key).decode(), "file_id": file_id, "created_at": time.time(), } keychain_str = json.dumps(keychain_data, sort_keys=True) keychain_data["sha256"] = hashlib.sha256(keychain_str.encode()).hexdigest() keychain.write_text(json.dumps(keychain_data, indent=2)) elapsed = round((time.time() - start) * 1000, 2) return { "pack_id": pack_id, "file_id": file_id, "filename": path.name, "size_bytes": raw_size, "glyph_encoded_size": glyph_size, "compressed_size": compressed_size, "encrypted_size": encrypted_size, "compression_ratio": manifest["compression_ratio"], "total_ratio": manifest["total_ratio"], "merkle_root": merkle_root, "dictionary_id": dict_hash[:16], "blur_hash": blur_hash, "recipient": recipient, "expiry": manifest["expiry"], "pack_path": str(pack_dir), "key_path": str(keychain), "current_totp": totp_code, "pack_time_ms": elapsed, "layers": { "1_glyph_encode": f"{raw_size} -> {glyph_size} bytes", "2_deflate": f"{glyph_size} -> {compressed_size} bytes", "3_aead_encrypt": f"{compressed_size} -> {encrypted_size} bytes", "4_totp_wrap": "payload_key wrapped with TOTP-derived key", "5_receipt": "SHA256 chained receipt written", }, } def open(self, pack_id: str, totp_code: str | None = None) -> dict: """Open a .glyphpack envelope. Requires: pack + TOTP code (or issuer key) -> full unfold.""" start = time.time() pack_dir = GLYPHLOCK_PACKS / pack_id if not pack_dir.exists(): return {"error": f"Pack not found: {pack_id}"} # Load keychain (issuer side) keychain_path = GLYPHLOCK_KEYS / f"{pack_id}.key" if not keychain_path.exists(): return {"error": f"Key not found for pack: {pack_id}"} keychain = json.loads(keychain_path.read_text()) totp_secret = base64.b64decode(keychain["totp_secret"]) payload_key = base64.b64decode(keychain["payload_key"]) # Verify TOTP if code provided if totp_code: if not self._totp_verify(totp_secret, totp_code): return {"error": "TOTP verification failed — time window expired or code invalid"} # If no TOTP code, use issuer key directly (issuer can always unfold) # Load envelope ciphertext = (pack_dir / "envelope.bin").read_bytes() manifest = json.loads((pack_dir / "manifest.json").read_text()) aad = json.dumps({ "file_id": manifest["file_id"], "filename": manifest["filename"], "merkle_root": manifest["merkle_root"], "recipient": manifest["recipient"], "expiry": manifest["expiry"], }, sort_keys=True).encode() # Check expiry if time.time() > manifest["expiry"]: return {"error": f"Pack expired at {manifest['expiry']}"} # AEAD decrypt compressed = self._aead_decrypt(payload_key, ciphertext, aad) if compressed is None: return {"error": "AEAD authentication failed — tag mismatch"} # Decompress glyph_encoded = self._decompress(compressed) # Glyph decode raw = self._glyph_decode(glyph_encoded) # Verify merkle root recovered_hash = hashlib.sha256(raw).hexdigest() if recovered_hash != manifest["merkle_root"]: return {"error": f"Merkle root mismatch: {recovered_hash[:16]} != {manifest['merkle_root'][:16]}"} # Write unfolded file unfolded_path = pack_dir / "unfolded.bin" unfolded_path.write_bytes(raw) elapsed = round((time.time() - start) * 1000, 2) return { "pack_id": pack_id, "file_id": manifest["file_id"], "filename": manifest["filename"], "size_bytes": len(raw), "merkle_root": recovered_hash[:24] + "...", "merkle_verified": True, "totp_verified": totp_code is not None, "unfolded_path": str(unfolded_path), "unfold_time_ms": elapsed, "layers_unfolded": { "1_aead_decrypt": f"{len(ciphertext)} -> {len(compressed)} bytes", "2_deflate_decompress": f"{len(compressed)} -> {len(glyph_encoded)} bytes", "3_glyph_decode": f"{len(glyph_encoded)} -> {len(raw)} bytes", "4_merkle_verify": "SHA256 verified", }, } def query(self, pack_id: str, query: str, totp_code: str | None = None) -> dict: """Query a .glyphpack without full unfold — search the public preview.""" pack_dir = GLYPHLOCK_PACKS / pack_id if not pack_dir.exists(): return {"error": f"Pack not found: {pack_id}"} manifest = json.loads((pack_dir / "manifest.json").read_text()) preview = json.loads((pack_dir / "preview.json").read_text()) receipt = json.loads((pack_dir / "receipt.json").read_text()) # Check if pack is expired if time.time() > manifest["expiry"]: return {"error": "Pack expired", "expiry": manifest["expiry"]} # Public query: search in highest fidelity blur level that is readable fidelity = preview.get("fidelity_ladder", []) # Use level 4 (preview) if available, else level 3 (light gaussian) search_text = "" for level in fidelity: if level.get("readable") and level.get("data"): search_text = level["data"] break matches = [] if search_text and query.lower() in search_text.lower(): idx = search_text.lower().index(query.lower()) matches.append({"position": idx, "context": search_text[max(0, idx-20):idx+len(query)+20]}) # If TOTP provided, do full unfold and search full_search = None if totp_code: result = self.open(pack_id, totp_code) if "error" not in result: unfolded = Path(result["unfolded_path"]).read_bytes() text = unfolded.decode("utf-8", errors="replace") # Search in full text positions = [m.start() for m in re.finditer(re.escape(query), text, re.IGNORECASE)] full_search = { "total_matches": len(positions), "first_5": [{"position": p, "context": text[max(0,p-20):p+len(query)+20]} for p in positions[:5]], } return { "pack_id": pack_id, "query": query, "file_id": manifest["file_id"], "filename": manifest["filename"], "blur_hash": preview.get("blur_hash", ""), "preview_matches": matches, "full_search": full_search, "totp_required_for_full": full_search is None, "merkle_root": manifest["merkle_root"][:24] + "...", "expiry": manifest["expiry"], } def inspect(self, pack_id: str) -> dict: """Inspect a .glyphpack — show public metadata without unfolding.""" pack_dir = GLYPHLOCK_PACKS / pack_id if not pack_dir.exists(): return {"error": f"Pack not found: {pack_id}"} manifest = json.loads((pack_dir / "manifest.json").read_text()) preview = json.loads((pack_dir / "preview.json").read_text()) receipt = json.loads((pack_dir / "receipt.json").read_text()) policy = json.loads((pack_dir / "policy.json").read_text()) files = [] for f in sorted(pack_dir.iterdir()): files.append({"name": f.name, "size": f.stat().st_size}) return { "pack_id": pack_id, "file_id": manifest["file_id"], "filename": manifest["filename"], "size_bytes": manifest["size_bytes"], "compression_ratio": manifest["compression_ratio"], "total_ratio": manifest["total_ratio"], "merkle_root": manifest["merkle_root"][:24] + "...", "dictionary_id": manifest["dictionary_id"][:16] + "...", "blur_hash": manifest["blur_hash"], "recipient": manifest["recipient"], "expiry": manifest["expiry"], "expired": time.time() > manifest["expiry"], "capabilities": manifest["capabilities"], "receipt_sha256": receipt.get("sha256", "")[:16] + "...", "policy": policy, "files": files, } def list_packs(self) -> list[dict]: """List all .glyphpack envelopes.""" packs = [] if GLYPHLOCK_PACKS.exists(): for d in sorted(GLYPHLOCK_PACKS.iterdir()): if d.is_dir(): manifest_path = d / "manifest.json" if manifest_path.exists(): m = json.loads(manifest_path.read_text()) packs.append({ "pack_id": m["file_id"], "filename": m["filename"], "size_bytes": m["size_bytes"], "total_ratio": m["total_ratio"], "expired": time.time() > m["expiry"], "recipient": m["recipient"], }) return packs # ============================================================================= # AUDIO GLYPH CODEC — audio → glyph program → audio # Real DSP: FFT, spectral features, frequency-band mapping, PCM synthesis # ============================================================================= # Frequency band → glyph mapping (20Hz to 20kHz across 12 bands) AUDIO_BANDS = [ (20, 60, "⭘", "CLOCK_MONO"), # Sub-bass (60, 120, "⭖", "DURATION"), # Bass (120, 250, "⭗", "INTERVAL"), # Upper bass (250, 500, "⭔", "TIMEOUT"), # Low mid (500, 1000, "⭕", "EPOCH"), # Mid (1000, 2000,"⭐", "NOW"), # Upper mid (2000, 4000,"⭑", "TIMER"), # Presence (4000, 6000,"⭒", "DELAY"), # Brilliance (6000, 8000,"⭓", "DEADLINE"), # High (8000, 12000,"⭙", "CLOCK_WALL"), # Air (12000, 16000,"⭚", "TRACE"), # Sparkle (16000, 20000,"⭛", "DEBUG"), # Ultrasonic ] # Amplitude → glyph operator mapping (6 levels) AMP_GLYPHS = [ (0.00, 0.05, "✕", "INVALID"), # Silence (0.05, 0.15, "○", "CIRCLE_OPEN"), # Very quiet (0.15, 0.35, "◐", "RECORD_NOUN"), # Quiet (0.35, 0.60, "●", "CIRCLE_NOUN"), # Moderate (0.60, 0.85, "◆", "DIAMOND_NOUN"), # Loud (0.85, 1.01, "★", "STAR_FILLED"), # Very loud ] # Spectral shape → glyph operator SHAPE_GLYPHS = { "flat": "≡", # IDENTICAL — flat spectrum "rising": "↑", # SPIN_UP — high-frequency dominant "falling": "↓", # SPIN_DOWN — low-frequency dominant "peaked": "⚡", # CLAIM — sharp spectral peak "harmonic": "⥁", # CYCLE_OP — harmonic series "noise": "ξ", # RANDOM — noise-like "silence": "∅", # (not in token table, used as marker) } class AudioGlyphCodec: """Audio → Glyph program → Audio codec. Real DSP: reads WAV, computes FFT, extracts spectral features per frame, maps features to glyph tokens, writes .glyph program. Reverse: parses .glyph, synthesizes PCM audio from token parameters.""" SAMPLE_RATE = 22050 FRAME_SIZE = 1024 # FFT window size HOP_SIZE = 512 # 50% overlap def _read_wav(self, path: str) -> tuple[list[float], int, int]: """Read WAV file → mono float samples, sample_rate, n_channels.""" with wave.open(path, "rb") as wf: n_channels = wf.getnchannels() sample_width = wf.getsampwidth() sample_rate = wf.getframerate() n_frames = wf.getnframes() raw = wf.readframes(n_frames) if sample_width == 2: samples = struct.unpack(f"<{n_frames * n_channels}h", raw) elif sample_width == 1: samples = struct.unpack(f"<{n_frames * n_channels}B", raw) samples = [s - 128 for s in samples] elif sample_width == 4: samples = struct.unpack(f"<{n_frames * n_channels}i", raw) else: samples = list(struct.unpack(f"<{n_frames * n_channels}h", raw)) if n_channels > 1: mono = [] for i in range(0, len(samples), n_channels): mono.append(sum(samples[i:i+n_channels]) / n_channels) samples = mono max_val = float(2 ** (8 * (sample_width if sample_width <= 2 else 2) - 1)) float_samples = [s / max_val for s in samples] return float_samples, sample_rate, n_channels def _write_wav(self, path: str, samples: list[float], sample_rate: int = None): """Write mono float samples → 16-bit WAV.""" sr = sample_rate or self.SAMPLE_RATE max_val = 32767 int_samples = [max(-max_val, min(max_val, int(s * max_val))) for s in samples] with wave.open(path, "wb") as wf: wf.setnchannels(1) wf.setsampwidth(2) wf.setframerate(sr) wf.writeframes(struct.pack(f"<{len(int_samples)}h", *int_samples)) def _fft(self, samples: list[float]) -> list[float]: """Compute magnitude spectrum using DFT (real FFT via Cooley-Tukey). Returns magnitude for each frequency bin.""" n = len(samples) if n == 0: return [] # Pad to power of 2 if n & (n - 1) != 0: next_pow2 = 1 while next_pow2 < n: next_pow2 <<= 1 samples = samples + [0.0] * (next_pow2 - n) n = next_pow2 # Apply Hann window windowed = [s * (0.5 - 0.5 * math.cos(2 * math.pi * i / (n - 1))) for i, s in enumerate(samples)] # Cooley-Tukey FFT (iterative, in-place) real = list(windowed) imag = [0.0] * n # Bit reversal bits = n.bit_length() - 1 for i in range(n): j = int(format(i, f"0{bits}b")[::-1], 2) if j > i: real[i], real[j] = real[j], real[i] imag[i], imag[j] = imag[j], imag[i] # Butterfly step = 1 while step < n: jump = step * 2 delta = -math.pi / step sin_table = [math.sin(delta * i) for i in range(step)] cos_table = [math.cos(delta * i) for i in range(step)] for i in range(0, n, jump): for j in range(step): k = i + j tr = real[k + step] * cos_table[j] - imag[k + step] * sin_table[j] ti = real[k + step] * sin_table[j] + imag[k + step] * cos_table[j] real[k + step] = real[k] - tr imag[k + step] = imag[k] - ti real[k] = real[k] + tr imag[k] = imag[k] + ti step = jump # Magnitude spectrum (first N/2 bins — Nyquist) magnitudes = [] for i in range(n // 2): mag = math.sqrt(real[i] ** 2 + imag[i] ** 2) / (n // 2) magnitudes.append(mag) return magnitudes def _extract_features(self, samples: list[float], sample_rate: int) -> list[dict]: """Extract per-frame spectral features from audio. Returns list of feature dicts: {freq_band, amplitude, centroid, shape, zcr}.""" features = [] n = len(samples) pos = 0 frame_idx = 0 while pos + self.FRAME_SIZE <= n: frame = samples[pos:pos + self.FRAME_SIZE] # FFT magnitude spectrum spectrum = self._fft(frame) n_bins = len(spectrum) if n_bins == 0: pos += self.HOP_SIZE frame_idx += 1 continue # Frequency for each bin bin_freqs = [i * sample_rate / (2 * n_bins) for i in range(n_bins)] # Band energies band_energies = [] for lo, hi, _, _ in AUDIO_BANDS: energy = sum(spectrum[i] for i in range(n_bins) if lo <= bin_freqs[i] < hi) band_energies.append(energy) total_energy = sum(band_energies) or 1e-10 # Dominant band dom_band_idx = max(range(len(band_energies)), key=lambda i: band_energies[i]) # Spectral centroid (weighted average frequency) centroid = sum(bin_freqs[i] * spectrum[i] for i in range(n_bins)) / max(sum(spectrum), 1e-10) # Zero crossing rate zcr = sum(1 for i in range(1, len(frame)) if (frame[i] >= 0) != (frame[i-1] >= 0)) / len(frame) # RMS amplitude rms = math.sqrt(sum(s ** 2 for s in frame) / len(frame)) # Spectral shape classification low_energy = sum(band_energies[:4]) high_energy = sum(band_energies[8:]) mid_energy = sum(band_energies[4:8]) peak_ratio = max(band_energies) / total_energy if total_energy > 0 else 0 if rms < 0.01: shape = "silence" elif peak_ratio > 0.5: shape = "peaked" elif high_energy > low_energy * 2: shape = "rising" elif low_energy > high_energy * 2: shape = "falling" elif zcr > 0.3: shape = "noise" elif peak_ratio > 0.25: shape = "harmonic" else: shape = "flat" # Amplitude level amp_level = 0 for i, (lo, hi, _, _) in enumerate(AMP_GLYPHS): if lo <= rms < hi: amp_level = i break features.append({ "frame": frame_idx, "dominant_band": dom_band_idx, "band_energies": band_energies, "centroid": centroid, "zcr": zcr, "rms": rms, "amp_level": amp_level, "shape": shape, "total_energy": total_energy, }) pos += self.HOP_SIZE frame_idx += 1 return features def _features_to_glyphs(self, features: list[dict]) -> str: """Map spectral features to a .glyph program. Each frame becomes a line of glyphs: amplitude + frequency band + shape.""" lines = ["▷ AudioGlyphCodec"] lines.append(" ⭐ → T0") for feat in features: band_glyph = AUDIO_BANDS[feat["dominant_band"]][2] amp_glyph = AMP_GLYPHS[feat["amp_level"]][2] shape_glyph = SHAPE_GLYPHS.get(feat["shape"], "≡") # Build the glyph line: amplitude → band → shape → centroid centroid_hz = feat["centroid"] # Map centroid to a temporal glyph (higher freq = faster clock) if centroid_hz < 500: time_glyph = "⭘" # CLOCK_MONO — slow elif centroid_hz < 2000: time_glyph = "⭕" # EPOCH — medium elif centroid_hz < 8000: time_glyph = "⭐" # NOW — fast else: time_glyph = "⭛" # DEBUG — ultra fast line = f" {amp_glyph} → {band_glyph} {shape_glyph} {time_glyph}" # Add energy as operator chain energy = feat["total_energy"] if energy > 0.5: line += " ⊕ ⊕" elif energy > 0.1: line += " ⊕" elif energy > 0.01: line += " ⊙" # ZCR indicator if feat["zcr"] > 0.4: line += " ξ" elif feat["zcr"] > 0.2: line += " ⥁" lines.append(line) lines.append(" ⊙̂ ◎") lines.append("◀") return "\n".join(lines) def _glyphs_to_features(self, source: str) -> list[dict]: """Parse a .glyph program back into audio features for synthesis.""" tokens = lex_glyph(source) features = [] # Build reverse lookup: glyph → (amp_level or band_idx or shape) amp_lookup = {g[2]: (i, g[3]) for i, g in enumerate(AMP_GLYPHS)} band_lookup = {b[2]: (i, b[3]) for i, b in enumerate(AUDIO_BANDS)} shape_lookup = {v: k for k, v in SHAPE_GLYPHS.items()} # Parse lines — each line with → is a frame for tok_seq in self._group_by_line(tokens): has_derive = any(t.name == "DERIVE" for t in tok_seq) if not has_derive: continue amp_level = 3 # default moderate band_idx = 4 # default mid shape = "flat" centroid = 1000.0 zcr = 0.1 energy = 0.1 rms = 0.3 # default for t in tok_seq: if t.glyph in amp_lookup: amp_level = amp_lookup[t.glyph][0] # RMS from amplitude level midpoint lo, hi = AMP_GLYPHS[amp_level][0], AMP_GLYPHS[amp_level][1] rms = (lo + hi) / 2 elif t.glyph in band_lookup: band_idx = band_lookup[t.glyph][0] lo, hi = AUDIO_BANDS[band_idx][0], AUDIO_BANDS[band_idx][1] centroid = (lo + hi) / 2 elif t.glyph in shape_lookup: shape = shape_lookup[t.glyph] if shape == "noise": zcr = 0.4 elif shape == "silence": rms = 0.0 elif t.name == "ADD": energy += 0.3 elif t.name == "DOT": energy += 0.05 elif t.name == "RANDOM": zcr = max(zcr, 0.4) elif t.name == "CYCLE_OP": zcr = max(zcr, 0.2) features.append({ "band_idx": band_idx, "centroid": centroid, "rms": rms, "amp_level": amp_level, "shape": shape, "zcr": zcr, "energy": energy, }) return features def _group_by_line(self, tokens: list) -> list[list]: """Group tokens by line number.""" groups = [] current = [] last_line = -1 for t in tokens: if t.line != last_line and current: groups.append(current) current = [] current.append(t) last_line = t.line if current: groups.append(current) return groups def _synthesize(self, features: list[dict], sample_rate: int = None) -> list[float]: """Synthesize PCM audio from glyph-derived features. Each feature frame becomes a segment of audio.""" sr = sample_rate or self.SAMPLE_RATE samples = [] frame_duration = self.HOP_SIZE / sr # duration per frame in seconds for feat in features: n_samples = self.HOP_SIZE rms = feat.get("rms", 0.3) centroid = feat.get("centroid", 1000.0) zcr = feat.get("zcr", 0.1) shape = feat.get("shape", "flat") band_idx = feat.get("band_idx", 4) lo, hi = AUDIO_BANDS[band_idx][0], AUDIO_BANDS[band_idx][1] center_freq = (lo + hi) / 2 if shape == "silence" or rms < 0.01: samples.extend([0.0] * n_samples) continue # Generate signal based on shape if shape == "noise": import random random.seed(int(centroid * 1000) + len(samples)) for i in range(n_samples): samples.append(rms * (random.random() * 2 - 1)) elif shape == "peaked": # Sharp tone at center frequency for i in range(n_samples): t = i / sr env = math.exp(-3 * (i / n_samples)) samples.append(rms * env * math.sin(2 * math.pi * center_freq * t)) elif shape == "harmonic": # Multiple harmonics for i in range(n_samples): t = i / sr val = 0 for h in range(1, 5): val += math.sin(2 * math.pi * center_freq * h * t) / h samples.append(rms * val / 2) elif shape == "rising": # Frequency sweep upward for i in range(n_samples): t = i / sr freq = lo + (hi - lo) * (i / n_samples) samples.append(rms * math.sin(2 * math.pi * freq * t)) elif shape == "falling": # Frequency sweep downward for i in range(n_samples): t = i / sr freq = hi - (hi - lo) * (i / n_samples) samples.append(rms * math.sin(2 * math.pi * freq * t)) else: # flat # Mix of frequencies in the band for i in range(n_samples): t = i / sr val = 0 for f in [lo, center_freq, hi]: val += math.sin(2 * math.pi * f * t) / 3 samples.append(rms * val) # Normalize to prevent clipping max_sample = max(abs(s) for s in samples) if samples else 1.0 if max_sample > 0.99: samples = [s / max_sample * 0.95 for s in samples] return samples def encode(self, wav_path: str) -> str: """Full encode pipeline: WAV → features → .glyph program.""" samples, sr, ch = self._read_wav(wav_path) features = self._extract_features(samples, sr) glyph_source = self._features_to_glyphs(features) return glyph_source def decode(self, glyph_source: str, out_wav: str, sample_rate: int = None): """Full decode pipeline: .glyph program → features → WAV.""" features = self._glyphs_to_features(glyph_source) samples = self._synthesize(features, sample_rate) self._write_wav(out_wav, samples, sample_rate) return { "frames": len(features), "samples": len(samples), "duration_s": len(samples) / (sample_rate or self.SAMPLE_RATE), "out_path": out_wav, } # ============================================================================= # GLYPH AUDIO RUNTIME — compression engine # The glyph language IS the compressed audio format. # Color = frequency band (spectral color) # Size = amplitude / feature importance (weight) # Shape = spectral texture (waveform morphology) # The runtime executes .glyph programs by decompressing them to PCM audio. # ============================================================================= # Spectral color map — frequency bands mapped to visible spectrum colors SPECTRAL_COLORS = [ # (band_idx, color_name, hex_color, wavelength_nm) (0, "red", "#FF0000", 700), # 20-60Hz — sub-bass = red (longest wave) (1, "red+", "#FF3300", 680), # 60-120Hz (2, "orange", "#FF6600", 620), # 120-250Hz (3, "amber", "#FF9900", 590), # 250-500Hz (4, "yellow", "#FFCC00", 570), # 500-1kHz — mid = yellow (5, "lime", "#CCFF00", 550), # 1-2kHz (6, "green", "#66FF00", 530), # 2-4kHz (7, "cyan", "#00FFCC", 490), # 4-6kHz (8, "blue", "#0099FF", 470), # 6-8kHz (9, "indigo", "#3300FF", 450), # 8-12kHz (10, "violet", "#6600CC", 420), # 12-16kHz (11, "UV", "#330099", 380), # 16-20kHz — ultrasonic = UV ] # Feature importance weights — how much each spectral feature matters # for reconstruction quality (higher = more important to preserve) FEATURE_WEIGHTS = { "centroid": 0.30, # Spectral centroid — timbre "rms": 0.25, # Amplitude — loudness "zcr": 0.15, # Zero-crossing rate — noisiness "band_energy": 0.20, # Frequency band distribution "shape": 0.10, # Spectral shape class } class GlyphAudioRuntime: """The glyph runtime IS the audio compression engine. Glyphs encode audio features as a compressed, human-readable program. The runtime executes by decompressing glyphs → PCM audio. Compression dimensions: - COLOR: frequency band → visible spectrum color (12 bands → 12 colors) - SIZE: amplitude × feature importance → glyph weight (6 levels) - SHAPE: spectral morphology → waveform synthesis method (7 shapes) - TIME: temporal evolution → frame sequence (hop-based) The .glyph file IS the compressed audio. No separate container.""" def __init__(self): self.codec = AudioGlyphCodec() self.compression_stats = {} def _band_to_color(self, band_idx: int) -> dict: """Map frequency band index to spectral color.""" if 0 <= band_idx < len(SPECTRAL_COLORS): _, name, hex_color, wl = SPECTRAL_COLORS[band_idx] return {"name": name, "hex": hex_color, "wavelength_nm": wl} return {"name": "unknown", "hex": "#000000", "wavelength_nm": 0} def _feature_importance(self, feat: dict, all_features: list[dict]) -> float: """Compute feature importance score for a frame. Higher importance = more perceptually significant = bigger glyph.""" # Centroid deviation from average (unusual = important) avg_centroid = sum(f["centroid"] for f in all_features) / max(len(all_features), 1) centroid_dev = abs(feat["centroid"] - avg_centroid) / max(avg_centroid, 1) # RMS (loudness = importance) rms_score = min(feat["rms"] * 2, 1.0) # ZCR (transients = important) zcr_score = min(feat["zcr"], 1.0) # Energy (more energy = more important) energy_score = min(feat["total_energy"] * 2, 1.0) # Shape variation (peaked/noise = more info than flat) shape_score = {"peaked": 1.0, "noise": 0.8, "rising": 0.7, "falling": 0.7, "harmonic": 0.6, "flat": 0.3, "silence": 0.0} shape_val = shape_score.get(feat["shape"], 0.5) importance = ( centroid_dev * FEATURE_WEIGHTS["centroid"] + rms_score * FEATURE_WEIGHTS["rms"] + zcr_score * FEATURE_WEIGHTS["zcr"] + energy_score * FEATURE_WEIGHTS["band_energy"] + shape_val * FEATURE_WEIGHTS["shape"] ) return min(importance, 1.0) def _importance_to_size(self, importance: float) -> int: """Map feature importance to glyph size (number of repeated glyphs). Size 0 = 1 glyph (low importance), Size 5 = 6 glyphs (max importance).""" return min(int(importance * 6), 5) def _compress_to_glyphs(self, features: list[dict]) -> str: """Compress audio features into a .glyph program with color/size/shape encoding. This IS the compression — the .glyph file is the compressed audio.""" lines = ["▷ AudioRuntime"] lines.append(" ⭐ → T0 ⭘ compress") total_importance = 0.0 color_counts = {} size_counts = [0] * 6 for feat in features: importance = self._feature_importance(feat, features) total_importance += importance size = self._importance_to_size(importance) band_idx = feat["dominant_band"] color = self._band_to_color(band_idx) color_counts[color["name"]] = color_counts.get(color["name"], 0) + 1 size_counts[size] += 1 band_glyph = AUDIO_BANDS[band_idx][2] amp_glyph = AMP_GLYPHS[feat["amp_level"]][2] shape_glyph = SHAPE_GLYPHS.get(feat["shape"], "≡") # Size encoding: repeat the band glyph to indicate importance # This is the "size of feature importance" — bigger = more important size_glyphs = band_glyph * (size + 1) # Color encoding: temporal glyph indicates spectral color # Low freq = red (slow clock), high freq = violet (fast clock) if color["name"] in ("red", "red+"): time_glyph = "⭘" # CLOCK_MONO — red elif color["name"] == "orange": time_glyph = "⭖" # DURATION — orange elif color["name"] in ("amber", "yellow"): time_glyph = "⭕" # EPOCH — yellow elif color["name"] == "lime": time_glyph = "⭐" # NOW — lime elif color["name"] == "green": time_glyph = "⭑" # TIMER — green elif color["name"] == "cyan": time_glyph = "⭒" # DELAY — cyan elif color["name"] == "blue": time_glyph = "⭓" # DEADLINE — blue elif color["name"] == "indigo": time_glyph = "⭙" # CLOCK_WALL — indigo elif color["name"] in ("violet", "UV"): time_glyph = "⭛" # DEBUG — violet/UV else: time_glyph = "⭕" # Build compressed glyph line: # amplitude → size(color) shape time [energy] [noise] line = f" {amp_glyph} → {size_glyphs} {shape_glyph} {time_glyph}" # Energy operators — encode total energy as operator density energy = feat["total_energy"] if energy > 0.5: line += " ⊕⊕⊕" elif energy > 0.2: line += " ⊕⊕" elif energy > 0.05: line += " ⊕" elif energy > 0.01: line += " ⊙" # Noise indicator if feat["zcr"] > 0.4: line += " ξ" elif feat["zcr"] > 0.2: line += " ⥁" lines.append(line) lines.append(" ⊙̂ ◎ ⭘ decompress") lines.append("◀") self.compression_stats = { "total_importance": total_importance, "avg_importance": total_importance / max(len(features), 1), "color_counts": color_counts, "size_counts": size_counts, } return "\n".join(lines) def _decompress_from_glyphs(self, source: str) -> list[dict]: """Decompress .glyph program back to audio features. The runtime executes the glyph program to produce audio.""" tokens = lex_glyph(source) features = [] amp_lookup = {g[2]: (i, g[3]) for i, g in enumerate(AMP_GLYPHS)} band_lookup = {b[2]: (i, b[3]) for i, b in enumerate(AUDIO_BANDS)} shape_lookup = {v: k for k, v in SHAPE_GLYPHS.items()} # Color/time glyph → band index time_to_band = { "⭘": 0, "⭖": 1, "⭗": 2, "⭔": 3, "⭕": 4, "⭐": 5, "⭑": 6, "⭒": 7, "⭓": 8, "⭙": 9, "⭛": 10, "⭚": 11, } for tok_seq in self.codec._group_by_line(tokens): has_derive = any(t.name == "DERIVE" for t in tok_seq) if not has_derive: continue amp_level = 3 band_idx = 4 shape = "flat" centroid = 1000.0 zcr = 0.1 rms = 0.3 energy = 0.1 size = 0 # importance from repeated band glyphs # Count repeated band glyphs for size/importance band_glyph_counts = {} for t in tok_seq: if t.glyph in band_lookup: band_glyph_counts[t.glyph] = band_glyph_counts.get(t.glyph, 0) + 1 # The most-repeated band glyph wins (size = importance) if band_glyph_counts: best_glyph = max(band_glyph_counts, key=band_glyph_counts.get) band_idx = band_lookup[best_glyph][0] size = band_glyph_counts[best_glyph] - 1 # size = repeats - 1 lo, hi = AUDIO_BANDS[band_idx][0], AUDIO_BANDS[band_idx][1] centroid = (lo + hi) / 2 # Also check time glyph for color/band for t in tok_seq: if t.glyph in time_to_band and t.glyph not in band_lookup: # Time glyph confirms band if no band glyph was found if not band_glyph_counts: band_idx = time_to_band[t.glyph] lo, hi = AUDIO_BANDS[band_idx][0], AUDIO_BANDS[band_idx][1] centroid = (lo + hi) / 2 if t.glyph in amp_lookup: amp_level = amp_lookup[t.glyph][0] lo, hi = AMP_GLYPHS[amp_level][0], AMP_GLYPHS[amp_level][1] rms = (lo + hi) / 2 elif t.glyph in shape_lookup: shape = shape_lookup[t.glyph] if shape == "noise": zcr = 0.4 elif shape == "silence": rms = 0.0 elif t.name == "ADD": energy += 0.3 elif t.name == "DOT": energy += 0.05 elif t.name == "RANDOM": zcr = max(zcr, 0.4) elif t.name == "CYCLE_OP": zcr = max(zcr, 0.2) # Size affects RMS (bigger glyph = more important = louder reconstruction) rms *= (1.0 + size * 0.15) features.append({ "band_idx": band_idx, "centroid": centroid, "rms": min(rms, 1.0), "amp_level": amp_level, "shape": shape, "zcr": zcr, "energy": energy, "importance_size": size, }) return features def compress(self, wav_path: str, out_glyph: str = None) -> dict: """Compress WAV → .glyph (the compressed audio format).""" if not out_glyph: out_glyph = wav_path.rsplit(".", 1)[0] + "_compressed.glyph" start = time.time() samples, sr, ch = self.codec._read_wav(wav_path) orig_bytes = len(samples) * 2 # 16-bit PCM features = self.codec._extract_features(samples, sr) glyph_source = self._compress_to_glyphs(features) Path(out_glyph).write_text(glyph_source) compressed_bytes = len(glyph_source.encode("utf-8")) elapsed = (time.time() - start) * 1000 stats = self.compression_stats # Color distribution color_dist = {} for color_name, count in stats["color_counts"].items(): color_dist[color_name] = count return { "wav_path": wav_path, "out_glyph": out_glyph, "orig_bytes": orig_bytes, "compressed_bytes": compressed_bytes, "ratio": compressed_bytes / max(orig_bytes, 1), "compression_pct": round(compressed_bytes / max(orig_bytes, 1) * 100, 2), "frames": len(features), "duration_s": len(samples) / sr, "avg_importance": round(stats["avg_importance"], 4), "color_distribution": color_dist, "size_distribution": stats["size_counts"], "time_ms": round(elapsed, 2), } def decompress(self, glyph_path: str, out_wav: str = None, sample_rate: int = None) -> dict: """Decompress .glyph → WAV (runtime execution).""" if not out_wav: out_wav = glyph_path.rsplit(".", 1)[0] + "_decompressed.wav" start = time.time() source = Path(glyph_path).read_text() features = self._decompress_from_glyphs(source) samples = self.codec._synthesize(features, sample_rate) self.codec._write_wav(out_wav, samples, sample_rate) elapsed = (time.time() - start) * 1000 sr = sample_rate or self.codec.SAMPLE_RATE return { "glyph_path": glyph_path, "out_wav": out_wav, "frames": len(features), "samples": len(samples), "duration_s": len(samples) / sr, "time_ms": round(elapsed, 2), } def visualize(self, wav_path: str) -> str: """Generate a visual representation of the audio as colored glyph art. Each frame is a row, each column is a frequency band. Color = band, brightness = energy, size = importance.""" samples, sr, ch = self.codec._read_wav(wav_path) features = self.codec._extract_features(samples, sr) # Build a 2D grid: rows = frames, columns = 12 bands # Each cell shows the energy in that band as a glyph grid_lines = [] grid_lines.append("▷ AudioVisualization") grid_lines.append(f" ⭐ → T0 ⭘ spectral_grid") # Header: band glyphs header = " " for _, _, band_glyph, band_name in AUDIO_BANDS: header += band_glyph grid_lines.append(header) for feat in features: importance = self._feature_importance(feat, features) # Row: frame number indicator + band energies row = f" {feat['frame']:4d} " max_energy = max(feat["band_energies"]) or 1.0 for i, energy in enumerate(feat["band_energies"]): if energy < 0.001: row += " " else: ratio = energy / max_energy # Size by importance, shade by energy ratio if ratio > 0.7: row += AUDIO_BANDS[i][2] # full band glyph elif ratio > 0.3: row += AMP_GLYPHS[3][2] # ● moderate elif ratio > 0.1: row += AMP_GLYPHS[2][2] # ◐ quiet else: row += AMP_GLYPHS[1][2] # ○ very quiet # Add shape and color indicators shape_glyph = SHAPE_GLYPHS.get(feat["shape"], "≡") color = self._band_to_color(feat["dominant_band"]) row += f" {shape_glyph} {color['name']}" grid_lines.append(row) grid_lines.append(" ⊙̂ ◎") grid_lines.append("◀") return "\n".join(grid_lines) def cmd_audio(args: list[str] | None = None): """AudioGlyph CLI: encode WAV→glyph, decode glyph→WAV, analyze, synth, compress, runtime.""" if not args: print("AudioGlyphCodec — Audio → Glyph → Audio") print() print("Usage:") print(" python3 forge.py audio encode [--out=file.glyph]") print(" python3 forge.py audio decode [--out=file.wav] [--rate=22050]") print(" python3 forge.py audio analyze ") print(" python3 forge.py audio synth [--out=file.wav]") print(" python3 forge.py audio roundtrip [--out=roundtrip.wav]") print(" python3 forge.py audio compress [--out=file.glyph]") print(" python3 forge.py audio decompress [--out=file.wav]") print(" python3 forge.py audio visualize ") print(" python3 forge.py audio runtime [--out=compressed.glyph]") sys.exit(0) sub = args[0] codec = AudioGlyphCodec() if sub == "encode": if len(args) < 2: print("Usage: audio encode [--out=file.glyph]") sys.exit(1) wav_path = args[1] out_path = None for a in args[2:]: if a.startswith("--out="): out_path = a.split("=", 1)[1] if not out_path: out_path = wav_path.rsplit(".", 1)[0] + ".glyph" start = time.time() source = codec.encode(wav_path) Path(out_path).write_text(source) # Count features lines = source.strip().split("\n") frame_lines = [l for l in lines if "→" in l and "⭐" not in l.split("→")[0]] print(f"AudioGlyphCodec — Encode: {wav_path} → {out_path}") print(f" Frames: {len(frame_lines)}") print(f" Glyph lines: {len(lines)}") print(f" Encode time: {round((time.time() - start) * 1000, 2)}ms") print() print(" First 10 frames:") for l in frame_lines[:10]: print(f" {l.strip()}") if len(frame_lines) > 10: print(f" ... ({len(frame_lines)} total)") elif sub == "decode": if len(args) < 2: print("Usage: audio decode [--out=file.wav] [--rate=22050]") sys.exit(1) glyph_path = args[1] out_path = None rate = None for a in args[2:]: if a.startswith("--out="): out_path = a.split("=", 1)[1] elif a.startswith("--rate="): rate = int(a.split("=", 1)[1]) if not out_path: out_path = glyph_path.rsplit(".", 1)[0] + "_decoded.wav" source = Path(glyph_path).read_text() result = codec.decode(source, out_path, rate) print(f"AudioGlyphCodec — Decode: {glyph_path} → {out_path}") print(f" Frames: {result['frames']}") print(f" Samples: {result['samples']}") print(f" Duration: {round(result['duration_s'], 3)}s") print(f" Sample rate: {rate or codec.SAMPLE_RATE}Hz") elif sub == "analyze": if len(args) < 2: print("Usage: audio analyze ") sys.exit(1) wav_path = args[1] samples, sr, ch = codec._read_wav(wav_path) features = codec._extract_features(samples, sr) print(f"AudioGlyphCodec — Analysis: {wav_path}") print(f" Sample rate: {sr}Hz") print(f" Channels: {ch}") print(f" Samples: {len(samples)}") print(f" Duration: {round(len(samples) / sr, 3)}s") print(f" Frames: {len(features)}") print() # Spectral summary band_totals = [0.0] * len(AUDIO_BANDS) shape_counts = {} amp_counts = [0] * len(AMP_GLYPHS) for f in features: for i, e in enumerate(f["band_energies"]): band_totals[i] += e shape_counts[f["shape"]] = shape_counts.get(f["shape"], 0) + 1 amp_counts[f["amp_level"]] += 1 print(" Frequency band distribution:") for i, (lo, hi, glyph, name) in enumerate(AUDIO_BANDS): total = band_totals[i] pct = total / max(sum(band_totals), 1) * 100 bar = "█" * int(pct / 2) print(f" {glyph} {name:20s} {lo:>6d}-{hi:>6d}Hz {pct:5.1f}% {bar}") print() print(" Spectral shapes:") for shape, count in sorted(shape_counts.items(), key=lambda x: -x[1]): glyph = SHAPE_GLYPHS.get(shape, "?") print(f" {glyph} {shape:12s} {count:4d} frames ({count/max(len(features),1)*100:.1f}%)") print() print(" Amplitude distribution:") for i, (lo, hi, glyph, name) in enumerate(AMP_GLYPHS): count = amp_counts[i] bar = "█" * int(count / max(len(features), 1) * 50) print(f" {glyph} {name:15s} {count:4d} {bar}") # Temporal evolution print() print(" Temporal evolution (every 10th frame):") for i in range(0, len(features), max(1, len(features) // 20)): f = features[i] band_glyph = AUDIO_BANDS[f["dominant_band"]][2] amp_glyph = AMP_GLYPHS[f["amp_level"]][2] shape_glyph = SHAPE_GLYPHS.get(f["shape"], "≡") print(f" frame {f['frame']:4d} {amp_glyph}{band_glyph}{shape_glyph} centroid={f['centroid']:.0f}Hz rms={f['rms']:.3f} zcr={f['zcr']:.2f}") elif sub == "roundtrip": if len(args) < 2: print("Usage: audio roundtrip [--out=roundtrip.wav]") sys.exit(1) wav_path = args[1] out_path = "roundtrip.wav" for a in args[2:]: if a.startswith("--out="): out_path = a.split("=", 1)[1] start = time.time() # Encode source = codec.encode(wav_path) # Decode result = codec.decode(source, out_path) elapsed = (time.time() - start) * 1000 # Compare orig_samples, orig_sr, _ = codec._read_wav(wav_path) new_samples, new_sr, _ = codec._read_wav(out_path) print(f"AudioGlyphCodec — Roundtrip: {wav_path} → glyph → {out_path}") print(f" Original: {len(orig_samples)} samples, {round(len(orig_samples)/orig_sr, 3)}s") print(f" Reconstructed: {len(new_samples)} samples, {round(len(new_samples)/new_sr, 3)}s") print(f" Total time: {round(elapsed, 2)}ms") print(f" Frames: {result['frames']}") # Spectral comparison orig_features = codec._extract_features(orig_samples, orig_sr) new_features = codec._extract_features(new_samples, new_sr) n = min(len(orig_features), len(new_features)) if n > 0: centroid_diff = sum(abs(orig_features[i]["centroid"] - new_features[i]["centroid"]) for i in range(n)) / n rms_diff = sum(abs(orig_features[i]["rms"] - new_features[i]["rms"]) for i in range(n)) / n print(f" Centroid diff: {round(centroid_diff, 1)}Hz avg") print(f" RMS diff: {round(rms_diff, 4)} avg") elif sub == "compress": if len(args) < 2: print("Usage: audio compress [--out=file.glyph]") sys.exit(1) wav_path = args[1] out_path = None for a in args[2:]: if a.startswith("--out="): out_path = a.split("=", 1)[1] rt = GlyphAudioRuntime() result = rt.compress(wav_path, out_path) print(f"GlyphAudioRuntime — Compress: {wav_path} → {result['out_glyph']}") print(f" Original: {result['orig_bytes']:,} bytes ({result['duration_s']:.2f}s)") print(f" Compressed: {result['compressed_bytes']:,} bytes") print(f" Ratio: {result['compression_pct']}% of original") print(f" Frames: {result['frames']}") print(f" Avg importance:{result['avg_importance']}") print(f" Time: {result['time_ms']}ms") print() print(" Color distribution (frequency → spectral color):") for color, count in sorted(result["color_distribution"].items(), key=lambda x: -x[1]): bar = "█" * min(count // 2, 30) print(f" {color:12s} {count:4d} frames {bar}") print() print(" Size distribution (feature importance → glyph size):") size_labels = ["S0 (1x)", "S1 (2x)", "S2 (3x)", "S3 (4x)", "S4 (5x)", "S5 (6x)"] for i, count in enumerate(result["size_distribution"]): bar = "█" * min(count // 2, 30) print(f" {size_labels[i]:10s} {count:4d} frames {bar}") elif sub == "decompress": if len(args) < 2: print("Usage: audio decompress [--out=file.wav] [--rate=22050]") sys.exit(1) glyph_path = args[1] out_path = None rate = None for a in args[2:]: if a.startswith("--out="): out_path = a.split("=", 1)[1] elif a.startswith("--rate="): rate = int(a.split("=", 1)[1]) rt = GlyphAudioRuntime() result = rt.decompress(glyph_path, out_path, rate) print(f"GlyphAudioRuntime — Decompress: {result['glyph_path']} → {result['out_wav']}") print(f" Frames: {result['frames']}") print(f" Samples: {result['samples']:,}") print(f" Duration: {result['duration_s']:.3f}s") print(f" Time: {result['time_ms']}ms") elif sub == "visualize": if len(args) < 2: print("Usage: audio visualize ") sys.exit(1) wav_path = args[1] rt = GlyphAudioRuntime() vis = rt.visualize(wav_path) print(vis) elif sub == "runtime": if len(args) < 2: print("Usage: audio runtime [--out=compressed.glyph]") sys.exit(1) wav_path = args[1] out_glyph = None for a in args[2:]: if a.startswith("--out="): out_glyph = a.split("=", 1)[1] rt = GlyphAudioRuntime() # Compress comp = rt.compress(wav_path, out_glyph) # Decompress out_wav = comp["out_glyph"].rsplit(".", 1)[0] + "_runtime.wav" decomp = rt.decompress(comp["out_glyph"], out_wav) # Compare orig_samples, orig_sr, _ = rt.codec._read_wav(wav_path) new_samples, new_sr, _ = rt.codec._read_wav(out_wav) print(f"GlyphAudioRuntime — Full Runtime: {wav_path}") print(f" ┌─ Compress") print(f" │ WAV: {comp['orig_bytes']:,} bytes ({comp['duration_s']:.2f}s)") print(f" │ Glyph: {comp['compressed_bytes']:,} bytes ({comp['compression_pct']}% ratio)") print(f" │ Frames: {comp['frames']}") print(f" │ Importance: {comp['avg_importance']}") print(f" │ Time: {comp['time_ms']}ms") print(f" ├─ Decompress (runtime execution)") print(f" │ Frames: {decomp['frames']}") print(f" │ Samples: {decomp['samples']:,}") print(f" │ Duration: {decomp['duration_s']:.3f}s") print(f" │ Time: {decomp['time_ms']}ms") print(f" └─ Result") print(f" Original: {len(orig_samples):,} samples") print(f" Reconstructed:{len(new_samples):,} samples") print(f" Total time: {round(comp['time_ms'] + decomp['time_ms'], 2)}ms") print() print(" Color spectrum (sound → color):") for color, count in sorted(comp["color_distribution"].items(), key=lambda x: -x[1]): pct = count / max(comp["frames"], 1) * 100 bar = "█" * int(pct / 2) print(f" {color:12s} {pct:5.1f}% {bar}") print() print(" Feature importance (size):") for i, count in enumerate(comp["size_distribution"]): pct = count / max(comp["frames"], 1) * 100 bar = "█" * int(pct / 2) print(f" Size {i} {pct:5.1f}% {bar}") else: print(f"Unknown audio subcommand: {sub}") sys.exit(1) def cmd_glyphlock(args: list[str] | None = None): """GlyphLock CLI: pack, open, query, inspect, list — time-gated glyph codec.""" if not args: print("GlyphLock - Time-Gated Glyph Dictionary Codec (GE²)") print() print("Usage:") print(" python3 forge.py glyphlock pack [--recipient=name] [--expiry=3600]") print(" python3 forge.py glyphlock open [--totp=code]") print(" python3 forge.py glyphlock query [--totp=code]") print(" python3 forge.py glyphlock inspect ") print(" python3 forge.py glyphlock list") print(" python3 forge.py glyphlock totp ") print(" python3 forge.py glyphlock blur [level]") sys.exit(0) sub = args[0] codec = GlyphLockCodec() if sub == "pack": if len(args) < 2: print("Usage: glyphlock pack [--recipient=name] [--expiry=3600]") sys.exit(1) filepath = args[1] recipient = "anonymous" expiry = 3600 for a in args[2:]: if a.startswith("--recipient="): recipient = a.split("=", 1)[1] elif a.startswith("--expiry="): expiry = int(a.split("=", 1)[1]) result = codec.pack(filepath, recipient, expiry) if "error" in result: print(f"Error: {result['error']}") sys.exit(1) print(f"GlyphLock - File packed into GE² envelope") print(f" Pack ID: {result['pack_id']}") print(f" File ID: {result['file_id']}") print(f" Filename: {result['filename']}") print(f" Original: {result['size_bytes']} bytes") print(f" Glyph encoded: {result['glyph_encoded_size']} bytes") print(f" Compressed: {result['compressed_size']} bytes") print(f" Encrypted: {result['encrypted_size']} bytes") print(f" Compression: {result['compression_ratio']}% of original") print(f" Total ratio: {result['total_ratio']}% of original") print(f" Merkle root: {result['merkle_root'][:24]}...") print(f" Dictionary: {result['dictionary_id']}...") print(f" Blur hash: {result['blur_hash']}") print(f" Recipient: {result['recipient']}") print(f" Expiry: {result['expiry']} ({expiry}s from now)") print(f" Current TOTP: {result['current_totp']}") print(f" Pack path: {result['pack_path']}") print(f" Key path: {result['key_path']}") print(f" Pack time: {result['pack_time_ms']}ms") print() print(" Layers:") for layer, desc in result["layers"].items(): print(f" {layer}: {desc}") print() print(" Access law:") print(" G alone -> priceable blur") print(" G + D + Kt -> unfold") print(" G + D + Kt + R -> accountable unfold") elif sub == "open": if len(args) < 2: print("Usage: glyphlock open [--totp=code]") sys.exit(1) pack_id = args[1] totp_code = None for a in args[2:]: if a.startswith("--totp="): totp_code = a.split("=", 1)[1] result = codec.open(pack_id, totp_code) if "error" in result: print(f"Error: {result['error']}") sys.exit(1) print(f"GlyphLock - Envelope unfolded") print(f" Pack ID: {result['pack_id']}") print(f" File ID: {result['file_id']}") print(f" Filename: {result['filename']}") print(f" Size: {result['size_bytes']} bytes") print(f" Merkle root: {result['merkle_root']}") print(f" Merkle verified: {result['merkle_verified']}") print(f" TOTP verified: {result['totp_verified']}") print(f" Unfolded to: {result['unfolded_path']}") print(f" Unfold time: {result['unfold_time_ms']}ms") print() print(" Layers unfolded:") for layer, desc in result["layers_unfolded"].items(): print(f" {layer}: {desc}") elif sub == "query": if len(args) < 3: print("Usage: glyphlock query [--totp=code]") sys.exit(1) pack_id = args[1] query = args[2] totp_code = None for a in args[3:]: if a.startswith("--totp="): totp_code = a.split("=", 1)[1] result = codec.query(pack_id, query, totp_code) if "error" in result: print(f"Error: {result['error']}") sys.exit(1) print(f"GlyphLock - Query: '{query}' in pack {pack_id}") print(f" File: {result['filename']}") print(f" Blur hash: {result['blur_hash']}") print(f" Preview matches: {len(result.get('preview_matches', []))}") for m in result.get("preview_matches", [])[:3]: print(f" pos {m['position']}: ...{m['context']}...") if result.get("full_search"): fs = result["full_search"] print(f" Full search: {fs['total_matches']} matches (TOTP verified)") for m in fs["first_5"][:3]: print(f" pos {m['position']}: ...{m['context']}...") else: print(f" Full search: requires TOTP code (--totp=XXXXXXXX)") elif sub == "inspect": if len(args) < 2: print("Usage: glyphlock inspect ") sys.exit(1) result = codec.inspect(args[1]) if "error" in result: print(f"Error: {result['error']}") sys.exit(1) print(f"GlyphLock - Pack inspection: {args[1]}") print(f" File ID: {result['file_id']}") print(f" Filename: {result['filename']}") print(f" Size: {result['size_bytes']} bytes") print(f" Compression: {result['compression_ratio']}% of original") print(f" Total ratio: {result['total_ratio']}% of original") print(f" Merkle root: {result['merkle_root']}") print(f" Dictionary: {result['dictionary_id']}") print(f" Blur hash: {result['blur_hash']}") print(f" Recipient: {result['recipient']}") print(f" Expired: {result['expired']}") print(f" Capabilities: {result['capabilities']}") print(f" Receipt SHA256: {result['receipt_sha256']}") print(f" Policy: {json.dumps(result['policy'])}") print(f" Files:") for f in result["files"]: print(f" {f['name']:20s} {f['size']:>10d} bytes") elif sub == "list": packs = codec.list_packs() print(f"GlyphLock - Packs ({len(packs)} total)") for p in packs: status = "EXPIRED" if p["expired"] else "active" print(f" {p['pack_id']} {p['filename']:30s} {p['size_bytes']:>10d} {p['total_ratio']:>6}% {status:8s} {p['recipient']}") elif sub == "totp": if len(args) < 2: print("Usage: glyphlock totp ") sys.exit(1) pack_id = args[1] keychain_path = GLYPHLOCK_KEYS / f"{pack_id}.key" if not keychain_path.exists(): print(f"Error: Key not found for pack: {pack_id}") sys.exit(1) keychain = json.loads(keychain_path.read_text()) totp_secret = base64.b64decode(keychain["totp_secret"]) codec2 = GlyphLockCodec() code = codec2._totp_generate(totp_secret) next_code = codec2._totp_generate(totp_secret, int(time.time()) + 30) print(f"GlyphLock - TOTP for pack {pack_id}") print(f" Current code: {code}") print(f" Next code: {next_code}") print(f" Step: 30s") print(f" Digits: 8") print(f" Valid window: ±30s") elif sub == "blur": if len(args) < 2: print("Usage: glyphlock blur [level 0-4]") sys.exit(1) pack_id = args[1] level_filter = int(args[2]) if len(args) > 2 else None pack_dir = GLYPHLOCK_PACKS / pack_id if not pack_dir.exists(): print(f"Error: Pack not found: {pack_id}") sys.exit(1) preview = json.loads((pack_dir / "preview.json").read_text()) manifest = json.loads((pack_dir / "manifest.json").read_text()) print(f"GlyphLock - Real Optical Blur for pack {pack_id}") print(f" File: {manifest['filename']}") print(f" Blur hash: {preview.get('blur_hash', '')}") print(f" Algorithm: {preview.get('blur_algorithm', '')}") print() for level in preview.get("fidelity_ladder", []): lvl = level["level"] if level_filter is not None and lvl != level_filter: continue readable = "READABLE" if level["readable"] else "BLURRED" print(f" Level {lvl}: {level['name']:20s} [{readable}]") data = level["data"] if lvl == 0: print(f" {data}") else: for line in data.split("\n")[:8]: print(f" {line}") if len(data.split("\n")) > 8: print(f" ... ({len(data)} chars total)") print() else: print(f"Unknown glyphlock subcommand: {sub}") sys.exit(1) # ============================================================================= # FORGE - Build tool (Hardhat/Forge style) # ============================================================================= PROJECT_STRUCTURE = { "src/": ".glyph and .over source files", "build/": "Compiled artifacts (JSON)", "test/": "Test vectors", "receipts/": "Signed receipts with SHA256", "snapshots/": "Policy snapshots", } def cmd_init(): """Initialize project structure.""" print("GlyphForge — Initializing project") print(f" Operator ratio: {len(OPERATORS)}/{len(GLYPH_TOKENS)} = {OPERATOR_RATIO:.1%}") print() for dirname, desc in PROJECT_STRUCTURE.items(): path = Path(dirname) path.mkdir(exist_ok=True) print(f" ✓ {dirname} — {desc}") # Write example .glyph file example_glyph = Path("src/example.glyph") if not example_glyph.exists(): example_glyph.write_text( "▷ HashVerify\n" " ◇ → H\n" " H ⊙ R\n" " R ≡ ◎\n" " ⊙̂ H\n" "◀\n" ) print(" ✓ src/example.glyph — example glyph program") # Write example .over file example_over = Path("src/example.over") if not example_over.exists(): example_over.write_text( "# OverLanguage workflow: verify and pay\n" "workflow: VerifyPay\n" "intent: verify artifact hash and issue payment receipt\n" "step 1: index file → local_index\n" "step 2: compute hash → merkle_root\n" "step 3: verify hash ≡ canonical → verified\n" "step 4: issue receipt → signed_receipt\n" "artifact: signed_receipt\n" "receipt: SHA256 chained from step 1 to step 4\n" "value: verified artifact with payment proof\n" ) print(" ✓ src/example.over — example workflow") # Write forge config config = Path("forge.json") if not config.exists(): config.write_text(json.dumps({ "compiler": "glyphforge", "version": "1.0.0", "operator_ratio": round(OPERATOR_RATIO, 4), "sources": {"glyph": "src/*.glyph", "over": "src/*.over"}, "output": "build/", "receipts": "receipts/", }, indent=2)) print(" ✓ forge.json — project config") print() print("Project initialized. Run: python3 forge.py build") def cmd_compile(filepath: str): """Compile a single .glyph or .over file.""" path = Path(filepath) if not path.exists(): print(f"Error: {filepath} not found") sys.exit(1) source = path.read_text() ext = path.suffix if ext == ".glyph": print(f"Compiling {filepath} (.glyph)") artifact = compile_glyph(source, filename=path.name) print(f" Glyphs: {artifact['glyph_count']}") print(f" Operators: {artifact['operator_count']} ({artifact['operator_ratio']:.1%})") print(f" Nodes: {artifact['node_count']}") print(f" Compile time: {artifact['compile_time_ms']}ms") print(f" SHA256: {artifact['sha256'][:16]}...") elif ext == ".over": print(f"Compiling {filepath} (.over)") artifact = compile_over(source, filename=path.name) print(f" Workflow: {artifact['workflow_name']}") print(f" Intent: {artifact['intent']}") print(f" Steps: {artifact['step_count']}") print(f" Receipts: {len(artifact['receipt_chain'])}") print(f" Merkle root: {artifact['merkle_root'][:16]}...") print(f" SHA256: {artifact['sha256'][:16]}...") else: print(f"Error: unknown file type {ext}") sys.exit(1) # Write build artifact build_dir = Path("build") build_dir.mkdir(exist_ok=True) out_name = path.stem + ".json" out_path = build_dir / out_name out_path.write_text(json.dumps(artifact, indent=2, ensure_ascii=False)) print(f" Output: {out_path}") return artifact def cmd_build(): """Compile all sources in src/.""" print("GlyphForge — Building all sources") print(f" Operator ratio: {len(OPERATORS)}/{len(GLYPH_TOKENS)} = {OPERATOR_RATIO:.1%}") print() src_dir = Path("src") if not src_dir.exists(): print("Error: src/ directory not found. Run: python3 forge.py init") sys.exit(1) glyph_files = sorted(src_dir.glob("*.glyph")) over_files = sorted(src_dir.glob("*.over")) all_files = glyph_files + over_files if not all_files: print("No .glyph or .over files found in src/") sys.exit(1) artifacts = [] for f in all_files: source = f.read_text() ext = f.suffix if ext == ".glyph": artifact = compile_glyph(source, filename=f.name) else: artifact = compile_over(source, filename=f.name) artifacts.append(artifact) print(f" ✓ {f.name} → build/{f.stem}.json (SHA256: {artifact['sha256'][:12]}...)") # Write combined build manifest build_dir = Path("build") manifest = { "build_time": time.time(), "file_count": len(all_files), "glyph_files": len(glyph_files), "over_files": len(over_files), "operator_ratio": round(OPERATOR_RATIO, 4), "artifacts": [ {"file": a["source_file"], "sha256": a["sha256"], "type": a["type"]} for a in artifacts ], } manifest_str = json.dumps(manifest, sort_keys=True) manifest["sha256"] = hashlib.sha256(manifest_str.encode()).hexdigest() (build_dir / "manifest.json").write_text(json.dumps(manifest, indent=2, ensure_ascii=False)) print() print(f"Built {len(all_files)} files. Manifest: build/manifest.json") print(f"Build SHA256: {manifest['sha256'][:16]}...") def cmd_test(): """Run all test vectors in test/.""" print("GlyphForge — Running tests") print() test_dir = Path("test") if not test_dir.exists(): print("No test/ directory. Creating with default tests...") test_dir.mkdir(exist_ok=True) # Write default test vectors (test_dir / "test_hash.glyph").write_text( "▷ HashTest\n ◇ → H\n H ⊙ R\n R ≡ ◎\n ⊙̂ H\n◀\n" ) (test_dir / "test_pay.glyph").write_text( "▷ PayTest\n ◇ → $\n $ Æ R\n R → ◎\n ¤ $\n◀\n" ) (test_dir / "test_verify.over").write_text( "workflow: TestVerify\n" "intent: test verification workflow\n" "step 1: hash file → file_hash\n" "step 2: check hash → result\n" "artifact: result\n" "value: test passes if hash verified\n" ) tests = sorted(test_dir.glob("*.glyph")) + sorted(test_dir.glob("*.over")) passed = 0 failed = 0 for t in tests: source = t.read_text() ext = t.suffix try: if ext == ".glyph": artifact = compile_glyph(source, filename=t.name) assert artifact["glyph_count"] > 0, "no glyphs found" assert artifact["sha256"], "no checksum" else: artifact = compile_over(source, filename=t.name) assert artifact["step_count"] > 0, "no steps" assert artifact["merkle_root"], "no merkle root" print(f" ✓ {t.name} — PASSED") passed += 1 except Exception as e: print(f" ✕ {t.name} — FAILED: {e}") failed += 1 print() print(f"Results: {passed} passed, {failed} failed, {passed + failed} total") def cmd_snapshot(): """Emit JSON policy snapshot with SHA256.""" print("GlyphForge — Emitting policy snapshot") print() build_dir = Path("build") if not build_dir.exists(): print("Error: no build/ directory. Run: python3 forge.py build first.") sys.exit(1) manifest_path = build_dir / "manifest.json" if not manifest_path.exists(): print("Error: no build manifest. Run: python3 forge.py build first.") sys.exit(1) manifest = json.loads(manifest_path.read_text()) snapshot = { "snapshot_time": time.time(), "build_sha256": manifest.get("sha256", ""), "file_count": manifest.get("file_count", 0), "operator_ratio": manifest.get("operator_ratio", 0), "artifacts": manifest.get("artifacts", []), "policy": { "mode": "production", "supervisor": "shared", "models": ["PCA", "KMeans", "SVM", "RandomForest", "GradientBoosting", "XGBoost"], "dry_run": False, }, } snapshot_str = json.dumps(snapshot, sort_keys=True) snapshot["sha256"] = hashlib.sha256(snapshot_str.encode()).hexdigest() snap_dir = Path("snapshots") snap_dir.mkdir(exist_ok=True) snap_name = f"snapshot_{int(time.time())}.json" snap_path = snap_dir / snap_name snap_path.write_text(json.dumps(snapshot, indent=2, ensure_ascii=False)) print(f" Snapshot: {snap_path}") print(f" SHA256: {snapshot['sha256']}") print(f" Files: {snapshot['file_count']}") print(f" Operator ratio: {snapshot['operator_ratio']:.1%}") print(f" Policy: {snapshot['policy']['mode']} / {snapshot['policy']['supervisor']} supervisor") def cmd_verify(receipt_path: str): """Verify a receipt or artifact checksum.""" print(f"GlyphForge — Verifying {receipt_path}") print() path = Path(receipt_path) if not path.exists(): print(f"Error: {receipt_path} not found") sys.exit(1) data = json.loads(path.read_text()) stored_hash = data.pop("sha256", None) if not stored_hash: print("Error: no sha256 field found") sys.exit(1) recomputed = hashlib.sha256(json.dumps(data, sort_keys=True).encode()).hexdigest() if recomputed == stored_hash: print(f" ✓ VALID — SHA256 matches") print(f" Stored: {stored_hash}") print(f" Recomputed: {recomputed}") print(f" Type: {data.get('type', 'unknown')}") if "merkle_root" in data: print(f" Merkle root: {data['merkle_root'][:16]}...") if "receipt_chain" in data: print(f" Receipt chain: {len(data['receipt_chain'])} entries") else: print(f" ✕ INVALID — SHA256 mismatch!") print(f" Stored: {stored_hash}") print(f" Recomputed: {recomputed}") def cmd_clean(): """Remove build artifacts.""" print("GlyphForge — Cleaning build artifacts") for dirname in ["build", "snapshots"]: d = Path(dirname) if d.exists(): for f in d.glob("*.json"): f.unlink() print(f" ✓ Cleaned {dirname}/") def main(): if len(sys.argv) < 2: print("GlyphForge — Compiler for .glyph and .over source files") print(f" Operators: {len(OPERATORS)}/{len(GLYPH_TOKENS)} = {OPERATOR_RATIO:.1%} of language") print() print("Commands:") print(" python3 forge.py init Initialize project") print(" python3 forge.py compile Compile .glyph or .over") print(" python3 forge.py build Build all sources in src/") print(" python3 forge.py test Run test vectors") print(" python3 forge.py snapshot Emit policy snapshot + SHA256") print(" python3 forge.py verify Verify checksum") print(" python3 forge.py clean Remove build artifacts") print(" python3 forge.py run Execute .over workflow with real I/O") print(" python3 forge.py jorki [args] JORKI file gateway CLI") print(" python3 forge.py glyphlock [args] GlyphLock time-gated codec CLI") sys.exit(0) cmd = sys.argv[1] if cmd == "init": cmd_init() elif cmd == "compile": if len(sys.argv) < 3: print("Usage: python3 forge.py compile ") sys.exit(1) cmd_compile(sys.argv[2]) elif cmd == "build": cmd_build() elif cmd == "test": cmd_test() elif cmd == "snapshot": cmd_snapshot() elif cmd == "verify": if len(sys.argv) < 3: print("Usage: python3 forge.py verify ") sys.exit(1) cmd_verify(sys.argv[2]) elif cmd == "clean": cmd_clean() elif cmd == "run": if len(sys.argv) < 3: print("Usage: python3 forge.py run [--key=value ...]") sys.exit(1) cmd_run(sys.argv[2], sys.argv[3:]) elif cmd == "jorki": cmd_jorki(sys.argv[2:]) elif cmd == "glyphlock": cmd_glyphlock(sys.argv[2:]) elif cmd == "audio": cmd_audio(sys.argv[2:]) else: print(f"Unknown command: {cmd}") sys.exit(1) if __name__ == "__main__": main()