Buckets:
| #!/usr/bin/env python3 | |
| """ | |
| Script to execute CA6 notebook cells and capture outputs | |
| """ | |
| import sys | |
| import io | |
| import warnings | |
| warnings.filterwarnings("ignore") | |
| # Set up matplotlib for non-interactive mode | |
| import matplotlib | |
| matplotlib.use("Agg") | |
| import matplotlib.pyplot as plt | |
| import torch | |
| import torch.nn as nn | |
| import torch.nn.functional as F | |
| import torch.optim as optim | |
| import numpy as np | |
| import seaborn as sns | |
| import pandas as pd | |
| from typing import Dict, List, Tuple, Optional, Any, Union | |
| from dataclasses import dataclass | |
| import json | |
| import random | |
| import time | |
| from collections import defaultdict | |
| from enum import Enum | |
| import itertools | |
| from abc import ABC, abstractmethod | |
| import re | |
| # Set random seeds | |
| torch.manual_seed(42) | |
| np.random.seed(42) | |
| random.seed(42) | |
| # Set plotting style | |
| plt.style.use("seaborn-v0_8") | |
| sns.set_palette("husl") | |
| print("=" * 80) | |
| print("CELL 1: Initial Setup and Imports") | |
| print("=" * 80) | |
| print("🚀 Advanced Systematic Generalization Analysis Started!") | |
| print("=" * 60) | |
| print(f"PyTorch version: {torch.__version__}") | |
| print(f"NumPy version: {np.__version__}") | |
| print(f"Matplotlib backend: {matplotlib.get_backend()}") | |
| print("Environment setup complete!") | |
| print() | |
| print("=" * 80) | |
| print("CELL 2: Advanced Neural Architectures") | |
| print("=" * 80) | |
| # Cell 2 content - Advanced Neural Architectures | |
| class SinusoidalPositionalEncoding(nn.Module): | |
| """Advanced sinusoidal positional encoding with learnable parameters""" | |
| def __init__(self, d_model: int, max_len: int = 5000, dropout: float = 0.1): | |
| super().__init__() | |
| self.dropout = nn.Dropout(p=dropout) | |
| # Learnable scaling factors | |
| self.scale = nn.Parameter(torch.ones(1)) | |
| self.phase_shift = nn.Parameter(torch.zeros(d_model)) | |
| pe = torch.zeros(max_len, d_model) | |
| position = torch.arange(0, max_len, dtype=torch.float).unsqueeze(1) | |
| div_term = torch.exp( | |
| torch.arange(0, d_model, 2).float() | |
| * (-torch.log(torch.tensor(10000.0)) / d_model) | |
| ) | |
| pe[:, 0::2] = torch.sin(position * div_term) | |
| pe[:, 1::2] = torch.cos(position * div_term) | |
| pe = pe.unsqueeze(0).transpose(0, 1) | |
| self.register_buffer("pe", pe) | |
| def forward(self, x: torch.Tensor) -> torch.Tensor: | |
| seq_len = x.size(0) | |
| pos_encoding = self.pe[:seq_len, :] * self.scale + self.phase_shift | |
| return self.dropout(x + pos_encoding) | |
| class CompositionalMultiHeadAttention(nn.Module): | |
| """Multi-head attention with explicit compositional structure""" | |
| def __init__(self, d_model: int, num_heads: int, dropout: float = 0.1): | |
| super().__init__() | |
| assert d_model % num_heads == 0 | |
| self.d_model = d_model | |
| self.num_heads = num_heads | |
| self.d_k = d_model // num_heads | |
| self.w_q = nn.Linear(d_model, d_model) | |
| self.w_k = nn.Linear(d_model, d_model) | |
| self.w_v = nn.Linear(d_model, d_model) | |
| self.w_o = nn.Linear(d_model, d_model) | |
| # Compositional bias parameters | |
| self.compositional_bias = nn.Parameter( | |
| torch.randn(num_heads, self.d_k, self.d_k) | |
| ) | |
| self.structural_attention = nn.Parameter(torch.randn(num_heads, 1, 1)) | |
| self.dropout = nn.Dropout(dropout) | |
| self.layer_norm = nn.LayerNorm(d_model) | |
| def forward( | |
| self, | |
| query: torch.Tensor, | |
| key: torch.Tensor, | |
| value: torch.Tensor, | |
| mask: Optional[torch.Tensor] = None, | |
| compositional_structure: Optional[torch.Tensor] = None, | |
| ) -> torch.Tensor: | |
| batch_size, seq_len = query.size(0), query.size(1) | |
| # Linear transformations | |
| Q = ( | |
| self.w_q(query) | |
| .view(batch_size, seq_len, self.num_heads, self.d_k) | |
| .transpose(1, 2) | |
| ) | |
| K = ( | |
| self.w_k(key) | |
| .view(batch_size, seq_len, self.num_heads, self.d_k) | |
| .transpose(1, 2) | |
| ) | |
| V = ( | |
| self.w_v(value) | |
| .view(batch_size, seq_len, self.num_heads, self.d_k) | |
| .transpose(1, 2) | |
| ) | |
| # Scaled dot-product attention | |
| scores = torch.matmul(Q, K.transpose(-2, -1)) / torch.sqrt( | |
| torch.tensor(self.d_k, dtype=torch.float) | |
| ) | |
| # Apply compositional bias | |
| if compositional_structure is not None: | |
| comp_bias = torch.matmul(compositional_structure, self.compositional_bias) | |
| scores = scores + comp_bias.unsqueeze(0) | |
| # Apply structural attention | |
| scores = scores * self.structural_attention | |
| if mask is not None: | |
| scores = scores.masked_fill(mask == 0, -1e9) | |
| attention_weights = F.softmax(scores, dim=-1) | |
| attention_weights = self.dropout(attention_weights) | |
| # Apply attention to values | |
| context = torch.matmul(attention_weights, V) | |
| context = ( | |
| context.transpose(1, 2).contiguous().view(batch_size, seq_len, self.d_model) | |
| ) | |
| output = self.w_o(context) | |
| return self.layer_norm(output + query) | |
| # Test the advanced architectures | |
| print("🧠 Testing Advanced Neural Architectures...") | |
| # Create test data | |
| batch_size, seq_len, d_model = 2, 10, 64 | |
| test_input = torch.randn(batch_size, seq_len, d_model) | |
| # Test positional encoding | |
| pos_encoding = SinusoidalPositionalEncoding(d_model) | |
| encoded_input = pos_encoding(test_input) | |
| print(f"✅ Positional encoding output shape: {encoded_input.shape}") | |
| # Test compositional attention | |
| compositional_attention = CompositionalMultiHeadAttention(d_model, num_heads=8) | |
| attention_output = compositional_attention(encoded_input, encoded_input, encoded_input) | |
| print(f"✅ Compositional attention output shape: {attention_output.shape}") | |
| print("🎉 Advanced neural architectures tested successfully!") | |
| print() | |
Xet Storage Details
- Size:
- 5.81 kB
- Xet hash:
- 267782961725e7a944477d3f780053ffec414158cd79bbcce930a2e49fb3f505
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.