Download cnn_bigru/__init__.py from PowerMachine/CNN-BiGRU: direct link, hf CLI and curl.
- Browser
- Download file 11.9 kB
-
https://huggingface.co/PowerMachine/CNN-BiGRU/resolve/main/cnn_bigru/__init__.py
- Command line
-
hf download hf://PowerMachine/CNN-BiGRU/cnn_bigru/__init__.py
-
curl -L -o __init__.py https://huggingface.co/PowerMachine/CNN-BiGRU/resolve/main/cnn_bigru/__init__.py
11.9 kB
| """ | |
| CNN-BiGRU — Modelo Multimodal Cooperativo Autoaprendível (v3.0) | |
| Implementação completa de um LLM multimodal com arquitetura CNN-BiGRU cooperativa: | |
| - Núcleo dual-stream CNN-BiGRU com 3 níveis de pontes (cross-attention, gated cells, fusão) | |
| - Encoders multimodais (imagem + áudio) com fusão gated | |
| - Gerador + Verificador + Camada anti-alucinação (lógica fuzzy de Łukasiewicz) | |
| - Auto-aprendizado: ajuste dinâmico de LR, spectral norm, orthogonal init | |
| - Hipóteses ativadas por punições (Gumbel-Softmax diferenciável) | |
| - Synergy search (N tentativas de combinação de hiperparâmetros) | |
| - Múltiplas funções de perda: L_G, L_V, L_AH, L_reg, L_EWC | |
| v2.0: | |
| - EWC (Elastic Weight Consolidation) para aprendizado contínuo | |
| - Context Window com cache KV (sliding window + sink) | |
| - RoPE (Rotary Position Embeddings) | |
| - TransformerBlock (CausalSelfAttention + FFN) | |
| - Self-attention final layer verdadeira (CLS-token + multi-head) | |
| - Weight tying opcional | |
| - Inferência autoregressiva REAL via GeneratorCNNBiGRU | |
| NOVO v3.0: | |
| - Cyclic Reasoning (Raciocínio Cíclico) com convergência antecipada | |
| - NLG (Natural Language Generation) com Transformer Decoder + Medusa MTP | |
| - NLP (Natural Language Processing) com 4 tarefas: SeqCls, TokCls, Span, Embed | |
| - W8A8 Quantization via SmoothQuant (alpha=0.5, per-channel) | |
| - VQ-VAE-2 Hierárquico (top + bottom codebooks, EMA update) | |
| - Multi-Token Prediction (MTP) com Medusa Heads | |
| - Multimodal Multi-Head Attention (cross-modal + modality gate) | |
| - Long Context Window (1M tokens) via chunked / ring attention | |
| - Monitor completo (treino, inferência, evolução, sistema) | |
| """ | |
| from __future__ import annotations | |
| __version__ = "3.0.0" | |
| __author__ = "CNN-BiGRU Project" | |
| # Versão e metadados | |
| __all__ = [ | |
| "__version__", | |
| "__author__", | |
| # Tokenizer | |
| "BBPETokenizer", | |
| # Utils | |
| "MemoryOptimizer", | |
| "SemanticEmbedder", | |
| "EWCConfig", | |
| "EWCState", | |
| "W8A8Config", | |
| "SmoothQuantizer", | |
| "quantize_model_w8a8", | |
| "estimate_memory_savings", | |
| "VQVAE2Config", | |
| "VQVAE2", | |
| "Monitor", | |
| "get_monitor", | |
| # Runtime | |
| "optimize_xeon_environment", | |
| "get_runtime_info", | |
| # Data | |
| "MultimodalStreamingDataset", | |
| "collate_multimodal", | |
| # Models | |
| "CooperativeCNNBiGRU", | |
| "MultimodalCNNBiGRU", | |
| "ImageEncoder", | |
| "AudioEncoder", | |
| "MultimodalFusion", | |
| "GeneratorCNNBiGRU", | |
| "VerifierCNNBiGRU", | |
| "AntiHallucinationLayer", | |
| "RotaryPositionEmbedding", | |
| "CausalSelfAttention", | |
| "TransformerBlock", | |
| "TransformerDecoderStack", | |
| "KVCache", | |
| "ContextWindowManager", | |
| "ContextWindowConfig", | |
| "LongContextConfig", | |
| "LongContextManager", | |
| "make_long_context_window", | |
| "CyclicReasoningConfig", | |
| "CyclicReasoning", | |
| "MedusaConfig", | |
| "MedusaMTP", | |
| "MedusaHead", | |
| "medusa_tree_decode", | |
| "NLGConfig", | |
| "NLGModule", | |
| "NLPConfig", | |
| "NLPModule", | |
| "SequenceClassificationHead", | |
| "TokenClassificationHead", | |
| "SpanDetectionHead", | |
| "EmbeddingHead", | |
| "MultimodalAttentionConfig", | |
| "MultimodalMultiHeadAttention", | |
| "CrossModalAttention", | |
| "ModalityGate", | |
| # Losses | |
| "LossConfig", | |
| "MultiLoss", | |
| # Training | |
| "TrainerConfig", | |
| "CooperativeTrainer", | |
| "AutoLearnConfig", | |
| "AutoLearner", | |
| "HypothesisConfig", | |
| "HypothesisController", | |
| "SynergySearcher", | |
| # Inference | |
| "generate_with_sampling", | |
| "evaluate_perplexity", | |
| ] | |
| # Lazy imports — apenas quando acessado | |
| def __getattr__(name: str): | |
| if name in ("BBPETokenizer",): | |
| from .tokenizer.bbpe_tokenizer import BBPETokenizer | |
| return BBPETokenizer | |
| if name in ("MemoryOptimizer",): | |
| from .utils.memory_optimizer import MemoryOptimizer | |
| return MemoryOptimizer | |
| if name in ("SemanticEmbedder",): | |
| from .utils.semantic_embeddings import SemanticEmbedder | |
| return SemanticEmbedder | |
| if name in ("EWCConfig", "EWCState"): | |
| from .utils.ewc import EWCConfig, EWCState | |
| if name == "EWCConfig": | |
| return EWCConfig | |
| return EWCState | |
| # W8A8 Quantization | |
| if name in ("W8A8Config", "SmoothQuantizer"): | |
| from .utils.quantization import W8A8Config, SmoothQuantizer | |
| if name == "W8A8Config": | |
| return W8A8Config | |
| return SmoothQuantizer | |
| if name == "quantize_model_w8a8": | |
| from .utils.quantization import quantize_model_w8a8 | |
| return quantize_model_w8a8 | |
| if name == "estimate_memory_savings": | |
| from .utils.quantization import estimate_memory_savings | |
| return estimate_memory_savings | |
| # VQ-VAE-2 | |
| if name in ("VQVAE2Config", "VQVAE2"): | |
| from .utils.vqvae2 import VQVAE2Config, VQVAE2 | |
| if name == "VQVAE2Config": | |
| return VQVAE2Config | |
| return VQVAE2 | |
| # Monitor | |
| if name in ("Monitor", "get_monitor"): | |
| from .utils.monitoring import Monitor, get_monitor | |
| if name == "Monitor": | |
| return Monitor | |
| return get_monitor | |
| # Runtime | |
| if name in ("optimize_xeon_environment", "get_runtime_info"): | |
| from .utils import xeon_runtime | |
| if name == "optimize_xeon_environment": | |
| return xeon_runtime.optimize_xeon_environment | |
| return xeon_runtime.get_runtime_info | |
| # Data | |
| if name in ("MultimodalStreamingDataset", "collate_multimodal"): | |
| from .data.streaming_dataset import MultimodalStreamingDataset, collate_multimodal | |
| if name == "MultimodalStreamingDataset": | |
| return MultimodalStreamingDataset | |
| return collate_multimodal | |
| # Models | |
| if name == "CooperativeCNNBiGRU": | |
| from .models.cooperative_bigru import CooperativeCNNBiGRU | |
| return CooperativeCNNBiGRU | |
| if name == "MultimodalCNNBiGRU": | |
| from .models.multimodal_model import MultimodalCNNBiGRU | |
| return MultimodalCNNBiGRU | |
| if name in ("ImageEncoder", "AudioEncoder", "MultimodalFusion"): | |
| from .models.multimodal_encoders import ImageEncoder, AudioEncoder, MultimodalFusion | |
| if name == "ImageEncoder": | |
| return ImageEncoder | |
| if name == "AudioEncoder": | |
| return AudioEncoder | |
| return MultimodalFusion | |
| if name in ("GeneratorCNNBiGRU", "VerifierCNNBiGRU", "AntiHallucinationLayer"): | |
| from .models.generator_verifier import ( | |
| GeneratorCNNBiGRU, | |
| VerifierCNNBiGRU, | |
| AntiHallucinationLayer, | |
| ) | |
| if name == "GeneratorCNNBiGRU": | |
| return GeneratorCNNBiGRU | |
| if name == "VerifierCNNBiGRU": | |
| return VerifierCNNBiGRU | |
| return AntiHallucinationLayer | |
| if name == "RotaryPositionEmbedding": | |
| from .models.rope import RotaryPositionEmbedding | |
| return RotaryPositionEmbedding | |
| if name in ("CausalSelfAttention", "TransformerBlock", "TransformerDecoderStack"): | |
| from .models.transformer_block import ( | |
| CausalSelfAttention, | |
| TransformerBlock, | |
| TransformerDecoderStack, | |
| ) | |
| if name == "CausalSelfAttention": | |
| return CausalSelfAttention | |
| if name == "TransformerBlock": | |
| return TransformerBlock | |
| return TransformerDecoderStack | |
| # Context window (incl. 1M tokens) | |
| if name in ("KVCache", "ContextWindowManager", "ContextWindowConfig"): | |
| from .models.context_window import KVCache, ContextWindowManager, ContextWindowConfig | |
| if name == "KVCache": | |
| return KVCache | |
| if name == "ContextWindowManager": | |
| return ContextWindowManager | |
| return ContextWindowConfig | |
| if name in ("LongContextConfig", "LongContextManager", "make_long_context_window"): | |
| from .models.context_window import ( | |
| LongContextConfig, | |
| LongContextManager, | |
| make_long_context_window, | |
| ) | |
| if name == "LongContextConfig": | |
| return LongContextConfig | |
| if name == "LongContextManager": | |
| return LongContextManager | |
| return make_long_context_window | |
| # Cyclic Reasoning | |
| if name in ("CyclicReasoningConfig", "CyclicReasoning"): | |
| from .models.cyclic_reasoning import CyclicReasoningConfig, CyclicReasoning | |
| if name == "CyclicReasoningConfig": | |
| return CyclicReasoningConfig | |
| return CyclicReasoning | |
| # Medusa MTP | |
| if name in ("MedusaConfig", "MedusaMTP", "MedusaHead", "medusa_tree_decode"): | |
| from .models.medusa_heads import ( | |
| MedusaConfig, | |
| MedusaMTP, | |
| MedusaHead, | |
| medusa_tree_decode, | |
| ) | |
| if name == "MedusaConfig": | |
| return MedusaConfig | |
| if name == "MedusaMTP": | |
| return MedusaMTP | |
| if name == "MedusaHead": | |
| return MedusaHead | |
| return medusa_tree_decode | |
| # NLG | |
| if name in ("NLGConfig", "NLGModule"): | |
| from .models.nlg import NLGConfig, NLGModule | |
| if name == "NLGConfig": | |
| return NLGConfig | |
| return NLGModule | |
| # NLP | |
| if name in ("NLPConfig", "NLPModule", | |
| "SequenceClassificationHead", "TokenClassificationHead", | |
| "SpanDetectionHead", "EmbeddingHead"): | |
| from .models.nlp import ( | |
| NLPConfig, | |
| NLPModule, | |
| SequenceClassificationHead, | |
| TokenClassificationHead, | |
| SpanDetectionHead, | |
| EmbeddingHead, | |
| ) | |
| if name == "NLPConfig": | |
| return NLPConfig | |
| if name == "NLPModule": | |
| return NLPModule | |
| if name == "SequenceClassificationHead": | |
| return SequenceClassificationHead | |
| if name == "TokenClassificationHead": | |
| return TokenClassificationHead | |
| if name == "SpanDetectionHead": | |
| return SpanDetectionHead | |
| return EmbeddingHead | |
| # Multimodal Attention | |
| if name in ("MultimodalAttentionConfig", "MultimodalMultiHeadAttention", | |
| "CrossModalAttention", "ModalityGate"): | |
| from .models.multimodal_attention import ( | |
| MultimodalAttentionConfig, | |
| MultimodalMultiHeadAttention, | |
| CrossModalAttention, | |
| ModalityGate, | |
| ) | |
| if name == "MultimodalAttentionConfig": | |
| return MultimodalAttentionConfig | |
| if name == "MultimodalMultiHeadAttention": | |
| return MultimodalMultiHeadAttention | |
| if name == "CrossModalAttention": | |
| return CrossModalAttention | |
| return ModalityGate | |
| # Losses | |
| if name in ("LossConfig", "MultiLoss"): | |
| from .losses.losses import LossConfig, MultiLoss | |
| if name == "LossConfig": | |
| return LossConfig | |
| return MultiLoss | |
| # Training | |
| if name in ("TrainerConfig", "CooperativeTrainer"): | |
| from .training.trainer import TrainerConfig, CooperativeTrainer | |
| if name == "TrainerConfig": | |
| return TrainerConfig | |
| return CooperativeTrainer | |
| if name in ("AutoLearnConfig", "AutoLearner"): | |
| from .training.auto_learner import AutoLearnConfig, AutoLearner | |
| if name == "AutoLearnConfig": | |
| return AutoLearnConfig | |
| return AutoLearner | |
| if name in ("HypothesisConfig", "HypothesisController", "SynergySearcher"): | |
| from .training.hypothesis_controller import ( | |
| HypothesisConfig, | |
| HypothesisController, | |
| SynergySearcher, | |
| ) | |
| if name == "HypothesisConfig": | |
| return HypothesisConfig | |
| if name == "HypothesisController": | |
| return HypothesisController | |
| return SynergySearcher | |
| # Inference | |
| if name in ("generate_with_sampling", "evaluate_perplexity"): | |
| from .inference.inference import generate_with_sampling, evaluate_perplexity | |
| if name == "generate_with_sampling": | |
| return generate_with_sampling | |
| return evaluate_perplexity | |
| raise AttributeError(f"module {__name__!r} has no attribute {name!r}") | |