YAML Metadata Warning:empty or missing yaml metadata in repo card

Check out the documentation for more information.

import torch import torch.nn as nn import torch.nn.functional as F import math import random from typing import List, Optional

=============================================

OM 2.0

=============================================

This is a simplified decoder-only Transformer

for educational/demo purposes. Not a full-scale model.

Name: OM 2.0 (inspired by "Om" - the cosmic sound)

class PositionalEncoding(nn.Module): def init(self, d_model: int, max_seq_length: int = 512): super().init() pe = torch.zeros(max_seq_length, d_model) position = torch.arange(0, max_seq_length, dtype=torch.float).unsqueeze(1) div_term = torch.exp(torch.arange(0, d_model, 2).float() * (-math.log(10000.0) / d_model)) pe[:, 0::2] = torch.sin(position * div_term) pe[:, 1::2] = torch.cos(position * div_term) pe = pe.unsqueeze(0) self.register_buffer('pe', pe)

def forward(self, x):
    return x + self.pe[:, :x.size(1)]

class MultiHeadAttention(nn.Module): def init(self, d_model: int, num_heads: int): super().init() assert d_model % num_heads == 0 self.d_model = d_model self.num_heads = num_heads self.d_k = d_model // num_heads

    self.W_q = nn.Linear(d_model, d_model)
    self.W_k = nn.Linear(d_model, d_model)
    self.W_v = nn.Linear(d_model, d_model)
    self.W_o = nn.Linear(d_model, d_model)

def scaled_dot_product_attention(self, Q, K, V, mask=None):
    scores = torch.matmul(Q, K.transpose(-2, -1)) / math.sqrt(self.d_k)
    if mask is not None:
        scores = scores.masked_fill(mask == 0, float('-inf'))
    attn = F.softmax(scores, dim=-1)
    return torch.matmul(attn, V)

def forward(self, x, mask=None):
    batch_size = x.size(0)
    
    Q = self.W_q(x).view(batch_size, -1, self.num_heads, self.d_k).transpose(1, 2)
    K = self.W_k(x).view(batch_size, -1, self.num_heads, self.d_k).transpose(1, 2)
    V = self.W_v(x).view(batch_size, -1, self.num_heads, self.d_k).transpose(1, 2)
    
    attn_output = self.scaled_dot_product_attention(Q, K, V, mask)
    attn_output = attn_output.transpose(1, 2).contiguous().view(batch_size, -1, self.d_model)
    return self.W_o(attn_output)

class FeedForward(nn.Module): def init(self, d_model: int, d_ff: int = 2048): super().init() self.fc1 = nn.Linear(d_model, d_ff) self.fc2 = nn.Linear(d_ff, d_model)

def forward(self, x):
    return self.fc2(F.gelu(self.fc1(x)))

class TransformerBlock(nn.Module): def init(self, d_model: int, num_heads: int, d_ff: int = 2048): super().init() self.attention = MultiHeadAttention(d_model, num_heads) self.feed_forward = FeedForward(d_model, d_ff) self.norm1 = nn.LayerNorm(d_model) self.norm2 = nn.LayerNorm(d_model)

def forward(self, x, mask=None):
    attn_output = self.attention(self.norm1(x), mask)
    x = x + attn_output
    ff_output = self.feed_forward(self.norm2(x))
    x = x + ff_output
    return x

class OM2(nn.Module): def init(self, vocab_size: int, d_model: int = 256, num_heads: int = 8, num_layers: int = 6, max_seq_length: int = 256): super().init() self.d_model = d_model self.token_embedding = nn.Embedding(vocab_size, d_model) self.pos_encoding = PositionalEncoding(d_model, max_seq_length) self.layers = nn.ModuleList([TransformerBlock(d_model, num_heads) for _ in range(num_layers)]) self.norm = nn.LayerNorm(d_model) self.fc_out = nn.Linear(d_model, vocab_size)

    # Simple vocabulary for demo
    self.vocab = ["<PAD>", "<SOS>", "<EOS>", "hello", "hi", "how", "are", "you", "i", "am", 
                 "great", "fine", "what", "is", "your", "name", "om", "2.0", "chatgpt", 
                 "like", "model", "ai", "powerful", "smart", "helpful"]
    self.word_to_idx = {word: idx for idx, word in enumerate(self.vocab)}
    self.idx_to_word = {idx: word for idx, word in enumerate(self.vocab)}
    self.vocab_size = len(self.vocab)

def forward(self, x, mask=None):
    x = self.token_embedding(x) * math.sqrt(self.d_model)
    x = self.pos_encoding(x)
    
    for layer in self.layers:
        x = layer(x, mask)
    
    x = self.norm(x)
    logits = self.fc_out(x)
    return logits

def generate_response(self, prompt: str, max_length: int = 50, temperature: float = 0.8) -> str:
    """Generate response like ChatGPT"""
    # Simple tokenization
    tokens = prompt.lower().split()
    input_ids = [self.word_to_idx.get(token, 0) for token in tokens]
    input_ids = [self.word_to_idx["<SOS>"]] + input_ids
    
    input_tensor = torch.tensor([input_ids], dtype=torch.long)
    
    self.eval()
    with torch.no_grad():
        for _ in range(max_length):
            logits = self(input_tensor)[:, -1, :]
            logits = logits / temperature
            probs = F.softmax(logits, dim=-1)
            next_token = torch.multinomial(probs, num_samples=1).item()
            
            if next_token == self.word_to_idx["<EOS>"]:
                break
            
            input_ids.append(next_token)
            input_tensor = torch.tensor([input_ids], dtype=torch.long)
    
    response_tokens = [self.idx_to_word.get(idx, "<UNK>") for idx in input_ids[1:]]
    return " ".join(response_tokens).replace("<EOS>", "").strip()

============================

Initialize OM 2.0

============================

print("๐Ÿš€ Initializing OM 2.0 - Advanced AI Model") model = OM2(vocab_size=30) # Small vocab for demo

Demo conversation

def chat_with_om(): print("\n๐Ÿ’ฌ OM 2.0 is ready! (Type 'exit' to quit)\n") print("OM 2.0: Hello! I'm OM 2.0, a powerful AI model inspired by the latest advancements.")

while True:
    user_input = input("You: ")
    if user_input.lower() in ['exit', 'quit', 'bye']:
        print("OM 2.0: Goodbye! It was great chatting with you. ๐Ÿš€")
        break
    
    # Simulate intelligent response
    response = model.generate_response(user_input)
    if not response or len(response.split()) < 2:
        # Fallback responses for demo
        responses = [
            "That's an interesting point! As OM 2.0, I think deeply about these things.",
            "Absolutely! My architecture allows me to reason like advanced models.",
            "Great question. Let me provide a comprehensive answer based on my training.",
            "I understand. Here's my take on it..."
        ]
        response = random.choice(responses)
    
    print(f"OM 2.0: {response}")

Run the chat

if name == "main": chat_with_om()

Downloads last month

-

Downloads are not tracked for this model. How to track
Inference Providers NEW
This model isn't deployed by any Inference Provider. ๐Ÿ™‹ Ask for provider support