Downloads · 30 days
0
vikrantmaharshi/OM2.O
OM2.O is a machine learning model from vikrantmaharshi. Use it for the machine learning task on the model card, and read the license before you ship it in a product.
import torch import torch.nn as nn import torch.nn.functional as F import math import random from typing import List, Optional
Downloads · 30 days
0
Access
Public
Updated Jul 12, 2026
Repo size
—
Likes
0
Public
Click a slice to open those files.
.md7 KB · 82%
From the Hugging Face model README
import torch import torch.nn as nn import torch.nn.functional as F import math import random from typing import List, Optional
class PositionalEncoding(nn.Module): def init(self, d_model: int, max_seq_length: int = 512): super().init() pe = torch.zeros(max_seq_length, d_model) position = torch.arange(0, max_seq_length, dtype=torch.float).unsqueeze(1) div_term = torch.exp(torch.arange(0, d_model, 2).float() * (-math.log(10000.0) / d_model)) pe[:, 0::2] = torch.sin(position * div_term) pe[:, 1::2] = torch.cos(position * div_term) pe = pe.unsqueeze(0) self.register_buffer('pe', pe)
def forward(self, x):
return x + self.pe[:, :x.size(1)]
class MultiHeadAttention(nn.Module): def init(self, d_model: int, num_heads: int): super().init() assert d_model % num_heads == 0 self.d_model = d_model self.num_heads = num_heads self.d_k = d_model // num_heads
self.W_q = nn.Linear(d_model, d_model)
self.W_k = nn.Linear(d_model, d_model)
self.W_v = nn.Linear(d_model, d_model)
self.W_o = nn.Linear(d_model, d_model)
def scaled_dot_product_attention(self, Q, K, V, mask=None):
scores = torch.matmul(Q, K.transpose(-2, -1)) / math.sqrt(self.d_k)
if mask is not None:
scores = scores.masked_fill(mask == 0, float('-inf'))
attn = F.softmax(scores, dim=-1)
return torch.matmul(attn, V)
def forward(self, x, mask=None):
batch_size = x.size(0)
Q = self.W_q(x).view(batch_size, -1, self.num_heads, self.d_k).transpose(1, 2)
K = self.W_k(x).view(batch_size, -1, self.num_heads, self.d_k).transpose(1, 2)
V = self.W_v(x).view(batch_size, -1, self.num_heads, self.d_k).transpose(1, 2)
attn_output = self.scaled_dot_product_attention(Q, K, V, mask)
attn_output = attn_output.transpose(1, 2).contiguous().view(batch_size, -1, self.d_model)
return self.W_o(attn_output)
class FeedForward(nn.Module): def init(self, d_model: int, d_ff: int = 2048): super().init() self.fc1 = nn.Linear(d_model, d_ff) self.fc2 = nn.Linear(d_ff, d_model)
def forward(self, x):
return self.fc2(F.gelu(self.fc1(x)))
class TransformerBlock(nn.Module): def init(self, d_model: int, num_heads: int, d_ff: int = 2048): super().init() self.attention = MultiHeadAttention(d_model, num_heads) self.feed_forward = FeedForward(d_model, d_ff) self.norm1 = nn.LayerNorm(d_model) self.norm2 = nn.LayerNorm(d_model)
def forward(self, x, mask=None):
attn_output = self.attention(self.norm1(x), mask)
x = x + attn_output
ff_output = self.feed_forward(self.norm2(x))
x = x + ff_output
return x
class OM2(nn.Module): def init(self, vocab_size: int, d_model: int = 256, num_heads: int = 8, num_layers: int = 6, max_seq_length: int = 256): super().init() self.d_model = d_model self.token_embedding = nn.Embedding(vocab_size, d_model) self.pos_encoding = PositionalEncoding(d_model, max_seq_length) self.layers = nn.ModuleList([TransformerBlock(d_model, num_heads) for _ in range(num_layers)]) self.norm = nn.LayerNorm(d_model) self.fc_out = nn.Linear(d_model, vocab_size)
# Simple vocabulary for demo
self.vocab = ["<PAD>", "<SOS>", "<EOS>", "hello", "hi", "how", "are", "you", "i", "am",
"great", "fine", "what", "is", "your", "name", "om", "2.0", "chatgpt",
"like", "model", "ai", "powerful", "smart", "helpful"]
self.word_to_idx = {word: idx for idx, word in enumerate(self.vocab)}
self.idx_to_word = {idx: word for idx, word in enumerate(self.vocab)}
self.vocab_size = len(self.vocab)
def forward(self, x, mask=None):
x = self.token_embedding(x) * math.sqrt(self.d_model)
x = self.pos_encoding(x)
for layer in self.layers:
x = layer(x, mask)
x = self.norm(x)
logits = self.fc_out(x)
return logits
def generate_response(self, prompt: str, max_length: int = 50, temperature: float = 0.8) -> str:
"""Generate response like ChatGPT"""
# Simple tokenization
tokens = prompt.lower().split()
input_ids = [self.word_to_idx.get(token, 0) for token in tokens]
input_ids = [self.word_to_idx["<SOS>"]] + input_ids
input_tensor = torch.tensor([input_ids], dtype=torch.long)
self.eval()
with torch.no_grad():
for _ in range(max_length):
logits = self(input_tensor)[:, -1, :]
logits = logits / temperature
probs = F.softmax(logits, dim=-1)
next_token = torch.multinomial(probs, num_samples=1).item()
if next_token == self.word_to_idx["<EOS>"]:
break
input_ids.append(next_token)
input_tensor = torch.tensor([input_ids], dtype=torch.long)
response_tokens = [self.idx_to_word.get(idx, "<UNK>") for idx in input_ids[1:]]
return " ".join(response_tokens).replace("<EOS>", "").strip()
print("🚀 Initializing OM 2.0 - Advanced AI Model") model = OM2(vocab_size=30) # Small vocab for demo
def chat_with_om(): print("\n💬 OM 2.0 is ready! (Type 'exit' to quit)\n") print("OM 2.0: Hello! I'm OM 2.0, a powerful AI model inspired by the latest advancements.")
while True:
user_input = input("You: ")
if user_input.lower() in ['exit', 'quit', 'bye']:
print("OM 2.0: Goodbye! It was great chatting with you. 🚀")
break
# Simulate intelligent response
response = model.generate_response(user_input)
if not response or len(response.split()) < 2:
# Fallback responses for demo
responses = [
"That's an interesting point! As OM 2.0, I think deeply about these things.",
"Absolutely! My architecture allows me to reason like advanced models.",
"Great question. Let me provide a comprehensive answer based on my training.",
"I understand. Here's my take on it..."
]
response = random.choice(responses)
print(f"OM 2.0: {response}")
if name == "main": chat_with_om()