Downloads · 30 days
0
Xdimitris15gmailcom/Di_mi_ai
Di_mi_ai is a machine learning model from Xdimitris15gmailcom. Use it for the machine learning task on the model card, and read the license before you ship it in a product.
import os import torch from transformers import ( AutoTokenizer, AutoModelForCausalLM, TextDataset, DataCollatorForLanguageModeling, Trainer, TrainingArguments )
Downloads · 30 days
0
Access
Public
Updated May 12, 2026
Repo size
—
Likes
0
Public
Click a slice to open those files.
.md3.2 KB · 68%
From the Hugging Face model README
import os import torch from transformers import ( AutoTokenizer, AutoModelForCausalLM, TextDataset, DataCollatorForLanguageModeling, Trainer, TrainingArguments )
def finetune_on_text(file_path, model_name="gpt2", output_dir="./fine_tuned_ai"): """ Fine-tunes a causal language model on a local .txt file. """ print(f"--- Initializing Fine-tuning for {model_name} ---")
# 1. Load Tokenizer and Model
tokenizer = AutoTokenizer.from_pretrained(model_name)
model = AutoModelForCausalLM.from_pretrained(model_name)
# GPT-2 does not have a padding token by default, so we use the EOS token
if tokenizer.pad_token is None:
tokenizer.pad_token = tokenizer.eos_token
# 2. Prepare the Dataset
# This class reads your .txt file and chunks it into token blocks for the model
def load_dataset(path, tokenizer, block_size=128):
return TextDataset(
tokenizer=tokenizer,
file_path=path,
block_size=block_size,
)
train_dataset = load_dataset(file_path, tokenizer)
# 3. Data Collator
# This prepares the batches and handles shifting labels for causal language modeling
data_collator = DataCollatorForLanguageModeling(
tokenizer=tokenizer,
mlm=False # Causal LM, not Masked LM
)
# 4. Define Training Arguments
training_args = TrainingArguments(
output_dir=output_dir,
overwrite_output_dir=True,
num_train_epochs=3, # Number of times to go through your data
per_device_train_batch_size=4, # Adjust based on your GPU/RAM
save_steps=500, # Save checkpoint every 500 steps
save_total_limit=2, # Only keep the 2 most recent checkpoints
logging_steps=10,
prediction_loss_only=True,
)
# 5. Initialize Trainer
trainer = Trainer(
model=model,
args=training_args,
data_collator=data_collator,
train_dataset=train_dataset,
)
# 6. Train and Save
print("Starting training...")
trainer.train()
trainer.save_model(output_dir)
tokenizer.save_pretrained(output_dir)
print(f"Training complete. Model saved to {output_dir}")
def test_model(model_path, prompt="Once upon a time"): """ Load the newly trained model and generate text. """ from transformers import pipeline print("\n--- Testing Fine-tuned Model ---") gen = pipeline("text-generation", model=model_path, tokenizer=model_path) print(gen(prompt, max_length=100)[0]['generated_text'])
if name == "main": # REQUIRED: Create a file named 'my_data.txt' with your training text data_file = "my_data.txt"
if not os.path.exists(data_file):
with open(data_file, "w") as f:
f.write("Artificial intelligence is a branch of computer science.\n" * 100)
print(f"Created a dummy {data_file}. Replace this with your actual data!")
# Run the process
# NOTE: This requires 'pip install transformers torch accelerate'
finetune_on_text(data_file)
# Test the result
test_model("./fine_tuned_ai", prompt="AI is basically")