#!/usr/bin/env python3
"""Manually add word-level tokens to vocab for coherent generation."""
import sys
sys.path.insert(0, 'c:/MONIKA')
from salience_os_seed.proto_lm._torch_impl import ProtoLanguageModel, TrainingConfig

# Load current model
m = ProtoLanguageModel(TrainingConfig(checkpoint_path=None, learning_enabled=False))

# Add common word tokens
words = [
    " I", " am", " is", " are", " was", " the", " a", " to", " and", " of",
    " you", " me", " we", " he", " she", " it", " that", " this", " my", " your",
    " can", " will", " would", " should", " could", " have", " has", " had",
    " do", " does", " did", " not", " be", " been", " being",
    " hello", " hi", " thank", " thanks", " please", " yes", " no",
    " good", " well", " better", " best", " help", " learn", " know", " think",
    " want", " need", " like", " love", " feel", " see", " hear", " say",
    " tell", " ask", " understand", " believe", " hope", " try", " make",
    " day", " time", " now", " today", " here", " there", " what", " how",
    " when", " where", " why", " who", " which", " very", " much", " more",
]

print(f"Starting vocab size: {m.vocab.size()}")
for word in words:
    if word not in m.vocab.token_to_id:
        m.vocab.tokens.append(word)
m.vocab._refresh_index()
m._ensure_capacity(m.vocab.size())
print(f"New vocab size: {m.vocab.size()}")

# Save
import torch
torch.save({
    'model': m.state_dict(),
    'optimizer': m.optimizer.state_dict(),
    'step': m.step,
    'vocab': {'tokens': m.vocab.tokens, 'merges': m.vocab.merges},
    'scheduler': None
}, 'storage/monika_fresh.pt')
print(f"Saved with {len(words)} new word tokens")
