#!/usr/bin/env python3
import sys
sys.path.insert(0, 'c:/MONIKA')
from salience_os_seed.proto_lm.vocab import Vocabulary
from salience_os_seed.proto_lm._torch_impl import ProtoLanguageModel, TrainingConfig

# Test new vocab
v = Vocabulary()
print(f"New vocab size: {v.size()}")
print(f"Sample tokens: {v.tokens[116:126]}")  # First 10 word tokens

# Create model with new vocab
m = ProtoLanguageModel(TrainingConfig(checkpoint_path=None))
print(f"\nModel vocab size: {m.vocab.size()}")

# Train on conversation
for text in [
    "Hello I am Monika",
    "Hello how are you",
    "I am learning to talk",
    "I want to help you",
    "Thank you for teaching me",
] * 10:
    m.training_step(text)

print(f"\nAfter training: loss={m._latest_loss:.4f}")

# Generate
result = m.sample("Hello I", max_tokens=10, temperature=0.5, repetition_penalty=1.5)
print(f"Generation: '{result}'")
