"""
Create sample training data for demonstration.
For real benchmarks, use actual datasets like WikiText, PTB, etc.
"""

import os


def create_sample_data(output_dir='./data'):
    """Create sample training data."""
    os.makedirs(output_dir, exist_ok=True)
    
    # Sample text data (Wikipedia-style)
    sample_texts = """
    The novel AI model uses a groundbreaking formula for information processing.
    Unlike traditional transformers, this model scores information based on novelty, retention, and payoff.
    The scoring formula combines multiple factors to determine the importance of each token.
    Novelty measures how much new information a token provides relative to context.
    Retention estimates the long-term value and memorability of information.
    Payoff computes the immediate utility and relevance of the current token.
    Continuity ensures that selected information maintains coherence with the existing context.
    Fatigue penalizes redundant information that has appeared recently.
    Time decay applies an exponential decay function based on the position in the sequence.
    The formula allows the model to dynamically prioritize information during processing.
    This approach differs fundamentally from standard attention mechanisms in transformers.
    Traditional attention uses dot-product similarity between queries and keys.
    Our formula-based approach considers multiple dimensions of information quality.
    The weights for novelty, retention, and payoff are learnable parameters.
    The model adapts these weights during training to optimize performance.
    Experimental results show promising improvements in information selection.
    The architecture maintains compatibility with existing transformer infrastructure.
    Training can be performed using standard optimization techniques.
    Gradient descent works well with the differentiable formula components.
    The memory buffer tracks recent embeddings for fatigue computation.
    Each component of the formula is computed using small neural networks.
    The novelty network compares current and context embeddings.
    The retention network evaluates future importance.
    The payoff network measures immediate relevance.
    The continuity network ensures semantic coherence.
    The fatigue network compares against recent items in memory.
    All components are differentiable and enable end-to-end training.
    The model supports standard language modeling tasks.
    Text generation uses the formula to guide token selection.
    The architecture scales well with increased model size.
    Larger models show improved performance on benchmarks.
    Evaluation metrics include perplexity and accuracy.
    The model can be fine-tuned for specific domains.
    Transfer learning works effectively with this architecture.
    """
    
    # Split into paragraphs
    paragraphs = [p.strip() for p in sample_texts.strip().split('\n') if p.strip()]
    
    # Write training data (80%)
    train_size = int(len(paragraphs) * 0.8)
    with open(os.path.join(output_dir, 'sample_train.txt'), 'w', encoding='utf-8') as f:
        f.write('\n\n'.join(paragraphs[:train_size]))
    
    # Write validation data (10%)
    val_size = int(len(paragraphs) * 0.1)
    with open(os.path.join(output_dir, 'sample_valid.txt'), 'w', encoding='utf-8') as f:
        f.write('\n\n'.join(paragraphs[train_size:train_size+val_size]))
    
    # Write test data (10%)
    with open(os.path.join(output_dir, 'sample_test.txt'), 'w', encoding='utf-8') as f:
        f.write('\n\n'.join(paragraphs[train_size+val_size:]))
    
    print(f"Created sample data in {output_dir}/")
    print(f"  - sample_train.txt ({train_size} paragraphs)")
    print(f"  - sample_valid.txt ({val_size} paragraphs)")
    print(f"  - sample_test.txt ({len(paragraphs) - train_size - val_size} paragraphs)")


if __name__ == '__main__':
    create_sample_data()



