#!/usr/bin/env python3
"""Download safe, open-license datasets for training."""

import urllib.request
import json
from pathlib import Path

datasets_dir = Path("C:\\MONIKA\\datasets")
datasets_dir.mkdir(exist_ok=True)

print("=== Downloading Open-License Datasets ===\n")

# 1. Simple Wikipedia - CC BY-SA 3.0
print("1. Downloading Simple Wikipedia sample...")
simple_wiki_url = "https://dumps.wikimedia.org/simplewiki/latest/simplewiki-latest-abstract.xml"
# Actually, let's use a text-based approach instead

# Let's create a curated dataset from public domain sources
print("Creating training dataset from public domain texts...")

# Sample texts that are definitely safe
training_texts = []

# Basic conversational patterns
conversations = [
    "Hello, how are you today?",
    "I am doing well, thank you for asking.",
    "What are you learning about?",
    "I am learning to understand language and communicate clearly.",
    "Can you help me with a question?",
    "Yes, I would be happy to help you.",
    "Tell me about the weather.",
    "The weather changes based on temperature, humidity, and atmospheric conditions.",
    "What is the meaning of life?",
    "Life has meaning through connections, growth, and contributing to the world.",
    "How do you learn new things?",
    "I learn through practice, repetition, and exposure to diverse examples.",
]

# Simple facts and knowledge
facts = [
    "The Earth orbits around the Sun once per year.",
    "Water freezes at zero degrees Celsius.",
    "Plants use photosynthesis to convert sunlight into energy.",
    "The human brain contains billions of neurons.",
    "Language allows humans to share complex ideas.",
    "Mathematics helps us understand patterns and relationships.",
    "History teaches us lessons from the past.",
    "Science discovers how the natural world works.",
    "Art expresses human creativity and emotion.",
    "Music combines rhythm, melody, and harmony.",
]

# Reasoning and explanations
reasoning = [
    "Because it rained, the ground is wet.",
    "If you practice regularly, you will improve your skills.",
    "When temperature drops below freezing, water turns to ice.",
    "Although it is difficult, persistence leads to success.",
    "Since the sun provides light, plants can grow.",
    "Therefore, we must consider all the evidence carefully.",
    "However, there are always exceptions to general rules.",
    "Moreover, combining different approaches often works best.",
    "Nevertheless, we should remain open to new possibilities.",
    "Consequently, our actions have effects on the future.",
]

# Combine all
all_texts = conversations + facts + reasoning

# Write to file
output_file = datasets_dir / "curated_training.txt"
with open(output_file, 'w', encoding='utf-8') as f:
    for text in all_texts:
        f.write(text + '\n')

print(f"✓ Created {output_file} with {len(all_texts)} examples")

# Also download some classic public domain literature
print("\n2. Downloading public domain texts from Project Gutenberg...")

gutenberg_urls = [
    ("https://www.gutenberg.org/files/1342/1342-0.txt", "pride_and_prejudice.txt"),  # Pride and Prejudice
    ("https://www.gutenberg.org/files/11/11-0.txt", "alice_wonderland.txt"),  # Alice in Wonderland
    ("https://www.gutenberg.org/files/1661/1661-0.txt", "sherlock_holmes.txt"),  # Sherlock Holmes
]

for url, filename in gutenberg_urls:
    try:
        print(f"  Downloading {filename}...")
        output_path = datasets_dir / filename
        urllib.request.urlretrieve(url, output_path)
        # Check size
        size = output_path.stat().st_size
        print(f"  ✓ Downloaded {filename} ({size:,} bytes)")
    except Exception as e:
        print(f"  ✗ Failed to download {filename}: {e}")

print("\n=== Dataset Download Complete ===")
print(f"Location: {datasets_dir}")
print("\nThese are all public domain or open-license texts.")
print("Safe for training without legal issues.")
