from datasets import load_dataset
from pathlib import Path

SYSTEM_PROMPT = "You are a helpful synthetic assistant that speaks concise English."


def build_user_prompt(instruction: str, context: str | None) -> str:
    instruction = (instruction or "").strip()
    context = (context or "").strip()
    if context:
        return f"{instruction}\n\nContext:\n{context}" if instruction else context
    return instruction or "Please respond helpfully to the user."


def format_segment(row: dict) -> str:
    instruction = row.get("instruction", "")
    context = row.get("context") or row.get("input") or ""
    response = (row.get("response") or "").strip()
    user_content = build_user_prompt(instruction, context).strip()
    return (
        f"<|system|>{SYSTEM_PROMPT}<|user|>{user_content}<|assistant|>{response}"
    )


def main() -> None:
    dataset = load_dataset("databricks/databricks-dolly-15k", split="train")
    output_path = Path("data/local_benchmarks/dolly15k_corpus.txt")
    output_path.parent.mkdir(parents=True, exist_ok=True)
    count = 0
    with output_path.open("w", encoding="utf-8") as handle:
        for row in dataset:
            segment = format_segment(row).strip()
            if not segment:
                continue
            handle.write(segment + "\n\n")
            count += 1
    print(f"Wrote {count} segments to {output_path}")


if __name__ == "__main__":
    main()
