#!/usr/bin/env python3
"""Assign answer letters to the item bank deterministically and validate it.

Input : items/item_bank.json   (authored content: correct, false, two distractors)
Output: data/questions.json    (frozen stimulus set with A-D letters)

Letter positions are assigned with a fixed seed so that the correct answer and the
supplied false answer are spread across positions A-D and are not correlated with
each other. This removes any confound between option position and the studied
manipulation, and makes the frozen stimulus set reproducible byte-for-byte.
"""

import json
import random
from collections import Counter
from pathlib import Path

SEED = 20260923
LETTERS = ["A", "B", "C", "D"]

ROOT = Path(__file__).resolve().parent.parent
IN_PATH = ROOT / "items" / "item_bank.json"
OUT_PATH = ROOT / "data" / "questions.json"


def main() -> None:
    bank = json.loads(IN_PATH.read_text(encoding="utf-8"))
    rng = random.Random(SEED)

    questions = []
    for item in bank["items"]:
        contents = [
            item["correct_answer"],
            item["false_answer"],
            item["distractor_1"],
            item["distractor_2"],
        ]
        if len(set(contents)) != 4:
            raise ValueError(f"{item['id']}: option texts are not distinct: {contents}")

        order = [0, 1, 2, 3]
        rng.shuffle(order)
        options = {LETTERS[position]: contents[source] for position, source in enumerate(order)}

        correct_letter = LETTERS[order.index(0)]
        false_letter = LETTERS[order.index(1)]

        questions.append(
            {
                "id": item["id"],
                "domain": item["domain"],
                "question": item["question"],
                "options": options,
                "correct_letter": correct_letter,
                "false_letter": false_letter,
                "correct_answer": item["correct_answer"],
                "false_answer": item["false_answer"],
            }
        )

    ids = [q["id"] for q in questions]
    if len(set(ids)) != len(ids):
        raise ValueError("duplicate item ids")

    payload = {
        "schema_version": "1.0",
        "seed": SEED,
        "n_items": len(questions),
        "stimulus_disclosure": bank["stimulus_disclosure"],
        "questions": questions,
    }
    OUT_PATH.parent.mkdir(parents=True, exist_ok=True)
    OUT_PATH.write_text(json.dumps(payload, indent=2, ensure_ascii=False) + "\n", encoding="utf-8")

    correct_positions = Counter(q["correct_letter"] for q in questions)
    false_positions = Counter(q["false_letter"] for q in questions)
    domains = Counter(q["domain"] for q in questions)

    print(f"wrote {OUT_PATH} with {len(questions)} questions")
    print("correct answer position:", dict(sorted(correct_positions.items())))
    print("false answer position  :", dict(sorted(false_positions.items())))
    print("domains                :", dict(sorted(domains.items())))


if __name__ == "__main__":
    main()
