diff --git a/export_nucleic_prompts.py b/export_nucleic_prompts.py new file mode 100644 index 0000000..a941753 --- /dev/null +++ b/export_nucleic_prompts.py @@ -0,0 +1,225 @@ +#!/usr/bin/env python3 +"""Export the first user prompt from each Nucleic transcript for purpose labeling.""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import unicodedata +from collections import Counter +from datetime import datetime, timezone +from pathlib import Path +from typing import Any + + +LABELS = ( + "planning", + "backendImpl", + "frontendImpl", + "quickFix", + "refactor", + "debugging", + "review", + "writing", +) + +LABEL_DEFINITIONS = { + "planning": "Architecture, design, migration strategy, or multi-step project planning.", + "backendImpl": "Server, API, data, algorithm, systems, or CLI implementation.", + "frontendImpl": "UI, views, styling, components, layout, or animation implementation.", + "quickFix": "A typo, one-liner, config tweak, or small contained bug fix.", + "refactor": "Restructuring, renaming, extraction, or cleanup without behavior change.", + "debugging": "Diagnosing a crash, failure, regression, or unexplained behavior.", + "review": "Reviewing, explaining, auditing, or answering questions about existing code.", + "writing": "Documentation, commit messages, summaries, formatting, or other prose work.", +} + + +class ExportError(RuntimeError): + """The transcript export could not be completed safely.""" + + +def normalized_prompt(prompt: str) -> str: + """Match the classifier dataset's stable prompt normalization.""" + + return " ".join(unicodedata.normalize("NFKC", prompt).split()) + + +def prompt_hash(prompt: str) -> str: + return hashlib.sha256(normalized_prompt(prompt).casefold().encode("utf-8")).hexdigest() + + +def canonical_json(value: dict[str, Any]) -> str: + return json.dumps(value, ensure_ascii=False, separators=(",", ":")) + + +def read_session(transcript: Path) -> tuple[dict[str, Any], str | None]: + header: dict[str, Any] | None = None + first_prompt: str | None = None + + with transcript.open(encoding="utf-8") as handle: + for line_number, line in enumerate(handle, start=1): + try: + value = json.loads(line) + except json.JSONDecodeError as error: + raise ExportError( + f"{transcript}:{line_number}: malformed JSON: {error.msg}" + ) from error + + if not isinstance(value, dict): + raise ExportError(f"{transcript}:{line_number}: expected a JSON object") + + if line_number == 1: + header = value + continue + + kind = value.get("kind") + if isinstance(kind, dict) and kind.get("type") == "userText": + text = kind.get("text") + if not isinstance(text, str) or not text.strip(): + raise ExportError( + f"{transcript}:{line_number}: first userText has no prompt text" + ) + first_prompt = text + break + + if header is None: + raise ExportError(f"{transcript}: transcript is empty") + return header, first_prompt + + +def session_sort_key(record: dict[str, Any]) -> tuple[str, str]: + return str(record.get("createdAt") or ""), str(record["sessionID"]) + + +def export(sessions_dir: Path, output: Path, manifest_path: Path) -> dict[str, Any]: + if not sessions_dir.is_dir(): + raise ExportError(f"sessions directory does not exist: {sessions_dir}") + + records: list[dict[str, Any]] = [] + omitted: list[dict[str, Any]] = [] + backend_counts: Counter[str] = Counter() + session_directories = sorted(path for path in sessions_dir.iterdir() if path.is_dir()) + + for session_dir in session_directories: + transcript = session_dir / "transcript.jsonl" + if not transcript.is_file(): + raise ExportError(f"missing transcript: {transcript}") + + header, prompt = read_session(transcript) + session_id = header.get("sessionID") + if session_id != session_dir.name: + raise ExportError( + f"{transcript}: header sessionID {session_id!r} does not match directory" + ) + + backend = str(header.get("backend") or "unknown") + backend_counts[backend] += 1 + common = { + "sessionID": session_id, + "createdAt": header.get("createdAt"), + "backend": backend, + } + + if prompt is None: + omitted.append({**common, "reason": "no_user_text"}) + continue + + records.append( + { + "schemaVersion": 1, + **common, + "promptHash": prompt_hash(prompt), + "prompt": prompt, + "purpose": None, + "labelNotes": None, + } + ) + + records.sort(key=session_sort_key) + omitted.sort(key=session_sort_key) + + output.parent.mkdir(parents=True, exist_ok=True) + output.write_text( + "".join(f"{canonical_json(record)}\n" for record in records), + encoding="utf-8", + ) + output_sha256 = hashlib.sha256(output.read_bytes()).hexdigest() + duplicate_session_prompts = sum( + count - 1 for count in Counter(record["promptHash"] for record in records).values() + if count > 1 + ) + + manifest = { + "schemaVersion": 1, + "generatedAt": datetime.now(timezone.utc).isoformat().replace("+00:00", "Z"), + "source": { + "kind": "nucleic_transcript_history", + "sessionsDirectory": str(sessions_dir), + "selection": "first non-empty AgentEvent kind=userText in each transcript", + "deduplicated": False, + }, + "output": { + "path": str(output), + "format": "jsonl", + "sha256": output_sha256, + }, + "counts": { + "sessionDirectories": len(session_directories), + "exportedPrompts": len(records), + "sessionsWithoutUserPrompt": len(omitted), + "duplicateSessionPrompts": duplicate_session_prompts, + "backends": dict(sorted(backend_counts.items())), + }, + "labelingContract": { + "field": "purpose", + "unlabeledValue": None, + "allowedValues": list(LABELS), + "generalIsNotALabel": True, + "definitions": LABEL_DEFINITIONS, + "instruction": ( + "Assign exactly one allowed purpose to every JSONL record. " + "Preserve all other fields and the original record order." + ), + }, + "sessionsWithoutUserPrompt": omitted, + } + manifest_path.parent.mkdir(parents=True, exist_ok=True) + manifest_path.write_text( + json.dumps(manifest, ensure_ascii=False, indent=2, sort_keys=True) + "\n", + encoding="utf-8", + ) + return manifest + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument( + "--sessions-dir", + type=Path, + required=True, + help="Nucleic Application Support sessions directory", + ) + parser.add_argument("--output", type=Path, required=True, help="Unlabeled JSONL output") + parser.add_argument( + "--manifest", + type=Path, + required=True, + help="JSON manifest describing the output and labeling contract", + ) + return parser.parse_args() + + +def main() -> None: + args = parse_args() + manifest = export( + args.sessions_dir.expanduser().resolve(), + args.output.expanduser().resolve(), + args.manifest.expanduser().resolve(), + ) + print(json.dumps(manifest["counts"], sort_keys=True)) + + +if __name__ == "__main__": + main()