#!/usr/bin/env python3 """ gen_automatability_data.py Labels the editor/src/*.h headers to produce training data for the automatability_strategy specialist. The six tiers (matching AutomatabilityAnnotation.strategy in Annotation.h): deterministic — Same input always produces same output. Pure formula, rule engine, or algorithmic transform. Zero parameters. Examples: validators, scorers, parsers, extractors, trackers. template — Deterministic composition of known structural patterns. Code generators driven by fixed templates; no ambiguity. Examples: code generators, builders, highlighters, register tools. specialist — Bounded output vocabulary (fixed N classes). Probabilistic but constrained. ~800KB model, microseconds. Examples: selectors with uncertainty, judges, advisors, annotators that make probabilistic choices. slm — Open vocabulary but local-scale. Complex transformation or interpretation where output tokens vary but scope is bounded. ~1-7GB. Examples: transpilers, type translators, optimizers. llm — Full generative reasoning. Cross-context, novel output, planning-level decisions. Examples: architect intake, decomposers, agent code gen, high-level orchestrators. human — Not yet formalizable. Requires manual judgment, approval, or editorial oversight. Examples: mutation approval, review surfaces, escalation. Input text format (what the specialist sees at inference time): "name=ClassName" This mirrors what Claude provides when calling add_skeleton_node — the class name is the primary signal. The Whetstone naming convention enforces semantic suffixes, making the name a strong predictor. Output: TSV with columns: labelhopstext label: integer index 0–5 0=deterministic 1=template 2=specialist 3=slm 4=llm 5=human hops: always 0 text: "name=ClassName" Usage: python3 specialists/scripts/gen_automatability_data.py \\ --src-dir editor/src \\ --train-out specialists/data/generated/automatability_train.tsv \\ --eval-out specialists/data/generated/automatability_eval.tsv """ import re import sys import random import argparse from pathlib import Path from collections import Counter # --------------------------------------------------------------------------- # Classification heuristics — suffix and prefix patterns on class names # --------------------------------------------------------------------------- # # Priority order: human > llm > slm > specialist > template > deterministic # (more specific / rarer tiers checked first to avoid being swallowed by # the large deterministic default) # human: requires manual judgment, approval workflows HUMAN_SUFFIXES = ( "Approval", "ReviewInterface", "ReviewSurface", "MutationPreview", "EscalationPlanner", "ApprovalGate", ) HUMAN_SUBSTRINGS = ( "Approval", "HumanReview", "ManualGate", "ReviewBoard", ) # llm: high-level reasoning, planning, novel generation LLM_PREFIXES = ( "Agent", # AgentCodeGen, AgentMutationApproval (covered by human above) ) LLM_SUFFIXES = ( "IntakeProcessor", "Decomposer", "ProjectPipeline", "ModuleDecomposer", "ProblemParser", "MultiLanguageOrchestrator", "WorkflowOrchestrator", "PolyglotOrchestrator", "PolyglotSuiteOrchestrator", "TechStackSelector", # open-ended tech selection = llm (vs bounded = specialist) "ScaffoldGenerator", # spec-to-scaffold = llm-driven ) LLM_SUBSTRINGS = ( "Planner", "Strategist", ) # slm: complex transformation, open output vocab but local scope SLM_SUFFIXES = ( "Transpiler", "TypeSystemTranslator", "TypeAwareMappings", "TransformEngine", "TransformEngineExtended", "Optimizer", "Lowering", ) SLM_SUBSTRINGS = ( "Transpil", "Transform", "Translat", # TranslationReport is excluded below via DETERMINISTIC_OVERRIDES ) # Names that would otherwise match a higher tier but are actually deterministic. # Checked before any tier classification. DETERMINISTIC_OVERRIDES = { "TranslationReport", # report doc, not a transformer "DWARFBoundaryAnnotator", # rule-based DWARF annotation "IntakeTextUtil", # text utility, not an intake processor } # specialist: bounded probabilistic — selectors, judges, advisors, annotators SPECIALIST_SUFFIXES = ( "Selector", "Judge", "Advisor", "Recommender", "Annotator", "Predictor", "Classifier", ) SPECIALIST_SUBSTRINGS = ( "Selector", "Judge", "Advisor", ) # template: code/output generation from fixed patterns TEMPLATE_PREFIXES = ( "Register", # RegisterXxxTools.h — tool registration templates "SyntaxHighlighter", ) TEMPLATE_SUFFIXES = ( "Generator", "Builder", "Factory", "Templates", "Highlighter", "Emitter", "Renderer", "Serializer", "Encoder", "Composer", "Formatter", # output formatters = deterministic, but code formatters = template "CodeGen", ) TEMPLATE_SUBSTRINGS = ( "Generator", "Builder", "Templates", ) # deterministic: pure computation — everything else # (validators, scorers, parsers, extractors, trackers, monitors, etc.) LABEL_ORDER = ["deterministic", "template", "specialist", "slm", "llm", "human"] LABEL_INDEX = {lbl: i for i, lbl in enumerate(LABEL_ORDER)} def classify(class_name: str) -> str: n = class_name # keep original case for prefix/suffix matching # Hard overrides — names that pattern-match a higher tier but are deterministic if n in DETERMINISTIC_OVERRIDES: return "deterministic" # human — must check before llm (some overlap on Agent* names) for suf in HUMAN_SUFFIXES: if n.endswith(suf): return "human" for sub in HUMAN_SUBSTRINGS: if sub in n: return "human" # llm — planning-level, open-ended for pre in LLM_PREFIXES: if n.startswith(pre): return "llm" for suf in LLM_SUFFIXES: if n.endswith(suf): return "llm" for sub in LLM_SUBSTRINGS: if sub in n: return "llm" # slm — complex transformation for suf in SLM_SUFFIXES: if n.endswith(suf): return "slm" for sub in SLM_SUBSTRINGS: if sub in n: return "slm" # specialist — bounded probabilistic decision for suf in SPECIALIST_SUFFIXES: if n.endswith(suf): return "specialist" for sub in SPECIALIST_SUBSTRINGS: if sub in n: return "specialist" # template — code/output generation for pre in TEMPLATE_PREFIXES: if n.startswith(pre): return "template" for suf in TEMPLATE_SUFFIXES: if n.endswith(suf): return "template" for sub in TEMPLATE_SUBSTRINGS: if sub in n: return "template" # deterministic — default return "deterministic" def header_to_class_name(path: Path) -> str: """Convert FileName.h → FileName (strip extension).""" return path.stem # --------------------------------------------------------------------------- # Main # --------------------------------------------------------------------------- def main(): ap = argparse.ArgumentParser() ap.add_argument("--src-dir", default="editor/src", help="Directory containing *.h headers (non-recursive)") ap.add_argument("--train-out", default="specialists/data/generated/automatability_train.tsv") ap.add_argument("--eval-out", default="specialists/data/generated/automatability_eval.tsv") ap.add_argument("--eval-frac", type=float, default=0.15) ap.add_argument("--seed", type=int, default=42) ap.add_argument("--show-labels", action="store_true", help="Print each header and its assigned label (for review)") args = ap.parse_args() src_dir = Path(args.src_dir) headers = sorted(src_dir.glob("*.h")) if not headers: print(f"No .h files found in {src_dir}", file=sys.stderr) sys.exit(1) print(f"Found {len(headers)} header files") all_examples: list[tuple[str, str]] = [] for h in headers: class_name = header_to_class_name(h) label = classify(class_name) all_examples.append((label, class_name)) if args.show_labels: print(f" {label:15s} {class_name}") counts = Counter(lbl for lbl, _ in all_examples) print("Raw class distribution:", dict(counts)) # Balance: use the median class size as target so we keep more deterministic # examples (the dominant real-world case) while still oversampling sparse tiers. sorted_counts = sorted(counts.values()) median_count = sorted_counts[len(sorted_counts) // 2] target = min(median_count * 2, max(sorted_counts)) target = max(target, 20) print(f"Balancing to {target} examples per class") rng = random.Random(args.seed) by_class: dict[str, list[str]] = {} for lbl, name in all_examples: by_class.setdefault(lbl, []).append(name) balanced: list[tuple[str, str]] = [] for lbl in LABEL_ORDER: names = by_class.get(lbl, []) if not names: print(f"WARNING: no examples for class '{lbl}'", file=sys.stderr) continue rng.shuffle(names) chosen = names[:target] while len(chosen) < target: chosen += names balanced.extend((lbl, name) for name in chosen[:target]) rng.shuffle(balanced) print(f"Balanced total: {len(balanced)} examples") n_eval = max(1, int(len(balanced) * args.eval_frac)) eval_set = balanced[:n_eval] train_set = balanced[n_eval:] Path(args.train_out).parent.mkdir(parents=True, exist_ok=True) Path(args.eval_out).parent.mkdir(parents=True, exist_ok=True) def write_tsv(path, rows): with open(path, "w") as f: for lbl, name in rows: idx = LABEL_INDEX[lbl] f.write(f"{idx}\t0\tname={name}\n") print(f"Wrote {len(rows)} rows → {path}") write_tsv(args.train_out, train_set) write_tsv(args.eval_out, eval_set) for split_name, split in [("train", train_set), ("eval", eval_set)]: c = Counter(lbl for lbl, _ in split) print(f" {split_name}: {dict(sorted(c.items()))}") if __name__ == "__main__": main()