Add top-gap weighted raw candidate selection telemetry (sprint 256)
This commit is contained in:
@@ -53,6 +53,8 @@ NATIVE_RAW_CANDIDATE_SEARCH="${WSTONE_NATIVE_RAW_CANDIDATE_SEARCH:-0}"
|
||||
NATIVE_RAW_CANDIDATE_MAX_VARIANTS="${WSTONE_NATIVE_RAW_CANDIDATE_MAX_VARIANTS:-8}"
|
||||
NATIVE_RAW_CANDIDATE_REQUIRE_UPLIFT="${WSTONE_NATIVE_RAW_CANDIDATE_REQUIRE_UPLIFT:-0}"
|
||||
NATIVE_RAW_HARDEN_TOP_GAPS="${WSTONE_NATIVE_RAW_HARDEN_TOP_GAPS:-0}"
|
||||
NATIVE_RAW_TOP_GAP_WEIGHTED_SELECT="${WSTONE_NATIVE_RAW_TOP_GAP_WEIGHTED_SELECT:-0}"
|
||||
NATIVE_RAW_TOP_GAP_BACKLOG_FILE="${WSTONE_NATIVE_RAW_TOP_GAP_BACKLOG_FILE:-}"
|
||||
EXTRA_NORMALIZED_REQUIREMENTS_FILE="${WSTONE_EXTRA_NORMALIZED_REQUIREMENTS_FILE:-}"
|
||||
EXTRA_TASKS_FILE="${WSTONE_EXTRA_TASKS_FILE:-}"
|
||||
CAPABILITY_SIGNALS_JSON="${WSTONE_CAPABILITY_SIGNALS_JSON:-}"
|
||||
@@ -392,6 +394,12 @@ else
|
||||
'{attempted:$attempted, applied:$applied, initial_task_count:$initial_task_count, target_min_task_count:$target_min_task_count}')"
|
||||
fi
|
||||
if [[ "$NATIVE_RAW_CANDIDATE_SEARCH" == "1" ]]; then
|
||||
top_gap_score_args=()
|
||||
use_top_gap_weighted_select=false
|
||||
if [[ "$NATIVE_RAW_TOP_GAP_WEIGHTED_SELECT" == "1" && -n "$NATIVE_RAW_TOP_GAP_BACKLOG_FILE" && -f "$NATIVE_RAW_TOP_GAP_BACKLOG_FILE" ]]; then
|
||||
top_gap_score_args=(--top-gaps "$NATIVE_RAW_TOP_GAP_BACKLOG_FILE")
|
||||
use_top_gap_weighted_select=true
|
||||
fi
|
||||
python3 "$ROOT_DIR/tools/mcp/synthesize_raw_candidate_requirement_variants.py" \
|
||||
--spec "$INPUT_FILE" \
|
||||
--profiles "$NATIVE_IMPACT_COVERAGE_PROFILES" \
|
||||
@@ -404,12 +412,15 @@ if [[ "$NATIVE_RAW_CANDIDATE_SEARCH" == "1" ]]; then
|
||||
--spec "$INPUT_FILE" \
|
||||
--profiles "$NATIVE_IMPACT_COVERAGE_PROFILES" \
|
||||
--tasks "$OUT_DIR/02ae_candidate_0_tasks.json" \
|
||||
"${top_gap_score_args[@]}" \
|
||||
--out "$OUT_DIR/02ae_candidate_0_score.json" >/dev/null
|
||||
best_variant="0"
|
||||
best_fail="$(jq '.failing_profile_count // 999' "$OUT_DIR/02ae_candidate_0_score.json")"
|
||||
best_task_count="$(jq '.task_count // 0' "$OUT_DIR/02ae_candidate_0_score.json")"
|
||||
best_top_gap_score="$(jq '.top_gap_score // 0' "$OUT_DIR/02ae_candidate_0_score.json")"
|
||||
baseline_fail="$best_fail"
|
||||
baseline_task_count="$best_task_count"
|
||||
baseline_top_gap_score="$best_top_gap_score"
|
||||
attempted=1
|
||||
successful=1
|
||||
|
||||
@@ -434,13 +445,28 @@ if [[ "$NATIVE_RAW_CANDIDATE_SEARCH" == "1" ]]; then
|
||||
--spec "$INPUT_FILE" \
|
||||
--profiles "$NATIVE_IMPACT_COVERAGE_PROFILES" \
|
||||
--tasks "$OUT_DIR/02ae_candidate_${variant}_tasks.json" \
|
||||
"${top_gap_score_args[@]}" \
|
||||
--out "$OUT_DIR/02ae_candidate_${variant}_score.json" >/dev/null
|
||||
cand_fail="$(jq '.failing_profile_count // 999' "$OUT_DIR/02ae_candidate_${variant}_score.json")"
|
||||
cand_task_count="$(jq '.task_count // 0' "$OUT_DIR/02ae_candidate_${variant}_score.json")"
|
||||
if [[ "$cand_fail" -lt "$best_fail" || ( "$cand_fail" -eq "$best_fail" && "$cand_task_count" -gt "$best_task_count" ) ]]; then
|
||||
cand_top_gap_score="$(jq '.top_gap_score // 0' "$OUT_DIR/02ae_candidate_${variant}_score.json")"
|
||||
should_select=false
|
||||
if [[ "$cand_fail" -lt "$best_fail" ]]; then
|
||||
should_select=true
|
||||
elif [[ "$cand_fail" -eq "$best_fail" ]]; then
|
||||
if [[ "$use_top_gap_weighted_select" == true && "$cand_top_gap_score" -gt "$best_top_gap_score" ]]; then
|
||||
should_select=true
|
||||
elif [[ "$cand_top_gap_score" -eq "$best_top_gap_score" && "$cand_task_count" -gt "$best_task_count" ]]; then
|
||||
should_select=true
|
||||
elif [[ "$use_top_gap_weighted_select" != true && "$cand_task_count" -gt "$best_task_count" ]]; then
|
||||
should_select=true
|
||||
fi
|
||||
fi
|
||||
if [[ "$should_select" == true ]]; then
|
||||
best_variant="$variant"
|
||||
best_fail="$cand_fail"
|
||||
best_task_count="$cand_task_count"
|
||||
best_top_gap_score="$cand_top_gap_score"
|
||||
TASKS="$cand_tasks"
|
||||
fi
|
||||
successful=$((successful + 1))
|
||||
@@ -456,10 +482,14 @@ if [[ "$NATIVE_RAW_CANDIDATE_SEARCH" == "1" ]]; then
|
||||
--arg best_variant "$best_variant" \
|
||||
--argjson baseline_failing_profile_count "$baseline_fail" \
|
||||
--argjson baseline_task_count "$baseline_task_count" \
|
||||
--argjson baseline_top_gap_score "$baseline_top_gap_score" \
|
||||
--argjson best_failing_profile_count "$best_fail" \
|
||||
--argjson best_task_count "$best_task_count" \
|
||||
--argjson best_top_gap_score "$best_top_gap_score" \
|
||||
--argjson available_variants "$variant_total" \
|
||||
'{enabled:$enabled, attempted_variants:$attempted, successful_variants:$successful, available_variants:$available_variants, selected_variant:$best_variant, baseline_failing_profile_count:$baseline_failing_profile_count, baseline_task_count:$baseline_task_count, best_failing_profile_count:$best_failing_profile_count, best_task_count:$best_task_count}')"
|
||||
--argjson top_gap_weighted_select "$use_top_gap_weighted_select" \
|
||||
--arg top_gap_backlog_file "$NATIVE_RAW_TOP_GAP_BACKLOG_FILE" \
|
||||
'{enabled:$enabled, attempted_variants:$attempted, successful_variants:$successful, available_variants:$available_variants, selected_variant:$best_variant, baseline_failing_profile_count:$baseline_failing_profile_count, baseline_task_count:$baseline_task_count, baseline_top_gap_score:$baseline_top_gap_score, best_failing_profile_count:$best_failing_profile_count, best_task_count:$best_task_count, best_top_gap_score:$best_top_gap_score, top_gap_weighted_select:$top_gap_weighted_select, top_gap_backlog_file:$top_gap_backlog_file}')"
|
||||
printf '%s\n' "$NATIVE_RAW_CANDIDATE_SEARCH_JSON" > "$OUT_DIR/02ae_raw_candidate_search.json"
|
||||
if [[ "$NATIVE_RAW_CANDIDATE_REQUIRE_UPLIFT" == "1" && "$best_fail" -ge "$baseline_fail" ]]; then
|
||||
echo "error: raw candidate search did not improve failing profile count" >&2
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
import argparse
|
||||
import json
|
||||
from pathlib import Path
|
||||
from typing import Dict, List
|
||||
from typing import Dict, List, Optional, Tuple
|
||||
|
||||
|
||||
def load_json(path: Path):
|
||||
@@ -35,6 +35,55 @@ def collect_task_unions(tasks: List[Dict]):
|
||||
return reasons, prereq_ops, contracts
|
||||
|
||||
|
||||
def parse_missing_signal(signal: str) -> Tuple[str, str]:
|
||||
if ":" not in signal:
|
||||
return signal, ""
|
||||
k, v = signal.split(":", 1)
|
||||
return k, v
|
||||
|
||||
|
||||
def signal_present(signal: str, tasks: List[Dict]) -> bool:
|
||||
if signal.startswith("native_task_count<"):
|
||||
try:
|
||||
min_count = int(signal.split("<", 1)[1])
|
||||
except (ValueError, IndexError):
|
||||
return False
|
||||
return len(tasks) >= min_count
|
||||
|
||||
kind, value = parse_missing_signal(signal)
|
||||
reasons, prereq_ops, contracts = collect_task_unions(tasks)
|
||||
|
||||
if kind == "missing_prerequisite_op":
|
||||
return value in set(prereq_ops)
|
||||
if kind == "missing_execution_contract":
|
||||
for ec in contracts:
|
||||
if isinstance(ec.get(value), bool):
|
||||
if ec.get(value):
|
||||
return True
|
||||
elif value in ec:
|
||||
return True
|
||||
return False
|
||||
if kind == "missing_reason_keyword":
|
||||
v = value.lower()
|
||||
return any(v in r for r in reasons)
|
||||
return False
|
||||
|
||||
|
||||
def load_top_gap_items(path: Optional[Path]) -> List[Dict]:
|
||||
if path is None or not path.exists():
|
||||
return []
|
||||
data = load_json(path)
|
||||
rows = list(data.get("prioritized_missing") or [])
|
||||
out: List[Dict] = []
|
||||
for row in rows:
|
||||
signal = str(row.get("missing", ""))
|
||||
if not signal:
|
||||
continue
|
||||
weight = int(row.get("count", 1) or 1)
|
||||
out.append({"missing": signal, "weight": weight})
|
||||
return out
|
||||
|
||||
|
||||
def check_profile(profile: Dict, tasks: List[Dict]) -> Dict:
|
||||
reasons, prereq_ops, contracts = collect_task_unions(tasks)
|
||||
missing: List[str] = []
|
||||
@@ -74,6 +123,7 @@ def main() -> int:
|
||||
p.add_argument("--profiles", required=True)
|
||||
p.add_argument("--tasks", required=True)
|
||||
p.add_argument("--out", required=True)
|
||||
p.add_argument("--top-gaps", default="", help="Optional raw gap backlog JSON with prioritized_missing.")
|
||||
args = p.parse_args()
|
||||
|
||||
spec_text = Path(args.spec).read_text(encoding="utf-8", errors="ignore")
|
||||
@@ -87,19 +137,45 @@ def main() -> int:
|
||||
|
||||
checks = [check_profile(pf, tasks) for pf in active]
|
||||
failing = [c for c in checks if not c.get("passed", False)]
|
||||
top_gap_items = load_top_gap_items(Path(args.top_gaps) if args.top_gaps else None)
|
||||
top_gap_signal_hits = 0
|
||||
top_gap_score = 0
|
||||
top_gap_signal_weight_total = 0
|
||||
for item in top_gap_items:
|
||||
sig = str(item.get("missing", ""))
|
||||
w = int(item.get("weight", 1) or 1)
|
||||
top_gap_signal_weight_total += w
|
||||
if signal_present(sig, tasks):
|
||||
top_gap_signal_hits += 1
|
||||
top_gap_score += w
|
||||
|
||||
result = {
|
||||
"status": "ok" if not failing else "fail",
|
||||
"task_count": len(tasks),
|
||||
"active_profile_count": len(active),
|
||||
"failing_profile_count": len(failing),
|
||||
"top_gap_signal_count": len(top_gap_items),
|
||||
"top_gap_signal_hits": top_gap_signal_hits,
|
||||
"top_gap_signal_weight_total": top_gap_signal_weight_total,
|
||||
"top_gap_score": top_gap_score,
|
||||
"checks": checks,
|
||||
}
|
||||
with Path(args.out).open("w", encoding="utf-8") as f:
|
||||
json.dump(result, f, indent=2, sort_keys=True)
|
||||
f.write("\n")
|
||||
|
||||
print(json.dumps({"status": result["status"], "task_count": result["task_count"], "failing_profile_count": result["failing_profile_count"], "out": args.out}, sort_keys=True))
|
||||
print(
|
||||
json.dumps(
|
||||
{
|
||||
"status": result["status"],
|
||||
"task_count": result["task_count"],
|
||||
"failing_profile_count": result["failing_profile_count"],
|
||||
"top_gap_score": result["top_gap_score"],
|
||||
"out": args.out,
|
||||
},
|
||||
sort_keys=True,
|
||||
)
|
||||
)
|
||||
return 0
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user