Steps 309-319: Sprint 11 Phases 11d-e — Kotlin/C# languages + workflow annotation foundation
Phase 11d (Steps 309-313): Kotlin + C# parsers and generators - KotlinParser (regex-based): fun, suspend fun, class, data class, val/var - KotlinGenerator: idiomatic Kotlin output with type mappings - CSharpParser (regex-based): methods, async, class, interface - CSharpGenerator: Allman braces, foreach, Task async, LINQ types - Pipeline integration for both languages, 10 parsers + 10 generators Phase 11e (Steps 314-319): Workflow annotation foundation - AnnotationInference: generalized multi-subject inference engine - Subject 9 routing annotations: ContextWidth, Review, Ambiguity, Automatability, Priority, ImplementationStatus - SkeletonAST: project specification before code exists - Architect tooling: createSkeleton, addSkeletonNode, getProjectModel, inferAnnotations RPCs + 4 MCP tools (42+ total) - Inference-to-routing bridge: complexity→ambiguity, getter→deterministic - TrainingDataExporter + TrainingDataGenerator scaffolding Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
This commit is contained in:
67
editor/src/TrainingDataExporter.h
Normal file
67
editor/src/TrainingDataExporter.h
Normal file
@@ -0,0 +1,67 @@
|
||||
#pragma once
|
||||
#include "TrainingDataGenerator.h"
|
||||
#include <nlohmann/json.hpp>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
#include <sstream>
|
||||
#include <map>
|
||||
|
||||
using json = nlohmann::json;
|
||||
|
||||
struct ExportStats {
|
||||
int totalPairs = 0;
|
||||
std::map<std::string, int> annotationCounts; // type -> count
|
||||
std::map<std::string, int> languageCounts; // language -> count
|
||||
};
|
||||
|
||||
class TrainingDataExporter {
|
||||
public:
|
||||
// HuggingFace instruction/input/output format (JSONL)
|
||||
static std::string exportHuggingFace(const std::vector<TrainingPair>& pairs) {
|
||||
std::ostringstream oss;
|
||||
for (const auto& p : pairs) {
|
||||
json line;
|
||||
line["instruction"] = "Add semantic annotations (Semanno format) to the following " +
|
||||
p.language + " code.";
|
||||
line["input"] = p.rawCode;
|
||||
line["output"] = p.annotatedCode;
|
||||
oss << line.dump() << "\n";
|
||||
}
|
||||
return oss.str();
|
||||
}
|
||||
|
||||
// Raw/annotated pairs (JSONL)
|
||||
static std::string exportPairsJSONL(const std::vector<TrainingPair>& pairs) {
|
||||
std::ostringstream oss;
|
||||
for (const auto& p : pairs) {
|
||||
json line;
|
||||
line["raw_code"] = p.rawCode;
|
||||
line["annotated_code"] = p.annotatedCode;
|
||||
line["language"] = p.language;
|
||||
line["annotations"] = p.annotations;
|
||||
oss << line.dump() << "\n";
|
||||
}
|
||||
return oss.str();
|
||||
}
|
||||
|
||||
// Compute statistics
|
||||
static ExportStats computeStats(const std::vector<TrainingPair>& pairs) {
|
||||
ExportStats stats;
|
||||
stats.totalPairs = (int)pairs.size();
|
||||
for (const auto& p : pairs) {
|
||||
stats.languageCounts[p.language]++;
|
||||
for (const auto& a : p.annotations) {
|
||||
stats.annotationCounts[a]++;
|
||||
}
|
||||
}
|
||||
return stats;
|
||||
}
|
||||
|
||||
static json statsToJson(const ExportStats& stats) {
|
||||
json j;
|
||||
j["totalPairs"] = stats.totalPairs;
|
||||
j["annotationCounts"] = stats.annotationCounts;
|
||||
j["languageCounts"] = stats.languageCounts;
|
||||
return j;
|
||||
}
|
||||
};
|
||||
Reference in New Issue
Block a user