mirror of
https://github.com/dotnet/skills.git
synced 2026-09-20 09:49:54 +08:00
15866 lines
646 KiB
JSON
15866 lines
646 KiB
JSON
{
|
|
"entries": {
|
|
"Efficiency": [
|
|
{
|
|
"benches": [
|
|
{
|
|
"overfittingScore": 0.29,
|
|
"notActivated": true,
|
|
"unit": "seconds",
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Time",
|
|
"value": 8.8
|
|
},
|
|
{
|
|
"overfittingScore": 0.29,
|
|
"notActivated": true,
|
|
"unit": "tokens",
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Tokens In",
|
|
"value": 13092.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Time",
|
|
"value": 2.2,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.29
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Tokens In",
|
|
"value": 12041.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.29
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Time",
|
|
"value": 6.7,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Tokens In",
|
|
"value": 12279.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"overfittingScore": 0.29,
|
|
"notActivated": true,
|
|
"unit": "seconds",
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Time",
|
|
"value": 1.5
|
|
},
|
|
{
|
|
"overfittingScore": 0.29,
|
|
"notActivated": true,
|
|
"unit": "tokens",
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Tokens In",
|
|
"value": 11958.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Time",
|
|
"value": 1.4,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.29
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Tokens In",
|
|
"value": 11967.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.29
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Time",
|
|
"value": 3.0,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Tokens In",
|
|
"value": 11774.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Time",
|
|
"value": 3.9,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.29
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Tokens In",
|
|
"value": 26217.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.29
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Time",
|
|
"value": 3.0,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.29
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Tokens In",
|
|
"value": 26146.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.29
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Time",
|
|
"value": 3.5,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Tokens In",
|
|
"value": 11830.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"overfittingScore": 0.29,
|
|
"notActivated": true,
|
|
"unit": "seconds",
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Time",
|
|
"value": 6.3
|
|
},
|
|
{
|
|
"overfittingScore": 0.29,
|
|
"notActivated": true,
|
|
"unit": "tokens",
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Tokens In",
|
|
"value": 12505.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Time",
|
|
"value": 6.3,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.29
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Tokens In",
|
|
"value": 12589.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.29
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Time",
|
|
"value": 7.1,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Tokens In",
|
|
"value": 12291.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Time",
|
|
"value": 6.2,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.29
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Tokens In",
|
|
"value": 26546.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.29
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Time",
|
|
"value": 9.8,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.29
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Tokens In",
|
|
"value": 26692.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.29
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Time",
|
|
"value": 11.4,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Tokens In",
|
|
"value": 12814.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Time",
|
|
"value": 7.2,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.29
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Tokens In",
|
|
"value": 55120.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.29
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Time",
|
|
"value": 6.9,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.29
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Tokens In",
|
|
"value": 26750.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.29
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Time",
|
|
"value": 2.5,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Tokens In",
|
|
"value": 11872.0,
|
|
"unit": "tokens"
|
|
}
|
|
],
|
|
"date": 1788627832039,
|
|
"commit": {
|
|
"author": {
|
|
"name": "Amaury Levé",
|
|
"username": "AbhitejJohn"
|
|
},
|
|
"message": "Improve non-passing dotnet-test scenarios (#1114)",
|
|
"id": "ac8f41264bdd557e58924a3110eea8e0917dcf4d",
|
|
"url": "https://github.com/dotnet/skills/commit/ac8f41264bdd557e58924a3110eea8e0917dcf4d",
|
|
"timestamp": "2026-09-04T10:21:25+00:00"
|
|
},
|
|
"tool": "customSmallerIsBetter",
|
|
"model": "gpt-5.3-codex"
|
|
},
|
|
{
|
|
"benches": [
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Time",
|
|
"value": 169.4,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Tokens In",
|
|
"value": 1363051.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Time",
|
|
"value": 114.3,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Tokens In",
|
|
"value": 426657.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Time",
|
|
"value": 172.1,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Tokens In",
|
|
"value": 628830.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"overfittingScore": 0.49,
|
|
"notActivated": true,
|
|
"unit": "seconds",
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Time",
|
|
"value": 52.6
|
|
},
|
|
{
|
|
"overfittingScore": 0.49,
|
|
"notActivated": true,
|
|
"unit": "tokens",
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Tokens In",
|
|
"value": 178956.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Time",
|
|
"value": 55.3,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Tokens In",
|
|
"value": 249307.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Time",
|
|
"value": 110.5,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Tokens In",
|
|
"value": 423481.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Time",
|
|
"value": 188.7,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Tokens In",
|
|
"value": 936677.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Time",
|
|
"value": 185.1,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Tokens In",
|
|
"value": 987345.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Time",
|
|
"value": 134.6,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Tokens In",
|
|
"value": 623444.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"overfittingScore": 0.49,
|
|
"notActivated": true,
|
|
"unit": "seconds",
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Time",
|
|
"value": 6.6
|
|
},
|
|
{
|
|
"overfittingScore": 0.49,
|
|
"notActivated": true,
|
|
"unit": "tokens",
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Tokens In",
|
|
"value": 11476.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Time",
|
|
"value": 14.3,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Tokens In",
|
|
"value": 25314.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Time",
|
|
"value": 11.0,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Tokens In",
|
|
"value": 11890.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Time",
|
|
"value": 12.7,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Tokens In",
|
|
"value": 24328.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Time",
|
|
"value": 7.4,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Tokens In",
|
|
"value": 11702.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Time",
|
|
"value": 32.4,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Tokens In",
|
|
"value": 106126.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Time",
|
|
"value": 158.6,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Tokens In",
|
|
"value": 817556.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Time",
|
|
"value": 212.0,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Tokens In",
|
|
"value": 784247.0,
|
|
"unit": "tokens"
|
|
}
|
|
],
|
|
"date": 1788627832128,
|
|
"commit": {
|
|
"author": {
|
|
"name": "Amaury Levé",
|
|
"username": "AbhitejJohn"
|
|
},
|
|
"message": "Improve non-passing dotnet-test scenarios (#1114)",
|
|
"id": "ac8f41264bdd557e58924a3110eea8e0917dcf4d",
|
|
"url": "https://github.com/dotnet/skills/commit/ac8f41264bdd557e58924a3110eea8e0917dcf4d",
|
|
"timestamp": "2026-09-04T10:21:25+00:00"
|
|
},
|
|
"tool": "customSmallerIsBetter",
|
|
"model": "mai-code-1-flash-picker"
|
|
},
|
|
{
|
|
"commit": {
|
|
"id": "ac8f41264bdd557e58924a3110eea8e0917dcf4d",
|
|
"timestamp": "2026-09-04T10:21:25+00:00",
|
|
"message": "Improve non-passing dotnet-test scenarios (#1114)",
|
|
"url": "https://github.com/dotnet/skills/commit/ac8f41264bdd557e58924a3110eea8e0917dcf4d",
|
|
"author": {
|
|
"name": "Amaury Levé",
|
|
"username": "AbhitejJohn"
|
|
}
|
|
},
|
|
"model": "claude-opus-5",
|
|
"date": 1788715663667,
|
|
"benches": [
|
|
{
|
|
"value": 168.9,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.45,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Time"
|
|
},
|
|
{
|
|
"value": 726342.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.45,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Tokens In"
|
|
},
|
|
{
|
|
"value": 200.2,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.45,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Time"
|
|
},
|
|
{
|
|
"value": 891200.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.45,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Tokens In"
|
|
},
|
|
{
|
|
"value": 214.6,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.45,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Time"
|
|
},
|
|
{
|
|
"value": 943874.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.45,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Tokens In"
|
|
},
|
|
{
|
|
"value": 183.2,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.45,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Time"
|
|
},
|
|
{
|
|
"value": 817352.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.45,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Tokens In"
|
|
},
|
|
{
|
|
"value": 131.9,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Time"
|
|
},
|
|
{
|
|
"value": 501034.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"value": 262.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.45,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Time"
|
|
},
|
|
{
|
|
"value": 920061.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.45,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Tokens In"
|
|
},
|
|
{
|
|
"value": 292.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.45,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Time"
|
|
},
|
|
{
|
|
"value": 815929.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.45,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Tokens In"
|
|
},
|
|
{
|
|
"value": 15.9,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.45,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Time"
|
|
},
|
|
{
|
|
"value": 40852.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.45,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Tokens In"
|
|
},
|
|
{
|
|
"value": 20.5,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.45,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Time"
|
|
},
|
|
{
|
|
"value": 41135.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.45,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Tokens In"
|
|
},
|
|
{
|
|
"value": 32.1,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Time"
|
|
},
|
|
{
|
|
"value": 74679.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"value": 21.7,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.45,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Time"
|
|
},
|
|
{
|
|
"value": 41329.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.45,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Tokens In"
|
|
},
|
|
{
|
|
"value": 24.7,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.45,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Time"
|
|
},
|
|
{
|
|
"value": 41498.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.45,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Tokens In"
|
|
},
|
|
{
|
|
"value": 30.8,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Time"
|
|
},
|
|
{
|
|
"value": 74462.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"value": 148.8,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.45,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Time"
|
|
},
|
|
{
|
|
"value": 499122.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.45,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Tokens In"
|
|
},
|
|
{
|
|
"value": 240.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.45,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Time"
|
|
},
|
|
{
|
|
"value": 615394.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.45,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Tokens In"
|
|
},
|
|
{
|
|
"value": 158.3,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Time"
|
|
},
|
|
{
|
|
"value": 405904.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Tokens In"
|
|
}
|
|
],
|
|
"tool": "customSmallerIsBetter"
|
|
},
|
|
{
|
|
"commit": {
|
|
"id": "ac8f41264bdd557e58924a3110eea8e0917dcf4d",
|
|
"timestamp": "2026-09-04T10:21:25+00:00",
|
|
"message": "Improve non-passing dotnet-test scenarios (#1114)",
|
|
"url": "https://github.com/dotnet/skills/commit/ac8f41264bdd557e58924a3110eea8e0917dcf4d",
|
|
"author": {
|
|
"name": "Amaury Levé",
|
|
"username": "AbhitejJohn"
|
|
}
|
|
},
|
|
"model": "claude-sonnet-5",
|
|
"date": 1788715663752,
|
|
"benches": [
|
|
{
|
|
"value": 320.5,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.44,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Time"
|
|
},
|
|
{
|
|
"value": 1786301.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.44,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Tokens In"
|
|
},
|
|
{
|
|
"value": 282.6,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.44,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Time"
|
|
},
|
|
{
|
|
"value": 1611896.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.44,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Tokens In"
|
|
},
|
|
{
|
|
"value": 228.7,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Time"
|
|
},
|
|
{
|
|
"value": 1804625.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"value": 330.2,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.44,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Time"
|
|
},
|
|
{
|
|
"value": 1758150.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.44,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Tokens In"
|
|
},
|
|
{
|
|
"value": 244.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.44,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Time"
|
|
},
|
|
{
|
|
"value": 1310756.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.44,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Tokens In"
|
|
},
|
|
{
|
|
"value": 88.0,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Time"
|
|
},
|
|
{
|
|
"value": 470565.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"value": 290.2,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.44,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Time"
|
|
},
|
|
{
|
|
"value": 2180647.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.44,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Tokens In"
|
|
},
|
|
{
|
|
"value": 219.1,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.44,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Time"
|
|
},
|
|
{
|
|
"value": 1552265.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.44,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Tokens In"
|
|
},
|
|
{
|
|
"value": 170.7,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Time"
|
|
},
|
|
{
|
|
"value": 930226.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"value": 16.3,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.44,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Time"
|
|
},
|
|
{
|
|
"value": 41105.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.44,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Tokens In"
|
|
},
|
|
{
|
|
"value": 14.2,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.44,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Time"
|
|
},
|
|
{
|
|
"value": 41033.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.44,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Tokens In"
|
|
},
|
|
{
|
|
"value": 19.5,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Time"
|
|
},
|
|
{
|
|
"value": 19387.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"value": 21.8,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.44,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Time"
|
|
},
|
|
{
|
|
"value": 41828.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.44,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Tokens In"
|
|
},
|
|
{
|
|
"value": 17.6,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.44,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Time"
|
|
},
|
|
{
|
|
"value": 41463.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.44,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Tokens In"
|
|
},
|
|
{
|
|
"value": 17.2,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Time"
|
|
},
|
|
{
|
|
"value": 37068.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"value": 208.6,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.44,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Time"
|
|
},
|
|
{
|
|
"value": 1336894.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.44,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Tokens In"
|
|
},
|
|
{
|
|
"value": 306.8,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.44,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Time"
|
|
},
|
|
{
|
|
"value": 2023186.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.44,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Tokens In"
|
|
},
|
|
{
|
|
"value": 167.3,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Time"
|
|
},
|
|
{
|
|
"value": 999251.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Tokens In"
|
|
}
|
|
],
|
|
"tool": "customSmallerIsBetter"
|
|
},
|
|
{
|
|
"commit": {
|
|
"id": "ac8f41264bdd557e58924a3110eea8e0917dcf4d",
|
|
"timestamp": "2026-09-04T10:21:25+00:00",
|
|
"message": "Improve non-passing dotnet-test scenarios (#1114)",
|
|
"url": "https://github.com/dotnet/skills/commit/ac8f41264bdd557e58924a3110eea8e0917dcf4d",
|
|
"author": {
|
|
"name": "Amaury Levé",
|
|
"username": "AbhitejJohn"
|
|
}
|
|
},
|
|
"model": "gpt-5.6-sol",
|
|
"date": 1788715663815,
|
|
"benches": [
|
|
{
|
|
"value": 160.8,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Time"
|
|
},
|
|
{
|
|
"value": 898613.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Tokens In"
|
|
},
|
|
{
|
|
"value": 330.1,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Time"
|
|
},
|
|
{
|
|
"value": 888811.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Tokens In"
|
|
},
|
|
{
|
|
"value": 185.0,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Time"
|
|
},
|
|
{
|
|
"value": 930406.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"value": 199.2,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Time"
|
|
},
|
|
{
|
|
"value": 1269755.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Tokens In"
|
|
},
|
|
{
|
|
"value": 252.3,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Time"
|
|
},
|
|
{
|
|
"value": 1669011.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Tokens In"
|
|
},
|
|
{
|
|
"value": 75.7,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Time"
|
|
},
|
|
{
|
|
"value": 220413.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"value": 214.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Time"
|
|
},
|
|
{
|
|
"value": 977047.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Tokens In"
|
|
},
|
|
{
|
|
"value": 213.3,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Time"
|
|
},
|
|
{
|
|
"value": 676273.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Tokens In"
|
|
},
|
|
{
|
|
"value": 124.9,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Time"
|
|
},
|
|
{
|
|
"value": 443993.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"value": 15.6,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Time"
|
|
},
|
|
{
|
|
"value": 27057.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Tokens In"
|
|
},
|
|
{
|
|
"value": 9.6,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Time"
|
|
},
|
|
{
|
|
"value": 26813.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Tokens In"
|
|
},
|
|
{
|
|
"value": 16.3,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Time"
|
|
},
|
|
{
|
|
"value": 62629.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"value": 14.8,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Time"
|
|
},
|
|
{
|
|
"value": 27096.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Tokens In"
|
|
},
|
|
{
|
|
"value": 16.9,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Time"
|
|
},
|
|
{
|
|
"value": 27315.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Tokens In"
|
|
},
|
|
{
|
|
"value": 13.6,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Time"
|
|
},
|
|
{
|
|
"value": 12725.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"value": 187.2,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Time"
|
|
},
|
|
{
|
|
"value": 725454.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Tokens In"
|
|
},
|
|
{
|
|
"value": 190.2,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Time"
|
|
},
|
|
{
|
|
"value": 1064339.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Tokens In"
|
|
},
|
|
{
|
|
"value": 115.3,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Time"
|
|
},
|
|
{
|
|
"value": 361053.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Tokens In"
|
|
}
|
|
],
|
|
"tool": "customSmallerIsBetter"
|
|
},
|
|
{
|
|
"commit": {
|
|
"message": "Improve non-passing dotnet-test scenarios (#1114)",
|
|
"id": "ac8f41264bdd557e58924a3110eea8e0917dcf4d",
|
|
"timestamp": "2026-09-04T10:21:25+00:00",
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Amaury Levé"
|
|
},
|
|
"url": "https://github.com/dotnet/skills/commit/ac8f41264bdd557e58924a3110eea8e0917dcf4d"
|
|
},
|
|
"tool": "customSmallerIsBetter",
|
|
"benches": [
|
|
{
|
|
"value": 296.1,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Time",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47
|
|
},
|
|
{
|
|
"value": 1191112.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Tokens In",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47
|
|
},
|
|
{
|
|
"value": 268.8,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47
|
|
},
|
|
{
|
|
"value": 1468179.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47
|
|
},
|
|
{
|
|
"value": 194.5,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Time"
|
|
},
|
|
{
|
|
"value": 646851.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"value": 131.9,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Time",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47
|
|
},
|
|
{
|
|
"value": 638800.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Tokens In",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47
|
|
},
|
|
{
|
|
"value": 133.8,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47
|
|
},
|
|
{
|
|
"value": 656098.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47
|
|
},
|
|
{
|
|
"value": 123.9,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Time"
|
|
},
|
|
{
|
|
"value": 502151.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"value": 184.5,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Time",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47
|
|
},
|
|
{
|
|
"value": 534850.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Tokens In",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47
|
|
},
|
|
{
|
|
"value": 205.0,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Time"
|
|
},
|
|
{
|
|
"value": 208064.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"value": 24.4,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Time",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47
|
|
},
|
|
{
|
|
"value": 30464.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Tokens In",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47
|
|
},
|
|
{
|
|
"value": 24.0,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47
|
|
},
|
|
{
|
|
"value": 30307.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47
|
|
},
|
|
{
|
|
"value": 15.2,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Time"
|
|
},
|
|
{
|
|
"value": 26908.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"value": 21.9,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Time",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47
|
|
},
|
|
{
|
|
"value": 30298.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Tokens In",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47
|
|
},
|
|
{
|
|
"value": 24.4,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47
|
|
},
|
|
{
|
|
"value": 30345.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47
|
|
},
|
|
{
|
|
"value": 25.9,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Time"
|
|
},
|
|
{
|
|
"value": 14341.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"value": 107.2,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Time",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47
|
|
},
|
|
{
|
|
"value": 412749.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Tokens In",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47
|
|
},
|
|
{
|
|
"value": 90.1,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47
|
|
},
|
|
{
|
|
"value": 375540.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47
|
|
},
|
|
{
|
|
"value": 98.7,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Time"
|
|
},
|
|
{
|
|
"value": 432459.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Tokens In"
|
|
}
|
|
],
|
|
"date": 1788790038987,
|
|
"model": "claude-sonnet-4.6"
|
|
},
|
|
{
|
|
"commit": {
|
|
"message": "Improve non-passing dotnet-test scenarios (#1114)",
|
|
"id": "ac8f41264bdd557e58924a3110eea8e0917dcf4d",
|
|
"timestamp": "2026-09-04T10:21:25+00:00",
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Amaury Levé"
|
|
},
|
|
"url": "https://github.com/dotnet/skills/commit/ac8f41264bdd557e58924a3110eea8e0917dcf4d"
|
|
},
|
|
"tool": "customSmallerIsBetter",
|
|
"benches": [
|
|
{
|
|
"value": 201.1,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Time",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"value": 620244.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Tokens In",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"value": 162.2,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"value": 987114.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"value": 103.0,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Time"
|
|
},
|
|
{
|
|
"value": 435285.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"value": 112.3,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Time",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"value": 432745.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Tokens In",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"value": 123.6,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"value": 734943.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"value": 72.1,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Time"
|
|
},
|
|
{
|
|
"value": 335341.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"value": 93.2,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Time",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"value": 377211.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Tokens In",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"value": 117.2,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"value": 553461.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"value": 146.8,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Time"
|
|
},
|
|
{
|
|
"value": 423449.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"value": 9.2,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Time",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"value": 26803.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Tokens In",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"value": 10.3,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"value": 26842.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"value": 8.6,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Time"
|
|
},
|
|
{
|
|
"value": 12304.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"value": 13.6,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Time",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"value": 27150.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Tokens In",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"value": 15.4,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"value": 27284.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"value": 13.9,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Time"
|
|
},
|
|
{
|
|
"value": 12859.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"value": 298.6,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Time"
|
|
},
|
|
{
|
|
"value": 1560635.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Tokens In"
|
|
},
|
|
{
|
|
"value": 167.5,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Time"
|
|
},
|
|
{
|
|
"value": 705439.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Tokens In"
|
|
},
|
|
{
|
|
"value": 145.0,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Time"
|
|
},
|
|
{
|
|
"value": 617010.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Tokens In"
|
|
}
|
|
],
|
|
"date": 1788790039073,
|
|
"model": "gpt-5.6-luna"
|
|
},
|
|
{
|
|
"model": "claude-haiku-4.5",
|
|
"tool": "customSmallerIsBetter",
|
|
"date": 1788885915193,
|
|
"commit": {
|
|
"id": "fbeeafe261b0fe704b9a95c58fd9fdbe34e96964",
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Viktor Hofer"
|
|
},
|
|
"message": "Delete msbuild-server skill (#1123)",
|
|
"timestamp": "2026-09-07T09:11:54+00:00",
|
|
"url": "https://github.com/dotnet/skills/commit/fbeeafe261b0fe704b9a95c58fd9fdbe34e96964"
|
|
},
|
|
"benches": [
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Time",
|
|
"value": 329.2
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Tokens In",
|
|
"value": 995915.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Time",
|
|
"value": 139.8
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Tokens In",
|
|
"value": 859650.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Time",
|
|
"value": 60.7
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Tokens In",
|
|
"value": 316936.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Time",
|
|
"value": 144.7
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Tokens In",
|
|
"value": 737887.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Time",
|
|
"value": 253.2
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Tokens In",
|
|
"value": 1020800.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Time",
|
|
"value": 346.2
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Tokens In",
|
|
"value": 1383047.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Time",
|
|
"value": 325.6
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Tokens In",
|
|
"value": 850142.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Time",
|
|
"value": 36.6
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Tokens In",
|
|
"value": 32495.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Time",
|
|
"value": 28.9
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Tokens In",
|
|
"value": 31589.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Time",
|
|
"value": 18.6
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Tokens In",
|
|
"value": 14601.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Time",
|
|
"value": 30.1
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Tokens In",
|
|
"value": 31800.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Time",
|
|
"value": 28.2
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Tokens In",
|
|
"value": 31548.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Time",
|
|
"value": 28.1
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Tokens In",
|
|
"value": 87186.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Time",
|
|
"value": 70.3
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Tokens In",
|
|
"value": 268405.0
|
|
}
|
|
]
|
|
},
|
|
{
|
|
"model": "gpt-5.3-codex",
|
|
"tool": "customSmallerIsBetter",
|
|
"date": 1788885915259,
|
|
"commit": {
|
|
"id": "fbeeafe261b0fe704b9a95c58fd9fdbe34e96964",
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Viktor Hofer"
|
|
},
|
|
"message": "Delete msbuild-server skill (#1123)",
|
|
"timestamp": "2026-09-07T09:11:54+00:00",
|
|
"url": "https://github.com/dotnet/skills/commit/fbeeafe261b0fe704b9a95c58fd9fdbe34e96964"
|
|
},
|
|
"benches": [
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Time",
|
|
"overfitting": "moderate",
|
|
"notActivated": true,
|
|
"value": 9.7,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.24
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Tokens In",
|
|
"overfitting": "moderate",
|
|
"notActivated": true,
|
|
"value": 12898.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.24
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.24,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Time",
|
|
"value": 7.6
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.24,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Tokens In",
|
|
"value": 12591.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Time",
|
|
"value": 15.6
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Tokens In",
|
|
"value": 13439.0
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Time",
|
|
"overfitting": "moderate",
|
|
"notActivated": true,
|
|
"value": 1.7,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.24
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Tokens In",
|
|
"overfitting": "moderate",
|
|
"notActivated": true,
|
|
"value": 11972.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.24
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.24,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Time",
|
|
"value": 9.5
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.24,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Tokens In",
|
|
"value": 82593.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Time",
|
|
"value": 3.4
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Tokens In",
|
|
"value": 11741.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.24,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Time",
|
|
"value": 4.0
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.24,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Tokens In",
|
|
"value": 26164.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.24,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Time",
|
|
"value": 4.0
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.24,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Tokens In",
|
|
"value": 26180.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Time",
|
|
"value": 12.3
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Tokens In",
|
|
"value": 61142.0
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Time",
|
|
"overfitting": "moderate",
|
|
"notActivated": true,
|
|
"value": 7.1,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.24
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Tokens In",
|
|
"overfitting": "moderate",
|
|
"notActivated": true,
|
|
"value": 12556.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.24
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.24,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Time",
|
|
"value": 8.3
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.24,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Tokens In",
|
|
"value": 12573.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Time",
|
|
"value": 8.3
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Tokens In",
|
|
"value": 12308.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.24,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Time",
|
|
"value": 7.5
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.24,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Tokens In",
|
|
"value": 26463.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.24,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Time",
|
|
"value": 8.2
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.24,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Tokens In",
|
|
"value": 26546.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Time",
|
|
"value": 12.7
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Tokens In",
|
|
"value": 12739.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Time",
|
|
"value": 3.6
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Tokens In",
|
|
"value": 26160.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Time",
|
|
"value": 6.8
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Tokens In",
|
|
"value": 26153.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Time",
|
|
"value": 2.1
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Tokens In",
|
|
"value": 11807.0
|
|
}
|
|
]
|
|
},
|
|
{
|
|
"model": "mai-code-1-flash-picker",
|
|
"tool": "customSmallerIsBetter",
|
|
"date": 1788885915332,
|
|
"commit": {
|
|
"id": "fbeeafe261b0fe704b9a95c58fd9fdbe34e96964",
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Viktor Hofer"
|
|
},
|
|
"message": "Delete msbuild-server skill (#1123)",
|
|
"timestamp": "2026-09-07T09:11:54+00:00",
|
|
"url": "https://github.com/dotnet/skills/commit/fbeeafe261b0fe704b9a95c58fd9fdbe34e96964"
|
|
},
|
|
"benches": [
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Time",
|
|
"overfitting": "moderate",
|
|
"notActivated": true,
|
|
"value": 216.0,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.46
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Tokens In",
|
|
"overfitting": "moderate",
|
|
"notActivated": true,
|
|
"value": 839597.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.46
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Time",
|
|
"value": 283.7
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Tokens In",
|
|
"value": 1306192.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Time",
|
|
"value": 218.2
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Tokens In",
|
|
"value": 625443.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Time",
|
|
"value": 102.7
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Tokens In",
|
|
"value": 428736.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Time",
|
|
"value": 82.6
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Tokens In",
|
|
"value": 272476.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Time",
|
|
"value": 81.1
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Tokens In",
|
|
"value": 268200.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Time",
|
|
"value": 314.0
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Tokens In",
|
|
"value": 1603181.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Time",
|
|
"value": 169.9
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Tokens In",
|
|
"value": 340941.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Time",
|
|
"value": 159.5
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Tokens In",
|
|
"value": 564683.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Time",
|
|
"value": 17.3
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Tokens In",
|
|
"value": 25673.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Time",
|
|
"value": 7.8
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Tokens In",
|
|
"value": 11434.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Time",
|
|
"value": 34.5
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Tokens In",
|
|
"value": 79269.0
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Time",
|
|
"overfitting": "moderate",
|
|
"notActivated": true,
|
|
"value": 64.1,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.46
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Tokens In",
|
|
"overfitting": "moderate",
|
|
"notActivated": true,
|
|
"value": 93970.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.46
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Time",
|
|
"value": 10.2
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Tokens In",
|
|
"value": 11925.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Time",
|
|
"value": 15.4
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Tokens In",
|
|
"value": 11863.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Time",
|
|
"value": 295.8
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Tokens In",
|
|
"value": 1118902.0
|
|
}
|
|
]
|
|
},
|
|
{
|
|
"commit": {
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Abhitej John"
|
|
},
|
|
"message": "Merge pull request #1149 from dotnet/abhitejjohn-vally-014-ci-proof",
|
|
"url": "https://github.com/dotnet/skills/commit/a8fece0fce5f8b0737754a332a1a7c6487e8f927",
|
|
"timestamp": "2026-09-09T17:53:08+00:00",
|
|
"id": "a8fece0fce5f8b0737754a332a1a7c6487e8f927"
|
|
},
|
|
"model": "claude-opus-4.8",
|
|
"benches": [
|
|
{
|
|
"overfitting": "moderate",
|
|
"value": 216.1,
|
|
"overfittingScore": 0.48,
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Time",
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"value": 813072.0,
|
|
"overfittingScore": 0.48,
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Tokens In",
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"value": 352.9,
|
|
"overfittingScore": 0.48,
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Time",
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"value": 1496628.0,
|
|
"overfittingScore": 0.48,
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Tokens In",
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"value": 236.2,
|
|
"overfittingScore": 0.48,
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Time",
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"value": 1192543.0,
|
|
"overfittingScore": 0.48,
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Tokens In",
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"value": 113.9,
|
|
"overfittingScore": 0.48,
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Time",
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"value": 624309.0,
|
|
"overfittingScore": 0.48,
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Tokens In",
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"value": 357.1,
|
|
"overfittingScore": 0.48,
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Time",
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"value": 1453621.0,
|
|
"overfittingScore": 0.48,
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Tokens In",
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"value": 31.3,
|
|
"overfittingScore": 0.48,
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Time",
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"value": 48068.0,
|
|
"overfittingScore": 0.48,
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Tokens In",
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"value": 26.1,
|
|
"overfittingScore": 0.48,
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Time",
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"value": 47544.0,
|
|
"overfittingScore": 0.48,
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Tokens In",
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"value": 21.1,
|
|
"overfittingScore": 0.48,
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Time",
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"value": 47316.0,
|
|
"overfittingScore": 0.48,
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Tokens In",
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"value": 25.3,
|
|
"overfittingScore": 0.48,
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Time",
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"value": 47686.0,
|
|
"overfittingScore": 0.48,
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Tokens In",
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"value": 233.3,
|
|
"overfittingScore": 0.48,
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Time",
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"value": 927724.0,
|
|
"overfittingScore": 0.48,
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Tokens In",
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"value": 221.1,
|
|
"overfittingScore": 0.48,
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Time",
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"value": 956622.0,
|
|
"overfittingScore": 0.48,
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Tokens In",
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"value": 262.3,
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Time",
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"value": 946114.0,
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Tokens In",
|
|
"unit": "tokens"
|
|
}
|
|
],
|
|
"date": 1789036360735,
|
|
"tool": "customSmallerIsBetter"
|
|
},
|
|
{
|
|
"tool": "customSmallerIsBetter",
|
|
"model": "gpt-5.6-luna",
|
|
"benches": [
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.28,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Time",
|
|
"value": 180.0
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.28,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Tokens In",
|
|
"value": 1008222.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.28,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Time",
|
|
"value": 108.8
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.28,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Tokens In",
|
|
"value": 624820.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Time",
|
|
"value": 150.9
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Tokens In",
|
|
"value": 698614.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.28,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Time",
|
|
"value": 129.8
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.28,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Tokens In",
|
|
"value": 505842.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.28,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Time",
|
|
"value": 153.5
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.28,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Tokens In",
|
|
"value": 699825.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Time",
|
|
"value": 77.8
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Tokens In",
|
|
"value": 335415.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.28,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Time",
|
|
"value": 105.7
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.28,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Tokens In",
|
|
"value": 471240.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.28,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Time",
|
|
"value": 155.8
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.28,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Tokens In",
|
|
"value": 431007.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Time",
|
|
"value": 101.7
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Tokens In",
|
|
"value": 302897.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.28,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Time",
|
|
"value": 12.3
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.28,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Tokens In",
|
|
"value": 26937.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.28,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Time",
|
|
"value": 10.3
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.28,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Tokens In",
|
|
"value": 26676.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Time",
|
|
"value": 8.8
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Tokens In",
|
|
"value": 12168.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.28,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Time",
|
|
"value": 15.1
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.28,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Tokens In",
|
|
"value": 27310.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.28,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Time",
|
|
"value": 16.7
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.28,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Tokens In",
|
|
"value": 27207.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Time",
|
|
"value": 14.1
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Tokens In",
|
|
"value": 12725.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Time",
|
|
"value": 251.6
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Tokens In",
|
|
"value": 1294540.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Time",
|
|
"value": 162.7
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Tokens In",
|
|
"value": 635611.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Time",
|
|
"value": 122.5
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Tokens In",
|
|
"value": 427083.0
|
|
}
|
|
],
|
|
"commit": {
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Abhitej John"
|
|
},
|
|
"id": "8a5a42d3e392b402768fc29416831643b79e402b",
|
|
"message": "Merge pull request #1153 from dotnet/abhitejjohn-vally-lock-concurrency-proof",
|
|
"timestamp": "2026-09-10T21:01:20+00:00",
|
|
"url": "https://github.com/dotnet/skills/commit/8a5a42d3e392b402768fc29416831643b79e402b"
|
|
},
|
|
"date": 1789129571363
|
|
},
|
|
{
|
|
"benches": [
|
|
{
|
|
"value": 246.6,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Time"
|
|
},
|
|
{
|
|
"value": 1237749.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"overfittingScore": 0.5,
|
|
"overfitting": "high",
|
|
"value": 230.7,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Time"
|
|
},
|
|
{
|
|
"overfittingScore": 0.5,
|
|
"overfitting": "high",
|
|
"value": 1618115.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Tokens In"
|
|
},
|
|
{
|
|
"overfittingScore": 0.5,
|
|
"overfitting": "high",
|
|
"value": 209.9,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Time"
|
|
},
|
|
{
|
|
"overfittingScore": 0.5,
|
|
"overfitting": "high",
|
|
"value": 1573939.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Tokens In"
|
|
},
|
|
{
|
|
"value": 111.7,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Time"
|
|
},
|
|
{
|
|
"value": 399191.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"overfittingScore": 0.5,
|
|
"overfitting": "high",
|
|
"value": 343.3,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Time"
|
|
},
|
|
{
|
|
"overfittingScore": 0.5,
|
|
"overfitting": "high",
|
|
"value": 1493703.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Tokens In"
|
|
},
|
|
{
|
|
"value": 281.0,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Time"
|
|
},
|
|
{
|
|
"value": 1165603.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"overfittingScore": 0.5,
|
|
"overfitting": "high",
|
|
"value": 31.9,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Time"
|
|
},
|
|
{
|
|
"overfittingScore": 0.5,
|
|
"overfitting": "high",
|
|
"value": 31658.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Tokens In"
|
|
},
|
|
{
|
|
"overfittingScore": 0.5,
|
|
"overfitting": "high",
|
|
"value": 28.8,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Time"
|
|
},
|
|
{
|
|
"overfittingScore": 0.5,
|
|
"overfitting": "high",
|
|
"value": 31373.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Tokens In"
|
|
},
|
|
{
|
|
"value": 16.7,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Time"
|
|
},
|
|
{
|
|
"value": 14323.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"overfittingScore": 0.5,
|
|
"overfitting": "high",
|
|
"value": 35.9,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Time"
|
|
},
|
|
{
|
|
"overfittingScore": 0.5,
|
|
"overfitting": "high",
|
|
"value": 32209.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Tokens In"
|
|
},
|
|
{
|
|
"overfittingScore": 0.5,
|
|
"overfitting": "high",
|
|
"value": 35.1,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Time"
|
|
},
|
|
{
|
|
"overfittingScore": 0.5,
|
|
"overfitting": "high",
|
|
"value": 32036.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Tokens In"
|
|
},
|
|
{
|
|
"value": 30.5,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Time"
|
|
},
|
|
{
|
|
"value": 15770.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"overfittingScore": 0.5,
|
|
"overfitting": "high",
|
|
"value": 332.6,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Time"
|
|
},
|
|
{
|
|
"overfittingScore": 0.5,
|
|
"overfitting": "high",
|
|
"value": 1329510.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Tokens In"
|
|
},
|
|
{
|
|
"value": 293.6,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Time"
|
|
},
|
|
{
|
|
"value": 936263.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Tokens In"
|
|
}
|
|
],
|
|
"date": 1789218680600,
|
|
"model": "claude-haiku-4.5",
|
|
"tool": "customSmallerIsBetter",
|
|
"commit": {
|
|
"timestamp": "2026-09-12T00:01:18+00:00",
|
|
"id": "4c72b17fa2c2aaa307d8f1e337ec75cab45ac5c7",
|
|
"message": "Merge pull request #1154 from dotnet/abhitejjohn-agentic-workflow-repair",
|
|
"url": "https://github.com/dotnet/skills/commit/4c72b17fa2c2aaa307d8f1e337ec75cab45ac5c7",
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Abhitej John"
|
|
}
|
|
}
|
|
},
|
|
{
|
|
"benches": [
|
|
{
|
|
"overfitting": "moderate",
|
|
"notActivated": true,
|
|
"value": 2.8,
|
|
"overfittingScore": 0.27,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Time"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"notActivated": true,
|
|
"value": 11919.0,
|
|
"overfittingScore": 0.27,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Tokens In"
|
|
},
|
|
{
|
|
"overfittingScore": 0.27,
|
|
"overfitting": "moderate",
|
|
"value": 3.2,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Time"
|
|
},
|
|
{
|
|
"overfittingScore": 0.27,
|
|
"overfitting": "moderate",
|
|
"value": 11942.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Tokens In"
|
|
},
|
|
{
|
|
"value": 3.4,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Time"
|
|
},
|
|
{
|
|
"value": 11693.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"notActivated": true,
|
|
"value": 5.6,
|
|
"overfittingScore": 0.27,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Time"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"notActivated": true,
|
|
"value": 49034.0,
|
|
"overfittingScore": 0.27,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Tokens In"
|
|
},
|
|
{
|
|
"overfittingScore": 0.27,
|
|
"overfitting": "moderate",
|
|
"value": 2.1,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Time"
|
|
},
|
|
{
|
|
"overfittingScore": 0.27,
|
|
"overfitting": "moderate",
|
|
"value": 11826.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Tokens In"
|
|
},
|
|
{
|
|
"value": 2.7,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Time"
|
|
},
|
|
{
|
|
"value": 11600.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"overfittingScore": 0.27,
|
|
"overfitting": "moderate",
|
|
"value": 11.2,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Time"
|
|
},
|
|
{
|
|
"overfittingScore": 0.27,
|
|
"overfitting": "moderate",
|
|
"value": 27040.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Tokens In"
|
|
},
|
|
{
|
|
"overfittingScore": 0.27,
|
|
"overfitting": "moderate",
|
|
"value": 4.1,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Time"
|
|
},
|
|
{
|
|
"overfittingScore": 0.27,
|
|
"overfitting": "moderate",
|
|
"value": 25937.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Tokens In"
|
|
},
|
|
{
|
|
"value": 40.7,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Time"
|
|
},
|
|
{
|
|
"value": 111308.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"notActivated": true,
|
|
"value": 6.6,
|
|
"overfittingScore": 0.27,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Time"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"notActivated": true,
|
|
"value": 12452.0,
|
|
"overfittingScore": 0.27,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Tokens In"
|
|
},
|
|
{
|
|
"overfittingScore": 0.27,
|
|
"overfitting": "moderate",
|
|
"value": 6.2,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Time"
|
|
},
|
|
{
|
|
"overfittingScore": 0.27,
|
|
"overfitting": "moderate",
|
|
"value": 12436.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Tokens In"
|
|
},
|
|
{
|
|
"value": 7.5,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Time"
|
|
},
|
|
{
|
|
"value": 12225.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"overfittingScore": 0.27,
|
|
"overfitting": "moderate",
|
|
"value": 7.5,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Time"
|
|
},
|
|
{
|
|
"overfittingScore": 0.27,
|
|
"overfitting": "moderate",
|
|
"value": 26454.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Tokens In"
|
|
},
|
|
{
|
|
"overfittingScore": 0.27,
|
|
"overfitting": "moderate",
|
|
"value": 8.2,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Time"
|
|
},
|
|
{
|
|
"overfittingScore": 0.27,
|
|
"overfitting": "moderate",
|
|
"value": 26407.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Tokens In"
|
|
},
|
|
{
|
|
"value": 13.1,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Time"
|
|
},
|
|
{
|
|
"value": 12868.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"value": 7.3,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Time"
|
|
},
|
|
{
|
|
"value": 26541.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Tokens In"
|
|
},
|
|
{
|
|
"value": 3.8,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Time"
|
|
},
|
|
{
|
|
"value": 26002.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Tokens In"
|
|
},
|
|
{
|
|
"value": 2.7,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Time"
|
|
},
|
|
{
|
|
"value": 11752.0,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Tokens In"
|
|
}
|
|
],
|
|
"date": 1789218680648,
|
|
"model": "gpt-5.3-codex",
|
|
"tool": "customSmallerIsBetter",
|
|
"commit": {
|
|
"timestamp": "2026-09-12T00:01:18+00:00",
|
|
"id": "4c72b17fa2c2aaa307d8f1e337ec75cab45ac5c7",
|
|
"message": "Merge pull request #1154 from dotnet/abhitejjohn-agentic-workflow-repair",
|
|
"url": "https://github.com/dotnet/skills/commit/4c72b17fa2c2aaa307d8f1e337ec75cab45ac5c7",
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Abhitej John"
|
|
}
|
|
}
|
|
},
|
|
{
|
|
"tool": "customSmallerIsBetter",
|
|
"model": "claude-opus-5",
|
|
"benches": [
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Time",
|
|
"overfitting": "moderate",
|
|
"value": 236.5,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 1197640.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"value": 183.2,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 897831.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Time",
|
|
"value": 328.1,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Tokens In",
|
|
"value": 1454748.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Time",
|
|
"overfitting": "moderate",
|
|
"value": 241.6,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 1059146.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"value": 176.3,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 893739.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Time",
|
|
"value": 112.2,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Tokens In",
|
|
"value": 427631.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Time",
|
|
"overfitting": "moderate",
|
|
"value": 200.2,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 699851.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"value": 286.9,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 1120550.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Time",
|
|
"value": 214.3,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Tokens In",
|
|
"value": 715191.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Time",
|
|
"overfitting": "moderate",
|
|
"value": 17.0,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 40627.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"value": 18.0,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 40735.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Time",
|
|
"value": 32.1,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Tokens In",
|
|
"value": 91615.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Time",
|
|
"overfitting": "moderate",
|
|
"value": 20.7,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 40957.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"value": 19.9,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 41009.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Time",
|
|
"value": 22.1,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Tokens In",
|
|
"value": 73489.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Time",
|
|
"overfitting": "moderate",
|
|
"value": 195.9,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 851465.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"value": 195.5,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 633129.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Time",
|
|
"value": 164.8,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Tokens In",
|
|
"value": 488885.0,
|
|
"unit": "tokens"
|
|
}
|
|
],
|
|
"date": 1789318508154,
|
|
"commit": {
|
|
"timestamp": "2026-09-12T07:43:41+00:00",
|
|
"url": "https://github.com/dotnet/skills/commit/4ecd7d9c76fa458807684771c2bfc7acf1e00ad3",
|
|
"message": "Merge pull request #1134 from dotnet/bot/weekly-version-sync",
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Abhitej John"
|
|
},
|
|
"id": "4ecd7d9c76fa458807684771c2bfc7acf1e00ad3"
|
|
}
|
|
},
|
|
{
|
|
"tool": "customSmallerIsBetter",
|
|
"model": "claude-sonnet-5",
|
|
"benches": [
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Time",
|
|
"overfitting": "moderate",
|
|
"value": 288.6,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 1823791.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"value": 225.6,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 1368201.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Time",
|
|
"value": 201.9,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Tokens In",
|
|
"value": 1086439.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Time",
|
|
"overfitting": "moderate",
|
|
"value": 138.2,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 788203.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"value": 141.3,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 808182.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Time",
|
|
"value": 82.2,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Tokens In",
|
|
"value": 400340.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Time",
|
|
"overfitting": "moderate",
|
|
"value": 142.1,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 1205155.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"value": 207.7,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 1710098.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Time",
|
|
"value": 210.9,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Tokens In",
|
|
"value": 1266344.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Time",
|
|
"overfitting": "moderate",
|
|
"value": 14.3,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 40762.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"value": 11.5,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 40547.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Time",
|
|
"value": 16.9,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Tokens In",
|
|
"value": 18991.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Time",
|
|
"overfitting": "moderate",
|
|
"value": 15.3,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 40809.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"value": 17.3,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 41013.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Time",
|
|
"value": 16.4,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Tokens In",
|
|
"value": 18976.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Time",
|
|
"overfitting": "moderate",
|
|
"value": 200.5,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 1318833.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"value": 184.1,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 1138211.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Time",
|
|
"value": 167.8,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Tokens In",
|
|
"value": 1087199.0,
|
|
"unit": "tokens"
|
|
}
|
|
],
|
|
"date": 1789318508226,
|
|
"commit": {
|
|
"timestamp": "2026-09-12T07:43:41+00:00",
|
|
"url": "https://github.com/dotnet/skills/commit/4ecd7d9c76fa458807684771c2bfc7acf1e00ad3",
|
|
"message": "Merge pull request #1134 from dotnet/bot/weekly-version-sync",
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Abhitej John"
|
|
},
|
|
"id": "4ecd7d9c76fa458807684771c2bfc7acf1e00ad3"
|
|
}
|
|
},
|
|
{
|
|
"tool": "customSmallerIsBetter",
|
|
"model": "gpt-5.6-sol",
|
|
"benches": [
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Time",
|
|
"overfitting": "moderate",
|
|
"value": 155.6,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.25
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 1134560.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.25
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"value": 176.6,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.25
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 791239.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.25
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Time",
|
|
"value": 111.1,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Tokens In",
|
|
"value": 628967.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Time",
|
|
"overfitting": "moderate",
|
|
"value": 209.2,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.25
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 1733964.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.25
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"value": 172.5,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.25
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 1137945.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.25
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Time",
|
|
"value": 67.2,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Tokens In",
|
|
"value": 214864.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Time",
|
|
"overfitting": "moderate",
|
|
"value": 200.3,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.25
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 524438.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.25
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"value": 159.3,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.25
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 533505.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.25
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Time",
|
|
"value": 213.0,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Tokens In",
|
|
"value": 647638.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Time",
|
|
"overfitting": "moderate",
|
|
"value": 9.9,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.25
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 26582.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.25
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"value": 14.1,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.25
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 26449.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.25
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Time",
|
|
"value": 8.5,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Tokens In",
|
|
"value": 12218.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Time",
|
|
"overfitting": "moderate",
|
|
"value": 15.8,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.25
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 26925.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.25
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"value": 11.7,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.25
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 26765.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.25
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Time",
|
|
"value": 28.8,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Tokens In",
|
|
"value": 62674.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Time",
|
|
"value": 209.8,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Tokens In",
|
|
"value": 635842.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Time",
|
|
"value": 158.4,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Tokens In",
|
|
"value": 562397.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Time",
|
|
"value": 139.6,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Tokens In",
|
|
"value": 673126.0,
|
|
"unit": "tokens"
|
|
}
|
|
],
|
|
"date": 1789318508294,
|
|
"commit": {
|
|
"timestamp": "2026-09-12T07:43:41+00:00",
|
|
"url": "https://github.com/dotnet/skills/commit/4ecd7d9c76fa458807684771c2bfc7acf1e00ad3",
|
|
"message": "Merge pull request #1134 from dotnet/bot/weekly-version-sync",
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Abhitej John"
|
|
},
|
|
"id": "4ecd7d9c76fa458807684771c2bfc7acf1e00ad3"
|
|
}
|
|
},
|
|
{
|
|
"date": 1789397101175,
|
|
"model": "claude-sonnet-5",
|
|
"commit": {
|
|
"url": "https://github.com/dotnet/skills/commit/4ecd7d9c76fa458807684771c2bfc7acf1e00ad3",
|
|
"timestamp": "2026-09-12T07:43:41+00:00",
|
|
"id": "4ecd7d9c76fa458807684771c2bfc7acf1e00ad3",
|
|
"message": "Merge pull request #1134 from dotnet/bot/weekly-version-sync",
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Abhitej John"
|
|
}
|
|
},
|
|
"benches": [
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.47,
|
|
"value": 282.7,
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Time",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.47,
|
|
"value": 1771803.0,
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Tokens In",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.47,
|
|
"value": 268.9,
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Time",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.47,
|
|
"value": 1673417.0,
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Tokens In",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"value": 146.2,
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Time"
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"value": 453065.0,
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.47,
|
|
"value": 209.5,
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Time",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.47,
|
|
"value": 1753160.0,
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Tokens In",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.47,
|
|
"value": 143.5,
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Time",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.47,
|
|
"value": 559335.0,
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Tokens In",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"value": 94.3,
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Time"
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"value": 462676.0,
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.47,
|
|
"value": 269.7,
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Time",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.47,
|
|
"value": 1972358.0,
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Tokens In",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.47,
|
|
"value": 179.5,
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Time",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.47,
|
|
"value": 1195301.0,
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Tokens In",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"value": 177.9,
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Time"
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"value": 950907.0,
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.47,
|
|
"value": 12.2,
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Time",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.47,
|
|
"value": 40578.0,
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Tokens In",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.47,
|
|
"value": 15.2,
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Time",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.47,
|
|
"value": 40775.0,
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Tokens In",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"value": 13.2,
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Time"
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"value": 18691.0,
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.47,
|
|
"value": 15.9,
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Time",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.47,
|
|
"value": 40908.0,
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Tokens In",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.47,
|
|
"value": 12.7,
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Time",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.47,
|
|
"value": 40641.0,
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Tokens In",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"value": 17.3,
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Time"
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"value": 19014.0,
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.47,
|
|
"value": 212.3,
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Time",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.47,
|
|
"value": 1164666.0,
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Tokens In",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.47,
|
|
"value": 225.2,
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Time",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.47,
|
|
"value": 1170665.0,
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Tokens In",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"value": 88.7,
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Time"
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"value": 518865.0,
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Tokens In"
|
|
}
|
|
],
|
|
"tool": "customSmallerIsBetter"
|
|
},
|
|
{
|
|
"date": 1789397101227,
|
|
"model": "gpt-5.6-luna",
|
|
"commit": {
|
|
"url": "https://github.com/dotnet/skills/commit/4ecd7d9c76fa458807684771c2bfc7acf1e00ad3",
|
|
"timestamp": "2026-09-12T07:43:41+00:00",
|
|
"id": "4ecd7d9c76fa458807684771c2bfc7acf1e00ad3",
|
|
"message": "Merge pull request #1134 from dotnet/bot/weekly-version-sync",
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Abhitej John"
|
|
}
|
|
},
|
|
"benches": [
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.28,
|
|
"value": 139.4,
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Time",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.28,
|
|
"value": 656346.0,
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Tokens In",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.28,
|
|
"value": 76.1,
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Time",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.28,
|
|
"value": 458730.0,
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Tokens In",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"value": 144.1,
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Time"
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"value": 447670.0,
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.28,
|
|
"value": 94.5,
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Time",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.28,
|
|
"value": 323031.0,
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Tokens In",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.28,
|
|
"value": 114.4,
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Time",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.28,
|
|
"value": 479728.0,
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Tokens In",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"value": 85.5,
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Time"
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"value": 269550.0,
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.28,
|
|
"value": 180.4,
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Time",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.28,
|
|
"value": 1156768.0,
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Tokens In",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.28,
|
|
"value": 142.0,
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Time",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.28,
|
|
"value": 509339.0,
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Tokens In",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"value": 254.1,
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Time"
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"value": 727574.0,
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.28,
|
|
"value": 11.4,
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Time",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.28,
|
|
"value": 26706.0,
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Tokens In",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.28,
|
|
"value": 12.4,
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Time",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.28,
|
|
"value": 26818.0,
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Tokens In",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"value": 10.1,
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Time"
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"value": 12178.0,
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.28,
|
|
"value": 20.3,
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Time",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.28,
|
|
"value": 27121.0,
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Tokens In",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.28,
|
|
"value": 13.2,
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Time",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.28,
|
|
"value": 27179.0,
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Tokens In",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"value": 10.5,
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Time"
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"value": 12519.0,
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Tokens In"
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.28,
|
|
"value": 171.8,
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Time",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.28,
|
|
"value": 860073.0,
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Tokens In",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.28,
|
|
"value": 172.3,
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Time",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.28,
|
|
"value": 958693.0,
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Tokens In",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"value": 186.6,
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Time"
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"value": 612822.0,
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Tokens In"
|
|
}
|
|
],
|
|
"tool": "customSmallerIsBetter"
|
|
},
|
|
{
|
|
"tool": "customSmallerIsBetter",
|
|
"model": "claude-haiku-4.5",
|
|
"date": 1789478285332,
|
|
"commit": {
|
|
"url": "https://github.com/dotnet/skills/commit/24f7cfbd42ad7bf52bcd67372816b982c38c64c6",
|
|
"author": {
|
|
"name": "Amaury Levé",
|
|
"username": "AbhitejJohn"
|
|
},
|
|
"timestamp": "2026-09-14T15:27:40+00:00",
|
|
"id": "24f7cfbd42ad7bf52bcd67372816b982c38c64c6",
|
|
"message": "Surface activation-only evaluation failures (#1163)"
|
|
},
|
|
"benches": [
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Time",
|
|
"notActivated": true,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48,
|
|
"unit": "seconds",
|
|
"value": 274.9
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Tokens In",
|
|
"notActivated": true,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48,
|
|
"unit": "tokens",
|
|
"value": 1007508.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Time",
|
|
"value": 331.8
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Tokens In",
|
|
"value": 1497526.0
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Time",
|
|
"notActivated": true,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48,
|
|
"unit": "seconds",
|
|
"value": 66.9
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Tokens In",
|
|
"notActivated": true,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48,
|
|
"unit": "tokens",
|
|
"value": 332029.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Time",
|
|
"value": 63.8
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Tokens In",
|
|
"value": 379272.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Time",
|
|
"value": 88.4
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Tokens In",
|
|
"value": 482826.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Time",
|
|
"value": 261.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Tokens In",
|
|
"value": 1020018.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Time",
|
|
"value": 291.1
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Tokens In",
|
|
"value": 1157414.0
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Time",
|
|
"notActivated": true,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48,
|
|
"unit": "seconds",
|
|
"value": 9.8
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Tokens In",
|
|
"notActivated": true,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48,
|
|
"unit": "tokens",
|
|
"value": 14012.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Time",
|
|
"value": 29.3
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Tokens In",
|
|
"value": 31556.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Time",
|
|
"value": 13.4
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Tokens In",
|
|
"value": 13972.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Time",
|
|
"value": 27.8
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Tokens In",
|
|
"value": 31187.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Time",
|
|
"value": 32.5
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Tokens In",
|
|
"value": 31626.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Time",
|
|
"value": 29.5
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Tokens In",
|
|
"value": 15484.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Time",
|
|
"value": 324.5
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Tokens In",
|
|
"value": 1057379.0
|
|
}
|
|
]
|
|
},
|
|
{
|
|
"tool": "customSmallerIsBetter",
|
|
"model": "gpt-5.3-codex",
|
|
"date": 1789478285409,
|
|
"commit": {
|
|
"url": "https://github.com/dotnet/skills/commit/24f7cfbd42ad7bf52bcd67372816b982c38c64c6",
|
|
"author": {
|
|
"name": "Amaury Levé",
|
|
"username": "AbhitejJohn"
|
|
},
|
|
"timestamp": "2026-09-14T15:27:40+00:00",
|
|
"id": "24f7cfbd42ad7bf52bcd67372816b982c38c64c6",
|
|
"message": "Surface activation-only evaluation failures (#1163)"
|
|
},
|
|
"benches": [
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Time",
|
|
"value": 10.3
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Tokens In",
|
|
"value": 26597.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Time",
|
|
"value": 11.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Tokens In",
|
|
"value": 26696.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Time",
|
|
"value": 10.4
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Tokens In",
|
|
"value": 12465.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Time",
|
|
"value": 9.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Tokens In",
|
|
"value": 81998.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Time",
|
|
"value": 1.5
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Tokens In",
|
|
"value": 11800.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Time",
|
|
"value": 3.4
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Tokens In",
|
|
"value": 11625.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Time",
|
|
"value": 12.7
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Tokens In",
|
|
"value": 27062.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Time",
|
|
"value": 12.6
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Tokens In",
|
|
"value": 27067.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Time",
|
|
"value": 3.8
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Tokens In",
|
|
"value": 11687.0
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Time",
|
|
"notActivated": true,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "seconds",
|
|
"value": 8.0
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Tokens In",
|
|
"notActivated": true,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "tokens",
|
|
"value": 12493.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Time",
|
|
"value": 7.2
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Tokens In",
|
|
"value": 12436.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Time",
|
|
"value": 9.2
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Tokens In",
|
|
"value": 12260.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Time",
|
|
"value": 9.9
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Tokens In",
|
|
"value": 26507.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Time",
|
|
"value": 9.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Tokens In",
|
|
"value": 12607.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Time",
|
|
"value": 14.4
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Tokens In",
|
|
"value": 12729.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Time",
|
|
"value": 6.8
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Tokens In",
|
|
"value": 26184.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Time",
|
|
"value": 8.6
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Tokens In",
|
|
"value": 26406.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Time",
|
|
"value": 2.6
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Tokens In",
|
|
"value": 11715.0
|
|
}
|
|
]
|
|
},
|
|
{
|
|
"date": 1789574378627,
|
|
"commit": {
|
|
"author": {
|
|
"name": "Abhitej John",
|
|
"username": "AbhitejJohn"
|
|
},
|
|
"timestamp": "2026-09-15T22:05:10+00:00",
|
|
"message": "Merge pull request #873 from dotnet/add-dotnet-refactoring-skills",
|
|
"url": "https://github.com/dotnet/skills/commit/26323a52990d0cbfc838117109b40e115aac891f",
|
|
"id": "26323a52990d0cbfc838117109b40e115aac891f"
|
|
},
|
|
"benches": [
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Time",
|
|
"value": 322.6
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Tokens In",
|
|
"value": 3148907.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Time",
|
|
"value": 295.8
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Tokens In",
|
|
"value": 1625042.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Time",
|
|
"value": 200.5
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Tokens In",
|
|
"value": 1394441.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Time",
|
|
"value": 207.4
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Tokens In",
|
|
"value": 1528701.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Time",
|
|
"value": 194.3
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Tokens In",
|
|
"value": 1388029.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Time",
|
|
"value": 81.4
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Tokens In",
|
|
"value": 437376.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Time",
|
|
"value": 139.8
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Tokens In",
|
|
"value": 928362.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Time",
|
|
"value": 198.7
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Tokens In",
|
|
"value": 1540348.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Time",
|
|
"value": 295.1
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Tokens In",
|
|
"value": 1763718.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Time",
|
|
"value": 13.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Tokens In",
|
|
"value": 40643.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Time",
|
|
"value": 13.8
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Tokens In",
|
|
"value": 40694.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Time",
|
|
"value": 14.8
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Tokens In",
|
|
"value": 18834.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Time",
|
|
"value": 13.7
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Tokens In",
|
|
"value": 40762.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Time",
|
|
"value": 17.3
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Tokens In",
|
|
"value": 41070.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Time",
|
|
"value": 20.6
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Tokens In",
|
|
"value": 19317.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Time",
|
|
"value": 231.6
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Tokens In",
|
|
"value": 1150106.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Time",
|
|
"value": 181.9
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Tokens In",
|
|
"value": 1132363.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Time",
|
|
"value": 100.9
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Tokens In",
|
|
"value": 632236.0
|
|
}
|
|
],
|
|
"model": "claude-sonnet-5",
|
|
"tool": "customSmallerIsBetter"
|
|
},
|
|
{
|
|
"date": 1789574378723,
|
|
"commit": {
|
|
"author": {
|
|
"name": "Abhitej John",
|
|
"username": "AbhitejJohn"
|
|
},
|
|
"timestamp": "2026-09-15T22:05:10+00:00",
|
|
"message": "Merge pull request #873 from dotnet/add-dotnet-refactoring-skills",
|
|
"url": "https://github.com/dotnet/skills/commit/26323a52990d0cbfc838117109b40e115aac891f",
|
|
"id": "26323a52990d0cbfc838117109b40e115aac891f"
|
|
},
|
|
"benches": [
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.24,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Time",
|
|
"value": 113.3
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.24,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Tokens In",
|
|
"value": 522778.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.24,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Time",
|
|
"value": 110.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.24,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Tokens In",
|
|
"value": 637289.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Time",
|
|
"value": 160.9
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Tokens In",
|
|
"value": 902108.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.24,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Time",
|
|
"value": 85.6
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.24,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Tokens In",
|
|
"value": 404761.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.24,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Time",
|
|
"value": 112.4
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.24,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Tokens In",
|
|
"value": 580188.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Time",
|
|
"value": 89.7
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Tokens In",
|
|
"value": 303894.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.24,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Time",
|
|
"value": 127.2
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.24,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Tokens In",
|
|
"value": 664530.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.24,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Time",
|
|
"value": 87.8
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.24,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Tokens In",
|
|
"value": 520656.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Time",
|
|
"value": 142.5
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Tokens In",
|
|
"value": 447716.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.24,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Time",
|
|
"value": 11.5
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.24,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Tokens In",
|
|
"value": 26786.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.24,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Time",
|
|
"value": 14.3
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.24,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Tokens In",
|
|
"value": 26664.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Time",
|
|
"value": 20.7
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Tokens In",
|
|
"value": 61807.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.24,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Time",
|
|
"value": 15.9
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.24,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Tokens In",
|
|
"value": 27237.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.24,
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Time",
|
|
"value": 16.2
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.24,
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Tokens In",
|
|
"value": 27027.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Time",
|
|
"value": 11.5
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Tokens In",
|
|
"value": 12501.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Time",
|
|
"value": 174.5
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Tokens In",
|
|
"value": 1174461.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Time",
|
|
"value": 291.0
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Tokens In",
|
|
"value": 1204315.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Time",
|
|
"value": 156.4
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Tokens In",
|
|
"value": 697603.0
|
|
}
|
|
],
|
|
"model": "gpt-5.6-luna",
|
|
"tool": "customSmallerIsBetter"
|
|
},
|
|
{
|
|
"benches": [
|
|
{
|
|
"overfittingScore": 0.45,
|
|
"overfitting": "moderate",
|
|
"value": 211.2,
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Time",
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"overfittingScore": 0.45,
|
|
"overfitting": "moderate",
|
|
"value": 1140439.0,
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Tokens In",
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"overfittingScore": 0.45,
|
|
"overfitting": "moderate",
|
|
"value": 187.5,
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Time",
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"overfittingScore": 0.45,
|
|
"overfitting": "moderate",
|
|
"value": 752031.0,
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Tokens In",
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"value": 186.6,
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Time",
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"value": 865867.0,
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Tokens In",
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"overfittingScore": 0.45,
|
|
"overfitting": "moderate",
|
|
"value": 121.9,
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Time",
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"overfittingScore": 0.45,
|
|
"overfitting": "moderate",
|
|
"value": 478666.0,
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Tokens In",
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"overfittingScore": 0.45,
|
|
"overfitting": "moderate",
|
|
"value": 179.0,
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Time",
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"overfittingScore": 0.45,
|
|
"overfitting": "moderate",
|
|
"value": 776652.0,
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Tokens In",
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"value": 126.4,
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Time",
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"value": 428921.0,
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Tokens In",
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"overfittingScore": 0.45,
|
|
"overfitting": "moderate",
|
|
"value": 237.8,
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Time",
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"overfittingScore": 0.45,
|
|
"overfitting": "moderate",
|
|
"value": 1175832.0,
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Tokens In",
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"overfittingScore": 0.45,
|
|
"overfitting": "moderate",
|
|
"value": 152.4,
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Time",
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"overfittingScore": 0.45,
|
|
"overfitting": "moderate",
|
|
"value": 758880.0,
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Tokens In",
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"value": 206.3,
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Time",
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"value": 665413.0,
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Tokens In",
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"overfittingScore": 0.45,
|
|
"overfitting": "moderate",
|
|
"value": 22.5,
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Time",
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"overfittingScore": 0.45,
|
|
"overfitting": "moderate",
|
|
"value": 40999.0,
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Tokens In",
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"overfittingScore": 0.45,
|
|
"overfitting": "moderate",
|
|
"value": 20.7,
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Time",
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"overfittingScore": 0.45,
|
|
"overfitting": "moderate",
|
|
"value": 40915.0,
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Tokens In",
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"value": 24.1,
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Time",
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"value": 73128.0,
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Tokens In",
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"overfittingScore": 0.45,
|
|
"overfitting": "moderate",
|
|
"value": 20.1,
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Time",
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"overfittingScore": 0.45,
|
|
"overfitting": "moderate",
|
|
"value": 40925.0,
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Tokens In",
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"overfittingScore": 0.45,
|
|
"overfitting": "moderate",
|
|
"value": 19.8,
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Time",
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"overfittingScore": 0.45,
|
|
"overfitting": "moderate",
|
|
"value": 40839.0,
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Tokens In",
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"value": 20.0,
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Time",
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"value": 18884.0,
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Tokens In",
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"overfittingScore": 0.45,
|
|
"overfitting": "moderate",
|
|
"value": 253.8,
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Time",
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"overfittingScore": 0.45,
|
|
"overfitting": "moderate",
|
|
"value": 1053517.0,
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Tokens In",
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"overfittingScore": 0.45,
|
|
"overfitting": "moderate",
|
|
"value": 190.8,
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Time",
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"overfittingScore": 0.45,
|
|
"overfitting": "moderate",
|
|
"value": 750379.0,
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Tokens In",
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"value": 115.4,
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Time",
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"value": 465995.0,
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Tokens In",
|
|
"unit": "tokens"
|
|
}
|
|
],
|
|
"commit": {
|
|
"message": "Merge pull request #1184 from dotnet/dependabot/npm_and_yarn/eng/evaluation-tools/microsoft/vally-cli-0.16.0",
|
|
"id": "8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3",
|
|
"timestamp": "2026-09-17T03:09:52+00:00",
|
|
"url": "https://github.com/dotnet/skills/commit/8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3",
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Abhitej John"
|
|
}
|
|
},
|
|
"date": 1789642372141,
|
|
"tool": "customSmallerIsBetter",
|
|
"model": "claude-opus-4.8"
|
|
},
|
|
{
|
|
"tool": "customSmallerIsBetter",
|
|
"model": "claude-sonnet-5",
|
|
"benches": [
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Time",
|
|
"value": 289.1,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Tokens In",
|
|
"value": 2414891.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Time",
|
|
"value": 123.8
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Tokens In",
|
|
"value": 367242.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Time",
|
|
"value": 151.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Tokens In",
|
|
"value": 796496.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Time",
|
|
"value": 230.6,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Tokens In",
|
|
"value": 1557727.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Time",
|
|
"value": 89.9
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Tokens In",
|
|
"value": 487462.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Time",
|
|
"value": 164.8,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Tokens In",
|
|
"value": 1103090.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Time",
|
|
"value": 276.2,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Tokens In",
|
|
"value": 1929143.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Time",
|
|
"value": 298.6
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Tokens In",
|
|
"value": 1941293.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Time",
|
|
"value": 15.5,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Tokens In",
|
|
"value": 40841.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Time",
|
|
"value": 13.6,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Tokens In",
|
|
"value": 40669.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Time",
|
|
"value": 15.2
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Tokens In",
|
|
"value": 18697.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Time",
|
|
"value": 18.2,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Tokens In",
|
|
"value": 41159.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Time",
|
|
"value": 17.8,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Tokens In",
|
|
"value": 41169.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Time",
|
|
"value": 19.9
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Tokens In",
|
|
"value": 19181.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Time",
|
|
"value": 204.9,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Tokens In",
|
|
"value": 1139819.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Time",
|
|
"value": 107.5,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Tokens In",
|
|
"value": 588746.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Time",
|
|
"value": 215.5
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Tokens In",
|
|
"value": 1392876.0
|
|
}
|
|
],
|
|
"commit": {
|
|
"message": "Merge pull request #1184 from dotnet/dependabot/npm_and_yarn/eng/evaluation-tools/microsoft/vally-cli-0.16.0",
|
|
"timestamp": "2026-09-17T03:09:52+00:00",
|
|
"url": "https://github.com/dotnet/skills/commit/8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3",
|
|
"author": {
|
|
"name": "Abhitej John",
|
|
"username": "AbhitejJohn"
|
|
},
|
|
"id": "8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3"
|
|
},
|
|
"date": 1789745909143
|
|
},
|
|
{
|
|
"tool": "customSmallerIsBetter",
|
|
"model": "gpt-5.6-luna",
|
|
"benches": [
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Time",
|
|
"value": 77.9,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Tokens In",
|
|
"value": 754146.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Time",
|
|
"value": 90.4,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Tokens In",
|
|
"value": 669966.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Time",
|
|
"value": 72.7
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Tokens In",
|
|
"value": 570209.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Time",
|
|
"value": 57.7,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Tokens In",
|
|
"value": 437048.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Time",
|
|
"value": 107.9,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Tokens In",
|
|
"value": 870474.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Time",
|
|
"value": 53.2
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Tokens In",
|
|
"value": 297415.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Time",
|
|
"value": 82.4,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Tokens In",
|
|
"value": 635001.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Time",
|
|
"value": 73.9,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Tokens In",
|
|
"value": 723325.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Time",
|
|
"value": 103.8
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Tokens In",
|
|
"value": 473377.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Time",
|
|
"value": 6.1,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Tokens In",
|
|
"value": 26594.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Time",
|
|
"value": 5.4,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Tokens In",
|
|
"value": 26434.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Time",
|
|
"value": 7.7
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Tokens In",
|
|
"value": 12357.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Time",
|
|
"value": 9.7,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Tokens In",
|
|
"value": 27399.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Time",
|
|
"value": 9.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Tokens In",
|
|
"value": 27152.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Time",
|
|
"value": 7.7
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Tokens In",
|
|
"value": 12594.0
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Time",
|
|
"value": 151.1,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Tokens In",
|
|
"value": 983739.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Time",
|
|
"value": 187.3,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Tokens In",
|
|
"value": 1035406.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"unit": "seconds",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Time",
|
|
"value": 59.7
|
|
},
|
|
{
|
|
"unit": "tokens",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Tokens In",
|
|
"value": 349026.0
|
|
}
|
|
],
|
|
"commit": {
|
|
"message": "Merge pull request #1184 from dotnet/dependabot/npm_and_yarn/eng/evaluation-tools/microsoft/vally-cli-0.16.0",
|
|
"timestamp": "2026-09-17T03:09:52+00:00",
|
|
"url": "https://github.com/dotnet/skills/commit/8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3",
|
|
"author": {
|
|
"name": "Abhitej John",
|
|
"username": "AbhitejJohn"
|
|
},
|
|
"id": "8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3"
|
|
},
|
|
"date": 1789745909242
|
|
},
|
|
{
|
|
"date": 1789836946087,
|
|
"benches": [
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Time",
|
|
"value": 331.6,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Tokens In",
|
|
"value": 1464135.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Time",
|
|
"overfitting": "moderate",
|
|
"value": 131.5,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 838673.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"value": 36.8,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 160964.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Time",
|
|
"value": 83.4,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Tokens In",
|
|
"value": 461020.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"notActivated": true,
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Time",
|
|
"value": 358.1,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.48,
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"notActivated": true,
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Tokens In",
|
|
"value": 1314352.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.48,
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"value": 232.2,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 882599.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Time",
|
|
"overfitting": "moderate",
|
|
"value": 30.6,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 31469.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"value": 33.3,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 31924.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Time",
|
|
"value": 16.9,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Tokens In",
|
|
"value": 14424.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Time",
|
|
"overfitting": "moderate",
|
|
"value": 28.9,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 31394.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"value": 30.4,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 31651.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Time",
|
|
"value": 20.4,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Tokens In",
|
|
"value": 14666.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Time",
|
|
"overfitting": "moderate",
|
|
"value": 280.5,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 916199.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Time",
|
|
"value": 218.7,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Tokens In",
|
|
"value": 1245112.0,
|
|
"unit": "tokens"
|
|
}
|
|
],
|
|
"commit": {
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Abhitej John"
|
|
},
|
|
"message": "Merge pull request #1184 from dotnet/dependabot/npm_and_yarn/eng/evaluation-tools/microsoft/vally-cli-0.16.0",
|
|
"url": "https://github.com/dotnet/skills/commit/8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3",
|
|
"timestamp": "2026-09-17T03:09:52+00:00",
|
|
"id": "8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3"
|
|
},
|
|
"tool": "customSmallerIsBetter",
|
|
"model": "claude-haiku-4.5"
|
|
},
|
|
{
|
|
"date": 1789836946152,
|
|
"benches": [
|
|
{
|
|
"notActivated": true,
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Time",
|
|
"value": 10.1,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.26,
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"notActivated": true,
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Tokens In",
|
|
"value": 13468.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.26,
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"value": 5.4,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 12492.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Time",
|
|
"value": 3.4,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Tokens In",
|
|
"value": 11667.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"notActivated": true,
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Time",
|
|
"value": 2.1,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.26,
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"notActivated": true,
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Tokens In",
|
|
"value": 11936.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.26,
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"value": 1.4,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 11845.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Time",
|
|
"value": 3.2,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Tokens In",
|
|
"value": 11637.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Time",
|
|
"overfitting": "moderate",
|
|
"value": 11.0,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 71012.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"value": 1.2,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 11854.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Time",
|
|
"value": 3.7,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Tokens In",
|
|
"value": 11726.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"notActivated": true,
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Time",
|
|
"value": 5.4,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.26,
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"notActivated": true,
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Tokens In",
|
|
"value": 12482.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.26,
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"value": 5.4,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 12460.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Time",
|
|
"value": 6.4,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Tokens In",
|
|
"value": 12168.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"notActivated": true,
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Time",
|
|
"value": 8.3,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.26,
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"notActivated": true,
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Tokens In",
|
|
"value": 12827.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.26,
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"value": 6.5,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 12573.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Time",
|
|
"value": 10.2,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Tokens In",
|
|
"value": 12719.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Time",
|
|
"value": 3.8,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Tokens In",
|
|
"value": 26022.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Time",
|
|
"value": 5.6,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Tokens In",
|
|
"value": 26424.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Time",
|
|
"value": 1.9,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Tokens In",
|
|
"value": 11709.0,
|
|
"unit": "tokens"
|
|
}
|
|
],
|
|
"commit": {
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Abhitej John"
|
|
},
|
|
"message": "Merge pull request #1184 from dotnet/dependabot/npm_and_yarn/eng/evaluation-tools/microsoft/vally-cli-0.16.0",
|
|
"url": "https://github.com/dotnet/skills/commit/8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3",
|
|
"timestamp": "2026-09-17T03:09:52+00:00",
|
|
"id": "8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3"
|
|
},
|
|
"tool": "customSmallerIsBetter",
|
|
"model": "gpt-5.3-codex"
|
|
},
|
|
{
|
|
"date": 1789836946226,
|
|
"benches": [
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Time",
|
|
"overfitting": "moderate",
|
|
"value": 110.7,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 1016723.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"value": 127.1,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 900753.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Time",
|
|
"value": 84.1,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Tokens In",
|
|
"value": 330906.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"notActivated": true,
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Time",
|
|
"value": 102.3,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.49,
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"notActivated": true,
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Tokens In",
|
|
"value": 627420.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.49,
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"value": 133.4,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 706238.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Time",
|
|
"value": 65.5,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Tokens In",
|
|
"value": 264852.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"notActivated": true,
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Time",
|
|
"value": 162.8,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.49,
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"notActivated": true,
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Tokens In",
|
|
"value": 1077304.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.49,
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"notActivated": true,
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Time",
|
|
"value": 6.3,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.49,
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"notActivated": true,
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Tokens In",
|
|
"value": 11415.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.49,
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"value": 4.9,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 11193.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Time",
|
|
"value": 9.8,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Tokens In",
|
|
"value": 11485.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"notActivated": true,
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Time",
|
|
"value": 13.1,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.49,
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"notActivated": true,
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Tokens In",
|
|
"value": 11561.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.49,
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"value": 16.9,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 83537.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Time",
|
|
"value": 11.0,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Tokens In",
|
|
"value": 11655.0,
|
|
"unit": "tokens"
|
|
},
|
|
{
|
|
"notActivated": true,
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Time",
|
|
"value": 170.2,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.49,
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"notActivated": true,
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Tokens In",
|
|
"value": 707500.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.49,
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Time",
|
|
"overfitting": "moderate",
|
|
"value": 146.2,
|
|
"unit": "seconds",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Tokens In",
|
|
"overfitting": "moderate",
|
|
"value": 982005.0,
|
|
"unit": "tokens",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Time",
|
|
"value": 65.2,
|
|
"unit": "seconds"
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Tokens In",
|
|
"value": 257754.0,
|
|
"unit": "tokens"
|
|
}
|
|
],
|
|
"commit": {
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Abhitej John"
|
|
},
|
|
"message": "Merge pull request #1184 from dotnet/dependabot/npm_and_yarn/eng/evaluation-tools/microsoft/vally-cli-0.16.0",
|
|
"url": "https://github.com/dotnet/skills/commit/8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3",
|
|
"timestamp": "2026-09-17T03:09:52+00:00",
|
|
"id": "8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3"
|
|
},
|
|
"tool": "customSmallerIsBetter",
|
|
"model": "mai-code-1.1-flash"
|
|
}
|
|
],
|
|
"Quality": [
|
|
{
|
|
"date": 1788627832039,
|
|
"verdictEvidence": [
|
|
{
|
|
"skillName": "technology-selection",
|
|
"skillKind": "invocable",
|
|
"state": "VALID_NO_CHANGE",
|
|
"stateReason": {
|
|
"code": "no_credible_preference_change",
|
|
"phase": "decision"
|
|
},
|
|
"passed": false,
|
|
"regressed": false,
|
|
"preferenceRegressed": false,
|
|
"reason": "Net win +66.7% (4W/2T/0L over 6 preference-eligible stimulus vote(s), sign test p=0.063), mean preference +26.7% across 6 paired run(s) — not credible — 2 of 6 preference-eligible stimulus vote(s) tied, leaving only 4 discordant preference vote(s). The sign test conditions on non-tie stimulus votes and cannot reach 0.05 below 5, so no record could have passed here — this is not a measured null. Either the skill is inert on these scenarios (make them discriminate) or the eval needs more distinct stimuli to clear the ties",
|
|
"gateEvidence": {
|
|
"stimulusVoteCount": 6,
|
|
"wins": 4,
|
|
"ties": 2,
|
|
"losses": 0,
|
|
"discordant": 4,
|
|
"direction": "better",
|
|
"pValue": 0.0625,
|
|
"alpha": 0.05,
|
|
"netWin": 0.6666666666666666,
|
|
"minimumNetWin": 0.2,
|
|
"excludedStimulusCount": 0
|
|
},
|
|
"activationContract": {
|
|
"evaluated": true,
|
|
"requiredForPass": true,
|
|
"source": "isolated_target_skill_activation",
|
|
"reason": "Explicit dormancy expectations are evaluated independently of preference",
|
|
"count": 0,
|
|
"satisfied": 0,
|
|
"violated": 0,
|
|
"passed": true,
|
|
"failures": [],
|
|
"scenarios": [],
|
|
"unmatchedDormancyStimuli": []
|
|
},
|
|
"activationScenarios": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "missing-activation",
|
|
"plugin": "plugin-no-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "missing-activation",
|
|
"plugin": "plugin-no-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "missing-activation",
|
|
"plugin": "plugin-no-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
}
|
|
],
|
|
"judgeRationales": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"direction": "better",
|
|
"rationale": "The core task was to actually build the app. Response A declined to build it (claiming it was blocked) and only provided a mock web-search stub with no real agent loop. Response B delivered a complete, functional implementation: a real DuckDuckGo web search tool, note-taking tools (including read_notes), a proper multi-iteration tool-calling loop with an iteration cap, and a structured final summary. Although both miss most rubric criteria (neither uses Microsoft.Agents.AI, observability, or token budgets), B is substantially more complete and usable, with a real search integration and iteration capping. B is much better overall. [Magnitude reduced to \"slightly-better\": the position-swapped pass judged this \"slightly-better\", so the more confident \"much-better\" was downgraded.]"
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"direction": "none",
|
|
"rationale": "Position-swap inconsistent (forward: tie, reverse: A). Defaulting to tie."
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"direction": "better",
|
|
"rationale": "Neither response produced final code (both correctly asked for missing project files/CSV). However B's plan aligns with nearly every rubric criterion explicitly (seed, TrainTestSplit, named metrics, PredictionEnginePool), while A's plan is generic and omits the reproducibility seed, train/test split specifics, named metrics, and the PredictionEnginePool best practice. B demonstrates substantially more correct technical intent. [Magnitude reduced to \"slightly-better\": the position-swapped pass judged this \"slightly-better\", so the more confident \"much-better\" was downgraded.]"
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"direction": "none",
|
|
"rationale": "Position-swap inconsistent (forward: B, reverse: A). Defaulting to tie."
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"direction": "better",
|
|
"rationale": "Both responses give solid, code-free RAG architectures with chunking, retrieval, grounding, and citations. However, the rubric specifically rewards Microsoft.Extensions.AI (IEmbeddingGenerator), Microsoft.Extensions.VectorData, an explicit similarity threshold, and an embedding cache. B hits all of these explicitly, while A misses the specific MS libraries and the explicit similarity threshold. A is more detailed and comprehensive in prose (phased rollout, acceptance criteria, richer data model), but on the criteria that matter here B is clearly more aligned. B wins on 4 criteria (two much-better) and ties on 2, making it substantially better against this rubric. [Magnitude reduced to \"slightly-better\": the position-swapped pass judged this \"slightly-better\", so the more confident \"much-better\" was downgraded.]"
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"direction": "better",
|
|
"rationale": "Both responses correctly chose ML.NET and neither delivered an actual implementation, which is the core failure. However, B was more methodical: it actually attempted to locate and read the project, discovering the real ChurnPredictor project and customers.csv via glob (whereas A did zero tool calls and simply claimed it lacked files that were in fact present). B's technology justification is also slightly stronger. Both fell short on the main implementation requirement, so the gap is small."
|
|
}
|
|
],
|
|
"links": [
|
|
{
|
|
"label": "Skill source",
|
|
"url": "https://github.com/dotnet/skills/blob/ac8f41264bdd557e58924a3110eea8e0917dcf4d/plugins/dotnet-ai/skills/technology-selection/SKILL.md"
|
|
},
|
|
{
|
|
"label": "Eval source",
|
|
"url": "https://github.com/dotnet/skills/blob/ac8f41264bdd557e58924a3110eea8e0917dcf4d/tests/dotnet-ai/technology-selection/eval.yaml"
|
|
},
|
|
{
|
|
"label": "Commit",
|
|
"url": "https://github.com/dotnet/skills/commit/ac8f41264bdd557e58924a3110eea8e0917dcf4d"
|
|
}
|
|
]
|
|
}
|
|
],
|
|
"tool": "customBiggerIsBetter",
|
|
"benches": [
|
|
{
|
|
"overfittingScore": 0.29,
|
|
"notActivated": true,
|
|
"unit": "Score (0-10)",
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Quality",
|
|
"value": 4.5
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Quality",
|
|
"value": 4.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.29
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Quality",
|
|
"value": 4.5,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"overfittingScore": 0.29,
|
|
"notActivated": true,
|
|
"unit": "Score (0-10)",
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Quality",
|
|
"value": 5.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Quality",
|
|
"value": 5.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.29
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Quality",
|
|
"value": 5.0,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Quality",
|
|
"value": 6.5,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.29
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Quality",
|
|
"value": 2.5,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.29
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Quality",
|
|
"value": 2.0,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"overfittingScore": 0.29,
|
|
"notActivated": true,
|
|
"unit": "Score (0-10)",
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Quality",
|
|
"value": 9.166666984558105
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Quality",
|
|
"value": 5.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.29
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Quality",
|
|
"value": 5.833333492279053,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Quality",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.29
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Quality",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.29
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Quality",
|
|
"value": 9.375,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Quality",
|
|
"value": 7.5,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.29
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Quality",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.29
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Quality",
|
|
"value": 7.5,
|
|
"unit": "Score (0-10)"
|
|
}
|
|
],
|
|
"commit": {
|
|
"author": {
|
|
"name": "Amaury Levé",
|
|
"username": "AbhitejJohn"
|
|
},
|
|
"message": "Improve non-passing dotnet-test scenarios (#1114)",
|
|
"id": "ac8f41264bdd557e58924a3110eea8e0917dcf4d",
|
|
"url": "https://github.com/dotnet/skills/commit/ac8f41264bdd557e58924a3110eea8e0917dcf4d",
|
|
"timestamp": "2026-09-04T10:21:25+00:00"
|
|
},
|
|
"model": "gpt-5.3-codex"
|
|
},
|
|
{
|
|
"date": 1788627832128,
|
|
"verdictEvidence": [
|
|
{
|
|
"skillName": "technology-selection",
|
|
"skillKind": "invocable",
|
|
"state": "VALID_NO_CHANGE",
|
|
"stateReason": {
|
|
"code": "no_credible_preference_change",
|
|
"phase": "decision"
|
|
},
|
|
"passed": false,
|
|
"regressed": false,
|
|
"preferenceRegressed": false,
|
|
"reason": "Net win +33.3% (3W/2T/1L over 6 preference-eligible stimulus vote(s), sign test p=0.312), mean preference +23.3% across 6 paired run(s) — not credible — 2 of 6 preference-eligible stimulus vote(s) tied, leaving only 4 discordant preference vote(s). The sign test conditions on non-tie stimulus votes and cannot reach 0.05 below 5, so no record could have passed here — this is not a measured null. Either the skill is inert on these scenarios (make them discriminate) or the eval needs more distinct stimuli to clear the ties",
|
|
"gateEvidence": {
|
|
"stimulusVoteCount": 6,
|
|
"wins": 3,
|
|
"ties": 2,
|
|
"losses": 1,
|
|
"discordant": 4,
|
|
"direction": "better",
|
|
"pValue": 0.31249999999999994,
|
|
"alpha": 0.05,
|
|
"netWin": 0.3333333333333333,
|
|
"minimumNetWin": 0.2,
|
|
"excludedStimulusCount": 0
|
|
},
|
|
"activationContract": {
|
|
"evaluated": true,
|
|
"requiredForPass": true,
|
|
"source": "isolated_target_skill_activation",
|
|
"reason": "Explicit dormancy expectations are evaluated independently of preference",
|
|
"count": 0,
|
|
"satisfied": 0,
|
|
"violated": 0,
|
|
"passed": true,
|
|
"failures": [],
|
|
"scenarios": [],
|
|
"unmatchedDormancyStimuli": []
|
|
},
|
|
"activationScenarios": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-no-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "missing-activation",
|
|
"plugin": "plugin-no-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-no-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "missing-activation",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-no-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-no-activity-observed"
|
|
}
|
|
],
|
|
"judgeRationales": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"direction": "better",
|
|
"rationale": "B is a successfully building .NET 10 implementation of an actual AI-agent architecture using the specifically requested Microsoft Agent Framework and Microsoft.Extensions.AI, with real tool integration. A builds and runs a useful deterministic web-search/note/summarization app, but does not meet the central framework and IChatClient requirements. B still misses the iteration cap, observability, and cost controls, and its full agent execution was not validated with an API key."
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"direction": "none",
|
|
"rationale": "Position-swap inconsistent (forward: tie, reverse: A). Defaulting to tie."
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"direction": "better",
|
|
"rationale": "B is methodologically and operationally stronger: deterministic training, held-out metrics, standard multiclass reporting, saved/loaded model support, and thread-safe pooled prediction. Its observed zero held-out accuracy is a serious warning that should have been investigated rather than presented as successful, so the overall advantage is not large; nevertheless A's unseeded, likely in-sample evaluation and singleton PredictionEngine are more fundamental production-quality omissions."
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"direction": "worse",
|
|
"rationale": "Both are sound high-level plans that honor the user's request to see a plan before implementation. A is modestly more complete and actionable, especially in its phased delivery, explicit backend services, re-indexing, citations, and operational hardening. Neither addresses the rubric's preferred Microsoft.Extensions.AI/VectorData abstractions, preventing a larger advantage."
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"direction": "better",
|
|
"rationale": "Both give valid code-free RAG plans with ingestion, retrieval, grounding, and citations. B better satisfies the requested .NET-specific technology selection and explicitly covers similarity thresholds, while A is somewhat more detailed on practical chunking and runtime flow."
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"direction": "none",
|
|
"rationale": "Position-swap inconsistent (forward: B, reverse: tie). Defaulting to tie."
|
|
}
|
|
],
|
|
"links": [
|
|
{
|
|
"label": "Skill source",
|
|
"url": "https://github.com/dotnet/skills/blob/ac8f41264bdd557e58924a3110eea8e0917dcf4d/plugins/dotnet-ai/skills/technology-selection/SKILL.md"
|
|
},
|
|
{
|
|
"label": "Eval source",
|
|
"url": "https://github.com/dotnet/skills/blob/ac8f41264bdd557e58924a3110eea8e0917dcf4d/tests/dotnet-ai/technology-selection/eval.yaml"
|
|
},
|
|
{
|
|
"label": "Commit",
|
|
"url": "https://github.com/dotnet/skills/commit/ac8f41264bdd557e58924a3110eea8e0917dcf4d"
|
|
}
|
|
]
|
|
}
|
|
],
|
|
"tool": "customBiggerIsBetter",
|
|
"benches": [
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Quality",
|
|
"value": 7.5,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Quality",
|
|
"value": 2.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Quality",
|
|
"value": 2.0,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"overfittingScore": 0.49,
|
|
"notActivated": true,
|
|
"unit": "Score (0-10)",
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Quality",
|
|
"value": 5.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Quality",
|
|
"value": 5.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Quality",
|
|
"value": 5.0,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Quality",
|
|
"value": 6.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Quality",
|
|
"value": 3.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Quality",
|
|
"value": 2.5,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"overfittingScore": 0.49,
|
|
"notActivated": true,
|
|
"unit": "Score (0-10)",
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Quality",
|
|
"value": 5.833333492279053
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Quality",
|
|
"value": 9.166666984558105,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Quality",
|
|
"value": 5.0,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Quality",
|
|
"value": 9.375,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Quality",
|
|
"value": 8.75,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Quality",
|
|
"value": 8.75,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Quality",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Quality",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)"
|
|
}
|
|
],
|
|
"commit": {
|
|
"author": {
|
|
"name": "Amaury Levé",
|
|
"username": "AbhitejJohn"
|
|
},
|
|
"message": "Improve non-passing dotnet-test scenarios (#1114)",
|
|
"id": "ac8f41264bdd557e58924a3110eea8e0917dcf4d",
|
|
"url": "https://github.com/dotnet/skills/commit/ac8f41264bdd557e58924a3110eea8e0917dcf4d",
|
|
"timestamp": "2026-09-04T10:21:25+00:00"
|
|
},
|
|
"model": "mai-code-1-flash-picker"
|
|
},
|
|
{
|
|
"verdictEvidence": [
|
|
{
|
|
"skillName": "technology-selection",
|
|
"skillKind": "invocable",
|
|
"state": "INVALID_INCONCLUSIVE",
|
|
"stateReason": {
|
|
"code": "unmatched_trajectories",
|
|
"phase": "comparison_pairing"
|
|
},
|
|
"passed": false,
|
|
"regressed": false,
|
|
"preferenceRegressed": false,
|
|
"reason": "Net win +75.0% (3W/1T/0L over 4 preference-eligible stimulus vote(s), sign test p=0.125), mean preference +45.0% across 4 paired run(s), 2 unmatched — inconclusive (unmatched trajectories)",
|
|
"gateEvidence": {
|
|
"stimulusVoteCount": 4,
|
|
"wins": 3,
|
|
"ties": 1,
|
|
"losses": 0,
|
|
"discordant": 3,
|
|
"direction": "better",
|
|
"pValue": 0.12500000000000003,
|
|
"alpha": 0.05,
|
|
"netWin": 0.75,
|
|
"minimumNetWin": 0.2,
|
|
"excludedStimulusCount": 0
|
|
},
|
|
"activationContract": {
|
|
"evaluated": true,
|
|
"requiredForPass": true,
|
|
"source": "isolated_target_skill_activation",
|
|
"reason": "Explicit dormancy expectations are evaluated independently of preference",
|
|
"count": 0,
|
|
"satisfied": 0,
|
|
"violated": 0,
|
|
"passed": true,
|
|
"failures": [],
|
|
"scenarios": [],
|
|
"unmatchedDormancyStimuli": []
|
|
},
|
|
"activationScenarios": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
}
|
|
],
|
|
"judgeRationales": [
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"direction": "better",
|
|
"rationale": "B directly implements the rubric's expected production LLM integration, DI, fixed generation settings, resilience, credential-safe setup, and model pinning, while still validating and testing the endpoint. A's self-contained endpoint may work functionally, but it misses the required AI-integration architecture entirely."
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"direction": "better",
|
|
"rationale": "Both are strong, appropriately scoped plans and satisfy almost all requirements. B is slightly stronger because it directly specifies the required VectorData.Abstractions/pgvector direction and has a more coherent MEAI-centric stack, while A offers useful project-specific inspection and concrete PDF parsing details."
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"direction": "none",
|
|
"rationale": "Position-swap inconsistent (forward: A, reverse: B). Defaulting to tie."
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"direction": "better",
|
|
"rationale": "Both are successful, verified ML.NET implementations and candidly flag the severe dataset-size limitation. B is marginally stronger overall as an implementation because it uses the framework-supported thread-safe prediction pool and model persistence while retaining an appropriate warning about invalid tiny-sample evaluation; A's explanation is marginally better."
|
|
}
|
|
],
|
|
"links": [
|
|
{
|
|
"label": "Skill source",
|
|
"url": "https://github.com/dotnet/skills/blob/ac8f41264bdd557e58924a3110eea8e0917dcf4d/plugins/dotnet-ai/skills/technology-selection/SKILL.md"
|
|
},
|
|
{
|
|
"label": "Eval source",
|
|
"url": "https://github.com/dotnet/skills/blob/ac8f41264bdd557e58924a3110eea8e0917dcf4d/tests/dotnet-ai/technology-selection/eval.yaml"
|
|
},
|
|
{
|
|
"label": "Commit",
|
|
"url": "https://github.com/dotnet/skills/commit/ac8f41264bdd557e58924a3110eea8e0917dcf4d"
|
|
}
|
|
]
|
|
}
|
|
],
|
|
"tool": "customBiggerIsBetter",
|
|
"benches": [
|
|
{
|
|
"value": 7.5,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.45,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Quality"
|
|
},
|
|
{
|
|
"value": 9.5,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.45,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Quality"
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.45,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Quality"
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.45,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Quality"
|
|
},
|
|
{
|
|
"value": 5.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Quality"
|
|
},
|
|
{
|
|
"value": 8.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.45,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Quality"
|
|
},
|
|
{
|
|
"value": 8.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.45,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Quality"
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.45,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Quality"
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.45,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Quality"
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Quality"
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.45,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Quality"
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.45,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Quality"
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Quality"
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.45,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Quality"
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.45,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Quality"
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Quality"
|
|
}
|
|
],
|
|
"date": 1788715663667,
|
|
"commit": {
|
|
"id": "ac8f41264bdd557e58924a3110eea8e0917dcf4d",
|
|
"timestamp": "2026-09-04T10:21:25+00:00",
|
|
"message": "Improve non-passing dotnet-test scenarios (#1114)",
|
|
"url": "https://github.com/dotnet/skills/commit/ac8f41264bdd557e58924a3110eea8e0917dcf4d",
|
|
"author": {
|
|
"name": "Amaury Levé",
|
|
"username": "AbhitejJohn"
|
|
}
|
|
},
|
|
"model": "claude-opus-5"
|
|
},
|
|
{
|
|
"verdictEvidence": [
|
|
{
|
|
"skillName": "technology-selection",
|
|
"skillKind": "invocable",
|
|
"state": "VALID_PASS",
|
|
"stateReason": {
|
|
"code": "credible_preference_improvement",
|
|
"phase": "decision"
|
|
},
|
|
"passed": true,
|
|
"regressed": false,
|
|
"preferenceRegressed": false,
|
|
"reason": "Net win +100.0% (6W/0T/0L over 6 preference-eligible stimulus vote(s), sign test p=0.016), mean preference +60.0% across 6 paired run(s) — credibly better",
|
|
"gateEvidence": {
|
|
"stimulusVoteCount": 6,
|
|
"wins": 6,
|
|
"ties": 0,
|
|
"losses": 0,
|
|
"discordant": 6,
|
|
"direction": "better",
|
|
"pValue": 0.015625000000000007,
|
|
"alpha": 0.05,
|
|
"netWin": 1,
|
|
"minimumNetWin": 0.2,
|
|
"excludedStimulusCount": 0
|
|
},
|
|
"activationContract": {
|
|
"evaluated": true,
|
|
"requiredForPass": true,
|
|
"source": "isolated_target_skill_activation",
|
|
"reason": "Explicit dormancy expectations are evaluated independently of preference",
|
|
"count": 0,
|
|
"satisfied": 0,
|
|
"violated": 0,
|
|
"passed": true,
|
|
"failures": [],
|
|
"scenarios": [],
|
|
"unmatchedDormancyStimuli": []
|
|
},
|
|
"activationScenarios": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
}
|
|
],
|
|
"judgeRationales": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"direction": "better",
|
|
"rationale": "Both appear to build clean, functional .NET 10 Agent Framework research apps with search and notes. B is stronger because it explicitly adds an iteration guardrail and output-token control; neither demonstrates the requested step-level observability."
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"direction": "better",
|
|
"rationale": "B fulfills the rubric's required Microsoft.Extensions.AI-based, DI-wired, configured and resilient LLM endpoint design. A provides a functional tested local extractive endpoint, but it misses every substantive AI-integration requirement in the stated rubric."
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"direction": "better",
|
|
"rationale": "B satisfies every listed implementation requirement, including reproducibility, held-out evaluation, and thread-safe pooled prediction. A has a functional ML.NET endpoint but misses the seed and held-out-split requirements and uses the explicitly disallowed singleton PredictionEngine."
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"direction": "better",
|
|
"rationale": "Both are good, appropriately stop at a plan, and cover the requested RAG architecture. B narrowly wins because it directly aligns with the required MEAI and Microsoft.Extensions.VectorData.Abstractions stack while retaining the requested Blazor, folder, and Postgres design. A is more detailed operationally and safer in some package specificity, but misses the explicit vector abstraction criterion."
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"direction": "better",
|
|
"rationale": "Both are solid, code-free RAG architecture plans covering ingestion, chunking, storage, grounded generation, and citations. B more completely and explicitly fulfills the requested .NET abstractions and relevance-control requirements, especially IEmbeddingGenerator and minimum-score rejection."
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"direction": "better",
|
|
"rationale": "Both fulfill the task well and validate live predictions. B is modestly stronger in production-oriented inference handling and candid evaluation of the extremely small dataset, while retaining the correct ML.NET approach."
|
|
}
|
|
],
|
|
"links": [
|
|
{
|
|
"label": "Skill source",
|
|
"url": "https://github.com/dotnet/skills/blob/ac8f41264bdd557e58924a3110eea8e0917dcf4d/plugins/dotnet-ai/skills/technology-selection/SKILL.md"
|
|
},
|
|
{
|
|
"label": "Eval source",
|
|
"url": "https://github.com/dotnet/skills/blob/ac8f41264bdd557e58924a3110eea8e0917dcf4d/tests/dotnet-ai/technology-selection/eval.yaml"
|
|
},
|
|
{
|
|
"label": "Commit",
|
|
"url": "https://github.com/dotnet/skills/commit/ac8f41264bdd557e58924a3110eea8e0917dcf4d"
|
|
}
|
|
]
|
|
}
|
|
],
|
|
"tool": "customBiggerIsBetter",
|
|
"benches": [
|
|
{
|
|
"value": 9.5,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.44,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Quality"
|
|
},
|
|
{
|
|
"value": 9.5,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.44,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Quality"
|
|
},
|
|
{
|
|
"value": 3.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Quality"
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.44,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Quality"
|
|
},
|
|
{
|
|
"value": 9.583333015441895,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.44,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Quality"
|
|
},
|
|
{
|
|
"value": 5.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Quality"
|
|
},
|
|
{
|
|
"value": 8.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.44,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Quality"
|
|
},
|
|
{
|
|
"value": 8.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.44,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Quality"
|
|
},
|
|
{
|
|
"value": 3.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Quality"
|
|
},
|
|
{
|
|
"value": 6.666666507720947,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.44,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Quality"
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.44,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Quality"
|
|
},
|
|
{
|
|
"value": 9.166666984558105,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Quality"
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.44,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Quality"
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.44,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Quality"
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Quality"
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.44,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Quality"
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.44,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Quality"
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Quality"
|
|
}
|
|
],
|
|
"date": 1788715663752,
|
|
"commit": {
|
|
"id": "ac8f41264bdd557e58924a3110eea8e0917dcf4d",
|
|
"timestamp": "2026-09-04T10:21:25+00:00",
|
|
"message": "Improve non-passing dotnet-test scenarios (#1114)",
|
|
"url": "https://github.com/dotnet/skills/commit/ac8f41264bdd557e58924a3110eea8e0917dcf4d",
|
|
"author": {
|
|
"name": "Amaury Levé",
|
|
"username": "AbhitejJohn"
|
|
}
|
|
},
|
|
"model": "claude-sonnet-5"
|
|
},
|
|
{
|
|
"verdictEvidence": [
|
|
{
|
|
"skillName": "technology-selection",
|
|
"skillKind": "invocable",
|
|
"state": "VALID_PASS",
|
|
"stateReason": {
|
|
"code": "credible_preference_improvement",
|
|
"phase": "decision"
|
|
},
|
|
"passed": true,
|
|
"regressed": false,
|
|
"preferenceRegressed": false,
|
|
"reason": "Net win +100.0% (6W/0T/0L over 6 preference-eligible stimulus vote(s), sign test p=0.016), mean preference +50.0% across 6 paired run(s) — credibly better",
|
|
"gateEvidence": {
|
|
"stimulusVoteCount": 6,
|
|
"wins": 6,
|
|
"ties": 0,
|
|
"losses": 0,
|
|
"discordant": 6,
|
|
"direction": "better",
|
|
"pValue": 0.015625000000000007,
|
|
"alpha": 0.05,
|
|
"netWin": 1,
|
|
"minimumNetWin": 0.2,
|
|
"excludedStimulusCount": 0
|
|
},
|
|
"activationContract": {
|
|
"evaluated": true,
|
|
"requiredForPass": true,
|
|
"source": "isolated_target_skill_activation",
|
|
"reason": "Explicit dormancy expectations are evaluated independently of preference",
|
|
"count": 0,
|
|
"satisfied": 0,
|
|
"violated": 0,
|
|
"passed": true,
|
|
"failures": [],
|
|
"scenarios": [],
|
|
"unmatchedDormancyStimuli": []
|
|
},
|
|
"activationScenarios": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
}
|
|
],
|
|
"judgeRationales": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"direction": "better",
|
|
"rationale": "Both responses produced a working .NET 10 build using Microsoft Agent Framework over IChatClient, and both verified the app runs (secret validation, usage handling). However, B addresses more rubric criteria substantively: bounded iterations/token budget, explicit tool schemas, DI-based logging/observability, retries, and timeouts. B also leveraged the technology-selection skill and made a more principled architectural decision. A relies on hosted web search (less explicit tooling) and shows no evidence of iteration caps, token budgets, or logging. B is meaningfully more complete against the rubric, though both are functional."
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"direction": "better",
|
|
"rationale": "The rubric clearly targets an LLM-based summarization endpoint built on Microsoft.Extensions.AI with proper DI, chat options, retries, config-based credentials, and a pinned dated model. Response B satisfies all six criteria, correctly interpreting the task as a summarization LLM operation and building the appropriate architecture, then verifying build and the 400 validation path. Response A built a deterministic string-truncation 'summarizer' that, while functional and verified end-to-end, meets none of the AI-specific rubric criteria. Both were methodical and error-free, but B aligns with the intended solution across the board.\""
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"direction": "better",
|
|
"rationale": "Both produce working ML.NET classifiers with train/test evaluation, multiclass metrics, validation, and a prediction endpoint. B is superior on the key differentiators: it uses the correct PredictionEnginePool pattern (A uses a singleton service), emphasizes deterministic seeding, and uses a category-balanced split producing more sensible 66.7% accuracy vs A's degenerate random-split 33.3%. B also exposes per-class scores. These give B a slight but clear overall edge.\""
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"direction": "better",
|
|
"rationale": "Both responses correctly deliver a plan before implementation, use Blazor, MEAI, pgvector, and the local file folder, with no hallucinated packages. B is more thorough: it explicitly names Microsoft.Extensions.VectorData, adds security/auth/access-control considerations, semantic chunking with page preservation, incremental re-indexing via hosted service, and an architecture diagram. It also correctly reasons about not needing an agent framework. A's workspace inspection was a nice touch, but B's plan quality is stronger. Neither names a concrete LLM provider. B is slightly better overall."
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"direction": "better",
|
|
"rationale": "Both responses are strong, well-structured architecture plans that correctly avoid writing code. B more explicitly hits the specific rubric criteria: naming Microsoft.Extensions.AI's IEmbeddingGenerator, Microsoft.Extensions.VectorData, an explicit configurable minimum similarity threshold, and explicit no-re-embedding-at-query-time guidance. A is comprehensive with more detail on observability and delivery, but is less precise on the specific technology names and the similarity threshold. B wins on several criteria while tying on chunking and attribution, giving it a slight overall edge."
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"direction": "better",
|
|
"rationale": "Both responses correctly chose ML.NET, avoided LLMs, and delivered working, building implementations with predictions, along with the important caveat about the tiny dataset. A has cleaner input validation and a clear churn endpoint. B goes further with a proper evaluation surface (metrics endpoint reporting accuracy/AUC/F1), model persistence, deterministic seeding, stratified fallback, IOptions configuration, and a thread-safe PredictionEnginePool—reflecting more methodical, production-grade engineering. Both articulate why ML.NET fits, but B's reasoning more directly addresses the LLM-vs-classic-ML tradeoff. The gap is modest since both satisfy all rubric items."
|
|
}
|
|
],
|
|
"links": [
|
|
{
|
|
"label": "Skill source",
|
|
"url": "https://github.com/dotnet/skills/blob/ac8f41264bdd557e58924a3110eea8e0917dcf4d/plugins/dotnet-ai/skills/technology-selection/SKILL.md"
|
|
},
|
|
{
|
|
"label": "Eval source",
|
|
"url": "https://github.com/dotnet/skills/blob/ac8f41264bdd557e58924a3110eea8e0917dcf4d/tests/dotnet-ai/technology-selection/eval.yaml"
|
|
},
|
|
{
|
|
"label": "Commit",
|
|
"url": "https://github.com/dotnet/skills/commit/ac8f41264bdd557e58924a3110eea8e0917dcf4d"
|
|
}
|
|
]
|
|
}
|
|
],
|
|
"tool": "customBiggerIsBetter",
|
|
"benches": [
|
|
{
|
|
"value": 3.5,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Quality"
|
|
},
|
|
{
|
|
"value": 3.5,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Quality"
|
|
},
|
|
{
|
|
"value": 2.5,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Quality"
|
|
},
|
|
{
|
|
"value": 6.25,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Quality"
|
|
},
|
|
{
|
|
"value": 7.916666507720947,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Quality"
|
|
},
|
|
{
|
|
"value": 5.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Quality"
|
|
},
|
|
{
|
|
"value": 4.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Quality"
|
|
},
|
|
{
|
|
"value": 6.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Quality"
|
|
},
|
|
{
|
|
"value": 3.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Quality"
|
|
},
|
|
{
|
|
"value": 9.166666984558105,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Quality"
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Quality"
|
|
},
|
|
{
|
|
"value": 9.166666984558105,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Quality"
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Quality"
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Quality"
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Quality"
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Quality"
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Quality"
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Quality"
|
|
}
|
|
],
|
|
"date": 1788715663815,
|
|
"commit": {
|
|
"id": "ac8f41264bdd557e58924a3110eea8e0917dcf4d",
|
|
"timestamp": "2026-09-04T10:21:25+00:00",
|
|
"message": "Improve non-passing dotnet-test scenarios (#1114)",
|
|
"url": "https://github.com/dotnet/skills/commit/ac8f41264bdd557e58924a3110eea8e0917dcf4d",
|
|
"author": {
|
|
"name": "Amaury Levé",
|
|
"username": "AbhitejJohn"
|
|
}
|
|
},
|
|
"model": "gpt-5.6-sol"
|
|
},
|
|
{
|
|
"benches": [
|
|
{
|
|
"value": 7.5,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Quality",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47
|
|
},
|
|
{
|
|
"value": 7.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47
|
|
},
|
|
{
|
|
"value": 2.5,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Quality"
|
|
},
|
|
{
|
|
"value": 9.583333015441895,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Quality",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47
|
|
},
|
|
{
|
|
"value": 9.583333015441895,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47
|
|
},
|
|
{
|
|
"value": 7.5,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Quality"
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Quality",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47
|
|
},
|
|
{
|
|
"value": 3.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Quality"
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Quality",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47
|
|
},
|
|
{
|
|
"value": 5.833333492279053,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Quality"
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Quality",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47
|
|
},
|
|
{
|
|
"value": 9.375,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Quality"
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Quality",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Quality"
|
|
}
|
|
],
|
|
"commit": {
|
|
"message": "Improve non-passing dotnet-test scenarios (#1114)",
|
|
"id": "ac8f41264bdd557e58924a3110eea8e0917dcf4d",
|
|
"timestamp": "2026-09-04T10:21:25+00:00",
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Amaury Levé"
|
|
},
|
|
"url": "https://github.com/dotnet/skills/commit/ac8f41264bdd557e58924a3110eea8e0917dcf4d"
|
|
},
|
|
"date": 1788790038987,
|
|
"verdictEvidence": [
|
|
{
|
|
"skillName": "technology-selection",
|
|
"skillKind": "invocable",
|
|
"state": "VALID_PASS",
|
|
"stateReason": {
|
|
"code": "credible_preference_improvement",
|
|
"phase": "decision"
|
|
},
|
|
"passed": true,
|
|
"regressed": false,
|
|
"preferenceRegressed": false,
|
|
"reason": "Net win +100.0% (6W/0T/0L over 6 preference-eligible stimulus vote(s), sign test p=0.016), mean preference +70.0% across 6 paired run(s) — credibly better",
|
|
"gateEvidence": {
|
|
"stimulusVoteCount": 6,
|
|
"wins": 6,
|
|
"ties": 0,
|
|
"losses": 0,
|
|
"discordant": 6,
|
|
"direction": "better",
|
|
"pValue": 0.015625000000000007,
|
|
"alpha": 0.05,
|
|
"netWin": 1,
|
|
"minimumNetWin": 0.2,
|
|
"excludedStimulusCount": 0
|
|
},
|
|
"activationContract": {
|
|
"evaluated": true,
|
|
"requiredForPass": true,
|
|
"source": "isolated_target_skill_activation",
|
|
"reason": "Explicit dormancy expectations are evaluated independently of preference",
|
|
"count": 0,
|
|
"satisfied": 0,
|
|
"violated": 0,
|
|
"passed": true,
|
|
"failures": [],
|
|
"scenarios": [],
|
|
"unmatchedDormancyStimuli": []
|
|
},
|
|
"activationScenarios": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-no-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
}
|
|
],
|
|
"judgeRationales": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"direction": "better",
|
|
"rationale": "Both builds compile cleanly and satisfy the basic app request, but B directly meets the key Agent Framework, bounded-loop, and explicit-schema requirements. A is a simpler IChatClient function-invocation implementation with simulated search and lacks the requested orchestration safeguards."
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"direction": "better",
|
|
"rationale": "Both deliver a built DI-based summarization endpoint, but B supplies the important explicit generation limits and a dated model pin, and shows a more deliberate resilience approach. A lacks evidence for those core operational safeguards."
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"direction": "better",
|
|
"rationale": "B more directly satisfies the requested operational ML.NET practices: deterministic setup, held-out metrics, and thread-safe pooled prediction. However, its unnecessary augmentation of the supplied CSV and apparent omission of Priority from the displayed feature pipeline are meaningful implementation concerns; A preserves and incorporates Priority and has reasonable cross-validation, so the overall gap is not large."
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"direction": "better",
|
|
"rationale": "B fulfills the key prescribed Microsoft.Extensions.AI and VectorData abstraction choices that A misses, while retaining all core requirements and adding practical RAG quality, security, citation, and resilience details. A is a reasonable generic architecture plan, but it is materially less aligned with the requested technology-selection rubric."
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"direction": "better",
|
|
"rationale": "Both are concise, code-free, credible RAG plans with ingestion, retrieval, grounded prompting, and citations. B better satisfies the requested .NET AI abstraction and relevance-control details by explicitly using IEmbeddingGenerator, VectorData, embedding caching, and a score threshold. A is somewhat more concrete on chunk sizing and project organization, but misses the threshold and is less decisive on the specified embedding abstraction."
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"direction": "better",
|
|
"rationale": "Both are successful, compiling ML.NET churn predictors with appropriate feature processing and explanations. B is modestly more production-conscious and evaluable, while A remains a credible complete implementation."
|
|
}
|
|
],
|
|
"links": [
|
|
{
|
|
"label": "Skill source",
|
|
"url": "https://github.com/dotnet/skills/blob/ac8f41264bdd557e58924a3110eea8e0917dcf4d/plugins/dotnet-ai/skills/technology-selection/SKILL.md"
|
|
},
|
|
{
|
|
"label": "Eval source",
|
|
"url": "https://github.com/dotnet/skills/blob/ac8f41264bdd557e58924a3110eea8e0917dcf4d/tests/dotnet-ai/technology-selection/eval.yaml"
|
|
},
|
|
{
|
|
"label": "Commit",
|
|
"url": "https://github.com/dotnet/skills/commit/ac8f41264bdd557e58924a3110eea8e0917dcf4d"
|
|
}
|
|
]
|
|
}
|
|
],
|
|
"model": "claude-sonnet-4.6",
|
|
"tool": "customBiggerIsBetter"
|
|
},
|
|
{
|
|
"benches": [
|
|
{
|
|
"value": 3.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Quality",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"value": 3.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"value": 3.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Quality"
|
|
},
|
|
{
|
|
"value": 8.333333015441895,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Quality",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"value": 5.833333492279053,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"value": 5.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Quality"
|
|
},
|
|
{
|
|
"value": 4.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Quality",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"value": 3.5,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"value": 3.5,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Quality"
|
|
},
|
|
{
|
|
"value": 9.166666984558105,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Quality",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"value": 5.833333492279053,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Quality"
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Quality",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"value": 9.375,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Quality"
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Quality"
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Quality"
|
|
},
|
|
{
|
|
"value": 6.666666507720947,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Quality"
|
|
}
|
|
],
|
|
"commit": {
|
|
"message": "Improve non-passing dotnet-test scenarios (#1114)",
|
|
"id": "ac8f41264bdd557e58924a3110eea8e0917dcf4d",
|
|
"timestamp": "2026-09-04T10:21:25+00:00",
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Amaury Levé"
|
|
},
|
|
"url": "https://github.com/dotnet/skills/commit/ac8f41264bdd557e58924a3110eea8e0917dcf4d"
|
|
},
|
|
"date": 1788790039073,
|
|
"verdictEvidence": [
|
|
{
|
|
"skillName": "technology-selection",
|
|
"skillKind": "invocable",
|
|
"state": "VALID_PASS",
|
|
"stateReason": {
|
|
"code": "credible_preference_improvement",
|
|
"phase": "decision"
|
|
},
|
|
"passed": true,
|
|
"regressed": false,
|
|
"preferenceRegressed": false,
|
|
"reason": "Net win +100.0% (6W/0T/0L over 6 preference-eligible stimulus vote(s), sign test p=0.016), mean preference +50.0% across 6 paired run(s) — credibly better",
|
|
"gateEvidence": {
|
|
"stimulusVoteCount": 6,
|
|
"wins": 6,
|
|
"ties": 0,
|
|
"losses": 0,
|
|
"discordant": 6,
|
|
"direction": "better",
|
|
"pValue": 0.015625000000000007,
|
|
"alpha": 0.05,
|
|
"netWin": 1,
|
|
"minimumNetWin": 0.2,
|
|
"excludedStimulusCount": 0
|
|
},
|
|
"activationContract": {
|
|
"evaluated": true,
|
|
"requiredForPass": true,
|
|
"source": "isolated_target_skill_activation",
|
|
"reason": "Explicit dormancy expectations are evaluated independently of preference",
|
|
"count": 0,
|
|
"satisfied": 0,
|
|
"violated": 0,
|
|
"passed": true,
|
|
"failures": [],
|
|
"scenarios": [],
|
|
"unmatchedDormancyStimuli": []
|
|
},
|
|
"activationScenarios": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
}
|
|
],
|
|
"judgeRationales": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"direction": "better",
|
|
"rationale": "Both produced a building .NET 10 console research agent using Microsoft.Agents.AI on top of IChatClient with web search and note-taking tools, and both validated the API-key error path. B additionally implements bounded tool iterations (an iteration cap) and researched the actual API surface (inspecting the XML docs) to fix real compile errors, showing a more careful, evidence-driven approach. A verified its build too but shows no iteration bounding. The gap is modest, so B is slightly better overall."
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"direction": "better",
|
|
"rationale": "Response B satisfies essentially all six rubric criteria: it uses Microsoft.Extensions.AI/IChatClient via DI, sets token limits/prompting, adds Polly retry and timeout, loads keys from config, and pins a model. It also verified a successful build after resolving package version conflicts. Response A implemented a trivial local sentence-selection summarizer that ignores the intended AI approach entirely and satisfies none of the rubric criteria, though it did verify the endpoint works. On result quality against the rubric, B is much better."
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"direction": "better",
|
|
"rationale": "Both responses successfully implement an ML.NET multiclass classifier with working prediction and metrics endpoints, verified via curl. They tie on ML.NET usage, train/test evaluation, and metrics reporting. However, B explicitly satisfies two rubric criteria that A shows no evidence of meeting: setting a random seed on MLContext (seed: 42) and using PredictionEnginePool for thread-safe inference. A instead uses cross-validation (robust but doesn't clearly hit the seed criterion) and shows no PredictionEnginePool usage. B aligns more closely with the specific best-practice rubric requirements, making it slightly better overall. A's only edge (LogLossReduction, CV) is minor by comparison."
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"direction": "better",
|
|
"rationale": "Both provide solid, well-organized plans that respect Blazor, Postgres/pgvector, and a file folder. However, B directly satisfies the key technical rubric criteria by naming Microsoft.Extensions.AI with IChatClient and concrete providers, and makes an informed architectural decision (IChatClient over Agent Framework). A stays generic ('provider abstraction') and never commits to MEAI or a concrete provider, missing central rubric points. B is clearly stronger on the decisive technology-selection criteria. [Magnitude reduced to \"slightly-better\": the position-swapped pass judged this \"slightly-better\", so the more confident \"much-better\" was downgraded.]"
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"direction": "better",
|
|
"rationale": "Both responses are strong, well-structured, code-free plans covering chunking, retrieval, thresholds, and citations. However, B directly satisfies the technology-selection rubric criteria by naming Microsoft.Extensions.AI, Microsoft.Extensions.VectorData, and an explicit embedding cache, whereas A stays generic on these points. B wins three criteria and ties three, making it the better response overall, though A remains a high-quality answer."
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"direction": "better",
|
|
"rationale": "Both responses correctly chose ML.NET, avoided LLMs, and implemented a working churn classification API over the provided CSV, verified via curl. B is slightly better: it includes a held-out evaluation split with a metrics endpoint, uses PredictionEnginePool for thread-safe serving, corrected the label pipeline, and gave a slightly more grounded technology justification explicitly contrasting with LLM suitability. A is solid and functional but simpler. The gap is modest."
|
|
}
|
|
],
|
|
"links": [
|
|
{
|
|
"label": "Skill source",
|
|
"url": "https://github.com/dotnet/skills/blob/ac8f41264bdd557e58924a3110eea8e0917dcf4d/plugins/dotnet-ai/skills/technology-selection/SKILL.md"
|
|
},
|
|
{
|
|
"label": "Eval source",
|
|
"url": "https://github.com/dotnet/skills/blob/ac8f41264bdd557e58924a3110eea8e0917dcf4d/tests/dotnet-ai/technology-selection/eval.yaml"
|
|
},
|
|
{
|
|
"label": "Commit",
|
|
"url": "https://github.com/dotnet/skills/commit/ac8f41264bdd557e58924a3110eea8e0917dcf4d"
|
|
}
|
|
]
|
|
}
|
|
],
|
|
"model": "gpt-5.6-luna",
|
|
"tool": "customBiggerIsBetter"
|
|
},
|
|
{
|
|
"tool": "customBiggerIsBetter",
|
|
"benches": [
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Quality",
|
|
"value": 2.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Quality",
|
|
"value": 6.666666507720947
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Quality",
|
|
"value": 9.583333015441895
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Quality",
|
|
"value": 5.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Quality",
|
|
"value": 4.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Quality",
|
|
"value": 2.5
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Quality",
|
|
"value": 3.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Quality",
|
|
"value": 9.166666984558105
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Quality",
|
|
"value": 10.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Quality",
|
|
"value": 5.833333492279053
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Quality",
|
|
"value": 10.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Quality",
|
|
"value": 10.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Quality",
|
|
"value": 8.75
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Quality",
|
|
"value": 10.0
|
|
}
|
|
],
|
|
"date": 1788885915193,
|
|
"verdictEvidence": [
|
|
{
|
|
"skillName": "technology-selection",
|
|
"skillKind": "invocable",
|
|
"state": "INVALID_INCONCLUSIVE",
|
|
"stateReason": {
|
|
"code": "unmatched_trajectories",
|
|
"phase": "comparison_pairing"
|
|
},
|
|
"passed": false,
|
|
"regressed": false,
|
|
"preferenceRegressed": false,
|
|
"reason": "Net win +100.0% (4W/0T/0L over 4 preference-eligible stimulus vote(s), sign test p=0.063), mean preference +40.0% across 4 paired run(s), 2 unmatched — inconclusive (unmatched trajectories)",
|
|
"gateEvidence": {
|
|
"stimulusVoteCount": 4,
|
|
"wins": 4,
|
|
"ties": 0,
|
|
"losses": 0,
|
|
"discordant": 4,
|
|
"direction": "better",
|
|
"pValue": 0.0625,
|
|
"alpha": 0.05,
|
|
"netWin": 1,
|
|
"minimumNetWin": 0.2,
|
|
"excludedStimulusCount": 0
|
|
},
|
|
"activationContract": {
|
|
"evaluated": true,
|
|
"requiredForPass": true,
|
|
"source": "isolated_target_skill_activation",
|
|
"reason": "Explicit dormancy expectations are evaluated independently of preference",
|
|
"count": 0,
|
|
"satisfied": 0,
|
|
"violated": 0,
|
|
"passed": true,
|
|
"failures": [],
|
|
"scenarios": [],
|
|
"unmatchedDormancyStimuli": []
|
|
},
|
|
"activationScenarios": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "missing-activation",
|
|
"plugin": "plugin-no-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-no-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "missing-activation",
|
|
"plugin": "plugin-no-activity-observed"
|
|
}
|
|
],
|
|
"judgeRationales": [
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"direction": "better",
|
|
"rationale": "Both miss the two central Microsoft.Extensions.AI/AddChatClient architecture requirements and neither adds retries or a pinned model. B nevertheless supplies a build-verified LLM-based endpoint with explicit generation controls and secure configuration-based authentication, whereas A is a functioning but non-AI extractive summarizer. A's runtime test is a positive, but does not offset B's additional rubric-aligned AI behavior."
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"direction": "better",
|
|
"rationale": "Both appear to provide a working ML.NET API, but B clearly demonstrates the key reproducibility, held-out evaluation, concrete metrics, and thread-safe PredictionEnginePool requirements. A's final response is largely unsupported summary claims and does not evidence these critical implementation details. [Magnitude reduced to \"slightly-better\": the position-swapped pass judged this \"slightly-better\", so the more confident \"much-better\" was downgraded.]"
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"direction": "better",
|
|
"rationale": "B is the stronger plan because it directly satisfies the requested MEAI and VectorData architectural choices and gives more actionable request/data-flow detail. However, its questionable NuGet guidance—especially `Npgsql.PluginTests`—is a material flaw, keeping the overall margin modest."
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"direction": "better",
|
|
"rationale": "Both are useful code-free RAG architectures covering ingestion, chunking, retrieval, LLM generation, and citations. B is more aligned with the .NET-specific requested architecture and importantly adds similarity-threshold/no-answer behavior and explicit IEmbeddingGenerator/VectorData abstractions. A remains solid but omits the threshold and the requested embedding abstraction."
|
|
}
|
|
],
|
|
"links": [
|
|
{
|
|
"label": "Skill source",
|
|
"url": "https://github.com/dotnet/skills/blob/fbeeafe261b0fe704b9a95c58fd9fdbe34e96964/plugins/dotnet-ai/skills/technology-selection/SKILL.md"
|
|
},
|
|
{
|
|
"label": "Eval source",
|
|
"url": "https://github.com/dotnet/skills/blob/fbeeafe261b0fe704b9a95c58fd9fdbe34e96964/tests/dotnet-ai/technology-selection/eval.yaml"
|
|
},
|
|
{
|
|
"label": "Commit",
|
|
"url": "https://github.com/dotnet/skills/commit/fbeeafe261b0fe704b9a95c58fd9fdbe34e96964"
|
|
}
|
|
]
|
|
}
|
|
],
|
|
"commit": {
|
|
"id": "fbeeafe261b0fe704b9a95c58fd9fdbe34e96964",
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Viktor Hofer"
|
|
},
|
|
"message": "Delete msbuild-server skill (#1123)",
|
|
"timestamp": "2026-09-07T09:11:54+00:00",
|
|
"url": "https://github.com/dotnet/skills/commit/fbeeafe261b0fe704b9a95c58fd9fdbe34e96964"
|
|
},
|
|
"model": "claude-haiku-4.5"
|
|
},
|
|
{
|
|
"tool": "customBiggerIsBetter",
|
|
"benches": [
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Quality",
|
|
"overfitting": "moderate",
|
|
"notActivated": true,
|
|
"value": 4.5,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.24
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.24,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Quality",
|
|
"value": 4.5
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Quality",
|
|
"value": 2.0
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Quality",
|
|
"overfitting": "moderate",
|
|
"notActivated": true,
|
|
"value": 5.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.24
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.24,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Quality",
|
|
"value": 5.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Quality",
|
|
"value": 5.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.24,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Quality",
|
|
"value": 2.5
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.24,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Quality",
|
|
"value": 4.5
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Quality",
|
|
"value": 2.0
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Quality",
|
|
"overfitting": "moderate",
|
|
"notActivated": true,
|
|
"value": 5.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.24
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.24,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Quality",
|
|
"value": 9.166666984558105
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Quality",
|
|
"value": 5.833333492279053
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.24,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Quality",
|
|
"value": 10.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.24,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Quality",
|
|
"value": 10.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Quality",
|
|
"value": 8.75
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Quality",
|
|
"value": 7.5
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Quality",
|
|
"value": 7.5
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Quality",
|
|
"value": 3.3333332538604736
|
|
}
|
|
],
|
|
"date": 1788885915259,
|
|
"verdictEvidence": [
|
|
{
|
|
"skillName": "technology-selection",
|
|
"skillKind": "invocable",
|
|
"state": "VALID_PASS",
|
|
"stateReason": {
|
|
"code": "credible_preference_improvement",
|
|
"phase": "decision"
|
|
},
|
|
"passed": true,
|
|
"regressed": false,
|
|
"preferenceRegressed": false,
|
|
"reason": "Net win +83.3% (5W/1T/0L over 6 preference-eligible stimulus vote(s), sign test p=0.031), mean preference +43.3% across 6 paired run(s) — credibly better",
|
|
"gateEvidence": {
|
|
"stimulusVoteCount": 6,
|
|
"wins": 5,
|
|
"ties": 1,
|
|
"losses": 0,
|
|
"discordant": 5,
|
|
"direction": "better",
|
|
"pValue": 0.03125,
|
|
"alpha": 0.05,
|
|
"netWin": 0.8333333333333334,
|
|
"minimumNetWin": 0.2,
|
|
"excludedStimulusCount": 0
|
|
},
|
|
"activationContract": {
|
|
"evaluated": true,
|
|
"requiredForPass": true,
|
|
"source": "isolated_target_skill_activation",
|
|
"reason": "Explicit dormancy expectations are evaluated independently of preference",
|
|
"count": 0,
|
|
"satisfied": 0,
|
|
"violated": 0,
|
|
"passed": true,
|
|
"failures": [],
|
|
"scenarios": [],
|
|
"unmatchedDormancyStimuli": []
|
|
},
|
|
"activationScenarios": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "missing-activation",
|
|
"plugin": "plugin-no-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "missing-activation",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "missing-activation",
|
|
"plugin": "plugin-no-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
}
|
|
],
|
|
"judgeRationales": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"direction": "better",
|
|
"rationale": "Both are single-turn scaffolds without real dotnet execution. B is meaningfully closer to the intended architecture, using Microsoft.Extensions.AI (IChatClient, AIFunctionFactory, ChatOptions) and MAF-style patterns, whereas A uses raw HTTP calls to the OpenAI Responses API with a hallucinated model name (gpt-5.6-terra). Both cap iterations and define tools with descriptions; neither adds observability logging or a token budget. B's foundation-layer choice gives it the edge, though the gap is modest since both miss several rubric points."
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"direction": "none",
|
|
"rationale": "Both responses failed identically: zero tool calls, no attempt to inspect the workspace, and only a request for the user to share files. Neither implemented the endpoint or satisfied any rubric criterion. The outputs are near-equivalent in quality and content, so this is a tie."
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"direction": "better",
|
|
"rationale": "Both responses failed the core task: neither produced an actual implementation. Response A got stuck on a tooling issue and gave up with only a vague offer to provide code. Response B also declined to write code (incorrectly claiming it needed files, though the files existed and A found them), but B at least articulated the correct technical approach in detail — ML.NET multiclass, train/test split with MicroAccuracy/MacroAccuracy/LogLoss, model persistence, and PredictionEnginePool DI — matching the rubric intent. Since no rubric criterion is fully satisfied by either (no real code), the differences are marginal, but B's plan aligns much more closely with the rubric requirements, making it slightly better."
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"direction": "better",
|
|
"rationale": "Both responses are very similar in quality: clear, well-structured plans presented before coding, using Blazor, pgvector, local file folder, ingestion/chunking, RAG, and ops concerns without hallucinating package names. Neither explicitly references the specific Microsoft.Extensions.AI / VectorData abstractions the rubric ideally wants. B is marginally better because it adds an explicit assumptions section (.NET 8, pgvector, LLM provider, text-vs-scanned PDFs), a clean layered solution structure, deployment approach, and slightly more precise chunking detail, making it a bit more actionable. The gap is small."
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"direction": "better",
|
|
"rationale": "Both give solid two-pipeline RAG architectures with chunking and citations. However, B satisfies nearly all the specific rubric criteria that A misses: it names Microsoft.Extensions.VectorData, Microsoft.Extensions.AI/IEmbeddingGenerator, a minimum similarity threshold, and explicit embedding caching. A is more verbose and slightly richer on chunking prose, but fails to select the specific .NET technologies and threshold/caching details the rubric rewards. B is much better overall on rubric fit while still concise and correct.\" [Magnitude reduced to \"slightly-better\": the position-swapped pass judged this \"slightly-better\", so the more confident \"much-better\" was downgraded.]"
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"direction": "better",
|
|
"rationale": "B directly answers the technology question (ML.NET) with grounded reasoning and sketches a concrete implementation, satisfying two rubric criteria fully. A punts entirely, asking for inputs without choosing a technology or explaining anything. Neither wrote actual code due to missing files, but B is far more responsive and correct."
|
|
}
|
|
],
|
|
"links": [
|
|
{
|
|
"label": "Skill source",
|
|
"url": "https://github.com/dotnet/skills/blob/fbeeafe261b0fe704b9a95c58fd9fdbe34e96964/plugins/dotnet-ai/skills/technology-selection/SKILL.md"
|
|
},
|
|
{
|
|
"label": "Eval source",
|
|
"url": "https://github.com/dotnet/skills/blob/fbeeafe261b0fe704b9a95c58fd9fdbe34e96964/tests/dotnet-ai/technology-selection/eval.yaml"
|
|
},
|
|
{
|
|
"label": "Commit",
|
|
"url": "https://github.com/dotnet/skills/commit/fbeeafe261b0fe704b9a95c58fd9fdbe34e96964"
|
|
}
|
|
]
|
|
}
|
|
],
|
|
"commit": {
|
|
"id": "fbeeafe261b0fe704b9a95c58fd9fdbe34e96964",
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Viktor Hofer"
|
|
},
|
|
"message": "Delete msbuild-server skill (#1123)",
|
|
"timestamp": "2026-09-07T09:11:54+00:00",
|
|
"url": "https://github.com/dotnet/skills/commit/fbeeafe261b0fe704b9a95c58fd9fdbe34e96964"
|
|
},
|
|
"model": "gpt-5.3-codex"
|
|
},
|
|
{
|
|
"tool": "customBiggerIsBetter",
|
|
"benches": [
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Quality",
|
|
"overfitting": "moderate",
|
|
"notActivated": true,
|
|
"value": 2.5,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.46
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Quality",
|
|
"value": 2.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Quality",
|
|
"value": 2.5
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Quality",
|
|
"value": 5.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Quality",
|
|
"value": 5.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Quality",
|
|
"value": 5.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Quality",
|
|
"value": 4.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Quality",
|
|
"value": 2.5
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Quality",
|
|
"value": 2.5
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Quality",
|
|
"value": 5.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Quality",
|
|
"value": 5.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Quality",
|
|
"value": 5.833333492279053
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Quality",
|
|
"overfitting": "moderate",
|
|
"notActivated": true,
|
|
"value": 8.75,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.46
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.46,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Quality",
|
|
"value": 8.75
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Quality",
|
|
"value": 8.75
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Quality",
|
|
"value": 10.0
|
|
}
|
|
],
|
|
"date": 1788885915332,
|
|
"verdictEvidence": [
|
|
{
|
|
"skillName": "technology-selection",
|
|
"skillKind": "invocable",
|
|
"state": "INVALID_INCONCLUSIVE",
|
|
"stateReason": {
|
|
"code": "unmatched_trajectories",
|
|
"phase": "comparison_pairing"
|
|
},
|
|
"passed": false,
|
|
"regressed": false,
|
|
"preferenceRegressed": false,
|
|
"reason": "Net win -40.0% (1W/1T/3L over 5 preference-eligible stimulus vote(s), sign test p=0.312), mean preference -4.0% across 5 paired run(s), 1 unmatched — inconclusive (unmatched trajectories)",
|
|
"gateEvidence": {
|
|
"stimulusVoteCount": 5,
|
|
"wins": 1,
|
|
"ties": 1,
|
|
"losses": 3,
|
|
"discordant": 4,
|
|
"direction": "worse",
|
|
"pValue": 0.31249999999999994,
|
|
"alpha": 0.05,
|
|
"netWin": -0.4,
|
|
"minimumNetWin": 0.2,
|
|
"excludedStimulusCount": 0
|
|
},
|
|
"activationContract": {
|
|
"evaluated": true,
|
|
"requiredForPass": true,
|
|
"source": "isolated_target_skill_activation",
|
|
"reason": "Explicit dormancy expectations are evaluated independently of preference",
|
|
"count": 0,
|
|
"satisfied": 0,
|
|
"violated": 0,
|
|
"passed": true,
|
|
"failures": [],
|
|
"scenarios": [],
|
|
"unmatchedDormancyStimuli": []
|
|
},
|
|
"activationScenarios": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "missing-activation",
|
|
"plugin": "plugin-no-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-no-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-no-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-no-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "missing-activation",
|
|
"plugin": "plugin-no-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "missing-activation",
|
|
"plugin": "plugin-no-activity-observed"
|
|
}
|
|
],
|
|
"judgeRationales": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"direction": "worse",
|
|
"rationale": "Both miss the two central requested Microsoft framework requirements and omit budgeting. A nevertheless provides clearer evidence of an actual bounded agent/tool orchestration loop, while B appears more like a predefined research workflow; B is only marginally better on visible run logging."
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"direction": "none",
|
|
"rationale": "Both agents successfully appear to add and test a working local summarization HTTP endpoint, but both miss every rubric requirement concerning Microsoft.Extensions.AI client integration, DI registration, generation options, retries, credential configuration, and pinned model version. B offers an optional maxSentences parameter and models, but that does not address the stated evaluation criteria."
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"direction": "better",
|
|
"rationale": "B satisfies the crucial production endpoint-pooling requirement and performs genuine held-out evaluation. A's endpoint works, but it replaces proper evaluation with in-sample scoring and presents the resulting perfect accuracy as meaningful, a substantial defect for the requested evaluation implementation."
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"direction": "worse",
|
|
"rationale": "Both are solid, concise plans responsive to the request. A is marginally stronger because it explicitly identifies viable concrete LLM provider options and more decisively recommends pgvector; B adds useful security and retrieval-quality details but remains provider-agnostic. Both miss the rubric's desired MEAI and VectorData abstraction/connector selections."
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"direction": "worse",
|
|
"rationale": "Both are solid, code-free RAG architecture plans and share the two notable omissions of an explicit similarity threshold and Microsoft.Extensions.AI. B is stronger on Markdown-aware chunking and project tailoring, while A is somewhat more complete and concrete on vector-store options, durable provenance/data modeling, and ingestion operational design."
|
|
}
|
|
],
|
|
"links": [
|
|
{
|
|
"label": "Skill source",
|
|
"url": "https://github.com/dotnet/skills/blob/fbeeafe261b0fe704b9a95c58fd9fdbe34e96964/plugins/dotnet-ai/skills/technology-selection/SKILL.md"
|
|
},
|
|
{
|
|
"label": "Eval source",
|
|
"url": "https://github.com/dotnet/skills/blob/fbeeafe261b0fe704b9a95c58fd9fdbe34e96964/tests/dotnet-ai/technology-selection/eval.yaml"
|
|
},
|
|
{
|
|
"label": "Commit",
|
|
"url": "https://github.com/dotnet/skills/commit/fbeeafe261b0fe704b9a95c58fd9fdbe34e96964"
|
|
}
|
|
]
|
|
}
|
|
],
|
|
"commit": {
|
|
"id": "fbeeafe261b0fe704b9a95c58fd9fdbe34e96964",
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Viktor Hofer"
|
|
},
|
|
"message": "Delete msbuild-server skill (#1123)",
|
|
"timestamp": "2026-09-07T09:11:54+00:00",
|
|
"url": "https://github.com/dotnet/skills/commit/fbeeafe261b0fe704b9a95c58fd9fdbe34e96964"
|
|
},
|
|
"model": "mai-code-1-flash-picker"
|
|
},
|
|
{
|
|
"verdictEvidence": [
|
|
{
|
|
"skillName": "technology-selection",
|
|
"skillKind": "invocable",
|
|
"state": "INVALID_INCONCLUSIVE",
|
|
"stateReason": {
|
|
"code": "unmatched_trajectories",
|
|
"phase": "comparison_pairing"
|
|
},
|
|
"passed": false,
|
|
"regressed": false,
|
|
"preferenceRegressed": false,
|
|
"reason": "Net win +100.0% (1W/0T/0L over 1 preference-eligible stimulus vote(s), sign test p=0.500), mean preference +40.0% across 1 paired run(s), 5 unmatched — inconclusive (unmatched trajectories)",
|
|
"gateEvidence": {
|
|
"stimulusVoteCount": 1,
|
|
"wins": 1,
|
|
"ties": 0,
|
|
"losses": 0,
|
|
"discordant": 1,
|
|
"direction": "better",
|
|
"pValue": 0.5,
|
|
"alpha": 0.05,
|
|
"netWin": 1,
|
|
"minimumNetWin": 0.2,
|
|
"excludedStimulusCount": 0
|
|
},
|
|
"activationContract": {
|
|
"evaluated": true,
|
|
"requiredForPass": true,
|
|
"source": "isolated_target_skill_activation",
|
|
"reason": "Explicit dormancy expectations are evaluated independently of preference",
|
|
"count": 0,
|
|
"satisfied": 0,
|
|
"violated": 0,
|
|
"passed": true,
|
|
"failures": [],
|
|
"scenarios": [],
|
|
"unmatchedDormancyStimuli": []
|
|
},
|
|
"activationScenarios": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-no-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
}
|
|
],
|
|
"judgeRationales": [
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"direction": "better",
|
|
"rationale": "Both are successful, verified ML.NET churn implementations with appropriate caveats about the five-row training set. B is modestly stronger overall due to explicit reproducibility, safe pooled prediction serving, model persistence, a useful ranked churn-flag endpoint, and demonstrated recovery from the small-sample AUC failure. The gap is not large because A also provides a correct, tested pipeline and clear technology rationale."
|
|
}
|
|
],
|
|
"links": [
|
|
{
|
|
"label": "Skill source",
|
|
"url": "https://github.com/dotnet/skills/blob/a8fece0fce5f8b0737754a332a1a7c6487e8f927/plugins/dotnet-ai/skills/technology-selection/SKILL.md"
|
|
},
|
|
{
|
|
"label": "Eval source",
|
|
"url": "https://github.com/dotnet/skills/blob/a8fece0fce5f8b0737754a332a1a7c6487e8f927/tests/dotnet-ai/technology-selection/eval.yaml"
|
|
},
|
|
{
|
|
"label": "Commit",
|
|
"url": "https://github.com/dotnet/skills/commit/a8fece0fce5f8b0737754a332a1a7c6487e8f927"
|
|
}
|
|
]
|
|
}
|
|
],
|
|
"commit": {
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Abhitej John"
|
|
},
|
|
"message": "Merge pull request #1149 from dotnet/abhitejjohn-vally-014-ci-proof",
|
|
"url": "https://github.com/dotnet/skills/commit/a8fece0fce5f8b0737754a332a1a7c6487e8f927",
|
|
"timestamp": "2026-09-09T17:53:08+00:00",
|
|
"id": "a8fece0fce5f8b0737754a332a1a7c6487e8f927"
|
|
},
|
|
"model": "claude-opus-4.8",
|
|
"tool": "customBiggerIsBetter",
|
|
"date": 1789036360735,
|
|
"benches": [
|
|
{
|
|
"overfitting": "moderate",
|
|
"value": 9.5,
|
|
"overfittingScore": 0.48,
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Quality",
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"value": 9.5,
|
|
"overfittingScore": 0.48,
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Quality",
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"overfittingScore": 0.48,
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Quality",
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"value": 9.583333015441895,
|
|
"overfittingScore": 0.48,
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Quality",
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"value": 8.0,
|
|
"overfittingScore": 0.48,
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Quality",
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"overfittingScore": 0.48,
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Quality",
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"value": 9.166666984558105,
|
|
"overfittingScore": 0.48,
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Quality",
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"overfittingScore": 0.48,
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Quality",
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"overfittingScore": 0.48,
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Quality",
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"overfittingScore": 0.48,
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Quality",
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"overfittingScore": 0.48,
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Quality",
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Quality",
|
|
"unit": "Score (0-10)"
|
|
}
|
|
]
|
|
},
|
|
{
|
|
"commit": {
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Abhitej John"
|
|
},
|
|
"id": "8a5a42d3e392b402768fc29416831643b79e402b",
|
|
"message": "Merge pull request #1153 from dotnet/abhitejjohn-vally-lock-concurrency-proof",
|
|
"timestamp": "2026-09-10T21:01:20+00:00",
|
|
"url": "https://github.com/dotnet/skills/commit/8a5a42d3e392b402768fc29416831643b79e402b"
|
|
},
|
|
"benches": [
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.28,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Quality",
|
|
"value": 3.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.28,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Quality",
|
|
"value": 5.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Quality",
|
|
"value": 2.5
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.28,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Quality",
|
|
"value": 7.916666507720947
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.28,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Quality",
|
|
"value": 9.583333015441895
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Quality",
|
|
"value": 5.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.28,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Quality",
|
|
"value": 4.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.28,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Quality",
|
|
"value": 5.5
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Quality",
|
|
"value": 3.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.28,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Quality",
|
|
"value": 10.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.28,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Quality",
|
|
"value": 10.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Quality",
|
|
"value": 5.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.28,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Quality",
|
|
"value": 10.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.28,
|
|
"overfitting": "moderate",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Quality",
|
|
"value": 10.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Quality",
|
|
"value": 9.375
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Quality",
|
|
"value": 10.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Quality",
|
|
"value": 9.166666984558105
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Quality",
|
|
"value": 10.0
|
|
}
|
|
],
|
|
"date": 1789129571363,
|
|
"tool": "customBiggerIsBetter",
|
|
"verdictEvidence": [
|
|
{
|
|
"skillName": "technology-selection",
|
|
"skillKind": "invocable",
|
|
"state": "VALID_PASS",
|
|
"stateReason": {
|
|
"code": "credible_preference_improvement",
|
|
"phase": "decision"
|
|
},
|
|
"passed": true,
|
|
"regressed": false,
|
|
"preferenceRegressed": false,
|
|
"reason": "Net win +83.3% (5W/1T/0L over 6 preference-eligible stimulus vote(s), sign test p=0.031), mean preference +43.3% across 6 paired run(s) — credibly better",
|
|
"gateEvidence": {
|
|
"stimulusVoteCount": 6,
|
|
"wins": 5,
|
|
"ties": 1,
|
|
"losses": 0,
|
|
"discordant": 5,
|
|
"direction": "better",
|
|
"pValue": 0.03125,
|
|
"alpha": 0.05,
|
|
"netWin": 0.8333333333333334,
|
|
"minimumNetWin": 0.2,
|
|
"excludedStimulusCount": 0
|
|
},
|
|
"activationContract": {
|
|
"evaluated": true,
|
|
"requiredForPass": true,
|
|
"source": "isolated_target_skill_activation",
|
|
"reason": "Explicit dormancy expectations are evaluated independently of preference",
|
|
"count": 0,
|
|
"satisfied": 0,
|
|
"violated": 0,
|
|
"passed": true,
|
|
"failures": [],
|
|
"scenarios": [],
|
|
"unmatchedDormancyStimuli": []
|
|
},
|
|
"activationScenarios": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
}
|
|
],
|
|
"judgeRationales": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"direction": "better",
|
|
"rationale": "Both responses produce a building .NET 10 console app using Microsoft Agent Framework over IChatClient with web search and note-taking tools. B's process was more methodical: it activated the technology-selection skill, consulted reference docs, and carefully inspected the Microsoft.Agents.AI API surface (AgentResponse accessors) to get correct code, plus it explicitly implements bounded tool iterations and retry handling. A's process was more haphazard — it struggled to find/edit the file, had a botched apply_patch, and used a crude sed insertion to fix a compile error, though it did ultimately build and validate the missing-key path. B edges ahead on the iteration-cap criterion and overall rigor, while other criteria are roughly tied. The advantage is modest."
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"direction": "better",
|
|
"rationale": "Response A built a deterministic extractive summarizer with no AI integration, failing essentially every rubric criterion which all presuppose an LLM/IChatClient-based implementation. While A's approach does technically produce a summary and runs successfully, it does not match the intended AI-based design the rubric evaluates. Response B correctly used Microsoft.Extensions.AI with IChatClient, DI registration, config-based API keys, and a pinned dated model version, satisfying the rubric across the board. B did not verify its endpoint end-to-end at runtime (it lacks an API key), but the code structure aligns fully with the requirements, making it much better overall.\""
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"direction": "better",
|
|
"rationale": "Both delivered working ML.NET classifiers with training, metrics, and prediction endpoints verified via smoke tests. B edges ahead by explicitly setting a reproducibility seed (seed 42) and using a clear train/test split matching the rubric, plus a hint at a pooled prediction service. A relied on cross-validation without a stated seed and produced suspiciously perfect (1.0) metrics indicating overfitting on the tiny dataset. Both had ambiguous evidence around PredictionEnginePool. Overall B better satisfies the reproducibility and train/test rubric points, making it slightly better."
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"direction": "better",
|
|
"rationale": "Both provide sound, code-free plans that use Blazor and a file folder as requested. However, B is substantially stronger against the rubric: it explicitly selects Microsoft.Extensions.AI as the abstraction layer, Microsoft.Extensions.AI.DataIngestion for chunking, and Microsoft.Extensions.VectorData.Abstractions with pgvector — the exact technology stack the rubric targets. A remains generic ('configurable provider', 'store embeddings in PostgreSQL') and never names these packages. B also offers richer operational, reliability, and phased implementation detail. Both correctly avoid unrequested infrastructure. Neither names a concrete LLM provider explicitly, a minor shared gap. Overall B is much better aligned with the rubric.\" [Magnitude reduced to \"slightly-better\": the position-swapped pass judged this \"slightly-better\", so the more confident \"much-better\" was downgraded.]"
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"direction": "better",
|
|
"rationale": "Both provide strong, well-structured plans meeting the 'no code' constraint. However B hits the specific rubric targets much more directly: it names Microsoft.Extensions.AI/IEmbeddingGenerator, Microsoft.Extensions.VectorData, explicit similarity thresholds, and embedding caching by content hash. A is solid and provider-agnostic but misses these specific technology selections and the explicit similarity threshold and caching details. B wins clearly on 4 of 6 criteria. [Magnitude reduced to \"slightly-better\": the position-swapped pass judged this \"slightly-better\", so the more confident \"much-better\" was downgraded.]"
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"direction": "none",
|
|
"rationale": "Position-swap inconsistent (forward: tie, reverse: A). Defaulting to tie."
|
|
}
|
|
],
|
|
"links": [
|
|
{
|
|
"label": "Skill source",
|
|
"url": "https://github.com/dotnet/skills/blob/8a5a42d3e392b402768fc29416831643b79e402b/plugins/dotnet-ai/skills/technology-selection/SKILL.md"
|
|
},
|
|
{
|
|
"label": "Eval source",
|
|
"url": "https://github.com/dotnet/skills/blob/8a5a42d3e392b402768fc29416831643b79e402b/tests/dotnet-ai/technology-selection/eval.yaml"
|
|
},
|
|
{
|
|
"label": "Commit",
|
|
"url": "https://github.com/dotnet/skills/commit/8a5a42d3e392b402768fc29416831643b79e402b"
|
|
}
|
|
]
|
|
}
|
|
],
|
|
"model": "gpt-5.6-luna"
|
|
},
|
|
{
|
|
"verdictEvidence": [
|
|
{
|
|
"skillName": "technology-selection",
|
|
"skillKind": "invocable",
|
|
"state": "INVALID_INCONCLUSIVE",
|
|
"stateReason": {
|
|
"code": "unmatched_trajectories",
|
|
"phase": "comparison_pairing"
|
|
},
|
|
"passed": false,
|
|
"regressed": false,
|
|
"preferenceRegressed": false,
|
|
"reason": "Net win +80.0% (4W/1T/0L over 5 preference-eligible stimulus vote(s), sign test p=0.063), mean preference +56.0% across 5 paired run(s), 1 unmatched — inconclusive (unmatched trajectories)",
|
|
"gateEvidence": {
|
|
"stimulusVoteCount": 5,
|
|
"wins": 4,
|
|
"ties": 1,
|
|
"losses": 0,
|
|
"discordant": 4,
|
|
"direction": "better",
|
|
"pValue": 0.0625,
|
|
"alpha": 0.05,
|
|
"netWin": 0.8,
|
|
"minimumNetWin": 0.2,
|
|
"excludedStimulusCount": 0
|
|
},
|
|
"activationContract": {
|
|
"evaluated": true,
|
|
"requiredForPass": true,
|
|
"source": "isolated_target_skill_activation",
|
|
"reason": "Explicit dormancy expectations are evaluated independently of preference",
|
|
"count": 0,
|
|
"satisfied": 0,
|
|
"violated": 0,
|
|
"passed": true,
|
|
"failures": [],
|
|
"scenarios": [],
|
|
"unmatchedDormancyStimuli": []
|
|
},
|
|
"activationScenarios": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "missing-activation",
|
|
"plugin": "plugin-no-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-no-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-no-activity-observed"
|
|
}
|
|
],
|
|
"judgeRationales": [
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"direction": "better",
|
|
"rationale": "B is materially closer to the requested AI-oriented architecture: it uses MEAI/IChatClient, configuration-based credentials, and explicit generation settings while providing a built endpoint. However, it does not demonstrate retries and uses an unversioned model alias; A is a functioning offline extractive endpoint but misses the AI-client requirements entirely."
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"direction": "better",
|
|
"rationale": "B directly satisfies the key reproducibility, held-out evaluation, multiclass metrics, and thread-safe PredictionEnginePool requirements. A appears to have a basic ML.NET implementation, but lacks evidence for seed/split correctness and uses a singleton PredictionEngine."
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"direction": "better",
|
|
"rationale": "Both appropriately stop at a plan and satisfy the core RAG, Blazor, PostgreSQL, and folder-based-document requirements. B is substantially more implementation-ready and aligned with the requested modern .NET AI abstractions, especially MEAI and VectorData/pgvector, while retaining sensible retrieval guardrails and citations."
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"direction": "better",
|
|
"rationale": "B satisfies every rubric item, particularly the required similarity threshold and Microsoft embedding abstraction, while providing a cohesive .NET-oriented design. A is also strong and more detailed on Markdown chunk handling, but misses those two explicit requirements. B's small pseudo-configuration/schema snippets are a minor tension with the no-code request, not enough to outweigh its stronger rubric coverage."
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"direction": "none",
|
|
"rationale": "Position-swap inconsistent (forward: B, reverse: A). Defaulting to tie."
|
|
}
|
|
],
|
|
"links": [
|
|
{
|
|
"label": "Skill source",
|
|
"url": "https://github.com/dotnet/skills/blob/4c72b17fa2c2aaa307d8f1e337ec75cab45ac5c7/plugins/dotnet-ai/skills/technology-selection/SKILL.md"
|
|
},
|
|
{
|
|
"label": "Eval source",
|
|
"url": "https://github.com/dotnet/skills/blob/4c72b17fa2c2aaa307d8f1e337ec75cab45ac5c7/tests/dotnet-ai/technology-selection/eval.yaml"
|
|
},
|
|
{
|
|
"label": "Commit",
|
|
"url": "https://github.com/dotnet/skills/commit/4c72b17fa2c2aaa307d8f1e337ec75cab45ac5c7"
|
|
}
|
|
]
|
|
}
|
|
],
|
|
"commit": {
|
|
"timestamp": "2026-09-12T00:01:18+00:00",
|
|
"id": "4c72b17fa2c2aaa307d8f1e337ec75cab45ac5c7",
|
|
"message": "Merge pull request #1154 from dotnet/abhitejjohn-agentic-workflow-repair",
|
|
"url": "https://github.com/dotnet/skills/commit/4c72b17fa2c2aaa307d8f1e337ec75cab45ac5c7",
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Abhitej John"
|
|
}
|
|
},
|
|
"date": 1789218680600,
|
|
"model": "claude-haiku-4.5",
|
|
"benches": [
|
|
{
|
|
"value": 2.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Quality"
|
|
},
|
|
{
|
|
"overfittingScore": 0.5,
|
|
"overfitting": "high",
|
|
"value": 9.166666984558105,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Quality"
|
|
},
|
|
{
|
|
"overfittingScore": 0.5,
|
|
"overfitting": "high",
|
|
"value": 6.666666507720947,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Quality"
|
|
},
|
|
{
|
|
"value": 5.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Quality"
|
|
},
|
|
{
|
|
"overfittingScore": 0.5,
|
|
"overfitting": "high",
|
|
"value": 3.5,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Quality"
|
|
},
|
|
{
|
|
"value": 3.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Quality"
|
|
},
|
|
{
|
|
"overfittingScore": 0.5,
|
|
"overfitting": "high",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Quality"
|
|
},
|
|
{
|
|
"overfittingScore": 0.5,
|
|
"overfitting": "high",
|
|
"value": 9.166666984558105,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Quality"
|
|
},
|
|
{
|
|
"value": 5.833333492279053,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Quality"
|
|
},
|
|
{
|
|
"overfittingScore": 0.5,
|
|
"overfitting": "high",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Quality"
|
|
},
|
|
{
|
|
"overfittingScore": 0.5,
|
|
"overfitting": "high",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Quality"
|
|
},
|
|
{
|
|
"value": 8.75,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Quality"
|
|
},
|
|
{
|
|
"overfittingScore": 0.5,
|
|
"overfitting": "high",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Quality"
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Quality"
|
|
}
|
|
],
|
|
"tool": "customBiggerIsBetter"
|
|
},
|
|
{
|
|
"verdictEvidence": [
|
|
{
|
|
"skillName": "technology-selection",
|
|
"skillKind": "invocable",
|
|
"state": "VALID_NO_CHANGE",
|
|
"stateReason": {
|
|
"code": "no_credible_preference_change",
|
|
"phase": "decision"
|
|
},
|
|
"passed": false,
|
|
"regressed": false,
|
|
"preferenceRegressed": false,
|
|
"reason": "Net win +66.7% (5W/0T/1L over 6 preference-eligible stimulus vote(s), sign test p=0.109), mean preference +36.7% across 6 paired run(s) — not credible (sign test p=0.109 > 0.05)",
|
|
"gateEvidence": {
|
|
"stimulusVoteCount": 6,
|
|
"wins": 5,
|
|
"ties": 0,
|
|
"losses": 1,
|
|
"discordant": 6,
|
|
"direction": "better",
|
|
"pValue": 0.10937500000000008,
|
|
"alpha": 0.05,
|
|
"netWin": 0.6666666666666666,
|
|
"minimumNetWin": 0.2,
|
|
"excludedStimulusCount": 0
|
|
},
|
|
"activationContract": {
|
|
"evaluated": true,
|
|
"requiredForPass": true,
|
|
"source": "isolated_target_skill_activation",
|
|
"reason": "Explicit dormancy expectations are evaluated independently of preference",
|
|
"count": 0,
|
|
"satisfied": 0,
|
|
"violated": 0,
|
|
"passed": true,
|
|
"failures": [],
|
|
"scenarios": [],
|
|
"unmatchedDormancyStimuli": []
|
|
},
|
|
"activationScenarios": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "missing-activation",
|
|
"plugin": "plugin-no-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "missing-activation",
|
|
"plugin": "plugin-no-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "missing-activation",
|
|
"plugin": "plugin-no-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
}
|
|
],
|
|
"judgeRationales": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"direction": "better",
|
|
"rationale": "Both responses failed to produce any code, instead punting back to the user without attempting the task. Neither satisfies any rubric criterion meaningfully. B is marginally better only because it explicitly names Microsoft Agent Framework, aligning slightly with the intended stack, whereas A stays generic. The difference is negligible; both are essentially failures."
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"direction": "better",
|
|
"rationale": "Neither response implemented the endpoint, so both fail every rubric criterion equally. However, Response A falsely claimed 'Implemented' while doing zero tool calls and never inspecting anything — a misleading and useless result. Response B actually investigated the project, hit a genuine tool failure (view failing despite glob finding files), and honestly reported being blocked, asking for the correct path rather than fabricating success. Response B's methodical approach and honest failure disclosure make it slightly better, though both ultimately delivered no working code."
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"direction": "better",
|
|
"rationale": "Both produce functionally similar ML.NET classifiers meeting most criteria (seed, train/test split, evaluation metrics, ML.NET). The decisive differences: B correctly uses PredictionEnginePool (thread-safe, the recommended ASP.NET Core pattern) satisfying criterion 5, while A uses a singleton PredictionEngine which is not thread-safe and fails that criterion. Additionally, B adds model persistence. However, A attempted to actually use the provided CSV and inspect the real project (though it was blocked by an environment file-write restriction and had to output code), whereas B fabricated its own training CSV rather than using the provided one, ignoring the actual project files. This partially offsets B's advantage. On balance, B is slightly better overall due to the PredictionEnginePool correctness which is an explicit rubric item, but not dramatically so given B ignored the provided data."
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"direction": "worse",
|
|
"rationale": "Both responses are very similar in quality: comprehensive, well-structured plans that correctly use Blazor, a file folder, Postgres, and RAG ingestion, both pausing before code. Neither names MEAI, VectorData.Abstractions, DataIngestion, or a concrete LLM provider, so both fall short on the more specific package rubric items equally. The differentiator is that A explicitly identifies pgvector as the vector storage mechanism in Postgres and mentions vector similarity indexes, making its plan more concrete on the vector-search requirement, whereas B leaves it as a vague 'Postgres-backed vector store adapter'. This gives A a slight edge overall."
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"direction": "better",
|
|
"rationale": "Both provide strong, well-structured RAG architecture plans without code. Response A is broader and more detailed across ingestion, operations, and rollout. However, the rubric specifically targets the Microsoft .NET AI stack, and Response B hits three technology-specific criteria (Microsoft.Extensions.VectorData, explicit similarity threshold, and Microsoft.Extensions.AI/IEmbeddingGenerator) that A misses or under-specifies. Since the task is .NET-specific and the rubric rewards these exact selections, B is slightly better overall despite A's greater breadth."
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"direction": "better",
|
|
"rationale": "Both correctly choose ML.NET and justify it, but the task explicitly asked to implement the feature. B delivers a complete, correct, production-quality implementation with training, evaluation, and a scoring endpoint, while A merely asks for the files without implementing anything. B fully satisfies the task; A does not."
|
|
}
|
|
],
|
|
"links": [
|
|
{
|
|
"label": "Skill source",
|
|
"url": "https://github.com/dotnet/skills/blob/4c72b17fa2c2aaa307d8f1e337ec75cab45ac5c7/plugins/dotnet-ai/skills/technology-selection/SKILL.md"
|
|
},
|
|
{
|
|
"label": "Eval source",
|
|
"url": "https://github.com/dotnet/skills/blob/4c72b17fa2c2aaa307d8f1e337ec75cab45ac5c7/tests/dotnet-ai/technology-selection/eval.yaml"
|
|
},
|
|
{
|
|
"label": "Commit",
|
|
"url": "https://github.com/dotnet/skills/commit/4c72b17fa2c2aaa307d8f1e337ec75cab45ac5c7"
|
|
}
|
|
]
|
|
}
|
|
],
|
|
"commit": {
|
|
"timestamp": "2026-09-12T00:01:18+00:00",
|
|
"id": "4c72b17fa2c2aaa307d8f1e337ec75cab45ac5c7",
|
|
"message": "Merge pull request #1154 from dotnet/abhitejjohn-agentic-workflow-repair",
|
|
"url": "https://github.com/dotnet/skills/commit/4c72b17fa2c2aaa307d8f1e337ec75cab45ac5c7",
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Abhitej John"
|
|
}
|
|
},
|
|
"date": 1789218680648,
|
|
"model": "gpt-5.3-codex",
|
|
"benches": [
|
|
{
|
|
"overfitting": "moderate",
|
|
"notActivated": true,
|
|
"value": 2.0,
|
|
"overfittingScore": 0.27,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Quality"
|
|
},
|
|
{
|
|
"overfittingScore": 0.27,
|
|
"overfitting": "moderate",
|
|
"value": 2.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Quality"
|
|
},
|
|
{
|
|
"value": 2.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Quality"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"notActivated": true,
|
|
"value": 5.0,
|
|
"overfittingScore": 0.27,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Quality"
|
|
},
|
|
{
|
|
"overfittingScore": 0.27,
|
|
"overfitting": "moderate",
|
|
"value": 5.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Quality"
|
|
},
|
|
{
|
|
"value": 5.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Quality"
|
|
},
|
|
{
|
|
"overfittingScore": 0.27,
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Quality"
|
|
},
|
|
{
|
|
"overfittingScore": 0.27,
|
|
"overfitting": "moderate",
|
|
"value": 2.5,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Quality"
|
|
},
|
|
{
|
|
"value": 7.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Quality"
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"notActivated": true,
|
|
"value": 5.0,
|
|
"overfittingScore": 0.27,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Quality"
|
|
},
|
|
{
|
|
"overfittingScore": 0.27,
|
|
"overfitting": "moderate",
|
|
"value": 9.166666984558105,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Quality"
|
|
},
|
|
{
|
|
"value": 5.833333492279053,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Quality"
|
|
},
|
|
{
|
|
"overfittingScore": 0.27,
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Quality"
|
|
},
|
|
{
|
|
"overfittingScore": 0.27,
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Quality"
|
|
},
|
|
{
|
|
"value": 9.375,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Quality"
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Quality"
|
|
},
|
|
{
|
|
"value": 8.333333015441895,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Quality"
|
|
},
|
|
{
|
|
"value": 7.5,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Quality"
|
|
}
|
|
],
|
|
"tool": "customBiggerIsBetter"
|
|
},
|
|
{
|
|
"benches": [
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Quality",
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Quality",
|
|
"value": 5.0,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Quality",
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Quality",
|
|
"value": 5.0,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Quality",
|
|
"overfitting": "moderate",
|
|
"value": 8.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"value": 8.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Quality",
|
|
"value": 3.0,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Quality",
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Quality",
|
|
"value": 9.166666984558105,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Quality",
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Quality",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Quality",
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Quality",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)"
|
|
}
|
|
],
|
|
"commit": {
|
|
"timestamp": "2026-09-12T07:43:41+00:00",
|
|
"url": "https://github.com/dotnet/skills/commit/4ecd7d9c76fa458807684771c2bfc7acf1e00ad3",
|
|
"message": "Merge pull request #1134 from dotnet/bot/weekly-version-sync",
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Abhitej John"
|
|
},
|
|
"id": "4ecd7d9c76fa458807684771c2bfc7acf1e00ad3"
|
|
},
|
|
"date": 1789318508154,
|
|
"model": "claude-opus-5",
|
|
"verdictEvidence": [
|
|
{
|
|
"skillName": "technology-selection",
|
|
"skillKind": "invocable",
|
|
"state": "VALID_PASS",
|
|
"stateReason": {
|
|
"code": "credible_preference_improvement",
|
|
"phase": "decision"
|
|
},
|
|
"passed": true,
|
|
"regressed": false,
|
|
"preferenceRegressed": false,
|
|
"reason": "Net win +100.0% (6W/0T/0L over 6 preference-eligible stimulus vote(s), sign test p=0.016), mean preference +60.0% across 6 paired run(s) — credibly better",
|
|
"gateEvidence": {
|
|
"stimulusVoteCount": 6,
|
|
"wins": 6,
|
|
"ties": 0,
|
|
"losses": 0,
|
|
"discordant": 6,
|
|
"direction": "better",
|
|
"pValue": 0.015625000000000007,
|
|
"alpha": 0.05,
|
|
"netWin": 1,
|
|
"minimumNetWin": 0.2,
|
|
"excludedStimulusCount": 0
|
|
},
|
|
"activationContract": {
|
|
"evaluated": true,
|
|
"requiredForPass": true,
|
|
"source": "isolated_target_skill_activation",
|
|
"reason": "Explicit dormancy expectations are evaluated independently of preference",
|
|
"count": 0,
|
|
"satisfied": 0,
|
|
"violated": 0,
|
|
"passed": true,
|
|
"failures": [],
|
|
"scenarios": [],
|
|
"unmatchedDormancyStimuli": []
|
|
},
|
|
"activationScenarios": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
}
|
|
],
|
|
"judgeRationales": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"direction": "better",
|
|
"rationale": "Both runs appear to produce a functioning, tested .NET 10 research agent with search, notes, and summaries. B is stronger on the requested production guardrails and rubric-specific details: explicit Extensions.AI foundation, loop cap, observability, and token-budget enforcement. The core implementation quality gap is meaningful but not overwhelming because A also appears complete and end-to-end verified."
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"direction": "better",
|
|
"rationale": "B directly satisfies every specified AI integration rubric item while still adding and validating an endpoint with input guardrails and resilience. A appears to deliver a working local extractive endpoint, but it deliberately substitutes a non-LLM implementation and consequently misses all six required integration criteria."
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"direction": "better",
|
|
"rationale": "B satisfies every explicit rubric item, including seeded reproducibility, held-out evaluation, and PredictionEnginePool-based serving. A is a plausible and candid ML.NET implementation, but misses the required holdout-split and pool-based endpoint design and does not demonstrate a seed."
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"direction": "better",
|
|
"rationale": "Both are strong, appropriately stop at a plan, and give practical RAG architectures. B is slightly stronger overall because it directly follows the specified MEAI VectorData abstraction requirement and adds useful operational safeguards (thresholding, retries/timeouts, secret handling, and privacy-conscious logging). A is more concrete about package-level implementation and is a close alternative."
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"direction": "better",
|
|
"rationale": "Both are concise, code-free, and solid RAG architecture plans. B is modestly stronger because it gives more implementation-relevant architectural detail around chunk integrity, persistent incremental embedding, threshold fallback behavior, and auditable source attribution without losing focus."
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"direction": "better",
|
|
"rationale": "Both are high-quality, working ML.NET solutions with appropriate caveats about the tiny dataset. B has a small quality advantage through more production-oriented scoring integration and more rigorous evaluation handling, while A remains correct and directly verified."
|
|
}
|
|
],
|
|
"links": [
|
|
{
|
|
"label": "Skill source",
|
|
"url": "https://github.com/dotnet/skills/blob/4ecd7d9c76fa458807684771c2bfc7acf1e00ad3/plugins/dotnet-ai/skills/technology-selection/SKILL.md"
|
|
},
|
|
{
|
|
"label": "Eval source",
|
|
"url": "https://github.com/dotnet/skills/blob/4ecd7d9c76fa458807684771c2bfc7acf1e00ad3/tests/dotnet-ai/technology-selection/eval.yaml"
|
|
},
|
|
{
|
|
"label": "Commit",
|
|
"url": "https://github.com/dotnet/skills/commit/4ecd7d9c76fa458807684771c2bfc7acf1e00ad3"
|
|
}
|
|
]
|
|
}
|
|
],
|
|
"tool": "customBiggerIsBetter"
|
|
},
|
|
{
|
|
"benches": [
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Quality",
|
|
"overfitting": "moderate",
|
|
"value": 7.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"value": 9.5,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Quality",
|
|
"value": 4.5,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Quality",
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"value": 9.583333015441895,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Quality",
|
|
"value": 5.0,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Quality",
|
|
"overfitting": "moderate",
|
|
"value": 8.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"value": 6.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Quality",
|
|
"value": 3.0,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Quality",
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"value": 9.166666984558105,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Quality",
|
|
"value": 9.166666984558105,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Quality",
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Quality",
|
|
"value": 9.375,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Quality",
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Quality",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)"
|
|
}
|
|
],
|
|
"commit": {
|
|
"timestamp": "2026-09-12T07:43:41+00:00",
|
|
"url": "https://github.com/dotnet/skills/commit/4ecd7d9c76fa458807684771c2bfc7acf1e00ad3",
|
|
"message": "Merge pull request #1134 from dotnet/bot/weekly-version-sync",
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Abhitej John"
|
|
},
|
|
"id": "4ecd7d9c76fa458807684771c2bfc7acf1e00ad3"
|
|
},
|
|
"date": 1789318508226,
|
|
"model": "claude-sonnet-5",
|
|
"verdictEvidence": [
|
|
{
|
|
"skillName": "technology-selection",
|
|
"skillKind": "invocable",
|
|
"state": "VALID_PASS",
|
|
"stateReason": {
|
|
"code": "credible_preference_improvement",
|
|
"phase": "decision"
|
|
},
|
|
"passed": true,
|
|
"regressed": false,
|
|
"preferenceRegressed": false,
|
|
"reason": "Net win +100.0% (6W/0T/0L over 6 preference-eligible stimulus vote(s), sign test p=0.016), mean preference +50.0% across 6 paired run(s) — credibly better",
|
|
"gateEvidence": {
|
|
"stimulusVoteCount": 6,
|
|
"wins": 6,
|
|
"ties": 0,
|
|
"losses": 0,
|
|
"discordant": 6,
|
|
"direction": "better",
|
|
"pValue": 0.015625000000000007,
|
|
"alpha": 0.05,
|
|
"netWin": 1,
|
|
"minimumNetWin": 0.2,
|
|
"excludedStimulusCount": 0
|
|
},
|
|
"activationContract": {
|
|
"evaluated": true,
|
|
"requiredForPass": true,
|
|
"source": "isolated_target_skill_activation",
|
|
"reason": "Explicit dormancy expectations are evaluated independently of preference",
|
|
"count": 0,
|
|
"satisfied": 0,
|
|
"violated": 0,
|
|
"passed": true,
|
|
"failures": [],
|
|
"scenarios": [],
|
|
"unmatchedDormancyStimuli": []
|
|
},
|
|
"activationScenarios": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
}
|
|
],
|
|
"judgeRationales": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"direction": "better",
|
|
"rationale": "B meets the central architectural requirement by using Microsoft Agent Framework over IChatClient and adds execution/usage guardrails and observability. A builds and has functional tools, but its explicit custom agent loop misses the primary orchestration requirement. [Magnitude reduced to \"slightly-better\": the position-swapped pass judged this \"slightly-better\", so the more confident \"much-better\" was downgraded.]"
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"direction": "better",
|
|
"rationale": "B satisfies every specified AI-integration criterion, builds successfully, and adds validation and model-failure handling. A provides a verified standalone endpoint, but its non-LLM extractive implementation misses all of the explicit Microsoft.Extensions.AI, client registration, options, resilience, credential, and model-version requirements."
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"direction": "better",
|
|
"rationale": "Both runs built and live-tested ML.NET classifiers with held-out metrics and prediction APIs, but B clearly meets the reproducibility and PredictionEnginePool requirements. A has a useful metrics endpoint, but that does not offset the lack of evidence for the explicitly required pool."
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"direction": "better",
|
|
"rationale": "Both are strong, appropriately scoped plans, but B better satisfies the rubric's explicit MEAI and VectorData.Abstractions requirements while preserving the requested Blazor, Postgres, and folder-based RAG design. A is slightly more concrete and dependable on conventional packages, but misses those specified abstraction choices."
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"direction": "better",
|
|
"rationale": "Both are strong, code-free architecture plans covering ingestion, semantic chunking, vector retrieval, grounded generation, and citations. B is modestly better because it explicitly prevents low-relevance retrieval from reaching the model and defines a grounded fallback; this is an important correctness safeguard for a documentation Q&A system."
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"direction": "better",
|
|
"rationale": "Both are successful, appropriate ML.NET implementations with verified endpoints. B is modestly stronger in production-oriented inference registration and in avoiding a potentially misleading in-sample metric on the tiny dataset."
|
|
}
|
|
],
|
|
"links": [
|
|
{
|
|
"label": "Skill source",
|
|
"url": "https://github.com/dotnet/skills/blob/4ecd7d9c76fa458807684771c2bfc7acf1e00ad3/plugins/dotnet-ai/skills/technology-selection/SKILL.md"
|
|
},
|
|
{
|
|
"label": "Eval source",
|
|
"url": "https://github.com/dotnet/skills/blob/4ecd7d9c76fa458807684771c2bfc7acf1e00ad3/tests/dotnet-ai/technology-selection/eval.yaml"
|
|
},
|
|
{
|
|
"label": "Commit",
|
|
"url": "https://github.com/dotnet/skills/commit/4ecd7d9c76fa458807684771c2bfc7acf1e00ad3"
|
|
}
|
|
]
|
|
}
|
|
],
|
|
"tool": "customBiggerIsBetter"
|
|
},
|
|
{
|
|
"benches": [
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Quality",
|
|
"overfitting": "moderate",
|
|
"value": 3.5,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.25
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"value": 3.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.25
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Quality",
|
|
"value": 3.0,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Quality",
|
|
"overfitting": "moderate",
|
|
"value": 7.916666507720947,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.25
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"value": 7.5,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.25
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Quality",
|
|
"value": 5.0,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Quality",
|
|
"overfitting": "moderate",
|
|
"value": 4.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.25
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"value": 4.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.25
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Quality",
|
|
"value": 3.0,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Quality",
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.25
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"value": 9.166666984558105,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.25
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Quality",
|
|
"value": 5.0,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Quality",
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.25
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.25
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Quality",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Quality",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Quality",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Quality",
|
|
"value": 9.166666984558105,
|
|
"unit": "Score (0-10)"
|
|
}
|
|
],
|
|
"commit": {
|
|
"timestamp": "2026-09-12T07:43:41+00:00",
|
|
"url": "https://github.com/dotnet/skills/commit/4ecd7d9c76fa458807684771c2bfc7acf1e00ad3",
|
|
"message": "Merge pull request #1134 from dotnet/bot/weekly-version-sync",
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Abhitej John"
|
|
},
|
|
"id": "4ecd7d9c76fa458807684771c2bfc7acf1e00ad3"
|
|
},
|
|
"date": 1789318508294,
|
|
"model": "gpt-5.6-sol",
|
|
"verdictEvidence": [
|
|
{
|
|
"skillName": "technology-selection",
|
|
"skillKind": "invocable",
|
|
"state": "VALID_PASS",
|
|
"stateReason": {
|
|
"code": "credible_preference_improvement",
|
|
"phase": "decision"
|
|
},
|
|
"passed": true,
|
|
"regressed": false,
|
|
"preferenceRegressed": false,
|
|
"reason": "Net win +100.0% (6W/0T/0L over 6 preference-eligible stimulus vote(s), sign test p=0.016), mean preference +50.0% across 6 paired run(s) — credibly better",
|
|
"gateEvidence": {
|
|
"stimulusVoteCount": 6,
|
|
"wins": 6,
|
|
"ties": 0,
|
|
"losses": 0,
|
|
"discordant": 6,
|
|
"direction": "better",
|
|
"pValue": 0.015625000000000007,
|
|
"alpha": 0.05,
|
|
"netWin": 1,
|
|
"minimumNetWin": 0.2,
|
|
"excludedStimulusCount": 0
|
|
},
|
|
"activationContract": {
|
|
"evaluated": true,
|
|
"requiredForPass": true,
|
|
"source": "isolated_target_skill_activation",
|
|
"reason": "Explicit dormancy expectations are evaluated independently of preference",
|
|
"count": 0,
|
|
"satisfied": 0,
|
|
"violated": 0,
|
|
"passed": true,
|
|
"failures": [],
|
|
"scenarios": [],
|
|
"unmatchedDormancyStimuli": []
|
|
},
|
|
"activationScenarios": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
}
|
|
],
|
|
"judgeRationales": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"direction": "better",
|
|
"rationale": "Both produced compiling .NET 10 console apps using Microsoft Agent Framework over IChatClient with web search and note-taking tools. B is stronger on several rubric-specific points: it explicitly caps iterations (MaximumIterationsPerRequest), adds logging infrastructure, and used a technology-selection skill to follow the agentic pattern (secure config via options binding/validation). A's approach was solid and it used OpenAI hosted web search cleanly, but it lacks visible iteration capping and step logging. Notably, B actually attempted to run the app (exiting on usage message), showing more verification. B edges ahead on the specific rubric criteria that differentiate quality."
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"direction": "better",
|
|
"rationale": "The rubric is entirely oriented around an AI/LLM-based summarization implementation using Microsoft.Extensions.AI IChatClient with DI, ChatOptions, retries, config-based secrets, and model versioning. Response B built exactly this: an Azure OpenAI IChatClient integration with structured output, validation, retries, timeout handling, and secure credential-based auth loaded from configuration. Response A instead built a trivial non-AI sentence-truncation summarizer that satisfies none of the rubric criteria. While A's endpoint works and returns something, it fundamentally misinterprets 'summarization' as real AI summarization and fails all six criteria. B is much better across the board."
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"direction": "better",
|
|
"rationale": "Both delivered working ML.NET classifiers that build cleanly with prediction and metrics endpoints. B more precisely satisfies the rubric: explicit fixed seed, canonical micro/macro accuracy and LogLoss metrics, model persistence, and notably PredictionEnginePool (a key rubric criterion A appears to miss with its singleton service). B also actually ran the API and verified predictions/metrics end-to-end, and honestly disclosed the unrepresentative metrics from the tiny dataset. A used cross-validation (reasonable for tiny data) but does not clearly use PredictionEnginePool or set a seed, and did not run the app to verify. B is slightly better overall, driven mainly by the PredictionEnginePool criterion."
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"direction": "better",
|
|
"rationale": "Both responses deliver strong, well-structured plans that honor the user's request for Blazor, Postgres, and a file folder before writing code. However, B better satisfies the specific technology-selection rubric criteria by explicitly naming Microsoft.Extensions.AI, concrete LLM providers, and Microsoft.Extensions.VectorData with pgvector, whereas A stays generic ('provider-neutral abstraction'). B's specificity on MEAI and VectorData gives it the edge, though A's plan is otherwise comparable in quality and coverage."
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"direction": "better",
|
|
"rationale": "Both responses are strong, well-structured, code-free architecture plans covering chunking, retrieval, thresholds, citations, and delivery plans. B edges ahead by explicitly naming the specific Microsoft abstractions (IEmbeddingGenerator, VectorData.Abstractions, DataIngestion), explicitly stating embedding caching, and including prompt-injection defense (treating docs as untrusted data). A explored the repo but the extra tool calls didn't materially improve the output. The differences are modest, so B is only slightly better overall.\""
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"direction": "better",
|
|
"rationale": "Both responses correctly select ML.NET and implement CSV-based binary classification with sound rationale. B is more thorough: it added held-out evaluation with metrics, model persistence, thread-safe PredictionEnginePool, and — importantly — actually ran the API and verified a live prediction, while A only confirmed a successful build. The gap is meaningful but not large since both meet the core rubric criteria."
|
|
}
|
|
],
|
|
"links": [
|
|
{
|
|
"label": "Skill source",
|
|
"url": "https://github.com/dotnet/skills/blob/4ecd7d9c76fa458807684771c2bfc7acf1e00ad3/plugins/dotnet-ai/skills/technology-selection/SKILL.md"
|
|
},
|
|
{
|
|
"label": "Eval source",
|
|
"url": "https://github.com/dotnet/skills/blob/4ecd7d9c76fa458807684771c2bfc7acf1e00ad3/tests/dotnet-ai/technology-selection/eval.yaml"
|
|
},
|
|
{
|
|
"label": "Commit",
|
|
"url": "https://github.com/dotnet/skills/commit/4ecd7d9c76fa458807684771c2bfc7acf1e00ad3"
|
|
}
|
|
]
|
|
}
|
|
],
|
|
"tool": "customBiggerIsBetter"
|
|
},
|
|
{
|
|
"verdictEvidence": [
|
|
{
|
|
"skillName": "technology-selection",
|
|
"skillKind": "invocable",
|
|
"state": "VALID_PASS",
|
|
"stateReason": {
|
|
"code": "credible_preference_improvement",
|
|
"phase": "decision"
|
|
},
|
|
"passed": true,
|
|
"regressed": false,
|
|
"preferenceRegressed": false,
|
|
"reason": "Net win +83.3% (5W/1T/0L over 6 preference-eligible stimulus vote(s), sign test p=0.031), mean preference +53.3% across 6 paired run(s) — credibly better",
|
|
"gateEvidence": {
|
|
"stimulusVoteCount": 6,
|
|
"wins": 5,
|
|
"ties": 1,
|
|
"losses": 0,
|
|
"discordant": 5,
|
|
"direction": "better",
|
|
"pValue": 0.03125,
|
|
"alpha": 0.05,
|
|
"netWin": 0.8333333333333334,
|
|
"minimumNetWin": 0.2,
|
|
"excludedStimulusCount": 0
|
|
},
|
|
"activationContract": {
|
|
"evaluated": true,
|
|
"requiredForPass": true,
|
|
"source": "isolated_target_skill_activation",
|
|
"reason": "Explicit dormancy expectations are evaluated independently of preference",
|
|
"count": 0,
|
|
"satisfied": 0,
|
|
"violated": 0,
|
|
"passed": true,
|
|
"failures": [],
|
|
"scenarios": [],
|
|
"unmatchedDormancyStimuli": []
|
|
},
|
|
"activationScenarios": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
}
|
|
],
|
|
"judgeRationales": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"direction": "better",
|
|
"rationale": "B meets the central framework requirement by using Microsoft Agent Framework atop IChatClient and also adds a token bound and timeout. Both build successfully and provide the required search/note capabilities, but A misses the specifically required orchestration layer. Neither clearly implements MaximumIterations or full tool-step observability. [Magnitude reduced to \"slightly-better\": the position-swapped pass judged this \"slightly-better\", so the more confident \"much-better\" was downgraded.]"
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"direction": "better",
|
|
"rationale": "B addresses every specified AI-integration criterion with an MEAI/DI-based, configured, resilient implementation. A delivers a functional dependency-free summarizer, but it misses the rubric's required AI client architecture and all associated configuration requirements."
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"direction": "better",
|
|
"rationale": "B meets every stated rubric requirement, including reproducibility, held-out evaluation, and safe pooled prediction engines. A is functional and has sensible cross-validation and metrics, but misses the explicit seed/train-test/pooling requirements."
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"direction": "better",
|
|
"rationale": "Both are good concise plans and honor the request to pause before implementation. B is stronger because it directly satisfies the MEAI and Microsoft.Extensions.VectorData.Abstractions requirements and adds useful citation, similarity-threshold, caching, and resilience considerations. A remains a viable, more conventional implementation plan but misses those specified abstractions."
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"direction": "none",
|
|
"rationale": "Position-swap inconsistent (forward: tie, reverse: B). Defaulting to tie."
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"direction": "better",
|
|
"rationale": "Both are correct, working ML.NET churn implementations with appropriate caveats for the five-row dataset. B has a modest quality edge in evaluation robustness and production-oriented prediction serving, while A remains a solid and verified solution."
|
|
}
|
|
],
|
|
"links": [
|
|
{
|
|
"label": "Skill source",
|
|
"url": "https://github.com/dotnet/skills/blob/4ecd7d9c76fa458807684771c2bfc7acf1e00ad3/plugins/dotnet-ai/skills/technology-selection/SKILL.md"
|
|
},
|
|
{
|
|
"label": "Eval source",
|
|
"url": "https://github.com/dotnet/skills/blob/4ecd7d9c76fa458807684771c2bfc7acf1e00ad3/tests/dotnet-ai/technology-selection/eval.yaml"
|
|
},
|
|
{
|
|
"label": "Commit",
|
|
"url": "https://github.com/dotnet/skills/commit/4ecd7d9c76fa458807684771c2bfc7acf1e00ad3"
|
|
}
|
|
]
|
|
}
|
|
],
|
|
"benches": [
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.47,
|
|
"value": 7.0,
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Quality",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.47,
|
|
"value": 7.0,
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Quality",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"value": 4.5,
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Quality"
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.47,
|
|
"value": 10.0,
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Quality",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.47,
|
|
"value": 6.666666507720947,
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Quality",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"value": 5.0,
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Quality"
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.47,
|
|
"value": 8.0,
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Quality",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.47,
|
|
"value": 8.0,
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Quality",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"value": 2.5,
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Quality"
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.47,
|
|
"value": 10.0,
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Quality",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.47,
|
|
"value": 10.0,
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Quality",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"value": 5.833333492279053,
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Quality"
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.47,
|
|
"value": 10.0,
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Quality",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.47,
|
|
"value": 10.0,
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Quality",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"value": 10.0,
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Quality"
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.47,
|
|
"value": 10.0,
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Quality",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.47,
|
|
"value": 10.0,
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Quality",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"value": 10.0,
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Quality"
|
|
}
|
|
],
|
|
"model": "claude-sonnet-5",
|
|
"date": 1789397101175,
|
|
"tool": "customBiggerIsBetter",
|
|
"commit": {
|
|
"url": "https://github.com/dotnet/skills/commit/4ecd7d9c76fa458807684771c2bfc7acf1e00ad3",
|
|
"timestamp": "2026-09-12T07:43:41+00:00",
|
|
"id": "4ecd7d9c76fa458807684771c2bfc7acf1e00ad3",
|
|
"message": "Merge pull request #1134 from dotnet/bot/weekly-version-sync",
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Abhitej John"
|
|
}
|
|
}
|
|
},
|
|
{
|
|
"verdictEvidence": [
|
|
{
|
|
"skillName": "technology-selection",
|
|
"skillKind": "invocable",
|
|
"state": "VALID_PASS",
|
|
"stateReason": {
|
|
"code": "credible_preference_improvement",
|
|
"phase": "decision"
|
|
},
|
|
"passed": true,
|
|
"regressed": false,
|
|
"preferenceRegressed": false,
|
|
"reason": "Net win +100.0% (6W/0T/0L over 6 preference-eligible stimulus vote(s), sign test p=0.016), mean preference +50.0% across 6 paired run(s) — credibly better",
|
|
"gateEvidence": {
|
|
"stimulusVoteCount": 6,
|
|
"wins": 6,
|
|
"ties": 0,
|
|
"losses": 0,
|
|
"discordant": 6,
|
|
"direction": "better",
|
|
"pValue": 0.015625000000000007,
|
|
"alpha": 0.05,
|
|
"netWin": 1,
|
|
"minimumNetWin": 0.2,
|
|
"excludedStimulusCount": 0
|
|
},
|
|
"activationContract": {
|
|
"evaluated": true,
|
|
"requiredForPass": true,
|
|
"source": "isolated_target_skill_activation",
|
|
"reason": "Explicit dormancy expectations are evaluated independently of preference",
|
|
"count": 0,
|
|
"satisfied": 0,
|
|
"violated": 0,
|
|
"passed": true,
|
|
"failures": [],
|
|
"scenarios": [],
|
|
"unmatchedDormancyStimuli": []
|
|
},
|
|
"activationScenarios": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"plugin": "plugin-activity-observed"
|
|
}
|
|
],
|
|
"judgeRationales": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"direction": "better",
|
|
"rationale": "Both responses successfully build a working .NET 10 console app using Microsoft.Agents.AI on top of Microsoft.Extensions.AI with web search and note-taking tools, and both compile successfully. B edges ahead by explicitly bounding the agentic loop to 8 iterations (satisfying the MaximumIterations criterion), whereas A shows no evidence of capping. B also demonstrated a more methodical process—inspecting the actual ChatClientAgent constructor signature and recovering from build errors deliberately—while A struggled with path resolution and got somewhat lucky. Neither implemented observability logging or a token/cost ceiling. The gap is modest but real, favoring B.\""
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"direction": "better",
|
|
"rationale": "The rubric is entirely oriented around a proper Microsoft.Extensions.AI / IChatClient LLM integration. Response B implemented exactly this: IChatClient abstraction, DI registration, retries, configuration-based key loading, and a dated model version, all verified to build. Response A instead built a naive extractive 'summarizer' that literally returns the input text unchanged (its smoke test echoed the full document), satisfying none of the rubric criteria. While Response A did verify its build and run a smoke test methodically, its output does not actually summarize and misses the entire intended architecture. Response B is much better across every criterion."
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"direction": "better",
|
|
"rationale": "Both responses successfully implemented an ML.NET-based ticket classifier with training, metrics, and a prediction endpoint, and both verified working endpoints via curl. However, B more directly satisfies several rubric criteria: it uses a proper held-out train/test split, an explicitly deterministic (seeded) pipeline, and critically implements a thread-safe PredictionEnginePool with input validation. A used cross-validation instead of a held-out split and showed no evidence of a seed or PredictionEnginePool. A's metrics (accuracy 1.0) come from tiny CV folds; B's held-out metrics are more honest given the 6-row dataset. B is the stronger, more idiomatic ASP.NET Core implementation, though A is a competent solution too."
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"direction": "better",
|
|
"rationale": "Both responses deliver a strong, well-structured plan before implementation, use Blazor, read from a file folder, and use pgvector in PostgreSQL. B is better aligned with the rubric's technology expectations: it explicitly selects Microsoft.Extensions.AI with IChatClient and a configurable concrete provider, correctly frames it as a RAG/language-generation workload rather than ML.NET, and adds valuable security/operational detail. A is solid but stays generic on the LLM abstraction. The gap is meaningful on the MEAI criterion but the overall plans are close in quality, so B is slightly better overall.\""
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"direction": "better",
|
|
"rationale": "Both responses are strong, well-organized plans meeting most criteria (chunking, threshold, attribution). B edges ahead by explicitly selecting the Microsoft.Extensions.AI/VectorData stack the rubric favors and by more clearly specifying embedding caching keyed on content hash + model version. A is excellent and comprehensive but relies on more generic component choices (Semantic Kernel 'or equivalent') and is less explicit about embedding caching. The gap is modest."
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"direction": "better",
|
|
"rationale": "Both responses correctly selected ML.NET, avoided LLMs, gave sound justifications, and produced working verified /predict endpoints. B is slightly more complete: it adds a held-out test split with evaluation metrics, an /at-risk endpoint, proper label conversion, and thread-safe PredictionEnginePool serving. The metrics B reported (accuracy 0.5, f1 0) suggest a weak model, but this is likely due to tiny dataset rather than a code flaw. Overall B's implementation is more rigorous and feature-complete, giving it a slight edge."
|
|
}
|
|
],
|
|
"links": [
|
|
{
|
|
"label": "Skill source",
|
|
"url": "https://github.com/dotnet/skills/blob/4ecd7d9c76fa458807684771c2bfc7acf1e00ad3/plugins/dotnet-ai/skills/technology-selection/SKILL.md"
|
|
},
|
|
{
|
|
"label": "Eval source",
|
|
"url": "https://github.com/dotnet/skills/blob/4ecd7d9c76fa458807684771c2bfc7acf1e00ad3/tests/dotnet-ai/technology-selection/eval.yaml"
|
|
},
|
|
{
|
|
"label": "Commit",
|
|
"url": "https://github.com/dotnet/skills/commit/4ecd7d9c76fa458807684771c2bfc7acf1e00ad3"
|
|
}
|
|
]
|
|
}
|
|
],
|
|
"benches": [
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.28,
|
|
"value": 3.0,
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Quality",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.28,
|
|
"value": 3.5,
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Quality",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"value": 2.5,
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Quality"
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.28,
|
|
"value": 7.916666507720947,
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Quality",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.28,
|
|
"value": 7.916666507720947,
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Quality",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"value": 5.0,
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Quality"
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.28,
|
|
"value": 4.0,
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Quality",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.28,
|
|
"value": 4.0,
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Quality",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"value": 3.0,
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Quality"
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.28,
|
|
"value": 10.0,
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Quality",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.28,
|
|
"value": 10.0,
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Quality",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"value": 5.0,
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Quality"
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.28,
|
|
"value": 10.0,
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Quality",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.28,
|
|
"value": 10.0,
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Quality",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"value": 9.375,
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Quality"
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.28,
|
|
"value": 10.0,
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Quality",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.28,
|
|
"value": 10.0,
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Quality",
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"value": 10.0,
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Quality"
|
|
}
|
|
],
|
|
"model": "gpt-5.6-luna",
|
|
"date": 1789397101227,
|
|
"tool": "customBiggerIsBetter",
|
|
"commit": {
|
|
"url": "https://github.com/dotnet/skills/commit/4ecd7d9c76fa458807684771c2bfc7acf1e00ad3",
|
|
"timestamp": "2026-09-12T07:43:41+00:00",
|
|
"id": "4ecd7d9c76fa458807684771c2bfc7acf1e00ad3",
|
|
"message": "Merge pull request #1134 from dotnet/bot/weekly-version-sync",
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Abhitej John"
|
|
}
|
|
}
|
|
},
|
|
{
|
|
"tool": "customBiggerIsBetter",
|
|
"commit": {
|
|
"url": "https://github.com/dotnet/skills/commit/24f7cfbd42ad7bf52bcd67372816b982c38c64c6",
|
|
"author": {
|
|
"name": "Amaury Levé",
|
|
"username": "AbhitejJohn"
|
|
},
|
|
"timestamp": "2026-09-14T15:27:40+00:00",
|
|
"id": "24f7cfbd42ad7bf52bcd67372816b982c38c64c6",
|
|
"message": "Surface activation-only evaluation failures (#1163)"
|
|
},
|
|
"date": 1789478285332,
|
|
"verdictEvidence": [
|
|
{
|
|
"skillName": "technology-selection",
|
|
"skillKind": "invocable",
|
|
"state": "INVALID_INCONCLUSIVE",
|
|
"stateReason": {
|
|
"code": "unmatched_trajectories",
|
|
"phase": "comparison_pairing"
|
|
},
|
|
"passed": false,
|
|
"regressed": false,
|
|
"preferenceRegressed": false,
|
|
"reason": "Net win +40.0% (2W/3T/0L over 5 preference-eligible stimulus vote(s), sign test p=0.250), mean preference +16.0% across 5 paired run(s), 1 unmatched — inconclusive (unmatched trajectories)",
|
|
"gateEvidence": {
|
|
"stimulusVoteCount": 5,
|
|
"wins": 2,
|
|
"ties": 3,
|
|
"losses": 0,
|
|
"discordant": 2,
|
|
"direction": "better",
|
|
"pValue": 0.25,
|
|
"alpha": 0.05,
|
|
"netWin": 0.4,
|
|
"minimumNetWin": 0.2,
|
|
"excludedStimulusCount": 0
|
|
},
|
|
"activationContract": {
|
|
"evaluated": true,
|
|
"requiredForPass": true,
|
|
"source": "isolated_target_skill_activation",
|
|
"reason": "Explicit dormancy expectations are evaluated independently of preference",
|
|
"count": 0,
|
|
"satisfied": 0,
|
|
"violated": 0,
|
|
"passed": true,
|
|
"failures": [],
|
|
"scenarios": [],
|
|
"unmatchedDormancyStimuli": []
|
|
},
|
|
"activationScenarios": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "missing-activation",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-no-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "missing-activation",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-no-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "missing-activation",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "missing-activation",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-no-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0
|
|
}
|
|
],
|
|
"judgeRationales": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"direction": "none",
|
|
"rationale": "Position-swap inconsistent (forward: A, reverse: B). Defaulting to tie."
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"direction": "none",
|
|
"rationale": "Position-swap inconsistent (forward: B, reverse: tie). Defaulting to tie."
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"direction": "better",
|
|
"rationale": "B more clearly satisfies the core ML.NET reproducibility, held-out evaluation, metric reporting, and thread-safe serving requirements, and its timeline shows a successful build and exercised endpoint. However, neither final answer actually shows the requested full implementation, and B's observed endpoint returns all-zero score values and appears to replace rather than plainly use the provided CSV, so the advantage is not overwhelming."
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"direction": "none",
|
|
"rationale": "Position-swap inconsistent (forward: A, reverse: B). Defaulting to tie."
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"direction": "better",
|
|
"rationale": "B more directly fulfills every stated rubric item and is especially well aligned with the requested .NET abstraction stack, while still providing a coherent ingestion/query architecture. A is strong and more detailed in several operational areas, and better obeys the no-code constraint; however, it misses the explicit Microsoft.Extensions.AI embedding abstraction. B's small pseudocode DI snippet is a minor violation of the no-code request, preventing a larger advantage."
|
|
}
|
|
],
|
|
"links": [
|
|
{
|
|
"label": "Skill source",
|
|
"url": "https://github.com/dotnet/skills/blob/24f7cfbd42ad7bf52bcd67372816b982c38c64c6/plugins/dotnet-ai/skills/technology-selection/SKILL.md"
|
|
},
|
|
{
|
|
"label": "Eval source",
|
|
"url": "https://github.com/dotnet/skills/blob/24f7cfbd42ad7bf52bcd67372816b982c38c64c6/tests/dotnet-ai/technology-selection/eval.yaml"
|
|
},
|
|
{
|
|
"label": "Commit",
|
|
"url": "https://github.com/dotnet/skills/commit/24f7cfbd42ad7bf52bcd67372816b982c38c64c6"
|
|
}
|
|
]
|
|
}
|
|
],
|
|
"benches": [
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Quality",
|
|
"notActivated": true,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48,
|
|
"unit": "Score (0-10)",
|
|
"value": 4.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Quality",
|
|
"value": 2.5
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Quality",
|
|
"notActivated": true,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48,
|
|
"unit": "Score (0-10)",
|
|
"value": 5.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Quality",
|
|
"value": 10.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Quality",
|
|
"value": 5.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Quality",
|
|
"value": 3.5
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Quality",
|
|
"value": 3.0
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Quality",
|
|
"notActivated": true,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48,
|
|
"unit": "Score (0-10)",
|
|
"value": 9.166666984558105
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Quality",
|
|
"value": 10.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Quality",
|
|
"value": 5.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Quality",
|
|
"value": 10.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Quality",
|
|
"value": 9.375
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Quality",
|
|
"value": 9.375
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Quality",
|
|
"value": 10.0
|
|
}
|
|
],
|
|
"model": "claude-haiku-4.5"
|
|
},
|
|
{
|
|
"tool": "customBiggerIsBetter",
|
|
"commit": {
|
|
"url": "https://github.com/dotnet/skills/commit/24f7cfbd42ad7bf52bcd67372816b982c38c64c6",
|
|
"author": {
|
|
"name": "Amaury Levé",
|
|
"username": "AbhitejJohn"
|
|
},
|
|
"timestamp": "2026-09-14T15:27:40+00:00",
|
|
"id": "24f7cfbd42ad7bf52bcd67372816b982c38c64c6",
|
|
"message": "Surface activation-only evaluation failures (#1163)"
|
|
},
|
|
"date": 1789478285409,
|
|
"verdictEvidence": [
|
|
{
|
|
"skillName": "technology-selection",
|
|
"skillKind": "invocable",
|
|
"state": "VALID_PASS",
|
|
"stateReason": {
|
|
"code": "credible_preference_improvement",
|
|
"phase": "decision"
|
|
},
|
|
"passed": true,
|
|
"regressed": false,
|
|
"preferenceRegressed": false,
|
|
"reason": "Net win +83.3% (5W/1T/0L over 6 preference-eligible stimulus vote(s), sign test p=0.031), mean preference +63.3% across 6 paired run(s) — credibly better",
|
|
"gateEvidence": {
|
|
"stimulusVoteCount": 6,
|
|
"wins": 5,
|
|
"ties": 1,
|
|
"losses": 0,
|
|
"discordant": 5,
|
|
"direction": "better",
|
|
"pValue": 0.03125,
|
|
"alpha": 0.05,
|
|
"netWin": 0.8333333333333334,
|
|
"minimumNetWin": 0.2,
|
|
"excludedStimulusCount": 0
|
|
},
|
|
"activationContract": {
|
|
"evaluated": true,
|
|
"requiredForPass": true,
|
|
"source": "isolated_target_skill_activation",
|
|
"reason": "Explicit dormancy expectations are evaluated independently of preference",
|
|
"count": 0,
|
|
"satisfied": 0,
|
|
"violated": 0,
|
|
"passed": true,
|
|
"failures": [],
|
|
"scenarios": [],
|
|
"unmatchedDormancyStimuli": []
|
|
},
|
|
"activationScenarios": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-no-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "missing-activation",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-no-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-no-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0
|
|
}
|
|
],
|
|
"judgeRationales": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"direction": "better",
|
|
"rationale": "B directly matches the intended architecture: Microsoft Agent Framework on top of Microsoft.Extensions.AI with a bounded iteration guardrail. A uses a different stack (Azure.AI.Agents.Persistent) with a raw polling loop and no iteration cap. Both fail the observability and cost-ceiling criteria equally, and tie on tool schemas. B wins clearly on three of the six criteria."
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"direction": "better",
|
|
"rationale": "Both responses failed to produce any implementation, so all technical rubric criteria are essentially unmet. However, B was more methodical: it explored the workspace, activated the technology-selection skill, actually found the project files, and correctly identified IChatClient via Microsoft.Extensions.AI as the appropriate approach before being blocked by a genuine file-read error. A did no investigation at all and falsely claimed 'Implemented' while immediately asking the user for files that were already present. B's approach was closer to correct and demonstrated real diagnostic effort, making it slightly better overall despite neither delivering working code."
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"direction": "better",
|
|
"rationale": "Both flagged missing workspace files, but A simply asked for files and delivered no implementation. B provided a complete, correct ML.NET implementation satisfying every rubric criterion (seeded MLContext, train/test split, proper multiclass metrics, PredictionEnginePool), while also noting it couldn't apply files directly. B is dramatically better."
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"direction": "none",
|
|
"rationale": "Position-swap inconsistent (forward: B, reverse: tie). Defaulting to tie."
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"direction": "better",
|
|
"rationale": "Both responses provide strong, code-free architecture plans covering ingestion, chunking, retrieval, citations, and rollout. However, B better matches the specific rubric targets: it selects Microsoft.Extensions.AI (IEmbeddingGenerator/IChatClient) and Microsoft.Extensions.VectorData explicitly, and calls out a minimum similarity threshold—three criteria where A is weaker or silent. A is more comprehensive in breadth (eval plan, observability, safety) but misses these specific technology and threshold points. Both tie on chunking, attribution, and caching. B wins overall by a modest margin due to closer rubric alignment."
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"direction": "better",
|
|
"rationale": "The task explicitly asked to implement the feature and explain the technology choice. Response B did both: it selected ML.NET with well-grounded reasoning and delivered a complete, correct binary classification pipeline using all the specified columns and the Churned label, plus production-safe integration advice. Response A stopped at asking for files/permission and only gave a technology recommendation without any implementation, failing the core 'implement it' requirement. Both chose the right technology, but B is far more complete and useful."
|
|
}
|
|
],
|
|
"links": [
|
|
{
|
|
"label": "Skill source",
|
|
"url": "https://github.com/dotnet/skills/blob/24f7cfbd42ad7bf52bcd67372816b982c38c64c6/plugins/dotnet-ai/skills/technology-selection/SKILL.md"
|
|
},
|
|
{
|
|
"label": "Eval source",
|
|
"url": "https://github.com/dotnet/skills/blob/24f7cfbd42ad7bf52bcd67372816b982c38c64c6/tests/dotnet-ai/technology-selection/eval.yaml"
|
|
},
|
|
{
|
|
"label": "Commit",
|
|
"url": "https://github.com/dotnet/skills/commit/24f7cfbd42ad7bf52bcd67372816b982c38c64c6"
|
|
}
|
|
]
|
|
}
|
|
],
|
|
"benches": [
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Quality",
|
|
"value": 9.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Quality",
|
|
"value": 9.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Quality",
|
|
"value": 2.5
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Quality",
|
|
"value": 6.666666507720947
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Quality",
|
|
"value": 5.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Quality",
|
|
"value": 5.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Quality",
|
|
"value": 9.5
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Quality",
|
|
"value": 9.5
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Quality",
|
|
"value": 2.0
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Quality",
|
|
"notActivated": true,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "Score (0-10)",
|
|
"value": 5.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Quality",
|
|
"value": 5.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Quality",
|
|
"value": 5.833333492279053
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Quality",
|
|
"value": 10.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.28,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Quality",
|
|
"value": 10.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Quality",
|
|
"value": 9.375
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Quality",
|
|
"value": 10.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Quality",
|
|
"value": 10.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Quality",
|
|
"value": 7.5
|
|
}
|
|
],
|
|
"model": "gpt-5.3-codex"
|
|
},
|
|
{
|
|
"benches": [
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Quality",
|
|
"value": 9.5
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Quality",
|
|
"value": 9.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Quality",
|
|
"value": 5.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Quality",
|
|
"value": 9.583333015441895
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Quality",
|
|
"value": 9.583333015441895
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Quality",
|
|
"value": 5.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Quality",
|
|
"value": 5.5
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Quality",
|
|
"value": 8.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Quality",
|
|
"value": 3.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Quality",
|
|
"value": 10.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Quality",
|
|
"value": 10.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Quality",
|
|
"value": 9.166666984558105
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Quality",
|
|
"value": 10.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Quality",
|
|
"value": 10.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Quality",
|
|
"value": 9.375
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Quality",
|
|
"value": 10.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.47,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Quality",
|
|
"value": 10.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Quality",
|
|
"value": 10.0
|
|
}
|
|
],
|
|
"model": "claude-sonnet-5",
|
|
"commit": {
|
|
"author": {
|
|
"name": "Abhitej John",
|
|
"username": "AbhitejJohn"
|
|
},
|
|
"timestamp": "2026-09-15T22:05:10+00:00",
|
|
"message": "Merge pull request #873 from dotnet/add-dotnet-refactoring-skills",
|
|
"url": "https://github.com/dotnet/skills/commit/26323a52990d0cbfc838117109b40e115aac891f",
|
|
"id": "26323a52990d0cbfc838117109b40e115aac891f"
|
|
},
|
|
"tool": "customBiggerIsBetter",
|
|
"verdictEvidence": [
|
|
{
|
|
"skillName": "technology-selection",
|
|
"skillKind": "invocable",
|
|
"state": "VALID_PASS",
|
|
"stateReason": {
|
|
"code": "credible_preference_improvement",
|
|
"phase": "decision"
|
|
},
|
|
"passed": true,
|
|
"regressed": false,
|
|
"preferenceRegressed": false,
|
|
"reason": "Net win +100.0% (6W/0T/0L over 6 preference-eligible stimulus vote(s), sign test p=0.016), mean preference +50.0% across 6 paired run(s) — credibly better",
|
|
"gateEvidence": {
|
|
"stimulusVoteCount": 6,
|
|
"wins": 6,
|
|
"ties": 0,
|
|
"losses": 0,
|
|
"discordant": 6,
|
|
"direction": "better",
|
|
"pValue": 0.015625000000000007,
|
|
"alpha": 0.05,
|
|
"netWin": 1,
|
|
"minimumNetWin": 0.2,
|
|
"excludedStimulusCount": 0
|
|
},
|
|
"activationContract": {
|
|
"evaluated": true,
|
|
"requiredForPass": true,
|
|
"source": "isolated_target_skill_activation",
|
|
"reason": "Explicit dormancy expectations are evaluated independently of preference",
|
|
"count": 0,
|
|
"satisfied": 0,
|
|
"violated": 0,
|
|
"passed": true,
|
|
"failures": [],
|
|
"scenarios": [],
|
|
"unmatchedDormancyStimuli": []
|
|
},
|
|
"activationScenarios": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
}
|
|
],
|
|
"judgeRationales": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"direction": "better",
|
|
"rationale": "Both compiled and implement the essential agent, search, and note-taking functionality. B more completely satisfies the requested operational safeguards and rubric requirements through iteration/token caps, explicit function metadata, and tool-call observability. A's DuckDuckGo HTML-result search is likely more useful as general web search than B's limited Instant Answer endpoint, so the overall advantage is meaningful but not large."
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"direction": "better",
|
|
"rationale": "B directly follows the AI-integration requirements: IChatClient/AddChatClient, explicit generation options, resilient calling, and credential-safe configuration, with a successful build. A is a functional self-contained endpoint, but it bypasses nearly every specified AI rubric requirement. B's missing demonstrated dated model pin prevents full compliance but does not outweigh its broad advantage."
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"direction": "better",
|
|
"rationale": "Both are functioning, verified ML.NET implementations with held-out evaluation and usable prediction endpoints. B is overall preferable because it satisfies the important ASP.NET concurrency requirement by using PredictionEnginePool and more clearly ensures reproducibility; A's richer metrics endpoint is a useful but smaller advantage."
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"direction": "better",
|
|
"rationale": "Both are strong, appropriately scoped plans. B is slightly better overall because it directly satisfies the requested Microsoft.Extensions.VectorData.Abstractions plus pgvector architecture and adds useful operational guardrails. A remains particularly practical, with a clear data model and implementation order, so the overall gap is modest."
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"direction": "better",
|
|
"rationale": "Both are strong, code-free RAG architectures that cover ingestion, semantic chunking, vector retrieval, grounding, and attribution. B is overall preferable because it explicitly includes the critical similarity-score gate and an abstention path, preventing irrelevant context from being presented as grounded answers. A is somewhat more detailed on chunking, observability, and rollout, but misses that rubric-critical retrieval safeguard."
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"direction": "better",
|
|
"rationale": "Both are correct, tested ML.NET implementations that satisfy the core task. B is modestly stronger due to more robust model-training/evaluation and concurrent-serving design, while A remains a valid end-to-end solution."
|
|
}
|
|
],
|
|
"links": [
|
|
{
|
|
"label": "Skill source",
|
|
"url": "https://github.com/dotnet/skills/blob/26323a52990d0cbfc838117109b40e115aac891f/plugins/dotnet-ai/skills/technology-selection/SKILL.md"
|
|
},
|
|
{
|
|
"label": "Eval source",
|
|
"url": "https://github.com/dotnet/skills/blob/26323a52990d0cbfc838117109b40e115aac891f/tests/dotnet-ai/technology-selection/eval.yaml"
|
|
},
|
|
{
|
|
"label": "Commit",
|
|
"url": "https://github.com/dotnet/skills/commit/26323a52990d0cbfc838117109b40e115aac891f"
|
|
}
|
|
]
|
|
}
|
|
],
|
|
"date": 1789574378627
|
|
},
|
|
{
|
|
"benches": [
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.24,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Quality",
|
|
"value": 3.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.24,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Quality",
|
|
"value": 3.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Quality",
|
|
"value": 5.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.24,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Quality",
|
|
"value": 6.25
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.24,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Quality",
|
|
"value": 7.916666507720947
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Quality",
|
|
"value": 5.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.24,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Quality",
|
|
"value": 3.5
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.24,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Quality",
|
|
"value": 3.5
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Quality",
|
|
"value": 3.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.24,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Quality",
|
|
"value": 10.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.24,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Quality",
|
|
"value": 9.166666984558105
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Quality",
|
|
"value": 5.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.24,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Quality",
|
|
"value": 10.0
|
|
},
|
|
{
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.24,
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Quality",
|
|
"value": 10.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Quality",
|
|
"value": 8.75
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Quality",
|
|
"value": 10.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Quality",
|
|
"value": 10.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Quality",
|
|
"value": 10.0
|
|
}
|
|
],
|
|
"model": "gpt-5.6-luna",
|
|
"commit": {
|
|
"author": {
|
|
"name": "Abhitej John",
|
|
"username": "AbhitejJohn"
|
|
},
|
|
"timestamp": "2026-09-15T22:05:10+00:00",
|
|
"message": "Merge pull request #873 from dotnet/add-dotnet-refactoring-skills",
|
|
"url": "https://github.com/dotnet/skills/commit/26323a52990d0cbfc838117109b40e115aac891f",
|
|
"id": "26323a52990d0cbfc838117109b40e115aac891f"
|
|
},
|
|
"tool": "customBiggerIsBetter",
|
|
"verdictEvidence": [
|
|
{
|
|
"skillName": "technology-selection",
|
|
"skillKind": "invocable",
|
|
"state": "VALID_NO_CHANGE",
|
|
"stateReason": {
|
|
"code": "no_credible_preference_change",
|
|
"phase": "decision"
|
|
},
|
|
"passed": false,
|
|
"regressed": false,
|
|
"preferenceRegressed": false,
|
|
"reason": "Net win +66.7% (4W/2T/0L over 6 preference-eligible stimulus vote(s), sign test p=0.063), mean preference +36.7% across 6 paired run(s) — not credible — 2 of 6 preference-eligible stimulus vote(s) tied, leaving only 4 discordant preference vote(s). The sign test conditions on non-tie stimulus votes and cannot reach 0.05 below 5, so no record could have passed here — this is not a measured null. Either the skill is inert on these scenarios (make them discriminate) or the eval needs more distinct stimuli to clear the ties",
|
|
"gateEvidence": {
|
|
"stimulusVoteCount": 6,
|
|
"wins": 4,
|
|
"ties": 2,
|
|
"losses": 0,
|
|
"discordant": 4,
|
|
"direction": "better",
|
|
"pValue": 0.0625,
|
|
"alpha": 0.05,
|
|
"netWin": 0.6666666666666666,
|
|
"minimumNetWin": 0.2,
|
|
"excludedStimulusCount": 0
|
|
},
|
|
"activationContract": {
|
|
"evaluated": true,
|
|
"requiredForPass": true,
|
|
"source": "isolated_target_skill_activation",
|
|
"reason": "Explicit dormancy expectations are evaluated independently of preference",
|
|
"count": 0,
|
|
"satisfied": 0,
|
|
"violated": 0,
|
|
"passed": true,
|
|
"failures": [],
|
|
"scenarios": [],
|
|
"unmatchedDormancyStimuli": []
|
|
},
|
|
"activationScenarios": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
}
|
|
],
|
|
"judgeRationales": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"direction": "none",
|
|
"rationale": "Position-swap inconsistent (forward: B, reverse: tie). Defaulting to tie."
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"direction": "better",
|
|
"rationale": "The task and rubric clearly expect an AI/LLM-based summarization endpoint built on Microsoft.Extensions.AI. Response B correctly interpreted this, using IChatClient, DI, configuration-based API keys, and a dated model version, satisfying most rubric criteria despite having to drop the retry client after an API mismatch. Response A implemented a deterministic extractive summarizer with no AI integration at all, failing nearly every rubric criterion. Although A's build/test flow worked and produced a functioning endpoint, it fundamentally misses the intended approach. B verified the build but didn't runtime-test (needs an API key), yet aligns far better with the requirements. B is much better overall."
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"direction": "better",
|
|
"rationale": "Both responses deliver working ML.NET classifiers with train/test evaluation and appropriate multiclass metrics, and both verified endpoints successfully. B distinctly satisfies two rubric criteria that A does not clearly address: it explicitly sets a deterministic seed and uses a thread-safe PredictionEnginePool (with model persistence to zip), which is the recommended ASP.NET Core pattern. A's summary and timeline give no indication of these, likely relying on a singleton PredictionEngine and no seed. These are meaningful correctness/best-practice advantages, giving B the edge, though both are otherwise comparable in quality."
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"direction": "better",
|
|
"rationale": "Both responses deliver strong, well-structured plans meeting most rubric criteria. B distinguishes itself by explicitly selecting the intended Microsoft.Extensions.AI and Microsoft.Extensions.VectorData.Abstractions stack with concrete providers, directly satisfying two criteria that A only addresses generically. A did inspect the repo (a nice touch), but the rubric prioritizes the specific technology selection, where B is clearly stronger. Overall B is slightly better."
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"direction": "better",
|
|
"rationale": "Both responses are well-structured, code-free architecture plans. However, B satisfies far more rubric criteria: it explicitly selects Microsoft.Extensions.AI (IEmbeddingGenerator/IChatClient), Microsoft.Extensions.VectorData, and includes a minimum similarity score threshold — all of which A only handles generically or omits. B also adds citation validation and more concrete caching. A is solid and technology-agnostic but misses the specific selections the rubric rewards. [Magnitude reduced to \"slightly-better\": the position-swapped pass judged this \"slightly-better\", so the more confident \"much-better\" was downgraded.]"
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"direction": "none",
|
|
"rationale": "Position-swap inconsistent (forward: tie, reverse: A). Defaulting to tie."
|
|
}
|
|
],
|
|
"links": [
|
|
{
|
|
"label": "Skill source",
|
|
"url": "https://github.com/dotnet/skills/blob/26323a52990d0cbfc838117109b40e115aac891f/plugins/dotnet-ai/skills/technology-selection/SKILL.md"
|
|
},
|
|
{
|
|
"label": "Eval source",
|
|
"url": "https://github.com/dotnet/skills/blob/26323a52990d0cbfc838117109b40e115aac891f/tests/dotnet-ai/technology-selection/eval.yaml"
|
|
},
|
|
{
|
|
"label": "Commit",
|
|
"url": "https://github.com/dotnet/skills/commit/26323a52990d0cbfc838117109b40e115aac891f"
|
|
}
|
|
]
|
|
}
|
|
],
|
|
"date": 1789574378723
|
|
},
|
|
{
|
|
"model": "claude-opus-4.8",
|
|
"tool": "customBiggerIsBetter",
|
|
"date": 1789642372141,
|
|
"commit": {
|
|
"message": "Merge pull request #1184 from dotnet/dependabot/npm_and_yarn/eng/evaluation-tools/microsoft/vally-cli-0.16.0",
|
|
"id": "8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3",
|
|
"timestamp": "2026-09-17T03:09:52+00:00",
|
|
"url": "https://github.com/dotnet/skills/commit/8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3",
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Abhitej John"
|
|
}
|
|
},
|
|
"verdictEvidence": [
|
|
{
|
|
"skillName": "technology-selection",
|
|
"skillKind": "invocable",
|
|
"state": "VALID_PASS",
|
|
"stateReason": {
|
|
"code": "credible_preference_improvement",
|
|
"phase": "decision"
|
|
},
|
|
"passed": true,
|
|
"regressed": false,
|
|
"preferenceRegressed": false,
|
|
"reason": "Net win +100.0% (6W/0T/0L over 6 preference-eligible stimulus vote(s), sign test p=0.016), mean preference +50.0% across 6 paired run(s) — credibly better",
|
|
"gateEvidence": {
|
|
"stimulusVoteCount": 6,
|
|
"wins": 6,
|
|
"ties": 0,
|
|
"losses": 0,
|
|
"discordant": 6,
|
|
"direction": "better",
|
|
"pValue": 0.015625000000000007,
|
|
"alpha": 0.05,
|
|
"netWin": 1,
|
|
"minimumNetWin": 0.2,
|
|
"excludedStimulusCount": 0
|
|
},
|
|
"activationContract": {
|
|
"evaluated": true,
|
|
"requiredForPass": true,
|
|
"source": "isolated_target_skill_activation",
|
|
"reason": "Explicit dormancy expectations are evaluated independently of preference",
|
|
"count": 0,
|
|
"satisfied": 0,
|
|
"violated": 0,
|
|
"passed": true,
|
|
"failures": [],
|
|
"scenarios": [],
|
|
"unmatchedDormancyStimuli": []
|
|
},
|
|
"activationScenarios": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
}
|
|
],
|
|
"judgeRationales": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"direction": "better",
|
|
"rationale": "Both appear to deliver buildable .NET 10 Agent Framework research agents with IChatClient and the requested tools. B more clearly addresses explicit schemas, observability, and output-token control, while neither demonstrates the required MaximumIterations cap. A's live search/note isolation test is a positive, but does not outweigh B's stronger rubric coverage."
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"direction": "better",
|
|
"rationale": "B directly fulfills every stated AI-integration rubric requirement and builds successfully. Although it could not make a successful live model call without Azure credentials, that is an environment limitation and its endpoint reaches the configured auth layer. A provides a functioning local summarizer, but it misses all six explicit MEAI, configuration, resilience, and pinned-model requirements."
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"direction": "better",
|
|
"rationale": "B satisfies the key reproducibility and thread-safe serving requirements explicitly while retaining the provided tiny dataset and accurately caveating its limitations. A's unseeded/unclear serving design and especially its replacement of the supplied CSV with invented training rows are material quality issues. [Magnitude reduced to \"slightly-better\": the position-swapped pass judged this \"slightly-better\", so the more confident \"much-better\" was downgraded.]"
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"direction": "better",
|
|
"rationale": "Both are strong, appropriately plan-only responses. B is better aligned with the requested .NET AI architecture by explicitly using MEAI and VectorData abstractions, while preserving the requested Blazor, Postgres, and folder-based workflow. A remains a sound alternative architecture with stronger concrete EF-oriented implementation detail, so the overall gap is modest."
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"direction": "better",
|
|
"rationale": "Both are solid, code-free RAG plans covering ingestion, structured chunking, retrieval, grounding, and citations. B more precisely fulfills the requested .NET abstractions and makes embedding caching and score-threshold behavior explicit; A remains a practical and largely complete alternative."
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"direction": "better",
|
|
"rationale": "Both are viable ML.NET implementations and explain the technology choice well. B is stronger overall because it treats the tiny source dataset honestly, provides a graceful evaluation caveat, and uses thread-safe PredictionEnginePool serving. A's alteration of the provided dataset and consequent perfect-metric claim is a meaningful quality concern, though its implementation otherwise appears functional."
|
|
}
|
|
],
|
|
"links": [
|
|
{
|
|
"label": "Skill source",
|
|
"url": "https://github.com/dotnet/skills/blob/8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3/plugins/dotnet-ai/skills/technology-selection/SKILL.md"
|
|
},
|
|
{
|
|
"label": "Eval source",
|
|
"url": "https://github.com/dotnet/skills/blob/8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3/tests/dotnet-ai/technology-selection/eval.yaml"
|
|
},
|
|
{
|
|
"label": "Commit",
|
|
"url": "https://github.com/dotnet/skills/commit/8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3"
|
|
}
|
|
]
|
|
}
|
|
],
|
|
"benches": [
|
|
{
|
|
"overfittingScore": 0.45,
|
|
"overfitting": "moderate",
|
|
"value": 7.5,
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Quality",
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"overfittingScore": 0.45,
|
|
"overfitting": "moderate",
|
|
"value": 9.5,
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Quality",
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"value": 5.0,
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Quality",
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"overfittingScore": 0.45,
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Quality",
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"overfittingScore": 0.45,
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Quality",
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"value": 5.0,
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Quality",
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"overfittingScore": 0.45,
|
|
"overfitting": "moderate",
|
|
"value": 8.0,
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Quality",
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"overfittingScore": 0.45,
|
|
"overfitting": "moderate",
|
|
"value": 8.0,
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Quality",
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"value": 3.0,
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Quality",
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"overfittingScore": 0.45,
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Quality",
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"overfittingScore": 0.45,
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Quality",
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"value": 5.0,
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Quality",
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"overfittingScore": 0.45,
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Quality",
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"overfittingScore": 0.45,
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Quality",
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Quality",
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"overfittingScore": 0.45,
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Quality",
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"overfittingScore": 0.45,
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Quality",
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"value": 10.0,
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Quality",
|
|
"unit": "Score (0-10)"
|
|
}
|
|
]
|
|
},
|
|
{
|
|
"benches": [
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Quality",
|
|
"value": 7.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Quality",
|
|
"value": 4.5
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Quality",
|
|
"value": 10.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Quality",
|
|
"value": 9.583333015441895,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Quality",
|
|
"value": 5.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Quality",
|
|
"value": 6.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Quality",
|
|
"value": 6.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Quality",
|
|
"value": 2.5
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Quality",
|
|
"value": 10.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Quality",
|
|
"value": 10.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Quality",
|
|
"value": 9.166666984558105
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Quality",
|
|
"value": 10.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Quality",
|
|
"value": 10.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Quality",
|
|
"value": 10.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Quality",
|
|
"value": 9.166666984558105,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Quality",
|
|
"value": 10.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Quality",
|
|
"value": 10.0
|
|
}
|
|
],
|
|
"date": 1789745909143,
|
|
"tool": "customBiggerIsBetter",
|
|
"model": "claude-sonnet-5",
|
|
"commit": {
|
|
"message": "Merge pull request #1184 from dotnet/dependabot/npm_and_yarn/eng/evaluation-tools/microsoft/vally-cli-0.16.0",
|
|
"timestamp": "2026-09-17T03:09:52+00:00",
|
|
"url": "https://github.com/dotnet/skills/commit/8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3",
|
|
"author": {
|
|
"name": "Abhitej John",
|
|
"username": "AbhitejJohn"
|
|
},
|
|
"id": "8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3"
|
|
},
|
|
"verdictEvidence": [
|
|
{
|
|
"skillName": "technology-selection",
|
|
"skillKind": "invocable",
|
|
"state": "VALID_NO_CHANGE",
|
|
"stateReason": {
|
|
"code": "no_credible_preference_change",
|
|
"phase": "decision"
|
|
},
|
|
"passed": false,
|
|
"regressed": false,
|
|
"preferenceRegressed": false,
|
|
"reason": "Net win +66.7% (5W/0T/1L over 6 preference-eligible stimulus vote(s), sign test p=0.109), mean preference +36.7% across 6 paired run(s) — not credible (sign test p=0.109 > 0.05)",
|
|
"gateEvidence": {
|
|
"stimulusVoteCount": 6,
|
|
"wins": 5,
|
|
"ties": 0,
|
|
"losses": 1,
|
|
"discordant": 6,
|
|
"direction": "better",
|
|
"pValue": 0.10937500000000008,
|
|
"alpha": 0.05,
|
|
"netWin": 0.6666666666666666,
|
|
"minimumNetWin": 0.2,
|
|
"excludedStimulusCount": 0
|
|
},
|
|
"activationContract": {
|
|
"evaluated": true,
|
|
"requiredForPass": true,
|
|
"source": "isolated_target_skill_activation",
|
|
"reason": "Explicit dormancy expectations are evaluated independently of preference",
|
|
"count": 0,
|
|
"satisfied": 0,
|
|
"violated": 0,
|
|
"passed": true,
|
|
"failures": [],
|
|
"scenarios": [],
|
|
"unmatchedDormancyStimuli": []
|
|
},
|
|
"activationScenarios": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-no-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
}
|
|
],
|
|
"judgeRationales": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"direction": "better",
|
|
"rationale": "Both projects build and provide the requested basic tools, but B follows the specified Microsoft Agent Framework architecture atop IChatClient and adds bounded, observable agent execution. A is a functional raw function-invocation app but misses the central orchestration requirement and lacks demonstrated safeguards. [Magnitude reduced to \"slightly-better\": the position-swapped pass judged this \"slightly-better\", so the more confident \"much-better\" was downgraded.]"
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"direction": "better",
|
|
"rationale": "B directly satisfies every requested AI-integration rubric, including DI, deterministic generation settings, resilience, credential-safe configuration, and a dated model deployment. A provides a functional tested local extractive endpoint, but misses the specified Microsoft.Extensions.AI architecture and all associated integration requirements."
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"direction": "better",
|
|
"rationale": "Both are working ML.NET implementations with verified endpoints and sensible small-data caveats. B more directly satisfies the specified reproducibility, held-out evaluation, and ASP.NET prediction-serving requirements, especially PredictionEnginePool. A offers useful cross-validation and an explicit training endpoint, but those do not outweigh B's clearer compliance with the rubric."
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"direction": "better",
|
|
"rationale": "Both are strong, concise plans that appropriately stop for confirmation before implementation. B is more directly aligned with the requested modern MEAI and VectorData abstractions and includes useful retrieval reliability details, while A is also technically sound and concrete."
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"direction": "better",
|
|
"rationale": "Both are strong, code-free architecture plans that satisfy nearly all requirements. B is modestly better because it clearly specifies the critical minimum-similarity rejection behavior and corresponding grounded no-answer path; A is otherwise at least as concrete and adds useful testing and operations detail."
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"direction": "worse",
|
|
"rationale": "Although both chose and justified the appropriate technology, A fulfills the central requirement against the supplied dataset and candidly identifies the small-data evaluation limitation. B's otherwise solid service design is undermined by replacing the user's data with fabricated records, making it materially less appropriate and its metrics inapplicable to the requested implementation. [Magnitude reduced to \"slightly-better\": the position-swapped pass judged this \"slightly-better\", so the more confident \"much-better\" was downgraded.]"
|
|
}
|
|
],
|
|
"links": [
|
|
{
|
|
"label": "Skill source",
|
|
"url": "https://github.com/dotnet/skills/blob/8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3/plugins/dotnet-ai/skills/technology-selection/SKILL.md"
|
|
},
|
|
{
|
|
"label": "Eval source",
|
|
"url": "https://github.com/dotnet/skills/blob/8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3/tests/dotnet-ai/technology-selection/eval.yaml"
|
|
},
|
|
{
|
|
"label": "Commit",
|
|
"url": "https://github.com/dotnet/skills/commit/8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3"
|
|
}
|
|
]
|
|
}
|
|
]
|
|
},
|
|
{
|
|
"benches": [
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Quality",
|
|
"value": 3.5,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Quality",
|
|
"value": 3.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Quality",
|
|
"value": 2.5
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Quality",
|
|
"value": 5.833333492279053,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Quality",
|
|
"value": 6.25,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Quality",
|
|
"value": 5.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Quality",
|
|
"value": 3.5,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Quality",
|
|
"value": 4.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Quality",
|
|
"value": 2.5
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Quality",
|
|
"value": 9.166666984558105,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Quality",
|
|
"value": 9.166666984558105,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Quality",
|
|
"value": 5.0
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Quality",
|
|
"value": 10.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Quality",
|
|
"value": 10.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Quality",
|
|
"value": 9.375
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Quality",
|
|
"value": 10.0,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Quality",
|
|
"value": 9.166666984558105,
|
|
"overfitting": "moderate",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"unit": "Score (0-10)",
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Quality",
|
|
"value": 10.0
|
|
}
|
|
],
|
|
"date": 1789745909242,
|
|
"tool": "customBiggerIsBetter",
|
|
"model": "gpt-5.6-luna",
|
|
"commit": {
|
|
"message": "Merge pull request #1184 from dotnet/dependabot/npm_and_yarn/eng/evaluation-tools/microsoft/vally-cli-0.16.0",
|
|
"timestamp": "2026-09-17T03:09:52+00:00",
|
|
"url": "https://github.com/dotnet/skills/commit/8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3",
|
|
"author": {
|
|
"name": "Abhitej John",
|
|
"username": "AbhitejJohn"
|
|
},
|
|
"id": "8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3"
|
|
},
|
|
"verdictEvidence": [
|
|
{
|
|
"skillName": "technology-selection",
|
|
"skillKind": "invocable",
|
|
"state": "VALID_PASS",
|
|
"stateReason": {
|
|
"code": "credible_preference_improvement",
|
|
"phase": "decision"
|
|
},
|
|
"passed": true,
|
|
"regressed": false,
|
|
"preferenceRegressed": false,
|
|
"reason": "Net win +83.3% (5W/1T/0L over 6 preference-eligible stimulus vote(s), sign test p=0.031), mean preference +43.3% across 6 paired run(s) — credibly better",
|
|
"gateEvidence": {
|
|
"stimulusVoteCount": 6,
|
|
"wins": 5,
|
|
"ties": 1,
|
|
"losses": 0,
|
|
"discordant": 5,
|
|
"direction": "better",
|
|
"pValue": 0.03125,
|
|
"alpha": 0.05,
|
|
"netWin": 0.8333333333333334,
|
|
"minimumNetWin": 0.2,
|
|
"excludedStimulusCount": 0
|
|
},
|
|
"activationContract": {
|
|
"evaluated": true,
|
|
"requiredForPass": true,
|
|
"source": "isolated_target_skill_activation",
|
|
"reason": "Explicit dormancy expectations are evaluated independently of preference",
|
|
"count": 0,
|
|
"satisfied": 0,
|
|
"violated": 0,
|
|
"passed": true,
|
|
"failures": [],
|
|
"scenarios": [],
|
|
"unmatchedDormancyStimuli": []
|
|
},
|
|
"activationScenarios": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
}
|
|
],
|
|
"judgeRationales": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"direction": "better",
|
|
"rationale": "Both built a working, building .NET 10 console research agent with two tools using the Microsoft Agent Framework. B is stronger on the explicit IChatClient foundation and on bounding the agentic loop via FunctionInvokingChatClient iteration caps, which it deliberately investigated and configured. B also had a more methodical process (activated the technology-selection skill, verified iteration config, tested missing-key handling). A got stuck on repeated apply_patch failures before recovering via python, whereas B progressed more smoothly. Neither fully implemented observability logging or an explicit token/cost budget. Overall B is slightly better."
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"direction": "better",
|
|
"rationale": "The rubric clearly expects an AI-backed summarization endpoint built on Microsoft.Extensions.AI with DI, ChatOptions, retry logic, config-based keys, and pinned model versions. Response B directly pursues this architecture and satisfies most criteria (MEAI/IChatClient, DI, retry, config-based key). Response A instead built a naive extractive string summarizer with no AI integration whatsoever, failing essentially all rubric criteria despite building and smoke-testing successfully. While A's execution was clean, it fundamentally misses the intended solution. B recovered from build errors (package version, .Use overload) methodically and ended with a successful build. B is much better against the rubric."
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"direction": "better",
|
|
"rationale": "Both use ML.NET and report multiclass metrics, but B satisfies three additional rubric criteria more clearly: it sets a random seed (42) for reproducibility, performs a proper train/test split, and uses PredictionEnginePool via Microsoft.Extensions.ML. A shows no seed, produced degenerate evaluation metrics (0 accuracy, null log loss reduction) suggesting improper/no held-out evaluation, and uses a vague custom prediction service rather than PredictionEnginePool. B is much better on the rubric that matters. [Magnitude reduced to \"slightly-better\": the position-swapped pass judged this \"slightly-better\", so the more confident \"much-better\" was downgraded.]"
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"direction": "better",
|
|
"rationale": "Both produce solid, well-structured RAG plans respecting Blazor, Postgres/pgvector, and file-folder constraints. B is superior because it concretely names the intended Microsoft.Extensions.AI abstraction and Microsoft.Extensions.VectorData.Abstractions packages that the rubric specifically rewards, while A stays generic ('APIs behind interfaces'). Both avoid hallucinated packages. The gap is meaningful but not huge since A's plan is otherwise equally comprehensive."
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"direction": "better",
|
|
"rationale": "Both responses provide strong, well-structured RAG plans covering chunking, thresholds, and citations. However, Response B hits the specific technology choices the rubric rewards: Microsoft.Extensions.AI (IChatClient/IEmbeddingGenerator), Microsoft.Extensions.VectorData, and an explicit embedding cache. Response A is technology-agnostic and misses these named abstractions and the caching plan. B satisfies all six criteria fully while A partially misses three.\" [Magnitude reduced to \"slightly-better\": the position-swapped pass judged this \"slightly-better\", so the more confident \"much-better\" was downgraded.]"
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"direction": "none",
|
|
"rationale": "Position-swap inconsistent (forward: B, reverse: tie). Defaulting to tie."
|
|
}
|
|
],
|
|
"links": [
|
|
{
|
|
"label": "Skill source",
|
|
"url": "https://github.com/dotnet/skills/blob/8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3/plugins/dotnet-ai/skills/technology-selection/SKILL.md"
|
|
},
|
|
{
|
|
"label": "Eval source",
|
|
"url": "https://github.com/dotnet/skills/blob/8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3/tests/dotnet-ai/technology-selection/eval.yaml"
|
|
},
|
|
{
|
|
"label": "Commit",
|
|
"url": "https://github.com/dotnet/skills/commit/8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3"
|
|
}
|
|
]
|
|
}
|
|
]
|
|
},
|
|
{
|
|
"commit": {
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Abhitej John"
|
|
},
|
|
"message": "Merge pull request #1184 from dotnet/dependabot/npm_and_yarn/eng/evaluation-tools/microsoft/vally-cli-0.16.0",
|
|
"url": "https://github.com/dotnet/skills/commit/8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3",
|
|
"timestamp": "2026-09-17T03:09:52+00:00",
|
|
"id": "8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3"
|
|
},
|
|
"benches": [
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Quality",
|
|
"value": 2.0,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Quality",
|
|
"overfitting": "moderate",
|
|
"value": 7.083333492279053,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"value": 5.416666507720947,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Quality",
|
|
"value": 5.0,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"notActivated": true,
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Quality",
|
|
"value": 2.5,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.48,
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"value": 2.5,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Quality",
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"value": 8.333333015441895,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Quality",
|
|
"value": 5.0,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Quality",
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Quality",
|
|
"value": 9.375,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Quality",
|
|
"overfitting": "moderate",
|
|
"value": 9.166666984558105,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.48
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Quality",
|
|
"value": 9.166666984558105,
|
|
"unit": "Score (0-10)"
|
|
}
|
|
],
|
|
"verdictEvidence": [
|
|
{
|
|
"skillName": "technology-selection",
|
|
"skillKind": "invocable",
|
|
"state": "INVALID_INCONCLUSIVE",
|
|
"stateReason": {
|
|
"code": "unmatched_trajectories",
|
|
"phase": "comparison_pairing"
|
|
},
|
|
"passed": false,
|
|
"regressed": false,
|
|
"preferenceRegressed": false,
|
|
"reason": "Net win +75.0% (3W/1T/0L over 4 preference-eligible stimulus vote(s), sign test p=0.125), mean preference +30.0% across 4 paired run(s), 2 unmatched — inconclusive (unmatched trajectories)",
|
|
"gateEvidence": {
|
|
"stimulusVoteCount": 4,
|
|
"wins": 3,
|
|
"ties": 1,
|
|
"losses": 0,
|
|
"discordant": 3,
|
|
"direction": "better",
|
|
"pValue": 0.12500000000000003,
|
|
"alpha": 0.05,
|
|
"netWin": 0.75,
|
|
"minimumNetWin": 0.2,
|
|
"excludedStimulusCount": 0
|
|
},
|
|
"activationContract": {
|
|
"evaluated": true,
|
|
"requiredForPass": true,
|
|
"source": "isolated_target_skill_activation",
|
|
"reason": "Explicit dormancy expectations are evaluated independently of preference",
|
|
"count": 0,
|
|
"satisfied": 0,
|
|
"violated": 0,
|
|
"passed": true,
|
|
"failures": [],
|
|
"scenarios": [],
|
|
"unmatchedDormancyStimuli": []
|
|
},
|
|
"activationScenarios": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "missing-activation",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-no-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-no-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "missing-activation",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-no-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-no-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
}
|
|
],
|
|
"judgeRationales": [
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"direction": "better",
|
|
"rationale": "B is closer to the intended AI-backed endpoint by providing DI, configuration-based Azure authentication, and generation controls, and it builds successfully. However, it misses several core rubric requirements: it bypasses Microsoft.Extensions.AI/AddChatClient, has no retries, and does not pin a model version. A is a viable self-contained extractive endpoint for the broad task but does not address the AI-specific requirements."
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"direction": "better",
|
|
"rationale": "Both are sound, user-responsive implementation plans that stop before code. B better matches the specified modern .NET AI and vector-data abstractions and gives a more actionable RAG data flow; A remains a valid but more generic architecture."
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"direction": "better",
|
|
"rationale": "Both are solid, code-free RAG plans that satisfy the core ingestion, chunking, retrieval, grounding, and endpoint requirements. B is better tailored to a .NET 10 architecture through explicit Microsoft.Extensions.AI and VectorData abstractions, clearer idempotent ingestion, and more traceable attribution. The gap is limited because A remains a complete and technically sound plan."
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"direction": "none",
|
|
"rationale": "Position-swap inconsistent (forward: A, reverse: B). Defaulting to tie."
|
|
}
|
|
],
|
|
"links": [
|
|
{
|
|
"label": "Skill source",
|
|
"url": "https://github.com/dotnet/skills/blob/8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3/plugins/dotnet-ai/skills/technology-selection/SKILL.md"
|
|
},
|
|
{
|
|
"label": "Eval source",
|
|
"url": "https://github.com/dotnet/skills/blob/8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3/tests/dotnet-ai/technology-selection/eval.yaml"
|
|
},
|
|
{
|
|
"label": "Commit",
|
|
"url": "https://github.com/dotnet/skills/commit/8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3"
|
|
}
|
|
]
|
|
}
|
|
],
|
|
"date": 1789836946087,
|
|
"model": "claude-haiku-4.5",
|
|
"tool": "customBiggerIsBetter"
|
|
},
|
|
{
|
|
"commit": {
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Abhitej John"
|
|
},
|
|
"message": "Merge pull request #1184 from dotnet/dependabot/npm_and_yarn/eng/evaluation-tools/microsoft/vally-cli-0.16.0",
|
|
"url": "https://github.com/dotnet/skills/commit/8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3",
|
|
"timestamp": "2026-09-17T03:09:52+00:00",
|
|
"id": "8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3"
|
|
},
|
|
"benches": [
|
|
{
|
|
"notActivated": true,
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Quality",
|
|
"value": 4.5,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.26,
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"value": 4.5,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Quality",
|
|
"value": 2.0,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"notActivated": true,
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Quality",
|
|
"value": 5.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.26,
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"value": 5.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Quality",
|
|
"value": 5.0,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Quality",
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"value": 2.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"name": "technology-selection/ML.NET classification on tabular data - Vanilla Quality",
|
|
"value": 2.0,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"notActivated": true,
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Quality",
|
|
"value": 5.833333492279053,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.26,
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"value": 9.166666984558105,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Quality",
|
|
"value": 5.833333492279053,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"notActivated": true,
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Quality",
|
|
"value": 9.375,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.26,
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.26
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Quality",
|
|
"value": 8.75,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Quality",
|
|
"value": 8.333333015441895,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Quality",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Quality",
|
|
"value": 3.3333332538604736,
|
|
"unit": "Score (0-10)"
|
|
}
|
|
],
|
|
"verdictEvidence": [
|
|
{
|
|
"skillName": "technology-selection",
|
|
"skillKind": "invocable",
|
|
"state": "VALID_PASS",
|
|
"stateReason": {
|
|
"code": "credible_preference_improvement",
|
|
"phase": "decision"
|
|
},
|
|
"passed": true,
|
|
"regressed": false,
|
|
"preferenceRegressed": false,
|
|
"reason": "Net win +83.3% (5W/1T/0L over 6 preference-eligible stimulus vote(s), sign test p=0.031), mean preference +63.3% across 6 paired run(s) — credibly better",
|
|
"gateEvidence": {
|
|
"stimulusVoteCount": 6,
|
|
"wins": 5,
|
|
"ties": 1,
|
|
"losses": 0,
|
|
"discordant": 5,
|
|
"direction": "better",
|
|
"pValue": 0.03125,
|
|
"alpha": 0.05,
|
|
"netWin": 0.8333333333333334,
|
|
"minimumNetWin": 0.2,
|
|
"excludedStimulusCount": 0
|
|
},
|
|
"activationContract": {
|
|
"evaluated": true,
|
|
"requiredForPass": true,
|
|
"source": "isolated_target_skill_activation",
|
|
"reason": "Explicit dormancy expectations are evaluated independently of preference",
|
|
"count": 0,
|
|
"satisfied": 0,
|
|
"violated": 0,
|
|
"passed": true,
|
|
"failures": [],
|
|
"scenarios": [],
|
|
"unmatchedDormancyStimuli": []
|
|
},
|
|
"activationScenarios": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "missing-activation",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-no-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "missing-activation",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-no-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-no-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "missing-activation",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-no-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "missing-activation",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-no-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
}
|
|
],
|
|
"judgeRationales": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"direction": "better",
|
|
"rationale": "Response A refused to produce any code, delivering nothing usable. Response B delivered a complete, mostly working .NET 10 console app with web search and note-taking tools, well-documented schemas, and clear run instructions. While B falls short on several rubric criteria (no real agent framework loop, no MaximumIterations, no observability, no token budget), it is vastly more useful than A's empty response."
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"direction": "better",
|
|
"rationale": "Both responses failed the core task: neither inspected the codebase nor wrote any code, and both falsely began with 'Implemented' despite doing nothing. Neither satisfies any rubric criterion. However, Response B is marginally more useful because it provides a concrete implementation plan (endpoint route, request/response shapes, validation, a service abstraction) that gives the user something actionable, whereas Response A only vaguely states it needs to inspect files. The difference is slight, and neither meets the substantive technical requirements."
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"direction": "better",
|
|
"rationale": "Response B inspected the workspace and delivered a complete, correct ML.NET implementation satisfying every rubric criterion, while Response A produced no implementation and merely asked for files. B is far superior.\""
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"direction": "none",
|
|
"rationale": "Position-swap inconsistent (forward: B, reverse: tie). Defaulting to tie."
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"direction": "better",
|
|
"rationale": "Both responses are well-structured, comprehensive RAG architecture plans covering ingestion, chunking, retrieval, citations, and phased rollout with equivalent quality on most criteria. B edges ahead by explicitly naming Microsoft.Extensions.AI for the embedding abstraction layer (matching the .NET 10 idiomatic tooling the rubric rewards) and by mentioning embedding caching. Neither hits the similarity threshold criterion. Overall B is slightly better due to more concrete .NET-specific tooling choices."
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"direction": "better",
|
|
"rationale": "B addresses all three rubric criteria with a correct technology choice, clear justification, and a concrete implementation plan. A simply asks for more files and delivers nothing, failing every criterion. While ideally a full code implementation would be provided, B is substantially more responsive and correct.\""
|
|
}
|
|
],
|
|
"links": [
|
|
{
|
|
"label": "Skill source",
|
|
"url": "https://github.com/dotnet/skills/blob/8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3/plugins/dotnet-ai/skills/technology-selection/SKILL.md"
|
|
},
|
|
{
|
|
"label": "Eval source",
|
|
"url": "https://github.com/dotnet/skills/blob/8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3/tests/dotnet-ai/technology-selection/eval.yaml"
|
|
},
|
|
{
|
|
"label": "Commit",
|
|
"url": "https://github.com/dotnet/skills/commit/8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3"
|
|
}
|
|
]
|
|
}
|
|
],
|
|
"date": 1789836946152,
|
|
"model": "gpt-5.3-codex",
|
|
"tool": "customBiggerIsBetter"
|
|
},
|
|
{
|
|
"commit": {
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Abhitej John"
|
|
},
|
|
"message": "Merge pull request #1184 from dotnet/dependabot/npm_and_yarn/eng/evaluation-tools/microsoft/vally-cli-0.16.0",
|
|
"url": "https://github.com/dotnet/skills/commit/8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3",
|
|
"timestamp": "2026-09-17T03:09:52+00:00",
|
|
"id": "8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3"
|
|
},
|
|
"benches": [
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Skilled Quality",
|
|
"overfitting": "moderate",
|
|
"value": 3.5,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"value": 2.5,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/Agentic workflow with guardrails - Vanilla Quality",
|
|
"value": 2.5,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"notActivated": true,
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Skilled Quality",
|
|
"value": 5.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.49,
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"value": 5.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/LLM integration with MEAI abstraction - Vanilla Quality",
|
|
"value": 5.0,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"notActivated": true,
|
|
"name": "technology-selection/ML.NET classification on tabular data - Skilled Quality",
|
|
"value": 4.5,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.49,
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"notActivated": true,
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Skilled Quality",
|
|
"value": 5.833333492279053,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.49,
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"value": 9.166666984558105,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/Natural-language scenario decomposition — RAG chatbot - Vanilla Quality",
|
|
"value": 5.833333492279053,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"notActivated": true,
|
|
"name": "technology-selection/RAG pipeline with vector search - Skilled Quality",
|
|
"value": 9.375,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.49,
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"value": 8.75,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/RAG pipeline with vector search - Vanilla Quality",
|
|
"value": 9.375,
|
|
"unit": "Score (0-10)"
|
|
},
|
|
{
|
|
"notActivated": true,
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Skilled Quality",
|
|
"value": 9.166666984558105,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.49,
|
|
"overfitting": "moderate"
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Plugin Quality",
|
|
"overfitting": "moderate",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)",
|
|
"overfittingScore": 0.49
|
|
},
|
|
{
|
|
"name": "technology-selection/Select ML.NET for tabular churn prediction - Vanilla Quality",
|
|
"value": 10.0,
|
|
"unit": "Score (0-10)"
|
|
}
|
|
],
|
|
"verdictEvidence": [
|
|
{
|
|
"skillName": "technology-selection",
|
|
"skillKind": "invocable",
|
|
"state": "INVALID_INCONCLUSIVE",
|
|
"stateReason": {
|
|
"code": "unmatched_trajectories",
|
|
"phase": "comparison_pairing"
|
|
},
|
|
"passed": false,
|
|
"regressed": false,
|
|
"preferenceRegressed": false,
|
|
"reason": "Net win -40.0% (1W/1T/3L over 5 preference-eligible stimulus vote(s), sign test p=0.312), mean preference -16.0% across 5 paired run(s), 1 unmatched — inconclusive (unmatched trajectories)",
|
|
"gateEvidence": {
|
|
"stimulusVoteCount": 5,
|
|
"wins": 1,
|
|
"ties": 1,
|
|
"losses": 3,
|
|
"discordant": 4,
|
|
"direction": "worse",
|
|
"pValue": 0.31249999999999994,
|
|
"alpha": 0.05,
|
|
"netWin": -0.4,
|
|
"minimumNetWin": 0.2,
|
|
"excludedStimulusCount": 0
|
|
},
|
|
"activationContract": {
|
|
"evaluated": true,
|
|
"requiredForPass": true,
|
|
"source": "isolated_target_skill_activation",
|
|
"reason": "Explicit dormancy expectations are evaluated independently of preference",
|
|
"count": 0,
|
|
"satisfied": 0,
|
|
"violated": 0,
|
|
"passed": true,
|
|
"failures": [],
|
|
"scenarios": [],
|
|
"unmatchedDormancyStimuli": []
|
|
},
|
|
"activationScenarios": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "activated",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-no-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "missing-activation",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-no-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
},
|
|
{
|
|
"scenarioName": "ML.NET classification on tabular data",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "missing-activation",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-no-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "missing-activation",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-no-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "missing-activation",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-no-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"expectation": "active",
|
|
"preferenceGateEligible": true,
|
|
"isolated": "missing-activation",
|
|
"isolatedActivationOnlyFailedRuns": 0,
|
|
"plugin": "plugin-activity-observed",
|
|
"pluginActivationOnlyFailedRuns": 0,
|
|
"invokedAgents": [],
|
|
"delegatedAgents": [],
|
|
"invokedSkills": [],
|
|
"isolatedTools": [],
|
|
"pluginTools": [],
|
|
"isolatedCompleted": null,
|
|
"pluginCompleted": null
|
|
}
|
|
],
|
|
"judgeRationales": [
|
|
{
|
|
"scenarioName": "Agentic workflow with guardrails",
|
|
"direction": "better",
|
|
"rationale": "Both successfully deliver buildable .NET 10 agent apps with search and notes. B is stronger on the rubric-critical safety requirement through an explicit eight-iteration cap and offers a more concrete, robust search implementation. Neither addresses a cost/token budget, and observability evidence remains limited."
|
|
},
|
|
{
|
|
"scenarioName": "LLM integration with MEAI abstraction",
|
|
"direction": "none",
|
|
"rationale": "Both agents delivered a working-style local summarization endpoint rather than the required Microsoft.Extensions.AI-based, DI-registered, configured, retrying, version-pinned AI integration. Thus both miss every stated rubric requirement."
|
|
},
|
|
{
|
|
"scenarioName": "Natural-language scenario decomposition — RAG chatbot",
|
|
"direction": "worse",
|
|
"rationale": "A is the stronger plan because it gives more actionable implementation and operational detail around extraction metadata, grounded RAG/citations, access control, synchronization, monitoring, and deployment. Both share the key rubric omissions of not choosing MEAI and Microsoft.Extensions.VectorData abstractions, so the advantage is modest."
|
|
},
|
|
{
|
|
"scenarioName": "RAG pipeline with vector search",
|
|
"direction": "worse",
|
|
"rationale": "Both deliver code-free, coherent RAG plans covering ingestion, structured chunking, embedding storage, retrieval, grounding, and citations. A is modestly stronger through more concrete chunking parameters, named vector-store choices, a conceptual data model, and clearer endpoint/quality-control detail. Neither addresses the requested Microsoft.Extensions.AI abstraction or an explicit similarity threshold."
|
|
},
|
|
{
|
|
"scenarioName": "Select ML.NET for tabular churn prediction",
|
|
"direction": "worse",
|
|
"rationale": "Both are viable compiled ML.NET churn implementations with appropriate technology choices. B's rationale and extra engineering features are stronger, but A is clearer, directly demonstrably implements the requested prediction endpoint against the supplied CSV, and avoids the concerning alteration of the provided dataset seen in B's run."
|
|
}
|
|
],
|
|
"links": [
|
|
{
|
|
"label": "Skill source",
|
|
"url": "https://github.com/dotnet/skills/blob/8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3/plugins/dotnet-ai/skills/technology-selection/SKILL.md"
|
|
},
|
|
{
|
|
"label": "Eval source",
|
|
"url": "https://github.com/dotnet/skills/blob/8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3/tests/dotnet-ai/technology-selection/eval.yaml"
|
|
},
|
|
{
|
|
"label": "Commit",
|
|
"url": "https://github.com/dotnet/skills/commit/8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3"
|
|
}
|
|
]
|
|
}
|
|
],
|
|
"date": 1789836946226,
|
|
"model": "mai-code-1.1-flash",
|
|
"tool": "customBiggerIsBetter"
|
|
}
|
|
],
|
|
"SkillValue": [
|
|
{
|
|
"judgeModel": "claude-opus-4.8",
|
|
"date": 1788627832039,
|
|
"commit": {
|
|
"author": {
|
|
"name": "Amaury Levé",
|
|
"username": "AbhitejJohn"
|
|
},
|
|
"message": "Improve non-passing dotnet-test scenarios (#1114)",
|
|
"id": "ac8f41264bdd557e58924a3110eea8e0917dcf4d",
|
|
"url": "https://github.com/dotnet/skills/commit/ac8f41264bdd557e58924a3110eea8e0917dcf4d",
|
|
"timestamp": "2026-09-04T10:21:25+00:00"
|
|
},
|
|
"skills": [
|
|
{
|
|
"timedOut": false,
|
|
"pairedN": 6,
|
|
"activationFired": 3,
|
|
"treatment": {
|
|
"timeMs": 5649.333333333333,
|
|
"tokensOut": 516.6666666666666,
|
|
"cacheRead": 15104.0,
|
|
"cacheWrite": 0.0,
|
|
"tokens": 24239.666666666668,
|
|
"n": 6,
|
|
"tokensIn": 23723.0
|
|
},
|
|
"baseAvailable": 6,
|
|
"baseline": {
|
|
"timeMs": 5706.5,
|
|
"tokensOut": 444.5,
|
|
"cacheRead": 4608.0,
|
|
"cacheWrite": 0.0,
|
|
"tokens": 12143.333333333334,
|
|
"n": 6,
|
|
"tokensIn": 11698.833333333334
|
|
},
|
|
"activationExpected": 6,
|
|
"hasPassData": true,
|
|
"treatmentFail": 4,
|
|
"baselineFail": 5,
|
|
"passTotal": 6,
|
|
"treatAvailable": 6,
|
|
"skill": "technology-selection"
|
|
}
|
|
],
|
|
"model": "gpt-5.3-codex"
|
|
},
|
|
{
|
|
"judgeModel": "gpt-5.6-terra",
|
|
"date": 1788627832128,
|
|
"commit": {
|
|
"author": {
|
|
"name": "Amaury Levé",
|
|
"username": "AbhitejJohn"
|
|
},
|
|
"message": "Improve non-passing dotnet-test scenarios (#1114)",
|
|
"id": "ac8f41264bdd557e58924a3110eea8e0917dcf4d",
|
|
"url": "https://github.com/dotnet/skills/commit/ac8f41264bdd557e58924a3110eea8e0917dcf4d",
|
|
"timestamp": "2026-09-04T10:21:25+00:00"
|
|
},
|
|
"skills": [
|
|
{
|
|
"timedOut": false,
|
|
"pairedN": 6,
|
|
"activationFired": 4,
|
|
"treatment": {
|
|
"timeMs": 98093.66666666667,
|
|
"tokensOut": 9264.666666666666,
|
|
"cacheRead": 508800.0,
|
|
"cacheWrite": 0.0,
|
|
"tokens": 555340.6666666666,
|
|
"n": 6,
|
|
"tokensIn": 546076.0
|
|
},
|
|
"baseAvailable": 6,
|
|
"baseline": {
|
|
"timeMs": 112080.16666666667,
|
|
"tokensOut": 9803.5,
|
|
"cacheRead": 388864.0,
|
|
"cacheWrite": 0.0,
|
|
"tokens": 429669.6666666667,
|
|
"n": 6,
|
|
"tokensIn": 419866.1666666667
|
|
},
|
|
"activationExpected": 6,
|
|
"hasPassData": true,
|
|
"treatmentFail": 4,
|
|
"baselineFail": 4,
|
|
"passTotal": 6,
|
|
"treatAvailable": 6,
|
|
"skill": "technology-selection"
|
|
}
|
|
],
|
|
"model": "mai-code-1-flash-picker"
|
|
},
|
|
{
|
|
"commit": {
|
|
"id": "ac8f41264bdd557e58924a3110eea8e0917dcf4d",
|
|
"timestamp": "2026-09-04T10:21:25+00:00",
|
|
"message": "Improve non-passing dotnet-test scenarios (#1114)",
|
|
"url": "https://github.com/dotnet/skills/commit/ac8f41264bdd557e58924a3110eea8e0917dcf4d",
|
|
"author": {
|
|
"name": "Amaury Levé",
|
|
"username": "AbhitejJohn"
|
|
}
|
|
},
|
|
"skills": [
|
|
{
|
|
"skill": "technology-selection",
|
|
"activationExpected": 6,
|
|
"baseline": {
|
|
"timeMs": 88293.25,
|
|
"tokens": 264019.75,
|
|
"cacheRead": 247166.25,
|
|
"n": 4,
|
|
"tokensOut": 5069.75,
|
|
"cacheWrite": 11759.75,
|
|
"tokensIn": 258950.0
|
|
},
|
|
"activationFired": 6,
|
|
"timedOut": false,
|
|
"baseAvailable": 4,
|
|
"baselineFail": 1,
|
|
"treatmentFail": 0,
|
|
"pairedN": 4,
|
|
"hasPassData": true,
|
|
"treatment": {
|
|
"timeMs": 100247.25,
|
|
"tokens": 381294.25,
|
|
"cacheRead": 356610.0,
|
|
"n": 4,
|
|
"tokensOut": 6302.75,
|
|
"cacheWrite": 18356.0,
|
|
"tokensIn": 374991.5
|
|
},
|
|
"treatAvailable": 6,
|
|
"passTotal": 4
|
|
}
|
|
],
|
|
"date": 1788715663667,
|
|
"judgeModel": "gpt-5.6-terra",
|
|
"model": "claude-opus-5"
|
|
},
|
|
{
|
|
"commit": {
|
|
"id": "ac8f41264bdd557e58924a3110eea8e0917dcf4d",
|
|
"timestamp": "2026-09-04T10:21:25+00:00",
|
|
"message": "Improve non-passing dotnet-test scenarios (#1114)",
|
|
"url": "https://github.com/dotnet/skills/commit/ac8f41264bdd557e58924a3110eea8e0917dcf4d",
|
|
"author": {
|
|
"name": "Amaury Levé",
|
|
"username": "AbhitejJohn"
|
|
}
|
|
},
|
|
"skills": [
|
|
{
|
|
"skill": "technology-selection",
|
|
"activationExpected": 6,
|
|
"baseline": {
|
|
"timeMs": 115241.5,
|
|
"tokens": 710187.0,
|
|
"cacheRead": 670992.8333333334,
|
|
"n": 6,
|
|
"tokensOut": 8715.5,
|
|
"cacheWrite": 30431.333333333332,
|
|
"tokensIn": 701471.5
|
|
},
|
|
"activationFired": 6,
|
|
"timedOut": false,
|
|
"baseAvailable": 6,
|
|
"baselineFail": 3,
|
|
"treatmentFail": 2,
|
|
"pairedN": 6,
|
|
"hasPassData": true,
|
|
"treatment": {
|
|
"timeMs": 197932.33333333334,
|
|
"tokens": 1190820.8333333333,
|
|
"cacheRead": 1145643.6666666667,
|
|
"n": 6,
|
|
"tokensOut": 12336.666666666666,
|
|
"cacheWrite": 32776.833333333336,
|
|
"tokensIn": 1178484.1666666667
|
|
},
|
|
"treatAvailable": 6,
|
|
"passTotal": 6
|
|
}
|
|
],
|
|
"date": 1788715663752,
|
|
"judgeModel": "gpt-5.6-terra",
|
|
"model": "claude-sonnet-5"
|
|
},
|
|
{
|
|
"commit": {
|
|
"id": "ac8f41264bdd557e58924a3110eea8e0917dcf4d",
|
|
"timestamp": "2026-09-04T10:21:25+00:00",
|
|
"message": "Improve non-passing dotnet-test scenarios (#1114)",
|
|
"url": "https://github.com/dotnet/skills/commit/ac8f41264bdd557e58924a3110eea8e0917dcf4d",
|
|
"author": {
|
|
"name": "Amaury Levé",
|
|
"username": "AbhitejJohn"
|
|
}
|
|
},
|
|
"skills": [
|
|
{
|
|
"skill": "technology-selection",
|
|
"activationExpected": 6,
|
|
"baseline": {
|
|
"timeMs": 88458.83333333333,
|
|
"tokens": 338536.5,
|
|
"cacheRead": 276611.3333333333,
|
|
"n": 6,
|
|
"tokensOut": 5737.666666666667,
|
|
"cacheWrite": 7059.333333333333,
|
|
"tokensIn": 332798.8333333333
|
|
},
|
|
"activationFired": 6,
|
|
"timedOut": false,
|
|
"baseAvailable": 6,
|
|
"baselineFail": 3,
|
|
"treatmentFail": 3,
|
|
"pairedN": 6,
|
|
"hasPassData": true,
|
|
"treatment": {
|
|
"timeMs": 131932.66666666666,
|
|
"tokens": 654170.3333333334,
|
|
"cacheRead": 553062.5,
|
|
"n": 6,
|
|
"tokensOut": 10317.0,
|
|
"cacheWrite": 10218.5,
|
|
"tokensIn": 643853.3333333334
|
|
},
|
|
"treatAvailable": 6,
|
|
"passTotal": 6
|
|
}
|
|
],
|
|
"date": 1788715663815,
|
|
"judgeModel": "claude-opus-4.8",
|
|
"model": "gpt-5.6-sol"
|
|
},
|
|
{
|
|
"commit": {
|
|
"message": "Improve non-passing dotnet-test scenarios (#1114)",
|
|
"id": "ac8f41264bdd557e58924a3110eea8e0917dcf4d",
|
|
"timestamp": "2026-09-04T10:21:25+00:00",
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Amaury Levé"
|
|
},
|
|
"url": "https://github.com/dotnet/skills/commit/ac8f41264bdd557e58924a3110eea8e0917dcf4d"
|
|
},
|
|
"skills": [
|
|
{
|
|
"baseline": {
|
|
"timeMs": 110536.0,
|
|
"tokensOut": 4208.333333333333,
|
|
"cacheRead": 277621.3333333333,
|
|
"cacheWrite": 12605.666666666666,
|
|
"n": 6,
|
|
"tokensIn": 300920.6666666667,
|
|
"tokens": 305129.0
|
|
},
|
|
"treatAvailable": 6,
|
|
"passTotal": 6,
|
|
"hasPassData": true,
|
|
"activationExpected": 6,
|
|
"pairedN": 6,
|
|
"baseAvailable": 6,
|
|
"treatmentFail": 1,
|
|
"activationFired": 6,
|
|
"timedOut": false,
|
|
"treatment": {
|
|
"timeMs": 127654.0,
|
|
"tokensOut": 5482.166666666667,
|
|
"cacheRead": 449948.6666666667,
|
|
"cacheWrite": 17590.5,
|
|
"n": 6,
|
|
"tokensIn": 467563.3333333333,
|
|
"tokens": 473045.5
|
|
},
|
|
"baselineFail": 4,
|
|
"skill": "technology-selection"
|
|
}
|
|
],
|
|
"date": 1788790038987,
|
|
"model": "claude-sonnet-4.6",
|
|
"judgeModel": "gpt-5.6-terra"
|
|
},
|
|
{
|
|
"commit": {
|
|
"message": "Improve non-passing dotnet-test scenarios (#1114)",
|
|
"id": "ac8f41264bdd557e58924a3110eea8e0917dcf4d",
|
|
"timestamp": "2026-09-04T10:21:25+00:00",
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Amaury Levé"
|
|
},
|
|
"url": "https://github.com/dotnet/skills/commit/ac8f41264bdd557e58924a3110eea8e0917dcf4d"
|
|
},
|
|
"skills": [
|
|
{
|
|
"baseline": {
|
|
"timeMs": 81568.0,
|
|
"tokensOut": 5289.166666666667,
|
|
"cacheRead": 273322.6666666667,
|
|
"cacheWrite": 27382.0,
|
|
"n": 6,
|
|
"tokensIn": 300752.1666666667,
|
|
"tokens": 306041.3333333333
|
|
},
|
|
"treatAvailable": 6,
|
|
"passTotal": 6,
|
|
"hasPassData": true,
|
|
"activationExpected": 6,
|
|
"pairedN": 6,
|
|
"baseAvailable": 6,
|
|
"treatmentFail": 3,
|
|
"activationFired": 6,
|
|
"timedOut": false,
|
|
"treatment": {
|
|
"timeMs": 121337.83333333333,
|
|
"tokensOut": 6522.0,
|
|
"cacheRead": 470214.3333333333,
|
|
"cacheWrite": 30665.833333333332,
|
|
"n": 6,
|
|
"tokensIn": 500942.6666666667,
|
|
"tokens": 507464.6666666667
|
|
},
|
|
"baselineFail": 5,
|
|
"skill": "technology-selection"
|
|
}
|
|
],
|
|
"date": 1788790039073,
|
|
"model": "gpt-5.6-luna",
|
|
"judgeModel": "claude-opus-4.8"
|
|
},
|
|
{
|
|
"model": "claude-haiku-4.5",
|
|
"judgeModel": "gpt-5.6-terra",
|
|
"skills": [
|
|
{
|
|
"activationExpected": 6,
|
|
"pairedN": 4,
|
|
"baseAvailable": 6,
|
|
"treatment": {
|
|
"cacheRead": 454570.25,
|
|
"tokensOut": 8786.75,
|
|
"tokensIn": 477399.5,
|
|
"cacheWrite": 22721.0,
|
|
"timeMs": 114901.75,
|
|
"n": 4,
|
|
"tokens": 486186.25
|
|
},
|
|
"timedOut": false,
|
|
"baseline": {
|
|
"cacheRead": 387235.5,
|
|
"tokensOut": 11394.0,
|
|
"tokensIn": 411060.0,
|
|
"cacheWrite": 23730.25,
|
|
"timeMs": 129243.0,
|
|
"n": 4,
|
|
"tokens": 422454.0
|
|
},
|
|
"treatmentFail": 2,
|
|
"skill": "technology-selection",
|
|
"passTotal": 4,
|
|
"baselineFail": 3,
|
|
"treatAvailable": 4,
|
|
"activationFired": 4,
|
|
"hasPassData": true
|
|
}
|
|
],
|
|
"date": 1788885915193,
|
|
"commit": {
|
|
"id": "fbeeafe261b0fe704b9a95c58fd9fdbe34e96964",
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Viktor Hofer"
|
|
},
|
|
"message": "Delete msbuild-server skill (#1123)",
|
|
"timestamp": "2026-09-07T09:11:54+00:00",
|
|
"url": "https://github.com/dotnet/skills/commit/fbeeafe261b0fe704b9a95c58fd9fdbe34e96964"
|
|
}
|
|
},
|
|
{
|
|
"model": "gpt-5.3-codex",
|
|
"judgeModel": "claude-opus-4.8",
|
|
"skills": [
|
|
{
|
|
"activationExpected": 6,
|
|
"pairedN": 6,
|
|
"baseAvailable": 6,
|
|
"treatment": {
|
|
"cacheRead": 10496.0,
|
|
"tokensOut": 430.5,
|
|
"tokensIn": 18938.333333333332,
|
|
"cacheWrite": 0.0,
|
|
"timeMs": 5607.166666666667,
|
|
"n": 6,
|
|
"tokens": 19368.833333333332
|
|
},
|
|
"timedOut": false,
|
|
"baseline": {
|
|
"cacheRead": 11093.333333333334,
|
|
"tokensOut": 678.3333333333334,
|
|
"tokensIn": 19851.0,
|
|
"cacheWrite": 0.0,
|
|
"timeMs": 9071.5,
|
|
"n": 6,
|
|
"tokens": 20529.333333333332
|
|
},
|
|
"treatmentFail": 5,
|
|
"skill": "technology-selection",
|
|
"passTotal": 6,
|
|
"baselineFail": 5,
|
|
"treatAvailable": 6,
|
|
"activationFired": 3,
|
|
"hasPassData": true
|
|
}
|
|
],
|
|
"date": 1788885915259,
|
|
"commit": {
|
|
"id": "fbeeafe261b0fe704b9a95c58fd9fdbe34e96964",
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Viktor Hofer"
|
|
},
|
|
"message": "Delete msbuild-server skill (#1123)",
|
|
"timestamp": "2026-09-07T09:11:54+00:00",
|
|
"url": "https://github.com/dotnet/skills/commit/fbeeafe261b0fe704b9a95c58fd9fdbe34e96964"
|
|
}
|
|
},
|
|
{
|
|
"model": "mai-code-1-flash-picker",
|
|
"judgeModel": "gpt-5.6-terra",
|
|
"skills": [
|
|
{
|
|
"activationExpected": 6,
|
|
"pairedN": 5,
|
|
"baseAvailable": 6,
|
|
"treatment": {
|
|
"cacheRead": 543923.2,
|
|
"tokensOut": 11082.6,
|
|
"tokensIn": 587148.8,
|
|
"cacheWrite": 0.0,
|
|
"timeMs": 142815.6,
|
|
"n": 5,
|
|
"tokens": 598231.4
|
|
},
|
|
"timedOut": false,
|
|
"baseline": {
|
|
"cacheRead": 269286.4,
|
|
"tokensOut": 10558.2,
|
|
"tokensIn": 299333.4,
|
|
"cacheWrite": 0.0,
|
|
"timeMs": 101740.2,
|
|
"n": 5,
|
|
"tokens": 309891.6
|
|
},
|
|
"treatmentFail": 4,
|
|
"skill": "technology-selection",
|
|
"passTotal": 5,
|
|
"baselineFail": 4,
|
|
"treatAvailable": 5,
|
|
"activationFired": 3,
|
|
"hasPassData": true
|
|
}
|
|
],
|
|
"date": 1788885915332,
|
|
"commit": {
|
|
"id": "fbeeafe261b0fe704b9a95c58fd9fdbe34e96964",
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Viktor Hofer"
|
|
},
|
|
"message": "Delete msbuild-server skill (#1123)",
|
|
"timestamp": "2026-09-07T09:11:54+00:00",
|
|
"url": "https://github.com/dotnet/skills/commit/fbeeafe261b0fe704b9a95c58fd9fdbe34e96964"
|
|
}
|
|
},
|
|
{
|
|
"commit": {
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Abhitej John"
|
|
},
|
|
"message": "Merge pull request #1149 from dotnet/abhitejjohn-vally-014-ci-proof",
|
|
"url": "https://github.com/dotnet/skills/commit/a8fece0fce5f8b0737754a332a1a7c6487e8f927",
|
|
"timestamp": "2026-09-09T17:53:08+00:00",
|
|
"id": "a8fece0fce5f8b0737754a332a1a7c6487e8f927"
|
|
},
|
|
"model": "claude-opus-4.8",
|
|
"skills": [
|
|
{
|
|
"treatmentFail": 0,
|
|
"skill": "technology-selection",
|
|
"hasPassData": true,
|
|
"baseAvailable": 1,
|
|
"baselineFail": 0,
|
|
"activationFired": 6,
|
|
"timedOut": false,
|
|
"treatAvailable": 6,
|
|
"passTotal": 1,
|
|
"baseline": {
|
|
"cacheWrite": 38188.0,
|
|
"timeMs": 262314.0,
|
|
"n": 1,
|
|
"tokens": 946114.0,
|
|
"cacheRead": 892513.0,
|
|
"tokensOut": 15351.0,
|
|
"tokensIn": 930763.0
|
|
},
|
|
"treatment": {
|
|
"cacheWrite": 33785.0,
|
|
"timeMs": 233325.0,
|
|
"n": 1,
|
|
"tokens": 927724.0,
|
|
"cacheRead": 876874.0,
|
|
"tokensOut": 17015.0,
|
|
"tokensIn": 910709.0
|
|
},
|
|
"pairedN": 1,
|
|
"activationExpected": 6
|
|
}
|
|
],
|
|
"date": 1789036360735,
|
|
"judgeModel": "gpt-5.6-terra"
|
|
},
|
|
{
|
|
"model": "gpt-5.6-luna",
|
|
"skills": [
|
|
{
|
|
"treatAvailable": 6,
|
|
"pairedN": 6,
|
|
"skill": "technology-selection",
|
|
"baseAvailable": 6,
|
|
"baselineFail": 4,
|
|
"activationExpected": 6,
|
|
"treatment": {
|
|
"cacheWrite": 30434.666666666668,
|
|
"tokensIn": 547790.0,
|
|
"timeMs": 115756.66666666667,
|
|
"tokens": 555681.8333333334,
|
|
"tokensOut": 7891.833333333333,
|
|
"cacheRead": 517290.3333333333,
|
|
"n": 6
|
|
},
|
|
"timedOut": false,
|
|
"hasPassData": true,
|
|
"treatmentFail": 3,
|
|
"baseline": {
|
|
"cacheWrite": 24474.666666666668,
|
|
"tokensIn": 292802.6666666667,
|
|
"timeMs": 79317.83333333333,
|
|
"tokens": 298150.3333333333,
|
|
"tokensOut": 5347.666666666667,
|
|
"cacheRead": 268284.0,
|
|
"n": 6
|
|
},
|
|
"activationFired": 6,
|
|
"passTotal": 6
|
|
}
|
|
],
|
|
"judgeModel": "claude-opus-4.8",
|
|
"commit": {
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Abhitej John"
|
|
},
|
|
"id": "8a5a42d3e392b402768fc29416831643b79e402b",
|
|
"message": "Merge pull request #1153 from dotnet/abhitejjohn-vally-lock-concurrency-proof",
|
|
"timestamp": "2026-09-10T21:01:20+00:00",
|
|
"url": "https://github.com/dotnet/skills/commit/8a5a42d3e392b402768fc29416831643b79e402b"
|
|
},
|
|
"date": 1789129571363
|
|
},
|
|
{
|
|
"judgeModel": "gpt-5.6-terra",
|
|
"skills": [
|
|
{
|
|
"baselineFail": 3,
|
|
"pairedN": 5,
|
|
"baseAvailable": 6,
|
|
"activationFired": 5,
|
|
"skill": "technology-selection",
|
|
"treatment": {
|
|
"cacheRead": 853540.0,
|
|
"tokensIn": 886325.6,
|
|
"tokens": 901039.0,
|
|
"timeMs": 194886.8,
|
|
"cacheWrite": 32617.4,
|
|
"n": 5,
|
|
"tokensOut": 14713.4
|
|
},
|
|
"activationExpected": 6,
|
|
"timedOut": false,
|
|
"treatAvailable": 5,
|
|
"treatmentFail": 1,
|
|
"passTotal": 5,
|
|
"hasPassData": true,
|
|
"baseline": {
|
|
"cacheRead": 475540.6,
|
|
"tokensIn": 495255.6,
|
|
"tokens": 506230.0,
|
|
"timeMs": 146688.0,
|
|
"cacheWrite": 19605.2,
|
|
"n": 5,
|
|
"tokensOut": 10974.4
|
|
}
|
|
}
|
|
],
|
|
"date": 1789218680600,
|
|
"model": "claude-haiku-4.5",
|
|
"commit": {
|
|
"timestamp": "2026-09-12T00:01:18+00:00",
|
|
"id": "4c72b17fa2c2aaa307d8f1e337ec75cab45ac5c7",
|
|
"message": "Merge pull request #1154 from dotnet/abhitejjohn-agentic-workflow-repair",
|
|
"url": "https://github.com/dotnet/skills/commit/4c72b17fa2c2aaa307d8f1e337ec75cab45ac5c7",
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Abhitej John"
|
|
}
|
|
}
|
|
},
|
|
{
|
|
"judgeModel": "claude-opus-4.8",
|
|
"skills": [
|
|
{
|
|
"baselineFail": 5,
|
|
"pairedN": 6,
|
|
"baseAvailable": 6,
|
|
"activationFired": 3,
|
|
"skill": "technology-selection",
|
|
"treatment": {
|
|
"cacheRead": 15701.333333333334,
|
|
"tokensIn": 24921.166666666668,
|
|
"tokens": 25573.333333333332,
|
|
"timeMs": 6829.333333333333,
|
|
"cacheWrite": 0.0,
|
|
"n": 6,
|
|
"tokensOut": 652.1666666666666
|
|
},
|
|
"activationExpected": 6,
|
|
"timedOut": false,
|
|
"treatAvailable": 6,
|
|
"treatmentFail": 3,
|
|
"passTotal": 6,
|
|
"hasPassData": true,
|
|
"baseline": {
|
|
"cacheRead": 18858.666666666668,
|
|
"tokensIn": 27631.5,
|
|
"tokens": 28574.333333333332,
|
|
"timeMs": 11672.833333333334,
|
|
"cacheWrite": 0.0,
|
|
"n": 6,
|
|
"tokensOut": 942.8333333333334
|
|
}
|
|
}
|
|
],
|
|
"date": 1789218680648,
|
|
"model": "gpt-5.3-codex",
|
|
"commit": {
|
|
"timestamp": "2026-09-12T00:01:18+00:00",
|
|
"id": "4c72b17fa2c2aaa307d8f1e337ec75cab45ac5c7",
|
|
"message": "Merge pull request #1154 from dotnet/abhitejjohn-agentic-workflow-repair",
|
|
"url": "https://github.com/dotnet/skills/commit/4c72b17fa2c2aaa307d8f1e337ec75cab45ac5c7",
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Abhitej John"
|
|
}
|
|
}
|
|
},
|
|
{
|
|
"model": "claude-opus-5",
|
|
"judgeModel": "gpt-5.6-terra",
|
|
"skills": [
|
|
{
|
|
"baseAvailable": 6,
|
|
"treatmentFail": 1,
|
|
"pairedN": 6,
|
|
"hasPassData": true,
|
|
"skill": "technology-selection",
|
|
"treatment": {
|
|
"tokensIn": 637919.5,
|
|
"cacheRead": 614519.0,
|
|
"timeMs": 151977.5,
|
|
"tokens": 648281.0,
|
|
"n": 6,
|
|
"tokensOut": 10361.5,
|
|
"cacheWrite": 23359.833333333332
|
|
},
|
|
"activationExpected": 6,
|
|
"baseline": {
|
|
"tokensIn": 532710.6666666666,
|
|
"cacheRead": 513925.0,
|
|
"timeMs": 145634.83333333334,
|
|
"tokens": 541926.5,
|
|
"n": 6,
|
|
"tokensOut": 9215.833333333334,
|
|
"cacheWrite": 18747.666666666668
|
|
},
|
|
"baselineFail": 3,
|
|
"timedOut": false,
|
|
"passTotal": 6,
|
|
"treatAvailable": 6,
|
|
"activationFired": 6
|
|
}
|
|
],
|
|
"date": 1789318508154,
|
|
"commit": {
|
|
"timestamp": "2026-09-12T07:43:41+00:00",
|
|
"url": "https://github.com/dotnet/skills/commit/4ecd7d9c76fa458807684771c2bfc7acf1e00ad3",
|
|
"message": "Merge pull request #1134 from dotnet/bot/weekly-version-sync",
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Abhitej John"
|
|
},
|
|
"id": "4ecd7d9c76fa458807684771c2bfc7acf1e00ad3"
|
|
}
|
|
},
|
|
{
|
|
"model": "claude-sonnet-5",
|
|
"judgeModel": "gpt-5.6-terra",
|
|
"skills": [
|
|
{
|
|
"baseAvailable": 6,
|
|
"treatmentFail": 2,
|
|
"pairedN": 6,
|
|
"hasPassData": true,
|
|
"skill": "technology-selection",
|
|
"treatment": {
|
|
"tokensIn": 859622.6666666666,
|
|
"cacheRead": 830653.0,
|
|
"timeMs": 133158.5,
|
|
"tokens": 869592.1666666666,
|
|
"n": 6,
|
|
"tokensOut": 9969.5,
|
|
"cacheWrite": 28920.0
|
|
},
|
|
"activationExpected": 6,
|
|
"baseline": {
|
|
"tokensIn": 637235.3333333334,
|
|
"cacheRead": 607538.6666666666,
|
|
"timeMs": 116027.16666666667,
|
|
"tokens": 646381.5,
|
|
"n": 6,
|
|
"tokensOut": 9146.166666666666,
|
|
"cacheWrite": 29652.333333333332
|
|
},
|
|
"baselineFail": 3,
|
|
"timedOut": false,
|
|
"passTotal": 6,
|
|
"treatAvailable": 6,
|
|
"activationFired": 6
|
|
}
|
|
],
|
|
"date": 1789318508226,
|
|
"commit": {
|
|
"timestamp": "2026-09-12T07:43:41+00:00",
|
|
"url": "https://github.com/dotnet/skills/commit/4ecd7d9c76fa458807684771c2bfc7acf1e00ad3",
|
|
"message": "Merge pull request #1134 from dotnet/bot/weekly-version-sync",
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Abhitej John"
|
|
},
|
|
"id": "4ecd7d9c76fa458807684771c2bfc7acf1e00ad3"
|
|
}
|
|
},
|
|
{
|
|
"model": "gpt-5.6-sol",
|
|
"judgeModel": "claude-opus-4.8",
|
|
"skills": [
|
|
{
|
|
"baseAvailable": 6,
|
|
"treatmentFail": 3,
|
|
"pairedN": 6,
|
|
"hasPassData": true,
|
|
"skill": "technology-selection",
|
|
"treatment": {
|
|
"tokensIn": 670688.0,
|
|
"cacheRead": 578901.3333333334,
|
|
"timeMs": 133439.0,
|
|
"tokens": 680385.1666666666,
|
|
"n": 6,
|
|
"tokensOut": 9697.166666666666,
|
|
"cacheWrite": 0.0
|
|
},
|
|
"activationExpected": 6,
|
|
"baseline": {
|
|
"tokensIn": 366308.6666666667,
|
|
"cacheRead": 313429.3333333333,
|
|
"timeMs": 94694.16666666667,
|
|
"tokens": 373247.8333333333,
|
|
"n": 6,
|
|
"tokensOut": 6939.166666666667,
|
|
"cacheWrite": 0.0
|
|
},
|
|
"baselineFail": 4,
|
|
"timedOut": false,
|
|
"passTotal": 6,
|
|
"treatAvailable": 6,
|
|
"activationFired": 6
|
|
}
|
|
],
|
|
"date": 1789318508294,
|
|
"commit": {
|
|
"timestamp": "2026-09-12T07:43:41+00:00",
|
|
"url": "https://github.com/dotnet/skills/commit/4ecd7d9c76fa458807684771c2bfc7acf1e00ad3",
|
|
"message": "Merge pull request #1134 from dotnet/bot/weekly-version-sync",
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Abhitej John"
|
|
},
|
|
"id": "4ecd7d9c76fa458807684771c2bfc7acf1e00ad3"
|
|
}
|
|
},
|
|
{
|
|
"date": 1789397101175,
|
|
"model": "claude-sonnet-5",
|
|
"commit": {
|
|
"url": "https://github.com/dotnet/skills/commit/4ecd7d9c76fa458807684771c2bfc7acf1e00ad3",
|
|
"timestamp": "2026-09-12T07:43:41+00:00",
|
|
"id": "4ecd7d9c76fa458807684771c2bfc7acf1e00ad3",
|
|
"message": "Merge pull request #1134 from dotnet/bot/weekly-version-sync",
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Abhitej John"
|
|
}
|
|
},
|
|
"judgeModel": "gpt-5.6-terra",
|
|
"skills": [
|
|
{
|
|
"hasPassData": true,
|
|
"baselineFail": 4,
|
|
"activationFired": 6,
|
|
"treatment": {
|
|
"tokensOut": 12889.833333333334,
|
|
"n": 6,
|
|
"timeMs": 167066.0,
|
|
"tokens": 1123912.1666666667,
|
|
"cacheWrite": 32773.666666666664,
|
|
"cacheRead": 1078189.6666666667,
|
|
"tokensIn": 1111022.3333333333
|
|
},
|
|
"activationExpected": 6,
|
|
"baseline": {
|
|
"tokensOut": 7333.666666666667,
|
|
"n": 6,
|
|
"timeMs": 89610.16666666667,
|
|
"tokens": 403869.6666666667,
|
|
"cacheWrite": 16815.833333333332,
|
|
"cacheRead": 379688.1666666667,
|
|
"tokensIn": 396536.0
|
|
},
|
|
"pairedN": 6,
|
|
"baseAvailable": 6,
|
|
"treatmentFail": 2,
|
|
"timedOut": false,
|
|
"skill": "technology-selection",
|
|
"treatAvailable": 6,
|
|
"passTotal": 6
|
|
}
|
|
]
|
|
},
|
|
{
|
|
"date": 1789397101227,
|
|
"model": "gpt-5.6-luna",
|
|
"commit": {
|
|
"url": "https://github.com/dotnet/skills/commit/4ecd7d9c76fa458807684771c2bfc7acf1e00ad3",
|
|
"timestamp": "2026-09-12T07:43:41+00:00",
|
|
"id": "4ecd7d9c76fa458807684771c2bfc7acf1e00ad3",
|
|
"message": "Merge pull request #1134 from dotnet/bot/weekly-version-sync",
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Abhitej John"
|
|
}
|
|
},
|
|
"judgeModel": "claude-opus-4.8",
|
|
"skills": [
|
|
{
|
|
"hasPassData": true,
|
|
"baselineFail": 4,
|
|
"activationFired": 6,
|
|
"treatment": {
|
|
"tokensOut": 7846.833333333333,
|
|
"n": 6,
|
|
"timeMs": 102984.33333333333,
|
|
"tokens": 508340.8333333333,
|
|
"cacheWrite": 27439.5,
|
|
"cacheRead": 472992.5,
|
|
"tokensIn": 500494.0
|
|
},
|
|
"activationExpected": 6,
|
|
"baseline": {
|
|
"tokensOut": 5823.833333333333,
|
|
"n": 6,
|
|
"timeMs": 115163.0,
|
|
"tokens": 347052.1666666667,
|
|
"cacheWrite": 24579.333333333332,
|
|
"cacheRead": 316595.0,
|
|
"tokensIn": 341228.3333333333
|
|
},
|
|
"pairedN": 6,
|
|
"baseAvailable": 6,
|
|
"treatmentFail": 3,
|
|
"timedOut": false,
|
|
"skill": "technology-selection",
|
|
"treatAvailable": 6,
|
|
"passTotal": 6
|
|
}
|
|
]
|
|
},
|
|
{
|
|
"judgeModel": "gpt-5.6-terra",
|
|
"model": "claude-haiku-4.5",
|
|
"date": 1789478285332,
|
|
"commit": {
|
|
"url": "https://github.com/dotnet/skills/commit/24f7cfbd42ad7bf52bcd67372816b982c38c64c6",
|
|
"author": {
|
|
"name": "Amaury Levé",
|
|
"username": "AbhitejJohn"
|
|
},
|
|
"timestamp": "2026-09-14T15:27:40+00:00",
|
|
"id": "24f7cfbd42ad7bf52bcd67372816b982c38c64c6",
|
|
"message": "Surface activation-only evaluation failures (#1163)"
|
|
},
|
|
"skills": [
|
|
{
|
|
"pairedN": 5,
|
|
"treatment": {
|
|
"tokens": 480950.8,
|
|
"cacheWrite": 22839.4,
|
|
"n": 5,
|
|
"timeMs": 128100.6,
|
|
"tokensIn": 471028.8,
|
|
"cacheRead": 448090.4,
|
|
"tokensOut": 9922.0
|
|
},
|
|
"baseAvailable": 6,
|
|
"baselineFail": 4,
|
|
"baseline": {
|
|
"tokens": 633444.4,
|
|
"cacheWrite": 25752.2,
|
|
"n": 5,
|
|
"timeMs": 150837.8,
|
|
"tokensIn": 621064.6,
|
|
"cacheRead": 595197.8,
|
|
"tokensOut": 12379.8
|
|
},
|
|
"activationExpected": 6,
|
|
"activationFired": 2,
|
|
"treatAvailable": 5,
|
|
"skill": "technology-selection",
|
|
"treatmentFail": 3,
|
|
"timedOut": false,
|
|
"hasPassData": true,
|
|
"passTotal": 5
|
|
}
|
|
]
|
|
},
|
|
{
|
|
"judgeModel": "claude-opus-4.8",
|
|
"model": "gpt-5.3-codex",
|
|
"date": 1789478285409,
|
|
"commit": {
|
|
"url": "https://github.com/dotnet/skills/commit/24f7cfbd42ad7bf52bcd67372816b982c38c64c6",
|
|
"author": {
|
|
"name": "Amaury Levé",
|
|
"username": "AbhitejJohn"
|
|
},
|
|
"timestamp": "2026-09-14T15:27:40+00:00",
|
|
"id": "24f7cfbd42ad7bf52bcd67372816b982c38c64c6",
|
|
"message": "Surface activation-only evaluation failures (#1163)"
|
|
},
|
|
"skills": [
|
|
{
|
|
"pairedN": 6,
|
|
"treatment": {
|
|
"tokens": 33473.5,
|
|
"cacheWrite": 0.0,
|
|
"n": 6,
|
|
"timeMs": 9450.5,
|
|
"tokensIn": 32710.166666666668,
|
|
"cacheRead": 21418.666666666668,
|
|
"tokensOut": 763.3333333333334
|
|
},
|
|
"baseAvailable": 6,
|
|
"baselineFail": 5,
|
|
"baseline": {
|
|
"tokens": 12080.166666666666,
|
|
"cacheWrite": 0.0,
|
|
"n": 6,
|
|
"timeMs": 7300.5,
|
|
"tokensIn": 11582.0,
|
|
"cacheRead": 4608.0,
|
|
"tokensOut": 498.1666666666667
|
|
},
|
|
"activationExpected": 6,
|
|
"activationFired": 5,
|
|
"treatAvailable": 6,
|
|
"skill": "technology-selection",
|
|
"treatmentFail": 2,
|
|
"timedOut": false,
|
|
"hasPassData": true,
|
|
"passTotal": 6
|
|
}
|
|
]
|
|
},
|
|
{
|
|
"date": 1789574378627,
|
|
"skills": [
|
|
{
|
|
"activationExpected": 6,
|
|
"skill": "technology-selection",
|
|
"baseline": {
|
|
"timeMs": 118868.0,
|
|
"tokensIn": 701961.5,
|
|
"tokens": 710987.0,
|
|
"cacheRead": 679100.8333333334,
|
|
"n": 6,
|
|
"cacheWrite": 22815.666666666668,
|
|
"tokensOut": 9025.5
|
|
},
|
|
"treatAvailable": 6,
|
|
"treatmentFail": 1,
|
|
"passTotal": 6,
|
|
"hasPassData": true,
|
|
"timedOut": false,
|
|
"pairedN": 6,
|
|
"treatment": {
|
|
"timeMs": 154680.16666666666,
|
|
"tokensIn": 1127161.0,
|
|
"tokens": 1139580.1666666667,
|
|
"cacheRead": 1094854.0,
|
|
"n": 6,
|
|
"cacheWrite": 32248.666666666668,
|
|
"tokensOut": 12419.166666666666
|
|
},
|
|
"baselineFail": 3,
|
|
"baseAvailable": 6,
|
|
"activationFired": 6
|
|
}
|
|
],
|
|
"commit": {
|
|
"author": {
|
|
"name": "Abhitej John",
|
|
"username": "AbhitejJohn"
|
|
},
|
|
"timestamp": "2026-09-15T22:05:10+00:00",
|
|
"message": "Merge pull request #873 from dotnet/add-dotnet-refactoring-skills",
|
|
"url": "https://github.com/dotnet/skills/commit/26323a52990d0cbfc838117109b40e115aac891f",
|
|
"id": "26323a52990d0cbfc838117109b40e115aac891f"
|
|
},
|
|
"judgeModel": "gpt-5.6-terra",
|
|
"model": "claude-sonnet-5"
|
|
},
|
|
{
|
|
"date": 1789574378723,
|
|
"skills": [
|
|
{
|
|
"activationExpected": 6,
|
|
"skill": "technology-selection",
|
|
"baseline": {
|
|
"timeMs": 96942.33333333333,
|
|
"tokensIn": 398276.1666666667,
|
|
"tokens": 404271.5,
|
|
"cacheRead": 375386.3333333333,
|
|
"n": 6,
|
|
"cacheWrite": 22831.833333333332,
|
|
"tokensOut": 5995.333333333333
|
|
},
|
|
"treatAvailable": 6,
|
|
"treatmentFail": 3,
|
|
"passTotal": 6,
|
|
"hasPassData": true,
|
|
"timedOut": false,
|
|
"pairedN": 6,
|
|
"treatment": {
|
|
"timeMs": 88004.66666666667,
|
|
"tokensIn": 463241.5,
|
|
"tokens": 470092.1666666667,
|
|
"cacheRead": 437165.5,
|
|
"n": 6,
|
|
"cacheWrite": 26018.0,
|
|
"tokensOut": 6850.666666666667
|
|
},
|
|
"baselineFail": 4,
|
|
"baseAvailable": 6,
|
|
"activationFired": 6
|
|
}
|
|
],
|
|
"commit": {
|
|
"author": {
|
|
"name": "Abhitej John",
|
|
"username": "AbhitejJohn"
|
|
},
|
|
"timestamp": "2026-09-15T22:05:10+00:00",
|
|
"message": "Merge pull request #873 from dotnet/add-dotnet-refactoring-skills",
|
|
"url": "https://github.com/dotnet/skills/commit/26323a52990d0cbfc838117109b40e115aac891f",
|
|
"id": "26323a52990d0cbfc838117109b40e115aac891f"
|
|
},
|
|
"judgeModel": "claude-opus-4.8",
|
|
"model": "gpt-5.6-luna"
|
|
},
|
|
{
|
|
"judgeModel": "gpt-5.6-terra",
|
|
"commit": {
|
|
"message": "Merge pull request #1184 from dotnet/dependabot/npm_and_yarn/eng/evaluation-tools/microsoft/vally-cli-0.16.0",
|
|
"id": "8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3",
|
|
"timestamp": "2026-09-17T03:09:52+00:00",
|
|
"url": "https://github.com/dotnet/skills/commit/8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3",
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Abhitej John"
|
|
}
|
|
},
|
|
"date": 1789642372141,
|
|
"skills": [
|
|
{
|
|
"baselineFail": 4,
|
|
"passTotal": 6,
|
|
"hasPassData": true,
|
|
"activationExpected": 6,
|
|
"baseAvailable": 6,
|
|
"treatment": {
|
|
"cacheRead": 623421.3333333334,
|
|
"timeMs": 144561.66666666666,
|
|
"tokensIn": 646304.5,
|
|
"n": 6,
|
|
"tokensOut": 8758.5,
|
|
"cacheWrite": 22841.833333333332,
|
|
"tokens": 655063.0
|
|
},
|
|
"baseline": {
|
|
"cacheRead": 395158.5,
|
|
"timeMs": 113123.16666666667,
|
|
"tokensIn": 412136.8333333333,
|
|
"n": 6,
|
|
"tokensOut": 7564.5,
|
|
"cacheWrite": 16946.333333333332,
|
|
"tokens": 419701.3333333333
|
|
},
|
|
"treatAvailable": 6,
|
|
"skill": "technology-selection",
|
|
"activationFired": 6,
|
|
"timedOut": false,
|
|
"pairedN": 6,
|
|
"treatmentFail": 2
|
|
}
|
|
],
|
|
"model": "claude-opus-4.8"
|
|
},
|
|
{
|
|
"judgeModel": "gpt-5.6-terra",
|
|
"model": "claude-sonnet-5",
|
|
"skills": [
|
|
{
|
|
"hasPassData": true,
|
|
"activationFired": 6,
|
|
"baselineFail": 3,
|
|
"baseAvailable": 6,
|
|
"timedOut": false,
|
|
"baseline": {
|
|
"cacheRead": 675312.5,
|
|
"cacheWrite": 20086.0,
|
|
"tokensIn": 695444.1666666666,
|
|
"n": 6,
|
|
"tokens": 704458.5,
|
|
"tokensOut": 9014.333333333334,
|
|
"timeMs": 127141.5
|
|
},
|
|
"treatmentFail": 2,
|
|
"treatment": {
|
|
"cacheRead": 884000.0,
|
|
"cacheWrite": 28097.333333333332,
|
|
"tokensIn": 912149.0,
|
|
"n": 6,
|
|
"tokens": 922716.0,
|
|
"tokensOut": 10567.0,
|
|
"timeMs": 140581.83333333334
|
|
},
|
|
"activationExpected": 6,
|
|
"pairedN": 6,
|
|
"treatAvailable": 6,
|
|
"skill": "technology-selection",
|
|
"passTotal": 6
|
|
}
|
|
],
|
|
"commit": {
|
|
"message": "Merge pull request #1184 from dotnet/dependabot/npm_and_yarn/eng/evaluation-tools/microsoft/vally-cli-0.16.0",
|
|
"timestamp": "2026-09-17T03:09:52+00:00",
|
|
"url": "https://github.com/dotnet/skills/commit/8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3",
|
|
"author": {
|
|
"name": "Abhitej John",
|
|
"username": "AbhitejJohn"
|
|
},
|
|
"id": "8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3"
|
|
},
|
|
"date": 1789745909143
|
|
},
|
|
{
|
|
"judgeModel": "claude-opus-4.8",
|
|
"model": "gpt-5.6-luna",
|
|
"skills": [
|
|
{
|
|
"hasPassData": true,
|
|
"activationFired": 6,
|
|
"baselineFail": 4,
|
|
"baseAvailable": 6,
|
|
"timedOut": false,
|
|
"baseline": {
|
|
"cacheRead": 258725.5,
|
|
"cacheWrite": 22008.833333333332,
|
|
"tokensIn": 280778.3333333333,
|
|
"n": 6,
|
|
"tokens": 285829.6666666667,
|
|
"tokensOut": 5051.333333333333,
|
|
"timeMs": 50809.833333333336
|
|
},
|
|
"treatmentFail": 3,
|
|
"treatment": {
|
|
"cacheRead": 442396.8333333333,
|
|
"cacheWrite": 27518.5,
|
|
"tokensIn": 469973.8333333333,
|
|
"n": 6,
|
|
"tokens": 477321.1666666667,
|
|
"tokensOut": 7347.333333333333,
|
|
"timeMs": 64151.666666666664
|
|
},
|
|
"activationExpected": 6,
|
|
"pairedN": 6,
|
|
"treatAvailable": 6,
|
|
"skill": "technology-selection",
|
|
"passTotal": 6
|
|
}
|
|
],
|
|
"commit": {
|
|
"message": "Merge pull request #1184 from dotnet/dependabot/npm_and_yarn/eng/evaluation-tools/microsoft/vally-cli-0.16.0",
|
|
"timestamp": "2026-09-17T03:09:52+00:00",
|
|
"url": "https://github.com/dotnet/skills/commit/8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3",
|
|
"author": {
|
|
"name": "Abhitej John",
|
|
"username": "AbhitejJohn"
|
|
},
|
|
"id": "8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3"
|
|
},
|
|
"date": 1789745909242
|
|
},
|
|
{
|
|
"skills": [
|
|
{
|
|
"treatment": {
|
|
"cacheRead": 428832.0,
|
|
"tokensOut": 6676.75,
|
|
"timeMs": 117880.75,
|
|
"tokensIn": 447757.0,
|
|
"n": 4,
|
|
"tokens": 454433.75,
|
|
"cacheWrite": 18819.25
|
|
},
|
|
"treatAvailable": 5,
|
|
"skill": "technology-selection",
|
|
"activationExpected": 6,
|
|
"baselineFail": 2,
|
|
"treatmentFail": 1,
|
|
"hasPassData": true,
|
|
"baseline": {
|
|
"cacheRead": 406562.75,
|
|
"tokensOut": 6246.0,
|
|
"timeMs": 84852.0,
|
|
"tokensIn": 427559.5,
|
|
"n": 4,
|
|
"tokens": 433805.5,
|
|
"cacheWrite": 20897.0
|
|
},
|
|
"activationFired": 4,
|
|
"timedOut": false,
|
|
"baseAvailable": 5,
|
|
"passTotal": 4,
|
|
"pairedN": 4
|
|
}
|
|
],
|
|
"date": 1789836946087,
|
|
"commit": {
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Abhitej John"
|
|
},
|
|
"message": "Merge pull request #1184 from dotnet/dependabot/npm_and_yarn/eng/evaluation-tools/microsoft/vally-cli-0.16.0",
|
|
"url": "https://github.com/dotnet/skills/commit/8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3",
|
|
"timestamp": "2026-09-17T03:09:52+00:00",
|
|
"id": "8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3"
|
|
},
|
|
"judgeModel": "gpt-5.6-terra",
|
|
"model": "claude-haiku-4.5"
|
|
},
|
|
{
|
|
"skills": [
|
|
{
|
|
"treatment": {
|
|
"cacheRead": 15104.0,
|
|
"tokensOut": 866.8333333333334,
|
|
"timeMs": 6784.666666666667,
|
|
"tokensIn": 23757.666666666668,
|
|
"n": 6,
|
|
"tokens": 24624.5,
|
|
"cacheWrite": 0.0
|
|
},
|
|
"treatAvailable": 6,
|
|
"skill": "technology-selection",
|
|
"activationExpected": 6,
|
|
"baselineFail": 5,
|
|
"treatmentFail": 3,
|
|
"hasPassData": true,
|
|
"baseline": {
|
|
"cacheRead": 4608.0,
|
|
"tokensOut": 355.6666666666667,
|
|
"timeMs": 4794.0,
|
|
"tokensIn": 11582.0,
|
|
"n": 6,
|
|
"tokens": 11937.666666666666,
|
|
"cacheWrite": 0.0
|
|
},
|
|
"activationFired": 2,
|
|
"timedOut": false,
|
|
"baseAvailable": 6,
|
|
"passTotal": 6,
|
|
"pairedN": 6
|
|
}
|
|
],
|
|
"date": 1789836946152,
|
|
"commit": {
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Abhitej John"
|
|
},
|
|
"message": "Merge pull request #1184 from dotnet/dependabot/npm_and_yarn/eng/evaluation-tools/microsoft/vally-cli-0.16.0",
|
|
"url": "https://github.com/dotnet/skills/commit/8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3",
|
|
"timestamp": "2026-09-17T03:09:52+00:00",
|
|
"id": "8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3"
|
|
},
|
|
"judgeModel": "claude-opus-4.8",
|
|
"model": "gpt-5.3-codex"
|
|
},
|
|
{
|
|
"skills": [
|
|
{
|
|
"treatment": {
|
|
"cacheRead": 403507.2,
|
|
"tokensOut": 11869.8,
|
|
"timeMs": 80518.2,
|
|
"tokensIn": 463054.0,
|
|
"n": 5,
|
|
"tokens": 474923.8,
|
|
"cacheWrite": 0.0
|
|
},
|
|
"treatAvailable": 6,
|
|
"skill": "technology-selection",
|
|
"activationExpected": 6,
|
|
"baselineFail": 3,
|
|
"treatmentFail": 3,
|
|
"hasPassData": true,
|
|
"baseline": {
|
|
"cacheRead": 145049.6,
|
|
"tokensOut": 5992.4,
|
|
"timeMs": 47122.8,
|
|
"tokensIn": 169338.0,
|
|
"n": 5,
|
|
"tokens": 175330.4,
|
|
"cacheWrite": 0.0
|
|
},
|
|
"activationFired": 1,
|
|
"timedOut": false,
|
|
"baseAvailable": 5,
|
|
"passTotal": 5,
|
|
"pairedN": 5
|
|
}
|
|
],
|
|
"date": 1789836946226,
|
|
"commit": {
|
|
"author": {
|
|
"username": "AbhitejJohn",
|
|
"name": "Abhitej John"
|
|
},
|
|
"message": "Merge pull request #1184 from dotnet/dependabot/npm_and_yarn/eng/evaluation-tools/microsoft/vally-cli-0.16.0",
|
|
"url": "https://github.com/dotnet/skills/commit/8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3",
|
|
"timestamp": "2026-09-17T03:09:52+00:00",
|
|
"id": "8bbfe7a4d1c5c0cd42cd04e38031779c75f2dda3"
|
|
},
|
|
"judgeModel": "gpt-5.6-terra",
|
|
"model": "mai-code-1.1-flash"
|
|
}
|
|
]
|
|
},
|
|
"lastUpdate": 1789836946226,
|
|
"repoUrl": "",
|
|
"schemaVersion": 2
|
|
}
|