{
  "schemaVersion": 1,
  "studyId": "felan-extensions-2026-09",
  "title": "Felan extension benchmark results \u2014 September 2026",
  "status": "complete",
  "methodology": {
    "outcomes": "Primary and secondary additive resource outcomes use the configured trial reducer within each case, then the ratio of reduced case sums. Legacy macro-mean case gains and ranges remain diagnostics.",
    "quality": "Quality is the candidate aggregate pass rate; baseline values and percentage-point changes remain available in the benchmark quality data.",
    "overallScore": null,
    "aggregation": {
      "headline": "ratioOfReducedSums",
      "trialReducer": "median",
      "diagnostics": "macroMean case-level gains and ranges",
      "zeroBaseline": "percentage unavailable"
    }
  },
  "primaryBenchmarks": [
    {
      "cohortId": "local-subagents",
      "benchmarkId": "subagents",
      "label": "Subagents organic exploration",
      "evidenceRole": "primary",
      "evidenceType": "recalculated-direct",
      "status": "eligible",
      "provider": [
        "openai-codex"
      ],
      "model": [
        "gpt-5.6-sol"
      ],
      "thinking": [
        "max"
      ],
      "felanVersion": [
        "0.21.1"
      ],
      "cases": [
        "subagents-codebase-exploration"
      ],
      "trials": 3,
      "aggregation": {
        "trials": "median",
        "cases": "ratioOfReducedSums"
      },
      "quality": {
        "baseline": {
          "runs": 3,
          "validRuns": 3,
          "passed": 3,
          "failed": 0,
          "invalid": 0,
          "state": "eligible",
          "gateResults": [
            {
              "metric": "quality.passRate",
              "min": 0.8,
              "value": 1,
              "pass": true
            }
          ]
        },
        "candidate": {
          "runs": 3,
          "validRuns": 3,
          "passed": 3,
          "failed": 0,
          "invalid": 0,
          "state": "eligible",
          "gateResults": [
            {
              "metric": "quality.passRate",
              "min": 0.8,
              "value": 1,
              "pass": true
            }
          ]
        }
      },
      "objectives": [
        {
          "metric": "cost.total",
          "role": "primary",
          "goal": "minimize",
          "status": "recalculated",
          "baselineValue": 1.2856670000000001,
          "candidateValue": 0.9811268400000001,
          "averageChangePercent": -23.68732805617629,
          "averageGainPercent": 23.68732805617629,
          "minGainPercent": 23.68732805617629,
          "maxGainPercent": 23.68732805617629,
          "comparedCases": 1,
          "expectedCases": 1,
          "evidence": "sources/provenance/subagents-cost-recalculation.json",
          "macroMeanBaselineValue": 1.2856670000000001,
          "macroMeanCandidateValue": 0.9811268400000001,
          "macroMeanChangePercent": -23.68732805617629,
          "macroMeanGainPercent": 23.68732805617629,
          "macroMinGainPercent": 23.68732805617629,
          "macroMaxGainPercent": 23.68732805617629,
          "aggregateBaselineValue": 1.2856670000000001,
          "aggregateCandidateValue": 0.9811268400000001,
          "aggregateChangePercent": -23.68732805617629,
          "aggregateGainPercent": 23.68732805617629,
          "headlineEstimator": "ratioOfReducedSums"
        }
      ],
      "selectedBatchId": "20260903-052334-7809",
      "sourceBatchId": "20260903-052334-7809",
      "reverificationId": null,
      "publicationEligible": true,
      "notes": [
        "Cost was recalculated by adding billed usage from all nine Luna child sessions to their corresponding root runs before aggregation.",
        "The prompt-token secondary objective is omitted."
      ],
      "qualityComparison": {
        "metric": "quality.passRate",
        "baselineValue": 1.0,
        "candidateValue": 1.0,
        "changePercentagePoints": 0.0
      },
      "report": "benchmarks/subagents/results.html"
    },
    {
      "cohortId": "local-markitdown-cost",
      "benchmarkId": "markitdown-cost",
      "label": "MarkItDown conversion",
      "evidenceRole": "primary",
      "evidenceType": "offline-reverification",
      "status": "eligible",
      "provider": [
        "openai-codex"
      ],
      "model": [
        "gpt-5.6-terra"
      ],
      "thinking": [
        "low"
      ],
      "felanVersion": [
        "0.19.0"
      ],
      "cases": [
        "markitdown-document-configuration",
        "markitdown-docx-extraction",
        "markitdown-pptx-extraction",
        "markitdown-xlsx-extraction"
      ],
      "trials": 3,
      "aggregation": {
        "trials": "median",
        "cases": "ratioOfReducedSums"
      },
      "quality": {
        "baseline": {
          "runs": 12,
          "validRuns": 12,
          "passed": 12,
          "failed": 0,
          "invalid": 0,
          "state": "eligible",
          "gateResults": [
            {
              "metric": "quality.passRate",
              "min": 0.8,
              "value": 1,
              "pass": true
            }
          ]
        },
        "candidate": {
          "runs": 12,
          "validRuns": 12,
          "passed": 12,
          "failed": 0,
          "invalid": 0,
          "state": "eligible",
          "gateResults": [
            {
              "metric": "quality.passRate",
              "min": 0.8,
              "value": 1,
              "pass": true
            }
          ]
        }
      },
      "objectives": [
        {
          "metric": "cost.total",
          "role": "primary",
          "goal": "minimize",
          "baselineValue": 0.013873299999999998,
          "candidateValue": 0.0095683,
          "averageChangePercent": -31.763528811240008,
          "averageGainPercent": 31.763528811240008,
          "minGainPercent": 18.485677936697634,
          "maxGainPercent": 47.760325770796975,
          "comparedCases": 4,
          "expectedCases": 4,
          "macroMeanBaselineValue": 0.013873299999999998,
          "macroMeanCandidateValue": 0.0095683,
          "macroMeanChangePercent": -31.763528811240008,
          "macroMeanGainPercent": 31.763528811240008,
          "macroMinGainPercent": 18.485677936697634,
          "macroMaxGainPercent": 47.760325770796975,
          "aggregateBaselineValue": 0.05549319999999999,
          "aggregateCandidateValue": 0.0382732,
          "aggregateChangePercent": -31.030829002472366,
          "aggregateGainPercent": 31.030829002472366,
          "headlineEstimator": "ratioOfReducedSums"
        },
        {
          "metric": "usage.promptTokens",
          "role": "secondary",
          "goal": "minimize",
          "baselineValue": 5392.25,
          "candidateValue": 4650.75,
          "averageChangePercent": -14.854257660239346,
          "averageGainPercent": 14.854257660239346,
          "minGainPercent": -6.4,
          "maxGainPercent": 32.65392781316348,
          "comparedCases": 4,
          "expectedCases": 4,
          "macroMeanBaselineValue": 5392.25,
          "macroMeanCandidateValue": 4650.75,
          "macroMeanChangePercent": -14.854257660239346,
          "macroMeanGainPercent": 14.854257660239346,
          "macroMinGainPercent": -6.4,
          "macroMaxGainPercent": 32.65392781316348,
          "aggregateBaselineValue": 21569,
          "aggregateCandidateValue": 18603,
          "aggregateChangePercent": -13.751217024433215,
          "aggregateGainPercent": 13.751217024433215,
          "headlineEstimator": "ratioOfReducedSums"
        }
      ],
      "selectedBatchId": "20260903-045947-5857",
      "sourceBatchId": "20260902-112808-33e8",
      "reverificationId": "20260903-045947-56b5",
      "publicationEligible": false,
      "report": "benchmarks/markitdown-cost/results.html",
      "json": "benchmarks/markitdown-cost/results.json",
      "notes": [],
      "qualityComparison": {
        "metric": "quality.passRate",
        "baselineValue": 1.0,
        "candidateValue": 1.0,
        "changePercentagePoints": 0.0
      }
    },
    {
      "cohortId": "local-output-style-concise",
      "benchmarkId": "output-style-concise",
      "label": "Concise output style",
      "evidenceRole": "primary",
      "evidenceType": "offline-reverification",
      "status": "eligible",
      "provider": [
        "openai-codex"
      ],
      "model": [
        "gpt-5.6-sol"
      ],
      "thinking": [
        "max"
      ],
      "felanVersion": [
        "0.20.2"
      ],
      "cases": [
        "output-style-planning",
        "output-style-review",
        "output-style-support"
      ],
      "trials": 3,
      "aggregation": {
        "trials": "median",
        "cases": "ratioOfReducedSums"
      },
      "quality": {
        "baseline": {
          "runs": 9,
          "validRuns": 9,
          "passed": 9,
          "failed": 0,
          "invalid": 0,
          "state": "eligible",
          "gateResults": [
            {
              "metric": "quality.passRate",
              "min": 0.8,
              "value": 1,
              "pass": true
            }
          ]
        },
        "candidate": {
          "runs": 9,
          "validRuns": 9,
          "passed": 9,
          "failed": 0,
          "invalid": 0,
          "state": "eligible",
          "gateResults": [
            {
              "metric": "quality.passRate",
              "min": 0.8,
              "value": 1,
              "pass": true
            }
          ]
        }
      },
      "objectives": [
        {
          "metric": "cost.total",
          "role": "primary",
          "goal": "minimize",
          "baselineValue": 0.076516,
          "candidateValue": 0.06541933333333333,
          "averageChangePercent": -14.890486501973443,
          "averageGainPercent": 14.890486501973443,
          "minGainPercent": 6.985581490433768,
          "maxGainPercent": 19.557637499242023,
          "comparedCases": 3,
          "expectedCases": 3,
          "macroMeanBaselineValue": 0.076516,
          "macroMeanCandidateValue": 0.06541933333333333,
          "macroMeanChangePercent": -14.890486501973443,
          "macroMeanGainPercent": 14.890486501973443,
          "macroMinGainPercent": 6.985581490433768,
          "macroMaxGainPercent": 19.557637499242023,
          "aggregateBaselineValue": 0.229548,
          "aggregateCandidateValue": 0.196258,
          "aggregateChangePercent": -14.502413438583659,
          "aggregateGainPercent": 14.502413438583659,
          "headlineEstimator": "ratioOfReducedSums"
        },
        {
          "metric": "usage.outputTokens",
          "role": "secondary",
          "goal": "minimize",
          "baselineValue": 1395.6666666666667,
          "candidateValue": 1166.6666666666667,
          "averageChangePercent": -15.981398627528208,
          "averageGainPercent": 15.981398627528208,
          "minGainPercent": 12.088974854932303,
          "maxGainPercent": 21.218074656188605,
          "comparedCases": 3,
          "expectedCases": 3,
          "macroMeanBaselineValue": 1395.6666666666667,
          "macroMeanCandidateValue": 1166.6666666666667,
          "macroMeanChangePercent": -15.981398627528208,
          "macroMeanGainPercent": 15.981398627528208,
          "macroMinGainPercent": 12.088974854932303,
          "macroMaxGainPercent": 21.218074656188605,
          "aggregateBaselineValue": 4187,
          "aggregateCandidateValue": 3500,
          "aggregateChangePercent": -16.40792930499164,
          "aggregateGainPercent": 16.40792930499164,
          "headlineEstimator": "ratioOfReducedSums"
        }
      ],
      "selectedBatchId": "20260903-045947-ba0e",
      "sourceBatchId": "20260902-121757-89d1",
      "reverificationId": "20260903-045947-56b5",
      "publicationEligible": false,
      "report": "benchmarks/output-style-concise/results.html",
      "json": "benchmarks/output-style-concise/results.json",
      "notes": [],
      "qualityComparison": {
        "metric": "quality.passRate",
        "baselineValue": 1.0,
        "candidateValue": 1.0,
        "changePercentagePoints": 0.0
      }
    },
    {
      "cohortId": "local-prewalk",
      "benchmarkId": "prewalk",
      "label": "Prewalk organic routing cost",
      "evidenceRole": "primary",
      "evidenceType": "offline-reverification",
      "status": "eligible",
      "provider": [
        "openai-codex"
      ],
      "model": [
        "gpt-5.6-sol"
      ],
      "thinking": [
        "max"
      ],
      "felanVersion": [
        "0.20.2"
      ],
      "cases": [
        "memory-summary-links",
        "prewalk-checkout"
      ],
      "trials": 3,
      "aggregation": {
        "trials": "median",
        "cases": "ratioOfReducedSums"
      },
      "quality": {
        "baseline": {
          "runs": 6,
          "validRuns": 6,
          "passed": 6,
          "failed": 0,
          "invalid": 0,
          "state": "eligible",
          "gateResults": [
            {
              "metric": "quality.passRate",
              "min": 0.8,
              "value": 1,
              "pass": true
            }
          ]
        },
        "candidate": {
          "runs": 6,
          "validRuns": 6,
          "passed": 6,
          "failed": 0,
          "invalid": 0,
          "state": "eligible",
          "gateResults": [
            {
              "metric": "quality.passRate",
              "min": 0.8,
              "value": 1,
              "pass": true
            }
          ]
        }
      },
      "objectives": [
        {
          "metric": "cost.total",
          "role": "primary",
          "goal": "minimize",
          "baselineValue": 4.262124999999999,
          "candidateValue": 1.449843,
          "averageChangePercent": -57.56549245867434,
          "averageGainPercent": 57.56549245867434,
          "minGainPercent": 43.50744763963154,
          "maxGainPercent": 71.62353727771713,
          "comparedCases": 2,
          "expectedCases": 2,
          "macroMeanBaselineValue": 4.262124999999999,
          "macroMeanCandidateValue": 1.449843,
          "macroMeanChangePercent": -57.56549245867434,
          "macroMeanGainPercent": 57.56549245867434,
          "macroMinGainPercent": 43.50744763963154,
          "macroMaxGainPercent": 71.62353727771713,
          "aggregateBaselineValue": 8.524249999999999,
          "aggregateCandidateValue": 2.899686,
          "aggregateChangePercent": -65.98309528697538,
          "aggregateGainPercent": 65.98309528697538,
          "headlineEstimator": "ratioOfReducedSums"
        }
      ],
      "selectedBatchId": "20260903-044517-8452",
      "sourceBatchId": "20260902-121757-b16d",
      "reverificationId": "20260903-044517-d1f1",
      "publicationEligible": false,
      "report": "benchmarks/prewalk/results.html",
      "json": "benchmarks/prewalk/results.json",
      "notes": [],
      "qualityComparison": {
        "metric": "quality.passRate",
        "baselineValue": 1.0,
        "candidateValue": 1.0,
        "changePercentagePoints": 0.0
      }
    },
    {
      "cohortId": "local-rtk",
      "benchmarkId": "rtk",
      "label": "RTK context reduction",
      "evidenceRole": "primary",
      "evidenceType": "offline-reverification",
      "status": "eligible-with-failure",
      "provider": [
        "openai-codex"
      ],
      "model": [
        "gpt-5.6-sol"
      ],
      "thinking": [
        "max"
      ],
      "felanVersion": [
        "0.20.0"
      ],
      "cases": [
        "prewalk-checkout",
        "rtk-felan-project-instructions"
      ],
      "trials": 3,
      "aggregation": {
        "trials": "median",
        "cases": "ratioOfReducedSums"
      },
      "quality": {
        "baseline": {
          "runs": 6,
          "validRuns": 6,
          "passed": 6,
          "failed": 0,
          "invalid": 0,
          "state": "eligible",
          "gateResults": [
            {
              "metric": "quality.passRate",
              "min": 0.8,
              "value": 1,
              "pass": true
            }
          ]
        },
        "candidate": {
          "runs": 6,
          "validRuns": 6,
          "passed": 5,
          "failed": 1,
          "invalid": 0,
          "state": "eligible",
          "gateResults": [
            {
              "metric": "quality.passRate",
              "min": 0.8,
              "value": 0.8333333333333334,
              "pass": true
            }
          ]
        }
      },
      "objectives": [
        {
          "metric": "cost.total",
          "role": "primary",
          "goal": "minimize",
          "baselineValue": 3.1299044999999994,
          "candidateValue": 2.2984535000000004,
          "averageChangePercent": -24.12217601810633,
          "averageGainPercent": 24.12217601810633,
          "minGainPercent": 15.397141844201368,
          "maxGainPercent": 32.84721019201129,
          "comparedCases": 2,
          "expectedCases": 2,
          "macroMeanBaselineValue": 3.1299044999999994,
          "macroMeanCandidateValue": 2.2984535000000004,
          "macroMeanChangePercent": -24.12217601810633,
          "macroMeanGainPercent": 24.12217601810633,
          "macroMinGainPercent": 15.397141844201368,
          "macroMaxGainPercent": 32.84721019201129,
          "aggregateBaselineValue": 6.259808999999999,
          "aggregateCandidateValue": 4.596907000000001,
          "aggregateChangePercent": -26.564740234086987,
          "aggregateGainPercent": 26.564740234086987,
          "headlineEstimator": "ratioOfReducedSums"
        },
        {
          "metric": "usage.promptTokens",
          "role": "secondary",
          "goal": "minimize",
          "baselineValue": 3620221.5,
          "candidateValue": 2151053.5,
          "averageChangePercent": -34.42765636669536,
          "averageGainPercent": 34.42765636669536,
          "minGainPercent": 21.345442121159167,
          "maxGainPercent": 47.50987061223156,
          "comparedCases": 2,
          "expectedCases": 2,
          "macroMeanBaselineValue": 3620221.5,
          "macroMeanCandidateValue": 2151053.5,
          "macroMeanChangePercent": -34.42765636669536,
          "macroMeanGainPercent": 34.42765636669536,
          "macroMinGainPercent": 21.345442121159167,
          "macroMaxGainPercent": 47.50987061223156,
          "aggregateBaselineValue": 7240443,
          "aggregateCandidateValue": 4302107,
          "aggregateChangePercent": -40.582268239664344,
          "aggregateGainPercent": 40.582268239664344,
          "headlineEstimator": "ratioOfReducedSums"
        }
      ],
      "selectedBatchId": "20260903-045947-e11d",
      "sourceBatchId": "20260902-121847-5fa4",
      "reverificationId": "20260903-045947-56b5",
      "publicationEligible": false,
      "report": "benchmarks/rtk/results.html",
      "json": "benchmarks/rtk/results.json",
      "notes": [
        "One candidate attempt timed out; the configured quality gate still passed.",
        "The displayed quality.passRate arm value of 1 is the macro mean of per-case trial medians; the aggregate candidate pass rate used by the gate is 5/6 (0.8333)."
      ],
      "qualityComparison": {
        "metric": "quality.passRate",
        "baselineValue": 1.0,
        "candidateValue": 0.8333333333333334,
        "changePercentagePoints": -16.666666666666664
      }
    },
    {
      "cohortId": "local-codebase-memory",
      "benchmarkId": "codebase-memory",
      "label": "Codebase Memory retrieval cost",
      "evidenceRole": "primary",
      "evidenceType": "derived-redacted-local-batch",
      "status": "eligible",
      "provider": [
        "openai-codex"
      ],
      "model": [
        "gpt-5.6-sol"
      ],
      "thinking": [
        "medium"
      ],
      "felanVersion": [
        "0.21.11"
      ],
      "cases": [
        "cbm-extension-architecture",
        "cbm-extension-config-scope",
        "cbm-project-instructions"
      ],
      "trials": 3,
      "aggregation": {
        "trials": "median",
        "cases": "ratioOfReducedSums"
      },
      "quality": {
        "baseline": {
          "runs": 9,
          "validRuns": 9,
          "passed": 9,
          "failed": 0,
          "invalid": 0,
          "state": "eligible",
          "gateResults": [
            {
              "metric": "quality.passRate",
              "min": 0.8,
              "value": 1,
              "pass": true
            }
          ]
        },
        "candidate": {
          "runs": 9,
          "validRuns": 9,
          "passed": 9,
          "failed": 0,
          "invalid": 0,
          "state": "eligible",
          "gateResults": [
            {
              "metric": "quality.passRate",
              "min": 0.8,
              "value": 1,
              "pass": true
            }
          ]
        }
      },
      "objectives": [
        {
          "metric": "cost.total",
          "role": "primary",
          "goal": "minimize",
          "baselineValue": 1.7583203333333337,
          "candidateValue": 1.666976666666667,
          "averageChangePercent": 0.1963961754850403,
          "averageGainPercent": -0.1963961754850403,
          "minGainPercent": -18.013738198418086,
          "maxGainPercent": 15.045995647481398,
          "comparedCases": 3,
          "expectedCases": 3,
          "macroMeanBaselineValue": 1.7583203333333337,
          "macroMeanCandidateValue": 1.666976666666667,
          "macroMeanChangePercent": 0.1963961754850403,
          "macroMeanGainPercent": -0.1963961754850403,
          "macroMinGainPercent": -18.013738198418086,
          "macroMaxGainPercent": 15.045995647481398,
          "aggregateBaselineValue": 5.274961000000001,
          "aggregateCandidateValue": 5.000930000000001,
          "aggregateChangePercent": -5.194938882012585,
          "aggregateGainPercent": 5.194938882012585,
          "headlineEstimator": "ratioOfReducedSums",
          "evidence": "sources/provenance/codebase-memory.json"
        },
        {
          "metric": "duration.stepsMs",
          "role": "secondary",
          "goal": "minimize",
          "baselineValue": 531465.6666666666,
          "candidateValue": 515404.3333333333,
          "averageChangePercent": -0.413738232914393,
          "averageGainPercent": 0.413738232914393,
          "minGainPercent": -8.28641114588226,
          "maxGainPercent": 8.3595128385207,
          "comparedCases": 3,
          "expectedCases": 3,
          "macroMeanBaselineValue": 531465.6666666666,
          "macroMeanCandidateValue": 515404.3333333333,
          "macroMeanChangePercent": -0.413738232914393,
          "macroMeanGainPercent": 0.413738232914393,
          "macroMinGainPercent": -8.28641114588226,
          "macroMaxGainPercent": 8.3595128385207,
          "aggregateBaselineValue": 1594397,
          "aggregateCandidateValue": 1546213,
          "aggregateChangePercent": -3.0220829567541836,
          "aggregateGainPercent": 3.0220829567541836,
          "headlineEstimator": "ratioOfReducedSums",
          "evidence": "sources/provenance/codebase-memory.json"
        }
      ],
      "selectedBatchId": "20260904-124556-6ffe",
      "sourceBatchId": "20260904-124556-6ffe",
      "reverificationId": null,
      "publicationEligible": true,
      "report": "benchmarks/codebase-memory/results.html",
      "json": "benchmarks/codebase-memory/results.json",
      "notes": [
        "The published result covers all 18 retained attempts across three cases.",
        "All attempts passed; one candidate architecture attempt received verifier reward 0.8 while the other rewards were 1.0."
      ],
      "qualityComparison": {
        "metric": "quality.passRate",
        "baselineValue": 1.0,
        "candidateValue": 1.0,
        "changePercentagePoints": 0.0
      }
    }
  ],
  "publicationAggregation": {
    "mode": "ratioOfReducedSums",
    "evidence": "sources/provenance/aggregate-gain-recalculation.json",
    "headline": "Aggregate resource outcome; macro case gains remain diagnostics."
  }
}
