{
  "schema": "ainglish.public-experiment-catalog.v1",
  "reviewed_on": "2026-09-06",
  "scope": "Editorial selection of synthetic local research, not proposal-settlement evidence or an independent evaluation.",
  "experiments": [
    {
      "id": "matched-learning",
      "title": "Does training on Ainglish help more than teaching the same ideas in English?",
      "date": "2026-09-06",
      "finding": "No selective benefit established in the small pilot.",
      "design": "One Qwen2.5-7B-Instruct model, one training seed, three model conditions. Read Ainglish without a reference in the prompt. The frozen score weights 96 rows containing 84 distinct cases; it is not 96 independent tasks.",
      "scores": [
        {
          "condition": "Base model, no project training",
          "correct": 72,
          "total": 96,
          "accuracy_pct": 75.0
        },
        {
          "condition": "Trained on Ainglish examples",
          "correct": 77,
          "total": 96,
          "accuracy_pct": 80.20833333333333
        },
        {
          "condition": "Trained on matched English examples",
          "correct": 77,
          "total": 96,
          "accuracy_pct": 80.20833333333333
        }
      ],
      "comparison": {
        "label": "Ainglish-trained minus English-trained, reading Ainglish",
        "delta_pp": 0.0,
        "interval_95": [
          -12.5,
          12.5
        ]
      },
      "guards_passed": false,
      "warnings": [
        "The prespecified boundary screen failed: -5.56 percentage points after Ainglish training versus matched English training.",
        "The interval is exploratory and clustered by 12 authored frames. The later duplicate audit did not change the frozen score."
      ],
      "limits": [
        "One cached model, one seed, small synthetic task families.",
        "Held-out framings and names, not held-out concepts or independent human tasks.",
        "Closed answer selection, not execution of real work. Token counts cover one reading turn only.",
        "No tokenizer change, no external-lab training receipt, no governance progression claim."
      ],
      "source": {
        "commit": "54d8d282762e7904a2441f883a46ca555a3b06aa",
        "sha256": "777010b3778b934ee538323f3ea02527afbc530fdeb15dca20341658f122fb4a",
        "result_url": "https://github.com/dexagon-ai/ainglish-evidence/blob/54d8d282762e7904a2441f883a46ca555a3b06aa/ratified-learning-pilot-2026-09-06/RESULT.json",
        "plan_url": "https://github.com/dexagon-ai/ainglish-evidence/blob/54d8d282762e7904a2441f883a46ca555a3b06aa/ratified-learning-pilot-2026-09-06/PLAN.json",
        "files_url": "https://github.com/dexagon-ai/ainglish-evidence/tree/54d8d282762e7904a2441f883a46ca555a3b06aa/ratified-learning-pilot-2026-09-06/",
        "result_file": "RESULT.json"
      }
    },
    {
      "id": "learning-and-retention",
      "title": "Can a broader teaching set improve unfamiliar wording without losing other skills?",
      "date": "2026-09-06",
      "finding": "An aggregate gain, but two family-level safeguards failed.",
      "design": "One Qwen2.5-7B-Instruct model and one training seed. The primary holdout has 252 distinct cases from 42 authored frames across six ratified families. The full campaign made 5,808 calls across five model conditions and several studies; those calls are not 5,808 independent test cases.",
      "scores": [
        {
          "condition": "Base model, no project training",
          "correct": 177,
          "total": 252,
          "accuracy_pct": 70.23809523809524
        },
        {
          "condition": "Trained on Ainglish examples",
          "correct": 226,
          "total": 252,
          "accuracy_pct": 89.68253968253968
        },
        {
          "condition": "Trained on matched English examples",
          "correct": 198,
          "total": 252,
          "accuracy_pct": 78.57142857142857
        }
      ],
      "comparison": {
        "label": "Ainglish-trained minus English-trained, reading Ainglish",
        "delta_pp": 11.11111111111111,
        "interval_95": [
          2.380952380952381,
          20.238095238095237
        ]
      },
      "guards_passed": false,
      "warnings": [
        "Updating instructions: -6.25 points versus matched English training, below the −5-point screen.",
        "English retention for alternatives: -8.33 points versus the base model, also below that screen.",
        "These point-estimate screens are not statistical proofs of non-inferiority. The aggregate gain does not cancel them."
      ],
      "limits": [
        "Synthetic closed-answer tasks authored within the project, not independent human tasks or a lab replication.",
        "The 95% interval resamples 42 authored frames, not 252 unrelated observations.",
        "One base-model family and a fixed tokenizer. Training weights cannot change that tokenizer’s segmentation.",
        "No demonstrated external adoption, governance settlement or general superiority claim."
      ],
      "source": {
        "commit": "54d8d282762e7904a2441f883a46ca555a3b06aa",
        "sha256": "945f33e674061a068a6984a8fb94410d51d5dd1b09f4028e1f60b32c19357990",
        "result_url": "https://github.com/dexagon-ai/ainglish-evidence/blob/54d8d282762e7904a2441f883a46ca555a3b06aa/usefulness-2026-09-06/RESEARCH-RESULTS.json",
        "plan_url": "https://github.com/dexagon-ai/ainglish-evidence/blob/54d8d282762e7904a2441f883a46ca555a3b06aa/usefulness-2026-09-06/PLAN.json",
        "files_url": "https://github.com/dexagon-ai/ainglish-evidence/tree/54d8d282762e7904a2441f883a46ca555a3b06aa/usefulness-2026-09-06/",
        "result_file": "RESEARCH-RESULTS.json"
      }
    },
    {
      "id": "reasoning-transfer",
      "title": "Does the learning advantage survive harder reasoning and different training seeds?",
      "date": "2026-09-06",
      "finding": "No repeatable advantage established over matched English training.",
      "design": "336 teaching cases per language, then 216 held-out cases from 36 newly authored reasoning frames across six families. Each language was trained with seeds 17, 29 and 43 on the same Qwen2.5-7B-Instruct base. These are three training seeds, not three model families. The wider campaign also tested 120 composition cases per language and condition: 4,704 target answers plus 84 controls.",
      "scores": [
        {
          "condition": "Base model, no project training",
          "correct": 129,
          "total": 216,
          "accuracy_pct": 59.72222222222222
        },
        {
          "condition": "Ainglish training, seed 17",
          "correct": 139,
          "total": 216,
          "accuracy_pct": 64.35185185185185
        },
        {
          "condition": "English training, seed 17",
          "correct": 156,
          "total": 216,
          "accuracy_pct": 72.22222222222223
        },
        {
          "condition": "Ainglish training, seed 29",
          "correct": 151,
          "total": 216,
          "accuracy_pct": 69.9074074074074
        },
        {
          "condition": "English training, seed 29",
          "correct": 150,
          "total": 216,
          "accuracy_pct": 69.44444444444444
        },
        {
          "condition": "Ainglish training, seed 43",
          "correct": 143,
          "total": 216,
          "accuracy_pct": 66.20370370370371
        },
        {
          "condition": "English training, seed 43",
          "correct": 148,
          "total": 216,
          "accuracy_pct": 68.51851851851852
        }
      ],
      "comparisons": [
        {
          "label": "Seed 17: Ainglish-trained minus English-trained, reading Ainglish",
          "delta_pp": -7.87037037037037,
          "interval_95": [
            -15.74074074074074,
            -1.388888888888889
          ]
        },
        {
          "label": "Seed 29: Ainglish-trained minus English-trained, reading Ainglish",
          "delta_pp": 0.4629629629629631,
          "interval_95": [
            -12.5,
            13.88888888888889
          ]
        },
        {
          "label": "Seed 43: Ainglish-trained minus English-trained, reading Ainglish",
          "delta_pp": -2.314814814814815,
          "interval_95": [
            -17.59259259259259,
            13.425925925925926
          ]
        }
      ],
      "guards_passed": false,
      "warnings": [
        "Seed 17 also failed the overall −5-point screen. None of the three seeds passed every family-level screen.",
        "Seed 17, Ainglish reading versus matched English training: alternatives -13.89 points; deadline -5.56 points; multiplicity -5.56 points; participants -5.56 points; unknown -16.67 points. These all fall below the −5-point screen.",
        "Seed 17, English retention versus the base model: alternatives -13.89 points; unknown -50.00 points. These all fall below the −5-point screen.",
        "Seed 29, Ainglish reading versus matched English training: deadline -30.56 points. These all fall below the −5-point screen.",
        "Seed 29, English retention versus the base model: deadline -16.67 points; unknown -33.33 points. These all fall below the −5-point screen.",
        "Seed 43, Ainglish reading versus matched English training: participants -11.11 points; unknown -19.44 points; update -27.78 points. These all fall below the −5-point screen.",
        "Seed 43, English retention versus the base model: unknown -22.22 points. These all fall below the −5-point screen."
      ],
      "limits": [
        "One base model, three training seeds, six ratified families, synthetic authored frames.",
        "Frame bootstrap describes this held-out frame collection, not all language or human understanding.",
        "Every seed and family is retained. English incumbent exposure differs; a future learning hypothesis does not nullify current costs or harm.",
        "A -5pp point screen is not statistical proof of non-inferiority.",
        "The newly authored reasoning frames still draw on known distinctions. Neither new names nor more seeds establish broad transfer.",
        "The changed holdout and curriculum prevent a causal comparison with the earlier +11.11-point study. Its result is preserved rather than overwritten.",
        "Composition results remain mixed and often weak. A correct answer on one distinction does not establish a correct joint plan."
      ],
      "source": {
        "commit": "925ffb89d13e7da6ac901095d72d95448f942199",
        "sha256": "4e2a794843868019b4eff4435a66f218b556f24cb195962e10ef2dfbfb62e092",
        "result_url": "https://github.com/dexagon-ai/ainglish-evidence/blob/925ffb89d13e7da6ac901095d72d95448f942199/learning-transfer-2026-09-06/RESULTS.json",
        "plan_url": "https://github.com/dexagon-ai/ainglish-evidence/blob/925ffb89d13e7da6ac901095d72d95448f942199/learning-transfer-2026-09-06/PLAN.json",
        "files_url": "https://github.com/dexagon-ai/ainglish-evidence/tree/925ffb89d13e7da6ac901095d72d95448f942199/learning-transfer-2026-09-06/",
        "result_file": "RESULTS.json"
      }
    }
  ]
}
