{
  "schema": "ainglish.public-experiment-catalog.v1",
  "reviewed_on": "2026-09-07",
  "scope": "Editorial selection of synthetic local research, not proposal-settlement evidence or an independent evaluation.",
  "experiments": [
    {
      "id": "matched-learning",
      "title": "Does training on Ainglish help more than teaching the same ideas in English?",
      "date": "2026-09-06",
      "finding": "No selective benefit established in the small pilot.",
      "design": "One Qwen2.5-7B-Instruct model, one training seed, three model conditions. Read Ainglish without a reference in the prompt. The frozen score weights 96 rows containing 84 distinct cases; it is not 96 independent tasks.",
      "scores": [
        {
          "condition": "Base model, no project training",
          "correct": 72,
          "total": 96,
          "accuracy_pct": 75.0
        },
        {
          "condition": "Trained on Ainglish examples",
          "correct": 77,
          "total": 96,
          "accuracy_pct": 80.20833333333333
        },
        {
          "condition": "Trained on matched English examples",
          "correct": 77,
          "total": 96,
          "accuracy_pct": 80.20833333333333
        }
      ],
      "comparison": {
        "label": "Ainglish-trained minus English-trained, reading Ainglish",
        "delta_pp": 0.0,
        "interval_95": [
          -12.5,
          12.5
        ]
      },
      "guards_passed": false,
      "warnings": [
        "The prespecified boundary screen failed: -5.56 percentage points after Ainglish training versus matched English training.",
        "The interval is exploratory and clustered by 12 authored frames. The later duplicate audit did not change the frozen score."
      ],
      "limits": [
        "One cached model, one seed, small synthetic task families.",
        "Held-out framings and names, not held-out concepts or independent human tasks.",
        "Closed answer selection, not execution of real work. Token counts cover one reading turn only.",
        "No tokenizer change, no external-lab training receipt, no governance progression claim."
      ],
      "source": {
        "commit": "54d8d282762e7904a2441f883a46ca555a3b06aa",
        "sha256": "777010b3778b934ee538323f3ea02527afbc530fdeb15dca20341658f122fb4a",
        "result_url": "https://github.com/dexagon-ai/ainglish-evidence/blob/54d8d282762e7904a2441f883a46ca555a3b06aa/ratified-learning-pilot-2026-09-06/RESULT.json",
        "plan_url": "https://github.com/dexagon-ai/ainglish-evidence/blob/54d8d282762e7904a2441f883a46ca555a3b06aa/ratified-learning-pilot-2026-09-06/PLAN.json",
        "files_url": "https://github.com/dexagon-ai/ainglish-evidence/tree/54d8d282762e7904a2441f883a46ca555a3b06aa/ratified-learning-pilot-2026-09-06/",
        "result_file": "RESULT.json"
      }
    },
    {
      "id": "learning-and-retention",
      "title": "Can a broader teaching set improve unfamiliar wording without losing other skills?",
      "date": "2026-09-06",
      "finding": "An aggregate gain, but two family-level safeguards failed.",
      "design": "One Qwen2.5-7B-Instruct model and one training seed. The primary holdout has 252 distinct cases from 42 authored frames across six ratified families. The full campaign made 5,808 calls across five model conditions and several studies; those calls are not 5,808 independent test cases.",
      "scores": [
        {
          "condition": "Base model, no project training",
          "correct": 177,
          "total": 252,
          "accuracy_pct": 70.23809523809524
        },
        {
          "condition": "Trained on Ainglish examples",
          "correct": 226,
          "total": 252,
          "accuracy_pct": 89.68253968253968
        },
        {
          "condition": "Trained on matched English examples",
          "correct": 198,
          "total": 252,
          "accuracy_pct": 78.57142857142857
        }
      ],
      "comparison": {
        "label": "Ainglish-trained minus English-trained, reading Ainglish",
        "delta_pp": 11.11111111111111,
        "interval_95": [
          2.380952380952381,
          20.238095238095237
        ]
      },
      "guards_passed": false,
      "warnings": [
        "Updating instructions: -6.25 points versus matched English training, below the −5-point screen.",
        "English retention for alternatives: -8.33 points versus the base model, also below that screen.",
        "These point-estimate screens are not statistical proofs of non-inferiority. The aggregate gain does not cancel them."
      ],
      "limits": [
        "Synthetic closed-answer tasks authored within the project, not independent human tasks or a lab replication.",
        "The 95% interval resamples 42 authored frames, not 252 unrelated observations.",
        "One base-model family and a fixed tokenizer. Training weights cannot change that tokenizer’s segmentation.",
        "No demonstrated external adoption, governance settlement or general superiority claim."
      ],
      "source": {
        "commit": "54d8d282762e7904a2441f883a46ca555a3b06aa",
        "sha256": "945f33e674061a068a6984a8fb94410d51d5dd1b09f4028e1f60b32c19357990",
        "result_url": "https://github.com/dexagon-ai/ainglish-evidence/blob/54d8d282762e7904a2441f883a46ca555a3b06aa/usefulness-2026-09-06/RESEARCH-RESULTS.json",
        "plan_url": "https://github.com/dexagon-ai/ainglish-evidence/blob/54d8d282762e7904a2441f883a46ca555a3b06aa/usefulness-2026-09-06/PLAN.json",
        "files_url": "https://github.com/dexagon-ai/ainglish-evidence/tree/54d8d282762e7904a2441f883a46ca555a3b06aa/usefulness-2026-09-06/",
        "result_file": "RESEARCH-RESULTS.json"
      }
    },
    {
      "id": "reasoning-transfer",
      "title": "Does the learning advantage survive harder reasoning and different training seeds?",
      "date": "2026-09-06",
      "finding": "No repeatable advantage established over matched English training.",
      "design": "336 teaching cases per language, then 216 held-out cases from 36 newly authored reasoning frames across six families. Each language was trained with seeds 17, 29 and 43 on the same Qwen2.5-7B-Instruct base. These are three training seeds, not three model families. The wider campaign also tested 120 composition cases per language and condition: 4,704 target answers plus 84 controls.",
      "scores": [
        {
          "condition": "Base model, no project training",
          "correct": 129,
          "total": 216,
          "accuracy_pct": 59.72222222222222
        },
        {
          "condition": "Ainglish training, seed 17",
          "correct": 139,
          "total": 216,
          "accuracy_pct": 64.35185185185185
        },
        {
          "condition": "English training, seed 17",
          "correct": 156,
          "total": 216,
          "accuracy_pct": 72.22222222222223
        },
        {
          "condition": "Ainglish training, seed 29",
          "correct": 151,
          "total": 216,
          "accuracy_pct": 69.9074074074074
        },
        {
          "condition": "English training, seed 29",
          "correct": 150,
          "total": 216,
          "accuracy_pct": 69.44444444444444
        },
        {
          "condition": "Ainglish training, seed 43",
          "correct": 143,
          "total": 216,
          "accuracy_pct": 66.20370370370371
        },
        {
          "condition": "English training, seed 43",
          "correct": 148,
          "total": 216,
          "accuracy_pct": 68.51851851851852
        }
      ],
      "comparisons": [
        {
          "label": "Seed 17: Ainglish-trained minus English-trained, reading Ainglish",
          "delta_pp": -7.87037037037037,
          "interval_95": [
            -15.74074074074074,
            -1.388888888888889
          ]
        },
        {
          "label": "Seed 29: Ainglish-trained minus English-trained, reading Ainglish",
          "delta_pp": 0.4629629629629631,
          "interval_95": [
            -12.5,
            13.88888888888889
          ]
        },
        {
          "label": "Seed 43: Ainglish-trained minus English-trained, reading Ainglish",
          "delta_pp": -2.314814814814815,
          "interval_95": [
            -17.59259259259259,
            13.425925925925926
          ]
        }
      ],
      "guards_passed": false,
      "warnings": [
        "Seed 17 also failed the overall −5-point screen. None of the three seeds passed every family-level screen.",
        "Seed 17, Ainglish reading versus matched English training: alternatives -13.89 points; deadline -5.56 points; multiplicity -5.56 points; participants -5.56 points; unknown -16.67 points. These all fall below the −5-point screen.",
        "Seed 17, English retention versus the base model: alternatives -13.89 points; unknown -50.00 points. These all fall below the −5-point screen.",
        "Seed 29, Ainglish reading versus matched English training: deadline -30.56 points. These all fall below the −5-point screen.",
        "Seed 29, English retention versus the base model: deadline -16.67 points; unknown -33.33 points. These all fall below the −5-point screen.",
        "Seed 43, Ainglish reading versus matched English training: participants -11.11 points; unknown -19.44 points; update -27.78 points. These all fall below the −5-point screen.",
        "Seed 43, English retention versus the base model: unknown -22.22 points. These all fall below the −5-point screen."
      ],
      "limits": [
        "One base model, three training seeds, six ratified families, synthetic authored frames.",
        "Frame bootstrap describes this held-out frame collection, not all language or human understanding.",
        "Every seed and family is retained. English incumbent exposure differs; a future learning hypothesis does not nullify current costs or harm.",
        "A -5pp point screen is not statistical proof of non-inferiority.",
        "The newly authored reasoning frames still draw on known distinctions. Neither new names nor more seeds establish broad transfer.",
        "The changed holdout and curriculum prevent a causal comparison with the earlier +11.11-point study. Its result is preserved rather than overwritten.",
        "Composition results remain mixed and often weak. A correct answer on one distinction does not establish a correct joint plan."
      ],
      "source": {
        "commit": "925ffb89d13e7da6ac901095d72d95448f942199",
        "sha256": "4e2a794843868019b4eff4435a66f218b556f24cb195962e10ef2dfbfb62e092",
        "result_url": "https://github.com/dexagon-ai/ainglish-evidence/blob/925ffb89d13e7da6ac901095d72d95448f942199/learning-transfer-2026-09-06/RESULTS.json",
        "plan_url": "https://github.com/dexagon-ai/ainglish-evidence/blob/925ffb89d13e7da6ac901095d72d95448f942199/learning-transfer-2026-09-06/PLAN.json",
        "files_url": "https://github.com/dexagon-ai/ainglish-evidence/tree/925ffb89d13e7da6ac901095d72d95448f942199/learning-transfer-2026-09-06/",
        "result_file": "RESULTS.json"
      }
    },
    {
      "id": "sender-receiver",
      "title": "Can a sender and receiver agree on a complete operational plan?",
      "type_label": "Reference-assisted sender–receiver experiment",
      "date": "2026-09-06",
      "finding": "No final exchange passed the declared complete-prose guard.",
      "design": "192 four-call episodes: one 32-pattern scenario in two languages across the base model and two preselected seed-17 adapters. Each episode includes a mandatory clarification. 768 target calls and 24 controls are not independent conversations with different agents.",
      "score_caption": "Final plan matches the brief AND meets the declared completed-prose format",
      "scores": [
        {
          "condition": "Base model · ainglish",
          "correct": 0,
          "total": 32,
          "accuracy_pct": 0.0
        },
        {
          "condition": "Base model · english",
          "correct": 0,
          "total": 32,
          "accuracy_pct": 0.0
        },
        {
          "condition": "Ainglish training seed 17 · ainglish",
          "correct": 0,
          "total": 32,
          "accuracy_pct": 0.0
        },
        {
          "condition": "Ainglish training seed 17 · english",
          "correct": 0,
          "total": 32,
          "accuracy_pct": 0.0
        },
        {
          "condition": "English training seed 17 · ainglish",
          "correct": 0,
          "total": 32,
          "accuracy_pct": 0.0
        },
        {
          "condition": "English training seed 17 · english",
          "correct": 0,
          "total": 32,
          "accuracy_pct": 0.0
        }
      ],
      "comparisons": [],
      "guards_passed": false,
      "warnings": [
        "Raw final-plan accuracy, ignoring the prose guard: base Ainglish 0/32 and English 3/32; Ainglish-trained 0/32 and 0/32; English-trained 2/32 and 1/32. Those are not successful complete-prose exchanges.",
        "Several calls hit their output cap, and sender messages sometimes contradicted the intended brief. A failed interface is not a clean estimate of the language’s inherent usefulness.",
        "Both guides and every clarification turn are included in the cost. Generated token IDs were not retained in this earlier study; do not claim an exact output-token recount."
      ],
      "limits": [
        "One base model, one preselected training seed, single-author synthetic instructions.",
        "Receiver proposes a plan; the runner simulates its correctness without executing external actions.",
        "Both guides are visible; this does not test unaided reading or external adoption.",
        "Tokenizers are fixed; learned weights cannot change segmentation.",
        "One authored scenario with 32 factorial patterns, not 32 independent task families. No real operational work was executed."
      ],
      "source": {
        "commit": "af79c32a68c287937feb537ecbedfb52a19bab29",
        "sha256": "0f06d0a30254eff954e6fbd2dfcedddf7add60775100aa4916509198d3ef1b53",
        "result_url": "https://github.com/dexagon-ai/ainglish-evidence/blob/af79c32a68c287937feb537ecbedfb52a19bab29/sender-receiver-2026-09-06/RESULTS.json",
        "plan_url": "https://github.com/dexagon-ai/ainglish-evidence/blob/af79c32a68c287937feb537ecbedfb52a19bab29/sender-receiver-2026-09-06/PLAN.json",
        "files_url": "https://github.com/dexagon-ai/ainglish-evidence/tree/af79c32a68c287937feb537ecbedfb52a19bab29/sender-receiver-2026-09-06/",
        "result_file": "RESULTS.json"
      }
    },
    {
      "id": "communication-diagnostic",
      "title": "Is the failure in the writer, reader, output format or token budget?",
      "type_label": "Reference-assisted communication diagnostic",
      "date": "2026-09-06",
      "finding": "Larger budgets removed truncation, but the declared training gate still failed.",
      "design": "One untrained Qwen2.5-7B-Instruct reader, three authored contexts, and 1-, 2- and 5-field conditions. Correct reference messages isolate the reader; a fixed-phrase writer and a handoff test follow for five fields. All 612 target calls and 16 controls completed without truncation, with exact input and output token IDs retained.",
      "score_caption": "Exact five-field interpretation of a correctly authored message, with a guide in the prompt",
      "scores": [
        {
          "condition": "Ainglish reference message",
          "correct": 44,
          "total": 96,
          "accuracy_pct": 45.833333333333336
        },
        {
          "condition": "English reference message",
          "correct": 58,
          "total": 96,
          "accuracy_pct": 60.416666666666664
        }
      ],
      "comparisons": [],
      "guards_passed": false,
      "warnings": [
        "Both five-field reader arms fall below the prospectively declared 80% floor. Communication-adapter training was held; a neutral format screen passing 8/8 does not cancel a failed task gate.",
        "The strict one-field score is 0/6 per arm because the model supplied extra keys. All six requested inclusion values per arm were correct in the separately labelled post-hoc audit.",
        "Terminal punctuation explains many unparsed writer outputs. Removing only that punctuation makes all 96 per arm parsable, but only 34/96 Ainglish and 53/96 English messages express the intended five choices. The frozen parser and failed gate remain unchanged."
      ],
      "limits": [
        "A narrower successor diagnostic, not a redefinition of the earlier free-prose result.",
        "Three authored contexts, not 114 independent reasoning templates or different model families.",
        "A fixed phrase table and explicit guides do not establish unaided reading, human understanding, independent replication or governance settlement.",
        "Different tasks, denominators and exposure conditions must not be pooled into one project success rate."
      ],
      "source": {
        "commit": "da975afbf8004f90c1861882953e61a00438c3a0",
        "sha256": "bc0ae885d57dde23f3f7f9f4a02158beabac533bbb72889ff0b1727cd379871d",
        "result_url": "https://github.com/dexagon-ai/ainglish-evidence/blob/da975afbf8004f90c1861882953e61a00438c3a0/communication-diagnostics-2026-09-06/RESULTS.json",
        "plan_url": "https://github.com/dexagon-ai/ainglish-evidence/blob/da975afbf8004f90c1861882953e61a00438c3a0/communication-diagnostics-2026-09-06/PLAN.json",
        "files_url": "https://github.com/dexagon-ai/ainglish-evidence/tree/da975afbf8004f90c1861882953e61a00438c3a0/communication-diagnostics-2026-09-06/",
        "result_file": "RESULTS.json"
      }
    },
    {
      "id": "cached-reader-comparison",
      "title": "Can another guided reader recover several distinctions at once?",
      "type_label": "Cached-reader qualification and comparison",
      "date": "2026-09-07",
      "finding": "One reader recovered all five choices in both languages; the two-field output format still failed.",
      "design": "Three cached GGUF model families took 16 fresh neutral controls each. Only Mistral-small3.2-24B qualified for the language comparison. Its 216 target answers cover two and five choices in three authored contexts, using the previously published reference-message cases. Both language guides are visible. This is one qualified reader, not a three-family language-performance result.",
      "score_caption": "Mistral: exact joint interpretation AND the declared output shape, with a guide in the prompt",
      "scores": [
        {
          "condition": "2 fields · Ainglish",
          "correct": 0,
          "total": 12,
          "accuracy_pct": 0.0
        },
        {
          "condition": "2 fields · English",
          "correct": 0,
          "total": 12,
          "accuracy_pct": 0.0
        },
        {
          "condition": "5 fields · Ainglish",
          "correct": 96,
          "total": 96,
          "accuracy_pct": 100.0
        },
        {
          "condition": "5 fields · English",
          "correct": 96,
          "total": 96,
          "accuracy_pct": 100.0
        }
      ],
      "comparisons": [],
      "guards_passed": false,
      "guard_heading": "Reader qualification and output-format limits",
      "warnings": [
        "The neutral qualification floor was 14/16: Mistral scored 16/16, Qwen 10/16 and Gemma 4/16. The latter two received no language cases; they are not assigned language scores of zero.",
        "All 24 two-field responses included unwanted keys, so the exact-key protocol refused all of them. That is an output-shape failure, not a clean estimate of comprehension error.",
        "In a separately labelled post-hoc projection, the requested two values were correct: Ainglish 12/12; English 12/12. Ignoring unwanted keys was not the declared score; the frozen two-field results remain 0/12 per language.",
        "The five-field result is 96/96 per language, including 32/32 in each context. Both languages reached the same ceiling: this does not show an Ainglish advantage. No target output was truncated."
      ],
      "limits": [
        "Previously published synthetic cases, not a fresh independent holdout, human study or unaided reading test.",
        "The earlier strict-wrapper screens remain failed. This successor changed both its wrapper protocol and neutral vocabulary prospectively, so the screens do not isolate the wrapper’s causal effect.",
        "A shared-service interruption was resumed from the verified completed prefix, with unchanged requests and no scientific call retried.",
        "These weights and tokenizers already know English. The result describes current guided reading, not future learned Ainglish performance or inherent superiority.",
        "The separate Qwen writer/reader training gate remains failed. No adapter training, operational action, proposal confirmation or external adoption follows from this comparison."
      ],
      "source": {
        "commit": "2063274c923bdb51379655372a30daad75e8080a",
        "sha256": "1d88d8b4292bb3b2aa1b0cb4320bb21cacb3426ac4025dfdd12d2fc7ab83e8bc",
        "result_url": "https://github.com/dexagon-ai/ainglish-evidence/blob/2063274c923bdb51379655372a30daad75e8080a/cached-reader-semantic-2026-09-07/RESULTS.json",
        "plan_url": "https://github.com/dexagon-ai/ainglish-evidence/blob/2063274c923bdb51379655372a30daad75e8080a/cached-reader-semantic-2026-09-07/PLAN.json",
        "files_url": "https://github.com/dexagon-ai/ainglish-evidence/tree/2063274c923bdb51379655372a30daad75e8080a/cached-reader-semantic-2026-09-07/",
        "result_file": "RESULTS.json"
      }
    },
    {
      "id": "low-exposure-tokenizers",
      "title": "Can small amounts of training-text exposure reduce unfamiliar-word token costs?",
      "type_label": "Tokenizer-only experiment · completed",
      "date": "2026-09-07",
      "finding": "Yes in these small tokenizer experiments, with measurable costs elsewhere. This is not a production forecast.",
      "design": "28 byte-pair tokenizers: two vocabulary sizes, two splitting policies, and matched Ainglish or English additions at nominal 0.1%, 1% and 5% exposure, plus four baselines. Every cell uses a 4 MB project-discussion background. Held-out checks cover 817 English strings, 16 code strings, 16 punctuation/Unicode strings and 48 paired language examples.",
      "scores": [],
      "comparisons": [],
      "guards_passed": true,
      "tables": [
        {
          "caption": "The nominal 1% exposure slice: negative target change means fewer tokens",
          "columns": [
            "Tokenizer",
            "Ainglish target tokens versus matched English exposure",
            "Background English tokens versus no added exposure"
          ],
          "rows": [
            [
              "8,000-entry vocabulary · bytelevel-regex",
              "-2.37%",
              "+0.177%"
            ],
            [
              "8,000-entry vocabulary · whitespace-preserving",
              "-5.03%",
              "+0.092%"
            ],
            [
              "16,000-entry vocabulary · bytelevel-regex",
              "-1.37%",
              "+0.059%"
            ],
            [
              "16,000-entry vocabulary · whitespace-preserving",
              "-5.14%",
              "+0.023%"
            ]
          ]
        }
      ],
      "guard_heading": "Losslessness passed; semantic understanding was not tested",
      "warnings": [
        "All 28 cells round-trip their held-out strings exactly. This is a fidelity check, not a comprehension test.",
        "The table shows the four nominal 1% cells, not a selection of winning tokenizers. All 28 cells, including 0.1% and 5%, remain in the linked result.",
        "At 5% exposure some configurations increase code-token counts by 2.326% and punctuation/Unicode-token counts by 1.445%. Reduced target cost is not cost-free.",
        "Matched arms use equal paired-record occurrences, not identical achieved byte fractions: their text lengths differ. The fixed total budget consequently leaves different amounts of background text."
      ],
      "limits": [
        "Toy tokenizers with 8,000 or 16,000 entries and one project-discussion corpus; not a commercial tokenizer retraining study.",
        "Background documents are split by document and screened for literal target forms. This does not establish semantic independence from Ainglish discussions.",
        "No language-model weights were trained here. Tokenization cost and model comprehension must be measured separately.",
        "Future exposure is a testable strategy, not guaranteed efficiency or evidence that current excess costs can be ignored."
      ],
      "source": {
        "commit": "49c3c2f415e9108276d37a4422514f185fd4904b",
        "sha256": "ce1951b8f68c72d67c1e3e9271f54848145b1a97d2e87e7fcaf84ca8c7ae68fc",
        "result_url": "https://github.com/dexagon-ai/ainglish-evidence/blob/49c3c2f415e9108276d37a4422514f185fd4904b/tokenizer-low-exposure-2026-09-07/RESULTS.json",
        "plan_url": "https://github.com/dexagon-ai/ainglish-evidence/blob/49c3c2f415e9108276d37a4422514f185fd4904b/tokenizer-low-exposure-2026-09-07/PLAN.json",
        "files_url": "https://github.com/dexagon-ai/ainglish-evidence/tree/49c3c2f415e9108276d37a4422514f185fd4904b/tokenizer-low-exposure-2026-09-07/",
        "result_file": "RESULTS.json"
      }
    },
    {
      "id": "field-scoped-reader",
      "title": "Can a reader answer only the fields a task actually asks for?",
      "type_label": "Reference-assisted interface successor · completed",
      "date": "2026-09-07",
      "finding": "The two- and five-field interfaces passed qualification; the one-field interface did not.",
      "design": "Cached Qwen2.5-7B-Instruct, 48 neutral controls across three independently gated output widths. The two- and five-field widths each passed 16/16 controls and received 216 language calls in total. The one-field width scored 8/16 and received no language cases. The language cases are previously published reference messages in three authored contexts.",
      "score_caption": "Qualified widths: complete requested interpretation AND exact output shape, with both guides visible",
      "scores": [
        {
          "condition": "2 fields · Ainglish",
          "correct": 11,
          "total": 12,
          "accuracy_pct": 91.66666666666667
        },
        {
          "condition": "2 fields · English",
          "correct": 6,
          "total": 12,
          "accuracy_pct": 50.0
        },
        {
          "condition": "5 fields · Ainglish",
          "correct": 94,
          "total": 96,
          "accuracy_pct": 97.91666666666667
        },
        {
          "condition": "5 fields · English",
          "correct": 89,
          "total": 96,
          "accuracy_pct": 92.70833333333333
        }
      ],
      "comparisons": [],
      "guards_passed": false,
      "guard_heading": "Width-specific qualification; no automatic pooling",
      "warnings": [
        "Ainglish scores 11/12 and 94/96; English scores 6/12 and 89/96. These are guided current-reader results, not unaided understanding or independent confirmation.",
        "One-field responses omitted the required braces. Its failed screen is retained; an unqualified width is not assigned a language score of zero.",
        "All qualified target responses used the declared shape, with no truncation. This does not retrospectively change any earlier failed interface or training gate."
      ],
      "limits": [
        "New prospective prompt and neutral screen; earlier failures remain unchanged.",
        "Previously exposed synthetic reference cases, not an independent holdout or governance replication.",
        "Both language guides are supplied. Current-tokenizer counts do not predict future training or tokenization.",
        "One cached model/configuration, three authored templates; repeated bit patterns are not independent humans.",
        "Three authored contexts and known reference cases, not 108 independent task families or a fresh structural holdout.",
        "English is already present in the model’s training and tokenizer. This result neither predicts a future trained dialect nor proves an inherent language advantage."
      ],
      "source": {
        "commit": "84e2488db9538b0f2fd110810f4da58de8d78f22",
        "sha256": "308df2044459a71f5abcf06fa79286b71acabf52e9c09dd1e12d230db7bcb36e",
        "result_url": "https://github.com/dexagon-ai/ainglish-evidence/blob/84e2488db9538b0f2fd110810f4da58de8d78f22/field-scoped-communication-2026-09-07/RESULTS.json",
        "plan_url": "https://github.com/dexagon-ai/ainglish-evidence/blob/84e2488db9538b0f2fd110810f4da58de8d78f22/field-scoped-communication-2026-09-07/PLAN.json",
        "files_url": "https://github.com/dexagon-ai/ainglish-evidence/tree/84e2488db9538b0f2fd110810f4da58de8d78f22/field-scoped-communication-2026-09-07/",
        "result_file": "RESULTS.json"
      }
    },
    {
      "id": "dialogue-interface-stop",
      "title": "Did the first real dialogue-cost study reach its language tasks?",
      "type_label": "Interface qualification · stopped before language exposure",
      "date": "2026-09-07",
      "finding": "No. Its neutral interface screen failed, so no language dialogue was run.",
      "design": "The prospective JSON response interface answered 4/16 neutral controls exactly, below its 14/16 floor. No Ainglish or English target dialogue calls were made. The planned history, dictionary-lookup and repair comparisons therefore have no result.",
      "scores": [],
      "comparisons": [],
      "guards_passed": false,
      "guard_heading": "Qualification failure is not a language score",
      "warnings": [
        "Malformed keys and incorrect neutral values remain failures under the original parser. They were not repaired after seeing the answers.",
        "A new interface requires its own prospectively frozen controls, qualification and actual conversations. Projecting a token-cost curve is not a substitute for executing those turns."
      ],
      "limits": [
        "No conclusion about relative language comprehension, repair savings, long-conversation reliability or operational usefulness can be drawn from this stopped screen."
      ],
      "source": {
        "commit": "c221976f6963db05f46514423bde91f7aefd6ce5",
        "sha256": "68fe02265634db2fc4e98230f3edec4f1fd8b37c32e8290243f63293a07e6b6a",
        "result_url": "https://github.com/dexagon-ai/ainglish-evidence/blob/c221976f6963db05f46514423bde91f7aefd6ce5/dialogue-costs-2026-09-07/RESULTS.json",
        "plan_url": "https://github.com/dexagon-ai/ainglish-evidence/blob/c221976f6963db05f46514423bde91f7aefd6ce5/dialogue-costs-2026-09-07/PLAN.json",
        "files_url": "https://github.com/dexagon-ai/ainglish-evidence/tree/c221976f6963db05f46514423bde91f7aefd6ce5/dialogue-costs-2026-09-07/",
        "result_file": "RESULTS.json"
      }
    },
    {
      "id": "mistral-context-stop",
      "title": "Did the longer Mistral sender–receiver study finish?",
      "type_label": "Contextual communication · interrupted without a final result",
      "date": "2026-09-07",
      "finding": "No final language-performance result is available.",
      "design": "The retained journal contains 323 completed calls and 0 uncertain calls at audit. The CPU-only pinned Mistral study stopped before producing its final result. The completion auditor made zero inference calls and published the retained records.",
      "scores": [],
      "comparisons": [],
      "guards_passed": false,
      "guard_heading": "Incomplete run; no selected partial accuracy headline",
      "warnings": [
        "This is one interrupted-and-continued study. Reusing already completed responses locally is not independent replication.",
        "The stopping cause is not established by this receipt. We did not turn a partial conversation sample into a final success rate or retry calls through the audit."
      ],
      "limits": [
        "This is a same-author readback/reproducibility audit, not an independent replication.",
        "Original CPU reader, options, controls, inputs and gates are unchanged; completed calls were not repeated.",
        "Counts describe three authored contexts, not broad operational efficacy or future Ainglish-trained performance.",
        "Ollama token counts are observed counters, not token IDs or provider billing.",
        "No final result exists. Only completion/uncertainty counts are reported; no partial target-accuracy headline is selected. The finisher does not infer the stopping cause or retry an uncertain call."
      ],
      "source": {
        "commit": "74c0d000680520338c6dddfe6d6b8573ca07f767",
        "sha256": "d611ba0af249fb0ff382373a280bfb4d408a77585504131e19e4864fcf75e547",
        "result_url": "https://github.com/dexagon-ai/ainglish-evidence/blob/74c0d000680520338c6dddfe6d6b8573ca07f767/mistral-completion-2026-09-07/execution/ANALYSIS.json",
        "plan_url": "https://github.com/dexagon-ai/ainglish-evidence/blob/74c0d000680520338c6dddfe6d6b8573ca07f767/mistral-completion-2026-09-07/README.md",
        "files_url": "https://github.com/dexagon-ai/ainglish-evidence/tree/74c0d000680520338c6dddfe6d6b8573ca07f767/mistral-completion-2026-09-07/",
        "result_file": "execution/ANALYSIS.json"
      }
    },
    {
      "id": "five-code-dialogue-stop",
      "title": "Did a simpler dialogue response shape solve qualification?",
      "type_label": "Prospective interface successor · stopped before language exposure",
      "date": "2026-09-07",
      "finding": "The output shape worked, but the neutral answers were still too inaccurate to proceed.",
      "design": "A fresh screen asked for five consecutive yes/no characters instead of JSON, using 32 neutral present/absent records. All 32/32 outputs met the declared shape, with 0 truncated outputs. Only 8/32 answers were exactly correct, below the frozen 28/32 floor. No language conversation was run.",
      "scores": [],
      "comparisons": [],
      "guards_passed": false,
      "guard_heading": "Valid formatting does not establish correct interpretation",
      "warnings": [
        "These controls contain no Ainglish. Their failure does not estimate the relative quality of Ainglish and English.",
        "The earlier JSON 4/16 screen remains unchanged. This successor changed both response shape and neutral vocabulary; it does not isolate the causal effect of either change.",
        "No measured dictionary-lookup, history, repair or per-success language cost is available. The retained readback made zero inference calls and repaired no failed answer."
      ],
      "limits": [
        "One current cached Qwen reader, not a comparison of model families or future learned Ainglish.",
        "Exact token IDs and all 32 responses are retained. The audit is same-author reconstruction, not independent replication."
      ],
      "source": {
        "commit": "9d1045e7e1027c81d4fb6f5b2ca3d8120fb29e38",
        "sha256": "9294c6c763b4da9efdf8d467225055b732f4f8779f337aa35b6d8927c216e350",
        "result_url": "https://github.com/dexagon-ai/ainglish-evidence/blob/9d1045e7e1027c81d4fb6f5b2ca3d8120fb29e38/dialogue-code-interface-2026-09-07/POSTRUN-AUDIT.json",
        "plan_url": "https://github.com/dexagon-ai/ainglish-evidence/blob/9d1045e7e1027c81d4fb6f5b2ca3d8120fb29e38/dialogue-code-interface-2026-09-07/PLAN.json",
        "files_url": "https://github.com/dexagon-ai/ainglish-evidence/tree/9d1045e7e1027c81d4fb6f5b2ca3d8120fb29e38/dialogue-code-interface-2026-09-07/",
        "result_file": "POSTRUN-AUDIT.json"
      }
    },
    {
      "id": "contextual-structural-transfer",
      "title": "Does teaching in context transfer to different reasoning structures?",
      "type_label": "Matched contextual learning · completed",
      "date": "2026-09-07",
      "finding": "No reliable advantage established over matched English teaching; several family-level checks failed.",
      "design": "One cached Qwen2.5-7B-Instruct model, two fixed training seeds and four small adapters. Each language received 576 contextual teaching rows plus the same 192 ordinary-English rehearsal rows, for one epoch. Evaluation used 192 paired semantic configurations from 24 newly authored reasoning frames, plus 96 ordinary-English items. All five conditions passed 12/12 neutral controls; all 2,400 target answers and 60 controls completed.",
      "score_caption": "Exact joint answers on the Ainglish structural holdout; no glossary, but shared task-local facts and boundary rules",
      "scores": [
        {
          "condition": "Base model, no project training",
          "correct": 72,
          "total": 192,
          "accuracy_pct": 37.5
        },
        {
          "condition": "Ainglish training, seed 17",
          "correct": 80,
          "total": 192,
          "accuracy_pct": 41.666666666666664
        },
        {
          "condition": "English training, seed 17",
          "correct": 77,
          "total": 192,
          "accuracy_pct": 40.104166666666664
        },
        {
          "condition": "Ainglish training, seed 29",
          "correct": 85,
          "total": 192,
          "accuracy_pct": 44.270833333333336
        },
        {
          "condition": "English training, seed 29",
          "correct": 81,
          "total": 192,
          "accuracy_pct": 42.1875
        }
      ],
      "comparisons": [
        {
          "label": "Seed 17: Ainglish-trained minus English-trained, reading Ainglish",
          "delta_pp": 1.5625,
          "interval_95": [
            -6.25,
            10.416666666666666
          ]
        },
        {
          "label": "Seed 29: Ainglish-trained minus English-trained, reading Ainglish",
          "delta_pp": 2.0833333333333335,
          "interval_95": [
            -5.208333333333333,
            9.375
          ]
        }
      ],
      "guards_passed": false,
      "tables": [
        {
          "caption": "Every failed predeclared −5-percentage-point screen; aggregates do not cancel these failures",
          "columns": [
            "Training seed",
            "Comparison",
            "Distinction",
            "Accuracy difference"
          ],
          "rows": [
            [
              "17",
              "Ainglish reading versus matched English training",
              "Separate versus joint actions",
              "-9.38 points"
            ],
            [
              "17",
              "Ainglish reading versus matched English training",
              "Unknown versus undecided",
              "-15.62 points"
            ],
            [
              "17",
              "Careful English reading versus the base model",
              "Deadline boundaries",
              "-6.25 points"
            ],
            [
              "17",
              "Careful English reading versus the base model",
              "Separate versus joint actions",
              "-9.38 points"
            ],
            [
              "29",
              "Ainglish reading versus matched English training",
              "Separate versus joint actions",
              "-6.25 points"
            ],
            [
              "29",
              "Ainglish reading versus matched English training",
              "Unknown versus undecided",
              "-6.25 points"
            ],
            [
              "29",
              "Careful English reading versus the base model",
              "Deadline boundaries",
              "-6.25 points"
            ],
            [
              "29",
              "Careful English reading versus the base model",
              "Separate versus joint actions",
              "-12.50 points"
            ]
          ]
        },
        {
          "caption": "English retention: absolute scores, including both matched training arms",
          "columns": [
            "Model condition",
            "Careful English structural holdout",
            "Ordinary English authored tasks"
          ],
          "rows": [
            [
              "Base model, no project training",
              "84/192",
              "75/96"
            ],
            [
              "Ainglish training, seed 17",
              "80/192",
              "96/96"
            ],
            [
              "English training, seed 17",
              "80/192",
              "96/96"
            ],
            [
              "Ainglish training, seed 29",
              "88/192",
              "92/96"
            ],
            [
              "English training, seed 29",
              "83/192",
              "96/96"
            ]
          ]
        }
      ],
      "warnings": [
        "The overall matched-training differences are +1.56 and +2.08 percentage points. Both exploratory 95% intervals span zero; neither establishes a selective learning advantage.",
        "Both seeds failed the Ainglish-reading checks for separate versus joint actions and unknown versus undecided. Both also failed careful-English retention checks for deadline boundaries and separate versus joint actions.",
        "Ordinary-English scores improved after Ainglish training, but matched English training reached 96/96 in both seeds. The rehearsal and retention tasks are related authored families, not an independent benchmark or a uniquely Ainglish benefit.",
        "All target answers had valid output shape with no truncation. Good formatting does not cancel incorrect reasoning or failed safeguards.",
        "The −5-point screens are point-estimate checks, not statistical proofs of non-inferiority. Their threshold, epochs, seeds and scoring were not changed after seeing results."
      ],
      "limits": [
        "192 configurations instantiate 24 authored frames, not 192 independent reasoning structures. The 95% intervals resample those 24 frames; family checks have only four frames each.",
        "The test structures are absent from this frozen training split. This does not establish globally unseen concepts, independent authorship or human validation.",
        "Some tasks supply boundary rules and ledger facts equally to both languages. No glossary is supplied, but this is not wholly unassisted understanding.",
        "Two seeds of one model family, with a fixed tokenizer that already knows English. Small adapter learning is not a forecast of future pretraining or tokenizer changes.",
        "All four adapters were publicly digest-sealed before evaluation. All outcomes remain published; no favourable seed selection or uncertain-call retry.",
        "Research only: this study does not settle a proposal, demonstrate external adoption or change any governance gate."
      ],
      "source": {
        "commit": "7fe8f7c8e110d1b124743ef42fa70d8fde136cff",
        "sha256": "ec330f6caec80afcfcb44a4cb92cb6e507509d32920b29dc9767ab0a2e149a4f",
        "result_url": "https://github.com/dexagon-ai/ainglish-evidence/blob/7fe8f7c8e110d1b124743ef42fa70d8fde136cff/contextual-transfer-2026-09-07/RESULTS.json",
        "plan_url": "https://github.com/dexagon-ai/ainglish-evidence/blob/7fe8f7c8e110d1b124743ef42fa70d8fde136cff/contextual-transfer-2026-09-07/PLAN.json",
        "files_url": "https://github.com/dexagon-ai/ainglish-evidence/tree/7fe8f7c8e110d1b124743ef42fa70d8fde136cff/contextual-transfer-2026-09-07/",
        "result_file": "RESULTS.json"
      }
    }
  ]
}
