{
  "title": "Askolia local Laya experiment, 23 September 2026",
  "artifactType": "Aggregate extract of a measured local experiment",
  "sourceReportSha256": "ba2723e744a6b7ce1b8fa92ae325021744c1c0121664fddc59b948ecb0a58fc4",
  "liveMeasurements": true,
  "status": "complete",
  "startedAt": "2026-09-23T11:49:19.424893+00:00",
  "finishedAt": "2026-09-23T11:49:37.678245+00:00",
  "config": {
    "laya": "0.3.7",
    "model": "convaiinnovations/laya-multilingual",
    "revision": "b4a904d1a2a54c822b829e24291d4b8f280fe43e",
    "upstream": "010bacef009c855ccba814b51f7c8e1d38ab5e3f",
    "minProbability": 0.8,
    "minConfidence": 0.8,
    "rawNoulThreshold": 0.5,
    "repetitions": 3,
    "threads": 4,
    "interopThreads": 1,
    "deviceRequested": "mps",
    "seed": 13,
    "maxLen": 1024,
    "headMaxLen": 256,
    "optionMaxTokens": 48,
    "excludedFromScoring": {
      "rag-holdout-injection-en": "Original end-to-end abstention label conflicts with retrieval-only relevance: refund-08 refutes the requested guarantee. Excluded after first-run label audit; raw outputs retained."
    }
  },
  "caseCount": 24,
  "corpora": [
    {
      "file": "jev-faq.json",
      "sha256": "4ff2cca8f7d54aa4d16863ec91b1fbefdbb905162af31f02df1fb7c1d562aad3",
      "version": "2026-09-23.2"
    },
    {
      "file": "jev-rag.json",
      "sha256": "b6dc8d3750f08b6a682ba2dc9a7294604b68caba4bdd627eb3c14c82a59ab4fd",
      "version": "2026-09-23.2"
    },
    {
      "file": "laya-stress.json",
      "sha256": "f434c48e97ddfbb059c8b18264469692c4824ce97a4a327d555e96e61ece2212",
      "version": "2026-09-23.1"
    }
  ],
  "cases": [
    {
      "id": "faq-tuning-delete-fr",
      "task": "faq",
      "language": "fr",
      "split": "tuning",
      "scoring": "primary",
      "scoringNote": null
    },
    {
      "id": "faq-tuning-human-request",
      "task": "faq",
      "language": "fr",
      "split": "tuning",
      "scoring": "primary",
      "scoringNote": null
    },
    {
      "id": "faq-tuning-context-en",
      "task": "faq",
      "language": "en",
      "split": "tuning",
      "scoring": "primary",
      "scoringNote": null
    },
    {
      "id": "faq-holdout-delete-en",
      "task": "faq",
      "language": "en",
      "split": "holdout",
      "scoring": "primary",
      "scoringNote": null
    },
    {
      "id": "faq-holdout-greeting-de",
      "task": "faq",
      "language": "de",
      "split": "holdout",
      "scoring": "primary",
      "scoringNote": null
    },
    {
      "id": "faq-holdout-account-specific-fr",
      "task": "faq",
      "language": "fr",
      "split": "holdout",
      "scoring": "primary",
      "scoringNote": null
    },
    {
      "id": "faq-holdout-hours-en",
      "task": "faq",
      "language": "en",
      "split": "holdout",
      "scoring": "primary",
      "scoringNote": null
    },
    {
      "id": "faq-holdout-injection-en",
      "task": "faq",
      "language": "en",
      "split": "holdout",
      "scoring": "primary",
      "scoringNote": null
    },
    {
      "id": "faq-holdout-unknown-locale",
      "task": "faq",
      "language": "en",
      "split": "holdout",
      "scoring": "primary",
      "scoringNote": null
    },
    {
      "id": "faq-holdout-delete-de",
      "task": "faq",
      "language": "de",
      "split": "holdout",
      "scoring": "primary",
      "scoringNote": null
    },
    {
      "id": "rag-tuning-delivery-fr",
      "task": "rag",
      "language": "fr",
      "split": "tuning",
      "scoring": "primary",
      "scoringNote": null
    },
    {
      "id": "rag-tuning-false-premise",
      "task": "rag",
      "language": "en",
      "split": "tuning",
      "scoring": "primary",
      "scoringNote": null
    },
    {
      "id": "rag-holdout-invoice-de",
      "task": "rag",
      "language": "de",
      "split": "holdout",
      "scoring": "primary",
      "scoringNote": null
    },
    {
      "id": "rag-holdout-combined-fr",
      "task": "rag",
      "language": "fr",
      "split": "holdout",
      "scoring": "primary",
      "scoringNote": null
    },
    {
      "id": "rag-holdout-no-evidence",
      "task": "rag",
      "language": "en",
      "split": "holdout",
      "scoring": "primary",
      "scoringNote": null
    },
    {
      "id": "rag-holdout-injection-en",
      "task": "rag",
      "language": "en",
      "split": "holdout",
      "scoring": "diagnostic",
      "scoringNote": "Original end-to-end abstention label conflicts with retrieval-only relevance: refund-08 refutes the requested guarantee. Excluded after first-run label audit; raw outputs retained."
    },
    {
      "id": "faq-laya-tuning-ambiguous-en",
      "task": "faq",
      "language": "en",
      "split": "tuning",
      "scoring": "primary",
      "scoringNote": null
    },
    {
      "id": "rag-laya-tuning-wrong-source-en",
      "task": "rag",
      "language": "en",
      "split": "tuning",
      "scoring": "primary",
      "scoringNote": null
    },
    {
      "id": "faq-laya-holdout-ambiguous-fr",
      "task": "faq",
      "language": "fr",
      "split": "holdout",
      "scoring": "primary",
      "scoringNote": null
    },
    {
      "id": "faq-laya-holdout-context-de",
      "task": "faq",
      "language": "de",
      "split": "holdout",
      "scoring": "primary",
      "scoringNote": null
    },
    {
      "id": "faq-laya-holdout-long-answer-en",
      "task": "faq",
      "language": "en",
      "split": "holdout",
      "scoring": "primary",
      "scoringNote": null
    },
    {
      "id": "rag-laya-holdout-pooled-overflow-de",
      "task": "rag",
      "language": "de",
      "split": "holdout",
      "scoring": "primary",
      "scoringNote": null
    },
    {
      "id": "rag-laya-holdout-passage-tail-fr",
      "task": "rag",
      "language": "fr",
      "split": "holdout",
      "scoring": "primary",
      "scoringNote": null
    },
    {
      "id": "rag-laya-holdout-no-evidence-fr",
      "task": "rag",
      "language": "fr",
      "split": "holdout",
      "scoring": "primary",
      "scoringNote": null
    }
  ],
  "protocolSha256": "ad76acc307bcce4fa29c521ceaae893e12c47c39351db80991c0a09229f7d122",
  "runnerSha256": "7e04553b3c74b86b82c0d49a17bbd1342b6236eee7c3c95e95153fed92af659c",
  "runtime": {
    "cpu": "Apple M5 Pro",
    "ramBytes": 68719476736,
    "device": "mps",
    "dtype": "torch.float32",
    "parameters": 321908995
  },
  "packages": {
    "laya": "0.3.7",
    "torch": "2.14.0",
    "transformers": "5.17.0"
  },
  "coldStart": {
    "importsMs": 1034.5344580709934,
    "modelLoadMs": 3791.565665975213,
    "firstPredictMs": 131.61412486806512,
    "totalMs": 4957.714248914272,
    "n": 1,
    "excludes": "Download, disk-cache eviction, hash validation and interpreter startup"
  },
  "warmup": {
    "caseArms": 34,
    "sdkCalls": 98,
    "repetitionsPerCaseArm": 1
  },
  "summary": [
    {
      "split": "holdout",
      "task": "faq",
      "arm": "choice",
      "distinctCases": 10,
      "scoredDistinctCases": 10,
      "outcomeSample": "Successful repetition 0, excluding diagnostic cases",
      "excludedCaseIds": [],
      "failedTrials": 0,
      "gated": {
        "n": 10,
        "rawCorrect": 9,
        "rawAccuracy": 0.9,
        "gatedCorrect": 4,
        "gatedAccuracy": 0.4,
        "automaticSelections": 0,
        "wrongAutomaticSelections": 0,
        "selectionPrecision": null,
        "abstentions": 10,
        "correctAbstentions": 4,
        "abstentionRecall": 1.0,
        "missedAnswerable": 6
      },
      "rawNoulAt05": null,
      "byLanguage": {
        "de": {
          "n": 3,
          "rawCorrect": 3,
          "rawAccuracy": 1.0,
          "gatedCorrect": 1,
          "gatedAccuracy": 0.3333333333333333,
          "automaticSelections": 0,
          "wrongAutomaticSelections": 0,
          "selectionPrecision": null,
          "abstentions": 3,
          "correctAbstentions": 1,
          "abstentionRecall": 1.0,
          "missedAnswerable": 2
        },
        "en": {
          "n": 5,
          "rawCorrect": 5,
          "rawAccuracy": 1.0,
          "gatedCorrect": 1,
          "gatedAccuracy": 0.2,
          "automaticSelections": 0,
          "wrongAutomaticSelections": 0,
          "selectionPrecision": null,
          "abstentions": 5,
          "correctAbstentions": 1,
          "abstentionRecall": 1.0,
          "missedAnswerable": 4
        },
        "fr": {
          "n": 2,
          "rawCorrect": 1,
          "rawAccuracy": 0.5,
          "gatedCorrect": 2,
          "gatedAccuracy": 1.0,
          "automaticSelections": 0,
          "wrongAutomaticSelections": 0,
          "selectionPrecision": null,
          "abstentions": 2,
          "correctAbstentions": 2,
          "abstentionRecall": 1.0,
          "missedAnswerable": 0
        }
      },
      "casesWithDecisionChanges": 0,
      "warmCaseLatency": {
        "n": 30,
        "p50Ms": 17.665292136371136,
        "p95Ms": 21.408332977443933,
        "maxMs": 22.017542272806168
      },
      "callsPerCase": [
        1
      ]
    },
    {
      "split": "holdout",
      "task": "rag",
      "arm": "pairwise",
      "distinctCases": 7,
      "scoredDistinctCases": 6,
      "outcomeSample": "Successful repetition 0, excluding diagnostic cases",
      "excludedCaseIds": [
        "rag-holdout-injection-en"
      ],
      "failedTrials": 0,
      "gated": {
        "n": 6,
        "truePositivePassages": 0,
        "falsePositivePassages": 0,
        "falseNegativePassages": 5,
        "precisionAt5": null,
        "recallAt5": 0.0,
        "mrrAnswerable": 0.0,
        "exactSetCorrect": 2,
        "abstentions": 6,
        "noEvidenceCases": 2,
        "correctAbstentions": 2,
        "noEvidenceFalseAccepts": 0
      },
      "rawNoulAt05": {
        "n": 6,
        "truePositivePassages": 0,
        "falsePositivePassages": 0,
        "falseNegativePassages": 5,
        "precisionAt5": null,
        "recallAt5": 0.0,
        "mrrAnswerable": 0.0,
        "exactSetCorrect": 2,
        "abstentions": 6,
        "noEvidenceCases": 2,
        "correctAbstentions": 2,
        "noEvidenceFalseAccepts": 0
      },
      "byLanguage": {
        "de": {
          "n": 2,
          "truePositivePassages": 0,
          "falsePositivePassages": 0,
          "falseNegativePassages": 2,
          "precisionAt5": null,
          "recallAt5": 0.0,
          "mrrAnswerable": 0.0,
          "exactSetCorrect": 0,
          "abstentions": 2,
          "noEvidenceCases": 0,
          "correctAbstentions": 0,
          "noEvidenceFalseAccepts": 0
        },
        "en": {
          "n": 1,
          "truePositivePassages": 0,
          "falsePositivePassages": 0,
          "falseNegativePassages": 0,
          "precisionAt5": null,
          "recallAt5": null,
          "mrrAnswerable": null,
          "exactSetCorrect": 1,
          "abstentions": 1,
          "noEvidenceCases": 1,
          "correctAbstentions": 1,
          "noEvidenceFalseAccepts": 0
        },
        "fr": {
          "n": 3,
          "truePositivePassages": 0,
          "falsePositivePassages": 0,
          "falseNegativePassages": 3,
          "precisionAt5": null,
          "recallAt5": 0.0,
          "mrrAnswerable": 0.0,
          "exactSetCorrect": 1,
          "abstentions": 3,
          "noEvidenceCases": 1,
          "correctAbstentions": 1,
          "noEvidenceFalseAccepts": 0
        }
      },
      "casesWithDecisionChanges": 0,
      "warmCaseLatency": {
        "n": 18,
        "p50Ms": 111.73345800489187,
        "p95Ms": 257.59758334606886,
        "maxMs": 257.59758334606886
      },
      "callsPerCase": [
        1,
        5,
        8,
        12
      ]
    },
    {
      "split": "holdout",
      "task": "rag",
      "arm": "pooled",
      "distinctCases": 7,
      "scoredDistinctCases": 6,
      "outcomeSample": "Successful repetition 0, excluding diagnostic cases",
      "excludedCaseIds": [
        "rag-holdout-injection-en"
      ],
      "failedTrials": 0,
      "gated": {
        "n": 6,
        "truePositivePassages": 0,
        "falsePositivePassages": 0,
        "falseNegativePassages": 5,
        "precisionAt5": null,
        "recallAt5": 0.0,
        "mrrAnswerable": 0.0,
        "exactSetCorrect": 2,
        "abstentions": 6,
        "noEvidenceCases": 2,
        "correctAbstentions": 2,
        "noEvidenceFalseAccepts": 0
      },
      "rawNoulAt05": {
        "n": 6,
        "truePositivePassages": 2,
        "falsePositivePassages": 3,
        "falseNegativePassages": 3,
        "precisionAt5": 0.4,
        "recallAt5": 0.4,
        "mrrAnswerable": 0.0625,
        "exactSetCorrect": 2,
        "abstentions": 5,
        "noEvidenceCases": 2,
        "correctAbstentions": 2,
        "noEvidenceFalseAccepts": 0
      },
      "byLanguage": {
        "de": {
          "n": 2,
          "truePositivePassages": 0,
          "falsePositivePassages": 0,
          "falseNegativePassages": 2,
          "precisionAt5": null,
          "recallAt5": 0.0,
          "mrrAnswerable": 0.0,
          "exactSetCorrect": 0,
          "abstentions": 2,
          "noEvidenceCases": 0,
          "correctAbstentions": 0,
          "noEvidenceFalseAccepts": 0
        },
        "en": {
          "n": 1,
          "truePositivePassages": 0,
          "falsePositivePassages": 0,
          "falseNegativePassages": 0,
          "precisionAt5": null,
          "recallAt5": null,
          "mrrAnswerable": null,
          "exactSetCorrect": 1,
          "abstentions": 1,
          "noEvidenceCases": 1,
          "correctAbstentions": 1,
          "noEvidenceFalseAccepts": 0
        },
        "fr": {
          "n": 3,
          "truePositivePassages": 0,
          "falsePositivePassages": 0,
          "falseNegativePassages": 3,
          "precisionAt5": null,
          "recallAt5": 0.0,
          "mrrAnswerable": 0.0,
          "exactSetCorrect": 1,
          "abstentions": 3,
          "noEvidenceCases": 1,
          "correctAbstentions": 1,
          "noEvidenceFalseAccepts": 0
        }
      },
      "casesWithDecisionChanges": 0,
      "warmCaseLatency": {
        "n": 18,
        "p50Ms": 112.43360443040729,
        "p95Ms": 760.9100830741227,
        "maxMs": 760.9100830741227
      },
      "callsPerCase": [
        1
      ]
    },
    {
      "split": "tuning",
      "task": "faq",
      "arm": "choice",
      "distinctCases": 4,
      "scoredDistinctCases": 4,
      "outcomeSample": "Successful repetition 0, excluding diagnostic cases",
      "excludedCaseIds": [],
      "failedTrials": 0,
      "gated": {
        "n": 4,
        "rawCorrect": 0,
        "rawAccuracy": 0.0,
        "gatedCorrect": 2,
        "gatedAccuracy": 0.5,
        "automaticSelections": 0,
        "wrongAutomaticSelections": 0,
        "selectionPrecision": null,
        "abstentions": 4,
        "correctAbstentions": 2,
        "abstentionRecall": 1.0,
        "missedAnswerable": 2
      },
      "rawNoulAt05": null,
      "byLanguage": {
        "en": {
          "n": 2,
          "rawCorrect": 0,
          "rawAccuracy": 0.0,
          "gatedCorrect": 1,
          "gatedAccuracy": 0.5,
          "automaticSelections": 0,
          "wrongAutomaticSelections": 0,
          "selectionPrecision": null,
          "abstentions": 2,
          "correctAbstentions": 1,
          "abstentionRecall": 1.0,
          "missedAnswerable": 1
        },
        "fr": {
          "n": 2,
          "rawCorrect": 0,
          "rawAccuracy": 0.0,
          "gatedCorrect": 1,
          "gatedAccuracy": 0.5,
          "automaticSelections": 0,
          "wrongAutomaticSelections": 0,
          "selectionPrecision": null,
          "abstentions": 2,
          "correctAbstentions": 1,
          "abstentionRecall": 1.0,
          "missedAnswerable": 1
        }
      },
      "casesWithDecisionChanges": 0,
      "warmCaseLatency": {
        "n": 12,
        "p50Ms": 18.007270758971572,
        "p95Ms": 27.209416963160038,
        "maxMs": 27.209416963160038
      },
      "callsPerCase": [
        1
      ]
    },
    {
      "split": "tuning",
      "task": "rag",
      "arm": "pairwise",
      "distinctCases": 3,
      "scoredDistinctCases": 3,
      "outcomeSample": "Successful repetition 0, excluding diagnostic cases",
      "excludedCaseIds": [],
      "failedTrials": 0,
      "gated": {
        "n": 3,
        "truePositivePassages": 1,
        "falsePositivePassages": 2,
        "falseNegativePassages": 2,
        "precisionAt5": 0.3333333333333333,
        "recallAt5": 0.3333333333333333,
        "mrrAnswerable": 0.3333333333333333,
        "exactSetCorrect": 0,
        "abstentions": 2,
        "noEvidenceCases": 0,
        "correctAbstentions": 0,
        "noEvidenceFalseAccepts": 0
      },
      "rawNoulAt05": {
        "n": 3,
        "truePositivePassages": 3,
        "falsePositivePassages": 3,
        "falseNegativePassages": 0,
        "precisionAt5": 0.5,
        "recallAt5": 1.0,
        "mrrAnswerable": 1.0,
        "exactSetCorrect": 2,
        "abstentions": 0,
        "noEvidenceCases": 0,
        "correctAbstentions": 0,
        "noEvidenceFalseAccepts": 0
      },
      "byLanguage": {
        "en": {
          "n": 2,
          "truePositivePassages": 1,
          "falsePositivePassages": 2,
          "falseNegativePassages": 1,
          "precisionAt5": 0.3333333333333333,
          "recallAt5": 0.5,
          "mrrAnswerable": 0.5,
          "exactSetCorrect": 0,
          "abstentions": 1,
          "noEvidenceCases": 0,
          "correctAbstentions": 0,
          "noEvidenceFalseAccepts": 0
        },
        "fr": {
          "n": 1,
          "truePositivePassages": 0,
          "falsePositivePassages": 0,
          "falseNegativePassages": 1,
          "precisionAt5": null,
          "recallAt5": 0.0,
          "mrrAnswerable": 0.0,
          "exactSetCorrect": 0,
          "abstentions": 1,
          "noEvidenceCases": 0,
          "correctAbstentions": 0,
          "noEvidenceFalseAccepts": 0
        }
      },
      "casesWithDecisionChanges": 0,
      "warmCaseLatency": {
        "n": 9,
        "p50Ms": 111.25975009053946,
        "p95Ms": 116.69037491083145,
        "maxMs": 116.69037491083145
      },
      "callsPerCase": [
        8
      ]
    },
    {
      "split": "tuning",
      "task": "rag",
      "arm": "pooled",
      "distinctCases": 3,
      "scoredDistinctCases": 3,
      "outcomeSample": "Successful repetition 0, excluding diagnostic cases",
      "excludedCaseIds": [],
      "failedTrials": 0,
      "gated": {
        "n": 3,
        "truePositivePassages": 1,
        "falsePositivePassages": 4,
        "falseNegativePassages": 2,
        "precisionAt5": 0.2,
        "recallAt5": 0.3333333333333333,
        "mrrAnswerable": 0.06666666666666667,
        "exactSetCorrect": 0,
        "abstentions": 2,
        "noEvidenceCases": 0,
        "correctAbstentions": 0,
        "noEvidenceFalseAccepts": 0
      },
      "rawNoulAt05": {
        "n": 3,
        "truePositivePassages": 1,
        "falsePositivePassages": 4,
        "falseNegativePassages": 2,
        "precisionAt5": 0.2,
        "recallAt5": 0.3333333333333333,
        "mrrAnswerable": 0.06666666666666667,
        "exactSetCorrect": 0,
        "abstentions": 2,
        "noEvidenceCases": 0,
        "correctAbstentions": 0,
        "noEvidenceFalseAccepts": 0
      },
      "byLanguage": {
        "en": {
          "n": 2,
          "truePositivePassages": 1,
          "falsePositivePassages": 4,
          "falseNegativePassages": 1,
          "precisionAt5": 0.2,
          "recallAt5": 0.5,
          "mrrAnswerable": 0.1,
          "exactSetCorrect": 0,
          "abstentions": 1,
          "noEvidenceCases": 0,
          "correctAbstentions": 0,
          "noEvidenceFalseAccepts": 0
        },
        "fr": {
          "n": 1,
          "truePositivePassages": 0,
          "falsePositivePassages": 0,
          "falseNegativePassages": 1,
          "precisionAt5": null,
          "recallAt5": 0.0,
          "mrrAnswerable": 0.0,
          "exactSetCorrect": 0,
          "abstentions": 1,
          "noEvidenceCases": 0,
          "correctAbstentions": 0,
          "noEvidenceFalseAccepts": 0
        }
      },
      "casesWithDecisionChanges": 0,
      "warmCaseLatency": {
        "n": 9,
        "p50Ms": 118.13933355733752,
        "p95Ms": 137.2209577821195,
        "maxMs": 137.2209577821195
      },
      "callsPerCase": [
        1
      ]
    }
  ],
  "memory": {
    "processPeakRssBytes": 2769534976,
    "rssAtEndBytes": 1250656256,
    "maxSampledMpsDriverBytes": 3648864256
  },
  "unmeasured": {
    "jev": "No TypeSafe access",
    "legacyLlm": "No online calls in this experiment",
    "generatedAnswerQuality": null,
    "financialSavings": null
  },
  "limitations": [
    "Synthetic, small and deliberately adversarial corpus; not representative customer traffic.",
    "Accuracy denominators count distinct cases; timing repetitions are not independent quality samples.",
    "Local Python SDK measurements on MPS, excluding HTTP, Convex, retrieval, generation and browser rendering.",
    "Jev prompts and thresholds were kept fixed; this is a direct substitution experiment, not optimized Laya performance.",
    "Neither Jev nor the legacy LLM was run in this experiment; no relative speed or financial saving was measured.",
    "A single cold-start sample excludes downloads and operating-system cache eviction.",
    "Process RSS and MPS driver memory overlap on unified memory and must not be added together.",
    "This aggregate extract does not contain the full runner, corpus or per-request token audit.",
    "One RAG evaluation case was excluded from the main scores after inspecting the first run; this is exploratory evidence, not a preregistered confirmatory study."
  ],
  "annotationReview": {
    "timing": "After the first measurement run",
    "excludedFromPrimaryScores": [
      "rag-holdout-injection-en"
    ],
    "reason": "The inherited empty gold set conflates final-answer abstention with passage relevance. A passage can legitimately correct the query premise. Raw cases and outputs were preserved, not relabelled; prompts, thresholds and weights were unchanged."
  }
}
