{
  "cohorts": {
    "full": {
      "bootstrap": {
        "conditions_on_fitted_models": true,
        "interval": "95% percentile, linear quantiles",
        "macro_f1_labels": 60,
        "normalization": "NFKC, casefold, whitespace collapse",
        "pairing": "Identical sampled cluster multiplicities for every method and seed in a replicate",
        "replicates": 2000,
        "rng": "numpy.random.default_rng / PCG64",
        "sampled_row_count_max": 3003,
        "sampled_row_count_min": 2953,
        "secondary_intervals": "Descriptive; no multiple-testing adjustment",
        "seed": 2026,
        "training_seed_uncertainty_in_interval": false,
        "unit": "normalized text cluster; sample the observed number of clusters with replacement",
        "zero_division": 0
      },
      "contrasts": [
        {
          "baseline": "base",
          "left": "lora-2400-mean",
          "metrics": {
            "accuracy": {
              "ci95_percentile": [
                0.4259238988061824,
                0.461306732407981
              ],
              "difference": 0.44395875364268106
            },
            "macro_f1": {
              "ci95_percentile": [
                0.37720748136639737,
                0.4166539161343707
              ],
              "difference": 0.3986528614920507
            }
          },
          "name": "primary-lora2400-minus-base",
          "priority": "primary",
          "right": "qwen-base",
          "size": 2400
        },
        {
          "baseline": "base",
          "left": "lora-600-mean",
          "metrics": {
            "accuracy": {
              "ci95_percentile": [
                0.3171776727604923,
                0.3502571276244637
              ],
              "difference": 0.33355749831876264
            },
            "macro_f1": {
              "ci95_percentile": [
                0.2846923694480344,
                0.32292624328240066
              ],
              "difference": 0.3056934328609028
            }
          },
          "name": "secondary-lora600-minus-base",
          "priority": "secondary-descriptive",
          "right": "qwen-base",
          "size": 600
        },
        {
          "baseline": "tfidf",
          "left": "lora-600-mean",
          "metrics": {
            "accuracy": {
              "ci95_percentile": [
                0.07525598240714348,
                0.11349957405575523
              ],
              "difference": 0.09414929388029591
            },
            "macro_f1": {
              "ci95_percentile": [
                0.1318070782850404,
                0.1798914641582536
              ],
              "difference": 0.15468159119690317
            }
          },
          "name": "secondary-lora600-minus-tfidf600",
          "priority": "secondary-descriptive",
          "right": "tfidf-600",
          "size": 600
        },
        {
          "baseline": "encoder",
          "left": "lora-600-mean",
          "metrics": {
            "accuracy": {
              "ci95_percentile": [
                -0.05066153306283083,
                -0.015375594152082622
              ],
              "difference": -0.03328850033624753
            },
            "macro_f1": {
              "ci95_percentile": [
                -0.01602434703235806,
                0.0313315231245057
              ],
              "difference": 0.007047494830933698
            }
          },
          "name": "secondary-lora600-minus-encoder600",
          "priority": "secondary-descriptive",
          "right": "encoder-600",
          "size": 600
        },
        {
          "baseline": "tfidf",
          "left": "lora-2400-mean",
          "metrics": {
            "accuracy": {
              "ci95_percentile": [
                0.04445578746169782,
                0.07508041534476287
              ],
              "difference": 0.059291638646043476
            },
            "macro_f1": {
              "ci95_percentile": [
                0.05705168634327283,
                0.10260042334251636
              ],
              "difference": 0.07884249777235075
            }
          },
          "name": "secondary-lora2400-minus-tfidf2400",
          "priority": "secondary-descriptive",
          "right": "tfidf-2400",
          "size": 2400
        },
        {
          "baseline": "encoder",
          "left": "lora-2400-mean",
          "metrics": {
            "accuracy": {
              "ci95_percentile": [
                -0.010564728275100568,
                0.0192722400435584
              ],
              "difference": 0.00448329970858552
            },
            "macro_f1": {
              "ci95_percentile": [
                -0.01644203957996825,
                0.029662700251698974
              ],
              "difference": 0.0043663065941428325
            }
          },
          "name": "secondary-lora2400-minus-encoder2400",
          "priority": "secondary-descriptive",
          "right": "encoder-2400",
          "size": 2400
        }
      ],
      "count": 2974,
      "excludedCount": 0,
      "label": "Full official test",
      "labelCount": 60,
      "loraMeans": {
        "2400": {
          "definition": "Arithmetic mean of three independently fitted seed scores; no prediction ensemble.",
          "members": [
            "lora-2400-s0",
            "lora-2400-s1",
            "lora-2400-s2"
          ],
          "metrics": {
            "accuracy": {
              "max": 0.7787491593813046,
              "mean": 0.752633938578794,
              "min": 0.7303295225285811,
              "sample_sd_across_training_seeds": 0.024433726340501297,
              "seed_values": [
                0.7787491593813046,
                0.7488231338264963,
                0.7303295225285811
              ]
            },
            "invalid_rate": {
              "max": 0.0023537323470073975,
              "mean": 0.0020174848688634837,
              "min": 0.0016812373907195697,
              "sample_sd_across_training_seeds": 0.0003362474781439139,
              "seed_values": [
                0.0023537323470073975,
                0.0016812373907195697,
                0.0020174848688634837
              ]
            },
            "macro_f1": {
              "max": 0.7379371049367126,
              "mean": 0.7021871160431004,
              "min": 0.6774568735626664,
              "sample_sd_across_training_seeds": 0.031710261825197955,
              "seed_values": [
                0.7379371049367126,
                0.6911673696299222,
                0.6774568735626664
              ]
            }
          }
        },
        "600": {
          "definition": "Arithmetic mean of three independently fitted seed scores; no prediction ensemble.",
          "members": [
            "lora-600-s0",
            "lora-600-s1",
            "lora-600-s2"
          ],
          "metrics": {
            "accuracy": {
              "max": 0.6809011432414257,
              "mean": 0.6422326832548756,
              "min": 0.6102891728312038,
              "sample_sd_across_training_seeds": 0.03578311475082846,
              "seed_values": [
                0.6102891728312038,
                0.6809011432414257,
                0.6355077336919973
              ]
            },
            "invalid_rate": {
              "max": 0.0070611970410221925,
              "mean": 0.005940372113875813,
              "min": 0.005043712172158709,
              "sample_sd_across_training_seeds": 0.00102725301388833,
              "seed_values": [
                0.0057162071284465365,
                0.005043712172158709,
                0.0070611970410221925
              ]
            },
            "macro_f1": {
              "max": 0.6319876544837586,
              "mean": 0.6092276874119524,
              "min": 0.591579247344328,
              "sample_sd_across_training_seeds": 0.020683462551573804,
              "seed_values": [
                0.591579247344328,
                0.6319876544837586,
                0.6041161604077707
              ]
            }
          }
        }
      },
      "methods": {
        "encoder-2400": {
          "accuracy": 0.7481506388702085,
          "count": 2974,
          "invalid_count": 0,
          "invalid_rate": 0.0,
          "macro_f1": 0.6978208094489575
        },
        "encoder-600": {
          "accuracy": 0.6755211835911231,
          "count": 2974,
          "invalid_count": 0,
          "invalid_rate": 0.0,
          "macro_f1": 0.6021801925810187
        },
        "lora-2400-s0": {
          "accuracy": 0.7787491593813046,
          "count": 2974,
          "invalid_count": 7,
          "invalid_rate": 0.0023537323470073975,
          "macro_f1": 0.7379371049367126
        },
        "lora-2400-s1": {
          "accuracy": 0.7488231338264963,
          "count": 2974,
          "invalid_count": 5,
          "invalid_rate": 0.0016812373907195697,
          "macro_f1": 0.6911673696299222
        },
        "lora-2400-s2": {
          "accuracy": 0.7303295225285811,
          "count": 2974,
          "invalid_count": 6,
          "invalid_rate": 0.0020174848688634837,
          "macro_f1": 0.6774568735626664
        },
        "lora-600-s0": {
          "accuracy": 0.6102891728312038,
          "count": 2974,
          "invalid_count": 17,
          "invalid_rate": 0.0057162071284465365,
          "macro_f1": 0.591579247344328
        },
        "lora-600-s1": {
          "accuracy": 0.6809011432414257,
          "count": 2974,
          "invalid_count": 15,
          "invalid_rate": 0.005043712172158709,
          "macro_f1": 0.6319876544837586
        },
        "lora-600-s2": {
          "accuracy": 0.6355077336919973,
          "count": 2974,
          "invalid_count": 21,
          "invalid_rate": 0.0070611970410221925,
          "macro_f1": 0.6041161604077707
        },
        "qwen-base": {
          "accuracy": 0.30867518493611296,
          "count": 2974,
          "invalid_count": 134,
          "invalid_rate": 0.04505716207128446,
          "macro_f1": 0.30353425455104965
        },
        "tfidf-2400": {
          "accuracy": 0.6933422999327505,
          "count": 2974,
          "invalid_count": 0,
          "invalid_rate": 0.0,
          "macro_f1": 0.6233446182707496
        },
        "tfidf-600": {
          "accuracy": 0.5480833893745797,
          "count": 2974,
          "invalid_count": 0,
          "invalid_rate": 0.0,
          "macro_f1": 0.45454609621504927
        }
      },
      "overlap": null,
      "textClusters": 2944,
      "zeroSupportLabels": [
        "cooking_query"
      ]
    },
    "overlapExcluded": {
      "bootstrap": {
        "conditions_on_fitted_models": true,
        "interval": "95% percentile, linear quantiles",
        "macro_f1_labels": 60,
        "normalization": "NFKC, casefold, whitespace collapse",
        "pairing": "Identical sampled cluster multiplicities for every method and seed in a replicate",
        "replicates": 2000,
        "rng": "numpy.random.default_rng / PCG64",
        "sampled_row_count_max": 2910,
        "sampled_row_count_min": 2882,
        "secondary_intervals": "Descriptive; no multiple-testing adjustment",
        "seed": 2026,
        "training_seed_uncertainty_in_interval": false,
        "unit": "normalized text cluster; sample the observed number of clusters with replacement",
        "zero_division": 0
      },
      "contrasts": [
        {
          "baseline": "base",
          "left": "lora-2400-mean",
          "metrics": {
            "accuracy": {
              "ci95_percentile": [
                0.4267309675036804,
                0.4629386075147575
              ],
              "difference": 0.44517392305920295
            },
            "macro_f1": {
              "ci95_percentile": [
                0.37522812948098483,
                0.41619701614815446
              ],
              "difference": 0.39704145233809096
            }
          },
          "name": "primary-lora2400-minus-base",
          "priority": "prespecified-overlap-sensitivity",
          "right": "qwen-base",
          "size": 2400
        },
        {
          "baseline": "base",
          "left": "lora-600-mean",
          "metrics": {
            "accuracy": {
              "ci95_percentile": [
                0.3162471741968351,
                0.35049510650546917
              ],
              "difference": 0.3334485141672426
            },
            "macro_f1": {
              "ci95_percentile": [
                0.2835710435747383,
                0.32192228539101586
              ],
              "difference": 0.3033941193077284
            }
          },
          "name": "secondary-lora600-minus-base",
          "priority": "prespecified-overlap-sensitivity",
          "right": "qwen-base",
          "size": 600
        },
        {
          "baseline": "tfidf",
          "left": "lora-600-mean",
          "metrics": {
            "accuracy": {
              "ci95_percentile": [
                0.07631764067905843,
                0.1153862772208779
              ],
              "difference": 0.0960608154803041
            },
            "macro_f1": {
              "ci95_percentile": [
                0.13742952921306126,
                0.1871142395425354
              ],
              "difference": 0.1610666233652865
            }
          },
          "name": "secondary-lora600-minus-tfidf600",
          "priority": "prespecified-overlap-sensitivity",
          "right": "tfidf-600",
          "size": 600
        },
        {
          "baseline": "encoder",
          "left": "lora-600-mean",
          "metrics": {
            "accuracy": {
              "ci95_percentile": [
                -0.05216654353946343,
                -0.015905809989199802
              ],
              "difference": -0.03386316516931587
            },
            "macro_f1": {
              "ci95_percentile": [
                -0.017517357536387577,
                0.03220652183458698
              ],
              "difference": 0.007159733570966664
            }
          },
          "name": "secondary-lora600-minus-encoder600",
          "priority": "prespecified-overlap-sensitivity",
          "right": "encoder-600",
          "size": 600
        },
        {
          "baseline": "tfidf",
          "left": "lora-2400-mean",
          "metrics": {
            "accuracy": {
              "ci95_percentile": [
                0.044801900531618574,
                0.07752914019180886
              ],
              "difference": 0.061621746141442
            },
            "macro_f1": {
              "ci95_percentile": [
                0.061478425817709764,
                0.11098085554729574
              ],
              "difference": 0.08443972448590886
            }
          },
          "name": "secondary-lora2400-minus-tfidf2400",
          "priority": "prespecified-overlap-sensitivity",
          "right": "tfidf-2400",
          "size": 2400
        },
        {
          "baseline": "encoder",
          "left": "lora-2400-mean",
          "metrics": {
            "accuracy": {
              "ci95_percentile": [
                -0.01023341382085774,
                0.019225461455873623
              ],
              "difference": 0.004952775858097169
            },
            "macro_f1": {
              "ci95_percentile": [
                -0.016826062745644417,
                0.03299271810450332
              ],
              "difference": 0.004996964897476586
            }
          },
          "name": "secondary-lora2400-minus-encoder2400",
          "priority": "prespecified-overlap-sensitivity",
          "right": "encoder-2400",
          "size": 2400
        }
      ],
      "count": 2894,
      "excludedCount": 80,
      "label": "Exact-text overlap excluded",
      "labelCount": 60,
      "loraMeans": {
        "2400": {
          "definition": "Arithmetic mean of three independently fitted seed scores; no prediction ensemble.",
          "members": [
            "lora-2400-s0",
            "lora-2400-s1",
            "lora-2400-s2"
          ],
          "metrics": {
            "accuracy": {
              "max": 0.7743607463718037,
              "mean": 0.7482146970744068,
              "min": 0.7263303386316516,
              "sample_sd_across_training_seeds": 0.024297150524839717,
              "seed_values": [
                0.7743607463718037,
                0.743953006219765,
                0.7263303386316516
              ]
            },
            "invalid_rate": {
              "max": 0.002073255010366275,
              "mean": 0.0019580741764570375,
              "min": 0.0017277125086385626,
              "sample_sd_across_training_seeds": 0.00019949905638895151,
              "seed_values": [
                0.002073255010366275,
                0.0017277125086385626,
                0.002073255010366275
              ]
            },
            "macro_f1": {
              "max": 0.7333181793682376,
              "mean": 0.6976840655696689,
              "min": 0.6732431972922516,
              "sample_sd_across_training_seeds": 0.031562912204517735,
              "seed_values": [
                0.7333181793682376,
                0.6864908200485174,
                0.6732431972922516
              ]
            }
          }
        },
        "600": {
          "definition": "Arithmetic mean of three independently fitted seed scores; no prediction ensemble.",
          "members": [
            "lora-600-s0",
            "lora-600-s1",
            "lora-600-s2"
          ],
          "metrics": {
            "accuracy": {
              "max": 0.6751900483759502,
              "mean": 0.6364892881824464,
              "min": 0.60573600552868,
              "sample_sd_across_training_seeds": 0.035402511441860775,
              "seed_values": [
                0.60573600552868,
                0.6751900483759502,
                0.6285418106427091
              ]
            },
            "invalid_rate": {
              "max": 0.007256392536281963,
              "mean": 0.006104584197189588,
              "min": 0.005183137525915688,
              "sample_sd_across_training_seeds": 0.0010556497799944344,
              "seed_values": [
                0.005874222529371113,
                0.005183137525915688,
                0.007256392536281963
              ]
            },
            "macro_f1": {
              "max": 0.6274212752963169,
              "mean": 0.6040367325393063,
              "min": 0.5863109107165674,
              "sample_sd_across_training_seeds": 0.02113128867175292,
              "seed_values": [
                0.5863109107165674,
                0.6274212752963169,
                0.5983780116050347
              ]
            }
          }
        }
      },
      "methods": {
        "encoder-2400": {
          "accuracy": 0.7432619212163096,
          "count": 2894,
          "invalid_count": 0,
          "invalid_rate": 0.0,
          "macro_f1": 0.6926871006721923
        },
        "encoder-600": {
          "accuracy": 0.6703524533517623,
          "count": 2894,
          "invalid_count": 0,
          "invalid_rate": 0.0,
          "macro_f1": 0.5968769989683397
        },
        "lora-2400-s0": {
          "accuracy": 0.7743607463718037,
          "count": 2894,
          "invalid_count": 6,
          "invalid_rate": 0.002073255010366275,
          "macro_f1": 0.7333181793682376
        },
        "lora-2400-s1": {
          "accuracy": 0.743953006219765,
          "count": 2894,
          "invalid_count": 5,
          "invalid_rate": 0.0017277125086385626,
          "macro_f1": 0.6864908200485174
        },
        "lora-2400-s2": {
          "accuracy": 0.7263303386316516,
          "count": 2894,
          "invalid_count": 6,
          "invalid_rate": 0.002073255010366275,
          "macro_f1": 0.6732431972922516
        },
        "lora-600-s0": {
          "accuracy": 0.60573600552868,
          "count": 2894,
          "invalid_count": 17,
          "invalid_rate": 0.005874222529371113,
          "macro_f1": 0.5863109107165674
        },
        "lora-600-s1": {
          "accuracy": 0.6751900483759502,
          "count": 2894,
          "invalid_count": 15,
          "invalid_rate": 0.005183137525915688,
          "macro_f1": 0.6274212752963169
        },
        "lora-600-s2": {
          "accuracy": 0.6285418106427091,
          "count": 2894,
          "invalid_count": 21,
          "invalid_rate": 0.007256392536281963,
          "macro_f1": 0.5983780116050347
        },
        "qwen-base": {
          "accuracy": 0.30304077401520385,
          "count": 2894,
          "invalid_count": 134,
          "invalid_rate": 0.04630269523151347,
          "macro_f1": 0.3006426132315779
        },
        "tfidf-2400": {
          "accuracy": 0.6865929509329648,
          "count": 2894,
          "invalid_count": 0,
          "invalid_rate": 0.0,
          "macro_f1": 0.61324434108376
        },
        "tfidf-600": {
          "accuracy": 0.5404284727021423,
          "count": 2894,
          "invalid_count": 0,
          "invalid_rate": 0.0,
          "macro_f1": 0.44297010917401985
        }
      },
      "overlap": {
        "excludedCount": 80,
        "policy": "Exclude test normalized texts present in frozen train-2400 OR dev; do not use the unused training pool.",
        "referenceDevCount": 2033,
        "referenceTrainCount": 2400,
        "referenceUniqueTexts": 4364,
        "retainedCount": 2894
      },
      "textClusters": 2876,
      "zeroSupportLabels": [
        "cooking_query",
        "general_greet"
      ]
    }
  },
  "examples": {
    "agreementCounts": {
      "all_three_lora_correct": 1914,
      "all_three_lora_correct_encoder_wrong": 226,
      "all_three_lora_same_wrong_label": 227,
      "all_three_lora_same_wrong_label_encoder_correct": 77,
      "all_three_lora_same_wrong_label_encoder_other_wrong_label": 57,
      "all_three_lora_same_wrong_label_encoder_same_wrong_label": 93,
      "all_three_lora_wrong_any_predicted_label": 463,
      "lora_seed_prediction_disagreement": 833
    },
    "agreementWarning": "Agreement counts describe overlap among three separate models; no majority vote, consensus prediction, or ensemble performance is computed.",
    "cohort": "full",
    "items": [
      {
        "encoder_2400": "calendar_query",
        "gold": "calendar_set",
        "id": "13457",
        "lora_seed_0": "calendar_query",
        "lora_seed_1": "calendar_query",
        "lora_seed_2": "calendar_query",
        "selection_rule": "First lexicographic ID in this top-three shared-confusion pair",
        "shared_pair_rank": 1,
        "utterance": "dime cuándo debo irme a un evento programado para que llegue a tiempo"
      },
      {
        "encoder_2400": "play_radio",
        "gold": "play_radio",
        "id": "9233",
        "lora_seed_0": "play_music",
        "lora_seed_1": "play_music",
        "lora_seed_2": "play_music",
        "selection_rule": "First lexicographic ID in this top-three shared-confusion pair",
        "shared_pair_rank": 2,
        "utterance": "abre spotify"
      },
      {
        "encoder_2400": "calendar_set",
        "gold": "calendar_query",
        "id": "7219",
        "lora_seed_0": "calendar_set",
        "lora_seed_1": "calendar_set",
        "lora_seed_2": "calendar_set",
        "selection_rule": "First lexicographic ID in this top-three shared-confusion pair",
        "shared_pair_rank": 3,
        "utterance": "pusiste el recordatorio sobre la reunión de mañana"
      },
      {
        "encoder_2400": "takeaway_order",
        "gold": "cooking_recipe",
        "id": "10090",
        "lora_seed_0": "cooking_recipe",
        "lora_seed_1": "cooking_recipe",
        "lora_seed_2": "cooking_recipe",
        "selection_rule": "First lexicographic ID where all three LoRA seeds are correct and the encoder is wrong",
        "utterance": "me puedes conseguir una receta de atún"
      },
      {
        "encoder_2400": "cooking_recipe",
        "gold": "cooking_recipe",
        "id": "10027",
        "lora_seed_0": "cooking_recipe",
        "lora_seed_1": "cooking_recipe",
        "lora_seed_2": "takeaway_order",
        "selection_rule": "First lexicographic ID where LoRA seed predictions differ",
        "utterance": "instrucciones para hacer una comida"
      }
    ],
    "pairRankingRule": "Shared pairs: descending number of rows with the identical wrong label in all three LoRA seeds; ties by (gold, predicted) lexicographic order. Pooled pairs: descending sum of the three individual error counts, then the same tie rule.",
    "postHoc": true,
    "scope": [
      "lora-2400-s0",
      "lora-2400-s1",
      "lora-2400-s2",
      "encoder-2400"
    ],
    "selectionWarning": "Five descriptive examples selected after test evaluation by the recorded ID-order rules. They are not a random sample, new predictions, or confirmatory evidence; controls above do not change their scope.",
    "trainingBudget": 2400
  },
  "latency": {
    "batchSize": 1,
    "count": 100,
    "ids": [
      "10058",
      "10129",
      "10162",
      "10453",
      "10458",
      "10496",
      "10697",
      "10730",
      "10830",
      "11377",
      "11928",
      "1211",
      "12185",
      "1223",
      "12322",
      "1234",
      "12470",
      "12840",
      "12901",
      "13063",
      "13110",
      "1327",
      "13481",
      "13647",
      "13797",
      "13806",
      "13848",
      "13870",
      "14012",
      "14145",
      "14346",
      "14397",
      "14574",
      "14849",
      "14888",
      "15397",
      "15436",
      "15836",
      "15909",
      "15930",
      "1615",
      "16206",
      "16319",
      "16421",
      "16699",
      "1680",
      "16833",
      "16903",
      "17080",
      "17093",
      "1732",
      "1901",
      "2009",
      "2112",
      "2161",
      "230",
      "2303",
      "2600",
      "2623",
      "3004",
      "318",
      "331",
      "3437",
      "3551",
      "3811",
      "4046",
      "406",
      "4681",
      "4822",
      "4836",
      "5064",
      "5094",
      "5182",
      "5232",
      "5249",
      "5290",
      "536",
      "562",
      "5621",
      "6259",
      "6540",
      "6898",
      "6912",
      "7184",
      "7541",
      "7699",
      "7727",
      "803",
      "8132",
      "8434",
      "8778",
      "8899",
      "9208",
      "9214",
      "925",
      "9428",
      "9838",
      "9876",
      "9892",
      "9934"
    ],
    "limitation": "Single shared-machine run; filesystem cache, system load and backend differences limit benchmark precision.",
    "lora2400MeanOfSeedMediansMs": 102.82912482701552,
    "lora2400ToEncoder2400MedianRatio": 13.478716061178558,
    "medianRatioDefinition": "Mean of the three LoRA-2400 medians divided by the encoder-2400 median; describes medians only.",
    "methods": {
      "encoder-2400": {
        "cold_model_loading_seconds": 5.750644915999146,
        "max_ms": 783.5627500026021,
        "median_ms": 7.6289999997243285,
        "min_ms": 4.733625013614073,
        "over_100ms_count": 12,
        "p95_ms": 370.95820799004287,
        "process_lifetime_peak_rss_bytes": 1016741888
      },
      "encoder-600": {
        "cold_model_loading_seconds": 4.298992666997947,
        "max_ms": 19.492374995024875,
        "median_ms": 5.941791503573768,
        "min_ms": 4.717665986390784,
        "over_100ms_count": 0,
        "p95_ms": 11.061166995204985,
        "process_lifetime_peak_rss_bytes": 1033027584
      },
      "lora-2400-s0": {
        "cold_model_loading_seconds": 1.6848436250002123,
        "max_ms": 120.72116698254831,
        "median_ms": 87.76047900028061,
        "min_ms": 79.46849998552352,
        "over_100ms_count": 7,
        "p95_ms": 103.39087501051836,
        "process_lifetime_peak_rss_bytes": 1364082688
      },
      "lora-2400-s1": {
        "cold_model_loading_seconds": 1.7083219590131193,
        "max_ms": 140.74704199447297,
        "median_ms": 104.86856249917764,
        "min_ms": 82.31670799432322,
        "over_100ms_count": 72,
        "p95_ms": 126.77770800655708,
        "process_lifetime_peak_rss_bytes": 1538506752
      },
      "lora-2400-s2": {
        "cold_model_loading_seconds": 1.4948612909938674,
        "max_ms": 281.5039169800002,
        "median_ms": 115.85833298158832,
        "min_ms": 84.71541697508655,
        "over_100ms_count": 91,
        "p95_ms": 176.6332919942215,
        "process_lifetime_peak_rss_bytes": 1404665856
      },
      "lora-600-s0": {
        "cold_model_loading_seconds": 1.9181890420150012,
        "max_ms": 160.39087501121685,
        "median_ms": 112.80956250266172,
        "min_ms": 82.42891699774191,
        "over_100ms_count": 80,
        "p95_ms": 138.0997920059599,
        "process_lifetime_peak_rss_bytes": 1239629824
      },
      "lora-600-s1": {
        "cold_model_loading_seconds": 6.370983667002292,
        "max_ms": 351.2558749935124,
        "median_ms": 113.00712499360088,
        "min_ms": 94.14683299837634,
        "over_100ms_count": 98,
        "p95_ms": 147.2149160108529,
        "process_lifetime_peak_rss_bytes": 1048313856
      },
      "lora-600-s2": {
        "cold_model_loading_seconds": 1.4596021250181366,
        "max_ms": 165.87691599852405,
        "median_ms": 109.03708299156278,
        "min_ms": 77.09845900535583,
        "over_100ms_count": 78,
        "p95_ms": 128.1229160085786,
        "process_lifetime_peak_rss_bytes": 2168913920
      },
      "qwen-base": {
        "cold_model_loading_seconds": 1.570847833994776,
        "max_ms": 128.11091600451618,
        "median_ms": 98.84564600361045,
        "min_ms": 70.68062500911765,
        "over_100ms_count": 43,
        "p95_ms": 112.52799999783747,
        "process_lifetime_peak_rss_bytes": 2269298688
      },
      "tfidf-2400": {
        "cold_model_loading_seconds": 0.6072952910035383,
        "max_ms": 0.31712499912828207,
        "median_ms": 0.14781249046791345,
        "min_ms": 0.13800000306218863,
        "over_100ms_count": 0,
        "p95_ms": 0.20550002227537334,
        "process_lifetime_peak_rss_bytes": 128401408
      },
      "tfidf-600": {
        "cold_model_loading_seconds": 0.4381652500014752,
        "max_ms": 0.3207910049241036,
        "median_ms": 0.15462501323781908,
        "min_ms": 0.14083299902267754,
        "over_100ms_count": 0,
        "p95_ms": 0.2382919774390757,
        "process_lifetime_peak_rss_bytes": 123486208
      }
    },
    "p95Definition": "Nearest rank: sorted observation at ceil(0.95 × 100), using one-based rank.",
    "sampling": "lowest SHA256(engineering-v1:ID), then canonical ID order",
    "split": "dev",
    "timingDefinition": "batch-one tokenization/feature extraction, inference, pooling/head when applicable, and output selection; local elapsed time; one warmup excluded",
    "warmupCount": 1,
    "warmupExcluded": true,
    "warmupId": "10058"
  },
  "resources": {
    "max_process_rss_bytes": 1349320704,
    "max_training_mlx_bytes": 3263566658,
    "measurementScope": "Training-plus-save time sums all 18 comparative LoRA fits; excludes development inference, initial model loading and engineering work. MLX allocator peaks and process-lifetime RSS overlap and must not be added.",
    "search_process_wall_seconds": 5926.002510083141,
    "search_training_and_save_seconds": 5881.035098291963,
    "selected": {
      "lora-2400-s0": {
        "adapter_bytes": 4986697,
        "mlx_peak_bytes": 3263501122,
        "observed": {
          "epochs": [
            {
              "complete": true,
              "epoch": 1,
              "examples": 2400,
              "ids_in_order_sha256": "7f5b3a2e3f747c33f1a6488269d9d20a946512c783720a202871698842a65d14",
              "unique_ids": 2400
            },
            {
              "complete": true,
              "epoch": 2,
              "examples": 2400,
              "ids_in_order_sha256": "ad24f4df71d8b8304ec834d4d579cc5b1bb52aa8a5c1da942bffff6a3c675b7f",
              "unique_ids": 2400
            }
          ],
          "examples": 4800,
          "final_reported_supervised_tokens": 21980,
          "microsteps": 1200,
          "optimizer_updates": 1200,
          "real_sequence_tokens": 1414844,
          "supervised_tokens": 21980,
          "trainable_parameters": 1245184
        },
        "training_seconds": 528.7362137079763
      },
      "lora-2400-s1": {
        "adapter_bytes": 4986697,
        "mlx_peak_bytes": 3263501122,
        "observed": {
          "epochs": [
            {
              "complete": true,
              "epoch": 1,
              "examples": 2400,
              "ids_in_order_sha256": "de84ce591a521f1d134035f40627f8e9a7a3527e279d7ec48358540e9ceac595",
              "unique_ids": 2400
            },
            {
              "complete": true,
              "epoch": 2,
              "examples": 2400,
              "ids_in_order_sha256": "d5cc564b3198aa619b614fecb501ace4c019a52e1d6119b0c8bbc3f46e9a7ae9",
              "unique_ids": 2400
            }
          ],
          "examples": 4800,
          "final_reported_supervised_tokens": 21980,
          "microsteps": 1200,
          "optimizer_updates": 1200,
          "real_sequence_tokens": 1414844,
          "supervised_tokens": 21980,
          "trainable_parameters": 1245184
        },
        "training_seconds": 550.8137194579758
      },
      "lora-2400-s2": {
        "adapter_bytes": 4986697,
        "mlx_peak_bytes": 3263566658,
        "observed": {
          "epochs": [
            {
              "complete": true,
              "epoch": 1,
              "examples": 2400,
              "ids_in_order_sha256": "2ba47b2d0dae05732189a7364a9319fafe162778804a193ff08b73bac70f0b20",
              "unique_ids": 2400
            },
            {
              "complete": true,
              "epoch": 2,
              "examples": 2400,
              "ids_in_order_sha256": "79c84e037949e48480b22dca990c79ad8594f239ece495b3432cf82847f09c37",
              "unique_ids": 2400
            }
          ],
          "examples": 4800,
          "final_reported_supervised_tokens": 21980,
          "microsteps": 1200,
          "optimizer_updates": 1200,
          "real_sequence_tokens": 1414844,
          "supervised_tokens": 21980,
          "trainable_parameters": 1245184
        },
        "training_seconds": 532.0316342080187
      },
      "lora-600-s0": {
        "adapter_bytes": 4986697,
        "mlx_peak_bytes": 3159431440,
        "observed": {
          "epochs": [
            {
              "complete": true,
              "epoch": 1,
              "examples": 600,
              "ids_in_order_sha256": "4791b10cd1d2e779d1fa7922e3d96c283ce191cb09cd0ef138524e6bf7129e7b",
              "unique_ids": 600
            },
            {
              "complete": true,
              "epoch": 2,
              "examples": 600,
              "ids_in_order_sha256": "7ce58e434e4fd6d171f8ed33089be1647f90a3e7b85453b0a4622c2d7b84fae9",
              "unique_ids": 600
            }
          ],
          "examples": 1200,
          "final_reported_supervised_tokens": 5494,
          "microsteps": 300,
          "optimizer_updates": 300,
          "real_sequence_tokens": 353548,
          "supervised_tokens": 5494,
          "trainable_parameters": 1245184
        },
        "training_seconds": 129.85348495902144
      },
      "lora-600-s1": {
        "adapter_bytes": 4986697,
        "mlx_peak_bytes": 3159480592,
        "observed": {
          "epochs": [
            {
              "complete": true,
              "epoch": 1,
              "examples": 600,
              "ids_in_order_sha256": "24b3e332697829c63c25a19e641fab74bafa09679d21a44cab9131b26ed7a737",
              "unique_ids": 600
            },
            {
              "complete": true,
              "epoch": 2,
              "examples": 600,
              "ids_in_order_sha256": "6acf63a6a31abf1f3ae1d5126680976e80aa076c01579160b89b88bcc1dd0c02",
              "unique_ids": 600
            }
          ],
          "examples": 1200,
          "final_reported_supervised_tokens": 5494,
          "microsteps": 300,
          "optimizer_updates": 300,
          "real_sequence_tokens": 353548,
          "supervised_tokens": 5494,
          "trainable_parameters": 1245184
        },
        "training_seconds": 130.17549654198228
      },
      "lora-600-s2": {
        "adapter_bytes": 4986697,
        "mlx_peak_bytes": 3159431440,
        "observed": {
          "epochs": [
            {
              "complete": true,
              "epoch": 1,
              "examples": 600,
              "ids_in_order_sha256": "3404a4f18f594e758f7bd102e73aff26b115e041a61321d08753c32d7c9b661a",
              "unique_ids": 600
            },
            {
              "complete": true,
              "epoch": 2,
              "examples": 600,
              "ids_in_order_sha256": "86d11ab40eb59e16c3540241bd63f99144c8086848726fc05d566a42b2016e8d",
              "unique_ids": 600
            }
          ],
          "examples": 1200,
          "final_reported_supervised_tokens": 5494,
          "microsteps": 300,
          "optimizer_updates": 300,
          "real_sequence_tokens": 353548,
          "supervised_tokens": 5494,
          "trainable_parameters": 1245184
        },
        "training_seconds": 131.4142886250047
      }
    },
    "trainable_parameters": [
      1245184
    ]
  },
  "schema": "lora-intent-adaptation-results-v1",
  "sourceHashes": {
    "algorithm": "SHA-256",
    "canonicalFiles": {
      "STUDY_PROTOCOL.md": "c57eb57c633d5ac7430e422d4a4319cd9eb73e0664341001eb5f40d4c6ae4f52",
      "data/prepared/manifest.json": "d0604dadefe770b0e163a75b7dff928c71883b55f9d67d6aa760f0391260b452",
      "reports/encoder-model-provenance.json": "6ce39e35a2d707c69328ab88a8eb25a3823faecd014907a6b589c12ef571236d",
      "reports/final-independent-audit.json": "b9089bd3502efe4e4ad088464ee4f20ec5818fd5f9efb805bd2b671e58c5c1bc",
      "reports/massive-provenance.json": "773b36f2dfd3df54c4c74e919ef05d6bb677218ee162e0ed9eca0a24d18c01c4",
      "reports/model-provenance.json": "e0360d87afeb2fd0761238dd7bc82315b899902b6fecd9963e451c22f7878887",
      "reports/posthoc-error-analysis.json": "794919866805c478a00eac575716b7b14406e734c05fac40a129ec5d0e51d4de",
      "studies/massive-es-2026-10-09/final/analysis/bootstrap-statistics.npz": "d0fec630d245c350bdafdc81cdb73d1348afde8ea5f76ad0d6605a9167c1acaa",
      "studies/massive-es-2026-10-09/final/analysis/report.json": "cbc53d4f14b8746ef0fd3b2149df508a6d7ec11003de521d2818e13bd48043e1",
      "studies/massive-es-2026-10-09/final/evaluations/encoder-2400/predictions.jsonl": "0d41c60e7f329b4e92a44af8cc74b56b705c15b8b8264a165ae707b403938125",
      "studies/massive-es-2026-10-09/final/evaluations/encoder-600/predictions.jsonl": "17eece8bb06c8ab9a95ca95cc299f0dbe5fb81861fb786ea5c5582f4d0b38224",
      "studies/massive-es-2026-10-09/final/evaluations/lora-2400-s0/predictions.jsonl": "4aee2d44d7d0cd65eda3dd253d405bc9f05c2e2195eb66f4b6a29c2de1fff86e",
      "studies/massive-es-2026-10-09/final/evaluations/lora-2400-s1/predictions.jsonl": "6da469e7cf427bf8a8a86792416ccace361effaf3e3eb981140d11c6c0b2bf19",
      "studies/massive-es-2026-10-09/final/evaluations/lora-2400-s2/predictions.jsonl": "f8fbce23cdbe0496d822d896cd3bbc0fddc007ccdd9ccd26132223a90e1957ee",
      "studies/massive-es-2026-10-09/final/evaluations/lora-600-s0/predictions.jsonl": "88bfa990fc2641b76f05d15e58bf6d7900e99485149ad9429f1b25559d0633d3",
      "studies/massive-es-2026-10-09/final/evaluations/lora-600-s1/predictions.jsonl": "b60dcefeac68f5dafd60a9af665c75007d6db0bbb19311cf52032e6b19c90f95",
      "studies/massive-es-2026-10-09/final/evaluations/lora-600-s2/predictions.jsonl": "2631376fe90ea8b7cc2512e8bc76f7f8c68d3242c6cadae3af95dcd50d8aad45",
      "studies/massive-es-2026-10-09/final/evaluations/qwen-base/predictions.jsonl": "65d4031bc25dae63c654d313feb9b8fe97bc7c93f84c4de70e9cc6ae98004bd2",
      "studies/massive-es-2026-10-09/final/evaluations/tfidf-2400/predictions.jsonl": "809b1ee12ca45d150e2c0b1a8ddecdd241d2ed0cd04fffb051402f71bd00826c",
      "studies/massive-es-2026-10-09/final/evaluations/tfidf-600/predictions.jsonl": "44c5238efefc7ee612ca4631852199c2794c1098516661c79574a230ac0936ab",
      "studies/massive-es-2026-10-09/final/freeze.json": "d5d4f78fd4cbdb101907d5ee56df0c3102486cdcb893434743eb8fb6bd614857",
      "studies/massive-es-2026-10-09/final/latency/encoder-2400.json": "1cf950c7a928be5f12b8973bd10a0bdef1526b2ea4b63e355097f7d387d1f3b4",
      "studies/massive-es-2026-10-09/final/latency/encoder-600.json": "f6edd8dad8de109d19cf80502f4c773aed03205a4ff67ab4f1e32675a757734a",
      "studies/massive-es-2026-10-09/final/latency/lora-2400-s0.json": "fd28b1caa1215353cad9d31acf5554c1b76b4ac041938141b7e3c2910083f7c1",
      "studies/massive-es-2026-10-09/final/latency/lora-2400-s1.json": "e164b4abe26c479032506c3200f3a7fb31f5da92e8f88dc40683d1a9c96b0233",
      "studies/massive-es-2026-10-09/final/latency/lora-2400-s2.json": "2a834e2127447710b84c76baae95eb42138b53d51ca4b24c9c46c68dfe17e32f",
      "studies/massive-es-2026-10-09/final/latency/lora-600-s0.json": "0f7e70fe39c760e747caa38ee702608b7492933a73d58a76b3f6c875273d1059",
      "studies/massive-es-2026-10-09/final/latency/lora-600-s1.json": "26ec0888fdf79e0eb029e9cdaa84d229a72eb6aa4c69a210cc0ad19f900ffa5e",
      "studies/massive-es-2026-10-09/final/latency/lora-600-s2.json": "f07d41687a4710e0103d070a33ea229abf87fa6fa4e98f637a61281b45e503b8",
      "studies/massive-es-2026-10-09/final/latency/qwen-base.json": "2e85de4284217b451ddd2540ef5d608ea9de165e25ac77024ee0c6052b077492",
      "studies/massive-es-2026-10-09/final/latency/tfidf-2400.json": "c44825b7744148c80c3438fab1aca5cd85ce50de0b3a499492ef4d8fdbae8521",
      "studies/massive-es-2026-10-09/final/latency/tfidf-600.json": "7e670700d80712ab4fe27c34073b13b672d804d3bea199865c990f5bf1f58bce",
      "studies/massive-es-2026-10-09/final/report/manifest.json": "669d35bac4a541bec86d1b5a76d79dae2c5bbcd895b2722c3a1ed3a568a3b4db",
      "studies/massive-es-2026-10-09/final/report/paired-contrasts.png": "0141cf35ab360031850a890f1fc301430a1a545f2a9a6e14a907541b7f11014a",
      "studies/massive-es-2026-10-09/final/report/paired-contrasts.svg": "a303f59c22b5cfaa5d5d4a37a3614fb851a7b6501ea85739694a7804ee031f5e",
      "studies/massive-es-2026-10-09/final/report/quality-latency.png": "e11b0ae786c5ae9651fd2bea8ca673ce19f4ffdec3ec600112d5914749a1ac1b",
      "studies/massive-es-2026-10-09/final/report/quality-latency.svg": "afdf447a6328362818c6a422e026edcc58275d9433ad2d56596f5e380bd51ef8",
      "studies/massive-es-2026-10-09/final/report/report.pdf": "ea4232354d852e8796ab7525da7befa63beaf64df105d98b246bf206642df479",
      "studies/massive-es-2026-10-09/final/report/resources.json": "06ceacae54d89d694a152dd96cb5a08f21a1aa74f75f5601a05744257f69d163",
      "studies/massive-es-2026-10-09/final/report/sample-efficiency.png": "d8b4c01373994a7154de3046c38b4a426553633eeba58ee4a114661bf18a36e5",
      "studies/massive-es-2026-10-09/final/report/sample-efficiency.svg": "2efad8a9c60f8af6fcf72e5a97446678d9d8eb28859eb50d0dfb465e2056d965"
    },
    "freezeId": "a488a4e8087899ff834b2efff59a110fb2064382b2d422f73f7f72899d24b5eb",
    "scope": "Byte hashes of the original local canonical evidence, identified by paths relative to the study repository root. Public repository copies may sanitize local paths; exported PDF and figures retain the original bytes."
  },
  "study": {
    "aiAssistance": "Developed with AI assistance; the author is responsible for the study design and interpretation.",
    "dataUrl": "/assets/lora-intent-adaptation/results.json",
    "dataset": {
      "archiveSha256": "4cba5faa11c71437928e17cb1b9b3d8b8e727e7ea363a3a9a8045e19c0491577",
      "archiveUrl": "https://amazon-massive-nlu-dataset.s3.amazonaws.com/amazon-massive-dataset-1.1.tar.gz",
      "attribution": "MASSIVE, FitzGerald et al. (2022), Amazon. Five Spanish utterances are reproduced unchanged from MASSIVE 1.1 es-ES; model outputs and post-hoc selection notes were added by this independent study.",
      "license": "CC BY 4.0",
      "licenseUrl": "https://creativecommons.org/licenses/by/4.0/",
      "locale": "es-ES",
      "name": "MASSIVE 1.1",
      "paperUrl": "https://arxiv.org/abs/2204.08582",
      "spanishSourceSha256": "310462a79fa181ff83c643a8d356c7b8155fd37a25e80a77ba3ca9b29305c4a5",
      "url": "https://github.com/alexa/massive"
    },
    "date": "2026-10-09",
    "finalMethodInstances": 11,
    "finalPredictions": 32714,
    "generation": {
      "greedy": true,
      "labelConstraint": false,
      "maxNewTokens": 32,
      "outputRepair": false,
      "thinking": false
    },
    "hardware": {
      "chip": "Apple M5 Pro",
      "unifiedMemoryGiB": 24
    },
    "interpretation": {
      "interval_scope": "Paired normalized-text cluster bootstrap conditions on the three fitted adapters; it does not capture the full distribution of training outcomes.",
      "primary": "Mean of three selected LoRA2400 seed macro-F1 scores minus unchanged Qwen macro-F1 on the complete official test.",
      "secondary": "Secondary and sensitivity intervals are descriptive without multiple-testing adjustment; retain negative results and both cohorts.",
      "seed_variation": "Sample standard deviation across the three fitted training seeds, reported separately; no seed was selected."
    },
    "labelCount": 60,
    "labels": [
      "alarm_query",
      "alarm_remove",
      "alarm_set",
      "audio_volume_down",
      "audio_volume_mute",
      "audio_volume_other",
      "audio_volume_up",
      "calendar_query",
      "calendar_remove",
      "calendar_set",
      "cooking_query",
      "cooking_recipe",
      "datetime_convert",
      "datetime_query",
      "email_addcontact",
      "email_query",
      "email_querycontact",
      "email_sendemail",
      "general_greet",
      "general_joke",
      "general_quirky",
      "iot_cleaning",
      "iot_coffee",
      "iot_hue_lightchange",
      "iot_hue_lightdim",
      "iot_hue_lightoff",
      "iot_hue_lighton",
      "iot_hue_lightup",
      "iot_wemo_off",
      "iot_wemo_on",
      "lists_createoradd",
      "lists_query",
      "lists_remove",
      "music_dislikeness",
      "music_likeness",
      "music_query",
      "music_settings",
      "news_query",
      "play_audiobook",
      "play_game",
      "play_music",
      "play_podcasts",
      "play_radio",
      "qa_currency",
      "qa_definition",
      "qa_factoid",
      "qa_maths",
      "qa_stock",
      "recommendation_events",
      "recommendation_locations",
      "recommendation_movies",
      "social_post",
      "social_query",
      "takeaway_order",
      "takeaway_query",
      "transport_query",
      "transport_taxi",
      "transport_ticket",
      "transport_traffic",
      "weather_query"
    ],
    "learningRatesSearched": [
      1e-05,
      3e-05,
      0.0001
    ],
    "limitations": [
      "One locale, task, quantized language model, short training budget and three fitted seeds; models differ in capacity and pretraining.",
      "An interval crossing zero establishes neither LoRA superiority nor equivalence to the frozen encoder.",
      "The prospective comparative plan followed known historical TF-IDF development results and a five-update engineering smoke run; this is not full preregistration.",
      "Exact-text overlap is addressed by a prespecified sensitivity cohort. Paraphrase overlap and pretraining exposure remain unknown.",
      "Latency comes from one run on a shared machine. Encoder-2400 has a lower median and a higher p95 than all three LoRA-2400 runs; the cause of its irregular tail is unknown."
    ],
    "liveInference": false,
    "locale": "es-ES",
    "lora": {
      "batchSize": 4,
      "epochs": 2,
      "lastLayers": 4,
      "projectionsPerLayer": 7,
      "rank": 8,
      "scale": 20,
      "trainableParameters": 1245184
    },
    "model": "Qwen3-1.7B, four-bit MLX weights",
    "models": {
      "encoder": {
        "repository": "sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2",
        "revision": "e8f8c211226b894fcb81acc59f3b34ba3efd5f42",
        "url": "https://huggingface.co/sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2/blob/e8f8c211226b894fcb81acc59f3b34ba3efd5f42/README.md"
      },
      "qwen": {
        "repository": "mlx-community/Qwen3-1.7B-4bit",
        "revision": "3b1b1768f8f8cf8351c712464f906e86c2b8269e",
        "url": "https://huggingface.co/mlx-community/Qwen3-1.7B-4bit/tree/3b1b1768f8f8cf8351c712464f906e86c2b8269e"
      }
    },
    "reportUrl": "/reports/lora-intent-adaptation.pdf",
    "repositoryUrl": "https://github.com/eduardoceto/lora-intent-adaptation",
    "sampling": {
      "description": "nested proportional largest-deficit allocation; minimum one example per label",
      "nested": true,
      "seed": 20261008
    },
    "savedResults": true,
    "scoring": "Exact whole-label matching after surrounding whitespace is removed; invalid labels count as errors. Macro-F1 averages all 60 frozen labels, including cooking_query with zero test support.",
    "selectedLearningRates": {
      "2400": 0.0001,
      "600": 0.0001
    },
    "selection": "One learning rate per budget selected by mean development macro-F1 across all three seeds; all three selected-rate seeds retained. Complete method selection frozen before final scoring.",
    "splitCounts": {
      "dev": 2033,
      "test": 2974,
      "train": 11514
    },
    "status": "Completed independent local comparative study",
    "title": "Spanish intent adaptation with LoRA",
    "trainingBudgets": [
      600,
      2400
    ],
    "trainingRuns": 18,
    "trainingSeeds": [
      0,
      1,
      2
    ]
  }
}
