{
  "revision": "91ff15aa854e8c5ca38d1bbf",
  "updated_at": "2026-10-04T18:31:07+00:00",
  "exported_at": "2026-10-04T18:33:26+00:00",
  "protocol": "3809e829ce612de2b66a",
  "claims": [
    {
      "id": "starter-validation-measurement",
      "title": "Untouched starter validation measurement",
      "kind": "observation",
      "status": "supported",
      "statement": "The untouched starter scored ROC-AUC 0.9572004065863107 on the frozen validation set; the recorded run used 400 GPU CatBoost trees and completed the candidate in 4.7448 seconds. This is a single measurement, not evidence of improvement or a mechanism.",
      "scope": "Airline satisfaction S6E10; frozen validation protocol, 489,744 training rows and 104,945 search-validation rows; RTX 2080 Ti physical index 1/logical device 0; starter configuration, seed 20261004, 180-second cap.",
      "supports": [
        "e-63f527c11cd4620a00b3c393"
      ],
      "opposes": [],
      "alternatives": [
        "GPU CatBoost is not bitwise deterministic, and the score may vary across runs.",
        "The separate task-brief starter measurement differs; measurements need not be directly comparable."
      ],
      "next_test": "Compare prespecified feature ablations against matched starter runs on the frozen validation protocol, repeating seeds where feasible to estimate run variation before deciding whether a difference is meaningful.",
      "updated_at": "2026-10-04T10:14:59+00:00"
    },
    {
      "id": "catboost-gpu-nondeterminism",
      "title": "CatBoost documentation describes GPU training as nondeterministic",
      "kind": "experience",
      "status": "supported",
      "statement": "The supplied CatBoost GPU documentation attributes nondeterministic training to floating-point summation order. This supports treating GPU score variation as possible but does not quantify variation for this task.",
      "scope": "CatBoost GPU training as described in the supplied documentation excerpt; no airline-specific score, configuration, or noise estimate.",
      "supports": [
        "e-19a209702a616e0cdc6a924e",
        "e-3674965902fffbb1cccb7a0d",
        "e-ab0caf3bac9abf38f3d4e681",
        "e-e51e973482c1fdc119bf73aa"
      ],
      "opposes": [],
      "alternatives": [
        "The documentation does not establish that any particular observed score difference is due to nondeterminism.",
        "Variation may also arise from differences in data, software, or execution configuration.",
        "The excerpt is documentation evidence, not an independent experimental replication."
      ],
      "next_test": "Repeat identical paired starter and feature-ablation runs with recorded seeds and configuration to estimate variation under the task protocol.",
      "updated_at": "2026-10-04T10:43:03+00:00"
    },
    {
      "id": "rating-summary-split-hypothesis",
      "title": "Rating summaries may provide useful tree splits",
      "kind": "hypothesis",
      "status": "untested",
      "statement": "An overall service-rating mean may improve ROC-AUC over raw features at starter settings. Two same-seed quick blocks favored mean-only over raw, and one adaptive search-validation mean run scored 0.9572675016680855; its difference from a separate valid root measurement was +0.0000670951. This does not establish a reliable validation improvement or a split-selection mechanism.",
      "scope": "Airline satisfaction S6E10; proposed rating-summary feature ablation versus the 400-tree GPU CatBoost starter on frozen validation data, under the same training pool and 180-second cap; no results supplied.",
      "supports": [
        "e-a88a499c59ab3499d264d09e"
      ],
      "opposes": [],
      "alternatives": [
        "The validation comparison is not a fresh matched control; its small difference may reflect GPU/run variation.",
        "Adaptive selection and quick results may overstate transfer to validation.",
        "A mean may alter finite-model split competition, but AUC alone cannot identify that mechanism.",
        "Evidence for the mean does not establish utility of dispersion, contrasts, or segment interactions."
      ],
      "next_test": "Run paired, prespecified mean-only and raw validation comparisons with repeated seeds where feasible. Distinguish mean-feature utility from run variation and adaptive-selection effects; change the default only if matched results support a practically relevant difference.",
      "updated_at": "2026-10-04T10:14:59+00:00"
    },
    {
      "id": "quick-rating-summary-ablation",
      "title": "Quick scores for raw ratings, mean-only, and full-profile feature modes",
      "kind": "observation",
      "status": "supported",
      "statement": "On the frozen quick split, the n002 mean-only arm scored ROC-AUC 0.9529834489267927 versus 0.9525373326142099 for raw features and 0.9523817935601727 for the full profile. The mean-over-raw difference was +0.0004461, close to the n001 difference of +0.0004461. Both same-seed blocks ranked mean above raw above profile; this is directional repeatability on one split, not independent-seed replication or evidence of a mechanism.",
      "scope": "Airline satisfaction S6E10; frozen quick split, 50,000 training and 10,000 development rows; seed 20261004; RTX 2080 Ti physical index 1/logical 0; 400-tree depth-6 GPU CatBoost; three single-run arms, 60-second cap.",
      "supports": [
        "e-a88a499c59ab3499d264d09e",
        "e-a991bea26e83fa3a5a1836d7"
      ],
      "opposes": [],
      "alternatives": [
        "The two blocks use the same seed and split and are not independent replication; GPU variation remains unquantified.",
        "Feature modes alter multiple inputs, so score differences do not identify which features matter or why.",
        "Quick results do not establish validation or final-set benefit."
      ],
      "next_test": "Run prespecified paired repeats with distinct recorded seeds if the quick interface permits, and compare raw, mean-only, and profile arms. Distinguish repeatable feature-mode differences from GPU/run variation; change the default only if the differences remain decision-relevant.",
      "updated_at": "2026-10-04T10:14:59+00:00"
    },
    {
      "id": "full-profile-versus-raw-quick",
      "title": "Full service-profile features scored below raw features in two quick blocks",
      "kind": "observation",
      "status": "supported",
      "statement": "The n002 full-profile arm scored 0.9523817935601727 versus 0.9525373326142099 for raw features, a difference of -0.0001555 ROC-AUC. The earlier n001 difference was -0.0000804. Both single-seed blocks ranked the tested profile below raw; they do not establish a reliable disadvantage or identify an individual feature effect.",
      "scope": "Airline satisfaction S6E10; frozen quick split, 50,000 training and 10,000 development rows; seed 20261004; RTX 2080 Ti physical index 1/logical 0; 400-tree depth-6 GPU CatBoost; full-profile versus raw single-run comparison, 60-second cap.",
      "supports": [
        "e-a88a499c59ab3499d264d09e",
        "e-a991bea26e83fa3a5a1836d7"
      ],
      "opposes": [],
      "alternatives": [
        "The blocks share seed and split and do not estimate independent-seed variation.",
        "The profile bundles multiple summaries, so redundancy or split competition may explain its score without any one feature being harmful.",
        "A small observed difference may be run variation."
      ],
      "next_test": "Repeat the prespecified profile-versus-raw comparison with distinct seeds and isolate profile components only if the bundle difference is stable. Distinguish bundle-level reproducibility from individual feature effects; omit the bundle only if repeated evidence supports that decision.",
      "updated_at": "2026-10-04T10:14:59+00:00"
    },
    {
      "id": "catboost-greedy-tree-structure-docs",
      "title": "CatBoost documentation describes greedy tree construction and symmetric trees",
      "kind": "experience",
      "status": "supported",
      "statement": "The supplied CatBoost documentation describes greedy tree construction and states that SymmetricTree builds level by level, splitting leaves at a level under a shared condition. This is design context only; it does not establish that any added rating summary or interaction improves airline ROC-AUC or works through a particular split-selection mechanism.",
      "scope": "CatBoost tree-structure selection as described in the supplied documentation excerpt; no airline-specific dataset, configuration, score, or feature-ablation result.",
      "supports": [
        "e-694e6783426c23859e7cc77f",
        "e-6a67c84e9cf53d815f701619",
        "e-80183cd00c072c1a68f8156d",
        "e-dfd3086fc47b41f87056d7e4",
        "e-f1cd25f0549c37e5830df239"
      ],
      "opposes": [],
      "alternatives": [
        "Documentation describes algorithm design, not the effect of a feature intervention.",
        "Score differences could reflect run variation or other consequences of changing the feature set.",
        "Literature novelty is unknown without relevant literature sources."
      ],
      "next_test": "Compare prespecified raw-rating and summary or interaction arms under identical settings with paired repeats; inspect selected splits separately if available. Distinguish score changes from the proposed split-selection explanation.",
      "updated_at": "2026-10-04T15:29:39+00:00"
    },
    {
      "id": "n002-mean-validation-measurement",
      "title": "Mean-feature arm validation measurements",
      "kind": "observation",
      "status": "supported",
      "statement": "The n002 mean-feature arm scored ROC-AUC 0.9572675016680855 with recorded candidate elapsed time 4.8972 seconds. In n004, the restored mean/400 configuration scored 0.9572919711976331; candidate elapsed time was 4.9475 seconds. These are separate measurements of the same configuration, not a matched feature ablation or evidence of improvement or mechanism.",
      "scope": "Airline satisfaction S6E10; n002 frozen search-validation evaluation, 489,744 training rows and 104,945 scoring rows; overall mean of 13 service ratings added to raw features; 400-tree depth-6 learning-rate-0.08 GPU Plain CatBoost, seed 20261004, RTX 2080 Ti physical index 1/logical device 0, 180-second cap.",
      "supports": [
        "e-67f4d6ec016c302d3bd22746",
        "e-a88a499c59ab3499d264d09e"
      ],
      "opposes": [],
      "alternatives": [
        "The measurements are separate runs; their small score difference may reflect GPU or run variation.",
        "These are adaptive search-validation measurements, not independent confirmation.",
        "Neither measurement identifies a mechanism or establishes performance on the sealed final set."
      ],
      "next_test": "Run matched raw-feature and mean-feature validation arms under the same protocol with repeated recorded seeds. Distinguish a stable mean-feature score difference from run variation; retain the mean only if matched evidence supports a decision-relevant difference.",
      "updated_at": "2026-10-04T10:55:37+00:00"
    },
    {
      "id": "n003-digital-segments-validation-measurement",
      "title": "Digital-segment feature arm validation score and runtime",
      "kind": "observation",
      "status": "supported",
      "statement": "The n003 digital-segment arm scored ROC-AUC 0.9573126801034795 on frozen search-validation rows. Candidate elapsed time was 5.1972 seconds, below the 180-second cap. The score was +0.0000452 versus the n002 mean-feature measurement, but this was not a fresh matched comparison and does not establish improvement or mechanism.",
      "scope": "Airline satisfaction S6E10; n003 frozen search-validation evaluation, 489,744 training rows and 104,945 scoring rows; overall service mean, digital-service mean, and digital mean gated by business travel, loyal customer, and business class; 400-tree depth-6 learning-rate-0.08 GPU Plain CatBoost, seed 20261004, RTX 2080 Ti physical index 1/logical device 0, 180-second cap.",
      "supports": [
        "e-29cb1794b2c71490a0d6e4a8"
      ],
      "opposes": [],
      "alternatives": [
        "The comparison to n002 is unmatched and the small difference may reflect GPU/run variation.",
        "This is adaptive search-validation evidence, not independent confirmation.",
        "The arm combines multiple feature changes and does not identify an individual gate effect or mechanism."
      ],
      "next_test": "Run matched validation arms for overall mean alone, overall plus ungated digital mean, and overall plus the three gates, using repeated recorded seeds where feasible. Distinguish gate utility from ungated-feature effects and run variation; change the default only if the net difference is repeatable and decision-relevant.",
      "updated_at": "2026-10-04T10:36:49+00:00"
    },
    {
      "id": "n003-digital-gates-quick-ablation",
      "title": "Reversed-order quick block repeats the digital-gate score ordering",
      "kind": "observation",
      "status": "supported",
      "statement": "In n005's reversed C/B/A quick block, the overall-mean arm scored 0.9529834894, the ungated digital-mean arm 0.9524785041, and the gated arm 0.9530102702. The gated arm exceeded the baseline by 0.0000268 and the ungated arm by 0.0005318; the ungated arm was 0.0005050 below baseline. These differences closely repeat n003's same-seed, same-split A/B/C block, but the blocks are not independent replication and do not establish reliable effects.",
      "scope": "Airline satisfaction S6E10; frozen quick split, 50,000 training and 10,000 scoring-only development rows; n003 ordered A/B/C arms, seed 20261004, RTX 2080 Ti physical index 1/logical device 0; 400-tree depth-6 learning-rate-0.08 GPU Plain CatBoost, 60-second cap.",
      "supports": [
        "e-29cb1794b2c71490a0d6e4a8",
        "e-a872fdf892d6b4233248a75f"
      ],
      "opposes": [],
      "alternatives": [
        "Both blocks use the same seed and quick split, so their close agreement is not independent replication.",
        "Extra numeric columns may alter split competition without encoding useful segment interactions.",
        "The small gated-versus-baseline difference may be run variation; the harness diagnostic is heuristic-only."
      ],
      "next_test": "Repeat the prespecified A/B/C comparison in paired blocks with distinct recorded seeds. Distinguish repeatable feature-mode differences from GPU/run variation and same-split dependence; change the feature default only if the net baseline comparison is reproducible and decision-relevant.",
      "updated_at": "2026-10-04T11:16:45+00:00"
    },
    {
      "id": "digital-gates-may-help-fixed-capacity-model",
      "title": "Digital-service gates repeat a small quick-score increment but miss the adoption margin",
      "kind": "hypothesis",
      "status": "mixed",
      "statement": "In n005, the gated bundle scored above both the overall-mean baseline and ungated digital-mean arm in a reversed-order quick block; the signs and near-identical differences also appeared in n003. However, the n005 gated-minus-baseline difference (+0.0000268) was below its prospectively specified 0.0002 engineering adoption margin. The evidence does not support retaining this bundle as the default under that decision rule, nor establish that all gates lack value or identify a segment mechanism.",
      "scope": "Airline satisfaction S6E10; proposed comparison of overall service mean against that baseline plus digital-service mean gated by business travel, loyalty, and class; 400-tree depth-6 learning-rate-0.08 GPU Plain CatBoost under the frozen quick and search-validation protocols; existing evidence uses seed 20261004 and is not a matched repeated validation comparison.",
      "supports": [
        "e-29cb1794b2c71490a0d6e4a8",
        "e-a872fdf892d6b4233248a75f"
      ],
      "opposes": [],
      "alternatives": [
        "Gates may expose useful finite-depth split directions conditional on the ungated digital mean, without improving net score over the overall-mean baseline.",
        "Added columns may change numeric borders or greedy split competition without useful semantic interaction value.",
        "Same-split, same-seed blocks do not estimate independent-seed variation or validate the segment constructs.",
        "The 0.0002 threshold is an engineering adoption rule, not a noise estimate or statistical significance threshold."
      ],
      "next_test": "If reconsidering the bundle, run prespecified paired repeats with distinct seeds and a matched validation A/B/C comparison. Distinguish net gate utility from run variation, ungated-feature effects, and added-column split competition; change the default only if the prespecified practical margin is met reproducibly.",
      "updated_at": "2026-10-04T11:16:45+00:00"
    },
    {
      "id": "ungated-digital-mean-not-beneficial-in-n003-quick-block",
      "title": "Ungated digital mean scored below the overall-mean arm in one quick block",
      "kind": "observation",
      "status": "supported",
      "statement": "In the n003 quick block, adding an ungated digital-service mean to the overall service mean scored 0.0005050 ROC-AUC below the overall-mean arm. This single comparison does not establish a reliable disadvantage or imply that digital summaries are universally unhelpful.",
      "scope": "Airline satisfaction S6E10; frozen quick split, 50,000 training and 10,000 scoring-only development rows; n003 overall-mean versus overall-plus-ungated-digital-mean arms, seed 20261004, RTX 2080 Ti physical index 1/logical device 0; 400-tree depth-6 learning-rate-0.08 GPU Plain CatBoost, 60-second cap.",
      "supports": [
        "e-29cb1794b2c71490a0d6e4a8"
      ],
      "opposes": [],
      "alternatives": [
        "One run per arm cannot quantify GPU variation.",
        "The score difference may reflect finite-model split competition rather than harmful information in the feature.",
        "The quick result does not establish the effect on validation or final data."
      ],
      "next_test": "Repeat matched overall-mean and ungated-digital-mean arms with distinct seeds. Distinguish a stable ungated-feature penalty from run variation and interactions with the gated bundle; omit the ungated feature only if repeated comparisons support that choice.",
      "updated_at": "2026-10-04T10:36:49+00:00"
    },
    {
      "id": "n004-fixed1600-capacity-quick-outcome",
      "title": "Quick scores for fixed-400 and fixed-1600-round feature arms",
      "kind": "observation",
      "status": "supported",
      "statement": "In the n004 ordered quick block, mean/400 scored 0.9529834894, raw/1600 scored 0.9506237039, and mean/1600 scored 0.9507057073. Both 1600-round arms scored below mean/400 by about 0.0023; mean/1600 exceeded raw/1600 by 0.0000820. This single-seed block does not estimate run variation or explain the score differences. The subsequent validation evaluation measured restored mean/400, not either 1600-round arm.",
      "scope": "Airline satisfaction S6E10; n004 frozen quick split, 50,000 training and 10,000 scoring-only development rows; ordered mean/400, raw/1600, mean/1600 arms, seed 20261004; RTX 2080 Ti physical index 1/logical device 0; depth-6 learning-rate-0.08 GPU Plain CatBoost, 60-second cap. The n004 validation run evaluated restored mean/400 only.",
      "supports": [
        "e-67f4d6ec016c302d3bd22746"
      ],
      "opposes": [],
      "alternatives": [
        "The comparison uses one quick split and seed; GPU variation and run order are unquantified.",
        "Quick uses fewer training rows than validation, so the result may not transfer to the full training pool.",
        "Without raw/400, the block does not identify a capacity-by-feature interaction.",
        "The score differences do not establish overfitting or a particular optimization mechanism."
      ],
      "next_test": "Repeat the fixed-round comparison with paired recorded seeds and include raw/400 to complete the feature-mode-by-round-count comparison. Distinguish run variation, quick-split-specific behavior, and a feature-mode interaction; reconsider increased rounds only if the repeated matched results improve on the appropriate 400-round baseline.",
      "updated_at": "2026-10-04T10:55:37+00:00"
    },
    {
      "id": "n004-fixed1600-improves-capacity-limited-ensemble",
      "title": "Increasing rounds to 1600 improves on mean/400",
      "kind": "hypothesis",
      "status": "refuted",
      "statement": "The n004 prediction that increasing from 400 to 1600 rounds at fixed depth 6 and learning rate 0.08 would improve ROC-AUC over mean/400 was not supported in the single quick block: both tested 1600-round arms scored lower. This scoped result does not establish that longer training is generally harmful or determine full-pool validation performance.",
      "scope": "Airline satisfaction S6E10; n004 frozen quick split, 50,000 training and 10,000 scoring-only development rows; comparison of mean/400 with raw/1600 and mean/1600; seed 20261004; RTX 2080 Ti physical index 1/logical device 0; depth-6 learning-rate-0.08 GPU Plain CatBoost, 60-second cap.",
      "supports": [],
      "opposes": [
        "e-67f4d6ec016c302d3bd22746"
      ],
      "alternatives": [
        "The single-seed result may include GPU variation.",
        "The fixed quick training subset may favor a different optimization trajectory than the full training pool.",
        "Overfitting or optimization-trajectory effects are possible but were not identified by learning curves or other controls."
      ],
      "next_test": "Repeat the prespecified round-count and feature-mode comparisons with paired seeds, including raw/400. Distinguish a reproducible quick-split decline from run variation and feature-mode interaction; retain 1600 rounds only if matched results improve on the corresponding 400-round arm.",
      "updated_at": "2026-10-04T10:55:37+00:00"
    },
    {
      "id": "n005-mean-validation-measurement",
      "title": "Mean-feature arm validation score and runtime in n005",
      "kind": "observation",
      "status": "supported",
      "statement": "The n005 delivered overall-mean arm scored ROC-AUC 0.9572758458565709 on frozen search-validation rows. Candidate elapsed time was 4.9980 seconds, below the 180-second cap. This was not a matched validation comparison against raw features or the digital-gate arms, and does not establish improvement or mechanism.",
      "scope": "Airline satisfaction S6E10; n005 frozen search-validation evaluation, 489,744 training rows and 104,945 scoring rows; overall service mean added to raw features; 400-tree depth-6 learning-rate-0.08 GPU Plain CatBoost, seed 20261004, RTX 2080 Ti physical index 1/logical device 0, 180-second cap.",
      "supports": [
        "e-a872fdf892d6b4233248a75f"
      ],
      "opposes": [],
      "alternatives": [
        "Historical mean-arm scores are separate runs and their small differences may reflect GPU or run variation.",
        "This is adaptive search-validation evidence, not independent confirmation.",
        "No raw-feature or digital-gate control was measured in this validation run."
      ],
      "next_test": "Run matched validation arms for raw features, mean-only, and the prespecified digital feature modes, with repeated recorded seeds where feasible. Distinguish feature-mode differences from run variation and adaptive-selection effects; change the default only if matched evidence supports a decision-relevant difference.",
      "updated_at": "2026-10-04T11:16:45+00:00"
    },
    {
      "id": "catboost-greedy-split-competition-hypothesis",
      "title": "Aggregate features may change candidate split selection",
      "kind": "hypothesis",
      "status": "untested",
      "statement": "CatBoost documentation describes selecting candidate trees and split scores during sequential boosting. This provides an algorithmic rationale for testing whether an aggregate rating feature changes candidate competition or the optimization path, but does not show that it improves airline ROC-AUC or that split competition explains any observed score difference.",
      "scope": "Algorithm-design context for CatBoost tree construction; no airline feature or score evidence; no task-specific mechanism identification.",
      "supports": [
        "e-6cb1d41624ff34e251702d27",
        "e-f653a1643ebe2654fbe2f15b"
      ],
      "opposes": [],
      "alternatives": [
        "An aggregate feature could expose a useful threshold not readily represented by individual raw-feature thresholds.",
        "Any score change could instead arise from altered candidate competition without useful interaction information.",
        "The documentation describes general algorithm design, not this dataset's behavior."
      ],
      "next_test": "Compare matched raw and aggregate-feature arms while recording selected splits or otherwise controlling candidate features. Distinguish useful aggregate split directions from generic changes in candidate competition; claim task-specific mechanism only if the control identifies it.",
      "updated_at": "2026-10-04T15:29:39+00:00"
    },
    {
      "id": "catboost-gpu-nondeterminism-magnitude-unknown",
      "title": "GPU training can be nondeterministic; task-specific variation is unquantified",
      "kind": "experience",
      "status": "supported",
      "statement": "The collected CatBoost documentation attributes GPU training nondeterminism to floating-point summation order. It establishes a possible source of variation, not its magnitude for this task or an explanation for any particular score difference.",
      "scope": "General CatBoost GPU training context; task uses RTX 2080 Ti physical index 1/logical device 0 under the frozen airline protocol; task-specific variation magnitude unknown.",
      "supports": [
        "e-cadf9184021af158fb695d8b"
      ],
      "opposes": [],
      "alternatives": [
        "The documented nondeterminism may be negligible for a particular task configuration.",
        "Variation can also arise from implementation, data ordering, or other uncontrolled run conditions.",
        "This source provides no uncertainty estimate for airline ROC-AUC."
      ],
      "next_test": "Repeat an otherwise identical candidate configuration across recorded runs and seeds under the authorized GPU protocol. Distinguish measurable run-to-run variation from feature-mode differences; defer small score-based decisions if variation is comparable to the observed difference.",
      "updated_at": "2026-10-04T11:22:25+00:00"
    },
    {
      "id": "n006-mean-validation-measurement",
      "title": "n006 validation measurement of the retained mean-feature arm",
      "kind": "observation",
      "status": "supported",
      "statement": "The n006 official search-validation evaluation measured the retained overall-mean arm at ROC-AUC 0.9572675425024442, with candidate elapsed time 5.0991 seconds. It evaluated 489,744 training rows and 104,945 scoring rows. This was not a count-feature evaluation or a matched comparison establishing improvement or mechanism.",
      "scope": "Airline satisfaction S6E10; n006 frozen search-validation evaluation, 489,744 training rows and 104,945 scoring rows; overall service mean added to raw features; 400-tree depth-6 learning-rate-0.08 GPU Plain CatBoost, seed 20261004, RTX 2080 Ti physical index 1/logical device 0, 180-second cap.",
      "supports": [
        "e-0b9cf72b7244cf181acd82e6"
      ],
      "opposes": [],
      "alternatives": [
        "The near-equality to earlier mean-arm scores is not an independent estimate of generalization or run variation.",
        "This adaptive search-validation measurement does not test the count feature; final remains sealed.",
        "The score does not identify a mechanism."
      ],
      "next_test": "If assessing the count feature on validation, run a matched mean-only versus mean-plus-count comparison under the same protocol, with recorded repeats where feasible. Distinguish count utility from run variation; change the feature decision only on repeatable, decision-relevant matched evidence.",
      "updated_at": "2026-10-04T11:37:14+00:00"
    },
    {
      "id": "n006-low-count-quick-observation",
      "title": "Mean-plus-low-count quick scores were below the mean-only control",
      "kind": "observation",
      "status": "supported",
      "statement": "In n006's B1/A/B2 quick block, the mean-plus-count arms scored 0.9527878400 and 0.9527878805, while the intervening mean-only control scored 0.9529835705. Both count-minus-control differences were about -0.0001957, below the prespecified +0.0002 adoption margin. The close B1/B2 scores do not quantify uncertainty or establish a reliable disadvantage.",
      "scope": "Airline satisfaction S6E10; n006 frozen quick split, 50,000 training and 10,000 scoring-only development rows; mean-only versus mean plus count of the exact 13 service ratings <=1; B1/A/B2 order, seed 20261004; RTX 2080 Ti physical index 1/logical device 0; 400-tree depth-6 learning-rate-0.08 GPU Plain CatBoost, 60-second cap.",
      "supports": [
        "e-0b9cf72b7244cf181acd82e6"
      ],
      "opposes": [],
      "alternatives": [
        "The two treatment fits share a split and seed and are not independent replication.",
        "The single fresh control and GPU nondeterminism leave score uncertainty unknown.",
        "A count may be redundant given raw ratings and their mean, or may alter finite-model split competition; these results do not distinguish those explanations.",
        "Quick results do not establish full-pool count performance."
      ],
      "next_test": "If reconsidering this augmentation, run paired comparisons with distinct recorded seeds and retain the same feature definition. Distinguish a repeatable conditional count effect from GPU/run variation; change the decision only if the prespecified practical adoption rule is met reproducibly.",
      "updated_at": "2026-10-04T11:37:14+00:00"
    },
    {
      "id": "n006-count-adoption-hypothesis",
      "title": "The exact low-rating count did not meet its quick adoption rule",
      "kind": "hypothesis",
      "status": "refuted",
      "statement": "For the n006 quick protocol, the hypothesis that adding only the count of the exact 13 service ratings <=1 to raw inputs and the overall mean would yield repeatable, decision-relevant utility was not supported: both valid count-minus-control differences were negative and below the prespecified +0.0002 margin. This retires the declared quick adoption prediction, not all count thresholds, service summaries, or full-pool effects.",
      "scope": "Airline satisfaction S6E10; n006 frozen quick split, 50,000 training and 10,000 scoring-only development rows; raw features plus overall service mean, compared with the same inputs plus count of the exact 13 service ratings <=1; B1/A/B2, seed 20261004; RTX 2080 Ti physical index 1/logical device 0; 400-tree depth-6 learning-rate-0.08 GPU Plain CatBoost, 60-second cap.",
      "supports": [],
      "opposes": [
        "e-0b9cf72b7244cf181acd82e6"
      ],
      "alternatives": [
        "The quick training subset may behave differently from full-pool training; no official count arm was evaluated.",
        "The result does not distinguish information redundancy from changes to numeric borders or greedy split competition.",
        "GPU variation is unquantified, and the two treatment fits are correlated."
      ],
      "next_test": "Do not retain this count as default under the stated quick adoption rule. Only revisit with prespecified distinct-seed paired repeats or a matched validation comparison; distinguish redundancy, split competition, and run variation, and reverse the decision only with reproducible decision-relevant benefit.",
      "updated_at": "2026-10-04T11:37:14+00:00"
    },
    {
      "id": "catboost-gpu-nondeterminism-documentation",
      "title": "CatBoost documentation describes GPU training as nondeterministic",
      "kind": "experience",
      "status": "supported",
      "statement": "A collected CatBoost documentation excerpt attributes GPU training nondeterminism to floating-point summation order. This describes implementation behavior generally; it does not measure run-to-run variation for this airline task or explain any particular score difference.",
      "scope": "CatBoost GPU training as described by the cited documentation; no task-specific dataset, configuration, or measured variance.",
      "supports": [
        "e-adc9e4bcf107479745d4fa7a"
      ],
      "opposes": [],
      "alternatives": [
        "The excerpt does not establish the magnitude or practical importance of variation under the frozen task configuration.",
        "Other run conditions could also affect observed score differences."
      ],
      "next_test": "Repeat identical authorized task runs and quantify score variation; distinguish task-specific variation from the documentation’s general implementation description.",
      "updated_at": "2026-10-04T11:42:51+00:00"
    },
    {
      "id": "greedy-split-selection-mechanism-context",
      "title": "CatBoost documentation describes greedy feature-split candidate selection",
      "kind": "experience",
      "status": "supported",
      "statement": "A collected CatBoost documentation excerpt describes greedy selection among feature-split candidates. This offers context for why added aggregate features could alter finite-model split competition, but is not evidence that this occurred or improved ROC-AUC in the airline experiments.",
      "scope": "CatBoost tree-structure selection as described by the cited documentation; no task-specific feature intervention or measured mechanism.",
      "supports": [
        "e-e5dd77cddf0e852a507007f4"
      ],
      "opposes": [],
      "alternatives": [
        "Documentation describes an algorithm, not the cause of any task score difference.",
        "Feature utility, redundancy, and run variation remain competing explanations."
      ],
      "next_test": "Compare candidate features with a suitable placebo-column or split diagnostic under matched task settings; attribute a mechanism only if the control distinguishes feature-specific utility from generic split competition.",
      "updated_at": "2026-10-04T11:42:51+00:00"
    },
    {
      "id": "catboost-parameter-roles-documentation",
      "title": "CatBoost documentation describes tree-growing policies and leaf-search thresholds",
      "kind": "experience",
      "status": "supported",
      "statement": "Collected CatBoost documentation describes SymmetricTree, Depthwise, and Lossguide tree-growing policies and states that min_data_in_leaf is supported on GPU for Depthwise and Lossguide. The excerpt describes this parameter as preventing split search in leaves below a sample-count threshold, not as guaranteeing a minimum child size. These descriptions motivate controlled comparisons but do not establish airline-specific performance or runtime.",
      "scope": "CatBoost parameter descriptions in the collected mutable master documentation excerpts accessed 2026-10-04; no pinned source revision, airline-specific configuration, or measured parameter effect.",
      "supports": [
        "e-1e647e8ab2fa569e6823e086",
        "e-207a732cc9bd82b1f5c91346",
        "e-80183cd00c072c1a68f8156d",
        "e-9d1392e76aeb95791ff0ee50",
        "e-cf8b600834398ec2d98bb197",
        "e-e2ca43fce116d81fcf1ca67d",
        "e-f003dbad00dd4f03847237bd"
      ],
      "opposes": [],
      "alternatives": [
        "The excerpts are mutable documentation, not a pinned description of the installed CatBoost version.",
        "Documented parameter semantics and GPU support do not establish task-specific benefit, runtime, or memory feasibility.",
        "Parameter descriptions do not identify a mechanism for any observed score difference."
      ],
      "next_test": "Compare a prespecified supported growth policy and leaf-search setting with a matched SymmetricTree control under the frozen airline protocol. Distinguish policy-associated score changes from run variation; only measured task-specific comparisons support performance claims.",
      "updated_at": "2026-10-04T16:42:14+00:00"
    },
    {
      "id": "learning-rate-iteration-trajectory-airline",
      "title": "Changing learning-rate and iteration trajectories may change airline ROC-AUC",
      "kind": "hypothesis",
      "status": "untested",
      "statement": "A controlled learning-rate and iteration-count comparison may identify a more useful training trajectory for the frozen airline task. Whether any trajectory improves ROC-AUC, and why, is untested; the documentation excerpts provide rationale for varying these parameters but no task-specific evidence.",
      "scope": "Airline satisfaction S6E10; frozen training and evaluation protocol on RTX 2080 Ti physical index 1/logical device 0; GPU CatBoost, with candidate learning-rate and iteration settings to be prespecified within the authorized budget; no task-specific trajectory measurement supplied.",
      "supports": [
        "e-1e647e8ab2fa569e6823e086",
        "e-9d1392e76aeb95791ff0ee50"
      ],
      "opposes": [],
      "alternatives": [
        "Differences may reflect GPU run variation rather than trajectory choice.",
        "An apparent quick-level effect may not transfer from the smaller quick training subset to the full training pool.",
        "Changing learning rate and iteration count together can confound their separate effects unless the comparison is designed to isolate them.",
        "The longer-round quick results recorded for a prior configuration do not establish overfitting or predict the outcome of a different trajectory."
      ],
      "next_test": "Compare prespecified learning-rate/iteration pairs against the 400-tree, depth-6, learning-rate-0.08 reference using matched quick runs and, if decision-relevant, separate harness evaluations. Distinguish trajectory effects from GPU variation; change the selected configuration only if matched evidence supports a practical improvement under the fixed time cap.",
      "updated_at": "2026-10-04T12:52:13+00:00"
    },
    {
      "id": "n007-shrinkage-trajectory-measurements",
      "title": "Quick and validation scores for the mean-feature learning-rate/iteration arms",
      "kind": "observation",
      "status": "supported",
      "statement": "In the n007 quick block, mean/400/.08 scored 0.9529833679, mean/400/.02 scored 0.9523508802, and mean/1600/.02 scored 0.9526854573. The 1600/.02 arm was 0.0002979106 below the quick reference and 0.0003345771 above the short low-rate control. The designated 1600/.02 arm scored 0.9571843964 on search validation and took 11.1143 seconds for the whole candidate CLI. The validation run was not a matched fresh control; these measurements establish neither improvement nor mechanism.",
      "scope": "Airline satisfaction S6E10; n007 frozen quick split (50,000 training, 10,000 scoring-only development rows) and frozen search-validation evaluation (489,744 training, 104,945 scoring rows); seed 20261004; RTX 2080 Ti physical index 1/logical device 0; GPU Plain CatBoost, mean feature, depth 6, l2_leaf_reg 3, comparing 400/.08, 400/.02, and 1600/.02; quick cap 60 seconds, validation cap 180 seconds.",
      "supports": [
        "e-4f93f545b0534311793ebd49"
      ],
      "opposes": [],
      "alternatives": [
        "Quick and validation use different training and scoring sets, so the quick ranking need not transfer.",
        "The validation score has no fresh matched control; historical comparisons are separate runs and GPU variation is unquantified.",
        "Learning rate and iteration count jointly affect tree-building paths; the scores do not identify optimization, regularization, or overfitting mechanisms.",
        "The quick differences are single-seed observations and do not establish reliable effects."
      ],
      "next_test": "Run separately authorized, matched validation evaluations for mean/400/.08 and mean/1600/.02 using the same frozen protocol and seed. Distinguish trajectory effects from run variation and quick-to-validation transfer; retain the slower trajectory only if matched results change the practical decision.",
      "updated_at": "2026-10-04T13:06:22+00:00"
    },
    {
      "id": "n007-compensated-low-rate-adoption",
      "title": "The compensated low-rate trajectory missed its quick adoption screen",
      "kind": "hypothesis",
      "status": "refuted",
      "statement": "For the n007 quick protocol, the prediction that mean/1600/.02 would exceed mean/400/.08 by at least 0.0002 ROC-AUC and also exceed mean/400/.02 was not supported: the first difference was -0.0002979106, while the second was +0.0003345771. This retires that prespecified quick adoption prediction only; the validation measurement was not a matched comparison.",
      "scope": "Airline satisfaction S6E10; n007 frozen quick split, 50,000 training and 10,000 scoring-only development rows; mean/400/.08 versus mean/400/.02 versus mean/1600/.02; seed 20261004; RTX 2080 Ti physical index 1/logical device 0; GPU Plain CatBoost, depth 6, l2_leaf_reg 3, 60-second cap.",
      "supports": [],
      "opposes": [
        "e-4f93f545b0534311793ebd49"
      ],
      "alternatives": [
        "The single quick block does not estimate run-to-run variation.",
        "The longer horizon improved over the short .02 control in this block, but does not establish why or whether further horizons would help.",
        "Quick-set behavior may differ from full-pool behavior; the unpaired validation score does not resolve that uncertainty."
      ],
      "next_test": "Do not adopt the 1600/.02 trajectory on this quick adoption rule. If reconsidered, use matched validation comparisons against mean/400/.08; distinguish practical transfer from GPU/run variation and reverse the decision only if prespecified matched evidence warrants it.",
      "updated_at": "2026-10-04T13:06:22+00:00"
    },
    {
      "id": "airline-depth-and-regularization-effects",
      "title": "Earlier depth/L2 interpretation is superseded by the matched depth-8/L2=3 arm",
      "kind": "hypothesis",
      "status": "superseded",
      "statement": "The earlier claim that the missing depth-8/L2=3 arm prevented separating depth and L2 effects is superseded. n010 supplies that arm: at L2=3, depth-8 scored 0.0003259148 above depth-6; at depth 8, L2=3 scored 0.0001026155 below L2=10. These are fixed-seed configuration contrasts, not evidence that either parameter caused the differences or that the small L2 contrast is equivalent.",
      "scope": "Airline satisfaction S6E10; frozen protocol on RTX 2080 Ti physical index 1/logical device 0; GPU CatBoost; candidate depth and regularization settings to be prespecified within the authorized budget; no task-specific depth or regularization measurement supplied.",
      "supports": [
        "e-5bc1b3a3245235c6fde1d37b",
        "e-e2ca43fce116d81fcf1ca67d",
        "e-e45b07a203abb6ab21700431"
      ],
      "opposes": [
        "e-5bc1b3a3245235c6fde1d37b"
      ],
      "alternatives": [
        "Fixed-seed GPU variation may contribute to the observed differences.",
        "Quick and full-pool results differ in training size and scoring population, so their ordering does not isolate a cause.",
        "Depth and L2 contrasts do not identify passenger interactions, split competition, or overfitting."
      ],
      "next_test": "For a decision that depends on the small L2 contrast, compare depth-8/L2=3 and depth-8/L2=10 with paired recorded seeds using identical features and full-pool protocol. This distinguishes persistence from GPU variation; revise the configuration choice only if the paired results meet a prespecified practical rule.",
      "updated_at": "2026-10-04T14:08:31+00:00"
    },
    {
      "id": "n008-depth8-l2-quick-and-validation-measurement",
      "title": "n008 depth-8/L2=10 scored lower on quick and higher than historical references on validation",
      "kind": "observation",
      "status": "supported",
      "statement": "In n008, the quick scores were 0.9529835299578185 for depth-6/L2=3, 0.9522140998522642 for depth-8/L2=3, and 0.9523157937896524 for depth-8/L2=10. The designated depth-8/L2=10 arm scored 0.9577204389488716 on search validation, numerically 0.0004284677512385038 above the historical n004 depth-6/L2=3 mean-feature score. Candidate elapsed time on validation was 6.1986 seconds. The validation comparison was not a fresh matched ablation; neither measurement identifies a mechanism.",
      "scope": "Airline satisfaction S6E10; n008 frozen quick split (50,000 training and 10,000 scoring-only rows) and search-validation evaluation (489,744 training and 104,945 scoring rows); raw features plus service_mean; seed 20261004; RTX 2080 Ti physical index 1/logical device 0; GPU Plain SymmetricTree CatBoost, 400 trees, learning rate 0.08; quick 60-second cap and validation 180-second cap. Quick arms: depth-6/L2=3, depth-8/L2=3, depth-8/L2=10; only depth-8/L2=10 received n008 validation evaluation.",
      "supports": [
        "e-5bc1b3a3245235c6fde1d37b"
      ],
      "opposes": [],
      "alternatives": [
        "The validation difference from historical references may reflect GPU variation or non-matched run conditions.",
        "Quick and validation differ in both training size and scoring population, so their score ordering does not isolate a training-size effect.",
        "The intervention changes model capacity and regularization; the scores do not reveal the causal mechanism."
      ],
      "next_test": "Measure matched full-pool depth-6/L2=3, depth-8/L2=3, and depth-8/L2=10 arms in separate authorized harness evaluations under identical features and seed. This distinguishes depth and L2-associated score differences from historical-run variation; retain depth 8 only if the matched comparisons change the practical decision.",
      "updated_at": "2026-10-04T13:28:46+00:00"
    },
    {
      "id": "n009-depth6-l2-3-validation-measurement",
      "title": "Fresh depth-6/L2=3 control scored 0.9572919 on search validation",
      "kind": "observation",
      "status": "supported",
      "statement": "The n009 raw-plus-service_mean depth-6/L2=3 control scored ROC-AUC 0.9572919086585254 on 489,744 training rows and 104,945 search-validation rows. Candidate elapsed time was 5.0922 seconds. Its score was 0.0004285303 below the n008 depth-8/L2=10 result. This is a fixed-seed sequential configuration contrast, not an independent confirmation or mechanism measurement.",
      "scope": "Airline satisfaction S6E10; n009 frozen search-validation evaluation, 489,744 training rows and 104,945 scoring rows; raw features plus service_mean; 400-tree depth-6 learning-rate-0.08 GPU Plain SymmetricTree CatBoost, L2=3, seed 20261004, RTX 2080 Ti physical index 1/logical device 0, 180-second cap.",
      "supports": [
        "e-e45b07a203abb6ab21700431"
      ],
      "opposes": [],
      "alternatives": [
        "The fixed-seed GPU fit may vary; the near-match to historical shallow validation is not an uncertainty estimate.",
        "The contrast with n008 jointly changes depth and L2 and cannot attribute the score difference to either parameter.",
        "Search-validation selection and the single evaluated treatment limit generalization beyond this protocol."
      ],
      "next_test": "Run the authorized full-pool depth-8/L2=3 arm with identical features, seed, and protocol. This separates a depth-associated contrast at L2=3 from an L2-associated contrast at depth 8; revise the configuration decision only from the matched measurements, recognizing fixed-seed uncertainty.",
      "updated_at": "2026-10-04T13:53:42+00:00"
    },
    {
      "id": "n009-depth8-l2-10-practical-contrast",
      "title": "Depth-8/L2=10 exceeded a fresh depth-6/L2=3 control by 0.0004285 validation AUC",
      "kind": "hypothesis",
      "status": "supported",
      "statement": "Under the n009 fixed-seed full-pool comparison, the immutable n008 depth-8/L2=10 score exceeded the fresh depth-6/L2=3 control by 0.00042853029034617407 ROC-AUC, meeting the prespecified 0.0002 practical screen. This supports the measured contrast only; it does not establish stable improvement, separate depth from L2, or identify a mechanism.",
      "scope": "Airline satisfaction S6E10; n008 depth-8/L2=10 versus n009 depth-6/L2=3, both raw features plus service_mean, 400 trees, learning rate 0.08, seed 20261004, frozen search-validation protocol with 489,744 training and 104,945 scoring rows, RTX 2080 Ti physical index 1/logical device 0; n009 180-second cap.",
      "supports": [
        "e-e45b07a203abb6ab21700431"
      ],
      "opposes": [],
      "alternatives": [
        "The contrast may reflect GPU/run variation; uncertainty is unknown.",
        "Depth and L2 changed together, so neither parameter's separate effect is identified.",
        "Quick measurements favored depth-6/L2=3 over depth-8/L2=10 in n008, so the full-pool contrast may depend on training or scoring conditions.",
        "The AUC contrast does not identify useful passenger interactions or another causal mechanism."
      ],
      "next_test": "Evaluate depth-8/L2=3 on the same full-pool protocol, then consider paired-seed A/C repeats if the configuration decision depends on this small fixed-seed contrast. These distinguish depth from L2 and persistence from GPU variation; change the default only if matched results remain practically useful.",
      "updated_at": "2026-10-04T13:53:42+00:00"
    },
    {
      "id": "n010-depth8-l2-3-validation-measurement",
      "title": "Depth-8/L2=3 scored 0.9576178 on search validation",
      "kind": "observation",
      "status": "supported",
      "statement": "The n010 raw-plus-service_mean depth-8/L2=3 arm scored ROC-AUC 0.9576178234932671 on 489,744 training rows and 104,945 search-validation rows. Candidate elapsed time was 6.3005 seconds. It exceeded the fresh n009 depth-6/L2=3 result by 0.0003259148 and was 0.0001026155 below n008 depth-8/L2=10. These are sequential fixed-seed configuration contrasts, not independent confirmation, equivalence, or mechanism measurements.",
      "scope": "Airline satisfaction S6E10; n010 frozen search-validation evaluation, 489,744 training rows and 104,945 scoring rows; raw features plus service_mean; 400-tree depth-8 learning-rate-0.08 GPU Plain SymmetricTree CatBoost, L2=3, seed 20261004, RTX 2080 Ti physical index 1/logical device 0, 180-second cap.",
      "supports": [
        "e-e618e1f97ed780a296794a51"
      ],
      "opposes": [],
      "alternatives": [
        "GPU/run variation may contribute to either contrast; uncertainty is unknown.",
        "Search-validation selection and the single evaluated treatment limit generalization beyond this protocol.",
        "AUC differences do not identify a partition, passenger interaction, or other causal mechanism."
      ],
      "next_test": "If the small depth-8 L2 contrast would change the chosen configuration, compare depth-8/L2=3 and depth-8/L2=10 with paired recorded seeds under the same full-pool protocol. This distinguishes a persistent L2-associated difference from run variation; change the choice only if results support a practically useful difference.",
      "updated_at": "2026-10-04T14:08:31+00:00"
    },
    {
      "id": "n010-depth8-l2-3-practical-retention",
      "title": "Depth-8/L2=3 passed the prespecified full-pool retention screens",
      "kind": "hypothesis",
      "status": "mixed",
      "statement": "Under the n010 fixed-seed full-pool comparison, depth-8/L2=3 scored 0.0003259148 above depth-6/L2=3 and 0.0001026155 below depth-8/L2=10. It passed the prespecified screens of at least +0.0002 over depth-6/L2=3 and no more than 0.0002 below depth-8/L2=10. This supports retention under those engineering criteria only; it does not establish stable improvement, equivalence, or mechanism.",
      "scope": "Airline satisfaction S6E10; n010 depth-8/L2=3 versus n009 depth-6/L2=3 and n008 depth-8/L2=10, all raw features plus service_mean, 400 trees, learning rate 0.08, seed 20261004, frozen search-validation protocol with 489,744 training and 104,945 scoring rows, RTX 2080 Ti physical index 1/logical device 0; n010 180-second cap.",
      "supports": [
        "e-e45b07a203abb6ab21700431",
        "e-e618e1f97ed780a296794a51"
      ],
      "opposes": [
        "e-5bc1b3a3245235c6fde1d37b"
      ],
      "alternatives": [
        "The single fixed-seed full-pool observations may reflect GPU/run variation; the heuristic comparison classification is not a significance certificate.",
        "The opposing quick ordering may arise from training size, scoring population, or their combination; it does not identify the cause.",
        "Adaptive search-validation selection may overstate transfer.",
        "The measured AUC contrast does not establish why the configurations differ."
      ],
      "next_test": "If configuration choice depends on this small fixed-seed contrast, run prespecified paired-seed comparisons of depth-8/L2=3 and depth-8/L2=10. This distinguishes persistence from GPU variation; retain the lower-cost/smaller-model option only if it remains practically acceptable under the chosen decision rule.",
      "updated_at": "2026-10-04T14:08:31+00:00"
    },
    {
      "id": "airline-depth-l2-matched-contrasts",
      "title": "Matched fixed-seed arms show positive depth-8 contrasts at L2=3",
      "kind": "hypothesis",
      "status": "mixed",
      "statement": "This successor corrects the predecessor's now-obsolete missing-arm interpretation: n009 depth-6/L2=3, n010 depth-8/L2=3, and n008 depth-8/L2=10 provide full-pool fixed-seed contrasts. The observed depth-8/L2=3 score exceeded depth-6/L2=3 by 0.0003259148; depth-8/L2=10 exceeded depth-8/L2=3 by 0.0001026155. These measurements separate the specified configuration contrasts arithmetically but do not identify causal parameter effects, interactions, stable gains, or equivalence.",
      "scope": "Airline satisfaction S6E10; n008/n009/n010 frozen search-validation evaluations, 489,744 training rows and 104,945 scoring rows; raw features plus service_mean; 400-tree GPU Plain SymmetricTree CatBoost, learning rate 0.08, seed 20261004, RTX 2080 Ti physical index 1/logical device 0; depth-6/L2=3, depth-8/L2=3, and depth-8/L2=10 configurations.",
      "supports": [
        "e-e45b07a203abb6ab21700431",
        "e-e618e1f97ed780a296794a51"
      ],
      "opposes": [
        "e-5bc1b3a3245235c6fde1d37b"
      ],
      "alternatives": [
        "GPU/run variation is unquantified and may explain some or all of the contrasts.",
        "Quick ordering opposes the full-pool ordering; the difference may involve training size, scoring population, or both.",
        "The missing depth-6/L2=10 cell leaves the full depth-by-L2 factorial interaction unidentified.",
        "AUC contrasts do not establish a model mechanism."
      ],
      "next_test": "Use paired recorded seeds for depth-8/L2=3 versus depth-8/L2=10 if their small difference affects configuration choice. This distinguishes repeatable L2-associated behavior from GPU variation; retain the preferred configuration only if the prespecified practical criterion is met.",
      "supersedes": "airline-depth-and-regularization-effects",
      "revision_reason": "n010 measured the previously missing depth-8/L2=3 full-pool arm, changing the scope from an unresolved missing comparison to three specified fixed-seed configuration contrasts. The earlier claim's scope is retained unchanged; its missing-arm interpretation is retired rather than silently rewritten. The new measurements still do not identify causal effects or resolve run uncertainty.",
      "updated_at": "2026-10-04T14:08:31+00:00"
    },
    {
      "id": "n011-dual-rating-validation-measurement",
      "title": "Dual numeric and categorical ratings scored 0.9577967 on search validation",
      "kind": "observation",
      "status": "supported",
      "statement": "The n011 dual-representation arm scored ROC-AUC 0.9577967296706902 on 104,945 search-validation rows after fitting 489,744 training rows. Candidate elapsed time was 10.3611 seconds. Its score was 0.0000762907 above historical n008, below the prespecified +0.0002 adoption margin. This was not a contemporaneous matched validation comparison and does not establish a mechanism.",
      "scope": "Airline satisfaction S6E10; n011 frozen search-validation evaluation, 489,744 training rows and 104,945 scoring rows; raw features plus service_mean, retaining 13 numeric service ratings and adding 13 native categorical rating views; 400-tree depth-8 learning-rate-0.08 GPU Plain SymmetricTree CatBoost, L2=10, seed 20261004, RTX 2080 Ti physical index 1/logical device 0, 180-second cap.",
      "supports": [
        "e-c2cd63e166d3607e88802a37"
      ],
      "opposes": [],
      "alternatives": [
        "GPU run variation is unquantified and may contribute to the small contrast with historical n008.",
        "The historical n008 anchor is not a contemporaneous control; adaptive selection may affect the observed contrast.",
        "The score does not establish that categorical equality splits were used or caused a change.",
        "The intervention adds 13 columns, so generic changes to split competition remain possible."
      ],
      "next_test": "Run a matched full-pool numeric-only versus dual-representation comparison with the same settings and recorded seeds. Distinguish a representation-associated score difference from run variation and historical-control differences; change the feature decision only if matched evidence meets the prespecified practical rule.",
      "updated_at": "2026-10-04T14:35:01+00:00"
    },
    {
      "id": "n011-dual-rating-representation-benefit",
      "title": "Dual rating views passed the quick screen but missed the validation adoption margin",
      "kind": "hypothesis",
      "status": "mixed",
      "statement": "In n011, the dual numeric-plus-categorical rating arm exceeded the numeric-only arm by 0.000229723 quick ROC-AUC and the categorical-replacement arm by 0.000353660; it narrowly passed the prespecified quick screen. Its official validation score exceeded historical n008 by only 0.000076291, missing the prespecified +0.0002 adoption margin. Evidence is mixed for practical benefit under these protocols; it does not identify why scores differed.",
      "scope": "Airline satisfaction S6E10; n011 comparison of numeric-only, categorical-replacement, and dual numeric-plus-categorical service-rating representations with service_mean retained; quick split of 50,000 training and 10,000 scoring-only rows and frozen search validation of 489,744 training and 104,945 scoring rows; seed 20261004; 400-tree depth-8 learning-rate-0.08 GPU Plain CatBoost, L2=10, RTX 2080 Ti physical index 1/logical device 0; quick cap 60 seconds and validation cap 180 seconds.",
      "supports": [
        "e-c2cd63e166d3607e88802a37"
      ],
      "opposes": [
        "e-c2cd63e166d3607e88802a37"
      ],
      "alternatives": [
        "Categorical equality candidates may complement numeric threshold candidates, but split usage and stable utility were not measured.",
        "Added columns may alter greedy split competition without providing useful additional information.",
        "GPU variation and the historical, unmatched validation anchor may explain part or all of the small observed contrast.",
        "Quick and validation protocols differ in training and scoring populations; their contrast does not identify a training-size effect."
      ],
      "next_test": "Use a separate authorized node for a matched full-pool numeric-only versus dual-representation comparison, with recorded paired seeds if feasible. This distinguishes persistent representation utility from fixed-seed variation and historical-anchor differences; retain dual views only if matched results meet a prespecified practical decision rule.",
      "updated_at": "2026-10-04T14:35:01+00:00"
    },
    {
      "id": "catboost-sum-models-api-feasibility",
      "title": "CatBoost documents a model-merging API for weighted tree blending",
      "kind": "experience",
      "status": "supported",
      "statement": "Collected CatBoost API excerpts document blending trees and counters from multiple trained models with configurable weights, including equal weights for averaging, and converting a compatible CatBoost model to CatBoostClassifier. This establishes documented API feasibility only; it does not demonstrate airline-task performance, arithmetic probability averaging, or the desired positive-class mapping.",
      "scope": "CatBoost sum_models API as described in the collected documentation excerpt captured 2026-10-04; no airline dataset, fitted configuration, measured score, or runtime result.",
      "supports": [
        "e-5d10410e5880fbc0e8e9b534",
        "e-69d441de6067c7317a97901e",
        "e-c3b2f5cae022cb55810a2785"
      ],
      "opposes": [],
      "alternatives": [
        "The excerpts do not guarantee behavior for every model type or categorical-counter configuration.",
        "Averaging raw margins and averaging probabilities are different ensemble methods.",
        "Conversion does not by itself verify positive-class mapping or prediction agreement.",
        "API feasibility does not imply improved ROC-AUC, calibration, or acceptable runtime."
      ],
      "next_test": "On identical supplied training rows, compare merged-model raw scores with explicitly weighted component raw scores, verify classes [0,1] and probability mapping, then test artifact reload and prediction. This distinguishes intended blending semantics and implementation feasibility from API assumptions; failures are implementation issues, not evidence about predictive utility.",
      "updated_at": "2026-10-04T15:12:34+00:00"
    },
    {
      "id": "catboost-merged-model-prediction-contract",
      "title": "CatBoost documents prediction inputs, probability output, and thread control",
      "kind": "experience",
      "status": "supported",
      "statement": "A collected CatBoost API excerpt states that prediction data must contain the model's features in the corresponding order, describes Probability output, and documents a thread_count parameter. This is API-use context, not evidence that any candidate artifact or class-column mapping has been verified.",
      "scope": "CatBoost predict API as described in the collected documentation excerpt captured 2026-10-04; no task-specific model artifact, prediction run, or class-mapping measurement.",
      "supports": [
        "e-a9413bb0a026a18cb909d7c7"
      ],
      "opposes": [],
      "alternatives": [
        "The excerpt does not verify installed-version behavior or a particular model's class ordering.",
        "Correct inference semantics do not establish an AUC improvement.",
        "A merged model may have categorical-counter details requiring separate checks."
      ],
      "next_test": "For any merged binary model, verify feature order, explicit thread limits, output shape and that the selected probability column corresponds to satisfaction=1; compare predictions against direct model probabilities. This distinguishes correct API use from feature-order or class-index mistakes before evaluating utility.",
      "updated_at": "2026-10-04T14:43:29+00:00"
    },
    {
      "id": "catboost-to-classifier-conversion-api",
      "title": "CatBoost documents conversion of compatible models to CatBoostClassifier",
      "kind": "experience",
      "status": "supported",
      "statement": "A collected CatBoost API excerpt documents conversion of a compatible CatBoost model to CatBoostClassifier. It provides an option for classifier-oriented artifact handling, but does not verify conversion, reload, or prediction behavior for a proposed merged model.",
      "scope": "CatBoost to_classifier API as described in the collected documentation excerpt captured 2026-10-04; no task-specific merged model or conversion test supplied.",
      "supports": [
        "e-2177017467f332763c6555d7"
      ],
      "opposes": [],
      "alternatives": [
        "Conversion is documented only for compatible loss functions.",
        "The excerpt does not establish merged-model compatibility in the installed version.",
        "Artifact conversion does not establish predictive benefit or acceptable whole-CLI timing."
      ],
      "next_test": "If a merged model requires classifier methods, convert it without refitting, save and reload it, then verify predict_proba shape and class mapping on supplied features. This distinguishes API feasibility from artifact compatibility; conversion failure is an implementation issue, not evidence against ensemble performance.",
      "updated_at": "2026-10-04T14:43:29+00:00"
    },
    {
      "id": "n012-fixed-restart-averaging-measurement",
      "title": "Restart-averaging validation measurement is unavailable after an implementation failure",
      "kind": "observation",
      "status": "supported",
      "statement": "The n012 validation evaluation has no score: the selected restart-averaging candidate could not be measured after a CatBoost merge-policy spelling error and the recorded tuning exceeded its selected trial budget. This is an unavailable measurement, not evidence for or against restart averaging. A separate 800-tree continuation control was fitted, but its score is also not supplied.",
      "scope": "Airline satisfaction S6E10; n012 selection attempt at frozen search-validation level, intended 489,744 training and 104,945 scoring rows; numeric+service_mean, depth-8/L2=10/learning-rate-0.08 GPU Plain CatBoost; intended equally weighted raw-margin average of 400-tree models at seeds 20261004 and 20261005 versus an 800-tree model at seed 20261004; RTX 2080 Ti physical index 1/logical device 0; 180-second validation cap; tuning budget 900 seconds and at most 3 trials.",
      "supports": [
        "e-0a2ca28dadc07d3eee0ebdf3"
      ],
      "opposes": [],
      "alternatives": [
        "The absent score reflects an implementation and trial-budget failure, not model efficacy.",
        "The supplied packet does not include a measured score for the successfully fitted continuation control.",
        "Technical merge audits on tree portions of the continuation model do not test restart efficacy."
      ],
      "next_test": "Only with a newly authorized node and budget, run the prespecified restart and continuation arms using the verified merge policy, saving components before merging. This distinguishes restart-averaging utility from continuation and component variation; change the decision only with actual matched harness scores, not the unavailable n012 result.",
      "updated_at": "2026-10-04T14:58:32+00:00"
    },
    {
      "id": "n013-restart-average-validation-measurement",
      "title": "Fixed restart average scored 0.9577151 on search validation",
      "kind": "observation",
      "status": "supported",
      "statement": "The n013 equal raw-margin average of two 400-tree restart models scored ROC-AUC 0.9577150798991513 on search validation, using 489,744 training rows and 104,945 scoring rows. Whole-candidate time was 10.1621 seconds. It was 0.0000053590 below historical n008 and missed the preregistered +0.0002 adoption threshold. Exact full-pool component and continuation scores were not supplied. This measurement does not establish an averaging mechanism or practical improvement.",
      "scope": "Airline satisfaction S6E10; n013 frozen search-validation evaluation, 489,744 training rows and 104,945 scoring rows; raw features plus service_mean; equal raw-margin average of two 400-tree depth-8, learning-rate-0.08, L2=10 GPU Plain SymmetricTree CatBoost models at seeds 20261004 and 20261005; RTX 2080 Ti physical index 1/logical device 0; 180-second cap.",
      "supports": [
        "e-91457c6271f25982c7045c5b"
      ],
      "opposes": [],
      "alternatives": [
        "The historical n008 comparison is not contemporaneous and cannot isolate the effect of restart averaging.",
        "The supplied full-pool result does not include scores for the exact A, B, or 800-tree continuation arms.",
        "GPU variation is unquantified; the small historical contrast may reflect run variation.",
        "The quick development comparison and full-pool evaluation differ in training and scoring populations."
      ],
      "next_test": "If restart averaging remains decision-relevant, measure its components and a prespecified continuation control in separate authorized harness evaluations under matched full-pool conditions. This distinguishes averaging utility from component and training-path differences; change the decision only on matched measurements meeting a prespecified practical rule.",
      "updated_at": "2026-10-04T15:21:37+00:00"
    },
    {
      "id": "n013-fixed-restart-practical-benefit",
      "title": "Fixed restart average did not meet its practical adoption screen",
      "kind": "hypothesis",
      "status": "mixed",
      "statement": "For n013, the quick-block restart average did not exceed the stronger component by the prespecified 0.0002 margin, while it exceeded the 800-tree continuation control. On search validation, the average missed its prespecified historical-n008-plus-0.0002 adoption threshold. The evidence opposes practical adoption of this fixed configuration under the stated screens, but does not establish that restart averaging is generally ineffective or identify why scores differed.",
      "scope": "Airline satisfaction S6E10; n013 comparison of equal raw-margin averaging of 400-tree models at seeds 20261004 and 20261005 against each component and an 800-tree continuation at seed 20261004; quick split of 50,000 training and 10,000 scoring-only rows and frozen search validation of 489,744 training and 104,945 scoring rows; numeric features plus service_mean; depth 8, learning rate 0.08, L2=10 GPU Plain SymmetricTree CatBoost; RTX 2080 Ti physical index 1/logical device 0; quick cap 60 seconds and validation cap 180 seconds.",
      "supports": [
        "e-91457c6271f25982c7045c5b"
      ],
      "opposes": [
        "e-91457c6271f25982c7045c5b"
      ],
      "alternatives": [
        "The quick average exceeded the continuation but not the stronger component; avoiding an adverse continuation result could account for that contrast without demonstrating restart complementarity.",
        "A weaker component may dilute the average; prediction-error correlation and complementarity were not measured.",
        "GPU variation, finite-sample ranking, and differences between quick and full-pool populations may affect the observed ordering.",
        "The comparison with historical n008 does not isolate restart averaging from inherited features and configuration."
      ],
      "next_test": "Only reconsider this configuration if separate authorized evaluations provide matched full-pool component and continuation comparisons, with prespecified practical thresholds and repeated paired seeds if feasible. This distinguishes component dilution and training-path effects from averaging utility and run variation; retain it only if matched results meet the practical rule.",
      "updated_at": "2026-10-04T15:21:37+00:00"
    },
    {
      "id": "catboost-gradient-boosting-score-documentation",
      "title": "CatBoost documentation describes sequential gradient boosting and split-score choices",
      "kind": "experience",
      "status": "supported",
      "statement": "The collected CatBoost documentation says trees are built sequentially to approximate negative loss gradients and describes score_function as a criterion used to select splits. It lists GPU support for all score types in the excerpt. This is API and algorithm context, not evidence that any score function or feature intervention improves airline ROC-AUC.",
      "scope": "CatBoost score-function and gradient-boosting descriptions in the collected documentation excerpt captured 2026-10-04; no airline-specific configuration, measured effect, or pinned source revision.",
      "supports": [
        "e-f653a1643ebe2654fbe2f15b"
      ],
      "opposes": [],
      "alternatives": [
        "The excerpt is documentation evidence rather than an independent implementation test.",
        "Supported GPU score types do not establish compatibility with every configuration or useful task-specific performance.",
        "The excerpt omits equations and notation-dependent detail."
      ],
      "next_test": "If score-function behavior becomes a research question, compare prespecified settings with matched data, features, and budgets. Distinguish score-function effects from run variation; do not infer a mechanism from AUC alone.",
      "updated_at": "2026-10-04T15:29:39+00:00"
    },
    {
      "id": "n014-raw-only-depth8-validation-measurement",
      "title": "Raw-only depth-8/L2=10 scored 0.9577846 on search validation",
      "kind": "observation",
      "status": "supported",
      "statement": "The n014 raw-only arm scored ROC-AUC 0.9577845705966306 on search validation, using 489,744 training rows and 104,945 scoring rows. Whole-candidate time was 6.0982 seconds. Its score was 0.0000641316 above the historical n008 raw-plus-service_mean score, within the prespecified 0.0002 engineering margin; this is not evidence of equivalence or a mechanism.",
      "scope": "Airline satisfaction S6E10; n014 frozen search-validation evaluation, 489,744 training rows and 104,945 scoring rows; 21 raw predictors including four native categories, without ID, target, or service_mean; 400-tree depth-8, L2=10, learning-rate-0.08 GPU Plain SymmetricTree CatBoost, seed 20261004, RTX 2080 Ti physical index 1/logical device 0; 180-second cap.",
      "supports": [
        "e-6cd38922818d8abed0b71a18"
      ],
      "opposes": [],
      "alternatives": [
        "The historical n008 mean-feature arm is not a fresh matched control, so the small observed difference may reflect GPU variation or run conditions.",
        "The raw-only arm scored lower than the historical mean-feature arm on quick; the reversal may depend on training or scoring population, or their combination.",
        "Removing a column changes the candidate feature set and greedy boosting path; the score alone does not identify why the result occurred."
      ],
      "next_test": "Run a fresh mean-feature control at identical settings and full-pool protocol. This distinguishes the effect of removing service_mean from historical-control and GPU variation; change the feature choice only if the matched result changes the prespecified practical decision.",
      "updated_at": "2026-10-04T15:47:29+00:00"
    },
    {
      "id": "n014-raw-retention-at-depth8",
      "title": "Raw-only depth-8 configuration passed the prespecified retention screen",
      "kind": "hypothesis",
      "status": "mixed",
      "statement": "In n014, raw-only depth-8/L2=10 scored 0.9577845706 on search validation, above the prespecified lower retention threshold of 0.9575204389 and below the raw-preference threshold of 0.9579204389. It therefore passed the stated engineering retention screen but did not demonstrate a practically material advantage over the historical mean-feature reference. The fresh n015 mean control scored 0.9577700474, only 0.0000145232 below the n014 raw score, also within the 0.0002 margin. Quick results favored mean, so these measurements do not establish equivalence, general redundancy of service_mean, or a mechanism.",
      "scope": "Airline satisfaction S6E10; n014 designated raw-only arm compared with immutable n008 raw-plus-service_mean reference; n014 frozen quick split (50,000 training and 10,000 scoring-only rows) and search validation (489,744 training and 104,945 scoring rows); 400-tree depth-8, L2=10, learning-rate-0.08 GPU Plain SymmetricTree CatBoost, seed 20261004, RTX 2080 Ti physical index 1/logical device 0; quick 60-second and validation 180-second caps.",
      "supports": [
        "e-6cd38922818d8abed0b71a18",
        "e-965fbcee80bf46cdd774d161"
      ],
      "opposes": [
        "e-5bc1b3a3245235c6fde1d37b"
      ],
      "alternatives": [
        "The n014 and n015 full-pool fits are sequential and single-seed; GPU variation is unquantified.",
        "Quick and validation differ in both training and scoring populations, so their opposing ordering does not isolate a population or training-size effect.",
        "Adding service_mean changes candidate feature competition and boosting paths; score differences do not identify why they occurred."
      ],
      "next_test": "If the small full-pool contrast affects the feature decision, run further matched raw-versus-mean full-pool comparisons with predeclared recorded seeds. This distinguishes persistent feature-associated ordering from GPU variation; change the decision only if results meet a prespecified practical rule.",
      "updated_at": "2026-10-04T16:08:29+00:00"
    },
    {
      "id": "catboost-gpu-training-nondeterminism-documentation",
      "title": "Collected CatBoost documentation describes GPU training as nondeterministic",
      "kind": "experience",
      "status": "supported",
      "statement": "Collected CatBoost documentation excerpts state that GPU training is nondeterministic because floating-point summations may occur in a nondeterministic order. The newer excerpt also describes GPU device IDs as zero-based, consistent with using logical device 0 under the authorized mask. These documentation statements do not provide an airline-specific variance estimate or significance threshold.",
      "scope": "CatBoost GPU training documentation excerpt captured 2026-10-04; no pinned source revision or task-specific estimate of score variation.",
      "supports": [
        "e-0230e1342ecdc9f92e56daad",
        "e-69071d65629d2ddd4239105e"
      ],
      "opposes": [],
      "alternatives": [
        "The excerpts are mutable documentation and may not exactly characterize the installed CatBoost version.",
        "They give no magnitude or distribution for ROC-AUC variation on this task.",
        "A score contrast may reflect configuration, run variation, or both."
      ],
      "next_test": "Use paired recorded-seed repetitions for decision-relevant GPU configuration contrasts where budget permits. Distinguish persistent ordering from observed run variation; do not treat the documentation statement as a statistical threshold.",
      "updated_at": "2026-10-04T16:42:14+00:00"
    },
    {
      "id": "n015-fresh-mean-validation-measurement",
      "title": "Fresh raw-plus-service_mean control scored 0.9577700 on search validation",
      "kind": "observation",
      "status": "supported",
      "statement": "The designated n015 raw-plus-service_mean arm scored ROC-AUC 0.9577700473603586 on search validation, using 489,744 training rows and 104,945 scoring rows. It was 0.0000145232 below the n014 raw-only score, within the prespecified absolute 0.0002 action margin. On the frozen quick split, mean scored 0.0004148383 above raw, reversing the ordering. The full-pool comparison is one fixed-seed realization, not evidence of equivalence or a mechanism.",
      "scope": "Airline satisfaction S6E10; n015 designated raw-plus-service_mean arm and n014 raw-only reference; n015 frozen quick split (50,000 training and 10,000 scoring-only rows) and search validation (489,744 training and 104,945 scoring rows); 21 raw predictors with four native categories, plus the row-local equal-weight mean of the exact 13 service ratings in the mean arm; 400-tree depth-8, L2=10, learning-rate-0.08 GPU Plain SymmetricTree CatBoost, seed 20261004, RTX 2080 Ti physical index 1/logical device 0; quick 60-second and validation 180-second caps.",
      "supports": [
        "e-965fbcee80bf46cdd774d161"
      ],
      "opposes": [],
      "alternatives": [
        "The full-pool contrast is between sequential runs and GPU variation is unknown.",
        "The quick ordering reversal may reflect different training or scoring populations, or both.",
        "AUC alone cannot distinguish aggregate-service utility from generic changes to feature competition or boosting paths."
      ],
      "next_test": "Run predeclared paired-seed raw-versus-mean full-pool evaluations if the feature choice depends on this small contrast. This distinguishes persistent ordering from run variation; retain a feature choice only if the paired results satisfy the practical decision rule.",
      "updated_at": "2026-10-04T16:08:29+00:00"
    },
    {
      "id": "n015-raw-mean-practical-separation",
      "title": "Fresh full-pool raw and mean scores fell within the practical action margin",
      "kind": "hypothesis",
      "status": "mixed",
      "statement": "The n015 practical prediction was met in the single full-pool realization: the fresh mean score was within 0.0002 of the n014 raw score. The quick comparison opposed that prediction, favoring mean by 0.0004148383. This mixed evidence does not establish equivalence, stable feature utility, or general service_mean redundancy.",
      "scope": "Airline satisfaction S6E10; n015 raw-versus-mean comparison against n014 raw reference; frozen quick split (50,000 training and 10,000 scoring-only rows) and search validation (489,744 training and 104,945 scoring rows); 400-tree depth-8, L2=10, learning-rate-0.08 GPU Plain SymmetricTree CatBoost, seed 20261004, RTX 2080 Ti physical index 1/logical device 0; 0.0002 absolute AUC action margin; quick 60-second and validation 180-second caps.",
      "supports": [
        "e-965fbcee80bf46cdd774d161"
      ],
      "opposes": [
        "e-965fbcee80bf46cdd774d161"
      ],
      "alternatives": [
        "The single full-pool contrast may be affected by GPU variation.",
        "Training and scoring populations both differ between quick and validation, preventing attribution of their reversal.",
        "The practical margin is an engineering convention, not a statistical equivalence test."
      ],
      "next_test": "Repeat matched raw-versus-mean full-pool comparisons under predeclared paired seeds. This distinguishes a practically small persistent contrast from seed/run variation and population-dependent ordering; reconsider the feature decision only if results consistently cross the action margin.",
      "updated_at": "2026-10-04T16:08:29+00:00"
    },
    {
      "id": "n016-depthwise-validation-measurement",
      "title": "Depthwise scored 0.9579240 on search validation",
      "kind": "observation",
      "status": "supported",
      "statement": "The designated raw-feature Depthwise arm scored ROC-AUC 0.9579240131255401 on 489,744 training rows and 104,945 search-validation rows. It exceeded the historical n014 SymmetricTree score by 0.0001394425, below the predeclared +0.0002 practical threshold. Its whole-candidate time was 7.6021 seconds, within the 180-second cap. This sequential, fixed-seed contrast does not establish equivalence, stable improvement, or a mechanism.",
      "scope": "Airline satisfaction S6E10; n016 designated Depthwise/min_data_in_leaf=1 arm compared with immutable n014 raw-only SymmetricTree reference; frozen search validation with 489,744 training and 104,945 scoring rows; 21 raw predictors including four native categories, without ID or target; 400-tree depth-8, L2=10, learning-rate-0.08 GPU Plain CatBoost, seed 20261004, RTX 2080 Ti physical index 1/logical device 0; 180-second cap.",
      "supports": [
        "e-7f750b5c9b9715c1340448b1"
      ],
      "opposes": [],
      "alternatives": [
        "The reference is historical and the sequential contrast may include GPU/run variation.",
        "Quick ordering was negative while validation ordering was positive; training and scoring populations both differ, so the reversal does not identify a population effect.",
        "The effective data partition changed with growth policy and realized tree complexity differed; the score does not isolate policy from these associated changes.",
        "AUC alone does not identify passenger-segment benefit or another mechanism."
      ],
      "next_test": "If growth policy remains decision-relevant, run a fresh matched SymmetricTree control and prespecified Depthwise comparison under the same full-pool protocol, with paired recorded seeds where feasible. This distinguishes a persistent policy-associated ordering from run variation and historical-control differences; change the choice only if results meet a predeclared practical rule.",
      "updated_at": "2026-10-04T16:34:19+00:00"
    },
    {
      "id": "n016-depthwise-practical-benefit",
      "title": "Depthwise did not meet its predeclared practical-improvement threshold",
      "kind": "hypothesis",
      "status": "refuted",
      "statement": "The n016 prediction that raw-feature Depthwise/min_data_in_leaf=1 would exceed the n014 SymmetricTree validation score by at least 0.0002 was not met: the observed difference was +0.0001394425. This rejects that scoped adoption prediction, not smaller persistent utility, other Depthwise settings, or a passenger-interaction mechanism.",
      "scope": "Airline satisfaction S6E10; n016 designated Depthwise/min_data_in_leaf=1 versus immutable n014 raw-only SymmetricTree reference; frozen search validation with 489,744 training and 104,945 scoring rows; 21 raw predictors including four native categories, without ID or target; 400-tree depth-8, L2=10, learning-rate-0.08 GPU Plain CatBoost, seed 20261004, RTX 2080 Ti physical index 1/logical device 0; 180-second cap; practical threshold +0.0002.",
      "supports": [],
      "opposes": [
        "e-7f750b5c9b9715c1340448b1"
      ],
      "alternatives": [
        "GPU variation is unquantified, and the comparison uses a historical rather than contemporaneous control.",
        "The heuristic harness comparison was classified uncertain and is not a significance certificate.",
        "Quick and validation orderings reverse, but their different training and scoring populations prevent attributing the reversal to training size alone.",
        "Policy-associated data partition and unequal realized tree complexity prevent attributing any contrast specifically to leaf-specific passenger effects."
      ],
      "next_test": "Do not adopt this setting on the basis of the observed sub-threshold contrast. If policy choice could change, compare fresh matched controls and paired recorded seeds; distinguish persistent practical benefit from run variation and policy-associated partition/complexity changes, and adopt only if the prespecified threshold is met.",
      "updated_at": "2026-10-04T16:34:19+00:00"
    },
    {
      "id": "n016-depthwise-runnable-feasibility",
      "title": "The designated Depthwise candidate ran within the evaluator caps",
      "kind": "experience",
      "status": "supported",
      "statement": "The n016 Depthwise candidate completed the frozen search-validation evaluation on the authorized RTX 2080 Ti and produced a model and ordered finite predictions, according to the supplied worker audit. Candidate elapsed time was 7.6021 seconds, below the 180-second cap. This is configuration-specific feasibility evidence, not evidence of general feasibility or predictive mechanism.",
      "scope": "Airline satisfaction S6E10; n016 delivered Depthwise/min_data_in_leaf=1 candidate, raw 21 predictors including four native categories; 400-tree depth-8, L2=10, learning-rate-0.08 GPU Plain CatBoost, seed 20261004; RTX 2080 Ti physical index 1/logical device 0; frozen search-validation evaluation and 180-second cap.",
      "supports": [
        "e-7f750b5c9b9715c1340448b1"
      ],
      "opposes": [],
      "alternatives": [
        "The evidence covers one candidate configuration and does not establish peak-memory behavior or an internal budget deadline.",
        "Two technical development-launch failures were reported; they are infrastructure/setup failures, not negative efficacy measurements."
      ],
      "next_test": "For any materially different policy, workload, or candidate implementation, verify GPU identity, CLI contract, model export, and whole-process timing under the applicable frozen evaluator; this distinguishes demonstrated configuration feasibility from assumptions about other workloads.",
      "updated_at": "2026-10-04T16:34:19+00:00"
    },
    {
      "id": "n017-fresh-symmetric-validation-measurement",
      "title": "Fresh raw SymmetricTree scored 0.9577846 on search validation",
      "kind": "observation",
      "status": "supported",
      "statement": "The designated n017 raw-feature SymmetricTree arm scored ROC-AUC 0.9577845705966306 on 489,744 training rows and 104,945 search-validation rows. This equals the reported n014 score at displayed precision and is 0.0001394425 below n016 Depthwise. It meets the prespecified strict absolute 0.0002 proximity prediction, but does not establish equivalence, stable policy ordering, or a mechanism. Whole-candidate time was 6.0993 seconds.",
      "scope": "Airline satisfaction S6E10; n017 designated fresh raw-only SymmetricTree arm compared with immutable n014 SymmetricTree and n016 Depthwise references; frozen search validation with 489,744 training and 104,945 scoring rows; 21 raw predictors including four native categories, without ID or target; 400-tree depth-8, L2=10, learning-rate-0.08 GPU Plain CatBoost, seed 20261004, RTX 2080 Ti physical index 1/logical device 0; 180-second cap.",
      "supports": [
        "e-15adc1988cb151046c60d6d8"
      ],
      "opposes": [],
      "alternatives": [
        "The n016 comparison is sequential and GPU variation is unknown; the harness heuristic classification is not a significance certificate.",
        "Quick and validation orderings differ, but their training and scoring populations differ too, so the reversal does not identify a population effect.",
        "Policy-associated data partition and realized tree complexity differed; AUC does not isolate policy from those changes or identify passenger-segment utility."
      ],
      "next_test": "If choosing between policies remains consequential, run fresh matched policy arms with paired recorded seeds under the same full-pool protocol. This distinguishes persistent policy-associated ordering from GPU/run variation; change the choice only if a prespecified practical rule is met.",
      "updated_at": "2026-10-04T17:08:33+00:00"
    },
    {
      "id": "n017-symmetric-depthwise-practical-proximity",
      "title": "Fresh SymmetricTree met the prespecified proximity margin to Depthwise",
      "kind": "hypothesis",
      "status": "supported",
      "statement": "The n017 practical prediction was met in one fixed-seed full-pool realization: fresh SymmetricTree scored 0.0001394425 below immutable n016 Depthwise, within the strict absolute 0.0002 proximity interval. This supports the scoped proximity prediction only; it does not establish equivalence, stable ordering, or a mechanism. The earlier Depthwise adoption prediction of at least +0.0002 over n014 remains unmet.",
      "scope": "Airline satisfaction S6E10; n017 designated fresh raw-only SymmetricTree versus immutable n016 raw-only Depthwise/min_data_in_leaf=1 and n014 raw-only SymmetricTree references; frozen search validation with 489,744 training and 104,945 scoring rows; 21 raw predictors including four native categories, without ID or target; 400-tree depth-8, L2=10, learning-rate-0.08 GPU Plain CatBoost, seed 20261004, RTX 2080 Ti physical index 1/logical device 0; 180-second cap; strict absolute proximity margin 0.0002.",
      "supports": [
        "e-15adc1988cb151046c60d6d8",
        "e-7f750b5c9b9715c1340448b1"
      ],
      "opposes": [],
      "alternatives": [
        "This is one fixed-seed realization and GPU variation is unquantified; proximity is not statistical equivalence.",
        "The historical Depthwise comparison and quick ordering do not form a contemporaneous paired block; no fresh Depthwise quick fit was made.",
        "Different partition and realized complexity prevent attribution to policy alone or to leaf-specific passenger interactions."
      ],
      "next_test": "If policy selection could change, evaluate matched SymmetricTree and Depthwise arms with paired recorded seeds. This distinguishes persistent practical ordering from run variation; retain a policy preference only if the prespecified practical decision rule is met.",
      "updated_at": "2026-10-04T17:08:33+00:00"
    },
    {
      "id": "n018-depth10-validation-measurement",
      "title": "Raw SymmetricTree depth 10 scored 0.9577367 on search validation",
      "kind": "observation",
      "status": "supported",
      "statement": "The designated raw-feature SymmetricTree depth-10 arm scored ROC-AUC 0.9577366851375627 on 489,744 training rows and 104,945 search-validation rows. It was 0.0000478855 below the immutable n017 depth-8 score and 0.0001873280 below n016 Depthwise. Whole-candidate time was 9.6110 seconds, within the 180-second cap. This single fixed-seed measurement does not establish stable ordering or a mechanism.",
      "scope": "Airline satisfaction S6E10; n018 designated raw21/four-native-category SymmetricTree depth-10 arm compared with immutable n017 raw-only SymmetricTree depth-8 and n016 raw-only Depthwise references; frozen search validation with 489,744 training and 104,945 scoring rows; 400 trees, L2=10, learning rate=0.08, seed 20261004, GPU Plain CatBoost on RTX 2080 Ti physical index 1/logical device 0; 180-second cap.",
      "supports": [
        "e-2147ecc3e3c9aa129b93bed1"
      ],
      "opposes": [],
      "alternatives": [
        "The official contrast is a sequential single-seed comparison; GPU variation is unknown.",
        "Depth changes greedy fitting paths and derived tree capacity, so the score does not isolate useful interaction order or another mechanism.",
        "Quick and validation populations differ in both fitting and scoring rows; their contrast does not identify a population effect."
      ],
      "next_test": "If depth selection remains consequential, run matched depth-8 and depth-10 full-pool arms with paired recorded seeds and identical settings. This distinguishes a persistent depth-associated ordering from GPU/run variation; change the default only if the predeclared practical rule is met.",
      "updated_at": "2026-10-04T17:26:26+00:00"
    },
    {
      "id": "n018-depth10-practical-benefit",
      "title": "Depth 10 missed the prespecified practical-improvement threshold",
      "kind": "hypothesis",
      "status": "refuted",
      "statement": "The n018 prediction that increasing depth from 8 to 10 would improve full-pool ROC-AUC by at least 0.0002 was not met: depth 10 scored 0.0000478855 below the immutable n017 depth-8 reference. The three quick fits also favored depth 8, with depth 10 and its same-seed repeat tied below the fresh depth-8 control. This rejects the scoped practical-benefit prediction, not deeper learners generally.",
      "scope": "Airline satisfaction S6E10; n018 designated raw21/four-native-category SymmetricTree depth-10 versus immutable n017 raw-only SymmetricTree depth-8; frozen quick split (50,000 training and 10,000 scoring-only rows) and search validation (489,744 training and 104,945 scoring rows); 400 trees, L2=10, learning rate=0.08, seed 20261004, GPU Plain CatBoost on RTX 2080 Ti physical index 1/logical device 0; quick 60-second and validation 180-second caps; practical improvement threshold +0.0002.",
      "supports": [],
      "opposes": [
        "e-2147ecc3e3c9aa129b93bed1"
      ],
      "alternatives": [
        "GPU variation is unquantified; the official result is one treatment realization against a historical control.",
        "Greater tree capacity may alter variance or optimization trajectory, but leaf counts alone do not diagnose either mechanism.",
        "This scope tests raw features and fixed 400-round, 0.08 learning-rate settings, not other depths, regularization, trajectories, or feature representations."
      ],
      "next_test": "Do not prioritize this fixed-round depth escalation as an improvement on current evidence. If reconsidered, run paired-seed depth-8/depth-10 full-pool comparisons. This distinguishes persistent ordering from run variation; reverse the decision only if the practical threshold is met consistently.",
      "updated_at": "2026-10-04T17:26:26+00:00"
    },
    {
      "id": "catboost-categorical-handling-docs",
      "title": "CatBoost documentation describes native handling of low-cardinality categories",
      "kind": "experience",
      "status": "supported",
      "statement": "A collected CatBoost documentation excerpt says low-cardinality categorical features may use one-hot encoding and that CTRs are not calculated for those features. This describes documented behavior, not verified behavior for a particular installed version or evidence of airline-task performance.",
      "scope": "CatBoost categorical-feature documentation excerpt captured 2026-10-04; no task-specific training configuration, cardinality audit, or measured score.",
      "supports": [
        "e-556fa2385067322f60ba8cd9"
      ],
      "opposes": [],
      "alternatives": [
        "The excerpt does not verify effective parameters or categorical handling in the installed version.",
        "Documented encoding behavior does not establish predictive utility.",
        "Training and prediction cardinalities or category preprocessing may affect the actual treatment."
      ],
      "next_test": "Audit the installed CatBoost version, effective one_hot_max_size, and observed cardinalities in training rows; compare native category treatments under matched settings. Distinguish documented behavior from actual configuration and task performance.",
      "updated_at": "2026-10-04T17:34:55+00:00"
    },
    {
      "id": "catboost-joint-category-candidate-hypothesis",
      "title": "Joint categorical features may change finite-model split access",
      "kind": "hypothesis",
      "status": "untested",
      "statement": "Collected CatBoost documentation describes greedy feature-split candidate selection. This motivates testing whether native joint categorical features make useful rating-by-segment distinctions more accessible, but no supplied task measurement tests this intervention or identifies such a mechanism.",
      "scope": "Algorithm-design rationale for native joint categorical features in CatBoost; no airline feature intervention or measured score evidence.",
      "supports": [
        "e-f505e02ddca3a160dfeaa9c5"
      ],
      "opposes": [],
      "alternatives": [
        "Native trees may already represent the relevant interactions without explicit joint categories.",
        "Any score difference could arise from changed split candidates or optimization paths rather than useful interaction information.",
        "Documentation is general context and does not verify version-specific GPU behavior or airline performance."
      ],
      "next_test": "Compare raw categorical inputs with prespecified native joint categories under identical training rows, settings, and recorded seeds. Distinguish useful joint-category access from generic changes to candidate competition or fitting variation; claim a mechanism only if an identifying control supports it.",
      "updated_at": "2026-10-04T17:34:55+00:00"
    },
    {
      "id": "n019-travel-joint-validation-measurement",
      "title": "Travel-conditioned rating equality views scored 0.9576975 on search validation",
      "kind": "observation",
      "status": "supported",
      "statement": "The designated n019 arm, retaining raw features and adding individual rating-level and travel-by-rating categorical views, scored ROC-AUC 0.9576975446690855 on 104,945 search-validation rows after fitting 489,744 rows. Candidate elapsed time was 15.3302 seconds. It scored 0.0000870259 below the historical n017 raw reference and missed the prespecified net adoption threshold by 0.0002870259. This is one fixed-seed measurement, not a mechanism test or evidence of equivalence.",
      "scope": "Airline satisfaction S6E10; n019 frozen search-validation evaluation, 489,744 training rows and 104,945 scoring rows; 21 raw predictors including four native categories, plus 13 categorical rating-level views and 13 travel-by-rating views; 400-tree depth-8 learning-rate-0.08 GPU Plain SymmetricTree CatBoost, L2=10, seed 20261004, one_hot_max_size=16, RTX 2080 Ti physical index 1/logical device 0, 180-second cap.",
      "supports": [
        "e-fc42c5f68b8e48715b48f7bf"
      ],
      "opposes": [],
      "alternatives": [
        "The comparison with n017 is historical rather than contemporaneous; GPU variation is unquantified.",
        "Adding 26 columns can change greedy split competition and boosting paths without demonstrating useful interaction information.",
        "The quick and validation scoring/training protocols differ; their contrasts do not identify a training-size effect.",
        "The harness comparison is heuristic-only and classified uncertain, not a significance certificate."
      ],
      "next_test": "If this representation remains decision-relevant, run a fresh matched full-pool raw-versus-designated-joint comparison with recorded paired seeds. This distinguishes a persistent feature-associated difference from run variation and the historical-control contrast; retain it only if the prespecified practical rule is met.",
      "updated_at": "2026-10-04T17:53:23+00:00"
    },
    {
      "id": "n019-travel-joint-practical-benefit",
      "title": "Travel-joint arm missed its prespecified practical screens",
      "kind": "hypothesis",
      "status": "refuted",
      "statement": "In n019, quick C−B was −0.0000996276 and C−A was +0.0002571925, failing the prespecified conjunction that both contrasts reach +0.0002. On search validation, C scored 0.9576975447, 0.0000870259 below the historical n017 raw reference and below the prespecified net adoption threshold. These measurements oppose practical adoption of this exact representation under the stated screens; they do not refute equality-only views or identify why scores differ.",
      "scope": "Airline satisfaction S6E10; n019 designated C (raw features plus 13 rating-level and 13 travel-by-rating categorical views) versus quick arms A raw and B raw plus rating-level views, and historical n017 raw reference; quick split of 50,000 training and 10,000 scoring-only rows and frozen search validation of 489,744 training and 104,945 scoring rows; seed 20261004; 400-tree depth-8, L2=10, learning-rate-0.08 GPU Plain SymmetricTree CatBoost; RTX 2080 Ti physical index 1/logical device 0; quick 60-second and validation 180-second caps; practical margin +0.0002.",
      "supports": [],
      "opposes": [
        "e-fc42c5f68b8e48715b48f7bf"
      ],
      "alternatives": [
        "The official raw comparison is historical, and GPU variation is unknown.",
        "Quick contrasts are single-seed and do not establish stable ordering.",
        "The arm adds both individual and joint views; the official result cannot isolate the incremental contribution of the joint block.",
        "Observed use of joint columns in quick models demonstrates access or split use, not efficacy."
      ],
      "next_test": "Do not adopt the n019 joint-view configuration on these screens. If it could change the final feature decision, use a fresh matched full-pool ablation with paired recorded seeds. This distinguishes persistent practical utility from fixed-seed variation and generic column competition; reverse the decision only if the predeclared margin is met.",
      "updated_at": "2026-10-04T17:53:23+00:00"
    },
    {
      "id": "catboost-grow-policy-documentation",
      "title": "CatBoost documentation describes Depthwise and min_data_in_leaf semantics",
      "kind": "experience",
      "status": "supported",
      "statement": "A collected CatBoost documentation excerpt says Depthwise grows non-terminal leaves level by level, choosing the best loss-improvement condition per leaf, and that min_data_in_leaf is available with Depthwise and Lossguide on CPU and GPU. It describes min_data_in_leaf as restricting further split search in leaves below the threshold; it does not guarantee a minimum size for both resulting children. This documents parameter semantics, not airline-task feasibility or efficacy.",
      "scope": "CatBoost common-parameter documentation excerpt captured 2026-10-04; no task-specific efficacy measurement, and support100 feasibility is unmeasured.",
      "supports": [
        "e-fcde345109c1dc93306afa76"
      ],
      "opposes": [],
      "alternatives": [
        "The excerpt documents intended behavior and does not verify the installed CatBoost version or effective runtime configuration.",
        "Compatibility does not establish that Depthwise or a particular min_data_in_leaf value fits within the frozen time budget.",
        "The excerpt does not establish airline-task predictive utility or a causal explanation for any score difference."
      ],
      "next_test": "If testing Depthwise, first verify installed-version GPU feasibility and output compatibility at the proposed settings; then compare a prespecified Depthwise arm with a matched SymmetricTree control. Distinguish parameter feasibility from task-specific ROC-AUC effects, and do not infer child-size guarantees from min_data_in_leaf.",
      "updated_at": "2026-10-04T18:01:07+00:00"
    },
    {
      "id": "n020-depthwise-support100-validation-measurement",
      "title": "Depthwise with min_data_in_leaf=100 scored 0.9578806 on search validation",
      "kind": "observation",
      "status": "supported",
      "statement": "The designated support100 arm scored ROC-AUC 0.9578806014199407 on 489,744 training rows and 104,945 search-validation rows. It was 0.0000434117 below the historical n016 support1 score, and its whole-candidate time was 6.1853 seconds, within the 180-second cap. This single fixed-seed measurement does not establish stable ordering or a mechanism.",
      "scope": "Airline satisfaction S6E10; n020 designated Depthwise/min_data_in_leaf=100 arm compared with immutable n016 Depthwise/min_data_in_leaf=1 reference; frozen search validation with 489,744 training and 104,945 scoring rows; 21 raw predictors including four native categories, without ID or target; 400-tree depth-8, L2=10, learning-rate-0.08 GPU Plain CatBoost, seed 20261004, RTX 2080 Ti physical index 1/logical device 0; 180-second cap.",
      "supports": [
        "e-38ce7174a007553b49416f26"
      ],
      "opposes": [],
      "alternatives": [
        "The comparison is sequential against a historical control; GPU variation is unknown.",
        "Changing the threshold altered realized tree leaf counts and model size as well as the candidate splits, so the score does not isolate a mechanism.",
        "AUC does not identify effects for particular passenger segments."
      ],
      "next_test": "If the support threshold remains decision-relevant, compare support1 and support100 in fresh matched full-pool arms with paired recorded seeds. This distinguishes persistent ordering from sequential-run variation; change the setting only if the prespecified practical rule is met.",
      "updated_at": "2026-10-04T18:31:07+00:00"
    },
    {
      "id": "n020-support100-practical-improvement",
      "title": "Support100 did not meet its prespecified practical-improvement threshold",
      "kind": "hypothesis",
      "status": "refuted",
      "statement": "The n020 prediction that changing min_data_in_leaf from 1 to 100 would improve full-pool ROC-AUC by at least 0.0002 over n016 was not met: the observed score was 0.0000434117 lower than n016. This rejects that scoped adoption prediction, not other thresholds or a general mechanism.",
      "scope": "Airline satisfaction S6E10; n020 designated Depthwise/min_data_in_leaf=100 versus immutable n016 Depthwise/min_data_in_leaf=1 reference; frozen search validation with 489,744 training and 104,945 scoring rows; 21 raw predictors including four native categories, without ID or target; 400-tree depth-8, L2=10, learning-rate-0.08 GPU Plain CatBoost, seed 20261004, RTX 2080 Ti physical index 1/logical device 0; 180-second cap; practical threshold +0.0002.",
      "supports": [],
      "opposes": [
        "e-38ce7174a007553b49416f26"
      ],
      "alternatives": [
        "GPU variation is unquantified and the reference was historical rather than contemporaneous.",
        "The comparison does not isolate effects of realized tree structure from the support threshold itself."
      ],
      "next_test": "Do not adopt support100 for the stated practical-improvement objective on this result. If revisiting the choice, run matched paired-seed support1/support100 evaluations; this distinguishes persistent ordering from run variation and changes the decision only if the practical threshold is met.",
      "updated_at": "2026-10-04T18:31:07+00:00"
    }
  ],
  "agenda": {
    "motivation": "Improve ROC-AUC under the frozen airline protocol while separating measured configuration contrasts from explanations about passenger interactions. The support100 treatment was evaluated, but its prespecified practical gain was not observed.",
    "method": "Treat the validated full-pool score as one fixed-seed measurement, not a significance result. The packet also reports negative quick comparisons and a repeat, but these do not replace the official full-pool measurement or establish independent-seed evidence. Keep the final holdout sealed.",
    "what_changed": "Added the n020 support100 search-validation measurement: 0.9578806014199407, 0.0000434117 below n016 support1 and below the +0.0002 adoption threshold. Retired the scoped practical-improvement prediction as refuted; the fixed-seed contrast does not establish equivalence or a mechanism. The evidence packet reports quick support100 results below support1, but no independent-seed inference follows.",
    "open_questions": [
      "Would matched paired-seed full-pool comparisons preserve the support1/support100 ordering?",
      "Can a controlled learning-rate/iteration trajectory or regularization setting improve practical ROC-AUC within the fixed time cap?",
      "Can a small GPU-compatible ensemble or principled categorical/ordinal representation improve the frozen validation score?",
      "Which configuration balances AUC and whole-CLI runtime under the cap?"
    ],
    "next_experiments": [
      "If support threshold selection remains consequential, compare support1 and support100 in matched paired-seed full-pool evaluations. This distinguishes persistent ordering from run variation; change the choice only if the prespecified practical rule is met.",
      "Compare a prespecified learning-rate/iteration trajectory with a matched fixed-iteration control. This distinguishes trajectory choice from tree count or added compute; change the configuration only if validation meets the practical rule within the cap.",
      "Test a distinct representation or small GPU-compatible ensemble against a matched raw-feature control. This distinguishes utility of the added modeling choice from sequential-run variation; retain it only if the practical criterion is met within the cap.",
      "Do not interpret AUC changes as evidence of a passenger-segment mechanism without an identifying control; preserve uncertainty when only fixed-seed comparisons are available."
    ]
  },
  "evidence": [
    {
      "id": "e-0230e1342ecdc9f92e56daad",
      "kind": "source",
      "title": "Training on GPU | CatBoost",
      "observed_at": "",
      "captured_at": "2026-10-04T15:53:49+00:00",
      "source_url": "https://catboost.ai/docs/en/features/training-on-gpu",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "3c36cbdf5cda8036043e97f64c22226ca0224791c618d53ff3e6a619816423f7",
      "packet_sha256": "7824945b0fd3c4b48e608db4ef061bf7bd0aad779ab8008a53892c98dc5fbbe2",
      "metrics": {}
    },
    {
      "id": "e-04ad4e8afcd31d5e35f25ae5",
      "kind": "source",
      "title": "Common parameters",
      "observed_at": "",
      "captured_at": "2026-10-04T13:52:15+00:00",
      "source_url": "https://raw.githubusercontent.com/catboost/catboost/master/catboost/docs/en/references/training-parameters/common.md",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "461c25229498478f5c1c29be033dda6e394f43f982fb9065751ee65c2daecf14",
      "packet_sha256": "887a249843e2791d44f7db4c45ee4da60986c5984139e37d71f1e2e585c495fd",
      "metrics": {}
    },
    {
      "id": "e-0a2ca28dadc07d3eee0ebdf3",
      "kind": "experiment",
      "title": "Fixed restart averaging with an equal-tree continuation control",
      "node": "n012",
      "parent": "n008",
      "observed_at": "2026-10-04T14:58:12+00:00",
      "captured_at": "2026-10-04T14:58:12+00:00",
      "commit": "3d8258edd1b00e7d370418a2e394bd2067dd31ed",
      "status": "failed",
      "measurement_status": "unavailable",
      "selection_eligible": false,
      "metric": "score",
      "level": "validation",
      "packet_sha256": "c7cb6d6cea3844cd6c400d3d7593e0914bfbaaf494e594e7189dec1f38ece0e9",
      "metrics": {}
    },
    {
      "id": "e-0b9cf72b7244cf181acd82e6",
      "kind": "experiment",
      "title": "Isolated lower-tail service consistency beyond the overall mean",
      "node": "n006",
      "parent": "n002",
      "observed_at": "2026-10-04T11:36:38+00:00",
      "captured_at": "2026-10-04T11:36:40+00:00",
      "commit": "cfa2f44baead3f04e806e87e10aa867ec1b212d3",
      "status": "done",
      "measurement_status": "validated",
      "selection_eligible": true,
      "score": 0.9572675425024442,
      "metric": "score",
      "level": "validation",
      "packet_sha256": "d2ffa69c6e77608f734dcc17770b68ac235e458ff07f78b67de0dab44a2275aa",
      "metrics": {
        "n": 104945,
        "train_n": 489744,
        "model_seed": 20261004,
        "candidate_elapsed_s": 5.099116491153836,
        "budget_s": 180,
        "protocol_sha256": "1ad5813ef657601e05e84583cf1d0efe637878028e669beb3fac35bd952b429d"
      }
    },
    {
      "id": "e-0dd06a8c64baaea9a5b4527d",
      "kind": "source",
      "title": "Common parameters",
      "observed_at": "",
      "captured_at": "2026-10-04T17:07:48+00:00",
      "source_url": "https://catboost.ai/docs/en/references/training-parameters/common",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "ce93814380d04b8935ecd890279eb021d13b609bebd17bc789dea3cd0b15e0e2",
      "packet_sha256": "1f2d6089423093e3b6b533b73e0c1e90125eb6d4e9db17605853f0dc81779120",
      "metrics": {}
    },
    {
      "id": "e-10401fce9a8713f6e1ce2523",
      "kind": "source",
      "title": "Common parameters",
      "observed_at": "",
      "captured_at": "2026-10-04T13:34:22+00:00",
      "source_url": "https://raw.githubusercontent.com/catboost/catboost/master/catboost/docs/en/references/training-parameters/common.md",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "01e0c3abf83ea9cf1481a77e2bd55112f5e440196732729fc8b3966bcfb7b011",
      "packet_sha256": "625dc061752b628053e6f6d43b3161d9b50e14e23c721ae60d868205177b3587",
      "metrics": {}
    },
    {
      "id": "e-1335c4df0ee85e11af4f66d7",
      "kind": "source",
      "title": "Choosing the tree structure | CatBoost",
      "observed_at": "",
      "captured_at": "2026-10-04T09:47:13+00:00",
      "source_url": "https://catboost.ai/docs/en/concepts/algorithm-main-stages_choose-tree-structure",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "ba8e7baaa4bfeaf796ee4383d69ce47a219a5289ba3ee913e8401207103c3081",
      "packet_sha256": "9b91810eb63a0158672fa067506466e99acebcae5c27576846ed749fdc0458d7",
      "metrics": {}
    },
    {
      "id": "e-15adc1988cb151046c60d6d8",
      "kind": "experiment",
      "title": "Fresh designated raw SymmetricTree full-pool policy control",
      "node": "n017",
      "parent": "n016",
      "observed_at": "2026-10-04T16:58:45+00:00",
      "captured_at": "2026-10-04T16:58:48+00:00",
      "commit": "0a09da870138362249d869eccad419aabbcea083",
      "status": "done",
      "measurement_status": "validated",
      "selection_eligible": true,
      "score": 0.9577845705966306,
      "metric": "score",
      "level": "validation",
      "packet_sha256": "05bedeb1bdef9fc41121c0fdaa4e6944b4f9c7683d9d45d253b5340e3463b591",
      "metrics": {
        "n": 104945,
        "train_n": 489744,
        "model_seed": 20261004,
        "candidate_elapsed_s": 6.099306438118219,
        "budget_s": 180,
        "protocol_sha256": "1ad5813ef657601e05e84583cf1d0efe637878028e669beb3fac35bd952b429d"
      }
    },
    {
      "id": "e-19a209702a616e0cdc6a924e",
      "kind": "source",
      "title": "Training on GPU | CatBoost",
      "observed_at": "",
      "captured_at": "2026-10-04T10:01:34+00:00",
      "source_url": "https://catboost.ai/docs/en/features/training-on-gpu",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "e58690d66e5f031e289eab0c41b4648ae283afe7d32863ba3b1c7bfe463d9555",
      "packet_sha256": "1e012fc10b3aa7457b42d47ec58b3fd8d0d24a45e06bf8326fe5d94e65de814c",
      "metrics": {}
    },
    {
      "id": "e-1e647e8ab2fa569e6823e086",
      "kind": "source",
      "title": "Common parameters",
      "observed_at": "",
      "captured_at": "2026-10-04T12:51:32+00:00",
      "source_url": "https://raw.githubusercontent.com/catboost/catboost/master/catboost/docs/en/references/training-parameters/common.md",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "04cf9c3e51a867e089672d316a03d99dcb9062affd94622c9d6eccedefe4b4e6",
      "packet_sha256": "25dddcee246886c8d135ab63642c0e7dcb693715d8908c090c87acfec5167ba2",
      "metrics": {}
    },
    {
      "id": "e-207a732cc9bd82b1f5c91346",
      "kind": "source",
      "title": "Common parameters",
      "observed_at": "",
      "captured_at": "2026-10-04T16:15:14+00:00",
      "source_url": "https://catboost.ai/docs/en/references/training-parameters/common",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "6104ea37a2776a7c94c4fc9af88ea141361c735f6afba06c8986109f38a88332",
      "packet_sha256": "df4d1f8868800b85b06ca11caec6b4006fc4a3d0b45ca3860b0e1f77096df72a",
      "metrics": {}
    },
    {
      "id": "e-2147ecc3e3c9aa129b93bed1",
      "kind": "experiment",
      "title": "Designated raw SymmetricTree depth10 with fixed-round depth8 control",
      "node": "n018",
      "parent": "n017",
      "observed_at": "2026-10-04T17:25:27+00:00",
      "captured_at": "2026-10-04T17:25:29+00:00",
      "commit": "e67631c9365d624f36f6e83af68945f3de356302",
      "status": "done",
      "measurement_status": "validated",
      "selection_eligible": true,
      "score": 0.9577366851375627,
      "metric": "score",
      "level": "validation",
      "packet_sha256": "26127c73db141e8090bc794e2fd5b131769bfb88310b8ad06f984daffaa74694",
      "metrics": {
        "n": 104945,
        "train_n": 489744,
        "model_seed": 20261004,
        "candidate_elapsed_s": 9.610986442305148,
        "budget_s": 180,
        "protocol_sha256": "1ad5813ef657601e05e84583cf1d0efe637878028e669beb3fac35bd952b429d"
      }
    },
    {
      "id": "e-2177017467f332763c6555d7",
      "kind": "source",
      "title": "to_classifier | CatBoost",
      "observed_at": "",
      "captured_at": "2026-10-04T14:42:59+00:00",
      "source_url": "https://catboost.ai/docs/en/concepts/python-reference_to_classifier",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "f56ade67c6b127a65225ae5ef383e914ea59679f4196a295fea02c8775ad3726",
      "packet_sha256": "9223ccadc6407385fc0cf625436a0a9c4a1c7120f07b0eb92f2146eff258549d",
      "metrics": {}
    },
    {
      "id": "e-29cb1794b2c71490a0d6e4a8",
      "kind": "experiment",
      "title": "Digital-service segment gates beyond an ungated main effect",
      "node": "n003",
      "parent": "n002",
      "observed_at": "2026-10-04T10:36:00+00:00",
      "captured_at": "2026-10-04T10:36:02+00:00",
      "commit": "7974f18ac9bfa2046a8f8f3614679e9ecfbbb368",
      "status": "done",
      "measurement_status": "validated",
      "selection_eligible": true,
      "score": 0.9573126801034795,
      "metric": "score",
      "level": "validation",
      "packet_sha256": "58d3b5774053d79563978f04c06b1726c85ea636ca29b420d166e231f3e4871f",
      "metrics": {
        "n": 104945,
        "train_n": 489744,
        "model_seed": 20261004,
        "candidate_elapsed_s": 5.197246262803674,
        "budget_s": 180,
        "protocol_sha256": "1ad5813ef657601e05e84583cf1d0efe637878028e669beb3fac35bd952b429d"
      }
    },
    {
      "id": "e-32de58413a9a17379cdda384",
      "kind": "source",
      "title": "Training on GPU | CatBoost",
      "observed_at": "",
      "captured_at": "2026-10-04T11:01:18+00:00",
      "source_url": "https://catboost.ai/docs/en/features/training-on-gpu",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "ce67867c33dc309568ec466d760a00d62092c65437a8e7f6acbdbc89acfd7e1d",
      "packet_sha256": "a48c5371089dd3e3fee1edd0cd0016597ce7cfbf7904fb0eb1bd7c0efb04d565",
      "metrics": {}
    },
    {
      "id": "e-3674965902fffbb1cccb7a0d",
      "kind": "source",
      "title": "Training on GPU | CatBoost",
      "observed_at": "",
      "captured_at": "2026-10-04T10:21:21+00:00",
      "source_url": "https://catboost.ai/docs/en/features/training-on-gpu",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "003d8806de6268af811551f4b9e55bb10dfe45a77237ad25e59b90acf9ebdd50",
      "packet_sha256": "d0cedda538ef9200674eb34efa22ead5d57132903e99b2a49c8b66446d9692d9",
      "metrics": {}
    },
    {
      "id": "e-38ce7174a007553b49416f26",
      "kind": "experiment",
      "title": "Designated Depthwise support100 with fresh support1 control and fixed-seed treatment repeat",
      "node": "n020",
      "parent": "n016",
      "observed_at": "2026-10-04T18:15:56+00:00",
      "captured_at": "2026-10-04T18:15:58+00:00",
      "commit": "b9e83b16dc2b3a422fddba7c0b87d05fd9d2285b",
      "status": "incomplete",
      "measurement_status": "validated",
      "selection_eligible": true,
      "score": 0.9578806014199407,
      "metric": "score",
      "level": "validation",
      "packet_sha256": "2b9bec21b6e1d83ce41e7191736ea0ad98e6f8ae8bfa79c1b573b963bba70b44",
      "metrics": {
        "n": 104945,
        "train_n": 489744,
        "model_seed": 20261004,
        "candidate_elapsed_s": 6.802453791722655,
        "budget_s": 180,
        "protocol_sha256": "1ad5813ef657601e05e84583cf1d0efe637878028e669beb3fac35bd952b429d"
      }
    },
    {
      "id": "e-4f93f545b0534311793ebd49",
      "kind": "experiment",
      "title": "Smaller-step boosting with a compensating horizon and short-path control",
      "node": "n007",
      "parent": "n004",
      "observed_at": "2026-10-04T13:05:22+00:00",
      "captured_at": "2026-10-04T13:05:24+00:00",
      "commit": "07966be0e441f167c1a6f6584d21c5a04a2dd6b8",
      "status": "done",
      "measurement_status": "validated",
      "selection_eligible": true,
      "score": 0.9571843963907822,
      "metric": "score",
      "level": "validation",
      "packet_sha256": "e2791f62461858e5555294c3116bdccfdc0350ed197b51f3ae256e29cf5bab9d",
      "metrics": {
        "n": 104945,
        "train_n": 489744,
        "model_seed": 20261004,
        "candidate_elapsed_s": 11.11427731718868,
        "budget_s": 180,
        "protocol_sha256": "1ad5813ef657601e05e84583cf1d0efe637878028e669beb3fac35bd952b429d"
      }
    },
    {
      "id": "e-556fa2385067322f60ba8cd9",
      "kind": "source",
      "title": "Categorical features | CatBoost",
      "observed_at": "",
      "captured_at": "2026-10-04T17:34:31+00:00",
      "source_url": "https://catboost.ai/docs/en/features/categorical-features",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "884c859c2b1438d0152da54c194702acd2f7cfbdec9baef86e9546f51d64ea2c",
      "packet_sha256": "8f58bed5f45d576b9c37d567e126069ccda1d80f2b7defb0bddc3bcf9f34f001",
      "metrics": {}
    },
    {
      "id": "e-5bc1b3a3245235c6fde1d37b",
      "kind": "experiment",
      "title": "Regularized depth8 with an isolated deeper-capacity control",
      "node": "n008",
      "parent": "n004",
      "observed_at": "2026-10-04T13:27:53+00:00",
      "captured_at": "2026-10-04T13:27:56+00:00",
      "commit": "698a0e64ef7dde0b1126dfda610f15beef51f2e1",
      "status": "done",
      "measurement_status": "validated",
      "selection_eligible": true,
      "score": 0.9577204389488716,
      "metric": "score",
      "level": "validation",
      "packet_sha256": "8d1841f155d18d49108b87f274c309e0e7838d861b62f1ee013cfa619298e46f",
      "metrics": {
        "n": 104945,
        "train_n": 489744,
        "model_seed": 20261004,
        "candidate_elapsed_s": 6.198641981929541,
        "budget_s": 180,
        "protocol_sha256": "1ad5813ef657601e05e84583cf1d0efe637878028e669beb3fac35bd952b429d"
      }
    },
    {
      "id": "e-5d10410e5880fbc0e8e9b534",
      "kind": "source",
      "title": "to_classifier | CatBoost",
      "observed_at": "",
      "captured_at": "2026-10-04T15:05:30+00:00",
      "source_url": "https://catboost.ai/docs/en/concepts/python-reference_to_classifier",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "e1092b42b6aeb8c4610d5a1cbb5f0c88f37d513a4b38ea4a50cf18d129b4dcb4",
      "packet_sha256": "0d45d5567d2f781006da2553744fb35996f50fdcf2099195be66093c190c3371",
      "metrics": {}
    },
    {
      "id": "e-63f527c11cd4620a00b3c393",
      "kind": "experiment",
      "title": "baseline",
      "node": "root",
      "observed_at": "2026-10-04T09:41:53+00:00",
      "captured_at": "2026-10-04T09:41:53+00:00",
      "commit": "520728d47beb2849b60ea7530f66b2234987a893",
      "status": "done",
      "measurement_status": "validated",
      "selection_eligible": true,
      "score": 0.9572004065863107,
      "metric": "score",
      "level": "validation",
      "packet_sha256": "8c5d35c6cfe0f845b9b9b9c58f90f94b4e5c0ceca38df42ce93a396f1c4dd2b3",
      "metrics": {
        "n": 104945,
        "train_n": 489744,
        "model_seed": 20261004,
        "candidate_elapsed_s": 4.744768015109003,
        "budget_s": 180,
        "protocol_sha256": "1ad5813ef657601e05e84583cf1d0efe637878028e669beb3fac35bd952b429d"
      }
    },
    {
      "id": "e-67f4d6ec016c302d3bd22746",
      "kind": "experiment",
      "title": "Capacity-conditioned overall-mean ablation at fixed boosting rounds",
      "node": "n004",
      "parent": "n002",
      "observed_at": "2026-10-04T10:54:54+00:00",
      "captured_at": "2026-10-04T10:54:57+00:00",
      "commit": "d7aca89bb74986b507362fbe320590949433795a",
      "status": "done",
      "measurement_status": "validated",
      "selection_eligible": true,
      "score": 0.9572919711976331,
      "metric": "score",
      "level": "validation",
      "packet_sha256": "aaa9c6b15c3d0a69794b871c357cd5c4f9ce5e6656566b6c3968d57964e227d1",
      "metrics": {
        "n": 104945,
        "train_n": 489744,
        "model_seed": 20261004,
        "candidate_elapsed_s": 4.947464779019356,
        "budget_s": 180,
        "protocol_sha256": "1ad5813ef657601e05e84583cf1d0efe637878028e669beb3fac35bd952b429d"
      }
    },
    {
      "id": "e-69071d65629d2ddd4239105e",
      "kind": "source",
      "title": "Training on GPU",
      "observed_at": "",
      "captured_at": "2026-10-04T16:41:47+00:00",
      "source_url": "https://catboost.ai/docs/en/features/training-on-gpu",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "921551734d427a1c6204063f1f80411bb11e3859778b49249835909d39349515",
      "packet_sha256": "9e888d63663db67e3a556d7971e1ad48c9e3502c34185763b7c7aa5132d41935",
      "metrics": {}
    },
    {
      "id": "e-694e6783426c23859e7cc77f",
      "kind": "source",
      "title": "Common parameters | CatBoost",
      "observed_at": "",
      "captured_at": "2026-10-04T10:21:21+00:00",
      "source_url": "https://catboost.ai/docs/en/references/training-parameters/common",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "b16440ed65347d3cf28d9e8ae1bd29ff2945bdcbf6e8ed74231531342f6260c9",
      "packet_sha256": "85d051929fca108c7cce59e6eb39d88cf5fc787f587273e3a631ae424df132da",
      "metrics": {}
    },
    {
      "id": "e-69d441de6067c7317a97901e",
      "kind": "source",
      "title": "sum_models | CatBoost",
      "observed_at": "",
      "captured_at": "2026-10-04T15:05:30+00:00",
      "source_url": "https://catboost.ai/docs/en/concepts/python-reference_sum_models",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "8258c18805467bae61bc4e8c618667d8766dd7fcdb14e001e05156f5aa16e62f",
      "packet_sha256": "69ca1af350cc96e9304d64b12df871ebd60ebaabb2c3d5eedaebc13dedb0ebc3",
      "metrics": {}
    },
    {
      "id": "e-6a67c84e9cf53d815f701619",
      "kind": "source",
      "title": "Common parameters | CatBoost",
      "observed_at": "",
      "captured_at": "2026-10-04T10:42:38+00:00",
      "source_url": "https://catboost.ai/docs/en/references/training-parameters/common",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "6b3fd68aaaa8c993090cd50ab6e42a639723ca5b1a34576f9f15a88ca0982bd4",
      "packet_sha256": "72456e25609fba58fec6865b9220667486ab980598c0074015f8689f3f78c2cb",
      "metrics": {}
    },
    {
      "id": "e-6cb1d41624ff34e251702d27",
      "kind": "source",
      "title": "Choosing the tree structure | CatBoost",
      "observed_at": "",
      "captured_at": "2026-10-04T11:21:59+00:00",
      "source_url": "https://catboost.ai/docs/en/concepts/algorithm-main-stages_choose-tree-structure",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "3b645accdea5e70cff89070c82444cdf11420d04447fcc44515456fc5f06678c",
      "packet_sha256": "e1eb84a634a284c65655f12a38470aba295d63b4302cf206405fe2d02946a255",
      "metrics": {}
    },
    {
      "id": "e-6cd38922818d8abed0b71a18",
      "kind": "experiment",
      "title": "Designated raw-only full-pool ablation at depth8 capacity",
      "node": "n014",
      "parent": "n008",
      "observed_at": "2026-10-04T15:46:56+00:00",
      "captured_at": "2026-10-04T15:46:56+00:00",
      "commit": "a21a547a1f00388209aa645e64f960dfb0f40f66",
      "status": "done",
      "measurement_status": "validated",
      "selection_eligible": true,
      "score": 0.9577845705966306,
      "metric": "score",
      "level": "validation",
      "packet_sha256": "4fc8baa6f831ccf51d89f0f7a34438b3fb8ea8b90934adf404e3b226427c755d",
      "metrics": {
        "n": 104945,
        "train_n": 489744,
        "model_seed": 20261004,
        "candidate_elapsed_s": 6.0982181606814265,
        "budget_s": 180,
        "protocol_sha256": "1ad5813ef657601e05e84583cf1d0efe637878028e669beb3fac35bd952b429d"
      }
    },
    {
      "id": "e-7f750b5c9b9715c1340448b1",
      "kind": "experiment",
      "title": "Designated GPU Depthwise with matched raw SymmetricTree control",
      "node": "n016",
      "parent": "n014",
      "observed_at": "2026-10-04T16:32:52+00:00",
      "captured_at": "2026-10-04T16:32:54+00:00",
      "commit": "ed0b71e78d8fff5194cbcc339946d3a705299718",
      "status": "done",
      "measurement_status": "validated",
      "selection_eligible": true,
      "score": 0.9579240131255401,
      "metric": "score",
      "level": "validation",
      "packet_sha256": "9c830ee4ba07a32deaa10d6e5b77a113e79d4b7d9012142bd39d1af1d9f0d43c",
      "metrics": {
        "n": 104945,
        "train_n": 489744,
        "model_seed": 20261004,
        "candidate_elapsed_s": 7.6021023308858275,
        "budget_s": 180,
        "protocol_sha256": "1ad5813ef657601e05e84583cf1d0efe637878028e669beb3fac35bd952b429d"
      }
    },
    {
      "id": "e-80183cd00c072c1a68f8156d",
      "kind": "source",
      "title": "Common parameters | CatBoost",
      "observed_at": "",
      "captured_at": "2026-10-04T15:28:56+00:00",
      "source_url": "https://catboost.ai/docs/en/references/training-parameters/common",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "182fb3a140b25b1108f9905de41811ca10b0c9dd61772cd18ad65f295c499e9e",
      "packet_sha256": "c0fb18a75da5bc139255836ede55fbcb01d052400a796c65b9e6c89773bc20c0",
      "metrics": {}
    },
    {
      "id": "e-83d8207cb56fcf7c93fdef42",
      "kind": "source",
      "title": "Training on GPU | CatBoost",
      "observed_at": "",
      "captured_at": "2026-10-04T15:05:30+00:00",
      "source_url": "https://catboost.ai/docs/en/features/training-on-gpu",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "d4e237dd4a6eb97743e5e52aae9bf39245c22c05f3864b40caf0ab8a136aaddb",
      "packet_sha256": "9c0122cd473e8c24477f92df5418ac5db58451ec4c9368ebc608433f924adea7",
      "metrics": {}
    },
    {
      "id": "e-84bd0566ff3af9afca6d961e",
      "kind": "source",
      "title": "Common parameters | CatBoost",
      "observed_at": "",
      "captured_at": "2026-10-04T11:01:18+00:00",
      "source_url": "https://catboost.ai/docs/en/references/training-parameters/common",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "4c0816b286abeec29363cc99ed082eccf9eec419d77fc3d576441f069811e0a4",
      "packet_sha256": "80c0cddd21f6cfce29a0bedb3a111e0ddd6438e36fc9c59c4e9981d477a53fe3",
      "metrics": {}
    },
    {
      "id": "e-8a473c323b8b0d8ef4a2d29c",
      "kind": "source",
      "title": "Training on GPU",
      "observed_at": "",
      "captured_at": "2026-10-04T13:52:15+00:00",
      "source_url": "https://raw.githubusercontent.com/catboost/catboost/master/catboost/docs/en/features/training-on-gpu.md",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "083c533dcbd00fc8dc808c31982171c0a85b43ec268cfb494006ba460576767e",
      "packet_sha256": "583e8155911763d322e8b6edef86d65a8eb0b11b096c53daa1703f32d1ae5640",
      "metrics": {}
    },
    {
      "id": "e-91457c6271f25982c7045c5b",
      "kind": "experiment",
      "title": "Repair-informed restart averaging with preserved component and continuation controls",
      "node": "n013",
      "parent": "n008",
      "observed_at": "2026-10-04T15:21:03+00:00",
      "captured_at": "2026-10-04T15:21:04+00:00",
      "commit": "23e74b8a3eb110c6716bd2b8bdd413fe4d6ac2dd",
      "status": "done",
      "measurement_status": "validated",
      "selection_eligible": true,
      "score": 0.9577150798991513,
      "metric": "score",
      "level": "validation",
      "packet_sha256": "4503736c366157ef1ec9d7ad431ce2b3755bccc05d077b6ce54864f5905476e4",
      "metrics": {
        "n": 104945,
        "train_n": 489744,
        "model_seed": 20261004,
        "candidate_elapsed_s": 10.162073895335197,
        "budget_s": 180,
        "protocol_sha256": "1ad5813ef657601e05e84583cf1d0efe637878028e669beb3fac35bd952b429d"
      }
    },
    {
      "id": "e-965fbcee80bf46cdd774d161",
      "kind": "experiment",
      "title": "Fresh designated mean full-pool control for the raw-versus-mean gap",
      "node": "n015",
      "parent": "n014",
      "observed_at": "2026-10-04T16:07:32+00:00",
      "captured_at": "2026-10-04T16:07:34+00:00",
      "commit": "8aea030a0d5384d7e13b435b19b6cb6efe9c5a41",
      "status": "done",
      "measurement_status": "validated",
      "selection_eligible": true,
      "score": 0.9577700473603586,
      "metric": "score",
      "level": "validation",
      "packet_sha256": "4d992f43d36596b3f511ff4812bd9f7d269a54f20896a91f180c37d03fcf0779",
      "metrics": {
        "n": 104945,
        "train_n": 489744,
        "model_seed": 20261004,
        "candidate_elapsed_s": 6.249048010446131,
        "budget_s": 180,
        "protocol_sha256": "1ad5813ef657601e05e84583cf1d0efe637878028e669beb3fac35bd952b429d"
      }
    },
    {
      "id": "e-99a1c93e31f38f05adcc50b5",
      "kind": "source",
      "title": "Common parameters | CatBoost",
      "observed_at": "",
      "captured_at": "2026-10-04T11:42:29+00:00",
      "source_url": "https://catboost.ai/docs/en/references/training-parameters/common",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "3b200869a870cb3d7d1eec3358200484199bad43507adb89322e74e73f5f4699",
      "packet_sha256": "364a160ca8e7aea395163bbd5a40d1831d1627e584f6058162a48dbd7c7759ca",
      "metrics": {}
    },
    {
      "id": "e-9d1392e76aeb95791ff0ee50",
      "kind": "source",
      "title": "Parameter tuning",
      "observed_at": "",
      "captured_at": "2026-10-04T12:51:32+00:00",
      "source_url": "https://raw.githubusercontent.com/catboost/catboost/master/catboost/docs/en/concepts/parameter-tuning.md",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "72e507d5cc6ff0ce697a849846f00a9a3f2299ccf811d0f5b62008e5dcf12377",
      "packet_sha256": "9b96d5fc427a3acb2aaa081aa08f10f957f7d43d3695b86111ae54a207523843",
      "metrics": {}
    },
    {
      "id": "e-a872fdf892d6b4233248a75f",
      "kind": "experiment",
      "title": "Reverse-order digital-gate replication with net-utility control",
      "node": "n005",
      "parent": "n003",
      "observed_at": "2026-10-04T11:16:03+00:00",
      "captured_at": "2026-10-04T11:16:04+00:00",
      "commit": "e48b3930da02381a2a5c6110c688e9b040cf04cd",
      "status": "done",
      "measurement_status": "validated",
      "selection_eligible": true,
      "score": 0.9572758458565709,
      "metric": "score",
      "level": "validation",
      "packet_sha256": "46322e5a92fc25f084234dc46a9a8017bfe02f06284224a752b12fe129a1e108",
      "metrics": {
        "n": 104945,
        "train_n": 489744,
        "model_seed": 20261004,
        "candidate_elapsed_s": 4.998024567961693,
        "budget_s": 180,
        "protocol_sha256": "1ad5813ef657601e05e84583cf1d0efe637878028e669beb3fac35bd952b429d"
      }
    },
    {
      "id": "e-a88a499c59ab3499d264d09e",
      "kind": "experiment",
      "title": "Matched summary replication with honest artifact compatibility",
      "node": "n002",
      "parent": "root",
      "observed_at": "2026-10-04T10:14:01+00:00",
      "captured_at": "2026-10-04T10:14:03+00:00",
      "commit": "44a50d708c20e1e2e1af6b13c984bc9638ea2f17",
      "status": "done",
      "measurement_status": "validated",
      "selection_eligible": true,
      "score": 0.9572675016680855,
      "metric": "score",
      "level": "validation",
      "packet_sha256": "5052e16515da1d192e3069149aac679f60b0061e88cb4233f1c78921582254ea",
      "metrics": {
        "n": 104945,
        "train_n": 489744,
        "model_seed": 20261004,
        "candidate_elapsed_s": 4.897205400280654,
        "budget_s": 180,
        "protocol_sha256": "1ad5813ef657601e05e84583cf1d0efe637878028e669beb3fac35bd952b429d"
      }
    },
    {
      "id": "e-a9413bb0a026a18cb909d7c7",
      "kind": "source",
      "title": "predict | CatBoost",
      "observed_at": "",
      "captured_at": "2026-10-04T14:42:59+00:00",
      "source_url": "https://catboost.ai/docs/en/concepts/python-reference_catboost_predict",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "d6b9ef74a3fbea7ec77c7f339af7489c0d467f0b67957cbb41872602e1e6eefb",
      "packet_sha256": "7a1c6eff3da1c4c73f079bc0dbf59b43209714e20aa3fd4acced96d2b031006c",
      "metrics": {}
    },
    {
      "id": "e-a991bea26e83fa3a5a1836d7",
      "kind": "experiment",
      "title": "Compact service profiles beyond overall rating mean",
      "node": "n001",
      "parent": "root",
      "observed_at": "2026-10-04T09:54:54+00:00",
      "captured_at": "2026-10-04T09:54:56+00:00",
      "commit": "d6335db7c15910d20e7d390b5845af13dc06cd34",
      "status": "failed",
      "measurement_status": "unavailable",
      "selection_eligible": false,
      "metric": "score",
      "level": "validation",
      "packet_sha256": "f718de2b5bc88f948c7545985933b9c95511e01ef0ed20614642a88b6a100841",
      "metrics": {}
    },
    {
      "id": "e-ab0caf3bac9abf38f3d4e681",
      "kind": "source",
      "title": "Training on GPU | CatBoost",
      "observed_at": "",
      "captured_at": "2026-10-04T09:47:13+00:00",
      "source_url": "https://catboost.ai/docs/en/features/training-on-gpu",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "04169fef11e82df508d822ffd78f2b991ee9e43e2910701b5e7b90d872b41aea",
      "packet_sha256": "3847fac52e8e8ea63e201508c7128f1292fcc15ab378ee3bb950143b03943581",
      "metrics": {}
    },
    {
      "id": "e-adc9e4bcf107479745d4fa7a",
      "kind": "source",
      "title": "Training on GPU | CatBoost",
      "observed_at": "",
      "captured_at": "2026-10-04T11:42:29+00:00",
      "source_url": "https://catboost.ai/docs/en/features/training-on-gpu",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "a87d23e6fd80be8e6a20c768c70063498c9ceefc31aedcc616db75c9a663eff8",
      "packet_sha256": "20c728c30e8fca6295bf9e0c0587d64327a91af71023cc8d9800859076e75297",
      "metrics": {}
    },
    {
      "id": "e-b9b057a83c6a6e2a083aa4e4",
      "kind": "source",
      "title": "Categorical features | CatBoost",
      "observed_at": "",
      "captured_at": "2026-10-04T14:16:06+00:00",
      "source_url": "https://catboost.ai/docs/en/features/categorical-features",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "d3ba708c3a664842d1bae813a4ac909b0644cd4b00703cbb129c6594c172dd67",
      "packet_sha256": "26d000edd6a7f1a454f2c9e7a5b57a7ebfed7ce43f751d068e5ad8bc5959c18a",
      "metrics": {}
    },
    {
      "id": "e-c2cd63e166d3607e88802a37",
      "kind": "experiment",
      "title": "Dual ordinal/categorical ratings with replacement control",
      "node": "n011",
      "parent": "n008",
      "observed_at": "2026-10-04T14:34:08+00:00",
      "captured_at": "2026-10-04T14:34:09+00:00",
      "commit": "ba5fef1b08d2e35ef9036cf3f68579590bbce7e1",
      "status": "done",
      "measurement_status": "validated",
      "selection_eligible": true,
      "score": 0.9577967296706902,
      "metric": "score",
      "level": "validation",
      "packet_sha256": "64e022ed65a08cb3df211f258a9fce4b0a94afe773611ebafff62cd50324e7ec",
      "metrics": {
        "n": 104945,
        "train_n": 489744,
        "model_seed": 20261004,
        "candidate_elapsed_s": 10.361053691245615,
        "budget_s": 180,
        "protocol_sha256": "1ad5813ef657601e05e84583cf1d0efe637878028e669beb3fac35bd952b429d"
      }
    },
    {
      "id": "e-c33efed018d2c3f828b4310b",
      "kind": "source",
      "title": "Training on GPU | CatBoost",
      "observed_at": "",
      "captured_at": "2026-10-04T14:16:06+00:00",
      "source_url": "https://catboost.ai/docs/en/features/training-on-gpu",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "768076c051c4c27db058a0d5bc628ef26039a9c39be660fb0432aa79acb1147f",
      "packet_sha256": "f725c6fe16ef22d16e872a836775d3f82c7bc405886abf32da7e52f9de8f0914",
      "metrics": {}
    },
    {
      "id": "e-c3b2f5cae022cb55810a2785",
      "kind": "source",
      "title": "sum_models | CatBoost",
      "observed_at": "",
      "captured_at": "2026-10-04T14:42:59+00:00",
      "source_url": "https://catboost.ai/docs/en/concepts/python-reference_sum_models",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "ba04ea8ca7af5f416e24dd16bea4ffeb8868639cf74ec65618a03cf5bf341429",
      "packet_sha256": "78deca83ccd2d218f39d9ff10dbaad366f5af84a04d7568005512d1e048829ff",
      "metrics": {}
    },
    {
      "id": "e-cadf9184021af158fb695d8b",
      "kind": "source",
      "title": "Training on GPU | CatBoost",
      "observed_at": "",
      "captured_at": "2026-10-04T11:21:59+00:00",
      "source_url": "https://catboost.ai/docs/en/features/training-on-gpu",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "ba5f7d4977ee08b3cdfd9278aee525368bf4216e31c5acb798cec5571aaead70",
      "packet_sha256": "6f0a49d2416b9db9bef7ad54cd86f9ea38cdd770bfdb0573038fa2c45f441f40",
      "metrics": {}
    },
    {
      "id": "e-cf8b600834398ec2d98bb197",
      "kind": "source",
      "title": "Common parameters",
      "observed_at": "",
      "captured_at": "2026-10-04T16:41:47+00:00",
      "source_url": "https://catboost.ai/docs/en/references/training-parameters/common",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "ede4aff1767e06f76a733c4f6fc3abc595c925c1ce96d45ecbcdbd93ad5f17c8",
      "packet_sha256": "32a8a1edf6f94b4628d25f6bea00a902c5cf3ddaebebde723b3b1dd29912475e",
      "metrics": {}
    },
    {
      "id": "e-dfd3086fc47b41f87056d7e4",
      "kind": "source",
      "title": "Choosing the tree structure | CatBoost",
      "observed_at": "",
      "captured_at": "2026-10-04T10:01:34+00:00",
      "source_url": "https://catboost.ai/docs/en/concepts/algorithm-main-stages_choose-tree-structure",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "43ab5569d6e8bdf857fe879b69d52f9f9a3ff82f0a0a915e044145cac34ab8b1",
      "packet_sha256": "519d25531476c0d599490eb13cc0c41e6e87e602671d38d505b9408a8577ffbd",
      "metrics": {}
    },
    {
      "id": "e-e09de2bfadc9ddd3da393276",
      "kind": "source",
      "title": "Choosing the tree structure | CatBoost",
      "observed_at": "",
      "captured_at": "2026-10-04T11:01:18+00:00",
      "source_url": "https://catboost.ai/docs/en/concepts/algorithm-main-stages_choose-tree-structure",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "acacbe31100affcb9e77b640c2e837842ed8757b5bc26e018e9bb84af98dc687",
      "packet_sha256": "f5566154ff13e20552e5e238e225ec2f6f59c12ac54c37a7e8686c3ac52f0c75",
      "metrics": {}
    },
    {
      "id": "e-e2ca43fce116d81fcf1ca67d",
      "kind": "source",
      "title": "Common parameters",
      "observed_at": "",
      "captured_at": "2026-10-04T13:12:11+00:00",
      "source_url": "https://raw.githubusercontent.com/catboost/catboost/master/catboost/docs/en/references/training-parameters/common.md",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "9b495708eb787b0988f15cf783d4c4c681dead00ae1f376a6e3cc86c1b18f435",
      "packet_sha256": "2a2dc2e76d22f9a8e8d2c6e9abe18aec26671c32a47d51c7d5c3f11774e2beff",
      "metrics": {}
    },
    {
      "id": "e-e45b07a203abb6ab21700431",
      "kind": "experiment",
      "title": "Fresh shallow full-pool control for the depth8 signal",
      "node": "n009",
      "parent": "n008",
      "observed_at": "2026-10-04T13:46:29+00:00",
      "captured_at": "2026-10-04T13:46:29+00:00",
      "commit": "1d52f783d1c1a2aea53e9e830bc75712135c4144",
      "status": "done",
      "measurement_status": "validated",
      "selection_eligible": true,
      "score": 0.9572919086585254,
      "metric": "score",
      "level": "validation",
      "packet_sha256": "662bc6221f4be19d8bebbb80348a49ec8a79197a34f061fb350394db539df98a",
      "metrics": {
        "n": 104945,
        "train_n": 489744,
        "model_seed": 20261004,
        "candidate_elapsed_s": 5.0922491904348135,
        "budget_s": 180,
        "protocol_sha256": "1ad5813ef657601e05e84583cf1d0efe637878028e669beb3fac35bd952b429d"
      }
    },
    {
      "id": "e-e51e973482c1fdc119bf73aa",
      "kind": "source",
      "title": "Training on GPU | CatBoost",
      "observed_at": "",
      "captured_at": "2026-10-04T10:42:38+00:00",
      "source_url": "https://catboost.ai/docs/en/features/training-on-gpu",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "b4fdb134b47d81f12607e7dbc4f6713da6547be4583baa3c5354e08ccf5a4c35",
      "packet_sha256": "434e589391988778953077cc1e611f9cf271b91bc2e3be08c7bc7ec916a666b9",
      "metrics": {}
    },
    {
      "id": "e-e5dd77cddf0e852a507007f4",
      "kind": "source",
      "title": "Choosing the tree structure | CatBoost",
      "observed_at": "",
      "captured_at": "2026-10-04T11:42:29+00:00",
      "source_url": "https://catboost.ai/docs/en/concepts/algorithm-main-stages_choose-tree-structure",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "74e7bd131425841b2f860574f2bda16fbdc8c67ea6df684110d5dd7822f7f32c",
      "packet_sha256": "5cf7d75b25ca0765792f65e07e07a1e0c34e59030df75497af1a877aa23261c0",
      "metrics": {}
    },
    {
      "id": "e-e618e1f97ed780a296794a51",
      "kind": "experiment",
      "title": "Missing full-pool depth8 control without stronger L2",
      "node": "n010",
      "parent": "n008",
      "observed_at": "2026-10-04T14:07:33+00:00",
      "captured_at": "2026-10-04T14:07:35+00:00",
      "commit": "5706cbfc7733c45f81428432beab1f294c8d4496",
      "status": "done",
      "measurement_status": "validated",
      "selection_eligible": true,
      "score": 0.9576178234932671,
      "metric": "score",
      "level": "validation",
      "packet_sha256": "67977ffa7a45dc8e6845aa016f99cb85b8a067d503fa0bf6f0d5bd5978067668",
      "metrics": {
        "n": 104945,
        "train_n": 489744,
        "model_seed": 20261004,
        "candidate_elapsed_s": 6.300525151193142,
        "budget_s": 180,
        "protocol_sha256": "1ad5813ef657601e05e84583cf1d0efe637878028e669beb3fac35bd952b429d"
      }
    },
    {
      "id": "e-e6a16215eac94c51a91d36fa",
      "kind": "source",
      "title": "Training on GPU",
      "observed_at": "",
      "captured_at": "2026-10-04T13:34:22+00:00",
      "source_url": "https://raw.githubusercontent.com/catboost/catboost/master/catboost/docs/en/features/training-on-gpu.md",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "d003b1d084c460ee4053882f8512af57bed2014bd51bdb4c15815909560b9e07",
      "packet_sha256": "8de0fcec6243a35bf4862f1e2bb927ac7d300ba9ba11dbf4d79830a3e0827bfd",
      "metrics": {}
    },
    {
      "id": "e-f003dbad00dd4f03847237bd",
      "kind": "source",
      "title": "Common parameters | CatBoost",
      "observed_at": "",
      "captured_at": "2026-10-04T15:53:49+00:00",
      "source_url": "https://catboost.ai/docs/en/references/training-parameters/common",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "9b4a2a87b14fba3c3ae28eddb515fac7a7b65e53560ec2bda80e95fe2a5baf46",
      "packet_sha256": "bf6ca37a868887b5fc4efac10249797a4f208da8b806c95eb9906f6872b46413",
      "metrics": {}
    },
    {
      "id": "e-f1cd25f0549c37e5830df239",
      "kind": "source",
      "title": "Choosing the tree structure | CatBoost",
      "observed_at": "",
      "captured_at": "2026-10-04T10:21:21+00:00",
      "source_url": "https://catboost.ai/docs/en/concepts/algorithm-main-stages_choose-tree-structure",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "575f32524bc5472ef849bd3442f31678e21a654d615840d2c67b50bb1ef4fc68",
      "packet_sha256": "163ca837f34e0817a4cfa13739968427085f6da0e2a178a3a266926e0763c22c",
      "metrics": {}
    },
    {
      "id": "e-f505e02ddca3a160dfeaa9c5",
      "kind": "source",
      "title": "Choosing the tree structure | CatBoost",
      "observed_at": "",
      "captured_at": "2026-10-04T17:34:31+00:00",
      "source_url": "https://catboost.ai/docs/en/concepts/algorithm-main-stages_choose-tree-structure",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "54f9fa20bfc5b97451687b373164c20759c18321f576ada2d45c012f700d50e2",
      "packet_sha256": "a358d60a1aef29204cf1214d12d8a34fd841a89032bf3e077f9f2ab22516bb4f",
      "metrics": {}
    },
    {
      "id": "e-f653a1643ebe2654fbe2f15b",
      "kind": "source",
      "title": "Score functions | CatBoost",
      "observed_at": "",
      "captured_at": "2026-10-04T15:28:56+00:00",
      "source_url": "https://catboost.ai/docs/en/concepts/algorithm-score-functions",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "2d256a3284b34fe930b2df0dfa226c1f8309c5f9aed8371ff0fe3646af6884f5",
      "packet_sha256": "f9dde59153121551de8bbdd1b385b94cce1cbb386f5985d1f89308a1fa9a4598",
      "metrics": {}
    },
    {
      "id": "e-fc42c5f68b8e48715b48f7bf",
      "kind": "experiment",
      "title": "Systematic travel-conditioned rating equality with ungated-view ablation",
      "node": "n019",
      "parent": "n017",
      "observed_at": "2026-10-04T17:52:11+00:00",
      "captured_at": "2026-10-04T17:52:11+00:00",
      "commit": "f9524ca5e5a980459e9c52e4431e8b30a48dd768",
      "status": "done",
      "measurement_status": "validated",
      "selection_eligible": true,
      "score": 0.9576975446690855,
      "metric": "score",
      "level": "validation",
      "packet_sha256": "9590c29f4e734aa9b744757069e673105b317d47267240a198892996aa5fb7fd",
      "metrics": {
        "n": 104945,
        "train_n": 489744,
        "model_seed": 20261004,
        "candidate_elapsed_s": 15.330239061266184,
        "budget_s": 180,
        "protocol_sha256": "1ad5813ef657601e05e84583cf1d0efe637878028e669beb3fac35bd952b429d"
      }
    },
    {
      "id": "e-fcde345109c1dc93306afa76",
      "kind": "source",
      "title": "Common parameters | CatBoost",
      "observed_at": "",
      "captured_at": "2026-10-04T18:00:34+00:00",
      "source_url": "https://catboost.ai/docs/en/references/training-parameters/common",
      "attribution": "Research-agent collected excerpt; external assertions require assessment.",
      "sha256": "335f9f80ca54e0574d801cd7bb7d4bc84603addcdf32fb71c12c4b593c928fcf",
      "packet_sha256": "3710746ae47072d04ebbeb077329908867622465574df62648341ff3b11d7194",
      "metrics": {}
    }
  ],
  "history": [
    {
      "revision": "91ff15aa854e8c5ca38d1bbf",
      "previous_revision": "fb8a04195f088e15f4e30f21",
      "updated_at": "2026-10-04T18:31:07+00:00",
      "claim_count": 60,
      "counts": {
        "supported": 41,
        "untested": 4,
        "mixed": 7,
        "refuted": 7,
        "superseded": 1
      },
      "changes": [
        {
          "claim": "n020-depthwise-support100-validation-measurement",
          "before": null,
          "after": "supported",
          "reason": "Added the n020 support100 search-validation measurement: 0.9578806014199407, 0.0000434117 below n016 support1 and below the +0.0002 adoption threshold. Retired the scoped practical-improvement prediction as refuted; the fixed-seed contrast does not establish equivalence or a mechanism. The evidence packet reports quick support100 results below support1, but no independent-seed inference follows.",
          "supersedes": ""
        },
        {
          "claim": "n020-support100-practical-improvement",
          "before": null,
          "after": "refuted",
          "reason": "Added the n020 support100 search-validation measurement: 0.9578806014199407, 0.0000434117 below n016 support1 and below the +0.0002 adoption threshold. Retired the scoped practical-improvement prediction as refuted; the fixed-seed contrast does not establish equivalence or a mechanism. The evidence packet reports quick support100 results below support1, but no independent-seed inference follows.",
          "supersedes": ""
        }
      ],
      "what_changed": "Added the n020 support100 search-validation measurement: 0.9578806014199407, 0.0000434117 below n016 support1 and below the +0.0002 adoption threshold. Retired the scoped practical-improvement prediction as refuted; the fixed-seed contrast does not establish equivalence or a mechanism. The evidence packet reports quick support100 results below support1, but no independent-seed inference follows."
    },
    {
      "revision": "fb8a04195f088e15f4e30f21",
      "previous_revision": "5c1a286fdecc30d55d7b2920",
      "updated_at": "2026-10-04T18:01:07+00:00",
      "claim_count": 58,
      "counts": {
        "supported": 40,
        "untested": 4,
        "mixed": 7,
        "refuted": 6,
        "superseded": 1
      },
      "changes": [
        {
          "claim": "catboost-grow-policy-documentation",
          "before": null,
          "after": "supported",
          "reason": "Added a bounded documentation-based experience: CatBoost describes Depthwise growth and min_data_in_leaf availability on GPU, while clarifying that the threshold restricts split search rather than guaranteeing child sizes. No new task measurement was supplied, so no empirical belief or configuration decision changed; support100 feasibility and any AUC effect remain unknown.",
          "supersedes": ""
        }
      ],
      "what_changed": "Added a bounded documentation-based experience: CatBoost describes Depthwise growth and min_data_in_leaf availability on GPU, while clarifying that the threshold restricts split search rather than guaranteeing child sizes. No new task measurement was supplied, so no empirical belief or configuration decision changed; support100 feasibility and any AUC effect remain unknown."
    },
    {
      "revision": "5c1a286fdecc30d55d7b2920",
      "previous_revision": "b63fc0c76ffa45c4f3f959e7",
      "updated_at": "2026-10-04T17:53:23+00:00",
      "claim_count": 57,
      "counts": {
        "supported": 39,
        "untested": 4,
        "mixed": 7,
        "refuted": 6,
        "superseded": 1
      },
      "changes": [
        {
          "claim": "n019-travel-joint-validation-measurement",
          "before": null,
          "after": "supported",
          "reason": "Added the n019 official measurement: the raw-plus-rating-level-plus-travel-joint arm scored 0.9576975447 and missed the prespecified net adoption threshold. Its quick joint increment over rating-level views was negative, so the scoped practical-benefit prediction is refuted. This weakens the case for adopting this exact joint representation, not for all interactions or equality-only views. No independent replication or final evaluation exists.",
          "supersedes": ""
        },
        {
          "claim": "n019-travel-joint-practical-benefit",
          "before": null,
          "after": "refuted",
          "reason": "Added the n019 official measurement: the raw-plus-rating-level-plus-travel-joint arm scored 0.9576975447 and missed the prespecified net adoption threshold. Its quick joint increment over rating-level views was negative, so the scoped practical-benefit prediction is refuted. This weakens the case for adopting this exact joint representation, not for all interactions or equality-only views. No independent replication or final evaluation exists.",
          "supersedes": ""
        }
      ],
      "what_changed": "Added the n019 official measurement: the raw-plus-rating-level-plus-travel-joint arm scored 0.9576975447 and missed the prespecified net adoption threshold. Its quick joint increment over rating-level views was negative, so the scoped practical-benefit prediction is refuted. This weakens the case for adopting this exact joint representation, not for all interactions or equality-only views. No independent replication or final evaluation exists."
    },
    {
      "revision": "b63fc0c76ffa45c4f3f959e7",
      "previous_revision": "335d60fe4afa7ad5206cca73",
      "updated_at": "2026-10-04T17:34:55+00:00",
      "claim_count": 55,
      "counts": {
        "supported": 38,
        "untested": 4,
        "mixed": 7,
        "refuted": 5,
        "superseded": 1
      },
      "changes": [
        {
          "claim": "catboost-categorical-handling-docs",
          "before": null,
          "after": "supported",
          "reason": "Added scoped experience that CatBoost documentation describes low-cardinality category handling, and an untested hypothesis motivating native joint-category experiments. No task score, implementation, or causal belief changed; supplied evidence contains no airline intervention measurements. Final holdout remains sealed.",
          "supersedes": ""
        },
        {
          "claim": "catboost-joint-category-candidate-hypothesis",
          "before": null,
          "after": "untested",
          "reason": "Added scoped experience that CatBoost documentation describes low-cardinality category handling, and an untested hypothesis motivating native joint-category experiments. No task score, implementation, or causal belief changed; supplied evidence contains no airline intervention measurements. Final holdout remains sealed.",
          "supersedes": ""
        }
      ],
      "what_changed": "Added scoped experience that CatBoost documentation describes low-cardinality category handling, and an untested hypothesis motivating native joint-category experiments. No task score, implementation, or causal belief changed; supplied evidence contains no airline intervention measurements. Final holdout remains sealed."
    },
    {
      "revision": "335d60fe4afa7ad5206cca73",
      "previous_revision": "432f2d876f02f8818c19244b",
      "updated_at": "2026-10-04T17:26:26+00:00",
      "claim_count": 53,
      "counts": {
        "supported": 37,
        "untested": 3,
        "mixed": 7,
        "refuted": 5,
        "superseded": 1
      },
      "changes": [
        {
          "claim": "n018-depth10-validation-measurement",
          "before": null,
          "after": "supported",
          "reason": "The designated n018 depth-10 validation arm scored 0.9577366851, 0.0000478855 below n017 depth 8 and below the +0.0002 practical-benefit threshold. Quick control and treatment results also favored depth 8, with the depth-10 repeat matching the treatment score. This lowers priority for this fixed-round depth escalation, without rejecting deeper learners generally. No service or interaction features were tested, and the final holdout remains sealed.",
          "supersedes": ""
        },
        {
          "claim": "n018-depth10-practical-benefit",
          "before": null,
          "after": "refuted",
          "reason": "The designated n018 depth-10 validation arm scored 0.9577366851, 0.0000478855 below n017 depth 8 and below the +0.0002 practical-benefit threshold. Quick control and treatment results also favored depth 8, with the depth-10 repeat matching the treatment score. This lowers priority for this fixed-round depth escalation, without rejecting deeper learners generally. No service or interaction features were tested, and the final holdout remains sealed.",
          "supersedes": ""
        }
      ],
      "what_changed": "The designated n018 depth-10 validation arm scored 0.9577366851, 0.0000478855 below n017 depth 8 and below the +0.0002 practical-benefit threshold. Quick control and treatment results also favored depth 8, with the depth-10 repeat matching the treatment score. This lowers priority for this fixed-round depth escalation, without rejecting deeper learners generally. No service or interaction features were tested, and the final holdout remains sealed."
    },
    {
      "revision": "432f2d876f02f8818c19244b",
      "previous_revision": "abf1a774b6cfa4e1e96d1796",
      "updated_at": "2026-10-04T17:08:33+00:00",
      "claim_count": 51,
      "counts": {
        "supported": 36,
        "untested": 3,
        "mixed": 7,
        "refuted": 4,
        "superseded": 1
      },
      "changes": [
        {
          "claim": "n017-fresh-symmetric-validation-measurement",
          "before": null,
          "after": "supported",
          "reason": "The n017 fresh SymmetricTree validation score, 0.9577845705966306, met its prespecified strict 0.0002 proximity prediction to n016 Depthwise and exactly recovered the n014 score at reported precision. This narrows the observed fixed-seed full-pool policy contrast but does not establish equivalence or stable ordering. The Depthwise practical-adoption threshold remains unmet. The parameter documentation adds no airline measurement. No evidence changes the service_mean or interaction-feature conclusions.",
          "supersedes": ""
        },
        {
          "claim": "n017-symmetric-depthwise-practical-proximity",
          "before": null,
          "after": "supported",
          "reason": "The n017 fresh SymmetricTree validation score, 0.9577845705966306, met its prespecified strict 0.0002 proximity prediction to n016 Depthwise and exactly recovered the n014 score at reported precision. This narrows the observed fixed-seed full-pool policy contrast but does not establish equivalence or stable ordering. The Depthwise practical-adoption threshold remains unmet. The parameter documentation adds no airline measurement. No evidence changes the service_mean or interaction-feature conclusions.",
          "supersedes": ""
        }
      ],
      "what_changed": "The n017 fresh SymmetricTree validation score, 0.9577845705966306, met its prespecified strict 0.0002 proximity prediction to n016 Depthwise and exactly recovered the n014 score at reported precision. This narrows the observed fixed-seed full-pool policy contrast but does not establish equivalence or stable ordering. The Depthwise practical-adoption threshold remains unmet. The parameter documentation adds no airline measurement. No evidence changes the service_mean or interaction-feature conclusions."
    },
    {
      "revision": "abf1a774b6cfa4e1e96d1796",
      "previous_revision": "ab481a3cba77d6c717735036",
      "updated_at": "2026-10-04T16:42:14+00:00",
      "claim_count": 49,
      "counts": {
        "supported": 34,
        "untested": 3,
        "mixed": 7,
        "refuted": 4,
        "superseded": 1
      },
      "changes": [
        {
          "claim": "catboost-gpu-training-nondeterminism-documentation",
          "before": "supported",
          "after": "supported",
          "reason": "Added a collected GPU-training excerpt supporting the existing documentation experience about nondeterminism, and a collected growth-policy excerpt supporting the existing parameter-semantics experience. These are source updates, not independent experiments or replications; the airline modeling decision and measured evidence remain unchanged.",
          "supersedes": ""
        },
        {
          "claim": "catboost-parameter-roles-documentation",
          "before": "supported",
          "after": "supported",
          "reason": "Added a collected GPU-training excerpt supporting the existing documentation experience about nondeterminism, and a collected growth-policy excerpt supporting the existing parameter-semantics experience. These are source updates, not independent experiments or replications; the airline modeling decision and measured evidence remain unchanged.",
          "supersedes": ""
        }
      ],
      "what_changed": "Added a collected GPU-training excerpt supporting the existing documentation experience about nondeterminism, and a collected growth-policy excerpt supporting the existing parameter-semantics experience. These are source updates, not independent experiments or replications; the airline modeling decision and measured evidence remain unchanged."
    },
    {
      "revision": "ab481a3cba77d6c717735036",
      "previous_revision": "19940b515a8328ea09c14a69",
      "updated_at": "2026-10-04T16:34:19+00:00",
      "claim_count": 49,
      "counts": {
        "supported": 34,
        "untested": 3,
        "mixed": 7,
        "refuted": 4,
        "superseded": 1
      },
      "changes": [
        {
          "claim": "n016-depthwise-validation-measurement",
          "before": null,
          "after": "supported",
          "reason": "The designated n016 Depthwise arm completed full-pool validation at 0.9579240131, +0.0001394425 over historical n014, missing its predeclared +0.0002 adoption threshold. This refutes that scoped practical-benefit prediction without establishing equivalence or harm. Quick ordering was negative, but quick and validation populations differ. Reported effective partition and realized tree complexity also differed, so no pure policy or passenger-mechanism attribution is warranted. Existing raw-versus-mean evidence is unchanged.",
          "supersedes": ""
        },
        {
          "claim": "n016-depthwise-practical-benefit",
          "before": null,
          "after": "refuted",
          "reason": "The designated n016 Depthwise arm completed full-pool validation at 0.9579240131, +0.0001394425 over historical n014, missing its predeclared +0.0002 adoption threshold. This refutes that scoped practical-benefit prediction without establishing equivalence or harm. Quick ordering was negative, but quick and validation populations differ. Reported effective partition and realized tree complexity also differed, so no pure policy or passenger-mechanism attribution is warranted. Existing raw-versus-mean evidence is unchanged.",
          "supersedes": ""
        },
        {
          "claim": "n016-depthwise-runnable-feasibility",
          "before": null,
          "after": "supported",
          "reason": "The designated n016 Depthwise arm completed full-pool validation at 0.9579240131, +0.0001394425 over historical n014, missing its predeclared +0.0002 adoption threshold. This refutes that scoped practical-benefit prediction without establishing equivalence or harm. Quick ordering was negative, but quick and validation populations differ. Reported effective partition and realized tree complexity also differed, so no pure policy or passenger-mechanism attribution is warranted. Existing raw-versus-mean evidence is unchanged.",
          "supersedes": ""
        }
      ],
      "what_changed": "The designated n016 Depthwise arm completed full-pool validation at 0.9579240131, +0.0001394425 over historical n014, missing its predeclared +0.0002 adoption threshold. This refutes that scoped practical-benefit prediction without establishing equivalence or harm. Quick ordering was negative, but quick and validation populations differ. Reported effective partition and realized tree complexity also differed, so no pure policy or passenger-mechanism attribution is warranted. Existing raw-versus-mean evidence is unchanged."
    },
    {
      "revision": "19940b515a8328ea09c14a69",
      "previous_revision": "6fc837e557608185819e07c2",
      "updated_at": "2026-10-04T16:16:05+00:00",
      "claim_count": 46,
      "counts": {
        "supported": 32,
        "untested": 3,
        "mixed": 7,
        "refuted": 3,
        "superseded": 1
      },
      "changes": [
        {
          "claim": "catboost-parameter-roles-documentation",
          "before": "supported",
          "after": "supported",
          "reason": "Added a collected CatBoost documentation excerpt describing growth-policy semantics and min_data_in_leaf availability and behavior. No task measurement or airline-performance belief changed; local runtime and efficacy remain untested.",
          "supersedes": ""
        }
      ],
      "what_changed": "Added a collected CatBoost documentation excerpt describing growth-policy semantics and min_data_in_leaf availability and behavior. No task measurement or airline-performance belief changed; local runtime and efficacy remain untested."
    },
    {
      "revision": "6fc837e557608185819e07c2",
      "previous_revision": "e437d783b3f8349e8afd5482",
      "updated_at": "2026-10-04T16:08:29+00:00",
      "claim_count": 46,
      "counts": {
        "supported": 32,
        "untested": 3,
        "mixed": 7,
        "refuted": 3,
        "superseded": 1
      },
      "changes": [
        {
          "claim": "n014-raw-retention-at-depth8",
          "before": "mixed",
          "after": "mixed",
          "reason": "The previously missing fresh full-pool mean control was measured: it scored 0.9577700474, 0.0000145232 below n014 raw and within the 0.0002 action margin. The n015 practical prediction was met on validation but opposed by the quick result, where mean led by 0.0004148383. This provides a provisional practical retention signal, not equivalence or a general feature decision; broader modeling branches remain open.",
          "supersedes": ""
        },
        {
          "claim": "n015-fresh-mean-validation-measurement",
          "before": null,
          "after": "supported",
          "reason": "The previously missing fresh full-pool mean control was measured: it scored 0.9577700474, 0.0000145232 below n014 raw and within the 0.0002 action margin. The n015 practical prediction was met on validation but opposed by the quick result, where mean led by 0.0004148383. This provides a provisional practical retention signal, not equivalence or a general feature decision; broader modeling branches remain open.",
          "supersedes": ""
        },
        {
          "claim": "n015-raw-mean-practical-separation",
          "before": null,
          "after": "mixed",
          "reason": "The previously missing fresh full-pool mean control was measured: it scored 0.9577700474, 0.0000145232 below n014 raw and within the 0.0002 action margin. The n015 practical prediction was met on validation but opposed by the quick result, where mean led by 0.0004148383. This provides a provisional practical retention signal, not equivalence or a general feature decision; broader modeling branches remain open.",
          "supersedes": ""
        }
      ],
      "what_changed": "The previously missing fresh full-pool mean control was measured: it scored 0.9577700474, 0.0000145232 below n014 raw and within the 0.0002 action margin. The n015 practical prediction was met on validation but opposed by the quick result, where mean led by 0.0004148383. This provides a provisional practical retention signal, not equivalence or a general feature decision; broader modeling branches remain open."
    },
    {
      "revision": "e437d783b3f8349e8afd5482",
      "previous_revision": "f8be1dc51b6d4bcd23efc4cc",
      "updated_at": "2026-10-04T15:54:19+00:00",
      "claim_count": 44,
      "counts": {
        "supported": 31,
        "untested": 3,
        "mixed": 6,
        "refuted": 3,
        "superseded": 1
      },
      "changes": [
        {
          "claim": "catboost-parameter-roles-documentation",
          "before": "supported",
          "after": "supported",
          "reason": "Added documentation-based context for Depthwise, SymmetricTree, and GPU nondeterminism. No airline measurement or mechanism evidence was added, so the feature and configuration decisions remain unchanged.",
          "supersedes": ""
        },
        {
          "claim": "catboost-gpu-training-nondeterminism-documentation",
          "before": null,
          "after": "supported",
          "reason": "Added documentation-based context for Depthwise, SymmetricTree, and GPU nondeterminism. No airline measurement or mechanism evidence was added, so the feature and configuration decisions remain unchanged.",
          "supersedes": ""
        }
      ],
      "what_changed": "Added documentation-based context for Depthwise, SymmetricTree, and GPU nondeterminism. No airline measurement or mechanism evidence was added, so the feature and configuration decisions remain unchanged."
    },
    {
      "revision": "f8be1dc51b6d4bcd23efc4cc",
      "previous_revision": "a3fa2f5b9e5979cddb621c93",
      "updated_at": "2026-10-04T15:47:29+00:00",
      "claim_count": 43,
      "counts": {
        "supported": 30,
        "untested": 3,
        "mixed": 6,
        "refuted": 3,
        "superseded": 1
      },
      "changes": [
        {
          "claim": "n014-raw-only-depth8-validation-measurement",
          "before": null,
          "after": "supported",
          "reason": "A full-pool raw-only depth-8/L2=10 evaluation is now available at 0.9577845706, 0.0000641 above historical n008 with service_mean; it passes the prespecified lower retention screen but not the raw-preference threshold. The quick raw-only score was 0.0004148 below the historical mean-feature quick result, preserving counterevidence. No matched full-pool mean control was obtained, so the feature decision remains provisional and no mechanism conclusion changed.",
          "supersedes": ""
        },
        {
          "claim": "n014-raw-retention-at-depth8",
          "before": null,
          "after": "mixed",
          "reason": "A full-pool raw-only depth-8/L2=10 evaluation is now available at 0.9577845706, 0.0000641 above historical n008 with service_mean; it passes the prespecified lower retention screen but not the raw-preference threshold. The quick raw-only score was 0.0004148 below the historical mean-feature quick result, preserving counterevidence. No matched full-pool mean control was obtained, so the feature decision remains provisional and no mechanism conclusion changed.",
          "supersedes": ""
        }
      ],
      "what_changed": "A full-pool raw-only depth-8/L2=10 evaluation is now available at 0.9577845706, 0.0000641 above historical n008 with service_mean; it passes the prespecified lower retention screen but not the raw-preference threshold. The quick raw-only score was 0.0004148 below the historical mean-feature quick result, preserving counterevidence. No matched full-pool mean control was obtained, so the feature decision remains provisional and no mechanism conclusion changed."
    },
    {
      "revision": "a3fa2f5b9e5979cddb621c93",
      "previous_revision": "ea3e2b3c27d1e3e427de7f61",
      "updated_at": "2026-10-04T15:29:39+00:00",
      "claim_count": 41,
      "counts": {
        "supported": 29,
        "untested": 3,
        "mixed": 5,
        "refuted": 3,
        "superseded": 1
      },
      "changes": [
        {
          "claim": "catboost-parameter-roles-documentation",
          "before": "supported",
          "after": "supported",
          "reason": "Updated the documentation-based parameter and tree-construction context and added a documentation-based description of gradient boosting and split-score selection. These are not new experiment results: no performance belief, adoption decision, or mechanism conclusion changed. The n013 restart-average result and its unresolved matched-component comparisons remain as previously recorded.",
          "supersedes": ""
        },
        {
          "claim": "catboost-greedy-tree-structure-docs",
          "before": "supported",
          "after": "supported",
          "reason": "Updated the documentation-based parameter and tree-construction context and added a documentation-based description of gradient boosting and split-score selection. These are not new experiment results: no performance belief, adoption decision, or mechanism conclusion changed. The n013 restart-average result and its unresolved matched-component comparisons remain as previously recorded.",
          "supersedes": ""
        },
        {
          "claim": "catboost-greedy-split-competition-hypothesis",
          "before": "untested",
          "after": "untested",
          "reason": "Updated the documentation-based parameter and tree-construction context and added a documentation-based description of gradient boosting and split-score selection. These are not new experiment results: no performance belief, adoption decision, or mechanism conclusion changed. The n013 restart-average result and its unresolved matched-component comparisons remain as previously recorded.",
          "supersedes": ""
        },
        {
          "claim": "catboost-gradient-boosting-score-documentation",
          "before": null,
          "after": "supported",
          "reason": "Updated the documentation-based parameter and tree-construction context and added a documentation-based description of gradient boosting and split-score selection. These are not new experiment results: no performance belief, adoption decision, or mechanism conclusion changed. The n013 restart-average result and its unresolved matched-component comparisons remain as previously recorded.",
          "supersedes": ""
        }
      ],
      "what_changed": "Updated the documentation-based parameter and tree-construction context and added a documentation-based description of gradient boosting and split-score selection. These are not new experiment results: no performance belief, adoption decision, or mechanism conclusion changed. The n013 restart-average result and its unresolved matched-component comparisons remain as previously recorded."
    },
    {
      "revision": "ea3e2b3c27d1e3e427de7f61",
      "previous_revision": "fb55162983f7a5258d966b6c",
      "updated_at": "2026-10-04T15:21:37+00:00",
      "claim_count": 40,
      "counts": {
        "supported": 28,
        "untested": 3,
        "mixed": 5,
        "refuted": 3,
        "superseded": 1
      },
      "changes": [
        {
          "claim": "n013-restart-average-validation-measurement",
          "before": null,
          "after": "supported",
          "reason": "The restart-average validation score is no longer unavailable: n013 measured 0.9577150799. It missed the historical-reference-plus-0.0002 adoption screen, while quick results showed it below the stronger component and above the continuation. Thus the fixed configuration has mixed evidence and is not justified for adoption by these results. n012 remains an unavailable measurement; n013 does not revise that historical failure. No mechanism or independent replication is established.",
          "supersedes": ""
        },
        {
          "claim": "n013-fixed-restart-practical-benefit",
          "before": null,
          "after": "mixed",
          "reason": "The restart-average validation score is no longer unavailable: n013 measured 0.9577150799. It missed the historical-reference-plus-0.0002 adoption screen, while quick results showed it below the stronger component and above the continuation. Thus the fixed configuration has mixed evidence and is not justified for adoption by these results. n012 remains an unavailable measurement; n013 does not revise that historical failure. No mechanism or independent replication is established.",
          "supersedes": ""
        }
      ],
      "what_changed": "The restart-average validation score is no longer unavailable: n013 measured 0.9577150799. It missed the historical-reference-plus-0.0002 adoption screen, while quick results showed it below the stronger component and above the continuation. Thus the fixed configuration has mixed evidence and is not justified for adoption by these results. n012 remains an unavailable measurement; n013 does not revise that historical failure. No mechanism or independent replication is established."
    },
    {
      "revision": "fb55162983f7a5258d966b6c",
      "previous_revision": "7827b849120a00a07033b627",
      "updated_at": "2026-10-04T15:12:34+00:00",
      "claim_count": 38,
      "counts": {
        "supported": 27,
        "untested": 3,
        "mixed": 4,
        "refuted": 3,
        "superseded": 1
      },
      "changes": [
        {
          "claim": "catboost-sum-models-api-feasibility",
          "before": "supported",
          "after": "supported",
          "reason": "Added collected API evidence supporting documented weighted model blending and compatible classifier conversion. This does not repair n012's unavailable validation measurement or change performance beliefs; no efficacy score was obtained, and no existing scoped performance claim is revised.",
          "supersedes": ""
        }
      ],
      "what_changed": "Added collected API evidence supporting documented weighted model blending and compatible classifier conversion. This does not repair n012's unavailable validation measurement or change performance beliefs; no efficacy score was obtained, and no existing scoped performance claim is revised."
    },
    {
      "revision": "7827b849120a00a07033b627",
      "previous_revision": "b707d29303d349c5b097957c",
      "updated_at": "2026-10-04T14:58:32+00:00",
      "claim_count": 38,
      "counts": {
        "supported": 27,
        "untested": 3,
        "mixed": 4,
        "refuted": 3,
        "superseded": 1
      },
      "changes": [
        {
          "claim": "n012-fixed-restart-averaging-measurement",
          "before": null,
          "after": "supported",
          "reason": "Recorded n012 as an unavailable validation measurement caused by an implementation failure and exhausted trial budget, not as a negative or positive modeling result. The continuation fit and non-fitting technical audits do not change performance beliefs. No existing scoped performance claim is revised.",
          "supersedes": ""
        }
      ],
      "what_changed": "Recorded n012 as an unavailable validation measurement caused by an implementation failure and exhausted trial budget, not as a negative or positive modeling result. The continuation fit and non-fitting technical audits do not change performance beliefs. No existing scoped performance claim is revised."
    },
    {
      "revision": "b707d29303d349c5b097957c",
      "previous_revision": "2fb61e81da103d97a2f14854",
      "updated_at": "2026-10-04T14:43:29+00:00",
      "claim_count": 37,
      "counts": {
        "supported": 26,
        "untested": 3,
        "mixed": 4,
        "refuted": 3,
        "superseded": 1
      },
      "changes": [
        {
          "claim": "catboost-sum-models-api-feasibility",
          "before": null,
          "after": "supported",
          "reason": "Added documentation-based experience claims for model merging, prediction, and classifier conversion. No task-specific experiment or score was supplied, so no performance claim, configuration decision, or prior claim status changes.",
          "supersedes": ""
        },
        {
          "claim": "catboost-merged-model-prediction-contract",
          "before": null,
          "after": "supported",
          "reason": "Added documentation-based experience claims for model merging, prediction, and classifier conversion. No task-specific experiment or score was supplied, so no performance claim, configuration decision, or prior claim status changes.",
          "supersedes": ""
        },
        {
          "claim": "catboost-to-classifier-conversion-api",
          "before": null,
          "after": "supported",
          "reason": "Added documentation-based experience claims for model merging, prediction, and classifier conversion. No task-specific experiment or score was supplied, so no performance claim, configuration decision, or prior claim status changes.",
          "supersedes": ""
        }
      ],
      "what_changed": "Added documentation-based experience claims for model merging, prediction, and classifier conversion. No task-specific experiment or score was supplied, so no performance claim, configuration decision, or prior claim status changes."
    },
    {
      "revision": "2fb61e81da103d97a2f14854",
      "previous_revision": "d0150975ddc63c6c8323fd09",
      "updated_at": "2026-10-04T14:35:01+00:00",
      "claim_count": 34,
      "counts": {
        "supported": 23,
        "untested": 3,
        "mixed": 4,
        "refuted": 3,
        "superseded": 1
      },
      "changes": [
        {
          "claim": "n011-dual-rating-validation-measurement",
          "before": null,
          "after": "supported",
          "reason": "Added the n011 validation measurement and revised the dual-representation interpretation from a quick-screen pass to mixed evidence for practical benefit: quick C−A was +0.000229723, while the validation contrast with historical n008 was +0.000076291, below the +0.0002 adoption margin. The result does not establish mechanism or justify default adoption. Existing rating-summary and other scoped claims are unchanged.",
          "supersedes": ""
        },
        {
          "claim": "n011-dual-rating-representation-benefit",
          "before": null,
          "after": "mixed",
          "reason": "Added the n011 validation measurement and revised the dual-representation interpretation from a quick-screen pass to mixed evidence for practical benefit: quick C−A was +0.000229723, while the validation contrast with historical n008 was +0.000076291, below the +0.0002 adoption margin. The result does not establish mechanism or justify default adoption. Existing rating-summary and other scoped claims are unchanged.",
          "supersedes": ""
        }
      ],
      "what_changed": "Added the n011 validation measurement and revised the dual-representation interpretation from a quick-screen pass to mixed evidence for practical benefit: quick C−A was +0.000229723, while the validation contrast with historical n008 was +0.000076291, below the +0.0002 adoption margin. The result does not establish mechanism or justify default adoption. Existing rating-summary and other scoped claims are unchanged."
    },
    {
      "revision": "d0150975ddc63c6c8323fd09",
      "previous_revision": "ed737712d9cef910f70cba82",
      "updated_at": "2026-10-04T14:16:21+00:00",
      "claim_count": 32,
      "counts": {
        "supported": 22,
        "untested": 3,
        "mixed": 3,
        "refuted": 3,
        "superseded": 1
      },
      "changes": [],
      "what_changed": "No task-specific claim or measured observation changes: the new packets contain external documentation only, with no supplied evaluation results. They support considering a controlled categorical-representation comparison and retaining uncertainty about GPU variation; they do not establish that one-hot treatment, rating interactions, or any feature improves AUC."
    },
    {
      "revision": "ed737712d9cef910f70cba82",
      "previous_revision": "f3104fadde7254f0afcde799",
      "updated_at": "2026-10-04T14:08:31+00:00",
      "claim_count": 32,
      "counts": {
        "supported": 22,
        "untested": 3,
        "mixed": 3,
        "refuted": 3,
        "superseded": 1
      },
      "changes": [
        {
          "claim": "n010-depth8-l2-3-validation-measurement",
          "before": null,
          "after": "supported",
          "reason": "Stored evidence status: n010-depth8-l2-3-practical-retention: requested supported, retained as mixed because both supporting and opposing references remain. No historical reference was removed.\n\nDraft interpretation (subject to the corrections above): n010 measured depth-8/L2=3 at ROC-AUC 0.9576178235. It was 0.0003259148 above n009 depth-6/L2=3 and 0.0001026155 below n008 depth-8/L2=10, passing the prespecified retention screens. The earlier missing-depth-8/L2=3 gap is closed and the predecessor claim is retired with its original scope preserved. Quick counterevidence, GPU uncertainty, adaptive selection, and mechanism uncertainty remain; no causal or robust-improvement claim is added.",
          "supersedes": ""
        },
        {
          "claim": "n010-depth8-l2-3-practical-retention",
          "before": null,
          "after": "mixed",
          "reason": "Stored evidence status: n010-depth8-l2-3-practical-retention: requested supported, retained as mixed because both supporting and opposing references remain. No historical reference was removed.\n\nDraft interpretation (subject to the corrections above): n010 measured depth-8/L2=3 at ROC-AUC 0.9576178235. It was 0.0003259148 above n009 depth-6/L2=3 and 0.0001026155 below n008 depth-8/L2=10, passing the prespecified retention screens. The earlier missing-depth-8/L2=3 gap is closed and the predecessor claim is retired with its original scope preserved. Quick counterevidence, GPU uncertainty, adaptive selection, and mechanism uncertainty remain; no causal or robust-improvement claim is added.",
          "supersedes": ""
        },
        {
          "claim": "airline-depth-and-regularization-effects",
          "before": "mixed",
          "after": "superseded",
          "reason": "Stored evidence status: n010-depth8-l2-3-practical-retention: requested supported, retained as mixed because both supporting and opposing references remain. No historical reference was removed.\n\nDraft interpretation (subject to the corrections above): n010 measured depth-8/L2=3 at ROC-AUC 0.9576178235. It was 0.0003259148 above n009 depth-6/L2=3 and 0.0001026155 below n008 depth-8/L2=10, passing the prespecified retention screens. The earlier missing-depth-8/L2=3 gap is closed and the predecessor claim is retired with its original scope preserved. Quick counterevidence, GPU uncertainty, adaptive selection, and mechanism uncertainty remain; no causal or robust-improvement claim is added.",
          "supersedes": ""
        },
        {
          "claim": "airline-depth-l2-matched-contrasts",
          "before": null,
          "after": "mixed",
          "reason": "n010 measured the previously missing depth-8/L2=3 full-pool arm, changing the scope from an unresolved missing comparison to three specified fixed-seed configuration contrasts. The earlier claim's scope is retained unchanged; its missing-arm interpretation is retired rather than silently rewritten. The new measurements still do not identify causal effects or resolve run uncertainty.",
          "supersedes": "airline-depth-and-regularization-effects"
        }
      ],
      "what_changed": "Stored evidence status: n010-depth8-l2-3-practical-retention: requested supported, retained as mixed because both supporting and opposing references remain. No historical reference was removed.\n\nDraft interpretation (subject to the corrections above): n010 measured depth-8/L2=3 at ROC-AUC 0.9576178235. It was 0.0003259148 above n009 depth-6/L2=3 and 0.0001026155 below n008 depth-8/L2=10, passing the prespecified retention screens. The earlier missing-depth-8/L2=3 gap is closed and the predecessor claim is retired with its original scope preserved. Quick counterevidence, GPU uncertainty, adaptive selection, and mechanism uncertainty remain; no causal or robust-improvement claim is added."
    },
    {
      "revision": "f3104fadde7254f0afcde799",
      "previous_revision": "714b2f3e173bd103b92a35cc",
      "updated_at": "2026-10-04T13:53:42+00:00",
      "claim_count": 29,
      "counts": {
        "supported": 21,
        "untested": 3,
        "mixed": 2,
        "refuted": 3
      },
      "changes": [
        {
          "claim": "n009-depth6-l2-3-validation-measurement",
          "before": null,
          "after": "supported",
          "reason": "Added the n009 fresh depth-6/L2=3 validation measurement and its contrast with n008 depth-8/L2=10. The prespecified 0.0002 full-pool comparison screen was met. This replaces the prior missing-fresh-control gap, but does not resolve the missing depth-8/L2=3 arm, the opposing quick pattern, run uncertainty, or mechanism. Historical evidence and counterevidence remain intact; no mechanism claim is upgraded.",
          "supersedes": ""
        },
        {
          "claim": "n009-depth8-l2-10-practical-contrast",
          "before": null,
          "after": "supported",
          "reason": "Added the n009 fresh depth-6/L2=3 validation measurement and its contrast with n008 depth-8/L2=10. The prespecified 0.0002 full-pool comparison screen was met. This replaces the prior missing-fresh-control gap, but does not resolve the missing depth-8/L2=3 arm, the opposing quick pattern, run uncertainty, or mechanism. Historical evidence and counterevidence remain intact; no mechanism claim is upgraded.",
          "supersedes": ""
        },
        {
          "claim": "airline-depth-and-regularization-effects",
          "before": "mixed",
          "after": "mixed",
          "reason": "Added the n009 fresh depth-6/L2=3 validation measurement and its contrast with n008 depth-8/L2=10. The prespecified 0.0002 full-pool comparison screen was met. This replaces the prior missing-fresh-control gap, but does not resolve the missing depth-8/L2=3 arm, the opposing quick pattern, run uncertainty, or mechanism. Historical evidence and counterevidence remain intact; no mechanism claim is upgraded.",
          "supersedes": ""
        }
      ],
      "what_changed": "Added the n009 fresh depth-6/L2=3 validation measurement and its contrast with n008 depth-8/L2=10. The prespecified 0.0002 full-pool comparison screen was met. This replaces the prior missing-fresh-control gap, but does not resolve the missing depth-8/L2=3 arm, the opposing quick pattern, run uncertainty, or mechanism. Historical evidence and counterevidence remain intact; no mechanism claim is upgraded."
    },
    {
      "revision": "714b2f3e173bd103b92a35cc",
      "previous_revision": "b2eb756aa2ceefc54860470d",
      "updated_at": "2026-10-04T13:28:46+00:00",
      "claim_count": 27,
      "counts": {
        "supported": 19,
        "untested": 3,
        "mixed": 2,
        "refuted": 3
      },
      "changes": [
        {
          "claim": "airline-depth-and-regularization-effects",
          "before": "untested",
          "after": "mixed",
          "reason": "The previously untested depth/regularization proposition now has mixed, scoped evidence: n008 quick scores were lower for both depth-8 arms than depth-6/L2=3, while the evaluated depth-8/L2=10 validation score exceeded historical references numerically. This does not establish improvement because there is no fresh full-pool control, no validation result for depth-8/L2=3, and the quick prediction for a practically useful C-versus-A gain failed. Preserve both measurements and prioritize matched full-pool controls before adopting or rejecting the treatment. No mechanism claim is upgraded.",
          "supersedes": ""
        },
        {
          "claim": "n008-depth8-l2-quick-and-validation-measurement",
          "before": null,
          "after": "supported",
          "reason": "The previously untested depth/regularization proposition now has mixed, scoped evidence: n008 quick scores were lower for both depth-8 arms than depth-6/L2=3, while the evaluated depth-8/L2=10 validation score exceeded historical references numerically. This does not establish improvement because there is no fresh full-pool control, no validation result for depth-8/L2=3, and the quick prediction for a practically useful C-versus-A gain failed. Preserve both measurements and prioritize matched full-pool controls before adopting or rejecting the treatment. No mechanism claim is upgraded.",
          "supersedes": ""
        }
      ],
      "what_changed": "The previously untested depth/regularization proposition now has mixed, scoped evidence: n008 quick scores were lower for both depth-8 arms than depth-6/L2=3, while the evaluated depth-8/L2=10 validation score exceeded historical references numerically. This does not establish improvement because there is no fresh full-pool control, no validation result for depth-8/L2=3, and the quick prediction for a practically useful C-versus-A gain failed. Preserve both measurements and prioritize matched full-pool controls before adopting or rejecting the treatment. No mechanism claim is upgraded."
    },
    {
      "revision": "b2eb756aa2ceefc54860470d",
      "previous_revision": "907f5578c6d792c55f892817",
      "updated_at": "2026-10-04T13:13:03+00:00",
      "claim_count": 26,
      "counts": {
        "supported": 18,
        "untested": 4,
        "mixed": 1,
        "refuted": 3
      },
      "changes": [
        {
          "claim": "catboost-parameter-roles-documentation",
          "before": "supported",
          "after": "supported",
          "reason": "Added documentation-based rationale for a depth intervention and regularization control. No new airline measurements were supplied, so no performance belief or configuration decision changes; trajectory and service-feature evidence remain as previously recorded.",
          "supersedes": ""
        },
        {
          "claim": "airline-depth-and-regularization-effects",
          "before": null,
          "after": "untested",
          "reason": "Added documentation-based rationale for a depth intervention and regularization control. No new airline measurements were supplied, so no performance belief or configuration decision changes; trajectory and service-feature evidence remain as previously recorded.",
          "supersedes": ""
        }
      ],
      "what_changed": "Added documentation-based rationale for a depth intervention and regularization control. No new airline measurements were supplied, so no performance belief or configuration decision changes; trajectory and service-feature evidence remain as previously recorded."
    },
    {
      "revision": "907f5578c6d792c55f892817",
      "previous_revision": "1e14e2965c0996663e047b10",
      "updated_at": "2026-10-04T13:06:22+00:00",
      "claim_count": 25,
      "counts": {
        "supported": 18,
        "untested": 3,
        "mixed": 1,
        "refuted": 3
      },
      "changes": [
        {
          "claim": "n007-shrinkage-trajectory-measurements",
          "before": null,
          "after": "supported",
          "reason": "Added the n007 quick-arm scores and the designated 1600/.02 search-validation score and runtime. The exact quick adoption prediction failed: 1600/.02 lost to 400/.08 but beat 400/.02. This does not refute other trajectories, establish overfitting, or show a validation improvement. Existing service-feature counterevidence and final holdout status are unchanged.",
          "supersedes": ""
        },
        {
          "claim": "n007-compensated-low-rate-adoption",
          "before": null,
          "after": "refuted",
          "reason": "Added the n007 quick-arm scores and the designated 1600/.02 search-validation score and runtime. The exact quick adoption prediction failed: 1600/.02 lost to 400/.08 but beat 400/.02. This does not refute other trajectories, establish overfitting, or show a validation improvement. Existing service-feature counterevidence and final holdout status are unchanged.",
          "supersedes": ""
        }
      ],
      "what_changed": "Added the n007 quick-arm scores and the designated 1600/.02 search-validation score and runtime. The exact quick adoption prediction failed: 1600/.02 lost to 400/.08 but beat 400/.02. This does not refute other trajectories, establish overfitting, or show a validation improvement. Existing service-feature counterevidence and final holdout status are unchanged."
    },
    {
      "revision": "1e14e2965c0996663e047b10",
      "previous_revision": "de674d430e1981ee1ae70264",
      "updated_at": "2026-10-04T12:52:13+00:00",
      "claim_count": 23,
      "counts": {
        "supported": 17,
        "untested": 3,
        "mixed": 1,
        "refuted": 2
      },
      "changes": [
        {
          "claim": "catboost-parameter-roles-documentation",
          "before": null,
          "after": "supported",
          "reason": "Added scoped documentation experience about CatBoost parameter descriptions and an untested trajectory hypothesis. No measured experimental claim, configuration decision, or feature default changed. Existing service-feature counterevidence remains intact; the supplied evidence does not justify asserting overfitting or a trajectory mechanism.",
          "supersedes": ""
        },
        {
          "claim": "learning-rate-iteration-trajectory-airline",
          "before": null,
          "after": "untested",
          "reason": "Added scoped documentation experience about CatBoost parameter descriptions and an untested trajectory hypothesis. No measured experimental claim, configuration decision, or feature default changed. Existing service-feature counterevidence remains intact; the supplied evidence does not justify asserting overfitting or a trajectory mechanism.",
          "supersedes": ""
        }
      ],
      "what_changed": "Added scoped documentation experience about CatBoost parameter descriptions and an untested trajectory hypothesis. No measured experimental claim, configuration decision, or feature default changed. Existing service-feature counterevidence remains intact; the supplied evidence does not justify asserting overfitting or a trajectory mechanism."
    },
    {
      "revision": "de674d430e1981ee1ae70264",
      "previous_revision": "176772dbe33d4c1b1e258b84",
      "updated_at": "2026-10-04T11:42:51+00:00",
      "claim_count": 21,
      "counts": {
        "supported": 16,
        "untested": 2,
        "mixed": 1,
        "refuted": 2
      },
      "changes": [
        {
          "claim": "catboost-gpu-nondeterminism-documentation",
          "before": null,
          "after": "supported",
          "reason": "Added scoped documentation experience: CatBoost describes GPU training as nondeterministic and tree construction as greedy feature-split selection. Neither excerpt supplies task-specific measurements or establishes a mechanism. No experimental claim or feature decision changes; the mean remains the default, the count has not met its adoption rule, and final remains sealed.",
          "supersedes": ""
        },
        {
          "claim": "greedy-split-selection-mechanism-context",
          "before": null,
          "after": "supported",
          "reason": "Added scoped documentation experience: CatBoost describes GPU training as nondeterministic and tree construction as greedy feature-split selection. Neither excerpt supplies task-specific measurements or establishes a mechanism. No experimental claim or feature decision changes; the mean remains the default, the count has not met its adoption rule, and final remains sealed.",
          "supersedes": ""
        }
      ],
      "what_changed": "Added scoped documentation experience: CatBoost describes GPU training as nondeterministic and tree construction as greedy feature-split selection. Neither excerpt supplies task-specific measurements or establishes a mechanism. No experimental claim or feature decision changes; the mean remains the default, the count has not met its adoption rule, and final remains sealed."
    },
    {
      "revision": "176772dbe33d4c1b1e258b84",
      "previous_revision": "c30165901b186b9bc04fce78",
      "updated_at": "2026-10-04T11:37:14+00:00",
      "claim_count": 19,
      "counts": {
        "supported": 14,
        "untested": 2,
        "mixed": 1,
        "refuted": 2
      },
      "changes": [
        {
          "claim": "n006-mean-validation-measurement",
          "before": null,
          "after": "supported",
          "reason": "n006 adds a controlled quick observation: both mean-plus-low-count scores were about 0.000196 below the intervening mean-only control, so the exact augmentation did not meet its +0.0002 adoption rule and the mean remains the default. Official validation evaluated mean only; its score and runtime update the mean measurement record, not evidence for the count. No mechanism or general count effect is established.",
          "supersedes": ""
        },
        {
          "claim": "n006-low-count-quick-observation",
          "before": null,
          "after": "supported",
          "reason": "n006 adds a controlled quick observation: both mean-plus-low-count scores were about 0.000196 below the intervening mean-only control, so the exact augmentation did not meet its +0.0002 adoption rule and the mean remains the default. Official validation evaluated mean only; its score and runtime update the mean measurement record, not evidence for the count. No mechanism or general count effect is established.",
          "supersedes": ""
        },
        {
          "claim": "n006-count-adoption-hypothesis",
          "before": null,
          "after": "refuted",
          "reason": "n006 adds a controlled quick observation: both mean-plus-low-count scores were about 0.000196 below the intervening mean-only control, so the exact augmentation did not meet its +0.0002 adoption rule and the mean remains the default. Official validation evaluated mean only; its score and runtime update the mean measurement record, not evidence for the count. No mechanism or general count effect is established.",
          "supersedes": ""
        }
      ],
      "what_changed": "n006 adds a controlled quick observation: both mean-plus-low-count scores were about 0.000196 below the intervening mean-only control, so the exact augmentation did not meet its +0.0002 adoption rule and the mean remains the default. Official validation evaluated mean only; its score and runtime update the mean measurement record, not evidence for the count. No mechanism or general count effect is established."
    },
    {
      "revision": "c30165901b186b9bc04fce78",
      "previous_revision": "b4734460d1fc5eb09e84c123",
      "updated_at": "2026-10-04T11:22:25+00:00",
      "claim_count": 16,
      "counts": {
        "supported": 12,
        "untested": 2,
        "mixed": 1,
        "refuted": 1
      },
      "changes": [
        {
          "claim": "catboost-greedy-split-competition-hypothesis",
          "before": null,
          "after": "untested",
          "reason": "Added algorithmic context that candidate split competition is a plausible alternative explanation for feature-mode differences, and documentation that GPU training can be nondeterministic. Neither source is airline performance evidence or quantifies task-specific noise, so no feature-default or mechanism belief changes.",
          "supersedes": ""
        },
        {
          "claim": "catboost-gpu-nondeterminism-magnitude-unknown",
          "before": null,
          "after": "supported",
          "reason": "Added algorithmic context that candidate split competition is a plausible alternative explanation for feature-mode differences, and documentation that GPU training can be nondeterministic. Neither source is airline performance evidence or quantifies task-specific noise, so no feature-default or mechanism belief changes.",
          "supersedes": ""
        }
      ],
      "what_changed": "Added algorithmic context that candidate split competition is a plausible alternative explanation for feature-mode differences, and documentation that GPU training can be nondeterministic. Neither source is airline performance evidence or quantifies task-specific noise, so no feature-default or mechanism belief changes."
    },
    {
      "revision": "b4734460d1fc5eb09e84c123",
      "previous_revision": "d7be0e335186df89e9357f27",
      "updated_at": "2026-10-04T11:16:45+00:00",
      "claim_count": 14,
      "counts": {
        "supported": 11,
        "untested": 1,
        "mixed": 1,
        "refuted": 1
      },
      "changes": [
        {
          "claim": "digital-gates-may-help-fixed-capacity-model",
          "before": "untested",
          "after": "mixed",
          "reason": "The gate bundle's quick-score ordering and differences repeated closely in reversed order, but its baseline increment remained far below the predeclared engineering adoption margin; the delivered default was mean-only. The new validation result is for mean-only and supplies no matched gate or raw control. No mechanism claim is resolved.",
          "supersedes": ""
        },
        {
          "claim": "n003-digital-gates-quick-ablation",
          "before": "supported",
          "after": "supported",
          "reason": "The gate bundle's quick-score ordering and differences repeated closely in reversed order, but its baseline increment remained far below the predeclared engineering adoption margin; the delivered default was mean-only. The new validation result is for mean-only and supplies no matched gate or raw control. No mechanism claim is resolved.",
          "supersedes": ""
        },
        {
          "claim": "n005-mean-validation-measurement",
          "before": null,
          "after": "supported",
          "reason": "The gate bundle's quick-score ordering and differences repeated closely in reversed order, but its baseline increment remained far below the predeclared engineering adoption margin; the delivered default was mean-only. The new validation result is for mean-only and supplies no matched gate or raw control. No mechanism claim is resolved.",
          "supersedes": ""
        }
      ],
      "what_changed": "The gate bundle's quick-score ordering and differences repeated closely in reversed order, but its baseline increment remained far below the predeclared engineering adoption margin; the delivered default was mean-only. The new validation result is for mean-only and supplies no matched gate or raw control. No mechanism claim is resolved."
    },
    {
      "revision": "d7be0e335186df89e9357f27",
      "previous_revision": "f2b38d282c6daca7153209e6",
      "updated_at": "2026-10-04T11:01:33+00:00",
      "claim_count": 13,
      "counts": {
        "supported": 10,
        "untested": 2,
        "refuted": 1
      },
      "changes": [],
      "what_changed": "No claim revisions: the new packets describe CatBoost GPU nondeterminism and greedy tree construction/parameters, but report no task-specific experiment measurements. They do not resolve the existing feature or round-count questions."
    },
    {
      "revision": "f2b38d282c6daca7153209e6",
      "previous_revision": "01c63f0420b07d78d29ec5aa",
      "updated_at": "2026-10-04T10:55:37+00:00",
      "claim_count": 13,
      "counts": {
        "supported": 10,
        "untested": 2,
        "refuted": 1
      },
      "changes": [
        {
          "claim": "n002-mean-validation-measurement",
          "before": "supported",
          "after": "supported",
          "reason": "Added n004 evidence that both tested 1600-round quick arms scored below mean/400; the proposed longer-round improvement was not supported within that quick scope. Validation subsequently measured restored mean/400 only, so it does not resolve 1600-round performance. The n004 mean/400 validation score is close to the earlier n002 measurement and does not establish a feature effect. Gate utility remains unresolved, and final remains sealed.",
          "supersedes": ""
        },
        {
          "claim": "n004-fixed1600-capacity-quick-outcome",
          "before": null,
          "after": "supported",
          "reason": "Added n004 evidence that both tested 1600-round quick arms scored below mean/400; the proposed longer-round improvement was not supported within that quick scope. Validation subsequently measured restored mean/400 only, so it does not resolve 1600-round performance. The n004 mean/400 validation score is close to the earlier n002 measurement and does not establish a feature effect. Gate utility remains unresolved, and final remains sealed.",
          "supersedes": ""
        },
        {
          "claim": "n004-fixed1600-improves-capacity-limited-ensemble",
          "before": null,
          "after": "refuted",
          "reason": "Added n004 evidence that both tested 1600-round quick arms scored below mean/400; the proposed longer-round improvement was not supported within that quick scope. Validation subsequently measured restored mean/400 only, so it does not resolve 1600-round performance. The n004 mean/400 validation score is close to the earlier n002 measurement and does not establish a feature effect. Gate utility remains unresolved, and final remains sealed.",
          "supersedes": ""
        }
      ],
      "what_changed": "Added n004 evidence that both tested 1600-round quick arms scored below mean/400; the proposed longer-round improvement was not supported within that quick scope. Validation subsequently measured restored mean/400 only, so it does not resolve 1600-round performance. The n004 mean/400 validation score is close to the earlier n002 measurement and does not establish a feature effect. Gate utility remains unresolved, and final remains sealed."
    },
    {
      "revision": "01c63f0420b07d78d29ec5aa",
      "previous_revision": "81087d017358807e1eff04d6",
      "updated_at": "2026-10-04T10:43:03+00:00",
      "claim_count": 11,
      "counts": {
        "supported": 9,
        "untested": 2
      },
      "changes": [
        {
          "claim": "catboost-gpu-nondeterminism",
          "before": "supported",
          "after": "supported",
          "reason": "Added CatBoost documentation context on GPU nondeterminism and tree-growing policy. No new task-specific measurements were supplied, so feature-utility beliefs and experimental decisions do not change; final remains sealed.",
          "supersedes": ""
        },
        {
          "claim": "catboost-greedy-tree-structure-docs",
          "before": "supported",
          "after": "supported",
          "reason": "Added CatBoost documentation context on GPU nondeterminism and tree-growing policy. No new task-specific measurements were supplied, so feature-utility beliefs and experimental decisions do not change; final remains sealed.",
          "supersedes": ""
        }
      ],
      "what_changed": "Added CatBoost documentation context on GPU nondeterminism and tree-growing policy. No new task-specific measurements were supplied, so feature-utility beliefs and experimental decisions do not change; final remains sealed."
    },
    {
      "revision": "81087d017358807e1eff04d6",
      "previous_revision": "1a15eda1c2e3af2d840c7826",
      "updated_at": "2026-10-04T10:36:49+00:00",
      "claim_count": 11,
      "counts": {
        "supported": 9,
        "untested": 2
      },
      "changes": [
        {
          "claim": "n003-digital-segments-validation-measurement",
          "before": null,
          "after": "supported",
          "reason": "Added the n003 validation measurement and quick A/B/C observations. The quick ungated digital arm scored below the overall-mean arm, while the gated arm was only slightly above it; the gated validation score exceeded the historical mean-only score by a small unmatched amount. These results motivate paired repeats, not a claim of robust utility. No mechanism was identified, and final remains sealed.",
          "supersedes": ""
        },
        {
          "claim": "n003-digital-gates-quick-ablation",
          "before": null,
          "after": "supported",
          "reason": "Added the n003 validation measurement and quick A/B/C observations. The quick ungated digital arm scored below the overall-mean arm, while the gated arm was only slightly above it; the gated validation score exceeded the historical mean-only score by a small unmatched amount. These results motivate paired repeats, not a claim of robust utility. No mechanism was identified, and final remains sealed.",
          "supersedes": ""
        },
        {
          "claim": "digital-gates-may-help-fixed-capacity-model",
          "before": null,
          "after": "untested",
          "reason": "Added the n003 validation measurement and quick A/B/C observations. The quick ungated digital arm scored below the overall-mean arm, while the gated arm was only slightly above it; the gated validation score exceeded the historical mean-only score by a small unmatched amount. These results motivate paired repeats, not a claim of robust utility. No mechanism was identified, and final remains sealed.",
          "supersedes": ""
        },
        {
          "claim": "ungated-digital-mean-not-beneficial-in-n003-quick-block",
          "before": null,
          "after": "supported",
          "reason": "Added the n003 validation measurement and quick A/B/C observations. The quick ungated digital arm scored below the overall-mean arm, while the gated arm was only slightly above it; the gated validation score exceeded the historical mean-only score by a small unmatched amount. These results motivate paired repeats, not a claim of robust utility. No mechanism was identified, and final remains sealed.",
          "supersedes": ""
        }
      ],
      "what_changed": "Added the n003 validation measurement and quick A/B/C observations. The quick ungated digital arm scored below the overall-mean arm, while the gated arm was only slightly above it; the gated validation score exceeded the historical mean-only score by a small unmatched amount. These results motivate paired repeats, not a claim of robust utility. No mechanism was identified, and final remains sealed."
    },
    {
      "revision": "1a15eda1c2e3af2d840c7826",
      "previous_revision": "e305299afc532e46fbf0f830",
      "updated_at": "2026-10-04T10:22:08+00:00",
      "claim_count": 7,
      "counts": {
        "supported": 6,
        "untested": 1
      },
      "changes": [
        {
          "claim": "catboost-gpu-nondeterminism",
          "before": "supported",
          "after": "supported",
          "reason": "No new airline measurements or ablations were supplied, so no empirical feature-effect belief or model decision changes. The documentation claims include the new excerpts: greedy tree construction and GPU nondeterminism are design context, not causal evidence for observed feature scores. Final remains sealed.",
          "supersedes": ""
        },
        {
          "claim": "catboost-greedy-tree-structure-docs",
          "before": "supported",
          "after": "supported",
          "reason": "No new airline measurements or ablations were supplied, so no empirical feature-effect belief or model decision changes. The documentation claims include the new excerpts: greedy tree construction and GPU nondeterminism are design context, not causal evidence for observed feature scores. Final remains sealed.",
          "supersedes": ""
        }
      ],
      "what_changed": "No new airline measurements or ablations were supplied, so no empirical feature-effect belief or model decision changes. The documentation claims include the new excerpts: greedy tree construction and GPU nondeterminism are design context, not causal evidence for observed feature scores. Final remains sealed."
    },
    {
      "revision": "e305299afc532e46fbf0f830",
      "previous_revision": "85943388f116298685f53030",
      "updated_at": "2026-10-04T10:14:59+00:00",
      "claim_count": 7,
      "counts": {
        "supported": 6,
        "untested": 1
      },
      "changes": [
        {
          "claim": "quick-rating-summary-ablation",
          "before": "supported",
          "after": "supported",
          "reason": "The quick mean-over-raw and mean-over-profile ranking now appears in two same-seed blocks with similar mean-over-raw differences, strengthening directional repeatability only on that quick split. The mean arm also has a valid search-validation score, +0.0000671 versus a separate root measurement, but the comparison is unmatched and uncertainty is unknown, so no reliable validation improvement is established. The tested full profile scored below raw in both quick blocks; this remains bundle-specific and not a causal feature diagnosis. Final remains sealed.",
          "supersedes": ""
        },
        {
          "claim": "rating-summary-split-hypothesis",
          "before": "untested",
          "after": "untested",
          "reason": "The quick mean-over-raw and mean-over-profile ranking now appears in two same-seed blocks with similar mean-over-raw differences, strengthening directional repeatability only on that quick split. The mean arm also has a valid search-validation score, +0.0000671 versus a separate root measurement, but the comparison is unmatched and uncertainty is unknown, so no reliable validation improvement is established. The tested full profile scored below raw in both quick blocks; this remains bundle-specific and not a causal feature diagnosis. Final remains sealed.",
          "supersedes": ""
        },
        {
          "claim": "catboost-greedy-tree-structure-docs",
          "before": "supported",
          "after": "supported",
          "reason": "The quick mean-over-raw and mean-over-profile ranking now appears in two same-seed blocks with similar mean-over-raw differences, strengthening directional repeatability only on that quick split. The mean arm also has a valid search-validation score, +0.0000671 versus a separate root measurement, but the comparison is unmatched and uncertainty is unknown, so no reliable validation improvement is established. The tested full profile scored below raw in both quick blocks; this remains bundle-specific and not a causal feature diagnosis. Final remains sealed.",
          "supersedes": ""
        },
        {
          "claim": "full-profile-versus-raw-quick",
          "before": "supported",
          "after": "supported",
          "reason": "The quick mean-over-raw and mean-over-profile ranking now appears in two same-seed blocks with similar mean-over-raw differences, strengthening directional repeatability only on that quick split. The mean arm also has a valid search-validation score, +0.0000671 versus a separate root measurement, but the comparison is unmatched and uncertainty is unknown, so no reliable validation improvement is established. The tested full profile scored below raw in both quick blocks; this remains bundle-specific and not a causal feature diagnosis. Final remains sealed.",
          "supersedes": ""
        },
        {
          "claim": "starter-validation-measurement",
          "before": "supported",
          "after": "supported",
          "reason": "The quick mean-over-raw and mean-over-profile ranking now appears in two same-seed blocks with similar mean-over-raw differences, strengthening directional repeatability only on that quick split. The mean arm also has a valid search-validation score, +0.0000671 versus a separate root measurement, but the comparison is unmatched and uncertainty is unknown, so no reliable validation improvement is established. The tested full profile scored below raw in both quick blocks; this remains bundle-specific and not a causal feature diagnosis. Final remains sealed.",
          "supersedes": ""
        },
        {
          "claim": "n002-mean-validation-measurement",
          "before": null,
          "after": "supported",
          "reason": "The quick mean-over-raw and mean-over-profile ranking now appears in two same-seed blocks with similar mean-over-raw differences, strengthening directional repeatability only on that quick split. The mean arm also has a valid search-validation score, +0.0000671 versus a separate root measurement, but the comparison is unmatched and uncertainty is unknown, so no reliable validation improvement is established. The tested full profile scored below raw in both quick blocks; this remains bundle-specific and not a causal feature diagnosis. Final remains sealed.",
          "supersedes": ""
        }
      ],
      "what_changed": "The quick mean-over-raw and mean-over-profile ranking now appears in two same-seed blocks with similar mean-over-raw differences, strengthening directional repeatability only on that quick split. The mean arm also has a valid search-validation score, +0.0000671 versus a separate root measurement, but the comparison is unmatched and uncertainty is unknown, so no reliable validation improvement is established. The tested full profile scored below raw in both quick blocks; this remains bundle-specific and not a causal feature diagnosis. Final remains sealed."
    },
    {
      "revision": "85943388f116298685f53030",
      "previous_revision": "8e8da7cec6eaf76f6d8462b3",
      "updated_at": "2026-10-04T10:02:03+00:00",
      "claim_count": 6,
      "counts": {
        "supported": 5,
        "untested": 1
      },
      "changes": [
        {
          "claim": "catboost-gpu-nondeterminism",
          "before": "supported",
          "after": "supported",
          "reason": "Added a fresh CatBoost GPU nondeterminism documentation reference to the existing experience claim and recorded documentation context about greedy split selection. The fresh retrieval is not independent replication. No empirical score belief or feature-selection decision changed.",
          "supersedes": ""
        },
        {
          "claim": "catboost-greedy-tree-structure-docs",
          "before": null,
          "after": "supported",
          "reason": "Added a fresh CatBoost GPU nondeterminism documentation reference to the existing experience claim and recorded documentation context about greedy split selection. The fresh retrieval is not independent replication. No empirical score belief or feature-selection decision changed.",
          "supersedes": ""
        }
      ],
      "what_changed": "Added a fresh CatBoost GPU nondeterminism documentation reference to the existing experience claim and recorded documentation context about greedy split selection. The fresh retrieval is not independent replication. No empirical score belief or feature-selection decision changed."
    },
    {
      "revision": "8e8da7cec6eaf76f6d8462b3",
      "previous_revision": "22ec584040f3649d3c3f6e7e",
      "updated_at": "2026-10-04T09:55:32+00:00",
      "claim_count": 5,
      "counts": {
        "supported": 4,
        "untested": 1
      },
      "changes": [
        {
          "claim": "rating-summary-split-hypothesis",
          "before": "untested",
          "after": "untested",
          "reason": "The supplied quick results show mean-only above raw and the full profile slightly below raw in one run each. The worker reports retaining mean-only as the default, but this is selection evidence only, not repeatable support. The validation measurement is unavailable because model integrity failed; this is a process failure, not evidence against the feature hypothesis. No claim about validation performance or mechanism changed.",
          "supersedes": ""
        },
        {
          "claim": "quick-rating-summary-ablation",
          "before": null,
          "after": "supported",
          "reason": "The supplied quick results show mean-only above raw and the full profile slightly below raw in one run each. The worker reports retaining mean-only as the default, but this is selection evidence only, not repeatable support. The validation measurement is unavailable because model integrity failed; this is a process failure, not evidence against the feature hypothesis. No claim about validation performance or mechanism changed.",
          "supersedes": ""
        },
        {
          "claim": "full-profile-versus-raw-quick",
          "before": null,
          "after": "supported",
          "reason": "The supplied quick results show mean-only above raw and the full profile slightly below raw in one run each. The worker reports retaining mean-only as the default, but this is selection evidence only, not repeatable support. The validation measurement is unavailable because model integrity failed; this is a process failure, not evidence against the feature hypothesis. No claim about validation performance or mechanism changed.",
          "supersedes": ""
        }
      ],
      "what_changed": "The supplied quick results show mean-only above raw and the full profile slightly below raw in one run each. The worker reports retaining mean-only as the default, but this is selection evidence only, not repeatable support. The validation measurement is unavailable because model integrity failed; this is a process failure, not evidence against the feature hypothesis. No claim about validation performance or mechanism changed."
    },
    {
      "revision": "22ec584040f3649d3c3f6e7e",
      "previous_revision": "9c2a8482717f5d37bbd167fa",
      "updated_at": "2026-10-04T09:47:49+00:00",
      "claim_count": 3,
      "counts": {
        "supported": 2,
        "untested": 1
      },
      "changes": [
        {
          "claim": "catboost-gpu-nondeterminism",
          "before": null,
          "after": "supported",
          "reason": "Added supplied documentation context that CatBoost GPU training may be nondeterministic and tree construction greedily selects splits. Neither excerpt reports an airline experiment, so the feature-benefit question remains untested; no performance belief changed.",
          "supersedes": ""
        },
        {
          "claim": "rating-summary-split-hypothesis",
          "before": null,
          "after": "untested",
          "reason": "Added supplied documentation context that CatBoost GPU training may be nondeterministic and tree construction greedily selects splits. Neither excerpt reports an airline experiment, so the feature-benefit question remains untested; no performance belief changed.",
          "supersedes": ""
        }
      ],
      "what_changed": "Added supplied documentation context that CatBoost GPU training may be nondeterministic and tree construction greedily selects splits. Neither excerpt reports an airline experiment, so the feature-benefit question remains untested; no performance belief changed."
    },
    {
      "revision": "9c2a8482717f5d37bbd167fa",
      "previous_revision": "",
      "updated_at": "2026-10-04T09:42:10+00:00",
      "claim_count": 1,
      "counts": {
        "supported": 1
      },
      "changes": [
        {
          "claim": "starter-validation-measurement",
          "before": null,
          "after": "supported",
          "reason": "Recorded one measured untouched-starter validation result (ROC-AUC 0.9572004065863107; candidate elapsed 4.7448 s). No feature intervention, ablation, replication, or mechanism test is present, so no belief about feature benefits changed.",
          "supersedes": ""
        }
      ],
      "what_changed": "Recorded one measured untouched-starter validation result (ROC-AUC 0.9572004065863107; candidate elapsed 4.7448 s). No feature intervention, ablation, replication, or mechanism test is present, so no belief about feature benefits changed."
    }
  ],
  "counts": {
    "supported": 41,
    "untested": 4,
    "mixed": 7,
    "refuted": 7,
    "superseded": 1
  },
  "kind_counts": {
    "observation": 25,
    "experience": 14,
    "hypothesis": 21
  },
  "evidence_counts": {
    "source": 43,
    "experiment": 21
  },
  "provenance_note": "Evidence index contains aggregate metadata and hashes, not raw packets, dataset rows or model-session logs. Multiple entries can describe the same experiment/source."
}
