{
  "blocks": [
    {
      "boundaries": "The method and normalization axes distinguish library-size scaling, TPM and LayerNorm; log transforms are separate calls.",
      "definition": "Transform numerical values using a declared scaling or statistical normalization rule.",
      "inputs": [
        {
          "meaning": "Documented input data operands.",
          "name": "values",
          "optional": false,
          "variadic": true
        }
      ],
      "label": "Normalization",
      "official_reference": "https://scikit-learn.org/stable/modules/preprocessing.html",
      "operation_id": "normalize",
      "outputs": [
        {
          "meaning": "Produced output data operands.",
          "name": "values",
          "optional": false,
          "variadic": true
        }
      ],
      "parameters": [
        "method",
        "axes",
        "scope"
      ],
      "reuse_record_ids": [
        "full_2026-07-06__rec_003517",
        "full_2026-07-06__rec_003852"
      ],
      "status": "source_backed",
      "supplemental_examples": [],
      "usage": [
        {
          "call_id": "normalize::1",
          "component": "RNA preprocessing",
          "condition": null,
          "evidence_ids": [
            "ed0cbc9219226da12cb74868c945901c59c51a869e0593f5bfcd24dc08383722"
          ],
          "inputs": {
            "values": [
              "rna_counts"
            ]
          },
          "outputs": {
            "values": [
              "rna_tpm"
            ]
          },
          "parameters": {
            "method": "TPM",
            "scope": "source-defined expression axes",
            "source_port_roles": [
              "measured_expression"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "normalize",
          "trajectory_id": "pretrain_multimodal_dca"
        },
        {
          "call_id": "esm_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "esm_loaded"
            ]
          },
          "outputs": {
            "values": [
              "esm_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_availability",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "esm_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "esm_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "esm_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_project",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "genept_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "genept_loaded"
            ]
          },
          "outputs": {
            "values": [
              "genept_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_availability",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "genept_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "genept_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "genept_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_project",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "string_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "string_loaded"
            ]
          },
          "outputs": {
            "values": [
              "string_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_availability",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "string_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "string_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "string_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_project",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "depmap_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "depmap_loaded"
            ]
          },
          "outputs": {
            "values": [
              "depmap_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_availability",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "depmap_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "depmap_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "depmap_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_project",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "cp_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "cp_loaded"
            ]
          },
          "outputs": {
            "values": [
              "cp_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_availability",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "cp_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "cp_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "cp_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_project",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "control_normalization::1",
          "component": "Expression preprocessing",
          "condition": "input dataset contains raw counts requiring preprocessing",
          "evidence_ids": [
            "2a372343d3eae52243f5266cdc612ce3e829405b347fedc2a492379f56290e1d"
          ],
          "inputs": {
            "values": [
              "control_pool"
            ]
          },
          "outputs": {
            "values": [
              "control_normalization::scaled_expression"
            ]
          },
          "parameters": {
            "method": "CP10K",
            "source_port_roles": [
              "measured counts or pre-normalized values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "control_normalization",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "target_normalization::1",
          "component": "Expression preprocessing",
          "condition": "input dataset contains raw counts requiring preprocessing",
          "evidence_ids": [
            "2a372343d3eae52243f5266cdc612ce3e829405b347fedc2a492379f56290e1d"
          ],
          "inputs": {
            "values": [
              "target_pool"
            ]
          },
          "outputs": {
            "values": [
              "target_normalization::scaled_expression"
            ]
          },
          "parameters": {
            "method": "CP10K",
            "source_port_roles": [
              "measured counts or pre-normalized values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "target_normalization",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "identity_encoder::2",
          "component": "Gene identity encoder",
          "condition": null,
          "evidence_ids": [
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0"
          ],
          "inputs": {
            "values": [
              "identity_encoder::looked_up_vectors"
            ]
          },
          "outputs": {
            "values": [
              "identity_encoded"
            ]
          },
          "parameters": {
            "axes": "source-defined feature axes",
            "method": "LayerNorm",
            "source_port_roles": [
              "gene indices",
              "shared raw embedding-table parameters; learned weights, not measured sample data"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "identity_encoder",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "value_encoder::2",
          "component": "Continuous value encoder",
          "condition": null,
          "evidence_ids": [
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0"
          ],
          "inputs": {
            "values": [
              "value_encoder::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "value_encoded"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "current expression values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "value_encoder",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "mask_encoder::2",
          "component": "Perturbation mask encoder",
          "condition": null,
          "evidence_ids": [
            "d4015be6a525bdcb39076445d9d4b72f2bc20449673edb27d686b896a0d20d0a"
          ],
          "inputs": {
            "values": [
              "mask_encoder::looked_up_vectors"
            ]
          },
          "outputs": {
            "values": [
              "mask_encoded"
            ]
          },
          "parameters": {
            "axes": "source-defined feature axes",
            "method": "LayerNorm",
            "source_port_roles": [
              "per-position reveal mask"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "mask_encoder",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "esm_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "esm_loaded"
            ]
          },
          "outputs": {
            "values": [
              "esm_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_availability",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "esm_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "esm_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "esm_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_project",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "genept_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "genept_loaded"
            ]
          },
          "outputs": {
            "values": [
              "genept_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_availability",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "genept_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "genept_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "genept_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_project",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "string_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "string_loaded"
            ]
          },
          "outputs": {
            "values": [
              "string_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_availability",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "string_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "string_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "string_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_project",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "depmap_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "depmap_loaded"
            ]
          },
          "outputs": {
            "values": [
              "depmap_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_availability",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "depmap_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "depmap_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "depmap_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_project",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "cp_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "cp_loaded"
            ]
          },
          "outputs": {
            "values": [
              "cp_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_availability",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "cp_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "cp_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "cp_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_project",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "control_normalization::1",
          "component": "Expression preprocessing",
          "condition": "input dataset contains raw counts requiring preprocessing",
          "evidence_ids": [
            "2a372343d3eae52243f5266cdc612ce3e829405b347fedc2a492379f56290e1d"
          ],
          "inputs": {
            "values": [
              "control_pool"
            ]
          },
          "outputs": {
            "values": [
              "control_normalization::scaled_expression"
            ]
          },
          "parameters": {
            "method": "CP10K",
            "source_port_roles": [
              "measured counts or pre-normalized values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "control_normalization",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "target_normalization::1",
          "component": "Expression preprocessing",
          "condition": "input dataset contains raw counts requiring preprocessing",
          "evidence_ids": [
            "2a372343d3eae52243f5266cdc612ce3e829405b347fedc2a492379f56290e1d"
          ],
          "inputs": {
            "values": [
              "target_pool"
            ]
          },
          "outputs": {
            "values": [
              "target_normalization::scaled_expression"
            ]
          },
          "parameters": {
            "method": "CP10K",
            "source_port_roles": [
              "measured counts or pre-normalized values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "target_normalization",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "identity_encoder::2",
          "component": "Gene identity encoder",
          "condition": null,
          "evidence_ids": [
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0"
          ],
          "inputs": {
            "values": [
              "identity_encoder::looked_up_vectors"
            ]
          },
          "outputs": {
            "values": [
              "identity_encoded"
            ]
          },
          "parameters": {
            "axes": "source-defined feature axes",
            "method": "LayerNorm",
            "source_port_roles": [
              "gene indices",
              "shared raw embedding-table parameters; learned weights, not measured sample data"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "identity_encoder",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "value_encoder::2",
          "component": "Continuous value encoder",
          "condition": null,
          "evidence_ids": [
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0"
          ],
          "inputs": {
            "values": [
              "value_encoder::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "value_encoded"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "current expression values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "value_encoder",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "mask_encoder::2",
          "component": "Perturbation mask encoder",
          "condition": null,
          "evidence_ids": [
            "d4015be6a525bdcb39076445d9d4b72f2bc20449673edb27d686b896a0d20d0a"
          ],
          "inputs": {
            "values": [
              "mask_encoder::looked_up_vectors"
            ]
          },
          "outputs": {
            "values": [
              "mask_encoded"
            ]
          },
          "parameters": {
            "axes": "source-defined feature axes",
            "method": "LayerNorm",
            "source_port_roles": [
              "per-position reveal mask"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "mask_encoder",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "esm_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "esm_loaded"
            ]
          },
          "outputs": {
            "values": [
              "esm_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_availability",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "esm_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "esm_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "esm_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_project",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "genept_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "genept_loaded"
            ]
          },
          "outputs": {
            "values": [
              "genept_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_availability",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "genept_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "genept_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "genept_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_project",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "string_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "string_loaded"
            ]
          },
          "outputs": {
            "values": [
              "string_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_availability",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "string_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "string_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "string_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_project",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "depmap_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "depmap_loaded"
            ]
          },
          "outputs": {
            "values": [
              "depmap_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_availability",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "depmap_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "depmap_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "depmap_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_project",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "cp_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "cp_loaded"
            ]
          },
          "outputs": {
            "values": [
              "cp_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_availability",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "cp_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "cp_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "cp_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_project",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "control_normalization::1",
          "component": "Expression preprocessing",
          "condition": "input dataset contains raw counts requiring preprocessing",
          "evidence_ids": [
            "2a372343d3eae52243f5266cdc612ce3e829405b347fedc2a492379f56290e1d"
          ],
          "inputs": {
            "values": [
              "control_pool"
            ]
          },
          "outputs": {
            "values": [
              "control_normalization::scaled_expression"
            ]
          },
          "parameters": {
            "method": "CP10K",
            "source_port_roles": [
              "measured counts or pre-normalized values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "control_normalization",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "target_normalization::1",
          "component": "Expression preprocessing",
          "condition": "input dataset contains raw counts requiring preprocessing",
          "evidence_ids": [
            "2a372343d3eae52243f5266cdc612ce3e829405b347fedc2a492379f56290e1d"
          ],
          "inputs": {
            "values": [
              "target_pool"
            ]
          },
          "outputs": {
            "values": [
              "target_normalization::scaled_expression"
            ]
          },
          "parameters": {
            "method": "CP10K",
            "source_port_roles": [
              "measured counts or pre-normalized values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "target_normalization",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "identity_encoder::2",
          "component": "Gene identity encoder",
          "condition": null,
          "evidence_ids": [
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0"
          ],
          "inputs": {
            "values": [
              "identity_encoder::looked_up_vectors"
            ]
          },
          "outputs": {
            "values": [
              "identity_encoded"
            ]
          },
          "parameters": {
            "axes": "source-defined feature axes",
            "method": "LayerNorm",
            "source_port_roles": [
              "gene indices",
              "shared raw embedding-table parameters; learned weights, not measured sample data"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "identity_encoder",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "value_encoder::2",
          "component": "Continuous value encoder",
          "condition": null,
          "evidence_ids": [
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0"
          ],
          "inputs": {
            "values": [
              "value_encoder::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "value_encoded"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "current expression values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "value_encoder",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "mask_encoder::2",
          "component": "Perturbation mask encoder",
          "condition": null,
          "evidence_ids": [
            "d4015be6a525bdcb39076445d9d4b72f2bc20449673edb27d686b896a0d20d0a"
          ],
          "inputs": {
            "values": [
              "mask_encoder::looked_up_vectors"
            ]
          },
          "outputs": {
            "values": [
              "mask_encoded"
            ]
          },
          "parameters": {
            "axes": "source-defined feature axes",
            "method": "LayerNorm",
            "source_port_roles": [
              "per-position reveal mask"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "mask_encoder",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "esm_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "esm_loaded"
            ]
          },
          "outputs": {
            "values": [
              "esm_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_availability",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "esm_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "esm_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "esm_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_project",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "genept_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "genept_loaded"
            ]
          },
          "outputs": {
            "values": [
              "genept_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_availability",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "genept_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "genept_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "genept_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_project",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "string_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "string_loaded"
            ]
          },
          "outputs": {
            "values": [
              "string_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_availability",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "string_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "string_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "string_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_project",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "depmap_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "depmap_loaded"
            ]
          },
          "outputs": {
            "values": [
              "depmap_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_availability",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "depmap_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "depmap_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "depmap_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_project",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "cp_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "cp_loaded"
            ]
          },
          "outputs": {
            "values": [
              "cp_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_availability",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "cp_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "cp_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "cp_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_project",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "control_normalization::1",
          "component": "Expression preprocessing",
          "condition": "input dataset contains raw counts requiring preprocessing",
          "evidence_ids": [
            "2a372343d3eae52243f5266cdc612ce3e829405b347fedc2a492379f56290e1d"
          ],
          "inputs": {
            "values": [
              "control_pool"
            ]
          },
          "outputs": {
            "values": [
              "control_normalization::scaled_expression"
            ]
          },
          "parameters": {
            "method": "CP10K",
            "source_port_roles": [
              "measured counts or pre-normalized values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "control_normalization",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "target_normalization::1",
          "component": "Expression preprocessing",
          "condition": "input dataset contains raw counts requiring preprocessing",
          "evidence_ids": [
            "2a372343d3eae52243f5266cdc612ce3e829405b347fedc2a492379f56290e1d"
          ],
          "inputs": {
            "values": [
              "target_pool"
            ]
          },
          "outputs": {
            "values": [
              "target_normalization::scaled_expression"
            ]
          },
          "parameters": {
            "method": "CP10K",
            "source_port_roles": [
              "measured counts or pre-normalized values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "target_normalization",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "identity_encoder::2",
          "component": "Gene identity encoder",
          "condition": null,
          "evidence_ids": [
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0"
          ],
          "inputs": {
            "values": [
              "identity_encoder::looked_up_vectors"
            ]
          },
          "outputs": {
            "values": [
              "identity_encoded"
            ]
          },
          "parameters": {
            "axes": "source-defined feature axes",
            "method": "LayerNorm",
            "source_port_roles": [
              "gene indices",
              "shared raw embedding-table parameters; learned weights, not measured sample data"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "identity_encoder",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "value_encoder::2",
          "component": "Continuous value encoder",
          "condition": null,
          "evidence_ids": [
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0"
          ],
          "inputs": {
            "values": [
              "value_encoder::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "value_encoded"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "current expression values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "value_encoder",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "mask_encoder::2",
          "component": "Perturbation mask encoder",
          "condition": null,
          "evidence_ids": [
            "d4015be6a525bdcb39076445d9d4b72f2bc20449673edb27d686b896a0d20d0a"
          ],
          "inputs": {
            "values": [
              "mask_encoder::looked_up_vectors"
            ]
          },
          "outputs": {
            "values": [
              "mask_encoded"
            ]
          },
          "parameters": {
            "axes": "source-defined feature axes",
            "method": "LayerNorm",
            "source_port_roles": [
              "per-position reveal mask"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "mask_encoder",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "esm_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "esm_loaded"
            ]
          },
          "outputs": {
            "values": [
              "esm_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_availability",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "esm_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "esm_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "esm_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_project",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "genept_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "genept_loaded"
            ]
          },
          "outputs": {
            "values": [
              "genept_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_availability",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "genept_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "genept_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "genept_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_project",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "string_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "string_loaded"
            ]
          },
          "outputs": {
            "values": [
              "string_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_availability",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "string_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "string_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "string_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_project",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "depmap_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "depmap_loaded"
            ]
          },
          "outputs": {
            "values": [
              "depmap_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_availability",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "depmap_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "depmap_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "depmap_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_project",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "cp_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "cp_loaded"
            ]
          },
          "outputs": {
            "values": [
              "cp_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_availability",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "cp_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "cp_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "cp_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_project",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "control_normalization::1",
          "component": "Expression preprocessing",
          "condition": "input dataset contains raw counts requiring preprocessing",
          "evidence_ids": [
            "2a372343d3eae52243f5266cdc612ce3e829405b347fedc2a492379f56290e1d"
          ],
          "inputs": {
            "values": [
              "control_pool"
            ]
          },
          "outputs": {
            "values": [
              "control_normalization::scaled_expression"
            ]
          },
          "parameters": {
            "method": "CP10K",
            "source_port_roles": [
              "measured counts or pre-normalized values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "control_normalization",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "target_normalization::1",
          "component": "Expression preprocessing",
          "condition": "input dataset contains raw counts requiring preprocessing",
          "evidence_ids": [
            "2a372343d3eae52243f5266cdc612ce3e829405b347fedc2a492379f56290e1d"
          ],
          "inputs": {
            "values": [
              "target_pool"
            ]
          },
          "outputs": {
            "values": [
              "target_normalization::scaled_expression"
            ]
          },
          "parameters": {
            "method": "CP10K",
            "source_port_roles": [
              "measured counts or pre-normalized values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "target_normalization",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "identity_encoder::2",
          "component": "Gene identity encoder",
          "condition": null,
          "evidence_ids": [
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0"
          ],
          "inputs": {
            "values": [
              "identity_encoder::looked_up_vectors"
            ]
          },
          "outputs": {
            "values": [
              "identity_encoded"
            ]
          },
          "parameters": {
            "axes": "source-defined feature axes",
            "method": "LayerNorm",
            "source_port_roles": [
              "gene indices",
              "shared raw embedding-table parameters; learned weights, not measured sample data"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "identity_encoder",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "value_encoder::2",
          "component": "Continuous value encoder",
          "condition": null,
          "evidence_ids": [
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0"
          ],
          "inputs": {
            "values": [
              "value_encoder::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "value_encoded"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "current expression values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "value_encoder",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "mask_encoder::2",
          "component": "Perturbation mask encoder",
          "condition": null,
          "evidence_ids": [
            "d4015be6a525bdcb39076445d9d4b72f2bc20449673edb27d686b896a0d20d0a"
          ],
          "inputs": {
            "values": [
              "mask_encoder::looked_up_vectors"
            ]
          },
          "outputs": {
            "values": [
              "mask_encoded"
            ]
          },
          "parameters": {
            "axes": "source-defined feature axes",
            "method": "LayerNorm",
            "source_port_roles": [
              "per-position reveal mask"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "mask_encoder",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "esm_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "esm_loaded"
            ]
          },
          "outputs": {
            "values": [
              "esm_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_availability",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "esm_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "esm_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "esm_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_project",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "genept_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "genept_loaded"
            ]
          },
          "outputs": {
            "values": [
              "genept_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_availability",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "genept_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "genept_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "genept_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_project",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "string_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "string_loaded"
            ]
          },
          "outputs": {
            "values": [
              "string_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_availability",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "string_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "string_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "string_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_project",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "depmap_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "depmap_loaded"
            ]
          },
          "outputs": {
            "values": [
              "depmap_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_availability",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "depmap_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "depmap_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "depmap_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_project",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "cp_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "cp_loaded"
            ]
          },
          "outputs": {
            "values": [
              "cp_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_availability",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "cp_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "cp_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "cp_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_project",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "control_normalization::1",
          "component": "Expression preprocessing",
          "condition": "input dataset contains raw counts requiring preprocessing",
          "evidence_ids": [
            "2a372343d3eae52243f5266cdc612ce3e829405b347fedc2a492379f56290e1d"
          ],
          "inputs": {
            "values": [
              "control_pool"
            ]
          },
          "outputs": {
            "values": [
              "control_normalization::scaled_expression"
            ]
          },
          "parameters": {
            "method": "CP10K",
            "source_port_roles": [
              "measured counts or pre-normalized values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "control_normalization",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "target_normalization::1",
          "component": "Expression preprocessing",
          "condition": "input dataset contains raw counts requiring preprocessing",
          "evidence_ids": [
            "2a372343d3eae52243f5266cdc612ce3e829405b347fedc2a492379f56290e1d"
          ],
          "inputs": {
            "values": [
              "target_pool"
            ]
          },
          "outputs": {
            "values": [
              "target_normalization::scaled_expression"
            ]
          },
          "parameters": {
            "method": "CP10K",
            "source_port_roles": [
              "measured counts or pre-normalized values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "target_normalization",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "identity_encoder::2",
          "component": "Gene identity encoder",
          "condition": null,
          "evidence_ids": [
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0"
          ],
          "inputs": {
            "values": [
              "identity_encoder::looked_up_vectors"
            ]
          },
          "outputs": {
            "values": [
              "identity_encoded"
            ]
          },
          "parameters": {
            "axes": "source-defined feature axes",
            "method": "LayerNorm",
            "source_port_roles": [
              "gene indices",
              "shared raw embedding-table parameters; learned weights, not measured sample data"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "identity_encoder",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "value_encoder::2",
          "component": "Continuous value encoder",
          "condition": null,
          "evidence_ids": [
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0"
          ],
          "inputs": {
            "values": [
              "value_encoder::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "value_encoded"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "current expression values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "value_encoder",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "mask_encoder::2",
          "component": "Perturbation mask encoder",
          "condition": null,
          "evidence_ids": [
            "d4015be6a525bdcb39076445d9d4b72f2bc20449673edb27d686b896a0d20d0a"
          ],
          "inputs": {
            "values": [
              "mask_encoder::looked_up_vectors"
            ]
          },
          "outputs": {
            "values": [
              "mask_encoded"
            ]
          },
          "parameters": {
            "axes": "source-defined feature axes",
            "method": "LayerNorm",
            "source_port_roles": [
              "per-position reveal mask"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "mask_encoder",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "esm_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "esm_loaded"
            ]
          },
          "outputs": {
            "values": [
              "esm_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_availability",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "esm_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "esm_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "esm_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_project",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "genept_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "genept_loaded"
            ]
          },
          "outputs": {
            "values": [
              "genept_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_availability",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "genept_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "genept_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "genept_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_project",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "string_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "string_loaded"
            ]
          },
          "outputs": {
            "values": [
              "string_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_availability",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "string_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "string_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "string_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_project",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "depmap_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "depmap_loaded"
            ]
          },
          "outputs": {
            "values": [
              "depmap_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_availability",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "depmap_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "depmap_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "depmap_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_project",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "cp_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "cp_loaded"
            ]
          },
          "outputs": {
            "values": [
              "cp_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_availability",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "cp_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "cp_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "cp_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_project",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "control_normalization::1",
          "component": "Expression preprocessing",
          "condition": "input dataset contains raw counts requiring preprocessing",
          "evidence_ids": [
            "2a372343d3eae52243f5266cdc612ce3e829405b347fedc2a492379f56290e1d"
          ],
          "inputs": {
            "values": [
              "control_pool"
            ]
          },
          "outputs": {
            "values": [
              "control_normalization::scaled_expression"
            ]
          },
          "parameters": {
            "method": "CP10K",
            "source_port_roles": [
              "measured counts or pre-normalized values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "control_normalization",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "target_normalization::1",
          "component": "Expression preprocessing",
          "condition": "input dataset contains raw counts requiring preprocessing",
          "evidence_ids": [
            "2a372343d3eae52243f5266cdc612ce3e829405b347fedc2a492379f56290e1d"
          ],
          "inputs": {
            "values": [
              "target_pool"
            ]
          },
          "outputs": {
            "values": [
              "target_normalization::scaled_expression"
            ]
          },
          "parameters": {
            "method": "CP10K",
            "source_port_roles": [
              "measured counts or pre-normalized values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "target_normalization",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "identity_encoder::2",
          "component": "Gene identity encoder",
          "condition": null,
          "evidence_ids": [
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0"
          ],
          "inputs": {
            "values": [
              "identity_encoder::looked_up_vectors"
            ]
          },
          "outputs": {
            "values": [
              "identity_encoded"
            ]
          },
          "parameters": {
            "axes": "source-defined feature axes",
            "method": "LayerNorm",
            "source_port_roles": [
              "gene indices",
              "shared raw embedding-table parameters; learned weights, not measured sample data"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "identity_encoder",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "value_encoder::2",
          "component": "Continuous value encoder",
          "condition": null,
          "evidence_ids": [
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0"
          ],
          "inputs": {
            "values": [
              "value_encoder::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "value_encoded"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "current expression values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "value_encoder",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "mask_encoder::2",
          "component": "Perturbation mask encoder",
          "condition": null,
          "evidence_ids": [
            "d4015be6a525bdcb39076445d9d4b72f2bc20449673edb27d686b896a0d20d0a"
          ],
          "inputs": {
            "values": [
              "mask_encoder::looked_up_vectors"
            ]
          },
          "outputs": {
            "values": [
              "mask_encoded"
            ]
          },
          "parameters": {
            "axes": "source-defined feature axes",
            "method": "LayerNorm",
            "source_port_roles": [
              "per-position reveal mask"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "mask_encoder",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "esm_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "esm_loaded"
            ]
          },
          "outputs": {
            "values": [
              "esm_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_availability",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "esm_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "esm_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "esm_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_project",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "genept_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "genept_loaded"
            ]
          },
          "outputs": {
            "values": [
              "genept_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_availability",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "genept_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "genept_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "genept_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_project",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "string_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "string_loaded"
            ]
          },
          "outputs": {
            "values": [
              "string_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_availability",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "string_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "string_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "string_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_project",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "depmap_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "depmap_loaded"
            ]
          },
          "outputs": {
            "values": [
              "depmap_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_availability",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "depmap_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "depmap_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "depmap_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_project",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "cp_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "cp_loaded"
            ]
          },
          "outputs": {
            "values": [
              "cp_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_availability",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "cp_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "cp_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "cp_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_project",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "control_normalization::1",
          "component": "Expression preprocessing",
          "condition": "input dataset contains raw counts requiring preprocessing",
          "evidence_ids": [
            "2a372343d3eae52243f5266cdc612ce3e829405b347fedc2a492379f56290e1d"
          ],
          "inputs": {
            "values": [
              "control_pool"
            ]
          },
          "outputs": {
            "values": [
              "control_normalization::scaled_expression"
            ]
          },
          "parameters": {
            "method": "CP10K",
            "source_port_roles": [
              "measured counts or pre-normalized values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "control_normalization",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "target_normalization::1",
          "component": "Expression preprocessing",
          "condition": "input dataset contains raw counts requiring preprocessing",
          "evidence_ids": [
            "2a372343d3eae52243f5266cdc612ce3e829405b347fedc2a492379f56290e1d"
          ],
          "inputs": {
            "values": [
              "target_pool"
            ]
          },
          "outputs": {
            "values": [
              "target_normalization::scaled_expression"
            ]
          },
          "parameters": {
            "method": "CP10K",
            "source_port_roles": [
              "measured counts or pre-normalized values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "target_normalization",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "identity_encoder::2",
          "component": "Gene identity encoder",
          "condition": null,
          "evidence_ids": [
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0"
          ],
          "inputs": {
            "values": [
              "identity_encoder::looked_up_vectors"
            ]
          },
          "outputs": {
            "values": [
              "identity_encoded"
            ]
          },
          "parameters": {
            "axes": "source-defined feature axes",
            "method": "LayerNorm",
            "source_port_roles": [
              "gene indices",
              "shared raw embedding-table parameters; learned weights, not measured sample data"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "identity_encoder",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "value_encoder::2",
          "component": "Continuous value encoder",
          "condition": null,
          "evidence_ids": [
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0"
          ],
          "inputs": {
            "values": [
              "value_encoder::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "value_encoded"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "current expression values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "value_encoder",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "mask_encoder::2",
          "component": "Perturbation mask encoder",
          "condition": null,
          "evidence_ids": [
            "d4015be6a525bdcb39076445d9d4b72f2bc20449673edb27d686b896a0d20d0a"
          ],
          "inputs": {
            "values": [
              "mask_encoder::looked_up_vectors"
            ]
          },
          "outputs": {
            "values": [
              "mask_encoded"
            ]
          },
          "parameters": {
            "axes": "source-defined feature axes",
            "method": "LayerNorm",
            "source_port_roles": [
              "per-position reveal mask"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "mask_encoder",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "esm_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "esm_loaded"
            ]
          },
          "outputs": {
            "values": [
              "esm_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_availability",
          "trajectory_id": "xcell_jurkat_active_inference"
        },
        {
          "call_id": "esm_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "esm_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "esm_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_project",
          "trajectory_id": "xcell_jurkat_active_inference"
        },
        {
          "call_id": "genept_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "genept_loaded"
            ]
          },
          "outputs": {
            "values": [
              "genept_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_availability",
          "trajectory_id": "xcell_jurkat_active_inference"
        },
        {
          "call_id": "genept_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "genept_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "genept_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_project",
          "trajectory_id": "xcell_jurkat_active_inference"
        },
        {
          "call_id": "string_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "string_loaded"
            ]
          },
          "outputs": {
            "values": [
              "string_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_availability",
          "trajectory_id": "xcell_jurkat_active_inference"
        },
        {
          "call_id": "string_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "string_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "string_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_project",
          "trajectory_id": "xcell_jurkat_active_inference"
        },
        {
          "call_id": "depmap_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "depmap_loaded"
            ]
          },
          "outputs": {
            "values": [
              "depmap_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_availability",
          "trajectory_id": "xcell_jurkat_active_inference"
        },
        {
          "call_id": "depmap_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "depmap_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "depmap_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_project",
          "trajectory_id": "xcell_jurkat_active_inference"
        },
        {
          "call_id": "cp_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "cp_loaded"
            ]
          },
          "outputs": {
            "values": [
              "cp_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_availability",
          "trajectory_id": "xcell_jurkat_active_inference"
        },
        {
          "call_id": "cp_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "cp_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "cp_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_project",
          "trajectory_id": "xcell_jurkat_active_inference"
        },
        {
          "call_id": "esm_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "esm_loaded"
            ]
          },
          "outputs": {
            "values": [
              "esm_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_availability",
          "trajectory_id": "ultra_melanocyte_post_tta_inference"
        },
        {
          "call_id": "esm_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "esm_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "esm_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_project",
          "trajectory_id": "ultra_melanocyte_post_tta_inference"
        },
        {
          "call_id": "genept_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "genept_loaded"
            ]
          },
          "outputs": {
            "values": [
              "genept_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_availability",
          "trajectory_id": "ultra_melanocyte_post_tta_inference"
        },
        {
          "call_id": "genept_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "genept_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "genept_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_project",
          "trajectory_id": "ultra_melanocyte_post_tta_inference"
        },
        {
          "call_id": "string_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "string_loaded"
            ]
          },
          "outputs": {
            "values": [
              "string_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_availability",
          "trajectory_id": "ultra_melanocyte_post_tta_inference"
        },
        {
          "call_id": "string_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "string_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "string_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_project",
          "trajectory_id": "ultra_melanocyte_post_tta_inference"
        },
        {
          "call_id": "depmap_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "depmap_loaded"
            ]
          },
          "outputs": {
            "values": [
              "depmap_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_availability",
          "trajectory_id": "ultra_melanocyte_post_tta_inference"
        },
        {
          "call_id": "depmap_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "depmap_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "depmap_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_project",
          "trajectory_id": "ultra_melanocyte_post_tta_inference"
        },
        {
          "call_id": "cp_availability::1",
          "component": "External embedding collator",
          "condition": "source lookup is present and optional LayerNorm is enabled",
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "values": [
              "cp_loaded"
            ]
          },
          "outputs": {
            "values": [
              "cp_availability::optionally_normalized_reference"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_availability",
          "trajectory_id": "ultra_melanocyte_post_tta_inference"
        },
        {
          "call_id": "cp_project::2",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "cp_project::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "cp_projected"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_project",
          "trajectory_id": "ultra_melanocyte_post_tta_inference"
        },
        {
          "call_id": "identity_encoder::2",
          "component": "Gene identity encoder",
          "condition": null,
          "evidence_ids": [
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0"
          ],
          "inputs": {
            "values": [
              "identity_encoder::looked_up_vectors"
            ]
          },
          "outputs": {
            "values": [
              "identity_encoded"
            ]
          },
          "parameters": {
            "axes": "source-defined feature axes",
            "method": "LayerNorm",
            "source_port_roles": [
              "gene indices",
              "shared raw embedding-table parameters; learned weights, not measured sample data"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "identity_encoder",
          "trajectory_id": "ultra_melanocyte_tta_single_step"
        },
        {
          "call_id": "value_encoder::2",
          "component": "Continuous value encoder",
          "condition": null,
          "evidence_ids": [
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0"
          ],
          "inputs": {
            "values": [
              "value_encoder::projected_values"
            ]
          },
          "outputs": {
            "values": [
              "value_encoded"
            ]
          },
          "parameters": {
            "method": "LayerNorm",
            "source_port_roles": [
              "current expression values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "value_encoder",
          "trajectory_id": "ultra_melanocyte_tta_single_step"
        },
        {
          "call_id": "mask_encoder::2",
          "component": "Perturbation mask encoder",
          "condition": null,
          "evidence_ids": [
            "d4015be6a525bdcb39076445d9d4b72f2bc20449673edb27d686b896a0d20d0a"
          ],
          "inputs": {
            "values": [
              "mask_encoder::looked_up_vectors"
            ]
          },
          "outputs": {
            "values": [
              "mask_encoded"
            ]
          },
          "parameters": {
            "axes": "source-defined feature axes",
            "method": "LayerNorm",
            "source_port_roles": [
              "per-position reveal mask"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "mask_encoder",
          "trajectory_id": "ultra_melanocyte_tta_single_step"
        }
      ]
    },
    {
      "boundaries": "Offset and base are instance parameters; normalization and binning use their own calls.",
      "definition": "Apply a declared logarithmic transform to numerical values.",
      "inputs": [
        {
          "meaning": "Documented input data operands.",
          "name": "values",
          "optional": false,
          "variadic": true
        }
      ],
      "label": "Log transformation",
      "official_reference": "https://numpy.org/doc/stable/reference/generated/numpy.log1p.html",
      "operation_id": "log_transform",
      "outputs": [
        {
          "meaning": "Produced output data operands.",
          "name": "values",
          "optional": false,
          "variadic": true
        }
      ],
      "parameters": [
        "method",
        "offset",
        "base"
      ],
      "reuse_record_ids": [
        "full_2026-07-06__rec_003517"
      ],
      "status": "source_backed",
      "supplemental_examples": [
        {
          "assembly": {
            "bypasses": [],
            "calls": [
              {
                "call_id": "log",
                "component": "OKR-Cell input embedding module",
                "condition": null,
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": {
                  "values": [
                    "expression"
                  ]
                },
                "operation_id": "log_transform",
                "outputs": {
                  "values": [
                    "logged"
                  ]
                },
                "parameters": {
                  "method": "log1p"
                },
                "source_step_id": "log",
                "status": "documented_operation",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "call_id": "hvg",
                "component": "OKR-Cell input embedding module",
                "condition": null,
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": {
                  "values": [
                    "logged",
                    "gene_ids"
                  ]
                },
                "operation_id": "select",
                "outputs": {
                  "values": [
                    "selected_values",
                    "selected_genes"
                  ]
                },
                "parameters": {
                  "criterion": "highly variable genes",
                  "method": "unspecified"
                },
                "source_step_id": "hvg",
                "status": "documented_operation",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "call_id": "gene_token_ids",
                "component": "OKR-Cell input embedding module",
                "condition": null,
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": {
                  "keys": [
                    "selected_genes"
                  ],
                  "table": [
                    "gene_vocabulary"
                  ]
                },
                "operation_id": "lookup",
                "outputs": {
                  "values": [
                    "gene_indices"
                  ]
                },
                "parameters": {
                  "resource_kind": "gene vocabulary"
                },
                "source_step_id": "gene_token_ids",
                "status": "documented_operation",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "call_id": "bins",
                "component": "OKR-Cell input embedding module",
                "condition": null,
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": {
                  "values": [
                    "selected_values"
                  ]
                },
                "operation_id": "bin",
                "outputs": {
                  "values": [
                    "bin_indices"
                  ]
                },
                "parameters": {
                  "bin_edges": "source-defined; formula not decoded",
                  "scope": "per_cell_nonzero_values",
                  "strategy": "equal_interval",
                  "zero_handling": "retain zero"
                },
                "source_step_id": "bins",
                "status": "documented_operation",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "call_id": "gene_lookup",
                "component": "OKR-Cell input embedding module",
                "condition": null,
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": {
                  "keys": [
                    "gene_indices"
                  ],
                  "table": [
                    "gene_table"
                  ]
                },
                "operation_id": "lookup",
                "outputs": {
                  "values": [
                    "gene_vectors"
                  ]
                },
                "parameters": {
                  "resource_kind": "embedding matrix",
                  "trainability": "unspecified"
                },
                "source_step_id": "gene_lookup",
                "status": "documented_operation",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "call_id": "value_lookup",
                "component": "OKR-Cell input embedding module",
                "condition": null,
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": {
                  "keys": [
                    "bin_indices"
                  ],
                  "table": [
                    "bin_table"
                  ]
                },
                "operation_id": "lookup",
                "outputs": {
                  "values": [
                    "value_vectors"
                  ]
                },
                "parameters": {
                  "resource_kind": "embedding matrix",
                  "trainability": "unspecified"
                },
                "source_step_id": "value_lookup",
                "status": "documented_operation",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "call_id": "fusion",
                "component": "OKR-Cell input embedding module",
                "condition": null,
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": {
                  "values": [
                    "gene_vectors",
                    "value_vectors"
                  ]
                },
                "operation_id": "add",
                "outputs": {
                  "values": [
                    "fused"
                  ]
                },
                "parameters": {
                  "alignment": "selected gene position and shared embedding feature coordinate"
                },
                "source_step_id": "fusion",
                "status": "documented_operation",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              }
            ],
            "contract_version": "operation-assembly-v1",
            "evidence": [
              {
                "kind": "paper_text",
                "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                "section_id": "sec_0022"
              }
            ],
            "library_release": "1.0.1",
            "library_sha256": "8baa240c279a84a1229eefd5f34e0527291086570378e20500f37bcb9f1f2caa",
            "lifecycle_phase": "unspecified",
            "model_role": "unresolved",
            "model_variant": "OKR-Cell input module",
            "nodes": [
              {
                "axis_semantics": null,
                "contextual_role": "sample_input",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Measured single-cell expression",
                "node_id": "expression",
                "origin": "source_annotation",
                "representation_type": "expression_values",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "sample_input",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Identifiers aligned to expression positions",
                "node_id": "gene_ids",
                "origin": "source_annotation",
                "representation_type": "gene_identifiers",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "intermediate_representation",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Log1p transformed expression",
                "node_id": "logged",
                "origin": "source_annotation",
                "representation_type": "log_expression",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "intermediate_representation",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "HVG-selected expression",
                "node_id": "selected_values",
                "origin": "source_annotation",
                "representation_type": "selected_expression",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "intermediate_representation",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Identifiers of the selected genes",
                "node_id": "selected_genes",
                "origin": "source_annotation",
                "representation_type": "selected_gene_identifiers",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "reference_resource",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Gene identifier to vocabulary-index mapping",
                "node_id": "gene_vocabulary",
                "origin": "source_annotation",
                "representation_type": "vocabulary_table",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "intermediate_representation",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Vocabulary identifiers for selected genes",
                "node_id": "gene_indices",
                "origin": "source_annotation",
                "representation_type": "gene_vocabulary_indices",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "intermediate_representation",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Equal-interval nonzero expression bins with zero preserved",
                "node_id": "bin_indices",
                "origin": "source_annotation",
                "representation_type": "expression_bin_indices",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "model_parameter",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Gene embedding-layer parameters emb_g; trainability unspecified in this section",
                "node_id": "gene_table",
                "origin": "source_annotation",
                "representation_type": "embedding_matrix",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "model_parameter",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Expression embedding-layer parameters emb_x; a separate table",
                "node_id": "bin_table",
                "origin": "source_annotation",
                "representation_type": "embedding_matrix",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "intermediate_representation",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Selected gene embedding vectors",
                "node_id": "gene_vectors",
                "origin": "source_annotation",
                "representation_type": "gene_embeddings",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "intermediate_representation",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Selected expression-bin embedding vectors",
                "node_id": "value_vectors",
                "origin": "source_annotation",
                "representation_type": "expression_embeddings",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "intermediate_representation",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Element-wise sum supplied to Transformer encoder blocks",
                "node_id": "fused",
                "origin": "source_annotation",
                "representation_type": "summed_input_embeddings",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              }
            ],
            "open_questions": [
              "Source title is marked WITHDRAWN in the inventory. This example describes the documented mechanism and changes no eligibility decision.",
              "Missing equations, HVG algorithm, matrix training status, special-token placement and phase-specific masking remain unspecified."
            ],
            "receipt_inputs": [
              {
                "node_id": "fused",
                "port_role": "input embeddings"
              }
            ],
            "recipient_component": "Transformer encoder blocks",
            "record_id": "full_2026-07-06__rec_001277",
            "source_steps": [
              {
                "component": "OKR-Cell input embedding module",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": [
                  {
                    "node_id": "expression",
                    "port_role": "values"
                  }
                ],
                "operation_type": "log_transform",
                "outputs": [
                  "logged"
                ],
                "step_id": "log",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "component": "OKR-Cell input embedding module",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": [
                  {
                    "node_id": "logged",
                    "port_role": "values"
                  },
                  {
                    "node_id": "gene_ids",
                    "port_role": "values"
                  }
                ],
                "operation_type": "select",
                "outputs": [
                  "selected_values",
                  "selected_genes"
                ],
                "step_id": "hvg",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "component": "OKR-Cell input embedding module",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": [
                  {
                    "node_id": "selected_genes",
                    "port_role": "keys"
                  },
                  {
                    "node_id": "gene_vocabulary",
                    "port_role": "table"
                  }
                ],
                "operation_type": "lookup",
                "outputs": [
                  "gene_indices"
                ],
                "step_id": "gene_token_ids",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "component": "OKR-Cell input embedding module",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": [
                  {
                    "node_id": "selected_values",
                    "port_role": "values"
                  }
                ],
                "operation_type": "bin",
                "outputs": [
                  "bin_indices"
                ],
                "step_id": "bins",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "component": "OKR-Cell input embedding module",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": [
                  {
                    "node_id": "gene_indices",
                    "port_role": "keys"
                  },
                  {
                    "node_id": "gene_table",
                    "port_role": "table"
                  }
                ],
                "operation_type": "lookup",
                "outputs": [
                  "gene_vectors"
                ],
                "step_id": "gene_lookup",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "component": "OKR-Cell input embedding module",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": [
                  {
                    "node_id": "bin_indices",
                    "port_role": "keys"
                  },
                  {
                    "node_id": "bin_table",
                    "port_role": "table"
                  }
                ],
                "operation_type": "lookup",
                "outputs": [
                  "value_vectors"
                ],
                "step_id": "value_lookup",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "component": "OKR-Cell input embedding module",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": [
                  {
                    "node_id": "gene_vectors",
                    "port_role": "values"
                  },
                  {
                    "node_id": "value_vectors",
                    "port_role": "values"
                  }
                ],
                "operation_type": "add",
                "outputs": [
                  "fused"
                ],
                "step_id": "fusion",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              }
            ],
            "task_configuration": "Section-only shared input embedding example",
            "trajectory_id": "input_embedding_section_example"
          },
          "evidence_ids": [
            "3e7bc526b93f082fe2fa11116e75653717b5fa87d40f2a62d606e3047f563504"
          ],
          "operation_id": "log_transform",
          "paper": "OKR-Cell",
          "record_id": "full_2026-07-06__rec_001277",
          "scope": "supplemental mechanism example; outside four-record assembly/reuse denominators",
          "source_status": "WITHDRAWN"
        }
      ],
      "usage": [
        {
          "call_id": "control_normalization::2",
          "component": "Expression preprocessing",
          "condition": "input dataset contains raw counts requiring preprocessing",
          "evidence_ids": [
            "2a372343d3eae52243f5266cdc612ce3e829405b347fedc2a492379f56290e1d"
          ],
          "inputs": {
            "values": [
              "control_normalization::scaled_expression"
            ]
          },
          "outputs": {
            "values": [
              "control_normalized"
            ]
          },
          "parameters": {
            "method": "log1p",
            "source_port_roles": [
              "measured counts or pre-normalized values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "control_normalization",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "target_normalization::2",
          "component": "Expression preprocessing",
          "condition": "input dataset contains raw counts requiring preprocessing",
          "evidence_ids": [
            "2a372343d3eae52243f5266cdc612ce3e829405b347fedc2a492379f56290e1d"
          ],
          "inputs": {
            "values": [
              "target_normalization::scaled_expression"
            ]
          },
          "outputs": {
            "values": [
              "target_normalized"
            ]
          },
          "parameters": {
            "method": "log1p",
            "source_port_roles": [
              "measured counts or pre-normalized values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "target_normalization",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "control_normalization::2",
          "component": "Expression preprocessing",
          "condition": "input dataset contains raw counts requiring preprocessing",
          "evidence_ids": [
            "2a372343d3eae52243f5266cdc612ce3e829405b347fedc2a492379f56290e1d"
          ],
          "inputs": {
            "values": [
              "control_normalization::scaled_expression"
            ]
          },
          "outputs": {
            "values": [
              "control_normalized"
            ]
          },
          "parameters": {
            "method": "log1p",
            "source_port_roles": [
              "measured counts or pre-normalized values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "control_normalization",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "target_normalization::2",
          "component": "Expression preprocessing",
          "condition": "input dataset contains raw counts requiring preprocessing",
          "evidence_ids": [
            "2a372343d3eae52243f5266cdc612ce3e829405b347fedc2a492379f56290e1d"
          ],
          "inputs": {
            "values": [
              "target_normalization::scaled_expression"
            ]
          },
          "outputs": {
            "values": [
              "target_normalized"
            ]
          },
          "parameters": {
            "method": "log1p",
            "source_port_roles": [
              "measured counts or pre-normalized values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "target_normalization",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "control_normalization::2",
          "component": "Expression preprocessing",
          "condition": "input dataset contains raw counts requiring preprocessing",
          "evidence_ids": [
            "2a372343d3eae52243f5266cdc612ce3e829405b347fedc2a492379f56290e1d"
          ],
          "inputs": {
            "values": [
              "control_normalization::scaled_expression"
            ]
          },
          "outputs": {
            "values": [
              "control_normalized"
            ]
          },
          "parameters": {
            "method": "log1p",
            "source_port_roles": [
              "measured counts or pre-normalized values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "control_normalization",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "target_normalization::2",
          "component": "Expression preprocessing",
          "condition": "input dataset contains raw counts requiring preprocessing",
          "evidence_ids": [
            "2a372343d3eae52243f5266cdc612ce3e829405b347fedc2a492379f56290e1d"
          ],
          "inputs": {
            "values": [
              "target_normalization::scaled_expression"
            ]
          },
          "outputs": {
            "values": [
              "target_normalized"
            ]
          },
          "parameters": {
            "method": "log1p",
            "source_port_roles": [
              "measured counts or pre-normalized values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "target_normalization",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "control_normalization::2",
          "component": "Expression preprocessing",
          "condition": "input dataset contains raw counts requiring preprocessing",
          "evidence_ids": [
            "2a372343d3eae52243f5266cdc612ce3e829405b347fedc2a492379f56290e1d"
          ],
          "inputs": {
            "values": [
              "control_normalization::scaled_expression"
            ]
          },
          "outputs": {
            "values": [
              "control_normalized"
            ]
          },
          "parameters": {
            "method": "log1p",
            "source_port_roles": [
              "measured counts or pre-normalized values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "control_normalization",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "target_normalization::2",
          "component": "Expression preprocessing",
          "condition": "input dataset contains raw counts requiring preprocessing",
          "evidence_ids": [
            "2a372343d3eae52243f5266cdc612ce3e829405b347fedc2a492379f56290e1d"
          ],
          "inputs": {
            "values": [
              "target_normalization::scaled_expression"
            ]
          },
          "outputs": {
            "values": [
              "target_normalized"
            ]
          },
          "parameters": {
            "method": "log1p",
            "source_port_roles": [
              "measured counts or pre-normalized values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "target_normalization",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "control_normalization::2",
          "component": "Expression preprocessing",
          "condition": "input dataset contains raw counts requiring preprocessing",
          "evidence_ids": [
            "2a372343d3eae52243f5266cdc612ce3e829405b347fedc2a492379f56290e1d"
          ],
          "inputs": {
            "values": [
              "control_normalization::scaled_expression"
            ]
          },
          "outputs": {
            "values": [
              "control_normalized"
            ]
          },
          "parameters": {
            "method": "log1p",
            "source_port_roles": [
              "measured counts or pre-normalized values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "control_normalization",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "target_normalization::2",
          "component": "Expression preprocessing",
          "condition": "input dataset contains raw counts requiring preprocessing",
          "evidence_ids": [
            "2a372343d3eae52243f5266cdc612ce3e829405b347fedc2a492379f56290e1d"
          ],
          "inputs": {
            "values": [
              "target_normalization::scaled_expression"
            ]
          },
          "outputs": {
            "values": [
              "target_normalized"
            ]
          },
          "parameters": {
            "method": "log1p",
            "source_port_roles": [
              "measured counts or pre-normalized values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "target_normalization",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "control_normalization::2",
          "component": "Expression preprocessing",
          "condition": "input dataset contains raw counts requiring preprocessing",
          "evidence_ids": [
            "2a372343d3eae52243f5266cdc612ce3e829405b347fedc2a492379f56290e1d"
          ],
          "inputs": {
            "values": [
              "control_normalization::scaled_expression"
            ]
          },
          "outputs": {
            "values": [
              "control_normalized"
            ]
          },
          "parameters": {
            "method": "log1p",
            "source_port_roles": [
              "measured counts or pre-normalized values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "control_normalization",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "target_normalization::2",
          "component": "Expression preprocessing",
          "condition": "input dataset contains raw counts requiring preprocessing",
          "evidence_ids": [
            "2a372343d3eae52243f5266cdc612ce3e829405b347fedc2a492379f56290e1d"
          ],
          "inputs": {
            "values": [
              "target_normalization::scaled_expression"
            ]
          },
          "outputs": {
            "values": [
              "target_normalized"
            ]
          },
          "parameters": {
            "method": "log1p",
            "source_port_roles": [
              "measured counts or pre-normalized values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "target_normalization",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "control_normalization::2",
          "component": "Expression preprocessing",
          "condition": "input dataset contains raw counts requiring preprocessing",
          "evidence_ids": [
            "2a372343d3eae52243f5266cdc612ce3e829405b347fedc2a492379f56290e1d"
          ],
          "inputs": {
            "values": [
              "control_normalization::scaled_expression"
            ]
          },
          "outputs": {
            "values": [
              "control_normalized"
            ]
          },
          "parameters": {
            "method": "log1p",
            "source_port_roles": [
              "measured counts or pre-normalized values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "control_normalization",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "target_normalization::2",
          "component": "Expression preprocessing",
          "condition": "input dataset contains raw counts requiring preprocessing",
          "evidence_ids": [
            "2a372343d3eae52243f5266cdc612ce3e829405b347fedc2a492379f56290e1d"
          ],
          "inputs": {
            "values": [
              "target_normalization::scaled_expression"
            ]
          },
          "outputs": {
            "values": [
              "target_normalized"
            ]
          },
          "parameters": {
            "method": "log1p",
            "source_port_roles": [
              "measured counts or pre-normalized values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "target_normalization",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "control_normalization::2",
          "component": "Expression preprocessing",
          "condition": "input dataset contains raw counts requiring preprocessing",
          "evidence_ids": [
            "2a372343d3eae52243f5266cdc612ce3e829405b347fedc2a492379f56290e1d"
          ],
          "inputs": {
            "values": [
              "control_normalization::scaled_expression"
            ]
          },
          "outputs": {
            "values": [
              "control_normalized"
            ]
          },
          "parameters": {
            "method": "log1p",
            "source_port_roles": [
              "measured counts or pre-normalized values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "control_normalization",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "target_normalization::2",
          "component": "Expression preprocessing",
          "condition": "input dataset contains raw counts requiring preprocessing",
          "evidence_ids": [
            "2a372343d3eae52243f5266cdc612ce3e829405b347fedc2a492379f56290e1d"
          ],
          "inputs": {
            "values": [
              "target_normalization::scaled_expression"
            ]
          },
          "outputs": {
            "values": [
              "target_normalized"
            ]
          },
          "parameters": {
            "method": "log1p",
            "source_port_roles": [
              "measured counts or pre-normalized values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "target_normalization",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "normalize::1",
          "component": "C2S preprocessing",
          "condition": null,
          "evidence_ids": [
            "4369c346796e71fe1a47af754545c6d56162a17e0a9e90121bd2da965f73c68a"
          ],
          "inputs": {
            "values": [
              "expression"
            ]
          },
          "outputs": {
            "values": [
              "normalized"
            ]
          },
          "parameters": {
            "method": "log1p",
            "source_port_roles": [
              "measured expression"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "normalize",
          "trajectory_id": "c2s_scale_fine_tuning_prompt_receipt"
        }
      ]
    },
    {
      "boundaries": "Bin construction, per-sample/global scope and zero handling are explicit instance parameters; learned vector quantization has a different algorithm.",
      "definition": "Assign numerical values to discrete interval identifiers.",
      "inputs": [
        {
          "meaning": "Documented input data operands.",
          "name": "values",
          "optional": false,
          "variadic": true
        }
      ],
      "label": "Binning",
      "official_reference": "https://scikit-learn.org/stable/modules/generated/sklearn.preprocessing.KBinsDiscretizer.html",
      "operation_id": "bin",
      "outputs": [
        {
          "meaning": "Produced output data operands.",
          "name": "values",
          "optional": false,
          "variadic": true
        }
      ],
      "parameters": [
        "strategy",
        "scope",
        "zero_handling",
        "bin_edges"
      ],
      "reuse_record_ids": [],
      "status": "source_backed",
      "supplemental_examples": [
        {
          "assembly": {
            "bypasses": [],
            "calls": [
              {
                "call_id": "log",
                "component": "OKR-Cell input embedding module",
                "condition": null,
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": {
                  "values": [
                    "expression"
                  ]
                },
                "operation_id": "log_transform",
                "outputs": {
                  "values": [
                    "logged"
                  ]
                },
                "parameters": {
                  "method": "log1p"
                },
                "source_step_id": "log",
                "status": "documented_operation",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "call_id": "hvg",
                "component": "OKR-Cell input embedding module",
                "condition": null,
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": {
                  "values": [
                    "logged",
                    "gene_ids"
                  ]
                },
                "operation_id": "select",
                "outputs": {
                  "values": [
                    "selected_values",
                    "selected_genes"
                  ]
                },
                "parameters": {
                  "criterion": "highly variable genes",
                  "method": "unspecified"
                },
                "source_step_id": "hvg",
                "status": "documented_operation",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "call_id": "gene_token_ids",
                "component": "OKR-Cell input embedding module",
                "condition": null,
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": {
                  "keys": [
                    "selected_genes"
                  ],
                  "table": [
                    "gene_vocabulary"
                  ]
                },
                "operation_id": "lookup",
                "outputs": {
                  "values": [
                    "gene_indices"
                  ]
                },
                "parameters": {
                  "resource_kind": "gene vocabulary"
                },
                "source_step_id": "gene_token_ids",
                "status": "documented_operation",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "call_id": "bins",
                "component": "OKR-Cell input embedding module",
                "condition": null,
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": {
                  "values": [
                    "selected_values"
                  ]
                },
                "operation_id": "bin",
                "outputs": {
                  "values": [
                    "bin_indices"
                  ]
                },
                "parameters": {
                  "bin_edges": "source-defined; formula not decoded",
                  "scope": "per_cell_nonzero_values",
                  "strategy": "equal_interval",
                  "zero_handling": "retain zero"
                },
                "source_step_id": "bins",
                "status": "documented_operation",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "call_id": "gene_lookup",
                "component": "OKR-Cell input embedding module",
                "condition": null,
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": {
                  "keys": [
                    "gene_indices"
                  ],
                  "table": [
                    "gene_table"
                  ]
                },
                "operation_id": "lookup",
                "outputs": {
                  "values": [
                    "gene_vectors"
                  ]
                },
                "parameters": {
                  "resource_kind": "embedding matrix",
                  "trainability": "unspecified"
                },
                "source_step_id": "gene_lookup",
                "status": "documented_operation",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "call_id": "value_lookup",
                "component": "OKR-Cell input embedding module",
                "condition": null,
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": {
                  "keys": [
                    "bin_indices"
                  ],
                  "table": [
                    "bin_table"
                  ]
                },
                "operation_id": "lookup",
                "outputs": {
                  "values": [
                    "value_vectors"
                  ]
                },
                "parameters": {
                  "resource_kind": "embedding matrix",
                  "trainability": "unspecified"
                },
                "source_step_id": "value_lookup",
                "status": "documented_operation",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "call_id": "fusion",
                "component": "OKR-Cell input embedding module",
                "condition": null,
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": {
                  "values": [
                    "gene_vectors",
                    "value_vectors"
                  ]
                },
                "operation_id": "add",
                "outputs": {
                  "values": [
                    "fused"
                  ]
                },
                "parameters": {
                  "alignment": "selected gene position and shared embedding feature coordinate"
                },
                "source_step_id": "fusion",
                "status": "documented_operation",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              }
            ],
            "contract_version": "operation-assembly-v1",
            "evidence": [
              {
                "kind": "paper_text",
                "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                "section_id": "sec_0022"
              }
            ],
            "library_release": "1.0.1",
            "library_sha256": "8baa240c279a84a1229eefd5f34e0527291086570378e20500f37bcb9f1f2caa",
            "lifecycle_phase": "unspecified",
            "model_role": "unresolved",
            "model_variant": "OKR-Cell input module",
            "nodes": [
              {
                "axis_semantics": null,
                "contextual_role": "sample_input",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Measured single-cell expression",
                "node_id": "expression",
                "origin": "source_annotation",
                "representation_type": "expression_values",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "sample_input",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Identifiers aligned to expression positions",
                "node_id": "gene_ids",
                "origin": "source_annotation",
                "representation_type": "gene_identifiers",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "intermediate_representation",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Log1p transformed expression",
                "node_id": "logged",
                "origin": "source_annotation",
                "representation_type": "log_expression",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "intermediate_representation",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "HVG-selected expression",
                "node_id": "selected_values",
                "origin": "source_annotation",
                "representation_type": "selected_expression",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "intermediate_representation",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Identifiers of the selected genes",
                "node_id": "selected_genes",
                "origin": "source_annotation",
                "representation_type": "selected_gene_identifiers",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "reference_resource",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Gene identifier to vocabulary-index mapping",
                "node_id": "gene_vocabulary",
                "origin": "source_annotation",
                "representation_type": "vocabulary_table",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "intermediate_representation",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Vocabulary identifiers for selected genes",
                "node_id": "gene_indices",
                "origin": "source_annotation",
                "representation_type": "gene_vocabulary_indices",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "intermediate_representation",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Equal-interval nonzero expression bins with zero preserved",
                "node_id": "bin_indices",
                "origin": "source_annotation",
                "representation_type": "expression_bin_indices",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "model_parameter",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Gene embedding-layer parameters emb_g; trainability unspecified in this section",
                "node_id": "gene_table",
                "origin": "source_annotation",
                "representation_type": "embedding_matrix",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "model_parameter",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Expression embedding-layer parameters emb_x; a separate table",
                "node_id": "bin_table",
                "origin": "source_annotation",
                "representation_type": "embedding_matrix",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "intermediate_representation",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Selected gene embedding vectors",
                "node_id": "gene_vectors",
                "origin": "source_annotation",
                "representation_type": "gene_embeddings",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "intermediate_representation",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Selected expression-bin embedding vectors",
                "node_id": "value_vectors",
                "origin": "source_annotation",
                "representation_type": "expression_embeddings",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "intermediate_representation",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Element-wise sum supplied to Transformer encoder blocks",
                "node_id": "fused",
                "origin": "source_annotation",
                "representation_type": "summed_input_embeddings",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              }
            ],
            "open_questions": [
              "Source title is marked WITHDRAWN in the inventory. This example describes the documented mechanism and changes no eligibility decision.",
              "Missing equations, HVG algorithm, matrix training status, special-token placement and phase-specific masking remain unspecified."
            ],
            "receipt_inputs": [
              {
                "node_id": "fused",
                "port_role": "input embeddings"
              }
            ],
            "recipient_component": "Transformer encoder blocks",
            "record_id": "full_2026-07-06__rec_001277",
            "source_steps": [
              {
                "component": "OKR-Cell input embedding module",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": [
                  {
                    "node_id": "expression",
                    "port_role": "values"
                  }
                ],
                "operation_type": "log_transform",
                "outputs": [
                  "logged"
                ],
                "step_id": "log",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "component": "OKR-Cell input embedding module",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": [
                  {
                    "node_id": "logged",
                    "port_role": "values"
                  },
                  {
                    "node_id": "gene_ids",
                    "port_role": "values"
                  }
                ],
                "operation_type": "select",
                "outputs": [
                  "selected_values",
                  "selected_genes"
                ],
                "step_id": "hvg",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "component": "OKR-Cell input embedding module",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": [
                  {
                    "node_id": "selected_genes",
                    "port_role": "keys"
                  },
                  {
                    "node_id": "gene_vocabulary",
                    "port_role": "table"
                  }
                ],
                "operation_type": "lookup",
                "outputs": [
                  "gene_indices"
                ],
                "step_id": "gene_token_ids",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "component": "OKR-Cell input embedding module",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": [
                  {
                    "node_id": "selected_values",
                    "port_role": "values"
                  }
                ],
                "operation_type": "bin",
                "outputs": [
                  "bin_indices"
                ],
                "step_id": "bins",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "component": "OKR-Cell input embedding module",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": [
                  {
                    "node_id": "gene_indices",
                    "port_role": "keys"
                  },
                  {
                    "node_id": "gene_table",
                    "port_role": "table"
                  }
                ],
                "operation_type": "lookup",
                "outputs": [
                  "gene_vectors"
                ],
                "step_id": "gene_lookup",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "component": "OKR-Cell input embedding module",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": [
                  {
                    "node_id": "bin_indices",
                    "port_role": "keys"
                  },
                  {
                    "node_id": "bin_table",
                    "port_role": "table"
                  }
                ],
                "operation_type": "lookup",
                "outputs": [
                  "value_vectors"
                ],
                "step_id": "value_lookup",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "component": "OKR-Cell input embedding module",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": [
                  {
                    "node_id": "gene_vectors",
                    "port_role": "values"
                  },
                  {
                    "node_id": "value_vectors",
                    "port_role": "values"
                  }
                ],
                "operation_type": "add",
                "outputs": [
                  "fused"
                ],
                "step_id": "fusion",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              }
            ],
            "task_configuration": "Section-only shared input embedding example",
            "trajectory_id": "input_embedding_section_example"
          },
          "evidence_ids": [
            "3e7bc526b93f082fe2fa11116e75653717b5fa87d40f2a62d606e3047f563504"
          ],
          "operation_id": "bin",
          "paper": "OKR-Cell",
          "record_id": "full_2026-07-06__rec_001277",
          "scope": "supplemental mechanism example; outside four-record assembly/reuse denominators",
          "source_status": "WITHDRAWN"
        }
      ],
      "usage": []
    },
    {
      "boundaries": "Vocabulary, segmentation and source alphabet belong to the instance. Numerical binning is a separate operation.",
      "definition": "Segment a symbolic sequence into discrete vocabulary tokens or indices.",
      "inputs": [
        {
          "meaning": "Documented input data operands.",
          "name": "values",
          "optional": false,
          "variadic": true
        }
      ],
      "label": "Tokenization",
      "official_reference": null,
      "operation_id": "tokenize",
      "outputs": [
        {
          "meaning": "Produced output data operands.",
          "name": "values",
          "optional": false,
          "variadic": true
        }
      ],
      "parameters": [
        "vocabulary",
        "segmentation",
        "source_alphabet"
      ],
      "reuse_record_ids": [
        "full_2026-07-06__rec_000771"
      ],
      "status": "source_backed",
      "supplemental_examples": [],
      "usage": [
        {
          "call_id": "tokenize_english::1",
          "component": "LLaMA tokenizer",
          "condition": null,
          "evidence_ids": [
            "9c0d8b9208d8d4fd297e09a703528eafc1ef47ac592336cf054290ee1a345f10",
            "66f967ac6c448ed7cef8aa2b5fab9ae881c5cd419add6c080ee4ee75e1f46aa8"
          ],
          "inputs": {
            "values": [
              "prompt"
            ]
          },
          "outputs": {
            "values": [
              "english_tokens"
            ]
          },
          "parameters": {
            "segmentation": "source tokenizer",
            "source_alphabet": "text",
            "source_port_roles": [
              "instruction_text"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "tokenize_english",
          "trajectory_id": "benchmark_fine_tuning"
        },
        {
          "call_id": "tokenize_dna_a::1",
          "component": "Nucleotide Transformer tokenizer",
          "condition": null,
          "evidence_ids": [
            "9c0d8b9208d8d4fd297e09a703528eafc1ef47ac592336cf054290ee1a345f10",
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "sequence_a"
            ]
          },
          "outputs": {
            "values": [
              "tokens_a"
            ]
          },
          "parameters": {
            "segmentation": "source tokenizer",
            "source_alphabet": "nucleotide",
            "source_port_roles": [
              "nucleotide_sequence"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "tokenize_dna_a",
          "trajectory_id": "benchmark_fine_tuning"
        },
        {
          "call_id": "tokenize_english::1",
          "component": "LLaMA tokenizer",
          "condition": null,
          "evidence_ids": [
            "9c0d8b9208d8d4fd297e09a703528eafc1ef47ac592336cf054290ee1a345f10",
            "66f967ac6c448ed7cef8aa2b5fab9ae881c5cd419add6c080ee4ee75e1f46aa8"
          ],
          "inputs": {
            "values": [
              "prompt"
            ]
          },
          "outputs": {
            "values": [
              "english_tokens"
            ]
          },
          "parameters": {
            "segmentation": "source tokenizer",
            "source_alphabet": "text",
            "source_port_roles": [
              "instruction_text"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "tokenize_english",
          "trajectory_id": "benchmark_inference"
        },
        {
          "call_id": "tokenize_dna_a::1",
          "component": "Nucleotide Transformer tokenizer",
          "condition": null,
          "evidence_ids": [
            "9c0d8b9208d8d4fd297e09a703528eafc1ef47ac592336cf054290ee1a345f10",
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "sequence_a"
            ]
          },
          "outputs": {
            "values": [
              "tokens_a"
            ]
          },
          "parameters": {
            "segmentation": "source tokenizer",
            "source_alphabet": "nucleotide",
            "source_port_roles": [
              "nucleotide_sequence"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "tokenize_dna_a",
          "trajectory_id": "benchmark_inference"
        },
        {
          "call_id": "tokenize_english::1",
          "component": "LLaMA tokenizer",
          "condition": null,
          "evidence_ids": [
            "9c0d8b9208d8d4fd297e09a703528eafc1ef47ac592336cf054290ee1a345f10",
            "66f967ac6c448ed7cef8aa2b5fab9ae881c5cd419add6c080ee4ee75e1f46aa8"
          ],
          "inputs": {
            "values": [
              "prompt"
            ]
          },
          "outputs": {
            "values": [
              "english_tokens"
            ]
          },
          "parameters": {
            "segmentation": "source tokenizer",
            "source_alphabet": "text",
            "source_port_roles": [
              "instruction_text"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "tokenize_english",
          "trajectory_id": "curated_dna_fine_tuning"
        },
        {
          "call_id": "tokenize_dna_a::1",
          "component": "Nucleotide Transformer tokenizer",
          "condition": null,
          "evidence_ids": [
            "9c0d8b9208d8d4fd297e09a703528eafc1ef47ac592336cf054290ee1a345f10",
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "sequence_a"
            ]
          },
          "outputs": {
            "values": [
              "tokens_a"
            ]
          },
          "parameters": {
            "segmentation": "source tokenizer",
            "source_alphabet": "nucleotide",
            "source_port_roles": [
              "nucleotide_sequence"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "tokenize_dna_a",
          "trajectory_id": "curated_dna_fine_tuning"
        },
        {
          "call_id": "tokenize_english::1",
          "component": "LLaMA tokenizer",
          "condition": null,
          "evidence_ids": [
            "9c0d8b9208d8d4fd297e09a703528eafc1ef47ac592336cf054290ee1a345f10",
            "66f967ac6c448ed7cef8aa2b5fab9ae881c5cd419add6c080ee4ee75e1f46aa8"
          ],
          "inputs": {
            "values": [
              "prompt"
            ]
          },
          "outputs": {
            "values": [
              "english_tokens"
            ]
          },
          "parameters": {
            "segmentation": "source tokenizer",
            "source_alphabet": "text",
            "source_port_roles": [
              "instruction_text"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "tokenize_english",
          "trajectory_id": "curated_dna_inference"
        },
        {
          "call_id": "tokenize_dna_a::1",
          "component": "Nucleotide Transformer tokenizer",
          "condition": null,
          "evidence_ids": [
            "9c0d8b9208d8d4fd297e09a703528eafc1ef47ac592336cf054290ee1a345f10",
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "sequence_a"
            ]
          },
          "outputs": {
            "values": [
              "tokens_a"
            ]
          },
          "parameters": {
            "segmentation": "source tokenizer",
            "source_alphabet": "nucleotide",
            "source_port_roles": [
              "nucleotide_sequence"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "tokenize_dna_a",
          "trajectory_id": "curated_dna_inference"
        },
        {
          "call_id": "tokenize_english::1",
          "component": "LLaMA tokenizer",
          "condition": null,
          "evidence_ids": [
            "9c0d8b9208d8d4fd297e09a703528eafc1ef47ac592336cf054290ee1a345f10",
            "66f967ac6c448ed7cef8aa2b5fab9ae881c5cd419add6c080ee4ee75e1f46aa8"
          ],
          "inputs": {
            "values": [
              "prompt"
            ]
          },
          "outputs": {
            "values": [
              "english_tokens"
            ]
          },
          "parameters": {
            "segmentation": "source tokenizer",
            "source_alphabet": "text",
            "source_port_roles": [
              "instruction_text"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "tokenize_english",
          "trajectory_id": "curated_rna_fine_tuning"
        },
        {
          "call_id": "tokenize_dna_a::1",
          "component": "Nucleotide Transformer tokenizer",
          "condition": null,
          "evidence_ids": [
            "9c0d8b9208d8d4fd297e09a703528eafc1ef47ac592336cf054290ee1a345f10",
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "sequence_a"
            ]
          },
          "outputs": {
            "values": [
              "tokens_a"
            ]
          },
          "parameters": {
            "segmentation": "source tokenizer",
            "source_alphabet": "nucleotide",
            "source_port_roles": [
              "nucleotide_sequence"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "tokenize_dna_a",
          "trajectory_id": "curated_rna_fine_tuning"
        },
        {
          "call_id": "tokenize_english::1",
          "component": "LLaMA tokenizer",
          "condition": null,
          "evidence_ids": [
            "9c0d8b9208d8d4fd297e09a703528eafc1ef47ac592336cf054290ee1a345f10",
            "66f967ac6c448ed7cef8aa2b5fab9ae881c5cd419add6c080ee4ee75e1f46aa8"
          ],
          "inputs": {
            "values": [
              "prompt"
            ]
          },
          "outputs": {
            "values": [
              "english_tokens"
            ]
          },
          "parameters": {
            "segmentation": "source tokenizer",
            "source_alphabet": "text",
            "source_port_roles": [
              "instruction_text"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "tokenize_english",
          "trajectory_id": "curated_rna_inference"
        },
        {
          "call_id": "tokenize_dna_a::1",
          "component": "Nucleotide Transformer tokenizer",
          "condition": null,
          "evidence_ids": [
            "9c0d8b9208d8d4fd297e09a703528eafc1ef47ac592336cf054290ee1a345f10",
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "sequence_a"
            ]
          },
          "outputs": {
            "values": [
              "tokens_a"
            ]
          },
          "parameters": {
            "segmentation": "source tokenizer",
            "source_alphabet": "nucleotide",
            "source_port_roles": [
              "nucleotide_sequence"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "tokenize_dna_a",
          "trajectory_id": "curated_rna_inference"
        },
        {
          "call_id": "tokenize_english::1",
          "component": "LLaMA tokenizer",
          "condition": null,
          "evidence_ids": [
            "9c0d8b9208d8d4fd297e09a703528eafc1ef47ac592336cf054290ee1a345f10",
            "66f967ac6c448ed7cef8aa2b5fab9ae881c5cd419add6c080ee4ee75e1f46aa8"
          ],
          "inputs": {
            "values": [
              "prompt"
            ]
          },
          "outputs": {
            "values": [
              "english_tokens"
            ]
          },
          "parameters": {
            "segmentation": "source tokenizer",
            "source_alphabet": "text",
            "source_port_roles": [
              "instruction_text"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "tokenize_english",
          "trajectory_id": "curated_protein_fine_tuning"
        },
        {
          "call_id": "tokenize_dna_a::1",
          "component": "Nucleotide Transformer tokenizer",
          "condition": null,
          "evidence_ids": [
            "9c0d8b9208d8d4fd297e09a703528eafc1ef47ac592336cf054290ee1a345f10",
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "sequence_a"
            ]
          },
          "outputs": {
            "values": [
              "tokens_a"
            ]
          },
          "parameters": {
            "segmentation": "source tokenizer",
            "source_alphabet": "nucleotide",
            "source_port_roles": [
              "nucleotide_sequence"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "tokenize_dna_a",
          "trajectory_id": "curated_protein_fine_tuning"
        },
        {
          "call_id": "tokenize_english::1",
          "component": "LLaMA tokenizer",
          "condition": null,
          "evidence_ids": [
            "9c0d8b9208d8d4fd297e09a703528eafc1ef47ac592336cf054290ee1a345f10",
            "66f967ac6c448ed7cef8aa2b5fab9ae881c5cd419add6c080ee4ee75e1f46aa8"
          ],
          "inputs": {
            "values": [
              "prompt"
            ]
          },
          "outputs": {
            "values": [
              "english_tokens"
            ]
          },
          "parameters": {
            "segmentation": "source tokenizer",
            "source_alphabet": "text",
            "source_port_roles": [
              "instruction_text"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "tokenize_english",
          "trajectory_id": "curated_protein_inference"
        },
        {
          "call_id": "tokenize_dna_a::1",
          "component": "Nucleotide Transformer tokenizer",
          "condition": null,
          "evidence_ids": [
            "9c0d8b9208d8d4fd297e09a703528eafc1ef47ac592336cf054290ee1a345f10",
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "sequence_a"
            ]
          },
          "outputs": {
            "values": [
              "tokens_a"
            ]
          },
          "parameters": {
            "segmentation": "source tokenizer",
            "source_alphabet": "nucleotide",
            "source_port_roles": [
              "nucleotide_sequence"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "tokenize_dna_a",
          "trajectory_id": "curated_protein_inference"
        },
        {
          "call_id": "tokenize_english::1",
          "component": "LLaMA tokenizer",
          "condition": null,
          "evidence_ids": [
            "9c0d8b9208d8d4fd297e09a703528eafc1ef47ac592336cf054290ee1a345f10",
            "66f967ac6c448ed7cef8aa2b5fab9ae881c5cd419add6c080ee4ee75e1f46aa8"
          ],
          "inputs": {
            "values": [
              "prompt"
            ]
          },
          "outputs": {
            "values": [
              "english_tokens"
            ]
          },
          "parameters": {
            "segmentation": "source tokenizer",
            "source_alphabet": "text",
            "source_port_roles": [
              "instruction_text"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "tokenize_english",
          "trajectory_id": "multiple_fine_tuning"
        },
        {
          "call_id": "tokenize_dna_a::1",
          "component": "Nucleotide Transformer tokenizer",
          "condition": null,
          "evidence_ids": [
            "9c0d8b9208d8d4fd297e09a703528eafc1ef47ac592336cf054290ee1a345f10",
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "sequence_a"
            ]
          },
          "outputs": {
            "values": [
              "tokens_a"
            ]
          },
          "parameters": {
            "segmentation": "source tokenizer",
            "source_alphabet": "nucleotide",
            "source_port_roles": [
              "nucleotide_sequence"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "tokenize_dna_a",
          "trajectory_id": "multiple_fine_tuning"
        },
        {
          "call_id": "tokenize_dna_b::1",
          "component": "Nucleotide Transformer tokenizer",
          "condition": null,
          "evidence_ids": [
            "9c0d8b9208d8d4fd297e09a703528eafc1ef47ac592336cf054290ee1a345f10",
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "sequence_b"
            ]
          },
          "outputs": {
            "values": [
              "tokens_b"
            ]
          },
          "parameters": {
            "segmentation": "source tokenizer",
            "source_alphabet": "nucleotide",
            "source_port_roles": [
              "nucleotide_sequence"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "tokenize_dna_b",
          "trajectory_id": "multiple_fine_tuning"
        },
        {
          "call_id": "tokenize_english::1",
          "component": "LLaMA tokenizer",
          "condition": null,
          "evidence_ids": [
            "9c0d8b9208d8d4fd297e09a703528eafc1ef47ac592336cf054290ee1a345f10",
            "66f967ac6c448ed7cef8aa2b5fab9ae881c5cd419add6c080ee4ee75e1f46aa8"
          ],
          "inputs": {
            "values": [
              "prompt"
            ]
          },
          "outputs": {
            "values": [
              "english_tokens"
            ]
          },
          "parameters": {
            "segmentation": "source tokenizer",
            "source_alphabet": "text",
            "source_port_roles": [
              "instruction_text"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "tokenize_english",
          "trajectory_id": "multiple_inference"
        },
        {
          "call_id": "tokenize_dna_a::1",
          "component": "Nucleotide Transformer tokenizer",
          "condition": null,
          "evidence_ids": [
            "9c0d8b9208d8d4fd297e09a703528eafc1ef47ac592336cf054290ee1a345f10",
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "sequence_a"
            ]
          },
          "outputs": {
            "values": [
              "tokens_a"
            ]
          },
          "parameters": {
            "segmentation": "source tokenizer",
            "source_alphabet": "nucleotide",
            "source_port_roles": [
              "nucleotide_sequence"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "tokenize_dna_a",
          "trajectory_id": "multiple_inference"
        },
        {
          "call_id": "tokenize_dna_b::1",
          "component": "Nucleotide Transformer tokenizer",
          "condition": null,
          "evidence_ids": [
            "9c0d8b9208d8d4fd297e09a703528eafc1ef47ac592336cf054290ee1a345f10",
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "sequence_b"
            ]
          },
          "outputs": {
            "values": [
              "tokens_b"
            ]
          },
          "parameters": {
            "segmentation": "source tokenizer",
            "source_alphabet": "nucleotide",
            "source_port_roles": [
              "nucleotide_sequence"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "tokenize_dna_b",
          "trajectory_id": "multiple_inference"
        },
        {
          "call_id": "tokenize_english::1",
          "component": "LLaMA tokenizer",
          "condition": null,
          "evidence_ids": [
            "9c0d8b9208d8d4fd297e09a703528eafc1ef47ac592336cf054290ee1a345f10",
            "66f967ac6c448ed7cef8aa2b5fab9ae881c5cd419add6c080ee4ee75e1f46aa8"
          ],
          "inputs": {
            "values": [
              "prompt"
            ]
          },
          "outputs": {
            "values": [
              "english_tokens"
            ]
          },
          "parameters": {
            "segmentation": "source tokenizer",
            "source_alphabet": "text",
            "source_port_roles": [
              "instruction_text"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "tokenize_english",
          "trajectory_id": "unconditioned_benchmark_fine_tuning"
        },
        {
          "call_id": "tokenize_dna_a::1",
          "component": "Nucleotide Transformer tokenizer",
          "condition": null,
          "evidence_ids": [
            "9c0d8b9208d8d4fd297e09a703528eafc1ef47ac592336cf054290ee1a345f10",
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "sequence_a"
            ]
          },
          "outputs": {
            "values": [
              "tokens_a"
            ]
          },
          "parameters": {
            "segmentation": "source tokenizer",
            "source_alphabet": "nucleotide",
            "source_port_roles": [
              "nucleotide_sequence"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "tokenize_dna_a",
          "trajectory_id": "unconditioned_benchmark_fine_tuning"
        },
        {
          "call_id": "tokenize_english::1",
          "component": "LLaMA tokenizer",
          "condition": null,
          "evidence_ids": [
            "9c0d8b9208d8d4fd297e09a703528eafc1ef47ac592336cf054290ee1a345f10",
            "66f967ac6c448ed7cef8aa2b5fab9ae881c5cd419add6c080ee4ee75e1f46aa8"
          ],
          "inputs": {
            "values": [
              "prompt"
            ]
          },
          "outputs": {
            "values": [
              "english_tokens"
            ]
          },
          "parameters": {
            "segmentation": "source tokenizer",
            "source_alphabet": "text",
            "source_port_roles": [
              "instruction_text"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "tokenize_english",
          "trajectory_id": "unconditioned_benchmark_inference"
        },
        {
          "call_id": "tokenize_dna_a::1",
          "component": "Nucleotide Transformer tokenizer",
          "condition": null,
          "evidence_ids": [
            "9c0d8b9208d8d4fd297e09a703528eafc1ef47ac592336cf054290ee1a345f10",
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "sequence_a"
            ]
          },
          "outputs": {
            "values": [
              "tokens_a"
            ]
          },
          "parameters": {
            "segmentation": "source tokenizer",
            "source_alphabet": "nucleotide",
            "source_port_roles": [
              "nucleotide_sequence"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "tokenize_dna_a",
          "trajectory_id": "unconditioned_benchmark_inference"
        }
      ]
    },
    {
      "boundaries": "The table is an explicit operand. Learned embedding matrices, fixed vocabularies and precomputed feature references differ through resource metadata.",
      "definition": "Retrieve entries from a table using supplied identifiers or indices.",
      "inputs": [
        {
          "meaning": "Identifiers or indices selecting entries.",
          "name": "keys",
          "optional": false,
          "variadic": true
        },
        {
          "meaning": "Referenced lookup table or embedding matrix.",
          "name": "table",
          "optional": false,
          "variadic": false
        }
      ],
      "label": "Table lookup",
      "official_reference": "https://docs.pytorch.org/docs/stable/generated/torch.nn.Embedding.html",
      "operation_id": "lookup",
      "outputs": [
        {
          "meaning": "Produced output data operands.",
          "name": "values",
          "optional": false,
          "variadic": true
        }
      ],
      "parameters": [
        "key_space",
        "resource_kind",
        "trainability"
      ],
      "reuse_record_ids": [
        "full_2026-07-06__rec_000771",
        "full_2026-07-06__rec_001319",
        "full_2026-07-06__rec_003517"
      ],
      "status": "source_backed",
      "supplemental_examples": [
        {
          "assembly": {
            "bypasses": [],
            "calls": [
              {
                "call_id": "log",
                "component": "OKR-Cell input embedding module",
                "condition": null,
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": {
                  "values": [
                    "expression"
                  ]
                },
                "operation_id": "log_transform",
                "outputs": {
                  "values": [
                    "logged"
                  ]
                },
                "parameters": {
                  "method": "log1p"
                },
                "source_step_id": "log",
                "status": "documented_operation",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "call_id": "hvg",
                "component": "OKR-Cell input embedding module",
                "condition": null,
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": {
                  "values": [
                    "logged",
                    "gene_ids"
                  ]
                },
                "operation_id": "select",
                "outputs": {
                  "values": [
                    "selected_values",
                    "selected_genes"
                  ]
                },
                "parameters": {
                  "criterion": "highly variable genes",
                  "method": "unspecified"
                },
                "source_step_id": "hvg",
                "status": "documented_operation",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "call_id": "gene_token_ids",
                "component": "OKR-Cell input embedding module",
                "condition": null,
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": {
                  "keys": [
                    "selected_genes"
                  ],
                  "table": [
                    "gene_vocabulary"
                  ]
                },
                "operation_id": "lookup",
                "outputs": {
                  "values": [
                    "gene_indices"
                  ]
                },
                "parameters": {
                  "resource_kind": "gene vocabulary"
                },
                "source_step_id": "gene_token_ids",
                "status": "documented_operation",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "call_id": "bins",
                "component": "OKR-Cell input embedding module",
                "condition": null,
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": {
                  "values": [
                    "selected_values"
                  ]
                },
                "operation_id": "bin",
                "outputs": {
                  "values": [
                    "bin_indices"
                  ]
                },
                "parameters": {
                  "bin_edges": "source-defined; formula not decoded",
                  "scope": "per_cell_nonzero_values",
                  "strategy": "equal_interval",
                  "zero_handling": "retain zero"
                },
                "source_step_id": "bins",
                "status": "documented_operation",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "call_id": "gene_lookup",
                "component": "OKR-Cell input embedding module",
                "condition": null,
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": {
                  "keys": [
                    "gene_indices"
                  ],
                  "table": [
                    "gene_table"
                  ]
                },
                "operation_id": "lookup",
                "outputs": {
                  "values": [
                    "gene_vectors"
                  ]
                },
                "parameters": {
                  "resource_kind": "embedding matrix",
                  "trainability": "unspecified"
                },
                "source_step_id": "gene_lookup",
                "status": "documented_operation",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "call_id": "value_lookup",
                "component": "OKR-Cell input embedding module",
                "condition": null,
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": {
                  "keys": [
                    "bin_indices"
                  ],
                  "table": [
                    "bin_table"
                  ]
                },
                "operation_id": "lookup",
                "outputs": {
                  "values": [
                    "value_vectors"
                  ]
                },
                "parameters": {
                  "resource_kind": "embedding matrix",
                  "trainability": "unspecified"
                },
                "source_step_id": "value_lookup",
                "status": "documented_operation",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "call_id": "fusion",
                "component": "OKR-Cell input embedding module",
                "condition": null,
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": {
                  "values": [
                    "gene_vectors",
                    "value_vectors"
                  ]
                },
                "operation_id": "add",
                "outputs": {
                  "values": [
                    "fused"
                  ]
                },
                "parameters": {
                  "alignment": "selected gene position and shared embedding feature coordinate"
                },
                "source_step_id": "fusion",
                "status": "documented_operation",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              }
            ],
            "contract_version": "operation-assembly-v1",
            "evidence": [
              {
                "kind": "paper_text",
                "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                "section_id": "sec_0022"
              }
            ],
            "library_release": "1.0.1",
            "library_sha256": "8baa240c279a84a1229eefd5f34e0527291086570378e20500f37bcb9f1f2caa",
            "lifecycle_phase": "unspecified",
            "model_role": "unresolved",
            "model_variant": "OKR-Cell input module",
            "nodes": [
              {
                "axis_semantics": null,
                "contextual_role": "sample_input",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Measured single-cell expression",
                "node_id": "expression",
                "origin": "source_annotation",
                "representation_type": "expression_values",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "sample_input",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Identifiers aligned to expression positions",
                "node_id": "gene_ids",
                "origin": "source_annotation",
                "representation_type": "gene_identifiers",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "intermediate_representation",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Log1p transformed expression",
                "node_id": "logged",
                "origin": "source_annotation",
                "representation_type": "log_expression",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "intermediate_representation",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "HVG-selected expression",
                "node_id": "selected_values",
                "origin": "source_annotation",
                "representation_type": "selected_expression",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "intermediate_representation",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Identifiers of the selected genes",
                "node_id": "selected_genes",
                "origin": "source_annotation",
                "representation_type": "selected_gene_identifiers",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "reference_resource",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Gene identifier to vocabulary-index mapping",
                "node_id": "gene_vocabulary",
                "origin": "source_annotation",
                "representation_type": "vocabulary_table",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "intermediate_representation",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Vocabulary identifiers for selected genes",
                "node_id": "gene_indices",
                "origin": "source_annotation",
                "representation_type": "gene_vocabulary_indices",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "intermediate_representation",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Equal-interval nonzero expression bins with zero preserved",
                "node_id": "bin_indices",
                "origin": "source_annotation",
                "representation_type": "expression_bin_indices",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "model_parameter",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Gene embedding-layer parameters emb_g; trainability unspecified in this section",
                "node_id": "gene_table",
                "origin": "source_annotation",
                "representation_type": "embedding_matrix",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "model_parameter",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Expression embedding-layer parameters emb_x; a separate table",
                "node_id": "bin_table",
                "origin": "source_annotation",
                "representation_type": "embedding_matrix",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "intermediate_representation",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Selected gene embedding vectors",
                "node_id": "gene_vectors",
                "origin": "source_annotation",
                "representation_type": "gene_embeddings",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "intermediate_representation",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Selected expression-bin embedding vectors",
                "node_id": "value_vectors",
                "origin": "source_annotation",
                "representation_type": "expression_embeddings",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "intermediate_representation",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Element-wise sum supplied to Transformer encoder blocks",
                "node_id": "fused",
                "origin": "source_annotation",
                "representation_type": "summed_input_embeddings",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              }
            ],
            "open_questions": [
              "Source title is marked WITHDRAWN in the inventory. This example describes the documented mechanism and changes no eligibility decision.",
              "Missing equations, HVG algorithm, matrix training status, special-token placement and phase-specific masking remain unspecified."
            ],
            "receipt_inputs": [
              {
                "node_id": "fused",
                "port_role": "input embeddings"
              }
            ],
            "recipient_component": "Transformer encoder blocks",
            "record_id": "full_2026-07-06__rec_001277",
            "source_steps": [
              {
                "component": "OKR-Cell input embedding module",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": [
                  {
                    "node_id": "expression",
                    "port_role": "values"
                  }
                ],
                "operation_type": "log_transform",
                "outputs": [
                  "logged"
                ],
                "step_id": "log",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "component": "OKR-Cell input embedding module",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": [
                  {
                    "node_id": "logged",
                    "port_role": "values"
                  },
                  {
                    "node_id": "gene_ids",
                    "port_role": "values"
                  }
                ],
                "operation_type": "select",
                "outputs": [
                  "selected_values",
                  "selected_genes"
                ],
                "step_id": "hvg",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "component": "OKR-Cell input embedding module",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": [
                  {
                    "node_id": "selected_genes",
                    "port_role": "keys"
                  },
                  {
                    "node_id": "gene_vocabulary",
                    "port_role": "table"
                  }
                ],
                "operation_type": "lookup",
                "outputs": [
                  "gene_indices"
                ],
                "step_id": "gene_token_ids",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "component": "OKR-Cell input embedding module",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": [
                  {
                    "node_id": "selected_values",
                    "port_role": "values"
                  }
                ],
                "operation_type": "bin",
                "outputs": [
                  "bin_indices"
                ],
                "step_id": "bins",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "component": "OKR-Cell input embedding module",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": [
                  {
                    "node_id": "gene_indices",
                    "port_role": "keys"
                  },
                  {
                    "node_id": "gene_table",
                    "port_role": "table"
                  }
                ],
                "operation_type": "lookup",
                "outputs": [
                  "gene_vectors"
                ],
                "step_id": "gene_lookup",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "component": "OKR-Cell input embedding module",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": [
                  {
                    "node_id": "bin_indices",
                    "port_role": "keys"
                  },
                  {
                    "node_id": "bin_table",
                    "port_role": "table"
                  }
                ],
                "operation_type": "lookup",
                "outputs": [
                  "value_vectors"
                ],
                "step_id": "value_lookup",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "component": "OKR-Cell input embedding module",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": [
                  {
                    "node_id": "gene_vectors",
                    "port_role": "values"
                  },
                  {
                    "node_id": "value_vectors",
                    "port_role": "values"
                  }
                ],
                "operation_type": "add",
                "outputs": [
                  "fused"
                ],
                "step_id": "fusion",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              }
            ],
            "task_configuration": "Section-only shared input embedding example",
            "trajectory_id": "input_embedding_section_example"
          },
          "evidence_ids": [
            "3e7bc526b93f082fe2fa11116e75653717b5fa87d40f2a62d606e3047f563504"
          ],
          "operation_id": "lookup",
          "paper": "OKR-Cell",
          "record_id": "full_2026-07-06__rec_001277",
          "scope": "supplemental mechanism example; outside four-record assembly/reuse denominators",
          "source_status": "WITHDRAWN"
        }
      ],
      "usage": [
        {
          "call_id": "embed_text::1",
          "component": "LM input embedding layer",
          "condition": null,
          "evidence_ids": [
            "660739b5499eb6ec1e09716fdb22b00a47796e7ef237c13d282f97feb739c977"
          ],
          "inputs": {
            "keys": [
              "text_tokens"
            ],
            "table": [
              "embed_text::lookup_table"
            ]
          },
          "outputs": {
            "values": [
              "text_features"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "token_indices"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_001319",
          "source_step_id": "embed_text",
          "trajectory_id": "cta_inference"
        },
        {
          "call_id": "embed_text::1",
          "component": "LM input embedding layer",
          "condition": null,
          "evidence_ids": [
            "660739b5499eb6ec1e09716fdb22b00a47796e7ef237c13d282f97feb739c977"
          ],
          "inputs": {
            "keys": [
              "text_tokens"
            ],
            "table": [
              "embed_text::lookup_table"
            ]
          },
          "outputs": {
            "values": [
              "text_features"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "token_indices"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_001319",
          "source_step_id": "embed_text",
          "trajectory_id": "dsp_inference"
        },
        {
          "call_id": "embed_english::1",
          "component": "English token embedding lookup",
          "condition": null,
          "evidence_ids": [
            "66f967ac6c448ed7cef8aa2b5fab9ae881c5cd419add6c080ee4ee75e1f46aa8"
          ],
          "inputs": {
            "keys": [
              "english_tokens"
            ],
            "table": [
              "embed_english::lookup_table"
            ]
          },
          "outputs": {
            "values": [
              "english_embeddings"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "token_ids"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "embed_english",
          "trajectory_id": "benchmark_fine_tuning"
        },
        {
          "call_id": "embed_english::1",
          "component": "English token embedding lookup",
          "condition": null,
          "evidence_ids": [
            "66f967ac6c448ed7cef8aa2b5fab9ae881c5cd419add6c080ee4ee75e1f46aa8"
          ],
          "inputs": {
            "keys": [
              "english_tokens"
            ],
            "table": [
              "embed_english::lookup_table"
            ]
          },
          "outputs": {
            "values": [
              "english_embeddings"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "token_ids"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "embed_english",
          "trajectory_id": "benchmark_inference"
        },
        {
          "call_id": "embed_english::1",
          "component": "English token embedding lookup",
          "condition": null,
          "evidence_ids": [
            "66f967ac6c448ed7cef8aa2b5fab9ae881c5cd419add6c080ee4ee75e1f46aa8"
          ],
          "inputs": {
            "keys": [
              "english_tokens"
            ],
            "table": [
              "embed_english::lookup_table"
            ]
          },
          "outputs": {
            "values": [
              "english_embeddings"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "token_ids"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "embed_english",
          "trajectory_id": "curated_dna_fine_tuning"
        },
        {
          "call_id": "embed_english::1",
          "component": "English token embedding lookup",
          "condition": null,
          "evidence_ids": [
            "66f967ac6c448ed7cef8aa2b5fab9ae881c5cd419add6c080ee4ee75e1f46aa8"
          ],
          "inputs": {
            "keys": [
              "english_tokens"
            ],
            "table": [
              "embed_english::lookup_table"
            ]
          },
          "outputs": {
            "values": [
              "english_embeddings"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "token_ids"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "embed_english",
          "trajectory_id": "curated_dna_inference"
        },
        {
          "call_id": "embed_english::1",
          "component": "English token embedding lookup",
          "condition": null,
          "evidence_ids": [
            "66f967ac6c448ed7cef8aa2b5fab9ae881c5cd419add6c080ee4ee75e1f46aa8"
          ],
          "inputs": {
            "keys": [
              "english_tokens"
            ],
            "table": [
              "embed_english::lookup_table"
            ]
          },
          "outputs": {
            "values": [
              "english_embeddings"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "token_ids"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "embed_english",
          "trajectory_id": "curated_rna_fine_tuning"
        },
        {
          "call_id": "embed_english::1",
          "component": "English token embedding lookup",
          "condition": null,
          "evidence_ids": [
            "66f967ac6c448ed7cef8aa2b5fab9ae881c5cd419add6c080ee4ee75e1f46aa8"
          ],
          "inputs": {
            "keys": [
              "english_tokens"
            ],
            "table": [
              "embed_english::lookup_table"
            ]
          },
          "outputs": {
            "values": [
              "english_embeddings"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "token_ids"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "embed_english",
          "trajectory_id": "curated_rna_inference"
        },
        {
          "call_id": "embed_english::1",
          "component": "English token embedding lookup",
          "condition": null,
          "evidence_ids": [
            "66f967ac6c448ed7cef8aa2b5fab9ae881c5cd419add6c080ee4ee75e1f46aa8"
          ],
          "inputs": {
            "keys": [
              "english_tokens"
            ],
            "table": [
              "embed_english::lookup_table"
            ]
          },
          "outputs": {
            "values": [
              "english_embeddings"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "token_ids"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "embed_english",
          "trajectory_id": "curated_protein_fine_tuning"
        },
        {
          "call_id": "embed_english::1",
          "component": "English token embedding lookup",
          "condition": null,
          "evidence_ids": [
            "66f967ac6c448ed7cef8aa2b5fab9ae881c5cd419add6c080ee4ee75e1f46aa8"
          ],
          "inputs": {
            "keys": [
              "english_tokens"
            ],
            "table": [
              "embed_english::lookup_table"
            ]
          },
          "outputs": {
            "values": [
              "english_embeddings"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "token_ids"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "embed_english",
          "trajectory_id": "curated_protein_inference"
        },
        {
          "call_id": "embed_english::1",
          "component": "English token embedding lookup",
          "condition": null,
          "evidence_ids": [
            "66f967ac6c448ed7cef8aa2b5fab9ae881c5cd419add6c080ee4ee75e1f46aa8"
          ],
          "inputs": {
            "keys": [
              "english_tokens"
            ],
            "table": [
              "embed_english::lookup_table"
            ]
          },
          "outputs": {
            "values": [
              "english_embeddings"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "token_ids"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "embed_english",
          "trajectory_id": "multiple_fine_tuning"
        },
        {
          "call_id": "embed_english::1",
          "component": "English token embedding lookup",
          "condition": null,
          "evidence_ids": [
            "66f967ac6c448ed7cef8aa2b5fab9ae881c5cd419add6c080ee4ee75e1f46aa8"
          ],
          "inputs": {
            "keys": [
              "english_tokens"
            ],
            "table": [
              "embed_english::lookup_table"
            ]
          },
          "outputs": {
            "values": [
              "english_embeddings"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "token_ids"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "embed_english",
          "trajectory_id": "multiple_inference"
        },
        {
          "call_id": "embed_english::1",
          "component": "English token embedding lookup",
          "condition": null,
          "evidence_ids": [
            "66f967ac6c448ed7cef8aa2b5fab9ae881c5cd419add6c080ee4ee75e1f46aa8"
          ],
          "inputs": {
            "keys": [
              "english_tokens"
            ],
            "table": [
              "embed_english::lookup_table"
            ]
          },
          "outputs": {
            "values": [
              "english_embeddings"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "token_ids"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "embed_english",
          "trajectory_id": "unconditioned_benchmark_fine_tuning"
        },
        {
          "call_id": "embed_english::1",
          "component": "English token embedding lookup",
          "condition": null,
          "evidence_ids": [
            "66f967ac6c448ed7cef8aa2b5fab9ae881c5cd419add6c080ee4ee75e1f46aa8"
          ],
          "inputs": {
            "keys": [
              "english_tokens"
            ],
            "table": [
              "embed_english::lookup_table"
            ]
          },
          "outputs": {
            "values": [
              "english_embeddings"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "token_ids"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "embed_english",
          "trajectory_id": "unconditioned_benchmark_inference"
        },
        {
          "call_id": "esm_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "esm_reference"
            ]
          },
          "outputs": {
            "values": [
              "esm_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_lookup",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "genept_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "genept_reference"
            ]
          },
          "outputs": {
            "values": [
              "genept_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_lookup",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "string_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "string_reference"
            ]
          },
          "outputs": {
            "values": [
              "string_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_lookup",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "depmap_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "depmap_reference"
            ]
          },
          "outputs": {
            "values": [
              "depmap_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_lookup",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "cp_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "cp_reference"
            ]
          },
          "outputs": {
            "values": [
              "cp_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_lookup",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "perturbation_gene_lookup::1",
          "component": "Shared gene identity embedding",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157",
            "b5eff5a12d80cd65dd77898ce77a73dffd581d92fed00bd344bf64c31b249ded"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "gene_table_raw"
            ]
          },
          "outputs": {
            "values": [
              "perturbation_gene_raw"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "perturbation gene identity",
              "learned embedding parameters"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "perturbation_gene_lookup",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "index_genes::1",
          "component": "Gene vocabulary collator",
          "condition": null,
          "evidence_ids": [
            "1444dee862bf8026e79488f22786da5382b640a656212364c3dafe9e00be512d",
            "301537ea3a5e15bff0703f9663ddc3693a97d9dd786ecb18693204c1bc812af2"
          ],
          "inputs": {
            "keys": [
              "selected_genes"
            ],
            "table": [
              "index_genes::lookup_table"
            ]
          },
          "outputs": {
            "values": [
              "gene_indices"
            ]
          },
          "parameters": {
            "resource_kind": "vocabulary mapping",
            "source_port_roles": [
              "selected gene identifiers"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "index_genes",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "identity_encoder::1",
          "component": "Gene identity encoder",
          "condition": null,
          "evidence_ids": [
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0"
          ],
          "inputs": {
            "keys": [
              "gene_indices"
            ],
            "table": [
              "gene_table_raw"
            ]
          },
          "outputs": {
            "values": [
              "identity_encoder::looked_up_vectors"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "gene indices",
              "shared raw embedding-table parameters; learned weights, not measured sample data"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "identity_encoder",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "mask_encoder::1",
          "component": "Perturbation mask encoder",
          "condition": null,
          "evidence_ids": [
            "d4015be6a525bdcb39076445d9d4b72f2bc20449673edb27d686b896a0d20d0a"
          ],
          "inputs": {
            "keys": [
              "training_mask"
            ],
            "table": [
              "mask_encoder::lookup_table"
            ]
          },
          "outputs": {
            "values": [
              "mask_encoder::looked_up_vectors"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "per-position reveal mask"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "mask_encoder",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "esm_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "esm_reference"
            ]
          },
          "outputs": {
            "values": [
              "esm_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_lookup",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "genept_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "genept_reference"
            ]
          },
          "outputs": {
            "values": [
              "genept_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_lookup",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "string_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "string_reference"
            ]
          },
          "outputs": {
            "values": [
              "string_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_lookup",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "depmap_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "depmap_reference"
            ]
          },
          "outputs": {
            "values": [
              "depmap_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_lookup",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "cp_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "cp_reference"
            ]
          },
          "outputs": {
            "values": [
              "cp_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_lookup",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "perturbation_gene_lookup::1",
          "component": "Shared gene identity embedding",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157",
            "b5eff5a12d80cd65dd77898ce77a73dffd581d92fed00bd344bf64c31b249ded"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "gene_table_raw"
            ]
          },
          "outputs": {
            "values": [
              "perturbation_gene_raw"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "perturbation gene identity",
              "learned embedding parameters"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "perturbation_gene_lookup",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "index_genes::1",
          "component": "Gene vocabulary collator",
          "condition": null,
          "evidence_ids": [
            "1444dee862bf8026e79488f22786da5382b640a656212364c3dafe9e00be512d",
            "301537ea3a5e15bff0703f9663ddc3693a97d9dd786ecb18693204c1bc812af2"
          ],
          "inputs": {
            "keys": [
              "selected_genes"
            ],
            "table": [
              "index_genes::lookup_table"
            ]
          },
          "outputs": {
            "values": [
              "gene_indices"
            ]
          },
          "parameters": {
            "resource_kind": "vocabulary mapping",
            "source_port_roles": [
              "selected gene identifiers"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "index_genes",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "identity_encoder::1",
          "component": "Gene identity encoder",
          "condition": null,
          "evidence_ids": [
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0"
          ],
          "inputs": {
            "keys": [
              "gene_indices"
            ],
            "table": [
              "gene_table_raw"
            ]
          },
          "outputs": {
            "values": [
              "identity_encoder::looked_up_vectors"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "gene indices",
              "shared raw embedding-table parameters; learned weights, not measured sample data"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "identity_encoder",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "mask_encoder::1",
          "component": "Perturbation mask encoder",
          "condition": null,
          "evidence_ids": [
            "d4015be6a525bdcb39076445d9d4b72f2bc20449673edb27d686b896a0d20d0a"
          ],
          "inputs": {
            "keys": [
              "training_mask"
            ],
            "table": [
              "mask_encoder::lookup_table"
            ]
          },
          "outputs": {
            "values": [
              "mask_encoder::looked_up_vectors"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "per-position reveal mask"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "mask_encoder",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "esm_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "esm_reference"
            ]
          },
          "outputs": {
            "values": [
              "esm_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_lookup",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "genept_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "genept_reference"
            ]
          },
          "outputs": {
            "values": [
              "genept_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_lookup",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "string_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "string_reference"
            ]
          },
          "outputs": {
            "values": [
              "string_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_lookup",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "depmap_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "depmap_reference"
            ]
          },
          "outputs": {
            "values": [
              "depmap_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_lookup",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "cp_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "cp_reference"
            ]
          },
          "outputs": {
            "values": [
              "cp_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_lookup",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "perturbation_gene_lookup::1",
          "component": "Shared gene identity embedding",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157",
            "b5eff5a12d80cd65dd77898ce77a73dffd581d92fed00bd344bf64c31b249ded"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "gene_table_raw"
            ]
          },
          "outputs": {
            "values": [
              "perturbation_gene_raw"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "perturbation gene identity",
              "learned embedding parameters"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "perturbation_gene_lookup",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "index_genes::1",
          "component": "Gene vocabulary collator",
          "condition": null,
          "evidence_ids": [
            "1444dee862bf8026e79488f22786da5382b640a656212364c3dafe9e00be512d",
            "301537ea3a5e15bff0703f9663ddc3693a97d9dd786ecb18693204c1bc812af2"
          ],
          "inputs": {
            "keys": [
              "selected_genes"
            ],
            "table": [
              "index_genes::lookup_table"
            ]
          },
          "outputs": {
            "values": [
              "gene_indices"
            ]
          },
          "parameters": {
            "resource_kind": "vocabulary mapping",
            "source_port_roles": [
              "selected gene identifiers"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "index_genes",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "identity_encoder::1",
          "component": "Gene identity encoder",
          "condition": null,
          "evidence_ids": [
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0"
          ],
          "inputs": {
            "keys": [
              "gene_indices"
            ],
            "table": [
              "gene_table_raw"
            ]
          },
          "outputs": {
            "values": [
              "identity_encoder::looked_up_vectors"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "gene indices",
              "shared raw embedding-table parameters; learned weights, not measured sample data"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "identity_encoder",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "mask_encoder::1",
          "component": "Perturbation mask encoder",
          "condition": null,
          "evidence_ids": [
            "d4015be6a525bdcb39076445d9d4b72f2bc20449673edb27d686b896a0d20d0a"
          ],
          "inputs": {
            "keys": [
              "training_mask"
            ],
            "table": [
              "mask_encoder::lookup_table"
            ]
          },
          "outputs": {
            "values": [
              "mask_encoder::looked_up_vectors"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "per-position reveal mask"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "mask_encoder",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "esm_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "esm_reference"
            ]
          },
          "outputs": {
            "values": [
              "esm_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_lookup",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "genept_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "genept_reference"
            ]
          },
          "outputs": {
            "values": [
              "genept_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_lookup",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "string_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "string_reference"
            ]
          },
          "outputs": {
            "values": [
              "string_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_lookup",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "depmap_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "depmap_reference"
            ]
          },
          "outputs": {
            "values": [
              "depmap_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_lookup",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "cp_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "cp_reference"
            ]
          },
          "outputs": {
            "values": [
              "cp_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_lookup",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "perturbation_gene_lookup::1",
          "component": "Shared gene identity embedding",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157",
            "b5eff5a12d80cd65dd77898ce77a73dffd581d92fed00bd344bf64c31b249ded"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "gene_table_raw"
            ]
          },
          "outputs": {
            "values": [
              "perturbation_gene_raw"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "perturbation gene identity",
              "learned embedding parameters"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "perturbation_gene_lookup",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "index_genes::1",
          "component": "Gene vocabulary collator",
          "condition": null,
          "evidence_ids": [
            "1444dee862bf8026e79488f22786da5382b640a656212364c3dafe9e00be512d",
            "301537ea3a5e15bff0703f9663ddc3693a97d9dd786ecb18693204c1bc812af2"
          ],
          "inputs": {
            "keys": [
              "selected_genes"
            ],
            "table": [
              "index_genes::lookup_table"
            ]
          },
          "outputs": {
            "values": [
              "gene_indices"
            ]
          },
          "parameters": {
            "resource_kind": "vocabulary mapping",
            "source_port_roles": [
              "selected gene identifiers"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "index_genes",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "identity_encoder::1",
          "component": "Gene identity encoder",
          "condition": null,
          "evidence_ids": [
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0"
          ],
          "inputs": {
            "keys": [
              "gene_indices"
            ],
            "table": [
              "gene_table_raw"
            ]
          },
          "outputs": {
            "values": [
              "identity_encoder::looked_up_vectors"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "gene indices",
              "shared raw embedding-table parameters; learned weights, not measured sample data"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "identity_encoder",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "mask_encoder::1",
          "component": "Perturbation mask encoder",
          "condition": null,
          "evidence_ids": [
            "d4015be6a525bdcb39076445d9d4b72f2bc20449673edb27d686b896a0d20d0a"
          ],
          "inputs": {
            "keys": [
              "training_mask"
            ],
            "table": [
              "mask_encoder::lookup_table"
            ]
          },
          "outputs": {
            "values": [
              "mask_encoder::looked_up_vectors"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "per-position reveal mask"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "mask_encoder",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "esm_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "esm_reference"
            ]
          },
          "outputs": {
            "values": [
              "esm_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_lookup",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "genept_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "genept_reference"
            ]
          },
          "outputs": {
            "values": [
              "genept_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_lookup",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "string_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "string_reference"
            ]
          },
          "outputs": {
            "values": [
              "string_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_lookup",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "depmap_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "depmap_reference"
            ]
          },
          "outputs": {
            "values": [
              "depmap_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_lookup",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "cp_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "cp_reference"
            ]
          },
          "outputs": {
            "values": [
              "cp_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_lookup",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "perturbation_gene_lookup::1",
          "component": "Shared gene identity embedding",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157",
            "b5eff5a12d80cd65dd77898ce77a73dffd581d92fed00bd344bf64c31b249ded"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "gene_table_raw"
            ]
          },
          "outputs": {
            "values": [
              "perturbation_gene_raw"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "perturbation gene identity",
              "learned embedding parameters"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "perturbation_gene_lookup",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "index_genes::1",
          "component": "Gene vocabulary collator",
          "condition": null,
          "evidence_ids": [
            "1444dee862bf8026e79488f22786da5382b640a656212364c3dafe9e00be512d",
            "301537ea3a5e15bff0703f9663ddc3693a97d9dd786ecb18693204c1bc812af2"
          ],
          "inputs": {
            "keys": [
              "selected_genes"
            ],
            "table": [
              "index_genes::lookup_table"
            ]
          },
          "outputs": {
            "values": [
              "gene_indices"
            ]
          },
          "parameters": {
            "resource_kind": "vocabulary mapping",
            "source_port_roles": [
              "selected gene identifiers"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "index_genes",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "identity_encoder::1",
          "component": "Gene identity encoder",
          "condition": null,
          "evidence_ids": [
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0"
          ],
          "inputs": {
            "keys": [
              "gene_indices"
            ],
            "table": [
              "gene_table_raw"
            ]
          },
          "outputs": {
            "values": [
              "identity_encoder::looked_up_vectors"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "gene indices",
              "shared raw embedding-table parameters; learned weights, not measured sample data"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "identity_encoder",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "mask_encoder::1",
          "component": "Perturbation mask encoder",
          "condition": null,
          "evidence_ids": [
            "d4015be6a525bdcb39076445d9d4b72f2bc20449673edb27d686b896a0d20d0a"
          ],
          "inputs": {
            "keys": [
              "training_mask"
            ],
            "table": [
              "mask_encoder::lookup_table"
            ]
          },
          "outputs": {
            "values": [
              "mask_encoder::looked_up_vectors"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "per-position reveal mask"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "mask_encoder",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "esm_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "esm_reference"
            ]
          },
          "outputs": {
            "values": [
              "esm_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_lookup",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "genept_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "genept_reference"
            ]
          },
          "outputs": {
            "values": [
              "genept_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_lookup",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "string_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "string_reference"
            ]
          },
          "outputs": {
            "values": [
              "string_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_lookup",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "depmap_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "depmap_reference"
            ]
          },
          "outputs": {
            "values": [
              "depmap_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_lookup",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "cp_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "cp_reference"
            ]
          },
          "outputs": {
            "values": [
              "cp_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_lookup",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "perturbation_gene_lookup::1",
          "component": "Shared gene identity embedding",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157",
            "b5eff5a12d80cd65dd77898ce77a73dffd581d92fed00bd344bf64c31b249ded"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "gene_table_raw"
            ]
          },
          "outputs": {
            "values": [
              "perturbation_gene_raw"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "perturbation gene identity",
              "learned embedding parameters"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "perturbation_gene_lookup",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "index_genes::1",
          "component": "Gene vocabulary collator",
          "condition": null,
          "evidence_ids": [
            "1444dee862bf8026e79488f22786da5382b640a656212364c3dafe9e00be512d",
            "301537ea3a5e15bff0703f9663ddc3693a97d9dd786ecb18693204c1bc812af2"
          ],
          "inputs": {
            "keys": [
              "selected_genes"
            ],
            "table": [
              "index_genes::lookup_table"
            ]
          },
          "outputs": {
            "values": [
              "gene_indices"
            ]
          },
          "parameters": {
            "resource_kind": "vocabulary mapping",
            "source_port_roles": [
              "selected gene identifiers"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "index_genes",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "identity_encoder::1",
          "component": "Gene identity encoder",
          "condition": null,
          "evidence_ids": [
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0"
          ],
          "inputs": {
            "keys": [
              "gene_indices"
            ],
            "table": [
              "gene_table_raw"
            ]
          },
          "outputs": {
            "values": [
              "identity_encoder::looked_up_vectors"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "gene indices",
              "shared raw embedding-table parameters; learned weights, not measured sample data"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "identity_encoder",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "mask_encoder::1",
          "component": "Perturbation mask encoder",
          "condition": null,
          "evidence_ids": [
            "d4015be6a525bdcb39076445d9d4b72f2bc20449673edb27d686b896a0d20d0a"
          ],
          "inputs": {
            "keys": [
              "training_mask"
            ],
            "table": [
              "mask_encoder::lookup_table"
            ]
          },
          "outputs": {
            "values": [
              "mask_encoder::looked_up_vectors"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "per-position reveal mask"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "mask_encoder",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "esm_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "esm_reference"
            ]
          },
          "outputs": {
            "values": [
              "esm_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_lookup",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "genept_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "genept_reference"
            ]
          },
          "outputs": {
            "values": [
              "genept_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_lookup",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "string_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "string_reference"
            ]
          },
          "outputs": {
            "values": [
              "string_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_lookup",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "depmap_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "depmap_reference"
            ]
          },
          "outputs": {
            "values": [
              "depmap_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_lookup",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "cp_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "cp_reference"
            ]
          },
          "outputs": {
            "values": [
              "cp_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_lookup",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "perturbation_gene_lookup::1",
          "component": "Shared gene identity embedding",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157",
            "b5eff5a12d80cd65dd77898ce77a73dffd581d92fed00bd344bf64c31b249ded"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "gene_table_raw"
            ]
          },
          "outputs": {
            "values": [
              "perturbation_gene_raw"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "perturbation gene identity",
              "learned embedding parameters"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "perturbation_gene_lookup",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "index_genes::1",
          "component": "Gene vocabulary collator",
          "condition": null,
          "evidence_ids": [
            "1444dee862bf8026e79488f22786da5382b640a656212364c3dafe9e00be512d",
            "301537ea3a5e15bff0703f9663ddc3693a97d9dd786ecb18693204c1bc812af2"
          ],
          "inputs": {
            "keys": [
              "selected_genes"
            ],
            "table": [
              "index_genes::lookup_table"
            ]
          },
          "outputs": {
            "values": [
              "gene_indices"
            ]
          },
          "parameters": {
            "resource_kind": "vocabulary mapping",
            "source_port_roles": [
              "selected gene identifiers"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "index_genes",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "identity_encoder::1",
          "component": "Gene identity encoder",
          "condition": null,
          "evidence_ids": [
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0"
          ],
          "inputs": {
            "keys": [
              "gene_indices"
            ],
            "table": [
              "gene_table_raw"
            ]
          },
          "outputs": {
            "values": [
              "identity_encoder::looked_up_vectors"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "gene indices",
              "shared raw embedding-table parameters; learned weights, not measured sample data"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "identity_encoder",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "mask_encoder::1",
          "component": "Perturbation mask encoder",
          "condition": null,
          "evidence_ids": [
            "d4015be6a525bdcb39076445d9d4b72f2bc20449673edb27d686b896a0d20d0a"
          ],
          "inputs": {
            "keys": [
              "training_mask"
            ],
            "table": [
              "mask_encoder::lookup_table"
            ]
          },
          "outputs": {
            "values": [
              "mask_encoder::looked_up_vectors"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "per-position reveal mask"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "mask_encoder",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "esm_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "esm_reference"
            ]
          },
          "outputs": {
            "values": [
              "esm_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_lookup",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "genept_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "genept_reference"
            ]
          },
          "outputs": {
            "values": [
              "genept_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_lookup",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "string_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "string_reference"
            ]
          },
          "outputs": {
            "values": [
              "string_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_lookup",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "depmap_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "depmap_reference"
            ]
          },
          "outputs": {
            "values": [
              "depmap_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_lookup",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "cp_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "cp_reference"
            ]
          },
          "outputs": {
            "values": [
              "cp_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_lookup",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "perturbation_gene_lookup::1",
          "component": "Shared gene identity embedding",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157",
            "b5eff5a12d80cd65dd77898ce77a73dffd581d92fed00bd344bf64c31b249ded"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "gene_table_raw"
            ]
          },
          "outputs": {
            "values": [
              "perturbation_gene_raw"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "perturbation gene identity",
              "learned embedding parameters"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "perturbation_gene_lookup",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "index_genes::1",
          "component": "Gene vocabulary collator",
          "condition": null,
          "evidence_ids": [
            "1444dee862bf8026e79488f22786da5382b640a656212364c3dafe9e00be512d",
            "301537ea3a5e15bff0703f9663ddc3693a97d9dd786ecb18693204c1bc812af2"
          ],
          "inputs": {
            "keys": [
              "selected_genes"
            ],
            "table": [
              "index_genes::lookup_table"
            ]
          },
          "outputs": {
            "values": [
              "gene_indices"
            ]
          },
          "parameters": {
            "resource_kind": "vocabulary mapping",
            "source_port_roles": [
              "selected gene identifiers"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "index_genes",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "identity_encoder::1",
          "component": "Gene identity encoder",
          "condition": null,
          "evidence_ids": [
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0"
          ],
          "inputs": {
            "keys": [
              "gene_indices"
            ],
            "table": [
              "gene_table_raw"
            ]
          },
          "outputs": {
            "values": [
              "identity_encoder::looked_up_vectors"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "gene indices",
              "shared raw embedding-table parameters; learned weights, not measured sample data"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "identity_encoder",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "mask_encoder::1",
          "component": "Perturbation mask encoder",
          "condition": null,
          "evidence_ids": [
            "d4015be6a525bdcb39076445d9d4b72f2bc20449673edb27d686b896a0d20d0a"
          ],
          "inputs": {
            "keys": [
              "training_mask"
            ],
            "table": [
              "mask_encoder::lookup_table"
            ]
          },
          "outputs": {
            "values": [
              "mask_encoder::looked_up_vectors"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "per-position reveal mask"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "mask_encoder",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "read_output_parameters::1",
          "component": "Shared gene embedding and bias lookup",
          "condition": null,
          "evidence_ids": [
            "b5eff5a12d80cd65dd77898ce77a73dffd581d92fed00bd344bf64c31b249ded"
          ],
          "inputs": {
            "keys": [
              "gene_indices"
            ],
            "table": [
              "gene_table_raw"
            ]
          },
          "outputs": {
            "values": [
              "raw_output_embeddings"
            ]
          },
          "parameters": {
            "resource_kind": "shared embedding matrix",
            "source_port_roles": [
              "output gene indices",
              "shared raw embedding weights"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "read_output_parameters",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "esm_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "esm_reference"
            ]
          },
          "outputs": {
            "values": [
              "esm_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_lookup",
          "trajectory_id": "xcell_jurkat_active_inference"
        },
        {
          "call_id": "genept_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "genept_reference"
            ]
          },
          "outputs": {
            "values": [
              "genept_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_lookup",
          "trajectory_id": "xcell_jurkat_active_inference"
        },
        {
          "call_id": "string_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "string_reference"
            ]
          },
          "outputs": {
            "values": [
              "string_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_lookup",
          "trajectory_id": "xcell_jurkat_active_inference"
        },
        {
          "call_id": "depmap_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "depmap_reference"
            ]
          },
          "outputs": {
            "values": [
              "depmap_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_lookup",
          "trajectory_id": "xcell_jurkat_active_inference"
        },
        {
          "call_id": "cp_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "cp_reference"
            ]
          },
          "outputs": {
            "values": [
              "cp_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_lookup",
          "trajectory_id": "xcell_jurkat_active_inference"
        },
        {
          "call_id": "perturbation_gene_lookup::1",
          "component": "Shared gene identity embedding",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157",
            "b5eff5a12d80cd65dd77898ce77a73dffd581d92fed00bd344bf64c31b249ded"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "gene_table_raw"
            ]
          },
          "outputs": {
            "values": [
              "perturbation_gene_raw"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "perturbation gene identity",
              "learned embedding parameters"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "perturbation_gene_lookup",
          "trajectory_id": "xcell_jurkat_active_inference"
        },
        {
          "call_id": "esm_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "esm_reference"
            ]
          },
          "outputs": {
            "values": [
              "esm_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_lookup",
          "trajectory_id": "ultra_melanocyte_post_tta_inference"
        },
        {
          "call_id": "genept_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "genept_reference"
            ]
          },
          "outputs": {
            "values": [
              "genept_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_lookup",
          "trajectory_id": "ultra_melanocyte_post_tta_inference"
        },
        {
          "call_id": "string_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "string_reference"
            ]
          },
          "outputs": {
            "values": [
              "string_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_lookup",
          "trajectory_id": "ultra_melanocyte_post_tta_inference"
        },
        {
          "call_id": "depmap_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "depmap_reference"
            ]
          },
          "outputs": {
            "values": [
              "depmap_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_lookup",
          "trajectory_id": "ultra_melanocyte_post_tta_inference"
        },
        {
          "call_id": "cp_lookup::1",
          "component": "External perturbation embedding loader",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "cp_reference"
            ]
          },
          "outputs": {
            "values": [
              "cp_loaded"
            ]
          },
          "parameters": {
            "resource_kind": "precomputed reference",
            "source_port_roles": [
              "reference collection",
              "perturbation gene lookup key"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_lookup",
          "trajectory_id": "ultra_melanocyte_post_tta_inference"
        },
        {
          "call_id": "perturbation_gene_lookup::1",
          "component": "Shared gene identity embedding",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157",
            "b5eff5a12d80cd65dd77898ce77a73dffd581d92fed00bd344bf64c31b249ded"
          ],
          "inputs": {
            "keys": [
              "p"
            ],
            "table": [
              "gene_table_raw"
            ]
          },
          "outputs": {
            "values": [
              "perturbation_gene_raw"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "perturbation gene identity",
              "learned embedding parameters"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "perturbation_gene_lookup",
          "trajectory_id": "ultra_melanocyte_post_tta_inference"
        },
        {
          "call_id": "index_genes::1",
          "component": "Gene vocabulary collator",
          "condition": null,
          "evidence_ids": [
            "1444dee862bf8026e79488f22786da5382b640a656212364c3dafe9e00be512d",
            "301537ea3a5e15bff0703f9663ddc3693a97d9dd786ecb18693204c1bc812af2"
          ],
          "inputs": {
            "keys": [
              "selected_genes"
            ],
            "table": [
              "index_genes::lookup_table"
            ]
          },
          "outputs": {
            "values": [
              "gene_indices"
            ]
          },
          "parameters": {
            "resource_kind": "vocabulary mapping",
            "source_port_roles": [
              "selected gene identifiers"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "index_genes",
          "trajectory_id": "ultra_melanocyte_tta_single_step"
        },
        {
          "call_id": "identity_encoder::1",
          "component": "Gene identity encoder",
          "condition": null,
          "evidence_ids": [
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0"
          ],
          "inputs": {
            "keys": [
              "gene_indices"
            ],
            "table": [
              "gene_table_raw"
            ]
          },
          "outputs": {
            "values": [
              "identity_encoder::looked_up_vectors"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "gene indices",
              "shared raw embedding-table parameters; learned weights, not measured sample data"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "identity_encoder",
          "trajectory_id": "ultra_melanocyte_tta_single_step"
        },
        {
          "call_id": "mask_encoder::1",
          "component": "Perturbation mask encoder",
          "condition": null,
          "evidence_ids": [
            "d4015be6a525bdcb39076445d9d4b72f2bc20449673edb27d686b896a0d20d0a"
          ],
          "inputs": {
            "keys": [
              "zero_mask"
            ],
            "table": [
              "mask_encoder::lookup_table"
            ]
          },
          "outputs": {
            "values": [
              "mask_encoder::looked_up_vectors"
            ]
          },
          "parameters": {
            "resource_kind": "embedding matrix",
            "source_port_roles": [
              "per-position reveal mask"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "mask_encoder",
          "trajectory_id": "ultra_melanocyte_tta_single_step"
        },
        {
          "call_id": "map_symbols::1",
          "component": "C2S gene-symbol mapper",
          "condition": null,
          "evidence_ids": [
            "4369c346796e71fe1a47af754545c6d56162a17e0a9e90121bd2da965f73c68a"
          ],
          "inputs": {
            "keys": [
              "gene_ids"
            ],
            "table": [
              "gtf"
            ]
          },
          "outputs": {
            "values": [
              "symbols"
            ]
          },
          "parameters": {
            "resource_kind": "identifier reference",
            "source_port_roles": [
              "measured feature IDs",
              "gene-symbol reference"
            ],
            "trainability": "unspecified at this component/phase boundary"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "map_symbols",
          "trajectory_id": "c2s_scale_fine_tuning_prompt_receipt"
        }
      ]
    },
    {
      "boundaries": "Linear or MLP realization, activation and parameter state remain instance properties. Declared post-projection normalization is a separate call.",
      "definition": "Map numerical input features into a declared output feature space using a documented learned mapping.",
      "inputs": [
        {
          "meaning": "Documented input data operands.",
          "name": "values",
          "optional": false,
          "variadic": true
        }
      ],
      "label": "Learned projection",
      "official_reference": "https://docs.pytorch.org/docs/stable/generated/torch.nn.Linear.html",
      "operation_id": "project",
      "outputs": [
        {
          "meaning": "Produced output data operands.",
          "name": "values",
          "optional": false,
          "variadic": true
        }
      ],
      "parameters": [
        "method",
        "activation",
        "parameter_state"
      ],
      "reuse_record_ids": [
        "full_2026-07-06__rec_000771",
        "full_2026-07-06__rec_001319",
        "full_2026-07-06__rec_003517"
      ],
      "status": "source_backed",
      "supplemental_examples": [],
      "usage": [
        {
          "call_id": "encode_cell::1",
          "component": "Q-Former MLP",
          "condition": null,
          "evidence_ids": [
            "24d4b535db6118301f0cd99d1fa0a145689c1fd0022901d01ec01dc7d813837e"
          ],
          "inputs": {
            "values": [
              "cell"
            ]
          },
          "outputs": {
            "values": [
              "encode_cell::mlp_output"
            ]
          },
          "parameters": {
            "method": "residual MLP",
            "source_port_roles": [
              "measured_expression"
            ]
          },
          "record_id": "full_2026-07-06__rec_001319",
          "source_step_id": "encode_cell",
          "trajectory_id": "cta_inference"
        },
        {
          "call_id": "encode_cell::1",
          "component": "Q-Former MLP",
          "condition": null,
          "evidence_ids": [
            "24d4b535db6118301f0cd99d1fa0a145689c1fd0022901d01ec01dc7d813837e"
          ],
          "inputs": {
            "values": [
              "cell"
            ]
          },
          "outputs": {
            "values": [
              "encode_cell::mlp_output"
            ]
          },
          "parameters": {
            "method": "residual MLP",
            "source_port_roles": [
              "measured_expression"
            ]
          },
          "record_id": "full_2026-07-06__rec_001319",
          "source_step_id": "encode_cell",
          "trajectory_id": "dsp_inference"
        },
        {
          "call_id": "project::1",
          "component": "LM-to-cell fully connected layer",
          "condition": null,
          "evidence_ids": [
            "f08b1f77bcbe86379154e542054b367fc3d5ab1bced317995b9c51b17d5ef647"
          ],
          "inputs": {
            "values": [
              "hidden"
            ]
          },
          "outputs": {
            "values": [
              "condition"
            ]
          },
          "parameters": {
            "method": "linear",
            "source_port_roles": [
              "upstream_hidden_state"
            ]
          },
          "record_id": "full_2026-07-06__rec_001319",
          "source_step_id": "project",
          "trajectory_id": "cpcg_training"
        },
        {
          "call_id": "project::1",
          "component": "LM-to-cell fully connected layer",
          "condition": null,
          "evidence_ids": [
            "f08b1f77bcbe86379154e542054b367fc3d5ab1bced317995b9c51b17d5ef647"
          ],
          "inputs": {
            "values": [
              "hidden"
            ]
          },
          "outputs": {
            "values": [
              "condition"
            ]
          },
          "parameters": {
            "method": "linear",
            "source_port_roles": [
              "upstream_hidden_state"
            ]
          },
          "record_id": "full_2026-07-06__rec_001319",
          "source_step_id": "project",
          "trajectory_id": "cpcg_inference"
        },
        {
          "call_id": "project_dna_a::1",
          "component": "Dense neural network in ChatNT projection model",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "encoded_a"
            ]
          },
          "outputs": {
            "values": [
              "projected_a"
            ]
          },
          "parameters": {
            "method": "source-defined learned projection",
            "source_port_roles": [
              "dna_features"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "project_dna_a",
          "trajectory_id": "benchmark_fine_tuning"
        },
        {
          "call_id": "project_dna_a::1",
          "component": "Dense neural network in ChatNT projection model",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "encoded_a"
            ]
          },
          "outputs": {
            "values": [
              "projected_a"
            ]
          },
          "parameters": {
            "method": "source-defined learned projection",
            "source_port_roles": [
              "dna_features"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "project_dna_a",
          "trajectory_id": "benchmark_inference"
        },
        {
          "call_id": "project_dna_a::1",
          "component": "Dense neural network in ChatNT projection model",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "encoded_a"
            ]
          },
          "outputs": {
            "values": [
              "projected_a"
            ]
          },
          "parameters": {
            "method": "source-defined learned projection",
            "source_port_roles": [
              "dna_features"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "project_dna_a",
          "trajectory_id": "curated_dna_fine_tuning"
        },
        {
          "call_id": "project_dna_a::1",
          "component": "Dense neural network in ChatNT projection model",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "encoded_a"
            ]
          },
          "outputs": {
            "values": [
              "projected_a"
            ]
          },
          "parameters": {
            "method": "source-defined learned projection",
            "source_port_roles": [
              "dna_features"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "project_dna_a",
          "trajectory_id": "curated_dna_inference"
        },
        {
          "call_id": "project_dna_a::1",
          "component": "Dense neural network in ChatNT projection model",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "encoded_a"
            ]
          },
          "outputs": {
            "values": [
              "projected_a"
            ]
          },
          "parameters": {
            "method": "source-defined learned projection",
            "source_port_roles": [
              "dna_features"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "project_dna_a",
          "trajectory_id": "curated_rna_fine_tuning"
        },
        {
          "call_id": "project_dna_a::1",
          "component": "Dense neural network in ChatNT projection model",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "encoded_a"
            ]
          },
          "outputs": {
            "values": [
              "projected_a"
            ]
          },
          "parameters": {
            "method": "source-defined learned projection",
            "source_port_roles": [
              "dna_features"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "project_dna_a",
          "trajectory_id": "curated_rna_inference"
        },
        {
          "call_id": "project_dna_a::1",
          "component": "Dense neural network in ChatNT projection model",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "encoded_a"
            ]
          },
          "outputs": {
            "values": [
              "projected_a"
            ]
          },
          "parameters": {
            "method": "source-defined learned projection",
            "source_port_roles": [
              "dna_features"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "project_dna_a",
          "trajectory_id": "curated_protein_fine_tuning"
        },
        {
          "call_id": "project_dna_a::1",
          "component": "Dense neural network in ChatNT projection model",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "encoded_a"
            ]
          },
          "outputs": {
            "values": [
              "projected_a"
            ]
          },
          "parameters": {
            "method": "source-defined learned projection",
            "source_port_roles": [
              "dna_features"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "project_dna_a",
          "trajectory_id": "curated_protein_inference"
        },
        {
          "call_id": "project_dna_a::1",
          "component": "Dense neural network in ChatNT projection model",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "encoded_a"
            ]
          },
          "outputs": {
            "values": [
              "projected_a"
            ]
          },
          "parameters": {
            "method": "source-defined learned projection",
            "source_port_roles": [
              "dna_features"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "project_dna_a",
          "trajectory_id": "multiple_fine_tuning"
        },
        {
          "call_id": "project_dna_b::1",
          "component": "Dense neural network in ChatNT projection model",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "encoded_b"
            ]
          },
          "outputs": {
            "values": [
              "projected_b"
            ]
          },
          "parameters": {
            "method": "source-defined learned projection",
            "source_port_roles": [
              "dna_features"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "project_dna_b",
          "trajectory_id": "multiple_fine_tuning"
        },
        {
          "call_id": "project_dna_a::1",
          "component": "Dense neural network in ChatNT projection model",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "encoded_a"
            ]
          },
          "outputs": {
            "values": [
              "projected_a"
            ]
          },
          "parameters": {
            "method": "source-defined learned projection",
            "source_port_roles": [
              "dna_features"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "project_dna_a",
          "trajectory_id": "multiple_inference"
        },
        {
          "call_id": "project_dna_b::1",
          "component": "Dense neural network in ChatNT projection model",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "encoded_b"
            ]
          },
          "outputs": {
            "values": [
              "projected_b"
            ]
          },
          "parameters": {
            "method": "source-defined learned projection",
            "source_port_roles": [
              "dna_features"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "project_dna_b",
          "trajectory_id": "multiple_inference"
        },
        {
          "call_id": "project_dna_a::1",
          "component": "Dense neural network in ChatNT projection model",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "encoded_a"
            ]
          },
          "outputs": {
            "values": [
              "projected_a"
            ]
          },
          "parameters": {
            "method": "source-defined learned projection",
            "source_port_roles": [
              "dna_features"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "project_dna_a",
          "trajectory_id": "unconditioned_benchmark_fine_tuning"
        },
        {
          "call_id": "project_dna_a::1",
          "component": "Dense neural network in ChatNT projection model",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "encoded_a"
            ]
          },
          "outputs": {
            "values": [
              "projected_a"
            ]
          },
          "parameters": {
            "method": "source-defined learned projection",
            "source_port_roles": [
              "dna_features"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "project_dna_a",
          "trajectory_id": "unconditioned_benchmark_inference"
        },
        {
          "call_id": "esm_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "esm_ready"
            ]
          },
          "outputs": {
            "values": [
              "esm_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_project",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "genept_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "genept_ready"
            ]
          },
          "outputs": {
            "values": [
              "genept_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_project",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "string_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "string_ready"
            ]
          },
          "outputs": {
            "values": [
              "string_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_project",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "depmap_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "depmap_ready"
            ]
          },
          "outputs": {
            "values": [
              "depmap_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_project",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "cp_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "cp_ready"
            ]
          },
          "outputs": {
            "values": [
              "cp_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_project",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "value_encoder::1",
          "component": "Continuous value encoder",
          "condition": null,
          "evidence_ids": [
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0"
          ],
          "inputs": {
            "values": [
              "mixed_expression"
            ]
          },
          "outputs": {
            "values": [
              "value_encoder::projected_values"
            ]
          },
          "parameters": {
            "activation": "ReLU",
            "method": "MLP",
            "source_port_roles": [
              "current expression values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "value_encoder",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "esm_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "esm_ready"
            ]
          },
          "outputs": {
            "values": [
              "esm_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_project",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "genept_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "genept_ready"
            ]
          },
          "outputs": {
            "values": [
              "genept_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_project",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "string_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "string_ready"
            ]
          },
          "outputs": {
            "values": [
              "string_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_project",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "depmap_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "depmap_ready"
            ]
          },
          "outputs": {
            "values": [
              "depmap_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_project",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "cp_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "cp_ready"
            ]
          },
          "outputs": {
            "values": [
              "cp_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_project",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "value_encoder::1",
          "component": "Continuous value encoder",
          "condition": null,
          "evidence_ids": [
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0"
          ],
          "inputs": {
            "values": [
              "mixed_expression"
            ]
          },
          "outputs": {
            "values": [
              "value_encoder::projected_values"
            ]
          },
          "parameters": {
            "activation": "ReLU",
            "method": "MLP",
            "source_port_roles": [
              "current expression values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "value_encoder",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "esm_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "esm_ready"
            ]
          },
          "outputs": {
            "values": [
              "esm_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_project",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "genept_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "genept_ready"
            ]
          },
          "outputs": {
            "values": [
              "genept_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_project",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "string_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "string_ready"
            ]
          },
          "outputs": {
            "values": [
              "string_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_project",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "depmap_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "depmap_ready"
            ]
          },
          "outputs": {
            "values": [
              "depmap_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_project",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "cp_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "cp_ready"
            ]
          },
          "outputs": {
            "values": [
              "cp_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_project",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "value_encoder::1",
          "component": "Continuous value encoder",
          "condition": null,
          "evidence_ids": [
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0"
          ],
          "inputs": {
            "values": [
              "mixed_expression"
            ]
          },
          "outputs": {
            "values": [
              "value_encoder::projected_values"
            ]
          },
          "parameters": {
            "activation": "ReLU",
            "method": "MLP",
            "source_port_roles": [
              "current expression values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "value_encoder",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "esm_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "esm_ready"
            ]
          },
          "outputs": {
            "values": [
              "esm_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_project",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "genept_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "genept_ready"
            ]
          },
          "outputs": {
            "values": [
              "genept_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_project",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "string_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "string_ready"
            ]
          },
          "outputs": {
            "values": [
              "string_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_project",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "depmap_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "depmap_ready"
            ]
          },
          "outputs": {
            "values": [
              "depmap_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_project",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "cp_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "cp_ready"
            ]
          },
          "outputs": {
            "values": [
              "cp_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_project",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "value_encoder::1",
          "component": "Continuous value encoder",
          "condition": null,
          "evidence_ids": [
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0"
          ],
          "inputs": {
            "values": [
              "mixed_expression"
            ]
          },
          "outputs": {
            "values": [
              "value_encoder::projected_values"
            ]
          },
          "parameters": {
            "activation": "ReLU",
            "method": "MLP",
            "source_port_roles": [
              "current expression values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "value_encoder",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "esm_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "esm_ready"
            ]
          },
          "outputs": {
            "values": [
              "esm_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_project",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "genept_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "genept_ready"
            ]
          },
          "outputs": {
            "values": [
              "genept_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_project",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "string_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "string_ready"
            ]
          },
          "outputs": {
            "values": [
              "string_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_project",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "depmap_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "depmap_ready"
            ]
          },
          "outputs": {
            "values": [
              "depmap_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_project",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "cp_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "cp_ready"
            ]
          },
          "outputs": {
            "values": [
              "cp_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_project",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "value_encoder::1",
          "component": "Continuous value encoder",
          "condition": null,
          "evidence_ids": [
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0"
          ],
          "inputs": {
            "values": [
              "mixed_expression"
            ]
          },
          "outputs": {
            "values": [
              "value_encoder::projected_values"
            ]
          },
          "parameters": {
            "activation": "ReLU",
            "method": "MLP",
            "source_port_roles": [
              "current expression values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "value_encoder",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "esm_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "esm_ready"
            ]
          },
          "outputs": {
            "values": [
              "esm_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_project",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "genept_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "genept_ready"
            ]
          },
          "outputs": {
            "values": [
              "genept_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_project",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "string_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "string_ready"
            ]
          },
          "outputs": {
            "values": [
              "string_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_project",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "depmap_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "depmap_ready"
            ]
          },
          "outputs": {
            "values": [
              "depmap_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_project",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "cp_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "cp_ready"
            ]
          },
          "outputs": {
            "values": [
              "cp_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_project",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "value_encoder::1",
          "component": "Continuous value encoder",
          "condition": null,
          "evidence_ids": [
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0"
          ],
          "inputs": {
            "values": [
              "mixed_expression"
            ]
          },
          "outputs": {
            "values": [
              "value_encoder::projected_values"
            ]
          },
          "parameters": {
            "activation": "ReLU",
            "method": "MLP",
            "source_port_roles": [
              "current expression values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "value_encoder",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "esm_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "esm_ready"
            ]
          },
          "outputs": {
            "values": [
              "esm_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_project",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "genept_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "genept_ready"
            ]
          },
          "outputs": {
            "values": [
              "genept_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_project",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "string_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "string_ready"
            ]
          },
          "outputs": {
            "values": [
              "string_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_project",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "depmap_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "depmap_ready"
            ]
          },
          "outputs": {
            "values": [
              "depmap_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_project",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "cp_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "cp_ready"
            ]
          },
          "outputs": {
            "values": [
              "cp_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_project",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "value_encoder::1",
          "component": "Continuous value encoder",
          "condition": null,
          "evidence_ids": [
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0"
          ],
          "inputs": {
            "values": [
              "mixed_expression"
            ]
          },
          "outputs": {
            "values": [
              "value_encoder::projected_values"
            ]
          },
          "parameters": {
            "activation": "ReLU",
            "method": "MLP",
            "source_port_roles": [
              "current expression values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "value_encoder",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "esm_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "esm_ready"
            ]
          },
          "outputs": {
            "values": [
              "esm_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_project",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "genept_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "genept_ready"
            ]
          },
          "outputs": {
            "values": [
              "genept_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_project",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "string_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "string_ready"
            ]
          },
          "outputs": {
            "values": [
              "string_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_project",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "depmap_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "depmap_ready"
            ]
          },
          "outputs": {
            "values": [
              "depmap_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_project",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "cp_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "cp_ready"
            ]
          },
          "outputs": {
            "values": [
              "cp_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_project",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "value_encoder::1",
          "component": "Continuous value encoder",
          "condition": null,
          "evidence_ids": [
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0"
          ],
          "inputs": {
            "values": [
              "mixed_expression"
            ]
          },
          "outputs": {
            "values": [
              "value_encoder::projected_values"
            ]
          },
          "parameters": {
            "activation": "ReLU",
            "method": "MLP",
            "source_port_roles": [
              "current expression values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "value_encoder",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "esm_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "esm_ready"
            ]
          },
          "outputs": {
            "values": [
              "esm_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_project",
          "trajectory_id": "xcell_jurkat_active_inference"
        },
        {
          "call_id": "genept_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "genept_ready"
            ]
          },
          "outputs": {
            "values": [
              "genept_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_project",
          "trajectory_id": "xcell_jurkat_active_inference"
        },
        {
          "call_id": "string_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "string_ready"
            ]
          },
          "outputs": {
            "values": [
              "string_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_project",
          "trajectory_id": "xcell_jurkat_active_inference"
        },
        {
          "call_id": "depmap_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "depmap_ready"
            ]
          },
          "outputs": {
            "values": [
              "depmap_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_project",
          "trajectory_id": "xcell_jurkat_active_inference"
        },
        {
          "call_id": "cp_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "cp_ready"
            ]
          },
          "outputs": {
            "values": [
              "cp_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_project",
          "trajectory_id": "xcell_jurkat_active_inference"
        },
        {
          "call_id": "esm_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "esm_ready"
            ]
          },
          "outputs": {
            "values": [
              "esm_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_project",
          "trajectory_id": "ultra_melanocyte_post_tta_inference"
        },
        {
          "call_id": "genept_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "genept_ready"
            ]
          },
          "outputs": {
            "values": [
              "genept_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_project",
          "trajectory_id": "ultra_melanocyte_post_tta_inference"
        },
        {
          "call_id": "string_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "string_ready"
            ]
          },
          "outputs": {
            "values": [
              "string_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_project",
          "trajectory_id": "ultra_melanocyte_post_tta_inference"
        },
        {
          "call_id": "depmap_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "depmap_ready"
            ]
          },
          "outputs": {
            "values": [
              "depmap_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_project",
          "trajectory_id": "ultra_melanocyte_post_tta_inference"
        },
        {
          "call_id": "cp_project::1",
          "component": "Source-specific prior MLP",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "values": [
              "cp_ready"
            ]
          },
          "outputs": {
            "values": [
              "cp_project::projected_values"
            ]
          },
          "parameters": {
            "activation": "LeakyReLU",
            "method": "MLP",
            "source_port_roles": [
              "source-specific vector"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_project",
          "trajectory_id": "ultra_melanocyte_post_tta_inference"
        },
        {
          "call_id": "value_encoder::1",
          "component": "Continuous value encoder",
          "condition": null,
          "evidence_ids": [
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0"
          ],
          "inputs": {
            "values": [
              "ntc_input"
            ]
          },
          "outputs": {
            "values": [
              "value_encoder::projected_values"
            ]
          },
          "parameters": {
            "activation": "ReLU",
            "method": "MLP",
            "source_port_roles": [
              "current expression values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "value_encoder",
          "trajectory_id": "ultra_melanocyte_tta_single_step"
        }
      ]
    },
    {
      "boundaries": "Axis and ordering require source support. Concatenation preserves distinct contributions; addition uses a separate call.",
      "definition": "Join ordered inputs along a declared axis.",
      "inputs": [
        {
          "meaning": "Documented input data operands.",
          "name": "values",
          "optional": false,
          "variadic": true
        }
      ],
      "label": "Concatenation",
      "official_reference": "https://docs.pytorch.org/docs/stable/generated/torch.cat.html",
      "operation_id": "concatenate",
      "outputs": [
        {
          "meaning": "Produced output data operands.",
          "name": "values",
          "optional": false,
          "variadic": true
        }
      ],
      "parameters": [
        "axis",
        "order"
      ],
      "reuse_record_ids": [
        "full_2026-07-06__rec_001319",
        "full_2026-07-06__rec_003517",
        "full_2026-07-06__rec_003852"
      ],
      "status": "source_backed",
      "supplemental_examples": [],
      "usage": [
        {
          "call_id": "assemble_input::1",
          "component": "InstructCell input mapping",
          "condition": null,
          "evidence_ids": [
            "742867a8658381c023c9b532b9771b309ad624dc8f4580fa4cc5a30abbe8662a"
          ],
          "inputs": {
            "values": [
              "text_features",
              "cell_features"
            ]
          },
          "outputs": {
            "values": [
              "mixed_features"
            ]
          },
          "parameters": {
            "axis": "sequence position",
            "order": "source-defined",
            "source_port_roles": [
              "text_slots",
              "cell_slots"
            ]
          },
          "record_id": "full_2026-07-06__rec_001319",
          "source_step_id": "assemble_input",
          "trajectory_id": "cta_inference"
        },
        {
          "call_id": "assemble_input::1",
          "component": "InstructCell input mapping",
          "condition": null,
          "evidence_ids": [
            "742867a8658381c023c9b532b9771b309ad624dc8f4580fa4cc5a30abbe8662a"
          ],
          "inputs": {
            "values": [
              "text_features",
              "cell_features"
            ]
          },
          "outputs": {
            "values": [
              "mixed_features"
            ]
          },
          "parameters": {
            "axis": "sequence position",
            "order": "source-defined",
            "source_port_roles": [
              "text_slots",
              "cell_slots"
            ]
          },
          "record_id": "full_2026-07-06__rec_001319",
          "source_step_id": "assemble_input",
          "trajectory_id": "dsp_inference"
        },
        {
          "call_id": "concat_t::1",
          "component": "MUPAD mIF structural conditioning",
          "condition": null,
          "evidence_ids": [
            "bca8cb6992d809fc1e66ba68b4dc2c6d0585f42fc0e7ea4326424c94fd91af39"
          ],
          "inputs": {
            "values": [
              "struct_latent",
              "noisy_mif_t"
            ]
          },
          "outputs": {
            "values": [
              "spatial_joint_t"
            ]
          },
          "parameters": {
            "axis": "unspecified",
            "order": "structural latent and current noisy state",
            "source_port_roles": [
              "fixed_structure_template",
              "changing_marker_state_at_t"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "concat_t",
          "trajectory_id": "finetune_mif_dual_condition"
        },
        {
          "call_id": "concat::1",
          "component": "Shared-CA condition assembly",
          "condition": null,
          "evidence_ids": [
            "88101d2dce98a7d30784efd305607aeef3695140dfcda29ef049d6a456235b7f"
          ],
          "inputs": {
            "values": [
              "image_condition",
              "text_condition",
              "rna_condition"
            ]
          },
          "outputs": {
            "values": [
              "joint_conditions"
            ]
          },
          "parameters": {
            "axis": "unspecified",
            "order": "source-defined modality order",
            "source_port_roles": [
              "image_embedding_source",
              "text_embedding_source",
              "rna_embedding_source"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "concat",
          "trajectory_id": "shared_ca_image_to_image_training"
        },
        {
          "call_id": "concat::1",
          "component": "Shared-CA condition assembly",
          "condition": null,
          "evidence_ids": [
            "88101d2dce98a7d30784efd305607aeef3695140dfcda29ef049d6a456235b7f"
          ],
          "inputs": {
            "values": [
              "image_condition",
              "text_condition",
              "rna_condition"
            ]
          },
          "outputs": {
            "values": [
              "joint_conditions"
            ]
          },
          "parameters": {
            "axis": "unspecified",
            "order": "source-defined modality order",
            "source_port_roles": [
              "image_embedding_source",
              "text_embedding_source",
              "rna_embedding_source"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "concat",
          "trajectory_id": "shared_ca_image_to_image_inference"
        },
        {
          "call_id": "concat::1",
          "component": "Shared-CA condition assembly",
          "condition": null,
          "evidence_ids": [
            "88101d2dce98a7d30784efd305607aeef3695140dfcda29ef049d6a456235b7f"
          ],
          "inputs": {
            "values": [
              "image_condition",
              "text_condition",
              "rna_condition"
            ]
          },
          "outputs": {
            "values": [
              "joint_conditions"
            ]
          },
          "parameters": {
            "axis": "unspecified",
            "order": "source-defined modality order",
            "source_port_roles": [
              "image_embedding_source",
              "text_embedding_source",
              "rna_embedding_source"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "concat",
          "trajectory_id": "shared_ca_text_to_image_training"
        },
        {
          "call_id": "concat::1",
          "component": "Shared-CA condition assembly",
          "condition": null,
          "evidence_ids": [
            "88101d2dce98a7d30784efd305607aeef3695140dfcda29ef049d6a456235b7f"
          ],
          "inputs": {
            "values": [
              "image_condition",
              "text_condition",
              "rna_condition"
            ]
          },
          "outputs": {
            "values": [
              "joint_conditions"
            ]
          },
          "parameters": {
            "axis": "unspecified",
            "order": "source-defined modality order",
            "source_port_roles": [
              "image_embedding_source",
              "text_embedding_source",
              "rna_embedding_source"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "concat",
          "trajectory_id": "shared_ca_text_to_image_inference"
        },
        {
          "call_id": "concat::1",
          "component": "Shared-CA condition assembly",
          "condition": null,
          "evidence_ids": [
            "88101d2dce98a7d30784efd305607aeef3695140dfcda29ef049d6a456235b7f"
          ],
          "inputs": {
            "values": [
              "image_condition",
              "text_condition",
              "rna_condition"
            ]
          },
          "outputs": {
            "values": [
              "joint_conditions"
            ]
          },
          "parameters": {
            "axis": "unspecified",
            "order": "source-defined modality order",
            "source_port_roles": [
              "image_embedding_source",
              "text_embedding_source",
              "rna_embedding_source"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "concat",
          "trajectory_id": "shared_ca_RNA_to_image_training"
        },
        {
          "call_id": "concat::1",
          "component": "Shared-CA condition assembly",
          "condition": null,
          "evidence_ids": [
            "88101d2dce98a7d30784efd305607aeef3695140dfcda29ef049d6a456235b7f"
          ],
          "inputs": {
            "values": [
              "image_condition",
              "text_condition",
              "rna_condition"
            ]
          },
          "outputs": {
            "values": [
              "joint_conditions"
            ]
          },
          "parameters": {
            "axis": "unspecified",
            "order": "source-defined modality order",
            "source_port_roles": [
              "image_embedding_source",
              "text_embedding_source",
              "rna_embedding_source"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "concat",
          "trajectory_id": "shared_ca_RNA_to_image_inference"
        },
        {
          "call_id": "prepend_cls::1",
          "component": "CLS input assembler",
          "condition": null,
          "evidence_ids": [
            "d22008964e20ecb7bd2329ec77e9bdbbbe40471a1608eaae05c8926d57004e7d",
            "1444dee862bf8026e79488f22786da5382b640a656212364c3dafe9e00be512d"
          ],
          "inputs": {
            "values": [
              "cls_parameter",
              "gene_tokens"
            ]
          },
          "outputs": {
            "values": [
              "tokens_with_cls"
            ]
          },
          "parameters": {
            "axis": "sequence position",
            "order": "CLS then gene positions",
            "source_port_roles": [
              "encoded gene tokens",
              "independent CLS parameter"
            ],
            "unexpanded_layout": "source also documents batch/cell reshaping; its placement relative to this boundary remains unexpanded"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "prepend_cls",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "prepend_cls::1",
          "component": "CLS input assembler",
          "condition": null,
          "evidence_ids": [
            "d22008964e20ecb7bd2329ec77e9bdbbbe40471a1608eaae05c8926d57004e7d",
            "1444dee862bf8026e79488f22786da5382b640a656212364c3dafe9e00be512d"
          ],
          "inputs": {
            "values": [
              "cls_parameter",
              "gene_tokens"
            ]
          },
          "outputs": {
            "values": [
              "tokens_with_cls"
            ]
          },
          "parameters": {
            "axis": "sequence position",
            "order": "CLS then gene positions",
            "source_port_roles": [
              "encoded gene tokens",
              "independent CLS parameter"
            ],
            "unexpanded_layout": "source also documents batch/cell reshaping; its placement relative to this boundary remains unexpanded"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "prepend_cls",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "prepend_cls::1",
          "component": "CLS input assembler",
          "condition": null,
          "evidence_ids": [
            "d22008964e20ecb7bd2329ec77e9bdbbbe40471a1608eaae05c8926d57004e7d",
            "1444dee862bf8026e79488f22786da5382b640a656212364c3dafe9e00be512d"
          ],
          "inputs": {
            "values": [
              "cls_parameter",
              "gene_tokens"
            ]
          },
          "outputs": {
            "values": [
              "tokens_with_cls"
            ]
          },
          "parameters": {
            "axis": "sequence position",
            "order": "CLS then gene positions",
            "source_port_roles": [
              "encoded gene tokens",
              "independent CLS parameter"
            ],
            "unexpanded_layout": "source also documents batch/cell reshaping; its placement relative to this boundary remains unexpanded"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "prepend_cls",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "prepend_cls::1",
          "component": "CLS input assembler",
          "condition": null,
          "evidence_ids": [
            "d22008964e20ecb7bd2329ec77e9bdbbbe40471a1608eaae05c8926d57004e7d",
            "1444dee862bf8026e79488f22786da5382b640a656212364c3dafe9e00be512d"
          ],
          "inputs": {
            "values": [
              "cls_parameter",
              "gene_tokens"
            ]
          },
          "outputs": {
            "values": [
              "tokens_with_cls"
            ]
          },
          "parameters": {
            "axis": "sequence position",
            "order": "CLS then gene positions",
            "source_port_roles": [
              "encoded gene tokens",
              "independent CLS parameter"
            ],
            "unexpanded_layout": "source also documents batch/cell reshaping; its placement relative to this boundary remains unexpanded"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "prepend_cls",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "prepend_cls::1",
          "component": "CLS input assembler",
          "condition": null,
          "evidence_ids": [
            "d22008964e20ecb7bd2329ec77e9bdbbbe40471a1608eaae05c8926d57004e7d",
            "1444dee862bf8026e79488f22786da5382b640a656212364c3dafe9e00be512d"
          ],
          "inputs": {
            "values": [
              "cls_parameter",
              "gene_tokens"
            ]
          },
          "outputs": {
            "values": [
              "tokens_with_cls"
            ]
          },
          "parameters": {
            "axis": "sequence position",
            "order": "CLS then gene positions",
            "source_port_roles": [
              "encoded gene tokens",
              "independent CLS parameter"
            ],
            "unexpanded_layout": "source also documents batch/cell reshaping; its placement relative to this boundary remains unexpanded"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "prepend_cls",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "prepend_cls::1",
          "component": "CLS input assembler",
          "condition": null,
          "evidence_ids": [
            "d22008964e20ecb7bd2329ec77e9bdbbbe40471a1608eaae05c8926d57004e7d",
            "1444dee862bf8026e79488f22786da5382b640a656212364c3dafe9e00be512d"
          ],
          "inputs": {
            "values": [
              "cls_parameter",
              "gene_tokens"
            ]
          },
          "outputs": {
            "values": [
              "tokens_with_cls"
            ]
          },
          "parameters": {
            "axis": "sequence position",
            "order": "CLS then gene positions",
            "source_port_roles": [
              "encoded gene tokens",
              "independent CLS parameter"
            ],
            "unexpanded_layout": "source also documents batch/cell reshaping; its placement relative to this boundary remains unexpanded"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "prepend_cls",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "prepend_cls::1",
          "component": "CLS input assembler",
          "condition": null,
          "evidence_ids": [
            "d22008964e20ecb7bd2329ec77e9bdbbbe40471a1608eaae05c8926d57004e7d",
            "1444dee862bf8026e79488f22786da5382b640a656212364c3dafe9e00be512d"
          ],
          "inputs": {
            "values": [
              "cls_parameter",
              "gene_tokens"
            ]
          },
          "outputs": {
            "values": [
              "tokens_with_cls"
            ]
          },
          "parameters": {
            "axis": "sequence position",
            "order": "CLS then gene positions",
            "source_port_roles": [
              "encoded gene tokens",
              "independent CLS parameter"
            ],
            "unexpanded_layout": "source also documents batch/cell reshaping; its placement relative to this boundary remains unexpanded"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "prepend_cls",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "concat_local_global::2",
          "component": "Expression decoder input assembler",
          "condition": null,
          "evidence_ids": [
            "ecdf72423cf768656361070c11ac6419dabdd3f6e5a178b59dcb251e2fe6dd64"
          ],
          "inputs": {
            "values": [
              "gene_state",
              "concat_local_global::expanded_cell_summary"
            ]
          },
          "outputs": {
            "values": [
              "local_global"
            ]
          },
          "parameters": {
            "axis": "feature coordinate",
            "order": "gene feature then expanded cell summary",
            "source_port_roles": [
              "local gene features",
              "global cell features"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "concat_local_global",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "prepend_cls::1",
          "component": "CLS input assembler",
          "condition": null,
          "evidence_ids": [
            "d22008964e20ecb7bd2329ec77e9bdbbbe40471a1608eaae05c8926d57004e7d",
            "1444dee862bf8026e79488f22786da5382b640a656212364c3dafe9e00be512d"
          ],
          "inputs": {
            "values": [
              "cls_parameter",
              "gene_tokens"
            ]
          },
          "outputs": {
            "values": [
              "tokens_with_cls"
            ]
          },
          "parameters": {
            "axis": "sequence position",
            "order": "CLS then gene positions",
            "source_port_roles": [
              "encoded gene tokens",
              "independent CLS parameter"
            ],
            "unexpanded_layout": "source also documents batch/cell reshaping; its placement relative to this boundary remains unexpanded"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "prepend_cls",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "concat_local_global::2",
          "component": "Expression decoder input assembler",
          "condition": null,
          "evidence_ids": [
            "ecdf72423cf768656361070c11ac6419dabdd3f6e5a178b59dcb251e2fe6dd64"
          ],
          "inputs": {
            "values": [
              "gene_state",
              "concat_local_global::expanded_cell_summary"
            ]
          },
          "outputs": {
            "values": [
              "local_global"
            ]
          },
          "parameters": {
            "axis": "feature coordinate",
            "order": "gene feature then expanded cell summary",
            "source_port_roles": [
              "local gene features",
              "global cell features"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "concat_local_global",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "prepend_cls::1",
          "component": "CLS input assembler",
          "condition": null,
          "evidence_ids": [
            "d22008964e20ecb7bd2329ec77e9bdbbbe40471a1608eaae05c8926d57004e7d",
            "1444dee862bf8026e79488f22786da5382b640a656212364c3dafe9e00be512d"
          ],
          "inputs": {
            "values": [
              "cls_parameter",
              "gene_tokens"
            ]
          },
          "outputs": {
            "values": [
              "tokens_with_cls"
            ]
          },
          "parameters": {
            "axis": "sequence position",
            "order": "CLS then gene positions",
            "source_port_roles": [
              "encoded gene tokens",
              "independent CLS parameter"
            ],
            "unexpanded_layout": "source also documents batch/cell reshaping; its placement relative to this boundary remains unexpanded"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "prepend_cls",
          "trajectory_id": "ultra_melanocyte_tta_single_step"
        }
      ]
    },
    {
      "boundaries": "Partition rule and axes are explicit; unknown tensor reshaping remains uncertain.",
      "definition": "Partition an input into selected positions, channels, patches or feature slots.",
      "inputs": [
        {
          "meaning": "Documented input data operands.",
          "name": "values",
          "optional": false,
          "variadic": true
        }
      ],
      "label": "Partition or split",
      "official_reference": "https://docs.pytorch.org/docs/stable/generated/torch.split.html",
      "operation_id": "partition",
      "outputs": [
        {
          "meaning": "Produced output data operands.",
          "name": "values",
          "optional": false,
          "variadic": true
        }
      ],
      "parameters": [
        "method",
        "axis",
        "partition_rule"
      ],
      "reuse_record_ids": [
        "full_2026-07-06__rec_001319",
        "full_2026-07-06__rec_003517",
        "full_2026-07-06__rec_003852"
      ],
      "status": "source_backed",
      "supplemental_examples": [],
      "usage": [
        {
          "call_id": "encode_cell::2",
          "component": "Q-Former MLP",
          "condition": null,
          "evidence_ids": [
            "24d4b535db6118301f0cd99d1fa0a145689c1fd0022901d01ec01dc7d813837e"
          ],
          "inputs": {
            "values": [
              "encode_cell::mlp_output"
            ]
          },
          "outputs": {
            "values": [
              "key_values"
            ]
          },
          "parameters": {
            "method": "latent feature slot split",
            "source_port_roles": [
              "measured_expression"
            ]
          },
          "record_id": "full_2026-07-06__rec_001319",
          "source_step_id": "encode_cell",
          "trajectory_id": "cta_inference"
        },
        {
          "call_id": "encode_cell::2",
          "component": "Q-Former MLP",
          "condition": null,
          "evidence_ids": [
            "24d4b535db6118301f0cd99d1fa0a145689c1fd0022901d01ec01dc7d813837e"
          ],
          "inputs": {
            "values": [
              "encode_cell::mlp_output"
            ]
          },
          "outputs": {
            "values": [
              "key_values"
            ]
          },
          "parameters": {
            "method": "latent feature slot split",
            "source_port_roles": [
              "measured_expression"
            ]
          },
          "record_id": "full_2026-07-06__rec_001319",
          "source_step_id": "encode_cell",
          "trajectory_id": "dsp_inference"
        },
        {
          "call_id": "tile::1",
          "component": "WSI tiling",
          "condition": null,
          "evidence_ids": [
            "3aac6376e2215904c435a2d32f9e11b2bcd312f398afe30b245b3570fc53af6e"
          ],
          "inputs": {
            "values": [
              "wsi"
            ]
          },
          "outputs": {
            "values": [
              "patches"
            ]
          },
          "parameters": {
            "method": "spatial tiling",
            "source_port_roles": [
              "histology_wsi"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "tile",
          "trajectory_id": "finetune_st_condition"
        },
        {
          "call_id": "tile::1",
          "component": "WSI tiling",
          "condition": null,
          "evidence_ids": [
            "3aac6376e2215904c435a2d32f9e11b2bcd312f398afe30b245b3570fc53af6e"
          ],
          "inputs": {
            "values": [
              "wsi"
            ]
          },
          "outputs": {
            "values": [
              "patches"
            ]
          },
          "parameters": {
            "method": "spatial tiling",
            "source_port_roles": [
              "histology_wsi"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "tile",
          "trajectory_id": "infer_st_condition"
        },
        {
          "call_id": "group::1",
          "component": "Multiplex channel grouping",
          "condition": null,
          "evidence_ids": [
            "bca8cb6992d809fc1e66ba68b4dc2c6d0585f42fc0e7ea4326424c94fd91af39"
          ],
          "inputs": {
            "values": [
              "mif_image"
            ]
          },
          "outputs": {
            "values": [
              "marker_groups"
            ]
          },
          "parameters": {
            "method": "channel grouping",
            "source_port_roles": [
              "multiplex_target"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "group",
          "trajectory_id": "finetune_mif_dual_condition"
        },
        {
          "call_id": "split_cls::1",
          "component": "Encoder output selector",
          "condition": null,
          "evidence_ids": [
            "ecdf72423cf768656361070c11ac6419dabdd3f6e5a178b59dcb251e2fe6dd64",
            "ab9b509d6feb33215ba4c864c73834b4de7beb52e9b9115da5295a6bd5ace00a"
          ],
          "inputs": {
            "values": [
              "final_state"
            ]
          },
          "outputs": {
            "values": [
              "gene_state",
              "cls_state"
            ]
          },
          "parameters": {
            "method": "CLS/gene position split",
            "source_port_roles": [
              "final sequence"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "split_cls",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "split_cls::1",
          "component": "Encoder output selector",
          "condition": null,
          "evidence_ids": [
            "ecdf72423cf768656361070c11ac6419dabdd3f6e5a178b59dcb251e2fe6dd64",
            "ab9b509d6feb33215ba4c864c73834b4de7beb52e9b9115da5295a6bd5ace00a"
          ],
          "inputs": {
            "values": [
              "final_state"
            ]
          },
          "outputs": {
            "values": [
              "gene_state",
              "cls_state"
            ]
          },
          "parameters": {
            "method": "CLS/gene position split",
            "source_port_roles": [
              "final sequence"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "split_cls",
          "trajectory_id": "ultra_training_tied_output_receipt"
        }
      ]
    },
    {
      "boundaries": "Criteria, selectors and randomness remain explicit instance details.",
      "definition": "Retain values or positions satisfying a documented selection rule.",
      "inputs": [
        {
          "meaning": "Documented input data operands.",
          "name": "values",
          "optional": false,
          "variadic": true
        },
        {
          "meaning": "Indices, masks or other selection controls.",
          "name": "selectors",
          "optional": true,
          "variadic": true
        }
      ],
      "label": "Selection or filtering",
      "official_reference": null,
      "operation_id": "select",
      "outputs": [
        {
          "meaning": "Produced output data operands.",
          "name": "values",
          "optional": false,
          "variadic": true
        }
      ],
      "parameters": [
        "criterion",
        "method",
        "forced_inclusions"
      ],
      "reuse_record_ids": [
        "full_2026-07-06__rec_003517",
        "full_2026-07-06__rec_003852"
      ],
      "status": "source_backed",
      "supplemental_examples": [
        {
          "assembly": {
            "bypasses": [],
            "calls": [
              {
                "call_id": "log",
                "component": "OKR-Cell input embedding module",
                "condition": null,
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": {
                  "values": [
                    "expression"
                  ]
                },
                "operation_id": "log_transform",
                "outputs": {
                  "values": [
                    "logged"
                  ]
                },
                "parameters": {
                  "method": "log1p"
                },
                "source_step_id": "log",
                "status": "documented_operation",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "call_id": "hvg",
                "component": "OKR-Cell input embedding module",
                "condition": null,
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": {
                  "values": [
                    "logged",
                    "gene_ids"
                  ]
                },
                "operation_id": "select",
                "outputs": {
                  "values": [
                    "selected_values",
                    "selected_genes"
                  ]
                },
                "parameters": {
                  "criterion": "highly variable genes",
                  "method": "unspecified"
                },
                "source_step_id": "hvg",
                "status": "documented_operation",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "call_id": "gene_token_ids",
                "component": "OKR-Cell input embedding module",
                "condition": null,
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": {
                  "keys": [
                    "selected_genes"
                  ],
                  "table": [
                    "gene_vocabulary"
                  ]
                },
                "operation_id": "lookup",
                "outputs": {
                  "values": [
                    "gene_indices"
                  ]
                },
                "parameters": {
                  "resource_kind": "gene vocabulary"
                },
                "source_step_id": "gene_token_ids",
                "status": "documented_operation",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "call_id": "bins",
                "component": "OKR-Cell input embedding module",
                "condition": null,
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": {
                  "values": [
                    "selected_values"
                  ]
                },
                "operation_id": "bin",
                "outputs": {
                  "values": [
                    "bin_indices"
                  ]
                },
                "parameters": {
                  "bin_edges": "source-defined; formula not decoded",
                  "scope": "per_cell_nonzero_values",
                  "strategy": "equal_interval",
                  "zero_handling": "retain zero"
                },
                "source_step_id": "bins",
                "status": "documented_operation",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "call_id": "gene_lookup",
                "component": "OKR-Cell input embedding module",
                "condition": null,
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": {
                  "keys": [
                    "gene_indices"
                  ],
                  "table": [
                    "gene_table"
                  ]
                },
                "operation_id": "lookup",
                "outputs": {
                  "values": [
                    "gene_vectors"
                  ]
                },
                "parameters": {
                  "resource_kind": "embedding matrix",
                  "trainability": "unspecified"
                },
                "source_step_id": "gene_lookup",
                "status": "documented_operation",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "call_id": "value_lookup",
                "component": "OKR-Cell input embedding module",
                "condition": null,
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": {
                  "keys": [
                    "bin_indices"
                  ],
                  "table": [
                    "bin_table"
                  ]
                },
                "operation_id": "lookup",
                "outputs": {
                  "values": [
                    "value_vectors"
                  ]
                },
                "parameters": {
                  "resource_kind": "embedding matrix",
                  "trainability": "unspecified"
                },
                "source_step_id": "value_lookup",
                "status": "documented_operation",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "call_id": "fusion",
                "component": "OKR-Cell input embedding module",
                "condition": null,
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": {
                  "values": [
                    "gene_vectors",
                    "value_vectors"
                  ]
                },
                "operation_id": "add",
                "outputs": {
                  "values": [
                    "fused"
                  ]
                },
                "parameters": {
                  "alignment": "selected gene position and shared embedding feature coordinate"
                },
                "source_step_id": "fusion",
                "status": "documented_operation",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              }
            ],
            "contract_version": "operation-assembly-v1",
            "evidence": [
              {
                "kind": "paper_text",
                "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                "section_id": "sec_0022"
              }
            ],
            "library_release": "1.0.1",
            "library_sha256": "8baa240c279a84a1229eefd5f34e0527291086570378e20500f37bcb9f1f2caa",
            "lifecycle_phase": "unspecified",
            "model_role": "unresolved",
            "model_variant": "OKR-Cell input module",
            "nodes": [
              {
                "axis_semantics": null,
                "contextual_role": "sample_input",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Measured single-cell expression",
                "node_id": "expression",
                "origin": "source_annotation",
                "representation_type": "expression_values",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "sample_input",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Identifiers aligned to expression positions",
                "node_id": "gene_ids",
                "origin": "source_annotation",
                "representation_type": "gene_identifiers",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "intermediate_representation",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Log1p transformed expression",
                "node_id": "logged",
                "origin": "source_annotation",
                "representation_type": "log_expression",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "intermediate_representation",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "HVG-selected expression",
                "node_id": "selected_values",
                "origin": "source_annotation",
                "representation_type": "selected_expression",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "intermediate_representation",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Identifiers of the selected genes",
                "node_id": "selected_genes",
                "origin": "source_annotation",
                "representation_type": "selected_gene_identifiers",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "reference_resource",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Gene identifier to vocabulary-index mapping",
                "node_id": "gene_vocabulary",
                "origin": "source_annotation",
                "representation_type": "vocabulary_table",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "intermediate_representation",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Vocabulary identifiers for selected genes",
                "node_id": "gene_indices",
                "origin": "source_annotation",
                "representation_type": "gene_vocabulary_indices",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "intermediate_representation",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Equal-interval nonzero expression bins with zero preserved",
                "node_id": "bin_indices",
                "origin": "source_annotation",
                "representation_type": "expression_bin_indices",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "model_parameter",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Gene embedding-layer parameters emb_g; trainability unspecified in this section",
                "node_id": "gene_table",
                "origin": "source_annotation",
                "representation_type": "embedding_matrix",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "model_parameter",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Expression embedding-layer parameters emb_x; a separate table",
                "node_id": "bin_table",
                "origin": "source_annotation",
                "representation_type": "embedding_matrix",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "intermediate_representation",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Selected gene embedding vectors",
                "node_id": "gene_vectors",
                "origin": "source_annotation",
                "representation_type": "gene_embeddings",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "intermediate_representation",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Selected expression-bin embedding vectors",
                "node_id": "value_vectors",
                "origin": "source_annotation",
                "representation_type": "expression_embeddings",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "intermediate_representation",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Element-wise sum supplied to Transformer encoder blocks",
                "node_id": "fused",
                "origin": "source_annotation",
                "representation_type": "summed_input_embeddings",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              }
            ],
            "open_questions": [
              "Source title is marked WITHDRAWN in the inventory. This example describes the documented mechanism and changes no eligibility decision.",
              "Missing equations, HVG algorithm, matrix training status, special-token placement and phase-specific masking remain unspecified."
            ],
            "receipt_inputs": [
              {
                "node_id": "fused",
                "port_role": "input embeddings"
              }
            ],
            "recipient_component": "Transformer encoder blocks",
            "record_id": "full_2026-07-06__rec_001277",
            "source_steps": [
              {
                "component": "OKR-Cell input embedding module",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": [
                  {
                    "node_id": "expression",
                    "port_role": "values"
                  }
                ],
                "operation_type": "log_transform",
                "outputs": [
                  "logged"
                ],
                "step_id": "log",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "component": "OKR-Cell input embedding module",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": [
                  {
                    "node_id": "logged",
                    "port_role": "values"
                  },
                  {
                    "node_id": "gene_ids",
                    "port_role": "values"
                  }
                ],
                "operation_type": "select",
                "outputs": [
                  "selected_values",
                  "selected_genes"
                ],
                "step_id": "hvg",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "component": "OKR-Cell input embedding module",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": [
                  {
                    "node_id": "selected_genes",
                    "port_role": "keys"
                  },
                  {
                    "node_id": "gene_vocabulary",
                    "port_role": "table"
                  }
                ],
                "operation_type": "lookup",
                "outputs": [
                  "gene_indices"
                ],
                "step_id": "gene_token_ids",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "component": "OKR-Cell input embedding module",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": [
                  {
                    "node_id": "selected_values",
                    "port_role": "values"
                  }
                ],
                "operation_type": "bin",
                "outputs": [
                  "bin_indices"
                ],
                "step_id": "bins",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "component": "OKR-Cell input embedding module",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": [
                  {
                    "node_id": "gene_indices",
                    "port_role": "keys"
                  },
                  {
                    "node_id": "gene_table",
                    "port_role": "table"
                  }
                ],
                "operation_type": "lookup",
                "outputs": [
                  "gene_vectors"
                ],
                "step_id": "gene_lookup",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "component": "OKR-Cell input embedding module",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": [
                  {
                    "node_id": "bin_indices",
                    "port_role": "keys"
                  },
                  {
                    "node_id": "bin_table",
                    "port_role": "table"
                  }
                ],
                "operation_type": "lookup",
                "outputs": [
                  "value_vectors"
                ],
                "step_id": "value_lookup",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "component": "OKR-Cell input embedding module",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": [
                  {
                    "node_id": "gene_vectors",
                    "port_role": "values"
                  },
                  {
                    "node_id": "value_vectors",
                    "port_role": "values"
                  }
                ],
                "operation_type": "add",
                "outputs": [
                  "fused"
                ],
                "step_id": "fusion",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              }
            ],
            "task_configuration": "Section-only shared input embedding example",
            "trajectory_id": "input_embedding_section_example"
          },
          "evidence_ids": [
            "3e7bc526b93f082fe2fa11116e75653717b5fa87d40f2a62d606e3047f563504"
          ],
          "operation_id": "select",
          "paper": "OKR-Cell",
          "record_id": "full_2026-07-06__rec_001277",
          "scope": "supplemental mechanism example; outside four-record assembly/reuse denominators",
          "source_status": "WITHDRAWN"
        }
      ],
      "usage": [
        {
          "call_id": "filter::1",
          "component": "RNA preprocessing",
          "condition": null,
          "evidence_ids": [
            "ed0cbc9219226da12cb74868c945901c59c51a869e0593f5bfcd24dc08383722"
          ],
          "inputs": {
            "values": [
              "rna_tpm"
            ]
          },
          "outputs": {
            "values": [
              "rna_filtered"
            ]
          },
          "parameters": {
            "criterion": "source-defined gene filter",
            "source_port_roles": [
              "normalized_expression"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "filter",
          "trajectory_id": "pretrain_multimodal_dca"
        },
        {
          "call_id": "select_genes::1",
          "component": "Gene subsampling collator",
          "condition": null,
          "evidence_ids": [
            "2a372343d3eae52243f5266cdc612ce3e829405b347fedc2a492379f56290e1d"
          ],
          "inputs": {
            "values": [
              "control_set",
              "target_set",
              "p",
              "gene_annotation_reference"
            ]
          },
          "outputs": {
            "values": [
              "selected_genes",
              "x_control",
              "x_target"
            ]
          },
          "parameters": {
            "criterion": "protein-coding and shared gene selection; aligned sources",
            "source_port_roles": [
              "control set",
              "target set",
              "forced-inclusion perturbation gene",
              "protein-coding feature-selection reference"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "select_genes",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "select_genes::1",
          "component": "Gene subsampling collator",
          "condition": null,
          "evidence_ids": [
            "2a372343d3eae52243f5266cdc612ce3e829405b347fedc2a492379f56290e1d"
          ],
          "inputs": {
            "values": [
              "control_set",
              "target_set",
              "p",
              "gene_annotation_reference"
            ]
          },
          "outputs": {
            "values": [
              "selected_genes",
              "x_control",
              "x_target"
            ]
          },
          "parameters": {
            "criterion": "protein-coding and shared gene selection; aligned sources",
            "source_port_roles": [
              "control set",
              "target set",
              "forced-inclusion perturbation gene",
              "protein-coding feature-selection reference"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "select_genes",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "select_genes::1",
          "component": "Gene subsampling collator",
          "condition": null,
          "evidence_ids": [
            "2a372343d3eae52243f5266cdc612ce3e829405b347fedc2a492379f56290e1d"
          ],
          "inputs": {
            "values": [
              "control_set",
              "target_set",
              "p",
              "gene_annotation_reference"
            ]
          },
          "outputs": {
            "values": [
              "selected_genes",
              "x_control",
              "x_target"
            ]
          },
          "parameters": {
            "criterion": "protein-coding and shared gene selection; aligned sources",
            "source_port_roles": [
              "control set",
              "target set",
              "forced-inclusion perturbation gene",
              "protein-coding feature-selection reference"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "select_genes",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "select_genes::1",
          "component": "Gene subsampling collator",
          "condition": null,
          "evidence_ids": [
            "2a372343d3eae52243f5266cdc612ce3e829405b347fedc2a492379f56290e1d"
          ],
          "inputs": {
            "values": [
              "control_set",
              "target_set",
              "p",
              "gene_annotation_reference"
            ]
          },
          "outputs": {
            "values": [
              "selected_genes",
              "x_control",
              "x_target"
            ]
          },
          "parameters": {
            "criterion": "protein-coding and shared gene selection; aligned sources",
            "source_port_roles": [
              "control set",
              "target set",
              "forced-inclusion perturbation gene",
              "protein-coding feature-selection reference"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "select_genes",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "select_genes::1",
          "component": "Gene subsampling collator",
          "condition": null,
          "evidence_ids": [
            "2a372343d3eae52243f5266cdc612ce3e829405b347fedc2a492379f56290e1d"
          ],
          "inputs": {
            "values": [
              "control_set",
              "target_set",
              "p",
              "gene_annotation_reference"
            ]
          },
          "outputs": {
            "values": [
              "selected_genes",
              "x_control",
              "x_target"
            ]
          },
          "parameters": {
            "criterion": "protein-coding and shared gene selection; aligned sources",
            "source_port_roles": [
              "control set",
              "target set",
              "forced-inclusion perturbation gene",
              "protein-coding feature-selection reference"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "select_genes",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "select_genes::1",
          "component": "Gene subsampling collator",
          "condition": null,
          "evidence_ids": [
            "2a372343d3eae52243f5266cdc612ce3e829405b347fedc2a492379f56290e1d"
          ],
          "inputs": {
            "values": [
              "control_set",
              "target_set",
              "p",
              "gene_annotation_reference"
            ]
          },
          "outputs": {
            "values": [
              "selected_genes",
              "x_control",
              "x_target"
            ]
          },
          "parameters": {
            "criterion": "protein-coding and shared gene selection; aligned sources",
            "source_port_roles": [
              "control set",
              "target set",
              "forced-inclusion perturbation gene",
              "protein-coding feature-selection reference"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "select_genes",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "select_genes::1",
          "component": "Gene subsampling collator",
          "condition": null,
          "evidence_ids": [
            "2a372343d3eae52243f5266cdc612ce3e829405b347fedc2a492379f56290e1d"
          ],
          "inputs": {
            "values": [
              "control_set",
              "target_set",
              "p",
              "gene_annotation_reference"
            ]
          },
          "outputs": {
            "values": [
              "selected_genes",
              "x_control",
              "x_target"
            ]
          },
          "parameters": {
            "criterion": "protein-coding and shared gene selection; aligned sources",
            "source_port_roles": [
              "control set",
              "target set",
              "forced-inclusion perturbation gene",
              "protein-coding feature-selection reference"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "select_genes",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "select_genes::1",
          "component": "Gene subsampling collator",
          "condition": null,
          "evidence_ids": [
            "2a372343d3eae52243f5266cdc612ce3e829405b347fedc2a492379f56290e1d"
          ],
          "inputs": {
            "values": [
              "control_set",
              "target_set",
              "p",
              "gene_annotation_reference"
            ]
          },
          "outputs": {
            "values": [
              "selected_genes",
              "x_control",
              "x_target"
            ]
          },
          "parameters": {
            "criterion": "protein-coding and shared gene selection; aligned sources",
            "source_port_roles": [
              "control set",
              "target set",
              "forced-inclusion perturbation gene",
              "protein-coding feature-selection reference"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "select_genes",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "select_new::1",
          "component": "Cumulative gene selector",
          "condition": null,
          "evidence_ids": [
            "347089ce1327f071c6790467ae19079691059ad17ba3628bb6804409718bf1d3"
          ],
          "inputs": {
            "values": [
              "ranks",
              "mask_initial",
              "next_fraction"
            ]
          },
          "outputs": {
            "values": [
              "new_positions"
            ]
          },
          "parameters": {
            "criterion": "rank threshold and previously unrevealed positions",
            "source_port_roles": [
              "fixed random gene ranks",
              "already revealed positions",
              "next scheduled threshold"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "select_new",
          "trajectory_id": "xcell_jurkat_active_inference"
        },
        {
          "call_id": "accept::2",
          "component": "Inference prediction acceptance",
          "condition": null,
          "evidence_ids": [
            "a79be46174c965e1ab88a0e4dffd45c6d51fc40851b08db84d1f3783b29e4dfd"
          ],
          "inputs": {
            "selectors": [
              "new_positions"
            ],
            "values": [
              "accept::nonnegative_predictions"
            ]
          },
          "outputs": {
            "values": [
              "accepted_values"
            ]
          },
          "parameters": {
            "criterion": "newly accepted gene positions",
            "source_port_roles": [
              "full current generated prediction",
              "newly accepted positions"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "accept",
          "trajectory_id": "xcell_jurkat_active_inference"
        },
        {
          "call_id": "select_new::1",
          "component": "Cumulative gene selector",
          "condition": null,
          "evidence_ids": [
            "347089ce1327f071c6790467ae19079691059ad17ba3628bb6804409718bf1d3"
          ],
          "inputs": {
            "values": [
              "ranks",
              "mask_initial",
              "next_fraction"
            ]
          },
          "outputs": {
            "values": [
              "new_positions"
            ]
          },
          "parameters": {
            "criterion": "rank threshold and previously unrevealed positions",
            "source_port_roles": [
              "fixed random gene ranks",
              "already revealed positions",
              "next scheduled threshold"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "select_new",
          "trajectory_id": "ultra_melanocyte_post_tta_inference"
        },
        {
          "call_id": "accept::2",
          "component": "Inference prediction acceptance",
          "condition": null,
          "evidence_ids": [
            "a79be46174c965e1ab88a0e4dffd45c6d51fc40851b08db84d1f3783b29e4dfd"
          ],
          "inputs": {
            "selectors": [
              "new_positions"
            ],
            "values": [
              "accept::nonnegative_predictions"
            ]
          },
          "outputs": {
            "values": [
              "accepted_values"
            ]
          },
          "parameters": {
            "criterion": "newly accepted gene positions",
            "source_port_roles": [
              "full current generated prediction",
              "newly accepted positions"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "accept",
          "trajectory_id": "ultra_melanocyte_post_tta_inference"
        },
        {
          "call_id": "sentence::2",
          "component": "cell2sentence converter",
          "condition": null,
          "evidence_ids": [
            "cbb198189e386a7a24affd7c4f9c9a95471d3f771849a17027c5dd357dbb620f",
            "4369c346796e71fe1a47af754545c6d56162a17e0a9e90121bd2da965f73c68a"
          ],
          "inputs": {
            "values": [
              "sentence::ranked_gene_identifiers"
            ]
          },
          "outputs": {
            "values": [
              "sentence::selected_ranked_gene_identifiers"
            ]
          },
          "parameters": {
            "criterion": "top-ranked expressed genes; source top-K",
            "source_port_roles": [
              "gene expression ranks",
              "gene-name vocabulary"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "sentence",
          "trajectory_id": "c2s_scale_fine_tuning_prompt_receipt"
        }
      ]
    },
    {
      "boundaries": "Tie handling, ascending/descending order and rank scope require evidence.",
      "definition": "Compute an ordering of values or identifiers using a declared ranking rule.",
      "inputs": [
        {
          "meaning": "Documented input data operands.",
          "name": "values",
          "optional": false,
          "variadic": true
        },
        {
          "meaning": "Identifiers explicitly aligned to the ranked values.",
          "name": "identifiers",
          "optional": true,
          "variadic": true
        }
      ],
      "label": "Ranking",
      "official_reference": null,
      "operation_id": "rank",
      "outputs": [
        {
          "meaning": "Produced output data operands.",
          "name": "values",
          "optional": false,
          "variadic": true
        }
      ],
      "parameters": [
        "order",
        "tie_handling",
        "scope"
      ],
      "reuse_record_ids": [
        "full_2026-07-06__rec_003517"
      ],
      "status": "source_backed",
      "supplemental_examples": [],
      "usage": [
        {
          "call_id": "sentence::1",
          "component": "cell2sentence converter",
          "condition": null,
          "evidence_ids": [
            "cbb198189e386a7a24affd7c4f9c9a95471d3f771849a17027c5dd357dbb620f",
            "4369c346796e71fe1a47af754545c6d56162a17e0a9e90121bd2da965f73c68a"
          ],
          "inputs": {
            "identifiers": [
              "symbols"
            ],
            "values": [
              "normalized"
            ]
          },
          "outputs": {
            "values": [
              "sentence::ranked_gene_identifiers"
            ]
          },
          "parameters": {
            "order": "descending expression",
            "source_port_roles": [
              "gene expression ranks",
              "gene-name vocabulary"
            ],
            "tie_handling": "unspecified"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "sentence",
          "trajectory_id": "c2s_scale_fine_tuning_prompt_receipt"
        }
      ]
    },
    {
      "boundaries": "Template, field order and lexical mapping are instance properties. A tokenization call follows when documented.",
      "definition": "Encode supplied data and context into an ordered textual representation.",
      "inputs": [
        {
          "meaning": "Documented input data operands.",
          "name": "values",
          "optional": false,
          "variadic": true
        }
      ],
      "label": "Text serialization",
      "official_reference": null,
      "operation_id": "serialize",
      "outputs": [
        {
          "meaning": "Produced output data operands.",
          "name": "values",
          "optional": false,
          "variadic": true
        }
      ],
      "parameters": [
        "template",
        "order",
        "lexical_mapping"
      ],
      "reuse_record_ids": [
        "full_2026-07-06__rec_003517"
      ],
      "status": "source_backed",
      "supplemental_examples": [],
      "usage": [
        {
          "call_id": "sentence::3",
          "component": "cell2sentence converter",
          "condition": null,
          "evidence_ids": [
            "cbb198189e386a7a24affd7c4f9c9a95471d3f771849a17027c5dd357dbb620f",
            "4369c346796e71fe1a47af754545c6d56162a17e0a9e90121bd2da965f73c68a"
          ],
          "inputs": {
            "values": [
              "sentence::selected_ranked_gene_identifiers"
            ]
          },
          "outputs": {
            "values": [
              "sentence"
            ]
          },
          "parameters": {
            "source_port_roles": [
              "gene expression ranks",
              "gene-name vocabulary"
            ],
            "template": "ordered gene-name sentence"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "sentence",
          "trajectory_id": "c2s_scale_fine_tuning_prompt_receipt"
        },
        {
          "call_id": "format_prompt::1",
          "component": "C2S prompt formatter",
          "condition": null,
          "evidence_ids": [
            "4369c346796e71fe1a47af754545c6d56162a17e0a9e90121bd2da965f73c68a"
          ],
          "inputs": {
            "values": [
              "sentence",
              "label",
              "expressed_gene_count"
            ]
          },
          "outputs": {
            "values": [
              "prompt"
            ]
          },
          "parameters": {
            "source_port_roles": [
              "ranked control-cell sentence",
              "perturbation label",
              "num genes template field"
            ],
            "template": "source prompt template"
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "format_prompt",
          "trajectory_id": "c2s_scale_fine_tuning_prompt_receipt"
        }
      ]
    },
    {
      "boundaries": "The symbolic target length, padding value and receiving mask remain explicit.",
      "definition": "Extend a sequence using declared padding values or special tokens.",
      "inputs": [
        {
          "meaning": "Documented input data operands.",
          "name": "values",
          "optional": false,
          "variadic": true
        }
      ],
      "label": "Padding",
      "official_reference": null,
      "operation_id": "pad",
      "outputs": [
        {
          "meaning": "Produced output data operands.",
          "name": "values",
          "optional": false,
          "variadic": true
        }
      ],
      "parameters": [
        "side",
        "value",
        "target_length"
      ],
      "reuse_record_ids": [
        "full_2026-07-06__rec_000771"
      ],
      "status": "source_backed",
      "supplemental_examples": [],
      "usage": [
        {
          "call_id": "pad_dna_a::1",
          "component": "ChatNT DNA input preparation",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "tokens_a"
            ]
          },
          "outputs": {
            "values": [
              "padded_tokens_a"
            ]
          },
          "parameters": {
            "source_port_roles": [
              "unpadded_token_ids"
            ],
            "target_length": "n_target_tokens",
            "value": "source padding token"
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "pad_dna_a",
          "trajectory_id": "benchmark_fine_tuning"
        },
        {
          "call_id": "pad_dna_a::1",
          "component": "ChatNT DNA input preparation",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "tokens_a"
            ]
          },
          "outputs": {
            "values": [
              "padded_tokens_a"
            ]
          },
          "parameters": {
            "source_port_roles": [
              "unpadded_token_ids"
            ],
            "target_length": "n_target_tokens",
            "value": "source padding token"
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "pad_dna_a",
          "trajectory_id": "benchmark_inference"
        },
        {
          "call_id": "pad_dna_a::1",
          "component": "ChatNT DNA input preparation",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "tokens_a"
            ]
          },
          "outputs": {
            "values": [
              "padded_tokens_a"
            ]
          },
          "parameters": {
            "source_port_roles": [
              "unpadded_token_ids"
            ],
            "target_length": "n_target_tokens",
            "value": "source padding token"
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "pad_dna_a",
          "trajectory_id": "curated_dna_fine_tuning"
        },
        {
          "call_id": "pad_dna_a::1",
          "component": "ChatNT DNA input preparation",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "tokens_a"
            ]
          },
          "outputs": {
            "values": [
              "padded_tokens_a"
            ]
          },
          "parameters": {
            "source_port_roles": [
              "unpadded_token_ids"
            ],
            "target_length": "n_target_tokens",
            "value": "source padding token"
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "pad_dna_a",
          "trajectory_id": "curated_dna_inference"
        },
        {
          "call_id": "pad_dna_a::1",
          "component": "ChatNT DNA input preparation",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "tokens_a"
            ]
          },
          "outputs": {
            "values": [
              "padded_tokens_a"
            ]
          },
          "parameters": {
            "source_port_roles": [
              "unpadded_token_ids"
            ],
            "target_length": "n_target_tokens",
            "value": "source padding token"
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "pad_dna_a",
          "trajectory_id": "curated_rna_fine_tuning"
        },
        {
          "call_id": "pad_dna_a::1",
          "component": "ChatNT DNA input preparation",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "tokens_a"
            ]
          },
          "outputs": {
            "values": [
              "padded_tokens_a"
            ]
          },
          "parameters": {
            "source_port_roles": [
              "unpadded_token_ids"
            ],
            "target_length": "n_target_tokens",
            "value": "source padding token"
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "pad_dna_a",
          "trajectory_id": "curated_rna_inference"
        },
        {
          "call_id": "pad_dna_a::1",
          "component": "ChatNT DNA input preparation",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "tokens_a"
            ]
          },
          "outputs": {
            "values": [
              "padded_tokens_a"
            ]
          },
          "parameters": {
            "source_port_roles": [
              "unpadded_token_ids"
            ],
            "target_length": "n_target_tokens",
            "value": "source padding token"
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "pad_dna_a",
          "trajectory_id": "curated_protein_fine_tuning"
        },
        {
          "call_id": "pad_dna_a::1",
          "component": "ChatNT DNA input preparation",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "tokens_a"
            ]
          },
          "outputs": {
            "values": [
              "padded_tokens_a"
            ]
          },
          "parameters": {
            "source_port_roles": [
              "unpadded_token_ids"
            ],
            "target_length": "n_target_tokens",
            "value": "source padding token"
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "pad_dna_a",
          "trajectory_id": "curated_protein_inference"
        },
        {
          "call_id": "pad_dna_a::1",
          "component": "ChatNT DNA input preparation",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "tokens_a"
            ]
          },
          "outputs": {
            "values": [
              "padded_tokens_a"
            ]
          },
          "parameters": {
            "source_port_roles": [
              "unpadded_token_ids"
            ],
            "target_length": "n_target_tokens",
            "value": "source padding token"
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "pad_dna_a",
          "trajectory_id": "multiple_fine_tuning"
        },
        {
          "call_id": "pad_dna_b::1",
          "component": "ChatNT DNA input preparation",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "tokens_b"
            ]
          },
          "outputs": {
            "values": [
              "padded_tokens_b"
            ]
          },
          "parameters": {
            "source_port_roles": [
              "unpadded_token_ids"
            ],
            "target_length": "n_target_tokens",
            "value": "source padding token"
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "pad_dna_b",
          "trajectory_id": "multiple_fine_tuning"
        },
        {
          "call_id": "pad_dna_a::1",
          "component": "ChatNT DNA input preparation",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "tokens_a"
            ]
          },
          "outputs": {
            "values": [
              "padded_tokens_a"
            ]
          },
          "parameters": {
            "source_port_roles": [
              "unpadded_token_ids"
            ],
            "target_length": "n_target_tokens",
            "value": "source padding token"
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "pad_dna_a",
          "trajectory_id": "multiple_inference"
        },
        {
          "call_id": "pad_dna_b::1",
          "component": "ChatNT DNA input preparation",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "tokens_b"
            ]
          },
          "outputs": {
            "values": [
              "padded_tokens_b"
            ]
          },
          "parameters": {
            "source_port_roles": [
              "unpadded_token_ids"
            ],
            "target_length": "n_target_tokens",
            "value": "source padding token"
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "pad_dna_b",
          "trajectory_id": "multiple_inference"
        },
        {
          "call_id": "pad_dna_a::1",
          "component": "ChatNT DNA input preparation",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "tokens_a"
            ]
          },
          "outputs": {
            "values": [
              "padded_tokens_a"
            ]
          },
          "parameters": {
            "source_port_roles": [
              "unpadded_token_ids"
            ],
            "target_length": "n_target_tokens",
            "value": "source padding token"
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "pad_dna_a",
          "trajectory_id": "unconditioned_benchmark_fine_tuning"
        },
        {
          "call_id": "pad_dna_a::1",
          "component": "ChatNT DNA input preparation",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f"
          ],
          "inputs": {
            "values": [
              "tokens_a"
            ]
          },
          "outputs": {
            "values": [
              "padded_tokens_a"
            ]
          },
          "parameters": {
            "source_port_roles": [
              "unpadded_token_ids"
            ],
            "target_length": "n_target_tokens",
            "value": "source padding token"
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "pad_dna_a",
          "trajectory_id": "unconditioned_benchmark_inference"
        }
      ]
    },
    {
      "boundaries": "Alignment and any scaling must be documented. Unknown fusion algebra stays unresolved.",
      "definition": "Add aligned numeric representations element by element.",
      "inputs": [
        {
          "meaning": "Documented input data operands.",
          "name": "values",
          "optional": false,
          "variadic": true
        }
      ],
      "label": "Element-wise addition",
      "official_reference": "https://docs.pytorch.org/docs/stable/generated/torch.add.html",
      "operation_id": "add",
      "outputs": [
        {
          "meaning": "Produced output data operands.",
          "name": "values",
          "optional": false,
          "variadic": true
        }
      ],
      "parameters": [
        "alignment",
        "scaling"
      ],
      "reuse_record_ids": [
        "full_2026-07-06__rec_003852"
      ],
      "status": "source_backed",
      "supplemental_examples": [
        {
          "assembly": {
            "bypasses": [],
            "calls": [
              {
                "call_id": "log",
                "component": "OKR-Cell input embedding module",
                "condition": null,
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": {
                  "values": [
                    "expression"
                  ]
                },
                "operation_id": "log_transform",
                "outputs": {
                  "values": [
                    "logged"
                  ]
                },
                "parameters": {
                  "method": "log1p"
                },
                "source_step_id": "log",
                "status": "documented_operation",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "call_id": "hvg",
                "component": "OKR-Cell input embedding module",
                "condition": null,
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": {
                  "values": [
                    "logged",
                    "gene_ids"
                  ]
                },
                "operation_id": "select",
                "outputs": {
                  "values": [
                    "selected_values",
                    "selected_genes"
                  ]
                },
                "parameters": {
                  "criterion": "highly variable genes",
                  "method": "unspecified"
                },
                "source_step_id": "hvg",
                "status": "documented_operation",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "call_id": "gene_token_ids",
                "component": "OKR-Cell input embedding module",
                "condition": null,
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": {
                  "keys": [
                    "selected_genes"
                  ],
                  "table": [
                    "gene_vocabulary"
                  ]
                },
                "operation_id": "lookup",
                "outputs": {
                  "values": [
                    "gene_indices"
                  ]
                },
                "parameters": {
                  "resource_kind": "gene vocabulary"
                },
                "source_step_id": "gene_token_ids",
                "status": "documented_operation",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "call_id": "bins",
                "component": "OKR-Cell input embedding module",
                "condition": null,
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": {
                  "values": [
                    "selected_values"
                  ]
                },
                "operation_id": "bin",
                "outputs": {
                  "values": [
                    "bin_indices"
                  ]
                },
                "parameters": {
                  "bin_edges": "source-defined; formula not decoded",
                  "scope": "per_cell_nonzero_values",
                  "strategy": "equal_interval",
                  "zero_handling": "retain zero"
                },
                "source_step_id": "bins",
                "status": "documented_operation",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "call_id": "gene_lookup",
                "component": "OKR-Cell input embedding module",
                "condition": null,
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": {
                  "keys": [
                    "gene_indices"
                  ],
                  "table": [
                    "gene_table"
                  ]
                },
                "operation_id": "lookup",
                "outputs": {
                  "values": [
                    "gene_vectors"
                  ]
                },
                "parameters": {
                  "resource_kind": "embedding matrix",
                  "trainability": "unspecified"
                },
                "source_step_id": "gene_lookup",
                "status": "documented_operation",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "call_id": "value_lookup",
                "component": "OKR-Cell input embedding module",
                "condition": null,
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": {
                  "keys": [
                    "bin_indices"
                  ],
                  "table": [
                    "bin_table"
                  ]
                },
                "operation_id": "lookup",
                "outputs": {
                  "values": [
                    "value_vectors"
                  ]
                },
                "parameters": {
                  "resource_kind": "embedding matrix",
                  "trainability": "unspecified"
                },
                "source_step_id": "value_lookup",
                "status": "documented_operation",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "call_id": "fusion",
                "component": "OKR-Cell input embedding module",
                "condition": null,
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": {
                  "values": [
                    "gene_vectors",
                    "value_vectors"
                  ]
                },
                "operation_id": "add",
                "outputs": {
                  "values": [
                    "fused"
                  ]
                },
                "parameters": {
                  "alignment": "selected gene position and shared embedding feature coordinate"
                },
                "source_step_id": "fusion",
                "status": "documented_operation",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              }
            ],
            "contract_version": "operation-assembly-v1",
            "evidence": [
              {
                "kind": "paper_text",
                "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                "section_id": "sec_0022"
              }
            ],
            "library_release": "1.0.1",
            "library_sha256": "8baa240c279a84a1229eefd5f34e0527291086570378e20500f37bcb9f1f2caa",
            "lifecycle_phase": "unspecified",
            "model_role": "unresolved",
            "model_variant": "OKR-Cell input module",
            "nodes": [
              {
                "axis_semantics": null,
                "contextual_role": "sample_input",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Measured single-cell expression",
                "node_id": "expression",
                "origin": "source_annotation",
                "representation_type": "expression_values",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "sample_input",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Identifiers aligned to expression positions",
                "node_id": "gene_ids",
                "origin": "source_annotation",
                "representation_type": "gene_identifiers",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "intermediate_representation",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Log1p transformed expression",
                "node_id": "logged",
                "origin": "source_annotation",
                "representation_type": "log_expression",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "intermediate_representation",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "HVG-selected expression",
                "node_id": "selected_values",
                "origin": "source_annotation",
                "representation_type": "selected_expression",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "intermediate_representation",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Identifiers of the selected genes",
                "node_id": "selected_genes",
                "origin": "source_annotation",
                "representation_type": "selected_gene_identifiers",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "reference_resource",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Gene identifier to vocabulary-index mapping",
                "node_id": "gene_vocabulary",
                "origin": "source_annotation",
                "representation_type": "vocabulary_table",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "intermediate_representation",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Vocabulary identifiers for selected genes",
                "node_id": "gene_indices",
                "origin": "source_annotation",
                "representation_type": "gene_vocabulary_indices",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "intermediate_representation",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Equal-interval nonzero expression bins with zero preserved",
                "node_id": "bin_indices",
                "origin": "source_annotation",
                "representation_type": "expression_bin_indices",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "model_parameter",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Gene embedding-layer parameters emb_g; trainability unspecified in this section",
                "node_id": "gene_table",
                "origin": "source_annotation",
                "representation_type": "embedding_matrix",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "model_parameter",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Expression embedding-layer parameters emb_x; a separate table",
                "node_id": "bin_table",
                "origin": "source_annotation",
                "representation_type": "embedding_matrix",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "intermediate_representation",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Selected gene embedding vectors",
                "node_id": "gene_vectors",
                "origin": "source_annotation",
                "representation_type": "gene_embeddings",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "intermediate_representation",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Selected expression-bin embedding vectors",
                "node_id": "value_vectors",
                "origin": "source_annotation",
                "representation_type": "expression_embeddings",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              },
              {
                "axis_semantics": null,
                "contextual_role": "intermediate_representation",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "information_content": "Element-wise sum supplied to Transformer encoder blocks",
                "node_id": "fused",
                "origin": "source_annotation",
                "representation_type": "summed_input_embeddings",
                "symbolic_shape": null,
                "uncertainty": "Exact tensor layout is not reconstructed in this section-only example."
              }
            ],
            "open_questions": [
              "Source title is marked WITHDRAWN in the inventory. This example describes the documented mechanism and changes no eligibility decision.",
              "Missing equations, HVG algorithm, matrix training status, special-token placement and phase-specific masking remain unspecified."
            ],
            "receipt_inputs": [
              {
                "node_id": "fused",
                "port_role": "input embeddings"
              }
            ],
            "recipient_component": "Transformer encoder blocks",
            "record_id": "full_2026-07-06__rec_001277",
            "source_steps": [
              {
                "component": "OKR-Cell input embedding module",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": [
                  {
                    "node_id": "expression",
                    "port_role": "values"
                  }
                ],
                "operation_type": "log_transform",
                "outputs": [
                  "logged"
                ],
                "step_id": "log",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "component": "OKR-Cell input embedding module",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": [
                  {
                    "node_id": "logged",
                    "port_role": "values"
                  },
                  {
                    "node_id": "gene_ids",
                    "port_role": "values"
                  }
                ],
                "operation_type": "select",
                "outputs": [
                  "selected_values",
                  "selected_genes"
                ],
                "step_id": "hvg",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "component": "OKR-Cell input embedding module",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": [
                  {
                    "node_id": "selected_genes",
                    "port_role": "keys"
                  },
                  {
                    "node_id": "gene_vocabulary",
                    "port_role": "table"
                  }
                ],
                "operation_type": "lookup",
                "outputs": [
                  "gene_indices"
                ],
                "step_id": "gene_token_ids",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "component": "OKR-Cell input embedding module",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": [
                  {
                    "node_id": "selected_values",
                    "port_role": "values"
                  }
                ],
                "operation_type": "bin",
                "outputs": [
                  "bin_indices"
                ],
                "step_id": "bins",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "component": "OKR-Cell input embedding module",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": [
                  {
                    "node_id": "gene_indices",
                    "port_role": "keys"
                  },
                  {
                    "node_id": "gene_table",
                    "port_role": "table"
                  }
                ],
                "operation_type": "lookup",
                "outputs": [
                  "gene_vectors"
                ],
                "step_id": "gene_lookup",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "component": "OKR-Cell input embedding module",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": [
                  {
                    "node_id": "bin_indices",
                    "port_role": "keys"
                  },
                  {
                    "node_id": "bin_table",
                    "port_role": "table"
                  }
                ],
                "operation_type": "lookup",
                "outputs": [
                  "value_vectors"
                ],
                "step_id": "value_lookup",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              },
              {
                "component": "OKR-Cell input embedding module",
                "evidence": [
                  {
                    "kind": "paper_text",
                    "quote": "#### 4.1.1 Input Embeddings\n\nThe input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X ∈ R N × G , where N denotes the total number of cells, G is the total count of genes, and X i,j represents the raw read count of gene j in cell i ) into a consistent latent representation through three interconnected modules, followed by feature fusion:\n\n1. Gene Tokenization. Each gene is treated as a basic information unit (analogous to a word in NLG) and assigned a unique integer ID id ( g j ). Special tokens (e.g., &lt; cls &gt; for cell representation aggregation, &lt; pad &gt; for input length padding) are included to support cross-study gene set integration. For cell i , the gene token sequence is defined as:\n\nEach gene is treated as a fundamental informational element (analogous to a token in natural language processing) and assigned a unique integer identifier id ( g j ). To support cross-study integration of gene sets, two special tokens are introduced: &lt; cls &gt; for aggregating cell-level representations and &lt; pad &gt; for padding sequences to a uniform length. For cell i , the gene token sequence is defined as:\n\n<!-- formula-not-decoded -->\n\nwhere M (predefined input length) is set to the number of selected highly variable genes (HVGs), a standard practice in single-cell transcriptomic analysis.\n\n2. Gene Expression Binning. To resolve scale inconsistencies across different sequencing batches-a challenge that cannot be fully addressed by TPM normalization or log1p transformation alone, we implement a value binning strategy with standardized parameters:\n2. 1). Preprocessing: Log1p transformation and HVG selection are performed first;\n3. 2). Binning Operation: Non-zero expression values of each cell are partitioned into O equal-interval bins, while zero values are retained as 0. The binned value x ( i ) j is formally defined as:\n\n<!-- formula-not-decoded -->\n\nThe final expression vector for cell i is x ( i ) e = [ x ( i ) 1 , x ( i ) 2 , ..., x ( i ) M ], ensuring consistent semantics of expression levels across batches.\n\n3. Embedding Fusion. Gene tokens and binned expression values are independently projected to D -dimensional vectors via embedding layers ( emb g , emb x ) respectively. The final input embedding for cell i : is obtained by element-wise summation of these two components, integrating both gene identity and expression magnitude information:\n\n<!-- formula-not-decoded -->\n\nwhere h ( i ) ∈ R M × D serves as the input to subsequent Transformer encoder blocks.\n\n",
                    "section_id": "sec_0022"
                  }
                ],
                "inputs": [
                  {
                    "node_id": "gene_vectors",
                    "port_role": "values"
                  },
                  {
                    "node_id": "value_vectors",
                    "port_role": "values"
                  }
                ],
                "operation_type": "add",
                "outputs": [
                  "fused"
                ],
                "step_id": "fusion",
                "uncertainty": "Section-only reconstruction; missing equations and training details remain unresolved."
              }
            ],
            "task_configuration": "Section-only shared input embedding example",
            "trajectory_id": "input_embedding_section_example"
          },
          "evidence_ids": [
            "3e7bc526b93f082fe2fa11116e75653717b5fa87d40f2a62d606e3047f563504"
          ],
          "operation_id": "add",
          "paper": "OKR-Cell",
          "record_id": "full_2026-07-06__rec_001277",
          "scope": "supplemental mechanism example; outside four-record assembly/reuse denominators",
          "source_status": "WITHDRAWN"
        }
      ],
      "usage": [
        {
          "call_id": "fuse::1",
          "component": "DCA modality fusion",
          "condition": null,
          "evidence_ids": [
            "a397204b6677562d7112753b0d2ca86a8eb6e62745fb7347e2b2f8070e27c88f"
          ],
          "inputs": {
            "values": [
              "image_contribution",
              "text_contribution",
              "rna_contribution"
            ]
          },
          "outputs": {
            "values": [
              "fused_contribution"
            ]
          },
          "parameters": {
            "alignment": "documented modality contributions",
            "source_port_roles": [
              "image_contribution",
              "text_contribution",
              "rna_contribution"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "fuse",
          "trajectory_id": "pretrain_multimodal_dca"
        }
      ]
    },
    {
      "boundaries": "Distribution parameters, prior/posterior origin, reparameterization and temperature remain instance details.",
      "definition": "Produce realized values or selected positions from a documented probability distribution.",
      "inputs": [
        {
          "meaning": "Distribution or its encoded parameters.",
          "name": "distribution",
          "optional": false,
          "variadic": false
        },
        {
          "meaning": "Base random values when explicitly supplied.",
          "name": "noise",
          "optional": true,
          "variadic": true
        },
        {
          "meaning": "Other documented sampling operands.",
          "name": "context",
          "optional": true,
          "variadic": true
        }
      ],
      "label": "Sampling",
      "official_reference": "https://docs.pytorch.org/docs/stable/distributions.html",
      "operation_id": "sample",
      "outputs": [
        {
          "meaning": "Produced output data operands.",
          "name": "values",
          "optional": false,
          "variadic": true
        }
      ],
      "parameters": [
        "method",
        "distribution",
        "temperature",
        "origin"
      ],
      "reuse_record_ids": [
        "full_2026-07-06__rec_000771",
        "full_2026-07-06__rec_001319"
      ],
      "status": "source_backed",
      "supplemental_examples": [],
      "usage": [
        {
          "call_id": "sample_posterior::1",
          "component": "CVAE reparameterized sampler",
          "condition": null,
          "evidence_ids": [
            "9c29bcd5ebf39c41bdb3d4c9a585c4c9dffc2fb7b99e5128a1ff7caf8ada67d3",
            "50ece99ad418c4e1e3d97f97b4a35fe850d99a67e006a67e4a4bd6a366c0a3f9"
          ],
          "inputs": {
            "distribution": [
              "posterior_parameters"
            ],
            "noise": [
              "base_noise"
            ]
          },
          "outputs": {
            "values": [
              "gene_latent",
              "library_latent"
            ]
          },
          "parameters": {
            "method": "reparameterized_sampling",
            "source_port_roles": [
              "posterior_parameters",
              "base_random_sample"
            ]
          },
          "record_id": "full_2026-07-06__rec_001319",
          "source_step_id": "sample_posterior",
          "trajectory_id": "cpcg_training"
        },
        {
          "call_id": "sample_prior::1",
          "component": "CVAE inference sampler",
          "condition": null,
          "evidence_ids": [
            "dfec5194d417e5e467322027f4d2107d549a83d0bbfc031c1494614e675b2c15"
          ],
          "inputs": {
            "distribution": [
              "prior"
            ]
          },
          "outputs": {
            "values": [
              "latent"
            ]
          },
          "parameters": {
            "method": "prior_sampling",
            "source_port_roles": [
              "sampling_distribution"
            ]
          },
          "record_id": "full_2026-07-06__rec_001319",
          "source_step_id": "sample_prior",
          "trajectory_id": "cpcg_inference"
        },
        {
          "call_id": "sample_token::1",
          "component": "ChatNT temperature sampling",
          "condition": null,
          "evidence_ids": [
            "dfdd5ff10cb42c401f0c344f517d4536c0473fa7e78688d2b07ceaca468ae4e2"
          ],
          "inputs": {
            "distribution": [
              "word_distribution_t"
            ]
          },
          "outputs": {
            "values": [
              "sampled_token_t"
            ]
          },
          "parameters": {
            "method": "temperature_token_sampling",
            "source_port_roles": [
              "next_token_probabilities"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "sample_token",
          "trajectory_id": "benchmark_inference"
        },
        {
          "call_id": "sample_token::1",
          "component": "ChatNT temperature sampling",
          "condition": null,
          "evidence_ids": [
            "dfdd5ff10cb42c401f0c344f517d4536c0473fa7e78688d2b07ceaca468ae4e2"
          ],
          "inputs": {
            "distribution": [
              "word_distribution_t"
            ]
          },
          "outputs": {
            "values": [
              "sampled_token_t"
            ]
          },
          "parameters": {
            "method": "temperature_token_sampling",
            "source_port_roles": [
              "next_token_probabilities"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "sample_token",
          "trajectory_id": "curated_dna_inference"
        },
        {
          "call_id": "sample_token::1",
          "component": "ChatNT temperature sampling",
          "condition": null,
          "evidence_ids": [
            "dfdd5ff10cb42c401f0c344f517d4536c0473fa7e78688d2b07ceaca468ae4e2"
          ],
          "inputs": {
            "distribution": [
              "word_distribution_t"
            ]
          },
          "outputs": {
            "values": [
              "sampled_token_t"
            ]
          },
          "parameters": {
            "method": "temperature_token_sampling",
            "source_port_roles": [
              "next_token_probabilities"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "sample_token",
          "trajectory_id": "curated_rna_inference"
        },
        {
          "call_id": "sample_token::1",
          "component": "ChatNT temperature sampling",
          "condition": null,
          "evidence_ids": [
            "dfdd5ff10cb42c401f0c344f517d4536c0473fa7e78688d2b07ceaca468ae4e2"
          ],
          "inputs": {
            "distribution": [
              "word_distribution_t"
            ]
          },
          "outputs": {
            "values": [
              "sampled_token_t"
            ]
          },
          "parameters": {
            "method": "temperature_token_sampling",
            "source_port_roles": [
              "next_token_probabilities"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "sample_token",
          "trajectory_id": "curated_protein_inference"
        },
        {
          "call_id": "sample_token::1",
          "component": "ChatNT temperature sampling",
          "condition": null,
          "evidence_ids": [
            "dfdd5ff10cb42c401f0c344f517d4536c0473fa7e78688d2b07ceaca468ae4e2"
          ],
          "inputs": {
            "distribution": [
              "word_distribution_t"
            ]
          },
          "outputs": {
            "values": [
              "sampled_token_t"
            ]
          },
          "parameters": {
            "method": "temperature_token_sampling",
            "source_port_roles": [
              "next_token_probabilities"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "sample_token",
          "trajectory_id": "multiple_inference"
        },
        {
          "call_id": "sample_token::1",
          "component": "ChatNT temperature sampling",
          "condition": null,
          "evidence_ids": [
            "dfdd5ff10cb42c401f0c344f517d4536c0473fa7e78688d2b07ceaca468ae4e2"
          ],
          "inputs": {
            "distribution": [
              "word_distribution_t"
            ]
          },
          "outputs": {
            "values": [
              "sampled_token_t"
            ]
          },
          "parameters": {
            "method": "temperature_token_sampling",
            "source_port_roles": [
              "next_token_probabilities"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "sample_token",
          "trajectory_id": "unconditioned_benchmark_inference"
        }
      ]
    },
    {
      "boundaries": "Query, key/value and mask are different ports. Question-dependent multi-stage resamplers retain their unresolved internal schedule.",
      "definition": "Update a query representation using a separately supplied key/value source and documented optional mask.",
      "inputs": [
        {
          "meaning": "Query source.",
          "name": "queries",
          "optional": false,
          "variadic": false
        },
        {
          "meaning": "Separately supplied source for keys and values.",
          "name": "key_values",
          "optional": false,
          "variadic": false
        },
        {
          "meaning": "Documented attention mask.",
          "name": "mask",
          "optional": true,
          "variadic": false
        }
      ],
      "label": "Cross-attention",
      "official_reference": "https://docs.pytorch.org/docs/stable/generated/torch.nn.MultiheadAttention.html",
      "operation_id": "cross_attention",
      "outputs": [
        {
          "meaning": "Produced output data operands.",
          "name": "values",
          "optional": false,
          "variadic": true
        }
      ],
      "parameters": [
        "method",
        "mask_semantics",
        "parameter_state"
      ],
      "reuse_record_ids": [
        "full_2026-07-06__rec_000771",
        "full_2026-07-06__rec_001319",
        "full_2026-07-06__rec_003517"
      ],
      "status": "source_backed",
      "supplemental_examples": [],
      "usage": [
        {
          "call_id": "resample_cell::2",
          "component": "Q-Former transformer stack",
          "condition": null,
          "evidence_ids": [
            "c3607903f64f84aedfbffbd15c827676013db20c4aa650672c05fd297d8a3cc2",
            "fa4d06cb286e061963ed42259a3cb4594e3349adc0ece6854b4c0b60d3380924"
          ],
          "inputs": {
            "key_values": [
              "key_values"
            ],
            "queries": [
              "resample_cell::self_attended_queries"
            ]
          },
          "outputs": {
            "values": [
              "cell_features"
            ]
          },
          "parameters": {
            "method": "query-to-source attention",
            "repeat_schedule": "motif repeated within the documented stack",
            "source_port_roles": [
              "initial_queries",
              "keys_and_values"
            ]
          },
          "record_id": "full_2026-07-06__rec_001319",
          "source_step_id": "resample_cell",
          "trajectory_id": "cta_inference"
        },
        {
          "call_id": "resample_cell::2",
          "component": "Q-Former transformer stack",
          "condition": null,
          "evidence_ids": [
            "c3607903f64f84aedfbffbd15c827676013db20c4aa650672c05fd297d8a3cc2",
            "fa4d06cb286e061963ed42259a3cb4594e3349adc0ece6854b4c0b60d3380924"
          ],
          "inputs": {
            "key_values": [
              "key_values"
            ],
            "queries": [
              "resample_cell::self_attended_queries"
            ]
          },
          "outputs": {
            "values": [
              "cell_features"
            ]
          },
          "parameters": {
            "method": "query-to-source attention",
            "repeat_schedule": "motif repeated within the documented stack",
            "source_port_roles": [
              "initial_queries",
              "keys_and_values"
            ]
          },
          "record_id": "full_2026-07-06__rec_001319",
          "source_step_id": "resample_cell",
          "trajectory_id": "dsp_inference"
        },
        {
          "call_id": "resample_dna_a::1",
          "component": "Unconditioned Perceiver resampler",
          "condition": null,
          "evidence_ids": [
            "cf2cb3e8e02c308c1da59bdeb751c533d2dbb31ae66e729991cc4ea2c9047352",
            "b2a310307da6950793dff7844f48ee9a07758d8c2baa00e20e9de62b6a03aa9e"
          ],
          "inputs": {
            "key_values": [
              "projected_a"
            ],
            "queries": [
              "queries_a"
            ]
          },
          "outputs": {
            "values": [
              "resampled_a"
            ]
          },
          "parameters": {
            "method": "source-defined attention/resampling",
            "source_port_roles": [
              "learnable_queries",
              "dna_source_embeddings"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "resample_dna_a",
          "trajectory_id": "unconditioned_benchmark_fine_tuning"
        },
        {
          "call_id": "resample_dna_a::1",
          "component": "Unconditioned Perceiver resampler",
          "condition": null,
          "evidence_ids": [
            "cf2cb3e8e02c308c1da59bdeb751c533d2dbb31ae66e729991cc4ea2c9047352",
            "b2a310307da6950793dff7844f48ee9a07758d8c2baa00e20e9de62b6a03aa9e"
          ],
          "inputs": {
            "key_values": [
              "projected_a"
            ],
            "queries": [
              "queries_a"
            ]
          },
          "outputs": {
            "values": [
              "resampled_a"
            ]
          },
          "parameters": {
            "method": "source-defined attention/resampling",
            "source_port_roles": [
              "learnable_queries",
              "dna_source_embeddings"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "resample_dna_a",
          "trajectory_id": "unconditioned_benchmark_inference"
        },
        {
          "call_id": "condition_state::1",
          "component": "X-Cell designated cross-attention block",
          "condition": null,
          "evidence_ids": [
            "5ecc89cd6886a88843925414941ab1210bb9ac03a1026b585fde6638ce2b017b"
          ],
          "inputs": {
            "key_values": [
              "context"
            ],
            "mask": [
              "external_mask"
            ],
            "queries": [
              "query_state"
            ]
          },
          "outputs": {
            "values": [
              "conditioned_state"
            ]
          },
          "parameters": {
            "method": "source-defined attention/resampling",
            "source_port_roles": [
              "query source",
              "key/value source",
              "key padding mask"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "condition_state",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "condition_state::1",
          "component": "X-Cell-Ultra designated cross-attention block",
          "condition": null,
          "evidence_ids": [
            "5ecc89cd6886a88843925414941ab1210bb9ac03a1026b585fde6638ce2b017b"
          ],
          "inputs": {
            "key_values": [
              "context"
            ],
            "mask": [
              "external_mask"
            ],
            "queries": [
              "query_state"
            ]
          },
          "outputs": {
            "values": [
              "conditioned_state"
            ]
          },
          "parameters": {
            "method": "source-defined attention/resampling",
            "source_port_roles": [
              "query source",
              "key/value source",
              "key padding mask"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "condition_state",
          "trajectory_id": "ultra_training_tied_output_receipt"
        }
      ]
    },
    {
      "boundaries": "Masks and repeated layer schedules remain instance details.",
      "definition": "Update a sequence using query, key and value information derived from that sequence.",
      "inputs": [
        {
          "meaning": "Documented input data operands.",
          "name": "values",
          "optional": false,
          "variadic": true
        }
      ],
      "label": "Self-attention",
      "official_reference": "https://docs.pytorch.org/docs/stable/generated/torch.nn.MultiheadAttention.html",
      "operation_id": "self_attention",
      "outputs": [
        {
          "meaning": "Produced output data operands.",
          "name": "values",
          "optional": false,
          "variadic": true
        }
      ],
      "parameters": [
        "mask_semantics",
        "repeat_schedule"
      ],
      "reuse_record_ids": [
        "full_2026-07-06__rec_001319"
      ],
      "status": "source_backed",
      "supplemental_examples": [],
      "usage": [
        {
          "call_id": "resample_cell::1",
          "component": "Q-Former transformer stack",
          "condition": null,
          "evidence_ids": [
            "c3607903f64f84aedfbffbd15c827676013db20c4aa650672c05fd297d8a3cc2",
            "fa4d06cb286e061963ed42259a3cb4594e3349adc0ece6854b4c0b60d3380924"
          ],
          "inputs": {
            "values": [
              "initial_queries"
            ]
          },
          "outputs": {
            "values": [
              "resample_cell::self_attended_queries"
            ]
          },
          "parameters": {
            "repeat_schedule": "within each documented repeated transformer block",
            "source_port_roles": [
              "initial_queries",
              "keys_and_values"
            ]
          },
          "record_id": "full_2026-07-06__rec_001319",
          "source_step_id": "resample_cell",
          "trajectory_id": "cta_inference"
        },
        {
          "call_id": "resample_cell::1",
          "component": "Q-Former transformer stack",
          "condition": null,
          "evidence_ids": [
            "c3607903f64f84aedfbffbd15c827676013db20c4aa650672c05fd297d8a3cc2",
            "fa4d06cb286e061963ed42259a3cb4594e3349adc0ece6854b4c0b60d3380924"
          ],
          "inputs": {
            "values": [
              "initial_queries"
            ]
          },
          "outputs": {
            "values": [
              "resample_cell::self_attended_queries"
            ]
          },
          "parameters": {
            "repeat_schedule": "within each documented repeated transformer block",
            "source_port_roles": [
              "initial_queries",
              "keys_and_values"
            ]
          },
          "record_id": "full_2026-07-06__rec_001319",
          "source_step_id": "resample_cell",
          "trajectory_id": "dsp_inference"
        }
      ]
    },
    {
      "boundaries": "Missingness derives from source availability. A valid zero vector supplies no missingness evidence.",
      "definition": "Replace unavailable values using a declared rule and retain documented availability information.",
      "inputs": [
        {
          "meaning": "Documented input data operands.",
          "name": "values",
          "optional": false,
          "variadic": true
        },
        {
          "meaning": "Source availability/lookup context.",
          "name": "availability_context",
          "optional": true,
          "variadic": true
        }
      ],
      "label": "Missing-value imputation",
      "official_reference": "https://scikit-learn.org/stable/modules/impute.html",
      "operation_id": "impute",
      "outputs": [
        {
          "meaning": "Produced output data operands.",
          "name": "values",
          "optional": false,
          "variadic": true
        },
        {
          "meaning": "Documented missingness indicator.",
          "name": "missingness",
          "optional": true,
          "variadic": true
        }
      ],
      "parameters": [
        "replacement",
        "availability_source"
      ],
      "reuse_record_ids": [
        "full_2026-07-06__rec_003517"
      ],
      "status": "source_backed",
      "supplemental_examples": [],
      "usage": [
        {
          "call_id": "esm_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "esm_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "esm_missing"
            ],
            "values": [
              "esm_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_availability",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "genept_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "genept_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "genept_missing"
            ],
            "values": [
              "genept_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_availability",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "string_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "string_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "string_missing"
            ],
            "values": [
              "string_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_availability",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "depmap_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "depmap_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "depmap_missing"
            ],
            "values": [
              "depmap_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_availability",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "cp_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "cp_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "cp_missing"
            ],
            "values": [
              "cp_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_availability",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "esm_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "esm_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "esm_missing"
            ],
            "values": [
              "esm_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_availability",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "genept_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "genept_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "genept_missing"
            ],
            "values": [
              "genept_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_availability",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "string_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "string_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "string_missing"
            ],
            "values": [
              "string_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_availability",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "depmap_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "depmap_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "depmap_missing"
            ],
            "values": [
              "depmap_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_availability",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "cp_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "cp_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "cp_missing"
            ],
            "values": [
              "cp_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_availability",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "esm_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "esm_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "esm_missing"
            ],
            "values": [
              "esm_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_availability",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "genept_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "genept_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "genept_missing"
            ],
            "values": [
              "genept_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_availability",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "string_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "string_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "string_missing"
            ],
            "values": [
              "string_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_availability",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "depmap_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "depmap_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "depmap_missing"
            ],
            "values": [
              "depmap_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_availability",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "cp_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "cp_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "cp_missing"
            ],
            "values": [
              "cp_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_availability",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "esm_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "esm_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "esm_missing"
            ],
            "values": [
              "esm_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_availability",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "genept_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "genept_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "genept_missing"
            ],
            "values": [
              "genept_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_availability",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "string_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "string_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "string_missing"
            ],
            "values": [
              "string_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_availability",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "depmap_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "depmap_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "depmap_missing"
            ],
            "values": [
              "depmap_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_availability",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "cp_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "cp_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "cp_missing"
            ],
            "values": [
              "cp_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_availability",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "esm_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "esm_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "esm_missing"
            ],
            "values": [
              "esm_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_availability",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "genept_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "genept_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "genept_missing"
            ],
            "values": [
              "genept_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_availability",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "string_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "string_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "string_missing"
            ],
            "values": [
              "string_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_availability",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "depmap_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "depmap_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "depmap_missing"
            ],
            "values": [
              "depmap_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_availability",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "cp_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "cp_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "cp_missing"
            ],
            "values": [
              "cp_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_availability",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "esm_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "esm_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "esm_missing"
            ],
            "values": [
              "esm_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_availability",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "genept_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "genept_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "genept_missing"
            ],
            "values": [
              "genept_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_availability",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "string_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "string_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "string_missing"
            ],
            "values": [
              "string_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_availability",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "depmap_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "depmap_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "depmap_missing"
            ],
            "values": [
              "depmap_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_availability",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "cp_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "cp_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "cp_missing"
            ],
            "values": [
              "cp_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_availability",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "esm_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "esm_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "esm_missing"
            ],
            "values": [
              "esm_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_availability",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "genept_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "genept_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "genept_missing"
            ],
            "values": [
              "genept_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_availability",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "string_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "string_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "string_missing"
            ],
            "values": [
              "string_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_availability",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "depmap_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "depmap_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "depmap_missing"
            ],
            "values": [
              "depmap_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_availability",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "cp_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "cp_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "cp_missing"
            ],
            "values": [
              "cp_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_availability",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "esm_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "esm_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "esm_missing"
            ],
            "values": [
              "esm_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_availability",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "genept_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "genept_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "genept_missing"
            ],
            "values": [
              "genept_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_availability",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "string_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "string_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "string_missing"
            ],
            "values": [
              "string_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_availability",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "depmap_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "depmap_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "depmap_missing"
            ],
            "values": [
              "depmap_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_availability",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "cp_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "cp_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "cp_missing"
            ],
            "values": [
              "cp_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_availability",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "esm_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "esm_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "esm_missing"
            ],
            "values": [
              "esm_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_availability",
          "trajectory_id": "xcell_jurkat_active_inference"
        },
        {
          "call_id": "genept_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "genept_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "genept_missing"
            ],
            "values": [
              "genept_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_availability",
          "trajectory_id": "xcell_jurkat_active_inference"
        },
        {
          "call_id": "string_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "string_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "string_missing"
            ],
            "values": [
              "string_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_availability",
          "trajectory_id": "xcell_jurkat_active_inference"
        },
        {
          "call_id": "depmap_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "depmap_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "depmap_missing"
            ],
            "values": [
              "depmap_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_availability",
          "trajectory_id": "xcell_jurkat_active_inference"
        },
        {
          "call_id": "cp_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "cp_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "cp_missing"
            ],
            "values": [
              "cp_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_availability",
          "trajectory_id": "xcell_jurkat_active_inference"
        },
        {
          "call_id": "esm_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "esm_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "esm_missing"
            ],
            "values": [
              "esm_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "esm_availability",
          "trajectory_id": "ultra_melanocyte_post_tta_inference"
        },
        {
          "call_id": "genept_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "genept_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "genept_missing"
            ],
            "values": [
              "genept_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "genept_availability",
          "trajectory_id": "ultra_melanocyte_post_tta_inference"
        },
        {
          "call_id": "string_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "string_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "string_missing"
            ],
            "values": [
              "string_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "string_availability",
          "trajectory_id": "ultra_melanocyte_post_tta_inference"
        },
        {
          "call_id": "depmap_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "depmap_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "depmap_missing"
            ],
            "values": [
              "depmap_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "depmap_availability",
          "trajectory_id": "ultra_melanocyte_post_tta_inference"
        },
        {
          "call_id": "cp_availability::2",
          "component": "External embedding collator",
          "condition": null,
          "evidence_ids": [
            "7ff9a77702f6f4b508e4c4679c1d468e6cd23edb3bd4e7f78ff5795b695c11f1"
          ],
          "inputs": {
            "availability_context": [
              "p"
            ],
            "values": [
              "cp_availability::optionally_normalized_reference"
            ]
          },
          "outputs": {
            "missingness": [
              "cp_missing"
            ],
            "values": [
              "cp_ready"
            ]
          },
          "parameters": {
            "availability_source": "source lookup presence",
            "replacement": "zero",
            "source_port_roles": [
              "loaded source vector",
              "lookup availability identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "cp_availability",
          "trajectory_id": "ultra_melanocyte_post_tta_inference"
        }
      ]
    },
    {
      "boundaries": "Bounds belong to the actual documented use; selection follows as another call.",
      "definition": "Restrict numeric values to declared bounds.",
      "inputs": [
        {
          "meaning": "Documented input data operands.",
          "name": "values",
          "optional": false,
          "variadic": true
        }
      ],
      "label": "Clamping",
      "official_reference": "https://docs.pytorch.org/docs/stable/generated/torch.clamp.html",
      "operation_id": "clamp",
      "outputs": [
        {
          "meaning": "Produced output data operands.",
          "name": "values",
          "optional": false,
          "variadic": true
        }
      ],
      "parameters": [
        "lower",
        "upper"
      ],
      "reuse_record_ids": [
        "full_2026-07-06__rec_003517"
      ],
      "status": "source_backed",
      "supplemental_examples": [],
      "usage": [
        {
          "call_id": "accept::1",
          "component": "Inference prediction acceptance",
          "condition": null,
          "evidence_ids": [
            "a79be46174c965e1ab88a0e4dffd45c6d51fc40851b08db84d1f3783b29e4dfd"
          ],
          "inputs": {
            "values": [
              "round_prediction"
            ]
          },
          "outputs": {
            "values": [
              "accept::nonnegative_predictions"
            ]
          },
          "parameters": {
            "lower": "zero",
            "source_port_roles": [
              "full current generated prediction",
              "newly accepted positions"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "accept",
          "trajectory_id": "xcell_jurkat_active_inference"
        },
        {
          "call_id": "accept::1",
          "component": "Inference prediction acceptance",
          "condition": null,
          "evidence_ids": [
            "a79be46174c965e1ab88a0e4dffd45c6d51fc40851b08db84d1f3783b29e4dfd"
          ],
          "inputs": {
            "values": [
              "round_prediction"
            ]
          },
          "outputs": {
            "values": [
              "accept::nonnegative_predictions"
            ]
          },
          "parameters": {
            "lower": "zero",
            "source_port_roles": [
              "full current generated prediction",
              "newly accepted positions"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "accept",
          "trajectory_id": "ultra_melanocyte_post_tta_inference"
        }
      ]
    },
    {
      "boundaries": "Indices, prior state, replacement values and updated masks stay distinct operands. Recurrent use remains phase scoped.",
      "definition": "Write supplied replacement values into designated positions while retaining documented unaffected values.",
      "inputs": [
        {
          "meaning": "Prior data state.",
          "name": "state",
          "optional": false,
          "variadic": true
        },
        {
          "meaning": "New values.",
          "name": "replacements",
          "optional": false,
          "variadic": true
        },
        {
          "meaning": "Update positions or masks.",
          "name": "indices",
          "optional": false,
          "variadic": true
        },
        {
          "meaning": "Other source-defined controls.",
          "name": "context",
          "optional": true,
          "variadic": true
        }
      ],
      "label": "Indexed state update",
      "official_reference": "https://docs.pytorch.org/docs/stable/generated/torch.Tensor.scatter_.html",
      "operation_id": "update",
      "outputs": [
        {
          "meaning": "Produced output data operands.",
          "name": "values",
          "optional": false,
          "variadic": true
        }
      ],
      "parameters": [
        "method",
        "preserve_unselected"
      ],
      "reuse_record_ids": [
        "full_2026-07-06__rec_003517"
      ],
      "status": "source_backed",
      "supplemental_examples": [],
      "usage": [
        {
          "call_id": "reveal_target::1",
          "component": "Training expression mixer",
          "condition": null,
          "evidence_ids": [
            "bfc3cb3b89891fce1c3c87ec4711445ee7c610c3539f6ac41e63bc6e4df8134f",
            "b2219fd688af43a3c418ed9b0da8421573eff83e4c8cbb0ad2816047f40de556"
          ],
          "inputs": {
            "context": [
              "replacement_fraction"
            ],
            "indices": [
              "reveal_positions"
            ],
            "replacements": [
              "x_target"
            ],
            "state": [
              "x_control"
            ]
          },
          "outputs": {
            "values": [
              "training_mask",
              "mixed_expression"
            ]
          },
          "parameters": {
            "method": "indexed replacement",
            "preserve_unselected": true,
            "source_port_roles": [
              "unrevealed source values",
              "revealed ground-truth source values",
              "sampled revealed positions",
              "reveal strategy"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "reveal_target",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "reveal_target::1",
          "component": "Training expression mixer",
          "condition": null,
          "evidence_ids": [
            "bfc3cb3b89891fce1c3c87ec4711445ee7c610c3539f6ac41e63bc6e4df8134f",
            "b2219fd688af43a3c418ed9b0da8421573eff83e4c8cbb0ad2816047f40de556"
          ],
          "inputs": {
            "context": [
              "replacement_fraction"
            ],
            "indices": [
              "reveal_positions"
            ],
            "replacements": [
              "x_target"
            ],
            "state": [
              "x_control"
            ]
          },
          "outputs": {
            "values": [
              "training_mask",
              "mixed_expression"
            ]
          },
          "parameters": {
            "method": "indexed replacement",
            "preserve_unselected": true,
            "source_port_roles": [
              "unrevealed source values",
              "revealed ground-truth source values",
              "sampled revealed positions",
              "reveal strategy"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "reveal_target",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "reveal_target::1",
          "component": "Training expression mixer",
          "condition": null,
          "evidence_ids": [
            "bfc3cb3b89891fce1c3c87ec4711445ee7c610c3539f6ac41e63bc6e4df8134f",
            "b2219fd688af43a3c418ed9b0da8421573eff83e4c8cbb0ad2816047f40de556"
          ],
          "inputs": {
            "context": [
              "replacement_fraction"
            ],
            "indices": [
              "reveal_positions"
            ],
            "replacements": [
              "x_target"
            ],
            "state": [
              "x_control"
            ]
          },
          "outputs": {
            "values": [
              "training_mask",
              "mixed_expression"
            ]
          },
          "parameters": {
            "method": "indexed replacement",
            "preserve_unselected": true,
            "source_port_roles": [
              "unrevealed source values",
              "revealed ground-truth source values",
              "sampled revealed positions",
              "reveal strategy"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "reveal_target",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "reveal_target::1",
          "component": "Training expression mixer",
          "condition": null,
          "evidence_ids": [
            "bfc3cb3b89891fce1c3c87ec4711445ee7c610c3539f6ac41e63bc6e4df8134f",
            "b2219fd688af43a3c418ed9b0da8421573eff83e4c8cbb0ad2816047f40de556"
          ],
          "inputs": {
            "context": [
              "replacement_fraction"
            ],
            "indices": [
              "reveal_positions"
            ],
            "replacements": [
              "x_target"
            ],
            "state": [
              "x_control"
            ]
          },
          "outputs": {
            "values": [
              "training_mask",
              "mixed_expression"
            ]
          },
          "parameters": {
            "method": "indexed replacement",
            "preserve_unselected": true,
            "source_port_roles": [
              "unrevealed source values",
              "revealed ground-truth source values",
              "sampled revealed positions",
              "reveal strategy"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "reveal_target",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "reveal_target::1",
          "component": "Training expression mixer",
          "condition": null,
          "evidence_ids": [
            "bfc3cb3b89891fce1c3c87ec4711445ee7c610c3539f6ac41e63bc6e4df8134f",
            "b2219fd688af43a3c418ed9b0da8421573eff83e4c8cbb0ad2816047f40de556"
          ],
          "inputs": {
            "context": [
              "replacement_fraction"
            ],
            "indices": [
              "reveal_positions"
            ],
            "replacements": [
              "x_target"
            ],
            "state": [
              "x_control"
            ]
          },
          "outputs": {
            "values": [
              "training_mask",
              "mixed_expression"
            ]
          },
          "parameters": {
            "method": "indexed replacement",
            "preserve_unselected": true,
            "source_port_roles": [
              "unrevealed source values",
              "revealed ground-truth source values",
              "sampled revealed positions",
              "reveal strategy"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "reveal_target",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "reveal_target::1",
          "component": "Training expression mixer",
          "condition": null,
          "evidence_ids": [
            "bfc3cb3b89891fce1c3c87ec4711445ee7c610c3539f6ac41e63bc6e4df8134f",
            "b2219fd688af43a3c418ed9b0da8421573eff83e4c8cbb0ad2816047f40de556"
          ],
          "inputs": {
            "context": [
              "replacement_fraction"
            ],
            "indices": [
              "reveal_positions"
            ],
            "replacements": [
              "x_target"
            ],
            "state": [
              "x_control"
            ]
          },
          "outputs": {
            "values": [
              "training_mask",
              "mixed_expression"
            ]
          },
          "parameters": {
            "method": "indexed replacement",
            "preserve_unselected": true,
            "source_port_roles": [
              "unrevealed source values",
              "revealed ground-truth source values",
              "sampled revealed positions",
              "reveal strategy"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "reveal_target",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "reveal_target::1",
          "component": "Training expression mixer",
          "condition": null,
          "evidence_ids": [
            "bfc3cb3b89891fce1c3c87ec4711445ee7c610c3539f6ac41e63bc6e4df8134f",
            "b2219fd688af43a3c418ed9b0da8421573eff83e4c8cbb0ad2816047f40de556"
          ],
          "inputs": {
            "context": [
              "replacement_fraction"
            ],
            "indices": [
              "reveal_positions"
            ],
            "replacements": [
              "x_target"
            ],
            "state": [
              "x_control"
            ]
          },
          "outputs": {
            "values": [
              "training_mask",
              "mixed_expression"
            ]
          },
          "parameters": {
            "method": "indexed replacement",
            "preserve_unselected": true,
            "source_port_roles": [
              "unrevealed source values",
              "revealed ground-truth source values",
              "sampled revealed positions",
              "reveal strategy"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "reveal_target",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "reveal_target::1",
          "component": "Training expression mixer",
          "condition": null,
          "evidence_ids": [
            "bfc3cb3b89891fce1c3c87ec4711445ee7c610c3539f6ac41e63bc6e4df8134f",
            "b2219fd688af43a3c418ed9b0da8421573eff83e4c8cbb0ad2816047f40de556"
          ],
          "inputs": {
            "context": [
              "replacement_fraction"
            ],
            "indices": [
              "reveal_positions"
            ],
            "replacements": [
              "x_target"
            ],
            "state": [
              "x_control"
            ]
          },
          "outputs": {
            "values": [
              "training_mask",
              "mixed_expression"
            ]
          },
          "parameters": {
            "method": "indexed replacement",
            "preserve_unselected": true,
            "source_port_roles": [
              "unrevealed source values",
              "revealed ground-truth source values",
              "sampled revealed positions",
              "reveal strategy"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "reveal_target",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "scatter_update::1",
          "component": "Cumulative diffusion state updater",
          "condition": null,
          "evidence_ids": [
            "347089ce1327f071c6790467ae19079691059ad17ba3628bb6804409718bf1d3",
            "a79be46174c965e1ab88a0e4dffd45c6d51fc40851b08db84d1f3783b29e4dfd"
          ],
          "inputs": {
            "context": [],
            "indices": [
              "new_positions"
            ],
            "replacements": [
              "accepted_values"
            ],
            "state": [
              "x_initial",
              "mask_initial"
            ]
          },
          "outputs": {
            "values": [
              "x_next",
              "mask_next"
            ]
          },
          "parameters": {
            "method": "indexed replacement",
            "preserve_unselected": true,
            "source_port_roles": [
              "prior accumulated state",
              "prior reveal mask",
              "new clamped predictions",
              "new accepted positions"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "scatter_update",
          "trajectory_id": "xcell_jurkat_active_inference"
        },
        {
          "call_id": "scatter_update::1",
          "component": "Cumulative diffusion state updater",
          "condition": null,
          "evidence_ids": [
            "347089ce1327f071c6790467ae19079691059ad17ba3628bb6804409718bf1d3",
            "a79be46174c965e1ab88a0e4dffd45c6d51fc40851b08db84d1f3783b29e4dfd"
          ],
          "inputs": {
            "context": [],
            "indices": [
              "new_positions"
            ],
            "replacements": [
              "accepted_values"
            ],
            "state": [
              "x_initial",
              "mask_initial"
            ]
          },
          "outputs": {
            "values": [
              "x_next",
              "mask_next"
            ]
          },
          "parameters": {
            "method": "indexed replacement",
            "preserve_unselected": true,
            "source_port_roles": [
              "prior accumulated state",
              "prior reveal mask",
              "new clamped predictions",
              "new accepted positions"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "scatter_update",
          "trajectory_id": "ultra_melanocyte_post_tta_inference"
        }
      ]
    },
    {
      "boundaries": "Matching basis and correspondence evidence are mandatory; equal shape supplies no correspondence.",
      "definition": "Align data using documented identifier, spatial or sample correspondences.",
      "inputs": [
        {
          "meaning": "Documented input data operands.",
          "name": "values",
          "optional": false,
          "variadic": true
        }
      ],
      "label": "Correspondence alignment",
      "official_reference": null,
      "operation_id": "align",
      "outputs": [
        {
          "meaning": "Produced output data operands.",
          "name": "values",
          "optional": false,
          "variadic": true
        }
      ],
      "parameters": [
        "basis",
        "method"
      ],
      "reuse_record_ids": [
        "full_2026-07-06__rec_003852"
      ],
      "status": "source_backed",
      "supplemental_examples": [],
      "usage": [
        {
          "call_id": "align::1",
          "component": "ST regional alignment and extraction",
          "condition": null,
          "evidence_ids": [
            "3aac6376e2215904c435a2d32f9e11b2bcd312f398afe30b245b3570fc53af6e"
          ],
          "inputs": {
            "values": [
              "st_matrix",
              "patches"
            ]
          },
          "outputs": {
            "values": [
              "regional_expression"
            ]
          },
          "parameters": {
            "basis": "spatial correspondence",
            "source_port_roles": [
              "measured_spatial_expression",
              "region_correspondence"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "align",
          "trajectory_id": "finetune_st_condition"
        },
        {
          "call_id": "align::1",
          "component": "ST regional alignment and extraction",
          "condition": null,
          "evidence_ids": [
            "3aac6376e2215904c435a2d32f9e11b2bcd312f398afe30b245b3570fc53af6e"
          ],
          "inputs": {
            "values": [
              "st_matrix",
              "patches"
            ]
          },
          "outputs": {
            "values": [
              "regional_expression"
            ]
          },
          "parameters": {
            "basis": "spatial correspondence",
            "source_port_roles": [
              "measured_spatial_expression",
              "region_correspondence"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "align",
          "trajectory_id": "infer_st_condition"
        }
      ]
    },
    {
      "boundaries": "Policy and lifecycle availability remain instance attributes.",
      "definition": "Transform source data according to a documented augmentation policy.",
      "inputs": [
        {
          "meaning": "Documented input data operands.",
          "name": "values",
          "optional": false,
          "variadic": true
        }
      ],
      "label": "Data augmentation",
      "official_reference": null,
      "operation_id": "augment",
      "outputs": [
        {
          "meaning": "Produced output data operands.",
          "name": "values",
          "optional": false,
          "variadic": true
        }
      ],
      "parameters": [
        "policy",
        "phase"
      ],
      "reuse_record_ids": [
        "full_2026-07-06__rec_003852"
      ],
      "status": "source_backed",
      "supplemental_examples": [],
      "usage": [
        {
          "call_id": "augment::1",
          "component": "HED color augmentation",
          "condition": null,
          "evidence_ids": [
            "bca8cb6992d809fc1e66ba68b4dc2c6d0585f42fc0e7ea4326424c94fd91af39"
          ],
          "inputs": {
            "values": [
              "he_image"
            ]
          },
          "outputs": {
            "values": [
              "he_augmented"
            ]
          },
          "parameters": {
            "policy": "histology color augmentation",
            "source_port_roles": [
              "training_histology"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "augment",
          "trajectory_id": "finetune_mif_dual_condition"
        }
      ]
    },
    {
      "boundaries": "The symbolic input/output axes and broadcasting rule require evidence; physical sizes stay in source quotations.",
      "definition": "Change layout or expand documented axes while retaining the underlying data correspondence.",
      "inputs": [
        {
          "meaning": "Documented input data operands.",
          "name": "values",
          "optional": false,
          "variadic": true
        }
      ],
      "label": "Reshaping or expansion",
      "official_reference": "https://docs.pytorch.org/docs/stable/generated/torch.reshape.html",
      "operation_id": "reshape",
      "outputs": [
        {
          "meaning": "Produced output data operands.",
          "name": "values",
          "optional": false,
          "variadic": true
        }
      ],
      "parameters": [
        "method",
        "axes"
      ],
      "reuse_record_ids": [
        "full_2026-07-06__rec_003517"
      ],
      "status": "source_backed",
      "supplemental_examples": [],
      "usage": [
        {
          "call_id": "concat_local_global::1",
          "component": "Expression decoder input assembler",
          "condition": null,
          "evidence_ids": [
            "ecdf72423cf768656361070c11ac6419dabdd3f6e5a178b59dcb251e2fe6dd64"
          ],
          "inputs": {
            "values": [
              "cls_state"
            ]
          },
          "outputs": {
            "values": [
              "concat_local_global::expanded_cell_summary"
            ]
          },
          "parameters": {
            "method": "expand cell summary across gene positions",
            "source_port_roles": [
              "local gene features",
              "global cell features"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "concat_local_global",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "concat_local_global::1",
          "component": "Expression decoder input assembler",
          "condition": null,
          "evidence_ids": [
            "ecdf72423cf768656361070c11ac6419dabdd3f6e5a178b59dcb251e2fe6dd64"
          ],
          "inputs": {
            "values": [
              "cls_state"
            ]
          },
          "outputs": {
            "values": [
              "concat_local_global::expanded_cell_summary"
            ]
          },
          "parameters": {
            "method": "expand cell summary across gene positions",
            "source_port_roles": [
              "local gene features",
              "global cell features"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "concat_local_global",
          "trajectory_id": "ultra_training_tied_output_receipt"
        }
      ]
    }
  ],
  "contract_version": "operation-library-v1",
  "library_sha256": "8baa240c279a84a1229eefd5f34e0527291086570378e20500f37bcb9f1f2caa",
  "pending": [
    {
      "occurrences": [
        {
          "call_id": "decode_next::1",
          "component": "Vicuna-7b English decoder",
          "condition": null,
          "evidence_ids": [
            "dfdd5ff10cb42c401f0c344f517d4536c0473fa7e78688d2b07ceaca468ae4e2"
          ],
          "inputs": {
            "source_operands": [
              "sampled_token_t",
              "cache_t"
            ]
          },
          "outputs": {
            "source_results": [
              "word_distribution_next",
              "cache_next"
            ]
          },
          "parameters": {
            "source_operation": "autoregressive_distribution_and_cache_update",
            "source_port_roles": [
              "previous_sampled_token",
              "prior_attention_state"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "decode_next",
          "trajectory_id": "benchmark_inference"
        },
        {
          "call_id": "decode_next::1",
          "component": "Vicuna-7b English decoder",
          "condition": null,
          "evidence_ids": [
            "dfdd5ff10cb42c401f0c344f517d4536c0473fa7e78688d2b07ceaca468ae4e2"
          ],
          "inputs": {
            "source_operands": [
              "sampled_token_t",
              "cache_t"
            ]
          },
          "outputs": {
            "source_results": [
              "word_distribution_next",
              "cache_next"
            ]
          },
          "parameters": {
            "source_operation": "autoregressive_distribution_and_cache_update",
            "source_port_roles": [
              "previous_sampled_token",
              "prior_attention_state"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "decode_next",
          "trajectory_id": "curated_dna_inference"
        },
        {
          "call_id": "decode_next::1",
          "component": "Vicuna-7b English decoder",
          "condition": null,
          "evidence_ids": [
            "dfdd5ff10cb42c401f0c344f517d4536c0473fa7e78688d2b07ceaca468ae4e2"
          ],
          "inputs": {
            "source_operands": [
              "sampled_token_t",
              "cache_t"
            ]
          },
          "outputs": {
            "source_results": [
              "word_distribution_next",
              "cache_next"
            ]
          },
          "parameters": {
            "source_operation": "autoregressive_distribution_and_cache_update",
            "source_port_roles": [
              "previous_sampled_token",
              "prior_attention_state"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "decode_next",
          "trajectory_id": "curated_rna_inference"
        },
        {
          "call_id": "decode_next::1",
          "component": "Vicuna-7b English decoder",
          "condition": null,
          "evidence_ids": [
            "dfdd5ff10cb42c401f0c344f517d4536c0473fa7e78688d2b07ceaca468ae4e2"
          ],
          "inputs": {
            "source_operands": [
              "sampled_token_t",
              "cache_t"
            ]
          },
          "outputs": {
            "source_results": [
              "word_distribution_next",
              "cache_next"
            ]
          },
          "parameters": {
            "source_operation": "autoregressive_distribution_and_cache_update",
            "source_port_roles": [
              "previous_sampled_token",
              "prior_attention_state"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "decode_next",
          "trajectory_id": "curated_protein_inference"
        },
        {
          "call_id": "decode_next::1",
          "component": "Vicuna-7b English decoder",
          "condition": null,
          "evidence_ids": [
            "dfdd5ff10cb42c401f0c344f517d4536c0473fa7e78688d2b07ceaca468ae4e2"
          ],
          "inputs": {
            "source_operands": [
              "sampled_token_t",
              "cache_t"
            ]
          },
          "outputs": {
            "source_results": [
              "word_distribution_next",
              "cache_next"
            ]
          },
          "parameters": {
            "source_operation": "autoregressive_distribution_and_cache_update",
            "source_port_roles": [
              "previous_sampled_token",
              "prior_attention_state"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "decode_next",
          "trajectory_id": "multiple_inference"
        },
        {
          "call_id": "decode_next::1",
          "component": "Vicuna-7b English decoder",
          "condition": null,
          "evidence_ids": [
            "dfdd5ff10cb42c401f0c344f517d4536c0473fa7e78688d2b07ceaca468ae4e2"
          ],
          "inputs": {
            "source_operands": [
              "sampled_token_t",
              "cache_t"
            ]
          },
          "outputs": {
            "source_results": [
              "word_distribution_next",
              "cache_next"
            ]
          },
          "parameters": {
            "source_operation": "autoregressive_distribution_and_cache_update",
            "source_port_roles": [
              "previous_sampled_token",
              "prior_attention_state"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "decode_next",
          "trajectory_id": "unconditioned_benchmark_inference"
        }
      ],
      "source_operation": "autoregressive_distribution_and_cache_update",
      "status": "unexpanded_source_boundary"
    },
    {
      "occurrences": [
        {
          "call_id": "combine_tokens::1",
          "component": "Input representation combiner",
          "condition": null,
          "evidence_ids": [
            "c05e6918877939105bd5a61b17803c04c034d1b34eb4f200437b7bd1ae3f7e36",
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0",
            "d4015be6a525bdcb39076445d9d4b72f2bc20449673edb27d686b896a0d20d0a"
          ],
          "inputs": {
            "source_operands": [
              "identity_encoded",
              "value_encoded",
              "mask_encoded"
            ]
          },
          "outputs": {
            "source_results": [
              "gene_tokens"
            ]
          },
          "parameters": {
            "source_operation": "combine_identity_value_and_mask",
            "source_port_roles": [
              "gene identity contribution",
              "expression contribution",
              "revealed-signal contribution"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "combine_tokens",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "combine_tokens::1",
          "component": "Input representation combiner",
          "condition": null,
          "evidence_ids": [
            "c05e6918877939105bd5a61b17803c04c034d1b34eb4f200437b7bd1ae3f7e36",
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0",
            "d4015be6a525bdcb39076445d9d4b72f2bc20449673edb27d686b896a0d20d0a"
          ],
          "inputs": {
            "source_operands": [
              "identity_encoded",
              "value_encoded",
              "mask_encoded"
            ]
          },
          "outputs": {
            "source_results": [
              "gene_tokens"
            ]
          },
          "parameters": {
            "source_operation": "combine_identity_value_and_mask",
            "source_port_roles": [
              "gene identity contribution",
              "expression contribution",
              "revealed-signal contribution"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "combine_tokens",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "combine_tokens::1",
          "component": "Input representation combiner",
          "condition": null,
          "evidence_ids": [
            "c05e6918877939105bd5a61b17803c04c034d1b34eb4f200437b7bd1ae3f7e36",
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0",
            "d4015be6a525bdcb39076445d9d4b72f2bc20449673edb27d686b896a0d20d0a"
          ],
          "inputs": {
            "source_operands": [
              "identity_encoded",
              "value_encoded",
              "mask_encoded"
            ]
          },
          "outputs": {
            "source_results": [
              "gene_tokens"
            ]
          },
          "parameters": {
            "source_operation": "combine_identity_value_and_mask",
            "source_port_roles": [
              "gene identity contribution",
              "expression contribution",
              "revealed-signal contribution"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "combine_tokens",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "combine_tokens::1",
          "component": "Input representation combiner",
          "condition": null,
          "evidence_ids": [
            "c05e6918877939105bd5a61b17803c04c034d1b34eb4f200437b7bd1ae3f7e36",
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0",
            "d4015be6a525bdcb39076445d9d4b72f2bc20449673edb27d686b896a0d20d0a"
          ],
          "inputs": {
            "source_operands": [
              "identity_encoded",
              "value_encoded",
              "mask_encoded"
            ]
          },
          "outputs": {
            "source_results": [
              "gene_tokens"
            ]
          },
          "parameters": {
            "source_operation": "combine_identity_value_and_mask",
            "source_port_roles": [
              "gene identity contribution",
              "expression contribution",
              "revealed-signal contribution"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "combine_tokens",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "combine_tokens::1",
          "component": "Input representation combiner",
          "condition": null,
          "evidence_ids": [
            "c05e6918877939105bd5a61b17803c04c034d1b34eb4f200437b7bd1ae3f7e36",
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0",
            "d4015be6a525bdcb39076445d9d4b72f2bc20449673edb27d686b896a0d20d0a"
          ],
          "inputs": {
            "source_operands": [
              "identity_encoded",
              "value_encoded",
              "mask_encoded"
            ]
          },
          "outputs": {
            "source_results": [
              "gene_tokens"
            ]
          },
          "parameters": {
            "source_operation": "combine_identity_value_and_mask",
            "source_port_roles": [
              "gene identity contribution",
              "expression contribution",
              "revealed-signal contribution"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "combine_tokens",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "combine_tokens::1",
          "component": "Input representation combiner",
          "condition": null,
          "evidence_ids": [
            "c05e6918877939105bd5a61b17803c04c034d1b34eb4f200437b7bd1ae3f7e36",
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0",
            "d4015be6a525bdcb39076445d9d4b72f2bc20449673edb27d686b896a0d20d0a"
          ],
          "inputs": {
            "source_operands": [
              "identity_encoded",
              "value_encoded",
              "mask_encoded"
            ]
          },
          "outputs": {
            "source_results": [
              "gene_tokens"
            ]
          },
          "parameters": {
            "source_operation": "combine_identity_value_and_mask",
            "source_port_roles": [
              "gene identity contribution",
              "expression contribution",
              "revealed-signal contribution"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "combine_tokens",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "combine_tokens::1",
          "component": "Input representation combiner",
          "condition": null,
          "evidence_ids": [
            "c05e6918877939105bd5a61b17803c04c034d1b34eb4f200437b7bd1ae3f7e36",
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0",
            "d4015be6a525bdcb39076445d9d4b72f2bc20449673edb27d686b896a0d20d0a"
          ],
          "inputs": {
            "source_operands": [
              "identity_encoded",
              "value_encoded",
              "mask_encoded"
            ]
          },
          "outputs": {
            "source_results": [
              "gene_tokens"
            ]
          },
          "parameters": {
            "source_operation": "combine_identity_value_and_mask",
            "source_port_roles": [
              "gene identity contribution",
              "expression contribution",
              "revealed-signal contribution"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "combine_tokens",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "combine_tokens::1",
          "component": "Input representation combiner",
          "condition": null,
          "evidence_ids": [
            "c05e6918877939105bd5a61b17803c04c034d1b34eb4f200437b7bd1ae3f7e36",
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0",
            "d4015be6a525bdcb39076445d9d4b72f2bc20449673edb27d686b896a0d20d0a"
          ],
          "inputs": {
            "source_operands": [
              "identity_encoded",
              "value_encoded",
              "mask_encoded"
            ]
          },
          "outputs": {
            "source_results": [
              "gene_tokens"
            ]
          },
          "parameters": {
            "source_operation": "combine_identity_value_and_mask",
            "source_port_roles": [
              "gene identity contribution",
              "expression contribution",
              "revealed-signal contribution"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "combine_tokens",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "combine_tokens::1",
          "component": "Input representation combiner",
          "condition": null,
          "evidence_ids": [
            "c05e6918877939105bd5a61b17803c04c034d1b34eb4f200437b7bd1ae3f7e36",
            "32548dc74779a3238291c4b615a9bc6f2893d6079f1a15eaab3c221119efa5f0",
            "d4015be6a525bdcb39076445d9d4b72f2bc20449673edb27d686b896a0d20d0a"
          ],
          "inputs": {
            "source_operands": [
              "identity_encoded",
              "value_encoded",
              "mask_encoded"
            ]
          },
          "outputs": {
            "source_results": [
              "gene_tokens"
            ]
          },
          "parameters": {
            "source_operation": "combine_identity_value_and_mask",
            "source_port_roles": [
              "gene identity contribution",
              "expression contribution",
              "revealed-signal contribution"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "combine_tokens",
          "trajectory_id": "ultra_melanocyte_tta_single_step"
        }
      ],
      "source_operation": "combine_identity_value_and_mask",
      "status": "unexpanded_source_boundary"
    },
    {
      "occurrences": [
        {
          "call_id": "condition_lm::1",
          "component": "InstructCell LM",
          "condition": null,
          "evidence_ids": [
            "4686d79d4c2ef77f5e9a169e0df5ad589d4703075206dcc8ce6f771fc055ee4e",
            "589f39e264d41269bc3ae107e0bf5ef4dda75175ae74a38d70e6308056fa3e33"
          ],
          "inputs": {
            "source_operands": [
              "prompt_tokens",
              "response_tokens"
            ]
          },
          "outputs": {
            "source_results": [
              "hidden"
            ]
          },
          "parameters": {
            "source_operation": "conditional_hidden_state_computation",
            "source_port_roles": [
              "instruction_context",
              "target_response_context"
            ]
          },
          "record_id": "full_2026-07-06__rec_001319",
          "source_step_id": "condition_lm",
          "trajectory_id": "cpcg_training"
        }
      ],
      "source_operation": "conditional_hidden_state_computation",
      "status": "unexpanded_source_boundary"
    },
    {
      "occurrences": [
        {
          "call_id": "predict_current::1",
          "component": "X-Cell current-round forward predictor",
          "condition": null,
          "evidence_ids": [
            "347089ce1327f071c6790467ae19079691059ad17ba3628bb6804409718bf1d3",
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "source_operands": [
              "x_initial",
              "mask_initial",
              "context",
              "external_mask"
            ]
          },
          "outputs": {
            "source_results": [
              "round_prediction"
            ]
          },
          "parameters": {
            "source_operation": "conditioned_full_profile_prediction",
            "source_port_roles": [
              "current state for first round; recursively x_previous later",
              "current reveal status",
              "perturbation conditioning",
              "context availability"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "predict_current",
          "trajectory_id": "xcell_jurkat_active_inference"
        },
        {
          "call_id": "predict_current::1",
          "component": "X-Cell-Ultra current-round forward predictor",
          "condition": null,
          "evidence_ids": [
            "347089ce1327f071c6790467ae19079691059ad17ba3628bb6804409718bf1d3",
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "source_operands": [
              "x_initial",
              "mask_initial",
              "context",
              "external_mask"
            ]
          },
          "outputs": {
            "source_results": [
              "round_prediction"
            ]
          },
          "parameters": {
            "source_operation": "conditioned_full_profile_prediction",
            "source_port_roles": [
              "current state for first round; recursively x_previous later",
              "current reveal status",
              "perturbation conditioning",
              "context availability"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "predict_current",
          "trajectory_id": "ultra_melanocyte_post_tta_inference"
        }
      ],
      "source_operation": "conditioned_full_profile_prediction",
      "status": "unexpanded_source_boundary"
    },
    {
      "occurrences": [
        {
          "call_id": "encode_dna_a::1",
          "component": "Nucleotide Transformer v2 DNA encoder",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f",
            "4495722c8f5e8fe8576dbe8da863731e8741ef1aa7ac42d16cb13fb1aecc3452"
          ],
          "inputs": {
            "source_operands": [
              "padded_tokens_a"
            ]
          },
          "outputs": {
            "source_results": [
              "encoded_a"
            ]
          },
          "parameters": {
            "source_operation": "contextual_sequence_encoding",
            "source_port_roles": [
              "input_token_ids"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "encode_dna_a",
          "trajectory_id": "benchmark_fine_tuning"
        },
        {
          "call_id": "encode_dna_a::1",
          "component": "Nucleotide Transformer v2 DNA encoder",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f",
            "4495722c8f5e8fe8576dbe8da863731e8741ef1aa7ac42d16cb13fb1aecc3452"
          ],
          "inputs": {
            "source_operands": [
              "padded_tokens_a"
            ]
          },
          "outputs": {
            "source_results": [
              "encoded_a"
            ]
          },
          "parameters": {
            "source_operation": "contextual_sequence_encoding",
            "source_port_roles": [
              "input_token_ids"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "encode_dna_a",
          "trajectory_id": "benchmark_inference"
        },
        {
          "call_id": "encode_dna_a::1",
          "component": "Nucleotide Transformer v2 DNA encoder",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f",
            "4495722c8f5e8fe8576dbe8da863731e8741ef1aa7ac42d16cb13fb1aecc3452"
          ],
          "inputs": {
            "source_operands": [
              "padded_tokens_a"
            ]
          },
          "outputs": {
            "source_results": [
              "encoded_a"
            ]
          },
          "parameters": {
            "source_operation": "contextual_sequence_encoding",
            "source_port_roles": [
              "input_token_ids"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "encode_dna_a",
          "trajectory_id": "curated_dna_fine_tuning"
        },
        {
          "call_id": "encode_dna_a::1",
          "component": "Nucleotide Transformer v2 DNA encoder",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f",
            "4495722c8f5e8fe8576dbe8da863731e8741ef1aa7ac42d16cb13fb1aecc3452"
          ],
          "inputs": {
            "source_operands": [
              "padded_tokens_a"
            ]
          },
          "outputs": {
            "source_results": [
              "encoded_a"
            ]
          },
          "parameters": {
            "source_operation": "contextual_sequence_encoding",
            "source_port_roles": [
              "input_token_ids"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "encode_dna_a",
          "trajectory_id": "curated_dna_inference"
        },
        {
          "call_id": "encode_dna_a::1",
          "component": "Nucleotide Transformer v2 DNA encoder",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f",
            "4495722c8f5e8fe8576dbe8da863731e8741ef1aa7ac42d16cb13fb1aecc3452"
          ],
          "inputs": {
            "source_operands": [
              "padded_tokens_a"
            ]
          },
          "outputs": {
            "source_results": [
              "encoded_a"
            ]
          },
          "parameters": {
            "source_operation": "contextual_sequence_encoding",
            "source_port_roles": [
              "input_token_ids"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "encode_dna_a",
          "trajectory_id": "curated_rna_fine_tuning"
        },
        {
          "call_id": "encode_dna_a::1",
          "component": "Nucleotide Transformer v2 DNA encoder",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f",
            "4495722c8f5e8fe8576dbe8da863731e8741ef1aa7ac42d16cb13fb1aecc3452"
          ],
          "inputs": {
            "source_operands": [
              "padded_tokens_a"
            ]
          },
          "outputs": {
            "source_results": [
              "encoded_a"
            ]
          },
          "parameters": {
            "source_operation": "contextual_sequence_encoding",
            "source_port_roles": [
              "input_token_ids"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "encode_dna_a",
          "trajectory_id": "curated_rna_inference"
        },
        {
          "call_id": "encode_dna_a::1",
          "component": "Nucleotide Transformer v2 DNA encoder",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f",
            "4495722c8f5e8fe8576dbe8da863731e8741ef1aa7ac42d16cb13fb1aecc3452"
          ],
          "inputs": {
            "source_operands": [
              "padded_tokens_a"
            ]
          },
          "outputs": {
            "source_results": [
              "encoded_a"
            ]
          },
          "parameters": {
            "source_operation": "contextual_sequence_encoding",
            "source_port_roles": [
              "input_token_ids"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "encode_dna_a",
          "trajectory_id": "curated_protein_fine_tuning"
        },
        {
          "call_id": "encode_dna_a::1",
          "component": "Nucleotide Transformer v2 DNA encoder",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f",
            "4495722c8f5e8fe8576dbe8da863731e8741ef1aa7ac42d16cb13fb1aecc3452"
          ],
          "inputs": {
            "source_operands": [
              "padded_tokens_a"
            ]
          },
          "outputs": {
            "source_results": [
              "encoded_a"
            ]
          },
          "parameters": {
            "source_operation": "contextual_sequence_encoding",
            "source_port_roles": [
              "input_token_ids"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "encode_dna_a",
          "trajectory_id": "curated_protein_inference"
        },
        {
          "call_id": "encode_dna_a::1",
          "component": "Nucleotide Transformer v2 DNA encoder",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f",
            "4495722c8f5e8fe8576dbe8da863731e8741ef1aa7ac42d16cb13fb1aecc3452"
          ],
          "inputs": {
            "source_operands": [
              "padded_tokens_a"
            ]
          },
          "outputs": {
            "source_results": [
              "encoded_a"
            ]
          },
          "parameters": {
            "source_operation": "contextual_sequence_encoding",
            "source_port_roles": [
              "input_token_ids"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "encode_dna_a",
          "trajectory_id": "multiple_fine_tuning"
        },
        {
          "call_id": "encode_dna_b::1",
          "component": "Nucleotide Transformer v2 DNA encoder",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f",
            "4495722c8f5e8fe8576dbe8da863731e8741ef1aa7ac42d16cb13fb1aecc3452"
          ],
          "inputs": {
            "source_operands": [
              "padded_tokens_b"
            ]
          },
          "outputs": {
            "source_results": [
              "encoded_b"
            ]
          },
          "parameters": {
            "source_operation": "contextual_sequence_encoding",
            "source_port_roles": [
              "input_token_ids"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "encode_dna_b",
          "trajectory_id": "multiple_fine_tuning"
        },
        {
          "call_id": "encode_dna_a::1",
          "component": "Nucleotide Transformer v2 DNA encoder",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f",
            "4495722c8f5e8fe8576dbe8da863731e8741ef1aa7ac42d16cb13fb1aecc3452"
          ],
          "inputs": {
            "source_operands": [
              "padded_tokens_a"
            ]
          },
          "outputs": {
            "source_results": [
              "encoded_a"
            ]
          },
          "parameters": {
            "source_operation": "contextual_sequence_encoding",
            "source_port_roles": [
              "input_token_ids"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "encode_dna_a",
          "trajectory_id": "multiple_inference"
        },
        {
          "call_id": "encode_dna_b::1",
          "component": "Nucleotide Transformer v2 DNA encoder",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f",
            "4495722c8f5e8fe8576dbe8da863731e8741ef1aa7ac42d16cb13fb1aecc3452"
          ],
          "inputs": {
            "source_operands": [
              "padded_tokens_b"
            ]
          },
          "outputs": {
            "source_results": [
              "encoded_b"
            ]
          },
          "parameters": {
            "source_operation": "contextual_sequence_encoding",
            "source_port_roles": [
              "input_token_ids"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "encode_dna_b",
          "trajectory_id": "multiple_inference"
        },
        {
          "call_id": "encode_dna_a::1",
          "component": "Nucleotide Transformer v2 DNA encoder",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f",
            "4495722c8f5e8fe8576dbe8da863731e8741ef1aa7ac42d16cb13fb1aecc3452"
          ],
          "inputs": {
            "source_operands": [
              "padded_tokens_a"
            ]
          },
          "outputs": {
            "source_results": [
              "encoded_a"
            ]
          },
          "parameters": {
            "source_operation": "contextual_sequence_encoding",
            "source_port_roles": [
              "input_token_ids"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "encode_dna_a",
          "trajectory_id": "unconditioned_benchmark_fine_tuning"
        },
        {
          "call_id": "encode_dna_a::1",
          "component": "Nucleotide Transformer v2 DNA encoder",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f",
            "4495722c8f5e8fe8576dbe8da863731e8741ef1aa7ac42d16cb13fb1aecc3452"
          ],
          "inputs": {
            "source_operands": [
              "padded_tokens_a"
            ]
          },
          "outputs": {
            "source_results": [
              "encoded_a"
            ]
          },
          "parameters": {
            "source_operation": "contextual_sequence_encoding",
            "source_port_roles": [
              "input_token_ids"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "encode_dna_a",
          "trajectory_id": "unconditioned_benchmark_inference"
        }
      ],
      "source_operation": "contextual_sequence_encoding",
      "status": "unexpanded_source_boundary"
    },
    {
      "occurrences": [
        {
          "call_id": "translate::1",
          "component": "MLP v theta learned continuous flow",
          "condition": null,
          "evidence_ids": [
            "b7ac03d5e2f40174fe6a4848449dcf184601f4cfc52ebe493a6265b15855f756",
            "facf59bfd5bc978c6ac9a0b3e2f2c51325dfd993c7f958956476b2a3edacda57"
          ],
          "inputs": {
            "source_operands": [
              "he_embedding"
            ]
          },
          "outputs": {
            "source_results": [
              "translated_ihc"
            ]
          },
          "parameters": {
            "source_operation": "continuous_embedding_translation",
            "source_port_roles": [
              "source_embedding"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "translate",
          "trajectory_id": "infer_ihc_condition"
        }
      ],
      "source_operation": "continuous_embedding_translation",
      "status": "unexpanded_source_boundary"
    },
    {
      "occurrences": [
        {
          "call_id": "finish_encoder::1",
          "component": "Remaining interleaved encoder layers",
          "condition": null,
          "evidence_ids": [
            "ecdf72423cf768656361070c11ac6419dabdd3f6e5a178b59dcb251e2fe6dd64",
            "5ecc89cd6886a88843925414941ab1210bb9ac03a1026b585fde6638ce2b017b"
          ],
          "inputs": {
            "source_operands": [
              "conditioned_state",
              "context",
              "external_mask"
            ]
          },
          "outputs": {
            "source_results": [
              "final_state"
            ]
          },
          "parameters": {
            "source_operation": "finish_interleaved_transformer_encoding",
            "source_port_roles": [
              "current hidden state",
              "remaining conditioning keys/values",
              "padding mask"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "finish_encoder",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "finish_encoder::1",
          "component": "Remaining interleaved encoder layers",
          "condition": null,
          "evidence_ids": [
            "ecdf72423cf768656361070c11ac6419dabdd3f6e5a178b59dcb251e2fe6dd64",
            "5ecc89cd6886a88843925414941ab1210bb9ac03a1026b585fde6638ce2b017b"
          ],
          "inputs": {
            "source_operands": [
              "conditioned_state",
              "context",
              "external_mask"
            ]
          },
          "outputs": {
            "source_results": [
              "final_state"
            ]
          },
          "parameters": {
            "source_operation": "finish_interleaved_transformer_encoding",
            "source_port_roles": [
              "current hidden state",
              "remaining conditioning keys/values",
              "padding mask"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "finish_encoder",
          "trajectory_id": "ultra_training_tied_output_receipt"
        }
      ],
      "source_operation": "finish_interleaved_transformer_encoding",
      "status": "unexpanded_source_boundary"
    },
    {
      "occurrences": [
        {
          "call_id": "gene_condition::1",
          "component": "Gene identity perturbation context encoder",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "source_operands": [
              "perturbation_gene_raw"
            ]
          },
          "outputs": {
            "source_results": [
              "gene_prior"
            ]
          },
          "parameters": {
            "source_operation": "gene_identity_context_encoding",
            "source_port_roles": [
              "raw gene identity condition"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "gene_condition",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "gene_condition::1",
          "component": "Gene identity perturbation context encoder",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "source_operands": [
              "perturbation_gene_raw"
            ]
          },
          "outputs": {
            "source_results": [
              "gene_prior"
            ]
          },
          "parameters": {
            "source_operation": "gene_identity_context_encoding",
            "source_port_roles": [
              "raw gene identity condition"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "gene_condition",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "gene_condition::1",
          "component": "Gene identity perturbation context encoder",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "source_operands": [
              "perturbation_gene_raw"
            ]
          },
          "outputs": {
            "source_results": [
              "gene_prior"
            ]
          },
          "parameters": {
            "source_operation": "gene_identity_context_encoding",
            "source_port_roles": [
              "raw gene identity condition"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "gene_condition",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "gene_condition::1",
          "component": "Gene identity perturbation context encoder",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "source_operands": [
              "perturbation_gene_raw"
            ]
          },
          "outputs": {
            "source_results": [
              "gene_prior"
            ]
          },
          "parameters": {
            "source_operation": "gene_identity_context_encoding",
            "source_port_roles": [
              "raw gene identity condition"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "gene_condition",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "gene_condition::1",
          "component": "Gene identity perturbation context encoder",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "source_operands": [
              "perturbation_gene_raw"
            ]
          },
          "outputs": {
            "source_results": [
              "gene_prior"
            ]
          },
          "parameters": {
            "source_operation": "gene_identity_context_encoding",
            "source_port_roles": [
              "raw gene identity condition"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "gene_condition",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "gene_condition::1",
          "component": "Gene identity perturbation context encoder",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "source_operands": [
              "perturbation_gene_raw"
            ]
          },
          "outputs": {
            "source_results": [
              "gene_prior"
            ]
          },
          "parameters": {
            "source_operation": "gene_identity_context_encoding",
            "source_port_roles": [
              "raw gene identity condition"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "gene_condition",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "gene_condition::1",
          "component": "Gene identity perturbation context encoder",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "source_operands": [
              "perturbation_gene_raw"
            ]
          },
          "outputs": {
            "source_results": [
              "gene_prior"
            ]
          },
          "parameters": {
            "source_operation": "gene_identity_context_encoding",
            "source_port_roles": [
              "raw gene identity condition"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "gene_condition",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "gene_condition::1",
          "component": "Gene identity perturbation context encoder",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "source_operands": [
              "perturbation_gene_raw"
            ]
          },
          "outputs": {
            "source_results": [
              "gene_prior"
            ]
          },
          "parameters": {
            "source_operation": "gene_identity_context_encoding",
            "source_port_roles": [
              "raw gene identity condition"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "gene_condition",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "gene_condition::1",
          "component": "Gene identity perturbation context encoder",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "source_operands": [
              "perturbation_gene_raw"
            ]
          },
          "outputs": {
            "source_results": [
              "gene_prior"
            ]
          },
          "parameters": {
            "source_operation": "gene_identity_context_encoding",
            "source_port_roles": [
              "raw gene identity condition"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "gene_condition",
          "trajectory_id": "xcell_jurkat_active_inference"
        },
        {
          "call_id": "gene_condition::1",
          "component": "Gene identity perturbation context encoder",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "source_operands": [
              "perturbation_gene_raw"
            ]
          },
          "outputs": {
            "source_results": [
              "gene_prior"
            ]
          },
          "parameters": {
            "source_operation": "gene_identity_context_encoding",
            "source_port_roles": [
              "raw gene identity condition"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "gene_condition",
          "trajectory_id": "ultra_melanocyte_post_tta_inference"
        }
      ],
      "source_operation": "gene_identity_context_encoding",
      "status": "unexpanded_source_boundary"
    },
    {
      "occurrences": [
        {
          "call_id": "score::1",
          "component": "Pathway enrichment computation",
          "condition": null,
          "evidence_ids": [
            "ed0cbc9219226da12cb74868c945901c59c51a869e0593f5bfcd24dc08383722"
          ],
          "inputs": {
            "source_operands": [
              "rna_filtered",
              "gene_signatures"
            ]
          },
          "outputs": {
            "source_results": [
              "pathway_scores"
            ]
          },
          "parameters": {
            "source_operation": "gene_set_enrichment",
            "source_port_roles": [
              "filtered_expression",
              "reference_gene_sets"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "score",
          "trajectory_id": "pretrain_multimodal_dca"
        },
        {
          "call_id": "score::1",
          "component": "ST pathway compression",
          "condition": null,
          "evidence_ids": [
            "3aac6376e2215904c435a2d32f9e11b2bcd312f398afe30b245b3570fc53af6e",
            "4bc86e9d1d9362650f38bbe0575033ee343575c72d3ed7a0da2e6320cce5d0d7"
          ],
          "inputs": {
            "source_operands": [
              "regional_expression",
              "gene_signatures"
            ]
          },
          "outputs": {
            "source_results": [
              "pathway_scores"
            ]
          },
          "parameters": {
            "source_operation": "gene_set_enrichment",
            "source_port_roles": [
              "regional_expression",
              "reference_gene_sets"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "score",
          "trajectory_id": "finetune_st_condition"
        },
        {
          "call_id": "score::1",
          "component": "ST pathway compression",
          "condition": null,
          "evidence_ids": [
            "3aac6376e2215904c435a2d32f9e11b2bcd312f398afe30b245b3570fc53af6e",
            "4bc86e9d1d9362650f38bbe0575033ee343575c72d3ed7a0da2e6320cce5d0d7"
          ],
          "inputs": {
            "source_operands": [
              "regional_expression",
              "gene_signatures"
            ]
          },
          "outputs": {
            "source_results": [
              "pathway_scores"
            ]
          },
          "parameters": {
            "source_operation": "gene_set_enrichment",
            "source_port_roles": [
              "regional_expression",
              "reference_gene_sets"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "score",
          "trajectory_id": "infer_st_condition"
        }
      ],
      "source_operation": "gene_set_enrichment",
      "status": "unexpanded_source_boundary"
    },
    {
      "occurrences": [
        {
          "call_id": "condition_lm::1",
          "component": "InstructCell LM",
          "condition": null,
          "evidence_ids": [
            "c6adc5d178ed689e311fdb92505bee529437338558141c01b201900703383ba3"
          ],
          "inputs": {
            "source_operands": [
              "prompt_tokens",
              "response_tokens"
            ]
          },
          "outputs": {
            "source_results": [
              "hidden"
            ]
          },
          "parameters": {
            "source_operation": "generated_text_feedback_and_hidden_state_computation",
            "source_port_roles": [
              "instruction_context",
              "generated_response_context"
            ]
          },
          "record_id": "full_2026-07-06__rec_001319",
          "source_step_id": "condition_lm",
          "trajectory_id": "cpcg_inference"
        }
      ],
      "source_operation": "generated_text_feedback_and_hidden_state_computation",
      "status": "unexpanded_source_boundary"
    },
    {
      "occurrences": [
        {
          "call_id": "encode_image::1",
          "component": "MUSK image encoder",
          "condition": null,
          "evidence_ids": [
            "2f344b287bc862c0717d113dbca55b0be574f35a70990afe284ee672a2b0c913"
          ],
          "inputs": {
            "source_operands": [
              "reference_image"
            ]
          },
          "outputs": {
            "source_results": [
              "image_condition"
            ]
          },
          "parameters": {
            "source_operation": "image_encoding",
            "source_port_roles": [
              "reference_image"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "encode_image",
          "trajectory_id": "pretrain_multimodal_dca"
        },
        {
          "call_id": "encode::1",
          "component": "MUSK image encoder",
          "condition": null,
          "evidence_ids": [
            "2f344b287bc862c0717d113dbca55b0be574f35a70990afe284ee672a2b0c913"
          ],
          "inputs": {
            "source_operands": [
              "reference_image"
            ]
          },
          "outputs": {
            "source_results": [
              "condition"
            ]
          },
          "parameters": {
            "source_operation": "image_encoding",
            "source_port_roles": [
              "sample_condition"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "encode",
          "trajectory_id": "infer_image"
        },
        {
          "call_id": "encode::1",
          "component": "MUSK image encoder",
          "condition": null,
          "evidence_ids": [
            "2f344b287bc862c0717d113dbca55b0be574f35a70990afe284ee672a2b0c913"
          ],
          "inputs": {
            "source_operands": [
              "reference_image"
            ]
          },
          "outputs": {
            "source_results": [
              "condition"
            ]
          },
          "parameters": {
            "source_operation": "image_encoding",
            "source_port_roles": [
              "sample_condition"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "encode",
          "trajectory_id": "infer_classification_augmentation"
        },
        {
          "call_id": "encode::1",
          "component": "MUSK image encoder",
          "condition": null,
          "evidence_ids": [
            "2f344b287bc862c0717d113dbca55b0be574f35a70990afe284ee672a2b0c913",
            "b7ac03d5e2f40174fe6a4848449dcf184601f4cfc52ebe493a6265b15855f756"
          ],
          "inputs": {
            "source_operands": [
              "he_image"
            ]
          },
          "outputs": {
            "source_results": [
              "he_embedding"
            ]
          },
          "parameters": {
            "source_operation": "image_encoding",
            "source_port_roles": [
              "input_histology"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "encode",
          "trajectory_id": "infer_ihc_condition"
        }
      ],
      "source_operation": "image_encoding",
      "status": "unexpanded_source_boundary"
    },
    {
      "occurrences": [
        {
          "call_id": "encode::1",
          "component": "Frozen VAE encoder",
          "condition": null,
          "evidence_ids": [
            "bcda2b9598259f605027d377fd28ca31fd7d1506049aaa287b3189ced2201f6c"
          ],
          "inputs": {
            "source_operands": [
              "clean_image"
            ]
          },
          "outputs": {
            "source_results": [
              "clean_latent"
            ]
          },
          "parameters": {
            "source_operation": "image_to_latent_encoding",
            "source_port_roles": [
              "image"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "encode",
          "trajectory_id": "pretrain_vae_encoding"
        },
        {
          "call_id": "structure::1",
          "component": "VAE H&E encoder",
          "condition": null,
          "evidence_ids": [
            "bca8cb6992d809fc1e66ba68b4dc2c6d0585f42fc0e7ea4326424c94fd91af39"
          ],
          "inputs": {
            "source_operands": [
              "he_image"
            ]
          },
          "outputs": {
            "source_results": [
              "struct_latent"
            ]
          },
          "parameters": {
            "source_operation": "image_to_latent_encoding",
            "source_port_roles": [
              "structural_histology"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "structure",
          "trajectory_id": "finetune_mif_dual_condition"
        }
      ],
      "source_operation": "image_to_latent_encoding",
      "status": "unexpanded_source_boundary"
    },
    {
      "occurrences": [
        {
          "call_id": "sample_controls::1",
          "component": "Control-to-control dataloader",
          "condition": null,
          "evidence_ids": [
            "37fa154aae79c2c31844e4c69d98548bc0c022462c3c29870ae17b71709601e0"
          ],
          "inputs": {
            "source_operands": [
              "ntc_pool"
            ]
          },
          "outputs": {
            "source_results": [
              "ntc_input",
              "ntc_target",
              "zero_mask"
            ]
          },
          "parameters": {
            "source_operation": "independent_ntc_set_sampling_and_zero_mask",
            "source_port_roles": [
              "target-domain control pool"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "sample_controls",
          "trajectory_id": "ultra_melanocyte_tta_single_step"
        }
      ],
      "source_operation": "independent_ntc_set_sampling_and_zero_mask",
      "status": "unexpanded_source_boundary"
    },
    {
      "occurrences": [
        {
          "call_id": "encode_groups::1",
          "component": "VAE marker-subset encoders",
          "condition": null,
          "evidence_ids": [
            "bca8cb6992d809fc1e66ba68b4dc2c6d0585f42fc0e7ea4326424c94fd91af39"
          ],
          "inputs": {
            "source_operands": [
              "marker_groups"
            ]
          },
          "outputs": {
            "source_results": [
              "mif_latents"
            ]
          },
          "parameters": {
            "source_operation": "independent_subset_latent_encoding",
            "source_port_roles": [
              "channel_subset_images"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "encode_groups",
          "trajectory_id": "finetune_mif_dual_condition"
        }
      ],
      "source_operation": "independent_subset_latent_encoding",
      "status": "unexpanded_source_boundary"
    },
    {
      "occurrences": [
        {
          "call_id": "initialize::1",
          "component": "Cumulative diffusion initializer",
          "condition": null,
          "evidence_ids": [
            "347089ce1327f071c6790467ae19079691059ad17ba3628bb6804409718bf1d3"
          ],
          "inputs": {
            "source_operands": [
              "inference_controls"
            ]
          },
          "outputs": {
            "source_results": [
              "x_initial",
              "mask_initial",
              "ranks"
            ]
          },
          "parameters": {
            "source_operation": "initialize_control_state_zero_mask_and_random_ranks",
            "source_port_roles": [
              "measured control values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "initialize",
          "trajectory_id": "xcell_jurkat_active_inference"
        },
        {
          "call_id": "initialize::1",
          "component": "Cumulative diffusion initializer",
          "condition": null,
          "evidence_ids": [
            "347089ce1327f071c6790467ae19079691059ad17ba3628bb6804409718bf1d3"
          ],
          "inputs": {
            "source_operands": [
              "inference_controls"
            ]
          },
          "outputs": {
            "source_results": [
              "x_initial",
              "mask_initial",
              "ranks"
            ]
          },
          "parameters": {
            "source_operation": "initialize_control_state_zero_mask_and_random_ranks",
            "source_port_roles": [
              "measured control values"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "initialize",
          "trajectory_id": "ultra_melanocyte_post_tta_inference"
        }
      ],
      "source_operation": "initialize_control_state_zero_mask_and_random_ranks",
      "status": "unexpanded_source_boundary"
    },
    {
      "occurrences": [
        {
          "call_id": "preceding_layers::1",
          "component": "X-Cell interleaved transformer encoder prefix",
          "condition": null,
          "evidence_ids": [
            "5ecc89cd6886a88843925414941ab1210bb9ac03a1026b585fde6638ce2b017b",
            "26cbe43fcf338175b9bfd731d1f4fe451091cfd62ef47a59ce6b6d3b36c9a7df",
            "16b955b34a6965c5d49893e29d511cf9de119212ff151108fa1fba83123e7c50",
            "568b01bb4868c4e1a142381a35fa7c21b74e888c297d228f5d2e804aa6c46ca1"
          ],
          "inputs": {
            "source_operands": [
              "tokens_with_cls",
              "context",
              "external_mask"
            ]
          },
          "outputs": {
            "source_results": [
              "query_state"
            ]
          },
          "parameters": {
            "source_operation": "interleaved_post_layer_norm_transformer_encoding",
            "source_port_roles": [
              "initial cell-gene sequence",
              "earlier cross-attention keys/values",
              "earlier cross-attention padding mask"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "preceding_layers",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "preceding_layers::1",
          "component": "X-Cell interleaved transformer encoder prefix",
          "condition": null,
          "evidence_ids": [
            "5ecc89cd6886a88843925414941ab1210bb9ac03a1026b585fde6638ce2b017b",
            "26cbe43fcf338175b9bfd731d1f4fe451091cfd62ef47a59ce6b6d3b36c9a7df",
            "16b955b34a6965c5d49893e29d511cf9de119212ff151108fa1fba83123e7c50",
            "568b01bb4868c4e1a142381a35fa7c21b74e888c297d228f5d2e804aa6c46ca1"
          ],
          "inputs": {
            "source_operands": [
              "tokens_with_cls",
              "context",
              "external_mask"
            ]
          },
          "outputs": {
            "source_results": [
              "query_state"
            ]
          },
          "parameters": {
            "source_operation": "interleaved_post_layer_norm_transformer_encoding",
            "source_port_roles": [
              "initial cell-gene sequence",
              "earlier cross-attention keys/values",
              "earlier cross-attention padding mask"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "preceding_layers",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "preceding_layers::1",
          "component": "X-Cell interleaved transformer encoder prefix",
          "condition": null,
          "evidence_ids": [
            "5ecc89cd6886a88843925414941ab1210bb9ac03a1026b585fde6638ce2b017b",
            "26cbe43fcf338175b9bfd731d1f4fe451091cfd62ef47a59ce6b6d3b36c9a7df",
            "16b955b34a6965c5d49893e29d511cf9de119212ff151108fa1fba83123e7c50",
            "568b01bb4868c4e1a142381a35fa7c21b74e888c297d228f5d2e804aa6c46ca1"
          ],
          "inputs": {
            "source_operands": [
              "tokens_with_cls",
              "context",
              "external_mask"
            ]
          },
          "outputs": {
            "source_results": [
              "query_state"
            ]
          },
          "parameters": {
            "source_operation": "interleaved_post_layer_norm_transformer_encoding",
            "source_port_roles": [
              "initial cell-gene sequence",
              "earlier cross-attention keys/values",
              "earlier cross-attention padding mask"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "preceding_layers",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "preceding_layers::1",
          "component": "X-Cell interleaved transformer encoder prefix",
          "condition": null,
          "evidence_ids": [
            "5ecc89cd6886a88843925414941ab1210bb9ac03a1026b585fde6638ce2b017b",
            "26cbe43fcf338175b9bfd731d1f4fe451091cfd62ef47a59ce6b6d3b36c9a7df",
            "16b955b34a6965c5d49893e29d511cf9de119212ff151108fa1fba83123e7c50",
            "568b01bb4868c4e1a142381a35fa7c21b74e888c297d228f5d2e804aa6c46ca1"
          ],
          "inputs": {
            "source_operands": [
              "tokens_with_cls",
              "context",
              "external_mask"
            ]
          },
          "outputs": {
            "source_results": [
              "query_state"
            ]
          },
          "parameters": {
            "source_operation": "interleaved_post_layer_norm_transformer_encoding",
            "source_port_roles": [
              "initial cell-gene sequence",
              "earlier cross-attention keys/values",
              "earlier cross-attention padding mask"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "preceding_layers",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "preceding_layers::1",
          "component": "X-Cell interleaved transformer encoder prefix",
          "condition": null,
          "evidence_ids": [
            "5ecc89cd6886a88843925414941ab1210bb9ac03a1026b585fde6638ce2b017b",
            "26cbe43fcf338175b9bfd731d1f4fe451091cfd62ef47a59ce6b6d3b36c9a7df",
            "16b955b34a6965c5d49893e29d511cf9de119212ff151108fa1fba83123e7c50",
            "568b01bb4868c4e1a142381a35fa7c21b74e888c297d228f5d2e804aa6c46ca1"
          ],
          "inputs": {
            "source_operands": [
              "tokens_with_cls",
              "context",
              "external_mask"
            ]
          },
          "outputs": {
            "source_results": [
              "query_state"
            ]
          },
          "parameters": {
            "source_operation": "interleaved_post_layer_norm_transformer_encoding",
            "source_port_roles": [
              "initial cell-gene sequence",
              "earlier cross-attention keys/values",
              "earlier cross-attention padding mask"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "preceding_layers",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        }
      ],
      "source_operation": "interleaved_post_layer_norm_transformer_encoding",
      "status": "unexpanded_source_boundary"
    },
    {
      "occurrences": [
        {
          "call_id": "preceding_layers::1",
          "component": "X-Cell-Ultra interleaved transformer encoder prefix",
          "condition": null,
          "evidence_ids": [
            "5ecc89cd6886a88843925414941ab1210bb9ac03a1026b585fde6638ce2b017b",
            "26cbe43fcf338175b9bfd731d1f4fe451091cfd62ef47a59ce6b6d3b36c9a7df",
            "16b955b34a6965c5d49893e29d511cf9de119212ff151108fa1fba83123e7c50",
            "568b01bb4868c4e1a142381a35fa7c21b74e888c297d228f5d2e804aa6c46ca1"
          ],
          "inputs": {
            "source_operands": [
              "tokens_with_cls",
              "context",
              "external_mask"
            ]
          },
          "outputs": {
            "source_results": [
              "query_state"
            ]
          },
          "parameters": {
            "source_operation": "interleaved_pre_rms_norm_swiglu_qk_norm_transformer_encoding",
            "source_port_roles": [
              "initial cell-gene sequence",
              "earlier cross-attention keys/values",
              "earlier cross-attention padding mask"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "preceding_layers",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "preceding_layers::1",
          "component": "X-Cell-Ultra interleaved transformer encoder prefix",
          "condition": null,
          "evidence_ids": [
            "5ecc89cd6886a88843925414941ab1210bb9ac03a1026b585fde6638ce2b017b",
            "26cbe43fcf338175b9bfd731d1f4fe451091cfd62ef47a59ce6b6d3b36c9a7df",
            "16b955b34a6965c5d49893e29d511cf9de119212ff151108fa1fba83123e7c50",
            "568b01bb4868c4e1a142381a35fa7c21b74e888c297d228f5d2e804aa6c46ca1"
          ],
          "inputs": {
            "source_operands": [
              "tokens_with_cls",
              "context",
              "external_mask"
            ]
          },
          "outputs": {
            "source_results": [
              "query_state"
            ]
          },
          "parameters": {
            "source_operation": "interleaved_pre_rms_norm_swiglu_qk_norm_transformer_encoding",
            "source_port_roles": [
              "initial cell-gene sequence",
              "earlier cross-attention keys/values",
              "earlier cross-attention padding mask"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "preceding_layers",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "preceding_layers::1",
          "component": "X-Cell-Ultra interleaved transformer encoder prefix",
          "condition": null,
          "evidence_ids": [
            "5ecc89cd6886a88843925414941ab1210bb9ac03a1026b585fde6638ce2b017b",
            "26cbe43fcf338175b9bfd731d1f4fe451091cfd62ef47a59ce6b6d3b36c9a7df",
            "16b955b34a6965c5d49893e29d511cf9de119212ff151108fa1fba83123e7c50",
            "568b01bb4868c4e1a142381a35fa7c21b74e888c297d228f5d2e804aa6c46ca1"
          ],
          "inputs": {
            "source_operands": [
              "tokens_with_cls",
              "context",
              "external_mask"
            ]
          },
          "outputs": {
            "source_results": [
              "query_state"
            ]
          },
          "parameters": {
            "source_operation": "interleaved_pre_rms_norm_swiglu_qk_norm_transformer_encoding",
            "source_port_roles": [
              "initial cell-gene sequence",
              "earlier cross-attention keys/values",
              "earlier cross-attention padding mask"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "preceding_layers",
          "trajectory_id": "ultra_training_tied_output_receipt"
        }
      ],
      "source_operation": "interleaved_pre_rms_norm_swiglu_qk_norm_transformer_encoding",
      "status": "unexpanded_source_boundary"
    },
    {
      "occurrences": [
        {
          "call_id": "decode_hidden::1",
          "component": "Expression decoder MLP",
          "condition": null,
          "evidence_ids": [
            "ecdf72423cf768656361070c11ac6419dabdd3f6e5a178b59dcb251e2fe6dd64"
          ],
          "inputs": {
            "source_operands": [
              "local_global"
            ]
          },
          "outputs": {
            "source_results": [
              "decoder_hidden"
            ]
          },
          "parameters": {
            "source_operation": "leaky_relu_mlp_decoding",
            "source_port_roles": [
              "concatenated local/global features"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "decode_hidden",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "decode_hidden::1",
          "component": "Expression decoder MLP",
          "condition": null,
          "evidence_ids": [
            "ecdf72423cf768656361070c11ac6419dabdd3f6e5a178b59dcb251e2fe6dd64"
          ],
          "inputs": {
            "source_operands": [
              "local_global"
            ]
          },
          "outputs": {
            "source_results": [
              "decoder_hidden"
            ]
          },
          "parameters": {
            "source_operation": "leaky_relu_mlp_decoding",
            "source_port_roles": [
              "concatenated local/global features"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "decode_hidden",
          "trajectory_id": "ultra_training_tied_output_receipt"
        }
      ],
      "source_operation": "leaky_relu_mlp_decoding",
      "status": "unexpanded_source_boundary"
    },
    {
      "occurrences": [
        {
          "call_id": "attend_image::1",
          "component": "DCA image attention stream",
          "condition": null,
          "evidence_ids": [
            "36a4f4911fe6bde54a23461d243131b836ee48d139ad0e6b0160fb2ba4c56554",
            "a397204b6677562d7112753b0d2ca86a8eb6e62745fb7347e2b2f8070e27c88f"
          ],
          "inputs": {
            "source_operands": [
              "image_condition",
              "h",
              "availability"
            ]
          },
          "outputs": {
            "source_results": [
              "image_contribution"
            ]
          },
          "parameters": {
            "source_operation": "modality_attention_stream_processing",
            "source_port_roles": [
              "modality_condition",
              "image_query",
              "modality_activity_control"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "attend_image",
          "trajectory_id": "pretrain_multimodal_dca"
        },
        {
          "call_id": "attend_text::1",
          "component": "DCA text attention stream",
          "condition": null,
          "evidence_ids": [
            "36a4f4911fe6bde54a23461d243131b836ee48d139ad0e6b0160fb2ba4c56554",
            "a397204b6677562d7112753b0d2ca86a8eb6e62745fb7347e2b2f8070e27c88f"
          ],
          "inputs": {
            "source_operands": [
              "text_condition",
              "h",
              "availability"
            ]
          },
          "outputs": {
            "source_results": [
              "text_contribution"
            ]
          },
          "parameters": {
            "source_operation": "modality_attention_stream_processing",
            "source_port_roles": [
              "modality_condition",
              "image_query",
              "modality_activity_control"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "attend_text",
          "trajectory_id": "pretrain_multimodal_dca"
        },
        {
          "call_id": "attend_rna::1",
          "component": "DCA RNA modality stream",
          "condition": null,
          "evidence_ids": [
            "36a4f4911fe6bde54a23461d243131b836ee48d139ad0e6b0160fb2ba4c56554",
            "a397204b6677562d7112753b0d2ca86a8eb6e62745fb7347e2b2f8070e27c88f",
            "aa043c88a985cb5ffd3ca884a536aba656ab16826147e73069f0d75ec41e02ca"
          ],
          "inputs": {
            "source_operands": [
              "rna_attention_condition",
              "h",
              "availability"
            ]
          },
          "outputs": {
            "source_results": [
              "rna_contribution"
            ]
          },
          "parameters": {
            "source_operation": "modality_attention_stream_processing",
            "source_port_roles": [
              "rna_stream_condition_boundary",
              "image_query",
              "modality_activity_control"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "attend_rna",
          "trajectory_id": "pretrain_multimodal_dca"
        }
      ],
      "source_operation": "modality_attention_stream_processing",
      "status": "unexpanded_source_boundary"
    },
    {
      "occurrences": [
        {
          "call_id": "identify::1",
          "component": "Shared-CA modality augmentation",
          "condition": null,
          "evidence_ids": [
            "88101d2dce98a7d30784efd305607aeef3695140dfcda29ef049d6a456235b7f"
          ],
          "inputs": {
            "source_operands": [
              "joint_conditions",
              "modality_embeddings"
            ]
          },
          "outputs": {
            "source_results": [
              "identified_conditions"
            ]
          },
          "parameters": {
            "source_operation": "modality_identity_augmentation",
            "source_port_roles": [
              "joint_conditions",
              "source_identity_context"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "identify",
          "trajectory_id": "shared_ca_image_to_image_training"
        },
        {
          "call_id": "identify::1",
          "component": "Shared-CA modality augmentation",
          "condition": null,
          "evidence_ids": [
            "88101d2dce98a7d30784efd305607aeef3695140dfcda29ef049d6a456235b7f"
          ],
          "inputs": {
            "source_operands": [
              "joint_conditions",
              "modality_embeddings"
            ]
          },
          "outputs": {
            "source_results": [
              "identified_conditions"
            ]
          },
          "parameters": {
            "source_operation": "modality_identity_augmentation",
            "source_port_roles": [
              "joint_conditions",
              "source_identity_context"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "identify",
          "trajectory_id": "shared_ca_image_to_image_inference"
        },
        {
          "call_id": "identify::1",
          "component": "Shared-CA modality augmentation",
          "condition": null,
          "evidence_ids": [
            "88101d2dce98a7d30784efd305607aeef3695140dfcda29ef049d6a456235b7f"
          ],
          "inputs": {
            "source_operands": [
              "joint_conditions",
              "modality_embeddings"
            ]
          },
          "outputs": {
            "source_results": [
              "identified_conditions"
            ]
          },
          "parameters": {
            "source_operation": "modality_identity_augmentation",
            "source_port_roles": [
              "joint_conditions",
              "source_identity_context"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "identify",
          "trajectory_id": "shared_ca_text_to_image_training"
        },
        {
          "call_id": "identify::1",
          "component": "Shared-CA modality augmentation",
          "condition": null,
          "evidence_ids": [
            "88101d2dce98a7d30784efd305607aeef3695140dfcda29ef049d6a456235b7f"
          ],
          "inputs": {
            "source_operands": [
              "joint_conditions",
              "modality_embeddings"
            ]
          },
          "outputs": {
            "source_results": [
              "identified_conditions"
            ]
          },
          "parameters": {
            "source_operation": "modality_identity_augmentation",
            "source_port_roles": [
              "joint_conditions",
              "source_identity_context"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "identify",
          "trajectory_id": "shared_ca_text_to_image_inference"
        },
        {
          "call_id": "identify::1",
          "component": "Shared-CA modality augmentation",
          "condition": null,
          "evidence_ids": [
            "88101d2dce98a7d30784efd305607aeef3695140dfcda29ef049d6a456235b7f"
          ],
          "inputs": {
            "source_operands": [
              "joint_conditions",
              "modality_embeddings"
            ]
          },
          "outputs": {
            "source_results": [
              "identified_conditions"
            ]
          },
          "parameters": {
            "source_operation": "modality_identity_augmentation",
            "source_port_roles": [
              "joint_conditions",
              "source_identity_context"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "identify",
          "trajectory_id": "shared_ca_RNA_to_image_training"
        },
        {
          "call_id": "identify::1",
          "component": "Shared-CA modality augmentation",
          "condition": null,
          "evidence_ids": [
            "88101d2dce98a7d30784efd305607aeef3695140dfcda29ef049d6a456235b7f"
          ],
          "inputs": {
            "source_operands": [
              "joint_conditions",
              "modality_embeddings"
            ]
          },
          "outputs": {
            "source_results": [
              "identified_conditions"
            ]
          },
          "parameters": {
            "source_operation": "modality_identity_augmentation",
            "source_port_roles": [
              "joint_conditions",
              "source_identity_context"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "identify",
          "trajectory_id": "shared_ca_RNA_to_image_inference"
        }
      ],
      "source_operation": "modality_identity_augmentation",
      "status": "unexpanded_source_boundary"
    },
    {
      "occurrences": [
        {
          "call_id": "decode_first::1",
          "component": "Vicuna-7b English decoder",
          "condition": null,
          "evidence_ids": [
            "3d3504602ecdd447ac3a8bc9b8c6029e3b572e30978ffa0813626ecc374ba6b0",
            "dfdd5ff10cb42c401f0c344f517d4536c0473fa7e78688d2b07ceaca468ae4e2"
          ],
          "inputs": {
            "source_operands": [
              "fused_input"
            ]
          },
          "outputs": {
            "source_results": [
              "word_distribution_t",
              "cache_t"
            ]
          },
          "parameters": {
            "source_operation": "next_token_distribution_and_attention_caching",
            "source_port_roles": [
              "multimodal_prompt_embeddings"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "decode_first",
          "trajectory_id": "benchmark_inference"
        },
        {
          "call_id": "decode_first::1",
          "component": "Vicuna-7b English decoder",
          "condition": null,
          "evidence_ids": [
            "3d3504602ecdd447ac3a8bc9b8c6029e3b572e30978ffa0813626ecc374ba6b0",
            "dfdd5ff10cb42c401f0c344f517d4536c0473fa7e78688d2b07ceaca468ae4e2"
          ],
          "inputs": {
            "source_operands": [
              "fused_input"
            ]
          },
          "outputs": {
            "source_results": [
              "word_distribution_t",
              "cache_t"
            ]
          },
          "parameters": {
            "source_operation": "next_token_distribution_and_attention_caching",
            "source_port_roles": [
              "multimodal_prompt_embeddings"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "decode_first",
          "trajectory_id": "curated_dna_inference"
        },
        {
          "call_id": "decode_first::1",
          "component": "Vicuna-7b English decoder",
          "condition": null,
          "evidence_ids": [
            "3d3504602ecdd447ac3a8bc9b8c6029e3b572e30978ffa0813626ecc374ba6b0",
            "dfdd5ff10cb42c401f0c344f517d4536c0473fa7e78688d2b07ceaca468ae4e2"
          ],
          "inputs": {
            "source_operands": [
              "fused_input"
            ]
          },
          "outputs": {
            "source_results": [
              "word_distribution_t",
              "cache_t"
            ]
          },
          "parameters": {
            "source_operation": "next_token_distribution_and_attention_caching",
            "source_port_roles": [
              "multimodal_prompt_embeddings"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "decode_first",
          "trajectory_id": "curated_rna_inference"
        },
        {
          "call_id": "decode_first::1",
          "component": "Vicuna-7b English decoder",
          "condition": null,
          "evidence_ids": [
            "3d3504602ecdd447ac3a8bc9b8c6029e3b572e30978ffa0813626ecc374ba6b0",
            "dfdd5ff10cb42c401f0c344f517d4536c0473fa7e78688d2b07ceaca468ae4e2"
          ],
          "inputs": {
            "source_operands": [
              "fused_input"
            ]
          },
          "outputs": {
            "source_results": [
              "word_distribution_t",
              "cache_t"
            ]
          },
          "parameters": {
            "source_operation": "next_token_distribution_and_attention_caching",
            "source_port_roles": [
              "multimodal_prompt_embeddings"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "decode_first",
          "trajectory_id": "curated_protein_inference"
        },
        {
          "call_id": "decode_first::1",
          "component": "Vicuna-7b English decoder",
          "condition": null,
          "evidence_ids": [
            "3d3504602ecdd447ac3a8bc9b8c6029e3b572e30978ffa0813626ecc374ba6b0",
            "dfdd5ff10cb42c401f0c344f517d4536c0473fa7e78688d2b07ceaca468ae4e2"
          ],
          "inputs": {
            "source_operands": [
              "fused_input"
            ]
          },
          "outputs": {
            "source_results": [
              "word_distribution_t",
              "cache_t"
            ]
          },
          "parameters": {
            "source_operation": "next_token_distribution_and_attention_caching",
            "source_port_roles": [
              "multimodal_prompt_embeddings"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "decode_first",
          "trajectory_id": "multiple_inference"
        },
        {
          "call_id": "decode_first::1",
          "component": "Vicuna-7b English decoder",
          "condition": null,
          "evidence_ids": [
            "3d3504602ecdd447ac3a8bc9b8c6029e3b572e30978ffa0813626ecc374ba6b0",
            "dfdd5ff10cb42c401f0c344f517d4536c0473fa7e78688d2b07ceaca468ae4e2"
          ],
          "inputs": {
            "source_operands": [
              "fused_input"
            ]
          },
          "outputs": {
            "source_results": [
              "word_distribution_t",
              "cache_t"
            ]
          },
          "parameters": {
            "source_operation": "next_token_distribution_and_attention_caching",
            "source_port_roles": [
              "multimodal_prompt_embeddings"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "decode_first",
          "trajectory_id": "unconditioned_benchmark_inference"
        }
      ],
      "source_operation": "next_token_distribution_and_attention_caching",
      "status": "unexpanded_source_boundary"
    },
    {
      "occurrences": [
        {
          "call_id": "semantic::1",
          "component": "MUSK image encoder",
          "condition": null,
          "evidence_ids": [
            "bca8cb6992d809fc1e66ba68b4dc2c6d0585f42fc0e7ea4326424c94fd91af39"
          ],
          "inputs": {
            "source_operands": [
              "he_image"
            ]
          },
          "outputs": {
            "source_results": [
              "semantic_embeddings"
            ]
          },
          "parameters": {
            "source_operation": "patch_semantic_encoding",
            "source_port_roles": [
              "semantic_histology"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "semantic",
          "trajectory_id": "finetune_mif_dual_condition"
        }
      ],
      "source_operation": "patch_semantic_encoding",
      "status": "unexpanded_source_boundary"
    },
    {
      "occurrences": [
        {
          "call_id": "insert_sequences::1",
          "component": "ChatNT placeholder substitution",
          "condition": null,
          "evidence_ids": [
            "3d3504602ecdd447ac3a8bc9b8c6029e3b572e30978ffa0813626ecc374ba6b0",
            "66f967ac6c448ed7cef8aa2b5fab9ae881c5cd419add6c080ee4ee75e1f46aa8"
          ],
          "inputs": {
            "source_operands": [
              "english_embeddings",
              "resampled_a"
            ]
          },
          "outputs": {
            "source_results": [
              "fused_input"
            ]
          },
          "parameters": {
            "source_operation": "positional_embedding_interleaving",
            "source_port_roles": [
              "english_positions_and_placeholder_layout",
              "replacement_for_sequence_a_placeholder"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "insert_sequences",
          "trajectory_id": "benchmark_fine_tuning"
        },
        {
          "call_id": "insert_sequences::1",
          "component": "ChatNT placeholder substitution",
          "condition": null,
          "evidence_ids": [
            "3d3504602ecdd447ac3a8bc9b8c6029e3b572e30978ffa0813626ecc374ba6b0",
            "66f967ac6c448ed7cef8aa2b5fab9ae881c5cd419add6c080ee4ee75e1f46aa8"
          ],
          "inputs": {
            "source_operands": [
              "english_embeddings",
              "resampled_a"
            ]
          },
          "outputs": {
            "source_results": [
              "fused_input"
            ]
          },
          "parameters": {
            "source_operation": "positional_embedding_interleaving",
            "source_port_roles": [
              "english_positions_and_placeholder_layout",
              "replacement_for_sequence_a_placeholder"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "insert_sequences",
          "trajectory_id": "benchmark_inference"
        },
        {
          "call_id": "insert_sequences::1",
          "component": "ChatNT placeholder substitution",
          "condition": null,
          "evidence_ids": [
            "3d3504602ecdd447ac3a8bc9b8c6029e3b572e30978ffa0813626ecc374ba6b0",
            "66f967ac6c448ed7cef8aa2b5fab9ae881c5cd419add6c080ee4ee75e1f46aa8"
          ],
          "inputs": {
            "source_operands": [
              "english_embeddings",
              "resampled_a"
            ]
          },
          "outputs": {
            "source_results": [
              "fused_input"
            ]
          },
          "parameters": {
            "source_operation": "positional_embedding_interleaving",
            "source_port_roles": [
              "english_positions_and_placeholder_layout",
              "replacement_for_sequence_a_placeholder"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "insert_sequences",
          "trajectory_id": "curated_dna_fine_tuning"
        },
        {
          "call_id": "insert_sequences::1",
          "component": "ChatNT placeholder substitution",
          "condition": null,
          "evidence_ids": [
            "3d3504602ecdd447ac3a8bc9b8c6029e3b572e30978ffa0813626ecc374ba6b0",
            "66f967ac6c448ed7cef8aa2b5fab9ae881c5cd419add6c080ee4ee75e1f46aa8"
          ],
          "inputs": {
            "source_operands": [
              "english_embeddings",
              "resampled_a"
            ]
          },
          "outputs": {
            "source_results": [
              "fused_input"
            ]
          },
          "parameters": {
            "source_operation": "positional_embedding_interleaving",
            "source_port_roles": [
              "english_positions_and_placeholder_layout",
              "replacement_for_sequence_a_placeholder"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "insert_sequences",
          "trajectory_id": "curated_dna_inference"
        },
        {
          "call_id": "insert_sequences::1",
          "component": "ChatNT placeholder substitution",
          "condition": null,
          "evidence_ids": [
            "3d3504602ecdd447ac3a8bc9b8c6029e3b572e30978ffa0813626ecc374ba6b0",
            "66f967ac6c448ed7cef8aa2b5fab9ae881c5cd419add6c080ee4ee75e1f46aa8"
          ],
          "inputs": {
            "source_operands": [
              "english_embeddings",
              "resampled_a"
            ]
          },
          "outputs": {
            "source_results": [
              "fused_input"
            ]
          },
          "parameters": {
            "source_operation": "positional_embedding_interleaving",
            "source_port_roles": [
              "english_positions_and_placeholder_layout",
              "replacement_for_sequence_a_placeholder"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "insert_sequences",
          "trajectory_id": "curated_rna_fine_tuning"
        },
        {
          "call_id": "insert_sequences::1",
          "component": "ChatNT placeholder substitution",
          "condition": null,
          "evidence_ids": [
            "3d3504602ecdd447ac3a8bc9b8c6029e3b572e30978ffa0813626ecc374ba6b0",
            "66f967ac6c448ed7cef8aa2b5fab9ae881c5cd419add6c080ee4ee75e1f46aa8"
          ],
          "inputs": {
            "source_operands": [
              "english_embeddings",
              "resampled_a"
            ]
          },
          "outputs": {
            "source_results": [
              "fused_input"
            ]
          },
          "parameters": {
            "source_operation": "positional_embedding_interleaving",
            "source_port_roles": [
              "english_positions_and_placeholder_layout",
              "replacement_for_sequence_a_placeholder"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "insert_sequences",
          "trajectory_id": "curated_rna_inference"
        },
        {
          "call_id": "insert_sequences::1",
          "component": "ChatNT placeholder substitution",
          "condition": null,
          "evidence_ids": [
            "3d3504602ecdd447ac3a8bc9b8c6029e3b572e30978ffa0813626ecc374ba6b0",
            "66f967ac6c448ed7cef8aa2b5fab9ae881c5cd419add6c080ee4ee75e1f46aa8"
          ],
          "inputs": {
            "source_operands": [
              "english_embeddings",
              "resampled_a"
            ]
          },
          "outputs": {
            "source_results": [
              "fused_input"
            ]
          },
          "parameters": {
            "source_operation": "positional_embedding_interleaving",
            "source_port_roles": [
              "english_positions_and_placeholder_layout",
              "replacement_for_sequence_a_placeholder"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "insert_sequences",
          "trajectory_id": "curated_protein_fine_tuning"
        },
        {
          "call_id": "insert_sequences::1",
          "component": "ChatNT placeholder substitution",
          "condition": null,
          "evidence_ids": [
            "3d3504602ecdd447ac3a8bc9b8c6029e3b572e30978ffa0813626ecc374ba6b0",
            "66f967ac6c448ed7cef8aa2b5fab9ae881c5cd419add6c080ee4ee75e1f46aa8"
          ],
          "inputs": {
            "source_operands": [
              "english_embeddings",
              "resampled_a"
            ]
          },
          "outputs": {
            "source_results": [
              "fused_input"
            ]
          },
          "parameters": {
            "source_operation": "positional_embedding_interleaving",
            "source_port_roles": [
              "english_positions_and_placeholder_layout",
              "replacement_for_sequence_a_placeholder"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "insert_sequences",
          "trajectory_id": "curated_protein_inference"
        },
        {
          "call_id": "insert_sequences::1",
          "component": "ChatNT placeholder substitution",
          "condition": null,
          "evidence_ids": [
            "3d3504602ecdd447ac3a8bc9b8c6029e3b572e30978ffa0813626ecc374ba6b0",
            "66f967ac6c448ed7cef8aa2b5fab9ae881c5cd419add6c080ee4ee75e1f46aa8"
          ],
          "inputs": {
            "source_operands": [
              "english_embeddings",
              "resampled_a",
              "resampled_b"
            ]
          },
          "outputs": {
            "source_results": [
              "fused_input"
            ]
          },
          "parameters": {
            "source_operation": "positional_embedding_interleaving",
            "source_port_roles": [
              "english_positions_and_placeholder_layout",
              "replacement_for_sequence_a_placeholder",
              "replacement_for_sequence_b_placeholder"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "insert_sequences",
          "trajectory_id": "multiple_fine_tuning"
        },
        {
          "call_id": "insert_sequences::1",
          "component": "ChatNT placeholder substitution",
          "condition": null,
          "evidence_ids": [
            "3d3504602ecdd447ac3a8bc9b8c6029e3b572e30978ffa0813626ecc374ba6b0",
            "66f967ac6c448ed7cef8aa2b5fab9ae881c5cd419add6c080ee4ee75e1f46aa8"
          ],
          "inputs": {
            "source_operands": [
              "english_embeddings",
              "resampled_a",
              "resampled_b"
            ]
          },
          "outputs": {
            "source_results": [
              "fused_input"
            ]
          },
          "parameters": {
            "source_operation": "positional_embedding_interleaving",
            "source_port_roles": [
              "english_positions_and_placeholder_layout",
              "replacement_for_sequence_a_placeholder",
              "replacement_for_sequence_b_placeholder"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "insert_sequences",
          "trajectory_id": "multiple_inference"
        },
        {
          "call_id": "insert_sequences::1",
          "component": "ChatNT placeholder substitution",
          "condition": null,
          "evidence_ids": [
            "3d3504602ecdd447ac3a8bc9b8c6029e3b572e30978ffa0813626ecc374ba6b0",
            "66f967ac6c448ed7cef8aa2b5fab9ae881c5cd419add6c080ee4ee75e1f46aa8"
          ],
          "inputs": {
            "source_operands": [
              "english_embeddings",
              "resampled_a"
            ]
          },
          "outputs": {
            "source_results": [
              "fused_input"
            ]
          },
          "parameters": {
            "source_operation": "positional_embedding_interleaving",
            "source_port_roles": [
              "english_positions_and_placeholder_layout",
              "replacement_for_sequence_a_placeholder"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "insert_sequences",
          "trajectory_id": "unconditioned_benchmark_fine_tuning"
        },
        {
          "call_id": "insert_sequences::1",
          "component": "ChatNT placeholder substitution",
          "condition": null,
          "evidence_ids": [
            "3d3504602ecdd447ac3a8bc9b8c6029e3b572e30978ffa0813626ecc374ba6b0",
            "66f967ac6c448ed7cef8aa2b5fab9ae881c5cd419add6c080ee4ee75e1f46aa8"
          ],
          "inputs": {
            "source_operands": [
              "english_embeddings",
              "resampled_a"
            ]
          },
          "outputs": {
            "source_results": [
              "fused_input"
            ]
          },
          "parameters": {
            "source_operation": "positional_embedding_interleaving",
            "source_port_roles": [
              "english_positions_and_placeholder_layout",
              "replacement_for_sequence_a_placeholder"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "insert_sequences",
          "trajectory_id": "unconditioned_benchmark_inference"
        }
      ],
      "source_operation": "positional_embedding_interleaving",
      "status": "unexpanded_source_boundary"
    },
    {
      "occurrences": [
        {
          "call_id": "encode_posterior::1",
          "component": "CVAE posterior encoder",
          "condition": null,
          "evidence_ids": [
            "4ea2234ed06e6e3c7636818175ef32d6c430973e09dcd37014085863aaa3dbff",
            "9c29bcd5ebf39c41bdb3d4c9a585c4c9dffc2fb7b99e5128a1ff7caf8ada67d3"
          ],
          "inputs": {
            "source_operands": [
              "target_cell",
              "condition"
            ]
          },
          "outputs": {
            "source_results": [
              "posterior_parameters"
            ]
          },
          "parameters": {
            "source_operation": "posterior_parameter_encoding",
            "source_port_roles": [
              "observed_target_cell",
              "condition"
            ]
          },
          "record_id": "full_2026-07-06__rec_001319",
          "source_step_id": "encode_posterior",
          "trajectory_id": "cpcg_training"
        }
      ],
      "source_operation": "posterior_parameter_encoding",
      "status": "unexpanded_source_boundary"
    },
    {
      "occurrences": [
        {
          "call_id": "resample_dna_a::1",
          "component": "English-aware Perceiver resampler",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f",
            "b2a310307da6950793dff7844f48ee9a07758d8c2baa00e20e9de62b6a03aa9e"
          ],
          "inputs": {
            "source_operands": [
              "queries_a",
              "projected_a",
              "english_embeddings"
            ]
          },
          "outputs": {
            "source_results": [
              "resampled_a"
            ]
          },
          "parameters": {
            "source_operation": "question_conditioned_cross_attention_resampling",
            "source_port_roles": [
              "learnable_queries",
              "dna_source_embeddings",
              "provisional_question_context_representation_and_port"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "resample_dna_a",
          "trajectory_id": "benchmark_fine_tuning"
        },
        {
          "call_id": "resample_dna_a::1",
          "component": "English-aware Perceiver resampler",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f",
            "b2a310307da6950793dff7844f48ee9a07758d8c2baa00e20e9de62b6a03aa9e"
          ],
          "inputs": {
            "source_operands": [
              "queries_a",
              "projected_a",
              "english_embeddings"
            ]
          },
          "outputs": {
            "source_results": [
              "resampled_a"
            ]
          },
          "parameters": {
            "source_operation": "question_conditioned_cross_attention_resampling",
            "source_port_roles": [
              "learnable_queries",
              "dna_source_embeddings",
              "provisional_question_context_representation_and_port"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "resample_dna_a",
          "trajectory_id": "benchmark_inference"
        },
        {
          "call_id": "resample_dna_a::1",
          "component": "English-aware Perceiver resampler",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f",
            "b2a310307da6950793dff7844f48ee9a07758d8c2baa00e20e9de62b6a03aa9e"
          ],
          "inputs": {
            "source_operands": [
              "queries_a",
              "projected_a",
              "english_embeddings"
            ]
          },
          "outputs": {
            "source_results": [
              "resampled_a"
            ]
          },
          "parameters": {
            "source_operation": "question_conditioned_cross_attention_resampling",
            "source_port_roles": [
              "learnable_queries",
              "dna_source_embeddings",
              "provisional_question_context_representation_and_port"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "resample_dna_a",
          "trajectory_id": "curated_dna_fine_tuning"
        },
        {
          "call_id": "resample_dna_a::1",
          "component": "English-aware Perceiver resampler",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f",
            "b2a310307da6950793dff7844f48ee9a07758d8c2baa00e20e9de62b6a03aa9e"
          ],
          "inputs": {
            "source_operands": [
              "queries_a",
              "projected_a",
              "english_embeddings"
            ]
          },
          "outputs": {
            "source_results": [
              "resampled_a"
            ]
          },
          "parameters": {
            "source_operation": "question_conditioned_cross_attention_resampling",
            "source_port_roles": [
              "learnable_queries",
              "dna_source_embeddings",
              "provisional_question_context_representation_and_port"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "resample_dna_a",
          "trajectory_id": "curated_dna_inference"
        },
        {
          "call_id": "resample_dna_a::1",
          "component": "English-aware Perceiver resampler",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f",
            "b2a310307da6950793dff7844f48ee9a07758d8c2baa00e20e9de62b6a03aa9e"
          ],
          "inputs": {
            "source_operands": [
              "queries_a",
              "projected_a",
              "english_embeddings"
            ]
          },
          "outputs": {
            "source_results": [
              "resampled_a"
            ]
          },
          "parameters": {
            "source_operation": "question_conditioned_cross_attention_resampling",
            "source_port_roles": [
              "learnable_queries",
              "dna_source_embeddings",
              "provisional_question_context_representation_and_port"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "resample_dna_a",
          "trajectory_id": "curated_rna_fine_tuning"
        },
        {
          "call_id": "resample_dna_a::1",
          "component": "English-aware Perceiver resampler",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f",
            "b2a310307da6950793dff7844f48ee9a07758d8c2baa00e20e9de62b6a03aa9e"
          ],
          "inputs": {
            "source_operands": [
              "queries_a",
              "projected_a",
              "english_embeddings"
            ]
          },
          "outputs": {
            "source_results": [
              "resampled_a"
            ]
          },
          "parameters": {
            "source_operation": "question_conditioned_cross_attention_resampling",
            "source_port_roles": [
              "learnable_queries",
              "dna_source_embeddings",
              "provisional_question_context_representation_and_port"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "resample_dna_a",
          "trajectory_id": "curated_rna_inference"
        },
        {
          "call_id": "resample_dna_a::1",
          "component": "English-aware Perceiver resampler",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f",
            "b2a310307da6950793dff7844f48ee9a07758d8c2baa00e20e9de62b6a03aa9e"
          ],
          "inputs": {
            "source_operands": [
              "queries_a",
              "projected_a",
              "english_embeddings"
            ]
          },
          "outputs": {
            "source_results": [
              "resampled_a"
            ]
          },
          "parameters": {
            "source_operation": "question_conditioned_cross_attention_resampling",
            "source_port_roles": [
              "learnable_queries",
              "dna_source_embeddings",
              "provisional_question_context_representation_and_port"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "resample_dna_a",
          "trajectory_id": "curated_protein_fine_tuning"
        },
        {
          "call_id": "resample_dna_a::1",
          "component": "English-aware Perceiver resampler",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f",
            "b2a310307da6950793dff7844f48ee9a07758d8c2baa00e20e9de62b6a03aa9e"
          ],
          "inputs": {
            "source_operands": [
              "queries_a",
              "projected_a",
              "english_embeddings"
            ]
          },
          "outputs": {
            "source_results": [
              "resampled_a"
            ]
          },
          "parameters": {
            "source_operation": "question_conditioned_cross_attention_resampling",
            "source_port_roles": [
              "learnable_queries",
              "dna_source_embeddings",
              "provisional_question_context_representation_and_port"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "resample_dna_a",
          "trajectory_id": "curated_protein_inference"
        },
        {
          "call_id": "resample_dna_a::1",
          "component": "English-aware Perceiver resampler",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f",
            "b2a310307da6950793dff7844f48ee9a07758d8c2baa00e20e9de62b6a03aa9e"
          ],
          "inputs": {
            "source_operands": [
              "query_parameter_collection_unresolved",
              "projected_a",
              "english_embeddings"
            ]
          },
          "outputs": {
            "source_results": [
              "resampled_a"
            ]
          },
          "parameters": {
            "source_operation": "question_conditioned_cross_attention_resampling",
            "source_port_roles": [
              "learnable_query_selection_for_sequence_a_identity_unresolved",
              "dna_source_embeddings",
              "provisional_question_context_representation_and_port"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "resample_dna_a",
          "trajectory_id": "multiple_fine_tuning"
        },
        {
          "call_id": "resample_dna_b::1",
          "component": "English-aware Perceiver resampler",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f",
            "b2a310307da6950793dff7844f48ee9a07758d8c2baa00e20e9de62b6a03aa9e"
          ],
          "inputs": {
            "source_operands": [
              "query_parameter_collection_unresolved",
              "projected_b",
              "english_embeddings"
            ]
          },
          "outputs": {
            "source_results": [
              "resampled_b"
            ]
          },
          "parameters": {
            "source_operation": "question_conditioned_cross_attention_resampling",
            "source_port_roles": [
              "learnable_query_selection_for_sequence_b_identity_unresolved",
              "dna_source_embeddings",
              "provisional_question_context_representation_and_port"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "resample_dna_b",
          "trajectory_id": "multiple_fine_tuning"
        },
        {
          "call_id": "resample_dna_a::1",
          "component": "English-aware Perceiver resampler",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f",
            "b2a310307da6950793dff7844f48ee9a07758d8c2baa00e20e9de62b6a03aa9e"
          ],
          "inputs": {
            "source_operands": [
              "query_parameter_collection_unresolved",
              "projected_a",
              "english_embeddings"
            ]
          },
          "outputs": {
            "source_results": [
              "resampled_a"
            ]
          },
          "parameters": {
            "source_operation": "question_conditioned_cross_attention_resampling",
            "source_port_roles": [
              "learnable_query_selection_for_sequence_a_identity_unresolved",
              "dna_source_embeddings",
              "provisional_question_context_representation_and_port"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "resample_dna_a",
          "trajectory_id": "multiple_inference"
        },
        {
          "call_id": "resample_dna_b::1",
          "component": "English-aware Perceiver resampler",
          "condition": null,
          "evidence_ids": [
            "91155052a7fae86e2d5355d468495f013a0cfc57d93df52c4047f38bf519232f",
            "b2a310307da6950793dff7844f48ee9a07758d8c2baa00e20e9de62b6a03aa9e"
          ],
          "inputs": {
            "source_operands": [
              "query_parameter_collection_unresolved",
              "projected_b",
              "english_embeddings"
            ]
          },
          "outputs": {
            "source_results": [
              "resampled_b"
            ]
          },
          "parameters": {
            "source_operation": "question_conditioned_cross_attention_resampling",
            "source_port_roles": [
              "learnable_query_selection_for_sequence_b_identity_unresolved",
              "dna_source_embeddings",
              "provisional_question_context_representation_and_port"
            ]
          },
          "record_id": "full_2026-07-06__rec_000771",
          "source_step_id": "resample_dna_b",
          "trajectory_id": "multiple_inference"
        }
      ],
      "source_operation": "question_conditioned_cross_attention_resampling",
      "status": "unexpanded_source_boundary"
    },
    {
      "occurrences": [
        {
          "call_id": "tokenize::1",
          "component": "C2S training tokenizer and collator",
          "condition": null,
          "evidence_ids": [
            "e874b5e850f3fddeea6ac3ffc4ed09d68a7bd2d7d3d0ccbc4445142371816fc7"
          ],
          "inputs": {
            "source_operands": [
              "prompt",
              "target_sentence"
            ]
          },
          "outputs": {
            "source_results": [
              "training_tokens",
              "response_labels"
            ]
          },
          "parameters": {
            "source_operation": "reported_independent_prompt_response_tokenization_and_masking",
            "source_port_roles": [
              "formatted control/perturbation prompt",
              "paired perturbed response"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "tokenize",
          "trajectory_id": "c2s_scale_fine_tuning_prompt_receipt"
        }
      ],
      "source_operation": "reported_independent_prompt_response_tokenization_and_masking",
      "status": "unexpanded_source_boundary"
    },
    {
      "occurrences": [
        {
          "call_id": "sample_reveal::1",
          "component": "Training diffusion mask sampler",
          "condition": null,
          "evidence_ids": [
            "b2219fd688af43a3c418ed9b0da8421573eff83e4c8cbb0ad2816047f40de556",
            "bfab58c391ecd6cb7ddedddaa46b8afaa05c93d15f1e0db2b871170a55577767"
          ],
          "inputs": {
            "source_operands": [
              "selected_genes"
            ]
          },
          "outputs": {
            "source_results": [
              "replacement_fraction",
              "reveal_positions"
            ]
          },
          "parameters": {
            "source_operation": "sample_reveal_fraction_and_positions",
            "source_port_roles": [
              "eligible gene positions"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "sample_reveal",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "sample_reveal::1",
          "component": "Training diffusion mask sampler",
          "condition": null,
          "evidence_ids": [
            "b2219fd688af43a3c418ed9b0da8421573eff83e4c8cbb0ad2816047f40de556",
            "bfab58c391ecd6cb7ddedddaa46b8afaa05c93d15f1e0db2b871170a55577767"
          ],
          "inputs": {
            "source_operands": [
              "selected_genes"
            ]
          },
          "outputs": {
            "source_results": [
              "replacement_fraction",
              "reveal_positions"
            ]
          },
          "parameters": {
            "source_operation": "sample_reveal_fraction_and_positions",
            "source_port_roles": [
              "eligible gene positions"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "sample_reveal",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "sample_reveal::1",
          "component": "Training diffusion mask sampler",
          "condition": null,
          "evidence_ids": [
            "b2219fd688af43a3c418ed9b0da8421573eff83e4c8cbb0ad2816047f40de556",
            "bfab58c391ecd6cb7ddedddaa46b8afaa05c93d15f1e0db2b871170a55577767"
          ],
          "inputs": {
            "source_operands": [
              "selected_genes"
            ]
          },
          "outputs": {
            "source_results": [
              "replacement_fraction",
              "reveal_positions"
            ]
          },
          "parameters": {
            "source_operation": "sample_reveal_fraction_and_positions",
            "source_port_roles": [
              "eligible gene positions"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "sample_reveal",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "sample_reveal::1",
          "component": "Training diffusion mask sampler",
          "condition": null,
          "evidence_ids": [
            "b2219fd688af43a3c418ed9b0da8421573eff83e4c8cbb0ad2816047f40de556",
            "bfab58c391ecd6cb7ddedddaa46b8afaa05c93d15f1e0db2b871170a55577767"
          ],
          "inputs": {
            "source_operands": [
              "selected_genes"
            ]
          },
          "outputs": {
            "source_results": [
              "replacement_fraction",
              "reveal_positions"
            ]
          },
          "parameters": {
            "source_operation": "sample_reveal_fraction_and_positions",
            "source_port_roles": [
              "eligible gene positions"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "sample_reveal",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "sample_reveal::1",
          "component": "Training diffusion mask sampler",
          "condition": null,
          "evidence_ids": [
            "b2219fd688af43a3c418ed9b0da8421573eff83e4c8cbb0ad2816047f40de556",
            "bfab58c391ecd6cb7ddedddaa46b8afaa05c93d15f1e0db2b871170a55577767"
          ],
          "inputs": {
            "source_operands": [
              "selected_genes"
            ]
          },
          "outputs": {
            "source_results": [
              "replacement_fraction",
              "reveal_positions"
            ]
          },
          "parameters": {
            "source_operation": "sample_reveal_fraction_and_positions",
            "source_port_roles": [
              "eligible gene positions"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "sample_reveal",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "sample_reveal::1",
          "component": "Training diffusion mask sampler",
          "condition": null,
          "evidence_ids": [
            "b2219fd688af43a3c418ed9b0da8421573eff83e4c8cbb0ad2816047f40de556",
            "bfab58c391ecd6cb7ddedddaa46b8afaa05c93d15f1e0db2b871170a55577767"
          ],
          "inputs": {
            "source_operands": [
              "selected_genes"
            ]
          },
          "outputs": {
            "source_results": [
              "replacement_fraction",
              "reveal_positions"
            ]
          },
          "parameters": {
            "source_operation": "sample_reveal_fraction_and_positions",
            "source_port_roles": [
              "eligible gene positions"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "sample_reveal",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "sample_reveal::1",
          "component": "Training diffusion mask sampler",
          "condition": null,
          "evidence_ids": [
            "b2219fd688af43a3c418ed9b0da8421573eff83e4c8cbb0ad2816047f40de556",
            "bfab58c391ecd6cb7ddedddaa46b8afaa05c93d15f1e0db2b871170a55577767"
          ],
          "inputs": {
            "source_operands": [
              "selected_genes"
            ]
          },
          "outputs": {
            "source_results": [
              "replacement_fraction",
              "reveal_positions"
            ]
          },
          "parameters": {
            "source_operation": "sample_reveal_fraction_and_positions",
            "source_port_roles": [
              "eligible gene positions"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "sample_reveal",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "sample_reveal::1",
          "component": "Training diffusion mask sampler",
          "condition": null,
          "evidence_ids": [
            "b2219fd688af43a3c418ed9b0da8421573eff83e4c8cbb0ad2816047f40de556",
            "bfab58c391ecd6cb7ddedddaa46b8afaa05c93d15f1e0db2b871170a55577767"
          ],
          "inputs": {
            "source_operands": [
              "selected_genes"
            ]
          },
          "outputs": {
            "source_results": [
              "replacement_fraction",
              "reveal_positions"
            ]
          },
          "parameters": {
            "source_operation": "sample_reveal_fraction_and_positions",
            "source_port_roles": [
              "eligible gene positions"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "sample_reveal",
          "trajectory_id": "ultra_training_tied_output_receipt"
        }
      ],
      "source_operation": "sample_reveal_fraction_and_positions",
      "status": "unexpanded_source_boundary"
    },
    {
      "occurrences": [
        {
          "call_id": "invert::1",
          "component": "DDIM inversion",
          "condition": null,
          "evidence_ids": [
            "bddff46394c4fcc83a0e6738749288aa303c4afd6bbb23fd8bff32347354e806"
          ],
          "inputs": {
            "source_operands": [
              "ff_image",
              "source_condition"
            ]
          },
          "outputs": {
            "source_results": [
              "inverted_latent"
            ]
          },
          "parameters": {
            "source_operation": "source_conditioned_image_inversion",
            "source_port_roles": [
              "source_image",
              "source_domain_condition"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "invert",
          "trajectory_id": "infer_ff_structure_guided"
        }
      ],
      "source_operation": "source_conditioned_image_inversion",
      "status": "unexpanded_source_boundary"
    },
    {
      "occurrences": [
        {
          "call_id": "source_maps::1",
          "component": "Parallel source-conditioned reconstruction pass",
          "condition": null,
          "evidence_ids": [
            "ad95b5d7dace429088343880882f1351df2e10530fcc73a62b5da712aa911578"
          ],
          "inputs": {
            "source_operands": [
              "source_state_t",
              "source_condition"
            ]
          },
          "outputs": {
            "source_results": [
              "attention_maps"
            ]
          },
          "parameters": {
            "source_operation": "source_self_attention_extraction",
            "source_port_roles": [
              "source_state_at_step_t",
              "source_domain_condition"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "source_maps",
          "trajectory_id": "infer_ff_structure_guided"
        }
      ],
      "source_operation": "source_self_attention_extraction",
      "status": "unexpanded_source_boundary"
    },
    {
      "occurrences": [
        {
          "call_id": "assemble_context::1",
          "component": "Perturbation context collator",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "source_operands": [
              "esm_projected",
              "esm_missing",
              "string_projected",
              "string_missing",
              "genept_projected",
              "genept_missing",
              "depmap_projected",
              "depmap_missing",
              "cp_projected",
              "cp_missing",
              "gene_prior"
            ]
          },
          "outputs": {
            "source_results": [
              "context",
              "external_mask"
            ]
          },
          "parameters": {
            "source_operation": "stack_source_tokens_and_align_availability_mask",
            "source_port_roles": [
              "esm token",
              "esm missingness",
              "string token",
              "string missingness",
              "genept token",
              "genept missingness",
              "depmap token",
              "depmap missingness",
              "cp token",
              "cp missingness",
              "learned gene identity token"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "assemble_context",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "assemble_context::1",
          "component": "Perturbation context collator",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "source_operands": [
              "esm_projected",
              "esm_missing",
              "string_projected",
              "string_missing",
              "genept_projected",
              "genept_missing",
              "depmap_projected",
              "depmap_missing",
              "cp_projected",
              "cp_missing",
              "gene_prior"
            ]
          },
          "outputs": {
            "source_results": [
              "context",
              "external_mask"
            ]
          },
          "parameters": {
            "source_operation": "stack_source_tokens_and_align_availability_mask",
            "source_port_roles": [
              "esm token",
              "esm missingness",
              "string token",
              "string missingness",
              "genept token",
              "genept missingness",
              "depmap token",
              "depmap missingness",
              "cp token",
              "cp missingness",
              "learned gene identity token"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "assemble_context",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "assemble_context::1",
          "component": "Perturbation context collator",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "source_operands": [
              "esm_projected",
              "esm_missing",
              "string_projected",
              "string_missing",
              "genept_projected",
              "genept_missing",
              "depmap_projected",
              "depmap_missing",
              "cp_projected",
              "cp_missing",
              "gene_prior"
            ]
          },
          "outputs": {
            "source_results": [
              "context",
              "external_mask"
            ]
          },
          "parameters": {
            "source_operation": "stack_source_tokens_and_align_availability_mask",
            "source_port_roles": [
              "esm token",
              "esm missingness",
              "string token",
              "string missingness",
              "genept token",
              "genept missingness",
              "depmap token",
              "depmap missingness",
              "cp token",
              "cp missingness",
              "learned gene identity token"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "assemble_context",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "assemble_context::1",
          "component": "Perturbation context collator",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "source_operands": [
              "esm_projected",
              "esm_missing",
              "string_projected",
              "string_missing",
              "genept_projected",
              "genept_missing",
              "depmap_projected",
              "depmap_missing",
              "cp_projected",
              "cp_missing",
              "gene_prior"
            ]
          },
          "outputs": {
            "source_results": [
              "context",
              "external_mask"
            ]
          },
          "parameters": {
            "source_operation": "stack_source_tokens_and_align_availability_mask",
            "source_port_roles": [
              "esm token",
              "esm missingness",
              "string token",
              "string missingness",
              "genept token",
              "genept missingness",
              "depmap token",
              "depmap missingness",
              "cp token",
              "cp missingness",
              "learned gene identity token"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "assemble_context",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "assemble_context::1",
          "component": "Perturbation context collator",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "source_operands": [
              "esm_projected",
              "esm_missing",
              "string_projected",
              "string_missing",
              "genept_projected",
              "genept_missing",
              "depmap_projected",
              "depmap_missing",
              "cp_projected",
              "cp_missing",
              "gene_prior"
            ]
          },
          "outputs": {
            "source_results": [
              "context",
              "external_mask"
            ]
          },
          "parameters": {
            "source_operation": "stack_source_tokens_and_align_availability_mask",
            "source_port_roles": [
              "esm token",
              "esm missingness",
              "string token",
              "string missingness",
              "genept token",
              "genept missingness",
              "depmap token",
              "depmap missingness",
              "cp token",
              "cp missingness",
              "learned gene identity token"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "assemble_context",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "assemble_context::1",
          "component": "Perturbation context collator",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "source_operands": [
              "esm_projected",
              "esm_missing",
              "string_projected",
              "string_missing",
              "genept_projected",
              "genept_missing",
              "depmap_projected",
              "depmap_missing",
              "cp_projected",
              "cp_missing",
              "gene_prior"
            ]
          },
          "outputs": {
            "source_results": [
              "context",
              "external_mask"
            ]
          },
          "parameters": {
            "source_operation": "stack_source_tokens_and_align_availability_mask",
            "source_port_roles": [
              "esm token",
              "esm missingness",
              "string token",
              "string missingness",
              "genept token",
              "genept missingness",
              "depmap token",
              "depmap missingness",
              "cp token",
              "cp missingness",
              "learned gene identity token"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "assemble_context",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "assemble_context::1",
          "component": "Perturbation context collator",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "source_operands": [
              "esm_projected",
              "esm_missing",
              "string_projected",
              "string_missing",
              "genept_projected",
              "genept_missing",
              "depmap_projected",
              "depmap_missing",
              "cp_projected",
              "cp_missing",
              "gene_prior"
            ]
          },
          "outputs": {
            "source_results": [
              "context",
              "external_mask"
            ]
          },
          "parameters": {
            "source_operation": "stack_source_tokens_and_align_availability_mask",
            "source_port_roles": [
              "esm token",
              "esm missingness",
              "string token",
              "string missingness",
              "genept token",
              "genept missingness",
              "depmap token",
              "depmap missingness",
              "cp token",
              "cp missingness",
              "learned gene identity token"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "assemble_context",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "assemble_context::1",
          "component": "Perturbation context collator",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "source_operands": [
              "esm_projected",
              "esm_missing",
              "string_projected",
              "string_missing",
              "genept_projected",
              "genept_missing",
              "depmap_projected",
              "depmap_missing",
              "cp_projected",
              "cp_missing",
              "gene_prior"
            ]
          },
          "outputs": {
            "source_results": [
              "context",
              "external_mask"
            ]
          },
          "parameters": {
            "source_operation": "stack_source_tokens_and_align_availability_mask",
            "source_port_roles": [
              "esm token",
              "esm missingness",
              "string token",
              "string missingness",
              "genept token",
              "genept missingness",
              "depmap token",
              "depmap missingness",
              "cp token",
              "cp missingness",
              "learned gene identity token"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "assemble_context",
          "trajectory_id": "ultra_training_tied_output_receipt"
        },
        {
          "call_id": "assemble_context::1",
          "component": "Perturbation context collator",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "source_operands": [
              "esm_projected",
              "esm_missing",
              "string_projected",
              "string_missing",
              "genept_projected",
              "genept_missing",
              "depmap_projected",
              "depmap_missing",
              "cp_projected",
              "cp_missing",
              "gene_prior"
            ]
          },
          "outputs": {
            "source_results": [
              "context",
              "external_mask"
            ]
          },
          "parameters": {
            "source_operation": "stack_source_tokens_and_align_availability_mask",
            "source_port_roles": [
              "esm token",
              "esm missingness",
              "string token",
              "string missingness",
              "genept token",
              "genept missingness",
              "depmap token",
              "depmap missingness",
              "cp token",
              "cp missingness",
              "learned gene identity token"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "assemble_context",
          "trajectory_id": "xcell_jurkat_active_inference"
        },
        {
          "call_id": "assemble_context::1",
          "component": "Perturbation context collator",
          "condition": null,
          "evidence_ids": [
            "98f3ac49ccd645bf9ee55b2dc68a5a5187816d33b055c3d6698c86c0d2d10157"
          ],
          "inputs": {
            "source_operands": [
              "esm_projected",
              "esm_missing",
              "string_projected",
              "string_missing",
              "genept_projected",
              "genept_missing",
              "depmap_projected",
              "depmap_missing",
              "cp_projected",
              "cp_missing",
              "gene_prior"
            ]
          },
          "outputs": {
            "source_results": [
              "context",
              "external_mask"
            ]
          },
          "parameters": {
            "source_operation": "stack_source_tokens_and_align_availability_mask",
            "source_port_roles": [
              "esm token",
              "esm missingness",
              "string token",
              "string missingness",
              "genept token",
              "genept missingness",
              "depmap token",
              "depmap missingness",
              "cp token",
              "cp missingness",
              "learned gene identity token"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "assemble_context",
          "trajectory_id": "ultra_melanocyte_post_tta_inference"
        }
      ],
      "source_operation": "stack_source_tokens_and_align_availability_mask",
      "status": "unexpanded_source_boundary"
    },
    {
      "occurrences": [
        {
          "call_id": "paired_sets::1",
          "component": "MMD dataloader",
          "condition": null,
          "evidence_ids": [
            "5dc9f1c6e0f6e3d4e16d17efbc1604e601af33f9bd6453b05f179b6383058404"
          ],
          "inputs": {
            "source_operands": [
              "target_normalized",
              "control_normalized"
            ]
          },
          "outputs": {
            "source_results": [
              "target_set",
              "control_set"
            ]
          },
          "parameters": {
            "source_operation": "stratified_target_set_and_matched_control_sampling",
            "source_port_roles": [
              "perturbed cells grouped by context, perturbation, batch",
              "matched NTC pool"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "paired_sets",
          "trajectory_id": "xcell_pisces_pretraining_cross_attention"
        },
        {
          "call_id": "paired_sets::1",
          "component": "MMD dataloader",
          "condition": null,
          "evidence_ids": [
            "5dc9f1c6e0f6e3d4e16d17efbc1604e601af33f9bd6453b05f179b6383058404"
          ],
          "inputs": {
            "source_operands": [
              "target_normalized",
              "control_normalized"
            ]
          },
          "outputs": {
            "source_results": [
              "target_set",
              "control_set"
            ]
          },
          "parameters": {
            "source_operation": "stratified_target_set_and_matched_control_sampling",
            "source_port_roles": [
              "perturbed cells grouped by context, perturbation, batch",
              "matched NTC pool"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "paired_sets",
          "trajectory_id": "ultra_high_effect_training_cross_attention"
        },
        {
          "call_id": "paired_sets::1",
          "component": "MMD dataloader",
          "condition": null,
          "evidence_ids": [
            "5dc9f1c6e0f6e3d4e16d17efbc1604e601af33f9bd6453b05f179b6383058404"
          ],
          "inputs": {
            "source_operands": [
              "target_normalized",
              "control_normalized"
            ]
          },
          "outputs": {
            "source_results": [
              "target_set",
              "control_set"
            ]
          },
          "parameters": {
            "source_operation": "stratified_target_set_and_matched_control_sampling",
            "source_port_roles": [
              "perturbed cells grouped by context, perturbation, batch",
              "matched NTC pool"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "paired_sets",
          "trajectory_id": "ultra_full_corpus_fine_tuning_cross_attention"
        },
        {
          "call_id": "paired_sets::1",
          "component": "MMD dataloader",
          "condition": null,
          "evidence_ids": [
            "5dc9f1c6e0f6e3d4e16d17efbc1604e601af33f9bd6453b05f179b6383058404"
          ],
          "inputs": {
            "source_operands": [
              "target_normalized",
              "control_normalized"
            ]
          },
          "outputs": {
            "source_results": [
              "target_set",
              "control_set"
            ]
          },
          "parameters": {
            "source_operation": "stratified_target_set_and_matched_control_sampling",
            "source_port_roles": [
              "perturbed cells grouped by context, perturbation, batch",
              "matched NTC pool"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "paired_sets",
          "trajectory_id": "xcell_replogle_fine_tuning_cross_attention"
        },
        {
          "call_id": "paired_sets::1",
          "component": "MMD dataloader",
          "condition": null,
          "evidence_ids": [
            "5dc9f1c6e0f6e3d4e16d17efbc1604e601af33f9bd6453b05f179b6383058404"
          ],
          "inputs": {
            "source_operands": [
              "target_normalized",
              "control_normalized"
            ]
          },
          "outputs": {
            "source_results": [
              "target_set",
              "control_set"
            ]
          },
          "parameters": {
            "source_operation": "stratified_target_set_and_matched_control_sampling",
            "source_port_roles": [
              "perturbed cells grouped by context, perturbation, batch",
              "matched NTC pool"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "paired_sets",
          "trajectory_id": "xcell_parse_fine_tuning_cross_attention"
        },
        {
          "call_id": "paired_sets::1",
          "component": "MMD dataloader",
          "condition": null,
          "evidence_ids": [
            "5dc9f1c6e0f6e3d4e16d17efbc1604e601af33f9bd6453b05f179b6383058404"
          ],
          "inputs": {
            "source_operands": [
              "target_normalized",
              "control_normalized"
            ]
          },
          "outputs": {
            "source_results": [
              "target_set",
              "control_set"
            ]
          },
          "parameters": {
            "source_operation": "stratified_target_set_and_matched_control_sampling",
            "source_port_roles": [
              "perturbed cells grouped by context, perturbation, batch",
              "matched NTC pool"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "paired_sets",
          "trajectory_id": "xcell_resting_jurkat_fine_tuning_cross_attention"
        },
        {
          "call_id": "paired_sets::1",
          "component": "MMD dataloader",
          "condition": null,
          "evidence_ids": [
            "5dc9f1c6e0f6e3d4e16d17efbc1604e601af33f9bd6453b05f179b6383058404"
          ],
          "inputs": {
            "source_operands": [
              "target_normalized",
              "control_normalized"
            ]
          },
          "outputs": {
            "source_results": [
              "target_set",
              "control_set"
            ]
          },
          "parameters": {
            "source_operation": "stratified_target_set_and_matched_control_sampling",
            "source_port_roles": [
              "perturbed cells grouped by context, perturbation, batch",
              "matched NTC pool"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "paired_sets",
          "trajectory_id": "xcell_pretraining_scalar_output_receipt"
        },
        {
          "call_id": "paired_sets::1",
          "component": "MMD dataloader",
          "condition": null,
          "evidence_ids": [
            "5dc9f1c6e0f6e3d4e16d17efbc1604e601af33f9bd6453b05f179b6383058404"
          ],
          "inputs": {
            "source_operands": [
              "target_normalized",
              "control_normalized"
            ]
          },
          "outputs": {
            "source_results": [
              "target_set",
              "control_set"
            ]
          },
          "parameters": {
            "source_operation": "stratified_target_set_and_matched_control_sampling",
            "source_port_roles": [
              "perturbed cells grouped by context, perturbation, batch",
              "matched NTC pool"
            ]
          },
          "record_id": "full_2026-07-06__rec_003517",
          "source_step_id": "paired_sets",
          "trajectory_id": "ultra_training_tied_output_receipt"
        }
      ],
      "source_operation": "stratified_target_set_and_matched_control_sampling",
      "status": "unexpanded_source_boundary"
    },
    {
      "occurrences": [
        {
          "call_id": "encode_text::1",
          "component": "MUSK text encoder",
          "condition": null,
          "evidence_ids": [
            "2f344b287bc862c0717d113dbca55b0be574f35a70990afe284ee672a2b0c913"
          ],
          "inputs": {
            "source_operands": [
              "caption"
            ]
          },
          "outputs": {
            "source_results": [
              "text_condition"
            ]
          },
          "parameters": {
            "source_operation": "text_encoding",
            "source_port_roles": [
              "condition_text"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "encode_text",
          "trajectory_id": "pretrain_multimodal_dca"
        },
        {
          "call_id": "encode::1",
          "component": "MUSK text encoder",
          "condition": null,
          "evidence_ids": [
            "2f344b287bc862c0717d113dbca55b0be574f35a70990afe284ee672a2b0c913"
          ],
          "inputs": {
            "source_operands": [
              "caption"
            ]
          },
          "outputs": {
            "source_results": [
              "condition"
            ]
          },
          "parameters": {
            "source_operation": "text_encoding",
            "source_port_roles": [
              "sample_condition"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "encode",
          "trajectory_id": "infer_text"
        },
        {
          "call_id": "encode::1",
          "component": "MUSK text encoder",
          "condition": null,
          "evidence_ids": [
            "2f344b287bc862c0717d113dbca55b0be574f35a70990afe284ee672a2b0c913"
          ],
          "inputs": {
            "source_operands": [
              "caption"
            ]
          },
          "outputs": {
            "source_results": [
              "condition"
            ]
          },
          "parameters": {
            "source_operation": "text_encoding",
            "source_port_roles": [
              "sample_condition"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "encode",
          "trajectory_id": "infer_clip_augmentation"
        },
        {
          "call_id": "encode::1",
          "component": "MUSK text encoder",
          "condition": null,
          "evidence_ids": [
            "2f344b287bc862c0717d113dbca55b0be574f35a70990afe284ee672a2b0c913",
            "e489d46a31fad7a335b84199271b4ececfcdc360fe2d7f5315200177b837a1fa"
          ],
          "inputs": {
            "source_operands": [
              "domain_prompt"
            ]
          },
          "outputs": {
            "source_results": [
              "domain_embedding"
            ]
          },
          "parameters": {
            "source_operation": "text_encoding",
            "source_port_roles": [
              "domain_identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "encode",
          "trajectory_id": "train_ff_domain_ff"
        },
        {
          "call_id": "encode::1",
          "component": "MUSK text encoder",
          "condition": null,
          "evidence_ids": [
            "2f344b287bc862c0717d113dbca55b0be574f35a70990afe284ee672a2b0c913",
            "e489d46a31fad7a335b84199271b4ececfcdc360fe2d7f5315200177b837a1fa"
          ],
          "inputs": {
            "source_operands": [
              "domain_prompt"
            ]
          },
          "outputs": {
            "source_results": [
              "domain_embedding"
            ]
          },
          "parameters": {
            "source_operation": "text_encoding",
            "source_port_roles": [
              "domain_identity"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "encode",
          "trajectory_id": "train_ff_domain_ffpe"
        },
        {
          "call_id": "source_encode::1",
          "component": "MUSK text encoder",
          "condition": null,
          "evidence_ids": [
            "2f344b287bc862c0717d113dbca55b0be574f35a70990afe284ee672a2b0c913",
            "bddff46394c4fcc83a0e6738749288aa303c4afd6bbb23fd8bff32347354e806"
          ],
          "inputs": {
            "source_operands": [
              "source_prompt"
            ]
          },
          "outputs": {
            "source_results": [
              "source_condition"
            ]
          },
          "parameters": {
            "source_operation": "text_encoding",
            "source_port_roles": [
              "source_domain"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "source_encode",
          "trajectory_id": "infer_ff_structure_guided"
        },
        {
          "call_id": "target_encode::1",
          "component": "MUSK text encoder",
          "condition": null,
          "evidence_ids": [
            "2f344b287bc862c0717d113dbca55b0be574f35a70990afe284ee672a2b0c913",
            "bddff46394c4fcc83a0e6738749288aa303c4afd6bbb23fd8bff32347354e806"
          ],
          "inputs": {
            "source_operands": [
              "target_prompt"
            ]
          },
          "outputs": {
            "source_results": [
              "target_condition"
            ]
          },
          "parameters": {
            "source_operation": "text_encoding",
            "source_port_roles": [
              "target_domain"
            ]
          },
          "record_id": "full_2026-07-06__rec_003852",
          "source_step_id": "target_encode",
          "trajectory_id": "infer_ff_structure_guided"
        }
      ],
      "source_operation": "text_encoding",
      "status": "unexpanded_source_boundary"
    }
  ],
  "records": {
    "full_2026-07-06__rec_000771": {
      "input_packet_sha256": "0f547b950d71f3f513d88759aa20034df2e60de2789f7dd77eec377c18ea2d6b",
      "paper": "ChatNT",
      "source_sha256": "f8550c2b4f96c61cc5ed326152f2f2dd958e0f97138a9ea40d0c9522ce22ce2c"
    },
    "full_2026-07-06__rec_001319": {
      "input_packet_sha256": "adb4f6ec56e4a24c4cbf5081f5cdf2160ad7430caebab21eafa063db75b18e4b",
      "paper": "InstructCell",
      "source_sha256": "21bf270ef3e5d1ef4d2472f27332787c216747089e18c835ce2bd927325fe544"
    },
    "full_2026-07-06__rec_003517": {
      "input_packet_sha256": "3745a8e1f9502c81e46906896ff5194ff20eaa7192337697333cb4e0467cb7ab",
      "paper": "X-Cell",
      "source_sha256": "47c64fd9297c1e1fcba44ef6cf0d6ce1eafd566eb3e5f0620c9432ffb1ccf2f9"
    },
    "full_2026-07-06__rec_003852": {
      "input_packet_sha256": "157a3f415b424ecda6d4f7dbb1d4378349a260ad24b00897d1d6601ecdf91ecc",
      "paper": "MUPAD",
      "source_sha256": "38d2c0372a0bd53fb7b613423707c7cf8f7293a9f7b7e884471da2643802f006"
    }
  },
  "release": "1.0.1",
  "report": {
    "canonical_migration": false,
    "cross_paper_reused_types": 8,
    "evidence_snapshot_sha256": "e114686edfd81474b37b7d9c81e25054f4e5869989fdfad545e9970a1ca6afdf",
    "library_sha256": "8baa240c279a84a1229eefd5f34e0527291086570378e20500f37bcb9f1f2caa",
    "lossless_source_restoration": true,
    "operation_types": 22,
    "pilot_input_sha256": "f5513907443f3cb95898c266830ae370caba2b97cacd740c31b137695d32e23a",
    "primitive_calls": 509,
    "record_count": 4,
    "release": "1.0.1",
    "source_sizes_modified": false,
    "status": "operation_constructor_pilot",
    "trajectory_count": 47,
    "unexpanded_source_labels": 32,
    "unexpanded_source_steps": 146
  }
}
