{
  "schema": "fak-study-adjacency/1",
  "id": "vllm-related-system-adjacency-2026-08-26-v1",
  "title": "vLLM related-system adjacency at 2026-08-26T22:35:00Z",
  "scope": {
    "bounded_meaning": "The six issue-mandated peer runtimes plus active, non-archived vllm-project satellites whose shipped code directly changes a named FAK decision about scheduling, KV/cache ownership, disaggregation, routing, speculative decoding, current Apple-Metal execution, or GGUF interoperability. It is not an ecosystem-wide serving survey.",
    "inclusion_criteria": [
      "Admit a vllm-project satellite only when it is active and non-archived, contains runtime or control-plane implementation rather than only prose/assets, and maps to a named vLLM mechanism or explicit frontier-changing FAK contrast.",
      "Always include the six issue-mandated peers: sgl-project/sglang, NVIDIA/TensorRT-LLM, ai-dynamo/dynamo, ggml-org/llama.cpp, flashinfer-ai/flashinfer, and llm-d/llm-d.",
      "For hardware and artifact satellites, retain only the current FAK envelope: Apple Metal execution and GGUF interoperability.",
      "For overlapping control planes, retain a repository only when its distinct ownership seam changes a FAK choice: deployment assembly, prefix-aware endpoint routing, semantic model routing, autoscaling/cache control, or attention-FFN disaggregation."
    ],
    "exclusion_criteria": [
      "Exclude application-layer or modality-expansion systems such as agentic-api and vllm-omni until the FAK text-runtime decision frontier includes those scopes.",
      "Exclude archived repositories and fork mirrors whose canonical implementation lives elsewhere, including vllm-bench, vllm-nccl, DeepGEMM, FlashMLA, flash-attention, MSA, and lm-evaluation-harness.",
      "Exclude documentation, community, website, asset, recipe, RFC, daily-summary, CI, dashboard, and performance-reporting repositories because they do not implement a runtime mechanism in this adjacency spine.",
      "Exclude hardware backends outside the current FAK operating envelope, including Ascend, Gaudi, Neuron, OpenVINO, TPU, and XPU satellites; reopen when a matched FAK hardware campaign names that backend.",
      "Exclude narrow model plugins and skills unless a vLLM mechanism classification demonstrates a decision-changing runtime contrast not already represented by the admitted peers.",
      "Exclude offline model preparation, evaluation, training, or storage-format libraries such as llm-compressor, compressed-tensors, guidellm, perf-eval, and vime; they may enter a separate artifact or evaluation adjacency."
    ]
  },
  "anchor": {
    "repository": {
      "owner": "vllm-project",
      "repo": "vllm"
    },
    "pin": {
      "revision": "f18d0ba90d972a852a351c98be3f42b31372cfe4",
      "cutoff": "2026-08-26T22:35:00Z",
      "observed_at": "2026-08-27T02:19:32Z"
    },
    "normalized_records": 53848,
    "sha256": "2a66d4876aee3811eb200c0884c6558a5f3ac86c90b6c7f8b92f45b85fe671b2",
    "notes": "Terminally complete anchor supplied by issue #9275 and confirmed in the owner comment observed at 2026-08-27T02:19:32Z; the large corpus is intentionally not duplicated in this worktree."
  },
  "declared_repositories": [
    {
      "owner": "ai-dynamo",
      "repo": "dynamo"
    },
    {
      "owner": "flashinfer-ai",
      "repo": "flashinfer"
    },
    {
      "owner": "ggml-org",
      "repo": "llama.cpp"
    },
    {
      "owner": "llm-d",
      "repo": "llm-d"
    },
    {
      "owner": "NVIDIA",
      "repo": "TensorRT-LLM"
    },
    {
      "owner": "sgl-project",
      "repo": "sglang"
    },
    {
      "owner": "vllm-project",
      "repo": "afd-plugin"
    },
    {
      "owner": "vllm-project",
      "repo": "aibrix"
    },
    {
      "owner": "vllm-project",
      "repo": "production-stack"
    },
    {
      "owner": "vllm-project",
      "repo": "router"
    },
    {
      "owner": "vllm-project",
      "repo": "semantic-router"
    },
    {
      "owner": "vllm-project",
      "repo": "speculators"
    },
    {
      "owner": "vllm-project",
      "repo": "vllm-gguf-plugin"
    },
    {
      "owner": "vllm-project",
      "repo": "vllm-metal"
    }
  ],
  "members": [
    {
      "name": "NVIDIA Dynamo",
      "repository": {
        "owner": "ai-dynamo",
        "repo": "dynamo"
      },
      "pin": {
        "revision": "6fa28c2f84e0ceab4c4cf60f147f63dff77b7f17",
        "cutoff": "2026-08-26T22:35:00Z",
        "observed_at": "2026-08-27T02:29:21Z"
      },
      "processed": true,
      "inclusion_rationale": "Issue-mandated distributed inference runtime and control plane with vLLM backend integration, KV-aware routing, disaggregated serving, and planner mechanisms.",
      "decision_relation": "Separates cluster routing/planning/cache-fabric ownership from vLLM's per-engine scheduling and tests which of those seams FAK should own natively versus integrate.",
      "freshness_notes": "Newest default-branch commit at or before the shared cutoff is 6fa28c2f84e0ceab4c4cf60f147f63dff77b7f17 (2026-08-26T22:33:47Z); GitHub metadata observed 2026-08-27T02:29:21Z. The cutoff pin is 38 minutes newer than the checked exhaustive inventory.",
      "partial_notes": "Existing exhaustive tree/corpus evidence is pinned to f494601ef16b2a17b35ca340c656287b2787971a at 2026-08-26T22:01:04Z. The delta to 6fa28c2 is not classified here, and no fak-studyforge terminal receipt exists.",
      "source_class_receipts": [
        {
          "class": "forge_history",
          "status": "complete",
          "terminal_receipt": "study-forge:sha256:9e49e551c3966f612159a773265363d753396d795ece243fdb59165a88676be2; file-sha256:6547f197b42a31ba78579d32f66de440f128d23e8a9ce417bdd72b0392577bd9; records:13979; bytes:45421438",
          "notes": "Complete validated study-forge corpus at the shared cutoff and pinned revision 6fa28c2f84e0ceab4c4cf60f147f63dff77b7f17; full corpus retained in allocated scratch."
        },
        {
          "class": "repository_metadata",
          "status": "complete",
          "terminal_receipt": "github-graphql:defaultBranchRef.history(first:1,until=2026-08-26T22:35:00Z); response-sha256:0579a280eb576396695dac7b0e0b4a43c1bf953e7cbedec1e3513bf7e786f53b",
          "notes": "Authoritative GitHub GraphQL returned the canonical identity, default branch, cutoff revision, and commit timestamp. The normalized one-line response digest is retained here; raw responses remain allocated scratch."
        },
        {
          "class": "runtime_tree",
          "status": "partial",
          "notes": "Reused docs/research/inventory/ai-dynamo-dynamo.json at stale pin f494601e; 5,401 files were exhaustive there, not at the cutoff pin."
        }
      ],
      "candidates": [
        {
          "id": "dynamo-kv-router-planner",
          "title": "KV-aware router and disaggregated planner ownership",
          "rationale": "Determines whether FAK should import routing signals and planner receipts while retaining native policy, cache, scheduling, and kernel ownership.",
          "repository_links": [
            {
              "owner": "ai-dynamo",
              "repo": "dynamo"
            }
          ],
          "vllm_mechanism_link": "vLLM worker/backend integration, block-level KV events, prefill/decode disaggregation, and request scheduling"
        }
      ]
    },
    {
      "name": "FlashInfer",
      "repository": {
        "owner": "flashinfer-ai",
        "repo": "flashinfer"
      },
      "pin": {
        "revision": "e4b7fa4b7c3ba5e17286d9c59f2bcf2ca07e0a6d",
        "cutoff": "2026-08-26T22:35:00Z",
        "observed_at": "2026-08-27T02:29:21Z"
      },
      "processed": true,
      "inclusion_rationale": "Issue-mandated attention and MoE kernel/runtime library used by vLLM and peer engines.",
      "decision_relation": "Changes the kernel-plan/workspace/JIT artifact frontier without changing FAK's rule that native execution, memory, scheduling, cache, and evidence remain FAK-owned.",
      "freshness_notes": "Newest default-branch commit at or before the shared cutoff is e4b7fa4b7c3ba5e17286d9c59f2bcf2ca07e0a6d (2026-08-26T21:25:43Z); GitHub metadata observed 2026-08-27T02:29:21Z. The cutoff pin is newer than the checked exhaustive inventory by four days.",
      "partial_notes": "The reusable inventory is pinned to fb28d7242b3506a2348265962041acc1fb56cca4 (2026-08-22T11:34:52Z). Forge evidence is legacy command read-back rather than a terminal study-forge receipt.",
      "source_class_receipts": [
        {
          "class": "forge_history",
          "status": "complete",
          "terminal_receipt": "study-forge:sha256:90ad57df79a5c82656c41aaeb644c807117f6c0cff1227053f3806ab6ea1bb4e; file-sha256:31b45236b77a1e279f909762075437b80629acfd1446e01f2d44ba8bbcd127a1; records:5203; bytes:17238820",
          "notes": "Complete validated study-forge corpus at the shared cutoff and pinned revision e4b7fa4b7c3ba5e17286d9c59f2bcf2ca07e0a6d; full corpus retained in allocated scratch."
        },
        {
          "class": "repository_metadata",
          "status": "complete",
          "terminal_receipt": "github-graphql:defaultBranchRef.history(first:1,until=2026-08-26T22:35:00Z); response-sha256:901583878dc508e37e7bc6877575e3bcea1e732ba38baab4961e6310491d4a47",
          "notes": "Authoritative GitHub GraphQL returned the canonical identity, default branch, cutoff revision, and commit timestamp. The normalized one-line response digest is retained here; raw responses remain allocated scratch."
        },
        {
          "class": "runtime_tree",
          "status": "partial",
          "notes": "Reused docs/research/inventory/flashinfer-ai-flashinfer.json at stale pin fb28d724; exact cutoff tree remains uncaptured."
        }
      ],
      "candidates": [
        {
          "id": "flashinfer-plan-run-kernel-contract",
          "title": "Reusable kernel plan/run workspace contracts",
          "rationale": "Changes how FAK can borrow optimized kernel contracts without yielding fak-native engine ownership.",
          "repository_links": [
            {
              "owner": "flashinfer-ai",
              "repo": "flashinfer"
            }
          ],
          "vllm_mechanism_link": "vLLM attention and MoE backend integration, kernel selection, and reusable execution plans"
        }
      ]
    },
    {
      "name": "llama.cpp",
      "repository": {
        "owner": "ggml-org",
        "repo": "llama.cpp"
      },
      "pin": {
        "revision": "925e1179947ea0c0ebfb0032df18af3a729822be",
        "cutoff": "2026-08-26T22:35:00Z",
        "observed_at": "2026-08-27T02:29:21Z"
      },
      "processed": true,
      "inclusion_rationale": "Issue-mandated portable native runtime and the strongest single-binary/GGUF contrast to vLLM's accelerator-first serving architecture.",
      "decision_relation": "Defines the portability, CPU/Metal, GGUF, quantization, and low-resource baseline that prevents a vLLM-only conclusion from overstating the serving frontier.",
      "freshness_notes": "Newest default-branch commit at or before the shared cutoff is 925e1179947ea0c0ebfb0032df18af3a729822be (2026-08-26T21:34:28Z); GitHub metadata observed 2026-08-27T02:29:21Z. The same revision is the newest commit at the shared cutoff; the committed inventory was generated at 22:20Z.",
      "partial_notes": "The machine index has a complete recursive tree and a complete census of open issues/PRs, but closed forge history, discussions, releases, labels, and milestones are not terminally captured through 22:35Z.",
      "source_class_receipts": [
        {
          "class": "forge_history",
          "status": "complete",
          "terminal_receipt": "study-forge:sha256:f2b36ab2c7d52df411d588df4c18ba136702de4edca778728247cc99c82afca3; file-sha256:f865ff875d3f4af44d954f5bae5ea4e746b5649d48b17202e6ff43fc7f8fef8e; records:34633; bytes:114986090",
          "notes": "Complete validated study-forge corpus at the shared cutoff and pinned revision 925e1179947ea0c0ebfb0032df18af3a729822be; full corpus retained in allocated scratch."
        },
        {
          "class": "repository_metadata",
          "status": "complete",
          "terminal_receipt": "github-graphql:defaultBranchRef.history(first:1,until=2026-08-26T22:35:00Z); response-sha256:bb7561b67dae0ce068cb1afd368457a03ca5ea9ab7b34fda9e801ae7fcc28638",
          "notes": "Authoritative GitHub GraphQL returned the canonical identity, default branch, cutoff revision, and commit timestamp. The normalized one-line response digest is retained here; raw responses remain allocated scratch."
        },
        {
          "class": "runtime_tree",
          "status": "complete",
          "terminal_receipt": "docs/research/inventory/ggml-org-llama-cpp.json#upstream.tree.truncated=false",
          "notes": "Exact cutoff revision recursive tree: 3,871 entries, 3,498 blobs, 373 trees, not truncated."
        }
      ],
      "candidates": [
        {
          "id": "llamacpp-portable-native-frontier",
          "title": "Portable GGUF single-binary runtime frontier",
          "rationale": "Keeps FAK's native portability and artifact decisions honest while llama.cpp remains benchmark/reference only unless explicitly selected.",
          "repository_links": [
            {
              "owner": "ggml-org",
              "repo": "llama.cpp"
            }
          ],
          "frontier_changing_contrast": "llama.cpp's CPU/GPU/Metal GGUF runtime and single-binary deployment change the portability and low-resource frontier relative to vLLM's accelerator-first distributed serving."
        }
      ]
    },
    {
      "name": "llm-d",
      "repository": {
        "owner": "llm-d",
        "repo": "llm-d"
      },
      "pin": {
        "revision": "bc20f73bd344b5a0faad5afca93831088aeee957",
        "cutoff": "2026-08-26T22:35:00Z",
        "observed_at": "2026-08-27T02:29:21Z"
      },
      "processed": true,
      "inclusion_rationale": "Issue-mandated Kubernetes inference serving assembly for vLLM with endpoint picking, KV events, flow control, autoscaling, and prefill/decode disaggregation.",
      "decision_relation": "Defines the integration boundary between FAK's policy/model choice and an external cluster control plane's endpoint and topology decisions.",
      "freshness_notes": "Newest default-branch commit at or before the shared cutoff is bc20f73bd344b5a0faad5afca93831088aeee957 (2026-08-26T16:44:48Z); GitHub metadata observed 2026-08-27T02:29:21Z. The cutoff pin is newer than the checked exhaustive inventory by about 15 hours.",
      "partial_notes": "The reusable inventory is pinned to 3243fcf1191348b55c7811267a98117f8b7a6910. No terminal study-forge receipt exists at either pin.",
      "source_class_receipts": [
        {
          "class": "forge_history",
          "status": "complete",
          "terminal_receipt": "study-forge:sha256:89d9fbfa7ed92f44f5d6a303ddf99cc004f15c0ce66b3115cfa94f3fca72064e; file-sha256:e1c78998333edc530101554d13ec90c8c40758a2dbe6b7971490ffd2fec83bab; records:2470; bytes:5259930",
          "notes": "Complete validated study-forge corpus at the shared cutoff and pinned revision bc20f73bd344b5a0faad5afca93831088aeee957; full corpus retained in allocated scratch."
        },
        {
          "class": "repository_metadata",
          "status": "complete",
          "terminal_receipt": "github-graphql:defaultBranchRef.history(first:1,until=2026-08-26T22:35:00Z); response-sha256:57c031e130d149f50ee163d60355ecbbd06ba15de421f960f58d76a11f5fa184",
          "notes": "Authoritative GitHub GraphQL returned the canonical identity, default branch, cutoff revision, and commit timestamp. The normalized one-line response digest is retained here; raw responses remain allocated scratch."
        },
        {
          "class": "runtime_tree",
          "status": "partial",
          "notes": "Reused docs/research/inventory/llm-d-llm-d.json at stale pin 3243fcf1; the exact cutoff tree is not captured."
        }
      ],
      "candidates": [
        {
          "id": "llmd-kv-event-endpoint-routing",
          "title": "KV-event-aware endpoint selection and ownership boundary",
          "rationale": "Changes whether FAK integrates external locality signals while remaining authoritative for policy and model selection.",
          "repository_links": [
            {
              "owner": "llm-d",
              "repo": "llm-d"
            }
          ],
          "vllm_mechanism_link": "vLLM serving pods, KV events, prefix locality, flow control, and prefill/decode disaggregation"
        }
      ]
    },
    {
      "name": "TensorRT-LLM",
      "repository": {
        "owner": "NVIDIA",
        "repo": "TensorRT-LLM"
      },
      "pin": {
        "revision": "c2f5f31912bf2257bfc42f79fa6621766cd4f9e4",
        "cutoff": "2026-08-26T22:35:00Z",
        "observed_at": "2026-08-27T02:29:21Z"
      },
      "processed": true,
      "inclusion_rationale": "Issue-mandated NVIDIA production runtime with compiled TensorRT execution, inflight batching, quantized kernels, and multi-GPU serving.",
      "decision_relation": "Defines the NVIDIA throughput/latency and compiled-engine frontier used to judge whether FAK should borrow a kernel/contract, bind an optional backend, or stay minimal.",
      "freshness_notes": "Newest default-branch commit at or before the shared cutoff is c2f5f31912bf2257bfc42f79fa6621766cd4f9e4 (2026-08-26T22:03:30Z); GitHub metadata observed 2026-08-27T02:29:21Z. The cutoff pin is more than a month newer than the existing deep study.",
      "partial_notes": "Only the July study at f4c5c935aa891b0826f73936c4831236cb6ff836 is reusable. There is no current machine inventory or terminal forge receipt.",
      "source_class_receipts": [
        {
          "class": "forge_history",
          "status": "complete",
          "terminal_receipt": "study-forge:sha256:b80845eff4b726afc73a8aa3c7c64274bbef235a8f1964b6156a17bbf4634ef2; file-sha256:2fd1d9a546567b402df6db0a1c6a953ac1c89f2008bca1fb7d56357c399daaf1; records:18376; bytes:87650376",
          "notes": "Complete validated study-forge corpus at the shared cutoff and pinned revision c2f5f31912bf2257bfc42f79fa6621766cd4f9e4; full corpus retained in allocated scratch."
        },
        {
          "class": "repository_metadata",
          "status": "complete",
          "terminal_receipt": "github-graphql:defaultBranchRef.history(first:1,until=2026-08-26T22:35:00Z); response-sha256:d45dd09104d1edd45b1737d57248bb9793e11f70f206a6eba4c3fa5b743985b6",
          "notes": "Authoritative GitHub GraphQL returned the canonical identity, default branch, cutoff revision, and commit timestamp. The normalized one-line response digest is retained here; raw responses remain allocated scratch."
        },
        {
          "class": "runtime_tree",
          "status": "missing",
          "notes": "No machine-readable runtime tree inventory exists for c2f5f319; docs/notes/CONCEPT-STUDY-TENSORRT-LLM-2026-07-18.md is stale source guidance only."
        }
      ],
      "candidates": [
        {
          "id": "tensorrt-llm-compiled-runtime-frontier",
          "title": "Compiled NVIDIA runtime and inflight-batching frontier",
          "rationale": "Changes FAK's borrow/bind/stay-minimal decisions for NVIDIA kernels, expert dispatch, quantization, and matched serving benchmarks.",
          "repository_links": [
            {
              "owner": "NVIDIA",
              "repo": "TensorRT-LLM"
            }
          ],
          "vllm_mechanism_link": "vLLM continuous batching, paged KV management, quantized execution, and disaggregated serving"
        }
      ]
    },
    {
      "name": "SGLang",
      "repository": {
        "owner": "sgl-project",
        "repo": "sglang"
      },
      "pin": {
        "revision": "7f27bf470824f452a34e866d22ab5e332a23e26f",
        "cutoff": "2026-08-26T22:35:00Z",
        "observed_at": "2026-08-27T02:29:21Z"
      },
      "processed": true,
      "inclusion_rationale": "Issue-mandated peer serving runtime and direct reference for RadixAttention, HiCache, chunked prefill, continuous batching, structured generation, and speculative decoding.",
      "decision_relation": "Tests vLLM-derived scheduler and KV conclusions against a different prefix-tree/cache hierarchy and exposes an external ride-mode comparator already integrated by FAK.",
      "freshness_notes": "Newest default-branch commit at or before the shared cutoff is 7f27bf470824f452a34e866d22ab5e332a23e26f (2026-08-26T22:24:26Z); GitHub metadata observed 2026-08-27T02:29:21Z. The cutoff pin is current; existing deep notes and mini-sglang inventory are older and narrower.",
      "partial_notes": "No full sgl-project/sglang machine inventory or terminal forge receipt exists. Prior studies cover selected caching and serving slices only; #9289 owns exhaustive completion.",
      "source_class_receipts": [
        {
          "class": "forge_history",
          "status": "complete",
          "terminal_receipt": "study-forge:sha256:f1636085e8c9bfbb3a0c20fb4f684d329f76b8b17f54627001b22970ecaf0c7a; file-sha256:2103ccac2d04cc23e6a0bd405bd5ac5f6e1d6005efa81c83dc28490fe7d00056; records:36658; bytes:164718108",
          "notes": "Complete validated study-forge corpus at the shared cutoff and pinned revision 7f27bf470824f452a34e866d22ab5e332a23e26f; full corpus retained in allocated scratch."
        },
        {
          "class": "repository_metadata",
          "status": "complete",
          "terminal_receipt": "github-graphql:defaultBranchRef.history(first:1,until=2026-08-26T22:35:00Z); response-sha256:114d9a9af75479da6cb8ca5bb4df91d4bc0ff3559d1225ac7dbf0d5309e7813f",
          "notes": "Authoritative GitHub GraphQL returned the canonical identity, default branch, cutoff revision, and commit timestamp. The normalized one-line response digest is retained here; raw responses remain allocated scratch."
        },
        {
          "class": "runtime_tree",
          "status": "missing",
          "notes": "No standard inventory for full sgl-project/sglang at 7f27bf47; existing mini-sglang map is a different repository and cannot satisfy this member."
        }
      ],
      "candidates": [
        {
          "id": "sglang-radix-hicache-contrast",
          "title": "RadixAttention and hierarchical cache contrast",
          "rationale": "Changes FAK decisions on prefix identity, hierarchical residency, eviction, and whether to ride or own each cache seam.",
          "repository_links": [
            {
              "owner": "sgl-project",
              "repo": "sglang"
            }
          ],
          "vllm_mechanism_link": "vLLM PagedAttention, prefix caching, continuous batching, chunked prefill, and scheduler admission"
        }
      ]
    },
    {
      "name": "vLLM attention-FFN disaggregation plugin",
      "repository": {
        "owner": "vllm-project",
        "repo": "afd-plugin"
      },
      "pin": {
        "revision": "b00e884c4e8eb542544d3eba623e74ca01252c92",
        "cutoff": "2026-08-26T22:35:00Z",
        "observed_at": "2026-08-27T02:29:21Z"
      },
      "processed": true,
      "inclusion_rationale": "Active vllm-project runtime plugin whose attention/FFN disaggregation is a distinct topology not represented by the required peers' repository identities.",
      "decision_relation": "Changes the granularity at which FAK compares prefill/decode and expert/attention disaggregation while preserving native ownership boundaries.",
      "freshness_notes": "Newest default-branch commit at or before the shared cutoff is b00e884c4e8eb542544d3eba623e74ca01252c92 (2026-08-26T09:51:05Z); GitHub metadata observed 2026-08-27T02:29:21Z. The cutoff pin is current metadata-only evidence.",
      "partial_notes": "No local tree inventory or forge corpus was captured; only terminal repository metadata and the mechanism-level candidate are claimed.",
      "source_class_receipts": [
        {
          "class": "forge_history",
          "status": "complete",
          "terminal_receipt": "study-forge:sha256:76e106727641b361c9aa00b129fc7f8b7f0a02c652001b24a946a7e8697dc989; file-sha256:e1f191632e40c167aa9012f1142c8c12561fe7eaddbb249a5e8100eb86a05d63; records:290; bytes:1153460",
          "notes": "Complete validated study-forge corpus at the shared cutoff and pinned revision b00e884c4e8eb542544d3eba623e74ca01252c92; full corpus retained in allocated scratch."
        },
        {
          "class": "repository_metadata",
          "status": "complete",
          "terminal_receipt": "github-graphql:defaultBranchRef.history(first:1,until=2026-08-26T22:35:00Z); response-sha256:5478d62c626fab3bfcb978ff3d46827a85270113f221a4f7ee3e3b06a57ea76c",
          "notes": "Authoritative GitHub GraphQL returned the canonical identity, default branch, cutoff revision, and commit timestamp. The normalized one-line response digest is retained here; raw responses remain allocated scratch."
        },
        {
          "class": "runtime_tree",
          "status": "missing",
          "notes": "No pinned runtime tree slice is retained in git or allocated scratch for this run."
        }
      ],
      "candidates": [
        {
          "id": "afd-attention-ffn-disaggregation",
          "title": "Attention/FFN disaggregation topology",
          "rationale": "Changes the topology frontier beyond conventional prefill/decode disaggregation and can alter FAK routing and receipt boundaries.",
          "repository_links": [
            {
              "owner": "vllm-project",
              "repo": "afd-plugin"
            }
          ],
          "vllm_mechanism_link": "vLLM worker execution split across attention and feed-forward stages"
        }
      ]
    },
    {
      "name": "AIBrix",
      "repository": {
        "owner": "vllm-project",
        "repo": "aibrix"
      },
      "pin": {
        "revision": "7540088967a00a00d9e954688543914ec84ec832",
        "cutoff": "2026-08-26T22:35:00Z",
        "observed_at": "2026-08-27T02:29:21Z"
      },
      "processed": true,
      "inclusion_rationale": "Active vllm-project infrastructure runtime with routing, autoscaling, cache/offload, and distributed inference control mechanisms distinct from deployment assembly alone.",
      "decision_relation": "Changes FAK decisions about external autoscaling and cache-aware control-plane integration without authorizing a second native control plane.",
      "freshness_notes": "Newest default-branch commit at or before the shared cutoff is 7540088967a00a00d9e954688543914ec84ec832 (2026-08-26T19:43:52Z); GitHub metadata observed 2026-08-27T02:29:21Z. The cutoff pin is current metadata-only evidence.",
      "partial_notes": "No local tree inventory or forge corpus was captured; classification is mechanism-level and partial.",
      "source_class_receipts": [
        {
          "class": "forge_history",
          "status": "complete",
          "terminal_receipt": "study-forge:sha256:498dfd478eb302fc630665c0b23e66ff728e3caad168851e84a0c3f6d9800c25; file-sha256:f43e945d1fcb0a9131bf9aed741cd0ba5a1fb65160f16a351c5c94898f4ead62; records:2690; bytes:8549752",
          "notes": "Complete validated study-forge corpus at the shared cutoff and pinned revision 7540088967a00a00d9e954688543914ec84ec832; full corpus retained in allocated scratch."
        },
        {
          "class": "repository_metadata",
          "status": "complete",
          "terminal_receipt": "github-graphql:defaultBranchRef.history(first:1,until=2026-08-26T22:35:00Z); response-sha256:8bf963bedc1b4fd3d55c0b32a840b25f50c23f1d0e01426dd5a2d85cb955caa3",
          "notes": "Authoritative GitHub GraphQL returned the canonical identity, default branch, cutoff revision, and commit timestamp. The normalized one-line response digest is retained here; raw responses remain allocated scratch."
        },
        {
          "class": "runtime_tree",
          "status": "missing",
          "notes": "No pinned runtime tree inventory is present."
        }
      ],
      "candidates": [
        {
          "id": "aibrix-autoscale-cache-control",
          "title": "Autoscaling and distributed cache-control frontier",
          "rationale": "Separates external cluster elasticity from the FAK-native request policy, scheduling, and cache-ownership decisions.",
          "repository_links": [
            {
              "owner": "vllm-project",
              "repo": "aibrix"
            }
          ],
          "vllm_mechanism_link": "vLLM serving deployment, request routing, KV cache/offload, and replica management"
        }
      ]
    },
    {
      "name": "vLLM Production Stack",
      "repository": {
        "owner": "vllm-project",
        "repo": "production-stack"
      },
      "pin": {
        "revision": "58a0935955d5b29f615c784a3533ff2433075bdd",
        "cutoff": "2026-08-26T22:35:00Z",
        "observed_at": "2026-08-27T02:29:21Z"
      },
      "processed": true,
      "inclusion_rationale": "Active vllm-project reference deployment that assembles cluster-wide Kubernetes serving and performance components around vLLM.",
      "decision_relation": "Changes deployment and operational-envelope conclusions that a single vLLM repository cannot establish.",
      "freshness_notes": "Newest default-branch commit at or before the shared cutoff is 58a0935955d5b29f615c784a3533ff2433075bdd (2026-08-18T19:39:11Z); GitHub metadata observed 2026-08-27T02:29:21Z. The cutoff pin is unchanged since August 18 and the partial forge capture is pinned exactly to it.",
      "partial_notes": "Scratch study-forge capture exhausted issues, pulls, releases, labels, and milestones, but Discussions returned HTTP 410 and PR endpoint reconciliation exceeded policy; status remains partial.",
      "source_class_receipts": [
        {
          "class": "forge_history",
          "status": "complete",
          "terminal_receipt": "study-forge:sha256:80a468dc2d1da9db5a752de66e69fce311274f533b8782739ba095b422e2b00c; file-sha256:57702b0369e5a9a68dfb407dd71af425aa11f0143cbe83eb93462f8d67da67a2; records:1088; bytes:4109839",
          "notes": "Complete validated study-forge corpus at the shared cutoff and pinned revision 58a0935955d5b29f615c784a3533ff2433075bdd; full corpus retained in allocated scratch."
        },
        {
          "class": "repository_metadata",
          "status": "complete",
          "terminal_receipt": "github-graphql:defaultBranchRef.history(first:1,until=2026-08-26T22:35:00Z); response-sha256:1e45cba00cf8e04e5a30ef31858375d287db6b2ab866d40bbcd3a84a81ef8083",
          "notes": "Authoritative GitHub GraphQL returned the canonical identity, default branch, cutoff revision, and commit timestamp. The normalized one-line response digest is retained here; raw responses remain allocated scratch."
        },
        {
          "class": "runtime_tree",
          "status": "missing",
          "notes": "No standard pinned runtime tree inventory was produced in this run."
        }
      ],
      "candidates": [
        {
          "id": "production-stack-kubernetes-topology",
          "title": "Cluster-wide deployment assembly",
          "rationale": "Changes which deployment/operations conclusions are attributable to vLLM itself versus its reference production assembly.",
          "repository_links": [
            {
              "owner": "vllm-project",
              "repo": "production-stack"
            }
          ],
          "vllm_mechanism_link": "vLLM Kubernetes serving topology, replicas, routing, observability, and operational recipes"
        }
      ]
    },
    {
      "name": "vLLM Router",
      "repository": {
        "owner": "vllm-project",
        "repo": "router"
      },
      "pin": {
        "revision": "1d10e71fb7bb4c0adc9f2c16ec77bf5dd4aa1586",
        "cutoff": "2026-08-26T22:35:00Z",
        "observed_at": "2026-08-27T02:29:21Z"
      },
      "processed": true,
      "inclusion_rationale": "Active vllm-project high-performance router with a distinct prefix/cache-aware endpoint-placement seam.",
      "decision_relation": "Directly changes FAK's cache-residency routing comparison and the integration boundary between request policy and backend placement.",
      "freshness_notes": "Newest default-branch commit at or before the shared cutoff is 1d10e71fb7bb4c0adc9f2c16ec77bf5dd4aa1586 (2026-08-18T09:16:56Z); GitHub metadata observed 2026-08-27T02:29:21Z. The cutoff pin is unchanged since August 18 and the partial forge capture is pinned exactly to it.",
      "partial_notes": "Scratch capture exhausted five enabled endpoints and accepted the bounded PR reconciliation, but Discussions returned HTTP 410; the overall receipt is partial.",
      "source_class_receipts": [
        {
          "class": "forge_history",
          "status": "complete",
          "terminal_receipt": "study-forge:sha256:1973cac7a70d66f60cb0f778f05561eb184b0e39f3e61dd03241c361a902a97c; file-sha256:bbc3cf9f2dba1e006a3eb35e825f9b0c728141c9970589689ac34bedba4c3543; records:232; bytes:736345",
          "notes": "Complete validated study-forge corpus at the shared cutoff and pinned revision 1d10e71fb7bb4c0adc9f2c16ec77bf5dd4aa1586; full corpus retained in allocated scratch."
        },
        {
          "class": "repository_metadata",
          "status": "complete",
          "terminal_receipt": "github-graphql:defaultBranchRef.history(first:1,until=2026-08-26T22:35:00Z); response-sha256:6000415ff90e6b649bd860b38b96980ff79252009cdcb221ac68bce0f66a201f",
          "notes": "Authoritative GitHub GraphQL returned the canonical identity, default branch, cutoff revision, and commit timestamp. The normalized one-line response digest is retained here; raw responses remain allocated scratch."
        },
        {
          "class": "runtime_tree",
          "status": "missing",
          "notes": "No standard pinned runtime tree inventory was produced in this run."
        }
      ],
      "candidates": [
        {
          "id": "router-prefix-aware-placement",
          "title": "Prefix-aware endpoint placement",
          "rationale": "Changes whether FAK should consume external residency signals or keep placement entirely inside its own fleet router.",
          "repository_links": [
            {
              "owner": "vllm-project",
              "repo": "router"
            }
          ],
          "vllm_mechanism_link": "vLLM request routing using prefix/KV residency and backend load"
        }
      ]
    },
    {
      "name": "vLLM Semantic Router",
      "repository": {
        "owner": "vllm-project",
        "repo": "semantic-router"
      },
      "pin": {
        "revision": "2aaf0a647e491e8dde22a4ce6888975e358c9008",
        "cutoff": "2026-08-26T22:35:00Z",
        "observed_at": "2026-08-27T02:36:25Z"
      },
      "processed": true,
      "inclusion_rationale": "Active vllm-project programmable mixture-of-models router whose semantic model-choice seam is distinct from prefix-aware endpoint placement.",
      "decision_relation": "Changes the boundary between FAK policy/model routing and an external semantic router before a request reaches vLLM.",
      "freshness_notes": "Newest default-branch commit at or before the shared cutoff is 2aaf0a647e491e8dde22a4ce6888975e358c9008 (2026-08-26T16:50:54Z); GitHub metadata observed 2026-08-27T02:36:25Z. The cutoff pin is current metadata-only evidence.",
      "partial_notes": "No local tree inventory or forge corpus was captured; only the exact GitHub cutoff pin and decision-changing contrast are claimed.",
      "source_class_receipts": [
        {
          "class": "forge_history",
          "status": "complete",
          "terminal_receipt": "study-forge:sha256:0955662d24595bed793fecd2b8ba64616b5961980ec16fdb00585a95af99792a; file-sha256:4b141321dabc1f2b2495f791a46804d699756aa87da3c0f66e24ddee394c7c33; records:3112; bytes:11625722",
          "notes": "Complete validated study-forge corpus at the shared cutoff and pinned revision 2aaf0a647e491e8dde22a4ce6888975e358c9008; full corpus retained in allocated scratch."
        },
        {
          "class": "repository_metadata",
          "status": "complete",
          "terminal_receipt": "github-graphql:defaultBranchRef.history(first:1,until=2026-08-26T22:35:00Z); response-sha256:80e031c77a6249d52c331027c34ca49a155ba7ed588533a1481da8dc721a2b38",
          "notes": "Authoritative GitHub GraphQL returned the canonical identity, default branch, cutoff revision, and commit timestamp. The normalized one-line response digest is retained here; raw responses remain allocated scratch."
        },
        {
          "class": "runtime_tree",
          "status": "missing",
          "notes": "No pinned runtime tree inventory is present."
        }
      ],
      "candidates": [
        {
          "id": "semantic-router-model-choice",
          "title": "Semantic model-choice control plane",
          "rationale": "Prevents conflating semantic model selection, policy routing, and cache-aware replica placement into one mechanism.",
          "repository_links": [
            {
              "owner": "vllm-project",
              "repo": "semantic-router"
            }
          ],
          "frontier_changing_contrast": "Semantic routing chooses among heterogeneous model endpoints before vLLM execution, unlike prefix-aware placement among replicas and unlike FAK's policy-constrained model routing."
        }
      ]
    },
    {
      "name": "vLLM Speculators",
      "repository": {
        "owner": "vllm-project",
        "repo": "speculators"
      },
      "pin": {
        "revision": "51f8e02f077c9336f9e4cc66155e22127f354c5d",
        "cutoff": "2026-08-26T22:35:00Z",
        "observed_at": "2026-08-27T02:29:21Z"
      },
      "processed": true,
      "inclusion_rationale": "Active vllm-project speculative-decoding library and the canonical satellite for drafter algorithms, acceptance controllers, and checkpoint compatibility.",
      "decision_relation": "Changes vLLM-derived speculative-decoding conclusions and maps directly to FAK native/ride-mode draft ownership and quality gates.",
      "freshness_notes": "Newest default-branch commit at or before the shared cutoff is 51f8e02f077c9336f9e4cc66155e22127f354c5d (2026-08-26T19:46:42Z); GitHub metadata observed 2026-08-27T02:29:21Z. The cutoff pin is current; the committed exhaustive map is one day stale.",
      "partial_notes": "The exact-pin scratch capture terminally traversed all six endpoints but failed exact PR identity reconciliation (1,736 symmetric difference \u003e 1,000), so it remains a validated partial rather than complete.",
      "source_class_receipts": [
        {
          "class": "forge_history",
          "status": "complete",
          "terminal_receipt": "study-forge:sha256:37d3929622cbda83b513febdd5bf601f6a4ca58c4e06c38a961ad99bb62d7b11; file-sha256:dd6210c8045fa95e2fdc298c765670b68e435555415497157c05633042cfa92e; records:1078; bytes:3839241",
          "notes": "Complete validated study-forge corpus at the shared cutoff and pinned revision 51f8e02f077c9336f9e4cc66155e22127f354c5d; full corpus retained in allocated scratch."
        },
        {
          "class": "repository_metadata",
          "status": "complete",
          "terminal_receipt": "github-graphql:defaultBranchRef.history(first:1,until=2026-08-26T22:35:00Z); response-sha256:ae7848b63e4245d5d6ed8c8915a0e19bd91dda17a42132cf2ed07dc08b461d3b",
          "notes": "Authoritative GitHub GraphQL returned the canonical identity, default branch, cutoff revision, and commit timestamp. The normalized one-line response digest is retained here; raw responses remain allocated scratch."
        },
        {
          "class": "runtime_tree",
          "status": "partial",
          "notes": "Reused docs/research/inventory/vllm-project-speculators.json at stale pin 0faffeb3; exact cutoff tree is not inventoried."
        }
      ],
      "candidates": [
        {
          "id": "speculators-dynamic-draft-controllers",
          "title": "Dynamic speculative-draft controllers",
          "rationale": "Changes FAK decisions on adaptive draft length, acceptance telemetry, checkpoint alias regression tests, and when the external engine owns speculation.",
          "repository_links": [
            {
              "owner": "vllm-project",
              "repo": "speculators"
            }
          ],
          "vllm_mechanism_link": "vLLM speculative decoding, draft-model loading, acceptance, and multi-token prediction"
        }
      ]
    },
    {
      "name": "vLLM GGUF plugin",
      "repository": {
        "owner": "vllm-project",
        "repo": "vllm-gguf-plugin"
      },
      "pin": {
        "revision": "fb973ad784f38b98b054e136bec3414b7cd8494d",
        "cutoff": "2026-08-26T22:35:00Z",
        "observed_at": "2026-08-27T02:29:21Z"
      },
      "processed": true,
      "inclusion_rationale": "Active vllm-project runtime plugin that loads GGUF artifacts, directly bridging the vLLM and llama.cpp/FAK artifact worlds.",
      "decision_relation": "Changes whether GGUF interoperability is a portable artifact contract or an engine-specific runtime capability.",
      "freshness_notes": "Newest default-branch commit at or before the shared cutoff is fb973ad784f38b98b054e136bec3414b7cd8494d (2026-08-20T15:17:36Z); GitHub metadata observed 2026-08-27T02:29:21Z. The cutoff pin is unchanged since August 20 and is metadata-only evidence.",
      "partial_notes": "No local tree inventory or forge corpus was captured; only exact repository metadata and the interoperability contrast are claimed.",
      "source_class_receipts": [
        {
          "class": "forge_history",
          "status": "complete",
          "terminal_receipt": "study-forge:sha256:e2c5eb1f769d6729dc4eee0899a52d8df116702ada123ff3a93a99ea4bbedfa0; file-sha256:7349113f2ff31fe8c0522a573d90eeb94e87e72b1692b32e6ed4e47a1531fd5a; records:140; bytes:392667",
          "notes": "Complete validated study-forge corpus at the shared cutoff and pinned revision fb973ad784f38b98b054e136bec3414b7cd8494d; full corpus retained in allocated scratch."
        },
        {
          "class": "repository_metadata",
          "status": "complete",
          "terminal_receipt": "github-graphql:defaultBranchRef.history(first:1,until=2026-08-26T22:35:00Z); response-sha256:e78d016fc2451101764ecba28ee58f733a05fa7915773a6fcd93a44d606e068c",
          "notes": "Authoritative GitHub GraphQL returned the canonical identity, default branch, cutoff revision, and commit timestamp. The normalized one-line response digest is retained here; raw responses remain allocated scratch."
        },
        {
          "class": "runtime_tree",
          "status": "missing",
          "notes": "No pinned runtime tree inventory is present."
        }
      ],
      "candidates": [
        {
          "id": "vllm-gguf-artifact-compatibility",
          "title": "GGUF artifact interoperability",
          "rationale": "Keeps artifact identity and quantization evidence separate from token-engine performance claims.",
          "repository_links": [
            {
              "owner": "vllm-project",
              "repo": "vllm-gguf-plugin"
            }
          ],
          "vllm_mechanism_link": "vLLM model loading and quantization-plugin interfaces",
          "frontier_changing_contrast": "GGUF is native to llama.cpp and FAK's portable artifact path but enters vLLM through a separately versioned plugin, changing compatibility and benchmark-envelope assumptions."
        }
      ]
    },
    {
      "name": "vLLM Metal",
      "repository": {
        "owner": "vllm-project",
        "repo": "vllm-metal"
      },
      "pin": {
        "revision": "db4d7c57c7d72f731f7010fd3b60fc25fc489229",
        "cutoff": "2026-08-26T22:35:00Z",
        "observed_at": "2026-08-27T02:29:21Z"
      },
      "processed": true,
      "inclusion_rationale": "Active vllm-project Apple-Silicon backend and the only hardware satellite admitted because Apple Metal is in FAK's current native operating envelope.",
      "decision_relation": "Changes the Apple execution and API-compatibility frontier while FAK retains ownership of native Metal kernels, memory, scheduling, cache, adaptation, and receipts.",
      "freshness_notes": "Newest default-branch commit at or before the shared cutoff is db4d7c57c7d72f731f7010fd3b60fc25fc489229 (2026-08-26T13:39:22Z); GitHub metadata observed 2026-08-27T02:29:21Z. The cutoff pin is current and the scratch forge capture is pinned exactly to it.",
      "partial_notes": "Scratch capture exhausted five enabled endpoints, but Discussions returned HTTP 410 and PR reconciliation exceeded policy (1,060 \u003e 1,000); no tree inventory exists.",
      "source_class_receipts": [
        {
          "class": "forge_history",
          "status": "complete",
          "terminal_receipt": "study-forge:sha256:2da0ed39efa76b19305d4f7039debb7245e6eb27101326e193f0dc7fcf17422d; file-sha256:c4e674cc13c122438aa5f862aa96de63b4586edaa693a525683da8994ede5902; records:1114; bytes:2288792",
          "notes": "Complete validated study-forge corpus at the shared cutoff and pinned revision db4d7c57c7d72f731f7010fd3b60fc25fc489229; full corpus retained in allocated scratch."
        },
        {
          "class": "repository_metadata",
          "status": "complete",
          "terminal_receipt": "github-graphql:defaultBranchRef.history(first:1,until=2026-08-26T22:35:00Z); response-sha256:99d629405d298eb28c0707403bca1bd692debf08a5c682a644c86712871ee3fd",
          "notes": "Authoritative GitHub GraphQL returned the canonical identity, default branch, cutoff revision, and commit timestamp. The normalized one-line response digest is retained here; raw responses remain allocated scratch."
        },
        {
          "class": "runtime_tree",
          "status": "missing",
          "notes": "No standard pinned runtime tree inventory was produced in this run."
        }
      ],
      "candidates": [
        {
          "id": "vllm-metal-apple-runtime",
          "title": "Apple-Silicon vLLM-compatible runtime",
          "rationale": "Defines an optional comparator and interoperability source without weakening fak-native ownership.",
          "repository_links": [
            {
              "owner": "vllm-project",
              "repo": "vllm-metal"
            }
          ],
          "vllm_mechanism_link": "vLLM scheduler/API semantics mapped onto an Apple Metal backend",
          "frontier_changing_contrast": "A vLLM-compatible Apple backend changes the comparison baseline, but FAK must remain native rather than silently route native-performance work through another engine."
        }
      ]
    }
  ]
}
