{
  "census_date": "2026-10-02",
  "papers": [
    {
      "id": "spacecore",
      "title": "A Case for Stateless Mobile Core Network Functions in Space",
      "authors": "Yuanjie Li; Hewu Li; Wei Liu; Lixin Liu; Yimei Chen; Jianping Wu; Qian Wu; Jun Liu; Zeqi Lai",
      "year": "2022",
      "venue": "SIGCOMM",
      "url": "https://raw.githubusercontent.com/yuanjieli/SpaceCore-SIGCOMM22/master/sigcomm22.pdf",
      "pdf_url": "https://raw.githubusercontent.com/yuanjieli/SpaceCore-SIGCOMM22/master/sigcomm22.pdf",
      "regime": "A · Communication constellation",
      "evidence": [
        "§3 function splits; §4 design; §5 implementation; §6 and Table 4, PDF pp.10–13"
      ],
      "figure_number": "1",
      "evidence_status": "Full primary text reviewed",
      "built_for_en": "Satellite-hosted 5G core functions serve terrestrial mobile users.",
      "built_for_zh": "衛星承載 5G core functions，服務地面行動用戶。",
      "problem_en": "High orbital mobility repeatedly migrates per-user core state and creates signaling overhead.",
      "problem_zh": "高速軌道移動反覆遷移用戶 core state，造成 signaling 負擔。",
      "design_en": "Decouple function execution from state; anchor service areas geographically; store authenticated user state at the device and retrieve it locally.",
      "design_zh": "拆開 function execution 與 state，以固定地理服務區域承接 mobility，讓裝置保存可信 user state 並提供本地擷取。",
      "model_en": "Function/state split taxonomy, geographic grid addressing, orbital dynamics, protocol state-machine analysis and cryptographic state tokens.",
      "model_zh": "function/state 切分分類、地理網格位址、軌道動態、protocol state machine 與加密 state token。",
      "evaluation_en": "Commodity Raspberry Pi 4 satellite prototype, UERANSIM device emulation and Open5GS home core; operational Tiantong/Inmarsat and terrestrial 5G signaling datasets; constellation-scale replay.",
      "evaluation_zh": "Raspberry Pi 4 衛星 prototype、UERANSIM 裝置 emulation 與 Open5GS home core；Tiantong/Inmarsat 和地面 5G signaling datasets；星座規模 replay。",
      "baselines_en": "5G NTN, SkyCore, Baoyun, DPCM; protocol-function splits provide further empirical comparisons.",
      "baselines_zh": "5G NTN、SkyCore、Baoyun、DPCM；額外以 protocol-function split 比較功能成本。",
      "result_en": "For Starlink-like replay, satellite signaling reductions are 122.2× versus 5G NTN, 17.5× versus SkyCore, 40.3× versus DPCM and 49.3× versus Baoyun (Table 4).",
      "result_zh": "Starlink 型 replay 的 satellite signaling 降低幅度：對 5G NTN 為 122.2×、SkyCore 為 17.5×、DPCM 為 40.3×、Baoyun 為 49.3×（Table 4）。",
      "limitations_en": "The deployed system evidence covers terrestrial protocol hardware and traces; constellation behavior follows orbital replay. Endpoint state requires device support and trustworthy token processing.",
      "limitations_zh": "實作證據涵蓋地面 protocol 硬體與 traces；星座行為由軌道 replay 建立。端點 state 需要裝置支援與可信 token 處理。",
      "figure_page_1based": 1,
      "figure_bbox_points": [
        315,
        142,
        561,
        242
      ],
      "mechanism_scope": "Mobile core / state placement",
      "deployment_regime": "A",
      "space_ground_path": true,
      "display_regime_en": "A · Communication constellation",
      "display_regime_zh": "A · 通訊星座",
      "primary_pdf_sha256": "f95bb0f14dec2b27bb130dfabfa84717d242d176679e282fa6ffcc2bc5a54f55",
      "figure_asset": "../figures/spacecore.png",
      "figure_acquisition": "Primary PDF crop",
      "figure_content_type": "image/png",
      "figure_bytes": 50879,
      "figure_sha256": "149ebd7ca4b7a4ac6c33e14afa1c1d71c11ec03ebdfb60cd98bf472d4fb55159"
    },
    {
      "id": "sate",
      "title": "SaTE: Low-Latency Traffic Engineering for Satellite Networks",
      "authors": "Hao Wu; Yizhan Han; Mohit Rajpal; Qizhen Zhang; Jingxian Wang",
      "year": "2025",
      "venue": "SIGCOMM",
      "url": "https://fardatalab.org/sigcomm25-wu.pdf",
      "pdf_url": "https://fardatalab.org/sigcomm25-wu.pdf",
      "regime": "A · Communication constellation",
      "evidence": [
        "§2 formulation; §3 GNN; §4 setup; §5.1–5.5; Appendix H.1 offline; PDF pp.9–12"
      ],
      "figure_number": "5",
      "evidence_status": "Full primary text reviewed",
      "built_for_en": "A controller allocates traffic across paths in a 4,236-satellite network.",
      "built_for_zh": "controller 在 4,236 顆衛星網路的路徑間分配 traffic。",
      "problem_en": "Solver latency makes allocations stale while links and traffic change.",
      "problem_zh": "links 與 traffic 持續變動，solver 計算時間讓 allocation 過期。",
      "design_en": "Use a heterogeneous graph of satellites, paths and traffic; three graph-attention modules predict allocations, with capacity trimming and topology/traffic/path pruning.",
      "design_zh": "以 satellite、path、traffic 建立 heterogeneous graph，三組 graph-attention module 預測 allocation，再進行 capacity trimming 與 topology/traffic/path pruning。",
      "model_en": "Path-based multi-commodity throughput maximization; supervised GNN labels from Gurobi; Determinantal Point Process topology sampling.",
      "model_zh": "path-based multi-commodity throughput maximization；Gurobi 提供 supervised GNN labels；Determinantal Point Process 抽樣 topology。",
      "evaluation_en": "FCC-derived orbital simulations; 3M population-weighted users and 1,000 gateways; Poisson arrivals 125–500 flows/s; scaled 200Mbps ISLs and 50Mbps access; Azure A100; train/test 4:1 across 10,000 snapshots.",
      "evaluation_zh": "FCC 軌道模擬；300 萬人口加權用戶與 1,000 gateways；Poisson arrivals 125–500 flows/s；縮放至 200Mbps ISL 與 50Mbps access；Azure A100；10,000 snapshots 以 4:1 分割 train/test。",
      "baselines_en": "Gurobi, POP, ECMP with Water Filling, backpressure Satellite Routing, Teal, HARP.",
      "baselines_zh": "Gurobi、POP、ECMP with Water Filling、backpressure Satellite Routing、Teal、HARP。",
      "result_en": "17ms mean allocation inference, 2,738× faster than Gurobi; online satisfied demand improves 23.5% with cross-shell lasers and 46.6% with ground relays. Incremental path calculation separately averages 56ms.",
      "result_zh": "allocation inference 平均 17ms，較 Gurobi 快 2,738×；online satisfied demand 在 cross-shell laser 與 ground relay 情境提升 23.5% 和 46.6%。額外 incremental path calculation 平均 56ms。",
      "limitations_en": "Gain depends on charging computation delay to every method; offline Gurobi remains the reference optimum. Scaled capacities express relative trends; aggregate throughput objective leaves individual-flow fairness as an extension.",
      "limitations_zh": "增益包含各方法 computation delay；offline Gurobi 提供最佳化參照。縮放容量呈現相對趨勢；aggregate throughput objective 的 individual-flow fairness 可進一步擴展。",
      "figure_page_1based": 6,
      "figure_bbox_points": [
        315,
        85,
        563,
        244
      ],
      "mechanism_scope": "Dynamic routing / TE",
      "deployment_regime": "A",
      "space_ground_path": false,
      "display_regime_en": "A · Communication constellation",
      "display_regime_zh": "A · 通訊星座",
      "primary_pdf_sha256": "ca1f5c966fd6d8f97b6558a1c6a5c0abc7946d94fa0ca973d40e05da7b479ef4",
      "figure_asset": "../figures/sate.png",
      "figure_acquisition": "Primary PDF crop",
      "figure_content_type": "image/png",
      "figure_bytes": 93614,
      "figure_sha256": "506af2134c412868370758d90ac105dc447b2be2ecb50294c7edf6b219630e33"
    },
    {
      "id": "tinyleo",
      "title": "Small-scale LEO Satellite Networking for Global-scale Demands",
      "authors": "Yuanjie Li; Yimei Chen; Jiabo Yang; Jinyao Zhang; Bowen Sun; Lixin Liu; Hewu Li; Jianping Wu; Zeqi Lai; Qian Wu; Jun Liu",
      "year": "2025",
      "venue": "SIGCOMM",
      "url": "https://tinyleo-toolkit.github.io/TinyLEO/static/pdfs/sigcomm25-tinyleo.pdf",
      "pdf_url": "https://tinyleo-toolkit.github.io/TinyLEO/static/pdfs/sigcomm25-tinyleo.pdf",
      "regime": "A · Communication constellation",
      "evidence": [
        "§4.1 Equations 2–4; §4 MPC and anycast; §5 toolkit; §6.1–6.3, Fig.15; PDF pp.9–12"
      ],
      "figure_number": "6",
      "evidence_status": "Full primary text reviewed",
      "built_for_en": "A provider synthesizes a sparse constellation for specific geographic demand.",
      "built_for_zh": "provider 依指定地理需求合成稀疏星座。",
      "problem_en": "Uniform orbital supply serves spatially uneven demand with excess satellites and complex control updates.",
      "problem_zh": "均勻 orbital supply 對應空間分佈集中的 demand，造成衛星供給浪費與 control 更新成本。",
      "design_en": "Combine diverse Earth-repeat orbits; approximate supply-demand matching; preserve geographic intents using orbital MPC; enforce geographic SRv6 segment anycast.",
      "design_zh": "組合多樣 Earth-repeat orbits，以近似演算法匹配 supply-demand，利用 orbital MPC 維持 geographic intent，再以 geographic SRv6 segment anycast 執行。",
      "model_en": "Integer-program constellation synthesis, greedy matching approximation, model predictive control and geographic reachability verification.",
      "model_zh": "integer-program 星座合成、greedy matching approximation、model predictive control 與 geographic reachability verification。",
      "evaluation_en": "64,800 candidate tracks, 4,050 cells, three real-demand scenarios; 6,793-satellite Jan2025 reference; packet-level 1,741-container StarryNet-derived hardware-in-loop. Each modeled satellite offers 3×200Gbps ISL and 96Gbps radio access.",
      "evaluation_zh": "64,800 candidate tracks、4,050 cells、三種 real-demand scenarios；2025 年 1 月 6,793 顆 reference；1,741-container 的 StarryNet 延伸 packet-level hardware-in-loop。模型每顆 satellite 配置 3×200Gbps ISL 與 96Gbps radio access。",
      "baselines_en": "Starlink, MegaReduce, Gurobi v12.0 synthesis terminated at two months; geo-routing/IP control comparisons in §6.2–6.3.",
      "baselines_zh": "Starlink、MegaReduce、Gurobi v12.0 合成執行兩個月後取 incumbent；§6.2–6.3 另比較 geo-routing/IP control。",
      "result_en": "Strict demand matching gives 1,763/3,344/1,066 satellites versus 6,793 for customer/backbone/Latin-America demand; 99% availability gives 1,391/3,184/865. Planning takes 6.5–7.7h; control costs improve by 1–3 orders.",
      "result_zh": "嚴格 demand matching 在 customer/backbone/Latin-America 需求採 1,763/3,344/1,066 顆，相對 reference 6,793 顆；99% availability 採 1,391/3,184/865 顆。planning 為 6.5–7.7h；control cost 改善 1–3 個數量級。",
      "limitations_en": "The 7.9× headline corresponds to regional demand at 99% availability. Capacity and orbital synthesis assumptions determine feasibility; global access traffic differs from all-to-all cluster traffic.",
      "limitations_zh": "7.9× headline 對應 regional demand 與 99% availability。capacity 與 orbital synthesis 假設共同決定 feasibility；全球 access traffic 與 cluster all-to-all traffic 具有各自需求。",
      "figure_page_1based": 4,
      "figure_bbox_points": [
        314,
        71,
        593,
        242
      ],
      "mechanism_scope": "Topology supply / geographic control",
      "deployment_regime": "A",
      "space_ground_path": false,
      "display_regime_en": "A · Communication constellation",
      "display_regime_zh": "A · 通訊星座",
      "primary_pdf_sha256": "61fc1bacba5dbbc9f1342d1362025323b52358a2be8d632dc2bf4a7bf9191e45",
      "figure_asset": "../figures/tinyleo.png",
      "figure_acquisition": "Primary PDF crop",
      "figure_content_type": "image/png",
      "figure_bytes": 187079,
      "figure_sha256": "59915774a87cdabd86697cdb808455000f0805a8c89f26aaf5c704457128d414"
    },
    {
      "id": "starcdn",
      "title": "StarCDN: Moving Content Delivery Networks to Space",
      "authors": "William X. Zheng; Aryan Taneja; Maleeha Masood; Anirudh Sabnis; Ramesh K. Sitaraman; Deepak Vasisht",
      "year": "2025",
      "venue": "SIGCOMM",
      "url": "https://maleehamasood.github.io/content/papers/sigcomm-2025.pdf",
      "pdf_url": "https://maleehamasood.github.io/content/papers/sigcomm-2025.pdf",
      "regime": "A · Communication constellation",
      "evidence": [
        "§3 mechanism; §4 SpaceGEN; §5.1 setup; §5.2–5.4; Fig.5/7/10; PDF pp.7–10"
      ],
      "figure_number": "5",
      "evidence_status": "Full primary text reviewed",
      "built_for_en": "LEO Internet users access content cached on moving satellites.",
      "built_for_zh": "LEO Internet 用戶存取移動衛星上的 cached content。",
      "problem_en": "Orbital motion changes the user population served by each cache and reduces locality.",
      "problem_zh": "orbital motion 改變各 cache 的服務用戶羣，降低 content locality。",
      "design_en": "Assign consistent-hash buckets across nearby satellites; relay fetches from neighboring replicas so content flows against orbital motion; generate correlated requests using SpaceGEN.",
      "design_zh": "在鄰近衛星配置 consistent-hash buckets，從 neighboring replicas relay fetch，讓 content 逆 orbital motion 流動；以 SpaceGEN 建立 correlated requests。",
      "model_en": "Consistent hashing and replica graph; trace-based cache replay with LRU; empirical locality and bucket/latency tradeoff.",
      "model_zh": "consistent hashing 與 replica graph；LRU trace-based cache replay；量測 locality 與 bucket/latency tradeoff。",
      "evaluation_en": "Akamai production object traces seed five-day synthetic traces; CosmicBeats 15s orbital time step; multiprocess TCP cache replayer; varied 10–100GB caches and L=4/9 buckets.",
      "evaluation_zh": "Akamai production object traces 作為五天 synthetic traces 種子；CosmicBeats 15s orbital time step；multiprocess TCP cache replayer；10–100GB caches 與 L=4/9 buckets。",
      "baselines_en": "Naive LRU, ideal static satellite cache; StarCDN hashing-only ablation; measured terrestrial CDN and regular Starlink idle latency.",
      "baselines_zh": "Naive LRU、ideal static satellite cache、hashing-only ablation，以及 terrestrial CDN 和 regular Starlink idle latency 量測。",
      "result_en": "Ground-to-space bandwidth falls to 20–25% of uncached access; modeled median latency 22ms versus observed idle Starlink 55ms. At 9.7% unavailable satellites, bandwidth savings remain 74%.",
      "result_zh": "ground-to-space bandwidth 降至 uncached access 的 20–25%；模型 median latency 22ms，相對 idle Starlink 量測 55ms。9.7% satellite unavailable 的情境仍保留 74% bandwidth savings。",
      "limitations_en": "Latency analysis estimates propagation; queue and satellite processing require additional measurements. More buckets increase ISL latency; failures reduce hit rate.",
      "limitations_zh": "latency analysis 估計 propagation；queue 與 satellite processing 需要額外量測。更多 buckets 增加 ISL latency；failure 降低 hit rate。",
      "figure_page_1based": 7,
      "figure_bbox_points": [
        54,
        76,
        562,
        262
      ],
      "mechanism_scope": "Orbital caching / state placement",
      "deployment_regime": "A",
      "space_ground_path": true,
      "display_regime_en": "A/C · Orbital content delivery",
      "display_regime_zh": "A／C · 軌道內容交付",
      "primary_pdf_sha256": "21b9593c992e4c2d32f774fba9515f4663412c69549e47afae3bc7c0a5338794",
      "figure_asset": "../figures/starcdn.png",
      "figure_acquisition": "Primary PDF crop",
      "figure_content_type": "image/png",
      "figure_bytes": 573187,
      "figure_sha256": "e01b1c8e702b985a7b1dcfd2365defd1f42d1b4c1ba6abb9e5263d3b1f54f8f0"
    },
    {
      "id": "deepspace",
      "title": "DeepSpace: Super Resolution Powered Efficient and Reliable Satellite Image Data Acquistion",
      "authors": "Chuanhao Sun; Yu Zhang; Bill Tao; Deepak Vasisht; Mahesh K. Marina",
      "year": "2025",
      "venue": "SIGCOMM",
      "url": "https://www.pure.ed.ac.uk/ws/portalfiles/portal/546427521/SunEtalSIGCOMM25DeepSpace.pdf",
      "pdf_url": "https://www.pure.ed.ac.uk/ws/portalfiles/portal/546427521/SunEtalSIGCOMM25DeepSpace.pdf",
      "regime": "E · Observation/contact fleet",
      "evidence": [
        "§3–4 mechanism; §5.1–5.3; §6.1–6.2; Table4; §7 downstream; PDF pp.8–12"
      ],
      "figure_number": "2",
      "evidence_status": "Full primary text reviewed",
      "built_for_en": "Earth-observation satellites transmit compact imagery for cloud reconstruction.",
      "built_for_zh": "Earth-observation satellites 傳送精簡 imagery，由 cloud 重建。",
      "problem_en": "Intermittent downlinks carry large images while satellite compute, storage and uplink budgets constrain compression.",
      "problem_zh": "intermittent downlink 承載大量 images，同時 satellite compute、storage 與 uplink budget 限制 compression。",
      "design_en": "Lightweight BLSH image hashing and adaptive sampling run onboard; cloud MoE wavelet-diffusion super resolution selects experts and checks reconstruction metadata.",
      "design_zh": "onboard 執行輕量 BLSH image hashing 與 adaptive sampling；cloud MoE wavelet-diffusion super resolution 選 expert，並以 metadata 檢查 reconstruction。",
      "model_en": "Image-distance geometry, compression-ratio optimization, SSIM-based fidelity constraints, diffusion models and post-hoc low-resolution/hash similarity checks.",
      "model_zh": "image-distance geometry、compression-ratio optimization、SSIM fidelity constraints、diffusion models，以及 post-hoc low-resolution/hash similarity checks。",
      "evaluation_en": "Five custom/public datasets: Planet-CAL/HK, FarmVibes, DEN-3/DEN-12; evaluate compression, mean/worst SSIM/PSNR, onboard processing/storage and downstream wildfire/cropland/plastic tasks.",
      "evaluation_zh": "五種 custom/public datasets：Planet-CAL/HK、FarmVibes、DEN-3/DEN-12；評估 compression、mean/worst SSIM/PSNR、onboard processing/storage 與 wildfire/cropland/plastic tasks。",
      "baselines_en": "Lanczos interpolation; CS ADMM, gOMP, CoSaMP; DSCN/DCSN as printed, VQ-VAE-2; Kodan, Earth+; WaveDiff and SR3 in result tables.",
      "baselines_zh": "Lanczos interpolation、CS ADMM/gOMP/CoSaMP、原文列 DSCN/DCSN、VQ-VAE-2、Kodan、Earth+、WaveDiff，與 result tables 中的 SR3。",
      "result_en": "Reported mean CR is 146.4–320 depending on dataset. Planet-HK at CR256 yields SSIM0.91 and PSNR37.3dB versus WaveDiff0.80/33.5dB; uplink around 10Kbps.",
      "result_zh": "依 dataset，mean CR 為 146.4–320。Planet-HK 在 CR256 達 SSIM0.91、PSNR37.3dB；WaveDiff 為 0.80/33.5dB；uplink 約10Kbps。",
      "limitations_en": "Cloud reconstruction performs most neural compute. Fidelity evidence measures images and downstream tasks; distributed LLM collectives require separate workloads.",
      "limitations_zh": "cloud reconstruction 承擔主要 neural compute。fidelity 證據涵蓋 images 與 downstream tasks；distributed LLM collectives 需要各自 workloads。",
      "figure_page_1based": 5,
      "figure_bbox_points": [
        315,
        82,
        563,
        225
      ],
      "mechanism_scope": "EO workload / downlink reduction",
      "deployment_regime": "E",
      "space_ground_path": true,
      "display_regime_en": "E/C · Observation and ground delivery",
      "display_regime_zh": "E／C · 觀測與地面交付",
      "primary_pdf_sha256": "4ee367c032c9d5e7dc2fbf39826b98d7c568af5db72f73571abd51c44f60e880",
      "figure_asset": "../figures/deepspace.png",
      "figure_acquisition": "Primary PDF crop",
      "figure_content_type": "image/png",
      "figure_bytes": 56864,
      "figure_sha256": "43b4b3219df492b1aac2e737fc76aba4c235ac592cdcc75bb9b1ff01109e1c40"
    },
    {
      "id": "dissecting-starlink",
      "title": "Dissecting the StarLink: Characterizing Queuing and Flow Dynamics in the Starlink Network",
      "authors": "Hendrik Cech; Nitinder Mohan; Jörg Ott",
      "year": "2026",
      "venue": "SIGCOMM",
      "url": "https://spearlab.nl/papers/2026/sigcomm26-dissect-starlink.pdf",
      "pdf_url": "https://spearlab.nl/papers/2026/sigcomm26-dissect-starlink.pdf",
      "regime": "A · Communication constellation",
      "evidence": [
        "§3 infrastructure; §4.1 queue; §4.2 capacity; §4.4 reset; §4.6 transport; PDF pp.2–12"
      ],
      "figure_number": "1",
      "evidence_status": "Full primary text reviewed",
      "built_for_en": "Transport designers characterize a commercial LEO access bottleneck packet by packet.",
      "built_for_zh": "transport designers 逐 packet 刻畫 commercial LEO access bottleneck。",
      "problem_en": "Macro throughput/RTT statistics obscure bandwidth allocation, queue policy and periodic resets.",
      "problem_zh": "macro throughput/RTT statistics 隱藏 bandwidth allocation、queue policy 與 periodic resets。",
      "design_en": "NetScalpel schedules synchronized UDP burst/rate/cooldown and TCP experiments around 15s cycles; reconstruct queues from paired timestamps and compare CCAs.",
      "design_zh": "NetScalpel 在 15s cycle 安排 synchronized UDP burst/rate/cooldown 與 TCP 實驗，以 paired timestamps 重建 queues 並比較 CCAs。",
      "model_en": "Event-based head-drop queue reconstruction, demand-dependent capacity ramp model, paired Wilcoxon tests and equivalence testing.",
      "model_zh": "event-based head-drop queue reconstruction、demand-dependent capacity ramp model、paired Wilcoxon 與 equivalence testing。",
      "evaluation_en": "Munich residential terminal plus US east/west validation; AWS Frankfurt near PoP; NTP-calibrated microsecond packet records and 5ms TCP_INFO; randomized experiments across reconfiguration intervals.",
      "evaluation_zh": "Munich residential terminal 與 US east/west 驗證；AWS Frankfurt 鄰近 PoP；NTP 校準 microsecond packet records 與 5ms TCP_INFO；跨 reconfiguration intervals randomize experiments。",
      "baselines_en": "Head-drop versus tail-drop models; RED/CoDel conceptual checks; BBRv1/v3, CUBIC, SatPipe and slow-start variants HyStart, HyStart++, SEARCH, SUSS.",
      "baselines_zh": "head-drop/tail-drop models；RED/CoDel 機制比較；BBRv1/v3、CUBIC、SatPipe，及 HyStart、HyStart++、SEARCH、SUSS slow-start variants。",
      "result_en": "Estimated queues ~1,500 DL/~4,000 UL packets; demand ramps 100/30Mbps baselines by 3.4×/2× in 400ms; bandwidth resets every 15s; flow delay isolation coexists with coupled DL loss.",
      "result_zh": "estimated queues 約 DL1,500、UL4,000 packets；demand 使100/30Mbps baseline 在400ms增加3.4×/2×；bandwidth 每15s reset；flow delay isolation 與 coupled DL loss 並存。",
      "limitations_en": "Observations describe Starlink implementation and measured terminals. Inter-satellite paths and internal operator state require further observability; demand-driven replay should preserve traffic dependence.",
      "limitations_zh": "觀察描述 Starlink implementation 與量測 terminals。inter-satellite path 和 operator state 需要更多 observability；demand-driven replay 應保留 traffic dependence。",
      "figure_page_1based": 3,
      "figure_bbox_points": [
        53,
        82,
        295,
        328
      ],
      "mechanism_scope": "Transport / queue measurement",
      "deployment_regime": "A",
      "space_ground_path": true,
      "display_regime_en": "A · Communication constellation",
      "display_regime_zh": "A · 通訊星座",
      "primary_pdf_sha256": "d870b8046911a31aef4f9a7ef14437681abf8dfb2540c93c3849392cb18d41b0",
      "figure_asset": "../figures/dissecting-starlink.png",
      "figure_acquisition": "Primary PDF crop",
      "figure_content_type": "image/png",
      "figure_bytes": 93443,
      "figure_sha256": "33e2b2b7eeef18c08b7843bc94d5fd2b13262f8cb8277cb0ac46e1995a314625"
    },
    {
      "id": "commsar",
      "title": "CommSAR: Enabling Bidirectional Communication in SAR Imaging Satellites via Shared Waveform",
      "authors": "Yimin Zhao; Zhenning Li; Hao Pan; Hua Li; Minhao Cui; Jie Xiong; Yihai Wei; Yang Liu; Haonan Zhao; Mohan Zhang; Weibo Wang; Guihai Chen; Kaiyu Liu; Linghe Kong",
      "year": "2026",
      "venue": "SIGCOMM",
      "url": "https://ym-zhao.com/publications/zhao2026commsar/commsar-sigcomm26.pdf",
      "pdf_url": "https://ym-zhao.com/publications/zhao2026commsar/commsar-sigcomm26.pdf",
      "regime": "A · Communication constellation",
      "evidence": [
        "§3 waveform; §4 prototype; §5.1 experimental split; §5.2–5.5; PDF pp.8–12"
      ],
      "figure_number": "1",
      "evidence_status": "Full primary text reviewed",
      "built_for_en": "SAR satellites exchange control data through the imaging waveform and reflected echoes.",
      "built_for_zh": "SAR satellites 藉 imaging waveform 與 reflected echo 交換 control data。",
      "problem_en": "Dedicated communication hardware, spectrum and power impose constellation-scale overhead.",
      "problem_zh": "專用 communication hardware、spectrum 與 power 增加 constellation-scale 成本。",
      "design_en": "Encode downlink data by chirp-start frequency offsets; programmable ground metasurface uses differential phase modulation for uplink; opposite-slope pilots compensate mobility.",
      "design_zh": "downlink 以 chirp-start frequency offset 編碼；ground programmable metasurface 以 differential phase modulation 建立 uplink；opposite-slope pilots 補償 mobility。",
      "model_en": "Linear-frequency-modulated signal equations, matched filtering and radar cross section, Doppler estimation, differential BPSK, BER versus SNR.",
      "model_zh": "linear-frequency-modulated signal equations、matched filtering/radar cross section、Doppler estimation、differential BPSK，以及 BER–SNR。",
      "evaluation_en": "Actual commercial satellite supplies uplink IQ echoes and downlink channel traces; proposed downlink waveform is tested with FPGA channel replay in anechoic chamber; UAV interleaved imaging A/B isolates waveform effects.",
      "evaluation_zh": "actual commercial satellite 提供 uplink IQ echo 與 downlink channel traces；proposed downlink waveform 由 FPGA channel replay 在 anechoic chamber 測試；UAV interleaved imaging A/B 隔離 waveform effects。",
      "baselines_en": "Standard SAR LFM imaging, conventional Doppler compensation, SARLink uplink as related-rate reference; per-component ablations.",
      "baselines_zh": "standard SAR LFM imaging、conventional Doppler compensation、SARLink uplink rate reference，以及 per-component ablations。",
      "result_en": "Maximum modeled/tested waveform rates 105Kbps DL and 112Kbps UL; imaging evidence uses resolution, PSLR and ISLR. Final PDF112Kbps governs the author-page 120Kbps discrepancy.",
      "result_zh": "waveform 最大設計/測試 rate：DL105Kbps、UL112Kbps；imaging 以 resolution、PSLR、ISLR 驗證。final PDF112Kbps 提供 author page120Kbps 的版本裁決依據。",
      "limitations_en": "Each component has its own flight/testbed tier. Sparse TT&C-like rates serve control traffic; orbital AI fabric needs a separate high-bandwidth optical design.",
      "limitations_zh": "各 component 對應各自 flight/testbed tier。TT&C 型 rate 服務 control traffic；orbital AI fabric 需要另行高 bandwidth optical design。",
      "figure_page_1based": 1,
      "figure_bbox_points": [
        316,
        188,
        564,
        340
      ],
      "mechanism_scope": "Physical link / shared sensing",
      "deployment_regime": "A",
      "space_ground_path": false,
      "display_regime_en": "A · Communication constellation",
      "display_regime_zh": "A · 通訊星座",
      "primary_pdf_sha256": "1ef95dcb89e72340bfaac984eb50f3f8f26a4bb56fd5854e5bad92093b6baf95",
      "figure_asset": "../figures/commsar.png",
      "figure_acquisition": "Primary PDF crop",
      "figure_content_type": "image/png",
      "figure_bytes": 662261,
      "figure_sha256": "e5f7d3f95e9e529e1cfe4817f0bdc57245d58a4495fff623fd12cc6682e63602"
    },
    {
      "id": "planet-iot",
      "title": "Planet-Scale IoT Connectivity via LEO Satellites",
      "authors": "Ziyue Zhang; Xianjin Xia; Ruonan Li; Jinhong Liu; Yuanqing Zheng; Linghe Kong; Mo Li",
      "year": "2026",
      "venue": "SIGCOMM",
      "url": "https://www4.comp.polyu.edu.hk/~csyqzheng/papers/sigcomm26-DtS.pdf",
      "pdf_url": "https://www4.comp.polyu.edu.hk/~csyqzheng/papers/sigcomm26-DtS.pdf",
      "regime": "A · Communication constellation",
      "evidence": [
        "§3 measurement; §4.2 design; §4.3 deployment; Appendix E capacity and G optimization; PDF pp.11–14"
      ],
      "figure_number": "17",
      "evidence_status": "Full primary text reviewed",
      "built_for_en": "Battery-powered ground IoT nodes send sparse sensor reports and bursty images through intermittent LEO contacts.",
      "built_for_zh": "battery-powered ground IoT nodes 透過 intermittent LEO contact 傳送 sensor report 與 bursty images。",
      "problem_en": "Retransmission overhead, unfavorable contact slots and prolonged radio monitoring inflate delay and energy.",
      "problem_zh": "retransmission overhead、低品質 contact slot 與長時間 radio monitoring 增加 delay/energy。",
      "design_en": "Coded NACK pipelines transmission and repeats compact missing-packet IDs; DtS-FC forecasts contact/link quality, defers costly slots and schedules sleep.",
      "design_zh": "coded NACK pipeline transmission 並重複 compact missing-packet IDs；DtS-FC 預測 contact/link quality，移轉高成本 slot 並安排 sleep。",
      "model_en": "Contact-window prediction; reliability deadline/repetition optimization for NACK; lookahead F-pass energy/latency flow-control optimization.",
      "model_zh": "contact-window prediction；NACK reliability deadline/repetition optimization；F-pass lookahead energy/latency flow-control optimization。",
      "evaluation_en": "Six-satellite X-SNO measurements; coded NACK tested on three nodes and production-equivalent ground satellite radios plus channel emulator; DtS-FC deployed on eight nodes for one month in Yunnan.",
      "evaluation_zh": "six-satellite X-SNO measurements；coded NACK 在 three nodes、production-equivalent ground satellite radio 與 channel emulator 測試；DtS-FC 在 Yunnan eight nodes 部署一個月。",
      "baselines_en": "Current operational SatIoT protocol; separate coded-NACK and DtS-FC evaluations.",
      "baselines_zh": "current operational SatIoT protocol；coded-NACK 與 DtS-FC 分開評估。",
      "result_en": "Ground-test coded NACK improves throughput 2.1× and reduces transmissions 75.1–83.3%; live DtS-FC cuts >20min packet fraction 23%→11% and energy/bit38%.",
      "result_zh": "ground-test coded NACK 提升 throughput2.1×、減少 transmissions75.1–83.3%；live DtS-FC 將 >20min packet fraction 從23%降至11%，energy/bit降低38%。",
      "limitations_en": "Production deployment validates ground-side DtS-FC; satellite-side NACK integration follows ground-test evidence. 200–300 concurrent-node capacity extrapolates from eight-node tests and simulation.",
      "limitations_zh": "production deployment 驗證 ground-side DtS-FC；satellite-side NACK integration 以 ground-test 證據銜接。200–300 concurrent-node capacity 由 eight-node tests 與 simulation 外推。",
      "figure_page_1based": 11,
      "figure_bbox_points": [
        54,
        82,
        296,
        207
      ],
      "mechanism_scope": "IoT transport / contact scheduling",
      "deployment_regime": "A",
      "space_ground_path": true,
      "display_regime_en": "A/C · Contact-limited IoT service",
      "display_regime_zh": "A／C · Contact-limited IoT 服務",
      "primary_pdf_sha256": "845de2ce8633926b6e186f7c3173216a5814833cf87318be28c0960df6d044fc",
      "figure_asset": "../figures/planet-iot.png",
      "figure_acquisition": "Primary PDF crop",
      "figure_content_type": "image/png",
      "figure_bytes": 97316,
      "figure_sha256": "6b0e2229758d04e9019c864258d96cf1b0ca766272e0d44d34471100685f9a29"
    },
    {
      "id": "loon",
      "title": "SDN in the Stratosphere: Loon’s Aerospace Mesh Network",
      "authors": "Frank Uyeda; Marc Alvidrez; Erik Kline; Bryce Petrini; Brian Barritt; David Mandle; Aswin Chandy Alexander",
      "year": "2022",
      "venue": "SIGCOMM",
      "url": "https://storage.googleapis.com/gweb-research2023-media/pubtools/6737.pdf",
      "pdf_url": "https://storage.googleapis.com/gweb-research2023-media/pubtools/6737.pdf",
      "regime": "X · Aerial adjacency",
      "evidence": [
        "§2 system; §3 predictive failure response; §4 control tiers; PDF pp.2–9"
      ],
      "figure_number": "5",
      "evidence_status": "Full primary text reviewed",
      "built_for_en": "Moving balloon base stations backhaul LTE service through steerable mesh links.",
      "built_for_zh": "移動 balloon base stations 透過 steerable mesh links 提供 LTE backhaul。",
      "problem_en": "Motion and atmospheric conditions alter link feasibility while control-channel reliability varies.",
      "problem_zh": "motion 與 atmospheric conditions 改變 link feasibility，control-channel reliability 也隨時間變動。",
      "design_en": "Temporospatial SDN predicts radio feasibility and compiles topology/routing intents; local tracking, redundant links and hybrid satcom/in-band control support reactive recovery.",
      "design_zh": "temporospatial SDN 預測 radio feasibility 並編譯 topology/routing intent；local tracking、redundant links 與 hybrid satcom/in-band control 支援 reactive recovery。",
      "model_en": "Physical link-budget models, constrained topology/routing solver, predictive control and operational distributions of failure/recovery.",
      "model_zh": "physical link-budget models、constrained topology/routing solver、predictive control 與 operational failure/recovery distributions。",
      "evaluation_en": "Three years of production operation across three continents; meshes routinely 20+ balloons over 3000+km; comparison of planned withdrawals with unexpected failures.",
      "evaluation_zh": "three years production operation、three continents；mesh 經常20+ balloons、跨度3000+km；比較 planned withdrawal 與 unexpected failure。",
      "baselines_en": "Operational planned link withdrawals versus failed links; architecture and transceiver-count variants.",
      "baselines_zh": "operational planned link withdrawals 與 failed links；architecture/transceiver-count variants。",
      "result_en": "Among routes recovered within 5min, anticipated failures recover 37.8% faster on average;75% control-plane recoveries complete within 20s and 92.4% reuse existing links.",
      "result_zh": "在5min內恢復的 routes 中，anticipated failure 平均快37.8%；75% control-plane recovery 在20s內完成；92.4%使用既有 links。",
      "limitations_en": "Stratospheric balloons form an adjacent NTN regime with wind-driven trajectories and atmosphere-dependent radio links. Orbital satellites require orbital/optical-specific parameterization.",
      "limitations_zh": "stratospheric balloon 屬 adjacent NTN regime，trajectory 受 wind 驅動、radio link 受 atmosphere 影響。orbital satellite 需 orbital/optical 專用參數。",
      "figure_page_1based": 5,
      "figure_bbox_points": [
        53,
        82,
        296,
        215
      ],
      "mechanism_scope": "Adjacent aerospace NTN / predictive control",
      "deployment_regime": "X",
      "space_ground_path": false,
      "display_regime_en": "X · Stratospheric adjacency",
      "display_regime_zh": "X · 平流層相鄰系統",
      "primary_pdf_sha256": "83fe33c046a1b15deabf54eba168a7d85eef0ec7c85e6d6eef6d80e7924ff7a7",
      "figure_asset": "../figures/loon.png",
      "figure_acquisition": "Primary PDF crop",
      "figure_content_type": "image/png",
      "figure_bytes": 57813,
      "figure_sha256": "e94a5faf8a26d56b23d0880bdb937042259c10a4835123eb36a35ad1c305dd7f"
    },
    {
      "id": "serval",
      "title": "Known Knowns and Unknowns: Near-realtime Earth Observation Via Query Bifurcation in Serval",
      "authors": "Bill Tao; Om Chabra; Ishani Janveja; Indranil Gupta; Deepak Vasisht",
      "year": "2024",
      "venue": "NSDI",
      "url": "https://www.usenix.org/system/files/nsdi24-tao.pdf",
      "pdf_url": "https://www.usenix.org/system/files/nsdi24-tao.pdf",
      "regime": "E · Observation/contact fleet",
      "evidence": [
        "§3 query bifurcation; §4 implementation; §5 evaluation, Table 1–2 and Fig.6–11; PDF pp.7–11"
      ],
      "figure_number": "4",
      "figure_page_1based": 7,
      "evidence_status": "Full primary text reviewed",
      "built_for_en": "Nearly 200 Planet Dove imaging satellites answer prioritized Earth-observation queries.",
      "built_for_zh": "接近 200 顆 Planet Dove 成像衛星迴答具優先級的 Earth-observation queries。",
      "problem_en": "Downlink queues place urgent imagery behind large background collections; onboard compute and energy budgets constrain image filtering.",
      "problem_zh": "downlink queue 將急迫影像排在大量背景資料之後；onboard compute 與 energy budget 限制 image filtering。",
      "design_en": "Bifurcate queries into slowly changing ground-precomputed predicates and dynamic onboard predicates; schedule compute and downlink by query priority.",
      "design_zh": "將 query 拆成由地面預先計算的緩慢變動 predicates，以及衛星執行的動態 predicates；依 query priority 安排 compute 與 downlink。",
      "model_en": "Boolean query composition, spatial intersections, contact schedules, compute queues and solar/battery energy accounting.",
      "model_zh": "Boolean query composition、空間交集、contact schedules、compute queues 與 solar/battery 能量帳。",
      "evaluation_en": "Planet metadata for ten million images over July 1–20, 2021; Jetson AGX Orin profiling at 15 W and 30 W; orbit/contact simulation with traditional and distributed ground stations, keeping aggregate downlink constant.",
      "evaluation_zh": "使用 2021 年 7 月 1–20 日一千萬張 Planet 影像 metadata；在 15 W 與 30 W 下量測 Jetson AGX Orin；以軌道/contact simulation 比較傳統與分散 ground stations，維持 aggregate downlink。",
      "baselines_en": "In-order delivery, in-order delivery with distributed ground stations; component ablations of compute, weather prediction and historical forest labels; OEC-inspired filtering comparison.",
      "baselines_zh": "in-order delivery、搭配 distributed ground stations 的 in-order delivery；compute、weather prediction、historical forest labels 的 component ablations；OEC-inspired filtering comparison。",
      "result_en": "At 15 W, traditional-station median latency drops from 78.2 h to 1.1 h; distributed-station median drops from 71.71 h to 0.03 h. The 47-minute P90 headline uses the distributed-station condition.",
      "result_zh": "15 W 時，傳統 ground-station median latency 從 78.2 h 降至 1.1 h；分散 ground-station median 從 71.71 h 降至 0.03 h。P90 47 分鐘的 headline 對應分散 ground-station 條件。",
      "limitations_en": "Results describe prioritized imagery and metadata predicates; cluster-scale LLM collectives require accelerator and fabric measurements. Distributed ground stations contribute separately to latency gains.",
      "limitations_zh": "結果涵蓋 priority imagery 與 metadata predicates；cluster-scale LLM collectives 需要 accelerator 與 fabric measurements。distributed ground stations 對 latency gains 提供獨立貢獻。",
      "figure_bbox_points": [
        53,
        58,
        297,
        211
      ],
      "mechanism_scope": "Earth-observation compute / priority scheduling",
      "deployment_regime": "E",
      "space_ground_path": true,
      "display_regime_en": "E/C · Observation and ground delivery",
      "display_regime_zh": "E／C · 觀測與地面交付",
      "primary_pdf_sha256": "1985416fb78cf18d0bd481b053809347219ce9510d4e31af265ed46fc721a258",
      "figure_asset": "../figures/serval.png",
      "figure_acquisition": "Primary PDF crop",
      "figure_content_type": "image/png",
      "figure_bytes": 69283,
      "figure_sha256": "de700f37461e82d9e775d974e083392dbb21c8ffb19219ef20f2b22c5d52888c"
    },
    {
      "id": "starrynet",
      "title": "StarryNet: Empowering Researchers to Evaluate Futuristic Integrated Space and Terrestrial Networks",
      "authors": "Zeqi Lai; Hewu Li; Yangtao Deng; Qian Wu; Jun Liu; Yuanjie Li; Jihao Li; Lixin Liu; Weisen Liu; Jianping Wu",
      "year": "2023",
      "venue": "NSDI",
      "url": "https://www.usenix.org/system/files/nsdi23-lai-zeqi.pdf",
      "pdf_url": "https://www.usenix.org/system/files/nsdi23-lai-zeqi.pdf",
      "regime": "A · Communication constellation",
      "evidence": [
        "§3–5 architecture; §6 fidelity, scalability and applications, PDF pp.8–13"
      ],
      "figure_number": "1",
      "figure_page_1based": 5,
      "evidence_status": "Full primary text reviewed",
      "built_for_en": "Containerized satellite and terrestrial network software runs against time-evolving constellation links.",
      "built_for_zh": "containerized 衛星與地面 network software 在隨時間演變的星座 links 上執行。",
      "problem_en": "Thousands of mobile nodes require synchronized geometry, network-state updates and realistic software execution on terrestrial hosts.",
      "problem_zh": "數千個 mobile nodes 需要同步 geometry、network-state updates 與地面 host 上的實際 software execution。",
      "design_en": "Combine public orbital information, physical-to-virtual mapping, Linux containers, traffic-control link updates and multi-host manager/worker orchestration.",
      "design_zh": "結合公開軌道資料、physical-to-virtual mapping、Linux containers、traffic-control link updates 與 multi-host manager/worker orchestration。",
      "model_en": "Orbit propagation, geometric link feasibility, time-dependent graph mapping, resource capping and distributed event synchronization.",
      "model_zh": "軌道推算、幾何 link feasibility、time-dependent graph mapping、resource capping 與 distributed event synchronization。",
      "evaluation_en": "Eight Dell PowerEdge R740 servers; Starlink, Kuiper and Telesat configurations; ping/iperf validation against live European Starlink traces and CoreMark checks for virtual compute capacity.",
      "evaluation_zh": "八臺 Dell PowerEdge R740；Starlink、Kuiper、Telesat configurations；以歐洲 live Starlink traces 的 ping/iperf 驗證，並用 CoreMark 檢查 virtual compute capacity。",
      "baselines_en": "Hypatia, StarPerf, live Starlink measurements and physical-device CoreMark reference.",
      "baselines_zh": "Hypatia、StarPerf、live Starlink measurements 與 physical-device CoreMark reference。",
      "result_en": "A 4,408-satellite configuration initializes in 21.2 minutes using seven workers; 1-second updates consume 39.6% host CPU in the reported setup.",
      "result_zh": "4,408-satellite configuration 使用七個 workers 在 21.2 分鐘完成初始化；報告 setup 的 1-second updates 消耗 39.6% host CPU。",
      "limitations_en": "Fidelity follows the configured orbit, topology, link model and available traces. Commercial scheduler inference and future optical-terminal behavior call for explicit calibration.",
      "limitations_zh": "fidelity 隨 configured orbit、topology、link model 與 available traces 決定。commercial scheduler inference 與 future optical-terminal behavior 需要明確 calibration。",
      "figure_bbox_points": [
        315,
        82,
        560,
        266
      ],
      "mechanism_scope": "Constellation network emulation",
      "deployment_regime": "A",
      "space_ground_path": false,
      "display_regime_en": "A · Communication constellation",
      "display_regime_zh": "A · 通訊星座",
      "primary_pdf_sha256": "620afe3d7bd9bd76009cb2e29e19281d42442e90a2e223d0c5e57de6ebcf709b",
      "figure_asset": "../figures/starrynet.png",
      "figure_acquisition": "Primary PDF crop",
      "figure_content_type": "image/png",
      "figure_bytes": 196062,
      "figure_sha256": "d7d7fb1b24e2bb8d4d7b194a8c28a7b3d02d2a1a236f88922a3030b7a2169d11"
    },
    {
      "id": "hypatia",
      "title": "Exploring the “Internet from space” with Hypatia",
      "authors": "Simon Kassing; Debopam Bhattacherjee; André Baptista Águas; Jens Eirik Saethre; Ankit Singla",
      "year": "2020",
      "venue": "IMC",
      "url": "https://bdebopam.github.io/papers/imc2020-hypatia.pdf",
      "pdf_url": "https://bdebopam.github.io/papers/imc2020-hypatia.pdf",
      "regime": "A · Communication constellation",
      "evidence": [
        "§3 framework; §4 architecture, fidelity and runtime; §5 TCP/UDP investigations, PDF pp.4–13"
      ],
      "figure_number": "2",
      "figure_page_1based": 5,
      "evidence_status": "Full primary text reviewed",
      "built_for_en": "Proposed Starlink, Kuiper and Telesat constellations carry packet-level Internet traffic.",
      "built_for_zh": "規劃中的 Starlink、Kuiper 與 Telesat 星座承載 packet-level Internet traffic。",
      "problem_en": "Orbital motion changes path delay, routing and link load while TCP reacts to the evolving packet sequence and queues.",
      "problem_zh": "orbital motion 改變 path delay、routing 與 link load，同時 TCP 對變動中的 packet sequence 與 queues 作出反應。",
      "design_en": "Precompute orbital and forwarding snapshots, feed them into ns-3 packet simulation, and visualize trajectories through Cesium.",
      "design_zh": "預先計算 orbital 與 forwarding snapshots，送入 ns-3 packet simulation，並以 Cesium 視覺化 trajectories。",
      "model_en": "Time-varying +Grid graph, shortest paths, continuous geometric propagation delay, discrete-event queues and transport state.",
      "model_zh": "time-varying +Grid graph、shortest paths、continuous geometric propagation delay、discrete-event queues 與 transport state。",
      "evaluation_en": "FCC/ITU planned constellation parameters; validation against NetworkX Floyd–Warshall paths and ns-3 ping; TCP CUBIC and UDP workloads; runtime scaling across traffic rates.",
      "evaluation_zh": "FCC/ITU 規劃星座參數；以 NetworkX Floyd–Warshall paths 與 ns-3 ping 驗證；TCP CUBIC、UDP workloads；比較 traffic rate 對 runtime scaling 的影響。",
      "baselines_en": "Analytical shortest-path distance, ns-3 ping consistency checks, alternative ground-station and routing configurations, TCP/UDP flow scenarios.",
      "baselines_zh": "analytical shortest-path distance、ns-3 ping consistency checks、替代 ground-station/routing configurations、TCP/UDP flow scenarios。",
      "result_en": "On one 2.26 GHz Xeon L5520 core, a 10 Gbps flow over 10 simulated seconds takes approximately 33 minutes for UDP and 100 minutes for TCP. Packet experiments expose motion-induced RTT changes and reordering.",
      "result_zh": "在單顆 2.26 GHz Xeon L5520 core 上，10 Gbps flow 的 10 simulated seconds 約需 UDP 33 分鐘、TCP 100 分鐘。packet experiments 揭示 motion-induced RTT changes 與 reordering。",
      "limitations_en": "The model represents specified planned topology and forwarding policies. Runtime cost grows with packet count; RF scheduling, live commercial TE and terminal dynamics require dedicated models.",
      "limitations_zh": "模型代表指定的 planned topology 與 forwarding policies。runtime cost 隨 packet count 增長；RF scheduling、live commercial TE 與 terminal dynamics 需要專用 models。",
      "figure_bbox_points": [
        313,
        78,
        564,
        213.5
      ],
      "mechanism_scope": "Packet-level simulation foundation",
      "deployment_regime": "A",
      "space_ground_path": false,
      "display_regime_en": "A · Communication constellation",
      "display_regime_zh": "A · 通訊星座",
      "primary_pdf_sha256": "ae54c1edac2cdf742f60b6f93b0fcaaf747cacd691227582e875ee0d2e049064",
      "figure_asset": "../figures/hypatia.png",
      "figure_acquisition": "Primary PDF crop",
      "figure_content_type": "image/png",
      "figure_bytes": 36104,
      "figure_sha256": "aaa9592bb6e8af1743a8b3a84ae9536d831eb70affbdb2938037ebc09ecfee54"
    },
    {
      "id": "motifs",
      "title": "Network topology design at 27,000 km/hour",
      "authors": "Debopam Bhattacherjee; Ankit Singla",
      "year": "2019",
      "venue": "CoNEXT",
      "url": "https://cspeedweb.web.engr.illinois.edu/assets/publications/bhattacherjee-conext2019.pdf",
      "pdf_url": "https://cspeedweb.web.engr.illinois.edu/assets/publications/bhattacherjee-conext2019.pdf",
      "regime": "A · Communication constellation",
      "evidence": [
        "§3 ILP; §4 motifs; §5 latitude-dependent motifs; §6–7 evaluation, Table 1 and Fig.7–11"
      ],
      "figure_number": "6",
      "figure_page_1based": 6,
      "evidence_status": "Full primary text reviewed",
      "built_for_en": "Dense LEO constellations provide low-latency high-capacity inter-satellite paths.",
      "built_for_zh": "密集 LEO constellations 提供低延遲、高容量的 inter-satellite paths。",
      "problem_en": "Bounded optical-terminal degree, link range and moving geometry create a combinatorial topology-design problem.",
      "problem_zh": "有限 optical-terminal degree、link range 與 moving geometry 形成 combinatorial topology-design problem。",
      "design_en": "Enumerate repeating local connectivity motifs, then customize motifs across latitude zones to preserve useful links through orbital motion.",
      "design_zh": "列舉 repeating local connectivity motifs，再依 latitude zones 調整 motifs，讓 orbital motion 期間維持有用 links。",
      "model_en": "Mixed integer linear programming; degree and visibility constraints; weighted objective Mα=α·stretch+hop count; symmetry reduction and exhaustive motif search.",
      "model_zh": "mixed integer linear programming、degree/visibility constraints；weighted objective Mα=α·stretch+hop count；symmetry reduction 與 exhaustive motif search。",
      "evaluation_en": "Population-weighted traffic among 1,000 cities, plus GDP-weighted traffic among 100 cities; inclined 40×40 constellation and planned Starlink/Kuiper; sweeps of laser range and acquisition time.",
      "evaluation_zh": "1,000 城市的 population-weighted traffic，加上 100 城市的 GDP-weighted traffic；inclined 40×40 constellation 與 planned Starlink/Kuiper；掃描 laser range 與 acquisition time。",
      "baselines_en": "+Grid neighbor connectivity; ILP on small city sets; uniform motif and latitude-dependent multi-motif variants.",
      "baselines_zh": "+Grid neighbor connectivity；small city sets 的 ILP；uniform motif 與 latitude-dependent multi-motif variants。",
      "result_en": "Multi-motif designs improve the weighted hop/stretch objective by up to 54% for Starlink and 45% for Kuiper relative to +Grid. These percentages describe the paper’s network-efficiency objective.",
      "result_zh": "相較 +Grid，multi-motif designs 對 weighted hop/stretch objective 的改進最高達 Starlink 54%、Kuiper 45%。這些比例對應論文的 network-efficiency objective。",
      "limitations_en": "Hop count serves as a capacity proxy; throughput and collective completion require explicit capacities, queues and workload placement. Motif symmetry assumes regular constellation structure.",
      "limitations_zh": "hop count 作為 capacity proxy；throughput 與 collective completion 需要明確 capacities、queues 與 workload placement。motif symmetry 假設規則星座結構。",
      "figure_bbox_points": [
        311,
        82,
        610,
        238
      ],
      "mechanism_scope": "Optical ISL topology design foundation",
      "deployment_regime": "A",
      "space_ground_path": false,
      "display_regime_en": "A · Communication constellation",
      "display_regime_zh": "A · 通訊星座",
      "primary_pdf_sha256": "b78c5739f4ab9028d6d30c1c0458b8240b38196498c28cb5279449ddab8c8eb7",
      "figure_asset": "../figures/motifs.png",
      "figure_acquisition": "Primary PDF crop",
      "figure_content_type": "image/png",
      "figure_bytes": 68604,
      "figure_sha256": "dd552006e81059588bf69017e641a117045df660ae7960aa406051884b0f2093"
    },
    {
      "id": "starperf",
      "title": "StarPerf: Characterizing Network Performance for Emerging Mega-Constellations",
      "authors": "Zeqi Lai; Hewu Li; Jihao Li",
      "year": "2020",
      "venue": "ICNP",
      "url": "https://icnp20.cs.ucr.edu/proceedings/main/StarPerf.pdf",
      "pdf_url": "https://icnp20.cs.ucr.edu/proceedings/main/StarPerf.pdf",
      "regime": "A · Communication constellation",
      "evidence": [
        "§III simulator; §IV constellation comparison; §V satellite/cloud relay; Fig.2–3, Fig.10–11"
      ],
      "figure_number": "2",
      "figure_page_1based": 3,
      "evidence_status": "Full primary text reviewed",
      "built_for_en": "Area-to-area satellite connectivity and hybrid cloud/satellite interactive communication.",
      "built_for_zh": "area-to-area 衛星 connectivity，以及 hybrid cloud/satellite interactive communication。",
      "problem_en": "Constellation scale, architectural options and path choice jointly determine attainable performance.",
      "problem_zh": "constellation scale、architectural options 與 path choice 共同決定 attainable performance。",
      "design_en": "Build orbital area-to-area performance models and constellation scaling; choose low-latency satellite or cloud relays using measured historical path information.",
      "design_zh": "建立 orbital area-to-area performance models 與 constellation scaling；以 measured historical path information 選擇 low-latency satellite/cloud relay。",
      "model_en": "Geometric constellation graph, hexagonal Earth regions, path latency/capacity profiling and average-latency relay selection.",
      "model_zh": "geometric constellation graph、hexagonal Earth regions、path latency/capacity profiling 與 average-latency relay selection。",
      "evaluation_en": "Constellation simulation plus laptops running WebRTC; traffic-control reproduces model-predicted satellite delay while cloud routes use measured terrestrial delays.",
      "evaluation_zh": "constellation simulation 搭配執行 WebRTC 的 laptops；traffic-control 重現模型預測的 satellite delay，cloud routes 使用 measured terrestrial delays。",
      "baselines_en": "Terrestrial cloud relay, satellite relay and hybrid relay-selection configurations; constellation architecture variants.",
      "baselines_zh": "terrestrial cloud relay、satellite relay、hybrid relay-selection configurations；constellation architecture variants。",
      "result_en": "The hybrid relay prototype reduces end-to-end interactive latency by up to 62% under the evaluated long-distance communication cases.",
      "result_zh": "hybrid relay prototype 在評估的 long-distance communication cases 下，end-to-end interactive latency 最高降低 62%。",
      "limitations_en": "Prototype satellite paths inherit simulated delay. Packet-level congestion and optical acquisition require additional calibration.",
      "limitations_zh": "prototype satellite paths 沿用 simulated delay。packet-level congestion 與 optical acquisition 需要額外 calibration。",
      "figure_bbox_points": [
        322,
        47,
        555,
        171
      ],
      "mechanism_scope": "Constellation performance simulation foundation",
      "deployment_regime": "A",
      "space_ground_path": false,
      "display_regime_en": "A · Communication constellation",
      "display_regime_zh": "A · 通訊星座",
      "primary_pdf_sha256": "f7a59c741b6ac4900bbbd820d05562221394c6f71a77677f08d959acf3ba1c60",
      "figure_asset": "../figures/starperf.png",
      "figure_acquisition": "Primary PDF crop",
      "figure_content_type": "image/png",
      "figure_bytes": 43090,
      "figure_sha256": "d6d554c8115b7fd0d3da9282696f15fe74aa03b4f1f45a009db513cf33f5d1d6"
    },
    {
      "id": "mobility-measurement",
      "title": "Deciphering the Enigma of Satellite Computing with COTS Devices: Measurement and Analysis",
      "authors": "Ruolin Xing; Mengwei Xu; Ao Zhou; Qing Li; Yiran Zhang; Feng Qian; Shangguang Wang",
      "year": "2024",
      "venue": "MobiCom",
      "url": "https://feng-qian.github.io/paper/sat_mobicom24.pdf",
      "pdf_url": "https://feng-qian.github.io/paper/sat_mobicom24.pdf",
      "regime": "E · Observation/contact fleet",
      "evidence": [
        "§2 setup and Table 1; §3 thermal; §4 energy; §5 performance, PDF pp.3–13"
      ],
      "figure_number": "1",
      "figure_page_1based": 3,
      "evidence_status": "Full primary text reviewed",
      "built_for_en": "BUPT-1 carries commodity edge computers in approximately 490 km sun-synchronous orbit.",
      "built_for_zh": "BUPT-1 在約 490 km sun-synchronous orbit 承載 commodity edge computers。",
      "problem_en": "Thermal coupling, sunlight-dependent energy and device execution interact under satellite operating conditions.",
      "problem_zh": "thermal coupling、sunlight-dependent energy 與 device execution 在 satellite operating conditions 下互相影響。",
      "design_en": "Instrument live devices and a matched terrestrial counterpart; compare controlled CPU/accelerator loads, temperature, battery and performance traces.",
      "design_zh": "量測 live devices 與匹配的 terrestrial counterpart；比較 controlled CPU/accelerator loads、temperature、battery 與 performance traces。",
      "model_en": "Energy accounting, orbit-aligned sunlight profiles, thermal-response analysis and matched workload microbenchmarks.",
      "model_zh": "energy accounting、orbit-aligned sunlight profiles、thermal-response analysis 與 matched workload microbenchmarks。",
      "evaluation_en": "Six months, over 1,000 experiment hours and 10 million telemetry lines; a 17.44 kg spacecraft with two Huawei Atlas 200 DK boards, two Raspberry Pi boards and two 115 Wh batteries.",
      "evaluation_zh": "六個月、超過 1,000 experiment hours 與一千萬 telemetry lines；17.44 kg spacecraft 配置兩臺 Huawei Atlas 200 DK、兩臺 Raspberry Pi 與兩顆 115 Wh batteries。",
      "baselines_en": "Ground counterpart with matching compute devices, thermal construction and workloads; load/thread/frequency and battery-depth sweeps.",
      "baselines_zh": "ground counterpart 使用匹配的 compute devices、thermal construction 與 workloads；掃描 load/thread/frequency 與 battery depth。",
      "result_en": "Thermal throttling yields up to 10% performance reduction in the measured devices; sustained 9 W operation over 10 hours creates temperatures above 30°C and instability in the reported configuration.",
      "result_zh": "量測裝置的 thermal throttling 造成最高 10% performance reduction；報告 configuration 中持續 9 W 執行 10 小時，使 temperature 超過 30°C 並產生 instability。",
      "limitations_en": "The six-month experiment grounds edge-device energy and thermal models. GPU-rich orbital clusters require device-specific radiation, radiator and sustained-power evidence.",
      "limitations_zh": "六個月實驗為 edge-device energy 與 thermal models 提供實證。GPU-rich orbital clusters 需要 device-specific radiation、radiator 與 sustained-power evidence。",
      "figure_bbox_points": [
        316,
        80,
        560,
        218
      ],
      "mechanism_scope": "Live in-orbit COTS compute measurement",
      "deployment_regime": "E",
      "space_ground_path": true,
      "display_regime_en": "D/E · Onboard hardware evidence",
      "display_regime_zh": "D／E · 星上硬體證據",
      "primary_pdf_sha256": "cafde4dbd740204c29c32dfbeeed18be5dcb8f03e98a9d6ea18e599897234ab3",
      "figure_asset": "../figures/mobility-measurement.png",
      "figure_acquisition": "Primary PDF crop",
      "figure_content_type": "image/png",
      "figure_bytes": 255157,
      "figure_sha256": "3165eeec77611baf335ee6ba63d5919db55e719292a05e4ddb5d6905f17d0c40"
    },
    {
      "id": "mosaic",
      "title": "Democratizing Direct-to-Cell Low Earth Orbit Satellite Networks",
      "authors": "Lixin Liu; Yuanjie Li; Hewu Li; Jiabo Yang; Wei Liu; Jingyi Lan; Yufeng Wang; Jiarui Li; Jianping Wu; Qian Wu; Jun Liu; Zeqi Lai",
      "year": "2024",
      "venue": "NSDI",
      "url": "https://llxsd.github.io/docs/nsdi24-liu.pdf",
      "pdf_url": "https://llxsd.github.io/docs/nsdi24-liu.pdf",
      "regime": "A · Communication constellation",
      "evidence": [
        "§2–3 function splits; §5 protocol; §6 prototype; §7 evaluation and Table 1, PDF pp.10–13"
      ],
      "figure_number": "13",
      "figure_page_1based": 10,
      "evidence_status": "Full primary text reviewed",
      "built_for_en": "Multiple mobile operators lease LEO satellites to serve regular phones and IoT devices.",
      "built_for_zh": "多個 mobile operators 租用 LEO satellites，服務 regular phones 與 IoT devices。",
      "problem_en": "Hop-by-hop cellular sessions couple satellite operators, mobile operators and devices as orbital movement repeatedly changes their relationship.",
      "problem_zh": "hop-by-hop cellular sessions 將 satellite operators、mobile operators 與 devices 耦合，orbital movement 反覆改變三者關係。",
      "design_en": "Signed pay-as-you-go service tokens let satellites locally authorize service; geographic cells and end-to-end mobile sessions stabilize service policy.",
      "design_zh": "signed pay-as-you-go service tokens 讓 satellites 在本地授權服務；geographic cells 與 end-to-end mobile sessions 穩定 service policy。",
      "model_en": "Protocol state machines, cryptographic trust tokens, function-split deadlines, geographic service mappings and orbital trace replay.",
      "model_zh": "protocol state machines、cryptographic trust tokens、function-split deadlines、geographic service mappings 與 orbital trace replay。",
      "evaluation_en": "Commodity cellular/SIM prototype and constellation-driven signaling simulations; evaluate multi-operator access, paging load and service resumption.",
      "evaluation_zh": "commodity cellular/SIM prototype 與 constellation-driven signaling simulations；評估 multi-operator access、paging load 與 service resumption。",
      "baselines_en": "3GPP transparent satellite pipe, onboard distributed-unit and onboard full-RAN function splits; SpaceCore comparison; module ablations.",
      "baselines_zh": "3GPP transparent satellite pipe、onboard distributed-unit、onboard full-RAN function splits；SpaceCore comparison；module ablations。",
      "result_en": "Reported service-resumption latency improves 4.71–14.25× and signaling costs improve 850–7,640× in the evaluated multi-tenant mobility conditions.",
      "result_zh": "在評估的 multi-tenant mobility 條件下，service-resumption latency 改善 4.71–14.25×，signaling costs 改善 850–7,640×。",
      "limitations_en": "Results concern access authorization and mobile-session control. Orbital datacenter collectives need separate high-bandwidth transport and accelerator experiments.",
      "limitations_zh": "結果涵蓋 access authorization 與 mobile-session control。orbital datacenter collectives 需要獨立的 high-bandwidth transport 與 accelerator experiments。",
      "figure_bbox_points": [
        53,
        65,
        296,
        161
      ],
      "mechanism_scope": "Direct-to-cell multi-tenant state/control",
      "deployment_regime": "A",
      "space_ground_path": true,
      "display_regime_en": "A · Communication constellation",
      "display_regime_zh": "A · 通訊星座",
      "primary_pdf_sha256": "fab5d7eedfda37805be322bbbadc95b996a70eb5e0d258f7614103057ccad416",
      "figure_asset": "../figures/mosaic.png",
      "figure_acquisition": "Primary PDF crop",
      "figure_content_type": "image/png",
      "figure_bytes": 268884,
      "figure_sha256": "9120e72a2e4b8366c5cfed96057126b5fb3d6f0132569cdb2a4ddcf88481c93b"
    },
    {
      "id": "making-sense",
      "title": "Making Sense of Constellations: Methodologies for Understanding Starlink’s Scheduling Algorithms",
      "authors": "Hammas Bin Tanveer; Mike Puchol; Rachee Singh; Antonio Bianchi; Rishab Nithyanand",
      "year": "2023",
      "venue": "CoNEXT Companion",
      "url": "https://par.nsf.gov/servlets/purl/10568751",
      "pdf_url": "https://par.nsf.gov/servlets/purl/10568751",
      "regime": "A · Communication constellation",
      "evidence": [
        "§3–4 obstruction-map methodology; §5–6 model; §7–8 scope, PDF pp.3–8"
      ],
      "figure_number": "3",
      "figure_page_1based": 4,
      "evidence_status": "Full primary text reviewed",
      "built_for_en": "Consumer Starlink terminals expose signals for inferring serving-satellite assignment.",
      "built_for_zh": "consumer Starlink terminals 提供可推估 serving-satellite assignment 的 signals。",
      "problem_en": "Public terminal observations must separate physical satellite visibility from the operator’s assignment decisions.",
      "problem_zh": "公開 terminal observations 需要分辨 physical satellite visibility 與 operator assignment decisions。",
      "design_en": "Difference consecutive obstruction maps across 15-second slots; align sky tracks with public orbital predictions; train a satellite-characteristic predictor.",
      "design_zh": "對連續 15-second slots 的 obstruction maps 作差分；將 sky tracks 與公開 orbital predictions 對齊；訓練 satellite-characteristic predictor。",
      "model_en": "Map differencing, dynamic time warping, orbital geometry, random-forest classification and feature-importance analysis.",
      "model_zh": "map differencing、dynamic time warping、orbital geometry、random-forest classification 與 feature-importance analysis。",
      "evaluation_en": "Real-terminal measurements at geographically distributed sites; 80% training split with five-fold cross validation, 20% holdout and top-k accuracy.",
      "evaluation_zh": "在 geographically distributed sites 量測 real terminals；80% training split 搭配 five-fold cross validation、20% holdout 與 top-k accuracy。",
      "baselines_en": "A predictor that ranks clusters by the number of visible available satellites.",
      "baselines_zh": "依 visible available satellites 數量對 clusters 排序的 predictor。",
      "result_en": "Top-5 allocated-satellite characteristic accuracy reaches 65%, versus 22% for the availability-count baseline.",
      "result_zh": "top-5 allocated-satellite characteristic accuracy 達 65%，availability-count baseline 為 22%。",
      "limitations_en": "The model predicts characteristic clusters in measured northern-latitude sites. Satellite identities, other latitude ranges and firmware revisions require fresh validation. Venue status is CoNEXT Companion.",
      "limitations_zh": "模型預測量測 northern-latitude sites 的 characteristic clusters。satellite identities、其他 latitude ranges 與 firmware revisions 需要新 validation。venue 為 CoNEXT Companion。",
      "figure_bbox_points": [
        53,
        83,
        561,
        190
      ],
      "mechanism_scope": "Commercial scheduler inference",
      "deployment_regime": "A",
      "space_ground_path": false,
      "display_regime_en": "A · Communication constellation",
      "display_regime_zh": "A · 通訊星座",
      "primary_pdf_sha256": "4297f89340ae423dc3cae58cab294b738360ecdd28f14fbd0dab07d9f771efb4",
      "figure_asset": "../figures/making-sense.png",
      "figure_acquisition": "Primary PDF crop",
      "figure_content_type": "image/png",
      "figure_bytes": 105630,
      "figure_sha256": "a5e2591785ccf720b0e0674d0f27672afdf00d6c7a9908c135686e1c1876b09f"
    },
    {
      "id": "sno",
      "title": "Dissecting the Performance of Satellite Network Operators",
      "authors": "Aravindh Raman; Matteo Varvello; Hyunseok Chang; Nishanth Sastry; Yasir Zaki",
      "year": "2023",
      "venue": "CoNEXT",
      "url": "https://arxiv.org/pdf/2310.15808",
      "pdf_url": "https://arxiv.org/pdf/2310.15808",
      "regime": "A · Communication constellation",
      "evidence": [
        "§3 operator identification and data; §4 network performance; §5 application experiments; appendix participant counts"
      ],
      "figure_number": "1",
      "figure_page_1based": 5,
      "evidence_status": "Full primary text reviewed",
      "built_for_en": "LEO, MEO and GEO satellite operators serve public Internet subscribers.",
      "built_for_zh": "LEO、MEO 與 GEO satellite operators 服務 public Internet subscribers。",
      "problem_en": "Operator-level measurements need robust access identification, broad vantage coverage and application-level evidence.",
      "problem_zh": "operator-level measurements 需要穩健的 access identification、廣泛 vantage coverage 與 application-level evidence。",
      "design_en": "Combine ASN/operator classification with latency-distribution filtering, M-Lab and RIPE Atlas data, then collect browser/video experiments from recruited subscribers.",
      "design_zh": "結合 ASN/operator classification、latency-distribution filtering、M-Lab 與 RIPE Atlas data，再向 recruited subscribers 收集 browser/video experiments。",
      "model_en": "Kernel density estimation, longitudinal empirical distributions, operator/access classification and measurement-sample validation.",
      "model_zh": "kernel density estimation、longitudinal empirical distributions、operator/access classification 與 measurement-sample validation。",
      "evaluation_en": "Public longitudinal datasets across 18 satellite operators plus Prolific participant recruitment. The survey identifies 57 satellite subscribers among 14,371 screened participants; application experiments use their validated subset.",
      "evaluation_zh": "跨 18 家 satellite operators 的公開 longitudinal datasets，搭配 Prolific participant recruitment。survey 在 14,371 位 screened participants 中辨識 57 位 satellite subscribers；application experiments 使用 validated subset。",
      "baselines_en": "Cross-operator LEO/MEO/GEO comparisons, terrestrial references and access-specific webpage/video comparisons.",
      "baselines_zh": "cross-operator LEO/MEO/GEO comparisons、terrestrial references 與 access-specific webpage/video comparisons。",
      "result_en": "Measured Starlink access adds approximately 30–40 ms relative to favorable terrestrial connectivity; remote PoP choice produces roughly doubled latency in the Philippines example.",
      "result_zh": "量測 Starlink access 相對條件佳的 terrestrial connectivity 增加約 30–40 ms；Philippines example 的 remote PoP choice 造成約兩倍 latency。",
      "limitations_en": "Public speed tests and recruited users define the sample. ASN filtering, single-flow TCP behavior and geolocation quality shape interpretation.",
      "limitations_zh": "public speed tests 與 recruited users 定義 sample。ASN filtering、single-flow TCP behavior 與 geolocation quality 影響 interpretation。",
      "figure_bbox_points": [
        53,
        97,
        561,
        202
      ],
      "mechanism_scope": "Satellite operator performance measurement",
      "deployment_regime": "A",
      "space_ground_path": false,
      "display_regime_en": "A · Communication constellation",
      "display_regime_zh": "A · 通訊星座",
      "primary_pdf_sha256": "98a7a399487503701c97727597dd4d7f6c7173343e8c501a6ebd2cfcc69335f1",
      "figure_asset": "../figures/sno.png",
      "figure_acquisition": "Primary PDF crop",
      "figure_content_type": "image/png",
      "figure_bytes": 30840,
      "figure_sha256": "99d72049ad9604dc5cdc254f5e4e6c1b1cc5c90dbc00e1b9588e0c4638c7476d"
    },
    {
      "id": "first-look",
      "title": "A First Look at Starlink Performance",
      "authors": "François Michel; Martino Trevisan; Danilo Giordano; Olivier Bonaventure",
      "year": "2022",
      "venue": "IMC",
      "url": "https://iris.polito.it/retrieve/handle/11583/2972615/609130",
      "pdf_url": "https://iris.polito.it/retrieve/handle/11583/2972615/609130",
      "regime": "A · Communication constellation",
      "evidence": [
        "§2 measurement setup, Table 1; §3 network; §4 web; §5 QUIC and emulator, PDF pp.3–8"
      ],
      "figure_number": "1",
      "figure_page_1based": 4,
      "evidence_status": "Full primary text reviewed",
      "built_for_en": "A regular Starlink subscription in Louvain-la-Neuve, Belgium carries Internet and web traffic.",
      "built_for_zh": "Belgium Louvain-la-Neuve 的 regular Starlink subscription 承載 Internet 與 web traffic。",
      "problem_en": "Interactive latency, packet loss, throughput and application behavior change under loaded satellite access.",
      "problem_zh": "interactive latency、packet loss、throughput 與 application behavior 隨 loaded satellite access 改變。",
      "design_en": "Run periodic RIPE-anchor pings and Ookla tests, controlled QUIC transfers and BrowserTime experiments; publish a trace-based ERRANT emulation model.",
      "design_zh": "執行 periodic RIPE-anchor pings、Ookla tests、controlled QUIC transfers 與 BrowserTime experiments；公開 trace-based ERRANT emulation model。",
      "model_en": "Empirical latency/throughput distributions, packet-loss reconstruction, load-dependent RTT and trace-driven access emulation.",
      "model_zh": "empirical latency/throughput distributions、packet-loss reconstruction、load-dependent RTT 與 trace-driven access emulation。",
      "evaluation_en": "Five months of latency probes to 11 RIPE anchors, four months of speed tests and top-120 Belgian websites; 100 MB HTTP/3 transfers and light real-time QUIC message streams.",
      "evaluation_zh": "五個月對 11 個 RIPE anchors 的 latency probes；四個月 speed tests 與 Belgian top-120 websites；100 MB HTTP/3 transfers 與 light real-time QUIC message streams。",
      "baselines_en": "Commercial GEO service, UCLouvain wired 1 Gbps reference, HTTP/2 TCP and HTTP/3 QUIC behavior.",
      "baselines_zh": "commercial GEO service、UCLouvain wired 1 Gbps reference、HTTP/2 TCP 與 HTTP/3 QUIC behavior。",
      "result_en": "The nearby-anchor minimum RTTs shown in the paper fall near 20–29 ms. Loaded-transfer experiments reveal queue-related latency and losses beyond those minimum values.",
      "result_zh": "論文中 nearby-anchor minimum RTTs 約為 20–29 ms。loaded-transfer experiments 揭示額外 queue-related latency 與 losses。",
      "limitations_en": "One Belgian terminal and the early-2022 deployment define this measurement regime. Subsequent constellation and terminal revisions benefit from fresh calibration.",
      "limitations_zh": "單一 Belgian terminal 與 early-2022 deployment 定義量測 regime。後續 constellation 與 terminal revisions 需要新 calibration。",
      "figure_bbox_points": [
        53,
        82,
        296,
        193
      ],
      "mechanism_scope": "Early real-Starlink access measurement",
      "deployment_regime": "A",
      "space_ground_path": false,
      "display_regime_en": "A · Communication constellation",
      "display_regime_zh": "A · 通訊星座",
      "primary_pdf_sha256": "3286d93e6d53ab5f32f17cd264666b2d17cf4a395e8d489da1010192627150ea",
      "figure_asset": "../figures/first-look.png",
      "figure_acquisition": "Primary PDF crop",
      "figure_content_type": "image/png",
      "figure_bytes": 28224,
      "figure_sha256": "1554380cd0280e9e20bdd0507aabbf48f1290a9483a9a8b6a59a4f9ed6e87d57"
    },
    {
      "id": "cosmosim",
      "title": "Assessing LEO Satellite Networks for National Emergency Failover",
      "authors": "Vaibhav Bhosale; Ying Zhang; Sameer Kapoor; Robin Kim; Miguel Schlicht; Muskaan Gupta; Ekaterina Tumanova; Zachary S. Bischof; Fabián E. Bustamante; Alberto Dainotti; Ahmed Saeed",
      "year": "2025",
      "venue": "IMC",
      "url": "https://saeed.github.io/files/cosmosim-imc25.pdf",
      "pdf_url": "https://saeed.github.io/files/cosmosim-imc25.pdf",
      "regime": "A · Communication constellation",
      "evidence": [
        "§3 capacity graph and allocation; §4 data; §5 six case studies, Table 1–2; PDF pp.4–11"
      ],
      "figure_number": "3",
      "figure_page_1based": 5,
      "evidence_status": "Full primary text reviewed",
      "built_for_en": "National satellite access supplements international connectivity after submarine-cable failures.",
      "built_for_zh": "national satellite access 在 submarine-cable failures 後補充 international connectivity。",
      "problem_en": "Population placement, RF beams, spectrum reuse, gateways and policy jointly cap country-scale backup capacity.",
      "problem_zh": "population placement、RF beams、spectrum reuse、gateways 與 policy 共同限制 country-scale backup capacity。",
      "design_en": "Build a capacity graph with beam and interference constraints; allocate terminals and RF resources, then solve global or policy-restricted capacity flows.",
      "design_zh": "建立含 beam 與 interference constraints 的 capacity graph；配置 terminals、RF resources，再解 global 或 policy-restricted capacity flows。",
      "model_en": "NetworkX max-flow/min-cost capacity bounds; geometric beam feasibility; greedy capacity balancing and terminal-placement heuristics.",
      "model_zh": "NetworkX max-flow/min-cost capacity bounds；geometric beam feasibility；greedy capacity balancing 與 terminal-placement heuristics。",
      "evaluation_en": "Six real cable-failure case studies using RIPE Atlas and Calypso route/cable mappings; modeled Starlink regulatory configurations and terminal-count sweeps up to 50,000.",
      "evaluation_zh": "六個真實 cable-failure case studies 使用 RIPE Atlas 與 Calypso route/cable mappings；採 modeled Starlink regulatory configurations，掃描最高 50,000 terminals。",
      "baselines_en": "Population-proportional terminal placement, greedy capacity balancing, global versus domestic-restricted TE and alternative RF configurations.",
      "baselines_zh": "population-proportional terminal placement、greedy capacity balancing、global/domestic-restricted TE 與 alternative RF configurations。",
      "result_en": "Table 1 bounds range from 41 Gbps for Tonga to 4,653 Gbps for South Africa; comparison to lost cable capacity ranges from 0.9% for Great Britain to 434% for Haiti.",
      "result_zh": "Table 1 的 bounds 從 Tonga 41 Gbps 到 South Africa 4,653 Gbps；與 lost cable capacity 比較從 Great Britain 0.9% 到 Haiti 434%。",
      "limitations_en": "Capacity values are optimistic graph bounds conditioned on terminal placement and RF configuration. The represented interference model covers intra-satellite reuse; weather, terrain and inter-satellite interference motivate refinement.",
      "limitations_zh": "capacity values 為以 terminal placement 與 RF configuration 為條件的 optimistic graph bounds。interference model 涵蓋 intra-satellite reuse；weather、terrain 與 inter-satellite interference 提供 refinement 方向。",
      "figure_bbox_points": [
        316,
        82,
        567,
        188
      ],
      "mechanism_scope": "Country-scale capacity / RF resource allocation",
      "deployment_regime": "A",
      "space_ground_path": false,
      "display_regime_en": "A · Communication constellation",
      "display_regime_zh": "A · 通訊星座",
      "primary_pdf_sha256": "cf034fa5b66366a58444988189396cf6cf227c90a5dd074a25760534d9cf32f6",
      "figure_asset": "../figures/cosmosim.png",
      "figure_acquisition": "Primary PDF crop",
      "figure_content_type": "image/png",
      "figure_bytes": 92600,
      "figure_sha256": "606edaf260eb626d9a38b4aeb3d2223cfc4a9470f872828e7362abdaf210110e"
    },
    {
      "id": "leocc",
      "title": "LeoCC: Making Internet Congestion Control Robust to LEO Satellite Dynamics",
      "authors": "Zeqi Lai; Zonglun Li; Qian Wu; Hewu Li; Jihao Li; Xin Xie; Yuanjie Li; Jun Liu; Jianping Wu",
      "year": "2025",
      "venue": "SIGCOMM",
      "url": "https://doi.org/10.1145/3718958.3750491",
      "pdf_url": "https://dl.acm.org/doi/pdf/10.1145/3718958.3750491",
      "regime": "A · Communication constellation",
      "evidence": [
        "§4.1 reconfiguration detector; §4.2 Kalman/max estimators; §5 LeoReplayer; §6.1–6.7 evaluation, Figure10–21; browser source PDF pp.4–13"
      ],
      "figure_number": 6,
      "figure_page_1based": 4,
      "evidence_status": "Complete public article-reader primary text reviewed",
      "built_for_en": "End-to-end TCP traffic traverses commercial LEO satellite access and terrestrial Internet paths.",
      "built_for_zh": "end-to-end TCP traffic 經過 commercial LEO satellite access 與 terrestrial Internet paths。",
      "problem_en": "Connection reconfiguration abruptly changes capacity and base RTT; intrinsic delay jitter and radio loss distort conventional congestion estimates.",
      "problem_zh": "connection reconfiguration 突然改變 capacity 與 base RTT；固有 delay jitter 與 radio loss 影響 conventional congestion estimates。",
      "design_en": "Detect ACK-response-interval outliers and PoP ICMP signals; reset obsolete samples at reconfiguration, combine Kalman/max bandwidth estimates and RTT-band estimates, and transition between dynamic cruise and reconfiguration adaptation.",
      "design_zh": "偵測 ACK-response-interval outliers 與 PoP ICMP signals；在 reconfiguration 更新過期 samples，結合 Kalman/max bandwidth estimates 與 RTT-band estimates，並在 dynamic cruise、reconfiguration adaptation 間轉換。",
      "model_en": "Piecewise reconfiguration intervals, state-machine rate control, Kalman filtering, moving max/min filters and delay-based bottleneck discrimination.",
      "model_zh": "piecewise reconfiguration intervals、state-machine rate control、Kalman filtering、moving max/min filters 與 delay-based bottleneck discrimination。",
      "evaluation_en": "Linux kernel implementation; three live Starlink terminals in Madrid, New Jersey and Cebu; four Oracle servers; over fifty two-minute tests per CCA. LeoReplayer records 4,800 two-minute traces with saturated UDP plus ICMP probes and replays identical conditions for controlled comparisons.",
      "evaluation_zh": "Linux kernel implementation；Madrid、New Jersey、Cebu 三個 live Starlink terminals，四個 Oracle servers；每個 CCA 超過五十次、每次至少兩分鐘測試。LeoReplayer 以 saturated UDP 加 ICMP probes 收集 4,800 筆兩分鐘 traces，並 replay 相同 conditions 作 controlled comparisons。",
      "baselines_en": "TCP Reno, CUBIC, Vegas, BBRv1, BBRv3, VIVACE, Proteus, Copa, Verus, Sage and SaTCP. SaTCP receives LeoCC’s reconfiguration detector as its handover signal; LeoCC-SS and other component ablations isolate the design.",
      "baselines_zh": "TCP Reno、CUBIC、Vegas、BBRv1、BBRv3、VIVACE、Proteus、Copa、Verus、Sage、SaTCP。SaTCP 使用 LeoCC reconfiguration detector 作 handover signal；LeoCC-SS 與其他 component ablations 隔離設計效益。",
      "result_en": "Reported throughput rises 85–494% versus CUBIC/Copa/BBRv3 and delay falls 44–56% versus BBRv1/VIVACE. Controlled average utilization is 95.2% for LeoCC and 95.8% for BBRv1; the main gain couples high utilization with lower delay.",
      "result_zh": "報告 throughput 對 CUBIC/Copa/BBRv3 增加 85–494%；delay 對 BBRv1/VIVACE 降低 44–56%。controlled average utilization 為 LeoCC 95.2%、BBRv1 95.8%；主要效益是高 utilization 搭配低 delay。",
      "limitations_en": "LeoReplayer relies on the measured Starlink ICMP/data queue separation and access-link behavior. Optical cluster fabrics need their own reconfiguration, queue and failure calibration. Fairness with BBRv1 reaches Jain 0.98–0.99 in the tested configuration.",
      "limitations_zh": "LeoReplayer 依賴量測 Starlink 的 ICMP/data queue separation 與 access-link behavior。optical cluster fabrics 需要其自身 reconfiguration、queue、failure calibration。測試 configuration 中與 BBRv1 的 fairness 達 Jain 0.98–0.99。",
      "mechanism_scope": "LEO access congestion control",
      "deployment_regime": "A",
      "space_ground_path": true,
      "display_regime_en": "A · Communication constellation",
      "display_regime_zh": "A · 通訊星座",
      "figure_bbox_points": [
        313.4,
        82.8,
        557.5,
        177.7
      ],
      "figure_acquisition_method": "ACM public eReader, native zoom twice, viewport screenshot cropped to original Figure 6 and source caption",
      "figure_visually_inspected": true,
      "figure_asset": "../figures/leocc.png",
      "figure_acquisition": "ACM public eReader, native zoom twice, viewport screenshot cropped to original Figure 6 and source caption",
      "figure_content_type": "image/png",
      "figure_bytes": 193945,
      "figure_sha256": "418df5871c3385f57b15bd6f4f6d7f7da6fad1b74d12013b8dcbdb7702bb1472"
    },
    {
      "id": "coorbit",
      "title": "Achieving Efficient Storage and Communication via Collaboration",
      "authors": "Ruichen Li; Yufan Wu; Zhengyi Hu; Sheng-Jyun Cai; Lang Wei; Qifan Yang; Ting Zhu",
      "year": "2026",
      "venue": "SIGCOMM",
      "url": "https://doi.org/10.1145/3789240.3829176",
      "pdf_url": "https://dl.acm.org/doi/pdf/10.1145/3789240.3829176",
      "regime": "E · Observation/contact fleet",
      "evidence": [
        "§3.1–3.4 design; §5.1–5.4 methodology; §6.1–6.5 results, PDF pp.8–12; Appendix A.3 Algorithm 1, PDF p.16"
      ],
      "figure_number": "1",
      "figure_page_1based": 1,
      "evidence_status": "Complete public article-reader primary text reviewed",
      "built_for_en": "Earth-observation satellites exchange compact reference embeddings through ground contacts and retain informative image tiles.",
      "built_for_zh": "Earth-observation satellites 經 ground contacts 交換 compact reference embeddings，並保存有資訊價值的 image tiles。",
      "problem_en": "Temporal image stability and overlapping satellite coverage consume storage and downlink capacity; short uplink contacts constrain reference placement.",
      "problem_zh": "影像的 temporal stability 與 overlapping satellite coverage 消耗 storage、downlink capacity；短 uplink contacts 限制 reference placement。",
      "design_en": "MobileNetV2 tile embeddings support adaptive cosine-distance change detection; TLE-derived lookup tables remove predicted overlap; a ground planner greedily refines high-change tiles under per-contact embedding budgets.",
      "design_zh": "MobileNetV2 tile embeddings 支援 adaptive cosine-distance change detection；TLE-derived lookup tables 移除預測 overlap；ground planner 在 per-contact embedding budgets 下以 greedy procedure 細化 high-change tiles。",
      "model_en": "Cosine similarity, adaptive threshold μ+σ, geometric revisit/overlap prediction, greedy priority-queue tile splitting and time-integrated storage measured in byte-hours.",
      "model_zh": "cosine similarity、adaptive threshold μ+σ、geometric revisit/overlap prediction、greedy priority-queue tile splitting，以及以 byte-hours 量測 time-integrated storage。",
      "evaluation_en": "Ground Jetson Xavier NX/AGX Orin profiling plus simulation of 81 Planet satellites and a scaled 200-satellite Satellogic constellation. Daily DynamicEarthNet images emulate successive orbital revisits; seven areas cover generic imagery, wildfire and building damage. Model downlink is 160 Mbps and uplink 32 Kbps.",
      "evaluation_zh": "地面 Jetson Xavier NX/AGX Orin profiling，加上 81 顆 Planet satellites 與 scaled 200-satellite Satellogic constellation 的 simulation。以 daily DynamicEarthNet images 模擬 successive orbital revisits；七個 areas 涵蓋 generic imagery、wildfire、building damage。model downlink 為 160 Mbps、uplink 為 32 Kbps。",
      "baselines_en": "Full-image/full-downlink storage baseline; LUT-only, ALP-only and full-system ablations; Earth+, DeepSpace and CCSDS 123.0-B-2 communication comparisons. Kodan/Serval/WaveDiff/SR3 appear in the related-system taxonomy.",
      "baselines_zh": "full-image/full-downlink storage baseline；LUT-only、ALP-only、full-system ablations；communication comparison 使用 Earth+、DeepSpace、CCSDS 123.0-B-2。Kodan/Serval/WaveDiff/SR3 列於 related-system taxonomy。",
      "result_en": "Australian AoI shows 108.6× lower storage byte-hours and 41.6× lower communication volume versus conventional full downlink. AGX Orin base processing stays below 2 s/image and both boards stay below 10 W. Wildfire-driven Pilbara storage improves another 10.9×.",
      "result_zh": "Australian AoI 相對 conventional full downlink 的 storage byte-hours 降低 108.6×，communication volume 降低 41.6×。AGX Orin base processing 維持每張 image 低於 2 s，兩款 boards 均低於 10 W。wildfire-driven Pilbara storage 額外改善 10.9×。",
      "limitations_en": "Collaboration follows ground-mediated contact schedules. Daily images substitute for same-day orbital revisits; temporal granularity, embedding-based change recall and TLE buffers define fidelity. The hardware evidence profiles terrestrial edge boards.",
      "limitations_zh": "collaboration 沿用 ground-mediated contact schedules。daily images 代替 same-day orbital revisits；temporal granularity、embedding-based change recall 與 TLE buffers 定義 fidelity。hardware evidence 量測地面 edge boards。",
      "mechanism_scope": "Earth-observation storage / reference placement",
      "deployment_regime": "E",
      "space_ground_path": true,
      "display_regime_en": "E/C · Observation and ground delivery",
      "display_regime_zh": "E／C · 觀測與地面交付",
      "figure_asset": "../figures/coorbit.png",
      "figure_acquisition": "Public publisher article reader capture",
      "figure_content_type": "image/png",
      "figure_bytes": 229388,
      "figure_sha256": "112ebf8e811eaad58373cd56947c2eb9989e7574a6382fba20e7895494057c78"
    },
    {
      "id": "sn2",
      "title": "Direct-to-Cell Satellite Network without Satellite Navigation",
      "authors": "Wei Liu; Yuanjie Li; Jingyi Lan; Hewu Li; Yimei Chen; Lixin Liu; Jiabo Yang; Xi Long; Li Ouyang; Minghao Tang; Jianping Wu; Qian Wu; Jun Liu; Zeqi Lai",
      "year": "2025",
      "venue": "SIGCOMM",
      "url": "https://doi.org/10.1145/3718958.3750522",
      "pdf_url": "https://dl.acm.org/doi/pdf/10.1145/3718958.3750522",
      "regime": "A · Communication constellation",
      "evidence": [
        "§4.1 relaxed timing/Doppler geometry; §4.2 monotonic signed timing; §4.3 policy authorization; §5 prototype; §6.1–6.4 evaluation, PDF pp.10–12, Figure17–25"
      ],
      "figure_number": "1",
      "figure_page_1based": 1,
      "evidence_status": "Complete public article-reader primary text reviewed",
      "built_for_en": "Regular phones and IoT devices access direct-to-cell LEO satellite services under navigation interference and heterogeneous geographic policies.",
      "built_for_zh": "regular phones 與 IoT devices 在 navigation interference 與 heterogeneous geographic policies 下，存取 direct-to-cell LEO satellite services。",
      "problem_en": "GNSS-derived location and timing couple radio synchronization, satellite authentication and policy authorization to an external navigation service.",
      "problem_zh": "GNSS 提供的 location、timing 將 radio synchronization、satellite authentication 與 policy authorization 耦合於 external navigation service。",
      "design_en": "Use serving-satellite delay/Doppler signals for incremental localization, relax geometric accuracy to satisfy radio constraints, derive monotonic trusted time from authenticated signed broadcasts, and authorize services according to the policy region consistent with the position bounds.",
      "design_zh": "利用 serving-satellite delay/Doppler signals 逐步定位，調整 geometric accuracy 以滿足 radio constraints，從 authenticated signed broadcasts 建立 monotonic trusted time，並依 position bounds 對應的 policy region 授權 services。",
      "model_en": "TDoA/Doppler localization, timing-advance and frequency-error feasible regions, monotonic timestamp ordering, geometric policy/beam intersection and protocol state machines.",
      "model_zh": "TDoA/Doppler localization、timing-advance/frequency-error feasible regions、monotonic timestamp ordering、geometric policy/beam intersection 與 protocol state machines。",
      "evaluation_en": "Amarisoft Callbox 3GPP-R17/18 IoT/NR-NTN prototype with channels calibrated to real RSRP/SNR and public ephemeris; COTS iPhone 15, Iridium GO and Iridium 9555 satellite tests; USRP B210 gps-sdr-sim and fake NTN radios exercise controlled interference/spoofing.",
      "evaluation_zh": "Amarisoft Callbox 3GPP-R17/18 IoT/NR-NTN prototype，以真實 RSRP/SNR 與 public ephemeris 校準 channels；iPhone 15、Iridium GO、Iridium 9555 的 COTS satellite tests；USRP B210 gps-sdr-sim 和 fake NTN radios 建立 controlled interference/spoofing。",
      "baselines_en": "3GPP NTN, Globalstar iPhone satellite access, Iridium GO with GNSS and Iridium 9555 delay/Doppler localization; GNSS availability and attack-state comparisons.",
      "baselines_zh": "3GPP NTN、Globalstar iPhone satellite access、搭配 GNSS 的 Iridium GO、使用 delay/Doppler localization 的 Iridium 9555；比較 GNSS availability 與 attack states。",
      "result_en": "Overall service activation is 7.2–23.5× faster with available GNSS and 4.4× faster than Iridium localization under GNSS disruption. First radio access is 7.4–12.3× faster with GNSS and 1.9× faster under disruption. Authentication costs up to 0.025% radio resource at 20 ms signed-broadcast periodicity.",
      "result_zh": "GNSS available 時，overall service activation 快 7.2–23.5×；GNSS disruption 時相對 Iridium localization 快 4.4×。first radio access 在搭配 GNSS 時快 7.4–12.3×，disruption 時快 1.9×。20 ms signed-broadcast periodicity 的 authentication cost 最高為 0.025% radio resource。",
      "limitations_en": "The abstract’s availability multiplier describes faster service activation in §6.1; uptime percentage requires a separate availability metric. Controlled spoofing results come from channel-calibrated protocol hardware, and the real-device observations ground legacy operator behavior.",
      "limitations_zh": "abstract 的 availability multiplier 在 §6.1 對應較快 service activation；uptime percentage 需要另外的 availability metric。controlled spoofing results 來自 channel-calibrated protocol hardware，real-device observations 為 legacy operator behavior 提供實證。",
      "mechanism_scope": "Direct-to-cell navigation / trust / authorization",
      "deployment_regime": "A",
      "space_ground_path": true,
      "display_regime_en": "A · Communication constellation",
      "display_regime_zh": "A · 通訊星座",
      "figure_asset": "../figures/sn2.png",
      "figure_acquisition": "Public publisher article reader capture",
      "figure_content_type": "image/png",
      "figure_bytes": 135204,
      "figure_sha256": "3d448b1428d20bc57f37637ef5523a91c99f39a66da188d5a2b3f56cbaa733eb"
    },
    {
      "id": "cosmac",
      "title": "CosMAC: Constellation-Aware Medium Access and Scheduling for IoT Satellites",
      "authors": "Jayanth Shenoy; Om Chabra; Tusher Chakraborty; Suraj Jog; Deepak Vasisht; Ranveer Chandra",
      "year": "2024",
      "venue": "MobiCom",
      "url": "https://doi.org/10.1145/3636534.3690657",
      "pdf_url": "https://deepakv.web.illinois.edu/assets/papers/CosMac_MobiCom_2024.pdf",
      "regime": "A · Communication constellation",
      "evidence": [
        "§4.2 Eq.1 overlap-aware probability; §4.3 Eq.2 Poisson/binomial flow control, PDF pp.5–7; §5.2–5.3 conflict graph/MWIS, PDF pp.8–10; §6–7 platform and calibration, PDF pp.10–12; §8 Figure10–11, PDF pp.12–14"
      ],
      "figure_number": "2",
      "figure_page_1based": 3,
      "figure_bbox_points": [
        53,
        81,
        562,
        236
      ],
      "evidence_status": "Full primary text reviewed",
      "figure_caption_en": "CosMAC combines device-local overlap-aware random access, satellite-beacon aggregate flow feedback, and a cloud-computed conflict-graph downlink schedule. This original overview separates uplink collisions from one-to-many downlink interference.",
      "figure_caption_zh": "CosMAC 結合 device-local overlap-aware random access、satellite-beacon aggregate flow feedback 與 cloud-computed conflict-graph downlink schedule。原始 overview 分開呈現 uplink collisions 和 one-to-many downlink interference。",
      "built_for_en": "Low-power, omnidirectional LoRa picosatellite constellations connect global terrestrial IoT devices through intermittent ground contacts.",
      "built_for_zh": "低功耗、omnidirectional LoRa picosatellite constellations 經 intermittent ground contacts 連接全球 terrestrial IoT devices。",
      "problem_en": "Devices inside overlapping satellite footprints cause multi-receiver uplink contention, while broadcast downlink transmissions interfere across distributed ground stations.",
      "problem_zh": "位於 overlapping satellite footprints 的 devices 造成 multi-receiver uplink contention；broadcast downlink transmissions 則在 distributed ground stations 間形成 interference。",
      "design_en": "Choose device transmission probability α divided by the sum of device counts across visible footprints. Beacon-based additive-increase/multiplicative-decrease updates α using channel activity and decoded-packet trends. A centralized conflict-graph scheduler chooses high-RSSI downlinks with receiver diversity.",
      "design_zh": "將 device transmission probability 設為 α 除以 visible footprints 的 device counts 總和。beacon-based additive-increase/multiplicative-decrease 根據 channel activity 與 decoded-packet trends 更新 α。centralized conflict-graph scheduler 選擇 high-RSSI downlinks 並配置 receiver diversity。",
      "model_en": "Poisson data generation and binomial transmission/collision probabilities; weighted maximum independent set on satellite-ground link vertices with interference, receiver exclusivity and diversity-distance edges; randomized approximation plus greedy reliability repair for K≥2 receivers.",
      "model_zh": "Poisson data generation、binomial transmission/collision probabilities；以 satellite-ground link vertices 構成 weighted maximum independent set，edges 描述 interference、receiver exclusivity 和 diversity distance；採 randomized approximation 與 greedy reliability repair 配置 K≥2 receivers。",
      "evaluation_en": "Three FOSSA picosatellites and two Spanish ground stations ground collision, power and RF-link calibration. CosmicBeats simulates 173 SWARM-derived satellites, 100,000 uniformly distributed devices and 1,048 TinyGS sites for eight hours at one-second epochs, with 100-byte packets generated 5/25/50/100 times daily.",
      "evaluation_zh": "三顆 FOSSA picosatellites 和兩個 Spanish ground stations 為 collision、power、RF-link calibration 提供實證。CosmicBeats 使用 173 顆 SWARM-derived satellites、100,000 個均勻分佈 devices、1,048 個 TinyGS sites，以 one-second epochs 模擬八小時；100-byte packets 每日產生 5/25/50/100 次。",
      "baselines_en": "UTPF, an Aloha transmission-probability function based on a single footprint; fixed-probability FP-Aloha with p=25/100000; L2D2 maximal-matching downlink scheduler; combined UTPF+L2D2 and FP-Aloha+L2D2 end-to-end baselines.",
      "baselines_zh": "UTPF：依 single footprint 設定 Aloha transmission-probability function；FP-Aloha 使用 fixed probability p=25/100000；L2D2 maximal-matching downlink scheduler；end-to-end baselines 為 UTPF+L2D2、FP-Aloha+L2D2。",
      "result_en": "The paper reports up to 6.5× aggregate end-to-end throughput. At 50 packets/device/day, Figure 10 reports 1,388 bps for CosMAC, 211 bps for UTPF+L2D2 and 922 bps for FP-Aloha+L2D2. Median downlink receiver diversity is 12 ground stations in simulation.",
      "result_zh": "論文報告 aggregate end-to-end throughput 最高提升 6.5×。每日每個 device 50 packets 時，Figure10 為 CosMAC 1,388 bps、UTPF+L2D2 211 bps、FP-Aloha+L2D2 922 bps。simulation 的 median downlink receiver diversity 為 12 個 ground stations。",
      "limitations_en": "Live experiments validate component behavior and simulator models; throughput gains come from large-scale simulation. CAD collision sensing uses a ground receiver with two transmitting satellites because launched firmware is fixed. Device-location estimates and distributed ground-station backhaul define additional assumptions.",
      "limitations_zh": "live experiments 驗證 component behavior 與 simulator models；throughput gains 來自 large-scale simulation。CAD collision sensing 使用地面 receiver 和兩顆 transmitting satellites，對應已發射衛星的固定 firmware。device-location estimates 與 distributed ground-station backhaul 定義額外 assumptions。",
      "mechanism_scope": "IoT picosatellite RF MAC / downlink scheduling",
      "deployment_regime": "A",
      "space_ground_path": true,
      "display_regime_en": "A/C · Contact-limited IoT service",
      "display_regime_zh": "A／C · Contact-limited IoT 服務",
      "primary_pdf_sha256": "819974e0068996c627b616753f1bce9a8193e0bc0aa7156c779dfc7dba0f5597",
      "figure_asset": "../figures/cosmac.png",
      "figure_acquisition": "Primary PDF crop",
      "figure_content_type": "image/png",
      "figure_bytes": 205224,
      "figure_sha256": "1a2c8f0613767c9f3cf897f7e0d7edd223224508b0dc01a19f435f3b25907b29"
    },
    {
      "id": "taccl",
      "title": "TACCL: Guiding Collective Algorithm Synthesis using Communication Sketches",
      "authors": "Aashaka Shah; Vijay Chidambaram; Meghan Cowan; Saeed Maleki; Madan Musuvathi; Todd Mytkowicz; Jacob Nelson; Olli Saarikivi; Rachee Singh",
      "year": "2023",
      "venue": "NSDI",
      "url": "https://www.usenix.org/conference/nsdi23/presentation/shah",
      "pdf_url": "https://www.usenix.org/system/files/nsdi23-shah.pdf",
      "regime": "T · Terrestrial reference",
      "evidence": [
        "§3 communication sketches, PDF pp.5–6; §4.1 Table1 α–β profiling, PDF p.7; §5.1 three-stage synthesis, PDF pp.8–9; §6 runtime, PDF pp.9–10; §7.1–7.4 and Figures6–10, PDF pp.10–14; AppendixB MILP, PDF pp.18–20"
      ],
      "figure_number": "1",
      "figure_page_1based": 3,
      "figure_bbox_points": [
        112,
        70,
        505,
        144
      ],
      "evidence_status": "Full primary text reviewed; official NSDI2023 paper entry verified",
      "figure_caption_en": "TACCL takes a communication sketch, measured topology and collective semantics through relaxed routing, heuristic ordering and exact scheduling, then lowers the schedule into a GPU runtime. The figure illustrates a concrete collective-aware synthesis baseline for transfer studies.",
      "figure_caption_zh": "TACCL 將 communication sketch、measured topology 與 collective semantics 送入 relaxed routing、heuristic ordering 和 exact scheduling，再將 schedule 編譯到 GPU runtime。此原始 figure 提供 transfer studies 的具體 collective-aware synthesis baseline。",
      "built_for_en": "Multi-node GPU clusters run topology- and message-size-specific AllGather, AllToAll and AllReduce for distributed model training.",
      "built_for_zh": "multi-node GPU clusters 以 topology- 與 message-size-specific AllGather、AllToAll、AllReduce 執行 distributed model training。",
      "problem_en": "Heterogeneous NVLink, PCIe and InfiniBand paths, shared-link contention and chunk dependencies create a large routing-and-scheduling search space for collective algorithms.",
      "problem_zh": "heterogeneous NVLink、PCIe、InfiniBand paths、shared-link contention 與 chunk dependencies 形成龐大的 collective routing-and-scheduling search space。",
      "design_en": "Designer communication sketches constrain logical topology, switch-hyperedge connection policies, algorithm symmetry and input size. TACCL profiles α–β link costs, solves relaxed-routing MILP, greedily orders chunks, then solves contiguity/exact-scheduling MILP and executes generated TACCL-EF in an NCCL-compatible GPU interpreter.",
      "design_zh": "designer communication sketches 設定 logical topology、switch-hyperedge connection policies、algorithm symmetry 和 input size。TACCL 量測 α–β link costs，解 relaxed-routing MILP、以 greedy procedure 排序 chunks，再解 contiguity/exact-scheduling MILP，並以 NCCL-compatible GPU interpreter 執行生成的 TACCL-EF。",
      "model_en": "Continuous-time MILP with binary chunk-send/link-use variables and real-valued arrival/send times; collective pre/postconditions and data dependencies; α+βs transmission cost; relaxed capacity lower bounds followed by heuristic ordering and capacity-valid scheduling. AllGather permits chunk replication.",
      "model_zh": "continuous-time MILP 使用 binary chunk-send/link-use variables 與 real-valued arrival/send times；包含 collective pre/postconditions、data dependencies、α+βs transmission cost；先求 relaxed capacity lower bounds，再以 heuristic ordering 形成 capacity-valid scheduling。AllGather 可複製 chunks。",
      "evaluation_en": "Actual NVIDIA V100 hardware: two DGX-2 nodes or up to four Azure NDv2 nodes in the main 32-GPU evaluation. Standalone collective bandwidth sweeps, sketch/runtime ablations and PyTorch training of Transformer-XL, BERT and an internal MoE workload. Table 2 measures synthesis time; larger synthesis-only tests reach 80 and 128 GPUs.",
      "evaluation_zh": "實際 NVIDIA V100 hardware：main 32-GPU evaluation 使用兩臺 DGX-2 或最多四臺 Azure NDv2。包含 standalone collective bandwidth sweeps、sketch/runtime ablations，以及 Transformer-XL、BERT、internal MoE workload 的 PyTorch training。Table2 量測 synthesis time；更大 synthesis-only tests 達 80、128 GPUs。",
      "baselines_en": "NCCL v2.8.4-1 on identical hardware; ablations vary logical inter-node connectivity, chunk size/partitioning, switch-hyperedge policy and runtime instances. SCCL, Blink and Plink provide related-synthesis context.",
      "baselines_zh": "相同 hardware 上的 NCCL v2.8.4-1；ablations 掃描 logical inter-node connectivity、chunk size/partitioning、switch-hyperedge policy、runtime instances。SCCL、Blink、Plink 提供 related-synthesis context。",
      "result_en": "The 6.7× headline is standalone AllGather on two DGX-2 nodes at small message sizes. Across evaluated batch sizes, Transformer-XL training gains are 11%–1.94× on two NDv2 nodes and 2%–1.44× on four; BERT gains are 12%–2.36× and 7%–1.74×. Internal MoE throughput rises 17%. DGX-2 AllReduce at ≥512 MB is up to 9% slower than NCCL.",
      "result_zh": "6.7× headline 是兩臺 DGX-2 的 small-message standalone AllGather。在評估的 batch sizes 下，Transformer-XL training 在兩臺 NDv2 的 gains 為 11%–1.94×，四臺為 2%–1.44×；BERT 對應為 12%–2.36×、7%–1.74×。internal MoE throughput 增加 17%。DGX-2 AllReduce 在 ≥512 MB 時，最高比 NCCL 慢 9%。",
      "limitations_en": "Evaluation uses fixed profiled terrestrial topologies and designer-restricted sketches. A 30-minute contiguity timeout occurs in one AllToAll case; 128-GPU synthesis takes about 11 hours. Orbital transfer needs time-indexed capacity, link acquisition, faults and transition costs, plus an executable collective dependency model.",
      "limitations_zh": "evaluation 使用固定 profiled terrestrial topologies 與 designer-restricted sketches。一個 AllToAll case 達 30-minute contiguity timeout；128-GPU synthesis 約需 11 小時。orbital transfer 需要 time-indexed capacity、link acquisition、faults、transition costs，以及可執行的 collective dependency model。",
      "mechanism_scope": "Terrestrial collective-synthesis transfer baseline",
      "deployment_regime": "T",
      "space_ground_path": false,
      "display_regime_en": "T · Terrestrial reference",
      "display_regime_zh": "T · 地面比較基準",
      "primary_pdf_sha256": "376bee50850bda3a434330299c2505a4ca3933ae4cd90cb53a3b6965b7b4101e",
      "figure_asset": "../figures/taccl.png",
      "figure_acquisition": "Primary PDF crop",
      "figure_content_type": "image/png",
      "figure_bytes": 31949,
      "figure_sha256": "17851e6e558463b9b7dfd5a47a1d521f5f2f24a1b002ae899de20d24c07a5c54"
    },
    {
      "id": "suncatcher",
      "title": "Towards a future space-based, highly scalable AI infrastructure system design",
      "authors": "Blaise Agüera y Arcas; Travis Beals; Maria Biggs; Jessica V. Bloom; Thomas Fischbacher; Konstantin Gromov; Urs Köster; Rishiraj Pravahan; James Manyika",
      "year": "2025",
      "venue": "arXiv",
      "url": "https://arxiv.org/abs/2511.19468",
      "pdf_url": "https://arxiv.org/pdf/2511.19468v2",
      "regime": "B · Tight AI formation",
      "evidence": [
        "v2 §1.2 p2",
        "§2.1 p2–3",
        "§2.2 p3–5",
        "§2.3 p5–6",
        "§3 p7",
        "§4.2 p8–9",
        "Table1 p11"
      ],
      "figure_number": 1,
      "figure_page_1based": 3,
      "figure_bbox_points": [
        304,
        85,
        529,
        225
      ],
      "built_for_en": "A compact sun-synchronous TPU formation supplies tightly coupled machine-learning compute.",
      "built_for_zh": "緊密編隊的 sun-synchronous TPU 叢集提供密切耦合的機器學習運算。",
      "problem_en": "Existing commercial optical ISLs provide 1–100 Gbps while the proposed ML fabric requires aggregate links around 10 Tbps.",
      "problem_zh": "既有商用 optical ISL 提供 1–100 Gbps，提出的 ML fabric 需要約 10 Tbps 的聚合 link。",
      "design_en": "Short-distance FSO links combine DWDM and spatial multiplexing; a free-fall formation preserves nearby neighbors; radiation-tested Trillium TPUs supply compute.",
      "design_zh": "短距離 FSO link 結合 DWDM 與 spatial multiplexing；自由落體編隊維持鄰近節點；經輻射測試的 Trillium TPU 提供運算。",
      "model_en": "Gaussian-beam link budgets, Hill–Clohessy–Wiltshire orbital dynamics with J2 correction, proton radiation tests, and launch-price learning curves.",
      "model_zh": "Gaussian beam link budget、具 J2 修正的 Hill–Clohessy–Wiltshire 軌道動力學、質子輻射測試，以及發射價格學習曲線。",
      "evaluation_en": "An 81-satellite numerical formation at 650 km and 1 km radius; a short-path optical bench; 67 MeV proton irradiation of Trillium and host components.",
      "evaluation_zh": "在 650 km 高度、1 km 半徑下模擬 81 顆衛星的編隊；短距離光學 bench；使用 67 MeV 質子照射 Trillium 與 host 元件。",
      "baselines_en": "Commercial Starlink/Mynaric optical specifications and photon-per-bit modulation bounds; terrestrial launched-power cost references.",
      "baselines_zh": "商用 Starlink/Mynaric 光學規格、各調變方式每 bit 光子數下界，以及地面能源成本參照。",
      "result_en": "The optical bench achieves 800 Gbps in one direction and 1.6 Tbps bidirectional. A 24-channel 400G DWDM design yields a 9.6 Tbps design estimate.",
      "result_zh": "光學 bench 達到單向 800 Gbps 與雙向 1.6 Tbps；24 個 400G DWDM channel 的設計估計為 9.6 Tbps。",
      "limitations_en": "The 10 Tbps fabric is a design estimate. TPU count per satellite, production radiator dimensions, trained-cluster performance, and fielded multi-satellite fabrics remain research parameters.",
      "limitations_zh": "10 Tbps fabric 屬於設計估計；每衛星 TPU 數量、量產 radiator 尺寸、叢集訓練效能，以及在軌多衛星 fabric 屬於後續研究參數。",
      "figure_caption_en": "The optical link sketch couples DWDM capacity, free-space propagation, and spatial multiplexing. The paper separately demonstrates 800 Gbps directional bench throughput and projects multi-Tbps designs.",
      "figure_caption_zh": "光學 link 圖串接 DWDM capacity、自由空間傳播與 spatial multiplexing；論文另以 bench 示範單向 800 Gbps，並估算 multi-Tbps 設計。",
      "evidence_status": "Full primary text reviewed",
      "mechanism_scope": "Compact formation",
      "deployment_regime": "B",
      "space_ground_path": false,
      "display_regime_en": "B · Tight AI formation",
      "display_regime_zh": "B · 緊密 AI 編隊",
      "primary_pdf_sha256": "0dee31ee1f2d9185d724e58d039a5df89c204797aad1cd006042ad31abd5b290",
      "figure_asset": "../figures/suncatcher.png",
      "figure_acquisition": "Primary PDF crop",
      "figure_content_type": "image/png",
      "figure_bytes": 48907,
      "figure_sha256": "2618235de8f0e9e87bef1636c88b6c9a7757f4414051be1a07ed458eae325fba"
    },
    {
      "id": "spacemoe",
      "title": "SpaceMoE: Realizing Distributed Mixture-of-Experts Inference over Space Networks",
      "authors": "Zhanwei Wang; Huiling Yang; Min Sheng; Khaled B. Letaief; Kaibin Huang",
      "year": "2026",
      "venue": "arXiv",
      "url": "https://arxiv.org/abs/2605.00515",
      "pdf_url": "https://arxiv.org/pdf/2605.00515v2",
      "regime": "D · Orbital compute mesh",
      "evidence": [
        "v2 §II pp3–4",
        "§IV–V pp5–10",
        "§VII pp10–12",
        "TableII p11"
      ],
      "figure_number": 3,
      "figure_page_1based": 5,
      "figure_bbox_points": [
        316,
        49,
        569,
        226
      ],
      "built_for_en": "A geographically distributed polar constellation executes autoregressive MoE token generation with small onboard computers.",
      "built_for_zh": "地理分散的極地衛星星座使用小型 onboard computer 執行 autoregressive MoE token generation。",
      "problem_en": "Expert activation probabilities and moving satellite paths jointly determine gateway-to-expert latency.",
      "problem_zh": "expert activation 機率與移動中的衛星路徑共同決定 gateway 到 expert 的延遲。",
      "design_en": "Ring-aligned subnets map layers to orbit segments. Central gateways and activation-aware expert assignment associate popular experts with low expected path latency.",
      "design_zh": "沿軌道方向的 ring subnet 將 layer 對應至軌道區段；中心 gateway 與 activation-aware expert assignment 將熱門 expert 放入較低 expected path latency 的節點。",
      "model_en": "Temporal graphs, Dijkstra routes, Bernoulli link feasibility, probability-proportional sampling, bottleneck approximation and placement ordering theorems.",
      "model_zh": "temporal graph、Dijkstra route、Bernoulli link feasibility、依權重抽樣、瓶頸近似與 placement 排序定理。",
      "evaluation_en": "A 1056-satellite constellation uses 33 planes ×32 satellites, 550 km, 87° inclination, 200 snapshots, 0.95 link survival, and ≥100 Gbps ISLs. LLaMA-MoE-3.5B activation traces span eight reasoning datasets. Effective modeled CPU throughput is 7.28 GFLOPS.",
      "evaluation_zh": "1056 顆衛星由 33 個 plane ×32 顆構成，採用 550 km、87° inclination、200 個 snapshot、0.95 link survival 與 ≥100 Gbps ISL。LLaMA-MoE-3.5B activation trace 涵蓋八個推理資料集；模型使用 7.28 GFLOPS 的有效 CPU throughput。",
      "baselines_en": "RandPlace, RandIntra and RandIntra-CG.",
      "baselines_zh": "RandPlace、RandIntra 與 RandIntra-CG。",
      "result_en": "Table II reports 1.02–1.07 s/token, compared with 3.34–3.37 s/token for RandIntra-CG and 5.28–5.30 s/token for RandPlace.",
      "result_zh": "Table II 報告 1.02–1.07 s/token；RandIntra-CG 為 3.34–3.37 s/token，RandPlace 為 5.28–5.30 s/token。",
      "limitations_en": "The experiment evaluates sampled topology snapshots and a propagation-dominated latency model. Packet queues, dynamic expert loading, arrival-rate SLOs and compact TPU formations require separate experiments.",
      "limitations_zh": "實驗評估抽樣 topology snapshot 與 propagation 主導的 latency model；packet queue、動態 expert loading、arrival-rate SLO 與緊密 TPU 編隊需要各自的實驗。",
      "figure_caption_en": "SpaceMoE maps layer sub-networks and expert placement onto a satellite constellation; its center gateways aggregate activation delivery between layer groups.",
      "figure_caption_zh": "SpaceMoE 將 layer sub-network 與 expert placement 對應到衛星星座；center gateway 聚合 layer group 之間的 activation delivery。",
      "evidence_status": "Full primary text reviewed",
      "mechanism_scope": "Distributed LEO compute",
      "deployment_regime": "D",
      "space_ground_path": false,
      "display_regime_en": "D · Orbital compute mesh",
      "display_regime_zh": "D · 軌道運算 mesh",
      "primary_pdf_sha256": "0f8024b0fb5c6542d8fba0430f76f0e26e0c6e34b7687f1ff42ebd9ff6ee0af0",
      "figure_asset": "../figures/spacemoe.png",
      "figure_acquisition": "Primary PDF crop",
      "figure_content_type": "image/png",
      "figure_bytes": 59778,
      "figure_sha256": "cb7cb9cf1af56813ca6640ae7ae3d3f082412c97ca516e2bc58cc0f00b6105fd"
    },
    {
      "id": "collaborative-llm",
      "title": "Communication-Efficient Collaborative LLM Inference over LEO Satellite Networks",
      "authors": "Songge Zhang; Wen Wu; Liang Li; Ye Wang; Xuemin Shen",
      "year": "2026",
      "venue": "arXiv",
      "url": "https://arxiv.org/abs/2604.04654",
      "pdf_url": "https://arxiv.org/pdf/2604.04654v1",
      "regime": "D · Orbital compute mesh",
      "evidence": [
        "§III–V pp3–8",
        "§VI-A/TableII/TableIII p9",
        "§VI-B Fig10/TableIV/V pp10–12",
        "WCSP2025 predecessor DOI10.1109/WCSP68525.2025.1010621"
      ],
      "figure_number": 1,
      "figure_page_1based": 2,
      "figure_bbox_points": [
        46,
        52,
        296,
        193
      ],
      "built_for_en": "A satellite chain jointly processes remote-sensing images with partitioned vision transformers.",
      "built_for_zh": "衛星 chain 使用分割的 vision transformer 協同處理 remote-sensing image。",
      "problem_en": "Limited onboard memory and activation traffic constrain model splitting and end-to-end delivery delay.",
      "problem_zh": "有限 onboard memory 與 activation traffic 約束 model splitting 和端到端結果傳送延遲。",
      "design_en": "Model splitting and compute–communication overlap combine with learned Gumbel-mask sparsification, quantization and entropy coding.",
      "design_zh": "model splitting 與 compute–communication overlap 結合學習式 Gumbel mask sparsification、quantization 與 entropy coding。",
      "model_en": "A mixed-integer nonlinear delay minimization is represented as a DAG search. An outer A* layer assignment uses inner compression optimization and a maximum-stage pipeline term.",
      "model_zh": "mixed-integer nonlinear delay minimization 轉為 DAG search；外層 A* layer assignment 使用內層 compression optimization 與 maximum-stage pipeline 項。",
      "evaluation_en": "Hardware profiling uses four Jetson AGX Orin devices and an RTX 4070 Ti server. Simulation uses a 12-satellite orbit with a five-node compute setup, 0.5 Gbps ISLs, ViT-B/L/H/G, EuroSAT and RESISC45.",
      "evaluation_zh": "硬體 profiling 使用四臺 Jetson AGX Orin 與 RTX 4070 Ti server；simulation 採用 12 顆衛星的軌道與五節點 compute 設定、0.5 Gbps ISL、ViT-B/L/H/G、EuroSAT 與 RESISC45。",
      "baselines_en": "Ground-only, single-satellite, uniform splitting, compute-proportional heuristic splitting and Top-k compression.",
      "baselines_zh": "Ground-only、single-satellite、uniform splitting、依運算量分配的 heuristic splitting，以及 Top-k compression。",
      "result_en": "The abstract reports up to 42% delay reduction and approximately 71% communication reduction. Accuracy tables use 80% sparsity and 8-bit quantization; 194 of 200 split configurations stay within one percentage point of baseline, with six configurations losing 1.0–1.5 points.",
      "result_zh": "abstract 報告最高 42% delay reduction 與約 71% communication reduction；accuracy table 使用 80% sparsity 與 8-bit quantization；200 組 split 中 194 組的 accuracy loss 維持於一個百分點內，六組 loss 為 1.0–1.5 個百分點。",
      "limitations_en": "The measured workload is image classification. Autoregressive decode latency, KV migration, transport competition and orbital topology changes require additional validation.",
      "limitations_zh": "量測 workload 為 image classification；autoregressive decode latency、KV migration、transport competition 與 orbital topology change 需要額外驗證。",
      "figure_caption_en": "The collaborative pipeline splits a model across orbital compute stages and ground assistance; the full evaluation uses remote-sensing ViT image classification and activation compression.",
      "figure_caption_zh": "協作 pipeline 將 model 分配至軌道 compute stage 與地面協助節點；完整評估使用遙測 ViT 影像分類與 activation compression。",
      "evidence_status": "Full primary text reviewed",
      "mechanism_scope": "Satellite vision inference",
      "deployment_regime": "D",
      "space_ground_path": false,
      "display_regime_en": "D · Orbital compute mesh",
      "display_regime_zh": "D · 軌道運算 mesh",
      "primary_pdf_sha256": "6999e81f4e54cc02426667bd08288eb1d9ed22a2bdb844fd21a4c927bf4c2b54",
      "figure_asset": "../figures/collaborative-llm.png",
      "figure_acquisition": "Primary PDF crop",
      "figure_content_type": "image/png",
      "figure_bytes": 146893,
      "figure_sha256": "ab09a8d5c41027592eee8b7f61b2a8944ef29cb0884f0aed69d525ddd5b7aa9c"
    },
    {
      "id": "cost-network",
      "title": "The Cost and Network Limits of Space-Based AI Compute",
      "authors": "Kees van Berkel",
      "year": "2026",
      "venue": "arXiv",
      "url": "https://arxiv.org/abs/2607.14172",
      "pdf_url": "https://arxiv.org/pdf/2607.14172v1",
      "regime": "D · Orbital compute mesh",
      "evidence": [
        "§III pp3–5",
        "§IV/Table3/Eq2–3 pp5–7",
        "§V/Fig8 p7"
      ],
      "figure_number": 8,
      "figure_page_1based": 7,
      "figure_bbox_points": [
        44,
        44,
        296,
        221
      ],
      "built_for_en": "A power-matched 8000-rack terrestrial facility and 8000-satellite orbital fabric support a 1T-parameter workload.",
      "built_for_zh": "功率對齊的 8000-rack 地面設施與 8000-satellite orbital fabric 支援 1T-parameter workload。",
      "problem_en": "Sparse torus cuts can restrict gradient synchronization despite plentiful accelerator compute.",
      "problem_zh": "即使 accelerator compute 充足，稀疏 torus 的 cut capacity 仍會限制 gradient synchronization。",
      "design_en": "An analytical comparison connects network bisection bandwidth and workload bisection intensity through a roofline model.",
      "design_zh": "分析比較將 network bisection bandwidth 與 workload bisection intensity 連結至 roofline model。",
      "model_en": "Cut capacity, arithmetic intensity, compute rooflines and collective latency/bandwidth estimates.",
      "model_zh": "cut capacity、arithmetic intensity、compute roofline，以及 collective latency/bandwidth estimate。",
      "evaluation_en": "A 1 GW analytical design point uses Clos, 2D torus and 3D torus; 100 Gbps orbital ISLs and a 10 Tbps sensitivity case.",
      "evaluation_zh": "1 GW 分析設計點使用 Clos、2D torus 與 3D torus；orbital ISL 採用 100 Gbps，並分析 10 Tbps 的 sensitivity case。",
      "baselines_en": "Terrestrial Clos with 800 Gbps links and 2:1 oversubscription; orbital 2D/3D torus configurations.",
      "baselines_zh": "採用 800 Gbps link 與 2:1 oversubscription 的地面 Clos，以及 orbital 2D/3D torus configuration。",
      "result_en": "Table 3 assigns 28800 TB/s bisection bandwidth to Clos, 2.25 TB/s to orbital 2D torus, and 10 TB/s to orbital 3D torus at 100 Gbps per ISL.",
      "result_zh": "Table3 在每條 ISL 為 100 Gbps 下，指定 Clos 為 28800 TB/s bisection bandwidth、orbital 2D torus 為 2.25 TB/s、orbital 3D torus 為 10 TB/s。",
      "limitations_en": "The model assumes a particular global data-parallel traffic mapping and theoretical torus links. Application cut traffic, feasible optical wrap-around links, terminal count, topology alternatives and measured training steps determine transferability.",
      "limitations_zh": "模型假設特定 global data-parallel traffic mapping 與理論 torus link；application cut traffic、可行 optical wrap-around link、terminal count、替代 topology，以及實測 training step 決定結果的適用範圍。",
      "figure_caption_en": "The roofline sensitivity compares modeled orbital network throughput and compute ceilings as link capacity increases; topology degree and cut-crossing traffic define the applicable training scenario.",
      "figure_caption_zh": "roofline sensitivity 比較 link capacity 增加時的軌道 network throughput 模型與 compute ceiling；topology degree 與 cut-crossing traffic 界定適用的 training scenario。",
      "evidence_status": "Full primary text reviewed",
      "mechanism_scope": "Orbital fabric feasibility",
      "deployment_regime": "D",
      "space_ground_path": false,
      "display_regime_en": "D/B/T · Feasibility comparison",
      "display_regime_zh": "D／B／T · 可行性比較",
      "primary_pdf_sha256": "fa97b0782b43aeeb739f47360d9f0624a841479d0bd3be7966c07f789b8af3f7",
      "figure_asset": "../figures/cost-network.png",
      "figure_acquisition": "Primary PDF crop",
      "figure_content_type": "image/png",
      "figure_bytes": 323607,
      "figure_sha256": "dbc01565b7f641e64cab71f8bc1cda0c35c4b99e5d168531c224502c490ebd5f"
    },
    {
      "id": "connectivity",
      "title": "Enabling Space Datacenter Connectivity via Satellite Constellations",
      "authors": "Louis Barbier; Oana Hotescu; Emmanuel Lochin; Jérôme Lacan",
      "year": "2026",
      "venue": "LEO-NET workshop",
      "url": "https://doi.org/10.1145/3789240.3827590",
      "pdf_url": "https://hal.science/hal-05700217/document",
      "regime": "C · Space–ground path",
      "evidence": [
        "§3–4/Fig3 pp2–4",
        "§5.1/Table2 pp4–5",
        "§5.2/Fig4–6 pp5–6"
      ],
      "figure_number": 3,
      "figure_page_1based": 3,
      "figure_bbox_points": [
        322,
        86,
        570,
        174
      ],
      "built_for_en": "Sun-synchronous compute clusters reach terrestrial clients through a separate Walker relay constellation.",
      "built_for_zh": "sun-synchronous compute cluster 透過獨立的 Walker relay constellation 連接地面 client。",
      "problem_en": "A narrow dawn-dusk orbital footprint creates intermittent access to globally distributed ground stations.",
      "problem_zh": "狹窄的 dawn-dusk orbital footprint 導致全球 ground station 的 access 呈現間歇性。",
      "design_en": "A hierarchical gateway shell connects SSO clusters with Starlink/OneWeb-like relays. Longest-duration feasible optical contacts guide link selection.",
      "design_zh": "階層式 gateway shell 連接 SSO cluster 與 Starlink/OneWeb 類型 relay；以可行 optical contact 的最長 duration 引導 link selection。",
      "model_en": "Temporal graph edges satisfy LoS, range, angular tracking, field-of-regard and minimum-duration conditions; Dijkstra routing, betweenness, path entropy, Jaccard overlap and route-change rate quantify geometry.",
      "model_zh": "temporal graph edge 滿足 LoS、range、angular tracking、field-of-regard 與 minimum-duration 條件；Dijkstra routing、betweenness、path entropy、Jaccard overlap 與 route-change rate 衡量幾何特性。",
      "evaluation_en": "A 24-hour simulation evaluates an 81-node SSO relay shell and 55 ground stations; alternatives include Starlink G7 (3360), G2 (720), and OneWeb (588).",
      "evaluation_zh": "24-hour simulation 評估 81-node SSO relay shell 與 55 個 ground station；替代 relay 包含 Starlink G7（3360）、G2（720）與 OneWeb（588）。",
      "baselines_en": "SSO-only, SSO+Starlink G7, SSO+Starlink G2, SSO+OneWeb; ground-to-SSO and Walker-to-SSO path experiments.",
      "baselines_zh": "SSO-only、SSO+Starlink G7、SSO+Starlink G2、SSO+OneWeb；ground-to-SSO 與 Walker-to-SSO path experiment。",
      "result_en": "Starlink G7 median inter-shell link duration stays below ten minutes; more than 90% of OneWeb links persist beyond twenty minutes. G2 provides the strongest path stability in the tested geometry.",
      "result_zh": "Starlink G7 的 median inter-shell link duration 低於十分鐘；超過 90% 的 OneWeb link 維持二十分鐘以上；在測試幾何下，G2 提供最強的 path stability。",
      "limitations_en": "All Walker nodes are relay-capable, yielding a connectivity upper bound. Ground-station competition, offered traffic, packet queues and LLM service metrics require an emulator extension.",
      "limitations_zh": "所有 Walker node 均具 relay 能力，形成 connectivity upper bound；ground-station competition、offered traffic、packet queue 與 LLM service metric 需要 emulator extension。",
      "figure_caption_en": "The architecture separates orbital compute clusters from relay satellites and ground stations; the 81-satellite SSO case represents a gateway shell connecting distributed clusters.",
      "figure_caption_zh": "架構區分軌道 compute cluster、relay satellite 與 ground station；81 顆 SSO 衛星的案例表示連接分散 cluster 的 gateway shell。",
      "evidence_status": "Full primary text reviewed",
      "mechanism_scope": "Orbital DC access",
      "deployment_regime": "C",
      "space_ground_path": true,
      "display_regime_en": "C · Space–ground path",
      "display_regime_zh": "C · 天地路徑",
      "primary_pdf_sha256": "a804af399eaa99082f8111e4bd378be3183b17f74754e876518f8c29f6c0fbb3",
      "figure_asset": "../figures/connectivity.png",
      "figure_acquisition": "Primary PDF crop",
      "figure_content_type": "image/png",
      "figure_bytes": 47209,
      "figure_sha256": "35404403b4e1295db92f870c7c7c22380778a4d0aae774eb9a48290e6350f37d"
    },
    {
      "id": "dark-clouds",
      "title": "Dark Clouds Rising in Low-Earth Orbit: On Environmental Limits to Massive Orbital AI",
      "authors": "Robin Ohs; Gregory F. Stock; Andreas Schmidt; Juan A. Fraire; Jörg Ott; Holger Hermanns",
      "year": "2026",
      "venue": "LEO-NET workshop",
      "url": "https://doi.org/10.1145/3789240.3827598",
      "pdf_url": "https://hal.science/hal-05671481/document",
      "regime": "B · Tight AI formation",
      "evidence": [
        "HALv2 paper§3/Table1 PDFp3–4",
        "§4/Table2 PDFp5",
        "Fig1 PDFp6",
        "§5 PDFp6–7"
      ],
      "figure_number": 1,
      "figure_page_1based": 6,
      "figure_bbox_points": [
        52,
        84,
        570,
        292
      ],
      "built_for_en": "A one-H100 orbital compute node couples electrical power, radiator mass, solar area and mission lifetime.",
      "built_for_zh": "搭載一個 H100 的 orbital compute node 連結 electrical power、radiator mass、solar area 與 mission lifetime。",
      "problem_en": "Lifecycle analysis must account for power infrastructure, waste-heat rejection and payload redundancy.",
      "problem_zh": "lifecycle analysis 需要納入 power infrastructure、waste-heat rejection 與 payload redundancy。",
      "design_en": "ESpaS-ODC adds ISS-calibrated radiator sizing, solar degradation, eclipse margins and shared-infrastructure cold spares.",
      "design_zh": "ESpaS-ODC 加入經 ISS 校準的 radiator sizing、solar degradation、eclipse margin，以及共享 infrastructure 的 cold spare。",
      "model_en": "A component mass/carbon ledger uses eclipse geometry, linear thermal scaling and carbon amortization over GPU service hours.",
      "model_zh": "component mass/carbon ledger 使用 eclipse geometry、linear thermal scaling，以及 GPU service hour 的 carbon amortization。",
      "evaluation_en": "A 510 km analytical reference node uses a 700 W H100 plus 300 W peripherals, ideal high-beta sunlight, mission durations 1–9years and redundancy 1–5.",
      "evaluation_zh": "510 km 分析 reference node 使用 700 W H100 加上 300 W peripheral、理想 high-beta sunlight、1–9year mission duration 與 1–5 redundancy。",
      "baselines_en": "Global-average terrestrial DC and Finland-grid green DC; Starship and Falcon9 launch assumptions.",
      "baselines_zh": "global-average terrestrial DC 與 Finland-grid green DC；Starship 與 Falcon9 launch assumption。",
      "result_en": "The reported reference case has 44.3 kg modeled component mass, including 30.5 kg radiator (68.8%). One full cold spare increases 3-year carbon per GPU hour by 40% under Starship and 34% under Falcon9.",
      "result_zh": "報告 reference case 的 modeled component mass 為 44.3 kg，包含 30.5 kg radiator（68.8%）；一組完整 cold spare 在三年期下，使 Starship 假設的 carbon per GPU hour 提升 40%，Falcon9 假設提升 34%。",
      "limitations_en": "The component ledger covers an ideal 100% duty cycle and a50W transceiver placeholder. Bus, harness, shielding and thermal transport require additional mass accounting; reliability benefits require measured failure rates.",
      "limitations_zh": "component ledger 涵蓋理想 100% duty cycle 與 50 W transceiver placeholder；bus、harness、shielding 與 thermal transport 需要額外 mass accounting，reliability benefit 需要實測 failure rate。",
      "figure_caption_en": "The lifecycle-carbon heatmap sweeps mission duration and cold-spare redundancy for a modeled 1-kW orbital compute unit under Starship and Falcon-9 launch scenarios. Colors and terrestrial thresholds share the plotted model boundary.",
      "figure_caption_zh": "生命週期 carbon heatmap 針對模型中的 1-kW 軌道 compute unit，在 Starship 與 Falcon-9 發射情境下掃描 mission duration 與 cold-spare redundancy；色彩與地面 threshold 共用圖中的模型邊界。",
      "evidence_status": "Full primary text reviewed",
      "mechanism_scope": "Orbital physical budget",
      "deployment_regime": "B",
      "space_ground_path": false,
      "display_regime_en": "B/D · Physical feasibility",
      "display_regime_zh": "B／D · 物理可行性",
      "primary_pdf_sha256": "65afb8e017e43c1991edbb4fc557c39454af3b8a353a1a928abb02b6c6a38c39",
      "figure_asset": "../figures/dark-clouds.png",
      "figure_acquisition": "Primary PDF crop",
      "figure_content_type": "image/png",
      "figure_bytes": 138636,
      "figure_sha256": "34c1344e96f207479f98947b9e9697dc9ddfd6347600b9b48a9b71e9f01da0e4"
    },
    {
      "id": "phoenix",
      "title": "In-Orbit Processing or Not? Sunlight-Aware Task Scheduling for Energy-Efficient Space Edge Computing Networks",
      "authors": "Weisen Liu; Zeqi Lai; Qian Wu; Hewu Li; Qi Zhang; Zonglun Li; Yuanjie Li; Jun Liu",
      "year": "2024",
      "venue": "INFOCOM",
      "url": "https://doi.org/10.1109/INFOCOM52122.2024.10621268",
      "pdf_url": "https://arxiv.org/pdf/2407.07337",
      "regime": "E · Observation/contact fleet",
      "figure_number": 5,
      "figure_page_1based": 4,
      "figure_bbox_points": [
        55,
        50,
        300,
        235
      ],
      "evidence": [
        "§III-C p4; Fig5",
        "§IV-A p4–5",
        "§IV-B p5–6; Algorithms1–3",
        "§V-A p7–8",
        "§V-B p8–9; Figures7–9"
      ],
      "built_for_en": "Earth-observation tasks choose ground processing, local satellite compute, or sunlit peer satellites.",
      "built_for_zh": "地球觀測任務選擇地面處理、本地衛星運算，或處於日照區的同儕衛星。",
      "problem_en": "Eclipse execution increases battery depth of discharge; task deadlines limit opportunities to defer computation until sunlight.",
      "problem_zh": "蝕期執行增加電池放電深度；任務 deadline 限制將運算延後至日照區的機會。",
      "design_en": "A mission controller assigns orbital subsets by sunlight budget. Onboard managers choose deadline-feasible ground, local sunlit, or peer execution and arrange local work by deadline and sunlight.",
      "design_zh": "mission controller 依日照預算分配軌道子集；星上 manager 選擇滿足 deadline 的地面、本地日照區或同儕執行，並依 deadline 與日照安排本地工作。",
      "model_en": "A slotted visibility graph and binary task variables minimize average battery energy subject to deadlines. Generalized-assignment hardness motivates a decomposition into knapsack orbit assignment, orbit-based offloading, and processing arrangement.",
      "model_zh": "時槽化可見性 graph 與二元任務變數在 deadline 限制下最小化平均電池能耗；generalized assignment 的複雜度促成 knapsack 軌道分配、orbit-based offloading 與處理安排的分解。",
      "evaluation_en": "StarryNet containers combine a Dell workstation with a Jetson AGX Orin hardware-in-the-loop node; Starlink and OneWeb orbital configurations, SatNOGS ground stations, four seasons, ship detection, and wildfire segmentation. Compute uses 30/50/60 W; GSL/ISL use 100 Mbps/1 Gbps and 16/10 W; solar and battery use 120 W and 60 Wh.",
      "evaluation_zh": "StarryNet container 結合 Dell workstation 與 Jetson AGX Orin hardware-in-the-loop 節點；涵蓋 Starlink、OneWeb 軌道組態、SatNOGS 地面站、四季、船隻偵測與野火分割。運算功率為 30/50/60 W；GSL/ISL 為 100 Mbps/1 Gbps 與 16/10 W；太陽能與電池為 120 W、60 Wh。",
      "baselines_en": "OEC, MHSPO, and L2D2 compare orbital processing, peer offloading, and direct ground delivery.",
      "baselines_zh": "OEC、MHSPO、L2D2 比較軌道處理、同儕 offloading 與直接地面傳輸。",
      "result_en": "Figure 7 reports up to 54.8% lower maximum depth of discharge. Figure 8 estimates battery lifespan up to 2.9× versus OEC and 5.3× versus MHSPO under the Starlink configuration.",
      "result_zh": "Figure 7 報告最大放電深度降低最高 54.8%；Figure 8 在 Starlink 組態下估計電池壽命最高為 OEC 的 2.9×、MHSPO 的 5.3×。",
      "limitations_en": "Battery lifespan follows a depth-of-discharge model calibrated from prior studies. The prototype measures terrestrial compute and emulates orbital connectivity; radiator, radiation faults, and tightly coupled collectives require additional models.",
      "limitations_zh": "電池壽命依據既有研究校準的放電深度模型；prototype 量測地面運算並模擬軌道連線；radiator、輻射故障與密切耦合 collective 需要追加模型。",
      "figure_caption_en": "PHOENIX combines a ground mission controller with onboard task managers. Sunlight prediction guides orbital subset assignment and deadline-feasible ground, local, or peer processing.",
      "figure_caption_zh": "PHOENIX 結合地面 mission controller 與星上 task manager；sunlight prediction 引導軌道子集分配，以及符合 deadline 的地面、本地或同儕處理。",
      "source_literal_fields": [
        "title"
      ],
      "evidence_status": "Full primary text reviewed",
      "mechanism_scope": "Wide-area edge compute",
      "deployment_regime": "E",
      "space_ground_path": true,
      "display_regime_en": "E · Observation/contact fleet",
      "display_regime_zh": "E · 觀測與 contact 艦隊",
      "primary_pdf_sha256": "3f96c3b67e63434ac2ce4fc448fffb0fc6079fe361646c125496798565a86c96",
      "figure_asset": "../figures/phoenix.png",
      "figure_acquisition": "Primary PDF crop",
      "figure_content_type": "image/png",
      "figure_bytes": 223829,
      "figure_sha256": "a6bf0d12c47c6b52dfd1534ca984d5f56c44744282626bac020c96aa92872941"
    },
    {
      "id": "spacesched",
      "title": "SpaceSched: A Constellation-Wide Scheduling System for Resolving Ground Track Congestion in Remote Sensing",
      "authors": "Zehua Sun; Tao Ni; Pengfei Hu; Tao Gu; Weitao Xu",
      "year": "2025",
      "venue": "MobiCom",
      "url": "https://doi.org/10.1145/3680207.3765249",
      "pdf_url": "https://taoni.xyz/taoni_files/paper/mobicom25_spacesched.pdf",
      "regime": "E · Observation/contact fleet",
      "figure_number": 6,
      "figure_page_1based": 5,
      "figure_bbox_points": [
        54,
        83,
        300,
        176
      ],
      "evidence": [
        "§3 p5; Fig6",
        "§4 p5–6",
        "§5 p6–8",
        "§6 p8–9",
        "§7.1 p9–10",
        "§7.2 p10–11; Figures10–11"
      ],
      "built_for_en": "Constellation-wide remote sensing coordinates camera attitude, active spacecraft, and prioritized image delivery.",
      "built_for_zh": "全星座遙測協調相機姿態、活躍衛星與影像傳輸優先級。",
      "problem_en": "Overlapping observation footprints concentrate coverage and create redundant imagery, satellite use, and downlink backlog.",
      "problem_zh": "重疊觀測 footprint 集中覆蓋，增加重複影像、衛星使用量與 downlink backlog。",
      "design_en": "TLE time compensation and attitude calibration feed a coverage distributor and satellite selector on the ground. An onboard two-class queue prioritizes imagery from regions below the coverage-map 25th percentile.",
      "design_zh": "TLE 時間補償與姿態校準提供地面 coverage distributor、satellite selector；星上雙類 queue 優先傳送 coverage map 第 25 百分位以下區域的影像。",
      "model_en": "Geographic coverage grids, attitude profiles, hard-soft penalty objectives, genetic population search, and binary subset selection balance coverage gain, active count, and attitude variation.",
      "model_zh": "地理覆蓋 grid、姿態 profile、hard-soft penalty objective、genetic population search 與二元子集選擇共同平衡覆蓋增益、活躍數量與姿態變化。",
      "evaluation_en": "CelesTrak TLEs drive 17 SKYSAT, 50 LEMUR, and 126 FLOCK satellite models. Ground optimization uses an i7/RTX 3080 workstation; onboard scheduling uses Jetson TX2. The study covers a North American region, stripmap and spotlight modes, and 6/12/24-hour horizons.",
      "evaluation_zh": "CelesTrak TLE 驅動 17 顆 SKYSAT、50 顆 LEMUR、126 顆 FLOCK 的模型；地面最佳化使用 i7/RTX 3080 workstation，星上 scheduling 使用 Jetson TX2；研究涵蓋北美區域、stripmap 與 spotlight 模式，以及 6/12/24 小時 horizon。",
      "baselines_en": "The plain constellation uses zero off-nadir angle, the entire satellite set, and a sequential downlink queue. Component sensitivity includes population generations, attitude ranges, step sizes, and coverage tolerance.",
      "baselines_zh": "plain constellation 使用零 off-nadir angle、全部衛星與循序 downlink queue；元件敏感度涵蓋 population generation、姿態範圍、step size 與覆蓋容忍度。",
      "result_en": "Figure 10 gives coverage 52.38/84.24/98.93% versus 29.24/55.17/65.63%, active counts 11/31/53 versus 17/50/126, and data-load reductions 15.83–23.43×. The SKYSAT horizon sweep in Figure 11 reaches 1.84× coverage and 36.46× lower data load.",
      "result_zh": "Figure 10 的覆蓋為 52.38/84.24/98.93%，對照 29.24/55.17/65.63%；活躍數量為 11/31/53，對照 17/50/126；資料負載降低 15.83–23.43×。Figure 11 的 SKYSAT horizon sweep 達到 1.84× 覆蓋與 36.46× 較低資料負載。",
      "limitations_en": "Ground-track congestion measures observation-footprint overlap. The load metric is the imagery fraction needed to attain 95% expected coverage; packet latency and transport goodput require separate measurements. The evaluated baseline is a plain constellation.",
      "limitations_zh": "ground-track congestion 量測觀測 footprint 重疊；load metric 為達到 95% 預期覆蓋所需的影像比例；packet latency 與 transport goodput 需要各自量測。評估 baseline 為 plain constellation。",
      "figure_caption_en": "SpaceSched couples TLE-based constellation coordination, ground coverage optimization, satellite subset selection, and onboard prioritized image queues. The target is overlapping observation coverage and imagery delivery.",
      "figure_caption_zh": "SpaceSched 串接 TLE-based 星座協調、地面覆蓋最佳化、衛星子集選擇與星上優先影像 queue；處理對象為重疊觀測覆蓋與影像傳輸。",
      "evidence_status": "Full primary text reviewed",
      "mechanism_scope": "Remote-sensing scheduling",
      "deployment_regime": "E",
      "space_ground_path": true,
      "display_regime_en": "E · Observation/contact fleet",
      "display_regime_zh": "E · 觀測與 contact 艦隊",
      "primary_pdf_sha256": "fc5570a5ddb7d144f29b55aa64300a10b5dd852b8cc0048f6eca16eee5cbe946",
      "figure_asset": "../figures/spacesched.png",
      "figure_acquisition": "Primary PDF crop",
      "figure_content_type": "image/png",
      "figure_bytes": 105649,
      "figure_sha256": "49752afa0f5084531270695946b930006d2fd1fe5d95f1519b255f1c5d7f3646"
    },
    {
      "id": "satpipe",
      "title": "SATPIPE: Deterministic TCP Adaptation for Highly Dynamic LEO Satellite Networks",
      "authors": "Ding Zhao; Xinyu Zhang; Myungjin Lee",
      "year": "2025",
      "venue": "INFOCOM",
      "url": "https://doi.org/10.1109/INFOCOM55648.2025.11044600",
      "pdf_url": "https://xyzhang.ucsd.edu/papers/Ding.Zhao_INFOCOM25_SatPipe.pdf",
      "regime": "A · Communication constellation",
      "figure_number": 11,
      "figure_page_1based": 5,
      "figure_bbox_points": [
        343,
        305,
        547,
        422
      ],
      "evidence": [
        "§III p3–4; Figures3–8",
        "§IV p5–7; Fig11; Algorithms1–2",
        "§V-A p7–8",
        "§V-B/C p8–9; Figures14–20"
      ],
      "built_for_en": "Sender-side TCP adaptation improves bulk transfers and video streaming over consumer Starlink access.",
      "built_for_zh": "sender-side TCP adaptation 改善 consumer Starlink access 的 bulk transfer 與影片串流。",
      "problem_en": "Periodic service interruptions create excess in-flight queues and inaccurate BBR bandwidth estimates, delaying recovery after link service resumes.",
      "problem_zh": "週期性服務中斷產生過量 in-flight queue 與偏移的 BBR bandwidth estimate，延長 link 恢復服務後的回復時間。",
      "design_en": "A BBR-derived state machine enters Queue Maintenance at predicted interruption times, temporarily sets CWND to zero, and forces RTT probing during recovery. NTP timing or ACK inter-packet-delay histograms estimate the 15-second phase.",
      "design_zh": "以 BBR 為基礎的 state machine 在預測中斷時間進入 Queue Maintenance，暫時將 CWND 設為零，並在恢復階段觸發 RTT probing；NTP 時間或 ACK inter-packet-delay histogram 估計 15 秒週期相位。",
      "model_en": "Deterministic phase estimation modulo 15 seconds, transport state machines, bandwidth-delay product, and application QoE functions connect interruption timing to queues and bitrate.",
      "model_zh": "modulo 15 秒的確定性相位估計、transport state machine、bandwidth-delay product 與 application QoE function 串接中斷時間、queue 與 bitrate。",
      "evaluation_en": "Live Gen-2 Starlink terminal in San Diego connects by 1 Gbps Ethernet to a Linux client. Six AWS servers cover North California, Oregon, Ohio, London, Singapore, and Canada; Linux 6.8.10, iperf3, competing flows, and modified dash.js deliver TCP and Big Buck Bunny DASH measurements.",
      "evaluation_zh": "聖地牙哥的實際 Gen-2 Starlink terminal 經 1 Gbps Ethernet 連接 Linux client；六個 AWS server 位於北加州、Oregon、Ohio、London、Singapore、Canada；Linux 6.8.10、iperf3、競爭 flow 與修改版 dash.js 提供 TCP 及 Big Buck Bunny DASH 量測。",
      "baselines_en": "Reno, CUBIC, BBR, and SaTCP; fairness, RTT, retransmission, throughput percentiles, bitrate, and rebuffering are evaluated.",
      "baselines_zh": "Reno、CUBIC、BBR、SaTCP；量測 fairness、RTT、retransmission、throughput percentile、bitrate 與 rebuffering。",
      "result_en": "Average throughput improves 9.4–38.2% over BBR across tested paths; 10th-percentile throughput improves up to 127.8%. The paper reports 24.7% lower retransmission ratio, 10.8% higher video bitrate, and 33.5% shorter rebuffering.",
      "result_zh": "測試路徑的平均 throughput 相對 BBR 改善 9.4–38.2%；第 10 百分位 throughput 最高改善 127.8%；論文報告 retransmission ratio 降低 24.7%、影片 bitrate 提升 10.8%、rebuffering 縮短 33.5%。",
      "limitations_en": "The 15-second timing describes measured Starlink access behavior. Other constellations and orbital cluster ISLs require timing calibration. Sender-visible traces identify interruption effects; spacecraft reassociation attribution benefits from operator telemetry.",
      "limitations_zh": "15 秒時間描述量測到的 Starlink access 行為；其他星座與軌道 cluster ISL 需要 timing calibration。sender-visible trace 提供中斷效應證據；衛星 reassociation 的歸因可由 operator telemetry 補強。",
      "figure_caption_en": "The state-machine comparison shows SatPipe adding Queue Maintenance to BBR. Interruption timing triggers sender restraint and RTT probing to restore accurate pacing after service resumes.",
      "figure_caption_zh": "state-machine 對照呈現 SatPipe 在 BBR 加入 Queue Maintenance；中斷時間觸發 sender restraint 與 RTT probing，於服務恢復後重建準確 pacing。",
      "evidence_status": "Full primary text reviewed",
      "mechanism_scope": "Live LEO access transport",
      "deployment_regime": "A",
      "space_ground_path": true,
      "display_regime_en": "A · Communication constellation",
      "display_regime_zh": "A · 通訊星座",
      "primary_pdf_sha256": "f59949bd50317d8998a889621179cbb133bf4f8385bae13ccfedd10e4b330de7",
      "figure_asset": "../figures/satpipe.png",
      "figure_acquisition": "Primary PDF crop",
      "figure_content_type": "image/png",
      "figure_bytes": 23959,
      "figure_sha256": "647df9ed25c6da22b9c676e1c3963fde0bf85961095b7bc451f52b2f2efe8ccd"
    },
    {
      "id": "secure-ntn",
      "title": "Secure Task Offloading and Resource Allocation Design for Multi-Layer Non-Terrestrial Networks",
      "authors": "Alejandro Flores; Isabella W. G. da Silva; Vu Nguyen Ha; Konstantinos Ntontin; Hien Quoc Ngo; Michail Matthaiou; Symeon Chatzinotas",
      "year": "2026",
      "venue": "INFOCOM",
      "url": "https://orbilu.uni.lu/handle/10993/68694",
      "pdf_url": "https://arxiv.org/pdf/2602.17192",
      "regime": "D · Orbital compute mesh",
      "figure_number": 1,
      "figure_page_1based": 2,
      "figure_bbox_points": [
        318,
        46,
        559,
        199
      ],
      "evidence": [
        "§II p2–3; Fig1",
        "§III p4–5; Algorithm1; Equations20–32",
        "§IV p5–6; Figures2–4",
        "Institutional publication metadata: INFOCOM2026; DOI10.1109/INFOCOM59046.2026.11571748"
      ],
      "built_for_en": "Remote IoT tasks reach satellite MEC through UAV relays and a HAPS coordinator that authenticates requests.",
      "built_for_zh": "遠端 IoT task 經 UAV relay 與負責認證 request 的 HAPS coordinator 送往衛星 MEC。",
      "problem_en": "Malicious offload requests consume limited orbital compute; cryptographic authentication overhead competes with stringent task deadlines.",
      "problem_zh": "惡意 offload request 消耗受限的軌道 compute；cryptographic authentication overhead 與嚴格 task deadline 競爭。",
      "design_en": "Secret-key signal tags support physical-layer authentication at HAPS. Admitted tasks enter joint satellite selection and compute-share optimization solved through alternating offload and resource-allocation subproblems.",
      "design_zh": "secret-key signal tag 支援 HAPS 的 physical-layer authentication；通過認證的 task 進入聯合衛星選擇與 compute-share 最佳化，交替求解 offload 與 resource-allocation 子問題。",
      "model_en": "Gaussian detection statistics determine false-alarm and detection probabilities. A min-max normalized-delay mixed-integer formulation relaxes selection into penalized linear programs and solves convex CPU allocations by block coordinate descent.",
      "model_zh": "Gaussian detection statistic 決定 false-alarm 與 detection probability；min-max normalized-delay mixed-integer formulation 將選擇變數放寬為帶 penalty 的 linear program，並透過 block coordinate descent 求解 convex CPU allocation。",
      "evaluation_en": "Numerical experiments use 50 legitimate and 25 malicious devices across 10×10 km², 25 UAVs, and 3 visible LEO servers with 10 Gcycles/s each. Tasks contain 10 Kb and 200 cycles/bit; IoT access uses 200 kHz at 2.1 GHz, HAPS–LEO uses 100 MHz at 28 GHz.",
      "evaluation_zh": "數值實驗使用 10×10 km² 區域中的 50 個合法裝置、25 個惡意裝置、25 個 UAV，以及各具 10 Gcycles/s 的 3 顆可見 LEO server；task 為 10 Kb、200 cycles/bit；IoT access 使用 2.1 GHz 的 200 kHz channel，HAPS–LEO 使用 28 GHz 的 100 MHz channel。",
      "baselines_en": "BP-PK bilinear pairing, ID-BC blockchain, and MSR-BC blockchain use authentication computation times from cited prior work.",
      "baselines_zh": "BP-PK bilinear pairing、ID-BC blockchain、MSR-BC blockchain 使用引用研究的 authentication computation time。",
      "result_en": "Figure 2 shows earlier deadline feasibility than the compared cryptographic schemes. Figure 3 reports legitimate admission above 95% with 25 UAVs under stringent false-alarm constraints and malicious admission around 1% at a 0.05 false-alarm target.",
      "result_zh": "Figure 2 顯示相對比較的 cryptographic scheme 更早達到 deadline feasibility；Figure 3 在 25 個 UAV 與嚴格 false-alarm 限制下報告超過 95% 的合法 admission，且 false-alarm target 為 0.05 時惡意 admission 約 1%。",
      "limitations_en": "The evaluation studies numerical channel, authentication, and compute-allocation models for small remote tasks. Coherent reception, channel knowledge, and solver placement form explicit assumptions; orbital AI collectives require workload-scale and transport validation.",
      "limitations_zh": "評估針對小型遠端 task 的 channel、authentication 與 compute-allocation 數值模型；coherent reception、channel knowledge 與 solver placement 構成明確假設；orbital AI collective 需要 workload-scale 與 transport validation。",
      "figure_caption_en": "The four-layer path carries IoT requests through UAV relays and a HAPS authentication coordinator to three LEO compute servers; the studied traffic consists of small remote offloading tasks.",
      "figure_caption_zh": "四層路徑將 IoT request 經 UAV relay 與 HAPS authentication coordinator 傳至三個 LEO compute server；研究流量為小型遠端 offloading task。",
      "source_literal_fields": [
        "title"
      ],
      "evidence_status": "Full primary text reviewed",
      "mechanism_scope": "Multi-layer remote task offload",
      "deployment_regime": "D",
      "space_ground_path": false,
      "display_regime_en": "D · Orbital compute mesh",
      "display_regime_zh": "D · 軌道運算 mesh",
      "primary_pdf_sha256": "ac6d5a1254b35924fe71e43cb2e66dd0fee06321126ab9dceb9367a0778c4f66",
      "figure_asset": "../figures/secure-ntn.png",
      "figure_acquisition": "Primary PDF crop",
      "figure_content_type": "image/png",
      "figure_bytes": 89388,
      "figure_sha256": "d8e8f3801d4206c8e848002fce554a552a6bbf098a4f024371cf765fc5956234"
    },
    {
      "id": "orbitalbrain",
      "title": "OrbitalBrain: A Distributed Framework for Training ML Models in Space",
      "authors": "Om Chabra; Chenning Li; Kevin Hsieh; Santiago Segarra; Behnaz Arzani; Peder Olsen; Ranveer Chandra",
      "year": "2026",
      "venue": "NINeS",
      "url": "https://drops.dagstuhl.de/entities/document/10.4230/OASIcs.NINeS.2026.5",
      "pdf_url": "https://drops.dagstuhl.de/storage/01oasics/oasics-vol139-nines2026/OASIcs.NINeS.2026.5/OASIcs.NINeS.2026.5.pdf",
      "regime": "E · Observation/contact fleet",
      "figure_number": 4,
      "figure_page_1based": 8,
      "figure_bbox_points": [
        106,
        99,
        507,
        356
      ],
      "evidence": [
        "§2.1 p4; Fig1",
        "§2.3 p7",
        "§3.1 p8–9; Equation1",
        "§3.2 p9–11; Equation2",
        "§3.3 p11–12; Equations3–5",
        "§3.4 p12–13; Algorithm1",
        "§4 p13–14",
        "§5.1 p14–15",
        "§5.2 p15–17; Tables2–3",
        "§5.3 p17–18; Table4",
        "§5.4 p20–21",
        "AppendixA p31–32; Equations6–9",
        "AppendixB p32"
      ],
      "built_for_en": "Earth-observation satellites incrementally train local image models, exchange weights and selected raw imagery, and contribute updates to a ground global model.",
      "built_for_zh": "地球觀測衛星增量訓練本地影像 model，交換 weight 與選定 raw imagery，並向地面 global model 提供 update。",
      "problem_en": "Limited downlink, energy, and storage combine with skewed geographic labels and stale local models to slow distributed training convergence.",
      "problem_zh": "受限的 downlink、energy、storage 結合偏斜的地理 label distribution 與陳舊本地 model，延長分散訓練 convergence。",
      "design_en": "A cloud planner profiles loss, staleness, predicted compute, and label distributions. It chooses local compute, shortest-path-tree model averaging rooted at the satellite with most ISLs, or utility-ranked raw-data transfer; ground contacts deliver schedules and model updates.",
      "design_zh": "cloud planner 記錄 loss、staleness、預測 compute 與 label distribution；選擇本地 compute、以 ISL 數量最多衛星為 root 的 shortest-path-tree model averaging，或 utility-ranked raw-data transfer；地面 contact 傳送 schedule 與 model update。",
      "model_en": "A binary resource-allocation formulation maximizes accuracy gain under window, energy, and storage constraints. The practical O(S² log S) greedy planner uses loss/staleness utility, a decaying aggregation threshold, and Jensen–Shannon label-divergence reduction. The appendix calls the formulation MILP; its accuracy function requires an explicit linear surrogate for a solver-ready MILP.",
      "model_zh": "二元 resource-allocation formulation 在 window、energy、storage 限制下最大化 accuracy gain；實際 O(S² log S) greedy planner 使用 loss/staleness utility、衰減 aggregation threshold 與 Jensen–Shannon label divergence 改善量。Appendix 將 formulation 稱為 MILP；accuracy function 需要明確 linear surrogate 才能形成 solver-ready MILP。",
      "evaluation_en": "CosmicBeats orbital traces drive FLUTE/OpenMPI learning simulation: Planet 207 and Spire 117 satellites, 12 ground stations, 24 hours, five-minute windows, 100 Mbps bounded ISLs, and 360 GB storage. Final-five-layer DenseNet-161/fMoW and ResNet-50/So2Sat adaptation uses a Jetson Orin Nano 4 GB compute model. Power assumptions include 7 W solar generation, 7.5 W GPU/ISL demand, and 50 W downlink TX.",
      "evaluation_zh": "CosmicBeats 軌道 trace 驅動 FLUTE/OpenMPI learning simulation：Planet 207 顆、Spire 117 顆衛星，12 個地面站、24 小時、五分鐘 window、100 Mbps 上限 ISL、360 GB storage；以 Jetson Orin Nano 4 GB compute model 調整 DenseNet-161/fMoW 與 ResNet-50/So2Sat 的最後五層；power assumption 包含 7 W solar generation、7.5 W GPU/ISL demand、50 W downlink TX。",
      "baselines_en": "BentPipe centralized training, SyncFL, AsyncFL, FedBuff, and FedSpace; component ablations isolate aggregation and raw-data transfer. Ideal references supply balanced labels and unrestricted ground connectivity.",
      "baselines_zh": "BentPipe centralized training、SyncFL、AsyncFL、FedBuff、FedSpace；元件 ablation 各自評估 aggregation 與 raw-data transfer；ideal reference 提供平衡 label 與充分 ground connectivity。",
      "result_en": "Table 3 reports 1.52–12.42× faster attainment of selected baselines’ own 24-hour final accuracy. Table 2 gives OrbitalBrain fMoW accuracy 52.8/59.2% versus BentPipe 47.3/50.2%, and So2Sat 47.9/47.1% versus 46.0/43.0%. Differences are percentage points.",
      "result_zh": "Table3 報告達到選定 baseline 各自 24 小時 final accuracy 的時間加速 1.52–12.42×；Table2 的 OrbitalBrain fMoW accuracy 為 52.8/59.2%，BentPipe 為 47.3/50.2%；So2Sat 為 47.9/47.1%，對照 46.0/43.0%；差值單位為 percentage point。",
      "limitations_en": "Evidence covers trace-driven vision-model adaptation and five-minute scheduling. Training-step GPU calibration, packet queues and loss, optical pointing, radiator budgets, and repeated LLM collectives require dedicated validation. Orbital forecasts are assumed accurate; compression uses zero modeled energy; model/data communication uses bounded link rates.",
      "limitations_zh": "證據涵蓋 trace-driven vision-model adaptation 與五分鐘 scheduling；training-step GPU calibration、packet queue 與 loss、optical pointing、radiator budget，以及重複 LLM collective 需要各自驗證。軌道 forecast 採準確假設；compression 的模型能耗設為零；model/data communication 採 bounded link rate。",
      "figure_caption_en": "OrbitalBrain places its profiler, aggregation planner, data-transfer planner, and executor in the ground cloud. Satellites exchange local weights or imagery over scheduled ISLs; ground contacts relay statistics and schedules.",
      "figure_caption_zh": "OrbitalBrain 將 profiler、aggregation planner、data-transfer planner 與 executor 置於地面 cloud；衛星經排程 ISL 交換本地 weight 或 imagery；ground contact relay statistic 與 schedule。",
      "evidence_status": "Full primary text reviewed",
      "mechanism_scope": "Distributed Earth-observation training",
      "deployment_regime": "E",
      "space_ground_path": true,
      "display_regime_en": "E · Observation/contact fleet",
      "display_regime_zh": "E · 觀測與 contact 艦隊",
      "primary_pdf_sha256": "8e01c52c870545a34cfd5a1d466ab38822d7d6a950fc3c8aa72ba073d8f16421",
      "figure_asset": "../figures/orbitalbrain.png",
      "figure_acquisition": "Primary PDF crop",
      "figure_content_type": "image/png",
      "figure_bytes": 222780,
      "figure_sha256": "5a6d1484059b22a0c828a48174cb54415e575fdd6bbec69d188cae15857f6007"
    },
    {
      "id": "topoopt",
      "title": "TopoOpt: Co-optimizing Network Topology and Parallelization Strategy for Distributed Training Jobs",
      "authors": "Weiyang Wang; Moein Khazraee; Zhizhen Zhong; Manya Ghobadi; Zhihao Jia; Dheevatsa Mudigere; Ying Zhang; Anthony Kewitsch",
      "year": "2023",
      "venue": "NSDI",
      "url": "https://www.usenix.org/conference/nsdi23/presentation/wang-weiyang",
      "pdf_url": "https://www.usenix.org/system/files/nsdi23-wang-weiyang.pdf",
      "regime": "T · Terrestrial reference",
      "figure_number": 5,
      "figure_page_1based": 5,
      "figure_bbox_points": [
        339,
        220,
        537,
        315
      ],
      "evidence": [
        "§3 paperp742/PDFp5; Figure5",
        "§4.1 paperp743/PDFp6; Figure6",
        "§4.2 paperp743–744/PDFp6–7; Algorithm1",
        "§4.3 paperp744–745/PDFp7–8; Algorithms2–3",
        "§5.1 paperp746/PDFp9",
        "§5.3 paperp747–748/PDFp10–11; Figure11",
        "§5.4 paperp749/PDFp12; Equation1",
        "§5.7 paperp750/PDFp13; Figure17",
        "§6 paperp750–751/PDFp13–14; Figures18–21",
        "AppendixB paperp756–757/PDFp19–20",
        "AppendixE.4 paperp759/PDFp22; Algorithm5"
      ],
      "built_for_en": "Terrestrial DNN training jobs receive dedicated direct-connect optical network partitions co-designed with model parallelization and collective routing.",
      "built_for_zh": "地面 DNN training job 配置專屬 direct-connect optical network partition，並與 model parallelization、collective routing 共同設計。",
      "problem_en": "A traffic-independent fabric spends bandwidth and hardware on connectivity that mismatches a training job’s AllReduce and model-parallel dependencies.",
      "problem_zh": "與 traffic 獨立的 fabric 將 bandwidth 與 hardware 分配至和 training job 的 AllReduce、model-parallel dependency 契合程度較低的 connectivity。",
      "design_en": "Offline alternating optimization couples FlexFlow MCMC parallelization search with TopologyFinder. Degree is split between AllReduce and model-parallel subgraphs; coprime TotientPerms rings carry collectives, repeated Blossom matching favors heavy model-parallel edges, and host NIC forwarding supplies multi-hop RoCEv2. Modified NCCL load-balances ring permutations.",
      "design_zh": "offline alternating optimization 結合 FlexFlow MCMC parallelization search 與 TopologyFinder；degree 分配至 AllReduce、model-parallel subgraph；互質 TotientPerms ring 承載 collective，重複 Blossom matching 偏好高流量 model-parallel edge，host NIC forwarding 提供 multi-hop RoCEv2；修改版 NCCL 在 ring permutation 之間 load balance。",
      "model_en": "A DNN task graph drives degree-constrained topology and routing search. TotientPerms uses gcd(p,n)=1; geometric permutation selection targets a small diameter. CoinChangeMod routes the ring-derived graph and k-shortest paths route model-parallel traffic. MCMC is a sampling search method; simulator and forwarding bandwidth-tax models estimate iteration time.",
      "model_zh": "DNN task graph 驅動 degree-constrained topology 與 routing search；TotientPerms 使用 gcd(p,n)=1，geometric permutation selection 尋找較小 diameter；CoinChangeMod 為 ring-derived graph 選路，k-shortest path 為 model-parallel traffic 選路；MCMC 為 sampling search method；simulator 與 forwarding bandwidth-tax model 估計 iteration time。",
      "evaluation_en": "FlexNet searches strategies; htsim-based FlexNetPacket evaluates packet behavior. Simulations include 128-server dedicated and 432-server shared clusters, four A100s per simulated server, degree 4/8, and six DNNs. The prototype uses 12 one-A100 servers, a Telescent patch panel, RoCEv2/PFC, and degree 4×25Gbps interfaces per server, totaling 100Gbps.",
      "evaluation_zh": "FlexNet 搜尋 strategy；以 htsim 為基礎的 FlexNetPacket 評估 packet behavior；simulation 包含 128-server 專屬與 432-server 共用 cluster，每個模擬 server 配置四張 A100、degree4/8、六種 DNN；prototype 使用 12 個各配一張 A100 的 server、Telescent patch panel、RoCEv2/PFC，以及每 server degree4×25Gbps interface，合計100Gbps。",
      "baselines_en": "Simulation compares similar-cost Fat-tree, 2:1 oversubscribed Fat-tree, Ideal Switch, OCS-reconfig, equal-degree/bandwidth SiP-ML, and Expander. Prototype compares Switch100Gbps and Switch25Gbps. OCS-reconfig assumes 10ms setup and 50ms demand refresh; sensitivity sweeps 1µs–10ms.",
      "baselines_zh": "simulation 比較 similar-cost Fat-tree、2:1 oversubscribed Fat-tree、Ideal Switch、OCS-reconfig、equal-degree/bandwidth SiP-ML、Expander；prototype 比較 Switch100Gbps、Switch25Gbps；OCS-reconfig 採10ms setup 與50ms demand refresh；sensitivity 掃描1µs–10ms。",
      "result_en": "The simulation headline is up to 3.4× shorter training iteration time versus similar-cost Fat-tree. Prototype throughput tracks Switch100Gbps across tested models; VGG reaches 90% target accuracy 2× faster than Switch25Gbps. A large simulated all-to-all stress case makes TopoOpt1.1× slower than Fat-tree, identifying forwarding-cost sensitivity.",
      "result_zh": "simulation headline 為相對 similar-cost Fat-tree 最高3.4× 較短 training iteration time；prototype 各測試 model 的 throughput 接近 Switch100Gbps；VGG 達到90% target accuracy 的時間相對 Switch25Gbps 加速2×；大型 simulated all-to-all stress case 使 TopoOpt 相對 Fat-tree 慢1.1×，提供 forwarding-cost sensitivity 證據。",
      "limitations_en": "Main TopoOpt configures a topology before a job and retains it during training, with reconfiguration for permanent failures. Terrestrial arbitrary optical connectivity and passive-switch costs define its scope. Orbital transfer requires physical candidate edges, pointing/range constraints, setup outage, terminal degree/power, and forecast error. Cost-equivalent comparisons and equal-capacity comparisons answer separate questions.",
      "limitations_zh": "主要 TopoOpt 在 job 開始前配置 topology，並於 training 期間維持該配置，永久 failure 透過 reconfiguration 修復；地面任意 optical connectivity 與 passive-switch cost 界定適用範圍；軌道移植需要 physical candidate edge、pointing/range constraint、setup outage、terminal degree/power、forecast error；cost-equivalent 與 equal-capacity comparison 各自回答其對應問題。",
      "figure_caption_en": "TopoOpt’s terrestrial interconnect exposes d optical interfaces per server to optical switching planes. A job receives a selected direct-connect graph; host forwarding carries traffic across multiple edges. Orbital adaptation constrains the graph to feasible optical contacts and charges setup and power budgets.",
      "figure_caption_zh": "TopoOpt 的地面 interconnect 讓每 server 的 d 個 optical interface 連接 optical switching plane；job 配置選定的 direct-connect graph，host forwarding 承載 multi-edge traffic；軌道 adaptation 將 graph 限制於可行 optical contact，並計入 setup 與 power budget。",
      "evidence_status": "Full primary text reviewed",
      "mechanism_scope": "Terrestrial optical AI fabric baseline",
      "deployment_regime": "T",
      "space_ground_path": false,
      "display_regime_en": "T · Terrestrial reference",
      "display_regime_zh": "T · 地面比較基準",
      "primary_pdf_sha256": "16c3f666460abcb56e34fcd8008712c966a8cdf7bde529e2b5db457d357ed135",
      "figure_asset": "../figures/topoopt.png",
      "figure_acquisition": "Primary PDF crop",
      "figure_content_type": "image/png",
      "figure_bytes": 38609,
      "figure_sha256": "dfcd58a6eaedeeb0ba84c6a6c215022ef33637be5565ba34825a5bb990f0f544"
    },
    {
      "id": "teccl",
      "title": "Rethinking Machine Learning Collective Communication as a Multi-Commodity Flow Problem",
      "authors": "Xuting Liu, Behnaz Arzani, Siva Kesava Reddy Kakarla, Liangyu Zhao, Vincent Liu, Miguel Castro, Srikanth Kandula, Luke Marshall",
      "year": "2024",
      "venue": "SIGCOMM",
      "url": "https://doi.org/10.1145/3651890.3672249",
      "pdf_url": "https://vincen.tl/files/liu24teccl.pdf",
      "regime": "T · Terrestrial reference",
      "transfer_category": "terrestrial reference",
      "source_tier": "full primary text",
      "evidence_status": "Full primary PDF: formulation, implementation, evaluation, and limitations audited",
      "evidence": [
        "§3.1 PDF pp5–6",
        "§4–5 PDF pp6–8",
        "§6 PDF pp8–11; Tables 3–4",
        "§8 PDF p12; Appendix H topology parameters"
      ],
      "built_for_en": "Centrally managed GPU training clusters synthesize collective schedules for a known topology and finite communication demand.",
      "built_for_zh": "集中管理的 GPU 訓練叢集，依已知拓樸與有限資料需求合成 collective 排程。",
      "problem_en": "Joint routing and scheduling must preserve chunk identity, replication, propagation delay, store-and-forward, and per-link capacity while delivering every requested chunk quickly.",
      "problem_zh": "聯合路由與排程需要同時維持 chunk 身分、複製、傳播延遲、儲存轉送與鏈路容量，並儘早交付各目的端需求。",
      "design_en": "Discrete epochs track each source chunk in buffers and transmissions. AllGather uses MILP; AllToAll uses a continuous LP because its demand admits a flow interpretation. An A* inspired sequence of bounded-horizon optimizations carries buffer and in-flight state across rounds; reverse DFS prunes surplus transmissions. MSCCL executes exported schedules.",
      "design_zh": "離散 epoch 追蹤來源 chunk 的 buffer 與傳輸狀態。AllGather 使用 MILP；AllToAll 使用連續 LP。A* 啟發式反覆解有限 horizon，延續 buffer 與在途資料；反向 DFS 清理額外傳輸。輸出排程交由 MSCCL 執行。",
      "model_en": "The α–β graph model has binary chunk-presence variables, delayed causality, multicast-style copy inequalities, destination satisfaction, and capacity bounds. The weighted receipt objective rewards earlier delivery. §5 already permits a capacity matrix per epoch and aggregation of multi-tenant demands.",
      "model_zh": "α–β 圖模型包含二元 chunk presence、延遲因果、複製限制、目的端需求與容量限制。接收獎勵隨 epoch 增長遞減。§5 已支援逐 epoch 容量矩陣與多租戶需求加總。",
      "evaluation_en": "Most results compute completion from synthesized schedules and profiled link parameters on DGX1, DGX2, NDv2 and two proprietary topologies. Real execution uses two AMD chassis with 32 GPUs. Gurobi 9.5.2 runs on an 80-core/160-thread Xeon Platinum 8380 VM with 512 GB RAM; comparisons harmonize TACCL switch transit cost.",
      "evaluation_zh": "主要評估依合成排程與量測鏈路參數計算完成時間，涵蓋 DGX1、DGX2、NDv2 與兩種內部拓樸。硬體執行採兩臺 AMD chassis、32 GPU。Gurobi 9.5.2 使用 80-core／160-thread Xeon VM、512 GB RAM，與 TACCL 比較時統一 switch transit cost。",
      "baselines_en": "TACCL, MSCCL/SCCL synthesis, and RCCL on the AMD testbed; ablations cover copies, epoch granularity, buffers, optimal/30%-gap early-stop MILP, and A* horizons. TACCL routing and scheduling each receive 2 or 4 hours according to topology scale.",
      "baselines_zh": "比較 TACCL、MSCCL／SCCL 合成，以及 AMD testbed 上的 RCCL；消融涵蓋複製、epoch 粒度、buffer、最佳 MILP／30% gap early-stop 與 A* horizon。TACCL 的 routing 與 scheduling 階段依規模各配置 2 或 4 小時。",
      "result_en": "The abstract reports AMD algorithm bandwidth up to 2.14× TACCL and 3.18× RCCL; §6.2 places AllGather near 3× RCCL at 1 MB and 1.5–2× for larger transfers. Table 4 reports 256-GPU AllToAll synthesis in 1500 s with 4× coarser epochs, and 256-GPU A* AllGather in 2.8 h with 2× epochs.",
      "result_zh": "摘要報告 AMD algorithm bandwidth 最高達 TACCL 的 2.14×、RCCL 的 3.18×；§6.2 的 1 MB AllGather 約為 RCCL 的 3×，較大傳輸約 1.5–2×。Table 4 的 256-GPU AllToAll 使用 4× epoch、耗時 1500 s；256-GPU A* AllGather 使用 2× epoch、耗時 2.8 h。",
      "limitations_en": "Performance depends on supplied α/β, epoch and chunk size. AllGather early-stop and A* have heuristic quality tradeoffs; large models can require roughly 350 GB memory. The paper evaluates AllGather and AllToAll, while its AllReduce construction combines component collectives and leaves reduction compute cost for further modeling.",
      "limitations_zh": "效能依賴外部 α／β、epoch 與 chunk size。AllGather early-stop 與 A* 提供品質／速度取捨；大型模型約需 350 GB 記憶體。實測著重 AllGather／AllToAll；AllReduce 組合式設計的 reduction compute cost 留待進一步建模。",
      "math_latex": "\\sum_{s,c}F_{s,i,j,k,c}\\le T_{ij}\\tau,\\qquad \\max\\sum_{k,s\\ne d}\\frac{R_{s,d,k}}{k+1}",
      "math_source_anchor": "§3.1, PDF pp5–6: Capacity constraints and The objective; source equations appear as named displays",
      "math_definitions_en": "F is a chunk transmission indicator; T is link capacity in chunks/s; τ is epoch duration; R records received demand. Binary identities preserve replicated chunk provenance.",
      "math_definitions_zh": "F 表示 chunk 傳輸；T 為每秒 chunk 容量；τ 為 epoch 時間；R 追蹤需求接收。二元身分維持複製資料的來源。",
      "residual_novelty_en": "Supply the same contact matrix to TE-CCL as a strong temporal-capacity baseline. Residual orbital novelty concerns executable schedules under delayed telemetry, terminal acquisition and correlated outages, deadline-specific collective contracts, finite radiation-constrained memory, and measured training iteration tails.",
      "residual_novelty_zh": "將相同 contact matrix 提供 TE-CCL，作為 temporal capacity 的強基準。軌道研究的增量可聚焦延遲 telemetry、terminal acquisition、關聯 outage 下的可執行排程、collective deadline contract、受限記憶體，以及 training iteration tail。",
      "figure_caption_en": "Figure 2 explains the three modeling requirements: chunk copies, timing with queueing, and store-and-forward.",
      "figure_caption_zh": "Figure 2 呈現三項建模需求：chunk 複製、含 queueing 的傳輸時間，以及儲存轉送。",
      "open_source": "https://github.com/microsoft/TE-CCL",
      "gpu_count": "32 AMD testbed; 256 modeled",
      "node_count": "2 AMD chassis",
      "evaluation_method": "Simulation, Hardware Testbed, Analytical Model",
      "traffic_pattern": "AllGather, All-to-All",
      "compute_memory_hw": "AMD Instinct MI250",
      "comm_libraries": "NCCL",
      "affiliations": "University of Pennsylvania, Microsoft Research, University of Washington, OpenAI, Microsoft",
      "source_program_url": "https://conferences.sigcomm.org/sigcomm/2024/program/",
      "source_acquired_date": "2026-10-02",
      "publication_type": "SIGCOMM main-track terrestrial baseline",
      "pdf_sha256": "5ce0ad153d82f93a53691092fbcb2af71018c213ec0b010f51ef3db13f90f339",
      "figure_number": "2",
      "figure_page_1based": 3,
      "figure_bbox_points": [
        53,
        80,
        559,
        195
      ],
      "figure_acquisition_method": "Original primary PDF figure crop rendered at 216 DPI",
      "figure_sha256": "3a9aeaf65c180008580ec4af8292b353e95101fb29d7c63053f8f3bd4af869db",
      "figure_bytes": 70679,
      "figure_visually_inspected": true,
      "figure_source_pdf_url": "https://vincen.tl/files/liu24teccl.pdf",
      "software_simulator": "Not Specified",
      "mechanism_scope": "T",
      "deployment_regime": "T",
      "space_ground_path": false,
      "display_regime_en": "T · Terrestrial reference",
      "display_regime_zh": "T · 地面比較基準",
      "primary_pdf_sha256": "5ce0ad153d82f93a53691092fbcb2af71018c213ec0b010f51ef3db13f90f339",
      "figure_asset": "../figures/teccl.png",
      "figure_acquisition": "Original primary PDF figure crop rendered at 216 DPI",
      "figure_content_type": "image/png"
    },
    {
      "id": "mccs",
      "title": "MCCS: A Service-based Approach to Collective Communication for Multi-Tenant Cloud",
      "authors": "Yongji Wu, Yechen Xu, Jingrong Chen, Zhaodong Wang, Ying Zhang, Matthew Lentz, Danyang Zhuo",
      "year": "2024",
      "venue": "SIGCOMM",
      "url": "https://doi.org/10.1145/3651890.3672252",
      "pdf_url": "https://www.yongjiwu.me/assets/pdf/sigcomm24-mccs.pdf",
      "regime": "T · Terrestrial reference",
      "transfer_category": "terrestrial reference",
      "source_tier": "full primary text",
      "evidence_status": "Full primary PDF: formulation, implementation, evaluation, and limitations audited",
      "evidence": [
        "§4 PDF pp4–7; Figure 4",
        "§6.1–6.4 PDF pp8–10",
        "§6.5 PDF pp10–11; Figure 11"
      ],
      "built_for_en": "Cloud providers manage collectives across tenant GPU jobs while preserving an application-facing NCCL-like interface.",
      "built_for_zh": "Cloud provider 管理跨租戶 GPU collective，同時保留應用程式的 NCCL 類介面。",
      "problem_en": "Tenant-local collective choices can collide on shared physical paths. Runtime strategy changes require common operation boundaries and correct GPU memory/stream ordering across processes.",
      "problem_zh": "租戶各自選擇的 collective 會在共享 physical path 競爭；執行期策略切換需要共同操作邊界與跨程序 GPU memory／stream ordering。",
      "design_en": "A shim forwards allocations and collective requests to per-host services through shared-memory queues. CUDA IPC handles share GPU buffers/events. Proxy engines control communicators; transport engines manage RDMA flows. A control-ring AllGather exchanges last-launched sequence numbers, drains through their maximum, then rebuilds connections.",
      "design_zh": "Shim 透過共享記憶體 queue 將 allocation 與 collective request 交給 host service。CUDA IPC handle 共享 GPU buffer／event；proxy engine 管理 communicator，transport engine 管理 RDMA flow。Control-ring AllGather 交換最後 launch 的 sequence number，執行至共同最大值再重建連線。",
      "model_en": "Provider policies use greedy locality-aware rings and Hedera-style best-fit flow assignment, round-robin across jobs for fairness, reserved routes for priority, and time windows aligned to high-priority idle periods. This is a service architecture with concrete heuristics and a synchronization invariant.",
      "model_zh": "Policy 採 locality-aware greedy ring、Hedera 式 best-fit flow assignment、跨 job round-robin、公平分配、priority reserved route，以及對齊高 priority idle period 的時間窗。核心為 service architecture、具體 heuristic 與同步 invariant。",
      "evaluation_en": "Four servers hold eight RTX 3090 GPUs and ConnectX-5 100 Gbps NICs. A self-wired SN2100 emulates two leaves/two spines at 2:1 oversubscription. NCCL-derived kernels and trace replay cover VGG-19 data parallelism and GPT-2.7B tensor parallelism. Flow simulation scales to 768 GPUs, 200 Gbps links, 50 ResNet-50 jobs, and five repetitions.",
      "evaluation_zh": "四臺 server 共八張 RTX 3090、ConnectX-5 100 Gbps NIC。SN2100 自連線模擬 two-leaf／two-spine、2:1 oversubscription。NCCL kernel 與 trace replay 評估 VGG-19 DP、GPT-2.7B TP；flow simulation 擴至 768 GPU、200 Gbps、50 個 ResNet-50 job，重複五次。",
      "baselines_en": "NCCL v2.17.1; NCCL(OR) with manually optimal rings; MCCS(-FA)/(-FFA) flow-assignment ablations; ECMP, FFA, PFA and PFA+TS for trace replay; random rings, optimal rings and optimal rings+FFA for simulation.",
      "baselines_zh": "基準包括 NCCL v2.17.1、人工設定 optimal ring 的 NCCL(OR)、MCCS(-FA)／(-FFA)；trace replay 比較 ECMP、FFA、PFA、PFA+TS；simulation 比較 random ring、optimal ring、optimal ring+FFA。",
      "result_en": "For 8–512 MB collectives, average algorithm bandwidth speedup is 1.6× on four GPUs and 2.4× on eight GPUs versus NCCL. Trace replay prioritizes VGG by 34% versus ECMP. Simulation reports 3.27×/3.43× average AllReduce speedup for random/compact placement. Small 512 KB AllGather incurs 63% lower bandwidth than NCCL(OR) on four GPUs, from 50–80 μs service latency.",
      "result_zh": "8–512 MB collective 的平均 algorithm bandwidth，四 GPU 達 NCCL 的 1.6×、八 GPU 達 2.4×。Trace replay 的 VGG priority 排程比 ECMP 快 34%。Simulation 的 random／compact placement 平均 AllReduce speedup 為 3.27×／3.43×。四 GPU 的 512 KB AllGather 比 NCCL(OR) bandwidth 低 63%，service latency 為 50–80 μs。",
      "limitations_en": "Prototype kernels cover ring AllReduce/AllGather; the hardware experiment uses eight consumer GPUs. Training results replay profiled traces, and large-scale results use per-flow-fairness simulation. Reconfiguration closes and creates connections, while monitoring and provider policy remain external inputs.",
      "limitations_zh": "Prototype kernel 涵蓋 ring AllReduce／AllGather，硬體規模為八張 consumer GPU。Training 結果來自 profile trace replay，大規模結果採 per-flow fairness 模擬。Reconfiguration 重建連線，監測與 policy 由外部提供。",
      "math_latex": "q^{\\star}=\\max_r q_r;\\qquad k\\le q^{\\star}\\Rightarrow\\operatorname{oldSchedule}(k),\\quad k>q^{\\star}\\Rightarrow\\operatorname{newSchedule}(k)",
      "math_source_anchor": "§4.2, PDF pp5–6, Figure 4: sequence-number barrier design invariant; notation transcribed for this survey",
      "math_definitions_en": "q_r is rank r last-launched collective sequence number at reconfiguration; all ranks complete the same prefix before changing strategy.",
      "math_definitions_zh": "q_r 為 rank r 接獲 reconfiguration 時最後 launch 的 sequence number；各 rank 完成相同 prefix 後切換策略。",
      "residual_novelty_en": "Give MCCS identical contact forecasts and a forecast-driven ring/FFA policy. Residual contribution is contact-deadline consistency with in-flight chunks, forecast-error-safe migration and resource reuse across orbital partitions, with end-to-end workload evidence beyond trace replay.",
      "residual_novelty_zh": "提供 MCCS 相同 contact forecast 與預測驅動 ring／FFA policy。增量研究可聚焦含在途 chunk 的 contact-deadline 一致性、forecast error 下的 migration、跨軌道 partition 資源延續，以及端到端 workload 證據。",
      "figure_caption_en": "Figure 1 shows the shift from tenant-owned libraries to a provider-owned collective service and the resulting control points.",
      "figure_caption_zh": "Figure 1 呈現 tenant library 到 provider collective service 的架構，以及對應 control point。",
      "open_source": "https://github.com/phoenix-dataplane/MCCS/",
      "gpu_count": "8 NVIDIA RTX 3090; 768 simulated",
      "node_count": "4 servers; 96 simulated hosts",
      "evaluation_method": "Hardware Testbed, Simulation",
      "traffic_pattern": "AllReduce, AllGather, Tensor Parallelism",
      "compute_memory_hw": "NVIDIA RTX 3090",
      "network_hw": "ConnectX-5, Mellanox switches",
      "network_topology": "Leaf-Spine",
      "transport_and_interconnect": "RDMA, RoCEv2",
      "routing_and_congestion_control": "ECMP",
      "comm_libraries": "NCCL",
      "affiliations": "Duke University, Meta",
      "source_program_url": "https://conferences.sigcomm.org/sigcomm/2024/program/",
      "source_acquired_date": "2026-10-02",
      "publication_type": "SIGCOMM main-track terrestrial baseline",
      "pdf_sha256": "06c8361dc6f1078ecbe619e93ad33cf8abc5378156453b7fcf934434f834a6c2",
      "figure_number": "1",
      "figure_page_1based": 1,
      "figure_bbox_points": [
        317,
        188,
        559,
        399
      ],
      "figure_acquisition_method": "Original primary PDF figure crop rendered at 216 DPI",
      "figure_sha256": "f55a72deb9e097c5078c3600c8ce3a7254831cc24c341aefec6a00df946df5b3",
      "figure_bytes": 82840,
      "figure_visually_inspected": true,
      "figure_source_pdf_url": "https://www.yongjiwu.me/assets/pdf/sigcomm24-mccs.pdf",
      "software_simulator": "Not Specified",
      "mechanism_scope": "T",
      "deployment_regime": "T",
      "space_ground_path": false,
      "display_regime_en": "T · Terrestrial reference",
      "display_regime_zh": "T · 地面比較基準",
      "primary_pdf_sha256": "06c8361dc6f1078ecbe619e93ad33cf8abc5378156453b7fcf934434f834a6c2",
      "figure_asset": "../figures/mccs.png",
      "figure_acquisition": "Original primary PDF figure crop rendered at 216 DPI",
      "figure_content_type": "image/png"
    },
    {
      "id": "crux",
      "title": "Crux: GPU-Efficient Communication Scheduling for Deep Learning Training",
      "authors": "Jiamin Cao, Yu Guan, Kun Qian, Jiaqi Gao, Wencong Xiao, Jianbo Dong, Binzhang Fu, Dennis Cai, Ennan Zhai",
      "year": "2024",
      "venue": "SIGCOMM",
      "url": "https://doi.org/10.1145/3651890.3672239",
      "pdf_url": "https://cs.stanford.edu/~keithw/sigcomm2024/sigcomm24-final380-acmpaginated.pdf",
      "regime": "T · Terrestrial reference",
      "transfer_category": "terrestrial reference",
      "source_tier": "full primary text",
      "evidence_status": "Full primary PDF: formulation, implementation, evaluation, and limitations audited",
      "evidence": [
        "§3 PDF pp4–5, Equations (1)–(2)",
        "§4 PDF pp5–8, Equations (3)–(4)",
        "§6 PDF pp9–12; Figures 19–25",
        "§7.1 PDF p12"
      ],
      "built_for_en": "Multi-job training clusters share inter-host paths and PCIe resources; the operator targets aggregate useful GPU computation.",
      "built_for_zh": "多個 training job 共享 inter-host path 與 PCIe 資源，營運者追求整體有效 GPU 計算量。",
      "problem_en": "Inter-job communication contention stalls GPUs. Aggregate utilization depends on which computation a unit of communication unlocks, iteration periodicity, computation overlap and the small number of physical priority levels.",
      "problem_zh": "跨 job 通訊競爭造成 GPU 等待。整體利用率取決於每單位通訊解鎖的計算、iteration 週期、compute overlap，以及有限 physical priority level。",
      "design_en": "Profile GPU intensity, select least-congested paths in descending intensity order, correct priorities for iteration and overlap effects, and compress priorities through a weighted DAG K-cut dynamic program. A daemon probes paths/profiles jobs; the transport layer steers RoCEv2 by UDP source ports and sets priorities, with PCIe semaphores for intra-host contention.",
      "design_zh": "Profile GPU intensity，依 intensity 排序選 least-congested path，按 iteration／overlap 修正 priority，透過 weighted DAG K-cut dynamic program 壓縮 priority。Daemon 探測與 profile；transport 用 UDP source port 導流 RoCEv2、設定 priority，PCIe semaphore 管理 host 內競爭。",
      "model_en": "A weighted multi-commodity-flow-derived objective maximizes useful computation. For a single fixed-capacity bottleneck, intensity equals per-iteration FLOPs divided by bottleneck communication time; the long-horizon theorem relates total computation to the integral of scheduled intensity. Priority compression minimizes weighted utilization loss under K hardware queues.",
      "model_zh": "目標由 weighted multi-commodity flow 引導，最大化有效計算。固定單一 bottleneck 的 intensity 為每 iteration FLOPs 除以瓶頸通訊時間；長 horizon theorem 將計算量連結到 scheduled intensity 積分。Priority compression 在 K 個 hardware queue 下最小化 weighted utilization loss。",
      "evaluation_en": "A 96-A100 testbed uses 12 hosts and a two-layer Clos. Real ResNet/BERT/GPT training explores network and PCIe contention. An α–β simulator replays two weeks of 2000+ GPU production traces on two-layer Clos and double-sided topologies with eight priority levels; individual compute durations come from actual GPUs.",
      "evaluation_zh": "96-A100 testbed 由 12 host 與 two-layer Clos 組成。ResNet／BERT／GPT 實際訓練涵蓋 network／PCIe contention。α–β simulator 重播 2000+ GPU、兩週 production trace，評估 Clos／double-sided topology、八 priority level，compute duration 來自 GPU 量測。",
      "baselines_en": "Real training compares default communication and each job running alone. Trace simulations compare Sincronia, TACCL* and CASSINI; TACCL* is an author-implemented inter-job adaptation selecting least-congested links and prioritizing longer paths. Microbenchmarks also include Varys and optimum components, with Crux-PA, Crux-PS-PA and full Crux ablations.",
      "baselines_zh": "實際訓練比較 default communication 與各 job 單獨執行。Trace simulation 比較 Sincronia、TACCL*、CASSINI；TACCL* 為作者實作的 inter-job adaptation，選 least-congested link 並優先較長 path。Microbenchmark 另含 Varys、最佳解及 Crux-PA／Crux-PS-PA／full 消融。",
      "result_en": "Hardware experiments report utilization improvements of 8.3–14.8% across scenarios; prioritized BERT JCT decreases by up to 33% while ResNet JCT can increase 1–3%. Simulation reports +13–23% on Clos and reported +4–7% on double-sided topology. Figure 23 shows Clos utilization 0.39/0.49/0.48 for Sincronia/TACCL*/CASSINI and 0.62 for full Crux.",
      "result_zh": "硬體情境報告 報告 utilization 改善 +8.3–14.8%；優先 BERT 的 JCT 最多減少 33%，ResNet JCT 增加 1–3%。Simulation 在 Clos 報告 +13–23%，double-sided 報告 +4–7%。Figure 23 的 Clos utilization：Sincronia 0.39、TACCL* 0.49、CASSINI 0.48、full Crux 0.62。",
      "limitations_en": "The single-bottleneck theorem assumes constant capacity; network-wide scheduling uses practical heuristics. Prioritization exchanges aggregate efficiency for some jobs’ slower completion. Profiling/rescheduling can take up to a minute per arrival/completion; correction factors depend on the selected reference job.",
      "limitations_zh": "Single-bottleneck theorem 假設固定容量；network-wide 排程採 heuristic。Priority 提升整體效率時，部分 job 完成時間增加。每次 arrival／completion 的 profiling／rescheduling 約一分鐘內；correction factor 依 reference job 改變。",
      "math_latex": "I_j=\\frac{W_j}{t_j},\\quad t_j=\\max_{e\\in E}\\frac{M_{j,e}}{B_e},\\quad P_j=k_jI_j",
      "math_source_anchor": "§3.2 Equation (2), PDF p4; §4.2 Equation (3), PDF p6",
      "math_definitions_en": "W is per-iteration computation; M is per-link traffic; B is link bandwidth; k corrects intensity for iteration and overlap effects.",
      "math_definitions_zh": "W 為每 iteration 計算量；M 為逐鏈路 traffic；B 為容量；k 修正 iteration／overlap 效果。",
      "residual_novelty_en": "Feed Crux the same future link capacities, then profile a forecast-aware intensity policy. Residual novelty requires time-varying computation availability, sunlight/thermal duty, finite contact windows and collective dependency release times, measured through both utilization and deadline/tail metrics.",
      "residual_novelty_zh": "將相同後續 link capacity 提供 Crux，加入 forecast-aware intensity policy。增量需結合 compute availability、sunlight／thermal duty、有限 contact window 與 collective dependency release time，同時量測 utilization、deadline 與 tail。",
      "figure_caption_en": "Figure 10 links GPU intensity to path selection, priority assignment and priority compression.",
      "figure_caption_zh": "Figure 10 將 GPU intensity 連結到 path selection、priority assignment 與 priority compression。",
      "gpu_count": "96 NVIDIA A100; 2000+ trace GPUs",
      "node_count": "12 servers",
      "evaluation_method": "Hardware Testbed, Simulation, Analytical Model",
      "traffic_pattern": "AllReduce",
      "compute_memory_hw": "NVIDIA A100",
      "network_topology": "Clos",
      "transport_and_interconnect": "RoCEv2, TCP/IP, NVLink, PCIe 4.0",
      "routing_and_congestion_control": "ECMP",
      "affiliations": "Alibaba Cloud",
      "source_program_url": "https://conferences.sigcomm.org/sigcomm/2024/program/",
      "source_acquired_date": "2026-10-02",
      "publication_type": "SIGCOMM main-track terrestrial baseline",
      "pdf_sha256": "8cfbbab673992cdd81fc7e8dfa663b6e444487994613054b373812b648cf1597",
      "figure_number": "10",
      "figure_page_1based": 5,
      "figure_bbox_points": [
        317,
        85,
        559,
        143
      ],
      "figure_acquisition_method": "Original primary PDF figure crop rendered at 216 DPI",
      "figure_sha256": "f0aa09cf745628123202ea89436c47ae3532e596093751a24020926249766afd",
      "figure_bytes": 23395,
      "figure_visually_inspected": true,
      "figure_source_pdf_url": "https://cs.stanford.edu/~keithw/sigcomm2024/sigcomm24-final380-acmpaginated.pdf",
      "software_simulator": "Not Specified",
      "mechanism_scope": "T",
      "deployment_regime": "T",
      "space_ground_path": false,
      "display_regime_en": "T · Terrestrial reference",
      "display_regime_zh": "T · 地面比較基準",
      "primary_pdf_sha256": "8cfbbab673992cdd81fc7e8dfa663b6e444487994613054b373812b648cf1597",
      "figure_asset": "../figures/crux.png",
      "figure_acquisition": "Original primary PDF figure crop rendered at 216 DPI",
      "figure_content_type": "image/png"
    },
    {
      "id": "optccl",
      "title": "OptCCL: Scalable Synthesis of Optimal Collective Communication Algorithms",
      "authors": "Richard Shapley, Rachit Agarwal, David Shmoys",
      "year": "2026",
      "venue": "SIGCOMM",
      "url": "https://doi.org/10.1145/3789240.3829207",
      "pdf_url": "https://www.cs.cornell.edu/~ragarwal/pubs/optccl.pdf",
      "regime": "T · Terrestrial reference",
      "transfer_category": "terrestrial reference",
      "source_tier": "full primary text",
      "evidence_status": "Full primary PDF: formulation, implementation, evaluation, and limitations audited",
      "evidence": [
        "§3–4 PDF pp3–7, Equations (1)–(9), Theorem 2",
        "§4.3 PDF p7; finite-chunk overhead",
        "§6 PDF pp8–9; Equation (14)",
        "§7–9 PDF pp9–12; Table 2"
      ],
      "built_for_en": "Collective synthesis targets heterogeneous host/network graphs, hardware store/copy/reduce capabilities and concurrent collectives.",
      "built_for_zh": "Collective synthesis 面向異質 host／network 圖、hardware store／copy／reduce 能力與 concurrent collective。",
      "problem_en": "Exact step-based MILPs couple routes, chunk timing and bandwidth, making synthesis expensive. A scalable optimizer needs precise replication/resource accounting and an explicit realization of continuous flows into executable chunks.",
      "problem_zh": "Step-based MILP 耦合 route、chunk timing 與 bandwidth，合成成本增加。可擴充 optimizer 需要精確的 replication／resource accounting，以及 continuous flow 到 chunk 排程的實現。",
      "design_en": "Represent each origin–destination pair as a commodity, with usage variables accounting for shared copies. Decouple spatial routing from temporal scheduling through average lifetime capacity. LP solutions gain feasibility cuts via selective vertex duplication; Mirrored Dantzig–Wolfe exploits full-input symmetry. Tree decomposition plus weighted fair queueing emits uniform chunks.",
      "design_zh": "每個 origin–destination pair 作為 commodity，usage variable 計算 shared copy。藉 collective lifetime 平均容量將 spatial routing 與 temporal scheduling 分解。Selective vertex duplication 加強 LP feasibility；Mirrored Dantzig–Wolfe 利用完整輸入 symmetry。Tree decomposition 與 weighted fair queueing 產生 uniform chunk。",
      "model_en": "Primary objective is collective makespan, with secondary weighted transmitted volume. LP(9) constrains lifetime usage; Theorem 2 relates it to the step LP within arbitrary ε. Ideal topologies satisfy the additional algorithm-feasibility conditions; other graphs are strengthened iteratively. Uniform finite-size chunk realization has a quantified makespan inflation bound.",
      "model_zh": "主要目標為 collective makespan，次要目標為 weighted transmitted volume。LP(9) 限制 lifetime usage；Theorem 2 將其與 step LP 連結至任意 ε。Ideal topology 滿足 algorithm-feasibility 條件，其餘 graph 反覆加強。有限 uniform chunk 的 makespan inflation 有明確 bound。",
      "evaluation_en": "Simulation models A100 eight-GPU hosts and DOE four-GPU hosts with rail-based interconnects, AllGather/AllToAll, and 64-GPU concurrent collectives. Synthesis runs on 32-core Xeon Gold 6234 with 314 GB RAM. Every competitor uses its largest chunk count completing within three hours.",
      "evaluation_zh": "Simulation 建模 A100 八-GPU host、DOE 四-GPU host、rail-based interconnect、AllGather／AllToAll，以及 64-GPU concurrent collective。合成使用 32-core Xeon Gold 6234、314 GB RAM；各 competitor 使用三小時內完成的最大 chunk count。",
      "baselines_en": "TE-CCL, SyCCL and TACCL public implementations; TACCL AllToAll is excluded following implementation issues, and SyCCL single-host runs encounter supported-topology limits. Multiple-collective ablations compare separately sequential (Obvious), individually synthesized/concurrently scaled (Overlaid), and joint (All-in-one), each using OptCCL.",
      "baselines_zh": "比較 TE-CCL、SyCCL、TACCL 公開實作；TACCL AllToAll 受 implementation issue 限制，SyCCL single-host 受支援拓樸限制。多 collective 比較 Obvious sequential、Overlaid concurrent scaling、All-in-one joint，三者皆使用 OptCCL。",
      "result_en": "AllGather algorithm bandwidth is 1–5× TE-CCL, 1–1.6× SyCCL and 1.2–18× TACCL; large-topology synthesis is 3–32× faster than SyCCL and 5–60× than TACCL. AllToAll gains are 1–1.3× with 2–500× lower synthesis time. Table 2 makespans relative to joint optimization are 6.25/1/1 with NVLink and 4.87/1.31/1 across Obvious/Overlaid/All-in-one. Finite chunks finish within 1% of routing optimum in evaluated cases.",
      "result_zh": "AllGather algorithm bandwidth 達 TE-CCL 的 1–5×、SyCCL 的 1–1.6×、TACCL 的 1.2–18×；大規模 synthesis 比 SyCCL 快 3–32×、TACCL 快 5–60×。AllToAll 為 1–1.3×，synthesis time 減少 2–500×。Table 2 的 Obvious／Overlaid／All-in-one 相對 makespan：有 NVLink 為 6.25／1／1，移除 NVLink 為 4.87／1.31／1。有限 chunk 評估結果距 routing optimum 1% 內。",
      "limitations_en": "Evidence consists of simulated schedule quality and measured synthesis runtime. Proofs and evaluation emphasize large, bandwidth-dominant messages on a given topology; propagation-dominant small messages, application integration, real GPU measurements, and weighted flow-time objectives are further directions stated in §9.",
      "limitations_zh": "證據包括 simulated schedule quality 與實測 synthesis runtime。Proof／evaluation 聚焦固定拓樸、large bandwidth-dominant message；propagation-dominant small message、application integration、GPU testbed 與 weighted flow-time 為 §9 的後續方向。",
      "math_latex": "\\sum_{t=0}^{T-1}\\sum_{e\\in L}\\sum_o w_{\\kappa(e,t),o}\\le R\\,b(L),\\qquad w_{e,o}\\ge f_{e,o,d}",
      "math_source_anchor": "§3 Equation (4), PDF p4; §4.1 Equation (8), PDF p5; Theorem 2",
      "math_definitions_en": "R is target collective makespan; b(L) is an aggregate resource bandwidth bound; w tracks actual transmitted usage shared by replicated origin data; f denotes pairwise commodity flow.",
      "math_definitions_zh": "R 為目標 collective makespan；b(L) 為 aggregate resource bandwidth；w 追蹤共用 origin data 的實際 usage；f 為 pairwise commodity flow。",
      "residual_novelty_en": "Provide OptCCL the same contact forecast and compare a temporal graph extension plus schedule executor. Residual novelty concerns contact deadlines and availability uncertainty that invalidate lifetime-average capacity guarantees, propagation-sensitive realizability, and an operational end-to-end LLM fabric.",
      "residual_novelty_zh": "提供 OptCCL 相同 contact forecast，加入 temporal graph extension 與 executor 作比較。增量聚焦 lifetime-average capacity 與 contact deadline／availability uncertainty 的關係、propagation-sensitive 排程實現，以及端到端 LLM fabric。",
      "figure_caption_en": "Figure 1 maps a rail topology to its directed resource graph and a time-expanded representation.",
      "figure_caption_zh": "Figure 1 將 rail topology 轉為 directed resource graph 與 time-expanded representation。",
      "open_source": "https://github.com/ml-collectives/optccl",
      "gpu_count": "up to 256 modeled GPUs; 64 concurrent-collective case",
      "node_count": "up to 64 modeled four-GPU hosts",
      "evaluation_method": "Simulation, Analytical Model",
      "traffic_pattern": "AllGather, All-to-All, ReduceScatter",
      "compute_memory_hw": "NVIDIA A100 (modeled)",
      "network_topology": "Rail-optimized",
      "transport_and_interconnect": "NVLink",
      "affiliations": "Cornell University",
      "source_program_url": "https://conferences.sigcomm.org/sigcomm/2026/program/papers/",
      "source_acquired_date": "2026-10-02",
      "publication_type": "SIGCOMM main-track terrestrial baseline",
      "pdf_sha256": "9204defdc2948cd1398dac553e69fc1b991bc4777c29b23e42b81414e8fa9224",
      "figure_number": "1",
      "figure_page_1based": 3,
      "figure_bbox_points": [
        53,
        80,
        298,
        209
      ],
      "figure_acquisition_method": "Original primary PDF figure crop rendered at 216 DPI",
      "figure_sha256": "72b900f514de01bbe8b9d360fd99245e97bce4c99d285096b11941aaf97bce46",
      "figure_bytes": 47775,
      "figure_visually_inspected": true,
      "figure_source_pdf_url": "https://www.cs.cornell.edu/~ragarwal/pubs/optccl.pdf",
      "software_simulator": "Not Specified",
      "mechanism_scope": "T",
      "deployment_regime": "T",
      "space_ground_path": false,
      "display_regime_en": "T · Terrestrial reference",
      "display_regime_zh": "T · 地面比較基準",
      "primary_pdf_sha256": "9204defdc2948cd1398dac553e69fc1b991bc4777c29b23e42b81414e8fa9224",
      "figure_asset": "../figures/optccl.png",
      "figure_acquisition": "Original primary PDF figure crop rendered at 216 DPI",
      "figure_content_type": "image/png"
    },
    {
      "id": "trivance",
      "title": "Trivance: Latency-Optimal AllReduce by Shortcutting Multiport Networks",
      "authors": "Anton Juerss, Vamsi Addanki, Stefan Schmid",
      "year": "2026",
      "venue": "SIGCOMM",
      "url": "https://doi.org/10.1145/3789240.3829181",
      "pdf_url": "https://arxiv.org/pdf/2602.17254",
      "regime": "T · Terrestrial reference",
      "transfer_category": "terrestrial reference",
      "source_tier": "full primary text",
      "evidence_status": "Full primary PDF: formulation, implementation, evaluation, and limitations audited",
      "evidence": [
        "§2 PDF pp3–4, Table 1",
        "§4–5 PDF pp6–9, Lemma 4.1/Theorem 4.3",
        "§6–7 PDF pp9–12",
        "Appendices E–G PDF pp16–20"
      ],
      "built_for_en": "AllReduce on bidirectional rings and regular multidimensional direct-connect tori, including TPU-like topology families.",
      "built_for_zh": "Bidirectional ring 與 regular multidimensional direct-connect torus 上的 AllReduce，涵蓋 TPU 類拓樸。",
      "problem_en": "Collective startup steps, routed distance and link congestion interact: using both ports can reduce the step count while full vectors increase transmitted bytes.",
      "problem_zh": "Collective startup step、route distance 與 link congestion 相互影響；同時使用雙 port 可減少 step，full-vector 傳輸增加 byte 量。",
      "design_en": "Each rank communicates simultaneously with peers at ±3^k distance and jointly reduces two incoming messages. A latency variant forwards full vectors in one phase; a bandwidth variant runs ReduceScatter then reverse AllGather with shrinking/growing blocks. Generalization handles arbitrary sizes and parallel dimensions.",
      "design_zh": "各 rank 同時與 ±3^k 距離的 peer 通訊，joint reduction 合併兩份 incoming message。Latency variant 單階段傳 full vector；bandwidth variant 先 ReduceScatter 再反向 AllGather，縮小／放大 block。設計擴展至任意規模與多維度。",
      "model_en": "A congestion-aware Hockney cost sums startup steps and per-step bytes times congestion. For powers of three, latency mode uses log3 n steps and m log3 n bytes per port; bandwidth mode uses 2 log3 n steps and total 2m(1−1/n) bytes per node. Bounds apply to multiport regular graphs under deterministic shortest paths.",
      "model_zh": "Congestion-aware Hockney cost 加總 startup step 與每 step byte×congestion。Power-of-three 規模下，latency mode 採 log3 n step、每 port m log3 n byte；bandwidth mode 採 2 log3 n step、每 node 共 2m(1−1/n) byte。Bound 適用 deterministic shortest path 的 regular multiport graph。",
      "evaluation_en": "Packet-level SST simulations vary rings, square/rectangular 2D tori and 3D tori; message sizes range 32 B–128 MiB. Default links are 800 Gbps with 100 ns latency, 100 ns per-hop processing, 8192 B packets and 1.5 μs startup per step. Appendices vary bandwidth up to 3.2 Tbps, latency and packet size.",
      "evaluation_zh": "Packet-level SST 模擬 ring、square／rectangular 2D torus、3D torus，message 為 32 B–128 MiB。預設 800 Gbps、100 ns link latency、100 ns per-hop processing、8192 B packet、1.5 μs startup；appendix 分析最高 3.2 Tbps 與 latency／packet sensitivity。",
      "baselines_en": "Bucket, Recursive Doubling, Swing and Bruck with latency/bandwidth variants; Bruck gains shortest-path routing, joint reduction and reordered AllGather. Recursive Doubling exploits all 2D ports; power-of-three comparisons use Bucket/Swing/Bruck.",
      "baselines_zh": "比較 Bucket、Recursive Doubling、Swing、Bruck 的 latency／bandwidth variant；Bruck 加入 shortest-path routing、joint reduction 與 AllGather ordering，Recursive Doubling 使用全部 2D port；power-of-three 情境比較 Bucket／Swing／Bruck。",
      "result_en": "Trivance improves latency-bound completion by 5–30%; the advantage extends through 512 KiB in rings, 8 MiB in multidimensional tori, and 32 MiB in high-bandwidth examples. Some 3D cases improve up to 20% through 128 MiB. §6.2 identifies crossover to Swing around 8 MiB in rectangular tori.",
      "result_zh": "Latency-bound 完成時間改善 5–30%；優勢延伸至 ring 的 512 KiB、多維 torus 的 8 MiB、高 bandwidth 情境的 32 MiB。部分 3D 情境在 128 MiB 內改善最高 20%。§6.2 的 rectangular torus 約於 8 MiB 轉向 Swing。",
      "limitations_en": "The study measures simulated collective completion on stable regular topologies. Arithmetic reduction, host/GPU execution and end-to-end training require hardware validation. The two variants explicitly trade startup latency against byte volume; size departures from powers of three increase routing/data overhead.",
      "limitations_zh": "研究量測穩定 regular topology 的 simulated collective completion。Arithmetic reduction、host／GPU execution 與端到端 training 需要 hardware validation。兩 variant 呈現 startup latency／byte volume 取捨；規模偏離 power-of-three 會增加 route／data overhead。",
      "math_latex": "C(m,A)=\\operatorname{steps}(A)\\alpha+\\sum_{k=0}^{\\operatorname{steps}(A)-1}\\beta m_kc_k;\\quad \\pi_{\\pm}(r,k)=(r\\pm3^k)\\bmod n",
      "math_source_anchor": "§2.1 PDF p3, congestion-aware cost display; §4.1 PDF p6, peer-distance display",
      "math_definitions_en": "α is startup cost; β is inverse link bandwidth; m_k is per-step message size; c_k is shared-link congestion; r is rank and n is ring size.",
      "math_definitions_zh": "α 為 startup cost；β 為容量倒數；m_k 為每 step message；c_k 為 link sharing congestion；r 為 rank，n 為 ring 規模。",
      "residual_novelty_en": "Use both Trivance variants on each forecast-defined stable topology slice. Residual orbital work must handle changing membership/port availability and propagation-time feasibility, plus measured small-message/all-expert traffic; a triple-distance ring schedule alone transfers directly from this prior art.",
      "residual_novelty_zh": "在 forecast 定義的穩定 topology slice 上使用兩種 Trivance variant。增量需處理 membership／port availability 改變、propagation-time feasibility，以及 small-message／expert traffic 實測；triple-distance ring schedule 本身可直接移植 prior art。",
      "figure_caption_en": "Figure 1 contrasts Recursive Doubling, Bruck and Trivance communication distance and congestion over steps.",
      "figure_caption_zh": "Figure 1 比較 Recursive Doubling、Bruck、Trivance 的逐 step 通訊距離與 congestion。",
      "open_source": "https://doi.org/10.5281/zenodo.21180442",
      "gpu_count": "modeled network nodes",
      "node_count": "rings, 2D tori, 3D tori",
      "evaluation_method": "Simulation, Analytical Model",
      "traffic_pattern": "AllReduce, ReduceScatter, AllGather",
      "network_topology": "3D Torus",
      "software_simulator": "SST",
      "affiliations": "Weizenbaum Institute, TU Berlin, Purdue University",
      "source_program_url": "https://conferences.sigcomm.org/sigcomm/2026/program/papers/",
      "source_acquired_date": "2026-10-02",
      "publication_type": "SIGCOMM main-track terrestrial baseline",
      "pdf_sha256": "5ce5f24ebed0f409c81e09cd60e4eb811b97b2851e6450903e63e78d60f89e6c",
      "figure_number": "1",
      "figure_page_1based": 2,
      "figure_bbox_points": [
        53,
        80,
        559,
        229
      ],
      "figure_acquisition_method": "Original primary PDF figure crop rendered at 216 DPI",
      "figure_sha256": "278327d8a48f992a328ccdd0fb2651b3ef502e96a966aa480342a416257ff9e5",
      "figure_bytes": 262849,
      "figure_visually_inspected": true,
      "figure_source_pdf_url": "https://arxiv.org/pdf/2602.17254",
      "mechanism_scope": "T",
      "deployment_regime": "T",
      "space_ground_path": false,
      "display_regime_en": "T · Terrestrial reference",
      "display_regime_zh": "T · 地面比較基準",
      "primary_pdf_sha256": "5ce5f24ebed0f409c81e09cd60e4eb811b97b2851e6450903e63e78d60f89e6c",
      "figure_asset": "../figures/trivance.png",
      "figure_acquisition": "Original primary PDF figure crop rendered at 216 DPI",
      "figure_content_type": "image/png"
    },
    {
      "id": "tdtcp",
      "title": "Time-division TCP for Reconfigurable Data Center Networks",
      "authors": "Shawn Shuoshuo Chen, Weiyang Wang, Christopher Canel, Srinivasan Seshan, Alex C. Snoeren, Peter Steenkiste",
      "year": "2022",
      "venue": "SIGCOMM",
      "url": "https://doi.org/10.1145/3544216.3544254",
      "pdf_url": "https://www.cs.cmu.edu/~srini/papers/papers/2022-Chen-sigcomm/2022-Chen-sigcomm.pdf",
      "regime": "T · Terrestrial reference",
      "transfer_category": "terrestrial reference",
      "source_tier": "full primary text",
      "evidence_status": "Full primary PDF: formulation, implementation, evaluation, and limitations audited",
      "evidence": [
        "§3–4 PDF pp4–8",
        "§5.1–5.4 PDF pp8–12; Figures 6–11",
        "Appendix A/C PDF pp14–17"
      ],
      "built_for_en": "Long-lived TCP flows traverse recurrent electrical/optical time-division paths with sharply different bandwidth and RTT.",
      "built_for_zh": "長壽命 TCP flow 在週期性 electrical／optical time-division path 間切換，各 path 的 bandwidth／RTT 差異顯著。",
      "problem_en": "One congestion state mixes measurements across path regimes, while independent sequence spaces create receive-window stalls. Transitions also mix data and ACK paths and induce reordering.",
      "problem_zh": "單一 congestion state 混合多種 path regime；獨立 sequence space 引發 receive-window stall。Transition 會混合 data／ACK path 並產生 reordering。",
      "design_en": "Maintain per-TDN congestion/RTT/in-flight state and a connection-wide sequence space. ToRs announce path changes by ICMP; tagged data/ACK state assigns acknowledgments to the correct TDN. Transition-aware recovery separates reordering and loss; mixed-path RTT samples are filtered. Implemented in Linux 5.8.",
      "design_zh": "保留逐 TDN congestion／RTT／in-flight state，共享 connection-wide sequence space。ToR 用 ICMP 宣告 path change；data／ACK tag 將 ACK 歸屬到正確 TDN。Transition-aware recovery 處理 reordering／loss，RTT sampling 篩選 mixed-path sample。實作於 Linux 5.8。",
      "model_en": "A piecewise transport model has one active TDN at each instant and reusable per-path CUBIC states. Correct accounting uses current-TDN, all-TDN, any-TDN or packet-specific states for each TCP operation; retransmission can use whichever path becomes available.",
      "model_zh": "Piecewise transport model 每時刻啟用一個 TDN，重用逐 path CUBIC state。TCP 操作採用 current-TDN、all-TDN、any-TDN 或 packet-specific accounting；retransmission 使用當下可用 path。",
      "evaluation_en": "Real kernel endpoints run 16 containers on each of two servers; a third server runs Click/DPDK Etalon. The emulated EPS/OCS paths are 10/100 Gbps with 100/40 μs RTT, 180 μs days and 20 μs reconfiguration. Physical ConnectX-3 40 Gbps hardware uses 20× time dilation; 16 synchronized flows run 40 s, primarily at a 6:1 EPS:OCS schedule ratio.",
      "evaluation_zh": "兩臺實體 server 各執行 16 container，第三臺 Click／DPDK Etalon 模擬網路。EPS／OCS 為 10／100 Gbps、100／40 μs RTT、180 μs day、20 μs reconfiguration。ConnectX-3 40 Gbps 硬體採 20× time dilation；16 flow 同步執行 40 s，主要採 6:1 EPS:OCS schedule ratio。",
      "baselines_en": "CUBIC, DCTCP, MPTCP with two subflows, reTCP and reTCP with dynamic VOQ resizing; analytical ideal-throughput and packet-only reference curves. Ablations vary latency/bandwidth and notification optimizations.",
      "baselines_zh": "比較 CUBIC、DCTCP、雙 subflow MPTCP、reTCP、dynamic VOQ resizing reTCP；參考 analytical ideal-throughput 與 packet-only 曲線。消融涵蓋 latency／bandwidth 與 notification optimization。",
      "result_en": "The representative hybrid setting improves long-flow throughput 24% over CUBIC/DCTCP and 41% over MPTCP, with the lowest VOQ occupancy among compared transports. Notification optimizations contribute 12.7% throughput. At the 90th percentile of transition retransmissions, TDTCP retransmits 7 packets versus CUBIC 15.",
      "result_zh": "代表 hybrid 情境的 long-flow throughput 比 CUBIC／DCTCP 高 24%、MPTCP 高 41%，VOQ occupancy 為比較 transport 中最低。Notification optimization 提升 12.7% throughput。Transition retransmission 的 90th percentile，TDTCP 為 7 packet、CUBIC 為 15。",
      "limitations_en": "Evaluation centers on long flows and two recurring path regimes, with network emulation and dilated timing. Endpoints require prompt TDN announcements and relatively stable conditions within a TDN. Larger regime sets, extreme availability ratios and short-RPC completion are additional evaluation scopes.",
      "limitations_zh": "評估聚焦 long flow 與兩種重複 path regime，採 network emulation／time dilation。Endpoint 需要及時 TDN announcement 與 TDN 內相對穩定條件。較多 regime、極端 availability ratio、short RPC completion 為額外評估範圍。",
      "math_latex": "S(t)=S_{\\operatorname{TDN}(t)},\\qquad RTT_{ij}=\\tfrac12 RTT_i+\\tfrac12 RTT_j,\\qquad i=j\\Rightarrow\\operatorname{acceptRTTSample}",
      "math_source_anchor": "§3.1–3.4 PDF pp4–6; §4.4 PDF p8: per-TDN state invariant and mixed-path RTT identity",
      "math_definitions_en": "S is congestion state; i identifies the data path and j the ACK path; matching-path samples update their own TDN estimates.",
      "math_definitions_zh": "S 為 congestion state；i 表示 data path，j 表示 ACK path；相符 path 的 sample 更新各自 TDN estimate。",
      "residual_novelty_en": "TDTCP already transfers path-specific congestion memory to predictable changing fabrics. Give it the same orbital path announcements; evaluate residual novelty around partial-window service, long propagation delays, terminal reacquisition and stale state across outages, using deadline goodput and collective iteration tails.",
      "residual_novelty_zh": "TDTCP 已將 path-specific congestion memory 用於可預測動態 fabric。提供相同 orbital path announcement 後，增量可評估 partial-window service、長 propagation delay、terminal reacquisition、outage 後 stale state，量測 deadline goodput 與 collective iteration tail。",
      "figure_caption_en": "Figure 3 depicts cross-TDN reordering from data and ACKs traversing high/low-latency regimes.",
      "figure_caption_zh": "Figure 3 說明 data／ACK 經過高／低 latency regime 時的 cross-TDN reordering。",
      "gpu_count": "Not Specified",
      "node_count": "3 physical servers; 32 containers",
      "evaluation_method": "Hardware Testbed",
      "traffic_pattern": "Elephant flows",
      "compute_memory_hw": "Intel Xeon E5-2680",
      "software_simulator": "Etalon emulator",
      "transport_and_interconnect": "TCP/IP, Ethernet",
      "routing_and_congestion_control": "DCTCP",
      "affiliations": "Carnegie Mellon University, MIT, UC San Diego",
      "source_program_url": "https://conferences.sigcomm.org/sigcomm/2022/program.html",
      "source_acquired_date": "2026-10-02",
      "publication_type": "SIGCOMM main-track terrestrial baseline",
      "pdf_sha256": "04d9159d69eb09a02b8668712906ca67f9334279ecd6105e913ff022a4fe1d81",
      "figure_number": "3",
      "figure_page_1based": 6,
      "figure_bbox_points": [
        53,
        80,
        299,
        229
      ],
      "figure_acquisition_method": "Original primary PDF figure crop rendered at 216 DPI",
      "figure_sha256": "59e701a6111a0468a616b7a2c4480b2e9344b6f646f9f84d36b994d772d40237",
      "figure_bytes": 66832,
      "figure_visually_inspected": true,
      "figure_source_pdf_url": "https://www.cs.cmu.edu/~srini/papers/papers/2022-Chen-sigcomm/2022-Chen-sigcomm.pdf",
      "mechanism_scope": "T",
      "deployment_regime": "T",
      "space_ground_path": false,
      "display_regime_en": "T · Terrestrial reference",
      "display_regime_zh": "T · 地面比較基準",
      "primary_pdf_sha256": "04d9159d69eb09a02b8668712906ca67f9334279ecd6105e913ff022a4fe1d81",
      "figure_asset": "../figures/tdtcp.png",
      "figure_acquisition": "Original primary PDF figure crop rendered at 216 DPI",
      "figure_content_type": "image/png"
    },
    {
      "id": "rotornet",
      "title": "Realizing RotorNet: Toward Practical Microsecond Scale Optical Networking",
      "authors": "William M. Mellette, Alex Forencich, Rukshani Athapathu, Alex C. Snoeren, George Papen, George Porter",
      "year": "2024",
      "venue": "SIGCOMM",
      "url": "https://doi.org/10.1145/3651890.3672273",
      "pdf_url": "https://cseweb.ucsd.edu/~snoeren/papers/rotornet-sigcomm24.pdf",
      "regime": "T · Terrestrial reference",
      "transfer_category": "terrestrial reference",
      "source_tier": "full primary text",
      "evidence_status": "Full primary PDF: formulation, implementation, evaluation, and limitations audited",
      "evidence": [
        "§3 PDF pp3–4; Figure 2",
        "§4 PDF pp5–8; Figures 8–10",
        "§5 PDF pp8–11; Figures 13–15",
        "§6 PDF pp11–12"
      ],
      "built_for_en": "An actual optically switched rack implements periodic connectivity and multi-hop Opera forwarding through commodity Linux endpoints.",
      "built_for_zh": "實際 optical rack 以 commodity Linux endpoint 實現週期連線與 multi-hop Opera forwarding。",
      "problem_en": "Fast optical topology changes require link reacquisition, accurate time synchronization, topology-aware NIC routing and masking of manufacturing-related signal dropouts across a full end-to-end stack.",
      "problem_zh": "快速 optical topology change 需要 link reacquisition、精確 time synchronization、topology-aware NIC routing，以及完整 stack 的 manufacturing signal dropout masking。",
      "design_en": "A passive rotating diffractive disk cycles restricted matchings. A 128-port switch partitions into four staggered 32-port sub-rotors. FPGA Corundum NICs use PTP to align queues, routing and guard masks to the rotor; defect windows are profiled and transmission pauses before the MAC.",
      "design_zh": "Passive rotating diffractive disk 循環 restricted matching。128-port switch 分成四個 staggered 32-port sub-rotor。Corundum FPGA NIC 以 PTP 對齊 queue／routing／guard mask；profile defect window，於 MAC 前暫停傳輸。",
      "model_en": "Periodic matchings form Opera expander topologies, with precomputed shortest viable routes and port removal ahead of reconfiguration to drain in-flight traffic. RotorNet bulk store-and-forward and Opera cut-through forwarding provide distinct architectural modes; this deployment focuses on cut-through/direct forwarding and measured link behavior.",
      "model_zh": "Periodic matching 形成 Opera expander topology，預先計算可行 shortest route，並於 reconfiguration 前移除 port 讓在途資料排空。RotorNet bulk store-and-forward 與 Opera cut-through 為兩種 mode；本部署聚焦 cut-through／direct forwarding 與 link behavior 量測。",
      "evaluation_en": "A manufactured 3U 128-port switch with 7 μs reconfiguration and 100% port yield connects 16 cluster servers plus one control server. Each cluster server uses eight 10 Gbps optical links and an FPGA NIC. End-to-end iperf3, ICMP ping, userspace UDP and all-512-path BER measurements test masking, loaded latency and synchronization.",
      "evaluation_zh": "實製 3U 128-port switch 的 reconfiguration 為 7 μs、port yield 100%，連接 16 cluster server 與一臺 control server。每臺八條 10 Gbps optical link、FPGA NIC。Iperf3、ICMP、userspace UDP、全部 512 path BER 評估 masking、loaded latency 與 synchronization。",
      "baselines_en": "Masking enabled/disabled and theoretical available link rate for TCP; ConnectX-5 back-to-back and through an electrical packet switch for ICMP/UDP latency; unloaded and interfering TCP traffic cases.",
      "baselines_zh": "TCP 比較 masking enabled／disabled 與理論 available link rate；ICMP／UDP latency 比較 ConnectX-5 back-to-back、electrical packet switch，以及 unloaded／interfering TCP traffic。",
      "result_en": "Masking raises a defective-slot iperf3 transfer from 0.893 to 0.995 of the available slot rate and removes its measured retransmissions. One-hop TCP reaches about 98% of ideal 2.5 Gbps, while multi-hop Opera reaches 9.3 Gbps on a 10 Gbps link. Rotor phase accuracy is ±5 μs; loaded overlap raises approximately the last 15% of latency samples by 15 μs.",
      "result_zh": "Masking 使 defect slot 的 iperf3 從 available slot rate 的 0.893 升至 0.995，測得 retransmission 歸零。One-hop TCP 約達理想 2.5 Gbps 的 98%；multi-hop Opera 在 10 Gbps link 達 9.3 Gbps。Rotor phase accuracy 為 ±5 μs；loaded overlap 使約最後 15% latency sample 增加 15 μs。",
      "limitations_en": "The prototype operates at 10 Gbps per optical lane, while its 128-port device is deployed through a 16-node partition. Large-scale load balancing, NIC external memory, RotorLB bulk forwarding and application/transport integration are continuing work. This is a terrestrial optical component validation.",
      "limitations_zh": "Prototype 每 optical lane 為 10 Gbps，128-port device 以 16-node partition 部署。Large-scale load balancing、NIC external memory、RotorLB bulk forwarding、application／transport integration 為後續工作。證據涵蓋 terrestrial optical component validation。",
      "math_latex": "\\operatorname{TX}_p(t)=1\\Longrightarrow t\\in\\operatorname{activeMatching}_p\\cap\\operatorname{guardMask}_p^{\\complement};\\qquad \\operatorname{routeArrival}<\\operatorname{reconfigurationDeadline}",
      "math_source_anchor": "§3.2–3.4 PDF p4; §4.5 PDF p8: timing/guard-mask design invariant, notation transcribed for this survey",
      "math_definitions_en": "p is an optical port; active matching determines its peer; guard masks remove switch and defect intervals; routes reserve enough time to finish before circuit change.",
      "math_definitions_zh": "p 為 optical port；active matching 決定 peer；guard mask 排除 switch／defect interval，route 留出 circuit change 前完成的時間。",
      "residual_novelty_en": "Transfer the timing, guard-band and full-stack validation discipline. Orbital use requires measured acquisition/tracking and radiation/thermal qualification for its own free-space terminals, followed by transport correctness across actual contact gaps; rotor-disk timing is a terrestrial component parameter.",
      "residual_novelty_zh": "可移植 timing、guard-band 與 full-stack validation 方法。軌道場景需量測自身 free-space terminal 的 acquisition／tracking 與 radiation／thermal qualification，再驗證實際 contact gap 的 transport correctness；rotor-disk timing 屬 terrestrial component 參數。",
      "figure_caption_en": "Figure 2 links Linux application, FPGA NIC, PTP time synchronization and the optical rotor dataplane.",
      "figure_caption_zh": "Figure 2 連結 Linux application、FPGA NIC、PTP time synchronization 與 optical rotor dataplane。",
      "gpu_count": "Not Specified",
      "node_count": "16 cluster servers plus 1 control server",
      "evaluation_method": "Hardware Testbed",
      "compute_memory_hw": "FPGA NIC",
      "network_hw": "ConnectX-5",
      "transport_and_interconnect": "TCP/IP, Ethernet",
      "affiliations": "inFocus Networks, UC San Diego",
      "source_program_url": "https://conferences.sigcomm.org/sigcomm/2024/program/",
      "source_acquired_date": "2026-10-02",
      "publication_type": "SIGCOMM main-track terrestrial baseline",
      "pdf_sha256": "7b4975d7ca8dafccc7d65b8ecc4e5bde0f7cdbaeff8df4680510e32c12665dfe",
      "figure_number": "2",
      "figure_page_1based": 4,
      "figure_bbox_points": [
        53,
        80,
        559,
        201
      ],
      "figure_acquisition_method": "Original primary PDF figure crop rendered at 216 DPI",
      "figure_sha256": "6208b035d4deed06ee5cc85bc09d60d4aeafd9ed60d677fe11c574893d430788",
      "figure_bytes": 57716,
      "figure_visually_inspected": true,
      "figure_source_pdf_url": "https://cseweb.ucsd.edu/~snoeren/papers/rotornet-sigcomm24.pdf",
      "software_simulator": "Not Specified",
      "mechanism_scope": "T",
      "deployment_regime": "T",
      "space_ground_path": false,
      "display_regime_en": "T · Terrestrial reference",
      "display_regime_zh": "T · 地面比較基準",
      "primary_pdf_sha256": "7b4975d7ca8dafccc7d65b8ecc4e5bde0f7cdbaeff8df4680510e32c12665dfe",
      "figure_asset": "../figures/rotornet.png",
      "figure_acquisition": "Original primary PDF figure crop rendered at 216 DPI",
      "figure_content_type": "image/png"
    },
    {
      "id": "flare",
      "title": "Unlocking Superior Performance in Reconfigurable Data Center Networks with Credit-Based Transport",
      "authors": "Federico De Marchi, Jialong Li, Ying Zhang, Wei Bai, Yiting Xia",
      "year": "2025",
      "venue": "SIGCOMM",
      "url": "https://doi.org/10.1145/3718958.3754342",
      "pdf_url": "https://pure.mpg.de/rest/items/item_3678859_1/component/file_3678860/content",
      "regime": "T · Terrestrial reference",
      "transfer_category": "terrestrial reference",
      "source_tier": "full primary text",
      "evidence_status": "Full primary PDF: formulation, implementation, evaluation, and limitations audited",
      "evidence": [
        "§3–4 PDF pp3–7; Equation (1)",
        "§5 PDF p8; implementation",
        "§6 PDF pp8–11; Figures 14–19",
        "§7–8 PDF pp11–12; Figures 20–21",
        "Appendix A–C PDF pp15–17"
      ],
      "built_for_en": "Opera-like varying-expander fabrics retain multi-hop routes during microsecond optical reconfiguration and carry general datacenter traffic.",
      "built_for_zh": "Opera 類 varying-expander fabric 在 microsecond optical reconfiguration 期間維持 multi-hop route，承載 general datacenter traffic。",
      "problem_en": "Credit allocation faces multiple changing bottlenecks. Credits and data must use the same path despite reconfiguration, while credits dropped later waste earlier link opportunities; path-length bias also affects flow fairness.",
      "problem_zh": "Credit allocation 面臨多個 changing bottleneck。Credit／data 在 reconfiguration 下需要相同 path；下游 credit drop 會浪費上游容量，path-length bias 影響 fairness。",
      "design_en": "Enforce reverse credit–data symmetry using embedded entry time slices and RTT-length slots. Congested credit queues probabilistically favor fewer remaining hops; topology-aware rate updates and tentative credits fill spare capacity. Hop-count jitter helps short flows, and Aeolus provides fast start.",
      "design_zh": "以 embedded entry time slice 與 RTT-length slot 維持 credit／data reverse-path symmetry。Congested credit queue 以機率偏好剩餘 hop 較少者；topology-aware rate update 與 tentative credit 使用剩餘容量。Hop-count jitter 照顧 short flow，Aeolus 提供 fast start。",
      "model_en": "The design is a stateless stochastic admission heuristic with per-flow receiver feedback. Admission probability is 2^(1−h) above the congestion threshold, where h is remaining hops; credit/data scheduling obeys contact-slot timing. Long-term fairness relies on repeated near-uniform short-path opportunities.",
      "model_zh": "設計採 stateless stochastic admission heuristic 與 receiver feedback。Congestion threshold 以上 admission probability 為 2^(1−h)，h 為 remaining hop；credit／data 排程服從 slot timing。Long-term fairness 依賴反覆近乎均勻的 short-path 機會。",
      "evaluation_en": "htsim models 108 ToRs, six 108-port OCSes and 648 hosts, 100 Gbps links, 500 ns inter-ToR propagation and 15/55 μs slices. Web-search/Hadoop/RPC workloads run at 30% host load with 35% stress. A DPDK/P4 testbed uses three Tofino2 switches, four dual-port ConnectX-6 servers, eight emulated ToRs, 40 Gbps and 100 μs slices.",
      "evaluation_zh": "Htsim 建模 108 ToR、六臺 108-port OCS、648 host、100 Gbps、500 ns inter-ToR propagation、15／55 μs slice。Web-search／Hadoop／RPC 採 30% host load、35% stress。DPDK／P4 testbed 採三臺 Tofino2、四臺 dual-port ConnectX-6 server、八 emulated ToR、40 Gbps、100 μs slice。",
      "baselines_en": "NDP, ExpressPass enhanced with Aeolus, TDTCP tuned for Opera, Bolt with TDTCP reordering handling, and a Shale-inspired HbH baseline with ideal loss recovery. A similar-cost 3:1 oversubscribed Clos runs Bolt; full settings are in Appendix A.",
      "baselines_zh": "比較 NDP、加 Aeolus 的 ExpressPass、Opera-tuned TDTCP、加入 TDTCP reordering handling 的 Bolt，以及 Shale-inspired、ideal loss recovery HbH。Similar-cost 3:1 oversubscribed Clos 使用 Bolt；完整參數於 Appendix A。",
      "result_en": "Across stated simulations, throughput reaches up to 2× NDP and 1.5× ExpressPass; tail FCT reductions reach 10×/15×/3.5× against ExpressPass/TDTCP/Bolt. Credit waste falls to 4.4% versus ExpressPass 14.2%. At 35% load, Opera+Flare sustains over 34% throughput and reduces long-flow FCT up to 2.48× versus Clos+Bolt. Hardware tests validate fairness/queues rather than reproducing all headline ratios.",
      "result_zh": "指定 simulation 中 throughput 最高為 NDP 的 2×、ExpressPass 的 1.5×；對 ExpressPass／TDTCP／Bolt 的 tail FCT reduction 最高 10×／15×／3.5×。Credit waste 為 4.4%，ExpressPass 為 14.2%。35% load 下 Opera+Flare throughput 超過 34%，long-flow FCT 比 Clos+Bolt 最多降低 2.48×。硬體測試驗證 fairness／queue，headline ratio 來自 simulation。",
      "limitations_en": "The protocol assumes time-synchronized circuit schedules and sufficient path lifetime for a round trip. The small hardware graph exhibits stronger path-length/fairness imbalance than the simulated topology. Priority scheduling and background-traffic policies require additional mechanisms; LLM collective end-to-end training is a further workload scope.",
      "limitations_zh": "Protocol 假設 time-synchronized circuit schedule 與足供 round trip 的 path lifetime。小硬體 graph 的 path-length／fairness imbalance 較大。Priority scheduling／background policy 需額外機制，LLM collective 端到端 training 屬後續 workload 範圍。",
      "math_latex": "P(h)=\\left(\\tfrac12\\right)^{h-1};\\qquad T_{\\operatorname{slice}}\\ge RTT_{\\operatorname{path}},\\quad \\operatorname{route}(\\operatorname{data})=\\operatorname{reverse}(\\operatorname{route}(\\operatorname{credit}))",
      "math_source_anchor": "§4.2 Equation (1), PDF p6; §4.1 PDF pp5–6 credit–data path symmetry",
      "math_definitions_en": "h is remaining hop count; admission is probabilistic above the queue congestion threshold. The slot inequality expresses the protocol timing design requirement.",
      "math_definitions_zh": "h 為 remaining hop count；queue congestion threshold 以上使用機率 admission。Slot inequality 表示 protocol timing requirement。",
      "residual_novelty_en": "Expose the same contact schedule to Flare and its path-symmetry control. Residual orbital novelty concerns long RTT versus contact duration, asymmetric acquisition/outage windows, collective-critical rather than shortest-hop credit priorities, and service guarantees under forecast error.",
      "residual_novelty_zh": "提供 Flare 相同 contact schedule 與 path-symmetry control。增量可研究長 RTT／contact duration 關係、asymmetric acquisition／outage window、collective-critical credit priority，以及 forecast error 下的 service guarantee。",
      "figure_caption_en": "Figure 7 separates regular and tentative credit queue thresholds and the hop-dependent admission probabilities.",
      "figure_caption_zh": "Figure 7 區分 regular／tentative credit queue threshold，並呈現 hop-dependent admission probability。",
      "gpu_count": "Not Specified",
      "node_count": "4 physical servers; 648 simulated hosts",
      "evaluation_method": "Simulation, Hardware Testbed",
      "traffic_pattern": "Incast, All-to-All, Elephant flows",
      "software_simulator": "htsim",
      "network_hw": "Tofino",
      "network_topology": "Clos",
      "transport_and_interconnect": "Ethernet",
      "routing_and_congestion_control": "ECMP",
      "affiliations": "Max Planck Institute for Informatics, Microsoft",
      "source_program_url": "https://conferences.sigcomm.org/sigcomm/2025/program/papers-info/",
      "source_acquired_date": "2026-10-02",
      "publication_type": "SIGCOMM main-track terrestrial baseline",
      "pdf_sha256": "ac03c89eab43c2fc242910d31c64b766dd89197105038d398dba1fbc9913e1d2",
      "figure_number": "7",
      "figure_page_1based": 6,
      "figure_bbox_points": [
        53,
        80,
        299,
        162
      ],
      "figure_acquisition_method": "Original primary PDF figure crop rendered at 216 DPI",
      "figure_sha256": "99f8ede4de856f5c0c8e93bdf7b5f369a7b0b7cf35dd96cf163f34565ba02340",
      "figure_bytes": 31524,
      "figure_visually_inspected": true,
      "figure_source_pdf_url": "https://pure.mpg.de/rest/items/item_3678859_1/component/file_3678860/content",
      "mechanism_scope": "T",
      "deployment_regime": "T",
      "space_ground_path": false,
      "display_regime_en": "T · Terrestrial reference",
      "display_regime_zh": "T · 地面比較基準",
      "primary_pdf_sha256": "ac03c89eab43c2fc242910d31c64b766dd89197105038d398dba1fbc9913e1d2",
      "figure_asset": "../figures/flare.png",
      "figure_acquisition": "Original primary PDF figure crop rendered at 216 DPI",
      "figure_content_type": "image/png"
    },
    {
      "id": "theseus",
      "title": "Theseus: Runtime-Adaptive GPU Collective Communication with Hot-Swappable Schedules",
      "authors": "Rui Ding, Xiandong Lu, Jiajun Wang, Xunpeng Liu, Feiyang Wang, Xuran Hao, Houyuan Zhu, Anyi Xu, Sinuo Cao, Haifeng Sun, Qun Huang, Jiamin Cao, Jiaqi Gao",
      "year": "2026",
      "venue": "SIGCOMM",
      "url": "https://doi.org/10.1145/3789240.3829134",
      "pdf_url": "https://dl.acm.org/doi/pdf/10.1145/3789240.3829134",
      "regime": "T · Terrestrial reference",
      "transfer_category": "terrestrial reference",
      "source_tier": "full primary ACM eReader text",
      "evidence_status": "Full public ACM eReader: all 20 PDF pages captured",
      "evidence": [
        "§3–6 PDF pp4–9; Algorithm 1/Table 2",
        "§8.1–8.2 PDF pp9–11; Table 3/Figures 11–12",
        "§8.3 PDF pp11–12; Figures 13–16",
        "§9 PDF p12; assumptions §3 PDF p5"
      ],
      "built_for_en": "Long-running GPU applications select among custom collective schedules as expert loads, NIC health and synthesis quality evolve.",
      "built_for_zh": "長期 GPU application 因 expert load、NIC health、synthesis quality 改變而切換 custom collective schedule。",
      "problem_en": "Runtime schedule choices need deterministic agreement across ranks, resource-efficient support for hundreds of candidates and small switching costs while preserving collective execution semantics.",
      "problem_zh": "Runtime schedule choice 需要跨 rank deterministic agreement、高效支援數百候選 schedule，並以小型 switching cost 維持 collective execution semantics。",
      "design_en": "An asynchronous selection context holds endogenous/exogenous attributes, schedules and policies. Every K requests, a custom AllReduce agrees on common context elements and minimum visible versions; each rank then deterministically selects the highest-scoring usable schedule. A four-layer resource tree and lazy delta migration reuse scratch buffers, schedule resources, buffer handles and low-level operations.",
      "design_zh": "Async selection context 保存內／外部 attribute、schedule 與 policy。每 K 個 request 以 custom AllReduce 協議共同 element 與最低可見 version，各 rank 依 deterministic rule 選最高 score 可用 schedule。四層 resource tree 與 lazy delta migration 重用 scratch、schedule resource、buffer handle、low-level operation。",
      "model_en": "The agreement protocol uses set intersection and minimum-version reduction; an FSM enforces consistent downward resource-tree traversal. Policies include MoE load divergence, fail-slow health maps, improving synthesizer output and uniform-exploration bandit selection. Theseus supplies a runtime mechanism; user policies define optimization goals.",
      "model_zh": "Agreement protocol 採 set intersection／minimum-version reduction，FSM 保持一致的向下 resource-tree traversal。Policy 包括 MoE load divergence、fail-slow health map、持續改善 synthesis output、uniform-exploration bandit。Theseus 提供 runtime mechanism，optimization goal 由 user policy 定義。",
      "evaluation_en": "Four nodes contain 32 A100 SXM4 80 GB GPUs, six NVSwitches per node and four ConnectX-5 dual-100 Gbps NICs per node on two-tier Clos. CUDA 12.4, PyTorch 2.2 and Megatron-LM 0.16 run Qwen3-30B-A3B EP8/EP32 and Llama3-8B DP; heFFTe exercises 16-GPU AllToAll. Agreement uses K=100; communicator tests sweep 2–32 GPUs and 10–1000 schedules.",
      "evaluation_zh": "四 node 共 32 A100 SXM4 80 GB，每 node 六 NVSwitch、四張 dual-100 Gbps ConnectX-5，採 two-tier Clos。CUDA 12.4、PyTorch 2.2、Megatron-LM 0.16 執行 Qwen3-30B-A3B EP8／EP32、Llama3-8B DP；heFFTe 執行 16-GPU AllToAll。K=100，communicator test 掃描 2–32 GPU 與 10–1000 schedule。",
      "baselines_en": "NCCL v2.19 with its default tuner and MSCCL++ v0.8; model experiments use Megatron-LM with NCCL, and fail-slow adaptation enabled/disabled. heFFTe compares default, waiting for final TE-CCL synthesis, and continuous adoption of intermediate schedules; resource-tree/delta-migration ablations isolate overhead.",
      "baselines_zh": "比較 default tuner 的 NCCL v2.19、MSCCL++ v0.8；model experiment 採 Megatron-LM＋NCCL、fail-slow adaptation enabled／disabled。HeFFTe 比較 default、等待 TE-CCL 最終排程、持續採用中間排程；resource-tree／delta-migration 消融隔離 overhead。",
      "result_en": "Static communicator bandwidth improves up to 1.61× NCCL. Qwen MoE AllToAll reaches 1.73× at high skewness while actual training improves about 1.07×. A fully throttled NIC yields Llama iteration time 2.03 s versus healthy 1.75 s, while the unadapted 10%-NIC case reaches 3.76 s. Delta migration averages 8±5.1 ms versus 206±107 ms normal setup; average request TTE is 0.13 ms and over 95% are below 10 μs. Abstract maxima are 2.46× dynamic communication and 1.84× job time.",
      "result_zh": "Static communicator bandwidth 最高達 NCCL 的 1.61×。Qwen MoE 高 skewness AllToAll 達 1.73×，實際 training 約 1.07×。NIC 完全限速時 Llama iteration 為 2.03 s，healthy 為 1.75 s；原策略在 NIC 10% capacity 為 3.76 s。Delta migration 平均 8±5.1 ms，normal setup 206±107 ms；平均 request TTE 0.13 ms、超過 95% 低於 10 μs。摘要 maxima 為 dynamic communication 2.46×、job time 1.84×。",
      "limitations_en": "Control agreement assumes a connected bootstrap network, consistent request ordering, intra-node peer access and GPUDirect RDMA. Membership repair uses manual/preemptive actions. Schedule quality and stability depend on user policy; arbitrary policies can oscillate, with hysteresis/dwell guardrails available. Rare setup and agreement requests produce longer TTE tails.",
      "limitations_zh": "Control agreement 假設 connected bootstrap network、一致 request order、host 內 peer access、GPUDirect RDMA。Membership repair 採人工／預先處理。Schedule quality／stability 依 user policy，policy 可產生 oscillation，提供 hysteresis／dwell guardrail。少數 setup／agreement request 增加 TTE tail。",
      "math_latex": "\\mathcal A^*=\\bigcap_r\\mathcal A_r,\\quad\\mathcal S^*=\\bigcap_r\\mathcal S_r,\\quad v_a^*=\\min_r v_{a,r};\\qquad s^*=\\arg\\max_{s\\in\\mathcal S^*,\\,u_s(C^*,R)}\\operatorname{score}_s(C^*,R)",
      "math_source_anchor": "§4.1 PDF p5 schedule-selection model; §5 Algorithm 1 PDF p7, lines 9–13; notation transcribed for this survey",
      "math_definitions_en": "A/S are visible attribute/schedule sets; v is a version; u is the usability predicate; common context plus deterministic tie-breaks produces consistent schedules.",
      "math_definitions_zh": "A／S 為可見 attribute／schedule set；v 為 version；u 為 usability predicate；共同 context 與 deterministic tie-break 產生一致 schedule。",
      "residual_novelty_en": "Give Theseus orbital contact/health forecasts as exogenous attributes and the same synthesized schedule candidates. Residual contribution must address control-path partitions and delayed context agreement, contact-deadline-correct swapping with in-flight dependencies, and useful end-to-end gains after the existing runtime gets equivalent forecasts.",
      "residual_novelty_zh": "將 orbital contact／health forecast 作為 Theseus exogenous attribute，提供相同合成候選 schedule。增量需處理 control-path partition、延遲 context agreement、含在途 dependency 的 contact-deadline-correct swapping，以及既有 runtime 取得同等 forecast 後的端到端收益。",
      "figure_caption_en": "Figure 6 shows selection context agreement, deterministic selection and resource-materializing execution.",
      "figure_caption_zh": "Figure 6 呈現 selection context agreement、deterministic selection 與 resource-materializing execution。",
      "open_source": "https://github.com/N2-Sys/Theseus",
      "gpu_count": "32 NVIDIA A100 SXM4 80GB",
      "node_count": "4 servers",
      "evaluation_method": "Hardware Testbed",
      "traffic_pattern": "All-to-All, AllGather, AllReduce, ReduceScatter",
      "compute_memory_hw": "NVIDIA A100",
      "network_topology": "Clos",
      "network_hw": "ConnectX-5, NVSwitch",
      "transport_and_interconnect": "RDMA, RoCEv2, NVLink",
      "comm_libraries": "NCCL, MSCCL++",
      "affiliations": "Peking University, University of Chinese Academy of Sciences, National University of Singapore, Alibaba Cloud",
      "source_program_url": "https://conferences.sigcomm.org/sigcomm/2026/program/papers/",
      "source_acquired_date": "2026-10-02",
      "publication_type": "SIGCOMM main-track terrestrial baseline",
      "figure_number": "6",
      "figure_page_1based": 4,
      "figure_bbox_points": [
        324.6,
        75,
        551.4,
        212.4
      ],
      "figure_sha256": "bbb3ab5d10031df59f6ffb16f30ca9b6eeb50555bd534bb373aa667123aa21a3",
      "pdf_sha256": "",
      "primary_text_sha256": "69a1790f3f63221459142a0cd2fdf368e5829857e3abbfca50bf4ed9a2d9776f",
      "figure_acquisition_method": "Original ACM eReader full-viewport screenshot followed by image crop",
      "figure_bytes": 108948,
      "figure_visually_inspected": true,
      "figure_source_pdf_url": "https://dl.acm.org/doi/pdf/10.1145/3789240.3829134",
      "software_simulator": "Not Specified",
      "mechanism_scope": "T",
      "deployment_regime": "T",
      "space_ground_path": false,
      "display_regime_en": "T · Terrestrial reference",
      "display_regime_zh": "T · 地面比較基準",
      "figure_provenance": {
        "id": "theseus",
        "source_pdf_url": "https://dl.acm.org/doi/pdf/10.1145/3789240.3829134",
        "source_reader_url": "https://dl.acm.org/doi/epdf/10.1145/3789240.3829134",
        "figure_number": 6,
        "figure_page_1based": 4,
        "figure_bbox_points": [
          324.6,
          75,
          551.4,
          212.4
        ],
        "capture_method": "Original ACM eReader full-viewport screenshot followed by image crop",
        "screenshot_bbox_pixels": [
          979,
          193,
          1357,
          422
        ],
        "figure_sha256": "bbb3ab5d10031df59f6ffb16f30ca9b6eeb50555bd534bb373aa667123aa21a3",
        "reader_text_sha256": "69a1790f3f63221459142a0cd2fdf368e5829857e3abbfca50bf4ed9a2d9776f",
        "bytes": 108948,
        "visually_inspected": true
      },
      "figure_asset": "../figures/theseus.png",
      "figure_acquisition": "Original ACM eReader full-viewport screenshot followed by image crop",
      "figure_content_type": "image/png"
    },
    {
      "id": "cachegen",
      "title": "CacheGen: KV Cache Compression and Streaming for Fast Large Language Model Serving",
      "authors": "Yuhan Liu; Hanchen Li; Yihua Cheng; Siddhant Ray; Yuyang Huang; Qizheng Zhang; Kuntai Du; Jiayi Yao; Shan Lu; Ganesh Ananthanarayanan; Michael Maire; Henry Hoffmann; Ari Holtzman; Junchen Jiang",
      "year": "2024",
      "venue": "SIGCOMM",
      "url": "https://doi.org/10.1145/3651890.3672274",
      "pdf_url": "https://cs.stanford.edu/~keithw/sigcomm2024/sigcomm24-final1571-acmpaginated.pdf",
      "regime": "T · Terrestrial reference",
      "arxiv": "2310.07240",
      "affiliations": "University of Chicago, Microsoft Research, Stanford University",
      "evidence_status": "Complete primary full text reviewed",
      "source_version": "Conference primary paper",
      "evidence": [
        "§3–§5 PDFp4–7: statistics, arithmetic coding, adaptive chunk streaming",
        "§7.1–§7.2 PDFp8–9: workload, hardware, baseline and main results",
        "§7.3–§7.5 PDFp10–11: bandwidth/concurrency/quality sensitivity and adaptation",
        "Figure1 PDFp2"
      ],
      "built_for_en": "Repeated long contexts are reused across terrestrial LLM serving requests through compressed KV bitstreams fetched from storage or another worker.",
      "built_for_zh": "地面 LLM serving request 重用長 context，透過從 storage 或其他 worker 取得的 compressed KV bitstream 載入狀態。",
      "problem_en": "Sending full KV tensors costs bandwidth; sending text instead incurs prefill computation. A serving system must choose a context representation that meets TTFT and response-quality targets under changing bandwidth.",
      "problem_zh": "完整 KV tensor 傳輸消耗 bandwidth；text 傳輸則增加 prefill computation。Serving system 需隨 bandwidth 變化選擇 context representation，以滿足 TTFT 與 response-quality target。",
      "design_en": "Change-based encoding exploits adjacent-token locality; layer-sensitive quantization and channel-layer probability models feed arithmetic coding. A streamer adapts each context chunk between compression levels and text recomputation, while GPU decoding overlaps transmission.",
      "design_zh": "Change-based encoding 利用相鄰 token locality；layer-sensitive quantization 與 channel-layer probability model 提供 arithmetic coding 輸入。Streamer 逐 context chunk 選擇 compression level 或 text recomputation，GPU decoding 與 transmission 重疊。",
      "model_en": "Tensor statistics and layer-quality sensitivity determine encoding. Online selection uses observed bandwidth, chunk size, available GPU computation, and a TTFT budget; experiments report quality–size and quality–TTFT tradeoffs rather than a constellation optimization model.",
      "model_zh": "Tensor statistics 與 layer-quality sensitivity 決定 encoding。Online selection 使用 observed bandwidth、chunk size、available GPU computation 與 TTFT budget；experiment 呈現 quality–size、quality–TTFT tradeoff。",
      "evaluation_en": "A four-A40 server evaluates Mistral-7B and long-context Llama-34B/70B variants on 662 contexts from LongChat, TriviaQA, WikiText, and NarrativeQA. The main comparison uses 3Gbps; sensitivity spans 0.4–400Gbps, context length and concurrency. Random per-chunk bandwidth traces cover 0.1–10Gbps, averaged across 20 traces.",
      "evaluation_zh": "四張 A40 的 server 評估 Mistral-7B 與長 context Llama-34B/70B variant，使用 LongChat、TriviaQA、WikiText、NarrativeQA 共 662 個 context。主要 comparison 採 3Gbps；sensitivity 涵蓋 0.4–400Gbps、context length 與 concurrency。逐 chunk 隨機 bandwidth trace 範圍為 0.1–10Gbps，平均 20 條 trace。",
      "baselines_en": "Uniform 3/4/8-bit KV quantization; text-context prefill through vLLM; H2O and LLMlingua context compression; encoding and adaptation ablations. H2O is evaluated with an idealized offline query-tensor assumption.",
      "baselines_zh": "Uniform 3/4/8-bit KV quantization；透過 vLLM 的 text-context prefill；H2O 與 LLMlingua context compression；encoding、adaptation ablation。H2O evaluation 採可離線取得 query tensor 的 idealized assumption。",
      "result_en": "At 3Gbps, measured TTFT improves 3.2–3.7× over default quantization and 3.1–4.7× over text prefill; KV size improves 3.5–4.3× at similar task quality. Under a 1s TTFT SLO, adaptation reduces violation rate from 81% to 8% in the reported random-bandwidth experiment.",
      "result_zh": "3Gbps 下，measured TTFT 相對 default quantization 加速 3.2–3.7×，相對 text prefill 加速 3.1–4.7×；similar task quality 下，KV size 縮小 3.5–4.3×。在 reported random-bandwidth experiment 的 1s TTFT SLO 下，adaptation 將 violation rate 從 81% 降至 8%。",
      "limitations_en": "The evidence establishes compression and bandwidth-adaptive context streaming on terrestrial hardware. Orbital residual novelty can couple chunk decisions to contact expiration, cache-commit completion, solar/thermal budgets, and losses across interrupted transfers. A fair orbital baseline supplies CacheGen the same contact forecast and compression choices as the proposed scheduler.",
      "limitations_zh": "Evidence 支持地面 hardware 的 compression 與 bandwidth-adaptive context streaming。Orbital residual novelty 可將 chunk decision 結合 contact expiration、cache-commit completion、solar/thermal budget 與 interrupted-transfer loss。公平 orbital baseline 需讓 CacheGen 與 proposed scheduler 取得相同 contact forecast、compression choice。",
      "figure_caption_en": "Original Figure 1 compares full KV-context sharing with compact KV bitstreams. CacheGen changes transferred representation; an orbital extension additionally schedules representation against the remaining contact capacity.",
      "figure_caption_zh": "原始 Figure 1 比較 full KV-context sharing 與 compact KV bitstream。CacheGen 改變 transferred representation；orbital extension 進一步依 remaining contact capacity 排程 representation。",
      "figure_number": 1,
      "figure_page_1based": 2,
      "figure_bbox_points": [
        52.758620689655174,
        83.16,
        296.10775862068965,
        219.12
      ],
      "figure_sha256": "30dd35da0b6ae92c5da2ca96156fc82da516f40f434a90a4bd95ea322b92ce2e",
      "mechanism_scope": "T — Terrestrial KV transfer reference",
      "deployment_regime": "T",
      "space_ground_path": false,
      "display_regime_en": "T · Terrestrial reference",
      "display_regime_zh": "T · 地面比較基準",
      "primary_pdf_sha256": "bff4b9d4f9bfdf0bb04d4f94677d009263d4e79b046515327354e961c6dc2e6d",
      "figure_provenance": {
        "id": "cachegen",
        "figure_number": 1,
        "figure_page_1based": 2,
        "figure_bbox_points": [
          52.758620689655174,
          83.16,
          296.10775862068965,
          219.12
        ],
        "pdf_sha256": "bff4b9d4f9bfdf0bb04d4f94677d009263d4e79b046515327354e961c6dc2e6d",
        "figure_sha256": "30dd35da0b6ae92c5da2ca96156fc82da516f40f434a90a4bd95ea322b92ce2e",
        "bytes": 93133,
        "dimensions": [
          845,
          472
        ],
        "capture_method": "Original primary PDF rasterized at 250 dpi and cropped to figure plus complete source caption",
        "visually_inspected": true,
        "source_pdf_url": "https://cs.stanford.edu/~keithw/sigcomm2024/sigcomm24-final1571-acmpaginated.pdf"
      },
      "figure_acquisition_method": "Original primary PDF rasterized at 250 dpi and cropped to figure plus complete source caption",
      "figure_asset": "../figures/cachegen.png",
      "figure_acquisition": "Original primary PDF rasterized at 250 dpi and cropped to figure plus complete source caption",
      "figure_content_type": "image/png",
      "figure_bytes": 93133
    },
    {
      "id": "kvserve",
      "title": "KVServe: Service-Aware KV Cache Compression for Communication-Efficient Disaggregated LLM Serving",
      "authors": "Zedong Liu; Xinyang Ma; Dejun Luo; Hairui Zhao; Bing Lu; Wenjing Huang; Yida Gu; Xingchen Liu; Zheng Wei; Jinyang Liu; Dingwen Tao; Guangming Tan",
      "year": "2026",
      "venue": "SIGCOMM",
      "url": "https://doi.org/10.1145/3789240.3829139",
      "pdf_url": "https://arxiv.org/pdf/2605.13734",
      "regime": "T · Terrestrial reference",
      "arxiv": "2605.13734",
      "affiliations": "University of Chinese Academy of Sciences, Institute of Computing Technology Chinese Academy of Sciences, Shanghai Jiao Tong University",
      "evidence_status": "Complete primary full text reviewed",
      "source_version": "arXiv2605.13734 primary full text; official SIGCOMM2026 conference identity",
      "evidence": [
        "§3.2 PDFp5 Equations1–3: constrained latency/quality model",
        "§5 PDFp6–7: Bayesian optimization and Pareto search",
        "§6 PDFp8–9: service-aware controller and bandit",
        "§7.1–§7.4 PDFp9–12: exact hardware, model, baseline and gains",
        "Figure6 PDFp5"
      ],
      "built_for_en": "Terrestrial PD-disaggregated serving and prefix-cache offloading select KV compression by workload quality and service latency.",
      "built_for_zh": "地面 PD-disaggregated serving 與 prefix-cache offloading 依 workload quality、service latency 選擇 KV compression。",
      "problem_en": "Fixed compression pipelines trade communication savings against GPU compression cost and task accuracy; the best profile changes with workload and effective bandwidth.",
      "problem_zh": "固定 compression pipeline 在 communication saving、GPU compression cost、task accuracy 之間取捨；最佳 profile 隨 workload 與 effective bandwidth 改變。",
      "design_en": "A composable Transform→Quantizer→Codec pool includes mixed-precision head-wise quantization. Constraint-aware Gaussian-process Bayesian optimization constructs a three-dimensional quality/ratio/throughput Pareto frontier. The controller combines an analytic benefit boundary with a small epsilon-greedy bandit and EWMA latency residuals.",
      "design_zh": "可組合 Transform→Quantizer→Codec pool 包含 mixed-precision head-wise quantization。Constraint-aware Gaussian-process Bayesian optimization 建構 quality/ratio/throughput 三維 Pareto frontier。Controller 結合 analytic benefit boundary、小型 epsilon-greedy bandit 與 EWMA latency residual。",
      "model_en": "Equation 1 models profile latency as T_p(c)=T_model(w)+V/s_p+V/(B·cr_p), with baseline T_0=T_model+V/B. Equations 2–3 minimize T_p subject to T_p≤T_SLO and q_p(w)≥q_min. V is original KV bytes, B effective bytes/s, s_p effective compression-plus-decompression bytes/s, and cr_p compression ratio. The model assumes a fixed model/serving configuration within a decision segment.",
      "model_zh": "Equation1 將 profile latency 建模為 T_p(c)=T_model(w)+V/s_p+V/(B·cr_p)，baseline 為 T_0=T_model+V/B。Equations2–3 在 T_p≤T_SLO、q_p(w)≥q_min 下最小化 T_p。V 為 original KV byte，B 為 effective bytes/s，s_p 為 effective compression-plus-decompression bytes/s，cr_p 為 compression ratio。Model 在 decision segment 內採固定 model/serving configuration。",
      "evaluation_en": "vLLM0.10.1 and lm-eval-harness evaluate Qwen2.5-7B/32B-Instruct and Llama-3.1-8B-Instruct. Profiling uses four A100-40GB GPUs; serving uses RTX4090/5090, RTXPro6000 and H100 tiers at 10/50/100Gbps. Four profiling datasets and two held-out QA datasets assess generalization; Linux/NIC rate control sweeps bandwidth.",
      "evaluation_zh": "vLLM0.10.1 與 lm-eval-harness 評估 Qwen2.5-7B/32B-Instruct、Llama-3.1-8B-Instruct。Profiling 使用四張 A100-40GB；serving 使用 RTX4090/5090、RTXPro6000、H100 tier，對應 10/50/100Gbps。四種 profiling dataset 與兩種 held-out QA dataset 評估 generalization；Linux/NIC rate control 掃描 bandwidth。",
      "baselines_en": "Uncompressed BF16; integrated CacheGen and KIVI modules; DuoAttention for pruning/quality comparisons; Unified versus service-aware profiles and offline/online controller ablations. The accuracy constraint is 97% relative to the BF16 task baseline.",
      "baselines_zh": "Uncompressed BF16；整合 CacheGen、KIVI module；DuoAttention 用於 pruning/quality comparison；Unified 與 service-aware profile，以及 offline/online controller ablation。Accuracy constraint 為相對 BF16 task baseline 的 97%。",
      "result_en": "Reported maxima are 9.13× lower JCT in PD serving on HotpotQA and 32.8× lower TTFT in prefix caching under the evaluated constrained-bandwidth settings. Table 1 reports 100.35% mean relative accuracy and 8.28× mean compression ratio for service-aware profiles; these are workload averages within that table.",
      "result_zh": "Reported maximum 為 evaluated constrained-bandwidth setting 下，HotpotQA 的 PD serving JCT 加速 9.13×，prefix caching TTFT 加速 32.8×。Table1 的 service-aware profile 呈現 100.35% mean relative accuracy 與 8.28× mean compression ratio；兩者為該 table 的 workload average。",
      "limitations_en": "Service-aware bandwidth/quality control already defines a strong compression baseline. Orbital novelty must add coupled placement, finite contact deadlines, battery/thermal cost, and transfer recovery semantics. Credit the measured version as arXiv2605.13734 with conference identity verified from the official SIGCOMM2026 program.",
      "limitations_zh": "Service-aware bandwidth/quality control 已提供強 compression baseline。Orbital novelty 需加入 coupled placement、finite contact deadline、battery/thermal cost 與 transfer recovery semantics。Measured version 標示為 arXiv2605.13734，conference identity 依 official SIGCOMM2026 program 核對。",
      "figure_caption_en": "Original Figure 6 links offline strategy profiling, online service-aware selection and the serving data path. For orbital work, retain these compression choices while exposing finite contact capacity and energy state to a joint scheduler.",
      "figure_caption_zh": "原始 Figure 6 串接 offline strategy profiling、online service-aware selection 與 serving data path。Orbital work 可保留這些 compression choice，並讓 joint scheduler 取得 finite contact capacity、energy state。",
      "figure_number": 6,
      "figure_page_1based": 5,
      "figure_bbox_points": [
        315.23275862068965,
        81.84,
        561.2198275862069,
        248.82
      ],
      "figure_sha256": "7b5d9a443cd85135796c1dd7d8ecea948864c38e028f75b9023da28db92ab734",
      "mechanism_scope": "T — Terrestrial service-aware compression reference",
      "deployment_regime": "T",
      "space_ground_path": false,
      "display_regime_en": "T · Terrestrial reference",
      "display_regime_zh": "T · 地面比較基準",
      "primary_pdf_sha256": "febf952c4f72dd36e75ca7d0a7422646c4d4e77929892b3886b8b375f3d68383",
      "figure_provenance": {
        "id": "kvserve",
        "figure_number": 6,
        "figure_page_1based": 5,
        "figure_bbox_points": [
          315.23275862068965,
          81.84,
          561.2198275862069,
          248.82
        ],
        "pdf_sha256": "febf952c4f72dd36e75ca7d0a7422646c4d4e77929892b3886b8b375f3d68383",
        "figure_sha256": "7b5d9a443cd85135796c1dd7d8ecea948864c38e028f75b9023da28db92ab734",
        "bytes": 181038,
        "dimensions": [
          854,
          580
        ],
        "capture_method": "Original primary PDF rasterized at 250 dpi and cropped to figure plus complete source caption",
        "visually_inspected": true,
        "source_pdf_url": "https://arxiv.org/pdf/2605.13734"
      },
      "figure_acquisition_method": "Original primary PDF rasterized at 250 dpi and cropped to figure plus complete source caption",
      "figure_asset": "../figures/kvserve.png",
      "figure_acquisition": "Original primary PDF rasterized at 250 dpi and cropped to figure plus complete source caption",
      "figure_content_type": "image/png",
      "figure_bytes": 181038
    },
    {
      "id": "dualpath",
      "title": "DualPath: Accelerating Agentic LLM Inference by Harvesting Disaggregated KV-Cache Storage I/O",
      "authors": "Yongtong Wu; Shaoyuan Chen; Yinmin Zhong; Rilin Huang; Yixuan Tan; Wentao Zhang; Liyue Zhang; Shangyan Zhou; Yuxuan Liu; Shunfeng Zhou; Mingxing Zhang; Xin Jin; Panpan Huang",
      "year": "2026",
      "venue": "SIGCOMM",
      "url": "https://doi.org/10.1145/3789240.3829159",
      "pdf_url": "https://arxiv.org/pdf/2602.21548",
      "regime": "T · Terrestrial reference",
      "arxiv": "2602.21548",
      "affiliations": "Peking University, Tsinghua University, DeepSeek-AI",
      "evidence_status": "Complete primary full text reviewed",
      "source_version": "arXiv2602.21548v2,26February2026; source title literal: DualPath: Breaking the Storage Bandwidth Bottleneck in Agentic LLM Inference; conference title from official2026 program",
      "evidence": [
        "§4.1–§4.2 PDFp4–6: dual path and bandwidth bounds",
        "§5–§6 PDFp6–9: traffic isolation and scheduling",
        "§7.2–§7.3 PDFp9–12: hardware, workload, fair Basic baseline and throughput",
        "§7.6 PDFp13:48P96D scaling",
        "Figure4 PDFp5"
      ],
      "built_for_en": "Multi-turn agentic inference reloads high-hit-rate KV state from SSD-backed distributed storage into separate prefill and decode workers.",
      "built_for_zh": "Multi-turn agentic inference 從 SSD-backed distributed storage 重新載入 high-hit-rate KV state 至分離的 prefill、decode worker。",
      "problem_en": "Prefill storage NICs saturate while decode storage NICs remain lightly used. Long contexts with short appends make cached-state transfer dominate execution and leave GPU compute capacity idle.",
      "problem_zh": "Prefill storage NIC 飽和，decode storage NIC 使用率較低。Long context 搭配 short append，使 cached-state transfer 主導 execution，並造成 GPU compute capacity 閒置。",
      "design_en": "Dual-path loading combines storage→prefill with storage→decode→prefill over the compute RDMA fabric. CNIC-centric traffic management isolates model-critical collectives; global routing balances storage queues, and a compute-quota scheduler balances attention execution across engines using chunked prefill.",
      "design_zh": "Dual-path loading 結合 storage→prefill 與透過 compute RDMA fabric 的 storage→decode→prefill。CNIC-centric traffic management 隔離 model-critical collective；global routing 平衡 storage queue；compute-quota scheduler 透過 chunked prefill 平衡各 engine 的 attention execution。",
      "model_en": "Bandwidth accounting bounds admissible P/D ratios using GPUs per node, storage-NIC bandwidth, compute-NIC bandwidth and DRAM traffic. The scheduler fits attention execution time from cached/appended tokens and packs FIFO requests under a compute quota. The transport design assigns model inference and KV traffic separate virtual-lane priorities.",
      "model_zh": "Bandwidth accounting 依 GPUs per node、storage-NIC bandwidth、compute-NIC bandwidth、DRAM traffic 推導可行 P/D ratio。Scheduler 以 cached/appended token 擬合 attention execution time，並在 compute quota 下以 FIFO packing 排程 request。Transport design 為 model inference、KV traffic 配置分離 virtual-lane priority。",
      "evaluation_en": "Hopper servers carry eight GPUs, eight 400Gbps compute NICs and one 400Gbps storage NIC; 3FS supplies SSD-backed KV storage. DS660B, an internal DS27B variant and Qwen2.5-32B replay three 500-trajectory agent datasets with 32K/48K/64K context bounds. The scale experiment reaches 48P96D, or 144 servers and 1152 GPUs.",
      "evaluation_zh": "Hopper server 各配置八張 GPU、八張400Gbps compute NIC、一張400Gbps storage NIC；3FS 提供 SSD-backed KV storage。DS660B、internal DS27B variant、Qwen2.5-32B 重播三組各500-trajectory 的 agent dataset，context bound 為32K/48K/64K。Scale experiment 達48P96D，亦即144 server、1152 GPU。",
      "baselines_en": "Primary gains compare DualPath with Basic, the same internal runtime. SGLang with HiCache/Mooncake/3FS is reported separately with configuration differences; Oracle bypasses disk reads and state transfers. Ablations isolate path choice, traffic management and compute scheduling.",
      "baselines_zh": "主要 gain 比較 DualPath 與同一 internal runtime 的 Basic。SGLang 搭配 HiCache/Mooncake/3FS 另列 configuration difference；Oracle 略過 disk read、state transfer。Ablation 分析 path choice、traffic management、compute scheduling。",
      "result_en": "Against Basic, the primary version reports up to 1.87× offline throughput and 1.96× average online agents/s while meeting TTFT/TPOT SLOs. The 1.87× maximum corresponds to DS660B; the large 48P96D experiment establishes scaling behavior under the authors’ internal framework.",
      "result_zh": "相對 Basic，primary version 在滿足 TTFT/TPOT SLO 下呈現最高1.87× offline throughput 與平均1.96× online agents/s。1.87× maximum 對應 DS660B；大型48P96D experiment 支持 authors’ internal framework 下的 scaling behavior。",
      "limitations_en": "The data path relies on a separately provisioned, high-capacity compute fabric with spare bandwidth between collectives. Orbital transfer must debit relay KV bytes against the same ISL capacity used by model traffic and account for expiration, terminal power and shared bottlenecks. Report February arXivv2 measurements separately from the official updated conference title.",
      "limitations_zh": "Data path 依賴獨立配置的 high-capacity compute fabric，以及 collective 之間的 spare bandwidth。Orbital transfer 需將 relay KV byte 計入 model traffic 共用的 ISL capacity，並納入 expiration、terminal power、shared bottleneck。February arXivv2 measurement 與 official updated conference title 分列標示。",
      "figure_caption_en": "Original Figure 4 exposes both KV loading paths, GPU/DRAM staging and compute/storage NICs. An orbital comparison must charge every relay hop against shared laser-link capacity and energy.",
      "figure_caption_zh": "原始 Figure 4 呈現兩種 KV loading path、GPU/DRAM staging 與 compute/storage NIC。Orbital comparison 需將每個 relay hop 計入共用 laser-link capacity、energy。",
      "figure_number": 4,
      "figure_page_1based": 5,
      "figure_bbox_points": [
        68.58620689655173,
        81.84,
        542.7543103448276,
        252.78
      ],
      "figure_sha256": "facc01be654f990e6e180575b3b68541edc66359982514aa741b8a77866a49f7",
      "mechanism_scope": "T — Terrestrial agentic KV I/O reference",
      "deployment_regime": "T",
      "space_ground_path": false,
      "display_regime_en": "T · Terrestrial reference",
      "display_regime_zh": "T · 地面比較基準",
      "primary_pdf_sha256": "6d135c6563a7b6db1c5f2b27de50dee65f1bbdb4507bc6ddf871b88e3e0d6aa2",
      "figure_provenance": {
        "id": "dualpath",
        "figure_number": 4,
        "figure_page_1based": 5,
        "figure_bbox_points": [
          68.58620689655173,
          81.84,
          542.7543103448276,
          252.78
        ],
        "pdf_sha256": "6d135c6563a7b6db1c5f2b27de50dee65f1bbdb4507bc6ddf871b88e3e0d6aa2",
        "figure_sha256": "facc01be654f990e6e180575b3b68541edc66359982514aa741b8a77866a49f7",
        "bytes": 118867,
        "dimensions": [
          1647,
          594
        ],
        "capture_method": "Original primary PDF rasterized at 250 dpi and cropped to figure plus complete source caption",
        "visually_inspected": true,
        "source_pdf_url": "https://arxiv.org/pdf/2602.21548"
      },
      "figure_acquisition_method": "Original primary PDF rasterized at 250 dpi and cropped to figure plus complete source caption",
      "figure_asset": "../figures/dualpath.png",
      "figure_acquisition": "Original primary PDF rasterized at 250 dpi and cropped to figure plus complete source caption",
      "figure_content_type": "image/png",
      "figure_bytes": 118867
    },
    {
      "id": "connex",
      "title": "Connex: Endpoint Mobility Primitives for Dynamic LLM Serving",
      "authors": "Yanying Lin; Vincent Liu; Tao Luo; Chengzhong Xu; Kejiang Ye",
      "year": "2026",
      "venue": "SIGCOMM",
      "url": "https://doi.org/10.1145/3789240.3829200",
      "pdf_url": "https://yanyinglin.github.io/assets/pdf/Connex-Sigcomm26.pdf",
      "regime": "T · Terrestrial reference",
      "arxiv": "",
      "affiliations": "Shenzhen Institutes of Advanced Technology, University of Chinese Academy of Sciences, University of Pennsylvania, University of Macau",
      "evidence_status": "Complete primary full text reviewed",
      "source_version": "Conference primary paper",
      "evidence": [
        "§3–§4 PDFp5–8: architecture, epoch routing, handover and credits",
        "§6.1–§6.2 PDFp8–10:20 A40 GPUs, dataset, source model identifier and goodput definition",
        "§6.3–§6.5 PDFp10–12: churn, reconfiguration and overload",
        "Official ACM eReader14pages independently cross-checks author camera-ready source",
        "Figure4 PDFp5"
      ],
      "built_for_en": "Elastic terrestrial serving changes worker membership while token streams, activations and KV transfers remain in flight.",
      "built_for_zh": "Elastic terrestrial serving 在 token stream、activation、KV transfer 持續傳輸時調整 worker membership。",
      "problem_en": "Fixed communication groups impose collective quiescence and route rebuilding during join, leave and migration. Reordered or duplicated in-flight state and shared queue pressure amplify P99 request latency during churn.",
      "problem_zh": "固定 communication group 在 join、leave、migration 時需要 collective quiescence 與 route rebuilding。In-flight state 的 reorder、duplicate 與共用 queue pressure 會在 churn 期間放大 P99 request latency。",
      "design_en": "A stable logical-endpoint mobility contract combines epoch/lease routing, prepare→cutover→commit handover, sequence-based deduplication, receiver credits and traffic-class isolation. ZeroMQ supplies control coordination; pooled UCX/RDMA connections and capability-aware transport selection carry tensor data.",
      "design_zh": "Stable logical-endpoint mobility contract 結合 epoch/lease routing、prepare→cutover→commit handover、sequence-based deduplication、receiver credit、traffic-class isolation。ZeroMQ 提供 control coordination；pooled UCX/RDMA connection 與 capability-aware transport selection 傳送 tensor data。",
      "model_en": "Explicit protocol state machines track logical IDs, route epochs, leases, acknowledged sequence ranges and receiver-buffer credits. The continuity guarantee assumes recoverable request state and available replacement endpoints; irreversible state loss is escalated to the serving framework.",
      "model_zh": "Explicit protocol state machine 追蹤 logical ID、route epoch、lease、acknowledged sequence range、receiver-buffer credit。Continuity guarantee 採 recoverable request state 與 available replacement endpoint；irreversible state loss 交由 serving framework 處理。",
      "evaluation_en": "Five servers each contain four A40-48GB GPUs and 100Gbps ConnectX-6 NICs behind one full-bisection switch. Splitwise-derived request lengths drive PD and PP workloads; joins/leaves,30–120s preemption intervals, bursty loads and directory tests to 2000 mock endpoints evaluate churn. The source labels its serving model Llama3-13B.",
      "evaluation_zh": "五臺 server 各配置四張 A40-48GB、100Gbps ConnectX-6 NIC，連接單一 full-bisection switch。Splitwise-derived request length 驅動 PD、PP workload；join/leave、30–120s preemption interval、bursty load 與2000個 mock endpoint 的 directory test 評估 churn。Source 將 serving model 標記為 Llama3-13B。",
      "baselines_en": "NCCL, MooncakeTE, NIXL and ZeroMQ; NCCL2.27.5 shrink and destroy/reinit microbenchmarks; compact/mobility header and credit-path microbenchmarks. Source goodput is the fraction of issued requests completing with serving-layer continuity.",
      "baselines_zh": "NCCL、MooncakeTE、NIXL、ZeroMQ；NCCL2.27.5 shrink、destroy/reinit microbenchmark；compact/mobility header、credit-path microbenchmark。Source goodput 定義為保持 serving-layer continuity 並完成的 issued-request fraction。",
      "result_en": "The paper reports up to 85% lower churn-induced P99 spikes and 100% continuity success at moderate load versus 0–28% for baselines, with less than 5% steady-state overhead. Pair-local blocking cutover is about 11ms in a cross-pod microbenchmark; NCCL shrink is about 30ms and destroy/reinit about 554ms in a same-node two-rank benchmark.",
      "result_zh": "Paper 呈現 churn-induced P99 spike 最高降低85%，moderate load 的 continuity success 為100%，baseline 為0–28%；steady-state overhead 低於5%。Cross-pod microbenchmark 的 pair-local blocking cutover 約11ms；same-node two-rank benchmark 的 NCCL shrink 約30ms，destroy/reinit 約554ms。",
      "limitations_en": "Generic endpoint migration, epoch routing and ordered handover already exist as current networking prior art. Orbital residual novelty concerns finite contact deadlines, predicted separation, disconnected control paths, state-size feasibility and energy-constrained placement. Equal physical links, migration opportunities and forecast information make Connex an informative baseline; SLO-qualified throughput is a separate reported metric.",
      "limitations_zh": "Generic endpoint migration、epoch routing、ordered handover 已構成 current networking prior art。Orbital residual novelty 涵蓋 finite contact deadline、predicted separation、disconnected control path、state-size feasibility、energy-constrained placement。相同 physical link、migration opportunity、forecast information 可形成有資訊價值的 Connex baseline；SLO-qualified throughput 為另一項 reported metric。",
      "figure_caption_en": "Original Figure 4 places the mobility contract between applications and transport/physical links. Orbital adaptation can extend the contract with a transfer deadline and contact-budget admission while retaining sequence and credit semantics.",
      "figure_caption_zh": "原始 Figure 4 將 mobility contract 置於 application 與 transport/physical link 之間。Orbital adaptation 可加入 transfer deadline、contact-budget admission，並保留 sequence、credit semantics。",
      "figure_number": 4,
      "figure_page_1based": 5,
      "figure_bbox_points": [
        321.82758620689657,
        84.48,
        553.3060344827586,
        297.66
      ],
      "figure_sha256": "febb5b150f1b35b491b8886131ea24f313613feee19d5f2aa4e7fe9afd8fb4d2",
      "mechanism_scope": "T — Terrestrial LLM mobility reference",
      "deployment_regime": "T",
      "space_ground_path": false,
      "display_regime_en": "T · Terrestrial reference",
      "display_regime_zh": "T · 地面比較基準",
      "primary_pdf_sha256": "7529d30be00b11f49a009f2c0b92d0c580a37585bf4a77cafdf7794189e72dc0",
      "figure_provenance": {
        "id": "connex",
        "figure_number": 4,
        "figure_page_1based": 5,
        "figure_bbox_points": [
          321.82758620689657,
          84.48,
          553.3060344827586,
          297.66
        ],
        "pdf_sha256": "7529d30be00b11f49a009f2c0b92d0c580a37585bf4a77cafdf7794189e72dc0",
        "figure_sha256": "febb5b150f1b35b491b8886131ea24f313613feee19d5f2aa4e7fe9afd8fb4d2",
        "bytes": 115631,
        "dimensions": [
          804,
          741
        ],
        "capture_method": "Original primary PDF rasterized at 250 dpi and cropped to figure plus complete source caption",
        "visually_inspected": true,
        "source_pdf_url": "https://yanyinglin.github.io/assets/pdf/Connex-Sigcomm26.pdf"
      },
      "figure_acquisition_method": "Original primary PDF rasterized at 250 dpi and cropped to figure plus complete source caption",
      "figure_asset": "../figures/connex.png",
      "figure_acquisition": "Original primary PDF rasterized at 250 dpi and cropped to figure plus complete source caption",
      "figure_content_type": "image/png",
      "figure_bytes": 115631
    },
    {
      "id": "janus",
      "title": "Janus: A Unified Distributed Training Framework for Sparse Mixture-of-Experts Models",
      "authors": "Juncai Liu; Jessie Hui Wang; Yimin Jiang",
      "year": "2023",
      "venue": "SIGCOMM",
      "url": "https://doi.org/10.1145/3603269.3604869",
      "pdf_url": "https://ymjiang.github.io/files/sigcomm2023_janus.pdf",
      "regime": "T · Terrestrial reference",
      "arxiv": "",
      "affiliations": "Tsinghua University, Zhongguancun Laboratory, ByteDance",
      "evidence_status": "Complete primary full text reviewed",
      "source_version": "Conference primary paper",
      "evidence": [
        "§3.2–§5.1 PDFp4–6: data-centric paradigm, buffers and Equation1",
        "§5.2–§5.3 PDFp7–8: topology-aware scheduling and prefetch",
        "§7.1–§7.5 PDFp9–11: Tutel baseline,32GPU setup and16GPU residual-MoE result",
        "Figure4 PDFp4"
      ],
      "built_for_en": "Distributed sparse-MoE training can choose an expert-centric token exchange or a data-centric expert-weight fetch for each MoE block.",
      "built_for_zh": "Distributed sparse-MoE training 可逐 MoE block 選擇 expert-centric token exchange 或 data-centric expert-weight fetch。",
      "problem_en": "Expert routing produces expensive AlltoAll token exchanges and synchronization. For long sequences and suitable expert size, fetching weights to resident tokens can transfer fewer bytes and overlap computation.",
      "problem_zh": "Expert routing 產生高成本 AlltoAll token exchange、synchronization。Long sequence 與合適 expert size 下，將 weight 取至 resident token 所在節點可降低 transferred byte，並與 computation 重疊。",
      "design_en": "Janus selects communication paradigm through a traffic-volume threshold. Credit-bounded expert buffers, hierarchical per-machine caching and gradient merging, fine-grained asynchronous fetch, PCIe/NVLink-aware pull ordering and layer-ahead prefetch overlap communication with compute.",
      "design_zh": "Janus 透過 traffic-volume threshold 選擇 communication paradigm。Credit-bounded expert buffer、hierarchical per-machine caching、gradient merging、fine-grained asynchronous fetch、PCIe/NVLink-aware pull ordering、layer-ahead prefetch 讓 communication 與 compute 重疊。",
      "model_en": "Equation 1 derives R=B·S·k/(4·n·H·E) as expert-centric/data-centric traffic ratio: B batch per worker, S sequence length, k top-k, n machines, H hidden dimension and E experts per worker. R>1 favors weight fetching under the model’s balanced token distribution and FFN shape assumptions. Iteration barriers retain synchronized parameter updates.",
      "model_zh": "Equation1 推導 expert-centric/data-centric traffic ratio R=B·S·k/(4·n·H·E)：B 為 batch per worker，S 為 sequence length，k 為 top-k，n 為 machine count，H 為 hidden dimension，E 為 expert per worker。Balanced token distribution 與指定 FFN shape assumption 下，R>1 偏好 weight fetching。Iteration barrier 維持 synchronized parameter update。",
      "evaluation_en": "Four machines contain 32 A100-SXM80GB GPUs,200Gbps NICs and NVLink/NVSwitch within each machine. Experiments train twelve-block MoE-BERT, MoE-GPT and MoE-Transformer-XL with 32 experts per MoE block; batch/sequence sensitivity and a pyramidal-residual MoE variant cover 16/32GPU configurations.",
      "evaluation_zh": "四臺 machine 共配置32張 A100-SXM80GB、200Gbps NIC，machine 內使用 NVLink/NVSwitch。Experiment 訓練十二個 block 的 MoE-BERT、MoE-GPT、MoE-Transformer-XL，每個 MoE block 配32個 expert；batch/sequence sensitivity 與 pyramidal-residual MoE variant 涵蓋16/32GPU configuration。",
      "baselines_en": "Tutel supplies the main optimized expert-centric comparison. Janus expert-centric/data-centric variants and topology-priority/prefetch ablations isolate traffic-paradigm and scheduling contributions.",
      "baselines_zh": "Tutel 提供主要 optimized expert-centric comparison。Janus expert-centric/data-centric variant 與 topology-priority/prefetch ablation 分析 traffic-paradigm、scheduling contribution。",
      "result_en": "The source reports up to 16× lower traffic and up to 2.06× training speedup. Its standard 32GPU configuration achieves 1.28×,1.48× and 1.52× iteration speedup for MoE-BERT/GPT/Transformer-XL versus Tutel; the 2.06× maximum arises in the 16GPU pyramidal-residual MoE experiment.",
      "result_zh": "Source 呈現 traffic 最高降低16×，training speedup 最高2.06×。Standard32GPU configuration 相對 Tutel 的 MoE-BERT/GPT/Transformer-XL iteration speedup 依序為1.28×、1.48×、1.52×；2.06× maximum 來自16GPU pyramidal-residual MoE experiment。",
      "limitations_en": "Moving expert weights toward tokens is established training prior art. Orbital novelty can jointly choose bytes, expert placement and update freshness across expiring contacts, charging repeated weight refresh and gradient return. Extend the same R-based selector with contact feasibility before assessing a new temporal policy.",
      "limitations_zh": "將 expert weight 移向 token 為既有 training prior art。Orbital novelty 可跨 expiring contact 聯合選擇 byte、expert placement、update freshness，並計入 repeated weight refresh、gradient return。評估新 temporal policy 前，可先為相同 R-based selector 加入 contact feasibility。",
      "figure_caption_en": "Original Figure 4 shows per-worker expert buffers and per-machine expert caches. Janus supplies a concrete weights-versus-tokens baseline; orbital versions also track weight freshness and deadline-feasible gradient return.",
      "figure_caption_zh": "原始 Figure 4 呈現 per-worker expert buffer 與 per-machine expert cache。Janus 提供具體 weights-versus-tokens baseline；orbital version 進一步追蹤 weight freshness、deadline-feasible gradient return。",
      "figure_number": 4,
      "figure_page_1based": 4,
      "figure_bbox_points": [
        325.125,
        83.16,
        550.0086206896551,
        216.48
      ],
      "figure_sha256": "1f923c53169be77992a8b17c7df54872f25d15c83174e5363b9f60bc6dfb9bdc",
      "mechanism_scope": "T — Terrestrial MoE weights-versus-tokens reference",
      "deployment_regime": "T",
      "space_ground_path": false,
      "display_regime_en": "T · Terrestrial reference",
      "display_regime_zh": "T · 地面比較基準",
      "primary_pdf_sha256": "71448f3ade69a01bd2a973900404eef20de4caa78e3a3b95766c0ba9247e65c2",
      "figure_provenance": {
        "id": "janus",
        "figure_number": 4,
        "figure_page_1based": 4,
        "figure_bbox_points": [
          325.125,
          83.16,
          550.0086206896551,
          216.48
        ],
        "pdf_sha256": "71448f3ade69a01bd2a973900404eef20de4caa78e3a3b95766c0ba9247e65c2",
        "figure_sha256": "1f923c53169be77992a8b17c7df54872f25d15c83174e5363b9f60bc6dfb9bdc",
        "bytes": 70152,
        "dimensions": [
          781,
          463
        ],
        "capture_method": "Original primary PDF rasterized at 250 dpi and cropped to figure plus complete source caption",
        "visually_inspected": true,
        "source_pdf_url": "https://ymjiang.github.io/files/sigcomm2023_janus.pdf"
      },
      "figure_acquisition_method": "Original primary PDF rasterized at 250 dpi and cropped to figure plus complete source caption",
      "figure_asset": "../figures/janus.png",
      "figure_acquisition": "Original primary PDF rasterized at 250 dpi and cropped to figure plus complete source caption",
      "figure_content_type": "image/png",
      "figure_bytes": 70152
    },
    {
      "id": "ubep",
      "title": "UBEP: Re-architecting Expert Parallelism Communication Library for Production Superpods",
      "authors": "Yipeng Liu; Chang Liu; Si Shen; Jiaqi Zheng; Mingfan Li; Yuyang Yang; Guanhua Li; Yuquan Zhang; Yimeng Xu; Zhongzhe Hu; Zhiyuan Huang; Qihang Duan; Junsong Wang; Wenkai Ling; Baochuan Yang; Xianzhi Yu; Han Bao; Yijie Chen; Guihai Chen",
      "year": "2026",
      "venue": "SIGCOMM",
      "url": "https://doi.org/10.1145/3789240.3829183",
      "pdf_url": "https://arxiv.org/pdf/2607.06202",
      "regime": "T · Terrestrial reference",
      "arxiv": "2607.06202",
      "affiliations": "Nanjing University, Huawei Technologies",
      "evidence_status": "Complete primary full text reviewed",
      "source_version": "Conference-formatted arXiv2607.06202 primary paper",
      "evidence": [
        "§2–§3 PDFp3–8: fabric model, hierarchy and Data-as-Flag",
        "§5.1–§5.5 PDFp8–11:256 NPU dies, exact baselines, operator/TPOT metrics",
        "§6 PDFp11–12: portability, atomic-memory assumptions",
        "Figure4 PDFp5"
      ],
      "built_for_en": "MoE inference dispatch on hierarchical superpod fabrics exposes globally addressable memory and low-latency remote access.",
      "built_for_zh": "Hierarchical superpod fabric 上的 MoE inference dispatch 使用 globally addressable memory、low-latency remote access。",
      "problem_en": "Bulk-synchronous dispatch kernels serialize metadata, payload and reordering phases; global barriers and distance-dependent access create idle accelerator cores and uneven expert traffic.",
      "problem_zh": "Bulk-synchronous dispatch kernel 將 metadata、payload、reordering phase 串行化；global barrier、distance-dependent access 造成 accelerator core 閒置與 expert traffic 失衡。",
      "design_en": "Kernel decomposition splits communication dependencies across accelerator vector cores. Hierarchical token scheduling balances hop-distance classes; Data-as-Flag combines payload readiness with atomic writes through Token-Flag Fusion, Data Checksum or Sentinel Polling.",
      "design_zh": "Kernel decomposition 將 communication dependency 分配至 accelerator vector core。Hierarchical token scheduling 平衡 hop-distance class；Data-as-Flag 透過 Token-Flag Fusion、Data Checksum、Sentinel Polling，結合 payload readiness 與 atomic write。",
      "model_en": "A distance hierarchy distinguishes intra-NPU, one-hop and two-hop access. Scheduling maps logical token work to physical vector cores using prefix-sum/virtual-matrix remapping, while dependency analysis overlaps token sending and address calculation. The primary fabric assumption includes 512B atomic remote writes and a unified global address space.",
      "model_zh": "Distance hierarchy 區分 intra-NPU、one-hop、two-hop access。Scheduling 透過 prefix-sum/virtual-matrix remapping 將 logical token work 配置至 physical vector core，dependency analysis 重疊 token sending、address calculation。主要 fabric assumption 包含512B atomic remote write、unified global address space。",
      "evaluation_en": "A production CM384 allocation contains 16 Ascend servers and 256 NPU dies. Experiments sweep 16–256 ranks, batch size, expert count and top-k; end-to-end serving covers Qwen3-30B, GLM-4.7, DeepSeek-R1 and DeepSeek-V3.2, spanning 30B–685B parameters. Operator latency and inference TPOT are separate metrics.",
      "evaluation_zh": "Production CM384 allocation 包含16臺 Ascend server、256個 NPU die。Experiment 掃描16–256 rank、batch size、expert count、top-k；end-to-end serving 涵蓋 Qwen3-30B、GLM-4.7、DeepSeek-R1、DeepSeek-V3.2，parameter scale 為30B–685B。Operator latency 與 inference TPOT 為分立 metric。",
      "baselines_en": "CANN EP on the same CM384 is the primary performance baseline. An unmapped UBEP variant and Stop-and-Wait/TFF/DC/SP synchronization variants support ablation. DeepEP on H800 is a protocol-capability reference with a separate hardware scope.",
      "baselines_zh": "相同 CM384 上的 CANN EP 為主要 performance baseline。UBEP mapping ablation 與 Stop-and-Wait/TFF/DC/SP synchronization variant 提供分項評估。H800 上的 DeepEP 為具有獨立 hardware scope 的 protocol-capability reference。",
      "result_en": "On identical CM384 hardware, dispatch operator latency decreases by up to 52.4% and inference TPOT by up to 11.1% versus CANN EP. Table 4 reports 35.3–40.8% effective-bandwidth improvement across 16–256 ranks; these measurements establish communication-library benefits under the superpod fabric.",
      "result_zh": "相同 CM384 hardware 下，相對 CANN EP 的 dispatch operator latency 最高降低52.4%，inference TPOT 最高降低11.1%。Table4 在16–256 rank 呈現35.3–40.8% effective-bandwidth improvement；measurement 支持 superpod fabric 下的 communication-library benefit。",
      "limitations_en": "Fine-grained MoE communication and topology-aware dispatch already supply a current strong baseline. Orbital protocols require explicit reliability, memory-ordering, acknowledgment and epoch semantics over optical packet links; the atomic-global-memory optimization transfers under a matching substrate. Residual novelty concerns expiring contacts and resource-coupled dispatch schedules.",
      "limitations_zh": "Fine-grained MoE communication、topology-aware dispatch 已提供 current strong baseline。Orbital protocol 需在 optical packet link 上明確處理 reliability、memory-ordering、acknowledgment、epoch semantics；atomic-global-memory optimization 適用於對應 substrate。Residual novelty 涵蓋 expiring contact、resource-coupled dispatch schedule。",
      "figure_caption_en": "Original Figure 4 joins kernel decomposition, hierarchical token scheduling and Data-as-Flag. Its measured target is MoE inference communication; an orbital implementation must specify the message protocol supporting these dependency semantics.",
      "figure_caption_zh": "原始 Figure 4 結合 kernel decomposition、hierarchical token scheduling、Data-as-Flag。Measured target 為 MoE inference communication；orbital implementation 需指定支持這些 dependency semantics 的 message protocol。",
      "figure_number": 4,
      "figure_page_1based": 5,
      "figure_bbox_points": [
        52.099137931034484,
        81.18,
        293.4698275862069,
        240.9
      ],
      "figure_sha256": "7e62e6806777f54247a5a708cc2adf03e527130f6a2549da233dadf5f9852116",
      "mechanism_scope": "T — Terrestrial MoE inference communication reference",
      "deployment_regime": "T",
      "space_ground_path": false,
      "display_regime_en": "T · Terrestrial reference",
      "display_regime_zh": "T · 地面比較基準",
      "primary_pdf_sha256": "7eac6b34151c8546f9d55d2e3b9abdfc49cab15cc283258d17e4fe1ad2b72abe",
      "figure_provenance": {
        "id": "ubep",
        "figure_number": 4,
        "figure_page_1based": 5,
        "figure_bbox_points": [
          52.099137931034484,
          81.18,
          293.4698275862069,
          240.9
        ],
        "pdf_sha256": "7eac6b34151c8546f9d55d2e3b9abdfc49cab15cc283258d17e4fe1ad2b72abe",
        "figure_sha256": "7e62e6806777f54247a5a708cc2adf03e527130f6a2549da233dadf5f9852116",
        "bytes": 84172,
        "dimensions": [
          838,
          554
        ],
        "capture_method": "Original primary PDF rasterized at 250 dpi and cropped to figure plus complete source caption",
        "visually_inspected": true,
        "source_pdf_url": "https://arxiv.org/pdf/2607.06202"
      },
      "figure_acquisition_method": "Original primary PDF rasterized at 250 dpi and cropped to figure plus complete source caption",
      "figure_asset": "../figures/ubep.png",
      "figure_acquisition": "Original primary PDF rasterized at 250 dpi and cropped to figure plus complete source caption",
      "figure_content_type": "image/png",
      "figure_bytes": 84172
    },
    {
      "id": "hyna",
      "title": "HyNA: Taming Tail Latency in MoE Training with Hybrid Switch Silicon",
      "authors": "Yang Liu; Tianxiang Liu; Haipeng Yao",
      "year": "2026",
      "venue": "SIGCOMM",
      "url": "https://doi.org/10.1145/3789240.3829178",
      "pdf_url": "https://dl.acm.org/doi/pdf/10.1145/3789240.3829178",
      "regime": "T · Terrestrial reference",
      "arxiv": "",
      "affiliations": "Beijing University of Posts and Telecommunications",
      "evidence_status": "Complete primary full text reviewed",
      "source_version": "Official ACM eReader full15-page conference paper",
      "evidence": [
        "§1.2 PDFp2: gradient synchronization scope",
        "§3–§5 PDFp4–8: hybrid silicon, protocol, FPGA and queueing",
        "§6.1–§6.3 PDFp9–11: two-generator closed-loop replay, baselines and phase results",
        "§6.4 PDFp11: convergence experiment",
        "§5.2 PDFp7:7nm synthesis area/power Table2",
        "Figure3 PDFp2"
      ],
      "built_for_en": "Terrestrial training gradient AllReduce uses in-switch accumulation with a fast integer pipeline and embedded exception cores.",
      "built_for_zh": "地面 training gradient AllReduce 使用 in-switch accumulation，由 fast integer pipeline 與 embedded exception core 處理。",
      "problem_en": "Host parameter-server incast and off-chip fallback for hash collisions amplify synchronization tails under sparse, bursty gradients. A single delayed gradient block extends the BSP barrier.",
      "problem_zh": "Host parameter-server incast 與 hash collision 的 off-chip fallback 在 sparse、bursty gradient 下放大 synchronization tail。單一 delayed gradient block 延長 BSP barrier。",
      "design_en": "HyNA couples an RMT line-rate INT32 aggregation path with on-chip RISC-V exception handling for collisions and FP32 overflow recovery. A precision-aware protocol dynamically quantizes gradient blocks, tracks aggregation slots and handles recovery through an on-chip exception loop.",
      "design_zh": "HyNA 結合 RMT line-rate INT32 aggregation path 與 on-chip RISC-V exception handling，處理 collision、FP32 overflow recovery。Precision-aware protocol 動態 quantize gradient block、追蹤 aggregation slot，並透過 on-chip exception loop 處理 recovery。",
      "model_en": "BSP synchronization time follows the slowest completed gradient block. Slot collisions and exception arrival bursts drive a queueing/throughput model; Little’s Law dimensions exception queues for 200µs sustained bursts. Cycle-level analysis and 7nm synthesis estimate switch bus, core, area and power costs.",
      "model_zh": "BSP synchronization time 由最慢完成的 gradient block 決定。Slot collision、exception arrival burst 驅動 queueing/throughput model；Little’s Law 為200µs sustained burst 配置 exception queue。Cycle-level analysis、7nm synthesis 估計 switch bus、core、area、power cost。",
      "evaluation_en": "An AlveoU280 FPGA at 250MHz implements 100Gbps aggregation. Two physical servers with ConnectX-5 NICs replay closed-loop gradient traces as 256 logical workers, preserving BSP acknowledgment barriers. Traces include Llama3-70B and DeepSeek-V2-236B; separate Llama3-8B/Qwen7B training on a C4 slice checks convergence over 5000 iterations.",
      "evaluation_zh": "AlveoU280 FPGA 以250MHz 實作100Gbps aggregation。兩臺配置 ConnectX-5 NIC 的 physical server，以256個 logical worker 重播 closed-loop gradient trace，保留 BSP acknowledgment barrier。Trace 包含 Llama3-70B、DeepSeek-V2-236B；另以 C4 slice 訓練 Llama3-8B/Qwen7B，透過5000 iteration 檢查 convergence。",
      "baselines_en": "BytePS measures host-mediated aggregation throughput; SwitchML measures static INA; ATP is emulated by redirecting collision packets to a fallback-server port. NCCL2.18 FP32 supplies the numerical-fidelity reference. Wire speed supplies a physical upper bound.",
      "baselines_zh": "BytePS 評估 host-mediated aggregation throughput；SwitchML 評估 static INA；ATP 透過將 collision packet redirect 至 fallback-server port 模擬。NCCL2.18 FP32 提供 numerical-fidelity reference；wire speed 提供 physical upper bound。",
      "result_en": "Measured goodput is 84.5Gbps versus 11.5Gbps BytePS and 60Gbps ATP:7.35× and 1.4×, respectively. DeepSeek-V2 gradient synchronization improves 1.6× over ATP and 2.3× over BytePS. Estimated silicon area overhead is 2.9%; the∼14% full-training benefit is a phase-fraction estimate.",
      "result_zh": "Measured goodput 為84.5Gbps，BytePS 為11.5Gbps、ATP 為60Gbps，依序對應7.35×、1.4×。DeepSeek-V2 gradient synchronization 相對 ATP 加速1.6×、相對 BytePS 加速2.3×。Estimated silicon area overhead 為2.9%；約14% full-training benefit 為依 phase fraction 推算的 estimate。",
      "limitations_en": "The explicitly evaluated MoE phase is gradient AllReduce. Expert-routing AlltoAll defines a parallel workload family. The 256-worker scale is hardware-in-the-loop emulation; FPGA datapath and ASIC synthesis supply separate evidence. Orbital INA research must cost onboard aggregator state, radiation recovery, energy and contact-aware partial-gradient completion.",
      "limitations_zh": "Explicitly evaluated MoE phase 為 gradient AllReduce。Expert-routing AlltoAll 屬另一 workload family。256-worker scale 為 hardware-in-the-loop emulation；FPGA datapath、ASIC synthesis 提供各自 evidence。Orbital INA research 需計入 onboard aggregator state、radiation recovery、energy、contact-aware partial-gradient completion。",
      "figure_caption_en": "Original Figure 3 keeps integer accumulation and exceptional FP32/collision handling inside the switch. The figure explains on-chip closure; orbital aggregation additionally requires a contact- and failure-aware lifetime for partial gradient state.",
      "figure_caption_zh": "原始 Figure 3 將 integer accumulation 與 exceptional FP32/collision handling 留在 switch 內。Figure 說明 on-chip closure；orbital aggregation 進一步需要依 contact、failure 定義 partial gradient state lifetime。",
      "figure_number": 3,
      "figure_page_1based": 2,
      "figure_bbox_points": [
        325.2,
        72,
        550.2,
        183.6
      ],
      "figure_sha256": "62f610092f66710edcbd463bff7f47fe1096988eee0355887d5b57f4653af5f2",
      "mechanism_scope": "T — Terrestrial gradient aggregation reference",
      "deployment_regime": "T",
      "space_ground_path": false,
      "display_regime_en": "T · Terrestrial reference",
      "display_regime_zh": "T · 地面比較基準",
      "figure_provenance": {
        "id": "hyna",
        "source_pdf_url": "https://dl.acm.org/doi/pdf/10.1145/3789240.3829178",
        "source_reader_url": "https://dl.acm.org/doi/epdf/10.1145/3789240.3829178",
        "figure_number": 3,
        "figure_page_1based": 2,
        "figure_bbox_points": [
          325.2,
          72,
          550.2,
          183.6
        ],
        "capture_method": "Original ACM eReader full-viewport screenshot followed by image crop",
        "screenshot_bbox_pixels": [
          980,
          189,
          1355,
          375
        ],
        "figure_sha256": "62f610092f66710edcbd463bff7f47fe1096988eee0355887d5b57f4653af5f2",
        "reader_text_sha256": "e922adadf355dcd0730f877af46065b3936035c054840edb6aa6114d0e53d1ab",
        "bytes": 61912,
        "visually_inspected": true
      },
      "figure_acquisition_method": "Original ACM eReader full-viewport screenshot followed by image crop",
      "primary_text_sha256": "e922adadf355dcd0730f877af46065b3936035c054840edb6aa6114d0e53d1ab",
      "figure_asset": "../figures/hyna.png",
      "figure_acquisition": "Original ACM eReader full-viewport screenshot followed by image crop",
      "figure_content_type": "image/png",
      "figure_bytes": 61912
    },
    {
      "id": "uav5g",
      "title": "Unveiling Low-Altitude 5G Performance: Linking Key Influencing Factors with UAV Flight Parameters",
      "authors": "Xinzhe Liu; Jianer Zhou; Xiaoyong Ni; Ke Luo; Zhenyu Li; Xiaofeng Tao; Weichao Li",
      "year": "2026",
      "venue": "SIGCOMM",
      "url": "https://doi.org/10.1145/3789240.3829198",
      "pdf_url": "https://dl.acm.org/doi/pdf/10.1145/3789240.3829198",
      "regime": "X · Aerial adjacency",
      "arxiv": "",
      "affiliations": "Pengcheng Laboratory, South China University of Technology, Institute of Computing Technology, Beijing University of Posts and Telecommunications",
      "evidence_status": "Complete primary full text reviewed",
      "source_version": "Official ACM eReader full17-page conference paper",
      "evidence": [
        "§3 PDFp3: matched four-operator platform and300GB dataset",
        "§4 PDFp4–6: SL/MO/RB/handover effects and pseudo-successful handover",
        "§5 PDFp7–11: controlled altitude, speed, environment and band analysis",
        "§6–§7 PDFp12: simulator implications and source scope",
        "AppendixD PDFp14–17: separate5G-A CA measurements",
        "Figure1 PDFp3"
      ],
      "built_for_en": "Aerial user equipment accesses commercial terrestrial sub-6GHz5G cells during controlled low-altitude UAV flights.",
      "built_for_zh": "Aerial user equipment 在受控 low-altitude UAV flight 中接取 commercial terrestrial sub-6GHz5G cell。",
      "problem_en": "Three-dimensional flight changes signal propagation and visible-cell sets. MIMO configuration, nominal handover success and aggregate throughput can hide the cross-layer causes of degraded application connectivity.",
      "problem_zh": "Three-dimensional flight 改變 signal propagation、visible-cell set。MIMO configuration、nominal handover success、aggregate throughput 可能遮蔽 application connectivity 下降的 cross-layer cause。",
      "design_en": "A DJI Matrice400 carries four matched smartphones and an XCAL mini-PC; iPerf3 cloud backends measure application performance alongside PHY/MAC/RRC indicators. Controlled routes vary altitude, speed, environment, time and band while holding other parameters fixed.",
      "design_zh": "DJI Matrice400 搭載四支配對 smartphone 與 XCAL mini-PC；iPerf3 cloud backend 同步量測 application performance、PHY/MAC/RRC indicator。Controlled route 逐項調整 altitude、speed、environment、time、band，並固定其他 parameter。",
      "model_en": "A cross-layer causal explanation maps flight parameters through path loss, multipath and cell visibility into spatial layers, modulation order, allocated resource blocks and handover events. The work is an empirical measurement/attribution study supported by controlled interventions and conditional comparisons.",
      "model_zh": "Cross-layer causal explanation 將 flight parameter 經 path loss、multipath、cell visibility 對應至 spatial layer、modulation order、allocated resource block、handover event。研究以 controlled intervention、conditional comparison 支持 empirical measurement、attribution。",
      "evaluation_en": "Shenzhen urban/rural campaigns span four operators and approximately 300GB of time-aligned traces. Main flights use altitudes below 400m and speeds up to 15m/s, single-carrier 5GSA on n41/n78/n79. An appendix measures a separate 5G-A carrier-aggregation hotspot at 120–500m. Four 500Mbps cloud backends and paired-device checks control measurement artifacts.",
      "evaluation_zh": "Shenzhen urban/rural campaign 涵蓋四個 operator、約300GB time-aligned trace。主要 flight 使用400m 以下 altitude、最高15m/s speed，以及 n41/n78/n79 single-carrier5GSA。Appendix 另量測120–500m 的5G-A carrier-aggregation hotspot。四個500Mbps cloud backend、paired-device check 控制 measurement artifact。",
      "baselines_en": "Matched ground runs, fixed-route altitude/speed controls, operator/band/environment comparisons, redundant GalaxyS25 checks and carrier-aggregation versus single-carrier measurements form the empirical baselines.",
      "baselines_zh": "Matched ground run、fixed-route altitude/speed control、operator/band/environment comparison、備援 GalaxyS25 check、carrier-aggregation 與 single-carrier measurement 構成 empirical baseline。",
      "result_en": "Across the measured operators, aerial runs show lower uplink/downlink throughput and higher RTT than ground runs. Actual spatial-layer count explains capacity better than configured MIMO scale. Pseudo-successful handovers exhibit completed signaling followed by uplink stalls; increasing altitude strengthens LoS dominance and expands competing visible cells.",
      "result_zh": "Measured operator 的 aerial run 呈現較 ground run 低的 uplink/downlink throughput、較高 RTT。Actual spatial-layer count 較 configured MIMO scale 更能解釋 capacity。Pseudo-successful handover 呈現 signaling 完成後的 uplink stall；altitude 增加使 LoS dominance 增強，並擴大 competing visible cell set。",
      "limitations_en": "Its positive scope is airborne clients of terrestrial cellular base stations. A satellite synthesis can carry over aligned PHY/RRC/application tracing and controlled mobility attribution; orbital propagation distance, kilometer-per-second motion, space RF beam scheduling and vacuum optical terminals require their own models. Count this paper in the aerial-adjacency ledger.",
      "limitations_zh": "其適用範圍為接取 terrestrial cellular base station 的 airborne client。Satellite synthesis 可沿用 aligned PHY/RRC/application tracing、controlled mobility attribution；orbital propagation distance、kilometer-per-second motion、space RF beam scheduling、vacuum optical terminal 各需對應 model。此 paper 列於 aerial-adjacency ledger。",
      "figure_caption_en": "Original Figure 1 connects the UAV measurement payload, terrestrial base stations and cloud probes. It visualizes the scope and reusable cross-layer measurement method for mobility studies.",
      "figure_caption_zh": "原始 Figure 1 串接 UAV measurement payload、terrestrial base station、cloud probe，圖像化呈現研究範圍與 mobility study 可沿用的 cross-layer measurement method。",
      "figure_number": 1,
      "figure_page_1based": 3,
      "figure_bbox_points": [
        48,
        78,
        298.2,
        238.2
      ],
      "figure_sha256": "89b1de79a17e7e101ee678eb6f3dcf15fe73c5b33b2859de7d3c849fb96eac6e",
      "mechanism_scope": "X — Aerial RF measurement adjacency",
      "deployment_regime": "X",
      "space_ground_path": false,
      "display_regime_en": "X · Aerial adjacency",
      "display_regime_zh": "X · 空中相鄰系統",
      "figure_provenance": {
        "id": "uav5g",
        "source_pdf_url": "https://dl.acm.org/doi/pdf/10.1145/3789240.3829198",
        "source_reader_url": "https://dl.acm.org/doi/epdf/10.1145/3789240.3829198",
        "figure_number": 1,
        "figure_page_1based": 3,
        "figure_bbox_points": [
          48,
          78,
          298.2,
          238.2
        ],
        "capture_method": "Original ACM eReader full-viewport screenshot followed by image crop",
        "screenshot_bbox_pixels": [
          518,
          198,
          935,
          466
        ],
        "figure_sha256": "89b1de79a17e7e101ee678eb6f3dcf15fe73c5b33b2859de7d3c849fb96eac6e",
        "reader_text_sha256": "548ba54eb00456a108dfd52d1b6876ee854dd0b8a6c20f3bc6461738f0c50c6f",
        "bytes": 170684,
        "visually_inspected": true
      },
      "figure_acquisition_method": "Original ACM eReader full-viewport screenshot followed by image crop",
      "primary_text_sha256": "548ba54eb00456a108dfd52d1b6876ee854dd0b8a6c20f3bc6461738f0c50c6f",
      "figure_asset": "../figures/uav5g.png",
      "figure_acquisition": "Original ACM eReader full-viewport screenshot followed by image crop",
      "figure_content_type": "image/png",
      "figure_bytes": 170684
    },
    {
      "id": "cyclops",
      "title": "Cyclops: An FSO-based Wireless Link for VR Headsets",
      "authors": "Himanshu Gupta; Max Curran; Jon Longtin; Torin Rockwell; Kai Zheng; Mallesham Dasari",
      "year": "2022",
      "venue": "SIGCOMM",
      "url": "https://doi.org/10.1145/3544216.3544255",
      "pdf_url": "https://www3.cs.stonybrook.edu/~mdasari/papers/sigcomm-2022-paper.pdf",
      "regime": "T · Terrestrial reference",
      "arxiv": "",
      "affiliations": "Stony Brook University",
      "evidence_status": "Complete primary full text reviewed",
      "source_version": "Conference primary paper",
      "evidence": [
        "§3–§4 PDFp3–7: joint TX/RX pointing and geometric calibration",
        "§5.1–§5.3 PDFp8–11:1.5–2m bench tests and9.4/23.5Gbps throughput",
        "§5.4 PDFp11–12:500 user traces and98.6% connectivity simulation",
        "Figure5 PDFp4",
        "Official SIGCOMM2022 DOI proceedings pages601–614"
      ],
      "built_for_en": "A short-range indoor FSO link supplies high-bandwidth data to a moving VR headset with translation and rotation.",
      "built_for_zh": "Short-range indoor FSO link 向具 translation、rotation 的 moving VR headset 提供 high-bandwidth data。",
      "problem_en": "Narrow-beam optical coupling must simultaneously align transmit and receive mirrors as a headset moves. Receiver angle and position jointly determine optical loss, so endpoint movement demands fast calibrated pointing.",
      "problem_zh": "Headset 移動時，narrow-beam optical coupling 需同時對準 transmit、receive mirror。Receiver angle、position 聯合決定 optical loss，endpoint movement 因此需要 fast calibrated pointing。",
      "design_en": "The tracking system supplies headset pose to a learned pointing function that outputs four steering voltages for TX/RX galvo mirrors. Two-stage geometric calibration learns each mirror assembly and their relative coordinate mapping; commodity SFP optics implement 10G and 25G links.",
      "design_zh": "Tracking system 提供 headset pose 至 learned pointing function，輸出 TX/RX galvo mirror 的四組 steering voltage。Two-stage geometric calibration 學習各 mirror assembly 與相對 coordinate mapping；commodity SFP optics 實作10G、25G link。",
      "model_en": "A calibrated geometric ray model maps mirror voltages into beam origin/direction, then solves the joint pointing inverse from six-dimensional headset pose. Lateral/angular tolerance and measured alignment delay drive a1ms-slot user-trace connectivity simulation.",
      "model_zh": "Calibrated geometric ray model 將 mirror voltage 對應至 beam origin/direction，再依 six-dimensional headset pose 解 joint pointing inverse。Lateral/angular tolerance、measured alignment delay 驅動1ms-slot user-trace connectivity simulation。",
      "evaluation_en": "Bench prototypes span 1.5–2m, using a linear rail, rotation stage and hand-held mixed movement.10G and 25G iPerf tests measure throughput and received optical power. Trace replay uses 500 one-minute head-motion traces from 50 viewers,10ms pose reports and measured 1–2ms steering latency.",
      "evaluation_zh": "Bench prototype 距離為1.5–2m，使用 linear rail、rotation stage、hand-held mixed movement。10G、25G iPerf test 量測 throughput、received optical power。Trace replay 使用50位 viewer 的500條一分鐘 head-motion trace、10ms pose report，以及 measured1–2ms steering latency。",
      "baselines_en": "Link-design alternatives compare collimated and diverging beams and 10G/25G optics. Pure linear, pure angular and mixed-motion runs identify tolerance limits; calibrated model prediction is checked against measured target spots. The trace study uses measured prototype limits.",
      "baselines_zh": "Link-design alternative 比較 collimated/diverging beam、10G/25G optics。Pure linear、pure angular、mixed-motion run 判讀 tolerance limit；calibrated model prediction 與 measured target spot 比較。Trace study 使用 measured prototype limit。",
      "result_en": "Measured 10G throughput reaches 9.4Gbps within the tested movement bounds;25G reaches 23.5Gbps. The 25G trace-driven simulation reports 98.6% connected 1ms slots across 500 traces, with per-trace connectivity 95–99.98%. These are short-range terminal and trace-model results.",
      "result_zh": "Measured10G throughput 在 tested movement bound 內達9.4Gbps；25G 達23.5Gbps。25G trace-driven simulation 在500條 trace 的 connected1ms slot 比例為98.6%，各 trace connectivity 為95–99.98%。數字對應 short-range terminal、trace-model result。",
      "limitations_en": "The positive scope is a terrestrial indoor optical terminal mechanism. Orbital reuse concerns pose-to-pointing calibration, tolerance/error decomposition and reacquisition accounting; longer range, spacecraft attitude jitter, radiation, solar background and thermal/mechanical qualification add physical models. Count this paper in the terminal-adjacency ledger.",
      "limitations_zh": "其適用範圍為 terrestrial indoor optical terminal mechanism。Orbital reuse 涵蓋 pose-to-pointing calibration、tolerance/error decomposition、reacquisition accounting；longer range、spacecraft attitude jitter、radiation、solar background、thermal/mechanical qualification 需增加 physical model。此 paper 列於 terminal-adjacency ledger。",
      "figure_caption_en": "Original Figure 5 shows tracking reports feeding a learned joint TX/RX mirror controller. It provides a terminal-calibration reference for optical mobility; orbital designs separately define attitude, range and acquisition envelopes.",
      "figure_caption_zh": "原始 Figure 5 呈現 tracking report 輸入 learned joint TX/RX mirror controller，提供 optical mobility 的 terminal-calibration reference；orbital design 另行定義 attitude、range、acquisition envelope。",
      "figure_number": 5,
      "figure_page_1based": 4,
      "figure_bbox_points": [
        51.439655172413794,
        84.48,
        282.2586206896552,
        344.52
      ],
      "figure_sha256": "7670b74eb4e4284d88f9016e1d48629d839d2ce7a49f7a2a59eb849fa11df8d9",
      "mechanism_scope": "X — Terrestrial optical terminal adjacency",
      "deployment_regime": "T",
      "space_ground_path": false,
      "display_regime_en": "T · Optical-terminal adjacency",
      "display_regime_zh": "T · 光學終端相鄰系統",
      "primary_pdf_sha256": "301f03c73666c1d3bde97e46f73646d7c492eacff2075603556e668247bab27b",
      "figure_provenance": {
        "id": "cyclops",
        "figure_number": 5,
        "figure_page_1based": 4,
        "figure_bbox_points": [
          51.439655172413794,
          84.48,
          282.2586206896552,
          344.52
        ],
        "pdf_sha256": "301f03c73666c1d3bde97e46f73646d7c492eacff2075603556e668247bab27b",
        "figure_sha256": "7670b74eb4e4284d88f9016e1d48629d839d2ce7a49f7a2a59eb849fa11df8d9",
        "bytes": 145273,
        "dimensions": [
          801,
          903
        ],
        "capture_method": "Original primary PDF rasterized at 250 dpi and cropped to figure plus complete source caption",
        "visually_inspected": true,
        "source_pdf_url": "https://www3.cs.stonybrook.edu/~mdasari/papers/sigcomm-2022-paper.pdf"
      },
      "figure_acquisition_method": "Original primary PDF rasterized at 250 dpi and cropped to figure plus complete source caption",
      "figure_asset": "../figures/cyclops.png",
      "figure_acquisition": "Original primary PDF rasterized at 250 dpi and cropped to figure plus complete source caption",
      "figure_content_type": "image/png",
      "figure_bytes": 145273
    },
    {
      "id": "harvest",
      "title": "Harvest: Adaptive Photonic Switching Schedules for Collective Communication in Scale-up Domains",
      "authors": "Mahir Rahman, Samuel Joseph, Nihar Kodkani, Behnaz Arzani, Vamsi Addanki",
      "year": "2026",
      "venue": "SIGCOMM",
      "url": "https://doi.org/10.1145/3789240.3829166",
      "pdf_url": "https://stygianet.cs.purdue.edu/papers/harvest-sigcomm26.pdf",
      "regime": "T · Terrestrial reference",
      "transfer_category": "terrestrial reference",
      "source_tier": "full primary PDF",
      "evidence_status": "Full primary PDF: formulation, system design, evaluation, and scope audited",
      "source_program_url": "https://conferences.sigcomm.org/sigcomm/2026/program/papers/",
      "source_acquired_date": "2026-10-02",
      "publication_type": "SIGCOMM main-track terrestrial baseline",
      "evidence": [
        "§3.2–3.4 PDF pp3–6; Equations (1)–(4)",
        "§4 PDF pp6–8; Equations (5)–(8), Algorithm 1, Theorem 1",
        "§5 PDF p8; Recursive Doubling structure",
        "§6 PDF pp9–12; Figures 5–11",
        "Appendix A PDF p16; Equations (9)–(14)",
        "Appendix C.4 PDF p22; application and tenancy scope"
      ],
      "built_for_en": "Photonic scale-up domains with typically 8–64 GPUs, predetermined step-wise collective traffic, bounded per-GPU optical port degree, and GPU multi-hop forwarding. A synchronized controller can choose each topology and the reconfiguration instants.",
      "built_for_zh": "面向典型 8–64 GPU photonic scale-up domain；collective traffic 依 step 預先給定，GPU optical port degree 有界，支援 GPU 多跳轉送。同步 controller 可選 topology 與 reconfiguration 時刻。",
      "problem_en": "Given an existing collective algorithm, its fixed step matrices M_i and data volumes m_i, link bandwidth, port degree and reconfiguration latency α_r, synthesize a topology per step minimizing collective completion time. A topology change trades a setup penalty against propagation distance and shared-link congestion.",
      "problem_zh": "輸入既有 collective algorithm、固定 step matrix M_i／data volume m_i、link bandwidth、port degree 與 reconfiguration latency α_r，求逐 step topology，使 collective completion time 最小。Topology change 以 setup penalty 換取較短 propagation distance 與較低 shared-link congestion。",
      "design_en": "Partition collective steps into contiguous intervals. Each interval receives one optimized topology from a degree-constrained maximum-concurrent-flow MISOCP; a dynamic program chooses interval boundaries and then the reconfiguration count. Recursive Doubling connectivity and interval-topology structure yield an analytical subproblem with polylogarithmic schedule synthesis and cached schedules.",
      "design_zh": "將 collective step 分割成連續 interval。每 interval 以 degree-constrained maximum-concurrent-flow MISOCP 求單一 topology；dynamic program 決定 interval boundary，再選 reconfiguration 次數。Recursive Doubling 的 connectivity 與 interval-topology 結構提供解析 subproblem，可用 polylogarithmic synthesis 並快取 schedule。",
      "model_en": "Equation (4) adds per-step launch cost α, δ times path distance, β m_i divided by maximum concurrent flow θ, and α_r for topology changes. Equations (5)–(8) define the interval topology optimization and DP recurrence. Directed integer edge multiplicities obey incoming/outgoing degree limits; commodity conservation and capacities constrain routing. The guarantee is optimal topology scheduling for the supplied fixed collective and model.",
      "model_zh": "Equation (4) 加總逐 step launch cost α、δ×path distance、β m_i／maximum concurrent flow θ，以及 topology change 的 α_r。Equations (5)–(8) 定義 interval topology optimization 與 DP recurrence。Directed integer edge multiplicity 受 incoming／outgoing degree 約束；commodity conservation 與 capacity 約束 routing。保證適用輸入的固定 collective 與給定 model 之 topology scheduling。",
      "evaluation_en": "ASTRA-sim packet-level extensions, flow/numerical optimization, and an eight-GPU emulation testbed cover 8–64-GPU scale-up domains; analytical Recursive Doubling runtime scales to 1024 modeled nodes. Simulated ports use 800 Gb/s; the eight BlueField-3 NICs use 100-Gb/s optics and GPUDirect RDMA. NCCL operations execute step by step; measured runtimes are summed with a supplied fixed switch penalty. Reconfiguration latency spans 10 ns–10 ms.",
      "evaluation_zh": "使用 ASTRA-sim packet-level extension、flow／numerical optimization 與 8-GPU emulation testbed，涵蓋 8–64-GPU scale-up domain；解析 Recursive Doubling runtime 延伸到 1024 modeled node。Simulation port 為 800 Gb/s；8 張 BlueField-3 NIC 使用 100-Gb/s optics 與 GPUDirect RDMA。NCCL 按 step 執行，量測 runtime 加上指定的固定 switch penalty。Reconfiguration latency 掃描 10 ns–10 ms。",
      "baselines_en": "Static rings, 2D/3D tori and generalized Kautz topologies; Birkhoff–von Neumann schedules reconnect communicating pairs at every step; best-of-static-and-BvN; Ring versus Recursive Doubling collective choices. Workloads include Recursive Doubling, Swing/Bine butterfly, Bruck AllReduce and All-to-All, direct All-to-All, binomial/binary-tree broadcast.",
      "baselines_zh": "Baseline 包含 static ring、2D／3D torus、generalized Kautz topology；Birkhoff–von Neumann schedule 每 step 重接通訊 pair；static／BvN 取較佳者；另比較 Ring 與 Recursive Doubling collective。Workload 包含 Recursive Doubling、Swing／Bine butterfly、Bruck AllReduce／All-to-All、direct All-to-All 與 binomial／binary-tree broadcast。",
      "result_en": "At <1-µs switch latency and 1–256-KB messages, packet simulation reports 6.4×, 4.7× and 20× over selected static Recursive Doubling, Swing and All-to-All executions. At 100 µs, 1–256-KB messages average 7.3×, 10× and 5.3× over BvN respectively. Hardware emulation shows about 3× over a static ring and an intermediate regime beating both extremes. Recursive Doubling DP computation is <20 µs up to 64 nodes and averages <35 µs up to 1024. The calibrated model uses α=30.32 µs and effective bandwidth 85.11 Gb/s.",
      "result_zh": "Switch latency <1 µs、message 1–256 KB 時，packet simulation 相對選定 static Recursive Doubling、Swing、All-to-All execution 報告 6.4×、4.7×、20×。100 µs、message 1–256 KB 時，相對 BvN 的平均 gain 各為 7.3×、10×、5.3×。Hardware emulation 相對 static ring 約 3×，中間 regime 同時超越兩端 baseline。Recursive Doubling DP 在 64 node 內 <20 µs，1024 node 內平均 <35 µs。校準 model 使用 α=30.32 µs 與 effective bandwidth 85.11 Gb/s。",
      "limitations_en": "The eight-GPU testbed emulates reconfigurable optics through NIC flow steering and additive switch penalties; a physically changing photonic switch and full training-job deployment require dedicated evidence. General MISOCP reaches up to 64 GPUs; the microsecond synthesis result concerns the structured Recursive Doubling path. Collective algorithm selection, shared-job switching, overlapping pipeline traffic and scale-out integration are extension areas. Source appendices are labelled supporting material.",
      "limitations_zh": "8-GPU testbed 以 NIC flow steering 與 additive switch penalty 模擬 reconfigurable optics；實際 photonic switch 變動與完整 training job deployment 各需專屬證據。General MISOCP 到 64 GPU；microsecond synthesis 結果適用結構化 Recursive Doubling path。Collective algorithm selection、shared-job switching、overlapping pipeline traffic 與 scale-out integration 是延伸方向。原論文 appendix 標為 supporting material。",
      "residual_novelty_en": "Harvest already optimizes setup-aware optical topology schedules. Give it identical orbital contact forecasts and measured PAT/setup α_r, and constrain eligible edges and epoch capacities by those forecasts. Residual novelty requires exogenous, forecast-bounded contact expiry plus dependency-safe progress commit, partial-result retention and recovery; its controller-selected interval boundaries provide a concrete comparator.",
      "residual_novelty_zh": "Harvest 已最佳化 setup-aware optical topology schedule。提供相同 orbital contact forecast 與量測 PAT／setup α_r，再用 forecast 限制 eligible edge 與 epoch capacity。增量需處理外生、forecast-bounded contact expiry，以及 dependency-safe progress commit、partial-result retention 與 recovery；controller 選定的 interval boundary 是具體 comparator。",
      "row_draft_en": "Harvest · topology scheduling within a fixed collective · DP plus MISOCP, structured Recursive Doubling fast path · compare with identical setup/contact input · orbital residual: expiry-bounded durable progress and recovery across contacts.",
      "row_draft_zh": "Harvest · 固定 collective 內的 topology scheduling · DP＋MISOCP、結構化 Recursive Doubling fast path · 使用相同 setup／contact input 比較 · orbital 增量：expiry-bounded durable progress 與跨 contact recovery。",
      "math_latex": "DP[a,k]=\\min_{a<b\\le s+1}\\{t_c(a,b-1)+DP[b,k-1]\\},\\qquad k^*=\\arg\\min_k\\{DP[1,k]+k\\alpha_r\\}",
      "math_source_anchor": "§4.2 Equation (7), PDF p6; §4.3 Equation (8), PDF p7",
      "math_definitions_en": "s is collective step count; a and b are interval boundaries; k counts reconfigurations; t_c is the optimized interval completion time; α_r is the per-change cost. DP excludes that additive cost until choosing k.",
      "math_definitions_zh": "s 為 collective step 數；a／b 為 interval boundary；k 為 reconfiguration 次數；t_c 為最佳化 interval completion time；α_r 為每次切換成本。DP 求值後加總切換成本選 k。",
      "affiliations": "Purdue University, Microsoft Research",
      "source_version": "Author-hosted SIGCOMM 2026 ACM-format paper, 25 pages including supporting appendices",
      "open_source": "https://github.com/STyGIANet/Harvest",
      "gpu_count": "8 physical GPUs; 8–64 simulated GPUs; structured DP up to 1024 modeled nodes",
      "node_count": "8 physical GPU endpoints",
      "evaluation_method": "Hardware Testbed, Simulation, Analytical Model",
      "compute_memory_hw": "NVIDIA GPUs; BlueField-3 NIC GPU forwarding emulation",
      "network_hw": "BlueField-3 DPU",
      "network_topology": "3D Torus",
      "software_simulator": "ASTRA-sim, Gurobi",
      "traffic_pattern": "AllReduce, All-to-All, AllGather",
      "transport_and_interconnect": "RDMA",
      "comm_libraries": "NCCL",
      "pdf_sha256": "069134729d495bcd5bb4d09349634e120cf0ed6799c1c792c84c5433bcee32df",
      "primary_text_sha256": "64e42949c4adb5ca1a878e0e825d9d006d52079785d85bba227014961c742fa0",
      "figure_source_pdf_url": "https://stygianet.cs.purdue.edu/papers/harvest-sigcomm26.pdf",
      "figure_sha256": "a813842221f4710d8ce0e72b0d1c4559e8ade6e9dbd94b1143606aec53b2d2bf",
      "figure_bytes": 111282,
      "figure_number": "2",
      "figure_page_1based": 4,
      "figure_bbox_points": [
        53,
        80,
        399,
        268
      ],
      "figure_acquisition_method": "Original primary PDF crop rendered at 216 DPI",
      "figure_visually_inspected": true,
      "figure_caption_en": "Figure 2 shows Recursive Doubling step demands, the selected physical topologies and the resulting congestion; direct optical reconnection trades congestion against setup.",
      "figure_caption_zh": "Figure 2 呈現 Recursive Doubling step demand、選定 physical topology 與其 congestion；direct optical reconnection 以 setup 成本降低 congestion。",
      "mechanism_scope": "T",
      "deployment_regime": "T",
      "space_ground_path": false,
      "display_regime_en": "T · Terrestrial reference",
      "display_regime_zh": "T · 地面比較基準",
      "primary_pdf_sha256": "069134729d495bcd5bb4d09349634e120cf0ed6799c1c792c84c5433bcee32df",
      "figure_asset": "../figures/harvest.png",
      "figure_acquisition": "Original primary PDF crop rendered at 216 DPI",
      "figure_content_type": "image/png"
    },
    {
      "id": "opus",
      "title": "Opus: Photonic Rail-Optimized Fabric in ML Datacenters",
      "authors": "Eric Ding, Barry Lyu, Bhaskar Kataria, Rachee Singh",
      "year": "2026",
      "venue": "SIGCOMM",
      "url": "https://doi.org/10.1145/3789240.3829187",
      "pdf_url": "https://arxiv.org/pdf/2602.12521",
      "regime": "T · Terrestrial reference",
      "transfer_category": "terrestrial reference",
      "source_tier": "full primary PDF",
      "evidence_status": "Full primary PDF: formulation, system design, evaluation, and scope audited",
      "source_program_url": "https://conferences.sigcomm.org/sigcomm/2026/program/papers/",
      "source_acquired_date": "2026-10-02",
      "publication_type": "SIGCOMM main-track terrestrial baseline",
      "evidence": [
        "§2–3 PDF pp2–5; traffic phases and idle-window characterization",
        "§4 PDF pp5–8; Algorithms 1–2, Figures 7–8",
        "§5.1 PDF p8–9; physical OCS firmware timeline",
        "§5.2 PDF pp8–9; Table 1, Figures 10–11",
        "§5.3 PDF pp9–11; Figures 12–15",
        "§5.4 PDF p11; Table 2, power/cost model",
        "§7 PDF p12; scope and fault discussion",
        "Appendix D PDF p18; EP topology/routing"
      ],
      "built_for_en": "Rail-optimized scale-out ML clusters with predictable hybrid-parallelism phases, strong scale-up connectivity, commodity one-to-one OCS circuits, per-rail orchestrators and an application-level coordination network.",
      "built_for_zh": "面向 rail-optimized scale-out ML cluster，具有可預測 hybrid-parallelism phase、強 scale-up connectivity、commodity one-to-one OCS circuit、逐 rail orchestrator 與 application-level coordination network。",
      "problem_en": "Electrical rails provide high fan-out connectivity at substantial power and cost. Replace their packet switches with optical circuits while retaining useful per-phase connectivity for TP/DP/PP/CP/EP. The controller assigns the same limited ports to different parallelism dimensions over an iteration and hides reconfiguration in idle windows.",
      "problem_zh": "Electrical rail 的高 fan-out connectivity 帶來顯著 power／cost。以 optical circuit 取代 packet switch，同時維持 TP／DP／PP／CP／EP 每 phase 所需 connectivity。Controller 在 iteration 中將有限 port 時間分配給各 parallelism dimension，並利用 idle window 隱藏 reconfiguration。",
      "design_en": "A PyTorch shim profiles the first five steps, intercepts collectives, classifies management versus data traffic and identifies phase boundaries. A per-job controller synchronizes ranks and issues topology IDs to per-rail OCS orchestrators. Locks and completion callbacks drain affected traffic before circuit changes. Provisioning starts the next configuration after the previous phase; per-stage sub-mappings permit asynchronous pipeline progress.",
      "design_zh": "PyTorch shim profile 前五 step、攔截 collective、分類 management／data traffic 並辨識 phase boundary。逐 job controller 同步 rank，向逐 rail OCS orchestrator 發送 topology ID。Lock 與 completion callback 在 circuit change 前排空受影響 traffic。Provisioning 於上一 phase 完成後啟動下一 configuration；逐 stage sub-mapping 支援 pipeline 各自 progress。",
      "model_en": "This is a system/protocol design with phase-indexed topology state and rank-ready counters. Exposed switching cost is Σ_i max(0,T_reconfig−T_window,i). Safety invariants serialize communication and reconfiguration on affected sub-mappings. Topology encoding captures up to nine symmetric parallelisms plus an asymmetric PP dimension. The power/cost model counts electrical ports, OCS ports and transceivers separately.",
      "model_zh": "採用 system／protocol design，維護 phase-indexed topology state 與 rank-ready counter。Exposed switching cost 為 Σ_i max(0,T_reconfig−T_window,i)。Safety invariant 在受影響 sub-mapping 上序列化 communication 與 reconfiguration。Topology encoding 容納九個 symmetric parallelism 與一個 asymmetric PP dimension。Power／cost model 分項計數 electrical port、OCS port 與 transceiver。",
      "evaluation_en": "Physical hardware: four dual-L40 servers, a 64-port Polatis 6000 OCS, dual ConnectX-6 Dx NICs per server and two 100-Gb/s rails, running six-layer Llama-3. Perlmutter emulation executes TorchTitan training up to 64 A100 GPUs with logically enforced circuit connectivity and injected switching delays. ASTRA-sim with Chakra traces evaluates dense/MoE iteration times up to 2048 modeled H200/B200 GPUs, 0–1000-ms switching and 100–1600-Gb/s scale-out links.",
      "evaluation_zh": "Physical hardware 包含四臺 dual-L40 server、64-port Polatis 6000 OCS、每 server 雙 ConnectX-6 Dx NIC 與兩條 100-Gb/s rail，執行 six-layer Llama-3。Perlmutter emulation 在至多 64 A100 執行 TorchTitan training，以邏輯限制 circuit connectivity 並注入 switching delay。ASTRA-sim＋Chakra trace 評估至多 2048 modeled H200／B200 的 dense／MoE iteration time，掃描 0–1000-ms switching 與 100–1600-Gb/s scale-out link。",
      "baselines_en": "Hardware validates phase reconfiguration and link recovery. Perlmutter compares native NCCL/EPS, Opus and Opus+Provisioning. Simulation compares static EPS with every potentially configured link active and ideal one-shot allocation of the same total bandwidth across parallelisms. Power/cost compares rail-optimized EPS and fat-tree EPS using declared per-port accounting.",
      "baselines_zh": "Hardware 驗證 phase reconfiguration 與 link recovery。Perlmutter 比較 native NCCL／EPS、Opus、Opus＋Provisioning。Simulation 比較所有可能 configured link 同時啟用的 static EPS，以及相同總 bandwidth 在 parallelism 間最佳分配的 ideal one-shot。Power／cost 以指定 per-port accounting 比較 rail-optimized EPS 與 fat-tree EPS。",
      "result_en": "Physical optics recover within the 200-ms observation interval, while NIC firmware reports link-up around 6 s, falling to about 3 s with auto-negotiation disabled. At 50-ms injected switching, Perlmutter provisioning reduces step overhead to about 1%/2% for the two Llama configs. Zero-switch control overhead at 64 GPUs falls from 6.13% to 0.79% with provisioning. Simulation reports 5.31% slower than EPS at 128 H200/100 ms, 2.49% at 512 B200/10 ms, and 11.22% at 2048 B200/10 ms. Estimated rail power savings are 23.9× H200 and 15.4× B200; cost savings are 4.3×/3.2×.",
      "result_zh": "Physical optics 在 200-ms observation interval 內恢復，NIC firmware link-up 約 6 s；停用 auto-negotiation 後約 3 s。50-ms injected switching 時，Perlmutter provisioning 將兩種 Llama config 的 step overhead 降到約 1%／2%。64 GPU 的 zero-switch control overhead 以 provisioning 從 6.13% 降到 0.79%。Simulation 相對 EPS 的 overhead：128 H200／100 ms 為 5.31%、512 B200／10 ms 為 2.49%、2048 B200／10 ms 為 11.22%。Estimated rail power saving 為 H200 23.9×、B200 15.4×；cost saving 為 4.3×／3.2×。",
      "limitations_en": "Physical OCS deployment exposes a NIC firmware recovery bottleneck; low-overhead millisecond results use injected-delay emulation or simulation. Simulated EPS has a higher total bandwidth budget and abstracts switch congestion. Energy/cost ratios are per-port estimates with fiber costs outside the accounted components. MoE scale-out All-to-All and shorter phase windows increase overhead: at 88.9% scale-out EP traffic, 256-GPU simulation reports 22.5% degradation at 10-ms switching.",
      "limitations_zh": "Physical OCS deployment 呈現 NIC firmware recovery bottleneck；millisecond switching 的 low-overhead 結果來自 injected-delay emulation 或 simulation。Simulated EPS 使用較高總 bandwidth budget，並以抽象 model 處理 switch congestion。Energy／cost ratio 為 per-port estimate，fiber cost 位於計算元件範圍之外。MoE scale-out All-to-All 與較短 phase window 提高 overhead：88.9% scale-out EP traffic、256-GPU simulation 在 10-ms switching 時報告 22.5% degradation。",
      "residual_novelty_en": "Opus establishes in-job, phase-boundary circuit reconfiguration and traffic-drain safety on real OCS hardware. An orbital comparator should use its locks, rank-ready synchronization, provisioning and fallback with identical contact forecasts. Residual novelty requires externally imposed contact expiry, forecast-error budgets, dependency-safe progress commit before expiry and reusable partial results across reconnection. Controller-chosen circuit timing supplies the terrestrial reference.",
      "residual_novelty_zh": "Opus 已在實體 OCS 展示 in-job、phase-boundary circuit reconfiguration 與 traffic-drain safety。Orbital comparator 應以相同 contact forecast 使用其 lock、rank-ready synchronization、provisioning 與 fallback。增量需處理外部強制 contact expiry、forecast-error budget、expiry 前的 dependency-safe progress commit，以及 reconnection 後重用 partial result。Controller 選定 circuit timing 提供 terrestrial reference。",
      "row_draft_en": "Opus · parallelism-driven photonic rails · real OCS plus emulated and simulated training · compare phase-boundary locking/provisioning with identical forecasts · orbital residual: externally imposed expiry and committed progress surviving reconnection.",
      "row_draft_zh": "Opus · parallelism-driven photonic rail · real OCS＋emulated／simulated training · 用相同 forecast 比較 phase-boundary locking／provisioning · orbital 增量：externally imposed expiry 與 reconnection 後保留的 committed progress。",
      "math_latex": "T_{\\mathrm{exposed}}=\\sum_i\\max\\{0,T_{\\mathrm{reconfig}}-T_{\\mathrm{window},i}\\}",
      "math_source_anchor": "§4.2 Provisioning, PDF p7: source displayed expression; safety invariants G1/G2 in the same section",
      "math_definitions_en": "T_reconfig is circuit setup duration; T_window,i is the post-phase idle window. Affected communication starts after configuration completion; reconfiguration starts after affected in-flight work completes.",
      "math_definitions_zh": "T_reconfig 為 circuit setup duration；T_window,i 為 phase 後 idle window。受影響 communication 在 configuration 完成後開始；reconfiguration 在受影響 in-flight work 完成後開始。",
      "affiliations": "Cornell University, University of Michigan",
      "source_version": "arXiv 2602.12521v3, 2026-07-03; SIGCOMM title and DOI verified through the official program",
      "open_source": "https://github.com/opusfabric/Opus",
      "gpu_count": "8 L40 physical; up to 64 A100 emulated; up to 2048 H200/B200 modeled",
      "node_count": "4 physical dual-GPU servers",
      "evaluation_method": "Hardware Testbed, Simulation, Analytical Model",
      "compute_memory_hw": "NVIDIA L40, NVIDIA A100, NVIDIA H200, NVIDIA B200",
      "network_topology": "Rail-optimized",
      "software_simulator": "ASTRA-sim",
      "traffic_pattern": "AllGather, ReduceScatter, All-to-All, Pipeline Parallelism, Tensor Parallelism, FSDP",
      "transport_and_interconnect": "RoCEv2, NVLink",
      "comm_libraries": "NCCL",
      "pdf_sha256": "58cd4018b5051c1dabaf195d8825754707bc403498ea94969c112a9ff00980c5",
      "primary_text_sha256": "2a5577611063c3e2a860a35186d698b8a117df25f4ac90e38a023fac8c5391bb",
      "figure_source_pdf_url": "https://arxiv.org/pdf/2602.12521",
      "figure_sha256": "be52d74af68b44e6ed1df0312f0b7b278e68f363ded423c1b9445e5409d709a7",
      "figure_bytes": 111798,
      "figure_number": "7",
      "figure_page_1based": 6,
      "figure_bbox_points": [
        53,
        80,
        559,
        232
      ],
      "figure_acquisition_method": "Original primary PDF crop rendered at 216 DPI",
      "figure_visually_inspected": true,
      "figure_caption_en": "Figure 7 shows the shim, per-job controller, per-rail OCS orchestrators and metadata tables coordinating phase-triggered topology changes and RDMA traffic.",
      "figure_caption_zh": "Figure 7 呈現 shim、逐 job controller、逐 rail OCS orchestrator 與 metadata table，協調 phase-triggered topology change 及 RDMA traffic。",
      "mechanism_scope": "T",
      "deployment_regime": "T",
      "space_ground_path": false,
      "display_regime_en": "T · Terrestrial reference",
      "display_regime_zh": "T · 地面比較基準",
      "primary_pdf_sha256": "58cd4018b5051c1dabaf195d8825754707bc403498ea94969c112a9ff00980c5",
      "figure_asset": "../figures/opus.png",
      "figure_acquisition": "Original primary PDF crop rendered at 216 DPI",
      "figure_content_type": "image/png"
    },
    {
      "id": "mixnet",
      "title": "MixNet: A Runtime Reconfigurable Optical-Electrical Fabric for Distributed Mixture-of-Experts Training",
      "authors": "Xudong Liao; Yijun Sun; Han Tian; Xinchen Wan; Yilun Jin; Zilong Wang; Zhenghang Ren; Xinyang Huang; Wenxue Li; Kin Fai Tse; Zhizhen Zhong; Guyue Liu; Ying Zhang; Xiaofeng Ye; Yiming Zhang; Kai Chen",
      "year": "2025",
      "venue": "SIGCOMM",
      "url": "https://doi.org/10.1145/3718958.3750465",
      "pdf_url": "https://xcwanandy.github.io/papers/2025/mixnet-sigcomm25.pdf",
      "regime": "T · Terrestrial reference",
      "arxiv": "2501.03905",
      "affiliations": "Hong Kong University of Science and Technology; Massachusetts Institute of Technology; Peking University; Meta; EmbedWay; Xiamen University",
      "evidence_status": "Complete primary full text reviewed",
      "source_version": "Author-hosted SIGCOMM2025 conference primary PDF, 21 pages",
      "evidence": [
        "§3 PDFp3–5:128 H800 production measurements, locality and timing",
        "§4–§5 PDFp5–8:regional OCS architecture, Algorithm1 and runtime",
        "§6–§7.1 PDFp8–9:32 A100 prototype and packet-level simulation",
        "§7.3–§7.4 PDFp10–11:iteration time and performance per dollar",
        "AppendixB.1 PDFp17 Eq1:conditional-matrix prediction",
        "AppendixC PDFp18–19:OCS and NIC activation timing, reported-time scope",
        "AppendixD.1 PDFp19 and D.7 PDFp21:model scales and setup sensitivity",
        "Figure6 PDFp5"
      ],
      "built_for_en": "Distributed MoE training on a terrestrial GPU cluster augments a global electrical packet fabric with runtime-reconfigurable regional optical circuits. TP remains inside the local scale-up domain; regional EP traffic uses OCS, while DP and PP use the electrical fabric.",
      "built_for_zh": "地面 GPU cluster 的 distributed MoE training 以 runtime-reconfigurable regional optical circuit 擴充 global electrical packet fabric。TP 留在 local scale-up domain；regional EP traffic 使用 OCS，DP 與 PP 使用 electrical fabric。",
      "problem_en": "Data-dependent expert routing creates spatially skewed and temporally changing all-to-all demand. A fixed fabric pays for uniform bisection bandwidth, while an optical design must allocate finite server ports and absorb reconfiguration time before collective phases.",
      "problem_zh": "Data-dependent expert routing 形成空間偏斜且隨時間改變的 all-to-all demand。固定 fabric 需配置均勻 bisection bandwidth；optical design 需分配有限 server port，並在 collective phase 前容納 reconfiguration time。",
      "design_en": "Regional topology controllers collect expert demand; a greedy bottleneck-pair algorithm allocates degree-bounded circuits and permutes NIC mappings for NUMA locality. The custom RDMA collective runtime delegates inter-server EP traffic through gateway GPUs, overlaps inter-host and intra-host transfers, and retains hierarchical DP all-reduce on EPS. Four collective matrices within a layer share identical or transposed structure; available compute phases hide later reconfigurations.",
      "design_zh": "Regional topology controller 收集 expert demand；greedy bottleneck-pair algorithm 配置 degree-bounded circuit，並依 NUMA locality 排列 NIC mapping。Custom RDMA collective runtime 透過 gateway GPU 代理 inter-server EP traffic、重疊 inter-host 與 intra-host transfer，並讓 hierarchical DP all-reduce 使用 EPS。同一 layer 的四個 collective matrix 具有相同或轉置結構；可用 compute phase 容納後續 reconfiguration。",
      "model_en": "Algorithm 1 uses expert demand E, server demand D, optical degree α and allocated circuit matrix C; the next allocation targets the largest D[i,j]/C[i,j] completion estimate and proceeds when both endpoints retain spare ports. AppendixB.1 fits a column-stochastic expert-transition matrix P by weighted squared prediction error over a recent window, then predicts next-layer loads from the current-layer distribution. The simulator converts a profiled computation/communication DAG into packet-level events.",
      "model_zh": "Algorithm1 使用 expert demand E、server demand D、optical degree α 與 allocated circuit matrix C；下一次 allocation 選擇 D[i,j]/C[i,j] completion estimate 最大的 pair，並在兩端仍有空閒 port 時配置 circuit。AppendixB.1 以近期 window 的 weighted squared prediction error 擬合 column-stochastic expert-transition matrix P，再由 current-layer distribution 預測 next-layer load。Simulator 將 profiled computation/communication DAG 轉為 packet-level event。",
      "evaluation_en": "Production profiling uses 128 H800 GPUs and 128 ConnectX-7 400Gbps NICs. The prototype has four servers,32 A100 GPUs,16 ConnectX-6 100Gbps NICs, a32×32 Polatis OCS and SN3700 Ethernet switch; each server assigns 3 NICs to OCS and 1 to EPS. RoCEv2, NCCL and ibverbs carry real Megatron-LM training of truncated Mixtral8×7B, LLaMA-MoE and Qwen-MoE models. FlexFlow plus htsim evaluates full Mixtral8×7B/8×22B, Qwen-MoE and DeepSeek-R1 configurations, normally 1024 GPUs,100–800Gbps links,1µs propagation and 25ms OCS setup; scale sweeps reach 32768 GPUs.",
      "evaluation_zh": "Production profiling 使用128張 H800 與128張 ConnectX-7 400Gbps NIC。Prototype 含四臺 server、32張 A100、16張 ConnectX-6 100Gbps NIC、32×32 Polatis OCS 與 SN3700 Ethernet switch；每臺 server 配置3張 NIC 至 OCS、1張至 EPS。RoCEv2、NCCL 與 ibverbs 執行真實 Megatron-LM training，model 為縮短 layer 的 Mixtral8×7B、LLaMA-MoE、Qwen-MoE。FlexFlow 加 htsim 評估完整 Mixtral8×7B/8×22B、Qwen-MoE、DeepSeek-R1 configuration，主要採1024張 GPU、100–800Gbps link、1µs propagation、25ms OCS setup；scale sweep 達32768張 GPU。",
      "baselines_en": "The prototype compares an ideal 4×100Gbps-per-server EPS switch configuration. Simulations compare full-bisection fat-tree,3:1 oversubscribed fat-tree, rail-optimized fabric and TopoOpt; fixed parallel strategies isolate the fabric contribution. Prediction alternatives use uniform demand and the previous layer distribution; setup and optical-degree sweeps expose hardware sensitivity.",
      "baselines_zh": "Prototype 比較每臺 server 4×100Gbps 的 ideal EPS switch configuration。Simulation 比較 full-bisection fat-tree、3:1 oversubscribed fat-tree、rail-optimized fabric 與 TopoOpt；固定 parallel strategy 量測 fabric contribution。Prediction alternative 採 uniform demand 與 previous-layer distribution；setup、optical-degree sweep 呈現 hardware sensitivity。",
      "result_en": "The reported 32-GPU prototype iteration times are comparable to the ideal EPS baseline under its adjusted activation-time accounting. Simulated networking performance per dollar improves 1.2–1.5× over fat-tree at 100Gbps and 1.9–2.3× at 400Gbps across four MoE configurations; these ratios measure network cost efficiency. Simulated iteration time improves up to 2.5× over TopoOpt. Commodity OCS setup averages 41.44–46.75ms across 1–16 pairs; separate NIC reactivation averages 5.67s and reaches 6.33s at P99.",
      "result_zh": "在調整 activation-time accounting 的條件下，reported32-GPU prototype iteration time 與 ideal EPS baseline 接近。四種 MoE configuration 的 simulated networking performance per dollar 相對 fat-tree，在100Gbps 提升1.2–1.5×、400Gbps 提升1.9–2.3×；這些 ratio 衡量 network cost efficiency。Simulated iteration time 相對 TopoOpt 最高加速2.5×。Commodity OCS setup 在1–16 pair 下平均41.44–46.75ms；另計的 NIC reactivation 平均5.67s、P99達6.33s。",
      "limitations_en": "AppendixC calculates prototype training time with NIC reactivation removed; the simulated 25ms optical setup is a separate assumption. Optical fiber circuits, region locality, available EPS fallback, fixed parallel strategies and constant propagation define the evaluated regime. Orbital transfer should preserve demand prediction and greedy port allocation while constraining circuits to the same forecast contact graph and charging optical PAT plus transceiver recovery. Residual research can optimize expert-placement and collective epoch commits against externally imposed contact expiry and forecast error, measuring durable optimizer progress, rollback bytes and terminal-energy cost.",
      "limitations_zh": "AppendixC 的 prototype training time 扣除 NIC reactivation；simulated25ms optical setup 則屬另一項 assumption。Optical fiber circuit、region locality、可用 EPS fallback、固定 parallel strategy 與 constant propagation 定義 evaluated regime。Orbital transfer 可保留 demand prediction、greedy port allocation，將 circuit 限制在相同 forecast contact graph，並計入 optical PAT 與 transceiver recovery。Residual research 可依外部決定的 contact expiry、forecast error，共同最佳化 expert placement 與 collective epoch commit，量測 durable optimizer progress、rollback byte、terminal-energy cost。",
      "figure_caption_en": "Original Figure 6 separates local scale-up connectivity, regional optical circuits and the global electrical fabric. This hierarchy already co-designs MoE communication and runtime optical reconfiguration; orbital adaptation adds finite contact validity, terminal acquisition and epoch progress constraints.",
      "figure_caption_zh": "原始 Figure6 將 local scale-up connectivity、regional optical circuit、global electrical fabric 串接。此 hierarchy 已共同設計 MoE communication 與 runtime optical reconfiguration；orbital adaptation 加入有限 contact validity、terminal acquisition 與 epoch progress constraint。",
      "formula_latex": "\\min_{P}\\sum_{i=1}^{k}w_i\\lVert Y_i-PX_i\\rVert_2^2\\quad\\mathrm{s.t.}\\quad0\\le P_{ab}\\le1,\\ \\sum_aP_{ab}=1",
      "formula_source": "AppendixB.1 Eq1 PDFp17; probability constraints stated immediately below Eq1",
      "formula_explanation_en": "P maps normalized expert-load distribution X_i in one layer to Y_i in the next; k is the recent window and w_i its temporal weights. This is traffic prediction. Circuit allocation uses Algorithm 1 and optical degree α.",
      "formula_explanation_zh": "P 將某 layer 的 normalized expert-load distribution X_i 對應至下一 layer 的 Y_i；k 為近期 window，w_i 為 temporal weight。此式描述 traffic prediction。Circuit allocation 使用 Algorithm1 與 optical degree α。",
      "figure_number": 6,
      "figure_page_1based": 5,
      "figure_bbox_points": [
        314,
        82,
        562,
        212
      ],
      "figure_sha256": "7c991846615b494b74766e64449f6020224da7285972e041f8b451954506b3fe",
      "math_latex": "\\min_{P}\\sum_{i=1}^{k}w_i\\lVert Y_i-PX_i\\rVert_2^2\\quad\\mathrm{s.t.}\\quad0\\le P_{ab}\\le1,\\ \\sum_aP_{ab}=1",
      "math_source_anchor": "AppendixB.1 Eq1 PDFp17; probability constraints stated immediately below Eq1",
      "math_definitions_en": "P maps normalized expert-load distribution X_i in one layer to Y_i in the next; k is the recent window and w_i its temporal weights. This is traffic prediction. Circuit allocation uses Algorithm 1 and optical degree α.",
      "math_definitions_zh": "P 將某 layer 的 normalized expert-load distribution X_i 對應至下一 layer 的 Y_i；k 為近期 window，w_i 為 temporal weight。此式描述 traffic prediction。Circuit allocation 使用 Algorithm1 與 optical degree α。",
      "mechanism_scope": "T — Terrestrial runtime optical MoE fabric reference",
      "deployment_regime": "T",
      "space_ground_path": false,
      "display_regime_en": "T · Terrestrial reference",
      "display_regime_zh": "T · 地面比較基準",
      "primary_pdf_sha256": "1ae9d2f862c321d3ab0555d8211b469e9aef3c857d5bda1bf8173eb8188ae43c",
      "figure_provenance": {
        "id": "mixnet",
        "figure_number": 6,
        "figure_page_1based": 5,
        "figure_bbox_points": [
          314,
          82,
          562,
          212
        ],
        "pdf_page_size_points": [
          612,
          792
        ],
        "capture": "Poppler 240dpi original PDF crop",
        "figure_sha256": "7c991846615b494b74766e64449f6020224da7285972e041f8b451954506b3fe",
        "figure_bytes": 61676,
        "figure_dimensions": [
          826,
          434
        ],
        "source_url": "https://xcwanandy.github.io/papers/2025/mixnet-sigcomm25.pdf",
        "primary_sha256": "1ae9d2f862c321d3ab0555d8211b469e9aef3c857d5bda1bf8173eb8188ae43c",
        "visually_inspected": true,
        "inspection_scope": "Complete architecture, labels, arrows and original figure caption retained"
      },
      "figure_acquisition_method": "Primary figure extraction",
      "figure_asset": "../figures/mixnet.png",
      "figure_acquisition": "Primary figure extraction",
      "figure_content_type": "image/png",
      "figure_bytes": 61676
    },
    {
      "id": "geoorchestra",
      "title": "GeoOrchestra: Orchestrating Heterogeneous Geo-Distributed Training with Network-Aware Scheduling",
      "authors": "Ting Liu; Qinghua Wu; Jun Zhou; Yuan Sun; Jingbin Yang; Jinglei Pei; Chen Zhang; Heng Pan; Zhenyu Li; Yunjie Liu",
      "year": "2026",
      "venue": "SIGCOMM",
      "url": "https://doi.org/10.1145/3789240.3829136",
      "pdf_url": "https://dl.acm.org/doi/pdf/10.1145/3789240.3829136",
      "regime": "T · Terrestrial reference",
      "arxiv": "",
      "affiliations": "Institute of Computing Technology, Chinese Academy of Sciences; Purple Mountain Laboratories; University of Chinese Academy of Sciences; Computer Network Information Center, Chinese Academy of Sciences",
      "evidence_status": "Complete primary full text reviewed",
      "source_version": "Official ACM eReader conference primary paper,14 PDF pages, proceedings1095–1108",
      "evidence": [
        "§3.2 PDFp5–6 Eqs1–7 and Algorithm1:performance/memory/cost pruning",
        "§3.3 PDFp6–8 Algorithm2 Eqs8–11:layer shifts, buffer depth and bandwidth search",
        "§3.4 PDFp8:time-slot guarantees, work-conserving interleaving, candidate failover",
        "§4.1–§4.2 PDFp8–9:Check-and-Commit launch and final receiver commit ACK",
        "§5.1 PDFp9 Tables3–4:hardware, WAN, model families and baselines",
        "§5.2–§5.4 PDFp10–11:model errors, scoped throughput and planner overhead",
        "§6 PDFp12:memory-asymmetry, batch-size and dense-model scope",
        "Figure5 PDFp5"
      ],
      "built_for_en": "Dense LLM training across heterogeneous terrestrial GPU clusters shares a constrained WAN. The planner combines resource selection, intra-DC parallelism, inter-DC pipeline partitioning, memory buffering and bandwidth allocation.",
      "built_for_zh": "異質地面 GPU cluster 的 dense LLM training 共用受限 WAN。Planner 結合 resource selection、intra-DC parallelism、inter-DC pipeline partitioning、memory buffering 與 bandwidth allocation。",
      "problem_en": "Compute and memory capabilities scale asymmetrically across GPU generations, while shared WAN contention changes the usefulness of a parallel plan. The system seeks low iteration latency subject to stage memory limits and user cost budgets, with stable multi-tenant bandwidth service.",
      "problem_zh": "各代 GPU 的 compute、memory capability 呈現獨立擴展，共用 WAN contention 也會改變 parallel plan 的效益。System 在 stage memory limit、user cost budget 下尋求低 iteration latency，並提供穩定 multi-tenant bandwidth service。",
      "design_en": "An Analyzer prunes resource subsets through optimistic latency and cost bounds. The Orchestrator greedily shifts contiguous layers away from straggler stages or expands activation buffers to hide WAN delay, then adjusts bandwidth toward the rate needed by the compute window. The Runtime combines virtual-hard-pipe time slots with work-conserving interleaving, candidate-plan failover, synchronized Check-and-Commit launch and RDMA gateway buffering through final receiver commit ACK.",
      "design_zh": "Analyzer 透過 optimistic latency、cost bound 修剪 resource subset。Orchestrator 以 greedy 方法將連續 layer 從 straggler stage 移出，或增加 activation buffer 以容納 WAN delay，再將 bandwidth 調整至 compute window 所需 rate。Runtime 結合 virtual-hard-pipe time slot、work-conserving interleaving、candidate-plan failover、同步 Check-and-Commit launch，以及保留至 final receiver commit ACK 的 RDMA gateway buffer。",
      "model_en": "Equations 1–4 minimize modeled iteration time over resource set R and parallel strategy S under peak memory bounds. Pruning uses a continuous compute-throughput relaxation, PP payload divided by guaranteed bandwidth plus RTT, CDI=T_base/T_lower>1 and a cost ceiling. Equation 9 bounds buffer depth by available memory; Equation 10 models bubble reduction through extra overlapped micro-batches. Each tenant receives B_min=|T_u|·BW_max/N from its allocated slots.",
      "model_zh": "Equations1–4 在 peak memory bound 下，依 resource set R、parallel strategy S 最小化 modeled iteration time。Pruning 使用 continuous compute-throughput relaxation、PP payload 除以 guaranteed bandwidth 再加 RTT、CDI=T_base/T_lower>1 與 cost ceiling。Equation9 依 available memory 限定 buffer depth；Equation10 以額外重疊 micro-batch 建模 bubble reduction。每個 tenant 依 allocated slot 取得 B_min=|T_u|·BW_max/N。",
      "evaluation_en": "A six-node physical testbed has 48 GPUs:24 H20-141GB and 24 V100-32GB. Each server has four ConnectX-7 100Gbps NICs; the cross-cluster WAN spans 2000km with 10Gbps capacity and 20ms RTT. Experiments cover dense OPT1.3B–175B and Qwen1.5 1.8B–72B families. Simulations extend to 256–1024 mixed V100/H20/A100/H100 GPUs across 4–16 clusters; two-cluster traces validate simulator error within 5%. Reported metrics include throughput, iteration/memory prediction error and planner latency.",
      "evaluation_zh": "六臺 node 的 physical testbed 共48張 GPU：24張 H20-141GB、24張 V100-32GB。每臺 server 配置四張 ConnectX-7 100Gbps NIC；cross-cluster WAN 橫跨2000km，capacity 為10Gbps、RTT為20ms。Experiment 涵蓋 dense OPT1.3B–175B、Qwen1.5 1.8B–72B family。Simulation 擴展至4–16 cluster 的256–1024張混合 V100/H20/A100/H100；two-cluster trace 驗證 simulator error 在5%內。Reported metric 包含 throughput、iteration/memory prediction error 與 planner latency。",
      "baselines_en": "Aceso, Varuna, DTFM and Sailor compare parallel planning and training throughput. Homogeneous physical, heterogeneous physical, two-cluster simulation validation and multi-cluster scaling are separate settings. Table 5 compares pruned search with the corresponding unpruned search. Pure-DP systems are discussed under the model-replica memory requirements of large-model V100 deployments.",
      "baselines_zh": "Aceso、Varuna、DTFM、Sailor 比較 parallel planning 與 training throughput。Homogeneous physical、heterogeneous physical、two-cluster simulation validation、multi-cluster scaling 各自定義 measurement setting。Table5 比較 pruned search 與對應 unpruned search。Pure-DP system 的討論依 large-model V100 deployment 的 model-replica memory requirement 界定。",
      "result_en": "Physical heterogeneous throughput improves up to 32% over Sailor; homogeneous physical throughput improves about 1.15× over Sailor and matches Aceso. Large-scale simulations report 1.63× over Sailor for Qwen and 1.8× for OPT; the abstract 1.6–1.8× headline aligns with this scaling evidence. Heterogeneous iteration-time prediction error stays below 12% and memory error below 18%. Table 5 search-plus-pruning takes 41.5/96.5/164.8s at 32/64/128 GPUs, versus 958.7/1748.2/3641.9s for unpruned search.",
      "result_zh": "Physical heterogeneous throughput 相對 Sailor 最高提升32%；homogeneous physical 相對 Sailor 約加速1.15×，並與 Aceso 接近。Large-scale simulation 的 Qwen 相對 Sailor 加速1.63×、OPT加速1.8×；abstract 的1.6–1.8× headline 對應此 scaling evidence。Heterogeneous iteration-time prediction error 低於12%，memory error 低於18%。Table5 的 search-plus-pruning 在32/64/128張 GPU 下需41.5/96.5/164.8s，unpruned search 則需958.7/1748.2/3641.9s。",
      "limitations_en": "The evaluated workload is dense training with fixed tensor shapes and sufficient batch depth; extra overlap benefits depend on memory headroom. Existing contributions cover joint strategy/bandwidth search, dynamic WAN slots, synchronized launch, gateway final commit ACK and failed-chunk restart or reroute. Orbital residual should couple externally imposed, forecast-bounded contact expiry with PAT setup, buffer/energy limits and dependency-consistent collective epoch commits. Dedicated isolation, recovery, durable-progress and power measurements can establish that residual.",
      "limitations_zh": "Evaluated workload 為 fixed tensor shape、足夠 batch depth 的 dense training；額外 overlap 效益依 memory headroom 決定。Existing contribution 涵蓋 joint strategy/bandwidth search、dynamic WAN slot、同步 launch、gateway final commit ACK，以及 failed-chunk restart 或 reroute。Orbital residual 可將外部決定且受 forecast uncertainty 限制的 contact expiry，結合 PAT setup、buffer/energy limit 與 dependency-consistent collective epoch commit。Dedicated isolation、recovery、durable-progress、power measurement 可建立此 residual。",
      "figure_caption_en": "Original Figure 5 connects heterogeneous clusters and user constraints to the Analyzer, Orchestrator and Runtime feedback loop. The paper already jointly plans computation and WAN resources; the orbital research question adds externally imposed contact deadlines and durable collective progress.",
      "figure_caption_zh": "原始 Figure5 將 heterogeneous cluster、user constraint 接至 Analyzer、Orchestrator、Runtime feedback loop。Paper 已共同規劃 computation、WAN resource；orbital research question 加入外部決定的 contact deadline 與 durable collective progress。",
      "formula_latex": "T_{\\mathrm{iter}}=(P-1)(T_{\\mathrm{step}}+T_{\\mathrm{comm}})+(M-1)T_{\\mathrm{step}}",
      "formula_source": "§3.2.1 Eq1 PDFp5; iteration-latency approximation for pruning and planning",
      "formula_explanation_en": "P is pipeline stage count, M micro-batch count, T_step the bottleneck stage latency and T_comm the exposed WAN communication during warm-up. Equation 2 adds M_peak=M_static+β·μ_act+M_frag≤M_limit; β is buffered micro-batch depth, μ_act activation memory per micro-batch and M_frag fragmentation overhead.",
      "formula_explanation_zh": "P 為 pipeline stage 數、M為 micro-batch 數、T_step為 bottleneck stage latency、T_comm為 warm-up 期間 exposed WAN communication。Equation2 加入 M_peak=M_static+β·μ_act+M_frag≤M_limit；β為 buffered micro-batch depth、μ_act為每個 micro-batch 的 activation memory、M_frag為 fragmentation overhead。",
      "figure_number": 5,
      "figure_page_1based": 5,
      "figure_bbox_points": [
        63,
        84,
        548,
        236.4
      ],
      "figure_sha256": "505b4b057ab578023d0c9f52f25e1d83d0925c25a6f775d106742144080932fa",
      "math_latex": "T_{\\mathrm{iter}}=(P-1)(T_{\\mathrm{step}}+T_{\\mathrm{comm}})+(M-1)T_{\\mathrm{step}}",
      "math_source_anchor": "§3.2.1 Eq1 PDFp5; iteration-latency approximation for pruning and planning",
      "math_definitions_en": "P is pipeline stage count, M micro-batch count, T_step the bottleneck stage latency and T_comm the exposed WAN communication during warm-up. Equation 2 adds M_peak=M_static+β·μ_act+M_frag≤M_limit; β is buffered micro-batch depth, μ_act activation memory per micro-batch and M_frag fragmentation overhead.",
      "math_definitions_zh": "P 為 pipeline stage 數、M為 micro-batch 數、T_step為 bottleneck stage latency、T_comm為 warm-up 期間 exposed WAN communication。Equation2 加入 M_peak=M_static+β·μ_act+M_frag≤M_limit；β為 buffered micro-batch depth、μ_act為每個 micro-batch 的 activation memory、M_frag為 fragmentation overhead。",
      "mechanism_scope": "T — Terrestrial heterogeneous WAN training reference",
      "deployment_regime": "T",
      "space_ground_path": false,
      "display_regime_en": "T · Terrestrial reference",
      "display_regime_zh": "T · 地面比較基準",
      "figure_provenance": {
        "id": "geoorchestra",
        "figure_number": 5,
        "figure_page_1based": 5,
        "figure_bbox_points": [
          63,
          84,
          548,
          236.4
        ],
        "pdf_page_size_points": [
          612,
          792
        ],
        "capture": "Official ACM eReader original PDF canvas screenshot crop",
        "figure_sha256": "505b4b057ab578023d0c9f52f25e1d83d0925c25a6f775d106742144080932fa",
        "figure_bytes": 179186,
        "figure_dimensions": [
          808,
          254
        ],
        "source_url": "https://dl.acm.org/doi/epdf/10.1145/3789240.3829136",
        "primary_sha256": "edf5ba391a74e528bfc8a180d116b4c2bf6dec5a1b70e3127ee26ac4836ea499",
        "viewport_crop_pixels": [
          543,
          200,
          1351,
          454
        ],
        "canvas_origin_pixels": [
          438,
          60
        ],
        "canvas_scale_pixels_per_point": 1.6666666666666667,
        "visually_inspected": true,
        "inspection_scope": "Complete architecture, labels, arrows and original figure caption retained"
      },
      "figure_acquisition_method": "Primary figure extraction",
      "figure_asset": "../figures/geoorchestra.png",
      "figure_acquisition": "Primary figure extraction",
      "figure_content_type": "image/png",
      "figure_bytes": 179186
    },
    {
      "id": "preccl",
      "title": "PReCCL: Performant and Resilient Collective Communication via Integrated Inband Telemetry and Workload Reallocation",
      "authors": "Zhiyong Chen, Kaihui Gao, Li Chen, Rui Yan, Zihan Yan, Fei Gui, Dan Li, Jiamin Cao, Jiaqi Gao",
      "year": "2026",
      "venue": "SIGCOMM",
      "url": "https://doi.org/10.1145/3789240.3829133",
      "pdf_url": "https://dl.acm.org/doi/pdf/10.1145/3789240.3829133",
      "regime": "T · Terrestrial reference",
      "transfer_category": "terrestrial reference",
      "source_tier": "full primary ACM eReader text",
      "evidence_status": "Full official ACM eReader: all 19 PDF pages acquired and reviewed",
      "source_version": "Official 19-page SIGCOMM 2026 conference paper, proceedings pp396–414",
      "evidence": [
        "§3.1 Figure 5 PDF p4: architecture",
        "§3.2.1 Equation 1 PDF p5: stall-count model; §3.4 PDF p6: epoch consistency",
        "§3.5 Algorithm 1/Equations 3–4 PDF pp6–7: cross-VT reallocation; Appendix A PDF p14: conditional formulation",
        "§4 PDF pp7–8 and Appendix F PDF p16: partial-VT fault scope and socket control",
        "§5–6.2 PDF pp8–9 and Appendix G PDF p16: NCCL v2.29/32-GPU implementation and measured training",
        "§6.3 PDF pp9–10; Appendix H PDF pp16–17: 300-job replay and 512–1024-GPU Multiverse",
        "§6.5–6.7 PDF pp10–12; Appendices I/K/M PDF pp17–18: overhead, production trial, applicability"
      ],
      "built_for_en": "Repeated training collectives share heterogeneous virtual topologies in multi-tenant GPU clusters, where one slow topology delays the entire collective.",
      "built_for_zh": "重複執行的 training collective 在 multi-tenant GPU cluster 共用異質 virtual topology；單一緩慢 topology 延長整個 collective completion。",
      "problem_en": "A collective communication task partitions tensor bytes across ordered GPU graphs called virtual topologies (VTs). Equal byte allocation leaves healthy VTs waiting for a congested or partly failed VT; the library needs a collective-level view and consistent runtime reallocation across ranks.",
      "problem_zh": "Collective communication task 將 tensor byte 分配至稱為 virtual topology（VT）的有序 GPU graph。等量 byte allocation 使 healthy VT 等待擁塞或部分故障的 VT；library 需要 collective-level view，以及跨 rank 一致的 runtime reallocation。",
      "design_en": "Four modules collect FIFO stall counts, choose valid cross-VT policies, compute allocation and enforce it at collective boundaries. Ring/Tree use independent contiguous tensor blocks; AlltoAll uses an intra-node, single-relay GPU policy that prices the extra hop. Epoch-tagged piggyback metadata carries VT ordering and checksum; all ranks validate and deterministically derive the next allocation after the current collective completes, retaining existing NCCL transport connections. Partial-VT failure triggers auxiliary socket signaling, masking, collective retry and lightweight recovery probes.",
      "design_zh": "四個 module 蒐集 FIFO stall count、選擇有效 cross-VT policy、計算 allocation，並在 collective boundary 執行。Ring／Tree 使用獨立 contiguous tensor block；AlltoAll 採 host 內 single-relay GPU policy，計入額外 hop。Epoch-tagged piggyback metadata 攜帶 VT ordering 與 checksum；各 rank 在當前 collective 完成後驗證 metadata，依 deterministic rule 產生下一 allocation，並保留既有 NCCL transport connection。Partial-VT failure 啟動 auxiliary socket signaling、masking、collective retry 與輕量 recovery probe。",
      "model_en": "Equation 1 estimates VT completion from stall count, bytes and fixed overhead through a fitted linear model. A TeXCP-inspired iterative policy shifts work toward faster VTs and normalizes the total byte allocation. The epoch invariant keeps each allocation immutable during execution and admits its successor after completed execution and validated common metadata. Appendix A models stable positive effective bandwidth within a collective and a min-max allocation objective; these assumptions define the analytical scope. Calibration refits every 10 s by default using a 1000-sample window, at least 100 samples, outlier filtering and EMA smoothing.",
      "model_zh": "Equation 1 以 fitted linear model，依 stall count、byte 與 fixed overhead 估計 VT completion。TeXCP-inspired iterative policy 將工作移至較快 VT，並 normalization total byte allocation。Epoch invariant 使執行中的 allocation 固定，且在 execution 完成與共同 metadata 驗證後允許 successor。Appendix A 使用 collective 內 stable、positive effective bandwidth 與 min-max allocation objective；這些假設定義 analytical scope。Calibration 預設每 10 s refit，採 1000-sample window、至少 100 sample、outlier filtering 與 EMA smoothing。",
      "evaluation_en": "The real testbed has four servers with 32 Hopper GPUs, 80 GB HBM per GPU, eight 400 Gbps ConnectX-7 NICs per server, NVLink/NVSwitch and a two-tier leaf–spine fabric. NCCL v2.29 and Megatron-LM run concurrent GPT, Qwen-MoE and BERT training. Multiverse simulates 512/1024 GPUs with Poisson job arrivals and random four-host placements; matched testbed bus-bandwidth points agree within 6%. A separate 300-job Crux-placement replay runs on 1024 H200 GPUs. The production trial compares 146 completed jobs over two weeks with a preceding two-week NCCL period on 1024 H200 GPUs across 256 servers; Appendix K labels this evidence observational.",
      "evaluation_zh": "真實 testbed 包含四臺 server、32 顆 Hopper GPU，每 GPU 80 GB HBM；每 server 八張 400 Gbps ConnectX-7，搭配 NVLink／NVSwitch 與 two-tier leaf–spine fabric。NCCL v2.29 和 Megatron-LM 執行並行 GPT、Qwen-MoE 與 BERT training。Multiverse 模擬 512／1024 GPU，採 Poisson job arrival 與 random four-host placement；匹配 testbed 的 bus-bandwidth point 差距在 6% 內。獨立 300-job Crux-placement replay 在 1024 顆 H200 執行。Production trial 在 256 臺 server、1024 顆 H200 上，將兩週內完成的 146 個 job 與前兩週 NCCL period 比較；Appendix K 將此證據標記為 observational。",
      "baselines_en": "Real training compares NCCL v2.29, SyCCL and MCCS. Large-scale simulation additionally compares Crux and FuseLink; Crux-placement replay compares Crux with NCCL against Crux with PReCCL. Network-layer experiments use ECMP/DCQCN, packet spraying and HPCC. Ablations replace stall-count estimation with direct completion-time measurement and the coordinated allocator with per-VT AIMD. Startup nccl-tests chooses algorithm/group-size activation thresholds.",
      "baselines_zh": "真實 training 比較 NCCL v2.29、SyCCL 和 MCCS。Large-scale simulation 額外比較 Crux 與 FuseLink；Crux-placement replay 比較 Crux 搭配 NCCL 和 Crux 搭配 PReCCL。Network-layer experiment 使用 ECMP／DCQCN、packet spraying 與 HPCC。Ablation 以 direct completion-time measurement 取代 stall-count estimation，並以 per-VT AIMD 取代 coordinated allocator。Startup nccl-tests 選擇 algorithm／group-size activation threshold。",
      "result_en": "Under the stated 32-GPU multi-tenant workload, average Ring-AllReduce/AlltoAll bus bandwidth improves 1.8×/2.1× over NCCL for messages above 64 MB; GPT/Qwen/BERT training speeds up 1.19×/1.21×/1.18×. The 300-job placement replay reduces median/P95 JCT by 5.5%/22.1%. The observational production trial reports median/P95 JCT reductions of 5.7%/51.2% and median NIC utilization improvement of 21.4%; gray-failure recovery drives much of the tail gain. At 64 MB, reported AlltoAll and Ring/Tree overheads are 0.11% and 0.97%. A partial NIC failure retains 78% of pre-failure bus bandwidth after reconvergence within five reported rounds.",
      "result_zh": "在指定 32-GPU multi-tenant workload、message 超過 64 MB 時，平均 Ring-AllReduce／AlltoAll bus bandwidth 相對 NCCL 提升 1.8×／2.1×；GPT／Qwen／BERT training speedup 為 1.19×／1.21×／1.18×。300-job placement replay 使 median／P95 JCT 降低 5.5%／22.1%。Observational production trial 報告 median／P95 JCT 降低 5.7%／51.2%，median NIC utilization 增加 21.4%；gray-failure recovery 驅動大部分 tail gain。64 MB 時 AlltoAll 與 Ring／Tree reported overhead 為 0.11% 和 0.97%。Partial NIC failure 在五個 reported round 內重新收斂，保留故障前 78% bus bandwidth。",
      "limitations_en": "Adaptation uses at least one preceding collective and stable effective bandwidth within each collective. Ring/Tree activation thresholds range from 8–64 MB across the reported algorithm/group sizes; single-shot and smaller operations use default allocation. Recovery assumes surviving VTs and a functioning auxiliary control path; full group disconnection uses checkpoint restart. AlltoAll evaluation uses symmetric per-rank data and bounded intra-node relay candidates. Production gains have observational before/after scope. The Appendix A convergence-rate arithmetic and the main/appendix averaging definitions merit independent proof/implementation reconciliation; the five-round result is reported empirical evidence. Configured NCCL timeout, retry duration and retried bytes define absolute recovery latency; tensor-output checks establish payload correctness beyond metadata checksums.",
      "limitations_zh": "Adaptation 使用至少一個 preceding collective，且各 collective 內 effective bandwidth 保持穩定。依 reported algorithm／group size，Ring／Tree activation threshold 為 8–64 MB；single-shot 與較小 operation 採 default allocation。Recovery 假設 surviving VT 和可運作 auxiliary control path；full group disconnection 採 checkpoint restart。AlltoAll evaluation 使用 symmetric per-rank data 與有限 host 內 relay candidate。Production gain 具 observational before／after scope。Appendix A convergence-rate arithmetic 與 main／appendix averaging definition 需獨立 proof／implementation reconciliation；five-round result 屬 reported empirical evidence。Configured NCCL timeout、retry duration 與 retried byte 定義 absolute recovery latency；tensor-output check 建立 metadata checksum 之外的 payload correctness。",
      "math_latex": "\\widehat T_i=\\alpha\\,\\mathrm{count}_i+\\beta\\,\\mathrm{byteSize}_i+\\Delta",
      "math_source_anchor": "§3.2.1 Equation 1, PDF p5/proceedings p400; fitted per-VT completion estimator",
      "math_definitions_en": "The estimated completion time T uses a declared time unit; count is a dimensionless stall-event count and byteSize is bytes. Alpha has time per stall event, beta has time per byte, and Delta has time units. Parameters depend on GPU group and architecture. This empirical estimator supplies the allocation policy and requires calibration to the replay hardware and contact regime.",
      "math_definitions_zh": "Estimated completion time T 採宣告的 time unit；count 是 dimensionless stall-event count，byteSize 單位為 byte。Alpha 單位為 time／stall event，beta 為 time／byte，Delta 為 time。Parameter 依 GPU group 與 architecture 改變。此 empirical estimator 提供 allocation policy 輸入，需依 replay hardware 與 contact regime 校準。",
      "residual_novelty_en": "PReCCL already supplies immutable epoch allocation, validated metadata, per-boundary workload shifting and partial-VT recovery. Give it identical contact forecasts and feasible VT candidates. The orbital residual is admission before exogenous contact expiry, progress ownership when a collective outlasts available service, delayed or partitioned control, and priced retransmission/terminal reacquisition. Compare reactive and forecast-equipped adaptations with identical byte, memory, setup and control budgets; measure actual step tails and tensor correctness. Give every adapter the same measured or modeled recovery-control service trace, failure evidence and forecast age; charge auxiliary signaling and retried payload to common capacity and time budgets.",
      "residual_novelty_zh": "PReCCL 已提供 immutable epoch allocation、validated metadata、per-boundary workload shifting 與 partial-VT recovery。向它提供相同 contact forecast 和 feasible VT candidate。Orbital residual 是 exogenous contact expiry 前的 admission、collective 超出 available service 時的 progress ownership、delayed 或 partitioned control，以及具成本的 retransmission／terminal reacquisition。以相同 byte、memory、setup 與 control budget 比較 reactive 和 forecast-equipped adaptation；量測實際 step tail 與 tensor correctness。各 adapter 取得相同 measured 或 modeled recovery-control service trace、failure evidence 和 forecast age；auxiliary signaling 與 retried payload 計入共同 capacity 和 time budget。",
      "figure_caption_en": "Figure 5 connects collective FIFO congestion/fault signals to inband telemetry, cross-VT policy selection, data reallocation and next-collective enforcement. These modules already implement collective-level runtime adaptation; the orbital comparison adds externally bounded contacts and progress commit under the same forecast.",
      "figure_caption_zh": "Figure 5 將 collective FIFO congestion／fault signal 連接至 inband telemetry、cross-VT policy selection、data reallocation 和 next-collective enforcement。這些 module 已實作 collective-level runtime adaptation；orbital comparison 在相同 forecast 下加入外部限定 contact 與 progress commit。",
      "open_source": "Not Specified",
      "gpu_count": "32 Hopper GPUs in lab; 1024 NVIDIA H200 GPUs in production",
      "node_count": "4 lab servers; 256 production servers",
      "evaluation_method": "Hardware Testbed, Simulation, Production Cluster",
      "traffic_pattern": "All-to-All, AllReduce, AllGather, ReduceScatter, Ring-AllReduce",
      "compute_memory_hw": "NVIDIA Hopper, NVIDIA H200",
      "network_topology": "Leaf-Spine, Rail-optimized",
      "network_hw": "ConnectX-7, NVSwitch, Tomahawk ASICs",
      "transport_and_interconnect": "RDMA, NVLink, PCIe 5.0",
      "routing_and_congestion_control": "ECMP, DCQCN, Packet Spraying, PFC",
      "comm_libraries": "NCCL",
      "software_simulator": "Multiverse",
      "affiliations": "Tsinghua University, Zhongguancun Laboratory, Alibaba Cloud",
      "source_program_url": "https://conferences.sigcomm.org/sigcomm/2026/program/papers/",
      "source_acquired_date": "2026-10-02",
      "publication_type": "SIGCOMM main-track terrestrial baseline",
      "pdf_sha256": "",
      "primary_text_sha256": "ee1ae791c1bdeb16c774117d350202cda6e94ed0a8c6b37c61f3ace86ddff53a",
      "figure_number": "5",
      "figure_page_1based": 4,
      "figure_bbox_points": [
        51.75,
        81.375,
        297.0,
        216.375
      ],
      "figure_acquisition_method": "Original official ACM eReader enlarged full-viewport screenshot followed by image crop",
      "figure_sha256": "a9f4204e8c43d419386d6e774c70fe69eaee1371b4595815daf1ff4d1f9b70a8",
      "figure_bytes": 187523,
      "figure_visually_inspected": true,
      "figure_source_pdf_url": "https://dl.acm.org/doi/pdf/10.1145/3789240.3829133",
      "mechanism_scope": "T",
      "deployment_regime": "T",
      "space_ground_path": false,
      "display_regime_en": "T · Terrestrial reference",
      "display_regime_zh": "T · 地面比較基準",
      "figure_provenance": {
        "id": "preccl",
        "source_reader_url": "https://dl.acm.org/doi/epdf/10.1145/3789240.3829133",
        "source_pdf_url": "https://dl.acm.org/doi/pdf/10.1145/3789240.3829133",
        "figure_number": 5,
        "figure_page_1based": 4,
        "figure_bbox_points": [
          51.75,
          81.375,
          297.0,
          216.375
        ],
        "screenshot_bbox_pixels": [
          270,
          277,
          924,
          637
        ],
        "page_canvas_bbox_pixels": [
          132,
          60,
          1632,
          2112
        ],
        "page_size_points": [
          612,
          792
        ],
        "capture_method": "Original official ACM eReader enlarged full-viewport screenshot followed by image crop",
        "capture_sha256": "8b023b9de5764a0790a1332f35c49ce60f99fb4c4504a352755faaf1052642d6",
        "reader_text_sha256": "ee1ae791c1bdeb16c774117d350202cda6e94ed0a8c6b37c61f3ace86ddff53a",
        "figure_sha256": "a9f4204e8c43d419386d6e774c70fe69eaee1371b4595815daf1ff4d1f9b70a8",
        "bytes": 187523,
        "visually_inspected": true
      },
      "figure_asset": "../figures/preccl.png",
      "figure_acquisition": "Original official ACM eReader enlarged full-viewport screenshot followed by image crop",
      "figure_content_type": "image/png"
    }
  ]
}
