@inproceedings{470684eb7d1c4057a61a71035e460f67,
title = "Demo: Split-and-Pipeline: Collaborative Large Model Inference on Edge Devices",
abstract = "Deploying and executing large model inference on edge devices is challenging due to their limited computational power and memory resources. To address this challenge, we present a novel Split-and-Pipeline, a collaborative inference scheme that partitions a large model into multiple submodels and executes them across distributed edge devices in a pipelined manner. The scheme parallelizes data transfer across multiple CPU cores to avoid transmission bottlenecks. We build a real-world testbed using NVIDIA Jetson series edge devices to demonstrate the proposed scheme, achieving 1.2×–3.0× throughput improvement over state-of-the-art baselines.",
keywords = "Edge Networks, Large Models, Split-and-Pipeline Inference",
author = "Zuguang Li and Dongyuan Ou and Wen Wu and Songge Zhang and Shaohua Wu and Xuemin Shen",
note = "Publisher Copyright: {\textcopyright} 2025 Copyright held by the owner/author(s).; 31st Annual International Conference on Mobile Computing and Networking, ACM MobiCom 2025 ; Conference date: 04-11-2025 Through 08-11-2025",
year = "2025",
month = nov,
day = "21",
doi = "10.1145/3680207.3765592",
language = "英语",
series = "ACM MobiCom 2025 - Proceedings of the 2025 the 31st Annual International Conference on Mobile Computing and Networking",
publisher = "Association for Computing Machinery, Inc",
pages = "1207--1209",
booktitle = "ACM MobiCom 2025 - Proceedings of the 2025 the 31st Annual International Conference on Mobile Computing and Networking",
}