diff --git a/benchmark/hicache/data_processing.py b/benchmark/hicache/data_processing.py index ea95dcaa3..8f72a0d95 100644 --- a/benchmark/hicache/data_processing.py +++ b/benchmark/hicache/data_processing.py @@ -21,6 +21,7 @@ from sglang.bench_serving import ( ) from sglang.lang.chat_template import get_chat_template, get_chat_template_by_model_path from sglang.srt.entrypoints.openai.protocol import ChatCompletionMessageContentPart +from sglang.utils import encode_video_base64 # type of content fields, can be only prompts or with images/videos MsgContent = Union[str, List[ChatCompletionMessageContentPart]] @@ -324,9 +325,15 @@ def sample_nextqa_requests( prompt_len = len(prompt_token_ids) output_len = fixed_output_len # max output len, not real output len + # video input + base64_data = encode_video_base64(video.path, video.num_frames) + + # NOTE: This will be replaced by the expanded length from the server + prompt_len += video.num_frames + # add to content content = [ - {"type": "video_url", "video_url": {"url": video}}, + {"type": "image_url", "image_url": {"url": base64_data}}, {"type": "text", "text": prompt}, ] diff --git a/docs/references/frontend/frontend_tutorial.ipynb b/docs/references/frontend/frontend_tutorial.ipynb index 98a9ca50b..836cab627 100644 --- a/docs/references/frontend/frontend_tutorial.ipynb +++ b/docs/references/frontend/frontend_tutorial.ipynb @@ -33,7 +33,7 @@ "from sglang import assistant, function, gen, system, user\n", "from sglang import image\n", "from sglang import RuntimeEndpoint\n", - "from sglang.lang.api import set_default_backend, video\n", + "from sglang.lang.api import set_default_backend\n", "from sglang.srt.utils import load_image\n", "from sglang.test.doc_patch import launch_server_cmd\n", "from sglang.utils import print_highlight, terminate_process, wait_for_server\n", @@ -421,11 +421,7 @@ { "cell_type": "code", "execution_count": null, - "metadata": { - "jupyter": { - "is_executing": true - } - }, + "metadata": {}, "outputs": [], "source": [ "@function\n", @@ -440,30 +436,6 @@ "print_highlight(state[\"answer\"])" ] }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "Ask a question about a video" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "@function\n", - "def video_qa(s, video_file, question):\n", - " s += user(video(video_file) + question)\n", - " s += assistant(gen(\"answer\", max_tokens=256))\n", - "\n", - "\n", - "video_url = \"https://raw.githubusercontent.com/sgl-project/sgl-test-files/refs/heads/main/videos/jobs_presenting_ipod.mp4\"\n", - "state = video_qa(video_url, \"What is in the video?\")\n", - "print_highlight(state[\"answer\"])" - ] - }, { "cell_type": "code", "execution_count": null, diff --git a/python/sglang/lang/api.py b/python/sglang/lang/api.py index 7be588d53..745c656ee 100644 --- a/python/sglang/lang/api.py +++ b/python/sglang/lang/api.py @@ -229,7 +229,7 @@ def image(expr: SglExpr): return SglImage(expr) -def video(path: str, num_frames: int = -1): +def video(path: str, num_frames: int): return SglVideo(path, num_frames) diff --git a/python/sglang/lang/backend/runtime_endpoint.py b/python/sglang/lang/backend/runtime_endpoint.py index a70d85792..1573ca68d 100644 --- a/python/sglang/lang/backend/runtime_endpoint.py +++ b/python/sglang/lang/backend/runtime_endpoint.py @@ -104,7 +104,6 @@ class RuntimeEndpoint(BaseBackend): def commit_lazy_operations(self, s: StreamExecutor): data = {"text": s.text_, "sampling_params": {"max_new_tokens": 0}} self._add_images(s, data) - self._add_videos(s, data) res = http_request( self.base_url + "/generate", json=data, @@ -116,7 +115,6 @@ class RuntimeEndpoint(BaseBackend): def fill_image(self, s: StreamExecutor): data = {"text": s.text_, "sampling_params": {"max_new_tokens": 0}} self._add_images(s, data) - res = http_request( self.base_url + "/generate", json=data, @@ -183,7 +181,6 @@ class RuntimeEndpoint(BaseBackend): data[item] = value self._add_images(s, data) - self._add_videos(s, data) res = http_request( self.base_url + "/generate", @@ -225,7 +222,6 @@ class RuntimeEndpoint(BaseBackend): data["stream"] = True self._add_images(s, data) - self._add_videos(s, data) res = http_request( self.base_url + "/generate", @@ -328,8 +324,6 @@ class RuntimeEndpoint(BaseBackend): def _generate_http_request(self, s: StreamExecutor, data): self._add_images(s, data) - self._add_videos(s, data) - res = http_request( self.base_url + "/generate", json=data, @@ -344,11 +338,6 @@ class RuntimeEndpoint(BaseBackend): assert len(s.images_) == 1, "Only support one image." data["image_data"] = s.images_[0][1] - def _add_videos(self, s: StreamExecutor, data): - if s.videos_: - assert len(s.videos_) == 1, "Only support one video." - data["video_data"] = s.videos_ - def _assert_success(self, res): if res.status_code != 200: try: diff --git a/python/sglang/lang/chat_template.py b/python/sglang/lang/chat_template.py index 851aea40d..80ea6d963 100644 --- a/python/sglang/lang/chat_template.py +++ b/python/sglang/lang/chat_template.py @@ -16,9 +16,7 @@ class ChatTemplate: role_prefix_and_suffix: Dict[str, Tuple[str, str]] stop_str: List[str] = () image_token: str = "" - video_token: str = "