"""Preflight checks for OpenAI-compatible LLM endpoints (GenSearcher + browse).""" from __future__ import annotations import os from typing import Tuple import requests def normalize_openai_v1_base(url: str) -> str: u = (url or "").strip().rstrip("/") if not u: return "" if not u.endswith("/v1"): u = u + "/v1" return u def check_v1_models(base_url_v1: str, api_key: str, timeout: float = 15.0) -> Tuple[bool, str]: """ GET {base}/models — standard OpenAI-compatible discovery (vLLM, etc.). """ if not base_url_v1: return False, "URL is empty" url = base_url_v1.rstrip("/") + "/models" headers = {"Authorization": f"Bearer {api_key or 'EMPTY'}"} try: r = requests.get(url, headers=headers, timeout=timeout) if r.status_code == 200: return True, "OK" return False, f"HTTP {r.status_code}: {r.text[:300]}" except requests.exceptions.ConnectionError as e: return False, f"Connection failed (nothing listening or blocked): {e}" except requests.exceptions.Timeout: return False, "Timeout — server not responding" except requests.exceptions.RequestException as e: return False, str(e) def is_localhost_url(url: str) -> bool: u = (url or "").lower() return "127.0.0.1" in u or "localhost" in u def llm_endpoint_status() -> str: """Human-readable markdown for Gradio banner.""" gen_base = normalize_openai_v1_base(os.environ.get("OPENAI_BASE_URL", "")) gen_key = os.environ.get("OPENAI_API_KEY", "EMPTY") browse_base = normalize_openai_v1_base(os.environ.get("BROWSE_SUMMARY_BASE_URL", "")) browse_key = os.environ.get("BROWSE_SUMMARY_API_KEY", os.environ.get("OPENAI_API_KEY", "EMPTY")) lines = ["### Endpoint checks", ""] if not gen_base: lines.append( "**GenSearcher LLM:** `OPENAI_BASE_URL` is **not set**.\n\n" "- **All compute in this Space (recommended for your case):** add a Space variable " "`START_VLLM_GENSEARCHER=1` (and enough GPU). The entrypoint starts **vLLM for Gen-Searcher-8B inside this " "same container** and sets `OPENAI_BASE_URL` to `http://127.0.0.1:8002/v1`. That is still **this Space** — " "not a second Hugging Face Space. The app talks to vLLM over **localhost** inside the container (normal for vLLM).\n\n" "- **Or** set `OPENAI_BASE_URL` yourself to any OpenAI-compatible **`…/v1`** URL (only if the model runs elsewhere).\n" ) else: ok, msg = check_v1_models(gen_base, gen_key) if ok: lines.append(f"**GenSearcher LLM** (`OPENAI_BASE_URL`): reachable — `{gen_base}`") else: lines.append( f"**GenSearcher LLM** (`OPENAI_BASE_URL`): **unreachable** — `{gen_base}`\n\n" f"- Detail: `{msg}`\n" ) if is_localhost_url(gen_base): lines.append( "- You are using **localhost / 127.0.0.1**. Inside a Hugging Face Space, that is **this container only**. " "Either set `START_VLLM_GENSEARCHER=1` (and enough GPU) to run vLLM here, " "or set `OPENAI_BASE_URL` to a **public** inference URL (your vLLM, TGI, etc.).\n" ) lines.append("") if os.environ.get("BROWSE_GENERATE_ENGINE", "").strip().lower() == "vllm": if not browse_base: lines.append( "**Browse summarizer:** `BROWSE_SUMMARY_BASE_URL` is **not set** (needed when `BROWSE_GENERATE_ENGINE=vllm`)." ) else: ok_b, msg_b = check_v1_models(browse_base, browse_key) if ok_b: lines.append(f"**Browse LLM:** OK — `{browse_base}`") else: lines.append( f"**Browse LLM:** **unreachable** — `{browse_base}` — `{msg_b}`" ) if is_localhost_url(browse_base): lines.append( "- Same **localhost** note: use an external Qwen3-VL server or `START_VLLM_BROWSE=1` with extra GPU.\n" ) return "\n".join(lines)