drowzeys's picture
v1.0 alpha: keys-Auto Receipts Studio (iPhone / may add Autonomous Lamp Skill)
2edb151 verified
Raw
History Blame Contribute Delete
4.51 kB
"""One-click start: optional Gemma vLLM (≤15GB), then LAN UI + browser."""
from __future__ import annotations
import os
import shutil
import socket
import subprocess
import sys
import time
import urllib.error
import urllib.request
import webbrowser
from pathlib import Path
ROOT = Path(__file__).resolve().parent.parent
HOST = "127.0.0.1"
PORT = int(os.environ.get("RECEIPT_UI_PORT", "7860"))
VLLM_PORT = int(os.environ.get("RECEIPT_VLLM_PORT", "8080"))
START_VLLM = os.environ.get("RECEIPT_START_VLLM", "1").lower() not in {"0", "false", "no"}
MAX_GB = os.environ.get("RECEIPT_VLLM_MAX_GB", "15") # unused if UTIL is set in serve-gemma.sh
def _lan_ip() -> str:
sock = socket.socket(socket.AF_INET, socket.SOCK_DGRAM)
try:
sock.connect(("192.0.2.1", 1))
return sock.getsockname()[0]
except OSError:
return "127.0.0.1"
finally:
sock.close()
def _port_up(port: int) -> bool:
sock = socket.socket()
sock.settimeout(0.4)
try:
sock.connect((HOST, port))
return True
except OSError:
return False
finally:
sock.close()
def _vllm_ready() -> bool:
try:
with urllib.request.urlopen(
f"http://127.0.0.1:{VLLM_PORT}/v1/models", timeout=2
) as response:
return response.status == 200
except (urllib.error.URLError, TimeoutError, OSError):
return False
def _can_serve_gemma() -> bool:
if shutil.which("vllm") is None:
return False
model = Path(os.environ.get("RECEIPT_GEMMA_PATH", str(Path.home() / "models-gemma4-12b-it")))
return (model / "config.json").is_file()
def _spawn(cmd: list[str], log: Path, *, bash: bool = False) -> None:
log.parent.mkdir(parents=True, exist_ok=True)
creation = 0
if sys.platform == "win32":
creation = getattr(subprocess, "CREATE_NEW_CONSOLE", 0)
argv = cmd
if bash:
argv = ["bash", *cmd]
with log.open("a", encoding="utf-8") as handle:
subprocess.Popen(
argv,
cwd=str(ROOT),
env=os.environ.copy(),
stdout=handle,
stderr=handle,
creationflags=creation,
start_new_session=sys.platform != "win32",
)
def _ensure_vllm() -> None:
if not START_VLLM:
return
if _vllm_ready():
print(f"Gemma already up on :{VLLM_PORT} (not restarted; 15GB cap applies on a fresh serve).")
return
if not _can_serve_gemma():
print("No local vLLM/Gemma — skip serve. Point .env at the GPU box.")
return
script = ROOT / "scripts" / "serve-gemma.sh"
if not script.is_file():
print(f"missing {script}", file=sys.stderr)
return
os.environ.setdefault("RECEIPT_VLLM_MAX_GB", MAX_GB)
print("Starting Gemma 4 12B vLLM at gpu_memory_utilization=0.15 (FP8, max-model-len 8192)…")
_spawn([str(script)], ROOT / "data" / "vllm-gemma.log", bash=True)
for _ in range(120):
if _vllm_ready():
print("Gemma ready.")
return
time.sleep(5)
print(
f"vLLM still starting. Watch {ROOT / 'data' / 'vllm-gemma.log'}",
file=sys.stderr,
)
def _spawn_ui() -> None:
env = os.environ.copy()
env["RECEIPT_UI_SHARE_LAN"] = "true"
env.setdefault("RECEIPT_IDLE_SECONDS", "5")
log = ROOT / "data" / "ui.log"
log.parent.mkdir(parents=True, exist_ok=True)
creation = 0
if sys.platform == "win32":
creation = getattr(subprocess, "CREATE_NEW_CONSOLE", 0)
with log.open("a", encoding="utf-8") as handle:
subprocess.Popen(
[sys.executable, "-m", "app.cli", "ui"],
cwd=str(ROOT),
env=env,
stdout=handle,
stderr=handle,
creationflags=creation,
start_new_session=sys.platform != "win32",
)
def main() -> None:
os.chdir(ROOT)
_ensure_vllm()
if not _port_up(PORT):
print("Starting Receipt Studio UI…")
_spawn_ui()
for _ in range(40):
if _port_up(PORT):
break
time.sleep(0.25)
else:
print(f"UI did not bind :{PORT}. See {ROOT / 'data' / 'ui.log'}", file=sys.stderr)
raise SystemExit(1)
lan = _lan_ip()
review = f"http://127.0.0.1:{PORT}"
phone = f"http://{lan}:{PORT}/phone"
print(f"Review: {review}")
print(f"Phone: {phone}")
webbrowser.open(review)
if __name__ == "__main__":
main()