dsh: multimodal video/screen-recording analyzer (GLM-5v + ffmpeg frames); fix File/UploadFile imports + ffmpeg in image
This commit is contained in:
@@ -6,6 +6,8 @@ ENV PYTHONDONTWRITEBYTECODE=1 \
|
|||||||
|
|
||||||
WORKDIR /app
|
WORKDIR /app
|
||||||
|
|
||||||
|
RUN apt-get update && apt-get install -y --no-install-recommends ffmpeg && rm -rf /var/lib/apt/lists/*
|
||||||
|
|
||||||
COPY requirements.txt .
|
COPY requirements.txt .
|
||||||
RUN pip install --no-cache-dir -r requirements.txt
|
RUN pip install --no-cache-dir -r requirements.txt
|
||||||
|
|
||||||
|
|||||||
77
dsh/app.py
77
dsh/app.py
@@ -16,6 +16,8 @@ import json
|
|||||||
import os
|
import os
|
||||||
import base64
|
import base64
|
||||||
import re
|
import re
|
||||||
|
import shutil
|
||||||
|
import subprocess
|
||||||
import threading
|
import threading
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Annotated
|
from typing import Annotated
|
||||||
@@ -23,7 +25,7 @@ from uuid import uuid4
|
|||||||
from datetime import datetime, timezone
|
from datetime import datetime, timezone
|
||||||
|
|
||||||
import httpx
|
import httpx
|
||||||
from fastapi import FastAPI, Form, Request
|
from fastapi import FastAPI, File, Form, Request, UploadFile
|
||||||
from fastapi.responses import HTMLResponse, StreamingResponse
|
from fastapi.responses import HTMLResponse, StreamingResponse
|
||||||
from fastapi.staticfiles import StaticFiles
|
from fastapi.staticfiles import StaticFiles
|
||||||
from fastapi.templating import Jinja2Templates
|
from fastapi.templating import Jinja2Templates
|
||||||
@@ -49,6 +51,7 @@ VISION_BASE = os.getenv("DSH_VISION_BASE", "https://openrouter.ai/api/v1")
|
|||||||
VISION_MODEL = os.getenv("DSH_VISION_MODEL", "openai/gpt-5")
|
VISION_MODEL = os.getenv("DSH_VISION_MODEL", "openai/gpt-5")
|
||||||
VISION_KEY = os.getenv("DSH_VISION_KEY", "")
|
VISION_KEY = os.getenv("DSH_VISION_KEY", "")
|
||||||
IMAGE_MODEL = os.getenv("DSH_IMAGE_MODEL", "openai/gpt-image-1")
|
IMAGE_MODEL = os.getenv("DSH_IMAGE_MODEL", "openai/gpt-image-1")
|
||||||
|
VIDEO_MODEL = os.getenv("DSH_VIDEO_MODEL", "z-ai/glm-5v-turbo")
|
||||||
|
|
||||||
# --- per-user conversations (multiple sessions, persist in the history volume) ---
|
# --- per-user conversations (multiple sessions, persist in the history volume) ---
|
||||||
HISTORY_LIMIT = 60
|
HISTORY_LIMIT = 60
|
||||||
@@ -150,6 +153,45 @@ async def _complete(messages) -> str:
|
|||||||
return data["choices"][0]["message"]["content"]
|
return data["choices"][0]["message"]["content"]
|
||||||
|
|
||||||
|
|
||||||
|
async def _vision_complete(messages) -> str:
|
||||||
|
"""Non-streaming OpenRouter call to the multimodal video/vision model."""
|
||||||
|
key = VISION_KEY
|
||||||
|
base = VISION_BASE
|
||||||
|
headers = {"Authorization": f"Bearer {key}"} if key else {}
|
||||||
|
url = f"{base.rstrip('/')}/chat/completions"
|
||||||
|
payload = {"model": VIDEO_MODEL, "messages": messages, "stream": False, "temperature": 0.3}
|
||||||
|
async with httpx.AsyncClient(timeout=240) as client:
|
||||||
|
r = await client.post(url, json=payload, headers=headers)
|
||||||
|
r.raise_for_status()
|
||||||
|
return r.json()["choices"][0]["message"]["content"]
|
||||||
|
|
||||||
|
|
||||||
|
def _extract_frames(video: str, max_frames: int = 6):
|
||||||
|
"""Sample up to N JPEG frames from a video with ffmpeg; return data URIs."""
|
||||||
|
outdir = f"/tmp/frames_{uuid4().hex}"
|
||||||
|
os.makedirs(outdir, exist_ok=True)
|
||||||
|
try:
|
||||||
|
dur = subprocess.run(
|
||||||
|
["ffprobe", "-v", "error", "-show_entries", "format=duration",
|
||||||
|
"-of", "default=nw=1:nk=1", video],
|
||||||
|
capture_output=True, text=True).stdout.strip()
|
||||||
|
dur = float(dur) if dur else 10
|
||||||
|
fps = max(0.1, min(1.0, max_frames / max(dur, 1)))
|
||||||
|
subprocess.run(
|
||||||
|
["ffmpeg", "-v", "error", "-i", video, "-vf", f"fps={fps}",
|
||||||
|
"-frames:v", str(max_frames), f"{outdir}/f%03d.jpg"],
|
||||||
|
capture_output=True)
|
||||||
|
frames = []
|
||||||
|
for f in sorted(os.listdir(outdir)):
|
||||||
|
p = os.path.join(outdir, f)
|
||||||
|
if os.path.isfile(p) and os.path.getsize(p) > 0:
|
||||||
|
with open(p, "rb") as fh:
|
||||||
|
frames.append("data:image/jpeg;base64," + base64.b64encode(fh.read()).decode())
|
||||||
|
return frames
|
||||||
|
finally:
|
||||||
|
shutil.rmtree(outdir, ignore_errors=True)
|
||||||
|
|
||||||
|
|
||||||
@app.post("/api/summarize/web")
|
@app.post("/api/summarize/web")
|
||||||
async def summarize_web(url: Annotated[str, Form()]) -> dict:
|
async def summarize_web(url: Annotated[str, Form()]) -> dict:
|
||||||
"""Fetch a URL and summarise its text content via the LLM."""
|
"""Fetch a URL and summarise its text content via the LLM."""
|
||||||
@@ -298,6 +340,39 @@ async def chat(message: Annotated[str, Form()],
|
|||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
@app.post("/api/tool/video")
|
||||||
|
async def analyze_video(file: Annotated[UploadFile, File()]) -> dict:
|
||||||
|
"""Analyze a video / screen recording: extract keyframes, send to the
|
||||||
|
multimodal GLM model, return a description."""
|
||||||
|
if file is None or not file.filename:
|
||||||
|
return {"ok": False, "error": "No video chosen."}
|
||||||
|
if not VISION_KEY:
|
||||||
|
return {"ok": False, "error": "No vision API key configured."}
|
||||||
|
vid = f"/tmp/video_{uuid4().hex}.mp4"
|
||||||
|
try:
|
||||||
|
data = await file.read()
|
||||||
|
if len(data) > 200 * 1024 * 1024:
|
||||||
|
return {"ok": False, "error": "Video too large (max 200 MB)."}
|
||||||
|
with open(vid, "wb") as fh:
|
||||||
|
fh.write(data)
|
||||||
|
frames = _extract_frames(vid, max_frames=6)
|
||||||
|
if not frames:
|
||||||
|
return {"ok": False, "error": "Could not extract frames — is this a video?"}
|
||||||
|
content = [
|
||||||
|
{"type": "text", "text": "Analyze this video/screen recording: describe what is shown, "
|
||||||
|
"the main actions, and any visible text or UI. Give a concise bullet summary."}
|
||||||
|
]
|
||||||
|
for b64 in frames:
|
||||||
|
content.append({"type": "image_url", "image_url": {"url": b64}})
|
||||||
|
summary = await _vision_complete([{"role": "user", "content": content}])
|
||||||
|
return {"ok": True, "frames": len(frames), "summary": summary}
|
||||||
|
except Exception as exc: # noqa: BLE001
|
||||||
|
return {"ok": False, "error": f"{type(exc).__name__}: {exc}"}
|
||||||
|
finally:
|
||||||
|
if os.path.exists(vid):
|
||||||
|
os.remove(vid)
|
||||||
|
|
||||||
|
|
||||||
@app.post("/api/tool/image")
|
@app.post("/api/tool/image")
|
||||||
async def generate_image(prompt: Annotated[str, Form()]) -> dict:
|
async def generate_image(prompt: Annotated[str, Form()]) -> dict:
|
||||||
"""Generate an image from a prompt via OpenRouter (gpt-image-1); save to workspace."""
|
"""Generate an image from a prompt via OpenRouter (gpt-image-1); save to workspace."""
|
||||||
|
|||||||
@@ -87,6 +87,8 @@
|
|||||||
<button type="button" id="sumDocs" style="border:1px solid #ddd;border-radius:8px;padding:6px 10px;cursor:pointer">Summarise file</button>
|
<button type="button" id="sumDocs" style="border:1px solid #ddd;border-radius:8px;padding:6px 10px;cursor:pointer">Summarise file</button>
|
||||||
<input id="genInput" type="text" placeholder="Describe an image to generate…" style="flex:2;min-width:200px;border:1px solid #ddd;border-radius:8px;padding:6px 10px" />
|
<input id="genInput" type="text" placeholder="Describe an image to generate…" style="flex:2;min-width:200px;border:1px solid #ddd;border-radius:8px;padding:6px 10px" />
|
||||||
<button type="button" id="genImg" style="border:1px solid #ddd;border-radius:8px;padding:6px 10px;cursor:pointer">Generate image</button>
|
<button type="button" id="genImg" style="border:1px solid #ddd;border-radius:8px;padding:6px 10px;cursor:pointer">Generate image</button>
|
||||||
|
<input type="file" id="vidIn" accept="video/*" title="Analyze a video / screen recording (GLM multimodal)" style="flex:1;min-width:150px;border:1px solid #ddd;border-radius:8px;padding:6px;">
|
||||||
|
<button type="button" id="genVid" style="border:1px solid #ddd;border-radius:8px;padding:6px 10px;cursor:pointer">Analyze video</button>
|
||||||
</div>
|
</div>
|
||||||
</div>
|
</div>
|
||||||
</div>
|
</div>
|
||||||
@@ -203,6 +205,20 @@
|
|||||||
}
|
}
|
||||||
document.getElementById('sumWeb').addEventListener('click', () => runTool('/api/summarize/web', 'urlInput', 'Summarise page', 'url'));
|
document.getElementById('sumWeb').addEventListener('click', () => runTool('/api/summarize/web', 'urlInput', 'Summarise page', 'url'));
|
||||||
document.getElementById('sumDocs').addEventListener('click', () => runTool('/api/summarize/docs', 'fileInput', 'Summarise file', 'path'));
|
document.getElementById('sumDocs').addEventListener('click', () => runTool('/api/summarize/docs', 'fileInput', 'Summarise file', 'path'));
|
||||||
|
document.getElementById('genVid').addEventListener('click', async () => {
|
||||||
|
const f = document.getElementById('vidIn').files[0];
|
||||||
|
if (!f) return;
|
||||||
|
addBubble('user', 'Analyze video: ' + f.name); addBubble('bot', '');
|
||||||
|
const b = logEl.lastElementChild.querySelector('.bubble'); b.textContent = '…analyzing';
|
||||||
|
const fd = new FormData(); fd.append('file', f);
|
||||||
|
try {
|
||||||
|
const res = await fetch('/api/tool/video', { method: 'POST', body: fd });
|
||||||
|
const d = await res.json();
|
||||||
|
b.textContent = d.ok ? (d.summary || '') : '[error: ' + (d.error || 'unknown') + ']';
|
||||||
|
} catch (err) { b.textContent = '[error: ' + err + ']'; }
|
||||||
|
logEl.scrollTop = logEl.scrollHeight;
|
||||||
|
});
|
||||||
|
|
||||||
document.getElementById('genImg').addEventListener('click', async () => {
|
document.getElementById('genImg').addEventListener('click', async () => {
|
||||||
const val = document.getElementById('genInput').value.trim(); if (!val) return;
|
const val = document.getElementById('genInput').value.trim(); if (!val) return;
|
||||||
addBubble('user', 'Generate: ' + val); addBubble('bot', '');
|
addBubble('user', 'Generate: ' + val); addBubble('bot', '');
|
||||||
|
|||||||
Reference in New Issue
Block a user