-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathhandler.py
More file actions
78 lines (64 loc) · 2.76 KB
/
Copy pathhandler.py
File metadata and controls
78 lines (64 loc) · 2.76 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
import runpod
import requests
import json
import os
LLAMA_BASE_URL = f"http://localhost:{os.environ.get('LLAMA_PORT', '8080')}"
def forward_to_llama(endpoint: str, payload: dict) -> dict | str:
"""Forward a request to the local llama-server."""
url = f"{LLAMA_BASE_URL}{endpoint}"
response = requests.post(url, json=payload, timeout=300)
response.raise_for_status()
return response.json()
def handler(job: dict) -> dict:
"""
RunPod Serverless handler.
Supports OpenAI-compatible input:
- job["input"]["messages"] -> /v1/chat/completions
- job["input"]["prompt"] -> /v1/completions
- job["input"]["endpoint"] -> raw passthrough to any llama endpoint
"""
job_input = job.get("input", {})
# --- OpenAI Chat Completions ---
if "messages" in job_input:
payload = {
"model": job_input.get("model", "local-model"),
"messages": job_input["messages"],
"max_tokens": job_input.get("max_tokens", 4096),
"temperature": job_input.get("temperature", 1.0),
"top_p": job_input.get("top_p", 0.95),
"top_k": job_input.get("top_k", 20),
"min_p": job_input.get("min_p", 0.0),
"presence_penalty": job_input.get("presence_penalty", 1.5),
"stop": job_input.get("stop", None),
"stream": False,
}
# Remove None values
payload = {k: v for k, v in payload.items() if v is not None}
result = forward_to_llama("/v1/chat/completions", payload)
return result
# --- OpenAI Text Completions ---
elif "prompt" in job_input:
payload = {
"model": job_input.get("model", "local-model"),
"prompt": job_input["prompt"],
"max_tokens": job_input.get("max_tokens", 4096),
"temperature": job_input.get("temperature", 1.0),
"top_p": job_input.get("top_p", 0.95),
"top_k": job_input.get("top_k", 20),
"stop": job_input.get("stop", None),
"stream": False,
}
payload = {k: v for k, v in payload.items() if v is not None}
result = forward_to_llama("/v1/completions", payload)
return result
# --- Raw passthrough to any llama-server endpoint ---
elif "endpoint" in job_input:
endpoint = job_input.pop("endpoint")
result = forward_to_llama(endpoint, job_input)
return result
else:
return {
"error": "Invalid input. Provide 'messages' (chat), 'prompt' (completion), or 'endpoint' (raw)."
}
if __name__ == "__main__":
runpod.serverless.start({"handler": handler})