| 7 | |
| 8 | |
| 9 | def inference_chat(chat, model, api_url, token): |
| 10 | headers = { |
| 11 | "Content-Type": "application/json", |
| 12 | "Authorization": f"Bearer {token}" |
| 13 | } |
| 14 | |
| 15 | data = { |
| 16 | "model": model, |
| 17 | "messages": [], |
| 18 | "max_tokens": 2048, |
| 19 | 'temperature': 0.0, |
| 20 | "seed": 1234 |
| 21 | } |
| 22 | |
| 23 | for role, content in chat: |
| 24 | data["messages"].append({"role": role, "content": content}) |
| 25 | |
| 26 | while True: |
| 27 | try: |
| 28 | res = requests.post(api_url, headers=headers, json=data) |
| 29 | res_json = res.json() |
| 30 | res_content = res_json['choices'][0]['message']['content'] |
| 31 | except: |
| 32 | print("Network Error:") |
| 33 | try: |
| 34 | print(res.json()) |
| 35 | except: |
| 36 | print("Request Failed") |
| 37 | else: |
| 38 | break |
| 39 | |
| 40 | return res_content |