| 176 | |
| 177 | |
| 178 | def get_eval(content: str, max_tokens: int): |
| 179 | while True: |
| 180 | try: |
| 181 | response = openai.ChatCompletion.create( |
| 182 | model='gpt-4-0314', |
| 183 | messages=[{ |
| 184 | 'role': 'system', |
| 185 | 'content': 'You are a helpful and precise assistant for checking the quality of the answer.' |
| 186 | }, { |
| 187 | 'role': 'user', |
| 188 | 'content': content, |
| 189 | }], |
| 190 | temperature=0.2, # TODO: figure out which temperature is best for evaluation |
| 191 | max_tokens=max_tokens, |
| 192 | ) |
| 193 | break |
| 194 | except openai.error.RateLimitError: |
| 195 | pass |
| 196 | except Exception as e: |
| 197 | print(e) |
| 198 | time.sleep(NUM_SECONDS_TO_SLEEP) |
| 199 | |
| 200 | return response['choices'][0]['message']['content'] |
| 201 | |
| 202 | |
| 203 | def parse_score(review): |