Send an async chat completion request via snowflake.cortex.complete and return parsed JSON output. See: https://docs.snowflake.com/en/user-guide/snowflake-cortex/cortex-llm-rest-api#complete-function Arguments: - prompt: The prompt to complete - schema: JSON schema of the desi
(
prompt: str,
schema: dict[str, Any],
model: str|None = None,
max_tokens: int = 4096,
temperature: float=0.0,
timeout: float=60.0)
| 78 | |
| 79 | |
| 80 | async def cortex_complete( |
| 81 | prompt: str, |
| 82 | schema: dict[str, Any], |
| 83 | model: str|None = None, |
| 84 | max_tokens: int = 4096, |
| 85 | temperature: float=0.0, |
| 86 | timeout: float=60.0) -> str: |
| 87 | """ |
| 88 | Send an async chat completion request via snowflake.cortex.complete and return parsed JSON output. |
| 89 | |
| 90 | See: https://docs.snowflake.com/en/user-guide/snowflake-cortex/cortex-llm-rest-api#complete-function |
| 91 | |
| 92 | Arguments: |
| 93 | - prompt: The prompt to complete |
| 94 | - schema: JSON schema of the desired response. |
| 95 | - model: The name of the model to use (if not specified, one will be chosen) |
| 96 | - max_tokens: A value between 1 and 4096 (inclusive) that controls the maximum number of tokens to output. Output is truncated after this number of tokens. |
| 97 | - temperature: A value from 0 to 1 (inclusive) that controls the randomness of the output of the language model by influencing which possible token is chosen at each step. |
| 98 | - timeout: Maximum time (in seconds) to wait for a response. |
| 99 | """ |
| 100 | if model is None: |
| 101 | model = "claude-3-5-sonnet" |
| 102 | response = await post( |
| 103 | "/api/v2/cortex/inference:complete", |
| 104 | { |
| 105 | "model": model, |
| 106 | "max_tokens": max_tokens, |
| 107 | "temperature": temperature, |
| 108 | "messages": [ |
| 109 | # The precise system prompt may need adjustment given a model. For example, a simpler prompt worked well for larger |
| 110 | # models but saying JSON twice helped for llama3.1-8b |
| 111 | # Alternatively, should explore using structured outputs support as outlined in: |
| 112 | # https://docs.snowflake.com/en/user-guide/snowflake-cortex/complete-structured-outputs |
| 113 | {"role": "system", "content": f"Provide a response in valid JSON that matches this JSON schema: {json.dumps(schema)}"}, |
| 114 | {"role": "user", "content": prompt}, |
| 115 | ], |
| 116 | "stream": False, |
| 117 | }, |
| 118 | timeout, |
| 119 | ) |
| 120 | try: |
| 121 | return SnowflakeProvider.clean_response(response.get("choices")[0].get("message").get("content").strip()) |
| 122 | except Exception as e: |
| 123 | logger.error(f"Error processing Snowflake response: {e}") |
| 124 | return {} |
| 125 | |
| 126 | async def post(api: str, request: dict, timeout: float) -> dict: |
| 127 | cfg = CONFIG.llm_endpoints.get("snowflake") |
no test coverage detected