MCPcopy Create free account
hub / github.com/nlweb-ai/NLWeb / cortex_complete

Function cortex_complete

AskAgent/python/llm_providers/snowflake.py:80–124  ·  view source on GitHub ↗

Send an async chat completion request via snowflake.cortex.complete and return parsed JSON output. See: https://docs.snowflake.com/en/user-guide/snowflake-cortex/cortex-llm-rest-api#complete-function Arguments: - prompt: The prompt to complete - schema: JSON schema of the desi

(
        prompt: str,
        schema: dict[str, Any],
        model: str|None = None,
        max_tokens: int = 4096,
        temperature: float=0.0,
        timeout: float=60.0)

Source from the content-addressed store, hash-verified

78
79
80async def cortex_complete(
81 prompt: str,
82 schema: dict[str, Any],
83 model: str|None = None,
84 max_tokens: int = 4096,
85 temperature: float=0.0,
86 timeout: float=60.0) -> str:
87 """
88 Send an async chat completion request via snowflake.cortex.complete and return parsed JSON output.
89
90 See: https://docs.snowflake.com/en/user-guide/snowflake-cortex/cortex-llm-rest-api#complete-function
91
92 Arguments:
93 - prompt: The prompt to complete
94 - schema: JSON schema of the desired response.
95 - model: The name of the model to use (if not specified, one will be chosen)
96 - max_tokens: A value between 1 and 4096 (inclusive) that controls the maximum number of tokens to output. Output is truncated after this number of tokens.
97 - temperature: A value from 0 to 1 (inclusive) that controls the randomness of the output of the language model by influencing which possible token is chosen at each step.
98 - timeout: Maximum time (in seconds) to wait for a response.
99 """
100 if model is None:
101 model = "claude-3-5-sonnet"
102 response = await post(
103 "/api/v2/cortex/inference:complete",
104 {
105 "model": model,
106 "max_tokens": max_tokens,
107 "temperature": temperature,
108 "messages": [
109 # The precise system prompt may need adjustment given a model. For example, a simpler prompt worked well for larger
110 # models but saying JSON twice helped for llama3.1-8b
111 # Alternatively, should explore using structured outputs support as outlined in:
112 # https://docs.snowflake.com/en/user-guide/snowflake-cortex/complete-structured-outputs
113 {"role": "system", "content": f"Provide a response in valid JSON that matches this JSON schema: {json.dumps(schema)}"},
114 {"role": "user", "content": prompt},
115 ],
116 "stream": False,
117 },
118 timeout,
119 )
120 try:
121 return SnowflakeProvider.clean_response(response.get("choices")[0].get("message").get("content").strip())
122 except Exception as e:
123 logger.error(f"Error processing Snowflake response: {e}")
124 return {}
125
126async def post(api: str, request: dict, timeout: float) -> dict:
127 cfg = CONFIG.llm_endpoints.get("snowflake")

Callers 1

get_completionMethod · 0.85

Calls 4

postFunction · 0.85
getMethod · 0.80
clean_responseMethod · 0.45
errorMethod · 0.45

Tested by

no test coverage detected