MCPcopy Create free account
hub / github.com/RUC-NLPIR/WebThinker / process_single_sequence

Function process_single_sequence

scripts/run_naive_rag_report.py:163–268  ·  view source on GitHub ↗

Process a single question through RAG pipeline

(
    question: str,
    client: AsyncOpenAI,
    semaphore: asyncio.Semaphore,
    args: argparse.Namespace,
    search_cache: Dict,
    url_cache: Dict,
)

Source from the content-addressed store, hash-verified

161
162
163async def process_single_sequence(
164 question: str,
165 client: AsyncOpenAI,
166 semaphore: asyncio.Semaphore,
167 args: argparse.Namespace,
168 search_cache: Dict,
169 url_cache: Dict,
170) -> Dict:
171 """Process a single question through RAG pipeline"""
172
173 results = {}
174 # Search for relevant documents
175 try:
176 if question in search_cache:
177 results = search_cache[question]
178 else:
179 if args.search_engine == "bing":
180 if not args.bing_subscription_key:
181 print(f"Error: Bing search engine is selected, but BING_SUBSCRIPTION_KEY is not provided for question: {question}")
182 results = {}
183 else:
184 results = await bing_web_search_async(question, args.bing_subscription_key, args.bing_endpoint)
185 elif args.search_engine == "serper":
186 if not args.serper_api_key:
187 print(f"Error: Serper search engine is selected, but SERPER_API_KEY is not provided for question: {question}")
188 results = {}
189 else:
190 results = await google_serper_search_async(question, args.serper_api_key)
191 else:
192 print(f"Error: Unknown search engine: {args.search_engine}")
193 results = {}
194
195 if results: # Only cache if results are not empty (i.e., search was successful)
196 search_cache[question] = results
197
198 except Exception as e:
199 print(f"Error during search for '{question}' using {args.search_engine}: {e}")
200 results = {}
201
202 # Extract and process relevant documents
203 relevant_info = []
204 if results:
205 if args.search_engine == "bing":
206 relevant_info = extract_relevant_info(results)[:args.top_k]
207 elif args.search_engine == "serper":
208 relevant_info = extract_relevant_info_serper(results)[:args.top_k]
209
210 # Fetch page content for each result
211 documents = []
212 for idx, doc_info in enumerate(relevant_info):
213 url = doc_info['url']
214 if url not in url_cache:
215 try:
216 contents = await fetch_page_content_async(
217 [url],
218 use_jina=args.use_jina,
219 jina_api_key=args.jina_api_key,
220 keep_links=args.keep_links

Callers 1

main_asyncFunction · 0.70

Calls 9

bing_web_search_asyncFunction · 0.90
extract_relevant_infoFunction · 0.90
fetch_page_content_asyncFunction · 0.90
generate_responseFunction · 0.70
extract_markdown_contentFunction · 0.70

Tested by

no test coverage detected