MCPcopy Create free account
hub / github.com/NanmiCoder/MediaCrawler / get_data_stats

Function get_data_stats

api/routers/data.py:191–230  ·  view source on GitHub ↗

Get data statistics

()

Source from the content-addressed store, hash-verified

189
190@router.get("/stats")
191async def get_data_stats():
192 """Get data statistics"""
193 if not DATA_DIR.exists():
194 return {"total_files": 0, "total_size": 0, "by_platform": {}, "by_type": {}}
195
196 stats = {
197 "total_files": 0,
198 "total_size": 0,
199 "by_platform": {},
200 "by_type": {}
201 }
202
203 supported_extensions = {".json", ".csv", ".xlsx", ".xls"}
204
205 for root, dirs, filenames in os.walk(DATA_DIR):
206 root_path = Path(root)
207 for filename in filenames:
208 file_path = root_path / filename
209 if file_path.suffix.lower() not in supported_extensions:
210 continue
211
212 try:
213 stat = file_path.stat()
214 stats["total_files"] += 1
215 stats["total_size"] += stat.st_size
216
217 # Statistics by type
218 file_type = file_path.suffix[1:].lower()
219 stats["by_type"][file_type] = stats["by_type"].get(file_type, 0) + 1
220
221 # Statistics by platform (inferred from path)
222 rel_path = str(file_path.relative_to(DATA_DIR))
223 for platform in ["xhs", "dy", "ks", "bili", "wb", "tieba", "zhihu"]:
224 if platform in rel_path.lower():
225 stats["by_platform"][platform] = stats["by_platform"].get(platform, 0) + 1
226 break
227 except Exception:
228 continue
229
230 return stats

Callers

nothing calls this directly

Calls 1

getMethod · 0.45

Tested by

no test coverage detected