MCPcopy Create free account
hub / github.com/NanmiCoder/MediaCrawler / get_file_content

Function get_file_content

api/routers/data.py:99–163  ·  view source on GitHub ↗

Get file content or preview

(file_path: str, preview: bool = True, limit: int = 100)

Source from the content-addressed store, hash-verified

97
98@router.get("/files/{file_path:path}")
99async def get_file_content(file_path: str, preview: bool = True, limit: int = 100):
100 """Get file content or preview"""
101 full_path = DATA_DIR / file_path
102
103 if not full_path.exists():
104 raise HTTPException(status_code=404, detail="File not found")
105
106 if not full_path.is_file():
107 raise HTTPException(status_code=400, detail="Not a file")
108
109 # Security check: ensure within DATA_DIR
110 try:
111 full_path.resolve().relative_to(DATA_DIR.resolve())
112 except ValueError:
113 raise HTTPException(status_code=403, detail="Access denied")
114
115 if preview:
116 # Return preview data
117 try:
118 if full_path.suffix == ".json":
119 with open(full_path, "r", encoding="utf-8") as f:
120 data = json.load(f)
121 if isinstance(data, list):
122 return {"data": data[:limit], "total": len(data)}
123 return {"data": data, "total": 1}
124 elif full_path.suffix == ".csv":
125 import csv
126 with open(full_path, "r", encoding="utf-8") as f:
127 reader = csv.DictReader(f)
128 rows = []
129 for i, row in enumerate(reader):
130 if i >= limit:
131 break
132 rows.append(row)
133 # Re-read to get total count
134 f.seek(0)
135 total = sum(1 for _ in f) - 1
136 return {"data": rows, "total": total}
137 elif full_path.suffix.lower() in (".xlsx", ".xls"):
138 import pandas as pd
139 # Read first limit rows
140 df = pd.read_excel(full_path, nrows=limit)
141 # Get total row count (only read first column to save memory)
142 df_count = pd.read_excel(full_path, usecols=[0])
143 total = len(df_count)
144 # Convert to list of dictionaries, handle NaN values
145 rows = df.where(pd.notnull(df), None).to_dict(orient='records')
146 return {
147 "data": rows,
148 "total": total,
149 "columns": list(df.columns)
150 }
151 else:
152 raise HTTPException(status_code=400, detail="Unsupported file type for preview")
153 except json.JSONDecodeError:
154 raise HTTPException(status_code=400, detail="Invalid JSON file")
155 except Exception as e:
156 raise HTTPException(status_code=500, detail=str(e))

Callers

nothing calls this directly

Calls 1

sumFunction · 0.85

Tested by

no test coverage detected