Calculate optimal batch size based on task data size and annotation result size
(self)
| 1262 | return row[1] |
| 1263 | |
| 1264 | def get_task_batch_size(self): |
| 1265 | """Calculate optimal batch size based on task data size and annotation result size""" |
| 1266 | # For SQLite, use default MAX_TASK_BATCH_SIZE |
| 1267 | if settings.DJANGO_DB == settings.DJANGO_DB_SQLITE: |
| 1268 | return settings.MAX_TASK_BATCH_SIZE |
| 1269 | |
| 1270 | # Get maximum task data size using the optimized index |
| 1271 | max_task_size = 0 |
| 1272 | with connection.cursor() as cursor: |
| 1273 | cursor.execute( |
| 1274 | """ |
| 1275 | SELECT id, |
| 1276 | octet_length(data::text) AS bytes |
| 1277 | FROM task |
| 1278 | WHERE project_id = %s |
| 1279 | ORDER BY octet_length(data::text) DESC |
| 1280 | LIMIT 1 |
| 1281 | """, |
| 1282 | [self.id], |
| 1283 | ) |
| 1284 | |
| 1285 | row = cursor.fetchone() |
| 1286 | if row and row[1]: |
| 1287 | max_task_size = row[1] |
| 1288 | |
| 1289 | # Get maximum annotation result size using the new optimized index |
| 1290 | max_annotation_size = self.get_max_annotation_result_size() |
| 1291 | |
| 1292 | # Use the larger of the two sizes for batch calculation |
| 1293 | max_data_size = max(max_task_size, max_annotation_size) |
| 1294 | |
| 1295 | if max_data_size == 0: |
| 1296 | return settings.MAX_TASK_BATCH_SIZE |
| 1297 | |
| 1298 | batch_size = settings.TASK_DATA_PER_BATCH // max_data_size |
| 1299 | |
| 1300 | if batch_size > settings.MAX_TASK_BATCH_SIZE: |
| 1301 | batch_size = settings.MAX_TASK_BATCH_SIZE |
| 1302 | elif batch_size < 1: |
| 1303 | batch_size = 1 |
| 1304 | |
| 1305 | logger.info( |
| 1306 | f'Project {self.id}: max task size {max_task_size} bytes, ' |
| 1307 | f'max annotation size {max_annotation_size} bytes, ' |
| 1308 | f'calculated batch size {batch_size}' |
| 1309 | ) |
| 1310 | return batch_size |
| 1311 | |
| 1312 | def __str__(self): |
| 1313 | return f'{self.title} (id={self.id})' or _('Business number %d') % self.pk |
no test coverage detected