MCPcopy Create free account
hub / github.com/PostHog/duckgres / seedBundledExtensions

Function seedBundledExtensions

server/server.go:1067–1125  ·  view source on GitHub ↗
(srcRoot, dstRoot string)

Source from the content-addressed store, hash-verified

1065 if err := LoadExtensions(db, cfg.Extensions); err != nil {
1066 slog.Warn("Failed to load some extensions.", "user", username, "error", err)
1067 }
1068
1069 // Enable query profiling so per-query operator timing can be extracted
1070 // and attached to OTEL trace spans. Standard mode adds sub-1% overhead
1071 // (just clock_gettime per operator boundary).
1072 // Output goes to a fixed temp file; in K8s mode the worker reads it
1073 // after each query and sends it to the control plane via gRPC trailer.
1074 //
1075 // These SETs are session-scoped in DuckDB and cannot be set globally,
1076 // so they only apply to whichever connection runs them now. In cluster
1077 // mode (sharedWarmMode) the worker evicts connections between sessions
1078 // (see duckdbservice.evictConnFromPool), so per-session re-application
1079 // is required — see ApplyProfilingSettings.
1080 for _, stmt := range ProfilingSetupSQL(ProfilingOutputPath) {
1081 if _, err := db.Exec(stmt); err != nil {
1082 slog.Warn("Failed to apply DuckDB profiling setting.", "stmt", stmt, "error", err)
1083 }
1084 }
1085
1086 // Configure cache_httpfs cache directory if the extension is loaded.
1087 // cache_httpfs wraps httpfs with a local disk cache, avoiding repeated S3/HTTP downloads.
1088 if hasCacheHTTPFS(cfg.Extensions) {
1089 cacheDir := filepath.Join(cfg.DataDir, "cache")
1090 if err := os.MkdirAll(cacheDir, 0750); err != nil {
1091 slog.Warn("Failed to create cache_httpfs cache directory.", "cache_directory", cacheDir, "error", err)
1092 } else if _, err := db.Exec(fmt.Sprintf("SET cache_httpfs_cache_directory = '%s/'", cacheDir)); err != nil {
1093 // NOTE: cache directory path comes from trusted server config (DataDir), not user input.
1094 slog.Warn("Failed to set cache_httpfs cache directory.", "cache_directory", cacheDir, "error", err)
1095 } else {
1096 slog.Debug("Set cache_httpfs cache directory.", "cache_directory", cacheDir)
1097 }
1098 }
1099
1100 return nil
1101}
1102
1103// applyHTTPFSRetryBudget widens httpfs's retry budget for transient S3 throttling.
1104//
1105// S3 answers a sustained request burst with HTTP 503 SlowDown ("please reduce
1106// your request rate"). That's pure throttling — transient and safe to retry —
1107// and httpfs DOES retry 503 (RunRequestWithRetry → ShouldRetry in duckdb's
1108// http_util.cpp includes ServiceUnavailable_503), but its defaults are tiny:
1109// http_retries=3, http_retry_wait_ms=100, http_retry_backoff=4 → a cumulative
1110// backoff of only ~0.5s. A throttle that outlasts ~0.5s exhausts all 3 attempts
1111// and httpfs throws fatally; the error then surfaces all the way out as a fatal
1112// XX000 (no upper layer reclassifies an HTTP 503 as retryable — classifyErrorCode
1113// falls through to XX000), failing e.g. a DuckLake events DELETE that range-GETs
1114// parquet data files.
1115//
1116// Raising the budget to retries=10, wait=500ms, backoff=2 widens the cumulative
1117// backoff substantially. DuckDB's retry loop sleeps
1118// retry_wait_ms * retry_backoff^(tries-2) ms before each retry from the 2nd
1119// onward (the 1st is immediate): waits 0, 500, 1k, 2k, 4k, 8k, 16k, 32k, 64k,
1120// 128k ms — ~255s (~4.3min) cumulative, ~128s on the final single wait. That rides
1121// out a typical SlowDown burst (which clears in seconds to low minutes); the
1122// tradeoff is a single throttled request can block ~2min on its last retry
1123// (~4.3min worst case) — acceptable for a backfill, where retrying beats failing
1124// the partition, and well under the 55m worker drain timeout / 3600s pod grace.

Calls 2

copyFileFunction · 0.85