From 67fc41f2e4fbd80d4b7705dfaa45fe7bddbcabd3 Mon Sep 17 00:00:00 2001 From: Pierre-Antoine Bannier Date: Fri, 11 Sep 2026 13:54:54 +0200 Subject: [PATCH] chore: remove EdgeTAM encoder profiler sam3_profile_edgetam_encode (437 lines) and its profile_edgetam example re-built the RepViT/FPN graphs stage by stage just to print timings. sam3_benchmark already reports per-model, per-backend latency. Drop the function, its public declaration, the example target, and the release packaging entry. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01M2qLmQ9ag49hT61V6i9qAq --- .github/workflows/release.yml | 4 +- examples/CMakeLists.txt | 3 - examples/profile_edgetam.cpp | 96 -------- sam3.cpp | 437 ---------------------------------- sam3.h | 23 -- 5 files changed, 2 insertions(+), 561 deletions(-) delete mode 100644 examples/profile_edgetam.cpp diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index efd8560..ead832e 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -90,7 +90,7 @@ jobs: cp sam3.h staging/${{ matrix.artifact }}/include/ cp -r ggml/include/* staging/${{ matrix.artifact }}/include/ - for bin in sam3_quantize sam3_benchmark sam3_profile_edgetam; do + for bin in sam3_quantize sam3_benchmark; do if [ -f "build/examples/${bin}" ]; then cp "build/examples/${bin}" staging/${{ matrix.artifact }}/bin/ fi @@ -121,7 +121,7 @@ jobs: Get-ChildItem -Path ggml/include -Filter "*.h" | Copy-Item -Destination staging/${{ matrix.artifact }}/include/ # Example binaries - foreach ($bin in @("sam3_quantize", "sam3_benchmark", "sam3_profile_edgetam")) { + foreach ($bin in @("sam3_quantize", "sam3_benchmark")) { $paths = @("build/examples/Release/${bin}.exe", "build/examples/${bin}.exe") foreach ($p in $paths) { if (Test-Path $p) { Copy-Item $p staging/${{ matrix.artifact }}/bin/; break } diff --git a/examples/CMakeLists.txt b/examples/CMakeLists.txt index 98c3c0e..88dcfe9 100644 --- a/examples/CMakeLists.txt +++ b/examples/CMakeLists.txt @@ -7,9 +7,6 @@ target_link_libraries(sam3_quantize PRIVATE ggml) add_executable(sam3_benchmark benchmark.cpp) target_link_libraries(sam3_benchmark PRIVATE sam3) -add_executable(sam3_profile_edgetam profile_edgetam.cpp) -target_link_libraries(sam3_profile_edgetam PRIVATE sam3) - # SDL2 + ImGui are optional — build only if SDL2 is found. find_package(SDL2 QUIET) diff --git a/examples/profile_edgetam.cpp b/examples/profile_edgetam.cpp deleted file mode 100644 index 6f326b9..0000000 --- a/examples/profile_edgetam.cpp +++ /dev/null @@ -1,96 +0,0 @@ -/** - * profile_edgetam — per-stage latency profiler for the EdgeTAM RepViT+FPN - * image encoder. - * - * Loads an EdgeTAM model and a test image, then profiles the forward pass - * broken into individually-timed sub-graphs (stem, stages 0-3, FPN neck). - * - * Usage: - * profile_edgetam [options] - * - * Options: - * --model EdgeTAM .ggml file (default: models/edgetam_f16.ggml) - * --image Test image (default: data/test_image.jpg) - * --n-threads CPU threads (default: 4) - * --n-warmup Warmup iterations (default: 2) - * --n-iter Timed iterations (default: 5) - * --cpu Force CPU backend - */ - -#include "sam3.h" - -#include -#include -#include -#include - -int main(int argc, char** argv) { - std::string model_path = "models/edgetam_f16.ggml"; - std::string image_path = "data/test_image.jpg"; - int n_threads = 4; - int n_warmup = 2; - int n_iter = 5; - bool use_gpu = true; - - for (int i = 1; i < argc; ++i) { - if (strcmp(argv[i], "--model") == 0 && i + 1 < argc) { - model_path = argv[++i]; - } else if (strcmp(argv[i], "--image") == 0 && i + 1 < argc) { - image_path = argv[++i]; - } else if (strcmp(argv[i], "--n-threads") == 0 && i + 1 < argc) { - n_threads = atoi(argv[++i]); - } else if (strcmp(argv[i], "--n-warmup") == 0 && i + 1 < argc) { - n_warmup = atoi(argv[++i]); - } else if (strcmp(argv[i], "--n-iter") == 0 && i + 1 < argc) { - n_iter = atoi(argv[++i]); - } else if (strcmp(argv[i], "--cpu") == 0) { - use_gpu = false; - } else { - fprintf(stderr, "Unknown option: %s\n", argv[i]); - fprintf(stderr, "Usage: %s [--model ] [--image ] " - "[--n-threads ] [--n-warmup ] [--n-iter ] [--cpu]\n", argv[0]); - return 1; - } - } - - fprintf(stderr, "Loading model: %s\n", model_path.c_str()); - fprintf(stderr, "Test image: %s\n", image_path.c_str()); - fprintf(stderr, "Backend: %s\n", use_gpu ? "Metal (GPU)" : "CPU"); - fprintf(stderr, "\n"); - - // Load model - sam3_params params; - params.model_path = model_path; - params.use_gpu = use_gpu; - params.n_threads = n_threads; - - auto model = sam3_load_model(params); - if (!model) { - fprintf(stderr, "ERROR: failed to load model\n"); - return 1; - } - - if (sam3_get_model_type(*model) != SAM3_MODEL_EDGETAM) { - fprintf(stderr, "ERROR: model is not EdgeTAM (model_type=%d)\n", - sam3_get_model_type(*model)); - return 1; - } - - // Load image - auto image = sam3_load_image(image_path); - if (image.data.empty()) { - fprintf(stderr, "ERROR: failed to load image '%s'\n", image_path.c_str()); - return 1; - } - fprintf(stderr, "Image: %dx%d (%d channels)\n\n", image.width, image.height, image.channels); - - // Profile - bool ok = sam3_profile_edgetam_encode(*model, image, n_threads, n_warmup, n_iter); - if (!ok) { - fprintf(stderr, "ERROR: profiling failed\n"); - return 1; - } - - fprintf(stderr, "\nDone.\n"); - return 0; -} diff --git a/sam3.cpp b/sam3.cpp index c7148a6..9c751ea 100644 --- a/sam3.cpp +++ b/sam3.cpp @@ -4978,443 +4978,6 @@ static bool edgetam_encode_image(sam3_state& state, return true; } -/***************************************************************************** -** EdgeTAM Profiling -** -** Profiles the RepViT backbone + FPN neck by building and timing separate -** sub-graphs for each pipeline stage. -*****************************************************************************/ - -// Helper: build a sub-graph, allocate, set inputs, compute, read outputs. -// Returns elapsed wall-clock time in microseconds. -struct sam3_profile_subgraph_result { - double elapsed_us = 0.0; - int n_nodes = 0; - bool ok = false; -}; - -// Print a per-op-type summary of a built graph. -static void sam3_profile_print_op_summary(struct ggml_cgraph* graph, const char* label) { - int n = ggml_graph_n_nodes(graph); - - // Count ops by name - std::map op_counts; - std::map op_elements; // total output elements - for (int i = 0; i < n; ++i) { - auto* node = ggml_graph_node(graph, i); - std::string name = ggml_op_name(node->op); - op_counts[name]++; - op_elements[name] += ggml_nelements(node); - } - - fprintf(stderr, "\n %-25s %5s %14s\n", "Op", "Count", "Out elements"); - fprintf(stderr, " %-25s %5s %14s\n", - "-------------------------", "-----", "--------------"); - for (auto& kv : op_counts) { - fprintf(stderr, " %-25s %5d %14lld\n", - kv.first.c_str(), kv.second, (long long)op_elements[kv.first]); - } - fprintf(stderr, " %-25s %5d\n", "TOTAL", n); -} - -// Helper: capture op counts from a graph. -static std::map sam3_profile_op_counts(struct ggml_cgraph* graph) { - std::map counts; - int n = ggml_graph_n_nodes(graph); - for (int i = 0; i < n; ++i) { - auto* node = ggml_graph_node(graph, i); - counts[ggml_op_name(node->op)]++; - } - return counts; -} - -// Helper: run a single sub-graph timing trial. -// Allocates, warms up, times n_iter runs, reads output tensors, then frees. -// output_tensors: list of tensors to read back BEFORE freeing the allocator. -// output_buffers: filled with read-back data (parallel to output_tensors). -static sam3_profile_subgraph_result sam3_profile_run_subgraph( - ggml_backend_t backend, - struct ggml_cgraph* graph, - struct ggml_tensor* input_tensor, - const float* input_data, - size_t input_bytes, - int n_threads, - int n_warmup, - int n_iter, - const std::vector& output_tensors = {}, - std::vector>* output_buffers = nullptr) { - sam3_profile_subgraph_result res; - res.n_nodes = ggml_graph_n_nodes(graph); - - auto* galloc = ggml_gallocr_new(ggml_backend_get_default_buffer_type(backend)); - if (!ggml_gallocr_reserve(galloc, graph) || !ggml_gallocr_alloc_graph(galloc, graph)) { - fprintf(stderr, "%s: graph alloc failed\n", __func__); - ggml_gallocr_free(galloc); - return res; - } - - // Warm up - for (int i = 0; i < n_warmup; ++i) { - ggml_backend_tensor_set(input_tensor, input_data, 0, input_bytes); - sam3_graph_compute(backend, graph, n_threads); - } - - // Timed runs - double total_us = 0.0; - for (int i = 0; i < n_iter; ++i) { - ggml_backend_tensor_set(input_tensor, input_data, 0, input_bytes); - auto t0 = std::chrono::high_resolution_clock::now(); - sam3_graph_compute(backend, graph, n_threads); - auto t1 = std::chrono::high_resolution_clock::now(); - total_us += std::chrono::duration(t1 - t0).count(); - } - - res.elapsed_us = total_us / n_iter; - res.ok = true; - - // Read back output tensors BEFORE freeing the allocator - if (output_buffers) { - output_buffers->resize(output_tensors.size()); - for (size_t i = 0; i < output_tensors.size(); ++i) { - int64_t n = ggml_nelements(output_tensors[i]); - (*output_buffers)[i].resize(n); - ggml_backend_tensor_get(output_tensors[i], (*output_buffers)[i].data(), - 0, n * sizeof(float)); - } - } - - ggml_gallocr_free(galloc); - return res; -} - -bool sam3_profile_edgetam_encode(const sam3_model& model, - const sam3_image& image, - int n_threads, - int n_warmup, - int n_iter) { - const auto& hp = model.hparams; - if (!hp.is_edgetam()) { - fprintf(stderr, "%s: model is not EdgeTAM\n", __func__); - return false; - } - - const int img_size = hp.img_size; - fprintf(stderr, "%s: profiling EdgeTAM RepViT+FPN on %dx%d image\n", - __func__, img_size, img_size); - fprintf(stderr, " warmup=%d, iterations=%d, threads=%d\n", - n_warmup, n_iter, n_threads); - fprintf(stderr, " backend: %s\n", ggml_backend_name(model.backend)); - - // Stage config - fprintf(stderr, " stages: [%d, %d, %d, %d] blocks, channels: [%d, %d, %d, %d]\n", - hp.repvit_stages[0], hp.repvit_stages[1], - hp.repvit_stages[2], hp.repvit_stages[3], - hp.repvit_channels[0], hp.repvit_channels[1], - hp.repvit_channels[2], hp.repvit_channels[3]); - - // ── Preprocess image ──────────────────────────────────────────────── - auto img_data = sam2_preprocess_image(image, img_size); - - // ════════════════════════════════════════════════════════════════════ - // PART 1: Full graph — op summary + total timing - // ════════════════════════════════════════════════════════════════════ - fprintf(stderr, "\n=== FULL GRAPH (RepViT + FPN) ===\n"); - { - const size_t buf_size = ggml_tensor_overhead() * 16384 + ggml_graph_overhead() * 2; - struct ggml_init_params gp = {buf_size, nullptr, true}; - auto* ctx0 = ggml_init(gp); - - auto* inp = ggml_new_tensor_4d(ctx0, GGML_TYPE_F32, img_size, img_size, 3, 1); - ggml_set_name(inp, "input_image"); - ggml_set_input(inp); - - struct ggml_tensor* stage_outs[4] = {}; - edgetam_build_repvit_graph(ctx0, inp, model, stage_outs); - - struct ggml_tensor* fpn_outs[4] = {}; - edgetam_build_fpn_neck_graph(ctx0, stage_outs, model, fpn_outs); - - int n_fpn = 4 - hp.scalp; - auto* graph = ggml_new_graph_custom(ctx0, 32768, false); - for (int i = 0; i < n_fpn; ++i) { - char name[64]; - snprintf(name, sizeof(name), "fpn_out_%d", i); - ggml_set_name(fpn_outs[i], name); - ggml_set_output(fpn_outs[i]); - ggml_build_forward_expand(graph, fpn_outs[i]); - } - - sam3_profile_print_op_summary(graph, "Full RepViT+FPN"); - - // Print output shapes - fprintf(stderr, "\n Output shapes:\n"); - for (int i = 0; i < n_fpn; ++i) { - fprintf(stderr, " fpn_out[%d]: [%lld, %lld, %lld, %lld]\n", i, - (long long)fpn_outs[i]->ne[0], (long long)fpn_outs[i]->ne[1], - (long long)fpn_outs[i]->ne[2], (long long)fpn_outs[i]->ne[3]); - } - - // Time it (no outputs needed from full graph) - auto r = sam3_profile_run_subgraph(model.backend, graph, inp, - img_data.data(), img_data.size() * sizeof(float), - n_threads, n_warmup, n_iter); - if (r.ok) { - fprintf(stderr, "\n Total: %.2f ms (%d nodes)\n", r.elapsed_us / 1000.0, r.n_nodes); - } - - ggml_free(ctx0); - } - - // ════════════════════════════════════════════════════════════════════ - // PART 2: Per-stage sub-graphs — build and time each separately - // ════════════════════════════════════════════════════════════════════ - - struct StageResult { - std::string name; - double ms; - int n_nodes; - int64_t out_shape[4]; - std::map op_counts; - }; - std::vector results; - - // Current feature data flowing between stages (CPU-side buffer, graph isolation) - std::vector cur_data = img_data; - int64_t cur_ne[4] = {img_size, img_size, 3, 1}; // [W, H, C, B] - - // ── STEM ──────────────────────────────────────────────────────────── - { - const size_t buf_size = ggml_tensor_overhead() * 4096 + ggml_graph_overhead(); - struct ggml_init_params gp = {buf_size, nullptr, true}; - auto* ctx0 = ggml_init(gp); - - auto* inp = ggml_new_tensor_4d(ctx0, GGML_TYPE_F32, cur_ne[0], cur_ne[1], cur_ne[2], cur_ne[3]); - ggml_set_name(inp, "stem_in"); - ggml_set_input(inp); - - const auto& repvit = model.repvit; - auto* x = ggml_conv_2d(ctx0, repvit.stem_conv1_w, inp, 2, 2, 1, 1, 1, 1); - x = edgetam_conv2d_bias(ctx0, x, repvit.stem_conv1_b); - x = ggml_gelu(ctx0, x); - x = ggml_conv_2d(ctx0, repvit.stem_conv2_w, x, 2, 2, 1, 1, 1, 1); - x = edgetam_conv2d_bias(ctx0, x, repvit.stem_conv2_b); - - ggml_set_name(x, "stem_out"); - ggml_set_output(x); - - auto* graph = ggml_new_graph_custom(ctx0, 4096, false); - ggml_build_forward_expand(graph, x); - - std::vector> out_bufs; - auto r = sam3_profile_run_subgraph(model.backend, graph, inp, - cur_data.data(), cur_data.size() * sizeof(float), - n_threads, n_warmup, n_iter, - {x}, &out_bufs); - - StageResult sr; - sr.name = "Stem (2 convs)"; - sr.ms = r.ok ? r.elapsed_us / 1000.0 : -1.0; - sr.n_nodes = r.n_nodes; - for (int i = 0; i < 4; ++i) sr.out_shape[i] = x->ne[i]; - sr.op_counts = sam3_profile_op_counts(graph); - results.push_back(sr); - - // Use read-back output for next stage - if (r.ok) { - cur_data = std::move(out_bufs[0]); - for (int i = 0; i < 4; ++i) cur_ne[i] = x->ne[i]; - } - - ggml_free(ctx0); - } - - // ── STAGES 0-3 ────────────────────────────────────────────────────── - // Store stage outputs for FPN (in [W, H, C, 1] layout — no permute needed) - std::vector stage_data[4]; - int64_t stage_ne[4][4] = {}; - - for (int s = 0; s < hp.repvit_num_stages; ++s) { - const size_t buf_size = ggml_tensor_overhead() * 8192 + ggml_graph_overhead(); - struct ggml_init_params gp = {buf_size, nullptr, true}; - auto* ctx0 = ggml_init(gp); - - auto* inp = ggml_new_tensor_4d(ctx0, GGML_TYPE_F32, cur_ne[0], cur_ne[1], cur_ne[2], cur_ne[3]); - ggml_set_name(inp, "stage_in"); - ggml_set_input(inp); - - auto* x = inp; - auto& stage = model.repvit.stages[s]; - - // Downsample at start of stages 1, 2, 3 - if (stage.has_downsample) { - x = edgetam_repvit_downsample_forward(ctx0, x, stage.downsample); - } - - // Process all blocks - for (int b = 0; b < (int)stage.blocks.size(); ++b) { - x = edgetam_repvit_block_forward(ctx0, x, stage.blocks[b]); - } - - // Output stays in [W, H, C, 1] — no permute - char oname[64]; - snprintf(oname, sizeof(oname), "stage_%d_out", s); - ggml_set_name(x, oname); - ggml_set_output(x); - - auto* graph = ggml_new_graph_custom(ctx0, 8192, false); - ggml_build_forward_expand(graph, x); - - std::vector> out_bufs; - auto r = sam3_profile_run_subgraph(model.backend, graph, inp, - cur_data.data(), cur_data.size() * sizeof(float), - n_threads, n_warmup, n_iter, - {x}, &out_bufs); - - char sname[64]; - snprintf(sname, sizeof(sname), "Stage %d (%d blk%s%s)", s, - (int)stage.blocks.size(), - stage.blocks.size() != 1 ? "s" : "", - stage.has_downsample ? " + ds" : ""); - StageResult sr; - sr.name = sname; - sr.ms = r.ok ? r.elapsed_us / 1000.0 : -1.0; - sr.n_nodes = r.n_nodes; - for (int i = 0; i < 4; ++i) sr.out_shape[i] = x->ne[i]; - sr.op_counts = sam3_profile_op_counts(graph); - results.push_back(sr); - - if (r.ok) { - cur_data = std::move(out_bufs[0]); - for (int i = 0; i < 4; ++i) cur_ne[i] = x->ne[i]; - - // Stage data for FPN (same [W, H, C, 1] layout) - stage_data[s] = cur_data; - for (int i = 0; i < 4; ++i) stage_ne[s][i] = x->ne[i]; - } - - ggml_free(ctx0); - } - - // ── FPN NECK ──────────────────────────────────────────────────────── - { - const size_t buf_size = ggml_tensor_overhead() * 8192 + ggml_graph_overhead(); - struct ggml_init_params gp = {buf_size, nullptr, true}; - auto* ctx0 = ggml_init(gp); - - // Create input tensors for each stage (graph isolation) - struct ggml_tensor* stage_inputs[4] = {}; - for (int i = 0; i < 4; ++i) { - stage_inputs[i] = ggml_new_tensor_4d(ctx0, GGML_TYPE_F32, - stage_ne[i][0], stage_ne[i][1], - stage_ne[i][2], stage_ne[i][3]); - char name[64]; - snprintf(name, sizeof(name), "fpn_stage_in_%d", i); - ggml_set_name(stage_inputs[i], name); - ggml_set_input(stage_inputs[i]); - } - - struct ggml_tensor* fpn_outs[4] = {}; - edgetam_build_fpn_neck_graph(ctx0, stage_inputs, model, fpn_outs); - - int n_fpn = 4 - hp.scalp; - auto* graph = ggml_new_graph_custom(ctx0, 8192, false); - for (int i = 0; i < n_fpn; ++i) { - char name[64]; - snprintf(name, sizeof(name), "fpn_out_%d", i); - ggml_set_name(fpn_outs[i], name); - ggml_set_output(fpn_outs[i]); - ggml_build_forward_expand(graph, fpn_outs[i]); - } - - // Allocate and set inputs - auto* galloc = ggml_gallocr_new(ggml_backend_get_default_buffer_type(model.backend)); - if (!ggml_gallocr_reserve(galloc, graph) || !ggml_gallocr_alloc_graph(galloc, graph)) { - fprintf(stderr, "%s: FPN graph alloc failed\n", __func__); - ggml_gallocr_free(galloc); - ggml_free(ctx0); - return false; - } - - // Warm up - for (int i = 0; i < n_warmup; ++i) { - for (int j = 0; j < 4; ++j) { - ggml_backend_tensor_set(stage_inputs[j], stage_data[j].data(), 0, - stage_data[j].size() * sizeof(float)); - } - sam3_graph_compute(model.backend, graph, n_threads); - } - - // Timed runs - double total_us = 0.0; - for (int i = 0; i < n_iter; ++i) { - for (int j = 0; j < 4; ++j) { - ggml_backend_tensor_set(stage_inputs[j], stage_data[j].data(), 0, - stage_data[j].size() * sizeof(float)); - } - auto t0 = std::chrono::high_resolution_clock::now(); - sam3_graph_compute(model.backend, graph, n_threads); - auto t1 = std::chrono::high_resolution_clock::now(); - total_us += std::chrono::duration(t1 - t0).count(); - } - double fpn_ms = total_us / n_iter / 1000.0; - - StageResult sr; - sr.name = "FPN neck"; - sr.ms = fpn_ms; - sr.n_nodes = ggml_graph_n_nodes(graph); - sr.out_shape[0] = sr.out_shape[1] = sr.out_shape[2] = sr.out_shape[3] = 0; - if (fpn_outs[0]) { - for (int i = 0; i < 4; ++i) sr.out_shape[i] = fpn_outs[0]->ne[i]; - } - sr.op_counts = sam3_profile_op_counts(graph); - results.push_back(sr); - - ggml_gallocr_free(galloc); - ggml_free(ctx0); - } - - // ════════════════════════════════════════════════════════════════════ - // PART 3: Summary table - // ════════════════════════════════════════════════════════════════════ - fprintf(stderr, "\n=== TIMING BREAKDOWN ===\n\n"); - - // First pass: compute total for percentage column - double total_ms = 0.0; - for (auto& sr : results) { - if (sr.ms > 0) total_ms += sr.ms; - } - - fprintf(stderr, " %-28s %8s %5s %5s %s\n", - "Stage", "Time(ms)", "%%", "Nodes", "Output Shape"); - fprintf(stderr, " %-28s %8s %5s %5s %s\n", - "----------------------------", "--------", "-----", "-----", - "--------------------"); - - for (auto& sr : results) { - char shape[64]; - snprintf(shape, sizeof(shape), "[%lld, %lld, %lld, %lld]", - (long long)sr.out_shape[0], (long long)sr.out_shape[1], - (long long)sr.out_shape[2], (long long)sr.out_shape[3]); - double pct = (total_ms > 0 && sr.ms > 0) ? 100.0 * sr.ms / total_ms : 0.0; - fprintf(stderr, " %-28s %8.2f %4.1f%% %5d %s\n", - sr.name.c_str(), sr.ms, pct, sr.n_nodes, shape); - } - fprintf(stderr, " %-28s %8.2f\n", "SUM (per-stage)", total_ms); - - // Per-op type breakdown for each stage - fprintf(stderr, "\n=== PER-STAGE OP COUNTS ===\n"); - for (auto& sr : results) { - fprintf(stderr, "\n %s (%.2f ms, %d nodes):\n", sr.name.c_str(), sr.ms, sr.n_nodes); - fprintf(stderr, " %-18s %5s\n", "Op", "Count"); - fprintf(stderr, " %-18s %5s\n", "------------------", "-----"); - for (auto& kv : sr.op_counts) { - fprintf(stderr, " %-18s %5d\n", kv.first.c_str(), kv.second); - } - } - - return true; -} - /***************************************************************************** ** EdgeTAM Spatial Perceiver ** diff --git a/sam3.h b/sam3.h index 2f5c374..705781d 100644 --- a/sam3.h +++ b/sam3.h @@ -275,26 +275,3 @@ sam3_image sam3_load_image(const std::string & path); bool sam3_save_mask(const sam3_mask & mask, const std::string & path); sam3_image sam3_decode_video_frame(const std::string & video_path, int frame_index); sam3_video_info sam3_get_video_info(const std::string & video_path); - -/* -** ── Profiling ─────────────────────────────────────────────────────────── -*/ - -/* -** Profile the EdgeTAM image encoder (RepViT backbone + FPN neck). -** -** Runs the full graph once for a total timing and op summary, then builds -** and times each stage as a separate sub-graph to produce a per-stage -** latency breakdown: -** - Stem (2 convolutions) -** - Stage 0..3 (downsample + RepViT blocks) -** - FPN neck (lateral convolutions + top-down fusion) -** -** n_warmup iterations are run before n_iter timed iterations. -** Results are printed to stderr. -*/ -bool sam3_profile_edgetam_encode(const sam3_model & model, - const sam3_image & image, - int n_threads = 4, - int n_warmup = 2, - int n_iter = 5);