You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

TensorFlow C++中多GPU并行运行不同模型预测的技术问询

Hey there! I’ve worked with TensorFlow C++ on multi-GPU setups before, so let’s walk through exactly how to get your two models running on separate GTX 1080 Ti GPUs—mirroring that handy with tf.device('/gpu:0') behavior you know from Python. Given your Debian 9, CUDA 9.1, cuDNN 7.1 setup, here’s a solid implementation plan:

Core Implementation Approach

The key here is to isolate each model’s graph and session to its target GPU. TensorFlow C++ lets you either bind an entire session to a GPU, or scope individual graph nodes to a specific device (just like the Python context manager). We’ll cover both methods below.

Step-by-Step Code & Examples

First, let’s tie into your existing code snippet where you’re parsing a GPU ID (int GPUID = std::stoi(pa...)). We’ll expand that to handle two GPUs.

Method 1: Bind Entire Session to a Specific GPU

This is the simplest approach if you want an entire model’s computation to run on one GPU. We’ll create separate sessions, each restricted to a single GPU:

#include <tensorflow/core/public/session.h>
#include <tensorflow/core/graph/default_device.h>
#include <tensorflow/core/framework/graph.pb.h>
#include <tensorflow/core/util/command_line_flags.h>

// Helper function to create a session locked to a specific GPU
std::unique_ptr<tensorflow::Session> create_gpu_session(int gpu_id) {
    tensorflow::SessionOptions options;
    tensorflow::ConfigProto& config = options.config;

    // Restrict this session to only see the target GPU
    config.set_allow_soft_placement(false); // No fallback to CPU if GPU ops fail
    auto* gpu_opts = config.mutable_gpu_options();
    gpu_opts->set_visible_device_list(std::to_string(gpu_id));
    // Optional: Limit memory usage to avoid OOM (adjust based on your models)
    gpu_opts->set_per_process_gpu_memory_fraction(0.45);

    tensorflow::Session* session_ptr = nullptr;
    tensorflow::Status status = tensorflow::NewSession(options, &session_ptr);
    if (!status.ok()) {
        std::cerr << "Failed to create session for GPU " << gpu_id << ": " 
                  << status.ToString() << std::endl;
        return nullptr;
    }
    return std::unique_ptr<tensorflow::Session>(session_ptr);
}

int main(int argc, char* argv[]) {
    // Parse GPU IDs (example: from command line args, matching your existing code)
    int gpu_id_model0 = 0;
    int gpu_id_model1 = 1;
    // If you're reading from args:
    // tensorflow::Flag flags[] = {
    //     tensorflow::Flag("gpu0", &gpu_id_model0, "GPU ID for model 0"),
    //     tensorflow::Flag("gpu1", &gpu_id_model1, "GPU ID for model 1"),
    // };
    // tensorflow::ParseFlags(&argc, argv, flags, sizeof(flags)/sizeof(flags[0]));

    // Create separate sessions for each GPU
    auto session_gpu0 = create_gpu_session(gpu_id_model0);
    auto session_gpu1 = create_gpu_session(gpu_id_model1);

    if (!session_gpu0 || !session_gpu1) {
        return EXIT_FAILURE;
    }

    // Load Model 0 into GPU 0's session
    tensorflow::GraphDef graph_def0;
    tensorflow::Status load_status0 = tensorflow::ReadBinaryProto(
        tensorflow::Env::Default(), "path/to/model0.pb", &graph_def0);
    if (!load_status0.ok()) {
        std::cerr << "Failed to load Model 0: " << load_status0.ToString() << std::endl;
        return EXIT_FAILURE;
    }
    if (!session_gpu0->Create(graph_def0).ok()) {
        std::cerr << "Failed to initialize Model 0 graph" << std::endl;
        return EXIT_FAILURE;
    }

    // Load Model 1 into GPU 1's session
    tensorflow::GraphDef graph_def1;
    tensorflow::Status load_status1 = tensorflow::ReadBinaryProto(
        tensorflow::Env::Default(), "path/to/model1.pb", &graph_def1);
    if (!load_status1.ok()) {
        std::cerr << "Failed to load Model 1: " << load_status1.ToString() << std::endl;
        return EXIT_FAILURE;
    }
    if (!session_gpu1->Create(graph_def1).ok()) {
        std::cerr << "Failed to initialize Model 1 graph" << std::endl;
        return EXIT_FAILURE;
    }

    // Run predictions on separate GPUs (example)
    // Model 0 prediction
    tensorflow::Tensor input_tensor0(tensorflow::DT_FLOAT, tensorflow::TensorShape({1, 224, 224, 3}));
    // Populate input_tensor0 with your data...
    std::vector<tensorflow::Tensor> outputs0;
    tensorflow::Status predict_status0 = session_gpu0->Run(
        {{"input_layer_name", input_tensor0}}, // Inputs
        {"output_layer_name"}, // Outputs to fetch
        {}, // No feeds
        &outputs0);
    if (!predict_status0.ok()) {
        std::cerr << "Model 0 prediction failed: " << predict_status0.ToString() << std::endl;
    }

    // Model 1 prediction
    tensorflow::Tensor input_tensor1(tensorflow::DT_FLOAT, tensorflow::TensorShape({1, 224, 224, 3}));
    // Populate input_tensor1 with your data...
    std::vector<tensorflow::Tensor> outputs1;
    tensorflow::Status predict_status1 = session_gpu1->Run(
        {{"input_layer_name", input_tensor1}},
        {"output_layer_name"},
        {},
        &outputs1);
    if (!predict_status1.ok()) {
        std::cerr << "Model 1 prediction failed: " << predict_status1.ToString() << std::endl;
    }

    // Cleanup
    session_gpu0->Close();
    session_gpu1->Close();

    return EXIT_SUCCESS;
}

Method 2: Scope Graph Nodes to a GPU (Exact Python Equivalent)

If you want fine-grained control over which nodes run on which GPU (just like with tf.device), use TensorFlow’s Scope API to assign devices during graph construction:

#include <tensorflow/core/public/session.h>
#include <tensorflow/core/ops/standard_ops.h>

int main() {
    // Create root scope
    tensorflow::Scope root = tensorflow::Scope::NewRootScope();

    // Create a scope for GPU 0 (all nodes under this run on /gpu:0)
    tensorflow::Scope gpu0_scope = root.WithDevice("/gpu:0");
    // Build Model 0 nodes under gpu0_scope
    auto input0 = tensorflow::Placeholder(gpu0_scope, tensorflow::DT_FLOAT, tensorflow::TensorShape({1, 224, 224, 3}));
    // Add your model layers here (example: a simple conv layer)
    auto conv0 = tensorflow::Conv2D(gpu0_scope, input0, 
        tensorflow::Const(gpu0_scope, tensorflow::Tensor(tensorflow::DT_FLOAT, tensorflow::TensorShape({3,3,3,32}))),
        {1,1,1,1}, "SAME");
    auto output0 = tensorflow::Relu(gpu0_scope, conv0);

    // Create a scope for GPU 1 (all nodes under this run on /gpu:1)
    tensorflow::Scope gpu1_scope = root.WithDevice("/gpu:1");
    // Build Model 1 nodes under gpu1_scope
    auto input1 = tensorflow::Placeholder(gpu1_scope, tensorflow::DT_FLOAT, tensorflow::TensorShape({1, 224, 224, 3}));
    auto conv1 = tensorflow::Conv2D(gpu1_scope, input1, 
        tensorflow::Const(gpu1_scope, tensorflow::Tensor(tensorflow::DT_FLOAT, tensorflow::TensorShape({3,3,3,64}))),
        {1,1,1,1}, "SAME");
    auto output1 = tensorflow::Relu(gpu1_scope, conv1);

    // Create a single session (no need to restrict GPU visibility, nodes are scoped)
    tensorflow::SessionOptions options;
    tensorflow::ConfigProto& config = options.config;
    config.set_allow_soft_placement(false);
    auto session = std::unique_ptr<tensorflow::Session>(tensorflow::NewSession(options));

    // Build the combined graph
    tensorflow::GraphDef graph_def;
    if (!root.ToGraphDef(&graph_def).ok()) {
        std::cerr << "Failed to build graph" << std::endl;
        return EXIT_FAILURE;
    }
    if (!session->Create(graph_def).ok()) {
        std::cerr << "Failed to initialize graph" << std::endl;
        return EXIT_FAILURE;
    }

    // Run predictions on both GPUs
    tensorflow::Tensor input_tensor0(tensorflow::DT_FLOAT, tensorflow::TensorShape({1,224,224,3}));
    tensorflow::Tensor input_tensor1(tensorflow::DT_FLOAT, tensorflow::TensorShape({1,224,224,3}));
    // Populate inputs...

    std::vector<tensorflow::Tensor> outputs0, outputs1;
    session->Run({{input0.name(), input_tensor0}}, {output0.name()}, {}, &outputs0);
    session->Run({{input1.name(), input_tensor1}}, {output1.name()}, {}, &outputs1);

    // Cleanup
    session->Close();
    return EXIT_SUCCESS;
}
Critical Notes for Your Setup
  • Isolate Graphs & Sessions: Always use separate GraphDef and Session instances for each model if using Method 1—this prevents cross-GPU resource conflicts.
  • CUDA/cuDNN Compatibility: Make sure your TensorFlow C++ library was compiled with CUDA 9.1 and cuDNN 7.1. Mismatched versions will cause runtime crashes or errors.
  • Memory Management: Use per_process_gpu_memory_fraction (as in Method 1) to cap each session’s GPU memory usage. Your 11GB GPUs should handle most models, but this avoids OOM if both models are memory-heavy.
  • Error Handling: Never skip checking tensorflow::Status returns—they’ll tell you exactly if a GPU is unavailable, a model failed to load, or a prediction crashed.

内容的提问来源于stack exchange,提问作者kerollos gamal

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.05.26 09:51:00