Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
25 changes: 4 additions & 21 deletions src/llm/language_model/continuous_batching/servable.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -17,7 +17,6 @@
#include <memory>
#include <stdexcept>
#include <string>
#include <thread>
#include <vector>

#include "../../../logging.hpp"
Expand Down Expand Up @@ -57,20 +56,6 @@ void ContinuousBatchingServable::logPerfMetrics(ov::genai::PerfMetrics& perfMetr
prefillSpeedTps);
}

// CB stepping thread writes metrics in _free_non_running_requests() slightly after
// pushing the final output. Yield briefly to close the race window.
// TODO: remove once GenAI's get_perf_metrics() blocks instead of asserting (fix in generation_stream.hpp)
static std::optional<ov::genai::PerfMetrics> tryGetPerfMetrics(const ov::genai::GenerationHandle& handle) {
for (int i = 0; i < 1000; ++i) {
try {
return handle->get_perf_metrics();
} catch (const ov::Exception&) {
std::this_thread::yield();
}
}
return std::nullopt;
}

void ContinuousBatchingServable::notifyExecutorThread() {
SPDLOG_LOGGER_TRACE(llm_calculator_logger, "Notifying executor thread");
if (properties->llmExecutorWrapper == nullptr) {
Expand Down Expand Up @@ -177,9 +162,8 @@ absl::Status ContinuousBatchingServable::prepareCompleteResponse(std::shared_ptr
auto status = GenAiServable::prepareCompleteResponse(executionContext);
if (status.ok() && llm_calculator_logger->should_log(spdlog::level::debug)) {
auto cbExecutionContext = std::static_pointer_cast<ContinuousBatchingServableExecutionContext>(executionContext);
auto perfMetrics = tryGetPerfMetrics(cbExecutionContext->generationHandle);
if (perfMetrics)
logPerfMetrics(*perfMetrics);
auto perfMetrics = cbExecutionContext->generationHandle->get_perf_metrics();
logPerfMetrics(perfMetrics);
}
return status;
}
Expand All @@ -190,9 +174,8 @@ absl::Status ContinuousBatchingServable::preparePartialResponse(std::shared_ptr<
!executionContext->sendLoopbackSignal &&
llm_calculator_logger->should_log(spdlog::level::debug)) {
auto cbExecutionContext = std::static_pointer_cast<ContinuousBatchingServableExecutionContext>(executionContext);
auto perfMetrics = tryGetPerfMetrics(cbExecutionContext->generationHandle);
if (perfMetrics)
logPerfMetrics(*perfMetrics);
auto perfMetrics = cbExecutionContext->generationHandle->get_perf_metrics();
logPerfMetrics(perfMetrics);
}
return status;
}
Expand Down
26 changes: 4 additions & 22 deletions src/llm/visual_language_model/continuous_batching/servable.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -19,7 +19,6 @@
#include <memory>
#include <stdexcept>
#include <string>
#include <thread>
#include <vector>

#include "src/port/rapidjson_document.hpp"
Expand All @@ -36,21 +35,6 @@

namespace ovms {

// CB stepping thread writes base perf metrics in _free_non_running_requests() slightly
// after pushing the final output; get_vlm_perf_metrics() calls get_perf_metrics() internally.
// Yield briefly to close the race window.
// TODO: remove once GenAI's get_perf_metrics() blocks instead of asserting (fix in generation_stream.hpp)
static std::optional<ov::genai::VLMPerfMetrics> tryGetVlmPerfMetrics(const ov::genai::GenerationHandle& handle) {
for (int i = 0; i < 1000; ++i) {
try {
return handle->get_vlm_perf_metrics();
} catch (const ov::Exception&) {
std::this_thread::yield();
}
}
return std::nullopt;
}

void VisualLanguageModelServable::logPerfMetrics(ov::genai::VLMPerfMetrics& perfMetrics) {
const size_t inputTokenCount = perfMetrics.get_num_input_tokens();
const size_t outputTokenCount = perfMetrics.get_num_generated_tokens();
Expand Down Expand Up @@ -100,9 +84,8 @@ absl::Status VisualLanguageModelServable::prepareCompleteResponse(std::shared_pt
auto status = GenAiServable::prepareCompleteResponse(executionContext);
if (status.ok() && llm_calculator_logger->should_log(spdlog::level::debug)) {
auto vlmExecutionContext = std::static_pointer_cast<VisualLanguageModelServableExecutionContext>(executionContext);
auto perfMetrics = tryGetVlmPerfMetrics(vlmExecutionContext->generationHandle);
if (perfMetrics)
logPerfMetrics(*perfMetrics);
auto perfMetrics = vlmExecutionContext->generationHandle->get_vlm_perf_metrics();
logPerfMetrics(perfMetrics);
}
return status;
}
Expand All @@ -113,9 +96,8 @@ absl::Status VisualLanguageModelServable::preparePartialResponse(std::shared_ptr
!executionContext->sendLoopbackSignal &&
llm_calculator_logger->should_log(spdlog::level::debug)) {
auto vlmExecutionContext = std::static_pointer_cast<VisualLanguageModelServableExecutionContext>(executionContext);
auto perfMetrics = tryGetVlmPerfMetrics(vlmExecutionContext->generationHandle);
if (perfMetrics)
logPerfMetrics(*perfMetrics);
auto perfMetrics = vlmExecutionContext->generationHandle->get_vlm_perf_metrics();
logPerfMetrics(perfMetrics);
}
return status;
}
Expand Down
Loading