Skip to content

Commit bd28ce2

Browse files
committed
feat(multimodal): report prompt prefill timings separately
- isolate MTMD prompt evaluation from text generation timings - synchronize pending backend work at explicit multimodal phase boundaries - measure text, image, and audio prompt chunks in one context interval - skip additional timing synchronization for fully cached prompts - label multimodal prompt metrics before completion resets the counters Signed-off-by: JamePeng <jame_peng@sina.com>
1 parent 161af08 commit bd28ce2

1 file changed

Lines changed: 27 additions & 0 deletions

File tree

‎llama_cpp/llama_multimodal.py‎

Lines changed: 27 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -1170,6 +1170,20 @@ def __call__(
11701170

11711171
n_past = llama.n_tokens
11721172

1173+
# MTMD evaluates prompt chunks before create_completion() starts
1174+
# its own timing interval. Settle any earlier asynchronous work and
1175+
# measure this prefill phase separately so that the completion reset
1176+
# cannot discard or misattribute its context timings.
1177+
perf_enabled = not llama.context_params.no_perf
1178+
has_prompt_work = any(
1179+
end_idx > n_past
1180+
for _, end_idx, _, _, _ in chunk_token_spans
1181+
)
1182+
prompt_evaluated = False
1183+
if perf_enabled and has_prompt_work:
1184+
llama._ctx.synchronize()
1185+
llama._ctx.reset_timings()
1186+
11731187
for start_idx, end_idx, chunk_ptr, chunk_type, media_id in chunk_token_spans:
11741188
# Skip previously matched chunks
11751189
if end_idx <= n_past:
@@ -1194,6 +1208,7 @@ def __call__(
11941208

11951209
# Text evaluation delegates shift and chunking to native llama.eval
11961210
llama.eval(tokens_to_eval)
1211+
prompt_evaluated = True
11971212
n_past = llama.n_tokens
11981213

11991214
elif self._is_image_chunk(chunk_type) or self._is_audio_chunk(chunk_type):
@@ -1256,6 +1271,18 @@ def __call__(
12561271
llama.input_ids[n_past : new_n_past.value] = media_id
12571272
n_past = new_n_past.value
12581273
llama.n_tokens = n_past
1274+
prompt_evaluated = True
1275+
1276+
if perf_enabled and prompt_evaluated:
1277+
# mtmd_helper_eval_chunk_single() bypasses LlamaContext.decode(),
1278+
# so explicitly settle its queued timing data at this boundary.
1279+
llama._ctx.synchronize()
1280+
if llama.verbose:
1281+
print(
1282+
f"{self.log_prefix}(__call__): Multimodal prompt timings:",
1283+
file=sys.stderr,
1284+
)
1285+
llama._ctx.print_timings()
12591286

12601287
# Extract the final, perfectly synchronized prompt sequence
12611288
prompt = llama.input_ids[: llama.n_tokens].tolist()

0 commit comments

Comments
 (0)