mirror of
https://github.com/ikawrakow/ik_llama.cpp.git
synced 2026-07-21 10:15:35 +00:00
prompt cache: Fix assertion that prompt cache does ot rewind to middle of image (#1913)
This commit is contained in:
@@ -1250,7 +1250,7 @@ const mtmd::input_chunk_ptr& server_tokens::find_chunk(size_t idx) const {
|
||||
if (it != map_idx_to_media.end()) {
|
||||
return it->second;
|
||||
}
|
||||
throw std::runtime_error("Chunk not found");
|
||||
throw std::runtime_error("Chunk not found, or idx is not the first token of a chunk");
|
||||
}
|
||||
|
||||
void server_tokens::push_back(llama_token tok) {
|
||||
@@ -1369,18 +1369,10 @@ void server_tokens::keep_first(size_t n) {
|
||||
if (n == tokens.size()) {
|
||||
return; // nothing to do
|
||||
}
|
||||
// we throw an error if we try to remove a token in the middle of an image
|
||||
// for ex. with input of 5 text tokens and 2 images:
|
||||
// [0] [1] [2] [3] [4] [img0] [img0] [img0] [img1] [img1]
|
||||
// n 1 2 3 4 5 6 7 8 9 10
|
||||
// allowed to resize ^ ^
|
||||
// disallowed to resize ^ ^ ^
|
||||
if (n > 0) {
|
||||
llama_token last_token = tokens[n - 1];
|
||||
// make sure we never remove tokens in the middle of an image
|
||||
if (last_token == LLAMA_TOKEN_NULL) {
|
||||
find_chunk(n - 1); // will throw an error if the token is not begin-of-chunk
|
||||
}
|
||||
// It is an internal error if the longest common prefix ends in the middle of an image
|
||||
llama_token first_removed_token = tokens[n];
|
||||
if (first_removed_token == LLAMA_TOKEN_NULL) {
|
||||
find_chunk(n); // will throw an error if the token is not begin-of-chunk
|
||||
}
|
||||
// remove all image chunks that are not used anymore
|
||||
for (auto it = map_idx_to_media.begin(); it != map_idx_to_media.end(); ) {
|
||||
|
||||
Reference in New Issue
Block a user