2023-12-02 19:04:51 +01:00
|
|
|
|
import { callPopup, main_api } from '../../../script.js';
|
|
|
|
|
import { getContext } from '../../extensions.js';
|
|
|
|
|
import { registerSlashCommand } from '../../slash-commands.js';
|
2024-04-13 20:33:19 +02:00
|
|
|
|
import { getFriendlyTokenizerName, getTextTokens, getTokenCountAsync, tokenizers } from '../../tokenizers.js';
|
2024-04-28 18:47:53 +02:00
|
|
|
|
import { resetScrollHeight, debounce } from '../../utils.js';
|
|
|
|
|
import { debounce_timeout } from '../../constants.js';
|
2023-07-20 19:32:15 +02:00
|
|
|
|
|
2023-11-06 10:04:47 +01:00
|
|
|
|
function rgb2hex(rgb) {
|
|
|
|
|
rgb = rgb.match(/^rgba?[\s+]?\([\s+]?(\d+)[\s+]?,[\s+]?(\d+)[\s+]?,[\s+]?(\d+)[\s+]?/i);
|
2023-12-02 19:04:51 +01:00
|
|
|
|
return (rgb && rgb.length === 4) ? '#' +
|
|
|
|
|
('0' + parseInt(rgb[1], 10).toString(16)).slice(-2) +
|
|
|
|
|
('0' + parseInt(rgb[2], 10).toString(16)).slice(-2) +
|
|
|
|
|
('0' + parseInt(rgb[3], 10).toString(16)).slice(-2) : '';
|
2023-11-06 10:04:47 +01:00
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
$('button').click(function () {
|
|
|
|
|
var hex = rgb2hex($('input').val());
|
|
|
|
|
$('.result').html(hex);
|
|
|
|
|
});
|
|
|
|
|
|
2023-07-20 19:32:15 +02:00
|
|
|
|
async function doTokenCounter() {
|
2023-11-06 19:25:59 +01:00
|
|
|
|
const { tokenizerName, tokenizerId } = getFriendlyTokenizerName(main_api);
|
2023-07-20 19:32:15 +02:00
|
|
|
|
const html = `
|
|
|
|
|
<div class="wide100p">
|
|
|
|
|
<h3>Token Counter</h3>
|
2023-11-06 19:25:59 +01:00
|
|
|
|
<div class="justifyLeft flex-container flexFlowColumn">
|
2023-07-20 19:32:15 +02:00
|
|
|
|
<h4>Type / paste in the box below to see the number of tokens in the text.</h4>
|
2023-11-06 19:25:59 +01:00
|
|
|
|
<p>Selected tokenizer: ${tokenizerName}</p>
|
2023-11-06 01:42:51 +01:00
|
|
|
|
<div>Input:</div>
|
2023-11-06 19:25:59 +01:00
|
|
|
|
<textarea id="token_counter_textarea" class="wide100p textarea_compact" rows="1"></textarea>
|
2023-07-20 19:32:15 +02:00
|
|
|
|
<div>Tokens: <span id="token_counter_result">0</span></div>
|
2023-11-06 19:25:59 +01:00
|
|
|
|
<hr>
|
2023-11-06 01:42:51 +01:00
|
|
|
|
<div>Tokenized text:</div>
|
|
|
|
|
<div id="tokenized_chunks_display" class="wide100p">—</div>
|
2023-11-06 19:25:59 +01:00
|
|
|
|
<hr>
|
2023-11-06 01:42:51 +01:00
|
|
|
|
<div>Token IDs:</div>
|
2024-03-27 23:27:00 +01:00
|
|
|
|
<textarea id="token_counter_ids" class="wide100p textarea_compact" readonly rows="1">—</textarea>
|
2023-07-20 19:32:15 +02:00
|
|
|
|
</div>
|
|
|
|
|
</div>`;
|
|
|
|
|
|
|
|
|
|
const dialog = $(html);
|
2024-04-13 20:33:19 +02:00
|
|
|
|
const countDebounced = debounce(async () => {
|
2023-11-05 21:45:37 +01:00
|
|
|
|
const text = String($('#token_counter_textarea').val());
|
|
|
|
|
const ids = main_api == 'openai' ? getTextTokens(tokenizers.OPENAI, text) : getTextTokens(tokenizerId, text);
|
|
|
|
|
|
|
|
|
|
if (Array.isArray(ids) && ids.length > 0) {
|
2023-11-06 01:42:51 +01:00
|
|
|
|
$('#token_counter_ids').text(`[${ids.join(', ')}]`);
|
2023-11-05 21:45:37 +01:00
|
|
|
|
$('#token_counter_result').text(ids.length);
|
2023-11-06 01:42:51 +01:00
|
|
|
|
|
|
|
|
|
if (Object.hasOwnProperty.call(ids, 'chunks')) {
|
|
|
|
|
drawChunks(Object.getOwnPropertyDescriptor(ids, 'chunks').value, ids);
|
|
|
|
|
}
|
2023-11-05 21:45:37 +01:00
|
|
|
|
} else {
|
2024-04-13 20:33:19 +02:00
|
|
|
|
const count = await getTokenCountAsync(text);
|
2023-11-05 21:45:37 +01:00
|
|
|
|
$('#token_counter_ids').text('—');
|
|
|
|
|
$('#token_counter_result').text(count);
|
2023-11-06 01:42:51 +01:00
|
|
|
|
$('#tokenized_chunks_display').text('—');
|
2023-11-05 21:45:37 +01:00
|
|
|
|
}
|
2023-11-06 19:25:59 +01:00
|
|
|
|
|
|
|
|
|
resetScrollHeight($('#token_counter_textarea'));
|
|
|
|
|
resetScrollHeight($('#token_counter_ids'));
|
2024-04-28 06:21:47 +02:00
|
|
|
|
}, debounce_timeout.relaxed);
|
2024-03-01 20:42:49 +01:00
|
|
|
|
dialog.find('#token_counter_textarea').on('input', () => countDebounced());
|
2023-07-20 19:32:15 +02:00
|
|
|
|
|
|
|
|
|
$('#dialogue_popup').addClass('wide_dialogue_popup');
|
2023-11-05 21:55:10 +01:00
|
|
|
|
callPopup(dialog, 'text', '', { wide: true, large: true });
|
2023-07-20 19:32:15 +02:00
|
|
|
|
}
|
|
|
|
|
|
2023-11-06 01:42:51 +01:00
|
|
|
|
/**
|
|
|
|
|
* Draws the tokenized chunks in the UI
|
|
|
|
|
* @param {string[]} chunks
|
|
|
|
|
* @param {number[]} ids
|
|
|
|
|
*/
|
|
|
|
|
function drawChunks(chunks, ids) {
|
|
|
|
|
const pastelRainbow = [
|
2023-11-06 10:04:47 +01:00
|
|
|
|
//main_text_color,
|
|
|
|
|
//italics_text_color,
|
|
|
|
|
//quote_text_color,
|
2023-11-06 01:42:51 +01:00
|
|
|
|
'#FFB3BA',
|
|
|
|
|
'#FFDFBA',
|
|
|
|
|
'#FFFFBA',
|
|
|
|
|
'#BFFFBF',
|
|
|
|
|
'#BAE1FF',
|
|
|
|
|
'#FFBAF3',
|
|
|
|
|
];
|
|
|
|
|
$('#tokenized_chunks_display').empty();
|
|
|
|
|
|
|
|
|
|
for (let i = 0; i < chunks.length; i++) {
|
|
|
|
|
let chunk = chunks[i].replace(/▁/g, ' '); // This is a leading space in sentencepiece. More info: Lower one eighth block (U+2581)
|
|
|
|
|
|
|
|
|
|
// If <0xHEX>, decode it
|
|
|
|
|
if (/^<0x[0-9A-F]+>$/i.test(chunk)) {
|
|
|
|
|
const code = parseInt(chunk.substring(3, chunk.length - 1), 16);
|
|
|
|
|
chunk = String.fromCodePoint(code);
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// If newline - insert a line break
|
|
|
|
|
if (chunk === '\n') {
|
|
|
|
|
$('#tokenized_chunks_display').append('<br>');
|
|
|
|
|
continue;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
const color = pastelRainbow[i % pastelRainbow.length];
|
2024-03-21 19:18:02 +01:00
|
|
|
|
const chunkHtml = $('<code></code>');
|
|
|
|
|
chunkHtml.css('background-color', color);
|
|
|
|
|
chunkHtml.text(chunk);
|
2023-11-06 01:42:51 +01:00
|
|
|
|
chunkHtml.attr('title', ids[i]);
|
|
|
|
|
$('#tokenized_chunks_display').append(chunkHtml);
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
2024-04-13 20:33:19 +02:00
|
|
|
|
async function doCount() {
|
2023-10-21 13:23:56 +02:00
|
|
|
|
// get all of the messages in the chat
|
|
|
|
|
const context = getContext();
|
|
|
|
|
const messages = context.chat.filter(x => x.mes && !x.is_system).map(x => x.mes);
|
|
|
|
|
|
|
|
|
|
//concat all the messages into a single string
|
|
|
|
|
const allMessages = messages.join(' ');
|
|
|
|
|
|
|
|
|
|
console.debug('All messages:', allMessages);
|
|
|
|
|
|
|
|
|
|
//toastr success with the token count of the chat
|
2024-04-13 20:33:19 +02:00
|
|
|
|
const count = await getTokenCountAsync(allMessages);
|
|
|
|
|
toastr.success(`Token count: ${count}`);
|
2023-10-21 13:23:56 +02:00
|
|
|
|
}
|
|
|
|
|
|
2023-07-20 19:32:15 +02:00
|
|
|
|
jQuery(() => {
|
|
|
|
|
const buttonHtml = `
|
|
|
|
|
<div id="token_counter" class="list-group-item flex-container flexGap5">
|
|
|
|
|
<div class="fa-solid fa-1 extensionsMenuExtensionButton" /></div>
|
|
|
|
|
Token Counter
|
|
|
|
|
</div>`;
|
|
|
|
|
$('#extensionsMenu').prepend(buttonHtml);
|
|
|
|
|
$('#token_counter').on('click', doTokenCounter);
|
2023-10-21 13:23:56 +02:00
|
|
|
|
registerSlashCommand('count', doCount, [], '– counts the number of tokens in the current chat', true, false);
|
2023-07-20 19:32:15 +02:00
|
|
|
|
});
|