2023-03-25 19:26:40 +01:00
|
|
|
#include "common.h"
|
2023-03-24 16:19:05 +01:00
|
|
|
|
2023-03-11 00:04:06 +01:00
|
|
|
#include <cassert>
|
|
|
|
#include <cstring>
|
2023-04-14 17:19:17 +02:00
|
|
|
#include <iostream>
|
2023-03-10 19:40:58 +01:00
|
|
|
#include <fstream>
|
2023-04-14 17:19:17 +02:00
|
|
|
#include <sstream>
|
2023-03-12 21:28:36 +01:00
|
|
|
#include <string>
|
2023-03-22 06:32:36 +01:00
|
|
|
#include <iterator>
|
|
|
|
#include <algorithm>
|
2023-04-14 17:19:17 +02:00
|
|
|
#include <regex>
|
2023-03-10 19:40:58 +01:00
|
|
|
|
2023-03-28 16:09:55 +02:00
|
|
|
#if defined (_WIN32)
|
2023-04-08 17:49:39 +02:00
|
|
|
#include <fcntl.h>
|
|
|
|
#include <io.h>
|
2023-03-28 16:09:55 +02:00
|
|
|
#pragma comment(lib,"kernel32.lib")
|
|
|
|
extern "C" __declspec(dllimport) void* __stdcall GetStdHandle(unsigned long nStdHandle);
|
|
|
|
extern "C" __declspec(dllimport) int __stdcall GetConsoleMode(void* hConsoleHandle, unsigned long* lpMode);
|
|
|
|
extern "C" __declspec(dllimport) int __stdcall SetConsoleMode(void* hConsoleHandle, unsigned long dwMode);
|
|
|
|
extern "C" __declspec(dllimport) int __stdcall SetConsoleCP(unsigned int wCodePageID);
|
|
|
|
extern "C" __declspec(dllimport) int __stdcall SetConsoleOutputCP(unsigned int wCodePageID);
|
2023-04-11 21:45:44 +02:00
|
|
|
extern "C" __declspec(dllimport) int __stdcall WideCharToMultiByte(unsigned int CodePage, unsigned long dwFlags,
|
|
|
|
const wchar_t * lpWideCharStr, int cchWideChar,
|
|
|
|
char * lpMultiByteStr, int cbMultiByte,
|
2023-04-08 17:49:39 +02:00
|
|
|
const char * lpDefaultChar, bool * lpUsedDefaultChar);
|
|
|
|
#define CP_UTF8 65001
|
2023-03-28 16:09:55 +02:00
|
|
|
#endif
|
2023-03-12 21:15:00 +01:00
|
|
|
|
2023-04-14 17:19:17 +02:00
|
|
|
void split_args(const std::string & args_string, std::vector<std::string> & output_args)
|
|
|
|
{
|
|
|
|
std::string current_arg = "";
|
|
|
|
bool in_quotes = false;
|
|
|
|
char quote_type;
|
|
|
|
|
|
|
|
for (char c : args_string) {
|
|
|
|
if (c == '"' || c == '\'') {
|
|
|
|
if (!in_quotes) {
|
|
|
|
in_quotes = true;
|
|
|
|
quote_type = c;
|
|
|
|
} else if (quote_type == c) {
|
|
|
|
in_quotes = false;
|
|
|
|
} else {
|
|
|
|
current_arg += c;
|
|
|
|
}
|
|
|
|
} else if (in_quotes) {
|
|
|
|
current_arg += c;
|
|
|
|
} else if (std::isspace(c)) {
|
|
|
|
if (current_arg != "") {
|
|
|
|
output_args.push_back(current_arg);
|
|
|
|
current_arg = "";
|
|
|
|
}
|
|
|
|
} else {
|
|
|
|
current_arg += c;
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
if (current_arg != "") {
|
|
|
|
output_args.push_back(current_arg);
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
std::string unescape(const std::string & str) {
|
|
|
|
return std::regex_replace(str, std::regex("\\\\n"), "\n");
|
|
|
|
}
|
|
|
|
|
2023-03-10 19:40:58 +01:00
|
|
|
bool gpt_params_parse(int argc, char ** argv, gpt_params & params) {
|
2023-03-17 18:47:35 +01:00
|
|
|
// determine sensible default number of threads.
|
|
|
|
// std::thread::hardware_concurrency may not be equal to the number of cores, or may return 0.
|
|
|
|
#ifdef __linux__
|
|
|
|
std::ifstream cpuinfo("/proc/cpuinfo");
|
|
|
|
params.n_threads = std::count(std::istream_iterator<std::string>(cpuinfo),
|
|
|
|
std::istream_iterator<std::string>(),
|
|
|
|
std::string("processor"));
|
|
|
|
#endif
|
|
|
|
if (params.n_threads == 0) {
|
|
|
|
params.n_threads = std::max(1, (int32_t) std::thread::hardware_concurrency());
|
|
|
|
}
|
|
|
|
|
2023-03-23 18:54:28 +01:00
|
|
|
bool invalid_param = false;
|
|
|
|
std::string arg;
|
2023-04-02 04:41:12 +02:00
|
|
|
gpt_params default_params;
|
|
|
|
|
2023-04-14 17:19:17 +02:00
|
|
|
// get additional arguments from config files
|
|
|
|
std::vector<std::string> args;
|
2023-03-10 19:40:58 +01:00
|
|
|
for (int i = 1; i < argc; i++) {
|
2023-03-23 18:54:28 +01:00
|
|
|
arg = argv[i];
|
2023-04-14 17:19:17 +02:00
|
|
|
if (arg == "--config") {
|
|
|
|
if (++i >= argc) {
|
|
|
|
invalid_param = true;
|
|
|
|
break;
|
|
|
|
}
|
|
|
|
std::ifstream file(argv[i]);
|
|
|
|
if (!file) {
|
|
|
|
fprintf(stderr, "error: failed to open file '%s'\n", argv[i]);
|
|
|
|
invalid_param = true;
|
|
|
|
break;
|
|
|
|
}
|
|
|
|
std::string args_string;
|
|
|
|
std::copy(std::istreambuf_iterator<char>(file), std::istreambuf_iterator<char>(), back_inserter(args_string));
|
|
|
|
if (args_string.back() == '\n') {
|
|
|
|
args_string.pop_back();
|
|
|
|
}
|
|
|
|
split_args(args_string, args);
|
|
|
|
for (int j = 0; j < args.size(); j++) {
|
|
|
|
args[j] = unescape(args[j]);
|
|
|
|
}
|
|
|
|
} else {
|
|
|
|
args.emplace_back(argv[i]);
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
// parse args
|
|
|
|
int args_c = static_cast<int>(args.size());
|
|
|
|
for (int i = 0; i < args_c && !invalid_param; i++) {
|
|
|
|
arg = args[i];
|
2023-03-10 19:40:58 +01:00
|
|
|
|
|
|
|
if (arg == "-s" || arg == "--seed") {
|
2023-04-14 17:19:17 +02:00
|
|
|
if (++i >= args_c) {
|
2023-03-23 18:54:28 +01:00
|
|
|
invalid_param = true;
|
|
|
|
break;
|
|
|
|
}
|
2023-04-14 17:19:17 +02:00
|
|
|
params.seed = std::stoi(args[i]);
|
2023-03-10 19:40:58 +01:00
|
|
|
} else if (arg == "-t" || arg == "--threads") {
|
2023-04-14 17:19:17 +02:00
|
|
|
if (++i >= args_c) {
|
2023-03-23 18:54:28 +01:00
|
|
|
invalid_param = true;
|
|
|
|
break;
|
|
|
|
}
|
2023-04-14 17:19:17 +02:00
|
|
|
params.n_threads = std::stoi(args[i]);
|
2023-03-10 19:40:58 +01:00
|
|
|
} else if (arg == "-p" || arg == "--prompt") {
|
2023-04-14 17:19:17 +02:00
|
|
|
if (++i >= args_c) {
|
2023-03-23 18:54:28 +01:00
|
|
|
invalid_param = true;
|
|
|
|
break;
|
|
|
|
}
|
2023-04-14 17:19:17 +02:00
|
|
|
params.prompt = args[i];
|
2023-03-12 21:28:36 +01:00
|
|
|
} else if (arg == "-f" || arg == "--file") {
|
2023-04-14 17:19:17 +02:00
|
|
|
if (++i >= args_c) {
|
2023-03-23 18:54:28 +01:00
|
|
|
invalid_param = true;
|
|
|
|
break;
|
|
|
|
}
|
2023-04-14 17:19:17 +02:00
|
|
|
std::ifstream file(args[i]);
|
2023-03-31 20:03:48 +02:00
|
|
|
if (!file) {
|
2023-04-14 17:19:17 +02:00
|
|
|
fprintf(stderr, "error: failed to open file '%s'\n", args[i].c_str());
|
2023-03-31 20:03:48 +02:00
|
|
|
invalid_param = true;
|
|
|
|
break;
|
|
|
|
}
|
2023-03-19 17:37:02 +01:00
|
|
|
std::copy(std::istreambuf_iterator<char>(file), std::istreambuf_iterator<char>(), back_inserter(params.prompt));
|
2023-03-19 18:04:44 +01:00
|
|
|
if (params.prompt.back() == '\n') {
|
|
|
|
params.prompt.pop_back();
|
|
|
|
}
|
2023-03-10 19:40:58 +01:00
|
|
|
} else if (arg == "-n" || arg == "--n_predict") {
|
2023-04-14 17:19:17 +02:00
|
|
|
if (++i >= args_c) {
|
2023-03-23 18:54:28 +01:00
|
|
|
invalid_param = true;
|
|
|
|
break;
|
|
|
|
}
|
2023-04-14 17:19:17 +02:00
|
|
|
params.n_predict = std::stoi(args[i]);
|
2023-03-10 19:40:58 +01:00
|
|
|
} else if (arg == "--top_k") {
|
2023-04-14 17:19:17 +02:00
|
|
|
if (++i >= args_c) {
|
2023-03-23 18:54:28 +01:00
|
|
|
invalid_param = true;
|
|
|
|
break;
|
|
|
|
}
|
2023-04-14 17:19:17 +02:00
|
|
|
params.top_k = std::stoi(args[i]);
|
2023-03-15 20:42:40 +01:00
|
|
|
} else if (arg == "-c" || arg == "--ctx_size") {
|
2023-04-14 17:19:17 +02:00
|
|
|
if (++i >= args_c) {
|
2023-03-23 18:54:28 +01:00
|
|
|
invalid_param = true;
|
|
|
|
break;
|
|
|
|
}
|
2023-04-14 17:19:17 +02:00
|
|
|
params.n_ctx = std::stoi(args[i]);
|
2023-03-24 22:17:37 +01:00
|
|
|
} else if (arg == "--memory_f32") {
|
|
|
|
params.memory_f16 = false;
|
2023-03-10 19:40:58 +01:00
|
|
|
} else if (arg == "--top_p") {
|
2023-04-14 17:19:17 +02:00
|
|
|
if (++i >= args_c) {
|
2023-03-23 18:54:28 +01:00
|
|
|
invalid_param = true;
|
|
|
|
break;
|
|
|
|
}
|
2023-04-14 17:19:17 +02:00
|
|
|
params.top_p = std::stof(args[i]);
|
2023-03-10 19:40:58 +01:00
|
|
|
} else if (arg == "--temp") {
|
2023-04-14 17:19:17 +02:00
|
|
|
if (++i >= args_c) {
|
2023-03-23 18:54:28 +01:00
|
|
|
invalid_param = true;
|
|
|
|
break;
|
|
|
|
}
|
2023-04-14 17:19:17 +02:00
|
|
|
params.temp = std::stof(args[i]);
|
2023-03-12 10:27:42 +01:00
|
|
|
} else if (arg == "--repeat_last_n") {
|
2023-04-14 17:19:17 +02:00
|
|
|
if (++i >= args_c) {
|
2023-03-23 18:54:28 +01:00
|
|
|
invalid_param = true;
|
|
|
|
break;
|
|
|
|
}
|
2023-04-14 17:19:17 +02:00
|
|
|
params.repeat_last_n = std::stoi(args[i]);
|
2023-03-12 10:27:42 +01:00
|
|
|
} else if (arg == "--repeat_penalty") {
|
2023-04-14 17:19:17 +02:00
|
|
|
if (++i >= args_c) {
|
2023-03-23 18:54:28 +01:00
|
|
|
invalid_param = true;
|
|
|
|
break;
|
|
|
|
}
|
2023-04-14 17:19:17 +02:00
|
|
|
params.repeat_penalty = std::stof(args[i]);
|
2023-03-10 19:40:58 +01:00
|
|
|
} else if (arg == "-b" || arg == "--batch_size") {
|
2023-04-14 17:19:17 +02:00
|
|
|
if (++i >= args_c) {
|
2023-03-23 18:54:28 +01:00
|
|
|
invalid_param = true;
|
|
|
|
break;
|
|
|
|
}
|
2023-04-14 17:19:17 +02:00
|
|
|
params.n_batch = std::stoi(args[i]);
|
2023-03-24 22:17:37 +01:00
|
|
|
params.n_batch = std::min(512, params.n_batch);
|
2023-03-25 20:36:22 +01:00
|
|
|
} else if (arg == "--keep") {
|
2023-04-14 17:19:17 +02:00
|
|
|
if (++i >= args_c) {
|
2023-03-25 20:36:22 +01:00
|
|
|
invalid_param = true;
|
|
|
|
break;
|
|
|
|
}
|
2023-04-14 17:19:17 +02:00
|
|
|
params.n_keep = std::stoi(args[i]);
|
2023-03-10 19:40:58 +01:00
|
|
|
} else if (arg == "-m" || arg == "--model") {
|
2023-04-14 17:19:17 +02:00
|
|
|
if (++i >= args_c) {
|
2023-03-23 18:54:28 +01:00
|
|
|
invalid_param = true;
|
|
|
|
break;
|
|
|
|
}
|
2023-04-14 17:19:17 +02:00
|
|
|
params.model = args[i];
|
2023-03-12 22:13:28 +01:00
|
|
|
} else if (arg == "-i" || arg == "--interactive") {
|
|
|
|
params.interactive = true;
|
2023-03-24 16:05:13 +01:00
|
|
|
} else if (arg == "--embedding") {
|
|
|
|
params.embedding = true;
|
2023-04-14 17:19:17 +02:00
|
|
|
} else if (arg == "--clean-interface") {
|
|
|
|
params.clean_interface = true;
|
2023-03-24 16:05:13 +01:00
|
|
|
} else if (arg == "--interactive-start") {
|
|
|
|
params.interactive = true;
|
2023-03-22 18:16:35 +01:00
|
|
|
} else if (arg == "--interactive-first") {
|
|
|
|
params.interactive_start = true;
|
2023-03-19 17:37:02 +01:00
|
|
|
} else if (arg == "-ins" || arg == "--instruct") {
|
2023-04-14 17:19:17 +02:00
|
|
|
fprintf(stderr, "\n\nWarning: instruct mode is deprecated! Use: \n"
|
|
|
|
"--clean-interface "
|
|
|
|
"--interactive-first "
|
|
|
|
"--keep -1 "
|
|
|
|
"--ins-prefix-bos "
|
|
|
|
"--ins-prefix \"\\n\\n### Instruction:\\n\\n\" "
|
|
|
|
"--ins-suffix \"\\n\\n### Response:\\n\\n\" "
|
|
|
|
"-r \"### Instruction:\\n\\n\" "
|
|
|
|
"\n\n");
|
|
|
|
// params.instruct = true;
|
|
|
|
params.clean_interface = true;
|
|
|
|
params.interactive_start = true;
|
|
|
|
params.n_keep = -1;
|
|
|
|
params.instruct_prefix_bos = true;
|
|
|
|
params.instruct_prefix = "\n\n### Instruction:\n\n";
|
|
|
|
params.instruct_suffix = "\n\n### Response:\n\n";
|
|
|
|
params.antiprompt.push_back("### Instruction:\n\n");
|
2023-03-12 22:13:28 +01:00
|
|
|
} else if (arg == "--color") {
|
|
|
|
params.use_color = true;
|
2023-04-14 17:19:17 +02:00
|
|
|
} else if (arg == "--disable-multiline") {
|
|
|
|
params.multiline_mode = false;
|
2023-03-24 16:19:05 +01:00
|
|
|
} else if (arg == "--mlock") {
|
|
|
|
params.use_mlock = true;
|
Rewrite loading code to try to satisfy everyone:
- Support all three formats (ggml, ggmf, ggjt). (However, I didn't
include the hack needed to support GPT4All files without conversion.
Those can still be used after converting them with convert.py from my
other PR.)
- Support both mmap and read (mmap is used by default, but can be
disabled with `--no-mmap`, and is automatically disabled for pre-ggjt
files or on platforms where mmap is not supported).
- Support multi-file models like before, but automatically determine the
number of parts rather than requiring `--n_parts`.
- Improve validation and error checking.
- Stop using the per-file type field (f16) entirely in favor of just
relying on the per-tensor type/size fields. This has no immediate
benefit, but makes it easier to experiment with different formats, and
should make it easier to support the new GPTQ-for-LLaMa models in the
future (I have some work in progress on that front).
- Support VirtualLock on Windows (using the same `--mlock` option as on
Unix).
- Indicate loading progress when using mmap + mlock. (Which led me
to the interesting observation that on my Linux machine, with a
warm file cache, mlock actually takes some time, whereas mmap
without mlock starts almost instantly...)
- To help implement this, move mlock support from ggml to the
loading code.
- madvise/PrefetchVirtualMemory support (based on #740)
- Switch from ifstream to the `fopen` family of functions to avoid
unnecessary copying and, when mmap is enabled, allow reusing the same
file descriptor for both metadata reads and mmap (whereas the existing
implementation opens the file a second time to mmap).
- Quantization now produces a single-file output even with multi-file
inputs (not really a feature as much as 'it was easier this way').
Implementation notes:
I tried to factor the code into more discrete pieces than before.
Regarding code style: I tried to follow the code style, but I'm naughty
and used a few advanced C++ features repeatedly:
- Destructors to make it easier to ensure everything gets cleaned up.
- Exceptions. I don't even usually use exceptions when writing C++, and
I can remove them if desired... but here they make the loading code
much more succinct while still properly handling a variety of errors,
ranging from API calls failing to integer overflow and allocation
failure. The exceptions are converted to error codes at the
API boundary.)
Co-authored-by: Pavol Rusnak <pavol@rusnak.io> (for the bit I copied from #740)
2023-04-08 21:24:37 +02:00
|
|
|
} else if (arg == "--no-mmap") {
|
|
|
|
params.use_mmap = false;
|
2023-03-24 22:17:37 +01:00
|
|
|
} else if (arg == "--mtest") {
|
|
|
|
params.mem_test = true;
|
2023-03-25 20:36:22 +01:00
|
|
|
} else if (arg == "--verbose-prompt") {
|
2023-03-25 16:16:50 +01:00
|
|
|
params.verbose_prompt = true;
|
2023-03-12 22:13:28 +01:00
|
|
|
} else if (arg == "-r" || arg == "--reverse-prompt") {
|
2023-04-14 17:19:17 +02:00
|
|
|
if (++i >= args_c) {
|
2023-03-23 18:54:28 +01:00
|
|
|
invalid_param = true;
|
|
|
|
break;
|
|
|
|
}
|
2023-04-14 17:19:17 +02:00
|
|
|
params.antiprompt.push_back(args[i]);
|
|
|
|
} else if (arg == "--stop-prompt") {
|
|
|
|
if (++i >= args_c) {
|
|
|
|
invalid_param = true;
|
|
|
|
break;
|
|
|
|
}
|
|
|
|
params.stopprompt.push_back(args[i]);
|
|
|
|
} else if (arg == "--rm-trailing-space-workaround") {
|
|
|
|
params.rm_trailing_space_workaround = true;
|
2023-03-21 17:27:42 +01:00
|
|
|
} else if (arg == "--perplexity") {
|
|
|
|
params.perplexity = true;
|
2023-03-19 19:22:48 +01:00
|
|
|
} else if (arg == "--ignore-eos") {
|
|
|
|
params.ignore_eos = true;
|
2023-03-21 16:42:43 +01:00
|
|
|
} else if (arg == "--n_parts") {
|
2023-04-14 17:19:17 +02:00
|
|
|
if (++i >= args_c) {
|
2023-03-23 18:54:28 +01:00
|
|
|
invalid_param = true;
|
|
|
|
break;
|
|
|
|
}
|
2023-04-14 17:19:17 +02:00
|
|
|
params.n_parts = std::stoi(args[i]);
|
2023-03-10 19:40:58 +01:00
|
|
|
} else if (arg == "-h" || arg == "--help") {
|
2023-04-14 17:19:17 +02:00
|
|
|
gpt_print_usage(argv[0], default_params);
|
2023-03-10 19:40:58 +01:00
|
|
|
exit(0);
|
2023-03-19 19:36:19 +01:00
|
|
|
} else if (arg == "--random-prompt") {
|
|
|
|
params.random_prompt = true;
|
2023-03-25 13:03:19 +01:00
|
|
|
} else if (arg == "--in-prefix") {
|
2023-04-14 17:19:17 +02:00
|
|
|
if (++i >= args_c) {
|
|
|
|
invalid_param = true;
|
|
|
|
break;
|
|
|
|
}
|
|
|
|
params.input_prefix = args[i];
|
|
|
|
} else if (arg == "--ins-prefix-bos") {
|
|
|
|
params.instruct_prefix_bos = true;
|
|
|
|
} else if (arg == "--ins-prefix") {
|
|
|
|
if (++i >= args_c) {
|
2023-03-25 13:42:09 +01:00
|
|
|
invalid_param = true;
|
|
|
|
break;
|
|
|
|
}
|
2023-04-14 17:19:17 +02:00
|
|
|
params.instruct_prefix = args[i];
|
|
|
|
} else if (arg == "--ins-suffix-bos") {
|
|
|
|
params.instruct_suffix_bos = true;
|
|
|
|
} else if (arg == "--ins-suffix") {
|
|
|
|
if (++i >= args_c) {
|
|
|
|
invalid_param = true;
|
|
|
|
break;
|
|
|
|
}
|
|
|
|
params.instruct_suffix = args[i];
|
2023-03-10 19:40:58 +01:00
|
|
|
} else {
|
|
|
|
fprintf(stderr, "error: unknown argument: %s\n", arg.c_str());
|
2023-04-14 17:19:17 +02:00
|
|
|
gpt_print_usage(argv[0], default_params);
|
2023-03-23 18:54:28 +01:00
|
|
|
exit(1);
|
2023-03-10 19:40:58 +01:00
|
|
|
}
|
|
|
|
}
|
2023-03-23 18:54:28 +01:00
|
|
|
if (invalid_param) {
|
|
|
|
fprintf(stderr, "error: invalid parameter for argument: %s\n", arg.c_str());
|
2023-04-14 17:19:17 +02:00
|
|
|
gpt_print_usage(argv[0], default_params);
|
2023-03-23 18:54:28 +01:00
|
|
|
exit(1);
|
|
|
|
}
|
2023-03-10 19:40:58 +01:00
|
|
|
|
|
|
|
return true;
|
|
|
|
}
|
|
|
|
|
2023-04-14 17:19:17 +02:00
|
|
|
void gpt_print_usage(char * argv_0, const gpt_params & params) {
|
|
|
|
fprintf(stderr, "usage: %s [options]\n", argv_0);
|
2023-03-10 19:40:58 +01:00
|
|
|
fprintf(stderr, "\n");
|
|
|
|
fprintf(stderr, "options:\n");
|
|
|
|
fprintf(stderr, " -h, --help show this help message and exit\n");
|
2023-03-12 22:13:28 +01:00
|
|
|
fprintf(stderr, " -i, --interactive run in interactive mode\n");
|
2023-03-22 18:16:35 +01:00
|
|
|
fprintf(stderr, " --interactive-first run in interactive mode and wait for input right away\n");
|
2023-04-14 17:19:17 +02:00
|
|
|
fprintf(stderr, " --clean-interface hides input prefix & suffix and displays '>' instead\n");
|
2023-03-12 22:13:28 +01:00
|
|
|
fprintf(stderr, " -r PROMPT, --reverse-prompt PROMPT\n");
|
2023-03-22 18:16:35 +01:00
|
|
|
fprintf(stderr, " run in interactive mode and poll user input upon seeing PROMPT (can be\n");
|
2023-03-19 20:33:06 +01:00
|
|
|
fprintf(stderr, " specified more than once for multiple prompts).\n");
|
2023-03-12 22:13:28 +01:00
|
|
|
fprintf(stderr, " --color colorise output to distinguish prompt and user input from generations\n");
|
2023-04-14 17:19:17 +02:00
|
|
|
fprintf(stderr, " --disable-multiline disable multiline mode (use Ctrl+D on Linux/Mac and Ctrl+Z then Return on Windows to toggle multiline)\n");
|
2023-03-22 06:47:15 +01:00
|
|
|
fprintf(stderr, " -s SEED, --seed SEED RNG seed (default: -1, use random seed for <= 0)\n");
|
2023-03-10 19:40:58 +01:00
|
|
|
fprintf(stderr, " -t N, --threads N number of threads to use during computation (default: %d)\n", params.n_threads);
|
|
|
|
fprintf(stderr, " -p PROMPT, --prompt PROMPT\n");
|
2023-03-19 19:36:19 +01:00
|
|
|
fprintf(stderr, " prompt to start generation with (default: empty)\n");
|
|
|
|
fprintf(stderr, " --random-prompt start with a randomized prompt.\n");
|
2023-03-25 13:03:19 +01:00
|
|
|
fprintf(stderr, " --in-prefix STRING string to prefix user inputs with (default: empty)\n");
|
2023-04-14 17:19:17 +02:00
|
|
|
fprintf(stderr, " --ins-prefix STRING (instruct) prefix user inputs with tokenized string (default: empty)\n");
|
|
|
|
fprintf(stderr, " --ins-prefix-bos (instruct) prepend bos token to instruct prefix.\n");
|
|
|
|
fprintf(stderr, " --ins-suffix STRING (instruct) suffix user inputs with tokenized string (default: empty)\n");
|
|
|
|
fprintf(stderr, " --ins-suffix-bos (instruct) prepend bos token to instruct suffix.\n");
|
2023-03-12 21:28:36 +01:00
|
|
|
fprintf(stderr, " -f FNAME, --file FNAME\n");
|
|
|
|
fprintf(stderr, " prompt file to start generation.\n");
|
2023-03-28 16:09:55 +02:00
|
|
|
fprintf(stderr, " -n N, --n_predict N number of tokens to predict (default: %d, -1 = infinity)\n", params.n_predict);
|
2023-03-10 19:40:58 +01:00
|
|
|
fprintf(stderr, " --top_k N top-k sampling (default: %d)\n", params.top_k);
|
2023-03-28 18:48:20 +02:00
|
|
|
fprintf(stderr, " --top_p N top-p sampling (default: %.1f)\n", (double)params.top_p);
|
2023-03-12 10:27:42 +01:00
|
|
|
fprintf(stderr, " --repeat_last_n N last n tokens to consider for penalize (default: %d)\n", params.repeat_last_n);
|
2023-03-28 18:48:20 +02:00
|
|
|
fprintf(stderr, " --repeat_penalty N penalize repeat sequence of tokens (default: %.1f)\n", (double)params.repeat_penalty);
|
2023-03-15 20:42:40 +01:00
|
|
|
fprintf(stderr, " -c N, --ctx_size N size of the prompt context (default: %d)\n", params.n_ctx);
|
2023-03-19 19:22:48 +01:00
|
|
|
fprintf(stderr, " --ignore-eos ignore end of stream token and continue generating\n");
|
2023-03-24 22:17:37 +01:00
|
|
|
fprintf(stderr, " --memory_f32 use f32 instead of f16 for memory key+value\n");
|
2023-03-28 18:48:20 +02:00
|
|
|
fprintf(stderr, " --temp N temperature (default: %.1f)\n", (double)params.temp);
|
2023-03-21 16:42:43 +01:00
|
|
|
fprintf(stderr, " --n_parts N number of model parts (default: -1 = determine from dimensions)\n");
|
2023-03-10 19:40:58 +01:00
|
|
|
fprintf(stderr, " -b N, --batch_size N batch size for prompt processing (default: %d)\n", params.n_batch);
|
2023-03-21 17:27:42 +01:00
|
|
|
fprintf(stderr, " --perplexity compute perplexity over the prompt\n");
|
2023-03-28 16:09:55 +02:00
|
|
|
fprintf(stderr, " --keep number of tokens to keep from the initial prompt (default: %d, -1 = all)\n", params.n_keep);
|
Rewrite loading code to try to satisfy everyone:
- Support all three formats (ggml, ggmf, ggjt). (However, I didn't
include the hack needed to support GPT4All files without conversion.
Those can still be used after converting them with convert.py from my
other PR.)
- Support both mmap and read (mmap is used by default, but can be
disabled with `--no-mmap`, and is automatically disabled for pre-ggjt
files or on platforms where mmap is not supported).
- Support multi-file models like before, but automatically determine the
number of parts rather than requiring `--n_parts`.
- Improve validation and error checking.
- Stop using the per-file type field (f16) entirely in favor of just
relying on the per-tensor type/size fields. This has no immediate
benefit, but makes it easier to experiment with different formats, and
should make it easier to support the new GPTQ-for-LLaMa models in the
future (I have some work in progress on that front).
- Support VirtualLock on Windows (using the same `--mlock` option as on
Unix).
- Indicate loading progress when using mmap + mlock. (Which led me
to the interesting observation that on my Linux machine, with a
warm file cache, mlock actually takes some time, whereas mmap
without mlock starts almost instantly...)
- To help implement this, move mlock support from ggml to the
loading code.
- madvise/PrefetchVirtualMemory support (based on #740)
- Switch from ifstream to the `fopen` family of functions to avoid
unnecessary copying and, when mmap is enabled, allow reusing the same
file descriptor for both metadata reads and mmap (whereas the existing
implementation opens the file a second time to mmap).
- Quantization now produces a single-file output even with multi-file
inputs (not really a feature as much as 'it was easier this way').
Implementation notes:
I tried to factor the code into more discrete pieces than before.
Regarding code style: I tried to follow the code style, but I'm naughty
and used a few advanced C++ features repeatedly:
- Destructors to make it easier to ensure everything gets cleaned up.
- Exceptions. I don't even usually use exceptions when writing C++, and
I can remove them if desired... but here they make the loading code
much more succinct while still properly handling a variety of errors,
ranging from API calls failing to integer overflow and allocation
failure. The exceptions are converted to error codes at the
API boundary.)
Co-authored-by: Pavol Rusnak <pavol@rusnak.io> (for the bit I copied from #740)
2023-04-08 21:24:37 +02:00
|
|
|
if (llama_mlock_supported()) {
|
2023-03-24 16:19:05 +01:00
|
|
|
fprintf(stderr, " --mlock force system to keep model in RAM rather than swapping or compressing\n");
|
|
|
|
}
|
Rewrite loading code to try to satisfy everyone:
- Support all three formats (ggml, ggmf, ggjt). (However, I didn't
include the hack needed to support GPT4All files without conversion.
Those can still be used after converting them with convert.py from my
other PR.)
- Support both mmap and read (mmap is used by default, but can be
disabled with `--no-mmap`, and is automatically disabled for pre-ggjt
files or on platforms where mmap is not supported).
- Support multi-file models like before, but automatically determine the
number of parts rather than requiring `--n_parts`.
- Improve validation and error checking.
- Stop using the per-file type field (f16) entirely in favor of just
relying on the per-tensor type/size fields. This has no immediate
benefit, but makes it easier to experiment with different formats, and
should make it easier to support the new GPTQ-for-LLaMa models in the
future (I have some work in progress on that front).
- Support VirtualLock on Windows (using the same `--mlock` option as on
Unix).
- Indicate loading progress when using mmap + mlock. (Which led me
to the interesting observation that on my Linux machine, with a
warm file cache, mlock actually takes some time, whereas mmap
without mlock starts almost instantly...)
- To help implement this, move mlock support from ggml to the
loading code.
- madvise/PrefetchVirtualMemory support (based on #740)
- Switch from ifstream to the `fopen` family of functions to avoid
unnecessary copying and, when mmap is enabled, allow reusing the same
file descriptor for both metadata reads and mmap (whereas the existing
implementation opens the file a second time to mmap).
- Quantization now produces a single-file output even with multi-file
inputs (not really a feature as much as 'it was easier this way').
Implementation notes:
I tried to factor the code into more discrete pieces than before.
Regarding code style: I tried to follow the code style, but I'm naughty
and used a few advanced C++ features repeatedly:
- Destructors to make it easier to ensure everything gets cleaned up.
- Exceptions. I don't even usually use exceptions when writing C++, and
I can remove them if desired... but here they make the loading code
much more succinct while still properly handling a variety of errors,
ranging from API calls failing to integer overflow and allocation
failure. The exceptions are converted to error codes at the
API boundary.)
Co-authored-by: Pavol Rusnak <pavol@rusnak.io> (for the bit I copied from #740)
2023-04-08 21:24:37 +02:00
|
|
|
if (llama_mmap_supported()) {
|
|
|
|
fprintf(stderr, " --no-mmap do not memory-map model (slower load but may reduce pageouts if not using mlock)\n");
|
|
|
|
}
|
2023-03-24 22:17:37 +01:00
|
|
|
fprintf(stderr, " --mtest compute maximum memory usage\n");
|
2023-03-25 16:16:50 +01:00
|
|
|
fprintf(stderr, " --verbose-prompt print prompt before generation\n");
|
2023-03-10 19:40:58 +01:00
|
|
|
fprintf(stderr, " -m FNAME, --model FNAME\n");
|
|
|
|
fprintf(stderr, " model path (default: %s)\n", params.model.c_str());
|
|
|
|
fprintf(stderr, "\n");
|
|
|
|
}
|
|
|
|
|
|
|
|
std::string gpt_random_prompt(std::mt19937 & rng) {
|
|
|
|
const int r = rng() % 10;
|
|
|
|
switch (r) {
|
|
|
|
case 0: return "So";
|
|
|
|
case 1: return "Once upon a time";
|
|
|
|
case 2: return "When";
|
|
|
|
case 3: return "The";
|
|
|
|
case 4: return "After";
|
|
|
|
case 5: return "If";
|
|
|
|
case 6: return "import";
|
|
|
|
case 7: return "He";
|
|
|
|
case 8: return "She";
|
|
|
|
case 9: return "They";
|
|
|
|
default: return "To";
|
|
|
|
}
|
|
|
|
|
|
|
|
return "The";
|
|
|
|
}
|
|
|
|
|
2023-03-22 06:32:36 +01:00
|
|
|
// TODO: not great allocating this every time
|
|
|
|
std::vector<llama_token> llama_tokenize(struct llama_context * ctx, const std::string & text, bool add_bos) {
|
2023-03-22 17:09:38 +01:00
|
|
|
// initialize to prompt numer of chars, since n_tokens <= n_prompt_chars
|
|
|
|
std::vector<llama_token> res(text.size() + (int)add_bos);
|
2023-03-22 06:32:36 +01:00
|
|
|
int n = llama_tokenize(ctx, text.c_str(), res.data(), res.size(), add_bos);
|
2023-03-22 17:09:38 +01:00
|
|
|
assert(n >= 0);
|
2023-03-22 06:32:36 +01:00
|
|
|
res.resize(n);
|
2023-03-10 19:40:58 +01:00
|
|
|
|
2023-03-22 06:32:36 +01:00
|
|
|
return res;
|
2023-03-10 19:40:58 +01:00
|
|
|
}
|
2023-03-28 16:09:55 +02:00
|
|
|
|
|
|
|
/* Keep track of current color of output, and emit ANSI code if it changes. */
|
|
|
|
void set_console_color(console_state & con_st, console_color_t color) {
|
|
|
|
if (con_st.use_color && con_st.color != color) {
|
|
|
|
switch(color) {
|
|
|
|
case CONSOLE_COLOR_DEFAULT:
|
|
|
|
printf(ANSI_COLOR_RESET);
|
|
|
|
break;
|
|
|
|
case CONSOLE_COLOR_PROMPT:
|
|
|
|
printf(ANSI_COLOR_YELLOW);
|
|
|
|
break;
|
|
|
|
case CONSOLE_COLOR_USER_INPUT:
|
|
|
|
printf(ANSI_BOLD ANSI_COLOR_GREEN);
|
|
|
|
break;
|
|
|
|
}
|
|
|
|
con_st.color = color;
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
#if defined (_WIN32)
|
|
|
|
void win32_console_init(bool enable_color) {
|
|
|
|
unsigned long dwMode = 0;
|
|
|
|
void* hConOut = GetStdHandle((unsigned long)-11); // STD_OUTPUT_HANDLE (-11)
|
|
|
|
if (!hConOut || hConOut == (void*)-1 || !GetConsoleMode(hConOut, &dwMode)) {
|
|
|
|
hConOut = GetStdHandle((unsigned long)-12); // STD_ERROR_HANDLE (-12)
|
|
|
|
if (hConOut && (hConOut == (void*)-1 || !GetConsoleMode(hConOut, &dwMode))) {
|
|
|
|
hConOut = 0;
|
|
|
|
}
|
|
|
|
}
|
|
|
|
if (hConOut) {
|
|
|
|
// Enable ANSI colors on Windows 10+
|
|
|
|
if (enable_color && !(dwMode & 0x4)) {
|
|
|
|
SetConsoleMode(hConOut, dwMode | 0x4); // ENABLE_VIRTUAL_TERMINAL_PROCESSING (0x4)
|
|
|
|
}
|
|
|
|
// Set console output codepage to UTF8
|
2023-04-08 17:49:39 +02:00
|
|
|
SetConsoleOutputCP(CP_UTF8);
|
2023-03-28 16:09:55 +02:00
|
|
|
}
|
|
|
|
void* hConIn = GetStdHandle((unsigned long)-10); // STD_INPUT_HANDLE (-10)
|
|
|
|
if (hConIn && hConIn != (void*)-1 && GetConsoleMode(hConIn, &dwMode)) {
|
2023-04-08 17:49:39 +02:00
|
|
|
// Set console input codepage to UTF16
|
|
|
|
_setmode(_fileno(stdin), _O_WTEXT);
|
2023-03-28 16:09:55 +02:00
|
|
|
}
|
|
|
|
}
|
2023-04-08 17:49:39 +02:00
|
|
|
|
|
|
|
// Convert a wide Unicode string to an UTF8 string
|
|
|
|
void win32_utf8_encode(const std::wstring & wstr, std::string & str) {
|
2023-04-11 21:45:44 +02:00
|
|
|
int size_needed = WideCharToMultiByte(CP_UTF8, 0, &wstr[0], (int)wstr.size(), NULL, 0, NULL, NULL);
|
|
|
|
std::string strTo(size_needed, 0);
|
|
|
|
WideCharToMultiByte(CP_UTF8, 0, &wstr[0], (int)wstr.size(), &strTo[0], size_needed, NULL, NULL);
|
|
|
|
str = strTo;
|
2023-04-08 17:49:39 +02:00
|
|
|
}
|
2023-03-28 16:09:55 +02:00
|
|
|
#endif
|
2023-04-14 17:19:17 +02:00
|
|
|
|
|
|
|
bool get_input_text(std::string & input_text, bool eof_toggled_multiline_mode) {
|
|
|
|
bool another_line = true;
|
|
|
|
bool is_eof_multiline_toggled = false;
|
|
|
|
do {
|
|
|
|
std::string line;
|
|
|
|
#if defined(_WIN32)
|
|
|
|
auto & stdcin = std::wcin;
|
|
|
|
std::wstring wline;
|
|
|
|
if (!std::getline(stdcin, wline)) {
|
|
|
|
// input stream is bad or EOF received
|
|
|
|
if (stdcin.bad()) {
|
|
|
|
fprintf(stderr, "%s: error: input stream bad\n", __func__);
|
|
|
|
return 1;
|
|
|
|
}
|
|
|
|
}
|
|
|
|
win32_utf8_encode(wline, line);
|
|
|
|
#else
|
|
|
|
auto & stdcin = std::cin;
|
|
|
|
if (!std::getline(stdcin, line)) {
|
|
|
|
// input stream is bad or EOF received
|
|
|
|
if (stdcin.bad()) {
|
|
|
|
fprintf(stderr, "%s: error: input stream bad\n", __func__);
|
|
|
|
return 1;
|
|
|
|
}
|
|
|
|
}
|
|
|
|
#endif
|
|
|
|
if (stdcin.eof()) {
|
|
|
|
stdcin.clear();
|
|
|
|
stdcin.seekg(0, std::ios::beg);
|
|
|
|
if (!eof_toggled_multiline_mode) {
|
|
|
|
another_line = false;
|
|
|
|
} else {
|
|
|
|
is_eof_multiline_toggled = !is_eof_multiline_toggled;
|
|
|
|
if (is_eof_multiline_toggled) {
|
|
|
|
input_text += line;
|
|
|
|
continue;
|
|
|
|
}
|
|
|
|
}
|
|
|
|
}
|
|
|
|
if (!eof_toggled_multiline_mode) {
|
|
|
|
if (line.empty() || line.back() != '\\') {
|
|
|
|
another_line = false;
|
|
|
|
} else {
|
|
|
|
line.pop_back(); // Remove the continue character
|
|
|
|
}
|
|
|
|
} else {
|
|
|
|
if (!is_eof_multiline_toggled) {
|
|
|
|
another_line = false;
|
|
|
|
}
|
|
|
|
}
|
|
|
|
input_text += line;
|
|
|
|
if (another_line) {
|
|
|
|
input_text += '\n'; // Append the line to the result
|
|
|
|
}
|
|
|
|
} while (another_line);
|
|
|
|
return true;
|
|
|
|
}
|