This didn't really feel so worth it afterwards, but I did untangle a bunch of stuff that should not have been tangled. The general gist of this change is that variant bullshit was causing a bunch of compile time, and it seems like the only way to deal with variant induced compile time is to keep variant types out of headers. Explicit template instantiation seems to do nothing for them. I also seem to have gotten some back-end time improvement from explicitly instantiating regex, but I don't know why. There is no corresponding front-end time improvement from it: regex is still at the top of the sinners list. **** Templates that took longest to instantiate: 15231 ms: std::basic_regex<char>::_M_compile (28 times, avg 543 ms) 15066 ms: std::__detail::_Compiler<std::regex_traits<char>>::_Compiler (28 times, avg 538 ms) 12571 ms: std::__detail::_Compiler<std::regex_traits<char>>::_M_disjunction (28 times, avg 448 ms) 12454 ms: std::__detail::_Compiler<std::regex_traits<char>>::_M_alternative (28 times, avg 444 ms) 12225 ms: std::__detail::_Compiler<std::regex_traits<char>>::_M_term (28 times, avg 436 ms) 11363 ms: nlohmann::basic_json<>::parse<const char *> (21 times, avg 541 ms) 10628 ms: nlohmann::basic_json<>::basic_json (109 times, avg 97 ms) 10134 ms: std::__detail::_Compiler<std::regex_traits<char>>::_M_atom (28 times, avg 361 ms) Back-end time before messing with the regex: **** Function sets that took longest to compile / optimize: 8076 ms: void boost::io::detail::put<$>(boost::io::detail::put_holder<$> cons... (177 times, avg 45 ms) 4382 ms: std::_Rb_tree<$>::_M_erase(std::_Rb_tree_node<$>*) (1247 times, avg 3 ms) 3137 ms: boost::stacktrace::detail::to_string_impl_base<boost::stacktrace::de... (137 times, avg 22 ms) 2896 ms: void boost::io::detail::mk_str<$>(std::__cxx11::basic_string<$>&, ch... (177 times, avg 16 ms) 2304 ms: std::_Rb_tree<$>::_M_get_insert_hint_unique_pos(std::_Rb_tree_const_... (210 times, avg 10 ms) 2116 ms: bool std::__detail::_Compiler<$>::_M_expression_term<$>(std::__detai... (112 times, avg 18 ms) 2051 ms: std::_Rb_tree_iterator<$> std::_Rb_tree<$>::_M_emplace_hint_unique<$... (244 times, avg 8 ms) 2037 ms: toml::result<$> toml::detail::sequence<$>::invoke<$>(toml::detail::l... (93 times, avg 21 ms) 1928 ms: std::__detail::_Compiler<$>::_M_quantifier() (28 times, avg 68 ms) 1859 ms: nlohmann::json_abi_v3_11_3::detail::serializer<$>::dump(nlohmann::js... (41 times, avg 45 ms) 1824 ms: std::_Function_handler<$>::_M_manager(std::_Any_data&, std::_Any_dat... (973 times, avg 1 ms) 1810 ms: std::__detail::_BracketMatcher<$>::_BracketMatcher(std::__detail::_B... (112 times, avg 16 ms) 1793 ms: nix::fetchers::GitInputScheme::fetch(nix::ref<$>, nix::fetchers::Inp... (1 times, avg 1793 ms) 1759 ms: std::_Rb_tree<$>::_M_get_insert_unique_pos(std::__cxx11::basic_strin... (281 times, avg 6 ms) 1722 ms: bool nlohmann::json_abi_v3_11_3::detail::parser<$>::sax_parse_intern... (19 times, avg 90 ms) 1677 ms: boost::io::basic_altstringbuf<$>::overflow(int) (194 times, avg 8 ms) 1674 ms: std::__cxx11::basic_string<$>::_M_mutate(unsigned long, unsigned lon... (249 times, avg 6 ms) 1660 ms: std::_Rb_tree_node<$>* std::_Rb_tree<$>::_M_copy<$>(std::_Rb_tree_no... (304 times, avg 5 ms) 1599 ms: bool nlohmann::json_abi_v3_11_3::detail::parser<$>::sax_parse_intern... (19 times, avg 84 ms) 1568 ms: void std::__detail::_Compiler<$>::_M_insert_bracket_matcher<$>(bool) (112 times, avg 14 ms) 1541 ms: std::__shared_ptr<$>::~__shared_ptr() (531 times, avg 2 ms) 1539 ms: nlohmann::json_abi_v3_11_3::detail::serializer<$>::dump_escaped(std:... (41 times, avg 37 ms) 1471 ms: void std::__detail::_Compiler<$>::_M_insert_character_class_matcher<... (112 times, avg 13 ms) After messing with the regex (notice std::__detail::_Compiler vanishes here, but I don't know why): **** Function sets that took longest to compile / optimize: 8054 ms: void boost::io::detail::put<$>(boost::io::detail::put_holder<$> cons... (177 times, avg 45 ms) 4313 ms: std::_Rb_tree<$>::_M_erase(std::_Rb_tree_node<$>*) (1217 times, avg 3 ms) 3259 ms: boost::stacktrace::detail::to_string_impl_base<boost::stacktrace::de... (137 times, avg 23 ms) 3045 ms: void boost::io::detail::mk_str<$>(std::__cxx11::basic_string<$>&, ch... (177 times, avg 17 ms) 2314 ms: std::_Rb_tree<$>::_M_get_insert_hint_unique_pos(std::_Rb_tree_const_... (207 times, avg 11 ms) 1923 ms: std::_Rb_tree_iterator<$> std::_Rb_tree<$>::_M_emplace_hint_unique<$... (216 times, avg 8 ms) 1817 ms: bool nlohmann::json_abi_v3_11_3::detail::parser<$>::sax_parse_intern... (18 times, avg 100 ms) 1816 ms: toml::result<$> toml::detail::sequence<$>::invoke<$>(toml::detail::l... (93 times, avg 19 ms) 1788 ms: nlohmann::json_abi_v3_11_3::detail::serializer<$>::dump(nlohmann::js... (40 times, avg 44 ms) 1749 ms: std::_Rb_tree<$>::_M_get_insert_unique_pos(std::__cxx11::basic_strin... (278 times, avg 6 ms) 1724 ms: std::__cxx11::basic_string<$>::_M_mutate(unsigned long, unsigned lon... (248 times, avg 6 ms) 1697 ms: boost::io::basic_altstringbuf<$>::overflow(int) (194 times, avg 8 ms) 1684 ms: nix::fetchers::GitInputScheme::fetch(nix::ref<$>, nix::fetchers::Inp... (1 times, avg 1684 ms) 1680 ms: std::_Rb_tree_node<$>* std::_Rb_tree<$>::_M_copy<$>(std::_Rb_tree_no... (303 times, avg 5 ms) 1589 ms: bool nlohmann::json_abi_v3_11_3::detail::parser<$>::sax_parse_intern... (18 times, avg 88 ms) 1483 ms: non-virtual thunk to boost::wrapexcept<$>::~wrapexcept() (181 times, avg 8 ms) 1447 ms: nlohmann::json_abi_v3_11_3::detail::serializer<$>::dump_escaped(std:... (40 times, avg 36 ms) 1441 ms: std::__shared_ptr<$>::~__shared_ptr() (496 times, avg 2 ms) 1420 ms: boost::stacktrace::basic_stacktrace<$>::init(unsigned long, unsigned... (137 times, avg 10 ms) 1396 ms: boost::basic_format<$>::~basic_format() (194 times, avg 7 ms) 1290 ms: std::__cxx11::basic_string<$>::_M_replace_cold(char*, unsigned long,... (231 times, avg 5 ms) 1258 ms: std::vector<$>::~vector() (354 times, avg 3 ms) 1222 ms: std::__cxx11::basic_string<$>::_M_replace(unsigned long, unsigned lo... (231 times, avg 5 ms) 1194 ms: std::_Rb_tree<$>::_M_get_insert_hint_unique_pos(std::_Rb_tree_const_... (49 times, avg 24 ms) 1186 ms: bool tao::pegtl::internal::sor<$>::match<$>(std::integer_sequence<$>... (1 times, avg 1186 ms) 1149 ms: std::__detail::_Executor<$>::_M_dfs(std::__detail::_Executor<$>::_Ma... (70 times, avg 16 ms) 1123 ms: toml::detail::sequence<$>::invoke(toml::detail::location&) (69 times, avg 16 ms) 1110 ms: nlohmann::json_abi_v3_11_3::basic_json<$>::json_value::destroy(nlohm... (55 times, avg 20 ms) 1079 ms: std::_Function_handler<$>::_M_manager(std::_Any_data&, std::_Any_dat... (541 times, avg 1 ms) 1033 ms: nlohmann::json_abi_v3_11_3::detail::lexer<$>::scan_number() (20 times, avg 51 ms) Change-Id: I10af282bcd4fc39c2d3caae3453e599e4639c70b
422 lines
10 KiB
C++
422 lines
10 KiB
C++
#include <cstring>
|
|
|
|
#include <openssl/crypto.h>
|
|
#include <openssl/md5.h>
|
|
#include <openssl/sha.h>
|
|
|
|
#include "args.hh"
|
|
#include "hash.hh"
|
|
#include "archive.hh"
|
|
#include "charptr-cast.hh"
|
|
#include "logging.hh"
|
|
#include "split.hh"
|
|
#include "strings.hh"
|
|
|
|
#include <sys/types.h>
|
|
#include <sys/stat.h>
|
|
#include <fcntl.h>
|
|
|
|
namespace nix {
|
|
|
|
static size_t regularHashSize(HashType type) {
|
|
switch (type) {
|
|
case HashType::MD5: return md5HashSize;
|
|
case HashType::SHA1: return sha1HashSize;
|
|
case HashType::SHA256: return sha256HashSize;
|
|
case HashType::SHA512: return sha512HashSize;
|
|
}
|
|
abort();
|
|
}
|
|
|
|
|
|
std::set<std::string> hashTypes = { "md5", "sha1", "sha256", "sha512" };
|
|
|
|
|
|
Hash::Hash(HashType type) : type(type)
|
|
{
|
|
hashSize = regularHashSize(type);
|
|
assert(hashSize <= maxHashSize);
|
|
memset(hash, 0, maxHashSize);
|
|
}
|
|
|
|
|
|
bool Hash::operator == (const Hash & h2) const
|
|
{
|
|
if (hashSize != h2.hashSize) return false;
|
|
for (unsigned int i = 0; i < hashSize; i++)
|
|
if (hash[i] != h2.hash[i]) return false;
|
|
return true;
|
|
}
|
|
|
|
|
|
bool Hash::operator != (const Hash & h2) const
|
|
{
|
|
return !(*this == h2);
|
|
}
|
|
|
|
|
|
bool Hash::operator < (const Hash & h) const
|
|
{
|
|
if (hashSize < h.hashSize) return true;
|
|
if (hashSize > h.hashSize) return false;
|
|
for (unsigned int i = 0; i < hashSize; i++) {
|
|
if (hash[i] < h.hash[i]) return true;
|
|
if (hash[i] > h.hash[i]) return false;
|
|
}
|
|
return false;
|
|
}
|
|
|
|
|
|
const std::string base16Chars = "0123456789abcdef";
|
|
|
|
|
|
static std::string printHash16(const Hash & hash)
|
|
{
|
|
std::string buf;
|
|
buf.reserve(hash.hashSize * 2);
|
|
for (unsigned int i = 0; i < hash.hashSize; i++) {
|
|
buf.push_back(base16Chars[hash.hash[i] >> 4]);
|
|
buf.push_back(base16Chars[hash.hash[i] & 0x0f]);
|
|
}
|
|
return buf;
|
|
}
|
|
|
|
|
|
// omitted: E O U T
|
|
const std::string base32Chars = "0123456789abcdfghijklmnpqrsvwxyz";
|
|
|
|
|
|
static std::string printHash32(const Hash & hash)
|
|
{
|
|
assert(hash.hashSize);
|
|
size_t len = hash.base32Len();
|
|
assert(len);
|
|
|
|
std::string s;
|
|
s.reserve(len);
|
|
|
|
for (int n = (int) len - 1; n >= 0; n--) {
|
|
unsigned int b = n * 5;
|
|
unsigned int i = b / 8;
|
|
unsigned int j = b % 8;
|
|
unsigned char c =
|
|
(hash.hash[i] >> j)
|
|
| (i >= hash.hashSize - 1 ? 0 : hash.hash[i + 1] << (8 - j));
|
|
s.push_back(base32Chars[c & 0x1f]);
|
|
}
|
|
|
|
return s;
|
|
}
|
|
|
|
|
|
std::string printHash16or32(const Hash & hash)
|
|
{
|
|
return hash.to_string(hash.type == HashType::MD5 ? Base::Base16 : Base::Base32, false);
|
|
}
|
|
|
|
|
|
std::string Hash::to_string(Base base, bool includeType) const
|
|
{
|
|
std::string s;
|
|
if (base == Base::SRI || includeType) {
|
|
s += printHashType(type);
|
|
s += base == Base::SRI ? '-' : ':';
|
|
}
|
|
switch (base) {
|
|
case Base::Base16:
|
|
s += printHash16(*this);
|
|
break;
|
|
case Base::Base32:
|
|
s += printHash32(*this);
|
|
break;
|
|
case Base::Base64:
|
|
case Base::SRI:
|
|
s += base64Encode(std::string_view(charptr_cast<const char *>(hash), hashSize));
|
|
break;
|
|
}
|
|
return s;
|
|
}
|
|
|
|
Hash Hash::dummy(HashType::SHA256);
|
|
|
|
Hash Hash::parseSRI(std::string_view original) {
|
|
auto rest = original;
|
|
|
|
// Parse the has type before the separater, if there was one.
|
|
auto hashRaw = splitPrefixTo(rest, '-');
|
|
if (!hashRaw)
|
|
throw BadHash("hash '%s' is not SRI", original);
|
|
HashType parsedType = parseHashType(*hashRaw);
|
|
|
|
return Hash(rest, parsedType, true);
|
|
}
|
|
|
|
// Mutates the string to eliminate the prefixes when found
|
|
static std::pair<std::optional<HashType>, bool> getParsedTypeAndSRI(std::string_view & rest)
|
|
{
|
|
bool isSRI = false;
|
|
|
|
// Parse the hash type before the separator, if there was one.
|
|
std::optional<HashType> optParsedType;
|
|
{
|
|
auto hashRaw = splitPrefixTo(rest, ':');
|
|
|
|
if (!hashRaw) {
|
|
hashRaw = splitPrefixTo(rest, '-');
|
|
if (hashRaw)
|
|
isSRI = true;
|
|
}
|
|
if (hashRaw)
|
|
optParsedType = parseHashType(*hashRaw);
|
|
}
|
|
|
|
return {optParsedType, isSRI};
|
|
}
|
|
|
|
Hash Hash::parseAnyPrefixed(std::string_view original)
|
|
{
|
|
auto rest = original;
|
|
auto [optParsedType, isSRI] = getParsedTypeAndSRI(rest);
|
|
|
|
// Either the string or user must provide the type, if they both do they
|
|
// must agree.
|
|
if (!optParsedType)
|
|
throw BadHash("hash '%s' does not include a type", rest);
|
|
|
|
return Hash(rest, *optParsedType, isSRI);
|
|
}
|
|
|
|
Hash Hash::parseAny(std::string_view original, std::optional<HashType> optType)
|
|
{
|
|
auto rest = original;
|
|
auto [optParsedType, isSRI] = getParsedTypeAndSRI(rest);
|
|
|
|
// Either the string or user must provide the type, if they both do they
|
|
// must agree.
|
|
if (!optParsedType && !optType)
|
|
throw BadHash("hash '%s' does not include a type, nor is the type otherwise known from context", rest);
|
|
else if (optParsedType && optType && *optParsedType != *optType)
|
|
throw BadHash("hash '%s' should have type '%s'", original, printHashType(*optType));
|
|
|
|
HashType hashType = optParsedType ? *optParsedType : *optType;
|
|
return Hash(rest, hashType, isSRI);
|
|
}
|
|
|
|
Hash Hash::parseNonSRIUnprefixed(std::string_view s, HashType type)
|
|
{
|
|
return Hash(s, type, false);
|
|
}
|
|
|
|
Hash::Hash(std::string_view rest, HashType type, bool isSRI)
|
|
: Hash(type)
|
|
{
|
|
if (!isSRI && rest.size() == base16Len()) {
|
|
|
|
auto parseHexDigit = [&](char c) {
|
|
if (c >= '0' && c <= '9') return c - '0';
|
|
if (c >= 'A' && c <= 'F') return c - 'A' + 10;
|
|
if (c >= 'a' && c <= 'f') return c - 'a' + 10;
|
|
throw BadHash("invalid base-16 hash '%s'", rest);
|
|
};
|
|
|
|
for (unsigned int i = 0; i < hashSize; i++) {
|
|
hash[i] =
|
|
parseHexDigit(rest[i * 2]) << 4
|
|
| parseHexDigit(rest[i * 2 + 1]);
|
|
}
|
|
}
|
|
|
|
else if (!isSRI && rest.size() == base32Len()) {
|
|
|
|
for (unsigned int n = 0; n < rest.size(); ++n) {
|
|
char c = rest[rest.size() - n - 1];
|
|
size_t digit;
|
|
for (digit = 0; digit < base32Chars.size(); ++digit) /* !!! slow */
|
|
if (base32Chars[digit] == c) break;
|
|
if (digit >= 32)
|
|
throw BadHash("invalid base-32 hash '%s'", rest);
|
|
unsigned int b = n * 5;
|
|
unsigned int i = b / 8;
|
|
unsigned int j = b % 8;
|
|
hash[i] |= digit << j;
|
|
|
|
if (i < hashSize - 1) {
|
|
hash[i + 1] |= digit >> (8 - j);
|
|
} else {
|
|
if (digit >> (8 - j))
|
|
throw BadHash("invalid base-32 hash '%s'", rest);
|
|
}
|
|
}
|
|
}
|
|
|
|
else if (isSRI || rest.size() == base64Len()) {
|
|
auto d = base64Decode(rest);
|
|
if (d.size() != hashSize)
|
|
throw BadHash("invalid %s hash '%s'", isSRI ? "SRI" : "base-64", rest);
|
|
assert(hashSize);
|
|
memcpy(hash, d.data(), hashSize);
|
|
}
|
|
|
|
else
|
|
throw BadHash("hash '%s' has wrong length for hash type '%s'", rest, printHashType(this->type));
|
|
}
|
|
|
|
Hash newHashAllowEmpty(std::string_view hashStr, std::optional<HashType> ht)
|
|
{
|
|
if (hashStr.empty()) {
|
|
if (!ht)
|
|
throw BadHash("empty hash requires explicit hash type");
|
|
Hash h(*ht);
|
|
warn("found empty hash, assuming '%s'", h.to_string(Base::SRI, true));
|
|
return h;
|
|
} else
|
|
return Hash::parseAny(hashStr, ht);
|
|
}
|
|
|
|
|
|
union Ctx
|
|
{
|
|
MD5_CTX md5;
|
|
SHA_CTX sha1;
|
|
SHA256_CTX sha256;
|
|
SHA512_CTX sha512;
|
|
};
|
|
|
|
|
|
static void start(HashType ht, Ctx & ctx)
|
|
{
|
|
if (ht == HashType::MD5) MD5_Init(&ctx.md5);
|
|
else if (ht == HashType::SHA1) SHA1_Init(&ctx.sha1);
|
|
else if (ht == HashType::SHA256) SHA256_Init(&ctx.sha256);
|
|
else if (ht == HashType::SHA512) SHA512_Init(&ctx.sha512);
|
|
}
|
|
|
|
|
|
static void update(HashType ht, Ctx & ctx,
|
|
std::string_view data)
|
|
{
|
|
if (ht == HashType::MD5) MD5_Update(&ctx.md5, data.data(), data.size());
|
|
else if (ht == HashType::SHA1) SHA1_Update(&ctx.sha1, data.data(), data.size());
|
|
else if (ht == HashType::SHA256) SHA256_Update(&ctx.sha256, data.data(), data.size());
|
|
else if (ht == HashType::SHA512) SHA512_Update(&ctx.sha512, data.data(), data.size());
|
|
}
|
|
|
|
|
|
static void finish(HashType ht, Ctx & ctx, unsigned char * hash)
|
|
{
|
|
if (ht == HashType::MD5) MD5_Final(hash, &ctx.md5);
|
|
else if (ht == HashType::SHA1) SHA1_Final(hash, &ctx.sha1);
|
|
else if (ht == HashType::SHA256) SHA256_Final(hash, &ctx.sha256);
|
|
else if (ht == HashType::SHA512) SHA512_Final(hash, &ctx.sha512);
|
|
}
|
|
|
|
|
|
Hash hashString(HashType ht, std::string_view s)
|
|
{
|
|
Ctx ctx;
|
|
Hash hash(ht);
|
|
start(ht, ctx);
|
|
update(ht, ctx, s);
|
|
finish(ht, ctx, hash.hash);
|
|
return hash;
|
|
}
|
|
|
|
|
|
Hash hashFile(HashType ht, const Path & path)
|
|
{
|
|
HashSink sink(ht);
|
|
sink << readFileSource(path);
|
|
return sink.finish().first;
|
|
}
|
|
|
|
|
|
HashSink::HashSink(HashType ht) : ht(ht)
|
|
{
|
|
ctx = new Ctx;
|
|
bytes = 0;
|
|
start(ht, *ctx);
|
|
}
|
|
|
|
HashSink::~HashSink()
|
|
{
|
|
bufPos = 0;
|
|
delete ctx;
|
|
}
|
|
|
|
void HashSink::writeUnbuffered(std::string_view data)
|
|
{
|
|
bytes += data.size();
|
|
update(ht, *ctx, data);
|
|
}
|
|
|
|
HashResult HashSink::finish()
|
|
{
|
|
flush();
|
|
Hash hash(ht);
|
|
nix::finish(ht, *ctx, hash.hash);
|
|
return HashResult(hash, bytes);
|
|
}
|
|
|
|
HashResult HashSink::currentHash()
|
|
{
|
|
flush();
|
|
Ctx ctx2 = *ctx;
|
|
Hash hash(ht);
|
|
nix::finish(ht, ctx2, hash.hash);
|
|
return HashResult(hash, bytes);
|
|
}
|
|
|
|
|
|
HashResult hashPath(
|
|
HashType ht, const Path & path, PathFilter & filter)
|
|
{
|
|
HashSink sink(ht);
|
|
sink << dumpPath(path, filter);
|
|
return sink.finish();
|
|
}
|
|
|
|
|
|
Hash compressHash(const Hash & hash, unsigned int newSize)
|
|
{
|
|
Hash h(hash.type);
|
|
h.hashSize = newSize;
|
|
for (unsigned int i = 0; i < hash.hashSize; ++i)
|
|
h.hash[i % newSize] ^= hash.hash[i];
|
|
return h;
|
|
}
|
|
|
|
|
|
std::optional<HashType> parseHashTypeOpt(std::string_view s)
|
|
{
|
|
if (s == "md5") return HashType::MD5;
|
|
else if (s == "sha1") return HashType::SHA1;
|
|
else if (s == "sha256") return HashType::SHA256;
|
|
else if (s == "sha512") return HashType::SHA512;
|
|
else return std::optional<HashType> {};
|
|
}
|
|
|
|
HashType parseHashType(std::string_view s)
|
|
{
|
|
auto opt_h = parseHashTypeOpt(s);
|
|
if (opt_h)
|
|
return *opt_h;
|
|
else
|
|
throw UsageError("unknown hash algorithm '%1%'", s);
|
|
}
|
|
|
|
std::string_view printHashType(HashType ht)
|
|
{
|
|
switch (ht) {
|
|
case HashType::MD5: return "md5";
|
|
case HashType::SHA1: return "sha1";
|
|
case HashType::SHA256: return "sha256";
|
|
case HashType::SHA512: return "sha512";
|
|
default:
|
|
// illegal hash type enum value internally, as opposed to external input
|
|
// which should be validated with nice error message.
|
|
assert(false);
|
|
}
|
|
}
|
|
|
|
}
|