ccminer-gostd-lite/cuda_nist5.cu

extern "C"
{
#include "sph/sph_blake.h"
#include "sph/sph_groestl.h"
#include "sph/sph_skein.h"
#include "sph/sph_jh.h"
#include "sph/sph_keccak.h"
}

#include "miner.h"

#include "cuda_helper.h"
#include "quark/cuda_quark.h"

static uint32_t *d_hash[MAX_GPUS];

// Original nist5hash Funktion aus einem miner Quelltext
extern "C" void nist5hash(void *state, const void *input)
{
    sph_blake512_context ctx_blake;
    sph_groestl512_context ctx_groestl;
    sph_jh512_context ctx_jh;
    sph_keccak512_context ctx_keccak;
    sph_skein512_context ctx_skein;
    
    uint8_t hash[64];

    sph_blake512_init(&ctx_blake);
    sph_blake512 (&ctx_blake, input, 80);
    sph_blake512_close(&ctx_blake, (void*) hash);
    
    sph_groestl512_init(&ctx_groestl);
    sph_groestl512 (&ctx_groestl, (const void*) hash, 64);
    sph_groestl512_close(&ctx_groestl, (void*) hash);

    sph_jh512_init(&ctx_jh);
    sph_jh512 (&ctx_jh, (const void*) hash, 64);
    sph_jh512_close(&ctx_jh, (void*) hash);

    sph_keccak512_init(&ctx_keccak);
    sph_keccak512 (&ctx_keccak, (const void*) hash, 64);
    sph_keccak512_close(&ctx_keccak, (void*) hash);

    sph_skein512_init(&ctx_skein);
    sph_skein512 (&ctx_skein, (const void*) hash, 64);
    sph_skein512_close(&ctx_skein, (void*) hash);

    memcpy(state, hash, 32);
}

static bool init[MAX_GPUS] = { 0 };

extern "C" int scanhash_nist5(int thr_id, struct work *work, uint32_t max_nonce, unsigned long *hashes_done)
{
	uint32_t _ALIGN(64) endiandata[20];
	uint32_t *pdata = work->data;
	uint32_t *ptarget = work->target;
	const uint32_t first_nonce = pdata[19];
	int res = 0;

	uint32_t throughput =  cuda_default_throughput(thr_id, 1 << 20); // 256*256*16
	if (init[thr_id]) throughput = min(throughput, max_nonce - first_nonce);

	if (opt_benchmark)
		((uint32_t*)ptarget)[7] = 0x00FF;

	if (!init[thr_id])
	{
		cudaDeviceSynchronize();
		cudaSetDevice(device_map[thr_id]);

		// Constants copy/init (no device alloc in these algos)
		quark_blake512_cpu_init(thr_id, throughput);
		quark_groestl512_cpu_init(thr_id, throughput);
		quark_jh512_cpu_init(thr_id, throughput);
		quark_keccak512_cpu_init(thr_id, throughput);
		quark_skein512_cpu_init(thr_id, throughput);

		// char[64] work space for hashes results
		CUDA_SAFE_CALL(cudaMalloc(&d_hash[thr_id], (size_t)64 * throughput));

		cuda_check_cpu_init(thr_id, throughput);
		init[thr_id] = true;
	}

#ifdef USE_STREAMS
	cudaStream_t stream[5];
	for (int i = 0; i < 5; i++)
		cudaStreamCreate(&stream[i]);
#endif

	for (int k=0; k < 20; k++)
		be32enc(&endiandata[k], pdata[k]);

	quark_blake512_cpu_setBlock_80(thr_id, endiandata);
	cuda_check_cpu_setTarget(ptarget);

	do {
		int order = 0;

		// Hash with CUDA
		quark_blake512_cpu_hash_80(thr_id, throughput, pdata[19], d_hash[thr_id]); order++;
		quark_groestl512_cpu_hash_64(thr_id, throughput, pdata[19], NULL, d_hash[thr_id], order++);
		quark_jh512_cpu_hash_64(thr_id, throughput, pdata[19], NULL, d_hash[thr_id], order++);
		quark_keccak512_cpu_hash_64(thr_id, throughput, pdata[19], NULL, d_hash[thr_id], order++);
		quark_skein512_cpu_hash_64(thr_id, throughput, pdata[19], NULL, d_hash[thr_id], order++);

		*hashes_done = pdata[19] - first_nonce + throughput;

		uint32_t foundNonce = cuda_check_hash(thr_id, throughput, pdata[19], d_hash[thr_id]);
		if (foundNonce != UINT32_MAX)
		{
			const uint32_t Htarg = ptarget[7];
			uint32_t vhash64[8];
			be32enc(&endiandata[19], foundNonce);
			nist5hash(vhash64, endiandata);

			if (vhash64[7] <= Htarg && fulltest(vhash64, ptarget)) {
				res = 1;
				uint32_t secNonce = cuda_check_hash_suppl(thr_id, throughput, pdata[19], d_hash[thr_id], 1);
				work_set_target_ratio(work, vhash64);
				if (secNonce != 0) {
					be32enc(&endiandata[19], secNonce);
					nist5hash(vhash64, endiandata);
					if (bn_hash_target_ratio(vhash64, ptarget) > work->shareratio)
						work_set_target_ratio(work, vhash64);
					pdata[21] = secNonce;
					res++;
				}
				pdata[19] = foundNonce;
				goto out;
			}
			else {
				gpulog(LOG_WARNING, thr_id, "result for %08x does not validate on CPU!", foundNonce);
			}
		}

		if ((uint64_t) throughput + pdata[19] >= max_nonce) {
			pdata[19] = max_nonce;
			break;
		}

		pdata[19] += throughput;

	} while (!work_restart[thr_id].restart);

out:
//	*hashes_done = pdata[19] - first_nonce;
#ifdef USE_STREAMS
	for (int i = 0; i < 5; i++)
		cudaStreamDestroy(stream[i]);
#endif

	return res;
}

// ressources cleanup
extern "C" void free_nist5(int thr_id)
{
	if (!init[thr_id])
		return;

	cudaThreadSynchronize();

	cudaFree(d_hash[thr_id]);

	quark_blake512_cpu_free(thr_id);
	quark_groestl512_cpu_free(thr_id);
	cuda_check_cpu_free(thr_id);
	init[thr_id] = false;

	cudaDeviceSynchronize();
}
committing a quick attempt at NIST5 (TalkCoin) 11 years ago			`extern "C"`
			`{`
			`#include "sph/sph_blake.h"`
			`#include "sph/sph_groestl.h"`
			`#include "sph/sph_skein.h"`
			`#include "sph/sph_jh.h"`
			`#include "sph/sph_keccak.h"`
x11: adapt some blake 256 opts to 512 one blake512: for the moment 6.2ms vs 7.12 before (+10%) 10 years ago			`}`

committing a quick attempt at NIST5 (TalkCoin) 11 years ago			`#include "miner.h"`
x11: adapt some blake 256 opts to 512 one blake512: for the moment 6.2ms vs 7.12 before (+10%) 10 years ago
Remove duplicated defines present in cuda_helper.h also add cudaDeviceReset() on Ctrl+C for nvprof 10 years ago			`#include "cuda_helper.h"`
cuda: header for common kernel functions (quark/x11) Was thinking about doing that since months ;) lets go 9 years ago			`#include "quark/cuda_quark.h"`
committing a quick attempt at NIST5 (TalkCoin) 11 years ago
Handle a maximum of 16 gpus (vs 8 before) Some cards have 2 gpus on board... 10 years ago			`static uint32_t *d_hash[MAX_GPUS];`
committing a quick attempt at NIST5 (TalkCoin) 11 years ago
			`// Original nist5hash Funktion aus einem miner Quelltext`
Remove duplicated defines present in cuda_helper.h also add cudaDeviceReset() on Ctrl+C for nvprof 10 years ago			`extern "C" void nist5hash(void state, const void input)`
committing a quick attempt at NIST5 (TalkCoin) 11 years ago			`{`
			`sph_blake512_context ctx_blake;`
			`sph_groestl512_context ctx_groestl;`
			`sph_jh512_context ctx_jh;`
			`sph_keccak512_context ctx_keccak;`
			`sph_skein512_context ctx_skein;`

Move common check_cpu functions to root 10 years ago			`uint8_t hash[64];`
committing a quick attempt at NIST5 (TalkCoin) 11 years ago
			`sph_blake512_init(&ctx_blake);`
			`sph_blake512 (&ctx_blake, input, 80);`
			`sph_blake512_close(&ctx_blake, (void*) hash);`

			`sph_groestl512_init(&ctx_groestl);`
			`sph_groestl512 (&ctx_groestl, (const void*) hash, 64);`
			`sph_groestl512_close(&ctx_groestl, (void*) hash);`

			`sph_jh512_init(&ctx_jh);`
			`sph_jh512 (&ctx_jh, (const void*) hash, 64);`
			`sph_jh512_close(&ctx_jh, (void*) hash);`

			`sph_keccak512_init(&ctx_keccak);`
			`sph_keccak512 (&ctx_keccak, (const void*) hash, 64);`
			`sph_keccak512_close(&ctx_keccak, (void*) hash);`

			`sph_skein512_init(&ctx_skein);`
			`sph_skein512 (&ctx_skein, (const void*) hash, 64);`
			`sph_skein512_close(&ctx_skein, (void*) hash);`

			`memcpy(state, hash, 32);`
			`}`

Handle a maximum of 16 gpus (vs 8 before) Some cards have 2 gpus on board... 10 years ago			`static bool init[MAX_GPUS] = { 0 };`
various small changes heavy: reduce by 256 threads default intensity to all -i 20 cuda: put static thread init bools outside the code (made once) api: fix nvml header to build without 10 years ago
start v1.7, apply new prototypes to all algos 9 years ago			`extern "C" int scanhash_nist5(int thr_id, struct work work, uint32_t max_nonce, unsigned long hashes_done)`
committing a quick attempt at NIST5 (TalkCoin) 11 years ago			`{`
start v1.7, apply new prototypes to all algos 9 years ago			`uint32_t _ALIGN(64) endiandata[20];`
			`uint32_t *pdata = work->data;`
			`uint32_t *ptarget = work->target;`
committing a quick attempt at NIST5 (TalkCoin) 11 years ago			`const uint32_t first_nonce = pdata[19];`
algos: add functions to free allocated resources Will be used later for algo switching not really tested yet... 9 years ago			`int res = 0;`
committing a quick attempt at NIST5 (TalkCoin) 11 years ago
intensity: do not reduce throughput before init Else the memory allocated could be less than required later btw, use the new "cuda" function to apply intensity/throughput 9 years ago			`uint32_t throughput = cuda_default_throughput(thr_id, 1 << 20); // 25625616`
			`if (init[thr_id]) throughput = min(throughput, max_nonce - first_nonce);`
Allow different intensity per device and clean the old variables, no more required 10 years ago
committing a quick attempt at NIST5 (TalkCoin) 11 years ago			`if (opt_benchmark)`
x11: adapt some blake 256 opts to 512 one blake512: for the moment 6.2ms vs 7.12 before (+10%) 10 years ago			`((uint32_t*)ptarget)[7] = 0x00FF;`
committing a quick attempt at NIST5 (TalkCoin) 11 years ago
			`if (!init[thr_id])`
			`{`
benchmark: store all algos results + cuda fixes Note: lyra2, lyra2v2 and script seems to have problems to coexist with other algos... to run after some of them... moved lyra2 first and skip scrypt/jane for the moment... Only stored in memory for now.. to display a table after the bench ccminer -a auto --benchmark Results may be exported later to a json file... 9 years ago			`cudaDeviceSynchronize();`
committing a quick attempt at NIST5 (TalkCoin) 11 years ago			`cudaSetDevice(device_map[thr_id]);`

algos: add functions to free allocated resources Will be used later for algo switching not really tested yet... 9 years ago			`// Constants copy/init (no device alloc in these algos)`
committing a quick attempt at NIST5 (TalkCoin) 11 years ago			`quark_blake512_cpu_init(thr_id, throughput);`
			`quark_groestl512_cpu_init(thr_id, throughput);`
			`quark_jh512_cpu_init(thr_id, throughput);`
			`quark_keccak512_cpu_init(thr_id, throughput);`
			`quark_skein512_cpu_init(thr_id, throughput);`
various small changes heavy: reduce by 256 threads default intensity to all -i 20 cuda: put static thread init bools outside the code (made once) api: fix nvml header to build without 10 years ago
algos: add functions to free allocated resources Will be used later for algo switching not really tested yet... 9 years ago			`// char[64] work space for hashes results`
			`CUDA_SAFE_CALL(cudaMalloc(&d_hash[thr_id], (size_t)64 * throughput));`
various small changes heavy: reduce by 256 threads default intensity to all -i 20 cuda: put static thread init bools outside the code (made once) api: fix nvml header to build without 10 years ago
Remove duplicated defines present in cuda_helper.h also add cudaDeviceReset() on Ctrl+C for nvprof 10 years ago			`cuda_check_cpu_init(thr_id, throughput);`
committing a quick attempt at NIST5 (TalkCoin) 11 years ago			`init[thr_id] = true;`
			`}`

algos: add functions to free allocated resources Will be used later for algo switching not really tested yet... 9 years ago			`#ifdef USE_STREAMS`
			`cudaStream_t stream[5];`
			`for (int i = 0; i < 5; i++)`
			`cudaStreamCreate(&stream[i]);`
			`#endif`

committing a quick attempt at NIST5 (TalkCoin) 11 years ago			`for (int k=0; k < 20; k++)`
remove uint32_t cast 10 years ago			`be32enc(&endiandata[k], pdata[k]);`
committing a quick attempt at NIST5 (TalkCoin) 11 years ago
blake80: some changes and launch bounds, no perf changes 10 years ago			`quark_blake512_cpu_setBlock_80(thr_id, endiandata);`
Remove duplicated defines present in cuda_helper.h also add cudaDeviceReset() on Ctrl+C for nvprof 10 years ago			`cuda_check_cpu_setTarget(ptarget);`
committing a quick attempt at NIST5 (TalkCoin) 11 years ago
			`do {`
			`int order = 0;`

Remove duplicated defines present in cuda_helper.h also add cudaDeviceReset() on Ctrl+C for nvprof 10 years ago			`// Hash with CUDA`
blake80: some changes and launch bounds, no perf changes 10 years ago			`quark_blake512_cpu_hash_80(thr_id, throughput, pdata[19], d_hash[thr_id]); order++;`
committing a quick attempt at NIST5 (TalkCoin) 11 years ago			`quark_groestl512_cpu_hash_64(thr_id, throughput, pdata[19], NULL, d_hash[thr_id], order++);`
			`quark_jh512_cpu_hash_64(thr_id, throughput, pdata[19], NULL, d_hash[thr_id], order++);`
			`quark_keccak512_cpu_hash_64(thr_id, throughput, pdata[19], NULL, d_hash[thr_id], order++);`
			`quark_skein512_cpu_hash_64(thr_id, throughput, pdata[19], NULL, d_hash[thr_id], order++);`

start v1.7, apply new prototypes to all algos 9 years ago			`*hashes_done = pdata[19] - first_nonce + throughput;`

checkhash: simplify the common function use klaus trivial function, the old code has always been a bit weird.. split cuda_check_cpu_hash_64 in two functions, keep old for branched stuff 10 years ago			`uint32_t foundNonce = cuda_check_hash(thr_id, throughput, pdata[19], d_hash[thr_id]);`
Check and submit multiple nonces in one loop Added to most algos, checkhash function scans a big range and can find multiple nonces at once if the difficulty is low. Stop ignoring them, submit second one if found... Clean the draft code for rc=2 implemented for blake and pentablake btw... fix the reduced displayed hashrate when a nonce is found... Signed-off-by: Tanguy Pruvot <tanguy.pruvot@gmail.com> 10 years ago			`if (foundNonce != UINT32_MAX)`
committing a quick attempt at NIST5 (TalkCoin) 11 years ago			`{`
Check and submit multiple nonces in one loop Added to most algos, checkhash function scans a big range and can find multiple nonces at once if the difficulty is low. Stop ignoring them, submit second one if found... Clean the draft code for rc=2 implemented for blake and pentablake btw... fix the reduced displayed hashrate when a nonce is found... Signed-off-by: Tanguy Pruvot <tanguy.pruvot@gmail.com> 10 years ago			`const uint32_t Htarg = ptarget[7];`
committing a quick attempt at NIST5 (TalkCoin) 11 years ago			`uint32_t vhash64[8];`
			`be32enc(&endiandata[19], foundNonce);`
			`nist5hash(vhash64, endiandata);`

Check and submit multiple nonces in one loop Added to most algos, checkhash function scans a big range and can find multiple nonces at once if the difficulty is low. Stop ignoring them, submit second one if found... Clean the draft code for rc=2 implemented for blake and pentablake btw... fix the reduced displayed hashrate when a nonce is found... Signed-off-by: Tanguy Pruvot <tanguy.pruvot@gmail.com> 10 years ago			`if (vhash64[7] <= Htarg && fulltest(vhash64, ptarget)) {`
algos: add functions to free allocated resources Will be used later for algo switching not really tested yet... 9 years ago			`res = 1;`
Check and submit multiple nonces in one loop Added to most algos, checkhash function scans a big range and can find multiple nonces at once if the difficulty is low. Stop ignoring them, submit second one if found... Clean the draft code for rc=2 implemented for blake and pentablake btw... fix the reduced displayed hashrate when a nonce is found... Signed-off-by: Tanguy Pruvot <tanguy.pruvot@gmail.com> 10 years ago			`uint32_t secNonce = cuda_check_hash_suppl(thr_id, throughput, pdata[19], d_hash[thr_id], 1);`
diff: use the new function in all algos 9 years ago			`work_set_target_ratio(work, vhash64);`
Check and submit multiple nonces in one loop Added to most algos, checkhash function scans a big range and can find multiple nonces at once if the difficulty is low. Stop ignoring them, submit second one if found... Clean the draft code for rc=2 implemented for blake and pentablake btw... fix the reduced displayed hashrate when a nonce is found... Signed-off-by: Tanguy Pruvot <tanguy.pruvot@gmail.com> 10 years ago			`if (secNonce != 0) {`
start v1.7, apply new prototypes to all algos 9 years ago			`be32enc(&endiandata[19], secNonce);`
			`nist5hash(vhash64, endiandata);`
			`if (bn_hash_target_ratio(vhash64, ptarget) > work->shareratio)`
diff: use the new function in all algos 9 years ago			`work_set_target_ratio(work, vhash64);`
Check and submit multiple nonces in one loop Added to most algos, checkhash function scans a big range and can find multiple nonces at once if the difficulty is low. Stop ignoring them, submit second one if found... Clean the draft code for rc=2 implemented for blake and pentablake btw... fix the reduced displayed hashrate when a nonce is found... Signed-off-by: Tanguy Pruvot <tanguy.pruvot@gmail.com> 10 years ago			`pdata[21] = secNonce;`
			`res++;`
			`}`
committing a quick attempt at NIST5 (TalkCoin) 11 years ago			`pdata[19] = foundNonce;`
algos: add functions to free allocated resources Will be used later for algo switching not really tested yet... 9 years ago			`goto out;`
Check and submit multiple nonces in one loop Added to most algos, checkhash function scans a big range and can find multiple nonces at once if the difficulty is low. Stop ignoring them, submit second one if found... Clean the draft code for rc=2 implemented for blake and pentablake btw... fix the reduced displayed hashrate when a nonce is found... Signed-off-by: Tanguy Pruvot <tanguy.pruvot@gmail.com> 10 years ago			`}`
			`else {`
benchmark: enhance the mem leak detection reduce "false" warnings, and ignore unrelated/small ones <= 1 MB On windows the gpu memory can be allocated by other processes + some cleanup in algos... (free/gpulog) 9 years ago			`gpulog(LOG_WARNING, thr_id, "result for %08x does not validate on CPU!", foundNonce);`
committing a quick attempt at NIST5 (TalkCoin) 11 years ago			`}`
			`}`

never interrupt global benchmark with found nonces fix some algo weird hashrates (like blake) and reset device between algos, for better accuracy but this reset doesnt seems enough to bench all algos correctly... to test on linux, could be a driver issue... heavy: fix first alloc and indent with tabs... 9 years ago			`if ((uint64_t) throughput + pdata[19] >= max_nonce) {`
			`pdata[19] = max_nonce;`
			`break;`
			`}`

committing a quick attempt at NIST5 (TalkCoin) 11 years ago			`pdata[19] += throughput;`

never interrupt global benchmark with found nonces fix some algo weird hashrates (like blake) and reset device between algos, for better accuracy but this reset doesnt seems enough to bench all algos correctly... to test on linux, could be a driver issue... heavy: fix first alloc and indent with tabs... 9 years ago			`} while (!work_restart[thr_id].restart);`
committing a quick attempt at NIST5 (TalkCoin) 11 years ago
algos: add functions to free allocated resources Will be used later for algo switching not really tested yet... 9 years ago			`out:`
nist5: reported ccminer hashrate was wrong 9 years ago			`// *hashes_done = pdata[19] - first_nonce;`
algos: add functions to free allocated resources Will be used later for algo switching not really tested yet... 9 years ago			`#ifdef USE_STREAMS`
			`for (int i = 0; i < 5; i++)`
			`cudaStreamDestroy(stream[i]);`
			`#endif`

			`return res;`
committing a quick attempt at NIST5 (TalkCoin) 11 years ago			`}`
algos: add functions to free allocated resources Will be used later for algo switching not really tested yet... 9 years ago
			`// ressources cleanup`
			`extern "C" void free_nist5(int thr_id)`
			`{`
			`if (!init[thr_id])`
			`return;`

benchmark: enhance the mem leak detection reduce "false" warnings, and ignore unrelated/small ones <= 1 MB On windows the gpu memory can be allocated by other processes + some cleanup in algos... (free/gpulog) 9 years ago			`cudaThreadSynchronize();`
algos: add functions to free allocated resources Will be used later for algo switching not really tested yet... 9 years ago
			`cudaFree(d_hash[thr_id]);`

use blake512 sp kernels on SM 5+ (80+64) import and keep my code for older archs, like skein 64 reduce the gap between our versions... +150kH x11 GTX 960 / +30kH 750Ti +900kH quark GTX 960 / +230kH 750Ti 9 years ago			`quark_blake512_cpu_free(thr_id);`
algos: free allocated mem for algo switch All can be freed propertly now, except script (reset) and lyra2 (leak) 9 years ago			`quark_groestl512_cpu_free(thr_id);`
algos: add functions to free allocated resources Will be used later for algo switching not really tested yet... 9 years ago			`cuda_check_cpu_free(thr_id);`
			`init[thr_id] = false;`

			`cudaDeviceSynchronize();`
benchmark: store all algos results + cuda fixes Note: lyra2, lyra2v2 and script seems to have problems to coexist with other algos... to run after some of them... moved lyra2 first and skip scrypt/jane for the moment... Only stored in memory for now.. to display a table after the bench ccminer -a auto --benchmark Results may be exported later to a json file... 9 years ago			`}`