ccminer/Algo256/keccak256.cu

/*
 * Keccak 256
 *
 */

extern "C"
{
#include "sph/sph_shavite.h"
#include "sph/sph_simd.h"
#include "sph/sph_keccak.h"

#include "miner.h"
}

#include "cuda_helper.h"

static uint32_t *d_hash[MAX_GPUS];

extern void keccak256_cpu_init(int thr_id, uint32_t threads);
extern void keccak256_setBlock_80(void *pdata,const void *ptarget);
extern uint32_t keccak256_cpu_hash_80(int thr_id, uint32_t threads, uint32_t startNounce, uint32_t *d_hash, int order);

// CPU Hash
extern "C" void keccak256_hash(void *state, const void *input)
{
	sph_keccak_context ctx_keccak;

	uint32_t hash[16];

	sph_keccak256_init(&ctx_keccak);
	sph_keccak256 (&ctx_keccak, input, 80);
	sph_keccak256_close(&ctx_keccak, (void*) hash);

	memcpy(state, hash, 32);
}

static bool init[MAX_GPUS] = { 0 };

extern "C" int scanhash_keccak256(int thr_id, uint32_t *pdata,
	const uint32_t *ptarget, uint32_t max_nonce,
	unsigned long *hashes_done)
{
	const uint32_t first_nonce = pdata[19];
	uint32_t throughput = device_intensity(thr_id, __func__, 1U << 21); // 256*256*8*4
	throughput = min(throughput, max_nonce - first_nonce);

	if (opt_benchmark)
		((uint32_t*)ptarget)[7] = 0x0005;

	if (!init[thr_id]) {
		cudaSetDevice(device_map[thr_id]);

		CUDA_SAFE_CALL(cudaMalloc(&d_hash[thr_id], 16 * sizeof(uint32_t) * throughput));
		keccak256_cpu_init(thr_id, (int) throughput);

		init[thr_id] = true;
	}

	uint32_t endiandata[20];
	for (int k=0; k < 20; k++) {
		be32enc(&endiandata[k], ((uint32_t*)pdata)[k]);
	}

	keccak256_setBlock_80((void*)endiandata, ptarget);
	do {
		int order = 0;

		uint32_t foundNonce = keccak256_cpu_hash_80(thr_id, (int) throughput, pdata[19], d_hash[thr_id], order++);
		if (foundNonce != UINT32_MAX)
		{
			uint32_t Htarg = ptarget[7];
			uint32_t vhash64[8];
			be32enc(&endiandata[19], foundNonce);
			keccak256_hash(vhash64, endiandata);

			if (vhash64[7] <= Htarg && fulltest(vhash64, ptarget)) {
				*hashes_done = foundNonce - first_nonce + 1;
				pdata[19] = foundNonce;
				return 1;
			}
			else {
				applog(LOG_DEBUG, "GPU #%d: result for nounce %08x does not validate on CPU!", thr_id, foundNonce);
			}
		}

		if ((uint64_t) pdata[19] + throughput > max_nonce) {
			break;
		}

		pdata[19] += throughput;

	} while (!work_restart[thr_id].restart);

	*hashes_done = pdata[19] - first_nonce;
	return 0;
}
Add proper keccak-256 (maxcoin) Cleaned from djm34 repo, tuned for the 750 Ti 10 years ago			`/*`
			`* Keccak 256`
			`*`
			`*/`

			`extern "C"`
			`{`
			`#include "sph/sph_shavite.h"`
			`#include "sph/sph_simd.h"`
			`#include "sph/sph_keccak.h"`

			`#include "miner.h"`
			`}`

			`#include "cuda_helper.h"`

Handle a maximum of 16 gpus (vs 8 before) Some cards have 2 gpus on board... 10 years ago			`static uint32_t *d_hash[MAX_GPUS];`
Add proper keccak-256 (maxcoin) Cleaned from djm34 repo, tuned for the 750 Ti 10 years ago
cleanup: use unsigned throughput parameters Yes, its a big commit, was waiting 1.6 to do that... Sorry for your possible merge issues ;) 10 years ago			`extern void keccak256_cpu_init(int thr_id, uint32_t threads);`
Add proper keccak-256 (maxcoin) Cleaned from djm34 repo, tuned for the 750 Ti 10 years ago			`extern void keccak256_setBlock_80(void pdata,const void ptarget);`
cleanup: use unsigned throughput parameters Yes, its a big commit, was waiting 1.6 to do that... Sorry for your possible merge issues ;) 10 years ago			`extern uint32_t keccak256_cpu_hash_80(int thr_id, uint32_t threads, uint32_t startNounce, uint32_t *d_hash, int order);`
Add proper keccak-256 (maxcoin) Cleaned from djm34 repo, tuned for the 750 Ti 10 years ago
			`// CPU Hash`
			`extern "C" void keccak256_hash(void state, const void input)`
			`{`
			`sph_keccak_context ctx_keccak;`

			`uint32_t hash[16];`

			`sph_keccak256_init(&ctx_keccak);`
			`sph_keccak256 (&ctx_keccak, input, 80);`
			`sph_keccak256_close(&ctx_keccak, (void*) hash);`

			`memcpy(state, hash, 32);`
			`}`

Handle a maximum of 16 gpus (vs 8 before) Some cards have 2 gpus on board... 10 years ago			`static bool init[MAX_GPUS] = { 0 };`
various small changes heavy: reduce by 256 threads default intensity to all -i 20 cuda: put static thread init bools outside the code (made once) api: fix nvml header to build without 10 years ago
Add proper keccak-256 (maxcoin) Cleaned from djm34 repo, tuned for the 750 Ti 10 years ago			`extern "C" int scanhash_keccak256(int thr_id, uint32_t *pdata,`
			`const uint32_t *ptarget, uint32_t max_nonce,`
			`unsigned long *hashes_done)`
			`{`
			`const uint32_t first_nonce = pdata[19];`
Allow different intensity per device and clean the old variables, no more required 10 years ago			`uint32_t throughput = device_intensity(thr_id, __func__, 1U << 21); // 2562568*4`
cleanup: use unsigned throughput parameters Yes, its a big commit, was waiting 1.6 to do that... Sorry for your possible merge issues ;) 10 years ago			`throughput = min(throughput, max_nonce - first_nonce);`
Add proper keccak-256 (maxcoin) Cleaned from djm34 repo, tuned for the 750 Ti 10 years ago
			`if (opt_benchmark)`
Reduce keccak, deep & anime intensity + handle groestl -i param default intensity was the max supported by the card, and perf is not really better. I prefer to let it one under for cards with lower memory (1GB) 10 years ago			`((uint32_t*)ptarget)[7] = 0x0005;`
Add proper keccak-256 (maxcoin) Cleaned from djm34 repo, tuned for the 750 Ti 10 years ago
			`if (!init[thr_id]) {`
			`cudaSetDevice(device_map[thr_id]);`

			`CUDA_SAFE_CALL(cudaMalloc(&d_hash[thr_id], 16 * sizeof(uint32_t) * throughput));`
Enhance stale work detection + throughput fixes seems to resolve solo mining lock on share. export also computed solo work diff in api (not perfect) In high rate algos, throughput should be unsigned... This fixes keccak, blake and doom problems And change terminal color of debug lines, to be selectable in putty, color code is not supported in windows but selection is ok there. 10 years ago			`keccak256_cpu_init(thr_id, (int) throughput);`
Add proper keccak-256 (maxcoin) Cleaned from djm34 repo, tuned for the 750 Ti 10 years ago
			`init[thr_id] = true;`
			`}`

			`uint32_t endiandata[20];`
			`for (int k=0; k < 20; k++) {`
			`be32enc(&endiandata[k], ((uint32_t*)pdata)[k]);`
			`}`

			`keccak256_setBlock_80((void*)endiandata, ptarget);`
			`do {`
			`int order = 0;`

Enhance stale work detection + throughput fixes seems to resolve solo mining lock on share. export also computed solo work diff in api (not perfect) In high rate algos, throughput should be unsigned... This fixes keccak, blake and doom problems And change terminal color of debug lines, to be selectable in putty, color code is not supported in windows but selection is ok there. 10 years ago			`uint32_t foundNonce = keccak256_cpu_hash_80(thr_id, (int) throughput, pdata[19], d_hash[thr_id], order++);`
Check and submit multiple nonces in one loop Added to most algos, checkhash function scans a big range and can find multiple nonces at once if the difficulty is low. Stop ignoring them, submit second one if found... Clean the draft code for rc=2 implemented for blake and pentablake btw... fix the reduced displayed hashrate when a nonce is found... Signed-off-by: Tanguy Pruvot <tanguy.pruvot@gmail.com> 10 years ago			`if (foundNonce != UINT32_MAX)`
Add proper keccak-256 (maxcoin) Cleaned from djm34 repo, tuned for the 750 Ti 10 years ago			`{`
Handle intensity param in all algos and add a check related to start/max nounce params 10 years ago			`uint32_t Htarg = ptarget[7];`
Check and submit multiple nonces in one loop Added to most algos, checkhash function scans a big range and can find multiple nonces at once if the difficulty is low. Stop ignoring them, submit second one if found... Clean the draft code for rc=2 implemented for blake and pentablake btw... fix the reduced displayed hashrate when a nonce is found... Signed-off-by: Tanguy Pruvot <tanguy.pruvot@gmail.com> 10 years ago			`uint32_t vhash64[8];`
Add proper keccak-256 (maxcoin) Cleaned from djm34 repo, tuned for the 750 Ti 10 years ago			`be32enc(&endiandata[19], foundNonce);`
			`keccak256_hash(vhash64, endiandata);`

			`if (vhash64[7] <= Htarg && fulltest(vhash64, ptarget)) {`
keccak: not compatible with second nonces (was broken) Use djm34 new uint2 method to get a +40% boost (115 to 153MH/s) 10 years ago			`*hashes_done = foundNonce - first_nonce + 1;`
Add proper keccak-256 (maxcoin) Cleaned from djm34 repo, tuned for the 750 Ti 10 years ago			`pdata[19] = foundNonce;`
keccak: not compatible with second nonces (was broken) Use djm34 new uint2 method to get a +40% boost (115 to 153MH/s) 10 years ago			`return 1;`
Check and submit multiple nonces in one loop Added to most algos, checkhash function scans a big range and can find multiple nonces at once if the difficulty is low. Stop ignoring them, submit second one if found... Clean the draft code for rc=2 implemented for blake and pentablake btw... fix the reduced displayed hashrate when a nonce is found... Signed-off-by: Tanguy Pruvot <tanguy.pruvot@gmail.com> 10 years ago			`}`
			`else {`
Add proper keccak-256 (maxcoin) Cleaned from djm34 repo, tuned for the 750 Ti 10 years ago			`applog(LOG_DEBUG, "GPU #%d: result for nounce %08x does not validate on CPU!", thr_id, foundNonce);`
			`}`
			`}`

Enhance stale work detection + throughput fixes seems to resolve solo mining lock on share. export also computed solo work diff in api (not perfect) In high rate algos, throughput should be unsigned... This fixes keccak, blake and doom problems And change terminal color of debug lines, to be selectable in putty, color code is not supported in windows but selection is ok there. 10 years ago			`if ((uint64_t) pdata[19] + throughput > max_nonce) {`
Add proper keccak-256 (maxcoin) Cleaned from djm34 repo, tuned for the 750 Ti 10 years ago			`break;`
			`}`

			`pdata[19] += throughput;`

			`} while (!work_restart[thr_id].restart);`

Enhance stale work detection + throughput fixes seems to resolve solo mining lock on share. export also computed solo work diff in api (not perfect) In high rate algos, throughput should be unsigned... This fixes keccak, blake and doom problems And change terminal color of debug lines, to be selectable in putty, color code is not supported in windows but selection is ok there. 10 years ago			`*hashes_done = pdata[19] - first_nonce;`
Add proper keccak-256 (maxcoin) Cleaned from djm34 repo, tuned for the 750 Ti 10 years ago			`return 0;`
			`}`