ccminer-gostd-lite/pentablake.cu

/**
 * Penta Blake
 */

#include <stdint.h>
#include <memory.h>
#include "miner.h"

extern "C" {
#include "sph/sph_blake.h"
}

/* hash by cpu with blake 256 */
extern "C" void pentablakehash(void *output, const void *input)
{
	unsigned char _ALIGN(128) hash[64];

	sph_blake512_context ctx;

	sph_blake512_init(&ctx);
	sph_blake512(&ctx, input, 80);
	sph_blake512_close(&ctx, hash);

	sph_blake512(&ctx, hash, 64);
	sph_blake512_close(&ctx, hash);

	sph_blake512(&ctx, hash, 64);
	sph_blake512_close(&ctx, hash);

	sph_blake512(&ctx, hash, 64);
	sph_blake512_close(&ctx, hash);

	sph_blake512(&ctx, hash, 64);
	sph_blake512_close(&ctx, hash);

	memcpy(output, hash, 32);
}

#include "cuda_helper.h"

static uint32_t *d_hash[MAX_GPUS];

extern void quark_blake512_cpu_init(int thr_id, uint32_t threads);
extern void quark_blake512_cpu_free(int thr_id);
extern void quark_blake512_cpu_setBlock_80(int thr_id, uint32_t *pdata);
extern void quark_blake512_cpu_hash_80(int thr_id, uint32_t threads, uint32_t startNounce, uint32_t *d_hash);
extern void quark_blake512_cpu_hash_64(int thr_id, uint32_t threads, uint32_t startNounce, uint32_t *d_nonceVector, uint32_t *d_hash, int order);

static bool init[MAX_GPUS] = { 0 };

extern "C" int scanhash_pentablake(int thr_id, struct work *work, uint32_t max_nonce, unsigned long *hashes_done)
{
	uint32_t _ALIGN(64) endiandata[20];
	uint32_t *pdata = work->data;
	uint32_t *ptarget = work->target;
	const uint32_t first_nonce = pdata[19];
	int rc = 0;
	uint32_t throughput =  cuda_default_throughput(thr_id, 1U << 19);
	if (init[thr_id]) throughput = min(throughput, max_nonce - first_nonce);

	if (opt_benchmark)
		ptarget[7] = 0x000F;

	if (!init[thr_id]) {
		cudaSetDevice(device_map[thr_id]);
		if (opt_cudaschedule == -1 && gpu_threads == 1) {
			cudaDeviceReset();
			// reduce cpu usage
			cudaSetDeviceFlags(cudaDeviceScheduleBlockingSync);
			CUDA_LOG_ERROR();
		}
		gpulog(LOG_INFO, thr_id, "Intensity set to %g, %u cuda threads", throughput2intensity(throughput), throughput);

		CUDA_SAFE_CALL(cudaMalloc(&d_hash[thr_id], (size_t) 64 * throughput));

		quark_blake512_cpu_init(thr_id, throughput);
		cuda_check_cpu_init(thr_id, throughput);
		CUDA_LOG_ERROR();

		init[thr_id] = true;
	}

	for (int k=0; k < 20; k++)
		be32enc(&endiandata[k], pdata[k]);

	quark_blake512_cpu_setBlock_80(thr_id, endiandata);
	cuda_check_cpu_setTarget(ptarget);

	do {
		int order = 0;

		// GPU HASH
		quark_blake512_cpu_hash_80(thr_id, throughput, pdata[19], d_hash[thr_id]); order++;
		quark_blake512_cpu_hash_64(thr_id, throughput, pdata[19], NULL, d_hash[thr_id], order++);
		quark_blake512_cpu_hash_64(thr_id, throughput, pdata[19], NULL, d_hash[thr_id], order++);
		quark_blake512_cpu_hash_64(thr_id, throughput, pdata[19], NULL, d_hash[thr_id], order++);
		quark_blake512_cpu_hash_64(thr_id, throughput, pdata[19], NULL, d_hash[thr_id], order++);

		*hashes_done = pdata[19] - first_nonce + throughput;

		uint32_t foundNonce = cuda_check_hash(thr_id, throughput, pdata[19], d_hash[thr_id]);
		if (foundNonce != UINT32_MAX)
		{
			uint32_t vhash[8];

			be32enc(&endiandata[19], foundNonce);
			pentablakehash(vhash, endiandata);

			if (vhash[7] <= ptarget[7] && fulltest(vhash, ptarget)) {
				rc = 1;
				work_set_target_ratio(work, vhash);
				pdata[19] = foundNonce;
				return rc;
			} else {
				gpulog(LOG_WARNING, thr_id, "result for %08x does not validate on CPU!", foundNonce);
			}
		}

		if ((uint64_t) throughput + pdata[19] >= max_nonce) {
			pdata[19] = max_nonce;
			break;
		}

		pdata[19] += throughput;

	} while (!work_restart[thr_id].restart);

	return rc;
}

// cleanup
void free_pentablake(int thr_id)
{
	if (!init[thr_id])
		return;

	cudaThreadSynchronize();

	cudaFree(d_hash[thr_id]);

	quark_blake512_cpu_free(thr_id);
	cuda_check_cpu_free(thr_id);

	cudaDeviceSynchronize();

	init[thr_id] = false;
}
Add pentablake algo (-a penta) Signed-off-by: Tanguy Pruvot <tanguy.pruvot@gmail.com> 10 years ago			`/**`
pentablake: use common blake kernels (quark) reduce the binary size and improve the speed... 9 years ago			`* Penta Blake`
Add pentablake algo (-a penta) Signed-off-by: Tanguy Pruvot <tanguy.pruvot@gmail.com> 10 years ago			`*/`

pentablake: use common blake kernels (quark) reduce the binary size and improve the speed... 9 years ago			`#include <stdint.h>`
			`#include <memory.h>`
Add pentablake algo (-a penta) Signed-off-by: Tanguy Pruvot <tanguy.pruvot@gmail.com> 10 years ago			`#include "miner.h"`

			`extern "C" {`
			`#include "sph/sph_blake.h"`
			`}`

			`/* hash by cpu with blake 256 */`
			`extern "C" void pentablakehash(void output, const void input)`
			`{`
pentablake: use common blake kernels (quark) reduce the binary size and improve the speed... 9 years ago			`unsigned char _ALIGN(128) hash[64];`

Add pentablake algo (-a penta) Signed-off-by: Tanguy Pruvot <tanguy.pruvot@gmail.com> 10 years ago			`sph_blake512_context ctx;`

			`sph_blake512_init(&ctx);`
			`sph_blake512(&ctx, input, 80);`
			`sph_blake512_close(&ctx, hash);`

			`sph_blake512(&ctx, hash, 64);`
pentablake: use common blake kernels (quark) reduce the binary size and improve the speed... 9 years ago			`sph_blake512_close(&ctx, hash);`
Add pentablake algo (-a penta) Signed-off-by: Tanguy Pruvot <tanguy.pruvot@gmail.com> 10 years ago
pentablake: use common blake kernels (quark) reduce the binary size and improve the speed... 9 years ago			`sph_blake512(&ctx, hash, 64);`
Add pentablake algo (-a penta) Signed-off-by: Tanguy Pruvot <tanguy.pruvot@gmail.com> 10 years ago			`sph_blake512_close(&ctx, hash);`

			`sph_blake512(&ctx, hash, 64);`
pentablake: use common blake kernels (quark) reduce the binary size and improve the speed... 9 years ago			`sph_blake512_close(&ctx, hash);`
Add pentablake algo (-a penta) Signed-off-by: Tanguy Pruvot <tanguy.pruvot@gmail.com> 10 years ago
pentablake: use common blake kernels (quark) reduce the binary size and improve the speed... 9 years ago			`sph_blake512(&ctx, hash, 64);`
Add pentablake algo (-a penta) Signed-off-by: Tanguy Pruvot <tanguy.pruvot@gmail.com> 10 years ago			`sph_blake512_close(&ctx, hash);`

			`memcpy(output, hash, 32);`
			`}`

			`#include "cuda_helper.h"`

Handle a maximum of 16 gpus (vs 8 before) Some cards have 2 gpus on board... 10 years ago			`static uint32_t *d_hash[MAX_GPUS];`
Add pentablake algo (-a penta) Signed-off-by: Tanguy Pruvot <tanguy.pruvot@gmail.com> 10 years ago
pentablake: use common blake kernels (quark) reduce the binary size and improve the speed... 9 years ago			`extern void quark_blake512_cpu_init(int thr_id, uint32_t threads);`
			`extern void quark_blake512_cpu_free(int thr_id);`
			`extern void quark_blake512_cpu_setBlock_80(int thr_id, uint32_t *pdata);`
			`extern void quark_blake512_cpu_hash_80(int thr_id, uint32_t threads, uint32_t startNounce, uint32_t *d_hash);`
			`extern void quark_blake512_cpu_hash_64(int thr_id, uint32_t threads, uint32_t startNounce, uint32_t d_nonceVector, uint32_t d_hash, int order);`
Add pentablake algo (-a penta) Signed-off-by: Tanguy Pruvot <tanguy.pruvot@gmail.com> 10 years ago
Handle a maximum of 16 gpus (vs 8 before) Some cards have 2 gpus on board... 10 years ago			`static bool init[MAX_GPUS] = { 0 };`
various small changes heavy: reduce by 256 threads default intensity to all -i 20 cuda: put static thread init bools outside the code (made once) api: fix nvml header to build without 10 years ago
start v1.7, apply new prototypes to all algos 9 years ago			`extern "C" int scanhash_pentablake(int thr_id, struct work work, uint32_t max_nonce, unsigned long hashes_done)`
Add pentablake algo (-a penta) Signed-off-by: Tanguy Pruvot <tanguy.pruvot@gmail.com> 10 years ago			`{`
start v1.7, apply new prototypes to all algos 9 years ago			`uint32_t _ALIGN(64) endiandata[20];`
			`uint32_t *pdata = work->data;`
			`uint32_t *ptarget = work->target;`
Add pentablake algo (-a penta) Signed-off-by: Tanguy Pruvot <tanguy.pruvot@gmail.com> 10 years ago			`const uint32_t first_nonce = pdata[19];`
			`int rc = 0;`
pentablake: use common blake kernels (quark) reduce the binary size and improve the speed... 9 years ago			`uint32_t throughput = cuda_default_throughput(thr_id, 1U << 19);`
intensity: do not reduce throughput before init Else the memory allocated could be less than required later btw, use the new "cuda" function to apply intensity/throughput 9 years ago			`if (init[thr_id]) throughput = min(throughput, max_nonce - first_nonce);`
Add pentablake algo (-a penta) Signed-off-by: Tanguy Pruvot <tanguy.pruvot@gmail.com> 10 years ago
			`if (opt_benchmark)`
use blake512 sp kernels on SM 5+ (80+64) import and keep my code for older archs, like skein 64 reduce the gap between our versions... +150kH x11 GTX 960 / +30kH 750Ti +900kH quark GTX 960 / +230kH 750Ti 9 years ago			`ptarget[7] = 0x000F;`
Add pentablake algo (-a penta) Signed-off-by: Tanguy Pruvot <tanguy.pruvot@gmail.com> 10 years ago
			`if (!init[thr_id]) {`
use blake512 sp kernels on SM 5+ (80+64) import and keep my code for older archs, like skein 64 reduce the gap between our versions... +150kH x11 GTX 960 / +30kH 750Ti +900kH quark GTX 960 / +230kH 750Ti 9 years ago			`cudaSetDevice(device_map[thr_id]);`
1.7.1 release set schedule flags to reduce linux cpu usage without MyStreamSynchronize() 9 years ago			`if (opt_cudaschedule == -1 && gpu_threads == 1) {`
			`cudaDeviceReset();`
			`// reduce cpu usage`
			`cudaSetDeviceFlags(cudaDeviceScheduleBlockingSync);`
			`CUDA_LOG_ERROR();`
			`}`
Show intensity on init for all algos 8 years ago			`gpulog(LOG_INFO, thr_id, "Intensity set to %g, %u cuda threads", throughput2intensity(throughput), throughput);`
pentablake: use common blake kernels (quark) reduce the binary size and improve the speed... 9 years ago
algos: free allocated mem for algo switch All can be freed propertly now, except script (reset) and lyra2 (leak) 9 years ago			`CUDA_SAFE_CALL(cudaMalloc(&d_hash[thr_id], (size_t) 64 * throughput));`
pentablake: use common blake kernels (quark) reduce the binary size and improve the speed... 9 years ago
			`quark_blake512_cpu_init(thr_id, throughput);`
			`cuda_check_cpu_init(thr_id, throughput);`
			`CUDA_LOG_ERROR();`

Add pentablake algo (-a penta) Signed-off-by: Tanguy Pruvot <tanguy.pruvot@gmail.com> 10 years ago			`init[thr_id] = true;`
			`}`

			`for (int k=0; k < 20; k++)`
			`be32enc(&endiandata[k], pdata[k]);`

pentablake: use common blake kernels (quark) reduce the binary size and improve the speed... 9 years ago			`quark_blake512_cpu_setBlock_80(thr_id, endiandata);`
			`cuda_check_cpu_setTarget(ptarget);`
Add pentablake algo (-a penta) Signed-off-by: Tanguy Pruvot <tanguy.pruvot@gmail.com> 10 years ago
			`do {`
			`int order = 0;`

			`// GPU HASH`
pentablake: use common blake kernels (quark) reduce the binary size and improve the speed... 9 years ago			`quark_blake512_cpu_hash_80(thr_id, throughput, pdata[19], d_hash[thr_id]); order++;`
			`quark_blake512_cpu_hash_64(thr_id, throughput, pdata[19], NULL, d_hash[thr_id], order++);`
			`quark_blake512_cpu_hash_64(thr_id, throughput, pdata[19], NULL, d_hash[thr_id], order++);`
			`quark_blake512_cpu_hash_64(thr_id, throughput, pdata[19], NULL, d_hash[thr_id], order++);`
			`quark_blake512_cpu_hash_64(thr_id, throughput, pdata[19], NULL, d_hash[thr_id], order++);`
Add pentablake algo (-a penta) Signed-off-by: Tanguy Pruvot <tanguy.pruvot@gmail.com> 10 years ago
start v1.7, apply new prototypes to all algos 9 years ago			`*hashes_done = pdata[19] - first_nonce + throughput;`

pentablake: use common blake kernels (quark) reduce the binary size and improve the speed... 9 years ago			`uint32_t foundNonce = cuda_check_hash(thr_id, throughput, pdata[19], d_hash[thr_id]);`
Check and submit multiple nonces in one loop Added to most algos, checkhash function scans a big range and can find multiple nonces at once if the difficulty is low. Stop ignoring them, submit second one if found... Clean the draft code for rc=2 implemented for blake and pentablake btw... fix the reduced displayed hashrate when a nonce is found... Signed-off-by: Tanguy Pruvot <tanguy.pruvot@gmail.com> 10 years ago			`if (foundNonce != UINT32_MAX)`
Add pentablake algo (-a penta) Signed-off-by: Tanguy Pruvot <tanguy.pruvot@gmail.com> 10 years ago			`{`
start v1.7, apply new prototypes to all algos 9 years ago			`uint32_t vhash[8];`
Add pentablake algo (-a penta) Signed-off-by: Tanguy Pruvot <tanguy.pruvot@gmail.com> 10 years ago
			`be32enc(&endiandata[19], foundNonce);`
start v1.7, apply new prototypes to all algos 9 years ago			`pentablakehash(vhash, endiandata);`
Add pentablake algo (-a penta) Signed-off-by: Tanguy Pruvot <tanguy.pruvot@gmail.com> 10 years ago
start v1.7, apply new prototypes to all algos 9 years ago			`if (vhash[7] <= ptarget[7] && fulltest(vhash, ptarget)) {`
Add pentablake algo (-a penta) Signed-off-by: Tanguy Pruvot <tanguy.pruvot@gmail.com> 10 years ago			`rc = 1;`
diff: use the new function in all algos 9 years ago			`work_set_target_ratio(work, vhash);`
Check and submit multiple nonces in one loop Added to most algos, checkhash function scans a big range and can find multiple nonces at once if the difficulty is low. Stop ignoring them, submit second one if found... Clean the draft code for rc=2 implemented for blake and pentablake btw... fix the reduced displayed hashrate when a nonce is found... Signed-off-by: Tanguy Pruvot <tanguy.pruvot@gmail.com> 10 years ago			`pdata[19] = foundNonce;`
			`return rc;`
warnings: use the right device id (device_map[thr_id]) 10 years ago			`} else {`
benchmark: enhance the mem leak detection reduce "false" warnings, and ignore unrelated/small ones <= 1 MB On windows the gpu memory can be allocated by other processes + some cleanup in algos... (free/gpulog) 9 years ago			`gpulog(LOG_WARNING, thr_id, "result for %08x does not validate on CPU!", foundNonce);`
Add pentablake algo (-a penta) Signed-off-by: Tanguy Pruvot <tanguy.pruvot@gmail.com> 10 years ago			`}`
			`}`

never interrupt global benchmark with found nonces fix some algo weird hashrates (like blake) and reset device between algos, for better accuracy but this reset doesnt seems enough to bench all algos correctly... to test on linux, could be a driver issue... heavy: fix first alloc and indent with tabs... 9 years ago			`if ((uint64_t) throughput + pdata[19] >= max_nonce) {`
			`pdata[19] = max_nonce;`
			`break;`
			`}`

Add pentablake algo (-a penta) Signed-off-by: Tanguy Pruvot <tanguy.pruvot@gmail.com> 10 years ago			`pdata[19] += throughput;`

never interrupt global benchmark with found nonces fix some algo weird hashrates (like blake) and reset device between algos, for better accuracy but this reset doesnt seems enough to bench all algos correctly... to test on linux, could be a driver issue... heavy: fix first alloc and indent with tabs... 9 years ago			`} while (!work_restart[thr_id].restart);`
Add pentablake algo (-a penta) Signed-off-by: Tanguy Pruvot <tanguy.pruvot@gmail.com> 10 years ago
			`return rc;`
			`}`
algos: add functions to free allocated resources Will be used later for algo switching not really tested yet... 9 years ago
			`// cleanup`
			`void free_pentablake(int thr_id)`
			`{`
			`if (!init[thr_id])`
			`return;`

various fixes for SM 2.1 and the benchmark X11+ algos and quark are not compatible for the moment but these ones are : Benchmark results for Gigabyte GTX 460 (SM 2.1 / 1 GB): blakecoin : 159090.5 kH/s, 1 MB, 1048576 thr. blake : 70208.9 kH/s, 1 MB, 1048576 thr. bmw : 122802.6 kH/s, 65 MB, 2097152 thr. deep : 3533.6 kH/s, 33 MB, 524288 thr. fugue256 : 43177.9 kH/s, 17 MB, 524288 thr. heavy : 4118.2 kH/s, 147 MB, 524032 thr. keccak : 18673.1 kH/s, 129 MB, 2097152 thr. luffa : 28816.0 kH/s, 257 MB, 4194304 thr. lyra2 : 213.7 kH/s, 570 MB, 65536 thr. mjollnir : 3895.6 kH/s, 147 MB, 524032 thr. nist5 : 1101.4 kH/s, 67 MB, 1048576 thr. penta : 501.6 kH/s, 21 MB, 327680 thr. skein : 5432.4 kH/s, 65 MB, 1048576 thr. skein2 : 6788.9 kH/s, 33 MB, 524288 thr. whirlpool : 688.5 kH/s, 33 MB, 524288 thr. zr5 : 122.5 kH/s, 86 MB, 262144 thr. 9 years ago			`cudaThreadSynchronize();`
algos: add functions to free allocated resources Will be used later for algo switching not really tested yet... 9 years ago
			`cudaFree(d_hash[thr_id]);`
pentablake: use common blake kernels (quark) reduce the binary size and improve the speed... 9 years ago
			`quark_blake512_cpu_free(thr_id);`
			`cuda_check_cpu_free(thr_id);`
algos: add functions to free allocated resources Will be used later for algo switching not really tested yet... 9 years ago
			`cudaDeviceSynchronize();`
various fixes for SM 2.1 and the benchmark X11+ algos and quark are not compatible for the moment but these ones are : Benchmark results for Gigabyte GTX 460 (SM 2.1 / 1 GB): blakecoin : 159090.5 kH/s, 1 MB, 1048576 thr. blake : 70208.9 kH/s, 1 MB, 1048576 thr. bmw : 122802.6 kH/s, 65 MB, 2097152 thr. deep : 3533.6 kH/s, 33 MB, 524288 thr. fugue256 : 43177.9 kH/s, 17 MB, 524288 thr. heavy : 4118.2 kH/s, 147 MB, 524032 thr. keccak : 18673.1 kH/s, 129 MB, 2097152 thr. luffa : 28816.0 kH/s, 257 MB, 4194304 thr. lyra2 : 213.7 kH/s, 570 MB, 65536 thr. mjollnir : 3895.6 kH/s, 147 MB, 524032 thr. nist5 : 1101.4 kH/s, 67 MB, 1048576 thr. penta : 501.6 kH/s, 21 MB, 327680 thr. skein : 5432.4 kH/s, 65 MB, 1048576 thr. skein2 : 6788.9 kH/s, 33 MB, 524288 thr. whirlpool : 688.5 kH/s, 33 MB, 524288 thr. zr5 : 122.5 kH/s, 86 MB, 262144 thr. 9 years ago
			`init[thr_id] = false;`
			`}`