ccminer/myriadgroestl.cpp

#include <string.h>
#include <stdint.h>
#include <cuda_runtime.h>
#include <openssl/sha.h>

#include "sph/sph_groestl.h"

#include "miner.h"

void myriadgroestl_cpu_init(int thr_id, uint32_t threads);
void myriadgroestl_cpu_free(int thr_id);
void myriadgroestl_cpu_setBlock(int thr_id, void *data, uint32_t *target);
void myriadgroestl_cpu_hash(int thr_id, uint32_t threads, uint32_t startNonce, uint32_t *resNonces);

void myriadhash(void *state, const void *input)
{
	uint32_t _ALIGN(64) hash[16];
	sph_groestl512_context ctx_groestl;
	SHA256_CTX sha256;

	sph_groestl512_init(&ctx_groestl);
	sph_groestl512(&ctx_groestl, input, 80);
	sph_groestl512_close(&ctx_groestl, hash);

	SHA256_Init(&sha256);
	SHA256_Update(&sha256,(unsigned char *)hash, 64);
	SHA256_Final((unsigned char *)hash, &sha256);

	memcpy(state, hash, 32);
}

static bool init[MAX_GPUS] = { 0 };

int scanhash_myriad(int thr_id, struct work *work, uint32_t max_nonce, unsigned long *hashes_done)
{
	uint32_t _ALIGN(64) endiandata[32];
	uint32_t *pdata = work->data;
	uint32_t *ptarget = work->target;
	uint32_t start_nonce = pdata[19];
	int dev_id = device_map[thr_id];
	int intensity = (device_sm[dev_id] >= 600) ? 20 : 18;
	uint32_t throughput = cuda_default_throughput(thr_id, 1U << intensity);
	if (init[thr_id]) throughput = min(throughput, max_nonce - start_nonce);

	if (opt_benchmark)
		ptarget[7] = 0x0000ff;

	// init
	if(!init[thr_id])
	{
		cudaSetDevice(dev_id);
		if (opt_cudaschedule == -1 && gpu_threads == 1) {
			cudaDeviceReset();
			// reduce cpu usage
			cudaSetDeviceFlags(cudaDeviceScheduleBlockingSync);
			CUDA_LOG_ERROR();
		}
		gpulog(LOG_INFO, thr_id, "Intensity set to %g, %u cuda threads", throughput2intensity(throughput), throughput);

		myriadgroestl_cpu_init(thr_id, throughput);
		init[thr_id] = true;
	}

	for (int k=0; k < 20; k++)
		be32enc(&endiandata[k], pdata[k]);

	myriadgroestl_cpu_setBlock(thr_id, endiandata, ptarget);

	do {
		memset(work->nonces, 0xff, sizeof(work->nonces));

		// GPU
		myriadgroestl_cpu_hash(thr_id, throughput, pdata[19], work->nonces);

		*hashes_done = pdata[19] - start_nonce + throughput;

		if (work->nonces[0] < UINT32_MAX && bench_algo < 0)
		{
			uint32_t _ALIGN(64) vhash[8];
			endiandata[19] = swab32(work->nonces[0]);
			myriadhash(vhash, endiandata);
			if (vhash[7] <= ptarget[7] && fulltest(vhash, ptarget)) {
				work->valid_nonces = 1;
				work_set_target_ratio(work, vhash);
				if (work->nonces[1] != UINT32_MAX) {
					endiandata[19] = swab32(work->nonces[1]);
					myriadhash(vhash, endiandata);
					bn_set_target_ratio(work, vhash, 1);
					work->valid_nonces = 2;
					pdata[19] = max(work->nonces[0], work->nonces[1]) + 1;
				} else {
					pdata[19] = work->nonces[0] + 1; // cursor
				}
				return work->valid_nonces;
			}
			else if (vhash[7] > ptarget[7]) {
				gpu_increment_reject(thr_id);
				if (!opt_quiet)
					gpulog(LOG_WARNING, thr_id, "result for %08x does not validate on CPU!", work->nonces[0]);
				pdata[19] = work->nonces[0] + 1;
				continue;
			}
		}

		if ((uint64_t) throughput + pdata[19] >= max_nonce) {
			pdata[19] = max_nonce;
			break;
		}
		pdata[19] += throughput;

	} while (!work_restart[thr_id].restart);

	*hashes_done = max_nonce - start_nonce;

	return 0;
}

// cleanup
void free_myriad(int thr_id)
{
	if (!init[thr_id])
		return;

	cudaThreadSynchronize();

	myriadgroestl_cpu_free(thr_id);
	init[thr_id] = false;

	cudaDeviceSynchronize();
}
min() and max(a,b) are not defined on linux, in fact max exists in jansson includes (in tree only) Add them to miner.h 10 years ago			`#include <string.h>`
			`#include <stdint.h>`
Various algos cleanup + lyra2 sec nonce fix 9 years ago			`#include <cuda_runtime.h>`
min() and max(a,b) are not defined on linux, in fact max exists in jansson includes (in tree only) Add them to miner.h 10 years ago			`#include <openssl/sha.h>`

Revision 0.6 with myriad-groestl and jackpot coin 10 years ago			`#include "sph/sph_groestl.h"`

			`#include "miner.h"`

cleanup: use unsigned throughput parameters Yes, its a big commit, was waiting 1.6 to do that... Sorry for your possible merge issues ;) 9 years ago			`void myriadgroestl_cpu_init(int thr_id, uint32_t threads);`
algos: add functions to free allocated resources Will be used later for algo switching not really tested yet... 9 years ago			`void myriadgroestl_cpu_free(int thr_id);`
myr-gr: handle a second nonce & more cleanup 8 years ago			`void myriadgroestl_cpu_setBlock(int thr_id, void data, uint32_t target);`
			`void myriadgroestl_cpu_hash(int thr_id, uint32_t threads, uint32_t startNonce, uint32_t *resNonces);`
Revision 0.6 with myriad-groestl and jackpot coin 10 years ago
myriad/groestl: some more cleanup + tabs... 9 years ago			`void myriadhash(void state, const void input)`
Revision 0.6 with myriad-groestl and jackpot coin 10 years ago			`{`
myriad/groestl: some more cleanup + tabs... 9 years ago			`uint32_t _ALIGN(64) hash[16];`
myr-gr: clean up 10 years ago			`sph_groestl512_context ctx_groestl;`
			`SHA256_CTX sha256;`
Revision 0.6 with myriad-groestl and jackpot coin 10 years ago
myr-gr: clean up 10 years ago			`sph_groestl512_init(&ctx_groestl);`
myriad/groestl: some more cleanup + tabs... 9 years ago			`sph_groestl512(&ctx_groestl, input, 80);`
			`sph_groestl512_close(&ctx_groestl, hash);`
Revision 0.6 with myriad-groestl and jackpot coin 10 years ago
myr-gr: clean up 10 years ago			`SHA256_Init(&sha256);`
myriad/groestl: some more cleanup + tabs... 9 years ago			`SHA256_Update(&sha256,(unsigned char *)hash, 64);`
			`SHA256_Final((unsigned char *)hash, &sha256);`
Revision 0.6 with myriad-groestl and jackpot coin 10 years ago
myriad/groestl: some more cleanup + tabs... 9 years ago			`memcpy(state, hash, 32);`
Revision 0.6 with myriad-groestl and jackpot coin 10 years ago			`}`

Handle a maximum of 16 gpus (vs 8 before) Some cards have 2 gpus on board... 10 years ago			`static bool init[MAX_GPUS] = { 0 };`
Revision 0.6 with myriad-groestl and jackpot coin 10 years ago
start v1.7, apply new prototypes to all algos 9 years ago			`int scanhash_myriad(int thr_id, struct work work, uint32_t max_nonce, unsigned long hashes_done)`
Allow different intensity per device and clean the old variables, no more required 10 years ago			`{`
myriad/groestl: some more cleanup + tabs... 9 years ago			`uint32_t _ALIGN(64) endiandata[32];`
start v1.7, apply new prototypes to all algos 9 years ago			`uint32_t *pdata = work->data;`
			`uint32_t *ptarget = work->target;`
rename skein2 to c++, no cuda kernel code and some other changes... 9 years ago			`uint32_t start_nonce = pdata[19];`
myr-gr: remove unused allocated memory + pascal tweak + cleanup... 8 years ago			`int dev_id = device_map[thr_id];`
			`int intensity = (device_sm[dev_id] >= 600) ? 20 : 18;`
			`uint32_t throughput = cuda_default_throughput(thr_id, 1U << intensity);`
intensity: do not reduce throughput before init Else the memory allocated could be less than required later btw, use the new "cuda" function to apply intensity/throughput 9 years ago			`if (init[thr_id]) throughput = min(throughput, max_nonce - start_nonce);`
bump to revision V1.1 with Killer Groestl 10 years ago
Revision 0.6 with myriad-groestl and jackpot coin 10 years ago			`if (opt_benchmark)`
never interrupt global benchmark with found nonces fix some algo weird hashrates (like blake) and reset device between algos, for better accuracy but this reset doesnt seems enough to bench all algos correctly... to test on linux, could be a driver issue... heavy: fix first alloc and indent with tabs... 9 years ago			`ptarget[7] = 0x0000ff;`
Revision 0.6 with myriad-groestl and jackpot coin 10 years ago
			`// init`
			`if(!init[thr_id])`
			`{`
myr-gr: remove unused allocated memory + pascal tweak + cleanup... 8 years ago			`cudaSetDevice(dev_id);`
1.7.1 release set schedule flags to reduce linux cpu usage without MyStreamSynchronize() 9 years ago			`if (opt_cudaschedule == -1 && gpu_threads == 1) {`
			`cudaDeviceReset();`
			`// reduce cpu usage`
			`cudaSetDeviceFlags(cudaDeviceScheduleBlockingSync);`
			`CUDA_LOG_ERROR();`
			`}`
Show intensity on init for all algos 8 years ago			`gpulog(LOG_INFO, thr_id, "Intensity set to %g, %u cuda threads", throughput2intensity(throughput), throughput);`

api: report throughput when default 10 years ago			`myriadgroestl_cpu_init(thr_id, throughput);`
Revision 0.6 with myriad-groestl and jackpot coin 10 years ago			`init[thr_id] = true;`
			`}`
Allow different intensity per device and clean the old variables, no more required 10 years ago
myriad/groestl: some more cleanup + tabs... 9 years ago			`for (int k=0; k < 20; k++)`
			`be32enc(&endiandata[k], pdata[k]);`
Revision 0.6 with myriad-groestl and jackpot coin 10 years ago
myr-gr: handle a second nonce & more cleanup 8 years ago			`myriadgroestl_cpu_setBlock(thr_id, endiandata, ptarget);`
Allow different intensity per device and clean the old variables, no more required 10 years ago
Revision 0.6 with myriad-groestl and jackpot coin 10 years ago			`do {`
migrate 2nd nonce storage of most algos This allow to keep pdata[19] as cursor between scans, and later, to sort them.. remains... heavy, scrypt, sia... 7 years ago			`memset(work->nonces, 0xff, sizeof(work->nonces));`
Various algos cleanup + lyra2 sec nonce fix 9 years ago
migrate 2nd nonce storage of most algos This allow to keep pdata[19] as cursor between scans, and later, to sort them.. remains... heavy, scrypt, sia... 7 years ago			`// GPU`
			`myriadgroestl_cpu_hash(thr_id, throughput, pdata[19], work->nonces);`
Revision 0.6 with myriad-groestl and jackpot coin 10 years ago
never interrupt global benchmark with found nonces fix some algo weird hashrates (like blake) and reset device between algos, for better accuracy but this reset doesnt seems enough to bench all algos correctly... to test on linux, could be a driver issue... heavy: fix first alloc and indent with tabs... 9 years ago			`*hashes_done = pdata[19] - start_nonce + throughput;`

migrate 2nd nonce storage of most algos This allow to keep pdata[19] as cursor between scans, and later, to sort them.. remains... heavy, scrypt, sia... 7 years ago			`if (work->nonces[0] < UINT32_MAX && bench_algo < 0)`
Revision 0.6 with myriad-groestl and jackpot coin 10 years ago			`{`
start v1.7, apply new prototypes to all algos 9 years ago			`uint32_t _ALIGN(64) vhash[8];`
migrate 2nd nonce storage of most algos This allow to keep pdata[19] as cursor between scans, and later, to sort them.. remains... heavy, scrypt, sia... 7 years ago			`endiandata[19] = swab32(work->nonces[0]);`
start v1.7, apply new prototypes to all algos 9 years ago			`myriadhash(vhash, endiandata);`
			`if (vhash[7] <= ptarget[7] && fulltest(vhash, ptarget)) {`
migrate 2nd nonce storage of most algos This allow to keep pdata[19] as cursor between scans, and later, to sort them.. remains... heavy, scrypt, sia... 7 years ago			`work->valid_nonces = 1;`
diff: use the new function in all algos 9 years ago			`work_set_target_ratio(work, vhash);`
migrate 2nd nonce storage of most algos This allow to keep pdata[19] as cursor between scans, and later, to sort them.. remains... heavy, scrypt, sia... 7 years ago			`if (work->nonces[1] != UINT32_MAX) {`
			`endiandata[19] = swab32(work->nonces[1]);`
myr-gr: handle a second nonce & more cleanup 8 years ago			`myriadhash(vhash, endiandata);`
diff: show by default, rework shares diff storage This will allow later more gpu candidates. Note: This is an unfinished work, we keep the previous behavior for now To finish this, all algos solutions should be migrated and submitted nonces attributes stored. Its required to handle the different share diff per nonce and fix the possible solved count error (if 1/2 nonces is solved). 8 years ago			`bn_set_target_ratio(work, vhash, 1);`
migrate 2nd nonce storage of most algos This allow to keep pdata[19] as cursor between scans, and later, to sort them.. remains... heavy, scrypt, sia... 7 years ago			`work->valid_nonces = 2;`
			`pdata[19] = max(work->nonces[0], work->nonces[1]) + 1;`
			`} else {`
			`pdata[19] = work->nonces[0] + 1; // cursor`
myr-gr: handle a second nonce & more cleanup 8 years ago			`}`
migrate 2nd nonce storage of most algos This allow to keep pdata[19] as cursor between scans, and later, to sort them.. remains... heavy, scrypt, sia... 7 years ago			`return work->valid_nonces;`
			`}`
			`else if (vhash[7] > ptarget[7]) {`
api: report per thread cpu hash checks (ACC/REJ) + update all algos for that... 7 years ago			`gpu_increment_reject(thr_id);`
			`if (!opt_quiet)`
			`gpulog(LOG_WARNING, thr_id, "result for %08x does not validate on CPU!", work->nonces[0]);`
migrate 2nd nonce storage of most algos This allow to keep pdata[19] as cursor between scans, and later, to sort them.. remains... heavy, scrypt, sia... 7 years ago			`pdata[19] = work->nonces[0] + 1;`
			`continue;`
Revision 0.6 with myriad-groestl and jackpot coin 10 years ago			`}`
			`}`

never interrupt global benchmark with found nonces fix some algo weird hashrates (like blake) and reset device between algos, for better accuracy but this reset doesnt seems enough to bench all algos correctly... to test on linux, could be a driver issue... heavy: fix first alloc and indent with tabs... 9 years ago			`if ((uint64_t) throughput + pdata[19] >= max_nonce) {`
Revision 0.6 with myriad-groestl and jackpot coin 10 years ago			`pdata[19] = max_nonce;`
Add intensity to last algos and fix quark speed 10 years ago			`break;`
			`}`
api: report throughput when default 10 years ago			`pdata[19] += throughput;`
Revision 0.6 with myriad-groestl and jackpot coin 10 years ago
Add intensity to last algos and fix quark speed 10 years ago			`} while (!work_restart[thr_id].restart);`

never interrupt global benchmark with found nonces fix some algo weird hashrates (like blake) and reset device between algos, for better accuracy but this reset doesnt seems enough to bench all algos correctly... to test on linux, could be a driver issue... heavy: fix first alloc and indent with tabs... 9 years ago			`*hashes_done = max_nonce - start_nonce;`

Revision 0.6 with myriad-groestl and jackpot coin 10 years ago			`return 0;`
			`}`

algos: add functions to free allocated resources Will be used later for algo switching not really tested yet... 9 years ago			`// cleanup`
			`void free_myriad(int thr_id)`
			`{`
			`if (!init[thr_id])`
			`return;`

benchmark: enhance the mem leak detection reduce "false" warnings, and ignore unrelated/small ones <= 1 MB On windows the gpu memory can be allocated by other processes + some cleanup in algos... (free/gpulog) 9 years ago			`cudaThreadSynchronize();`
algos: add functions to free allocated resources Will be used later for algo switching not really tested yet... 9 years ago
			`myriadgroestl_cpu_free(thr_id);`
			`init[thr_id] = false;`

			`cudaDeviceSynchronize();`
intensity: do not reduce throughput before init Else the memory allocated could be less than required later btw, use the new "cuda" function to apply intensity/throughput 9 years ago			`}`