Mentions légales du service

Skip to content
Snippets Groups Projects
codelet_zgemm.c 4.26 KiB
Newer Older
 * @file starpu/codelet_zgemm.c
Mathieu Faverge's avatar
Mathieu Faverge committed
 * @copyright 2009-2014 The University of Tennessee and The University of
 *                      Tennessee Research Foundation. All rights reserved.
 * @copyright 2012-2020 Bordeaux INP, CNRS (LaBRI UMR 5800), Inria,
 *                      Univ. Bordeaux. All rights reserved.
 * @brief Chameleon zgemm StarPU codelet
 * @comment This file has been automatically generated
PRUVOST Florent's avatar
PRUVOST Florent committed
 *          from Plasma 2.5.0 for CHAMELEON 0.9.2
 * @author Hatem Ltaief
 * @author Jakub Kurzak
 * @author Mathieu Faverge
 * @author Emmanuel Agullo
 * @author Cedric Castagnede
BARROS DE ASSIS Lucas's avatar
BARROS DE ASSIS Lucas committed
 * @author Lucas Barros de Assis
#include "chameleon_starpu.h"
#include "runtime_codelet_z.h"
#if !defined(CHAMELEON_SIMULATION)
static void cl_zgemm_cpu_func(void *descr[], void *cl_arg)
{
    cham_trans_t transA;
    cham_trans_t transB;
    CHAMELEON_Complex64_t alpha;
    CHAMELEON_Complex64_t beta;
    tileA = cti_interface_get(descr[0]);
    tileB = cti_interface_get(descr[1]);
    tileC = cti_interface_get(descr[2]);
BARROS DE ASSIS Lucas's avatar
BARROS DE ASSIS Lucas committed

    starpu_codelet_unpack_args(cl_arg, &transA, &transB, &m, &n, &k, &alpha, &beta);
    TCORE_zgemm( transA, transB,
                 m, n, k,
                 alpha, tileA, tileB,
                 beta,  tileC );
#ifdef CHAMELEON_USE_CUDA
static void cl_zgemm_cuda_func(void *descr[], void *cl_arg)
{
    cham_trans_t transA;
    cham_trans_t transB;
    int m;
    int n;
    int k;
    cuDoubleComplex alpha;
    cuDoubleComplex beta;
    tileA = cti_interface_get(descr[0]);
    tileB = cti_interface_get(descr[1]);
    tileC = cti_interface_get(descr[2]);
BARROS DE ASSIS Lucas's avatar
BARROS DE ASSIS Lucas committed

    starpu_codelet_unpack_args(cl_arg, &transA, &transB, &m, &n, &k, &alpha, &beta);
Mathieu Faverge's avatar
Mathieu Faverge committed
    RUNTIME_getStream( stream );
        &alpha, tileA->mat, tileA->ld,
                tileB->mat, tileB->ld,
        &beta,  tileC->mat, tileC->ld,

#ifndef STARPU_CUDA_ASYNC
    cudaStreamSynchronize( stream );
#endif

    return;
}
#endif /* defined(CHAMELEON_USE_CUDA) */
#endif /* !defined(CHAMELEON_SIMULATION) */
CODELETS(zgemm, 3, cl_zgemm_cpu_func, cl_zgemm_cuda_func, STARPU_CUDA_ASYNC)

/**
 *
 * @ingroup INSERT_TASK_Complex64_t
 *
 */
void INSERT_TASK_zgemm(const RUNTIME_option_t *options,
                      cham_trans_t transA, cham_trans_t transB,
                      int m, int n, int k, int nb,
                      CHAMELEON_Complex64_t alpha, const CHAM_desc_t *A, int Am, int An,
                                                   const CHAM_desc_t *B, int Bm, int Bn,
                      CHAMELEON_Complex64_t beta,  const CHAM_desc_t *C, int Cm, int Cn)
{
    (void)nb;
    struct starpu_codelet *codelet = &cl_zgemm;
    void (*callback)(void*) = options->profiling ? cl_zgemm_callback : NULL;
    starpu_option_request_t* schedopt = (starpu_option_request_t *)(options->request->schedopt);
    int workerid = (schedopt == NULL) ? -1 : schedopt->workerid;

    CHAMELEON_BEGIN_ACCESS_DECLARATION;
    CHAMELEON_ACCESS_R(A, Am, An);
    CHAMELEON_ACCESS_R(B, Bm, Bn);
    CHAMELEON_ACCESS_RW(C, Cm, Cn);
    CHAMELEON_END_ACCESS_DECLARATION;

    starpu_insert_task(
        starpu_mpi_codelet(codelet),
        STARPU_VALUE,    &transA,            sizeof(int),
        STARPU_VALUE,    &transB,            sizeof(int),
        STARPU_VALUE,    &m,                 sizeof(int),
        STARPU_VALUE,    &n,                 sizeof(int),
        STARPU_VALUE,    &k,                 sizeof(int),
        STARPU_VALUE,    &alpha,             sizeof(CHAMELEON_Complex64_t),
        STARPU_R,         RTBLKADDR(A, CHAMELEON_Complex64_t, Am, An),
        STARPU_R,         RTBLKADDR(B, CHAMELEON_Complex64_t, Bm, Bn),
        STARPU_VALUE,    &beta,              sizeof(CHAMELEON_Complex64_t),
        STARPU_RW,        RTBLKADDR(C, CHAMELEON_Complex64_t, Cm, Cn),
        STARPU_PRIORITY,  options->priority,
        STARPU_CALLBACK,  callback,
#if defined(CHAMELEON_CODELETS_HAVE_NAME)
        STARPU_NAME, "zgemm",
#endif
        0);