Skip to content

Performance improvement opportunities #3

Description

@cianciosa

There are several places where the code using critical for reductions. For instance here

#pragma omp parallel default(none) shared(N1_dot, E1_dot, N2_dot, E2_dot, params, IONS, ss, CS, std::cout) firstprivate(NSP,NCP,DT)

#pragma omp parallel default(none) shared(N1_dot, E1_dot, N2_dot, E2_dot, params, IONS, ss, CS, std::cout) firstprivate(NSP,NCP,DT)
{
// Private variables:
    double N1_dot_private = 0;
    double N2_dot_private = 0;
    double E1_dot_private = 0;
    double E2_dot_private = 0;

    #pragma omp for
    for(int ii=0; ii<NSP; ii++)
    {
        // Boundary 1:
        if ( IONS->at(ss).f1(ii) == 1 )
        {
            double dE1 = IONS->at(ss).dE1(ii);
            double a_p = IONS->at(ss).a_p(ii);

            // Accumulate fluxes:
            N1_dot_private += (NCP/DT)*a_p;
            E1_dot_private += (NCP/DT)*a_p*dE1;

        }

        // Boundary 2:
        if ( IONS->at(ss).f2(ii) == 1 )
        {
            double dE2 = IONS->at(ss).dE2(ii);
            double a_p = IONS->at(ss).a_p(ii);

            // Accumulate fluxes:
            N2_dot_private += (NCP/DT)*a_p;
            E2_dot_private += (NCP/DT)*a_p*dE2;

        } // if

    } // omp for

    #pragma omp critical
    {
        N1_dot += N1_dot_private;
        N2_dot += N2_dot_private;
        E1_dot += E1_dot_private;
        E2_dot += E2_dot_private;
    }

} // omp parallel

can be rewritten to use openmp reduction clause.

const auto temp = NCP/DT;
#pragma omp parallel for default(none) shared(params, IONS, ss, temp) firstprivate(NSP) reduction(+:N1_dot, E1_dot, N2_dot, E2_dot)
for(int ii=0; ii<NSP; ii++)
{
    // Boundary 1:
    if ( IONS->at(ss).f1(ii) == 1 )
    {
        double dE1 = IONS->at(ss).dE1(ii);
        double a_p = IONS->at(ss).a_p(ii);

        // Accumulate fluxes:
        N1_dot += temp*a_p;
        E1_dot += temp*a_p*dE1;

     }

    // Boundary 2:
    if ( IONS->at(ss).f2(ii) == 1 )
    {
        double dE2 = IONS->at(ss).dE2(ii);
        double a_p = IONS->at(ss).a_p(ii);

        // Accumulate fluxes:
        N2_dot += temp*a_p;
        E2_dot += temp*a_p*dE2;

    } // if
} // omp parallel

Activity

Sign up for free to join this conversation on GitHub. Already have an account? Sign in to comment

Metadata

Metadata

Assignees

No one assigned

    Labels

    No labels
    No labels

    Type

    No type

    Projects

    No projects

      Milestone

      No milestone

      Relationships

      None yet

      Development

      No branches or pull requests

      Issue actions