Skip to content
New issue

Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.

By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.

Already on GitHub? Sign in to your account

Sacado local deep copy #9917

Merged
merged 4 commits into from
Nov 15, 2023
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
374 changes: 374 additions & 0 deletions packages/sacado/src/KokkosExp_View_Fad.hpp
Original file line number Diff line number Diff line change
Expand Up @@ -672,6 +672,380 @@ as_view_of_rank_n(View<T, Args...>) {

} // namespace Kokkos

namespace Kokkos {
namespace Experimental {

template <class DT, class... DP>
void KOKKOS_INLINE_FUNCTION local_deep_copy_contiguous(
const View<DT, DP...>& dst,
typename ViewTraits<DT, DP...>::const_value_type& value,
typename std::enable_if<(
( std::is_same< typename ViewTraits<DT,DP...>::specialize,
Kokkos::Impl::ViewSpecializeSacadoFad >::value
||
std::is_same< typename ViewTraits<DT,DP...>::specialize,
Kokkos::Impl::ViewSpecializeSacadoFadContiguous >::value )
&& (unsigned(ViewTraits<DT, DP...>::rank) == 1))>::type*)
{
if (dst.data() == nullptr)
return;

for (size_t i0 = 0; i0 < dst.extent(0); ++i0)
dst(i0) = value;
}

template <class DT, class... DP>
void KOKKOS_INLINE_FUNCTION local_deep_copy_contiguous(
const View<DT, DP...>& dst,
typename ViewTraits<DT, DP...>::const_value_type& value,
typename std::enable_if<(
( std::is_same< typename ViewTraits<DT,DP...>::specialize,
Kokkos::Impl::ViewSpecializeSacadoFad >::value
||
std::is_same< typename ViewTraits<DT,DP...>::specialize,
Kokkos::Impl::ViewSpecializeSacadoFadContiguous >::value )
&& (unsigned(ViewTraits<DT, DP...>::rank) == 2))>::type*)
{
if (dst.data() == nullptr)
return;

for (size_t i0 = 0; i0 < dst.extent(0); ++i0)
for (size_t i1 = 0; i1 < dst.extent(1); ++i1)
dst(i0,i1) = value;
}

template <class DT, class... DP>
void KOKKOS_INLINE_FUNCTION local_deep_copy_contiguous(
const View<DT, DP...>& dst,
typename ViewTraits<DT, DP...>::const_value_type& value,
typename std::enable_if<(
( std::is_same< typename ViewTraits<DT,DP...>::specialize,
Kokkos::Impl::ViewSpecializeSacadoFad >::value
||
std::is_same< typename ViewTraits<DT,DP...>::specialize,
Kokkos::Impl::ViewSpecializeSacadoFadContiguous >::value )
&& (unsigned(ViewTraits<DT, DP...>::rank) == 3))>::type*)
{
if (dst.data() == nullptr)
return;

for (size_t i0 = 0; i0 < dst.extent(0); ++i0)
for (size_t i1 = 0; i1 < dst.extent(1); ++i1)
for (size_t i2 = 0; i2 < dst.extent(2); ++i2)
dst(i0,i1,i2) = value;
}

template <class DT, class... DP>
void KOKKOS_INLINE_FUNCTION local_deep_copy_contiguous(
const View<DT, DP...>& dst,
typename ViewTraits<DT, DP...>::const_value_type& value,
typename std::enable_if<(
( std::is_same< typename ViewTraits<DT,DP...>::specialize,
Kokkos::Impl::ViewSpecializeSacadoFad >::value
||
std::is_same< typename ViewTraits<DT,DP...>::specialize,
Kokkos::Impl::ViewSpecializeSacadoFadContiguous >::value )
&& (unsigned(ViewTraits<DT, DP...>::rank) == 4))>::type*)
{
if (dst.data() == nullptr)
return;

for (size_t i0 = 0; i0 < dst.extent(0); ++i0)
for (size_t i1 = 0; i1 < dst.extent(1); ++i1)
for (size_t i2 = 0; i2 < dst.extent(2); ++i2)
for (size_t i3 = 0; i3 < dst.extent(3); ++i3)
dst(i0,i1,i2,i3) = value;
}

template <class DT, class... DP>
void KOKKOS_INLINE_FUNCTION local_deep_copy_contiguous(
const View<DT, DP...>& dst,
typename ViewTraits<DT, DP...>::const_value_type& value,
typename std::enable_if<(
( std::is_same< typename ViewTraits<DT,DP...>::specialize,
Kokkos::Impl::ViewSpecializeSacadoFad >::value
||
std::is_same< typename ViewTraits<DT,DP...>::specialize,
Kokkos::Impl::ViewSpecializeSacadoFadContiguous >::value )
&& (unsigned(ViewTraits<DT, DP...>::rank) == 5))>::type*)
{
if (dst.data() == nullptr)
return;

for (size_t i0 = 0; i0 < dst.extent(0); ++i0)
for (size_t i1 = 0; i1 < dst.extent(1); ++i1)
for (size_t i2 = 0; i2 < dst.extent(2); ++i2)
for (size_t i3 = 0; i3 < dst.extent(3); ++i3)
for (size_t i4 = 0; i4 < dst.extent(4); ++i4)
dst(i0,i1,i2,i3,i4) = value;
}

template <class DT, class... DP>
void KOKKOS_INLINE_FUNCTION local_deep_copy_contiguous(
const View<DT, DP...>& dst,
typename ViewTraits<DT, DP...>::const_value_type& value,
typename std::enable_if<(
( std::is_same< typename ViewTraits<DT,DP...>::specialize,
Kokkos::Impl::ViewSpecializeSacadoFad >::value
||
std::is_same< typename ViewTraits<DT,DP...>::specialize,
Kokkos::Impl::ViewSpecializeSacadoFadContiguous >::value )
&& (unsigned(ViewTraits<DT, DP...>::rank) == 6))>::type*)
{
if (dst.data() == nullptr)
return;

for (size_t i0 = 0; i0 < dst.extent(0); ++i0)
for (size_t i1 = 0; i1 < dst.extent(1); ++i1)
for (size_t i2 = 0; i2 < dst.extent(2); ++i2)
for (size_t i3 = 0; i3 < dst.extent(3); ++i3)
for (size_t i4 = 0; i4 < dst.extent(4); ++i4)
for (size_t i5 = 0; i5 < dst.extent(5); ++i5)
dst(i0,i1,i2,i3,i4,i5) = value;
}

template <class DT, class... DP>
void KOKKOS_INLINE_FUNCTION local_deep_copy_contiguous(
const View<DT, DP...>& dst,
typename ViewTraits<DT, DP...>::const_value_type& value,
typename std::enable_if<(
( std::is_same< typename ViewTraits<DT,DP...>::specialize,
Kokkos::Impl::ViewSpecializeSacadoFad >::value
||
std::is_same< typename ViewTraits<DT,DP...>::specialize,
Kokkos::Impl::ViewSpecializeSacadoFadContiguous >::value )
&& (unsigned(ViewTraits<DT, DP...>::rank) == 7))>::type*)
{
if (dst.data() == nullptr)
return;

for (size_t i0 = 0; i0 < dst.extent(0); ++i0)
for (size_t i1 = 0; i1 < dst.extent(1); ++i1)
for (size_t i2 = 0; i2 < dst.extent(2); ++i2)
for (size_t i3 = 0; i3 < dst.extent(3); ++i3)
for (size_t i4 = 0; i4 < dst.extent(4); ++i4)
for (size_t i5 = 0; i5 < dst.extent(5); ++i5)
for (size_t i6 = 0; i6 < dst.extent(6); ++i6)
dst(i0,i1,i2,i3,i4,i5,i6) = value;
}

template <class TeamType, class DT, class... DP>
void KOKKOS_INLINE_FUNCTION local_deep_copy_contiguous(
const TeamType& team, const View<DT, DP...>& dst,
typename ViewTraits<DT, DP...>::const_value_type& value,
typename std::enable_if<(
( std::is_same< typename ViewTraits<DT,DP...>::specialize,
Kokkos::Impl::ViewSpecializeSacadoFad >::value
||
std::is_same< typename ViewTraits<DT,DP...>::specialize,
Kokkos::Impl::ViewSpecializeSacadoFadContiguous >::value )
&& (unsigned(ViewTraits<DT, DP...>::rank) == 1))>::type*)
{
if (dst.data() == nullptr)
return;

const size_t N = dst.extent(0);

team.team_barrier();
Kokkos::parallel_for(Kokkos::TeamThreadRange(team, N),
[&](const int& i) { dst(i) = value; });
team.team_barrier();
}

template <class TeamType, class DT, class... DP>
void KOKKOS_INLINE_FUNCTION local_deep_copy_contiguous(
const TeamType& team, const View<DT, DP...>& dst,
typename ViewTraits<DT, DP...>::const_value_type& value,
typename std::enable_if<(
( std::is_same< typename ViewTraits<DT,DP...>::specialize,
Kokkos::Impl::ViewSpecializeSacadoFad >::value
||
std::is_same< typename ViewTraits<DT,DP...>::specialize,
Kokkos::Impl::ViewSpecializeSacadoFadContiguous >::value )
&& (unsigned(ViewTraits<DT, DP...>::rank) == 2))>::type*)
{
if (dst.data() == nullptr)
return;

const size_t N = dst.extent(0) * dst.extent(1);

team.team_barrier();
Kokkos::parallel_for(Kokkos::TeamThreadRange(team, N), [&](const int& i) {
int i0 = i % dst.extent(0);
int i1 = i / dst.extent(0);
dst(i0, i1) = value;
});
team.team_barrier();
}

template <class TeamType, class DT, class... DP>
void KOKKOS_INLINE_FUNCTION local_deep_copy_contiguous(
const TeamType& team, const View<DT, DP...>& dst,
typename ViewTraits<DT, DP...>::const_value_type& value,
typename std::enable_if<(
( std::is_same< typename ViewTraits<DT,DP...>::specialize,
Kokkos::Impl::ViewSpecializeSacadoFad >::value
||
std::is_same< typename ViewTraits<DT,DP...>::specialize,
Kokkos::Impl::ViewSpecializeSacadoFadContiguous >::value )
&& (unsigned(ViewTraits<DT, DP...>::rank) == 3))>::type*)
{
if (dst.data() == nullptr)
return;

const size_t N = dst.extent(0) * dst.extent(1) * dst.extent(2);

team.team_barrier();
Kokkos::parallel_for(Kokkos::TeamThreadRange(team, N), [&](const int& i) {
int i0 = i % dst.extent(0);
int itmp = i / dst.extent(0);
int i1 = itmp % dst.extent(1);
int i2 = itmp / dst.extent(1);
dst(i0, i1, i2) = value;
});
team.team_barrier();
}

template <class TeamType, class DT, class... DP>
void KOKKOS_INLINE_FUNCTION local_deep_copy_contiguous(
const TeamType& team, const View<DT, DP...>& dst,
typename ViewTraits<DT, DP...>::const_value_type& value,
typename std::enable_if<(
( std::is_same< typename ViewTraits<DT,DP...>::specialize,
Kokkos::Impl::ViewSpecializeSacadoFad >::value
||
std::is_same< typename ViewTraits<DT,DP...>::specialize,
Kokkos::Impl::ViewSpecializeSacadoFadContiguous >::value )
&& (unsigned(ViewTraits<DT, DP...>::rank) == 4))>::type*)
{
if (dst.data() == nullptr)
return;

const size_t N =
dst.extent(0) * dst.extent(1) * dst.extent(2) * dst.extent(3);

team.team_barrier();
Kokkos::parallel_for(Kokkos::TeamThreadRange(team, N), [&](const int& i) {
int i0 = i % dst.extent(0);
int itmp = i / dst.extent(0);
int i1 = itmp % dst.extent(1);
itmp = itmp / dst.extent(1);
int i2 = itmp % dst.extent(2);
int i3 = itmp / dst.extent(2);
dst(i0, i1, i2, i3) = value;
});
team.team_barrier();
}

template <class TeamType, class DT, class... DP>
void KOKKOS_INLINE_FUNCTION local_deep_copy_contiguous(
const TeamType& team, const View<DT, DP...>& dst,
typename ViewTraits<DT, DP...>::const_value_type& value,
typename std::enable_if<(
( std::is_same< typename ViewTraits<DT,DP...>::specialize,
Kokkos::Impl::ViewSpecializeSacadoFad >::value
||
std::is_same< typename ViewTraits<DT,DP...>::specialize,
Kokkos::Impl::ViewSpecializeSacadoFadContiguous >::value )
&& (unsigned(ViewTraits<DT, DP...>::rank) == 5))>::type*)
{
if (dst.data() == nullptr)
return;

const size_t N = dst.extent(0) * dst.extent(1) * dst.extent(2) *
dst.extent(3) * dst.extent(4);

team.team_barrier();
Kokkos::parallel_for(Kokkos::TeamThreadRange(team, N), [&](const int& i) {
int i0 = i % dst.extent(0);
int itmp = i / dst.extent(0);
int i1 = itmp % dst.extent(1);
itmp = itmp / dst.extent(1);
int i2 = itmp % dst.extent(2);
itmp = itmp / dst.extent(2);
int i3 = itmp % dst.extent(3);
int i4 = itmp / dst.extent(3);
dst(i0, i1, i2, i3, i4) = value;
});
team.team_barrier();
}

template <class TeamType, class DT, class... DP>
void KOKKOS_INLINE_FUNCTION local_deep_copy_contiguous(
const TeamType& team, const View<DT, DP...>& dst,
typename ViewTraits<DT, DP...>::const_value_type& value,
typename std::enable_if<(
( std::is_same< typename ViewTraits<DT,DP...>::specialize,
Kokkos::Impl::ViewSpecializeSacadoFad >::value
||
std::is_same< typename ViewTraits<DT,DP...>::specialize,
Kokkos::Impl::ViewSpecializeSacadoFadContiguous >::value )
&& (unsigned(ViewTraits<DT, DP...>::rank) == 6))>::type*)
{
if (dst.data() == nullptr)
return;

const size_t N = dst.extent(0) * dst.extent(1) * dst.extent(2) *
dst.extent(3) * dst.extent(4) * dst.extent(5);

team.team_barrier();
Kokkos::parallel_for(Kokkos::TeamThreadRange(team, N), [&](const int& i) {
int i0 = i % dst.extent(0);
int itmp = i / dst.extent(0);
int i1 = itmp % dst.extent(1);
itmp = itmp / dst.extent(1);
int i2 = itmp % dst.extent(2);
itmp = itmp / dst.extent(2);
int i3 = itmp % dst.extent(3);
itmp = itmp / dst.extent(3);
int i4 = itmp % dst.extent(4);
int i5 = itmp / dst.extent(4);
dst(i0, i1, i2, i3, i4, i5) = value;
});
team.team_barrier();
}

template <class TeamType, class DT, class... DP>
void KOKKOS_INLINE_FUNCTION local_deep_copy_contiguous(
const TeamType& team, const View<DT, DP...>& dst,
typename ViewTraits<DT, DP...>::const_value_type& value,
typename std::enable_if<(
( std::is_same< typename ViewTraits<DT,DP...>::specialize,
Kokkos::Impl::ViewSpecializeSacadoFad >::value
||
std::is_same< typename ViewTraits<DT,DP...>::specialize,
Kokkos::Impl::ViewSpecializeSacadoFadContiguous >::value )
&& (unsigned(ViewTraits<DT, DP...>::rank) == 7))>::type*)
{
if (dst.data() == nullptr)
return;

const size_t N = dst.extent(0) * dst.extent(1) * dst.extent(2) *
dst.extent(3) * dst.extent(4) * dst.extent(5) *
dst.extent(6);

team.team_barrier();
Kokkos::parallel_for(Kokkos::TeamThreadRange(team, N), [&](const int& i) {
int i0 = i % dst.extent(0);
int itmp = i / dst.extent(0);
int i1 = itmp % dst.extent(1);
itmp = itmp / dst.extent(1);
int i2 = itmp % dst.extent(2);
itmp = itmp / dst.extent(2);
int i3 = itmp % dst.extent(3);
itmp = itmp / dst.extent(3);
int i4 = itmp % dst.extent(4);
itmp = itmp / dst.extent(4);
int i5 = itmp % dst.extent(5);
int i6 = itmp / dst.extent(5);
dst(i0, i1, i2, i3, i4, i5, i6) = value;
});
team.team_barrier();
}

} // namespace Experimental
} // namespace Kokkos

//----------------------------------------------------------------------------

namespace Kokkos {
Expand Down
Loading