Fix typos

This commit is contained in:
Mike Taves
2024-08-02 00:06:24 +00:00
committed by Charles Schlosser
parent fd98cc49f1
commit c593e9e948
26 changed files with 38 additions and 38 deletions

View File

@@ -898,7 +898,7 @@ containing the real part of the complex values of the original tensor.
### (Operation) imag()
Returns a tensor with the same dimensions as the orginal tensor
Returns a tensor with the same dimensions as the original tensor
containing the imaginary part of the complex values of the original
tensor.
@@ -910,7 +910,7 @@ exponent.
The type of the exponent, Scalar, is always the same as the type of the
tensor coefficients. For example, only integer exponents can be used in
conjuntion with tensors of integer values.
conjunction with tensors of integer values.
You can use cast() to lift this restriction. For example this computes
cubic roots of an int Tensor:

View File

@@ -104,10 +104,10 @@ struct TTPanelSize {
static EIGEN_CONSTEXPR StorageIndex TileSizeDimM = LocalThreadSizeM * WorkLoadPerThreadM;
// TileSizeDimN: determines the tile size for the n dimension
static EIGEN_CONSTEXPR StorageIndex TileSizeDimN = LocalThreadSizeN * WorkLoadPerThreadN;
// LoadPerThreadLhs: determines workload per thread for loading Lhs Tensor. This must be divisable by packetsize
// LoadPerThreadLhs: determines workload per thread for loading Lhs Tensor. This must be divisible by packetsize
static EIGEN_CONSTEXPR StorageIndex LoadPerThreadLhs =
((TileSizeDimK * WorkLoadPerThreadM * WorkLoadPerThreadN) / (TileSizeDimN));
// LoadPerThreadRhs: determines workload per thread for loading Rhs Tensor. This must be divisable by packetsize
// LoadPerThreadRhs: determines workload per thread for loading Rhs Tensor. This must be divisible by packetsize
static EIGEN_CONSTEXPR StorageIndex LoadPerThreadRhs =
((TileSizeDimK * WorkLoadPerThreadM * WorkLoadPerThreadN) / (TileSizeDimM));
// BC : determines if supporting bank conflict is required
@@ -674,7 +674,7 @@ class TensorContractionKernel {
for (StorageIndex wLPTN = 0; wLPTN < Properties::WorkLoadPerThreadN / PrivateNStride; wLPTN++) {
// output leading dimension
StorageIndex outputLD = 0;
// When local memory is used the PrivateNstride is always 1 because the coalesed access on N is loaded into Local
// When local memory is used the PrivateNstride is always 1 because the coalesced access on N is loaded into Local
// memory and extracting from local to global is the same as no transposed version. However, when local memory is
// not used and RHS is transposed we packetize the load for RHS.
EIGEN_UNROLL_LOOP
@@ -898,7 +898,7 @@ class TensorContractionKernel {
EIGEN_CONSTEXPR StorageIndex LSD = InputBlockProperties::is_rhs ? LSDR : LSDL;
static_assert(((LocalOffset % (TileSizeDimNC / InputBlockProperties::nc_stride) == 0) &&
(LocalOffset % (Properties::TileSizeDimK / InputBlockProperties::c_stride) == 0)),
" LocalOffset must be divisable by stride");
" LocalOffset must be divisible by stride");
const StorageIndex &NC = InputBlockProperties::is_rhs ? triple_dim.N : triple_dim.M;
StorageIndex localThreadNC = local_index.first;
StorageIndex localThreadC = local_index.second;

View File

@@ -36,7 +36,7 @@ EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE T* constCast(const T* data) {
// used for referring to a Pointer on TensorEvaluator class. While the TensorExpression
// is a device-agnostic type and need MakePointer class for type conversion,
// the TensorEvaluator class can be specialized for a device, hence it is possible
// to construct different types of temproray storage memory in TensorEvaluator
// to construct different types of temporary storage memory in TensorEvaluator
// for different devices by specializing the following StorageMemory class.
template <typename T, typename device>
struct StorageMemory : MakePointer<T> {};

View File

@@ -136,7 +136,7 @@ class UniformRandomGenerator {
// thread but for SYCL ((CLOCK * 6364136223846793005ULL) + 0xda3e39cb94b95bdbULL) is passed to each thread and each
// thread adds the (global_thread_id* 6364136223846793005ULL) for itself only once, in order to complete the
// construction similar to CUDA Therefore, the thread Id injection is not available at this stage.
// However when the operator() is called the thread ID will be available. So inside the opeator,
// However when the operator() is called the thread ID will be available. So inside the operator,
// we add the thrreadID, BlockId,... (which is equivalent of i)
// to the seed and construct the unique m_state per thead similar to cuda.
m_exec_once = false;

View File

@@ -433,7 +433,7 @@ struct PartialReducerLauncher {
EIGEN_CONSTEXPR Index localRange = PannelParameters::LocalThreadSizeP * PannelParameters::LocalThreadSizeR;
// In this step, we force the code not to be more than 2-step reduction:
// Our empirical research shows that if each thread reduces at least 64
// elemnts individually, we get better performance. However, this can change
// elements individually, we get better performance. However, this can change
// on different platforms. In this step we force the code not to be
// morthan step reduction: Our empirical research shows that for inner_most
// dim reducer, it is better to have 8 group in a reduce dimension for sizes
@@ -495,7 +495,7 @@ struct FullReducer<Self, Op, Eigen::SyclDevice, Vectorizable> {
typename Self::Index inputSize = self.impl().dimensions().TotalSize();
// In this step we force the code not to be more than 2-step reduction:
// Our empirical research shows that if each thread reduces at least 512
// elemnts individually, we get better performance.
// elements individually, we get better performance.
const Index reductionPerThread = 2048;
// const Index num_work_group =
Index reductionGroup = dev.getPowerOfTwo(