/* These are memory addresses as would be seen by one or more EEPROM *chipsstrungontheI2Cbus,usuallybymanipulatingpins1-3ofa *setofEEPROMdevices.Theyformacontinuousmemoryspace. * *TheI2Cdeviceaddressincludesthedevicetypeidentifier,1010b, *whichisareservedvalueandindicatesthatthisisanI2CEEPROM *device.Italsoincludesthetop3bitsofthe19bitEEPROMmemory *address,namelybits18,17,and16.Thismakesupthe7bit *addresssentontheI2Cbuswithbit0beingthedirectionbit, *whichisnotrepresentedhere,andsentbythehardwaredirectly. * *Forinstance, *50h=1010000b=>devicetypeidentifier1010b,bits18:16=000b,address0. *54h=1010100b=>--"--,bits18:16=100b,address40000h. *56h=1010110b=>--"--,bits18:16=110b,address60000h. *DependingonthesizeoftheI2CEEPROMdevice(s),bits18:16may *addressmemoryinadeviceoradeviceontheI2Cbus,dependingon *thestatusofpins1-3.Seetopofamdgpu_eeprom.c. * *TheRAStableliveseitherataddress0oraddress40000hofEEPROM.
*/ #define EEPROM_I2C_MADDR_0 0x0 #define EEPROM_I2C_MADDR_4 0x40000
/* Given a zero-based index of an EEPROM RAS record, yields the EEPROM *offsetoffofRAS_TABLE_START.Thatis,thisissomethingyoucan *addtocontrol->i2c_address,andthentellI2Clayertoread *from/writetothere._Nisthesocalledabsoluteindex, *becauseitstartsrightafterthetableheader.
*/ #define RAS_INDEX_TO_OFFSET(_C, _N) ((_C)->ras_record_offset + \
(_N) * RAS_TABLE_RECORD_SIZE)
/* i2c may be unstable in gpu reset */
down_read(&adev->reset_domain->sem);
res = amdgpu_eeprom_write(adev->pm.ras_eeprom_i2c_bus,
control->i2c_address +
control->ras_header_offset,
buf, RAS_TABLE_HEADER_SIZE);
up_read(&adev->reset_domain->sem);
if (res < 0) {
dev_err(adev->dev, "Failed to write EEPROM table header:%d",
res);
} elseif (res < RAS_TABLE_HEADER_SIZE) {
dev_err(adev->dev, "Short write:%d out of %d\n", res,
RAS_TABLE_HEADER_SIZE);
res = -EIO;
} else {
res = 0;
}
/* i2c may be unstable in gpu reset */
down_read(&adev->reset_domain->sem);
res = amdgpu_eeprom_write(adev->pm.ras_eeprom_i2c_bus,
control->i2c_address +
control->ras_info_offset,
buf, RAS_TABLE_V2_1_INFO_SIZE);
up_read(&adev->reset_domain->sem);
if (res < 0) {
dev_err(adev->dev, "Failed to write EEPROM table ras info:%d",
res);
} elseif (res < RAS_TABLE_V2_1_INFO_SIZE) {
dev_err(adev->dev, "Short write:%d out of %d\n", res,
RAS_TABLE_V2_1_INFO_SIZE);
res = -EIO;
} else {
res = 0;
}
buf = kcalloc(num, RAS_TABLE_RECORD_SIZE, GFP_KERNEL); if (!buf) return -ENOMEM;
/* Encode all of them in one go.
*/
pp = buf; for (i = 0; i < num; i++, pp += RAS_TABLE_RECORD_SIZE) {
__encode_table_record_to_buf(control, &record[i], pp);
/* a, first record index to write into. *b,lastrecordindextowriteinto. *a=firstindextoread(fri)+numberofrecordsinthetable, *b=a+@num-1. *LetN=control->ras_max_num_record_count,thenwehave, *case0:0<=a<=b<N, *justappend@numrecordsstartingata; *case1:0<=a<N<=b, *append(N-a)recordsstartingata,and *appendtheremainder,b%N+1,startingat0. *case2:0<=fri<N<=a<=b,thenmoduloNwegettwosubcases, *case2a:0<=a<=b<N *appendnumrecordsstartingata;andfixfriifboverwroteit, *andsincea<=b,ifboverwroteitthenamust'vealso, *andifbdidn'toverwriteit,thenadidn'talso. *case2b:0<=b<a<N *writenumrecordsstartingata,whichwrapsaround0=N *andoverwritefriunconditionally.Nowfromcase2a, *thismeansthatbeclipsedfritooverwriteitandwrap *around0again,i.e.b=2N+rpremoduloN,soweunconditionally *setfri=b+1(modN). *Now,sincefriisupdatedineverycase,exceptthetrivialcase0, *thenumberofrecordspresentinthetableafterwriting,is, *num_recs-1=b-fri(modN),andwetakethepositivevalue, *byaddinganarbitrarymultipleofNbeforetakingthemoduloN *asshownbelow.
*/
a = control->ras_fri + control->ras_num_recs;
b = a + num - 1; if (b < control->ras_max_record_count) {
res = __amdgpu_ras_eeprom_write(control, buf, a, num);
} elseif (a < control->ras_max_record_count) {
u32 g0, g1;
g0 = control->ras_max_record_count - a;
g1 = b % control->ras_max_record_count + 1;
res = __amdgpu_ras_eeprom_write(control, buf, a, g0); if (res) goto Out;
res = __amdgpu_ras_eeprom_write(control,
buf + g0 * RAS_TABLE_RECORD_SIZE, 0, g1); if (res) goto Out; if (g1 > control->ras_fri)
control->ras_fri = g1 % control->ras_max_record_count;
} else {
a %= control->ras_max_record_count;
b %= control->ras_max_record_count;
if (a <= b) { /* Note that, b - a + 1 = num. */
res = __amdgpu_ras_eeprom_write(control, buf, a, num); if (res) goto Out; if (b >= control->ras_fri)
control->ras_fri = (b + 1) % control->ras_max_record_count;
} else {
u32 g0, g1;
/* b < a, which means, we write from *atotheendofthetable,andfrom *thestartofthetabletob.
*/
g0 = control->ras_max_record_count - a;
g1 = b + 1;
res = __amdgpu_ras_eeprom_write(control, buf, a, g0); if (res) goto Out;
res = __amdgpu_ras_eeprom_write(control,
buf + g0 * RAS_TABLE_RECORD_SIZE, 0, g1); if (res) goto Out;
control->ras_fri = g1 % control->ras_max_record_count;
}
}
control->ras_num_recs = 1 + (control->ras_max_record_count + b
- control->ras_fri)
% control->ras_max_record_count;
/*old asics only save pa to eeprom like before*/ if (IP_VERSION_MAJ(amdgpu_ip_version(adev, UMC_HWIP, 0)) < 12)
control->ras_num_pa_recs += num; else
control->ras_num_mca_recs += num;
if (num == 0) {
dev_err(adev->dev, "will not append 0 records\n"); return -EINVAL;
} elseif (num > control->ras_max_record_count) {
dev_err(adev->dev, "cannot append %d records than the size of table %d\n",
num, control->ras_max_record_count); return -EINVAL;
}
if (adev->gmc.gmc_funcs->query_mem_partition_mode)
nps = adev->gmc.gmc_funcs->query_mem_partition_mode(adev);
/* set the new channel index flag */ for (i = 0; i < num; i++)
record[i].retired_page |= (nps << UMC_NPS_SHIFT);
mutex_lock(&control->ras_tbl_mutex);
res = amdgpu_ras_eeprom_append_table(control, record, num); if (!res)
res = amdgpu_ras_eeprom_update_header(control); if (!res)
amdgpu_ras_debugfs_set_ret_size(control);
mutex_unlock(&control->ras_tbl_mutex);
/* clear channel index flag, the flag is only saved on eeprom */ for (i = 0; i < num; i++)
record[i].retired_page &= ~(nps << UMC_NPS_SHIFT);
if (num == 0) {
dev_err(adev->dev, "will not read 0 records\n"); return -EINVAL;
} elseif (num > control->ras_num_recs) {
dev_err(adev->dev, "too many records to read:%d available:%d\n",
num, control->ras_num_recs); return -EINVAL;
}
buf = kcalloc(num, RAS_TABLE_RECORD_SIZE, GFP_KERNEL); if (!buf) return -ENOMEM;
/* Determine how many records to read, from the first record *index,fri,totheendofthetable,andfromthebeginning *ofthetable,suchthatthetotalnumberofrecordsis *@num,andwehandlewraparoundwhenfri>0and *fri+num>RAS_MAX_RECORD_COUNT. * *Firstwecomputetheindexofthelastelement *whichwouldbefetchedfromeachregion, *g0isin[fri,fri+num-1],and *g1isin[0,RAS_MAX_RECORD_COUNT-1]. *Then,ifg0<RAS_MAX_RECORD_COUNT,theindexof *thelastelementtofetch,wesetg0to_thenumber_ *ofelementstofetch,@num,sinceweknowthatthelast *indexedtobefetcheddoesnotexceedthetable. * *If,however,g0>=RAS_MAX_RECORD_COUNT,then *wesetg0tothenumberofelementstoread *untiltheendofthetable,andg1tothenumberof *elementstoreadfromthebeginningofthetable.
*/
g0 = control->ras_fri + num - 1;
g1 = g0 % control->ras_max_record_count; if (g0 < control->ras_max_record_count) {
g0 = num;
g1 = 0;
} else {
g0 = control->ras_max_record_count - control->ras_fri;
g1 += 1;
}
mutex_lock(&control->ras_tbl_mutex);
res = __amdgpu_ras_eeprom_read(control, buf, control->ras_fri, g0); if (res) goto Out; if (g1) {
res = __amdgpu_ras_eeprom_read(control,
buf + g0 * RAS_TABLE_RECORD_SIZE, 0, g1); if (res) goto Out;
}
res = 0;
/* Read up everything? Then transform.
*/
pp = buf; for (i = 0; i < num; i++, pp += RAS_TABLE_RECORD_SIZE) {
__decode_table_record_from_buf(control, &record[i], pp);
uint32_t amdgpu_ras_eeprom_max_record_count(struct amdgpu_ras_eeprom_control *control)
{ /* get available eeprom table version first before eeprom table init */
amdgpu_ras_set_eeprom_table_version(control);
if (control->tbl_hdr.version >= RAS_TABLE_VER_V2_1) return RAS_MAX_RECORD_COUNT_V2_1; else return RAS_MAX_RECORD_COUNT;
}
/* Find the starting record index
*/
s = *pos - strlen(tbl_hdr_str) - tbl_hdr_fmt_size -
strlen(rec_hdr_str);
s = s / rec_hdr_fmt_size;
r = *pos - strlen(tbl_hdr_str) - tbl_hdr_fmt_size -
strlen(rec_hdr_str);
r = r % rec_hdr_fmt_size;
for ( ; size > 0 && s < control->ras_num_recs; s++) {
u32 ai = RAS_RI_TO_AI(control, s); /* Read a single record
*/
res = __amdgpu_ras_eeprom_read(control, dare, ai, 1); if (res) goto Out;
__decode_table_record_from_buf(control, &record, dare);
snprintf(data, sizeof(data), rec_hdr_fmt,
s,
RAS_INDEX_TO_OFFSET(control, ai),
record_err_type_str[record.err_type],
record.bank,
record.ts,
record.offset,
record.mem_channel,
record.mcumc_id,
record.retired_page);
data_len = min_t(size_t, rec_hdr_fmt_size - r, size); if (copy_to_user(buf, &data[r], data_len)) {
res = -EFAULT; goto Out;
}
buf += data_len;
size -= data_len;
*pos += data_len;
r = 0;
}
}
res = 0;
Out:
mutex_unlock(&control->ras_tbl_mutex); return res < 0 ? res : orig_size - size;
}
if (hdr->header == RAS_TABLE_HDR_VAL) {
dev_dbg(adev->dev, "Found existing EEPROM table with %d records",
control->ras_num_bad_pages);
if (hdr->version >= RAS_TABLE_VER_V2_1) {
res = __read_table_ras_info(control); if (res) return res;
}
res = __verify_ras_table_checksum(control); if (res)
dev_err(adev->dev, "RAS table incorrect checksum or error:%d\n",
res);
/* Warn if we are at 90% of the threshold or above
*/ if (10 * control->ras_num_bad_pages >= 9 * ras->bad_page_cnt_threshold)
dev_warn(adev->dev, "RAS records:%u exceeds 90%% of threshold:%d",
control->ras_num_bad_pages,
ras->bad_page_cnt_threshold);
} elseif (hdr->header == RAS_TABLE_HDR_BAD &&
amdgpu_bad_page_threshold != 0) { if (hdr->version >= RAS_TABLE_VER_V2_1) {
res = __read_table_ras_info(control); if (res) return res;
}
res = __verify_ras_table_checksum(control); if (res) {
dev_err(adev->dev, "RAS Table incorrect checksum or error:%d\n",
res); return -EINVAL;
} if (ras->bad_page_cnt_threshold >= control->ras_num_bad_pages) { /* This means that, the threshold was increased since *thelasttimethesystemwasbooted,andnow, *ras->bad_page_cnt_threshold-control->num_recs>0, *sothatatleastonemorerecordcanbesaved, *beforethepagecountthresholdisreached.
*/
dev_info(adev->dev, "records:%d threshold:%d, resetting " "RAS table header signature",
control->ras_num_bad_pages,
ras->bad_page_cnt_threshold);
res = amdgpu_ras_eeprom_correct_header_tag(control,
RAS_TABLE_HDR_VAL);
} else {
dev_warn(adev->dev, "RAS records:%d exceed threshold:%d\n",
control->ras_num_bad_pages, ras->bad_page_cnt_threshold); if ((amdgpu_bad_page_threshold == -1) ||
(amdgpu_bad_page_threshold == -2)) {
res = 0;
dev_warn(adev->dev, "Please consult AMD Service Action Guide (SAG) for appropriate service procedures\n");
} else {
ras->is_rma = true;
dev_warn(adev->dev, "User defined threshold is set, runtime service will be halt when threshold is reached\n");
}
}
}
Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.
Bemerkung:
Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.