{
  "slug": "ml-disk-failure-prediction",
  "title": "Adding performance and location data lifted 10-day disk failure prediction to 0.95 MCC across 380,000 disks",
  "kind": "dataset",
  "summary": "Lu et al., FAST 2020, studied 380,000 hard disks in 64 sites of one operator over about 70 days. With SMART, performance, and location data together, a CNN-LSTM scored 0.95 MCC for a 10-day prediction horizon, against 0.77 for the next best method, a random forest. Rows from Backblaze (2016) and Google (FAST 2007) show the SMART gap in other fleets.",
  "license": "Small derived table of figures printed in Lu et al. (FAST 2020), Backblaze (6 October 2016), and Pinheiro et al. (FAST 2007), with credit. Not a Creative Commons license. The USENIX proceedings state that rights to individual papers remain with the author or the author's employer. This file is not a copy of the papers and not the data set.",
  "licenseUrl": "https://www.usenix.org/system/files/fast20-lu.pdf",
  "sourceUrl": "https://www.usenix.org/system/files/fast20-lu.pdf",
  "accessed": "2026-10-10",
  "rowCount": 13,
  "columns": [
    {
      "name": "measure",
      "type": "string",
      "description": "The quantity as the source names it."
    },
    {
      "name": "value_low",
      "type": "number",
      "description": "Lower end of a printed range. Empty when the source prints a single value."
    },
    {
      "name": "value_high",
      "type": "number",
      "description": "Single printed value, or the upper end of a range. For 'more than half' it is the stated bound."
    },
    {
      "name": "unit",
      "type": "string",
      "description": "Unit of the value columns: disks, blocks, files, percent of disks, percent of mismatches, or probability."
    },
    {
      "name": "scope",
      "type": "string",
      "description": "Population and window the value applies to."
    },
    {
      "name": "note",
      "type": "string",
      "description": "What the value is not, or the source wording behind it."
    },
    {
      "name": "citation_id",
      "type": "string",
      "description": "Id of the opened source in the citations list."
    }
  ],
  "downloadPath": "/data/datasets/ml-disk-failure-prediction.csv",
  "humanPage": "https://hesela.com/analyses/ml-disk-failure-prediction/",
  "limits": [
    "Scores are the paper's own, from one operator, about 70 days, and a 10-day horizon.",
    "The number of failed disks is not stated in the text read.",
    "The paper prints 2.6 million device hours, which does not match 380,000 disks over about 70 days.",
    "Portability across sites held for the CNN-LSTM but not always for RF and GBDT.",
    "The Backblaze and Google rows measure SMART warnings, not model performance."
  ],
  "citations": [
    {
      "id": "lu-fast20-pdf",
      "title": "Making Disk Failure Predictions SMARTer!, Lu, Luo, Patel, Yao, Tiwari, and Shi, FAST 2020 (USENIX PDF, pages 151 to 167)",
      "url": "https://www.usenix.org/system/files/fast20-lu.pdf",
      "accessed": "2026-10-10"
    },
    {
      "id": "backblaze-smart-2016",
      "title": "What SMART Stats Tell Us About Hard Drives, Backblaze, 6 October 2016",
      "url": "https://www.backblaze.com/blog/what-smart-stats-indicate-hard-drive-failures/",
      "accessed": "2026-10-10"
    },
    {
      "id": "pinheiro-usenix-html",
      "title": "Failure Trends in a Large Disk Drive Population, Pinheiro, Weber, and Barroso, FAST 2007 (USENIX HTML proceedings)",
      "url": "https://www.usenix.org/legacy/events/fast07/tech/full_papers/pinheiro/pinheiro_html/index.html",
      "accessed": "2026-10-09"
    }
  ]
}