{
  "report_type": "ryedore.benchmark.v1",
  "generated_utc": "2026-08-10T04:55:29Z",
  "source_results": "/tmp/anomaly_FD003/results.json",
  "environment": {
    "python": "3.11.15",
    "platform": "Linux-6.6.87.2-microsoft-standard-WSL2-x86_64-with-glibc2.35",
    "numpy": "1.26.4",
    "tensorflow": "2.20.0",
    "torch": "2.5.1+cu124",
    "xgboost": "3.2.0",
    "code_ref": "no_git"
  },
  "benchmark": {
    "benchmark": "cmapss_fd003_timeseries_anomaly",
    "dataset": "CMAPSS FD003 (NASA public domain) \u2014 REAL near-failure label (RUL<=30)",
    "real_anomaly_rate": 0.1301,
    "question": "On a genuine TIME-SERIES, does the temporal BiLSTM beat per-row IsolationForest for real degradation/anomaly detection? (ai4i tabular said no)",
    "setup": {
      "model": "ModelFactory 'default' BiLSTM+attention",
      "lookback": 10,
      "seeds": [
        11,
        23,
        42
      ],
      "held_out": "stratified 0.20 on REAL label",
      "sequences": "engine-aware (never cross engine boundaries)",
      "label": "RUL<=30 => anomaly",
      "gpu": "/physical_device:GPU:0 capped 3072MB",
      "tf": "2.20.0"
    },
    "isolation": {
      "writes": "scratch only (/tmp/anomaly_FD003)",
      "production_outputs_touched": false,
      "db_writes": false
    },
    "results": {
      "prod_pseudo (BiLSTM on pseudo-labels, MEAN)": {
        "rmse": "n/a",
        "score": "n/a",
        "notes": "{\"auc_roc\": {\"mean\": 0.9535, \"std\": 0.0053}, \"auc_pr\": {\"mean\": 0.7186, \"std\": 0.0394}, \"f1_best\": {\"mean\": 0.7515, \"std\": 0.0096}}"
      },
      "prod_pseudo_ENSEMBLE": {
        "rmse": "n/a",
        "score": "n/a",
        "notes": "{\"auc_roc\": 0.9566532258064517, \"auc_pr\": 0.7015812540727715, \"f1_best\": 0.7580419580419581}"
      },
      "supervised_real (BiLSTM on REAL labels, MEAN) [UPPER BOUND]": {
        "rmse": "n/a",
        "score": "n/a",
        "notes": "{\"auc_roc\": {\"mean\": 0.9921, \"std\": 0.0006}, \"auc_pr\": {\"mean\": 0.9561, \"std\": 0.0024}, \"f1_best\": {\"mean\": 0.8769, \"std\": 0.0061}}"
      },
      "supervised_real_ENSEMBLE": {
        "rmse": "n/a",
        "score": "n/a",
        "notes": "{\"auc_roc\": 0.9925422686511397, \"auc_pr\": 0.9586646063792129, \"f1_best\": 0.8741935483870967}"
      },
      "iso_direct (IsolationForest score, non-DL)": {
        "rmse": "n/a",
        "score": "n/a",
        "notes": "{\"auc_roc\": 0.9614954384107611, \"auc_pr\": 0.7692993440576704, \"f1_best\": 0.7356643356643356}"
      },
      "fusion (BiLSTM-pseudo-ens + iso)": {
        "rmse": "n/a",
        "score": "n/a",
        "notes": "{\"auc_roc\": 0.9608505106488977, \"auc_pr\": 0.7496509800323774, \"f1_best\": 0.7608391608391608}"
      }
    },
    "summary_vs_real_labels": {
      "prod_pseudo (BiLSTM on pseudo-labels, MEAN)": {
        "auc_roc": {
          "mean": 0.9535,
          "std": 0.0053
        },
        "auc_pr": {
          "mean": 0.7186,
          "std": 0.0394
        },
        "f1_best": {
          "mean": 0.7515,
          "std": 0.0096
        }
      },
      "prod_pseudo_ENSEMBLE": {
        "auc_roc": 0.9566532258064517,
        "auc_pr": 0.7015812540727715,
        "f1_best": 0.7580419580419581
      },
      "supervised_real (BiLSTM on REAL labels, MEAN) [UPPER BOUND]": {
        "auc_roc": {
          "mean": 0.9921,
          "std": 0.0006
        },
        "auc_pr": {
          "mean": 0.9561,
          "std": 0.0024
        },
        "f1_best": {
          "mean": 0.8769,
          "std": 0.0061
        }
      },
      "supervised_real_ENSEMBLE": {
        "auc_roc": 0.9925422686511397,
        "auc_pr": 0.9586646063792129,
        "f1_best": 0.8741935483870967
      },
      "iso_direct (IsolationForest score, non-DL)": {
        "auc_roc": 0.9614954384107611,
        "auc_pr": 0.7692993440576704,
        "f1_best": 0.7356643356643356
      },
      "fusion (BiLSTM-pseudo-ens + iso)": {
        "auc_roc": 0.9608505106488977,
        "auc_pr": 0.7496509800323774,
        "f1_best": 0.7608391608391608
      }
    }
  },
  "integrity": {
    "sha256": "557b52f1084baf88312f3ac5c300a08626319f794bb6b806f426bc496c92f7cc",
    "canonicalization": "json sort_keys, separators (',',':'), integrity block excluded",
    "signature_alg": "RSA-PSS/SHA-256",
    "signature_b64": "U9mhm4qrrF7snb4HvwvINC0zi3aCYWY+0elYWhCCFB9F8O18TcP55uskkKJ7iYe9J+VBkw6ozmNuLwVqF0jKF7GFrB1vFUu0okkOUF1xum+zKPrNbKDSvzvIO1ikpEl6YgTUiAQdhxpFjMG8OEVLe/g7wty4Mi/oL2xfCRJ2vVjOZNeBqXV1QUSFERVUjgAFX4T5m2XpGLk/YabpzOv4R4/3jG8AoGpwZ5R8sSJoxw/yyF8QcTsUIpRXcHq0XEnEtoTim2DDsWnfvdO2QDfCXFEmrjWmPQKsw6cp2gR+aJc2L5LfQ1dVszg4GBfzoYrbCBb6V9LdqMDRwMYlI0UoiA==",
    "public_key_fingerprint_md5": "2bc036d1b625a16a8fe65648450c1031",
    "signed_utc": "2026-08-10T04:55:29Z",
    "verify": "python scripts/benchmark/generate_benchmark_report.py --verify <this file>",
    "note": "Recompute sha256 over the canonical payload (integrity excluded) and verify signature_b64 against storage/superadmin/keys/license_public.pem."
  }
}