Interactive research checklist · no training
Change the information regime, inspect input footprints, and export a protocol you can finish before running a real experiment.
These are explicitly chosen teaching assumptions. No head-to-head accuracy, latency or peak-memory results have been measured.
Native-inspired context sizes are 25 × 25, 7 × 7 and one full scene. The common patch control applies only in matched mode. Every format uses the same selected retained bands in this accounting; real native preprocessing needs a separate declaration.
| Illustrative format | One call | Raw input buffer | Naive whole-map input appearances |
|---|---|---|---|
| CNN / HybridSN-inspired25 × 25 patch | 16 patches16 prediction targets | 1.83 MiB30,000 values per patch | 491,520,000scalar input appearances |
| SpectralFormer-inspired25 × 25 patch | 16 patches16 prediction targets | 1.83 MiB30,000 values per patch | 491,520,000scalar input appearances |
| MambaHSI-inspired25 × 25 patch | 16 patches16 prediction targets | 1.83 MiB30,000 values per patch | 491,520,000scalar input appearances |
Naive whole-map accounting assumes one same-size zero-padded patch per scene pixel. It counts logical input appearances, including padding, rather than disk reads, MACs or simultaneously allocated memory. Efficient implementations can reuse input and features. Whole-image calls and patch calls produce different numbers of targets.
L = 49 (bands + one class token). Per-sequence component arithmetic, independent of the batch field.
These are different tensor components, not total model memory or a speed ranking. The state buffer is not a Mamba training-memory estimate. FlashAttention need not materialize the full score matrix. Actual activations, Q/K/V, convolutions, scan buffers, gradients, optimiser states, kernel workspace and batches are omitted.
Proposed upper bound: 39 single-device runs, 39.0 device-hours across three models. Each model gets 8 one-seed search trials plus 5 final-training seeds, at most 60 minutes per run. This is an unexecuted budget, not measured compute or a cost quote.
The exported JSON leaves dataset identity, split hashes, label budget, model revisions and hardware empty for you to supply. The example seeds are 100–104. No experiment is submitted or started.
Static default example. Enable JavaScript to change assumptions and export.
{
"schemaVersion": "hsi-comparison-protocol-1.0",
"status": "unexecuted-protocol-and-exact-toy-accounting",
"notABenchmark": true,
"settings": {
"mode": "matched",
"inspect": "cnn",
"split": "scene",
"target": "hidden",
"pretrain": "none",
"scaling": "train",
"side": 128,
"bands": 48,
"patch": 25,
"batch": 16,
"dtype": "fp32",
"trials": 8,
"seeds": 5,
"minutes": 60,
"axis": "spectral",
"heads": 4,
"dim": 64,
"state": 16
},
"question": "Compare adapted implementations under a common input extent and data regime.",
"data": {
"datasetId": null,
"sceneSplitManifestSha256": null,
"splitUnit": "scene",
"targetCovariatesAtTraining": "hidden",
"preprocessingFitScope": "train",
"pretraining": "none",
"inputBands": 48,
"labelBudgetPerClass": null,
"trainValidationTestIds": null,
"testLabelsUsedForSelection": false
},
"models": [
{
"id": "cnn",
"label": "CNN / HybridSN-inspired",
"context": "25 × 25 patch",
"sourceRevision": null,
"adaptationRequired": true,
"inputAccounting": {
"id": "cnn",
"name": "CNN / HybridSN-inspired",
"dense": false,
"context": "25 × 25 patch",
"patch": 25,
"inputValuesPerSample": 30000,
"inputBytesPerCall": 1920000,
"predictionTargetsPerCall": 16,
"callSamples": 16,
"naiveDenseMapInputValues": 491520000,
"overlap": {
"first": 625,
"second": 625,
"intersection": 425,
"union": 825,
"fraction": 0.68
}
}
},
{
"id": "transformer",
"label": "SpectralFormer-inspired",
"context": "25 × 25 patch",
"sourceRevision": null,
"adaptationRequired": true,
"inputAccounting": {
"id": "transformer",
"name": "SpectralFormer-inspired",
"dense": false,
"context": "25 × 25 patch",
"patch": 25,
"inputValuesPerSample": 30000,
"inputBytesPerCall": 1920000,
"predictionTargetsPerCall": 16,
"callSamples": 16,
"naiveDenseMapInputValues": 491520000,
"overlap": {
"first": 625,
"second": 625,
"intersection": 425,
"union": 825,
"fraction": 0.68
}
}
},
{
"id": "mamba",
"label": "MambaHSI-inspired",
"context": "25 × 25 patch",
"sourceRevision": null,
"adaptationRequired": true,
"inputAccounting": {
"id": "mamba",
"name": "MambaHSI-inspired",
"dense": false,
"context": "25 × 25 patch",
"patch": 25,
"inputValuesPerSample": 30000,
"inputBytesPerCall": 1920000,
"predictionTargetsPerCall": 16,
"callSamples": 16,
"naiveDenseMapInputValues": 491520000,
"overlap": {
"first": 625,
"second": 625,
"intersection": 425,
"union": 825,
"fraction": 0.68
}
}
}
],
"selection": {
"metric": "validation macro-F1",
"checkpointRule": "best validation metric within fixed cap; first checkpoint on ties",
"testSet": "evaluate once after configuration and checkpoint selection",
"search": {
"trialCountPerModel": 8,
"searchSeedsPerTrial": 1,
"finalSeeds": 5,
"perRunCapMinutes": 60,
"upperBoundRunsThreeModels": 39,
"upperBoundDeviceHoursThreeModels": 39
},
"seedList": [
100,
101,
102,
103,
104
],
"seedNote": "Example final-training seeds only. Publish separate split/search/initialisation seeds and all runs."
},
"compute": {
"hardware": null,
"softwareLock": null,
"precision": "fp32",
"trainingCapMinutes": 60,
"latencyProtocol": {
"warmup": 20,
"measuredRepeats": 100,
"report": [
"median",
"p95",
"IQR"
],
"scope": [
"model-only",
"full-scene end-to-end"
],
"include": [
"preprocessing",
"patch extraction or tiling",
"host-device transfers",
"stitching"
],
"acceleratorSynchronization": true,
"coldCompileTimeSeparate": true
},
"genericAllocationIllustration": {
"tokens": 49,
"axis": "spectral",
"bytesPerScalar": 4,
"materializedScoresBytes": 38416,
"oneTokenFeatureBytes": 12544,
"oneStreamingStateBytes": 4096
}
},
"outputsRequired": [
"per-class precision/recall/F1",
"macro-F1",
"overall accuracy",
"confusion matrix",
"per-scene results",
"paired seed results",
"training/search time",
"whole-scene latency",
"peak allocated and reserved device memory",
"host memory",
"failed and OOM runs"
],
"limitations": [
"Matched input extent requires adapted models. Patch-restricting MambaHSI changes its whole-image method; this is not an exact paper reproduction.",
"Scene-disjoint is a design intention here. Before running, verify scene IDs, provenance, duplicates and pretraining contamination."
],
"measurements": {
"accuracy": null,
"latency": null,
"peakMemory": null
},
"unfilledRequiredFields": [
"datasetId",
"sceneSplitManifestSha256",
"labelBudgetPerClass",
"trainValidationTestIds",
"all model sourceRevision values",
"hardware",
"softwareLock"
]
}