DESED and DataSED precomputed caches for Cover First, Disagree Softly: Rethinking Mismatch-First Active Learning for Frame-Level Audio Classification
{"This":[0],"record":[1],"provides":[2],"the":[3,12,49,75,81,96,99,106,110],"precomputed":[4,53],"DESED":[5,101],"and":[6,46,52,59,67,102],"DataSED":[7,103],"dataset":[8],"caches":[9,29],"used":[10],"in":[11,15,109],"experiments":[13,79],"reported":[14],"\\"Cover":[16],"First,":[17],"Disagree":[18],"Softly:":[19],"Rethinking":[20],"Mismatch-First":[21],"Active":[22],"Learning":[23],"for":[24,48],"Frame-Level":[25],"Audio":[26],"Classification.\\"":[27],"The":[28],"contain":[30],"frame-level":[31],"audio":[32,86,90],"embeddings":[33,45],"extracted":[34,100],"using":[35,80],"PANNs":[36],"Cnn14_DecisionLevelMax,":[37],"corresponding":[38],"multi-label":[39],"annotations,":[40],"valid":[41],"frame":[42],"counts,":[43],"segment-level":[44],"labels":[47],"training":[50],"pool,":[51],"Euclidean":[54],"distance":[55],"matrices.":[56],"Training,":[57],"validation,":[58],"test":[60],"splits":[61],"are":[62],"included,":[63],"together":[64],"with":[65],"class":[66],"segment":[68],"metadata.":[69],"These":[70],"files":[71],"support":[72],"reproduction":[73],"of":[74],"paper's":[76],"active":[77],"learning":[78],"accompanying":[82],"code,":[83],"without":[84],"repeating":[85],"feature":[87],"extraction.":[88],"Raw":[89],"is":[91],"not":[92],"included.":[93],"To":[94],"use":[95],"caches,":[97],"place":[98],"directories":[104],"under":[105],"CACHE_ROOT":[107],"configured":[108],"code":[111],"repository.":[112]}
Authors
- Shiqi Zhang
- Tuomas Virtanen
Publication Details
- Journal
- arXiv (Cornell University)
- Published
- 2026-09-18
- DOI
- https://doi.org/10.5281/zenodo.22825177
- Primary Topic
- Music and Audio Processing
- Type
- preprint