Spaces:
Running
Running
File size: 66,732 Bytes
1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 3f2f820 1cc1b05 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314 315 316 317 318 319 320 321 322 323 324 325 326 327 328 329 330 331 332 333 334 335 336 337 338 339 340 341 342 343 344 345 346 347 348 349 350 351 352 353 354 355 356 357 358 359 360 361 362 363 364 365 366 367 368 369 370 371 372 373 374 375 376 377 378 379 380 381 382 383 384 385 386 387 388 389 390 391 392 393 394 395 396 397 398 399 400 401 402 403 404 405 406 407 408 409 410 411 412 413 414 415 416 417 418 419 420 421 422 423 424 425 426 427 428 429 430 431 432 433 434 435 436 437 438 439 440 441 442 443 444 445 446 447 448 449 450 451 452 453 454 455 456 457 458 459 460 461 462 463 464 465 466 467 468 469 470 471 472 473 474 475 476 477 478 479 480 481 482 483 484 485 486 487 488 489 490 491 492 493 494 495 496 497 498 499 500 501 502 503 504 505 506 507 508 509 510 511 512 513 514 515 516 517 518 519 520 521 522 523 524 525 526 527 528 529 530 531 532 533 534 535 536 537 538 539 540 541 542 543 544 545 546 547 548 549 550 551 552 553 554 555 556 557 558 559 560 561 562 563 564 565 566 567 568 569 570 571 572 573 574 575 576 577 578 579 580 581 582 583 584 585 586 587 588 589 590 591 592 593 594 595 596 597 598 599 600 601 602 603 604 605 606 607 608 609 610 611 612 613 614 615 616 617 618 619 620 621 622 623 624 625 626 627 628 629 630 631 632 633 634 635 636 637 638 639 640 641 642 643 644 645 646 647 648 649 650 651 652 653 654 655 656 657 658 659 660 661 662 663 664 665 666 667 668 669 670 671 672 673 674 675 676 677 678 679 680 681 682 683 684 685 686 687 688 689 690 691 692 693 694 695 696 697 698 699 700 701 702 703 704 705 706 707 708 709 710 711 712 713 714 715 716 717 718 719 720 721 722 723 724 725 726 727 728 729 730 731 732 733 734 735 736 737 738 739 740 741 742 743 744 745 746 747 748 749 750 751 752 753 754 755 756 757 758 759 760 761 762 763 764 765 766 767 768 769 770 771 772 773 774 775 776 777 778 779 780 781 782 783 784 785 786 787 788 789 790 791 792 793 794 795 796 797 798 799 800 801 802 803 804 805 806 807 808 809 810 811 812 813 814 815 816 817 818 819 820 821 822 823 824 825 826 827 828 829 830 831 832 833 834 835 836 837 838 839 840 841 842 843 844 845 846 847 848 849 850 851 852 853 854 855 856 857 858 859 860 861 862 863 864 865 866 867 868 869 870 871 872 873 874 875 876 877 878 879 880 881 882 883 884 885 886 887 888 889 890 891 892 893 894 895 896 897 898 899 900 901 902 903 904 905 906 907 908 909 910 911 912 913 914 915 916 917 918 919 920 921 922 923 924 925 926 927 928 929 930 931 932 933 934 935 936 937 938 939 940 941 942 943 944 945 946 947 948 949 950 951 952 953 954 955 956 957 958 959 960 961 962 963 964 965 966 967 968 969 970 971 972 973 974 975 976 977 978 979 980 981 982 983 984 985 986 987 988 989 990 991 992 993 994 995 996 997 998 999 1000 1001 1002 1003 1004 1005 1006 1007 1008 1009 1010 1011 1012 1013 1014 1015 1016 1017 1018 1019 1020 1021 1022 1023 1024 1025 1026 1027 1028 1029 1030 1031 1032 1033 1034 1035 1036 1037 1038 1039 1040 1041 1042 1043 1044 1045 1046 1047 1048 1049 1050 1051 1052 1053 1054 1055 1056 1057 1058 1059 1060 1061 1062 1063 1064 1065 1066 1067 1068 1069 1070 1071 1072 1073 1074 1075 1076 1077 1078 1079 1080 1081 1082 1083 1084 1085 1086 1087 1088 1089 1090 1091 1092 1093 1094 1095 1096 1097 1098 1099 1100 1101 1102 1103 1104 1105 1106 1107 1108 1109 1110 1111 1112 1113 1114 1115 1116 1117 1118 1119 1120 1121 1122 1123 1124 1125 1126 1127 1128 1129 1130 1131 1132 1133 1134 1135 1136 1137 1138 1139 1140 1141 1142 1143 1144 1145 1146 1147 1148 1149 1150 1151 1152 1153 1154 1155 1156 1157 1158 1159 1160 1161 1162 1163 1164 1165 1166 1167 1168 1169 1170 | # feature_extractor_cpu.py β HF Spaces CPU Basic variant.
# Identical to feature_extractor.py. shifts=5 is preserved as required.
# Expected processing time on CPU Basic (2 vCPU): ~33β136 min per track
# depending on track length. This is expected β do not reduce shifts.
# No GPU code. No spaces.GPU decorator. Do not import the 'spaces' package.
import numpy as np
import librosa, torch, openl3, warnings
from demucs.pretrained import get_model
from demucs.apply import apply_model
from scipy.stats import entropy
from scipy.signal import butter, sosfilt, resample_poly
import pyloudnorm as pyln
warnings.filterwarnings('ignore'); EPS = 1e-10
# βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
# ENG_DIMS β ordered names for every element of the eng_vec returned by
# extract_mir_features(). Used by lane_model.py and analyze_submissions.py
# so that dimension names and indices are never hardcoded in two places.
#
# DESIGN RATIONALE β SPLITTING COMPOSITE DIMENSIONS
# ββββββββββββββββββββββββββββββββββββββββββββββββββ
# Every dimension here is intended to be atomic: one measurable quantity,
# one conceptual meaning. Dimensions that previously bundled distinct concepts
# have been decomposed. The affected cases are documented below.
#
# 1. vocal_prominence (old) β vocal_level + vocal_presence + vocal_brightness
# The old single ratio (vocal RMS / total RMS) conflated three independent
# questions: how loud are the vocals when they sing? how much of the song
# has singing at all? how upfront/forward do the vocals sit spectrally?
# A track can have loud but infrequent vocals (high level, low presence),
# or frequent but spectrally buried vocals (high presence, low brightness).
# These are now three separate dimensions.
#
# 2. drum_punch (kept, modified in v4) + drum_prominence (new in v3)
# drum_punch is the crest factor of the drum stem gated by drum_confidence.
# drum_prominence is the drum stem's RMS share of the total mix.
#
# Before v4, drum_punch was a raw crest factor: peak(drum_stem) /
# rms(drum_stem). This produced false positives for songs with no drums:
# htdemucs leaks guitar strum transients into the drum stem, giving the
# stem a tiny RMS denominator but real peaks β inflated crest factor.
# drum_prominence did not protect against this because it measures the
# stem's share of the total mix, not the stem's internal peak-to-RMS ratio.
#
# Fix (v4): drum_punch is now multiplied by drum_confidence [0,1], a
# frequency-domain drum detection score. Supporting sub-dimensions:
#
# drum_kick_presence β (v5, unchanged in v6) Drum stem's dominance in the
# 40β120 Hz kick band, measured as:
# RMS(drum_stem, 40β120 Hz) / RMS(full_mix, 40β120 Hz)
# When a real kick drum is present, the drum stem
# provides most of the mix's sub-bass energy (ratio
# 0.50β0.90). Guitar bleed: kick_dom β 0.05β0.30.
#
# drum_cymbal_presence β (v6 REDESIGN) HPSS-based sustained energy in the
# 4β10 kHz band of the drum stem. Measures crash and
# ride cymbal ring-out as opposed to transient hi-hat
# clicks (which go to hihat_presence below).
#
# Formula:
# sustained_frac = H_energy(4-10kHz) / total_energy(4-10kHz)
# band_prom = RMS(mag, 4-10kHz) / RMS(mag, all)
# cymbal_presence = (sustained_frac Γ band_prom) / CYMBAL_SAT
# (clamped to [0, 1])
#
# WHY v5 dominance ratio (8-16 kHz) FAILED:
# The v5 formula = RMS(drum_stem, 8-16kHz) / RMS(mix, 8-16kHz)
# collapses when a bright guitar or synth pad adds energy
# at 8-16 kHz alongside real hi-hats. The mix denominator
# inflates while the drum numerator is unchanged, pushing the
# ratio below the lane mean even with cymbals throughout.
# This caused "cymbal presence thinner than what I feature"
# rejections for tracks verified to have hi-hats and crashes.
#
# The HPSS H component captures ring-out / shimmer β the
# characteristic of crashes and rides β regardless of what
# other sources contribute to the full mix at those frequencies.
#
# hihat_presence β (v6 NEW) Onset density of the HPSS percussive (P)
# component in 8β14 kHz, as a fraction of HIHAT_ONSET_SAT.
#
# Hi-hats are short transients β appear in P component.
# Guitar chord rings at 8-14 kHz are sustained β appear in H.
# Soft-masking by P/(mag+EPS) then zeroing non-hihat bins
# isolates percussive 8-14 kHz content. Onset detection on
# this signal counts hi-hat events per second.
#
# Guitar staccato transients at 8-14 kHz are much weaker
# than real hi-hat transients and fall below the
# onset_detect delta threshold.
#
# drum_confidence β (v6 updated) Composite [0,1] drum detection score:
# 0.55 Γ clamp(kick_dom / DRUM_KICK_SAT, 0, 1)
# + 0.30 Γ hihat_presence (already in [0,1])
# + 0.15 Γ cymbal_presence (already in [0,1])
#
# Hi-hat weight raised vs. v5 cymbal weight (0.30 vs 0.40)
# because onset density at 8-14 kHz is near-impossible to
# trigger without real drum content. A drumless acoustic
# track cannot produce rapid regular transients in that band.
#
# drum_punch (index 2): drum_crest_raw Γ drum_confidence (unchanged gate mechanism).
#
# 3. bass_tightness (kept) + bass_prominence (new in v3)
# Identical reasoning to drum_punch/drum_prominence above.
#
# 4. guitar_energy (old) β other_prominence (renamed in v3)
# With htdemucs_6s (v8), the "other" stem is now cleaner: guitar and piano
# are extracted into dedicated stems (indices 4 and 5). other_prominence now
# reflects only synths, strings, pads, and miscellaneous non-vocal content.
#
# 5. egtr_presence + egtr_level (v8 NEW)
# Two new dimensions operating on the dedicated guitar stem from htdemucs_6s.
# See below for full rationale.
#
# RENAMES (done in v3):
# complexity β harmonic_complexity (chroma entropy β general complexity)
# dynamics β lufs_variation (std(LUFS) β all kinds of dynamics)
#
# Index history:
# Indices 0β25: identical to v3.
# Indices 26β27: added in v4, semantics updated in v5 (dominance ratio).
# Index 26: unchanged in v6 (kick dominance ratio).
# Index 27: v6 redesign β was 8-16kHz dominance ratio, now HPSS sustained
# fraction in 4-10kHz. INCOMPATIBLE with v5 cache entries.
# Index 28: v6 updated weights (was 0.60Γkick+0.40Γcymbal).
# Index 29: v6 NEW β hihat_presence.
# Indices 30β31: v8 NEW β egtr_presence, egtr_level.
# htdemucs switched from 4-stem (htdemucs) to 6-stem (htdemucs_6s).
# total_rms now includes guitar_rms + piano_rms (more accurate
# stem-sum denominator). All v7 cache entries are incompatible.
# v9 UPDATE β egtr_presence formula revised (same indices 30β31, new values):
# _compute_guitar_dims gains a third sub-dimension: tonal_sustain.
# tonal_sustain = H_mid_energy / (H_mid_energy + P_mid_energy + EPS)
# normalized to EGTR_SUSTAIN_SAT (0.70). Captures sustained guitar
# notes (legato lines, held power chords) that are tonal β H
# component β invisible to picking_density (percussive-only).
# Constants recalibrated for semi-clean and moderate-tempo guitar:
# EGTR_PICK_ONSET_SAT 4.0 β 3.0 (was too fast; 3/sec = 8th
# notes at 90 BPM)
# EGTR_FLATNESS_SAT 0.25 β 0.18 (was calibrated for heavy
# distortion; clean/crunch
# electric sits 0.06β0.14)
# Weights revised: 0.45Γpicking + 0.25Γflatness + 0.30Γsustain
# (was 0.65Γpicking + 0.35Γflatness β over-weighted fast picking).
# All v8 cache entries are incompatible (egtr_presence values shift).
# βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
ENG_DIMS = [
"lufs", # 0: Integrated loudness (LUFS)
"crest_factor", # 1: Peak-to-RMS ratio (global mix)
"drum_punch", # 2: Drum stem crest factor Γ drum_confidence (gated)
"drum_level", # 3: Drum stem RMS / total mix RMS [v10: renamed from drum_prominence]
"stereo_width", # 4: Mid-side energy ratio
"brightness", # 5: Spectral centroid of mix (Hz)
"harmonic_complexity", # 6: Chroma entropy (harmonic diversity only)
"clip_num_events", # 7: Number of separate clipping events
"clip_total", # 8: Total clipped samples
"clip_ratio", # 9: Clipped samples / total samples
"clip_max_run", # 10: Longest consecutive clipping run (samples)
"clip_mean_run", # 11: Mean clipping run length across all events
"other_level", # 12: Other stem (synth/pad/strings) RMS / total mix RMS [v10: renamed from other_prominence]
"vocal_level", # 13: Vocal stem RMS / total mix RMS (loudness when present)
"vocal_presence", # 14: Fraction of frames with active vocal signal (0β1)
"vocal_brightness", # 15: Spectral centroid of vocal stem (Hz)
"chorus_lift", # 16: Max short-term LUFS minus median (LU)
"crispness", # 17: Zero-crossing rate (high-frequency texture)
"clarity", # 18: Spectral flatness (tonal vs. noise-like character)
"length_s", # 19: Track length in seconds
"lufs_variation", # 20: Std of short-term LUFS
"peak_lufs", # 21: Max short-term LUFS
"rolloff", # 22: Spectral rolloff (Hz)
"bass_tightness", # 23: Bass stem crest factor
"bass_level", # 24: Bass stem RMS / total mix RMS [v10: renamed from bass_prominence]
"phase_corr", # 25: L/R phase correlation
# ββ Drum sub-dimensions (v4+, semantics per version noted) βββββββββββββββ
"drum_kick_level", # 26: RMS(drum stem, 40β120 Hz) / RMS(full mix, 40β120 Hz) [v10: renamed from drum_kick_presence]
"drum_cymbal_presence",# 27: HPSS sustained fraction Γ band prominence in 4β10 kHz [v6]
"drum_confidence", # 28: 0.55Γkick_conf + 0.30Γhihat_conf + 0.15Γcymbal_conf [v6]
"hihat_presence", # 29: Onset density of percussive 8β14 kHz drum component [v6 NEW]
# ββ Electric guitar dimensions (v8 NEW β htdemucs_6s guitar stem) ββββββββ
"egtr_presence", # 30: Composite electric guitar detection score [0,1] [v8 NEW]
"egtr_level", # 31: egtr_presence Γ (guitar_rms / total_rms) β guitar mix level [v8 NEW]
# ββ Stem presence + level dimensions (v10 NEW) ββββββββββββββββββββββββββββ
"drum_presence", # 32: Fraction of 1-sec chunks with active drum stem signal [v10 NEW]
"other_presence", # 33: Fraction of 1-sec chunks with active other stem signal [v10 NEW]
"bass_presence", # 34: Fraction of 1-sec chunks with active bass stem signal [v10 NEW]
"drum_kick_presence", # 35: Onset density in 40β120 Hz kick band / DRUM_KICK_ONSET_SAT [v10 NEW]
"drum_cymbal_level", # 36: RMS(drum stem, 4β10 kHz) / total mix RMS [v10 NEW]
"hihat_level", # 37: RMS(drum stem, 8β14 kHz) / total mix RMS [v10 NEW]
]
# ββ Vocal activity detection threshold βββββββββββββββββββββββββββββββββββββββ
# Frames whose RMS exceeds this value are counted as active vocal frames.
# 0.01 linear β -40 dBFS.
VOCAL_ACTIVITY_THRESHOLD = 0.01
# ββ Drum band analysis constants (v6) ββββββββββββββββββββββββββββββββββββββββ
#
# KICK (dominance ratio, unchanged from v5):
# DRUM_KICK_SAT = 0.55
# The kick dominance ratio at which kick_conf saturates at 1.0. Real kick
# drum: drum stem provides 50β90 % of mix kick-band energy β conf = 1.0.
# Acoustic guitar bleed: kick_dom β 0.05β0.30 β conf β€ 0.55.
#
# CYMBAL (HPSS sustained fraction, 4β10 kHz, v6):
# CYMBAL_BAND_FMIN / FMAX = 4000β10000 Hz
# Crash and ride cymbal fundamental energy. Crashes ring in 3β8 kHz;
# rides shimmer in 5β12 kHz. 4-10 kHz captures the core sustained content
# of both without extending into the hi-hat domain (8-14 kHz).
# CYMBAL_SAT = 0.20
# Value of (sustained_frac Γ band_prom) at which cymbal_presence = 1.0.
# Estimated: crash-heavy track β sustained_frac β 0.60, band_prom β 0.35
# β product β 0.21. Kick+snare only β product β 0.05β0.10.
# Tune up if lane refs with strong crashes score < 0.70; tune down if
# kick-only tracks score > 0.40.
#
# HI-HAT (HPSS onset density, 8β14 kHz, v6 NEW):
# HIHAT_BAND_FMIN / FMAX = 8000β14000 Hz
# Closed and open hi-hat fundamental energy. Concentrated above 8 kHz;
# stopping at 14 kHz avoids excessive noise-floor sensitivity at the
# top of the drum stem frequency range.
# HIHAT_ONSET_SAT = 5.0 (onsets/sec)
# Onset rate at which hihat_presence = 1.0. 16th notes at 75 bpm = 5/sec;
# 8th notes at 150 bpm = 5/sec. Most steady hi-hat patterns in rock/pop
# fall in the 3β8 onsets/sec range. Tune down if lane refs with active
# hi-hats score < 0.60; tune up if drumless tracks score > 0.15.
#
# DRUM CONFIDENCE WEIGHTS (v6):
# DRUM_KICK_WEIGHT = 0.55 (was 0.60 in v5)
# DRUM_HIHAT_WEIGHT = 0.30 (new; replaces cymbal_weight contribution)
# DRUM_CYMBAL_WEIGHT = 0.15 (new; crash/ride as secondary gate input)
# Sum = 1.0 β fully normalized confidence score.
#
# Hi-hat raised to 0.30: onset density at 8-14 kHz is nearly impossible to
# produce without a real drum kit. Acoustic guitars, hand percussion, and
# electronic pads without a dedicated hi-hat pattern will all score near 0.
# This makes drum_confidence a much stronger gate against false positives.
#
DRUM_KICK_FMIN = 40.0 # Hz β kick fundamental lower bound
DRUM_KICK_FMAX = 120.0 # Hz β kick fundamental upper bound
DRUM_KICK_SAT = 0.55 # kick dominance ratio at kick_conf = 1.0
CYMBAL_BAND_FMIN = 4000.0 # Hz β crash/ride lower bound
CYMBAL_BAND_FMAX = 10000.0 # Hz β crash/ride upper bound
CYMBAL_SAT = 0.20 # (sustained_frac Γ band_prom) at cymbal_presence = 1.0
HIHAT_BAND_FMIN = 8000.0 # Hz β hi-hat lower bound
HIHAT_BAND_FMAX = 14000.0 # Hz β hi-hat upper bound
HIHAT_ONSET_SAT = 5.0 # onsets/sec at hihat_presence = 1.0
DRUM_KICK_WEIGHT = 0.55 # weight for kick_conf in drum_confidence
DRUM_HIHAT_WEIGHT = 0.30 # weight for hihat_presence in drum_confidence
DRUM_CYMBAL_WEIGHT = 0.15 # weight for cymbal_presence in drum_confidence
# ββ Electric guitar detection constants (v9) ββββββββββββββββββββββββββββββββββ
#
# EGTR_MID_FMIN / FMAX = 200β5000 Hz
# Core range of electric guitar fundamentals and dominant harmonics.
# Low E string = 82 Hz; high E string = 1318 Hz. Upper harmonics and
# distortion products extend to 5 kHz. Starting at 200 Hz avoids bleed
# from the bass stem into the guitar stem. Stopping at 5 kHz keeps the
# measure out of the hi-hat / pick-noise band (handled by drum dims).
#
# EGTR_PICK_ONSET_SAT = 3.0 (onsets/sec) [v9: was 4.0]
# Onset rate at which picking_density = 1.0. 8th notes at 90 bpm = 3/sec;
# 16th notes at 45 bpm = 3/sec. 4.0 was calibrated for fast-picking rock
# styles; moderate-tempo rhythm guitar (power chords, palm-muted chugging)
# sits at 2β3 onsets/sec and was being under-scored.
# Tune DOWN further if lane refs with active rhythm guitar score < 0.50.
# Tune UP if piano-only or pad-heavy tracks score above 0.30.
#
# EGTR_FLATNESS_SAT = 0.18 [v9: was 0.25]
# Mid-band spectral flatness at which mid_saturation = 1.0.
# Clean electric guitar (Fender/Strat clean tone): 0.06β0.14.
# Semi-clean / light crunch: 0.08β0.18.
# Crunchy/overdriven: 0.12β0.28. Full distortion: 0.20β0.40.
# Piano: 0.02β0.08. Acoustic guitar (clean): 0.03β0.10.
# 0.25 was only saturable by heavily overdriven guitar β clean and semi-clean
# electric guitar scored mid_saturation β€ 0.40 even with guitar throughout.
# 0.18 rewards any electric guitar content, not only distorted playing.
#
# EGTR_SUSTAIN_SAT = 0.70 [v9 NEW]
# H-component energy fraction at which tonal_sustain = 1.0.
# tonal_sustain = H_mid_energy / (H_mid_energy + P_mid_energy)
# where H = HPSS sustained component, P = HPSS percussive component,
# both measured in the 200β5000 Hz mid-band of the guitar stem.
#
# WHY THIS DIMENSION:
# picking_density and mid_saturation both measure the *attack/transient*
# character of guitar. A guitar playing sustained power chords, legato
# lines, or moderate-tempo chord progressions puts most of its energy into
# the H (sustained) component β the P (percussive) component only captures
# the brief pick attack. The entire sustain period is invisible to
# picking_density and adds no flatness signal. tonal_sustain explicitly
# captures this: as long as guitar notes are ringing, H_mid_energy is high
# relative to P_mid_energy, and tonal_sustain stays elevated regardless of
# tempo or gain stage.
#
# Expected ranges (guitar stem of htdemucs_6s):
# Electric guitar (any gain, any tempo): H_frac β 0.60β0.85 β sustain 0.86β1.0
# Acoustic guitar: H_frac β 0.55β0.75 β sustain 0.79β1.0
# Drum bleed in guitar stem: H_frac β 0.15β0.35 β sustain 0.21β0.50
# Low-level bleed / near-silence: Both near EPS β ratio noisy but
# egtr_level β 0 (guitar_rms near 0)
#
# 0.70 saturates at an H_frac consistent with guitar that is sustaining
# cleanly. Tracks with only transient bleed (drums) will score 0.21β0.50.
# Tune UP if non-guitar tracks score tonal_sustain > 0.60 consistently.
#
# WEIGHTS (v9):
# EGTR_PICK_WEIGHT = 0.45 (was 0.65) β rhythmic attack events
# EGTR_FLAT_WEIGHT = 0.25 (was 0.35) β harmonic saturation / gain stage
# EGTR_SUSTAIN_WEIGHT = 0.30 (NEW) β sustained tonal guitar content
# Sum = 1.00
#
# Rationale for reducing EGTR_PICK_WEIGHT from 0.65:
# The previous 0.65 weight made picking_density the near-exclusive gate.
# A track with guitar throughout but moderate tempo and clean tone could
# score egtr_presence β 0.62 even with guitar clearly audible in the mix,
# because picking_density was ~0.70 and mid_saturation was ~0.35.
# Redistributing weight to tonal_sustain captures guitar presence more
# robustly across gain stages and tempos.
#
# TUNING GUIDANCE (post first training run):
# Check the printed standards table for egtr_presence and egtr_level.
# Lane refs with active electric guitar: egtr_presence β₯ 0.65 (ideally 0.75β0.95)
# Non-lane refs without guitar: egtr_presence β€ 0.25
# If guitar tracks still score 0.30β0.50: lower EGTR_PICK_ONSET_SAT (3.0 β 2.0)
# and/or lower EGTR_FLATNESS_SAT (0.18 β 0.12).
# If piano/pad tracks score above 0.40: raise EGTR_SUSTAIN_SAT (0.70 β 0.80)
# or reduce EGTR_SUSTAIN_WEIGHT (0.30 β 0.20).
#
EGTR_MID_FMIN = 200.0 # Hz β guitar mid-band lower bound
EGTR_MID_FMAX = 5000.0 # Hz β guitar mid-band upper bound
EGTR_PICK_ONSET_SAT = 3.0 # onsets/sec at picking_density = 1.0 [v9: was 4.0]
EGTR_FLATNESS_SAT = 0.18 # flatness value at mid_saturation = 1.0 [v9: was 0.25]
EGTR_SUSTAIN_SAT = 0.70 # H-frac at tonal_sustain = 1.0 [v9 NEW]
EGTR_PICK_WEIGHT = 0.45 # weight for picking_density in egtr_presence [v9: was 0.65]
EGTR_FLAT_WEIGHT = 0.25 # weight for mid_saturation in egtr_presence [v9: was 0.35]
EGTR_SUSTAIN_WEIGHT = 0.30 # weight for tonal_sustain in egtr_presence [v9 NEW]
# ββ Stem temporal presence + kick onset constants (v10) βββββββββββββββββββββββ
DRUM_ACTIVITY_THRESHOLD = 0.01 # RMS threshold for an active drum chunk
BASS_ACTIVITY_THRESHOLD = 0.01 # RMS threshold for an active bass chunk
OTHER_ACTIVITY_THRESHOLD = 0.01 # RMS threshold for an active other chunk
DRUM_KICK_ONSET_SAT = 3.0 # kick onsets/sec at drum_kick_presence = 1.0
device = "cuda" if torch.cuda.is_available() else "cpu"
demucs_model = get_model('htdemucs_6s') # v8: 6-stem model (adds guitar + piano stems)
demucs_model.to(device)
# htdemucs_6s stem order: 0=drums, 1=bass, 2=other, 3=vocals, 4=guitar, 5=piano
_STEM_DRUMS = 0
_STEM_BASS = 1
_STEM_OTHER = 2
_STEM_VOCALS = 3
_STEM_GUITAR = 4 # v8 NEW β dedicated electric/acoustic guitar stem
_STEM_PIANO = 5 # v8 NEW β dedicated piano stem
def get_stems(path):
wav, _ = librosa.load(path, sr=demucs_model.samplerate, mono=False)
if wav.ndim == 1: wav = np.stack([wav, wav])
ref = torch.from_numpy(wav).to(device)
ref_std = ref.std() + EPS
ref = (ref - ref.mean()) / ref_std
with torch.no_grad():
sources = apply_model(demucs_model, ref.unsqueeze(0), shifts=5, split=True)[0]
return (sources * ref_std.cpu().item()).cpu().numpy()
def _measure_clipping(y, threshold=0.999):
"""
Returns all five clipping dimensions as individual floats.
threshold=0.999 (~-0.009 dBFS) avoids flagging legitimately loud
transient peaks that just touch near-full-scale without flat-topping.
Returns: (num_events, total_samples, ratio, max_run, mean_run)
"""
clipped = np.abs(y) >= threshold
total = int(np.sum(clipped))
if total == 0:
return 0.0, 0.0, 0.0, 0.0, 0.0
padded = np.concatenate(([False], clipped, [False]))
diff = np.diff(padded.astype(int))
run_starts = np.where(diff == 1)[0]
run_ends = np.where(diff == -1)[0]
run_lengths = run_ends - run_starts
return (
float(len(run_lengths)),
float(total),
float(total / len(y)),
float(np.max(run_lengths)),
float(np.mean(run_lengths)),
)
def _band_rms_abs(y: np.ndarray, sr: int, fmin: float, fmax: float) -> float:
"""
Absolute RMS of y in the [fmin, fmax] Hz band (not normalized).
Used by _compute_drum_bands for kick dominance ratio.
Returns 0.0 if signal is silent or band is invalid.
"""
nyq = sr / 2.0
lo = max(fmin, 1.0)
hi = min(fmax, nyq - 1.0)
if lo >= hi:
return 0.0
sos = butter(2, [lo, hi], btype='bandpass', fs=sr, output='sos')
filtered = sosfilt(sos, y)
return float(np.sqrt(np.mean(filtered**2)))
# βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
# ITU-R BS.1770-4 / BS.1775-2 LOUDNESS AND TRUE PEAK IMPLEMENTATION
#
# WHY THE OLD FORMULA WAS WRONG
# βββββββββββββββββββββββββββββββ
# The original approximation used:
# lufs β -0.691 + 10 Γ log10(mean(yΒ²))
# This is RMS-based power in dB, not integrated LUFS. It is missing two
# mandatory stages defined in ITU-R BS.1770:
#
# 1. K-weighting pre-filter β a two-stage IIR filter that:
# Stage 1: high-shelf boost (+4 dB at high frequencies) modelling
# the acoustic effect of the listener's head on a microphone.
# Stage 2: high-pass filter (RLB weighting) removing DC and sub-20 Hz
# content that inflates RMS but carries no perceived loudness.
# Together these raise typical music by +1 to +5 dB before squaring.
#
# 2. Gating β BS.1770-4 integrates only "active" blocks:
# Absolute gate: blocks below -70 LUFS excluded (silence/noise floor).
# Relative gate: second pass excludes blocks more than -10 LU below
# the ungated mean (rejects very quiet passages).
# Without gating, quiet intros/outros and noise floors pull the
# integrated value down, causing systematic under-reading of 2β5 dB.
#
# PRACTICAL IMPACT ON THIS SYSTEM
# ββββββββββββββββββββββββββββββββ
# Error: 2β4 dB under-reading of integrated LUFS, varying by program material.
# β’ The Spotify normalization gain calculation was off by that same margin.
# β’ Lane engineering standards for lufs and peak_lufs were calibrated against
# the wrong values (consistently wrong, so relative comparisons partially
# survived β but absolute statements about headroom were incorrect).
# β’ short-term LUFS (used for peak_lufs, lufs_variation, chorus_lift) had the
# same error AND lacked K-weighting, so spectral character affected the
# "loudness" reading in a non-perceptual way.
# β’ true_peak_dbfs was never returned by the modal worker, making the
# tp_headroom constraint in apply_streaming_normalization() dead code.
#
# FIX SUMMARY (v7)
# ββββββββββββββββ
# _compute_integrated_lufs() β pyloudnorm Meter.integrated_loudness()
# Full BS.1770-4: K-weighting + gating.
# _compute_short_term_lufs_array() β K-weighted sliding 3s/1s windows, no gate.
# Same K-weighting as integrated. Replaces the
# raw librosa.feature.rms()-based st_lufs blocks.
# _true_peak_dbfs() β 4Γ polyphase oversampling (BS.1775-2) to
# detect intersample peaks. Returned as third
# element of extract_mir_features() so the modal
# worker can include it in the response dict and
# apply_streaming_normalization() can enforce the
# -1.0 dBFS true peak ceiling correctly.
#
# NOTE: pyloudnorm must be installed: pip install pyloudnorm
# βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
def _kweight_sos(sr: int) -> np.ndarray:
"""
BS.1770 K-weighting filter as a stacked SOS array for scipy.signal.sosfilt.
Computed directly from the analog prototype parameters in BS.1770 Annex 1
using the Audio EQ Cookbook (Bristow-Johnson) bilinear transform formulas.
Does NOT touch pyloudnorm internals β meter._filters internal structure
has changed between pyloudnorm versions and is not part of its public API.
Two cascaded stages:
Stage 1 β Pre-filter (high-shelf, +4 dB at high frequencies):
Models the free-field to diffuse-field acoustic correction for the
effect of the listener's head on a mounted microphone. Raises
high-frequency energy before mean-square integration so that HF
content is weighted appropriately by the loudness measure.
Parameters: f0 = 1500 Hz, dBgain = +4 dB, Q = 1/sqrt(2)
Stage 2 β RLB weighting (2nd-order Butterworth high-pass, ~38.1 Hz):
Removes DC and very low-frequency energy that carries no perceptual
loudness but would otherwise inflate the mean-square sum.
Parameters: fc = 38.13494637408312 Hz
Returns a (2, 6) ndarray β two SOS sections β ready for sosfilt().
"""
# Stage 1: Pre-filter high-shelf (Audio EQ Cookbook high-shelf formula)
f0 = 1500.0 # Hz β shelf midpoint frequency
dBgain = 4.0 # dB β shelf gain
Q = 1.0 / np.sqrt(2) # = 0.7071... (Butterworth Q)
A = 10.0 ** (dBgain / 40.0) # sqrt of linear power gain
w0 = 2.0 * np.pi * f0 / sr # digital frequency (radians/sample)
alpha = np.sin(w0) / (2.0 * Q) # bandwidth term
# High-shelf numerator / denominator (Audio EQ Cookbook, high-shelf)
b0 = A * ((A + 1) + (A - 1) * np.cos(w0) + 2 * np.sqrt(A) * alpha)
b1 = -2 * A * ((A - 1) + (A + 1) * np.cos(w0))
b2 = A * ((A + 1) + (A - 1) * np.cos(w0) - 2 * np.sqrt(A) * alpha)
a0 = ((A + 1) - (A - 1) * np.cos(w0) + 2 * np.sqrt(A) * alpha)
a1 = 2 * ((A - 1) - (A + 1) * np.cos(w0))
a2 = ((A + 1) - (A - 1) * np.cos(w0) - 2 * np.sqrt(A) * alpha)
# Second-order section: [b0/a0, b1/a0, b2/a0, 1.0, a1/a0, a2/a0]
sos_shelf = np.array([[b0/a0, b1/a0, b2/a0, 1.0, a1/a0, a2/a0]])
# Stage 2: RLB high-pass (2nd-order Butterworth)
fc = 38.13494637408312 # Hz
sos_hp = butter(2, fc / (sr / 2.0), btype='high', output='sos')
return np.vstack([sos_shelf, sos_hp]) # shape (2, 6)
def _compute_integrated_lufs(y_stereo: np.ndarray, sr: int) -> float:
"""
ITU-R BS.1770-4 integrated LUFS (K-weighted, gated).
y_stereo: (2, n_samples) stereo array at sr Hz.
Returns integrated LUFS as float; returns -70.0 for silent / fully-gated
signals (pyloudnorm returns -inf in those cases).
"""
meter = pyln.Meter(sr)
data = y_stereo.T.astype(np.float64) # pyloudnorm expects (n_samples, n_ch)
result = meter.integrated_loudness(data)
return float(result) if np.isfinite(result) else -70.0
def _compute_short_term_lufs_array(y_stereo: np.ndarray, sr: int,
window_s: float = 3.0,
hop_s: float = 1.0) -> np.ndarray:
"""
K-weighted short-term loudness in sliding windows (BS.1770, no gating).
Applies the same K-weighting filters as integrated LUFS but without the
absolute/relative gating step, matching the BS.1770 short-term definition.
Uses the BS.1770 stereo channel-sum formula (G_L = G_R = 1.0):
L_window = -0.691 + 10 Γ log10(mean_sq_L + mean_sq_R)
window_s = 3.0, hop_s = 1.0 matches the original librosa.feature.rms()
block framing that peak_lufs, lufs_variation, and chorus_lift are based on.
y_stereo: (2, n_samples) stereo array at sr Hz.
Returns array of per-window LUFS values (float64).
"""
# Apply K-weighting (BS.1770 pre-filter + RLB high-pass) to each channel.
# _kweight_sos() returns a single stacked (2, 6) SOS array computed
# directly from the BS.1770 analog prototype β no pyloudnorm internals used.
sos = _kweight_sos(sr)
data = y_stereo.astype(np.float64).copy() # (2, n_samples)
for ch in range(data.shape[0]):
data[ch] = sosfilt(sos, data[ch])
n_samples = data.shape[1]
win_len = int(window_s * sr)
hop_len = max(int(hop_s * sr), 1)
results = []
for start in range(0, max(n_samples - win_len + 1, 1), hop_len):
block = data[:, start : start + win_len] # (2, win_len)
ms_sum = float(np.sum(np.mean(block ** 2, axis=1))) # sum over L, R
results.append(-0.691 + 10.0 * np.log10(ms_sum + EPS))
return np.array(results, dtype=np.float64) if results else np.array([-70.0])
def _true_peak_dbfs(y_stereo: np.ndarray) -> float:
"""
ITU-R BS.1775-2 true peak level (dBFS), measured via 4Γ oversampling.
Standard sample-domain peak measurement misses intersample peaks β peaks
that occur between samples and can exceed the sample-domain maximum after
reconstruction. 4Γ oversampling with a polyphase anti-aliasing filter
(scipy.signal.resample_poly) captures these peaks reliably.
Spotify, Apple Music, YouTube, and Tidal all measure true peak at β₯ 4Γ
oversampling when deciding whether and by how much to apply limiting during
normalization. -1.0 dBFS TP is the standard limiting ceiling for most DSPs.
y_stereo: (2, n_samples) stereo or (n_samples,) mono array.
Returns peak in dBFS. Typical brick-wall limited master: -0.5 to -0.1 dBFS.
Silent signal: returns -120.0.
"""
oversampled = resample_poly(y_stereo, 4, 1)
peak_linear = float(np.max(np.abs(oversampled)))
if peak_linear < EPS:
return -120.0
return float(20.0 * np.log10(peak_linear))
# βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
# CANONICAL STREAMING NORMALIZATION
#
# Single source of truth used by both lane_model.py (training) and
# analyze_submissions.py (inference). Having one implementation guarantees
# that training and scoring apply identical gain logic β any future change to
# normalization behaviour only needs to be made in one place.
#
# Previously duplicated as:
# _norm_eng() in lane_model.py (training)
# apply_streaming_normalization() in analyze_submissions.py (inference)
# Those two copies were kept manually in sync. They are now replaced by this
# single function, imported by both callers.
# βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
def apply_streaming_normalization(
eng: np.ndarray,
target_lufs: float,
true_peak_dbfs: "float | None" = None,
) -> "tuple[np.ndarray, float]":
"""
Simulate streaming-platform loudness normalization on an eng vector.
Computes the gain the platform would apply to reach target_lufs, then
shifts the lufs and peak_lufs dimensions by that gain.
Only attenuates: gain is capped at 0.0 dB (platforms never boost quiet
masters β Spotify, Apple Music, etc. leave them quieter than the target).
true_peak_dbfs (BS.1775-2, 4Γ oversampled):
If provided and finite, the gain is also capped so that true peak
never exceeds -1.0 dBFS post-normalization, matching the hardware
limiter ceiling used by most streaming DSPs. This cap only activates
for masters with intersample peaks above -1.0 dBFS. When None or
non-finite, the cap is skipped (equivalent to tp_headroom = 0.0,
which is harmless because the 0.0 attenuation cap already applies).
Raises ValueError if "lufs" or "peak_lufs" are absent from ENG_DIMS.
Callers must treat this as a fatal configuration error β it means
ENG_DIMS was modified without updating this function.
Parameters
----------
eng : np.ndarray
Raw engineering feature vector (32 elements, pre-normalization).
target_lufs : float
Streaming loudness target in LUFS (e.g. -14.0 for Spotify).
true_peak_dbfs : float | None
BS.1775-2 true peak level of the master in dBFS, or None if
unavailable (true peak cap is skipped when None).
Returns
-------
eng_normed : np.ndarray
Copy of eng with lufs and peak_lufs shifted by gain_db.
gain_db : float
Gain applied (always <= 0.0).
"""
try:
lufs_idx = ENG_DIMS.index("lufs")
peak_idx = ENG_DIMS.index("peak_lufs")
except ValueError as exc:
raise ValueError(
f"apply_streaming_normalization: required dimension not found in ENG_DIMS: {exc}"
) from exc
# True peak cap: only active when true_peak_dbfs is provided and finite.
# When true_peak > -1.0 dBFS (intersample peak above limiter ceiling),
# tp_headroom is negative, further restricting the gain so the normalized
# true peak lands at -1.0 dBFS. When true_peak <= -1.0 dBFS, tp_headroom
# is non-negative and the existing 0.0 cap is the binding constraint.
tp_headroom = (
(-1.0 - true_peak_dbfs)
if (true_peak_dbfs is not None and np.isfinite(true_peak_dbfs))
else 0.0
)
gain_db = min(target_lufs - eng[lufs_idx], tp_headroom, 0.0)
out = eng.copy()
out[lufs_idx] += gain_db
out[peak_idx] += gain_db
return out, gain_db
def _compute_drum_bands(drum_mono: np.ndarray, mix_mono: np.ndarray, sr: int,
total_rms: float) -> tuple:
"""
Returns (kick_dom, cymbal_presence, hihat_presence, cymbal_level, hihat_level).
kick_dom β dominance ratio (unchanged from v5):
RMS(drum_stem, 40β120 Hz) / RMS(mix, 40β120 Hz)
cymbal_presence β HPSS sustained fraction Γ band prominence in 4β10 kHz [0,1]
hihat_presence β onset density of HPSS percussive 8β14 kHz component [0,1]
cymbal_level β RMS(drum_stem, 4β10 kHz) / total_rms [v10 NEW]
hihat_level β RMS(drum_stem, 8β14 kHz) / total_rms [v10 NEW]
WHY HPSS INSTEAD OF DOMINANCE RATIO FOR CYMBAL / HI-HAT (v6):
The v5 formula drum_cymbal_presence = RMS(drum_8-16kHz) / RMS(mix_8-16kHz)
conflated hi-hats and crashes and failed in mixed-source tracks. When a bright
guitar or synth pad adds energy at 8-16 kHz alongside real drum content, the
mix denominator inflates while the drum numerator is unchanged β the ratio drops
below the lane mean even with cymbals audibly present throughout the track.
HPSS (Harmonic-Percussive Source Separation, librosa.decompose.hpss) applies
a horizontal median filter to the magnitude spectrogram to extract sustained
tonal content (H) and a vertical median filter for transient broadband content (P):
H (harmonic/sustained): crash ring-out, ride shimmer, guitar chord sustain.
P (percussive/transient): hi-hat clicks, kick/snare attacks.
The H component is not affected by other instruments' level in the full mix β
it measures what is sustained within the drum stem's own spectrogram. A crash
ring-out is always captured in H regardless of guitar/synth in the full mix.
CYMBAL (crash/ride): measured via sustained fraction Γ band prominence.
sustained_frac = H_energy(4-10kHz) / total_energy(4-10kHz)
β high when crash ring-out dominates 4-10kHz drum content
β lower for snare-only (snare ring has less sustained relative energy in band)
band_prom = RMS(mag_4-10kHz) / RMS(mag_all_freqs)
β scales by how much 4-10kHz energy the drum stem actually carries
β near-silent drum stem (no drums) β band_prom β 0 β product β 0
Combined: cymbal_presence = (sustained_frac Γ band_prom) / CYMBAL_SAT
HI-HAT: measured via onset density of percussive 8-14 kHz signal.
1. Apply soft percussive mask P/(mag+EPS) to complex STFT D.
2. Zero all frequency bins outside 8-14 kHz.
3. Reconstruct time-domain hi-hat signal via iSTFT.
4. Run librosa onset detection on reconstructed signal.
5. onset_rate (onsets/sec) / HIHAT_ONSET_SAT β [0,1].
Guitar chord rings at 8-14 kHz are sustained β H component β soft_p low
for those bins β soft-masked out. Guitar staccato transients at 8-14 kHz
are much weaker than real hi-hat transients β below onset_detect threshold.
"""
# ββ Kick dominance (unchanged from v5) ββββββββββββββββββββββββββββββββββββ
kick_mix = _band_rms_abs(mix_mono, sr, DRUM_KICK_FMIN, DRUM_KICK_FMAX)
kick_drum = _band_rms_abs(drum_mono, sr, DRUM_KICK_FMIN, DRUM_KICK_FMAX)
kick_dom = float(min(kick_drum / (kick_mix + EPS), 1.0)) if kick_mix >= EPS else 0.0
# ββ STFT + HPSS on drum stem ββββββββββββββββββββββββββββββββββββββββββββββ
n_fft = 2048
hop = 512
D = librosa.stft(drum_mono, n_fft=n_fft, hop_length=hop) # complex (1025, n_frames)
mag = np.abs(D) # magnitude spectrogram
H, P = librosa.decompose.hpss(mag, kernel_size=31, power=2.0) # H=sustained, P=transient
freqs = librosa.fft_frequencies(sr=sr, n_fft=n_fft) # shape (1025,)
# ββ Cymbal presence: HPSS sustained energy in 4β10 kHz βββββββββββββββββββ
cymbal_mask = (freqs >= CYMBAL_BAND_FMIN) & (freqs <= CYMBAL_BAND_FMAX)
H_cym = H[cymbal_mask, :] # sustained component in cymbal band
mag_cym = mag[cymbal_mask, :] # total magnitude in cymbal band
H_cym_energy = float(np.mean(H_cym ** 2))
band_energy = float(np.mean(mag_cym ** 2)) + EPS
sustained_frac = H_cym_energy / band_energy # fraction of cymbal-band energy that is sustained
band_rms = float(np.sqrt(np.mean(mag_cym ** 2)))
total_mag_rms = float(np.sqrt(np.mean(mag ** 2))) + EPS
band_prom = band_rms / total_mag_rms # cymbal band's prominence in overall drum stem
cymbal_presence = float(min((sustained_frac * band_prom) / CYMBAL_SAT, 1.0))
# ββ Hi-hat presence: onset density of percussive 8β14 kHz component ββββββ
hihat_mask = (freqs >= HIHAT_BAND_FMIN) & (freqs <= HIHAT_BAND_FMAX)
# Soft-mask: amplify percussive content, suppress sustained content per bin.
# soft_p[f,t] = P[f,t] / (mag[f,t] + EPS) β [0,1] β how percussive each bin is.
soft_p = P / (mag + EPS)
D_hihat = D * soft_p # complex STFT weighted toward percussive content
D_hihat[~hihat_mask, :] = 0.0 # zero non-hihat frequencies
# iSTFT reconstructs time-domain signal containing only percussive 8-14 kHz content.
hihat_sig = librosa.istft(D_hihat, hop_length=hop, length=len(drum_mono))
oenv = librosa.onset.onset_strength(y=hihat_sig, sr=sr, hop_length=hop)
onsets = librosa.onset.onset_detect(onset_envelope=oenv, sr=sr, hop_length=hop)
duration = len(drum_mono) / sr
onset_rate = len(onsets) / max(duration, 1.0)
hihat_presence = float(min(onset_rate / HIHAT_ONSET_SAT, 1.0))
# ββ Cymbal band level: absolute amplitude share in full mix βββββββββββββββ
cymbal_level = float(_band_rms_abs(drum_mono, sr, CYMBAL_BAND_FMIN, CYMBAL_BAND_FMAX)
/ (total_rms + EPS))
# ββ Hi-hat band level: absolute amplitude share in full mix ββββββββββββββ
hihat_level = float(_band_rms_abs(drum_mono, sr, HIHAT_BAND_FMIN, HIHAT_BAND_FMAX)
/ (total_rms + EPS))
return kick_dom, cymbal_presence, hihat_presence, cymbal_level, hihat_level
def _compute_stem_presence(stem_stereo: np.ndarray, sr: int, threshold: float,
chunk_s: float = 1.0) -> float:
"""
Fraction of 1-second chunks in which the stem's RMS exceeds threshold.
1.0 = stem active throughout the entire track.
0.5 = stem active in roughly half the track.
0.0 = stem silent throughout.
threshold = 0.01 (linear, β -40 dBFS) matches VOCAL_ACTIVITY_THRESHOLD.
stem_stereo: (2, n_samples) stereo stem array at sr Hz.
"""
mono = (stem_stereo[0] + stem_stereo[1]) / 2.0
chunk_len = max(int(chunk_s * sr), 1)
n_chunks = max(len(mono) // chunk_len, 1)
active = sum(
1 for i in range(n_chunks)
if np.sqrt(np.mean(mono[i * chunk_len : (i + 1) * chunk_len] ** 2)) > threshold
)
return float(active / n_chunks)
def _compute_kick_onset_presence(drum_mono: np.ndarray, sr: int) -> float:
"""
Onset density of percussive events in the kick band (40β120 Hz), normalised
to DRUM_KICK_ONSET_SAT. Returns a value in [0, 1].
Measures *rhythmic density* of kick hits β distinct from drum_kick_level
which measures low-end amplitude. A bass note scores high level, near-zero
presence. A kick drum scores high on both.
"""
if np.sqrt(np.mean(drum_mono ** 2)) < EPS:
return 0.0
lo = max(DRUM_KICK_FMIN, 1.0)
hi = min(DRUM_KICK_FMAX, sr / 2.0 - 1.0)
if lo >= hi:
return 0.0
hop = 512
sos = butter(2, [lo, hi], btype='bandpass', fs=sr, output='sos')
kick_sig = sosfilt(sos, drum_mono)
oenv = librosa.onset.onset_strength(y=kick_sig, sr=sr, hop_length=hop)
onsets = librosa.onset.onset_detect(onset_envelope=oenv, sr=sr, hop_length=hop)
duration = len(drum_mono) / sr
onset_rate = len(onsets) / max(duration, 1.0)
return float(min(onset_rate / DRUM_KICK_ONSET_SAT, 1.0))
def _compute_guitar_dims(guitar_mono: np.ndarray, sr: int) -> tuple:
"""
Returns (egtr_presence, picking_density, mid_saturation, tonal_sustain).
All values in [0, 1].
egtr_presence β composite [0,1] electric guitar detection score (v9):
EGTR_PICK_WEIGHT Γ picking_density
+ EGTR_FLAT_WEIGHT Γ mid_saturation
+ EGTR_SUSTAIN_WEIGHT Γ tonal_sustain
picking_density β onset density of the HPSS percussive component in
200β5000 Hz, normalized to EGTR_PICK_ONSET_SAT (3.0).
Detects pick/strum attack events per second.
Electric guitar: 1.5β5+ onsets/sec. Sustained pads: 0β0.5.
Piano also produces picking onsets; mid_saturation and
tonal_sustain together gate against false-positive piano.
mid_saturation β spectral flatness of the guitar stem in 200β5000 Hz,
normalized to EGTR_FLATNESS_SAT (0.18).
Clean electric guitar: 0.06β0.14 (mid_sat β 0.33β0.78).
Semi-clean / light crunch: 0.08β0.18 (mid_sat β 0.44β1.0).
Crunchy/overdriven: 0.12β0.28 (mid_sat β 0.67β1.0).
Piano: 0.02β0.08 (mid_sat β 0.11β0.44).
Acoustic guitar: 0.03β0.10 (mid_sat β 0.17β0.56).
Rewards harmonic saturation from magnetic pickups and
overdrive β the acoustic signature of electric amplification.
tonal_sustain β fraction of mid-band energy in the HPSS H (sustained)
component, normalized to EGTR_SUSTAIN_SAT (0.70).
Formula:
H_mid_energy = mean(H[mid_mask]Β²)
P_mid_energy = mean(P[mid_mask]Β²)
raw_sustain = H_mid_energy / (H_mid_energy + P_mid_energy + EPS)
tonal_sustain = min(raw_sustain / EGTR_SUSTAIN_SAT, 1.0)
WHY THIS DIMENSION (v9):
picking_density and mid_saturation both measure attack/
transient character. A guitar playing sustained power chords,
legato lines, or moderate-tempo chord progressions puts most
of its energy into the H (sustained) HPSS component β the
P (percussive) component only captures the brief pick attack.
The sustained note body is invisible to picking_density and
adds no flatness signal relative to its RMS contribution.
tonal_sustain captures this: as long as guitar notes are
ringing, H_mid_energy stays elevated relative to P_mid_energy,
regardless of tempo or gain stage.
Expected ranges (guitar stem, htdemucs_6s):
Electric guitar (any gain, any tempo): H_frac β 0.60β0.85
Acoustic guitar: H_frac β 0.55β0.75
Drum bleed in guitar stem: H_frac β 0.15β0.35
Near-silence (minimal bleed): ratio noisy but
egtr_level β 0
WHY HPSS ON GUITAR STEM (v8, unchanged):
Even with htdemucs_6s providing a dedicated guitar stem, there is residual
bleed from other instruments. HPSS separates the percussive (transient
attack) component from the harmonic (sustained) component within the guitar
stem itself. Extracting the P component in 200-5000 Hz and counting onsets
isolates picking/strumming events even with stem bleed, because instrument
bleed (synths, pads, vocals) tends to be sustained (β H component) rather
than transient (β P component) in this frequency range.
ACOUSTIC VS ELECTRIC LIMITATION (unchanged from v8):
Clean acoustic and clean electric guitar score similarly on picking_density.
mid_saturation partially distinguishes them (electric pickups add resonance
that raises flatness vs acoustic). tonal_sustain is similar for both.
The CLAP embedding (already in the feature vector) carries the residual
discrimination; the SVM learns from CLAP + these dims together.
Intermediate values (picking_density, mid_saturation, tonal_sustain) are
returned for debug/tuning but are NOT added to ENG_DIMS β only egtr_presence
and egtr_level are scoreable dimensions, preserving the atomicity principle.
"""
if np.sqrt(np.mean(guitar_mono ** 2)) < EPS:
# Silent guitar stem β no guitar in the track.
return 0.0, 0.0, 0.0, 0.0
n_fft = 2048
hop = 512
D = librosa.stft(guitar_mono, n_fft=n_fft, hop_length=hop)
mag = np.abs(D)
H, P = librosa.decompose.hpss(mag, kernel_size=31, power=2.0)
freqs = librosa.fft_frequencies(sr=sr, n_fft=n_fft)
mid_mask = (freqs >= EGTR_MID_FMIN) & (freqs <= EGTR_MID_FMAX)
# ββ Picking density: onset rate of percussive mid-band content ββββββββββββ
# Soft-mask the complex STFT toward percussive content, isolate mid band.
soft_p = P / (mag + EPS) # per-bin percussiveness score [0,1]
D_pick = D * soft_p # complex STFT weighted toward transients
D_pick[~mid_mask, :] = 0.0 # zero bins outside guitar mid band
pick_sig = librosa.istft(D_pick, hop_length=hop, length=len(guitar_mono))
oenv = librosa.onset.onset_strength(y=pick_sig, sr=sr, hop_length=hop)
onsets = librosa.onset.onset_detect(onset_envelope=oenv, sr=sr, hop_length=hop)
duration = len(guitar_mono) / sr
onset_rate = len(onsets) / max(duration, 1.0)
picking_density = float(min(onset_rate / EGTR_PICK_ONSET_SAT, 1.0))
# ββ Mid-band spectral flatness: harmonic saturation from pickups/overdrive β
# Flatness = geometric_mean / arithmetic_mean of magnitude spectrum.
# Ranges [0, 1]: 0 = perfectly tonal (sine-like), 1 = flat noise.
# Electric guitar pickups and overdrive add intermodulation products that
# raise flatness compared to clean acoustic guitar or piano.
mag_mid = mag[mid_mask, :]
if mag_mid.size == 0 or mag_mid.shape[0] == 0:
mid_saturation = 0.0
else:
geo_mean = np.exp(np.mean(np.log(mag_mid + EPS), axis=0)) # per-frame geo mean
arith_mean = np.mean(mag_mid, axis=0) + EPS # per-frame arith mean
frame_flat = geo_mean / arith_mean # flatness [0,1] per frame
raw_flatness = float(np.mean(frame_flat))
mid_saturation = float(min(raw_flatness / EGTR_FLATNESS_SAT, 1.0))
# ββ Tonal sustain: fraction of mid-band energy in sustained H component βββ
# Captures guitar notes that are ringing/sustaining β the body of each note
# after the pick attack, which is entirely invisible to picking_density.
# H (harmonic/sustained) component of HPSS in 200β5000 Hz.
# P (percussive/transient) component in same band (pick attacks only).
# raw_sustain = H_energy / (H_energy + P_energy) β [0, 1].
# When guitar notes sustain: H >> P, raw_sustain high β tonal_sustain high.
# When only drum transients bleed in: P >> H, raw_sustain low.
H_mid = H[mid_mask, :]
P_mid = P[mid_mask, :]
H_mid_energy = float(np.mean(H_mid ** 2))
P_mid_energy = float(np.mean(P_mid ** 2))
raw_sustain = H_mid_energy / (H_mid_energy + P_mid_energy + EPS)
tonal_sustain = float(min(raw_sustain / EGTR_SUSTAIN_SAT, 1.0))
egtr_presence = float(
EGTR_PICK_WEIGHT * picking_density +
EGTR_FLAT_WEIGHT * mid_saturation +
EGTR_SUSTAIN_WEIGHT * tonal_sustain
)
return egtr_presence, picking_density, mid_saturation, tonal_sustain
def extract_mir_features(path):
"""
Returns (l3_vec, eng_vec, true_peak_dbfs).
eng_vec is a 38-element array whose indices correspond to ENG_DIMS.
Indices 0β25 are unchanged from cache version 3.
Indices 26β29 are drum sub-dimensions (v4βv6, unchanged in v8):
26 drum_kick_level β kick dominance ratio (v5, renamed v10)
27 drum_cymbal_presence β HPSS sustained fraction Γ band prom in 4β10 kHz (v6)
28 drum_confidence β 0.55Γkick + 0.30Γhihat + 0.15Γcymbal (v6)
29 hihat_presence β onset density of percussive 8β14 kHz (v6 NEW)
Indices 30β31 are electric guitar dimensions (v8 NEW):
30 egtr_presence β composite guitar detection score [0,1]
31 egtr_level β egtr_presence Γ (guitar_rms / total_rms)
Indices 32β37 are stem presence + level dimensions (v10 NEW):
32 drum_presence β fraction of 1-sec chunks with active drum signal
33 other_presence β fraction of 1-sec chunks with active other signal
34 bass_presence β fraction of 1-sec chunks with active bass signal
35 drum_kick_presence β kick onset density in 40β120 Hz / DRUM_KICK_ONSET_SAT
36 drum_cymbal_level β RMS(drum, 4β10 kHz) / total_rms
37 hihat_level β RMS(drum, 8β14 kHz) / total_rms
drum_punch (index 2) is gated by drum_confidence:
drum_punch = crest_factor(drum_stem) Γ drum_confidence
v8 changes vs v7:
htdemucs_6s: model switched from htdemucs (4-stem) to htdemucs_6s (6-stem).
New stems: guitar (_STEM_GUITAR=4), piano (_STEM_PIANO=5).
other_prominence (idx 12) now reflects only synths/pads/strings β
guitar and piano are no longer blended into the other stem.
total_rms: now includes guitar_rms + piano_rms (previously only 4 stems).
This makes all stem-share ratios (drum_prominence, bass_prominence,
other_prominence, vocal_level) slightly different from v7.
egtr_presence (idx 30): HPSS-based picking density + mid-band spectral
flatness on the dedicated guitar stem. See _compute_guitar_dims().
egtr_level (idx 31): egtr_presence Γ (guitar_rms / total_rms).
Scales presence by the guitar stem's share of the total mix RMS.
A guitar that is present but buried low in the mix scores lower than
one that is both detectable and prominent.
eng_vec now has 32 elements. All v7 cache entries are incompatible.
true_peak_dbfs is the ITU-R BS.1775-2 true peak level of the full mix
(4Γ oversampled, in dBFS). Returned so the caller can enforce the -1.0 dBFS
true peak ceiling in apply_streaming_normalization() correctly.
Expected outcomes by source type (approximate, v9 constants):
Distorted electric guitar prominent in mix:
egtr_presence β 0.75β1.00, egtr_level β 0.20β0.50
Clean/semi-clean electric guitar throughout:
egtr_presence β 0.55β0.85, egtr_level β 0.10β0.35
Acoustic guitar only (no electric):
egtr_presence β 0.40β0.70, egtr_level β 0.05β0.20
Piano-led track (no guitar):
egtr_presence β 0.10β0.35, egtr_level β 0.01β0.10
Pad/synth-led track (no guitar):
egtr_presence β 0.05β0.25, egtr_level β 0.00β0.08
"""
y, sr = librosa.load(path, sr=44100, mono=True)
y_stereo, _ = librosa.load(path, sr=sr, mono=False)
if y_stereo.ndim == 1: # guard: mono input
y_stereo = np.stack([y_stereo, y_stereo])
emb_l3, _ = openl3.get_audio_embedding(
y, sr, content_type="music", embedding_size=512, verbose=False
)
l3_vec = np.mean(emb_l3, axis=0)
src = get_stems(path) # shape: (6, 2, n_samples) β stereo stems at 44100 Hz
# ββ Stem RMS values βββββββββββββββββββββββββββββββββββββββββββββββββββββββ
drum_rms = np.sqrt(np.mean(src[_STEM_DRUMS] **2))
bass_rms = np.sqrt(np.mean(src[_STEM_BASS] **2))
other_rms = np.sqrt(np.mean(src[_STEM_OTHER] **2))
voc_rms = np.sqrt(np.mean(src[_STEM_VOCALS]**2))
guitar_rms = np.sqrt(np.mean(src[_STEM_GUITAR]**2))
piano_rms = np.sqrt(np.mean(src[_STEM_PIANO] **2))
# v8: total_rms includes all 6 stems for accurate mix-share ratios.
total_rms = drum_rms + bass_rms + other_rms + voc_rms + guitar_rms + piano_rms + EPS
# ββ Drum crest factor (raw, before gating) ββββββββββββββββββββββββββββββββ
drum_crest_raw = np.max(np.abs(src[_STEM_DRUMS])) / (drum_rms + EPS)
# ββ Drum sub-band analysis (v6, unchanged in v8) ββββββββββββββββββββββββββ
drum_mono = (src[_STEM_DRUMS][0] + src[_STEM_DRUMS][1]) / 2.0
drum_kick_lvl, drum_cymbal_pres, hihat_pres, drum_cymbal_lvl, hihat_lvl = \
_compute_drum_bands(drum_mono, y, sr, total_rms)
# drum_confidence: weighted sum of three [0,1] drum indicators.
kick_conf = min(drum_kick_lvl / DRUM_KICK_SAT, 1.0)
hihat_conf = hihat_pres # already in [0, 1]
cymbal_conf = drum_cymbal_pres # already in [0, 1]
drum_conf = float(
DRUM_KICK_WEIGHT * kick_conf +
DRUM_HIHAT_WEIGHT * hihat_conf +
DRUM_CYMBAL_WEIGHT * cymbal_conf
)
# Gate drum_punch by confidence: suppresses false positives from guitar bleed.
drum_punch_val = drum_crest_raw * drum_conf
# ββ Bass crest factor βββββββββββββββββββββββββββββββββββββββββββββββββββββ
bass_tightness_val = np.max(np.abs(src[_STEM_BASS])) / (bass_rms + EPS)
# ββ Mix-level ratios (stem RMS / total RMS) βββββββββββββββββββββββββββββββ
drum_lvl = drum_rms / total_rms
bass_lvl = bass_rms / total_rms
other_lvl = other_rms / total_rms
voc_level = voc_rms / total_rms
# ββ Vocal sub-dimensions ββββββββββββββββββββββββββββββββββββββββββββββββββ
voc_mono = (src[_STEM_VOCALS][0] + src[_STEM_VOCALS][1]) / 2.0
voc_rms_frames = librosa.feature.rms(
y=voc_mono, frame_length=2048, hop_length=512
)[0]
voc_presence = float(np.mean(voc_rms_frames > VOCAL_ACTIVITY_THRESHOLD))
voc_brightness = float(np.mean(
librosa.feature.spectral_centroid(y=voc_mono, sr=sr)
))
# ββ Electric guitar sub-dimensions (v9 updated) ββββββββββββββββββββββββββ
guitar_mono = (src[_STEM_GUITAR][0] + src[_STEM_GUITAR][1]) / 2.0
egtr_pres, _picking, _flatness, _sustain = _compute_guitar_dims(guitar_mono, sr)
egtr_level_val = float(egtr_pres * (guitar_rms / total_rms))
# ββ Stem temporal presence (v10 NEW) βββββββββββββββββββββββββββββββββββββ
drum_presence_val = _compute_stem_presence(
src[_STEM_DRUMS], sr, DRUM_ACTIVITY_THRESHOLD
)
other_presence_val = _compute_stem_presence(
src[_STEM_OTHER], sr, OTHER_ACTIVITY_THRESHOLD
)
bass_presence_val = _compute_stem_presence(
src[_STEM_BASS], sr, BASS_ACTIVITY_THRESHOLD
)
# ββ Kick onset density (v10 NEW) βββββββββββββββββββββββββββββββββββββββββ
drum_kick_pres_val = _compute_kick_onset_presence(drum_mono, sr)
# ββ Mid-side stereo width βββββββββββββββββββββββββββββββββββββββββββββββββ
mid, side = (y_stereo[0] + y_stereo[1]) / 2, (y_stereo[0] - y_stereo[1]) / 2
ms_ratio = (np.sqrt(np.mean(side**2)) /
(np.sqrt(np.mean(mid**2)) + np.sqrt(np.mean(side**2)) + EPS))
# ββ Short-term loudness (K-weighted, BS.1770 β no gating) ββββββββββββββββ
st_lufs = _compute_short_term_lufs_array(y_stereo, sr, window_s=3.0, hop_s=1.0)
# ββ Clipping βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
c_events, c_total, c_ratio, c_max, c_mean = _measure_clipping(y)
# ββ Assemble eng_vec β must match ENG_DIMS order exactly βββββββββββββββββ
eng_vec = np.array([
_compute_integrated_lufs(y_stereo, sr), # 0 lufs (ITU-R BS.1770-4 β K-weighted, gated)
np.max(np.abs(y)) / (np.sqrt(np.mean(y**2)) + EPS), # 1 crest_factor
drum_punch_val, # 2 drum_punch (gated)
drum_lvl, # 3 drum_level (v10: renamed from drum_prominence)
ms_ratio, # 4 stereo_width
librosa.feature.spectral_centroid(y=y, sr=sr).mean(), # 5 brightness
entropy(np.mean(librosa.feature.chroma_cqt(y=y, sr=sr), axis=1)), # 6 harmonic_complexity
c_events, # 7 clip_num_events
c_total, # 8 clip_total
c_ratio, # 9 clip_ratio
c_max, # 10 clip_max_run
c_mean, # 11 clip_mean_run
other_lvl, # 12 other_level (v10: renamed from other_prominence)
voc_level, # 13 vocal_level
voc_presence, # 14 vocal_presence
voc_brightness, # 15 vocal_brightness
np.max(st_lufs) - np.median(st_lufs), # 16 chorus_lift
np.mean(librosa.feature.zero_crossing_rate(y)), # 17 crispness
np.mean(librosa.feature.spectral_flatness(y=y)), # 18 clarity
len(y) / sr, # 19 length_s
np.std(st_lufs), # 20 lufs_variation
np.max(st_lufs), # 21 peak_lufs
np.mean(librosa.feature.spectral_rolloff(y=y, sr=sr)), # 22 rolloff
bass_tightness_val, # 23 bass_tightness
bass_lvl, # 24 bass_level (v10: renamed from bass_prominence)
np.nan_to_num(np.corrcoef(y_stereo[0], y_stereo[1])[0, 1]), # 25 phase_corr
drum_kick_lvl, # 26 drum_kick_level (v10: renamed from drum_kick_presence)
drum_cymbal_pres, # 27 drum_cymbal_presence (v6)
drum_conf, # 28 drum_confidence (v6)
hihat_pres, # 29 hihat_presence (v6)
egtr_pres, # 30 egtr_presence (v8 NEW)
egtr_level_val, # 31 egtr_level (v8 NEW)
drum_presence_val, # 32 drum_presence (v10 NEW)
other_presence_val, # 33 other_presence (v10 NEW)
bass_presence_val, # 34 bass_presence (v10 NEW)
drum_kick_pres_val, # 35 drum_kick_presence (v10 NEW)
drum_cymbal_lvl, # 36 drum_cymbal_level (v10 NEW)
hihat_lvl, # 37 hihat_level (v10 NEW)
])
# ββ True peak (BS.1775-2, 4Γ oversampled) ββββββββββββββββββββββββββββββββ
# Returned as third element so the modal worker / analyze_submissions.py
# can enforce the -1.0 dBFS true peak ceiling in normalization simulation.
true_peak = _true_peak_dbfs(y_stereo)
return l3_vec, eng_vec, true_peak
|