from dataclasses import dataclass from src.about import Tasks def fields(raw_class): return [v for k, v in raw_class.__dict__.items() if k[:2] != "__" and k[-2:] != "__"] # These classes are for user facing column names, # to avoid having to change them all around the code # when a modif is needed @dataclass class ColumnContent: raw_name: str name: str type: str displayed_by_default: bool hidden: bool = False never_hidden: bool = False ## For the queue columns in the submission tab @dataclass(frozen=True) class BenchRawColumn: # Queue column model = ColumnContent("hf_model", "Model", "str", True) task = ColumnContent("task", "Task", "markdown", False) train_len = ColumnContent("train_len", "Train Samples", "number", False) test_len = ColumnContent("test_len", "Test Samples", "number", False) train_proc_time = ColumnContent("train_proc_time", "Train Processing (s)", "number", False) test_proc_time = ColumnContent("test_proc_time", "Test Processing (s)", "number", False) input_length = ColumnContent("input_length", "Number of Samples", "number", False) acc = ColumnContent("accuracy", "Accuracy", "number", True) mcc = ColumnContent("mcc", "MCC", "number", True) infer_time = ColumnContent("inference_time", "Inference Time (s)", "number", False) embds_dim = ColumnContent("embeddings_dim", "Embds Dim", "number", False) model_params = ColumnContent("model_params", "Model Params (M)", "number", True) vram_model = ColumnContent("vram_model", "VRAM Model (MB)", "number", False) classification_report = ColumnContent("classification_report", "Classification Report", "str", False, hidden=True) config = ColumnContent("config", "Config", "str", False, hidden=True) max_context_len = ColumnContent("max_context_size", "Max Context Length (bp)", "number", False) dataset_name = ColumnContent("dataset", "Dataset Name", "str", False) weighted_f1 = ColumnContent("weighted f1", "Weighted F1", "number", True) BenchTableModel = BenchRawColumn() # Column selection COLS = [c.name for c in fields(BenchRawColumn) if not c.hidden] METRICS_COLS = [BenchRawColumn.acc.name, BenchRawColumn.mcc.name, BenchRawColumn.weighted_f1.name] BENCHMARK_COLS = [t.value.col_name for t in Tasks] # For summarizing model performance per task type COLS_TO_AVERAGE = ['Accuracy', 'MCC', 'Weighted F1', 'Train Processing (s)', 'Test Processing (s)', 'Inference Time (s)', 'Number of Samples', 'Train Samples', 'Test Samples'] COLS_DEPEND_ON_MODEL = ['Model Params (M)', 'Embds Dim', 'VRAM Model (MB)', "Max Context Length (bp)", "Dataset Name"] TASK_TYPE_MAP = { "H2AFZ": "Histone", "H3K27ac": "Histone", "H3K27me3": "Histone", "H3K36me3": "Histone", "H3K4me1": "Histone", "H3K4me2": "Histone", "H3K4me3": "Histone", "H3K9ac": "Histone", "H3K9me3": "Histone", "H4K20me1": "Histone", "splice_sites_donors": "Splicing", "splice_sites_acceptors": "Splicing", "splice_sites_all": "Splicing", "promoter_no_tata": "Promoter", "promoter_tata": "Promoter", "promoter_all": "Promoter", "enhancers": "Enhancer", "enhancers_types": "Enhancer", "variant_effect_causal_eqtl": "SNP Classification", "variant_effect_pathogenic_clinvar": "SNP Classification", "variant_effect_pathogenic_omim": "SNP Classification" }