From facb1f2e5b7d5cbdb4ff88ca0a21c2080527874e Mon Sep 17 00:00:00 2001 From: Tarun Kumar Vangani Date: Wed, 12 Nov 2025 22:25:40 +0800 Subject: [PATCH 01/25] feat: Implement dynamic routing for MoE blocks Co-authored-by: aider (gemini/gemini-2.5-pro) --- model_utils.py | 78 ++++++++++++++++++++++++++++++++++++++++++++++++-- 1 file changed, 75 insertions(+), 3 deletions(-) diff --git a/model_utils.py b/model_utils.py index 836bce6..ec93e52 100644 --- a/model_utils.py +++ b/model_utils.py @@ -105,24 +105,96 @@ def patched_forward_zeroed_experts(self, hidden_states: torch.Tensor) -> torch.T final_hidden_states = final_hidden_states.reshape(batch_size, sequence_length, hidden_dim) return final_hidden_states, router_logits -def apply_pruning(model, experts_to_prune, mode="zero"): +def patched_forward_dynamic_routing(self, hidden_states: torch.Tensor) -> torch.Tensor: + """ + Patched forward method for dynamic routing based on cumulative probability threshold. + Selects a variable number of experts up to self.top_k. + """ + batch_size, sequence_length, hidden_dim = hidden_states.shape + hidden_states = hidden_states.view(-1, hidden_dim) + router_logits = self.gate(hidden_states) + + pruned_experts = getattr(self, "pruned_experts", None) + if pruned_experts is not None and len(pruned_experts) > 0: + idxs = torch.tensor(pruned_experts, device=router_logits.device, dtype=torch.long) + router_logits[:, idxs] = float('-inf') + + routing_probs = F.softmax(router_logits, dim=1, dtype=torch.float) + # self.top_k is used as the max number of experts to consider + routing_weights, selected_experts = torch.topk(routing_probs, self.top_k, dim=-1) + + # Dynamic routing based on cumulative probability + threshold = getattr(self, "dynamic_routing_threshold", 0.8) + cumulative_weights = torch.cumsum(routing_weights, dim=-1) + + # Create a mask to select experts until the cumulative probability exceeds the threshold + shifted_cumsum = torch.zeros_like(cumulative_weights) + shifted_cumsum[..., 1:] = cumulative_weights[..., :-1] + selection_mask = shifted_cumsum < threshold + + # Apply the mask to the routing weights + routing_weights = routing_weights * selection_mask + + if self.norm_topk_prob: + # Normalize routing weights for the selected experts + routing_weights_sum = routing_weights.sum(dim=-1, keepdim=True) + # Avoid division by zero + routing_weights_sum = torch.where(routing_weights_sum == 0, 1.0, routing_weights_sum) + routing_weights /= routing_weights_sum + + routing_weights = routing_weights.to(hidden_states.dtype) + + final_hidden_states = torch.zeros( + (batch_size * sequence_length, hidden_dim), dtype=hidden_states.dtype, device=hidden_states.device + ) + + expert_mask = torch.nn.functional.one_hot(selected_experts, num_classes=self.num_experts).permute(2, 1, 0) + + # Apply the dynamic selection mask to the expert_mask + selection_mask_broadcast = selection_mask.permute(1, 0).unsqueeze(0) + expert_mask = expert_mask * selection_mask_broadcast + + expert_hit = torch.greater(expert_mask.sum(dim=(-1, -2)), 0).nonzero() + for expert_idx in expert_hit: + expert_layer = self.experts[expert_idx] + idx, top_x = torch.where(expert_mask[expert_idx].squeeze(0)) + + current_state = hidden_states[None, top_x].reshape(-1, hidden_dim) + current_hidden_states = expert_layer(current_state) * routing_weights[top_x, idx, None] + + final_hidden_states.index_add_(0, top_x, current_hidden_states.to(hidden_states.dtype)) + + shared_expert_output = self.shared_expert(hidden_states) + shared_expert_output = F.sigmoid(self.shared_expert_gate(hidden_states)) * shared_expert_output + + final_hidden_states = final_hidden_states + shared_expert_output + + final_hidden_states = final_hidden_states.reshape(batch_size, sequence_length, hidden_dim) + return final_hidden_states, router_logits + +def apply_pruning(model, experts_to_prune, mode="zero", dynamic_routing_threshold=0.8): """ Monkey patch each OlmoeSparseMoeBlock forward and assign pruned experts per layer. Args: model: The model to prune. experts_to_prune: A dictionary where keys are layer indices and values are lists of expert indices to prune. - mode: "mask" (mask logits) or "zero" (zero out expert outputs) + mode: "mask" (mask logits), "zero" (zero out expert outputs), or "dynamic" (dynamic routing). + dynamic_routing_threshold (float): Cumulative probability threshold for dynamic routing. """ if mode == "mask": patch_fn = patched_forward_masked_experts elif mode == "zero": patch_fn = patched_forward_zeroed_experts + elif mode == "dynamic": + patch_fn = patched_forward_dynamic_routing else: - raise ValueError(f"Unknown mode: {mode}. Use 'mask' or 'zero'.") + raise ValueError(f"Unknown mode: {mode}. Use 'mask', 'zero', or 'dynamic'.") for layer_idx, layer in enumerate(model.model.layers): moe_block = layer.mlp if hasattr(moe_block, "gate") and hasattr(moe_block, "experts"): moe_block.pruned_experts = experts_to_prune.get(layer_idx, []) + if mode == "dynamic": + moe_block.dynamic_routing_threshold = dynamic_routing_threshold moe_block.forward = patch_fn.__get__(moe_block, moe_block.__class__) From 0c348c1856036b659513c33789d9e1cbedb59517 Mon Sep 17 00:00:00 2001 From: Tarun Kumar Vangani Date: Wed, 12 Nov 2025 22:28:03 +0800 Subject: [PATCH 02/25] feat: Add dynamic routing pruning mode and config Co-authored-by: aider (gemini/gemini-2.5-pro) --- evaluation.py | 10 ++++++++-- run_evaluation.sh | 2 +- 2 files changed, 9 insertions(+), 3 deletions(-) diff --git a/evaluation.py b/evaluation.py index 4ae5abc..8ed29a2 100644 --- a/evaluation.py +++ b/evaluation.py @@ -18,8 +18,9 @@ def parse_args(): parser.add_argument('--use_pruned_model', action='store_true', help='Whether to use pruned model') parser.add_argument('--pruned_metadata', type=str, default=None, help='Path to pruned expert metadata JSON') parser.add_argument('--mode', type=str, choices=['least', 'mode'], default='least', help='Strategy for selecting experts to prune') - parser.add_argument('--pruning_method', type=str, choices=['mask', 'zero'], default='zero', help='Method to use for pruning experts') + parser.add_argument('--pruning_method', type=str, choices=['mask', 'zero', 'dynamic'], default='zero', help='Method to use for pruning experts') parser.add_argument('--k', type=int, default=5, help='Number of experts to prune per layer') + parser.add_argument('--dynamic_routing_threshold', type=float, default=0.8, help='Cumulative probability threshold for dynamic routing') parser.add_argument('--device', type=str, default='cuda', help='Device for model') parser.add_argument('--output_file', type=str, default=None, help='File to save results JSON') @@ -39,7 +40,12 @@ def main(): top_k=args.k, mode=args.mode ) - apply_pruning(model.model, experts_to_prune, mode=args.pruning_method) + apply_pruning( + model.model, + experts_to_prune, + mode=args.pruning_method, + dynamic_routing_threshold=args.dynamic_routing_threshold + ) # Prepare arguments for simple_evaluate eval_kwargs = dict( diff --git a/run_evaluation.sh b/run_evaluation.sh index ec71467..8275175 100644 --- a/run_evaluation.sh +++ b/run_evaluation.sh @@ -7,7 +7,7 @@ python /mnt/ssd/shared/giangntt/expert_analysis/clean_code/evaluation.py \ --use_pruned_model \ --pruned_metadata "/mnt/ssd/shared/giangntt/expert_analysis/clean_code/prune_experts/super_experts_ids.json" \ --mode least \ - --pruning_method zero \ + --pruning_method dynamic \ --k 10 \ --device cuda \ # --output_file "/mnt/ssd/shared/giangntt/expert_analysis/clean_code/results.json" From 855d8e35735c32631fac9d90f2309d825d7c08ca Mon Sep 17 00:00:00 2001 From: Tarun Kumar Vangani Date: Wed, 12 Nov 2025 22:39:18 +0800 Subject: [PATCH 03/25] feat: Add script for dynamic routing evaluation on original model Co-authored-by: aider (gemini/gemini-2.5-pro) --- run_dynamic_eval.sh | 22 ++++++++++++++++++++++ 1 file changed, 22 insertions(+) create mode 100644 run_dynamic_eval.sh diff --git a/run_dynamic_eval.sh b/run_dynamic_eval.sh new file mode 100644 index 0000000..c9b9ff5 --- /dev/null +++ b/run_dynamic_eval.sh @@ -0,0 +1,22 @@ +#!/bin/bash + +# This script runs evaluation using the dynamic routing method on the original model. +# +# Why `--use_pruned_model` and `--pruned_metadata` are used: +# The `apply_pruning` function in `evaluation.py`, which patches the model for +# dynamic routing, is only called when the `--use_pruned_model` flag is set. +# This requires a metadata file, but we can prevent any experts from being +# pruned by setting `--k 0`. +# +# This setup allows us to apply the dynamic routing logic to the full, original model. + +python evaluation.py \ + --tasks gsm8k \ + --batch_size 8 \ + --limit 100 \ + --use_pruned_model \ + --pruned_metadata "prune_experts/super_experts_ids.json" \ + --mode least \ + --pruning_method dynamic \ + --k 0 \ + --device cuda From f861180d2303739d5d591f1d29217cb21827e2a8 Mon Sep 17 00:00:00 2001 From: Tarun Kumar Vangani Date: Wed, 12 Nov 2025 23:54:16 +0800 Subject: [PATCH 04/25] chore: Add debug logs for MoE layer patching and threshold Co-authored-by: aider (gemini/gemini-2.5-pro) --- model_utils.py | 2 ++ 1 file changed, 2 insertions(+) diff --git a/model_utils.py b/model_utils.py index ec93e52..3ef916e 100644 --- a/model_utils.py +++ b/model_utils.py @@ -194,7 +194,9 @@ def apply_pruning(model, experts_to_prune, mode="zero", dynamic_routing_threshol for layer_idx, layer in enumerate(model.model.layers): moe_block = layer.mlp if hasattr(moe_block, "gate") and hasattr(moe_block, "experts"): + print(f"INFO: Patching MoE layer {layer_idx} with mode '{mode}'.") moe_block.pruned_experts = experts_to_prune.get(layer_idx, []) if mode == "dynamic": + print(f"INFO: Setting dynamic_routing_threshold to {dynamic_routing_threshold} for layer {layer_idx}.") moe_block.dynamic_routing_threshold = dynamic_routing_threshold moe_block.forward = patch_fn.__get__(moe_block, moe_block.__class__) From 302c895a43d055d53ff2650a230627e7363fbbce Mon Sep 17 00:00:00 2001 From: Tarun Kumar Vangani Date: Thu, 13 Nov 2025 00:05:49 +0800 Subject: [PATCH 05/25] fix: Access dynamic routing threshold directly Co-authored-by: aider (gemini/gemini-2.5-pro) --- model_utils.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/model_utils.py b/model_utils.py index 3ef916e..69f7082 100644 --- a/model_utils.py +++ b/model_utils.py @@ -124,7 +124,8 @@ def patched_forward_dynamic_routing(self, hidden_states: torch.Tensor) -> torch. routing_weights, selected_experts = torch.topk(routing_probs, self.top_k, dim=-1) # Dynamic routing based on cumulative probability - threshold = getattr(self, "dynamic_routing_threshold", 0.8) + # The threshold is set on the MoE block during the patching process. + threshold = self.dynamic_routing_threshold cumulative_weights = torch.cumsum(routing_weights, dim=-1) # Create a mask to select experts until the cumulative probability exceeds the threshold From c89798d19f828b0293f4070f37c9e38eb50ed97f Mon Sep 17 00:00:00 2001 From: Tarun Kumar Vangani Date: Thu, 13 Nov 2025 00:19:35 +0800 Subject: [PATCH 06/25] feat: Add debug logging for dynamic MoE routing decisions Co-authored-by: aider (gemini/gemini-2.5-pro) --- model_utils.py | 15 +++++++++++++++ 1 file changed, 15 insertions(+) diff --git a/model_utils.py b/model_utils.py index 69f7082..50e92a7 100644 --- a/model_utils.py +++ b/model_utils.py @@ -136,6 +136,20 @@ def patched_forward_dynamic_routing(self, hidden_states: torch.Tensor) -> torch. # Apply the mask to the routing weights routing_weights = routing_weights * selection_mask + # DEBUG: Log routing info for the first token to verify the logic + if not hasattr(self, '_debug_logged'): + token_idx = 0 + print("\n" + "="*50) + print(f"DEBUG: Dynamic Routing for first token in batch (Layer: {self.layer_id if hasattr(self, 'layer_id') else 'Unknown'})") + print(f" Threshold: {threshold}") + print(f" Top-k routing weights: {routing_weights[token_idx].tolist()}") + print(f" Cumulative weights: {cumulative_weights[token_idx].tolist()}") + print(f" Selection mask: {selection_mask[token_idx].tolist()}") + num_selected = selection_mask[token_idx].sum().item() + print(f" => Number of experts selected: {num_selected}") + print("="*50 + "\n") + self._debug_logged = True + if self.norm_topk_prob: # Normalize routing weights for the selected experts routing_weights_sum = routing_weights.sum(dim=-1, keepdim=True) @@ -195,6 +209,7 @@ def apply_pruning(model, experts_to_prune, mode="zero", dynamic_routing_threshol for layer_idx, layer in enumerate(model.model.layers): moe_block = layer.mlp if hasattr(moe_block, "gate") and hasattr(moe_block, "experts"): + moe_block.layer_id = layer_idx # For debug logging print(f"INFO: Patching MoE layer {layer_idx} with mode '{mode}'.") moe_block.pruned_experts = experts_to_prune.get(layer_idx, []) if mode == "dynamic": From 723086d1c0f24b3e5cdf83aca5c809ae95bcde8f Mon Sep 17 00:00:00 2001 From: Tarun Kumar Vangani Date: Thu, 13 Nov 2025 00:32:10 +0800 Subject: [PATCH 07/25] fix: Align dynamic routing weight normalization for stability Co-authored-by: aider (gemini/gemini-2.5-pro) --- model_utils.py | 5 +---- 1 file changed, 1 insertion(+), 4 deletions(-) diff --git a/model_utils.py b/model_utils.py index 50e92a7..f9b5c33 100644 --- a/model_utils.py +++ b/model_utils.py @@ -152,10 +152,7 @@ def patched_forward_dynamic_routing(self, hidden_states: torch.Tensor) -> torch. if self.norm_topk_prob: # Normalize routing weights for the selected experts - routing_weights_sum = routing_weights.sum(dim=-1, keepdim=True) - # Avoid division by zero - routing_weights_sum = torch.where(routing_weights_sum == 0, 1.0, routing_weights_sum) - routing_weights /= routing_weights_sum + routing_weights /= routing_weights.sum(dim=-1, keepdim=True) routing_weights = routing_weights.to(hidden_states.dtype) From 21c0ced21783e2b477f8ff45093b8a793d829ee7 Mon Sep 17 00:00:00 2001 From: Tarun Kumar Vangani Date: Thu, 13 Nov 2025 00:38:53 +0800 Subject: [PATCH 08/25] fix: Add epsilon to routing weight normalization for stability Co-authored-by: aider (gemini/gemini-2.5-pro) --- model_utils.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/model_utils.py b/model_utils.py index f9b5c33..58ec652 100644 --- a/model_utils.py +++ b/model_utils.py @@ -152,7 +152,9 @@ def patched_forward_dynamic_routing(self, hidden_states: torch.Tensor) -> torch. if self.norm_topk_prob: # Normalize routing weights for the selected experts - routing_weights /= routing_weights.sum(dim=-1, keepdim=True) + # Add a small epsilon for numerical stability to prevent division by zero. + routing_weights_sum = routing_weights.sum(dim=-1, keepdim=True) + routing_weights = routing_weights / (routing_weights_sum + 1e-6) routing_weights = routing_weights.to(hidden_states.dtype) From 64f49160532b0dda005a57cf816ec9c7fdacc971 Mon Sep 17 00:00:00 2001 From: Tarun Kumar Vangani Date: Thu, 13 Nov 2025 00:48:17 +0800 Subject: [PATCH 09/25] fix: Resolve slicing bug that pruned all experts when top_k is 0 Co-authored-by: aider (gemini/gemini-2.5-pro) --- common_utils.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/common_utils.py b/common_utils.py index 206d742..418b373 100644 --- a/common_utils.py +++ b/common_utils.py @@ -24,6 +24,9 @@ def get_topk_experts_from_json(path, top_k=5, mode="most", criterion=None, out_p experts_to_prune = {} for k, v in sorted_experts.items(): k_int = int(k) + if top_k == 0: + experts_to_prune[k_int] = [] + continue v_int = [int(x) for x in v] if mode == "most": selected = v_int[:top_k] @@ -42,4 +45,4 @@ def get_topk_experts_from_json(path, top_k=5, mode="most", criterion=None, out_p (f" -> saved to '{out_path}'" if out_path else "")) print(f"Layers: {num_layers}, total experts selected: {total_experts}") - return experts_to_prune \ No newline at end of file + return experts_to_prune From 1517aaf85b158151fca5b139e4ab9047306b48ac Mon Sep 17 00:00:00 2001 From: Tarun Kumar Vangani Date: Thu, 13 Nov 2025 01:38:55 +0800 Subject: [PATCH 10/25] feat: Save dynamic routing expert activation report to output file Co-authored-by: aider (gemini/gemini-2.5-pro) --- evaluation.py | 17 +++++++++++++++++ model_utils.py | 21 +++++++-------------- 2 files changed, 24 insertions(+), 14 deletions(-) diff --git a/evaluation.py b/evaluation.py index 8ed29a2..82ce66f 100644 --- a/evaluation.py +++ b/evaluation.py @@ -61,12 +61,29 @@ def main(): results = simple_evaluate(**eval_kwargs) print(results) + expert_activation_report = {} + if args.pruning_method == "dynamic": + total_avg = 0 + num_layers = 0 + for i, layer in enumerate(model.model.layers): + moe_block = layer.mlp + if hasattr(moe_block, "num_activated_experts_log") and moe_block.num_activated_experts_log: + avg_experts = sum(moe_block.num_activated_experts_log) / len(moe_block.num_activated_experts_log) + expert_activation_report[f"layer_{i}"] = f"{avg_experts:.2f}" + total_avg += avg_experts + num_layers += 1 + if num_layers > 0: + expert_activation_report["overall_average"] = f"{total_avg / num_layers:.2f}" + if args.output_file: output_file = args.output_file output_data = { "config": vars(args), "results": results["results"] } + if expert_activation_report: + output_data["expert_activation_report"] = expert_activation_report + with open(output_file, "w") as f: json.dump(output_data, f, indent=4) print(f"Results and config saved to {output_file}") diff --git a/model_utils.py b/model_utils.py index 58ec652..4f01ac1 100644 --- a/model_utils.py +++ b/model_utils.py @@ -136,19 +136,11 @@ def patched_forward_dynamic_routing(self, hidden_states: torch.Tensor) -> torch. # Apply the mask to the routing weights routing_weights = routing_weights * selection_mask - # DEBUG: Log routing info for the first token to verify the logic - if not hasattr(self, '_debug_logged'): - token_idx = 0 - print("\n" + "="*50) - print(f"DEBUG: Dynamic Routing for first token in batch (Layer: {self.layer_id if hasattr(self, 'layer_id') else 'Unknown'})") - print(f" Threshold: {threshold}") - print(f" Top-k routing weights: {routing_weights[token_idx].tolist()}") - print(f" Cumulative weights: {cumulative_weights[token_idx].tolist()}") - print(f" Selection mask: {selection_mask[token_idx].tolist()}") - num_selected = selection_mask[token_idx].sum().item() - print(f" => Number of experts selected: {num_selected}") - print("="*50 + "\n") - self._debug_logged = True + # Track the average number of experts activated in this forward pass + if hasattr(self, "num_activated_experts_log"): + num_selected_experts_per_token = selection_mask.sum(dim=-1).float() + avg_experts_per_batch = num_selected_experts_per_token.mean().item() + self.num_activated_experts_log.append(avg_experts_per_batch) if self.norm_topk_prob: # Normalize routing weights for the selected experts @@ -208,10 +200,11 @@ def apply_pruning(model, experts_to_prune, mode="zero", dynamic_routing_threshol for layer_idx, layer in enumerate(model.model.layers): moe_block = layer.mlp if hasattr(moe_block, "gate") and hasattr(moe_block, "experts"): - moe_block.layer_id = layer_idx # For debug logging + moe_block.layer_id = layer_idx print(f"INFO: Patching MoE layer {layer_idx} with mode '{mode}'.") moe_block.pruned_experts = experts_to_prune.get(layer_idx, []) if mode == "dynamic": + moe_block.num_activated_experts_log = [] print(f"INFO: Setting dynamic_routing_threshold to {dynamic_routing_threshold} for layer {layer_idx}.") moe_block.dynamic_routing_threshold = dynamic_routing_threshold moe_block.forward = patch_fn.__get__(moe_block, moe_block.__class__) From 1ce3512d6e9958ad92644e95d8977b6b32756aa3 Mon Sep 17 00:00:00 2001 From: Tarun Kumar Vangani Date: Thu, 13 Nov 2025 01:46:00 +0800 Subject: [PATCH 11/25] fix: Fix AttributeError when accessing Qwen2MoeForCausalLM layers Co-authored-by: aider (gemini/gemini-2.5-pro) --- evaluation.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/evaluation.py b/evaluation.py index 82ce66f..98ec781 100644 --- a/evaluation.py +++ b/evaluation.py @@ -65,7 +65,7 @@ def main(): if args.pruning_method == "dynamic": total_avg = 0 num_layers = 0 - for i, layer in enumerate(model.model.layers): + for i, layer in enumerate(model.model.model.layers): moe_block = layer.mlp if hasattr(moe_block, "num_activated_experts_log") and moe_block.num_activated_experts_log: avg_experts = sum(moe_block.num_activated_experts_log) / len(moe_block.num_activated_experts_log) From e79a3b5b1a3935277be95bc0dbc56b7b5a8aaab1 Mon Sep 17 00:00:00 2001 From: Tarun Kumar Vangani Date: Thu, 13 Nov 2025 01:57:11 +0800 Subject: [PATCH 12/25] chore: Update dynamic eval script for MMLU with threshold --- run_dynamic_eval.sh | 12 +++++++++--- 1 file changed, 9 insertions(+), 3 deletions(-) diff --git a/run_dynamic_eval.sh b/run_dynamic_eval.sh index c9b9ff5..361be2d 100644 --- a/run_dynamic_eval.sh +++ b/run_dynamic_eval.sh @@ -11,12 +11,18 @@ # This setup allows us to apply the dynamic routing logic to the full, original model. python evaluation.py \ - --tasks gsm8k \ + --tasks mmlu \ --batch_size 8 \ - --limit 100 \ + --limit 1 \ --use_pruned_model \ --pruned_metadata "prune_experts/super_experts_ids.json" \ --mode least \ --pruning_method dynamic \ --k 0 \ - --device cuda + --dynamic_routing_threshold 0.2 \ + --device cuda \ + --output_file "results_dynamic_mmlu.json" + + + + From 46a83599ebf1fce6f90b4d225f9d4050f78090f7 Mon Sep 17 00:00:00 2001 From: Tarun Kumar Vangani Date: Thu, 13 Nov 2025 01:57:14 +0800 Subject: [PATCH 13/25] feat: Add script to loop dynamic routing thresholds for evaluation Co-authored-by: aider (gemini/gemini-2.5-pro) --- run_dynamic_eval.sh | 42 ++++++++++++++++++++++++++++++------------ 1 file changed, 30 insertions(+), 12 deletions(-) diff --git a/run_dynamic_eval.sh b/run_dynamic_eval.sh index 361be2d..e0104b8 100644 --- a/run_dynamic_eval.sh +++ b/run_dynamic_eval.sh @@ -10,18 +10,36 @@ # # This setup allows us to apply the dynamic routing logic to the full, original model. -python evaluation.py \ - --tasks mmlu \ - --batch_size 8 \ - --limit 1 \ - --use_pruned_model \ - --pruned_metadata "prune_experts/super_experts_ids.json" \ - --mode least \ - --pruning_method dynamic \ - --k 0 \ - --dynamic_routing_threshold 0.2 \ - --device cuda \ - --output_file "results_dynamic_mmlu.json" +# Define the thresholds to test +THRESHOLDS="0.05 0.1 0.15 0.2 0.25" + +# Loop over each threshold and run the evaluation +for threshold in $THRESHOLDS +do + echo "Running evaluation with dynamic routing threshold: $threshold" + + # Construct the output filename + output_filename="results_dynamic_mmlu_${threshold}.json" + + # Run the evaluation script + python evaluation.py \ + --tasks mmlu \ + --batch_size 8 \ + --limit 1 \ + --use_pruned_model \ + --pruned_metadata "prune_experts/super_experts_ids.json" \ + --mode least \ + --pruning_method dynamic \ + --k 0 \ + --dynamic_routing_threshold "$threshold" \ + --device cuda \ + --output_file "$output_filename" + + echo "Evaluation finished for threshold $threshold. Results saved to $output_filename" + echo "----------------------------------------------------" +done + +echo "All evaluations complete." From 8f20ff957ac68e9c6e2b9ba3469fddc8dfd7ebe7 Mon Sep 17 00:00:00 2001 From: Tarun Kumar Vangani Date: Thu, 13 Nov 2025 11:51:40 +0800 Subject: [PATCH 14/25] feat: Add moe_analysis script --- moe_analyis.py | 426 +++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 426 insertions(+) create mode 100644 moe_analyis.py diff --git a/moe_analyis.py b/moe_analyis.py new file mode 100644 index 0000000..671ece4 --- /dev/null +++ b/moe_analyis.py @@ -0,0 +1,426 @@ +import marimo + +__generated_with = "0.17.7" +app = marimo.App(width="medium") + + +@app.cell +def _(): + import marimo as mo + import pandas as pd + import altair as alt + import os + import json + return alt, json, mo, os, pd + + +@app.cell +def _(json, os): + def load_threshold_data(results_folder: str) -> dict: + """ + Loads JSON data from files in the specified results folder. + + Each JSON file is expected to contain a 'config' key with a + 'dynamic_routing_threshold' sub-key. The full JSON data is stored + in a dictionary using this threshold as the key. + + Args: + results_folder (str): The path to the folder containing JSON result files. + + Returns: + dict: A dictionary where keys are dynamic routing thresholds and values + are the full JSON data loaded from the corresponding file. + Returns an empty dictionary if the folder does not exist or + no valid data is found. + """ + loaded_data = {} + if os.path.exists(results_folder) and os.path.isdir(results_folder): + for filename in os.listdir(results_folder): + if filename.endswith(".json"): + filepath = os.path.join(results_folder, filename) + try: + with open(filepath, 'r') as f: + data = json.load(f) + if 'config' in data and 'dynamic_routing_threshold' in data['config']: + threshold_key = data['config']['dynamic_routing_threshold'] + loaded_data[threshold_key] = data + else: + print(f"Warning: 'config' or 'dynamic_routing_threshold' key not found in {filename}") + except json.JSONDecodeError: + print(f"Error decoding JSON from {filename}") + except Exception as e: + print(f"Error reading {filename}: {e}") + else: + print(f"The folder '{results_folder}' does not exist or is not a directory.") + return loaded_data + + results_folder = "results" + loaded_threshold_data = load_threshold_data(results_folder) + + loaded_threshold_data + return (loaded_threshold_data,) + + +@app.cell +def _(loaded_threshold_data): + loaded_threshold_data[0.1] + return + + +@app.function +def extract_mmlu_categories(loaded_threshold_data): + unique_mmlu_categories = set() + + for threshold_str, data in loaded_threshold_data.items(): + results = data.get('results', {}) + + # Add overall MMLU category + if 'mmlu' in results: + unique_mmlu_categories.add('MMLU Overall') + + # Add MMLU subcategories + for cat_key, cat_data in results.items(): + if cat_key.startswith('mmlu_') and 'acc,none' in cat_data: + alias = cat_data.get('alias', cat_key).replace(' - ', '') + unique_mmlu_categories.add(alias.strip()) + + mmlu_categories_list = sorted(list(unique_mmlu_categories)) + return mmlu_categories_list + + +@app.cell +def _(loaded_threshold_data): + mmlu_categories_list = extract_mmlu_categories(loaded_threshold_data) + mmlu_categories_list + return + + +@app.cell +def _(): + # Grouping MMLU sub-categories into broader topics for analysis. + # These groupings are based on the standard MMLU benchmark structure. + mmlu_category_groups = { + "STEM": [ + "abstract_algebra", "anatomy", "astronomy", "college_biology", + "college_chemistry", "college_computer_science", "college_mathematics", + "college_medicine", "college_physics", "computer_security", + "conceptual_physics", "econometrics", "electrical_engineering", + "elementary_mathematics", "high_school_biology", "high_school_chemistry", + "high_school_computer_science", "high_school_mathematics", + "high_school_physics", "high_school_statistics", "machine_learning", + "medical_genetics", "nutrition", "professional_medicine", "virology", + "clinical_knowledge" + ], + "Humanities": [ + "formal_logic", "high_school_european_history", "high_school_geography", + "high_school_us_history", "high_school_world_history", "international_law", + "jurisprudence", "logical_fallacies", "moral_disputes", + "moral_scenarios", "philosophy", "prehistory", "professional_law", + "world_religions" + ], + "Social Sciences": [ + "business_ethics", "global_facts", "high_school_government_and_politics", + "high_school_macroeconomics", "high_school_microeconomics", + "high_school_psychology", "human_aging", "human_sexuality", + "management", "marketing", "professional_accounting", + "professional_psychology", "public_relations", "security_studies", + "sociology", "us_foreign_policy" + ], + "Other": [ + "miscellaneous" + ], + "Overall": [ + "MMLU Overall" + ] + } + + mmlu_category_groups + return (mmlu_category_groups,) + + +@app.cell +def _(alt, mo, pd): + def create_single_mmlu_group_chart(group_name: str, df_mmlu_acc: pd.DataFrame, mmlu_category_groups: dict): + """ + Creates an Altair chart for a specific MMLU category group, showing accuracy + across different dynamic routing thresholds for its subcategories. + + Args: + group_name (str): The name of the MMLU category group (e.g., "STEM", "Humanities"). + df_mmlu_acc (pd.DataFrame): DataFrame containing MMLU accuracy data, + with columns 'threshold', 'category', 'accuracy', 'stderr'. + mmlu_category_groups (dict): A dictionary mapping group names to lists of MMLU subcategories. + + Returns: + alt.Chart or mo.md: An Altair chart object for the specified group, + or a marimo markdown message if data is not available. + """ + if df_mmlu_acc is None or df_mmlu_acc.empty: + return mo.md("No MMLU accuracy data available to create category-specific charts.") + + if group_name not in mmlu_category_groups: + return mo.md(f"Error: MMLU category group '{group_name}' not found.") + + subcategories_list = mmlu_category_groups[group_name] + filtered_df = df_mmlu_acc[df_mmlu_acc['category'].isin(subcategories_list)] + + if filtered_df.empty: + return mo.md(f"No data found for MMLU category group: **{group_name}**.") + + # Create the base chart for the current group + base_chart = alt.Chart(filtered_df).encode( + x=alt.X('threshold:Q', title='Dynamic Routing Threshold'), + y=alt.Y('accuracy:Q', title='Accuracy', scale=alt.Scale(zero=False)), + color=alt.Color('category:N', title='MMLU Subcategory'), + tooltip=[ + alt.Tooltip('threshold', title='Threshold'), + alt.Tooltip('category', title='Subcategory'), + alt.Tooltip('accuracy', title='Accuracy', format='.3f'), + alt.Tooltip('stderr', title='Std. Error', format='.3f') + ] + ).properties( + title=f'{group_name} MMLU Accuracy Across Thresholds' + ) + + # Line chart + line_chart = base_chart.mark_line(point=True) + + # Error bars + error_bars = base_chart.mark_errorbar(extent='stderr').encode( + yError='stderr:Q' + ) + + # Combine line chart and error bars, make interactive + combined_group_chart = (line_chart + error_bars).interactive() + return combined_group_chart + return (create_single_mmlu_group_chart,) + + +@app.cell +def _(create_single_mmlu_group_chart, df_mmlu_acc, mmlu_category_groups): + humanities_chart = create_single_mmlu_group_chart("Humanities", df_mmlu_acc, mmlu_category_groups) + humanities_chart + return + + +@app.cell +def _(alt, conti, loaded_threshold_data, mo, pd): + # Prepare data for plotting + plot_data = [] + for threshold_str, data in loaded_threshold_data.items(): + if threshold_str == "0.2": + conti + try: + threshold = float(threshold_str) + results = data.get('results', {}) + + # Overall MMLU + if 'mmlu' in results: + plot_data.append({ + 'threshold': threshold, + 'category': 'MMLU Overall', + 'accuracy': float(results['mmlu']['acc,none']), + 'stderr': float(results['mmlu']['acc_stderr,none']) + }) + + # All MMLU subcategories + for cat_key, cat_data in results.items(): + # Check if it's an MMLU subcategory (starts with 'mmlu_' but is not the overall 'mmlu' key) + if cat_key.startswith('mmlu_') and 'acc,none' in cat_data: + # Extract alias, removing leading ' - ' if present + alias = cat_data.get('alias', cat_key).replace(' - ', '') + plot_data.append({ + 'threshold': threshold, + 'category': alias, + 'accuracy': float(cat_data['acc,none']), + 'stderr': float(cat_data['acc_stderr,none']) + }) + except (ValueError, KeyError) as e: + print(f"Skipping data for threshold '{threshold_str}' due to error: {e}") + continue + + if not plot_data: + mo.md("No valid MMLU accuracy data found for plotting.") + else: + df_mmlu_acc = pd.DataFrame(plot_data) + + # Create the Altair chart + chart = alt.Chart(df_mmlu_acc).mark_line(point=True).encode( + x=alt.X('threshold:Q', title='Dynamic Routing Threshold'), + y=alt.Y('accuracy:Q', title='Accuracy', scale=alt.Scale(zero=False)), + color=alt.Color('category:N', title='MMLU Category'), + tooltip=[ + alt.Tooltip('threshold', title='Threshold'), + alt.Tooltip('category', title='Category'), + alt.Tooltip('accuracy', title='Accuracy', format='.3f'), + alt.Tooltip('stderr', title='Std. Error', format='.3f') + ] + ).properties( + title='MMLU Accuracy Across Different Dynamic Routing Thresholds' + ) + + # Add error bars + error_bars = alt.Chart(df_mmlu_acc).mark_errorbar(extent='stderr').encode( + x='threshold:Q', + y='accuracy:Q', + yError='stderr:Q', + color='category:N', + tooltip=[ + alt.Tooltip('threshold', title='Threshold'), + alt.Tooltip('category', title='Category'), + alt.Tooltip('accuracy', title='Accuracy', format='.3f'), + alt.Tooltip('stderr', title='Std. Error', format='.3f') + ] + ) + + # Combine the line chart and error bars + combined_chart = (chart + error_bars).interactive() + combined_chart + return combined_chart, df_mmlu_acc + + +@app.cell +def _(combined_chart): + combined_chart + return + + +@app.cell +def _(loaded_threshold_data): + loaded_threshold_data[0.25] + return + + +@app.cell +def _(): + [ + " abstract_algebra", + " anatomy", + " astronomy", + " business_ethics", + " clinical_knowledge", + " college_biology", + " college_chemistry", + " college_computer_science", + " college_mathematics", + " college_medicine", + " college_physics", + " computer_security", + " conceptual_physics", + " econometrics", + " electrical_engineering", + " elementary_mathematics", + " formal_logic", + " global_facts", + " high_school_biology", + " high_school_chemistry", + " high_school_computer_science", + " high_school_european_history", + " high_school_geography", + " high_school_government_and_politics", + " high_school_macroeconomics", + " high_school_mathematics", + " high_school_microeconomics", + " high_school_physics", + " high_school_psychology", + " high_school_statistics", + " high_school_us_history", + " high_school_world_history", + " human_aging", + " human_sexuality", + " international_law", + " jurisprudence", + " logical_fallacies", + " machine_learning", + " management", + " marketing", + " medical_genetics", + " miscellaneous", + " moral_disputes", + " moral_scenarios", + " nutrition", + " philosophy", + " prehistory", + " professional_accounting", + " professional_law", + " professional_medicine", + " professional_psychology", + " public_relations", + " security_studies", + " sociology", + " us_foreign_policy", + " virology", + " world_religions", + "MMLU Overall", + "humanities", + "other", + "social sciences", + "stem" + ] + return + + +@app.cell +def _(): + return + + +@app.cell +def _(alt, mo): + def create_mmlu_category_charts(df_mmlu_acc, mmlu_category_groups): + category_charts = [] + + if df_mmlu_acc is not None and not df_mmlu_acc.empty: + for group_name, subcategories_list in mmlu_category_groups.items(): + # Filter the DataFrame for the current higher-level category's subcategories + filtered_df = df_mmlu_acc[df_mmlu_acc['category'].isin(subcategories_list)] + + if not filtered_df.empty: + # Create the base chart for the current group + base_chart = alt.Chart(filtered_df).encode( + x=alt.X('threshold:Q', title='Dynamic Routing Threshold'), + y=alt.Y('accuracy:Q', title='Accuracy', scale=alt.Scale(zero=False)), + color=alt.Color('category:N', title='MMLU Subcategory'), + tooltip=[ + alt.Tooltip('threshold', title='Threshold'), + alt.Tooltip('category', title='Subcategory'), + alt.Tooltip('accuracy', title='Accuracy', format='.3f'), + alt.Tooltip('stderr', title='Std. Error', format='.3f') + ] + ).properties( + title=f'{group_name} MMLU Accuracy Across Thresholds' + ) + + # Line chart + line_chart = base_chart.mark_line(point=True) + + # Error bars + error_bars = base_chart.mark_errorbar(extent='stderr').encode( + yError='stderr:Q' + ) + + # Combine line chart and error bars, make interactive + combined_group_chart = (line_chart + error_bars).interactive() + category_charts.append(combined_group_chart) + else: + # Assuming 'mo' (marimo) is available in the notebook environment + category_charts.append(mo.md("No MMLU accuracy data available to create category-specific charts.")) + + return category_charts + return (create_mmlu_category_charts,) + + +@app.cell +def _(create_mmlu_category_charts, df_mmlu_acc, mmlu_category_groups): + category_charts = create_mmlu_category_charts(df_mmlu_acc, mmlu_category_groups) + category_charts + return + + +@app.cell +def _(): + return + + +if __name__ == "__main__": + app.run() From a984ca458fbb241c015d936ab4a59dab2e5df3d3 Mon Sep 17 00:00:00 2001 From: Tarun Kumar Vangani Date: Thu, 13 Nov 2025 11:51:42 +0800 Subject: [PATCH 15/25] fix: Strip whitespace from MMLU category names Co-authored-by: aider (gemini/gemini-2.5-pro) --- moe_analyis.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/moe_analyis.py b/moe_analyis.py index 671ece4..fe02bd9 100644 --- a/moe_analyis.py +++ b/moe_analyis.py @@ -231,7 +231,7 @@ def _(alt, conti, loaded_threshold_data, mo, pd): alias = cat_data.get('alias', cat_key).replace(' - ', '') plot_data.append({ 'threshold': threshold, - 'category': alias, + 'category': alias.strip(), 'accuracy': float(cat_data['acc,none']), 'stderr': float(cat_data['acc_stderr,none']) }) From beffad88065549cdeeec164339a6c715189f64cb Mon Sep 17 00:00:00 2001 From: Tarun Kumar Vangani Date: Thu, 13 Nov 2025 11:57:45 +0800 Subject: [PATCH 16/25] fix: Robustly parse MMLU category aliases from result data Co-authored-by: aider (gemini/gemini-2.5-pro) --- moe_analyis.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/moe_analyis.py b/moe_analyis.py index fe02bd9..21408a0 100644 --- a/moe_analyis.py +++ b/moe_analyis.py @@ -81,7 +81,7 @@ def extract_mmlu_categories(loaded_threshold_data): # Add MMLU subcategories for cat_key, cat_data in results.items(): if cat_key.startswith('mmlu_') and 'acc,none' in cat_data: - alias = cat_data.get('alias', cat_key).replace(' - ', '') + alias = cat_data.get('alias', cat_key).split('-')[-1] unique_mmlu_categories.add(alias.strip()) mmlu_categories_list = sorted(list(unique_mmlu_categories)) @@ -228,7 +228,7 @@ def _(alt, conti, loaded_threshold_data, mo, pd): # Check if it's an MMLU subcategory (starts with 'mmlu_' but is not the overall 'mmlu' key) if cat_key.startswith('mmlu_') and 'acc,none' in cat_data: # Extract alias, removing leading ' - ' if present - alias = cat_data.get('alias', cat_key).replace(' - ', '') + alias = cat_data.get('alias', cat_key).split('-')[-1] plot_data.append({ 'threshold': threshold, 'category': alias.strip(), From 678313c8e2b830543c75b0193ae75e339283e276 Mon Sep 17 00:00:00 2001 From: Tarun Kumar Vangani Date: Thu, 13 Nov 2025 12:09:28 +0800 Subject: [PATCH 17/25] refactor: Consolidate MMLU category chart generation and display --- moe_analyis.py | 148 ++++--------------------------------------------- 1 file changed, 10 insertions(+), 138 deletions(-) diff --git a/moe_analyis.py b/moe_analyis.py index 21408a0..d15a2fe 100644 --- a/moe_analyis.py +++ b/moe_analyis.py @@ -139,77 +139,11 @@ def _(): @app.cell -def _(alt, mo, pd): - def create_single_mmlu_group_chart(group_name: str, df_mmlu_acc: pd.DataFrame, mmlu_category_groups: dict): - """ - Creates an Altair chart for a specific MMLU category group, showing accuracy - across different dynamic routing thresholds for its subcategories. - - Args: - group_name (str): The name of the MMLU category group (e.g., "STEM", "Humanities"). - df_mmlu_acc (pd.DataFrame): DataFrame containing MMLU accuracy data, - with columns 'threshold', 'category', 'accuracy', 'stderr'. - mmlu_category_groups (dict): A dictionary mapping group names to lists of MMLU subcategories. - - Returns: - alt.Chart or mo.md: An Altair chart object for the specified group, - or a marimo markdown message if data is not available. - """ - if df_mmlu_acc is None or df_mmlu_acc.empty: - return mo.md("No MMLU accuracy data available to create category-specific charts.") - - if group_name not in mmlu_category_groups: - return mo.md(f"Error: MMLU category group '{group_name}' not found.") - - subcategories_list = mmlu_category_groups[group_name] - filtered_df = df_mmlu_acc[df_mmlu_acc['category'].isin(subcategories_list)] - - if filtered_df.empty: - return mo.md(f"No data found for MMLU category group: **{group_name}**.") - - # Create the base chart for the current group - base_chart = alt.Chart(filtered_df).encode( - x=alt.X('threshold:Q', title='Dynamic Routing Threshold'), - y=alt.Y('accuracy:Q', title='Accuracy', scale=alt.Scale(zero=False)), - color=alt.Color('category:N', title='MMLU Subcategory'), - tooltip=[ - alt.Tooltip('threshold', title='Threshold'), - alt.Tooltip('category', title='Subcategory'), - alt.Tooltip('accuracy', title='Accuracy', format='.3f'), - alt.Tooltip('stderr', title='Std. Error', format='.3f') - ] - ).properties( - title=f'{group_name} MMLU Accuracy Across Thresholds' - ) - - # Line chart - line_chart = base_chart.mark_line(point=True) - - # Error bars - error_bars = base_chart.mark_errorbar(extent='stderr').encode( - yError='stderr:Q' - ) - - # Combine line chart and error bars, make interactive - combined_group_chart = (line_chart + error_bars).interactive() - return combined_group_chart - return (create_single_mmlu_group_chart,) - - -@app.cell -def _(create_single_mmlu_group_chart, df_mmlu_acc, mmlu_category_groups): - humanities_chart = create_single_mmlu_group_chart("Humanities", df_mmlu_acc, mmlu_category_groups) - humanities_chart - return - - -@app.cell -def _(alt, conti, loaded_threshold_data, mo, pd): +def _(alt, loaded_threshold_data, mo, pd): # Prepare data for plotting plot_data = [] for threshold_str, data in loaded_threshold_data.items(): - if threshold_str == "0.2": - conti + try: threshold = float(threshold_str) results = data.get('results', {}) @@ -292,76 +226,8 @@ def _(loaded_threshold_data): @app.cell -def _(): - [ - " abstract_algebra", - " anatomy", - " astronomy", - " business_ethics", - " clinical_knowledge", - " college_biology", - " college_chemistry", - " college_computer_science", - " college_mathematics", - " college_medicine", - " college_physics", - " computer_security", - " conceptual_physics", - " econometrics", - " electrical_engineering", - " elementary_mathematics", - " formal_logic", - " global_facts", - " high_school_biology", - " high_school_chemistry", - " high_school_computer_science", - " high_school_european_history", - " high_school_geography", - " high_school_government_and_politics", - " high_school_macroeconomics", - " high_school_mathematics", - " high_school_microeconomics", - " high_school_physics", - " high_school_psychology", - " high_school_statistics", - " high_school_us_history", - " high_school_world_history", - " human_aging", - " human_sexuality", - " international_law", - " jurisprudence", - " logical_fallacies", - " machine_learning", - " management", - " marketing", - " medical_genetics", - " miscellaneous", - " moral_disputes", - " moral_scenarios", - " nutrition", - " philosophy", - " prehistory", - " professional_accounting", - " professional_law", - " professional_medicine", - " professional_psychology", - " public_relations", - " security_studies", - " sociology", - " us_foreign_policy", - " virology", - " world_religions", - "MMLU Overall", - "humanities", - "other", - "social sciences", - "stem" - ] - return - - -@app.cell -def _(): +def _(mmlu_category_groups): + mmlu_category_groups.keys() return @@ -414,6 +280,12 @@ def create_mmlu_category_charts(df_mmlu_acc, mmlu_category_groups): def _(create_mmlu_category_charts, df_mmlu_acc, mmlu_category_groups): category_charts = create_mmlu_category_charts(df_mmlu_acc, mmlu_category_groups) category_charts + return (category_charts,) + + +@app.cell +def _(category_charts, mo): + mo.vstack(category_charts) return From fe37ddead66da9c4fcd86a744cf784658d3a6ae9 Mon Sep 17 00:00:00 2001 From: Tarun Kumar Vangani Date: Thu, 13 Nov 2025 12:10:08 +0800 Subject: [PATCH 18/25] fix: Handle MMLU data parsing errors more gracefully per category Co-authored-by: aider (gemini/gemini-2.5-pro) --- moe_analyis.py | 38 ++++++++++++++++++++++---------------- 1 file changed, 22 insertions(+), 16 deletions(-) diff --git a/moe_analyis.py b/moe_analyis.py index d15a2fe..2dcb1da 100644 --- a/moe_analyis.py +++ b/moe_analyis.py @@ -143,33 +143,39 @@ def _(alt, loaded_threshold_data, mo, pd): # Prepare data for plotting plot_data = [] for threshold_str, data in loaded_threshold_data.items(): - try: threshold = float(threshold_str) results = data.get('results', {}) # Overall MMLU if 'mmlu' in results: - plot_data.append({ - 'threshold': threshold, - 'category': 'MMLU Overall', - 'accuracy': float(results['mmlu']['acc,none']), - 'stderr': float(results['mmlu']['acc_stderr,none']) - }) + try: + plot_data.append({ + 'threshold': threshold, + 'category': 'MMLU Overall', + 'accuracy': float(results['mmlu']['acc,none']), + 'stderr': float(results['mmlu']['acc_stderr,none']) + }) + except (ValueError, KeyError, TypeError) as e: + print(f"Could not process overall MMLU for threshold {threshold}: {e}") + # All MMLU subcategories for cat_key, cat_data in results.items(): # Check if it's an MMLU subcategory (starts with 'mmlu_' but is not the overall 'mmlu' key) if cat_key.startswith('mmlu_') and 'acc,none' in cat_data: - # Extract alias, removing leading ' - ' if present - alias = cat_data.get('alias', cat_key).split('-')[-1] - plot_data.append({ - 'threshold': threshold, - 'category': alias.strip(), - 'accuracy': float(cat_data['acc,none']), - 'stderr': float(cat_data['acc_stderr,none']) - }) - except (ValueError, KeyError) as e: + try: + # Extract alias, removing leading ' - ' if present + alias = cat_data.get('alias', cat_key).split('-')[-1] + plot_data.append({ + 'threshold': threshold, + 'category': alias.strip(), + 'accuracy': float(cat_data['acc,none']), + 'stderr': float(cat_data['acc_stderr,none']) + }) + except (ValueError, KeyError, TypeError) as e: + print(f"Could not process {cat_key} for threshold {threshold}: {e}") + except (ValueError, TypeError) as e: print(f"Skipping data for threshold '{threshold_str}' due to error: {e}") continue From 6b4eb8e8d59a094f6b1a287769feb93849048a96 Mon Sep 17 00:00:00 2001 From: Tarun Kumar Vangani Date: Thu, 13 Nov 2025 13:39:16 +0800 Subject: [PATCH 19/25] feat: Add expert activation vs. threshold plot Co-authored-by: aider (gemini/gemini-2.5-pro) --- moe_analyis.py | 45 +++++++++++++++++++++++++++++++++++++++++++-- 1 file changed, 43 insertions(+), 2 deletions(-) diff --git a/moe_analyis.py b/moe_analyis.py index 2dcb1da..5b4bb3a 100644 --- a/moe_analyis.py +++ b/moe_analyis.py @@ -296,8 +296,49 @@ def _(category_charts, mo): @app.cell -def _(): - return +def _(alt, loaded_threshold_data, mo, pd): + """ + Processes expert activation data and creates a plot showing the + average number of activated experts versus the dynamic routing threshold. + """ + expert_data = [] + for threshold_str, data in loaded_threshold_data.items(): + try: + threshold = float(threshold_str) + report = data.get('expert_activation_report', {}) + avg_experts = report.get('overall_average') + + if avg_experts is not None: + expert_data.append({ + 'threshold': threshold, + 'average_experts': float(avg_experts), + }) + except (ValueError, TypeError) as e: + print(f"Could not process expert activation data for threshold '{threshold_str}': {e}") + continue + + chart = None + df_expert_activation = None + if expert_data: + df_expert_activation = pd.DataFrame(expert_data) + + # Create the chart + chart = alt.Chart(df_expert_activation).mark_line(point=True).encode( + x=alt.X('threshold:Q', title='Dynamic Routing Threshold'), + y=alt.Y('average_experts:Q', title='Average Activated Experts'), + tooltip=[ + alt.Tooltip('threshold', title='Threshold'), + alt.Tooltip('average_experts', title='Avg. Activated Experts', format='.2f') + ] + ).properties( + title='Average Activated Experts vs. Dynamic Routing Threshold' + ).interactive() + + chart + else: + mo.md("No expert activation data found to plot.") + + return chart, df_expert_activation if __name__ == "__main__": From df62952a23dc0d9d1736b3b6e0772c8ada6c0e94 Mon Sep 17 00:00:00 2001 From: Tarun Kumar Vangani Date: Thu, 13 Nov 2025 13:42:10 +0800 Subject: [PATCH 20/25] refactor: Modularize expert activation chart creation --- moe_analyis.py | 77 ++++++++++++++++++++++++++++++++++---------------- 1 file changed, 52 insertions(+), 25 deletions(-) diff --git a/moe_analyis.py b/moe_analyis.py index 5b4bb3a..c86a5cc 100644 --- a/moe_analyis.py +++ b/moe_analyis.py @@ -216,7 +216,7 @@ def _(alt, loaded_threshold_data, mo, pd): # Combine the line chart and error bars combined_chart = (chart + error_bars).interactive() combined_chart - return combined_chart, df_mmlu_acc + return chart, combined_chart, df_mmlu_acc @app.cell @@ -296,33 +296,49 @@ def _(category_charts, mo): @app.cell -def _(alt, loaded_threshold_data, mo, pd): +def _(alt, chart, loaded_threshold_data, mo, pd): """ Processes expert activation data and creates a plot showing the average number of activated experts versus the dynamic routing threshold. """ - expert_data = [] - for threshold_str, data in loaded_threshold_data.items(): - try: - threshold = float(threshold_str) - report = data.get('expert_activation_report', {}) - avg_experts = report.get('overall_average') - - if avg_experts is not None: - expert_data.append({ - 'threshold': threshold, - 'average_experts': float(avg_experts), - }) - except (ValueError, TypeError) as e: - print(f"Could not process expert activation data for threshold '{threshold_str}': {e}") - continue - chart = None - df_expert_activation = None - if expert_data: - df_expert_activation = pd.DataFrame(expert_data) + def _extract_expert_activation_data(loaded_threshold_data): + """ + Extracts expert activation data from the loaded threshold data. + + Args: + loaded_threshold_data (dict): A dictionary containing data for various thresholds. + + Returns: + list: A list of dictionaries, each containing 'threshold' and 'average_experts'. + """ + expert_data = [] + for threshold_str, data in loaded_threshold_data.items(): + try: + threshold = float(threshold_str) + report = data.get('expert_activation_report', {}) + avg_experts = report.get('overall_average') + + if avg_experts is not None: + expert_data.append({ + 'threshold': threshold, + 'average_experts': float(avg_experts), + }) + except (ValueError, TypeError) as e: + print(f"Could not process expert activation data for threshold '{threshold_str}': {e}") + continue + return expert_data - # Create the chart + def _create_expert_activation_chart(df_expert_activation): + """ + Creates an Altair chart showing average activated experts vs. dynamic routing threshold. + + Args: + df_expert_activation (pd.DataFrame): DataFrame with 'threshold' and 'average_experts' columns. + + Returns: + alt.Chart: An Altair chart object. + """ chart = alt.Chart(df_expert_activation).mark_line(point=True).encode( x=alt.X('threshold:Q', title='Dynamic Routing Threshold'), y=alt.Y('average_experts:Q', title='Average Activated Experts'), @@ -333,12 +349,23 @@ def _(alt, loaded_threshold_data, mo, pd): ).properties( title='Average Activated Experts vs. Dynamic Routing Threshold' ).interactive() + return chart + + expert_data_list = _extract_expert_activation_data(loaded_threshold_data) + if expert_data_list: + df_expert_activation = pd.DataFrame(expert_data_list) + expert_chart = _create_expert_activation_chart(df_expert_activation) chart else: - mo.md("No expert activation data found to plot.") - - return chart, df_expert_activation + mo.md("No expert activation data foundto plot.") + expert_chart + return + + +@app.cell +def _(): + return if __name__ == "__main__": From b94f5c212a8e9470140fb724ed4111edf0304623 Mon Sep 17 00:00:00 2001 From: Tarun Kumar Vangani Date: Thu, 13 Nov 2025 13:42:12 +0800 Subject: [PATCH 21/25] feat: Add plot for average expert activation per layer Co-authored-by: aider (gemini/gemini-2.5-pro) --- moe_analyis.py | 55 ++++++++++++++++++++++++++++++++++++++++++++++++-- 1 file changed, 53 insertions(+), 2 deletions(-) diff --git a/moe_analyis.py b/moe_analyis.py index c86a5cc..90695a5 100644 --- a/moe_analyis.py +++ b/moe_analyis.py @@ -364,8 +364,59 @@ def _create_expert_activation_chart(df_expert_activation): @app.cell -def _(): - return +def _(alt, loaded_threshold_data, mo, pd): + """ + Processes layer-specific expert activation data and creates a plot showing + the average number of activated experts per layer, faceted by the dynamic + routing threshold. + """ + layer_activation_data = [] + + for threshold_str, data in loaded_threshold_data.items(): + try: + threshold = float(threshold_str) + report = data.get('expert_activation_report', {}) + + for key, value in report.items(): + if key.startswith('layer_'): + try: + layer_index = int(key.split('_')[1]) + activation = float(value) + layer_activation_data.append({ + 'threshold': threshold, + 'layer': layer_index, + 'average_experts': activation + }) + except (ValueError, IndexError): + print(f"Could not parse layer data for key '{key}' in threshold '{threshold_str}'") + continue + except (ValueError, TypeError) as e: + print(f"Could not process expert activation data for threshold '{threshold_str}': {e}") + continue + + df_layer_activation = None + layer_chart = None + if layer_activation_data: + df_layer_activation = pd.DataFrame(layer_activation_data) + + layer_chart = alt.Chart(df_layer_activation).mark_line(point=True).encode( + x=alt.X('layer:Q', title='Layer Index', axis=alt.Axis(tickMinStep=1)), + y=alt.Y('average_experts:Q', title='Average Activated Experts'), + color=alt.Color('threshold:N', title='Routing Threshold'), + tooltip=[ + alt.Tooltip('threshold', title='Threshold'), + alt.Tooltip('layer', title='Layer'), + alt.Tooltip('average_experts', title='Avg. Activated Experts', format='.2f') + ] + ).properties( + title='Average Expert Activation per Layer for Different Thresholds' + ).interactive() + + layer_chart + else: + mo.md("No layer-specific expert activation data found to plot.") + + return df_layer_activation, layer_chart if __name__ == "__main__": From 1a36b512719f93ebd23be70a72a409b2546c3dd0 Mon Sep 17 00:00:00 2001 From: Tarun Kumar Vangani Date: Thu, 13 Nov 2025 13:45:05 +0800 Subject: [PATCH 22/25] refactor: Reorient layer activation plot to threshold on X, layers as lines Co-authored-by: aider (gemini/gemini-2.5-pro) --- moe_analyis.py | 77 ++++++++++++++++++++++++++++---------------------- 1 file changed, 44 insertions(+), 33 deletions(-) diff --git a/moe_analyis.py b/moe_analyis.py index 90695a5..c445029 100644 --- a/moe_analyis.py +++ b/moe_analyis.py @@ -366,52 +366,63 @@ def _create_expert_activation_chart(df_expert_activation): @app.cell def _(alt, loaded_threshold_data, mo, pd): """ - Processes layer-specific expert activation data and creates a plot showing - the average number of activated experts per layer, faceted by the dynamic - routing threshold. + Processes layer-specific expert activation data and creates a plot of + average activated experts vs. threshold, with a line for each layer. """ - layer_activation_data = [] - for threshold_str, data in loaded_threshold_data.items(): - try: - threshold = float(threshold_str) - report = data.get('expert_activation_report', {}) - - for key, value in report.items(): - if key.startswith('layer_'): - try: - layer_index = int(key.split('_')[1]) - activation = float(value) - layer_activation_data.append({ - 'threshold': threshold, - 'layer': layer_index, - 'average_experts': activation - }) - except (ValueError, IndexError): - print(f"Could not parse layer data for key '{key}' in threshold '{threshold_str}'") - continue - except (ValueError, TypeError) as e: - print(f"Could not process expert activation data for threshold '{threshold_str}': {e}") - continue + def _extract_layer_activation_data(loaded_threshold_data): + """ + Extracts layer-specific expert activation data from loaded threshold data. + """ + layer_activation_data = [] + for threshold_str, data in loaded_threshold_data.items(): + try: + threshold = float(threshold_str) + report = data.get('expert_activation_report', {}) - df_layer_activation = None - layer_chart = None - if layer_activation_data: - df_layer_activation = pd.DataFrame(layer_activation_data) + for key, value in report.items(): + if key.startswith('layer_'): + try: + layer_index = int(key.split('_')[1]) + activation = float(value) + layer_activation_data.append({ + 'threshold': threshold, + 'layer': layer_index, + 'average_experts': activation + }) + except (ValueError, IndexError): + print(f"Could not parse layer data for key '{key}' in threshold '{threshold_str}'") + continue + except (ValueError, TypeError) as e: + print(f"Could not process expert activation data for threshold '{threshold_str}': {e}") + continue + return layer_activation_data - layer_chart = alt.Chart(df_layer_activation).mark_line(point=True).encode( - x=alt.X('layer:Q', title='Layer Index', axis=alt.Axis(tickMinStep=1)), + def _create_layer_activation_chart(df_layer_activation): + """ + Creates an Altair chart for layer-specific expert activation. + """ + chart = alt.Chart(df_layer_activation).mark_line(point=True).encode( + x=alt.X('threshold:Q', title='Dynamic Routing Threshold'), y=alt.Y('average_experts:Q', title='Average Activated Experts'), - color=alt.Color('threshold:N', title='Routing Threshold'), + color=alt.Color('layer:N', title='Layer'), tooltip=[ alt.Tooltip('threshold', title='Threshold'), alt.Tooltip('layer', title='Layer'), alt.Tooltip('average_experts', title='Avg. Activated Experts', format='.2f') ] ).properties( - title='Average Expert Activation per Layer for Different Thresholds' + title='Average Expert Activation vs. Threshold per Layer' ).interactive() + return chart + layer_data_list = _extract_layer_activation_data(loaded_threshold_data) + + df_layer_activation = None + layer_chart = None + if layer_data_list: + df_layer_activation = pd.DataFrame(layer_data_list) + layer_chart = _create_layer_activation_chart(df_layer_activation) layer_chart else: mo.md("No layer-specific expert activation data found to plot.") From 62980940ff997bfc841bba75cf1d256b23977b3c Mon Sep 17 00:00:00 2001 From: Tarun Kumar Vangani Date: Thu, 13 Nov 2025 13:50:01 +0800 Subject: [PATCH 23/25] Run analyis in loop with larger batch size --- run_dynamic_eval.sh | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/run_dynamic_eval.sh b/run_dynamic_eval.sh index e0104b8..eaa6608 100644 --- a/run_dynamic_eval.sh +++ b/run_dynamic_eval.sh @@ -11,7 +11,7 @@ # This setup allows us to apply the dynamic routing logic to the full, original model. # Define the thresholds to test -THRESHOLDS="0.05 0.1 0.15 0.2 0.25" +THRESHOLDS="0.3 0.35 0.4 0.45 0.5" # Loop over each threshold and run the evaluation for threshold in $THRESHOLDS @@ -24,8 +24,8 @@ do # Run the evaluation script python evaluation.py \ --tasks mmlu \ - --batch_size 8 \ - --limit 1 \ + --batch_size 12 \ + --limit 300 \ --use_pruned_model \ --pruned_metadata "prune_experts/super_experts_ids.json" \ --mode least \ From 9c1caeccdfae3c5634866d90ed047ea2210d2d77 Mon Sep 17 00:00:00 2001 From: Tarun Kumar Vangani Date: Thu, 13 Nov 2025 13:50:20 +0800 Subject: [PATCH 24/25] Visualizations for Dynamic MoE --- moe_analyis.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/moe_analyis.py b/moe_analyis.py index c445029..094f57b 100644 --- a/moe_analyis.py +++ b/moe_analyis.py @@ -426,8 +426,8 @@ def _create_layer_activation_chart(df_layer_activation): layer_chart else: mo.md("No layer-specific expert activation data found to plot.") - - return df_layer_activation, layer_chart + layer_chart + return if __name__ == "__main__": From 9b6976570b2411473ba3b567a595ec0f038b3ac6 Mon Sep 17 00:00:00 2001 From: Tarun Kumar Vangani Date: Thu, 13 Nov 2025 13:50:45 +0800 Subject: [PATCH 25/25] Add Results for Different Thresholds and Visualization --- moe_analysis_notebook.html | 396 +++++++++++++++++++++++++ results/results_dynamic_mmlu_0.05.json | 356 ++++++++++++++++++++++ results/results_dynamic_mmlu_0.1.json | 356 ++++++++++++++++++++++ results/results_dynamic_mmlu_0.15.json | 356 ++++++++++++++++++++++ results/results_dynamic_mmlu_0.2.json | 356 ++++++++++++++++++++++ results/results_dynamic_mmlu_0.25.json | 356 ++++++++++++++++++++++ results/results_dynamic_mmlu_0.3.json | 356 ++++++++++++++++++++++ results/results_dynamic_mmlu_0.35.json | 356 ++++++++++++++++++++++ results/results_dynamic_mmlu_0.4.json | 356 ++++++++++++++++++++++ 9 files changed, 3244 insertions(+) create mode 100644 moe_analysis_notebook.html create mode 100644 results/results_dynamic_mmlu_0.05.json create mode 100644 results/results_dynamic_mmlu_0.1.json create mode 100644 results/results_dynamic_mmlu_0.15.json create mode 100644 results/results_dynamic_mmlu_0.2.json create mode 100644 results/results_dynamic_mmlu_0.25.json create mode 100644 results/results_dynamic_mmlu_0.3.json create mode 100644 results/results_dynamic_mmlu_0.35.json create mode 100644 results/results_dynamic_mmlu_0.4.json diff --git a/moe_analysis_notebook.html b/moe_analysis_notebook.html new file mode 100644 index 0000000..adbf566 --- /dev/null +++ b/moe_analysis_notebook.html @@ -0,0 +1,396 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + moe analyis + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
+ +
+ + + + + + + diff --git a/results/results_dynamic_mmlu_0.05.json b/results/results_dynamic_mmlu_0.05.json new file mode 100644 index 0000000..3475ce4 --- /dev/null +++ b/results/results_dynamic_mmlu_0.05.json @@ -0,0 +1,356 @@ +{ + "config": { + "tasks": [ + "mmlu" + ], + "batch_size": 12, + "limit": 300, + "use_pruned_model": true, + "pruned_metadata": "prune_experts/super_experts_ids.json", + "mode": "least", + "pruning_method": "dynamic", + "k": 0, + "dynamic_routing_threshold": 0.05, + "device": "cuda", + "output_file": "results_dynamic_mmlu_0.05.json" + }, + "results": { + "mmlu": { + "acc,none": 0.5484048404840484, + "acc_stderr,none": 0.004576226983895873, + "alias": "mmlu" + }, + "mmlu_humanities": { + "acc,none": 0.5645796064400715, + "acc_stderr,none": 0.00889584517226482, + "alias": " - humanities" + }, + "mmlu_formal_logic": { + "alias": " - formal_logic", + "acc,none": 0.2698412698412698, + "acc_stderr,none": 0.0397015827323517 + }, + "mmlu_high_school_european_history": { + "alias": " - high_school_european_history", + "acc,none": 0.6848484848484848, + "acc_stderr,none": 0.03627730575022407 + }, + "mmlu_high_school_us_history": { + "alias": " - high_school_us_history", + "acc,none": 0.6862745098039216, + "acc_stderr,none": 0.03256685484460391 + }, + "mmlu_high_school_world_history": { + "alias": " - high_school_world_history", + "acc,none": 0.7046413502109705, + "acc_stderr,none": 0.029696338713422813 + }, + "mmlu_international_law": { + "alias": " - international_law", + "acc,none": 0.6942148760330579, + "acc_stderr,none": 0.04205953933884123 + }, + "mmlu_jurisprudence": { + "alias": " - jurisprudence", + "acc,none": 0.6574074074074074, + "acc_stderr,none": 0.04587904741301815 + }, + "mmlu_logical_fallacies": { + "alias": " - logical_fallacies", + "acc,none": 0.6257668711656442, + "acc_stderr,none": 0.03802068102899616 + }, + "mmlu_moral_disputes": { + "alias": " - moral_disputes", + "acc,none": 0.5866666666666667, + "acc_stderr,none": 0.028478055207315906 + }, + "mmlu_moral_scenarios": { + "alias": " - moral_scenarios", + "acc,none": 0.23333333333333334, + "acc_stderr,none": 0.024459979523511425 + }, + "mmlu_philosophy": { + "alias": " - philosophy", + "acc,none": 0.5833333333333334, + "acc_stderr,none": 0.02851131064391753 + }, + "mmlu_prehistory": { + "alias": " - prehistory", + "acc,none": 0.6466666666666666, + "acc_stderr,none": 0.02764374949090691 + }, + "mmlu_professional_law": { + "alias": " - professional_law", + "acc,none": 0.4166666666666667, + "acc_stderr,none": 0.028511310643917525 + }, + "mmlu_world_religions": { + "alias": " - world_religions", + "acc,none": 0.7426900584795322, + "acc_stderr,none": 0.03352799844161867 + }, + "mmlu_other": { + "acc,none": 0.5828877005347594, + "acc_stderr,none": 0.009405032896180549, + "alias": " - other" + }, + "mmlu_business_ethics": { + "alias": " - business_ethics", + "acc,none": 0.69, + "acc_stderr,none": 0.046482319871173176 + }, + "mmlu_clinical_knowledge": { + "alias": " - clinical_knowledge", + "acc,none": 0.5773584905660377, + "acc_stderr,none": 0.030402331445769502 + }, + "mmlu_college_medicine": { + "alias": " - college_medicine", + "acc,none": 0.5086705202312138, + "acc_stderr,none": 0.03811890988940413 + }, + "mmlu_global_facts": { + "alias": " - global_facts", + "acc,none": 0.39, + "acc_stderr,none": 0.04902071300001973 + }, + "mmlu_human_aging": { + "alias": " - human_aging", + "acc,none": 0.57847533632287, + "acc_stderr,none": 0.033141902221106585 + }, + "mmlu_management": { + "alias": " - management", + "acc,none": 0.6796116504854369, + "acc_stderr,none": 0.04620284082280041 + }, + "mmlu_marketing": { + "alias": " - marketing", + "acc,none": 0.8162393162393162, + "acc_stderr,none": 0.0253721396717229 + }, + "mmlu_medical_genetics": { + "alias": " - medical_genetics", + "acc,none": 0.61, + "acc_stderr,none": 0.04902071300001973 + }, + "mmlu_miscellaneous": { + "alias": " - miscellaneous", + "acc,none": 0.6866666666666666, + "acc_stderr,none": 0.02682505913963038 + }, + "mmlu_nutrition": { + "alias": " - nutrition", + "acc,none": 0.5766666666666667, + "acc_stderr,none": 0.028573804116352318 + }, + "mmlu_professional_accounting": { + "alias": " - professional_accounting", + "acc,none": 0.40425531914893614, + "acc_stderr,none": 0.029275532159704753 + }, + "mmlu_professional_medicine": { + "alias": " - professional_medicine", + "acc,none": 0.5514705882352942, + "acc_stderr,none": 0.030211479609121628 + }, + "mmlu_virology": { + "alias": " - virology", + "acc,none": 0.5, + "acc_stderr,none": 0.03892494720807614 + }, + "mmlu_social_sciences": { + "acc,none": 0.6205761316872428, + "acc_stderr,none": 0.00960184684075155, + "alias": " - social sciences" + }, + "mmlu_econometrics": { + "alias": " - econometrics", + "acc,none": 0.3333333333333333, + "acc_stderr,none": 0.04434600701584929 + }, + "mmlu_high_school_geography": { + "alias": " - high_school_geography", + "acc,none": 0.6818181818181818, + "acc_stderr,none": 0.033184773338453294 + }, + "mmlu_high_school_government_and_politics": { + "alias": " - high_school_government_and_politics", + "acc,none": 0.772020725388601, + "acc_stderr,none": 0.030276909945178256 + }, + "mmlu_high_school_macroeconomics": { + "alias": " - high_school_macroeconomics", + "acc,none": 0.49666666666666665, + "acc_stderr,none": 0.02891510401902553 + }, + "mmlu_high_school_microeconomics": { + "alias": " - high_school_microeconomics", + "acc,none": 0.5714285714285714, + "acc_stderr,none": 0.0321453685978864 + }, + "mmlu_high_school_psychology": { + "alias": " - high_school_psychology", + "acc,none": 0.6933333333333334, + "acc_stderr,none": 0.026666666666666665 + }, + "mmlu_human_sexuality": { + "alias": " - human_sexuality", + "acc,none": 0.5801526717557252, + "acc_stderr,none": 0.04328577215262974 + }, + "mmlu_professional_psychology": { + "alias": " - professional_psychology", + "acc,none": 0.54, + "acc_stderr,none": 0.02882306768491568 + }, + "mmlu_public_relations": { + "alias": " - public_relations", + "acc,none": 0.6363636363636364, + "acc_stderr,none": 0.04607582090719978 + }, + "mmlu_security_studies": { + "alias": " - security_studies", + "acc,none": 0.6244897959183674, + "acc_stderr,none": 0.031001209039894864 + }, + "mmlu_sociology": { + "alias": " - sociology", + "acc,none": 0.7562189054726368, + "acc_stderr,none": 0.03036049015401464 + }, + "mmlu_us_foreign_policy": { + "alias": " - us_foreign_policy", + "acc,none": 0.8, + "acc_stderr,none": 0.04020151261036849 + }, + "mmlu_stem": { + "acc,none": 0.4469820554649266, + "acc_stderr,none": 0.00877226790621949, + "alias": " - stem" + }, + "mmlu_abstract_algebra": { + "alias": " - abstract_algebra", + "acc,none": 0.33, + "acc_stderr,none": 0.04725815626252609 + }, + "mmlu_anatomy": { + "alias": " - anatomy", + "acc,none": 0.5037037037037037, + "acc_stderr,none": 0.04319223625811333 + }, + "mmlu_astronomy": { + "alias": " - astronomy", + "acc,none": 0.5526315789473685, + "acc_stderr,none": 0.040463368839782535 + }, + "mmlu_college_biology": { + "alias": " - college_biology", + "acc,none": 0.5902777777777778, + "acc_stderr,none": 0.04112490974670787 + }, + "mmlu_college_chemistry": { + "alias": " - college_chemistry", + "acc,none": 0.35, + "acc_stderr,none": 0.04793724854411023 + }, + "mmlu_college_computer_science": { + "alias": " - college_computer_science", + "acc,none": 0.39, + "acc_stderr,none": 0.04902071300001973 + }, + "mmlu_college_mathematics": { + "alias": " - college_mathematics", + "acc,none": 0.34, + "acc_stderr,none": 0.04760952285695233 + }, + "mmlu_college_physics": { + "alias": " - college_physics", + "acc,none": 0.43137254901960786, + "acc_stderr,none": 0.049280995972875316 + }, + "mmlu_computer_security": { + "alias": " - computer_security", + "acc,none": 0.75, + "acc_stderr,none": 0.04351941398892446 + }, + "mmlu_conceptual_physics": { + "alias": " - conceptual_physics", + "acc,none": 0.4340425531914894, + "acc_stderr,none": 0.03240038086792751 + }, + "mmlu_electrical_engineering": { + "alias": " - electrical_engineering", + "acc,none": 0.4827586206896552, + "acc_stderr,none": 0.04164188720169378 + }, + "mmlu_elementary_mathematics": { + "alias": " - elementary_mathematics", + "acc,none": 0.38666666666666666, + "acc_stderr,none": 0.028163138908196883 + }, + "mmlu_high_school_biology": { + "alias": " - high_school_biology", + "acc,none": 0.5966666666666667, + "acc_stderr,none": 0.028370197016959926 + }, + "mmlu_high_school_chemistry": { + "alias": " - high_school_chemistry", + "acc,none": 0.46798029556650245, + "acc_stderr,none": 0.03510766597959214 + }, + "mmlu_high_school_computer_science": { + "alias": " - high_school_computer_science", + "acc,none": 0.62, + "acc_stderr,none": 0.04878317312145634 + }, + "mmlu_high_school_mathematics": { + "alias": " - high_school_mathematics", + "acc,none": 0.32222222222222224, + "acc_stderr,none": 0.028493465091028538 + }, + "mmlu_high_school_physics": { + "alias": " - high_school_physics", + "acc,none": 0.3509933774834437, + "acc_stderr,none": 0.038969819642573705 + }, + "mmlu_high_school_statistics": { + "alias": " - high_school_statistics", + "acc,none": 0.3194444444444444, + "acc_stderr,none": 0.03179876342176851 + }, + "mmlu_machine_learning": { + "alias": " - machine_learning", + "acc,none": 0.35714285714285715, + "acc_stderr,none": 0.045479609997643805 + } + }, + "expert_activation_report": { + "layer_0": "1.17", + "layer_1": "1.07", + "layer_2": "1.06", + "layer_3": "1.04", + "layer_4": "1.04", + "layer_5": "1.04", + "layer_6": "1.03", + "layer_7": "1.02", + "layer_8": "1.02", + "layer_9": "1.02", + "layer_10": "1.02", + "layer_11": "1.02", + "layer_12": "1.01", + "layer_13": "1.00", + "layer_14": "1.00", + "layer_15": "1.01", + "layer_16": "1.00", + "layer_17": "1.01", + "layer_18": "1.01", + "layer_19": "1.01", + "layer_20": "1.02", + "layer_21": "1.04", + "layer_22": "1.03", + "layer_23": "1.07", + "overall_average": "1.03" + } +} \ No newline at end of file diff --git a/results/results_dynamic_mmlu_0.1.json b/results/results_dynamic_mmlu_0.1.json new file mode 100644 index 0000000..6060135 --- /dev/null +++ b/results/results_dynamic_mmlu_0.1.json @@ -0,0 +1,356 @@ +{ + "config": { + "tasks": [ + "mmlu" + ], + "batch_size": 12, + "limit": 300, + "use_pruned_model": true, + "pruned_metadata": "prune_experts/super_experts_ids.json", + "mode": "least", + "pruning_method": "dynamic", + "k": 0, + "dynamic_routing_threshold": 0.1, + "device": "cuda", + "output_file": "results_dynamic_mmlu_0.1.json" + }, + "results": { + "mmlu": { + "acc,none": 0.5511551155115512, + "acc_stderr,none": 0.004566236928462437, + "alias": "mmlu" + }, + "mmlu_humanities": { + "acc,none": 0.5713774597495528, + "acc_stderr,none": 0.00886657028410953, + "alias": " - humanities" + }, + "mmlu_formal_logic": { + "alias": " - formal_logic", + "acc,none": 0.2777777777777778, + "acc_stderr,none": 0.04006168083848877 + }, + "mmlu_high_school_european_history": { + "alias": " - high_school_european_history", + "acc,none": 0.6666666666666666, + "acc_stderr,none": 0.036810508691615514 + }, + "mmlu_high_school_us_history": { + "alias": " - high_school_us_history", + "acc,none": 0.7401960784313726, + "acc_stderr,none": 0.030778554678693247 + }, + "mmlu_high_school_world_history": { + "alias": " - high_school_world_history", + "acc,none": 0.7257383966244726, + "acc_stderr,none": 0.029041333510597976 + }, + "mmlu_international_law": { + "alias": " - international_law", + "acc,none": 0.7024793388429752, + "acc_stderr,none": 0.04173349148083503 + }, + "mmlu_jurisprudence": { + "alias": " - jurisprudence", + "acc,none": 0.6203703703703703, + "acc_stderr,none": 0.04691521224077746 + }, + "mmlu_logical_fallacies": { + "alias": " - logical_fallacies", + "acc,none": 0.6441717791411042, + "acc_stderr,none": 0.03761521380046734 + }, + "mmlu_moral_disputes": { + "alias": " - moral_disputes", + "acc,none": 0.5866666666666667, + "acc_stderr,none": 0.028478055207315906 + }, + "mmlu_moral_scenarios": { + "alias": " - moral_scenarios", + "acc,none": 0.25666666666666665, + "acc_stderr,none": 0.02526044198731046 + }, + "mmlu_philosophy": { + "alias": " - philosophy", + "acc,none": 0.6033333333333334, + "acc_stderr,none": 0.02829149642514491 + }, + "mmlu_prehistory": { + "alias": " - prehistory", + "acc,none": 0.6233333333333333, + "acc_stderr,none": 0.02802226115126559 + }, + "mmlu_professional_law": { + "alias": " - professional_law", + "acc,none": 0.4, + "acc_stderr,none": 0.028331529878993136 + }, + "mmlu_world_religions": { + "alias": " - world_religions", + "acc,none": 0.7660818713450293, + "acc_stderr,none": 0.032467217651178305 + }, + "mmlu_other": { + "acc,none": 0.5889992360580596, + "acc_stderr,none": 0.009363828692678213, + "alias": " - other" + }, + "mmlu_business_ethics": { + "alias": " - business_ethics", + "acc,none": 0.71, + "acc_stderr,none": 0.045604802157206865 + }, + "mmlu_clinical_knowledge": { + "alias": " - clinical_knowledge", + "acc,none": 0.6037735849056604, + "acc_stderr,none": 0.030102793781791155 + }, + "mmlu_college_medicine": { + "alias": " - college_medicine", + "acc,none": 0.5317919075144508, + "acc_stderr,none": 0.03804749744364767 + }, + "mmlu_global_facts": { + "alias": " - global_facts", + "acc,none": 0.38, + "acc_stderr,none": 0.04878317312145634 + }, + "mmlu_human_aging": { + "alias": " - human_aging", + "acc,none": 0.5829596412556054, + "acc_stderr,none": 0.03309266936071724 + }, + "mmlu_management": { + "alias": " - management", + "acc,none": 0.7184466019417476, + "acc_stderr,none": 0.044532548363264673 + }, + "mmlu_marketing": { + "alias": " - marketing", + "acc,none": 0.7948717948717948, + "acc_stderr,none": 0.02645350805404033 + }, + "mmlu_medical_genetics": { + "alias": " - medical_genetics", + "acc,none": 0.59, + "acc_stderr,none": 0.04943110704237104 + }, + "mmlu_miscellaneous": { + "alias": " - miscellaneous", + "acc,none": 0.7233333333333334, + "acc_stderr,none": 0.025870931391123536 + }, + "mmlu_nutrition": { + "alias": " - nutrition", + "acc,none": 0.5933333333333334, + "acc_stderr,none": 0.02840750341836662 + }, + "mmlu_professional_accounting": { + "alias": " - professional_accounting", + "acc,none": 0.4078014184397163, + "acc_stderr,none": 0.02931601177634362 + }, + "mmlu_professional_medicine": { + "alias": " - professional_medicine", + "acc,none": 0.5147058823529411, + "acc_stderr,none": 0.030359697079046163 + }, + "mmlu_virology": { + "alias": " - virology", + "acc,none": 0.4939759036144578, + "acc_stderr,none": 0.03892212195333041 + }, + "mmlu_social_sciences": { + "acc,none": 0.6230452674897119, + "acc_stderr,none": 0.00961018670984974, + "alias": " - social sciences" + }, + "mmlu_econometrics": { + "alias": " - econometrics", + "acc,none": 0.30701754385964913, + "acc_stderr,none": 0.04339138322579864 + }, + "mmlu_high_school_geography": { + "alias": " - high_school_geography", + "acc,none": 0.6717171717171717, + "acc_stderr,none": 0.03345678422756777 + }, + "mmlu_high_school_government_and_politics": { + "alias": " - high_school_government_and_politics", + "acc,none": 0.7668393782383419, + "acc_stderr,none": 0.03051611137147603 + }, + "mmlu_high_school_macroeconomics": { + "alias": " - high_school_macroeconomics", + "acc,none": 0.5366666666666666, + "acc_stderr,none": 0.028837890554337213 + }, + "mmlu_high_school_microeconomics": { + "alias": " - high_school_microeconomics", + "acc,none": 0.5588235294117647, + "acc_stderr,none": 0.03225294232399634 + }, + "mmlu_high_school_psychology": { + "alias": " - high_school_psychology", + "acc,none": 0.6833333333333333, + "acc_stderr,none": 0.026901833738451123 + }, + "mmlu_human_sexuality": { + "alias": " - human_sexuality", + "acc,none": 0.6335877862595419, + "acc_stderr,none": 0.0422587545196964 + }, + "mmlu_professional_psychology": { + "alias": " - professional_psychology", + "acc,none": 0.55, + "acc_stderr,none": 0.028770804599878932 + }, + "mmlu_public_relations": { + "alias": " - public_relations", + "acc,none": 0.6090909090909091, + "acc_stderr,none": 0.046737523336702363 + }, + "mmlu_security_studies": { + "alias": " - security_studies", + "acc,none": 0.6326530612244898, + "acc_stderr,none": 0.030862144921087586 + }, + "mmlu_sociology": { + "alias": " - sociology", + "acc,none": 0.7313432835820896, + "acc_stderr,none": 0.03134328358208955 + }, + "mmlu_us_foreign_policy": { + "alias": " - us_foreign_policy", + "acc,none": 0.82, + "acc_stderr,none": 0.03861229196653691 + }, + "mmlu_stem": { + "acc,none": 0.4433931484502447, + "acc_stderr,none": 0.008757392582262784, + "alias": " - stem" + }, + "mmlu_abstract_algebra": { + "alias": " - abstract_algebra", + "acc,none": 0.39, + "acc_stderr,none": 0.04902071300001973 + }, + "mmlu_anatomy": { + "alias": " - anatomy", + "acc,none": 0.5111111111111111, + "acc_stderr,none": 0.043182754919779784 + }, + "mmlu_astronomy": { + "alias": " - astronomy", + "acc,none": 0.5723684210526315, + "acc_stderr,none": 0.04026097083296561 + }, + "mmlu_college_biology": { + "alias": " - college_biology", + "acc,none": 0.5972222222222222, + "acc_stderr,none": 0.041014055198424236 + }, + "mmlu_college_chemistry": { + "alias": " - college_chemistry", + "acc,none": 0.3, + "acc_stderr,none": 0.04605661864718382 + }, + "mmlu_college_computer_science": { + "alias": " - college_computer_science", + "acc,none": 0.37, + "acc_stderr,none": 0.048523658709390974 + }, + "mmlu_college_mathematics": { + "alias": " - college_mathematics", + "acc,none": 0.39, + "acc_stderr,none": 0.04902071300001973 + }, + "mmlu_college_physics": { + "alias": " - college_physics", + "acc,none": 0.38235294117647056, + "acc_stderr,none": 0.048355036961072254 + }, + "mmlu_computer_security": { + "alias": " - computer_security", + "acc,none": 0.72, + "acc_stderr,none": 0.045126085985421296 + }, + "mmlu_conceptual_physics": { + "alias": " - conceptual_physics", + "acc,none": 0.4425531914893617, + "acc_stderr,none": 0.03246956919789957 + }, + "mmlu_electrical_engineering": { + "alias": " - electrical_engineering", + "acc,none": 0.4827586206896552, + "acc_stderr,none": 0.04164188720169378 + }, + "mmlu_elementary_mathematics": { + "alias": " - elementary_mathematics", + "acc,none": 0.37666666666666665, + "acc_stderr,none": 0.02802226115126559 + }, + "mmlu_high_school_biology": { + "alias": " - high_school_biology", + "acc,none": 0.5966666666666667, + "acc_stderr,none": 0.028370197016959926 + }, + "mmlu_high_school_chemistry": { + "alias": " - high_school_chemistry", + "acc,none": 0.47783251231527096, + "acc_stderr,none": 0.03514528562175008 + }, + "mmlu_high_school_computer_science": { + "alias": " - high_school_computer_science", + "acc,none": 0.59, + "acc_stderr,none": 0.04943110704237104 + }, + "mmlu_high_school_mathematics": { + "alias": " - high_school_mathematics", + "acc,none": 0.3148148148148148, + "acc_stderr,none": 0.02831753349606653 + }, + "mmlu_high_school_physics": { + "alias": " - high_school_physics", + "acc,none": 0.33774834437086093, + "acc_stderr,none": 0.038615575462551656 + }, + "mmlu_high_school_statistics": { + "alias": " - high_school_statistics", + "acc,none": 0.3055555555555556, + "acc_stderr,none": 0.0314155462940254 + }, + "mmlu_machine_learning": { + "alias": " - machine_learning", + "acc,none": 0.33035714285714285, + "acc_stderr,none": 0.04464285714285714 + } + }, + "expert_activation_report": { + "layer_0": "1.91", + "layer_1": "1.80", + "layer_2": "1.66", + "layer_3": "1.41", + "layer_4": "1.35", + "layer_5": "1.33", + "layer_6": "1.30", + "layer_7": "1.16", + "layer_8": "1.17", + "layer_9": "1.16", + "layer_10": "1.15", + "layer_11": "1.12", + "layer_12": "1.09", + "layer_13": "1.07", + "layer_14": "1.09", + "layer_15": "1.11", + "layer_16": "1.10", + "layer_17": "1.15", + "layer_18": "1.16", + "layer_19": "1.25", + "layer_20": "1.27", + "layer_21": "1.35", + "layer_22": "1.34", + "layer_23": "1.45", + "overall_average": "1.29" + } +} \ No newline at end of file diff --git a/results/results_dynamic_mmlu_0.15.json b/results/results_dynamic_mmlu_0.15.json new file mode 100644 index 0000000..30544ea --- /dev/null +++ b/results/results_dynamic_mmlu_0.15.json @@ -0,0 +1,356 @@ +{ + "config": { + "tasks": [ + "mmlu" + ], + "batch_size": 12, + "limit": 300, + "use_pruned_model": true, + "pruned_metadata": "prune_experts/super_experts_ids.json", + "mode": "least", + "pruning_method": "dynamic", + "k": 0, + "dynamic_routing_threshold": 0.15, + "device": "cuda", + "output_file": "results_dynamic_mmlu_0.15.json" + }, + "results": { + "mmlu": { + "acc,none": 0.571965529886322, + "acc_stderr,none": 0.004542259782918584, + "alias": "mmlu" + }, + "mmlu_humanities": { + "acc,none": 0.5889087656529517, + "acc_stderr,none": 0.00881548824744021, + "alias": " - humanities" + }, + "mmlu_formal_logic": { + "alias": " - formal_logic", + "acc,none": 0.30158730158730157, + "acc_stderr,none": 0.04104947269903394 + }, + "mmlu_high_school_european_history": { + "alias": " - high_school_european_history", + "acc,none": 0.6909090909090909, + "acc_stderr,none": 0.03608541011573964 + }, + "mmlu_high_school_us_history": { + "alias": " - high_school_us_history", + "acc,none": 0.7303921568627451, + "acc_stderr,none": 0.031145570659486796 + }, + "mmlu_high_school_world_history": { + "alias": " - high_school_world_history", + "acc,none": 0.7257383966244726, + "acc_stderr,none": 0.029041333510597976 + }, + "mmlu_international_law": { + "alias": " - international_law", + "acc,none": 0.7520661157024794, + "acc_stderr,none": 0.039418975265163046 + }, + "mmlu_jurisprudence": { + "alias": " - jurisprudence", + "acc,none": 0.6574074074074074, + "acc_stderr,none": 0.04587904741301815 + }, + "mmlu_logical_fallacies": { + "alias": " - logical_fallacies", + "acc,none": 0.6687116564417178, + "acc_stderr,none": 0.03697983910025588 + }, + "mmlu_moral_disputes": { + "alias": " - moral_disputes", + "acc,none": 0.6066666666666667, + "acc_stderr,none": 0.028250090846760858 + }, + "mmlu_moral_scenarios": { + "alias": " - moral_scenarios", + "acc,none": 0.25666666666666665, + "acc_stderr,none": 0.02526044198731046 + }, + "mmlu_philosophy": { + "alias": " - philosophy", + "acc,none": 0.6333333333333333, + "acc_stderr,none": 0.02786867328338387 + }, + "mmlu_prehistory": { + "alias": " - prehistory", + "acc,none": 0.6533333333333333, + "acc_stderr,none": 0.027522498482247464 + }, + "mmlu_professional_law": { + "alias": " - professional_law", + "acc,none": 0.43, + "acc_stderr,none": 0.02863096997084751 + }, + "mmlu_world_religions": { + "alias": " - world_religions", + "acc,none": 0.7485380116959064, + "acc_stderr,none": 0.03327504423846844 + }, + "mmlu_other": { + "acc,none": 0.5985485103132162, + "acc_stderr,none": 0.00932653488572079, + "alias": " - other" + }, + "mmlu_business_ethics": { + "alias": " - business_ethics", + "acc,none": 0.69, + "acc_stderr,none": 0.046482319871173176 + }, + "mmlu_clinical_knowledge": { + "alias": " - clinical_knowledge", + "acc,none": 0.5924528301886792, + "acc_stderr,none": 0.03024223380085444 + }, + "mmlu_college_medicine": { + "alias": " - college_medicine", + "acc,none": 0.5433526011560693, + "acc_stderr,none": 0.03798106566014504 + }, + "mmlu_global_facts": { + "alias": " - global_facts", + "acc,none": 0.39, + "acc_stderr,none": 0.04902071300001973 + }, + "mmlu_human_aging": { + "alias": " - human_aging", + "acc,none": 0.5964125560538116, + "acc_stderr,none": 0.03292802819330314 + }, + "mmlu_management": { + "alias": " - management", + "acc,none": 0.7766990291262136, + "acc_stderr,none": 0.041235531898914324 + }, + "mmlu_marketing": { + "alias": " - marketing", + "acc,none": 0.8205128205128205, + "acc_stderr,none": 0.025140935950335442 + }, + "mmlu_medical_genetics": { + "alias": " - medical_genetics", + "acc,none": 0.64, + "acc_stderr,none": 0.048241815132442176 + }, + "mmlu_miscellaneous": { + "alias": " - miscellaneous", + "acc,none": 0.7033333333333334, + "acc_stderr,none": 0.026416749751055384 + }, + "mmlu_nutrition": { + "alias": " - nutrition", + "acc,none": 0.6133333333333333, + "acc_stderr,none": 0.028163138908196883 + }, + "mmlu_professional_accounting": { + "alias": " - professional_accounting", + "acc,none": 0.42907801418439717, + "acc_stderr,none": 0.029525914302558586 + }, + "mmlu_professional_medicine": { + "alias": " - professional_medicine", + "acc,none": 0.5257352941176471, + "acc_stderr,none": 0.030332578094555054 + }, + "mmlu_virology": { + "alias": " - virology", + "acc,none": 0.4819277108433735, + "acc_stderr,none": 0.03889951252827222 + }, + "mmlu_social_sciences": { + "acc,none": 0.648559670781893, + "acc_stderr,none": 0.009466441844659251, + "alias": " - social sciences" + }, + "mmlu_econometrics": { + "alias": " - econometrics", + "acc,none": 0.34210526315789475, + "acc_stderr,none": 0.04462917535336936 + }, + "mmlu_high_school_geography": { + "alias": " - high_school_geography", + "acc,none": 0.7424242424242424, + "acc_stderr,none": 0.031156269519646826 + }, + "mmlu_high_school_government_and_politics": { + "alias": " - high_school_government_and_politics", + "acc,none": 0.7512953367875648, + "acc_stderr,none": 0.03119584087770025 + }, + "mmlu_high_school_macroeconomics": { + "alias": " - high_school_macroeconomics", + "acc,none": 0.5366666666666666, + "acc_stderr,none": 0.028837890554337213 + }, + "mmlu_high_school_microeconomics": { + "alias": " - high_school_microeconomics", + "acc,none": 0.5840336134453782, + "acc_stderr,none": 0.03201650100739608 + }, + "mmlu_high_school_psychology": { + "alias": " - high_school_psychology", + "acc,none": 0.7233333333333334, + "acc_stderr,none": 0.025870931391123536 + }, + "mmlu_human_sexuality": { + "alias": " - human_sexuality", + "acc,none": 0.6564885496183206, + "acc_stderr,none": 0.041649760719448814 + }, + "mmlu_professional_psychology": { + "alias": " - professional_psychology", + "acc,none": 0.57, + "acc_stderr,none": 0.02863096997084751 + }, + "mmlu_public_relations": { + "alias": " - public_relations", + "acc,none": 0.6272727272727273, + "acc_stderr,none": 0.04631381319425461 + }, + "mmlu_security_studies": { + "alias": " - security_studies", + "acc,none": 0.710204081632653, + "acc_stderr,none": 0.029043088683304373 + }, + "mmlu_sociology": { + "alias": " - sociology", + "acc,none": 0.7562189054726368, + "acc_stderr,none": 0.03036049015401464 + }, + "mmlu_us_foreign_policy": { + "alias": " - us_foreign_policy", + "acc,none": 0.76, + "acc_stderr,none": 0.04292346959909278 + }, + "mmlu_stem": { + "acc,none": 0.4730831973898858, + "acc_stderr,none": 0.008769855322103346, + "alias": " - stem" + }, + "mmlu_abstract_algebra": { + "alias": " - abstract_algebra", + "acc,none": 0.35, + "acc_stderr,none": 0.04793724854411023 + }, + "mmlu_anatomy": { + "alias": " - anatomy", + "acc,none": 0.5185185185185185, + "acc_stderr,none": 0.043163785995113245 + }, + "mmlu_astronomy": { + "alias": " - astronomy", + "acc,none": 0.6118421052631579, + "acc_stderr,none": 0.03965842097512745 + }, + "mmlu_college_biology": { + "alias": " - college_biology", + "acc,none": 0.6180555555555556, + "acc_stderr,none": 0.04062990784146671 + }, + "mmlu_college_chemistry": { + "alias": " - college_chemistry", + "acc,none": 0.33, + "acc_stderr,none": 0.04725815626252609 + }, + "mmlu_college_computer_science": { + "alias": " - college_computer_science", + "acc,none": 0.41, + "acc_stderr,none": 0.04943110704237104 + }, + "mmlu_college_mathematics": { + "alias": " - college_mathematics", + "acc,none": 0.37, + "acc_stderr,none": 0.048523658709390974 + }, + "mmlu_college_physics": { + "alias": " - college_physics", + "acc,none": 0.39215686274509803, + "acc_stderr,none": 0.04858083574266346 + }, + "mmlu_computer_security": { + "alias": " - computer_security", + "acc,none": 0.72, + "acc_stderr,none": 0.045126085985421296 + }, + "mmlu_conceptual_physics": { + "alias": " - conceptual_physics", + "acc,none": 0.4851063829787234, + "acc_stderr,none": 0.03267151848924783 + }, + "mmlu_electrical_engineering": { + "alias": " - electrical_engineering", + "acc,none": 0.5379310344827586, + "acc_stderr,none": 0.041546596717075446 + }, + "mmlu_elementary_mathematics": { + "alias": " - elementary_mathematics", + "acc,none": 0.36333333333333334, + "acc_stderr,none": 0.027814616968981937 + }, + "mmlu_high_school_biology": { + "alias": " - high_school_biology", + "acc,none": 0.66, + "acc_stderr,none": 0.027395285584216902 + }, + "mmlu_high_school_chemistry": { + "alias": " - high_school_chemistry", + "acc,none": 0.5024630541871922, + "acc_stderr,none": 0.035179450386910595 + }, + "mmlu_high_school_computer_science": { + "alias": " - high_school_computer_science", + "acc,none": 0.67, + "acc_stderr,none": 0.04725815626252609 + }, + "mmlu_high_school_mathematics": { + "alias": " - high_school_mathematics", + "acc,none": 0.34814814814814815, + "acc_stderr,none": 0.02904560029061618 + }, + "mmlu_high_school_physics": { + "alias": " - high_school_physics", + "acc,none": 0.3708609271523179, + "acc_stderr,none": 0.039439666991836285 + }, + "mmlu_high_school_statistics": { + "alias": " - high_school_statistics", + "acc,none": 0.37962962962962965, + "acc_stderr,none": 0.03309682581119039 + }, + "mmlu_machine_learning": { + "alias": " - machine_learning", + "acc,none": 0.35714285714285715, + "acc_stderr,none": 0.045479609997643805 + } + }, + "expert_activation_report": { + "layer_0": "2.72", + "layer_1": "2.68", + "layer_2": "2.45", + "layer_3": "2.14", + "layer_4": "1.95", + "layer_5": "1.91", + "layer_6": "1.89", + "layer_7": "1.56", + "layer_8": "1.58", + "layer_9": "1.53", + "layer_10": "1.57", + "layer_11": "1.47", + "layer_12": "1.44", + "layer_13": "1.42", + "layer_14": "1.42", + "layer_15": "1.39", + "layer_16": "1.35", + "layer_17": "1.49", + "layer_18": "1.47", + "layer_19": "1.66", + "layer_20": "1.71", + "layer_21": "1.85", + "layer_22": "1.83", + "layer_23": "2.02", + "overall_average": "1.77" + } +} \ No newline at end of file diff --git a/results/results_dynamic_mmlu_0.2.json b/results/results_dynamic_mmlu_0.2.json new file mode 100644 index 0000000..12be653 --- /dev/null +++ b/results/results_dynamic_mmlu_0.2.json @@ -0,0 +1,356 @@ +{ + "config": { + "tasks": [ + "mmlu" + ], + "batch_size": 12, + "limit": 300, + "use_pruned_model": true, + "pruned_metadata": "prune_experts/super_experts_ids.json", + "mode": "least", + "pruning_method": "dynamic", + "k": 0, + "dynamic_routing_threshold": 0.2, + "device": "cuda", + "output_file": "results_dynamic_mmlu_0.2.json" + }, + "results": { + "mmlu": { + "acc,none": 0.5951595159515951, + "acc_stderr,none": 0.004494672096858826, + "alias": "mmlu" + }, + "mmlu_humanities": { + "acc,none": 0.6110912343470483, + "acc_stderr,none": 0.008756741372506894, + "alias": " - humanities" + }, + "mmlu_formal_logic": { + "alias": " - formal_logic", + "acc,none": 0.3253968253968254, + "acc_stderr,none": 0.04190596438871138 + }, + "mmlu_high_school_european_history": { + "alias": " - high_school_european_history", + "acc,none": 0.6787878787878788, + "acc_stderr,none": 0.03646204963253815 + }, + "mmlu_high_school_us_history": { + "alias": " - high_school_us_history", + "acc,none": 0.7303921568627451, + "acc_stderr,none": 0.031145570659486796 + }, + "mmlu_high_school_world_history": { + "alias": " - high_school_world_history", + "acc,none": 0.7679324894514767, + "acc_stderr,none": 0.02747974455080847 + }, + "mmlu_international_law": { + "alias": " - international_law", + "acc,none": 0.7107438016528925, + "acc_stderr,none": 0.0413911272763546 + }, + "mmlu_jurisprudence": { + "alias": " - jurisprudence", + "acc,none": 0.6944444444444444, + "acc_stderr,none": 0.044531975073749855 + }, + "mmlu_logical_fallacies": { + "alias": " - logical_fallacies", + "acc,none": 0.7116564417177914, + "acc_stderr,none": 0.03559039531617345 + }, + "mmlu_moral_disputes": { + "alias": " - moral_disputes", + "acc,none": 0.63, + "acc_stderr,none": 0.027921294063982024 + }, + "mmlu_moral_scenarios": { + "alias": " - moral_scenarios", + "acc,none": 0.2866666666666667, + "acc_stderr,none": 0.02615166012679861 + }, + "mmlu_philosophy": { + "alias": " - philosophy", + "acc,none": 0.6466666666666666, + "acc_stderr,none": 0.02764374949090691 + }, + "mmlu_prehistory": { + "alias": " - prehistory", + "acc,none": 0.6733333333333333, + "acc_stderr,none": 0.027122634635122614 + }, + "mmlu_professional_law": { + "alias": " - professional_law", + "acc,none": 0.47, + "acc_stderr,none": 0.028863651326417043 + }, + "mmlu_world_religions": { + "alias": " - world_religions", + "acc,none": 0.7894736842105263, + "acc_stderr,none": 0.03126781714663182 + }, + "mmlu_other": { + "acc,none": 0.6283422459893048, + "acc_stderr,none": 0.00919395126918091, + "alias": " - other" + }, + "mmlu_business_ethics": { + "alias": " - business_ethics", + "acc,none": 0.73, + "acc_stderr,none": 0.04461960433384737 + }, + "mmlu_clinical_knowledge": { + "alias": " - clinical_knowledge", + "acc,none": 0.6188679245283019, + "acc_stderr,none": 0.02989060968628662 + }, + "mmlu_college_medicine": { + "alias": " - college_medicine", + "acc,none": 0.5664739884393064, + "acc_stderr,none": 0.03778621079092051 + }, + "mmlu_global_facts": { + "alias": " - global_facts", + "acc,none": 0.41, + "acc_stderr,none": 0.04943110704237104 + }, + "mmlu_human_aging": { + "alias": " - human_aging", + "acc,none": 0.6278026905829597, + "acc_stderr,none": 0.03244305283008735 + }, + "mmlu_management": { + "alias": " - management", + "acc,none": 0.7669902912621359, + "acc_stderr,none": 0.041858325989283136 + }, + "mmlu_marketing": { + "alias": " - marketing", + "acc,none": 0.8205128205128205, + "acc_stderr,none": 0.025140935950335442 + }, + "mmlu_medical_genetics": { + "alias": " - medical_genetics", + "acc,none": 0.63, + "acc_stderr,none": 0.048523658709390974 + }, + "mmlu_miscellaneous": { + "alias": " - miscellaneous", + "acc,none": 0.76, + "acc_stderr,none": 0.02469885513168684 + }, + "mmlu_nutrition": { + "alias": " - nutrition", + "acc,none": 0.66, + "acc_stderr,none": 0.027395285584216902 + }, + "mmlu_professional_accounting": { + "alias": " - professional_accounting", + "acc,none": 0.4432624113475177, + "acc_stderr,none": 0.02963483847376595 + }, + "mmlu_professional_medicine": { + "alias": " - professional_medicine", + "acc,none": 0.5808823529411765, + "acc_stderr,none": 0.02997280717046463 + }, + "mmlu_virology": { + "alias": " - virology", + "acc,none": 0.5180722891566265, + "acc_stderr,none": 0.03889951252827222 + }, + "mmlu_social_sciences": { + "acc,none": 0.6757201646090535, + "acc_stderr,none": 0.009218033232235717, + "alias": " - social sciences" + }, + "mmlu_econometrics": { + "alias": " - econometrics", + "acc,none": 0.2894736842105263, + "acc_stderr,none": 0.04266339443159392 + }, + "mmlu_high_school_geography": { + "alias": " - high_school_geography", + "acc,none": 0.7626262626262627, + "acc_stderr,none": 0.030313710538198924 + }, + "mmlu_high_school_government_and_politics": { + "alias": " - high_school_government_and_politics", + "acc,none": 0.8134715025906736, + "acc_stderr,none": 0.0281120912101175 + }, + "mmlu_high_school_macroeconomics": { + "alias": " - high_school_macroeconomics", + "acc,none": 0.5833333333333334, + "acc_stderr,none": 0.02851131064391753 + }, + "mmlu_high_school_microeconomics": { + "alias": " - high_school_microeconomics", + "acc,none": 0.6134453781512605, + "acc_stderr,none": 0.031631458075523776 + }, + "mmlu_high_school_psychology": { + "alias": " - high_school_psychology", + "acc,none": 0.74, + "acc_stderr,none": 0.025366873297069253 + }, + "mmlu_human_sexuality": { + "alias": " - human_sexuality", + "acc,none": 0.732824427480916, + "acc_stderr,none": 0.03880848301082397 + }, + "mmlu_professional_psychology": { + "alias": " - professional_psychology", + "acc,none": 0.5966666666666667, + "acc_stderr,none": 0.028370197016959926 + }, + "mmlu_public_relations": { + "alias": " - public_relations", + "acc,none": 0.6636363636363637, + "acc_stderr,none": 0.04525393596302509 + }, + "mmlu_security_studies": { + "alias": " - security_studies", + "acc,none": 0.7061224489795919, + "acc_stderr,none": 0.02916273841024971 + }, + "mmlu_sociology": { + "alias": " - sociology", + "acc,none": 0.7910447761194029, + "acc_stderr,none": 0.02874829893172869 + }, + "mmlu_us_foreign_policy": { + "alias": " - us_foreign_policy", + "acc,none": 0.78, + "acc_stderr,none": 0.041633319989322654 + }, + "mmlu_stem": { + "acc,none": 0.4884176182707993, + "acc_stderr,none": 0.00877683486319255, + "alias": " - stem" + }, + "mmlu_abstract_algebra": { + "alias": " - abstract_algebra", + "acc,none": 0.31, + "acc_stderr,none": 0.04648231987117317 + }, + "mmlu_anatomy": { + "alias": " - anatomy", + "acc,none": 0.5185185185185185, + "acc_stderr,none": 0.043163785995113245 + }, + "mmlu_astronomy": { + "alias": " - astronomy", + "acc,none": 0.6513157894736842, + "acc_stderr,none": 0.038781398887976125 + }, + "mmlu_college_biology": { + "alias": " - college_biology", + "acc,none": 0.6666666666666666, + "acc_stderr,none": 0.03942082639927217 + }, + "mmlu_college_chemistry": { + "alias": " - college_chemistry", + "acc,none": 0.37, + "acc_stderr,none": 0.048523658709390974 + }, + "mmlu_college_computer_science": { + "alias": " - college_computer_science", + "acc,none": 0.34, + "acc_stderr,none": 0.04760952285695233 + }, + "mmlu_college_mathematics": { + "alias": " - college_mathematics", + "acc,none": 0.4, + "acc_stderr,none": 0.0492365963917331 + }, + "mmlu_college_physics": { + "alias": " - college_physics", + "acc,none": 0.4019607843137255, + "acc_stderr,none": 0.04878608714466993 + }, + "mmlu_computer_security": { + "alias": " - computer_security", + "acc,none": 0.71, + "acc_stderr,none": 0.045604802157206865 + }, + "mmlu_conceptual_physics": { + "alias": " - conceptual_physics", + "acc,none": 0.46808510638297873, + "acc_stderr,none": 0.03261936918467376 + }, + "mmlu_electrical_engineering": { + "alias": " - electrical_engineering", + "acc,none": 0.5310344827586206, + "acc_stderr,none": 0.041586327620978254 + }, + "mmlu_elementary_mathematics": { + "alias": " - elementary_mathematics", + "acc,none": 0.38333333333333336, + "acc_stderr,none": 0.02811757974289911 + }, + "mmlu_high_school_biology": { + "alias": " - high_school_biology", + "acc,none": 0.69, + "acc_stderr,none": 0.02674667484725186 + }, + "mmlu_high_school_chemistry": { + "alias": " - high_school_chemistry", + "acc,none": 0.5221674876847291, + "acc_stderr,none": 0.03514528562175008 + }, + "mmlu_high_school_computer_science": { + "alias": " - high_school_computer_science", + "acc,none": 0.64, + "acc_stderr,none": 0.048241815132442176 + }, + "mmlu_high_school_mathematics": { + "alias": " - high_school_mathematics", + "acc,none": 0.4222222222222222, + "acc_stderr,none": 0.030114442019668074 + }, + "mmlu_high_school_physics": { + "alias": " - high_school_physics", + "acc,none": 0.3708609271523179, + "acc_stderr,none": 0.039439666991836285 + }, + "mmlu_high_school_statistics": { + "alias": " - high_school_statistics", + "acc,none": 0.4027777777777778, + "acc_stderr,none": 0.0334488738299786 + }, + "mmlu_machine_learning": { + "alias": " - machine_learning", + "acc,none": 0.375, + "acc_stderr,none": 0.04595091388086298 + } + }, + "expert_activation_report": { + "layer_0": "3.31", + "layer_1": "3.31", + "layer_2": "3.11", + "layer_3": "2.82", + "layer_4": "2.57", + "layer_5": "2.54", + "layer_6": "2.45", + "layer_7": "2.06", + "layer_8": "2.05", + "layer_9": "2.06", + "layer_10": "2.10", + "layer_11": "1.98", + "layer_12": "1.94", + "layer_13": "1.95", + "layer_14": "1.98", + "layer_15": "1.87", + "layer_16": "1.77", + "layer_17": "1.98", + "layer_18": "1.90", + "layer_19": "2.18", + "layer_20": "2.24", + "layer_21": "2.35", + "layer_22": "2.37", + "layer_23": "2.58", + "overall_average": "2.31" + } +} \ No newline at end of file diff --git a/results/results_dynamic_mmlu_0.25.json b/results/results_dynamic_mmlu_0.25.json new file mode 100644 index 0000000..5cd1d8c --- /dev/null +++ b/results/results_dynamic_mmlu_0.25.json @@ -0,0 +1,356 @@ +{ + "config": { + "tasks": [ + "mmlu" + ], + "batch_size": 12, + "limit": 300, + "use_pruned_model": true, + "pruned_metadata": "prune_experts/super_experts_ids.json", + "mode": "least", + "pruning_method": "dynamic", + "k": 0, + "dynamic_routing_threshold": 0.25, + "device": "cuda", + "output_file": "results_dynamic_mmlu_0.25.json" + }, + "results": { + "mmlu": { + "acc,none": 0.6068023469013568, + "acc_stderr,none": 0.004464612500571555, + "alias": "mmlu" + }, + "mmlu_humanities": { + "acc,none": 0.6196779964221825, + "acc_stderr,none": 0.008652691513969508, + "alias": " - humanities" + }, + "mmlu_formal_logic": { + "alias": " - formal_logic", + "acc,none": 0.3253968253968254, + "acc_stderr,none": 0.04190596438871138 + }, + "mmlu_high_school_european_history": { + "alias": " - high_school_european_history", + "acc,none": 0.6727272727272727, + "acc_stderr,none": 0.036639749943912406 + }, + "mmlu_high_school_us_history": { + "alias": " - high_school_us_history", + "acc,none": 0.75, + "acc_stderr,none": 0.03039153369274154 + }, + "mmlu_high_school_world_history": { + "alias": " - high_school_world_history", + "acc,none": 0.7721518987341772, + "acc_stderr,none": 0.027303484599069387 + }, + "mmlu_international_law": { + "alias": " - international_law", + "acc,none": 0.71900826446281, + "acc_stderr,none": 0.04103203830514515 + }, + "mmlu_jurisprudence": { + "alias": " - jurisprudence", + "acc,none": 0.6944444444444444, + "acc_stderr,none": 0.044531975073749855 + }, + "mmlu_logical_fallacies": { + "alias": " - logical_fallacies", + "acc,none": 0.7239263803680982, + "acc_stderr,none": 0.035123852837050454 + }, + "mmlu_moral_disputes": { + "alias": " - moral_disputes", + "acc,none": 0.64, + "acc_stderr,none": 0.02775911673437953 + }, + "mmlu_moral_scenarios": { + "alias": " - moral_scenarios", + "acc,none": 0.2833333333333333, + "acc_stderr,none": 0.026059845940064916 + }, + "mmlu_philosophy": { + "alias": " - philosophy", + "acc,none": 0.6666666666666666, + "acc_stderr,none": 0.027262027336984355 + }, + "mmlu_prehistory": { + "alias": " - prehistory", + "acc,none": 0.7033333333333334, + "acc_stderr,none": 0.026416749751055384 + }, + "mmlu_professional_law": { + "alias": " - professional_law", + "acc,none": 0.45, + "acc_stderr,none": 0.028770804599878935 + }, + "mmlu_world_religions": { + "alias": " - world_religions", + "acc,none": 0.8245614035087719, + "acc_stderr,none": 0.0291708855007277 + }, + "mmlu_other": { + "acc,none": 0.6405653170359052, + "acc_stderr,none": 0.009082926063340935, + "alias": " - other" + }, + "mmlu_business_ethics": { + "alias": " - business_ethics", + "acc,none": 0.71, + "acc_stderr,none": 0.045604802157206865 + }, + "mmlu_clinical_knowledge": { + "alias": " - clinical_knowledge", + "acc,none": 0.6264150943396226, + "acc_stderr,none": 0.029773082713319812 + }, + "mmlu_college_medicine": { + "alias": " - college_medicine", + "acc,none": 0.5664739884393064, + "acc_stderr,none": 0.03778621079092051 + }, + "mmlu_global_facts": { + "alias": " - global_facts", + "acc,none": 0.36, + "acc_stderr,none": 0.048241815132442176 + }, + "mmlu_human_aging": { + "alias": " - human_aging", + "acc,none": 0.6278026905829597, + "acc_stderr,none": 0.03244305283008735 + }, + "mmlu_management": { + "alias": " - management", + "acc,none": 0.7669902912621359, + "acc_stderr,none": 0.041858325989283136 + }, + "mmlu_marketing": { + "alias": " - marketing", + "acc,none": 0.8547008547008547, + "acc_stderr,none": 0.02308663508684137 + }, + "mmlu_medical_genetics": { + "alias": " - medical_genetics", + "acc,none": 0.68, + "acc_stderr,none": 0.046882617226215076 + }, + "mmlu_miscellaneous": { + "alias": " - miscellaneous", + "acc,none": 0.7666666666666667, + "acc_stderr,none": 0.024459979523511425 + }, + "mmlu_nutrition": { + "alias": " - nutrition", + "acc,none": 0.6933333333333334, + "acc_stderr,none": 0.026666666666666665 + }, + "mmlu_professional_accounting": { + "alias": " - professional_accounting", + "acc,none": 0.4574468085106383, + "acc_stderr,none": 0.029719281272236827 + }, + "mmlu_professional_medicine": { + "alias": " - professional_medicine", + "acc,none": 0.6176470588235294, + "acc_stderr,none": 0.029520095697687675 + }, + "mmlu_virology": { + "alias": " - virology", + "acc,none": 0.5060240963855421, + "acc_stderr,none": 0.03892212195333041 + }, + "mmlu_social_sciences": { + "acc,none": 0.6810699588477366, + "acc_stderr,none": 0.00915898993705285, + "alias": " - social sciences" + }, + "mmlu_econometrics": { + "alias": " - econometrics", + "acc,none": 0.32456140350877194, + "acc_stderr,none": 0.044045561573747664 + }, + "mmlu_high_school_geography": { + "alias": " - high_school_geography", + "acc,none": 0.7777777777777778, + "acc_stderr,none": 0.029620227874790486 + }, + "mmlu_high_school_government_and_politics": { + "alias": " - high_school_government_and_politics", + "acc,none": 0.8186528497409327, + "acc_stderr,none": 0.027807032360686077 + }, + "mmlu_high_school_macroeconomics": { + "alias": " - high_school_macroeconomics", + "acc,none": 0.5633333333333334, + "acc_stderr,none": 0.028682840061780172 + }, + "mmlu_high_school_microeconomics": { + "alias": " - high_school_microeconomics", + "acc,none": 0.6008403361344538, + "acc_stderr,none": 0.031811100324139197 + }, + "mmlu_high_school_psychology": { + "alias": " - high_school_psychology", + "acc,none": 0.7433333333333333, + "acc_stderr,none": 0.025260441987310457 + }, + "mmlu_human_sexuality": { + "alias": " - human_sexuality", + "acc,none": 0.7251908396946565, + "acc_stderr,none": 0.03915345408847834 + }, + "mmlu_professional_psychology": { + "alias": " - professional_psychology", + "acc,none": 0.5966666666666667, + "acc_stderr,none": 0.028370197016959926 + }, + "mmlu_public_relations": { + "alias": " - public_relations", + "acc,none": 0.6545454545454545, + "acc_stderr,none": 0.045546196175410524 + }, + "mmlu_security_studies": { + "alias": " - security_studies", + "acc,none": 0.746938775510204, + "acc_stderr,none": 0.027833023871399704 + }, + "mmlu_sociology": { + "alias": " - sociology", + "acc,none": 0.8059701492537313, + "acc_stderr,none": 0.02796267760476895 + }, + "mmlu_us_foreign_policy": { + "alias": " - us_foreign_policy", + "acc,none": 0.8, + "acc_stderr,none": 0.04020151261036849 + }, + "mmlu_stem": { + "acc,none": 0.5073409461663948, + "acc_stderr,none": 0.008791607591326177, + "alias": " - stem" + }, + "mmlu_abstract_algebra": { + "alias": " - abstract_algebra", + "acc,none": 0.35, + "acc_stderr,none": 0.04793724854411023 + }, + "mmlu_anatomy": { + "alias": " - anatomy", + "acc,none": 0.5407407407407407, + "acc_stderr,none": 0.04304979692464244 + }, + "mmlu_astronomy": { + "alias": " - astronomy", + "acc,none": 0.6513157894736842, + "acc_stderr,none": 0.038781398887976125 + }, + "mmlu_college_biology": { + "alias": " - college_biology", + "acc,none": 0.7013888888888888, + "acc_stderr,none": 0.03827052357950753 + }, + "mmlu_college_chemistry": { + "alias": " - college_chemistry", + "acc,none": 0.43, + "acc_stderr,none": 0.049756985195624305 + }, + "mmlu_college_computer_science": { + "alias": " - college_computer_science", + "acc,none": 0.43, + "acc_stderr,none": 0.049756985195624305 + }, + "mmlu_college_mathematics": { + "alias": " - college_mathematics", + "acc,none": 0.39, + "acc_stderr,none": 0.04902071300001973 + }, + "mmlu_college_physics": { + "alias": " - college_physics", + "acc,none": 0.3627450980392157, + "acc_stderr,none": 0.047840607041056527 + }, + "mmlu_computer_security": { + "alias": " - computer_security", + "acc,none": 0.73, + "acc_stderr,none": 0.04461960433384737 + }, + "mmlu_conceptual_physics": { + "alias": " - conceptual_physics", + "acc,none": 0.4723404255319149, + "acc_stderr,none": 0.03263597118409762 + }, + "mmlu_electrical_engineering": { + "alias": " - electrical_engineering", + "acc,none": 0.5724137931034483, + "acc_stderr,none": 0.041227371113703344 + }, + "mmlu_elementary_mathematics": { + "alias": " - elementary_mathematics", + "acc,none": 0.4266666666666667, + "acc_stderr,none": 0.028603050929618474 + }, + "mmlu_high_school_biology": { + "alias": " - high_school_biology", + "acc,none": 0.72, + "acc_stderr,none": 0.02596627604487782 + }, + "mmlu_high_school_chemistry": { + "alias": " - high_school_chemistry", + "acc,none": 0.5024630541871922, + "acc_stderr,none": 0.035179450386910595 + }, + "mmlu_high_school_computer_science": { + "alias": " - high_school_computer_science", + "acc,none": 0.58, + "acc_stderr,none": 0.04960449637488582 + }, + "mmlu_high_school_mathematics": { + "alias": " - high_school_mathematics", + "acc,none": 0.4148148148148148, + "acc_stderr,none": 0.03003984245406923 + }, + "mmlu_high_school_physics": { + "alias": " - high_school_physics", + "acc,none": 0.37748344370860926, + "acc_stderr,none": 0.03958027231121572 + }, + "mmlu_high_school_statistics": { + "alias": " - high_school_statistics", + "acc,none": 0.4675925925925926, + "acc_stderr,none": 0.03402801581358966 + }, + "mmlu_machine_learning": { + "alias": " - machine_learning", + "acc,none": 0.39285714285714285, + "acc_stderr,none": 0.04635550135609972 + } + }, + "expert_activation_report": { + "layer_0": "3.63", + "layer_1": "3.63", + "layer_2": "3.55", + "layer_3": "3.23", + "layer_4": "3.12", + "layer_5": "3.05", + "layer_6": "2.99", + "layer_7": "2.52", + "layer_8": "2.57", + "layer_9": "2.64", + "layer_10": "2.68", + "layer_11": "2.58", + "layer_12": "2.47", + "layer_13": "2.51", + "layer_14": "2.45", + "layer_15": "2.40", + "layer_16": "2.25", + "layer_17": "2.41", + "layer_18": "2.37", + "layer_19": "2.70", + "layer_20": "2.65", + "layer_21": "2.77", + "layer_22": "2.80", + "layer_23": "3.00", + "overall_average": "2.79" + } +} \ No newline at end of file diff --git a/results/results_dynamic_mmlu_0.3.json b/results/results_dynamic_mmlu_0.3.json new file mode 100644 index 0000000..9883832 --- /dev/null +++ b/results/results_dynamic_mmlu_0.3.json @@ -0,0 +1,356 @@ +{ + "config": { + "tasks": [ + "mmlu" + ], + "batch_size": 12, + "limit": 300, + "use_pruned_model": true, + "pruned_metadata": "prune_experts/super_experts_ids.json", + "mode": "least", + "pruning_method": "dynamic", + "k": 0, + "dynamic_routing_threshold": 0.3, + "device": "cuda", + "output_file": "results_dynamic_mmlu_0.3.json" + }, + "results": { + "mmlu": { + "acc,none": 0.6169783645031169, + "acc_stderr,none": 0.004435145518823452, + "alias": "mmlu" + }, + "mmlu_humanities": { + "acc,none": 0.6232558139534884, + "acc_stderr,none": 0.00860960391323884, + "alias": " - humanities" + }, + "mmlu_formal_logic": { + "alias": " - formal_logic", + "acc,none": 0.3253968253968254, + "acc_stderr,none": 0.04190596438871138 + }, + "mmlu_high_school_european_history": { + "alias": " - high_school_european_history", + "acc,none": 0.6848484848484848, + "acc_stderr,none": 0.03627730575022407 + }, + "mmlu_high_school_us_history": { + "alias": " - high_school_us_history", + "acc,none": 0.7598039215686274, + "acc_stderr,none": 0.029983733055913633 + }, + "mmlu_high_school_world_history": { + "alias": " - high_school_world_history", + "acc,none": 0.759493670886076, + "acc_stderr,none": 0.02782078198114972 + }, + "mmlu_international_law": { + "alias": " - international_law", + "acc,none": 0.7603305785123967, + "acc_stderr,none": 0.038968789850704115 + }, + "mmlu_jurisprudence": { + "alias": " - jurisprudence", + "acc,none": 0.7222222222222222, + "acc_stderr,none": 0.04330043749650743 + }, + "mmlu_logical_fallacies": { + "alias": " - logical_fallacies", + "acc,none": 0.7361963190184049, + "acc_stderr,none": 0.03462419931615622 + }, + "mmlu_moral_disputes": { + "alias": " - moral_disputes", + "acc,none": 0.66, + "acc_stderr,none": 0.027395285584216902 + }, + "mmlu_moral_scenarios": { + "alias": " - moral_scenarios", + "acc,none": 0.2633333333333333, + "acc_stderr,none": 0.025471401031969185 + }, + "mmlu_philosophy": { + "alias": " - philosophy", + "acc,none": 0.6666666666666666, + "acc_stderr,none": 0.027262027336984355 + }, + "mmlu_prehistory": { + "alias": " - prehistory", + "acc,none": 0.6833333333333333, + "acc_stderr,none": 0.026901833738451123 + }, + "mmlu_professional_law": { + "alias": " - professional_law", + "acc,none": 0.47, + "acc_stderr,none": 0.028863651326417043 + }, + "mmlu_world_religions": { + "alias": " - world_religions", + "acc,none": 0.8187134502923976, + "acc_stderr,none": 0.02954774168764006 + }, + "mmlu_other": { + "acc,none": 0.6550802139037433, + "acc_stderr,none": 0.00897307309882773, + "alias": " - other" + }, + "mmlu_business_ethics": { + "alias": " - business_ethics", + "acc,none": 0.73, + "acc_stderr,none": 0.04461960433384737 + }, + "mmlu_clinical_knowledge": { + "alias": " - clinical_knowledge", + "acc,none": 0.6641509433962264, + "acc_stderr,none": 0.029067220146644885 + }, + "mmlu_college_medicine": { + "alias": " - college_medicine", + "acc,none": 0.5838150289017341, + "acc_stderr,none": 0.037585177754049494 + }, + "mmlu_global_facts": { + "alias": " - global_facts", + "acc,none": 0.36, + "acc_stderr,none": 0.048241815132442176 + }, + "mmlu_human_aging": { + "alias": " - human_aging", + "acc,none": 0.6412556053811659, + "acc_stderr,none": 0.032190792004199886 + }, + "mmlu_management": { + "alias": " - management", + "acc,none": 0.7961165048543689, + "acc_stderr,none": 0.0398913985953177 + }, + "mmlu_marketing": { + "alias": " - marketing", + "acc,none": 0.8589743589743589, + "acc_stderr,none": 0.022801382534597486 + }, + "mmlu_medical_genetics": { + "alias": " - medical_genetics", + "acc,none": 0.67, + "acc_stderr,none": 0.04725815626252609 + }, + "mmlu_miscellaneous": { + "alias": " - miscellaneous", + "acc,none": 0.7866666666666666, + "acc_stderr,none": 0.02369131349654085 + }, + "mmlu_nutrition": { + "alias": " - nutrition", + "acc,none": 0.6966666666666667, + "acc_stderr,none": 0.026585019936500975 + }, + "mmlu_professional_accounting": { + "alias": " - professional_accounting", + "acc,none": 0.4574468085106383, + "acc_stderr,none": 0.029719281272236827 + }, + "mmlu_professional_medicine": { + "alias": " - professional_medicine", + "acc,none": 0.6544117647058824, + "acc_stderr,none": 0.0288881931039887 + }, + "mmlu_virology": { + "alias": " - virology", + "acc,none": 0.5060240963855421, + "acc_stderr,none": 0.03892212195333041 + }, + "mmlu_social_sciences": { + "acc,none": 0.6946502057613169, + "acc_stderr,none": 0.009029716740566076, + "alias": " - social sciences" + }, + "mmlu_econometrics": { + "alias": " - econometrics", + "acc,none": 0.2982456140350877, + "acc_stderr,none": 0.04303684033537313 + }, + "mmlu_high_school_geography": { + "alias": " - high_school_geography", + "acc,none": 0.7878787878787878, + "acc_stderr,none": 0.029126522834586777 + }, + "mmlu_high_school_government_and_politics": { + "alias": " - high_school_government_and_politics", + "acc,none": 0.8238341968911918, + "acc_stderr,none": 0.02749350424454809 + }, + "mmlu_high_school_macroeconomics": { + "alias": " - high_school_macroeconomics", + "acc,none": 0.5866666666666667, + "acc_stderr,none": 0.028478055207315906 + }, + "mmlu_high_school_microeconomics": { + "alias": " - high_school_microeconomics", + "acc,none": 0.6218487394957983, + "acc_stderr,none": 0.031499305777849054 + }, + "mmlu_high_school_psychology": { + "alias": " - high_school_psychology", + "acc,none": 0.7733333333333333, + "acc_stderr,none": 0.02421260961795184 + }, + "mmlu_human_sexuality": { + "alias": " - human_sexuality", + "acc,none": 0.7022900763358778, + "acc_stderr,none": 0.04010358942462197 + }, + "mmlu_professional_psychology": { + "alias": " - professional_psychology", + "acc,none": 0.6366666666666667, + "acc_stderr,none": 0.027814616968981933 + }, + "mmlu_public_relations": { + "alias": " - public_relations", + "acc,none": 0.6454545454545455, + "acc_stderr,none": 0.04582004841505413 + }, + "mmlu_security_studies": { + "alias": " - security_studies", + "acc,none": 0.7387755102040816, + "acc_stderr,none": 0.028123429335142832 + }, + "mmlu_sociology": { + "alias": " - sociology", + "acc,none": 0.835820895522388, + "acc_stderr,none": 0.026193923544454094 + }, + "mmlu_us_foreign_policy": { + "alias": " - us_foreign_policy", + "acc,none": 0.8, + "acc_stderr,none": 0.04020151261036849 + }, + "mmlu_stem": { + "acc,none": 0.5171288743882545, + "acc_stderr,none": 0.008804305609521144, + "alias": " - stem" + }, + "mmlu_abstract_algebra": { + "alias": " - abstract_algebra", + "acc,none": 0.38, + "acc_stderr,none": 0.04878317312145634 + }, + "mmlu_anatomy": { + "alias": " - anatomy", + "acc,none": 0.562962962962963, + "acc_stderr,none": 0.042849586397534056 + }, + "mmlu_astronomy": { + "alias": " - astronomy", + "acc,none": 0.6118421052631579, + "acc_stderr,none": 0.03965842097512745 + }, + "mmlu_college_biology": { + "alias": " - college_biology", + "acc,none": 0.6805555555555556, + "acc_stderr,none": 0.0389907368735733 + }, + "mmlu_college_chemistry": { + "alias": " - college_chemistry", + "acc,none": 0.45, + "acc_stderr,none": 0.05 + }, + "mmlu_college_computer_science": { + "alias": " - college_computer_science", + "acc,none": 0.44, + "acc_stderr,none": 0.049888765156985884 + }, + "mmlu_college_mathematics": { + "alias": " - college_mathematics", + "acc,none": 0.38, + "acc_stderr,none": 0.04878317312145634 + }, + "mmlu_college_physics": { + "alias": " - college_physics", + "acc,none": 0.4411764705882353, + "acc_stderr,none": 0.049406356306056644 + }, + "mmlu_computer_security": { + "alias": " - computer_security", + "acc,none": 0.72, + "acc_stderr,none": 0.045126085985421296 + }, + "mmlu_conceptual_physics": { + "alias": " - conceptual_physics", + "acc,none": 0.5276595744680851, + "acc_stderr,none": 0.03263597118409762 + }, + "mmlu_electrical_engineering": { + "alias": " - electrical_engineering", + "acc,none": 0.593103448275862, + "acc_stderr,none": 0.040937939812662354 + }, + "mmlu_elementary_mathematics": { + "alias": " - elementary_mathematics", + "acc,none": 0.43, + "acc_stderr,none": 0.02863096997084751 + }, + "mmlu_high_school_biology": { + "alias": " - high_school_biology", + "acc,none": 0.7433333333333333, + "acc_stderr,none": 0.025260441987310457 + }, + "mmlu_high_school_chemistry": { + "alias": " - high_school_chemistry", + "acc,none": 0.49261083743842365, + "acc_stderr,none": 0.03517603540361012 + }, + "mmlu_high_school_computer_science": { + "alias": " - high_school_computer_science", + "acc,none": 0.59, + "acc_stderr,none": 0.04943110704237104 + }, + "mmlu_high_school_mathematics": { + "alias": " - high_school_mathematics", + "acc,none": 0.42592592592592593, + "acc_stderr,none": 0.0301491356013659 + }, + "mmlu_high_school_physics": { + "alias": " - high_school_physics", + "acc,none": 0.3841059602649007, + "acc_stderr,none": 0.03971301814719192 + }, + "mmlu_high_school_statistics": { + "alias": " - high_school_statistics", + "acc,none": 0.4537037037037037, + "acc_stderr,none": 0.033953227263757976 + }, + "mmlu_machine_learning": { + "alias": " - machine_learning", + "acc,none": 0.39285714285714285, + "acc_stderr,none": 0.04635550135609972 + } + }, + "expert_activation_report": { + "layer_0": "3.77", + "layer_1": "3.82", + "layer_2": "3.77", + "layer_3": "3.61", + "layer_4": "3.53", + "layer_5": "3.46", + "layer_6": "3.45", + "layer_7": "3.00", + "layer_8": "3.02", + "layer_9": "3.11", + "layer_10": "3.12", + "layer_11": "3.08", + "layer_12": "3.02", + "layer_13": "2.97", + "layer_14": "2.92", + "layer_15": "2.81", + "layer_16": "2.65", + "layer_17": "2.80", + "layer_18": "2.76", + "layer_19": "3.06", + "layer_20": "3.00", + "layer_21": "3.08", + "layer_22": "3.12", + "layer_23": "3.30", + "overall_average": "3.18" + } +} \ No newline at end of file diff --git a/results/results_dynamic_mmlu_0.35.json b/results/results_dynamic_mmlu_0.35.json new file mode 100644 index 0000000..f9a3757 --- /dev/null +++ b/results/results_dynamic_mmlu_0.35.json @@ -0,0 +1,356 @@ +{ + "config": { + "tasks": [ + "mmlu" + ], + "batch_size": 12, + "limit": 300, + "use_pruned_model": true, + "pruned_metadata": "prune_experts/super_experts_ids.json", + "mode": "least", + "pruning_method": "dynamic", + "k": 0, + "dynamic_routing_threshold": 0.35, + "device": "cuda", + "output_file": "results_dynamic_mmlu_0.35.json" + }, + "results": { + "mmlu": { + "acc,none": 0.6206453978731207, + "acc_stderr,none": 0.004417465524477762, + "alias": "mmlu" + }, + "mmlu_humanities": { + "acc,none": 0.631484794275492, + "acc_stderr,none": 0.008505050815020863, + "alias": " - humanities" + }, + "mmlu_formal_logic": { + "alias": " - formal_logic", + "acc,none": 0.31746031746031744, + "acc_stderr,none": 0.04163453031302857 + }, + "mmlu_high_school_european_history": { + "alias": " - high_school_european_history", + "acc,none": 0.6787878787878788, + "acc_stderr,none": 0.03646204963253815 + }, + "mmlu_high_school_us_history": { + "alias": " - high_school_us_history", + "acc,none": 0.7843137254901961, + "acc_stderr,none": 0.028867431449849337 + }, + "mmlu_high_school_world_history": { + "alias": " - high_school_world_history", + "acc,none": 0.7890295358649789, + "acc_stderr,none": 0.0265583725026619 + }, + "mmlu_international_law": { + "alias": " - international_law", + "acc,none": 0.7603305785123967, + "acc_stderr,none": 0.038968789850704115 + }, + "mmlu_jurisprudence": { + "alias": " - jurisprudence", + "acc,none": 0.7314814814814815, + "acc_stderr,none": 0.04284467968052193 + }, + "mmlu_logical_fallacies": { + "alias": " - logical_fallacies", + "acc,none": 0.7177914110429447, + "acc_stderr,none": 0.03536117886664741 + }, + "mmlu_moral_disputes": { + "alias": " - moral_disputes", + "acc,none": 0.6766666666666666, + "acc_stderr,none": 0.027050608391385764 + }, + "mmlu_moral_scenarios": { + "alias": " - moral_scenarios", + "acc,none": 0.24, + "acc_stderr,none": 0.02469885513168684 + }, + "mmlu_philosophy": { + "alias": " - philosophy", + "acc,none": 0.66, + "acc_stderr,none": 0.027395285584216902 + }, + "mmlu_prehistory": { + "alias": " - prehistory", + "acc,none": 0.7066666666666667, + "acc_stderr,none": 0.026330094490574305 + }, + "mmlu_professional_law": { + "alias": " - professional_law", + "acc,none": 0.5033333333333333, + "acc_stderr,none": 0.02891510401902553 + }, + "mmlu_world_religions": { + "alias": " - world_religions", + "acc,none": 0.8304093567251462, + "acc_stderr,none": 0.02878210810540169 + }, + "mmlu_other": { + "acc,none": 0.6608097784568373, + "acc_stderr,none": 0.008960681943423046, + "alias": " - other" + }, + "mmlu_business_ethics": { + "alias": " - business_ethics", + "acc,none": 0.71, + "acc_stderr,none": 0.045604802157206865 + }, + "mmlu_clinical_knowledge": { + "alias": " - clinical_knowledge", + "acc,none": 0.6716981132075471, + "acc_stderr,none": 0.028901593612411777 + }, + "mmlu_college_medicine": { + "alias": " - college_medicine", + "acc,none": 0.5895953757225434, + "acc_stderr,none": 0.03750757044895538 + }, + "mmlu_global_facts": { + "alias": " - global_facts", + "acc,none": 0.41, + "acc_stderr,none": 0.04943110704237104 + }, + "mmlu_human_aging": { + "alias": " - human_aging", + "acc,none": 0.6457399103139013, + "acc_stderr,none": 0.032100621541349836 + }, + "mmlu_management": { + "alias": " - management", + "acc,none": 0.8252427184466019, + "acc_stderr,none": 0.03760178006026618 + }, + "mmlu_marketing": { + "alias": " - marketing", + "acc,none": 0.8717948717948718, + "acc_stderr,none": 0.021901905115073377 + }, + "mmlu_medical_genetics": { + "alias": " - medical_genetics", + "acc,none": 0.66, + "acc_stderr,none": 0.04760952285695234 + }, + "mmlu_miscellaneous": { + "alias": " - miscellaneous", + "acc,none": 0.79, + "acc_stderr,none": 0.023555243542102446 + }, + "mmlu_nutrition": { + "alias": " - nutrition", + "acc,none": 0.68, + "acc_stderr,none": 0.026977012386927034 + }, + "mmlu_professional_accounting": { + "alias": " - professional_accounting", + "acc,none": 0.475177304964539, + "acc_stderr,none": 0.0297907192438297 + }, + "mmlu_professional_medicine": { + "alias": " - professional_medicine", + "acc,none": 0.6580882352941176, + "acc_stderr,none": 0.02881472242225415 + }, + "mmlu_virology": { + "alias": " - virology", + "acc,none": 0.5120481927710844, + "acc_stderr,none": 0.038913644958358196 + }, + "mmlu_social_sciences": { + "acc,none": 0.6979423868312757, + "acc_stderr,none": 0.009015406279237463, + "alias": " - social sciences" + }, + "mmlu_econometrics": { + "alias": " - econometrics", + "acc,none": 0.3157894736842105, + "acc_stderr,none": 0.04372748290278002 + }, + "mmlu_high_school_geography": { + "alias": " - high_school_geography", + "acc,none": 0.7878787878787878, + "acc_stderr,none": 0.029126522834586777 + }, + "mmlu_high_school_government_and_politics": { + "alias": " - high_school_government_and_politics", + "acc,none": 0.8082901554404145, + "acc_stderr,none": 0.02840895362624522 + }, + "mmlu_high_school_macroeconomics": { + "alias": " - high_school_macroeconomics", + "acc,none": 0.58, + "acc_stderr,none": 0.02854322544954485 + }, + "mmlu_high_school_microeconomics": { + "alias": " - high_school_microeconomics", + "acc,none": 0.6218487394957983, + "acc_stderr,none": 0.031499305777849054 + }, + "mmlu_high_school_psychology": { + "alias": " - high_school_psychology", + "acc,none": 0.78, + "acc_stderr,none": 0.02395648228514071 + }, + "mmlu_human_sexuality": { + "alias": " - human_sexuality", + "acc,none": 0.7175572519083969, + "acc_stderr,none": 0.039484061257683584 + }, + "mmlu_professional_psychology": { + "alias": " - professional_psychology", + "acc,none": 0.6433333333333333, + "acc_stderr,none": 0.02770216390105995 + }, + "mmlu_public_relations": { + "alias": " - public_relations", + "acc,none": 0.6727272727272727, + "acc_stderr,none": 0.04494290866252085 + }, + "mmlu_security_studies": { + "alias": " - security_studies", + "acc,none": 0.746938775510204, + "acc_stderr,none": 0.027833023871399704 + }, + "mmlu_sociology": { + "alias": " - sociology", + "acc,none": 0.8407960199004975, + "acc_stderr,none": 0.025870646766169153 + }, + "mmlu_us_foreign_policy": { + "alias": " - us_foreign_policy", + "acc,none": 0.79, + "acc_stderr,none": 0.040936018074033236 + }, + "mmlu_stem": { + "acc,none": 0.5151712887438825, + "acc_stderr,none": 0.008794696237094786, + "alias": " - stem" + }, + "mmlu_abstract_algebra": { + "alias": " - abstract_algebra", + "acc,none": 0.44, + "acc_stderr,none": 0.049888765156985884 + }, + "mmlu_anatomy": { + "alias": " - anatomy", + "acc,none": 0.5259259259259259, + "acc_stderr,none": 0.043135316967505756 + }, + "mmlu_astronomy": { + "alias": " - astronomy", + "acc,none": 0.6118421052631579, + "acc_stderr,none": 0.03965842097512745 + }, + "mmlu_college_biology": { + "alias": " - college_biology", + "acc,none": 0.6736111111111112, + "acc_stderr,none": 0.03921067198982268 + }, + "mmlu_college_chemistry": { + "alias": " - college_chemistry", + "acc,none": 0.41, + "acc_stderr,none": 0.04943110704237104 + }, + "mmlu_college_computer_science": { + "alias": " - college_computer_science", + "acc,none": 0.43, + "acc_stderr,none": 0.049756985195624305 + }, + "mmlu_college_mathematics": { + "alias": " - college_mathematics", + "acc,none": 0.4, + "acc_stderr,none": 0.0492365963917331 + }, + "mmlu_college_physics": { + "alias": " - college_physics", + "acc,none": 0.4019607843137255, + "acc_stderr,none": 0.04878608714466993 + }, + "mmlu_computer_security": { + "alias": " - computer_security", + "acc,none": 0.72, + "acc_stderr,none": 0.045126085985421296 + }, + "mmlu_conceptual_physics": { + "alias": " - conceptual_physics", + "acc,none": 0.5531914893617021, + "acc_stderr,none": 0.03250053684365844 + }, + "mmlu_electrical_engineering": { + "alias": " - electrical_engineering", + "acc,none": 0.5724137931034483, + "acc_stderr,none": 0.041227371113703344 + }, + "mmlu_elementary_mathematics": { + "alias": " - elementary_mathematics", + "acc,none": 0.43333333333333335, + "acc_stderr,none": 0.028657565120703166 + }, + "mmlu_high_school_biology": { + "alias": " - high_school_biology", + "acc,none": 0.75, + "acc_stderr,none": 0.025041771123531665 + }, + "mmlu_high_school_chemistry": { + "alias": " - high_school_chemistry", + "acc,none": 0.5221674876847291, + "acc_stderr,none": 0.03514528562175008 + }, + "mmlu_high_school_computer_science": { + "alias": " - high_school_computer_science", + "acc,none": 0.59, + "acc_stderr,none": 0.04943110704237104 + }, + "mmlu_high_school_mathematics": { + "alias": " - high_school_mathematics", + "acc,none": 0.3962962962962963, + "acc_stderr,none": 0.02982261945853398 + }, + "mmlu_high_school_physics": { + "alias": " - high_school_physics", + "acc,none": 0.3973509933774834, + "acc_stderr,none": 0.039955240076816834 + }, + "mmlu_high_school_statistics": { + "alias": " - high_school_statistics", + "acc,none": 0.4351851851851852, + "acc_stderr,none": 0.03381200005643526 + }, + "mmlu_machine_learning": { + "alias": " - machine_learning", + "acc,none": 0.38392857142857145, + "acc_stderr,none": 0.04616143075028549 + } + }, + "expert_activation_report": { + "layer_0": "3.86", + "layer_1": "3.88", + "layer_2": "3.86", + "layer_3": "3.84", + "layer_4": "3.80", + "layer_5": "3.71", + "layer_6": "3.68", + "layer_7": "3.41", + "layer_8": "3.37", + "layer_9": "3.46", + "layer_10": "3.45", + "layer_11": "3.50", + "layer_12": "3.38", + "layer_13": "3.32", + "layer_14": "3.32", + "layer_15": "3.20", + "layer_16": "3.03", + "layer_17": "3.15", + "layer_18": "3.11", + "layer_19": "3.37", + "layer_20": "3.27", + "layer_21": "3.33", + "layer_22": "3.37", + "layer_23": "3.47", + "overall_average": "3.46" + } +} \ No newline at end of file diff --git a/results/results_dynamic_mmlu_0.4.json b/results/results_dynamic_mmlu_0.4.json new file mode 100644 index 0000000..bbffe58 --- /dev/null +++ b/results/results_dynamic_mmlu_0.4.json @@ -0,0 +1,356 @@ +{ + "config": { + "tasks": [ + "mmlu" + ], + "batch_size": 12, + "limit": 300, + "use_pruned_model": true, + "pruned_metadata": "prune_experts/super_experts_ids.json", + "mode": "least", + "pruning_method": "dynamic", + "k": 0, + "dynamic_routing_threshold": 0.4, + "device": "cuda", + "output_file": "results_dynamic_mmlu_0.4.json" + }, + "results": { + "mmlu": { + "acc,none": 0.6242207554088742, + "acc_stderr,none": 0.004418014800603493, + "alias": "mmlu" + }, + "mmlu_humanities": { + "acc,none": 0.6329159212880143, + "acc_stderr,none": 0.008512830566508249, + "alias": " - humanities" + }, + "mmlu_formal_logic": { + "alias": " - formal_logic", + "acc,none": 0.3253968253968254, + "acc_stderr,none": 0.04190596438871138 + }, + "mmlu_high_school_european_history": { + "alias": " - high_school_european_history", + "acc,none": 0.6727272727272727, + "acc_stderr,none": 0.036639749943912406 + }, + "mmlu_high_school_us_history": { + "alias": " - high_school_us_history", + "acc,none": 0.7794117647058824, + "acc_stderr,none": 0.029102254389674065 + }, + "mmlu_high_school_world_history": { + "alias": " - high_school_world_history", + "acc,none": 0.7932489451476793, + "acc_stderr,none": 0.026361651668389073 + }, + "mmlu_international_law": { + "alias": " - international_law", + "acc,none": 0.768595041322314, + "acc_stderr,none": 0.03849856098794088 + }, + "mmlu_jurisprudence": { + "alias": " - jurisprudence", + "acc,none": 0.7685185185185185, + "acc_stderr,none": 0.040774947092526284 + }, + "mmlu_logical_fallacies": { + "alias": " - logical_fallacies", + "acc,none": 0.7239263803680982, + "acc_stderr,none": 0.035123852837050454 + }, + "mmlu_moral_disputes": { + "alias": " - moral_disputes", + "acc,none": 0.6733333333333333, + "acc_stderr,none": 0.027122634635122614 + }, + "mmlu_moral_scenarios": { + "alias": " - moral_scenarios", + "acc,none": 0.25333333333333335, + "acc_stderr,none": 0.025152082937711873 + }, + "mmlu_philosophy": { + "alias": " - philosophy", + "acc,none": 0.66, + "acc_stderr,none": 0.027395285584216902 + }, + "mmlu_prehistory": { + "alias": " - prehistory", + "acc,none": 0.6933333333333334, + "acc_stderr,none": 0.026666666666666665 + }, + "mmlu_professional_law": { + "alias": " - professional_law", + "acc,none": 0.49666666666666665, + "acc_stderr,none": 0.02891510401902553 + }, + "mmlu_world_religions": { + "alias": " - world_religions", + "acc,none": 0.8362573099415205, + "acc_stderr,none": 0.02838091959614585 + }, + "mmlu_other": { + "acc,none": 0.6623376623376623, + "acc_stderr,none": 0.008956600936248938, + "alias": " - other" + }, + "mmlu_business_ethics": { + "alias": " - business_ethics", + "acc,none": 0.73, + "acc_stderr,none": 0.04461960433384737 + }, + "mmlu_clinical_knowledge": { + "alias": " - clinical_knowledge", + "acc,none": 0.6754716981132075, + "acc_stderr,none": 0.02881561571343209 + }, + "mmlu_college_medicine": { + "alias": " - college_medicine", + "acc,none": 0.5895953757225434, + "acc_stderr,none": 0.03750757044895538 + }, + "mmlu_global_facts": { + "alias": " - global_facts", + "acc,none": 0.39, + "acc_stderr,none": 0.04902071300001973 + }, + "mmlu_human_aging": { + "alias": " - human_aging", + "acc,none": 0.6457399103139013, + "acc_stderr,none": 0.032100621541349836 + }, + "mmlu_management": { + "alias": " - management", + "acc,none": 0.8252427184466019, + "acc_stderr,none": 0.03760178006026618 + }, + "mmlu_marketing": { + "alias": " - marketing", + "acc,none": 0.8589743589743589, + "acc_stderr,none": 0.022801382534597486 + }, + "mmlu_medical_genetics": { + "alias": " - medical_genetics", + "acc,none": 0.66, + "acc_stderr,none": 0.04760952285695234 + }, + "mmlu_miscellaneous": { + "alias": " - miscellaneous", + "acc,none": 0.79, + "acc_stderr,none": 0.023555243542102446 + }, + "mmlu_nutrition": { + "alias": " - nutrition", + "acc,none": 0.6933333333333334, + "acc_stderr,none": 0.026666666666666665 + }, + "mmlu_professional_accounting": { + "alias": " - professional_accounting", + "acc,none": 0.4858156028368794, + "acc_stderr,none": 0.029815494483682023 + }, + "mmlu_professional_medicine": { + "alias": " - professional_medicine", + "acc,none": 0.6544117647058824, + "acc_stderr,none": 0.0288881931039887 + }, + "mmlu_virology": { + "alias": " - virology", + "acc,none": 0.5120481927710844, + "acc_stderr,none": 0.038913644958358196 + }, + "mmlu_social_sciences": { + "acc,none": 0.702880658436214, + "acc_stderr,none": 0.008991064564584969, + "alias": " - social sciences" + }, + "mmlu_econometrics": { + "alias": " - econometrics", + "acc,none": 0.32456140350877194, + "acc_stderr,none": 0.044045561573747664 + }, + "mmlu_high_school_geography": { + "alias": " - high_school_geography", + "acc,none": 0.7929292929292929, + "acc_stderr,none": 0.02886977846026699 + }, + "mmlu_high_school_government_and_politics": { + "alias": " - high_school_government_and_politics", + "acc,none": 0.8031088082901554, + "acc_stderr,none": 0.028697873971860723 + }, + "mmlu_high_school_macroeconomics": { + "alias": " - high_school_macroeconomics", + "acc,none": 0.6033333333333334, + "acc_stderr,none": 0.02829149642514491 + }, + "mmlu_high_school_microeconomics": { + "alias": " - high_school_microeconomics", + "acc,none": 0.6512605042016807, + "acc_stderr,none": 0.03095663632856658 + }, + "mmlu_high_school_psychology": { + "alias": " - high_school_psychology", + "acc,none": 0.7833333333333333, + "acc_stderr,none": 0.023825046699671875 + }, + "mmlu_human_sexuality": { + "alias": " - human_sexuality", + "acc,none": 0.7099236641221374, + "acc_stderr,none": 0.03980066246467765 + }, + "mmlu_professional_psychology": { + "alias": " - professional_psychology", + "acc,none": 0.6266666666666667, + "acc_stderr,none": 0.027972487412192524 + }, + "mmlu_public_relations": { + "alias": " - public_relations", + "acc,none": 0.6727272727272727, + "acc_stderr,none": 0.04494290866252085 + }, + "mmlu_security_studies": { + "alias": " - security_studies", + "acc,none": 0.7510204081632653, + "acc_stderr,none": 0.027682979522960276 + }, + "mmlu_sociology": { + "alias": " - sociology", + "acc,none": 0.8308457711442786, + "acc_stderr,none": 0.026508590656233278 + }, + "mmlu_us_foreign_policy": { + "alias": " - us_foreign_policy", + "acc,none": 0.82, + "acc_stderr,none": 0.03861229196653691 + }, + "mmlu_stem": { + "acc,none": 0.5213703099510604, + "acc_stderr,none": 0.008810630628527192, + "alias": " - stem" + }, + "mmlu_abstract_algebra": { + "alias": " - abstract_algebra", + "acc,none": 0.39, + "acc_stderr,none": 0.04902071300001973 + }, + "mmlu_anatomy": { + "alias": " - anatomy", + "acc,none": 0.5555555555555556, + "acc_stderr,none": 0.04292596718256977 + }, + "mmlu_astronomy": { + "alias": " - astronomy", + "acc,none": 0.631578947368421, + "acc_stderr,none": 0.03925523381052935 + }, + "mmlu_college_biology": { + "alias": " - college_biology", + "acc,none": 0.6666666666666666, + "acc_stderr,none": 0.03942082639927217 + }, + "mmlu_college_chemistry": { + "alias": " - college_chemistry", + "acc,none": 0.45, + "acc_stderr,none": 0.05 + }, + "mmlu_college_computer_science": { + "alias": " - college_computer_science", + "acc,none": 0.42, + "acc_stderr,none": 0.04960449637488583 + }, + "mmlu_college_mathematics": { + "alias": " - college_mathematics", + "acc,none": 0.44, + "acc_stderr,none": 0.049888765156985884 + }, + "mmlu_college_physics": { + "alias": " - college_physics", + "acc,none": 0.4019607843137255, + "acc_stderr,none": 0.04878608714466993 + }, + "mmlu_computer_security": { + "alias": " - computer_security", + "acc,none": 0.74, + "acc_stderr,none": 0.0440844002276808 + }, + "mmlu_conceptual_physics": { + "alias": " - conceptual_physics", + "acc,none": 0.5531914893617021, + "acc_stderr,none": 0.03250053684365844 + }, + "mmlu_electrical_engineering": { + "alias": " - electrical_engineering", + "acc,none": 0.6, + "acc_stderr,none": 0.040824829046386304 + }, + "mmlu_elementary_mathematics": { + "alias": " - elementary_mathematics", + "acc,none": 0.43333333333333335, + "acc_stderr,none": 0.028657565120703166 + }, + "mmlu_high_school_biology": { + "alias": " - high_school_biology", + "acc,none": 0.73, + "acc_stderr,none": 0.02567483835226067 + }, + "mmlu_high_school_chemistry": { + "alias": " - high_school_chemistry", + "acc,none": 0.5073891625615764, + "acc_stderr,none": 0.03517603540361012 + }, + "mmlu_high_school_computer_science": { + "alias": " - high_school_computer_science", + "acc,none": 0.6, + "acc_stderr,none": 0.0492365963917331 + }, + "mmlu_high_school_mathematics": { + "alias": " - high_school_mathematics", + "acc,none": 0.42592592592592593, + "acc_stderr,none": 0.0301491356013659 + }, + "mmlu_high_school_physics": { + "alias": " - high_school_physics", + "acc,none": 0.39072847682119205, + "acc_stderr,none": 0.039837983066598075 + }, + "mmlu_high_school_statistics": { + "alias": " - high_school_statistics", + "acc,none": 0.4583333333333333, + "acc_stderr,none": 0.03398110890294638 + }, + "mmlu_machine_learning": { + "alias": " - machine_learning", + "acc,none": 0.39285714285714285, + "acc_stderr,none": 0.04635550135609972 + } + }, + "expert_activation_report": { + "layer_0": "3.91", + "layer_1": "3.94", + "layer_2": "3.88", + "layer_3": "3.94", + "layer_4": "3.90", + "layer_5": "3.82", + "layer_6": "3.82", + "layer_7": "3.67", + "layer_8": "3.67", + "layer_9": "3.71", + "layer_10": "3.72", + "layer_11": "3.73", + "layer_12": "3.64", + "layer_13": "3.61", + "layer_14": "3.62", + "layer_15": "3.55", + "layer_16": "3.33", + "layer_17": "3.38", + "layer_18": "3.38", + "layer_19": "3.58", + "layer_20": "3.45", + "layer_21": "3.50", + "layer_22": "3.53", + "layer_23": "3.61", + "overall_average": "3.66" + } +} \ No newline at end of file