logan7000/cogrpo-n3-union-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupA-qwen25-best Reinforcement Learning • 242k • Updated 30 days ago • 53
logan7000/cogrpo-n3-union-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupA-qwen25-end Reinforcement Learning • 242k • Updated 30 days ago • 20
logan7000/cogrpo-n3-union-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupB-llama32-best Reinforcement Learning • 175k • Updated 30 days ago • 67
logan7000/cogrpo-n3-union-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupB-llama32-end Reinforcement Learning • 175k • Updated 30 days ago • 26
logan7000/cogrpo-n3-union-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupC-qwen3-best Reinforcement Learning • 2B • Updated 30 days ago • 55
logan7000/cogrpo-n3-union-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupC-qwen3-end Reinforcement Learning • 2B • Updated 30 days ago • 28
logan7000/cogrpo-n3-selfpeers-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupA-qwen25-best Reinforcement Learning • 242k • Updated 30 days ago • 46
logan7000/cogrpo-n3-selfpeers-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupA-qwen25-end Reinforcement Learning • 242k • Updated 30 days ago • 11
logan7000/cogrpo-n3-selfpeers-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupB-llama32-best Reinforcement Learning • 175k • Updated 30 days ago • 34
logan7000/cogrpo-n3-selfpeers-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupB-llama32-end Reinforcement Learning • 175k • Updated 30 days ago • 16
logan7000/cogrpo-n3-selfpeers-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupC-qwen3-best Reinforcement Learning • 2B • Updated 30 days ago • 37
logan7000/cogrpo-n3-selfpeers-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupC-qwen3-end Reinforcement Learning • 2B • Updated 30 days ago • 28
logan7000/cogrpo-n3-pooledmaj-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupA-qwen25-best Reinforcement Learning • 242k • Updated 26 days ago • 45
logan7000/cogrpo-n3-pooledmaj-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupA-qwen25-end Reinforcement Learning • 242k • Updated 26 days ago • 30
logan7000/cogrpo-n3-pooledmaj-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupB-llama32-best Reinforcement Learning • 175k • Updated 26 days ago • 47
logan7000/cogrpo-n3-pooledmaj-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupB-llama32-end Reinforcement Learning • 175k • Updated 26 days ago • 31
logan7000/cogrpo-n3-pooledmaj-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupC-qwen3-best Reinforcement Learning • 2B • Updated 26 days ago • 54
logan7000/cogrpo-n3-pooledmaj-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupC-qwen3-end Reinforcement Learning • 2B • Updated 26 days ago • 26
logan7000/cogrpo-n3-randompeer-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupA-qwen25-best Reinforcement Learning • 242k • Updated 26 days ago • 67
logan7000/cogrpo-n3-randompeer-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupA-qwen25-end Reinforcement Learning • 242k • Updated 26 days ago • 29
logan7000/cogrpo-n3-randompeer-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupB-llama32-best Reinforcement Learning • 175k • Updated 26 days ago • 46
logan7000/cogrpo-n3-randompeer-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupB-llama32-end Reinforcement Learning • 175k • Updated 26 days ago • 20
logan7000/cogrpo-n3-randompeer-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupC-qwen3-best Reinforcement Learning • 2B • Updated 26 days ago • 54
logan7000/cogrpo-n3-randompeer-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupC-qwen3-end Reinforcement Learning • 2B • Updated 26 days ago • 22
logan7000/cogrpo-n3-ring-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupA-qwen25-best Reinforcement Learning • 242k • Updated 26 days ago • 53
logan7000/cogrpo-n3-ring-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupA-qwen25-end Reinforcement Learning • 242k • Updated 26 days ago • 23