logan7000/cogrpo-n3-union-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupA-qwen25-best Reinforcement Learning • 242k • Updated Aug 6 • 7
logan7000/cogrpo-n3-union-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupA-qwen25-end Reinforcement Learning • 242k • Updated Aug 6 • 9
logan7000/cogrpo-n3-union-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupB-llama32-best Reinforcement Learning • 175k • Updated Aug 6 • 13
logan7000/cogrpo-n3-union-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupB-llama32-end Reinforcement Learning • 175k • Updated Aug 6 • 9
logan7000/cogrpo-n3-union-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupC-qwen3-best Reinforcement Learning • 2B • Updated Aug 6 • 14
logan7000/cogrpo-n3-union-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupC-qwen3-end Reinforcement Learning • 2B • Updated Aug 6 • 11
logan7000/cogrpo-n3-selfpeers-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupA-qwen25-best Reinforcement Learning • 242k • Updated Aug 6 • 18
logan7000/cogrpo-n3-selfpeers-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupA-qwen25-end Reinforcement Learning • 242k • Updated Aug 6 • 4
logan7000/cogrpo-n3-selfpeers-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupB-llama32-best Reinforcement Learning • 175k • Updated Aug 6 • 8
logan7000/cogrpo-n3-selfpeers-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupB-llama32-end Reinforcement Learning • 175k • Updated Aug 6 • 6
logan7000/cogrpo-n3-selfpeers-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupC-qwen3-best Reinforcement Learning • 2B • Updated Aug 6 • 10
logan7000/cogrpo-n3-selfpeers-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupC-qwen3-end Reinforcement Learning • 2B • Updated Aug 6 • 5
logan7000/cogrpo-n3-pooledmaj-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupA-qwen25-best Reinforcement Learning • 242k • Updated 29 days ago • 45
logan7000/cogrpo-n3-pooledmaj-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupA-qwen25-end Reinforcement Learning • 242k • Updated 29 days ago • 31
logan7000/cogrpo-n3-pooledmaj-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupB-llama32-best Reinforcement Learning • 175k • Updated 29 days ago • 47
logan7000/cogrpo-n3-pooledmaj-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupB-llama32-end Reinforcement Learning • 175k • Updated 29 days ago • 32
logan7000/cogrpo-n3-pooledmaj-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupC-qwen3-best Reinforcement Learning • 2B • Updated 29 days ago • 55
logan7000/cogrpo-n3-pooledmaj-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupC-qwen3-end Reinforcement Learning • 2B • Updated 29 days ago • 27
logan7000/cogrpo-n3-randompeer-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupA-qwen25-best Reinforcement Learning • 242k • Updated 29 days ago • 68
logan7000/cogrpo-n3-randompeer-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupA-qwen25-end Reinforcement Learning • 242k • Updated 29 days ago • 30
logan7000/cogrpo-n3-randompeer-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupB-llama32-best Reinforcement Learning • 175k • Updated 29 days ago • 47
logan7000/cogrpo-n3-randompeer-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupB-llama32-end Reinforcement Learning • 175k • Updated 29 days ago • 20
logan7000/cogrpo-n3-randompeer-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupC-qwen3-best Reinforcement Learning • 2B • Updated 29 days ago • 55
logan7000/cogrpo-n3-randompeer-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupC-qwen3-end Reinforcement Learning • 2B • Updated 29 days ago • 22
logan7000/cogrpo-n3-ring-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupA-qwen25-best Reinforcement Learning • 242k • Updated 29 days ago • 54
logan7000/cogrpo-n3-ring-qwen25-3b-x-llama32-3b-x-qwen3-1p7b-math345-groupA-qwen25-end Reinforcement Learning • 242k • Updated 29 days ago • 23