Upload folder using huggingface_hub
Browse files- .gitattributes +17 -0
- model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/.metadata +3 -0
- model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__0_0.distcp +3 -0
- model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__0_1.distcp +3 -0
- model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__1_0.distcp +3 -0
- model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__1_1.distcp +3 -0
- model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__2_0.distcp +3 -0
- model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__2_1.distcp +3 -0
- model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__3_0.distcp +3 -0
- model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__3_1.distcp +3 -0
- model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__4_0.distcp +3 -0
- model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__4_1.distcp +3 -0
- model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__5_0.distcp +3 -0
- model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__5_1.distcp +3 -0
- model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__6_0.distcp +3 -0
- model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__6_1.distcp +3 -0
- model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__7_0.distcp +3 -0
- model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__7_1.distcp +3 -0
- model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/common.pt +3 -0
- model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/metadata.json +1 -0
- model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/latest_checkpointed_iteration.txt +1 -0
- model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/linked_runs.txt +3 -0
    	
        .gitattributes
    CHANGED
    
    | @@ -729,3 +729,20 @@ model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3 | |
| 729 | 
             
            model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-1e-2-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__6_1.distcp filter=lfs diff=lfs merge=lfs -text
         | 
| 730 | 
             
            model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-1e-2-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__7_0.distcp filter=lfs diff=lfs merge=lfs -text
         | 
| 731 | 
             
            model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-1e-2-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__7_1.distcp filter=lfs diff=lfs merge=lfs -text
         | 
|  | |
|  | |
|  | |
|  | |
|  | |
|  | |
|  | |
|  | |
|  | |
|  | |
|  | |
|  | |
|  | |
|  | |
|  | |
|  | |
|  | 
|  | |
| 729 | 
             
            model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-1e-2-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__6_1.distcp filter=lfs diff=lfs merge=lfs -text
         | 
| 730 | 
             
            model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-1e-2-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__7_0.distcp filter=lfs diff=lfs merge=lfs -text
         | 
| 731 | 
             
            model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-1e-2-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__7_1.distcp filter=lfs diff=lfs merge=lfs -text
         | 
| 732 | 
            +
            model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/.metadata filter=lfs diff=lfs merge=lfs -text
         | 
| 733 | 
            +
            model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__0_0.distcp filter=lfs diff=lfs merge=lfs -text
         | 
| 734 | 
            +
            model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__0_1.distcp filter=lfs diff=lfs merge=lfs -text
         | 
| 735 | 
            +
            model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__1_0.distcp filter=lfs diff=lfs merge=lfs -text
         | 
| 736 | 
            +
            model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__1_1.distcp filter=lfs diff=lfs merge=lfs -text
         | 
| 737 | 
            +
            model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__2_0.distcp filter=lfs diff=lfs merge=lfs -text
         | 
| 738 | 
            +
            model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__2_1.distcp filter=lfs diff=lfs merge=lfs -text
         | 
| 739 | 
            +
            model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__3_0.distcp filter=lfs diff=lfs merge=lfs -text
         | 
| 740 | 
            +
            model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__3_1.distcp filter=lfs diff=lfs merge=lfs -text
         | 
| 741 | 
            +
            model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__4_0.distcp filter=lfs diff=lfs merge=lfs -text
         | 
| 742 | 
            +
            model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__4_1.distcp filter=lfs diff=lfs merge=lfs -text
         | 
| 743 | 
            +
            model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__5_0.distcp filter=lfs diff=lfs merge=lfs -text
         | 
| 744 | 
            +
            model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__5_1.distcp filter=lfs diff=lfs merge=lfs -text
         | 
| 745 | 
            +
            model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__6_0.distcp filter=lfs diff=lfs merge=lfs -text
         | 
| 746 | 
            +
            model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__6_1.distcp filter=lfs diff=lfs merge=lfs -text
         | 
| 747 | 
            +
            model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__7_0.distcp filter=lfs diff=lfs merge=lfs -text
         | 
| 748 | 
            +
            model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__7_1.distcp filter=lfs diff=lfs merge=lfs -text
         | 
    	
        model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/.metadata
    ADDED
    
    | @@ -0,0 +1,3 @@ | |
|  | |
|  | |
|  | 
|  | |
| 1 | 
            +
            version https://git-lfs.github.com/spec/v1
         | 
| 2 | 
            +
            oid sha256:caeb1141c9f8abf3c06685f87e04efa9873d1470f96454fb5fe3aca8a031211a
         | 
| 3 | 
            +
            size 2395258
         | 
    	
        model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__0_0.distcp
    ADDED
    
    | @@ -0,0 +1,3 @@ | |
|  | |
|  | |
|  | 
|  | |
| 1 | 
            +
            version https://git-lfs.github.com/spec/v1
         | 
| 2 | 
            +
            oid sha256:4270980a2e352589e87e38b42b770ae46ec8bbf5d6268e6bf4b40b909182d83c
         | 
| 3 | 
            +
            size 885540371
         | 
    	
        model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__0_1.distcp
    ADDED
    
    | @@ -0,0 +1,3 @@ | |
|  | |
|  | |
|  | 
|  | |
| 1 | 
            +
            version https://git-lfs.github.com/spec/v1
         | 
| 2 | 
            +
            oid sha256:0ba6703f55a165549a48c6480a04de29e666b99b2d8b4dcf08c53052e275d1f5
         | 
| 3 | 
            +
            size 885552026
         | 
    	
        model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__1_0.distcp
    ADDED
    
    | @@ -0,0 +1,3 @@ | |
|  | |
|  | |
|  | 
|  | |
| 1 | 
            +
            version https://git-lfs.github.com/spec/v1
         | 
| 2 | 
            +
            oid sha256:ea541a49c83f601bf7cccccfc0ebb1e59dd67fc2d4b8d5753f952a59a255db9c
         | 
| 3 | 
            +
            size 911423083
         | 
    	
        model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__1_1.distcp
    ADDED
    
    | @@ -0,0 +1,3 @@ | |
|  | |
|  | |
|  | 
|  | |
| 1 | 
            +
            version https://git-lfs.github.com/spec/v1
         | 
| 2 | 
            +
            oid sha256:507b4d0bd71f80a61d070ca30f04bdfc460b7bdbaf67f81e33ef9ec5946a7d21
         | 
| 3 | 
            +
            size 911406732
         | 
    	
        model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__2_0.distcp
    ADDED
    
    | @@ -0,0 +1,3 @@ | |
|  | |
|  | |
|  | 
|  | |
| 1 | 
            +
            version https://git-lfs.github.com/spec/v1
         | 
| 2 | 
            +
            oid sha256:236234aab9759d878db8e319eaf33a999de25740edd5d3f2a0af5959c3dd21bb
         | 
| 3 | 
            +
            size 885360269
         | 
    	
        model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__2_1.distcp
    ADDED
    
    | @@ -0,0 +1,3 @@ | |
|  | |
|  | |
|  | 
|  | |
| 1 | 
            +
            version https://git-lfs.github.com/spec/v1
         | 
| 2 | 
            +
            oid sha256:81874333fa4a063956a80de15a3be65490bc624836ba6e8d6105014444cae103
         | 
| 3 | 
            +
            size 885374462
         | 
    	
        model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__3_0.distcp
    ADDED
    
    | @@ -0,0 +1,3 @@ | |
|  | |
|  | |
|  | 
|  | |
| 1 | 
            +
            version https://git-lfs.github.com/spec/v1
         | 
| 2 | 
            +
            oid sha256:cf01b121a92d0c7a522738aa0d7c416c6a2b14b07bc42cc4b28e3dec167ad641
         | 
| 3 | 
            +
            size 884637222
         | 
    	
        model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__3_1.distcp
    ADDED
    
    | @@ -0,0 +1,3 @@ | |
|  | |
|  | |
|  | 
|  | |
| 1 | 
            +
            version https://git-lfs.github.com/spec/v1
         | 
| 2 | 
            +
            oid sha256:7d3a870285732a99f9ba75bfe4381d50b98efa3609fc79253fb4c6be8d5c57dc
         | 
| 3 | 
            +
            size 884615144
         | 
    	
        model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__4_0.distcp
    ADDED
    
    | @@ -0,0 +1,3 @@ | |
|  | |
|  | |
|  | 
|  | |
| 1 | 
            +
            version https://git-lfs.github.com/spec/v1
         | 
| 2 | 
            +
            oid sha256:8d6e209db06717aae7d943b331278d46aee10a0dfcf519572e08e20f1de7ac0c
         | 
| 3 | 
            +
            size 912018309
         | 
    	
        model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__4_1.distcp
    ADDED
    
    | @@ -0,0 +1,3 @@ | |
|  | |
|  | |
|  | 
|  | |
| 1 | 
            +
            version https://git-lfs.github.com/spec/v1
         | 
| 2 | 
            +
            oid sha256:afd180564eaecca5b8be38775540d358c95a3d5bebae98db26dfdc67b90296b2
         | 
| 3 | 
            +
            size 912043541
         | 
    	
        model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__5_0.distcp
    ADDED
    
    | @@ -0,0 +1,3 @@ | |
|  | |
|  | |
|  | 
|  | |
| 1 | 
            +
            version https://git-lfs.github.com/spec/v1
         | 
| 2 | 
            +
            oid sha256:92cbb4a8efc98f445b0b20a2549b6ef7ecf6a73dc8fb27b4865de0f9b7e68b9d
         | 
| 3 | 
            +
            size 884541063
         | 
    	
        model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__5_1.distcp
    ADDED
    
    | @@ -0,0 +1,3 @@ | |
|  | |
|  | |
|  | 
|  | |
| 1 | 
            +
            version https://git-lfs.github.com/spec/v1
         | 
| 2 | 
            +
            oid sha256:161ad606fc406d6d39c62b9c64f5c2847f1b3e4473c327e73684fb4d2d4cd09c
         | 
| 3 | 
            +
            size 884575757
         | 
    	
        model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__6_0.distcp
    ADDED
    
    | @@ -0,0 +1,3 @@ | |
|  | |
|  | |
|  | 
|  | |
| 1 | 
            +
            version https://git-lfs.github.com/spec/v1
         | 
| 2 | 
            +
            oid sha256:3f99128f0bec4366279876ed59f151f401f44366ff569e57240f07a8fdde27d0
         | 
| 3 | 
            +
            size 885158308
         | 
    	
        model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__6_1.distcp
    ADDED
    
    | @@ -0,0 +1,3 @@ | |
|  | |
|  | |
|  | 
|  | |
| 1 | 
            +
            version https://git-lfs.github.com/spec/v1
         | 
| 2 | 
            +
            oid sha256:e727e41a7ea6142621b7c5959f96ebb8b72cfcd25941d9a53e3a93e0db198f6a
         | 
| 3 | 
            +
            size 885199310
         | 
    	
        model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__7_0.distcp
    ADDED
    
    | @@ -0,0 +1,3 @@ | |
|  | |
|  | |
|  | 
|  | |
| 1 | 
            +
            version https://git-lfs.github.com/spec/v1
         | 
| 2 | 
            +
            oid sha256:74b1d498350afe622f3eae909c9abc57f6ccf3ad32ce82d0ea5ae387ba3e40bf
         | 
| 3 | 
            +
            size 884710920
         | 
    	
        model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/__7_1.distcp
    ADDED
    
    | @@ -0,0 +1,3 @@ | |
|  | |
|  | |
|  | 
|  | |
| 1 | 
            +
            version https://git-lfs.github.com/spec/v1
         | 
| 2 | 
            +
            oid sha256:b50b39d9f2407cc892a56054c48d97da8e41a785263cfb95ff3ba6a31a01f8c7
         | 
| 3 | 
            +
            size 884673072
         | 
    	
        model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/common.pt
    ADDED
    
    | @@ -0,0 +1,3 @@ | |
|  | |
|  | |
|  | 
|  | |
| 1 | 
            +
            version https://git-lfs.github.com/spec/v1
         | 
| 2 | 
            +
            oid sha256:1c9ebb7252219998941c3eee322296c5bda13958518831991137ec5856a95747
         | 
| 3 | 
            +
            size 19239
         | 
    	
        model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/iter_0004768/metadata.json
    ADDED
    
    | @@ -0,0 +1 @@ | |
|  | 
|  | |
| 1 | 
            +
            {"sharded_backend": "torch_dist", "sharded_backend_version": 1, "common_backend": "torch", "common_backend_version": 1}
         | 
    	
        model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/latest_checkpointed_iteration.txt
    ADDED
    
    | @@ -0,0 +1 @@ | |
|  | 
|  | |
| 1 | 
            +
            4768
         | 
    	
        model/dev-muon-mamba_moe-0.5b-q16-kv2-hybrid0.16-ep-16-sep-0-top2-cf-2-bias-1e-3-bf16-ep4-mp2-pp1-lr-2e-3-minlr-7e-7-bs-1024-gpus-8-seqlen-8192/linked_runs.txt
    ADDED
    
    | @@ -0,0 +1,3 @@ | |
|  | |
|  | |
|  | 
|  | |
| 1 | 
            +
            2025.05.27-00.49.27
         | 
| 2 | 
            +
            2025.05.27-13.35.55
         | 
| 3 | 
            +
            2025.05.27-13.44.37
         | 
