aleegis commited on
Commit
2489b90
·
verified ·
1 Parent(s): 4dbf9ed

Training in progress, step 3000, checkpoint

Browse files
last-checkpoint/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:075716c5570f8c8debc7e50f763274610603cfb6ff5e43c2dffe8173ad31612f
3
  size 73911112
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2f69215fa71abb8fa8004f1511df840de48cbb3e3e17f08478b4f4af80fb8b9e
3
  size 73911112
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:974d4ccf14aeae4d7e93b5874ce0039922f9f93a84e8c2be820bbb1cb8fd458d
3
  size 148053947
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d0e2f819b7454826ec6d8441dde5ffb4c03ce1cfd9a01fd30c049fb2905f63d5
3
  size 148053947
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:8d976af33a7826852ea9cb113ae65fe4244073dfc233a330c2cdaa7606d5b926
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9f39041d7c88d2860c2f85f64f19c797d2c674c05ba74004cf9743a32559628b
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:fb69807b0cc213740e86e6add3784f51b695f07900aa8a435d08a7fff4f32bd7
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:095865ee0dba2422fa75ed17304220ae17502f490054466dcc6d644f9f447b2a
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.30938466827088346,
6
  "eval_steps": 500,
7
- "global_step": 2700,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -1898,6 +1898,216 @@
1898
  "learning_rate": 3.0352986867686007e-06,
1899
  "loss": 1.6689,
1900
  "step": 2700
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1901
  }
1902
  ],
1903
  "logging_steps": 10,
@@ -1912,12 +2122,12 @@
1912
  "should_evaluate": false,
1913
  "should_log": false,
1914
  "should_save": true,
1915
- "should_training_stop": false
1916
  },
1917
  "attributes": {}
1918
  }
1919
  },
1920
- "total_flos": 3.526925859422208e+17,
1921
  "train_batch_size": 16,
1922
  "trial_name": null,
1923
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.3437607425232039,
6
  "eval_steps": 500,
7
+ "global_step": 3000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
1898
  "learning_rate": 3.0352986867686007e-06,
1899
  "loss": 1.6689,
1900
  "step": 2700
1901
+ },
1902
+ {
1903
+ "epoch": 0.3105305374126275,
1904
+ "grad_norm": 0.641806960105896,
1905
+ "learning_rate": 2.8388671026199522e-06,
1906
+ "loss": 1.5412,
1907
+ "step": 2710
1908
+ },
1909
+ {
1910
+ "epoch": 0.3116764065543715,
1911
+ "grad_norm": 0.6084569096565247,
1912
+ "learning_rate": 2.6488203809326207e-06,
1913
+ "loss": 1.3932,
1914
+ "step": 2720
1915
+ },
1916
+ {
1917
+ "epoch": 0.3128222756961155,
1918
+ "grad_norm": 0.5853345990180969,
1919
+ "learning_rate": 2.4651842509905487e-06,
1920
+ "loss": 1.4632,
1921
+ "step": 2730
1922
+ },
1923
+ {
1924
+ "epoch": 0.3139681448378595,
1925
+ "grad_norm": 0.5775859951972961,
1926
+ "learning_rate": 2.2879835741861586e-06,
1927
+ "loss": 1.3812,
1928
+ "step": 2740
1929
+ },
1930
+ {
1931
+ "epoch": 0.31511401397960354,
1932
+ "grad_norm": 0.5420204997062683,
1933
+ "learning_rate": 2.1172423406545516e-06,
1934
+ "loss": 1.5721,
1935
+ "step": 2750
1936
+ },
1937
+ {
1938
+ "epoch": 0.31625988312134756,
1939
+ "grad_norm": 0.48948049545288086,
1940
+ "learning_rate": 1.9529836660256096e-06,
1941
+ "loss": 1.6311,
1942
+ "step": 2760
1943
+ },
1944
+ {
1945
+ "epoch": 0.31740575226309153,
1946
+ "grad_norm": 0.6526772975921631,
1947
+ "learning_rate": 1.7952297882945003e-06,
1948
+ "loss": 1.7197,
1949
+ "step": 2770
1950
+ },
1951
+ {
1952
+ "epoch": 0.31855162140483556,
1953
+ "grad_norm": 0.5312083959579468,
1954
+ "learning_rate": 1.6440020648110067e-06,
1955
+ "loss": 1.3684,
1956
+ "step": 2780
1957
+ },
1958
+ {
1959
+ "epoch": 0.3196974905465796,
1960
+ "grad_norm": 0.6139872074127197,
1961
+ "learning_rate": 1.4993209693881183e-06,
1962
+ "loss": 1.5396,
1963
+ "step": 2790
1964
+ },
1965
+ {
1966
+ "epoch": 0.3208433596883236,
1967
+ "grad_norm": 0.6308534741401672,
1968
+ "learning_rate": 1.3612060895301759e-06,
1969
+ "loss": 1.4659,
1970
+ "step": 2800
1971
+ },
1972
+ {
1973
+ "epoch": 0.3219892288300676,
1974
+ "grad_norm": 0.5623085498809814,
1975
+ "learning_rate": 1.2296761237810207e-06,
1976
+ "loss": 1.4104,
1977
+ "step": 2810
1978
+ },
1979
+ {
1980
+ "epoch": 0.3231350979718116,
1981
+ "grad_norm": 0.5706859827041626,
1982
+ "learning_rate": 1.104748879192552e-06,
1983
+ "loss": 1.5481,
1984
+ "step": 2820
1985
+ },
1986
+ {
1987
+ "epoch": 0.32428096711355564,
1988
+ "grad_norm": 0.5466167330741882,
1989
+ "learning_rate": 9.864412689139123e-07,
1990
+ "loss": 1.4769,
1991
+ "step": 2830
1992
+ },
1993
+ {
1994
+ "epoch": 0.32542683625529967,
1995
+ "grad_norm": 0.5880963802337646,
1996
+ "learning_rate": 8.747693099017129e-07,
1997
+ "loss": 1.5516,
1998
+ "step": 2840
1999
+ },
2000
+ {
2001
+ "epoch": 0.32657270539704364,
2002
+ "grad_norm": 0.664244532585144,
2003
+ "learning_rate": 7.697481207516289e-07,
2004
+ "loss": 1.2943,
2005
+ "step": 2850
2006
+ },
2007
+ {
2008
+ "epoch": 0.32771857453878767,
2009
+ "grad_norm": 0.8473872542381287,
2010
+ "learning_rate": 6.713919196515317e-07,
2011
+ "loss": 1.2507,
2012
+ "step": 2860
2013
+ },
2014
+ {
2015
+ "epoch": 0.3288644436805317,
2016
+ "grad_norm": 0.8418035507202148,
2017
+ "learning_rate": 5.797140224566122e-07,
2018
+ "loss": 1.4377,
2019
+ "step": 2870
2020
+ },
2021
+ {
2022
+ "epoch": 0.3300103128222757,
2023
+ "grad_norm": 0.6410139203071594,
2024
+ "learning_rate": 4.947268408866113e-07,
2025
+ "loss": 1.4249,
2026
+ "step": 2880
2027
+ },
2028
+ {
2029
+ "epoch": 0.3311561819640197,
2030
+ "grad_norm": 0.6374297142028809,
2031
+ "learning_rate": 4.1644188084548063e-07,
2032
+ "loss": 1.5552,
2033
+ "step": 2890
2034
+ },
2035
+ {
2036
+ "epoch": 0.3323020511057637,
2037
+ "grad_norm": 0.701400637626648,
2038
+ "learning_rate": 3.4486974086366253e-07,
2039
+ "loss": 1.5406,
2040
+ "step": 2900
2041
+ },
2042
+ {
2043
+ "epoch": 0.33344792024750775,
2044
+ "grad_norm": 0.6102722883224487,
2045
+ "learning_rate": 2.800201106632205e-07,
2046
+ "loss": 1.5778,
2047
+ "step": 2910
2048
+ },
2049
+ {
2050
+ "epoch": 0.3345937893892518,
2051
+ "grad_norm": 0.6845527291297913,
2052
+ "learning_rate": 2.219017698460002e-07,
2053
+ "loss": 1.5503,
2054
+ "step": 2920
2055
+ },
2056
+ {
2057
+ "epoch": 0.33573965853099574,
2058
+ "grad_norm": 0.5155937075614929,
2059
+ "learning_rate": 1.7052258670501308e-07,
2060
+ "loss": 1.4669,
2061
+ "step": 2930
2062
+ },
2063
+ {
2064
+ "epoch": 0.33688552767273977,
2065
+ "grad_norm": 0.8121209740638733,
2066
+ "learning_rate": 1.2588951715921116e-07,
2067
+ "loss": 1.4042,
2068
+ "step": 2940
2069
+ },
2070
+ {
2071
+ "epoch": 0.3380313968144838,
2072
+ "grad_norm": 0.8292280435562134,
2073
+ "learning_rate": 8.800860381173448e-08,
2074
+ "loss": 1.415,
2075
+ "step": 2950
2076
+ },
2077
+ {
2078
+ "epoch": 0.3391772659562278,
2079
+ "grad_norm": 0.6214461326599121,
2080
+ "learning_rate": 5.688497513188229e-08,
2081
+ "loss": 1.2916,
2082
+ "step": 2960
2083
+ },
2084
+ {
2085
+ "epoch": 0.3403231350979718,
2086
+ "grad_norm": 0.6468619704246521,
2087
+ "learning_rate": 3.2522844760762836e-08,
2088
+ "loss": 1.3698,
2089
+ "step": 2970
2090
+ },
2091
+ {
2092
+ "epoch": 0.3414690042397158,
2093
+ "grad_norm": 0.5640342831611633,
2094
+ "learning_rate": 1.4925510940844156e-08,
2095
+ "loss": 1.336,
2096
+ "step": 2980
2097
+ },
2098
+ {
2099
+ "epoch": 0.34261487338145985,
2100
+ "grad_norm": 0.4522778391838074,
2101
+ "learning_rate": 4.095356069439005e-09,
2102
+ "loss": 1.3447,
2103
+ "step": 2990
2104
+ },
2105
+ {
2106
+ "epoch": 0.3437607425232039,
2107
+ "grad_norm": 0.560656726360321,
2108
+ "learning_rate": 3.384637615733155e-11,
2109
+ "loss": 1.4884,
2110
+ "step": 3000
2111
  }
2112
  ],
2113
  "logging_steps": 10,
 
2122
  "should_evaluate": false,
2123
  "should_log": false,
2124
  "should_save": true,
2125
+ "should_training_stop": true
2126
  },
2127
  "attributes": {}
2128
  }
2129
  },
2130
+ "total_flos": 3.91880651046912e+17,
2131
  "train_batch_size": 16,
2132
  "trial_name": null,
2133
  "trial_params": null