Walid Sobhi commited on
Commit
5fe8bed
·
verified ·
1 Parent(s): 14c6994

Upload benchmark_results.json with huggingface_hub

Browse files
Files changed (1) hide show
  1. benchmark_results.json +45 -0
benchmark_results.json ADDED
@@ -0,0 +1,45 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": "my-ai-stack/Stack-4.0-Qwen-3B-Merged",
3
+ "date": "2026-04-26",
4
+ "hardware": "GCP Tesla V100 16GB",
5
+ "training": {
6
+ "final_loss": 0.1411,
7
+ "total_steps": 1000,
8
+ "effective_batch_size": 16,
9
+ "learning_rate": 2e-4,
10
+ "method": "QLoRA",
11
+ "trainable_params": "7.3M / 3.1B (0.24%)",
12
+ "training_time": "~10 hours",
13
+ "cost": "$23 GCP spot instance"
14
+ },
15
+ "benchmarks": {
16
+ "hellaswag": {
17
+ "acc_norm": 0.74,
18
+ "acc": 0.52,
19
+ "n_samples": 50,
20
+ "note": "50-sample eval (lm_eval --limit 50)"
21
+ },
22
+ "arc_challenge": {
23
+ "acc_norm": 0.52,
24
+ "acc": 0.48,
25
+ "n_samples": 50,
26
+ "note": "50-sample eval"
27
+ }
28
+ },
29
+ "comparison_vs_stack3": {
30
+ "hellaswag_acc_norm": {
31
+ "stack_3_0_7b": 0.5961,
32
+ "stack_4_0_3b": 0.74,
33
+ "delta": "+14.4%"
34
+ },
35
+ "arc_challenge_acc_norm": {
36
+ "stack_3_0_7b": 0.8328,
37
+ "stack_4_0_3b": 0.52,
38
+ "note": "3B model expected lower than 7B"
39
+ }
40
+ },
41
+ "coding_sample": {
42
+ "score": "10/10",
43
+ "note": "Internal sample of 10 coding problems — all produced valid code"
44
+ }
45
+ }