didulantha commited on
Commit
e8cec49
·
verified ·
1 Parent(s): b5ab33c

Upload 7 files

Browse files
Files changed (7) hide show
  1. README.md +94 -0
  2. config.json +24 -0
  3. metadata.json +19 -0
  4. model.safetensors +3 -0
  5. special_tokens_map.json +7 -0
  6. tokenizer_config.json +58 -0
  7. vocab.txt +0 -0
README.md ADDED
@@ -0,0 +1,94 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ language: en
3
+ license: apache-2.0
4
+ tags:
5
+ - text-classification
6
+ - spam-detection
7
+ - distilbert
8
+ - sms
9
+ datasets:
10
+ - sms_spam
11
+ metrics:
12
+ - accuracy
13
+ - f1
14
+ - precision
15
+ - recall
16
+ widget:
17
+ - text: "Congratulations! You've won a FREE prize. Call now!"
18
+ - text: "Hey, are we still meeting for lunch tomorrow?"
19
+ - text: "URGENT: Your account has been compromised!"
20
+ - text: "Can you pick up milk on the way home?"
21
+ ---
22
+
23
+ # SMS Spam Detector
24
+
25
+ ## Model Description
26
+
27
+ Fine-tuned DistilBERT model for detecting spam SMS messages.
28
+
29
+ **Performance:**
30
+ - Accuracy: 0.9916 (99.16%)
31
+ - Precision: 0.9730
32
+ - Recall: 0.9643
33
+ - F1-Score: 0.9686
34
+ - ROC-AUC: 0.9990
35
+
36
+ ## Quick Start
37
+
38
+ ```python
39
+ from transformers import pipeline
40
+
41
+ classifier = pipeline("text-classification", model="YOUR_USERNAME/sms-spam-detector")
42
+ result = classifier("Win a free prize now!")
43
+ print(result)
44
+ ```
45
+
46
+ ## Detailed Usage
47
+
48
+ ```python
49
+ from transformers import DistilBertTokenizer, DistilBertForSequenceClassification
50
+ import torch
51
+
52
+ tokenizer = DistilBertTokenizer.from_pretrained("YOUR_USERNAME/sms-spam-detector")
53
+ model = DistilBertForSequenceClassification.from_pretrained("YOUR_USERNAME/sms-spam-detector")
54
+
55
+ def predict(text):
56
+ inputs = tokenizer(text, return_tensors="pt", truncation=True, max_length=128)
57
+ outputs = model(**inputs)
58
+ probs = torch.softmax(outputs.logits, dim=1)
59
+ return "SPAM" if probs[0][1] > 0.5 else "HAM"
60
+
61
+ print(predict("Free prize!"))
62
+ ```
63
+
64
+ ## Training Data
65
+
66
+ - Dataset: SMS Spam Collection (5,574 messages)
67
+ - Train/Val/Test split: 70/15/15
68
+ - Ham: 86.6%, Spam: 13.4%
69
+
70
+ ## Training Details
71
+
72
+ - Base model: distilbert-base-uncased
73
+ - Epochs: 3
74
+ - Batch size: 16
75
+ - Learning rate: 2e-05
76
+ - Optimizer: AdamW
77
+
78
+ ## Limitations
79
+
80
+ - Trained on English SMS messages only
81
+ - May not generalize well to other languages
82
+ - Performance may degrade on heavily obfuscated spam
83
+
84
+ ## Citation
85
+
86
+ ```bibtex
87
+ @misc{sms-spam-detector,
88
+ author = {Isuru Didulantha},
89
+ title = {SMS Spam Detector},
90
+ year = {2025},
91
+ publisher = {HuggingFace},
92
+ url = {https://huggingface.co/didulantha/sms-spam-detector}
93
+ }
94
+ ```
config.json ADDED
@@ -0,0 +1,24 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "activation": "gelu",
3
+ "architectures": [
4
+ "DistilBertForSequenceClassification"
5
+ ],
6
+ "attention_dropout": 0.1,
7
+ "dim": 768,
8
+ "dropout": 0.1,
9
+ "hidden_dim": 3072,
10
+ "initializer_range": 0.02,
11
+ "max_position_embeddings": 512,
12
+ "model_type": "distilbert",
13
+ "n_heads": 12,
14
+ "n_layers": 6,
15
+ "pad_token_id": 0,
16
+ "problem_type": "single_label_classification",
17
+ "qa_dropout": 0.1,
18
+ "seq_classif_dropout": 0.2,
19
+ "sinusoidal_pos_embds": false,
20
+ "tie_weights_": true,
21
+ "torch_dtype": "float32",
22
+ "transformers_version": "4.52.4",
23
+ "vocab_size": 30522
24
+ }
metadata.json ADDED
@@ -0,0 +1,19 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model_name": "sms-spam-detector",
3
+ "base_model": "distilbert-base-uncased",
4
+ "task": "text-classification",
5
+ "language": "en",
6
+ "dataset": "SMS Spam Collection",
7
+ "metrics": {
8
+ "accuracy": 0.9916267942583732,
9
+ "precision": 0.972972972972973,
10
+ "recall": 0.9642857142857143,
11
+ "f1_score": 0.968609865470852,
12
+ "roc_auc": 0.9990380820836622
13
+ },
14
+ "training_params": {
15
+ "epochs": 3,
16
+ "batch_size": 16,
17
+ "learning_rate": 2e-05
18
+ }
19
+ }
model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:16cab3d0cfcfad0aea145d597d0a15334efcf897fb696605fcadba28a4b5c603
3
+ size 267832560
special_tokens_map.json ADDED
@@ -0,0 +1,7 @@
 
 
 
 
 
 
 
 
1
+ {
2
+ "cls_token": "[CLS]",
3
+ "mask_token": "[MASK]",
4
+ "pad_token": "[PAD]",
5
+ "sep_token": "[SEP]",
6
+ "unk_token": "[UNK]"
7
+ }
tokenizer_config.json ADDED
@@ -0,0 +1,58 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "added_tokens_decoder": {
3
+ "0": {
4
+ "content": "[PAD]",
5
+ "lstrip": false,
6
+ "normalized": false,
7
+ "rstrip": false,
8
+ "single_word": false,
9
+ "special": true
10
+ },
11
+ "100": {
12
+ "content": "[UNK]",
13
+ "lstrip": false,
14
+ "normalized": false,
15
+ "rstrip": false,
16
+ "single_word": false,
17
+ "special": true
18
+ },
19
+ "101": {
20
+ "content": "[CLS]",
21
+ "lstrip": false,
22
+ "normalized": false,
23
+ "rstrip": false,
24
+ "single_word": false,
25
+ "special": true
26
+ },
27
+ "102": {
28
+ "content": "[SEP]",
29
+ "lstrip": false,
30
+ "normalized": false,
31
+ "rstrip": false,
32
+ "single_word": false,
33
+ "special": true
34
+ },
35
+ "103": {
36
+ "content": "[MASK]",
37
+ "lstrip": false,
38
+ "normalized": false,
39
+ "rstrip": false,
40
+ "single_word": false,
41
+ "special": true
42
+ }
43
+ },
44
+ "clean_up_tokenization_spaces": true,
45
+ "cls_token": "[CLS]",
46
+ "do_basic_tokenize": true,
47
+ "do_lower_case": true,
48
+ "extra_special_tokens": {},
49
+ "mask_token": "[MASK]",
50
+ "model_max_length": 512,
51
+ "never_split": null,
52
+ "pad_token": "[PAD]",
53
+ "sep_token": "[SEP]",
54
+ "strip_accents": null,
55
+ "tokenize_chinese_chars": true,
56
+ "tokenizer_class": "DistilBertTokenizer",
57
+ "unk_token": "[UNK]"
58
+ }
vocab.txt ADDED
The diff for this file is too large to render. See raw diff