-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathconfig.yaml
More file actions
41 lines (34 loc) · 1.65 KB
/
Copy pathconfig.yaml
File metadata and controls
41 lines (34 loc) · 1.65 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
Process:
Stream: False # setting to true will be an expensive operation, stream data in batches, defined below.
# If set to false, process will read the input ticket query from [Data][input_query_path]
Stream_batch: 500 # 500 new tickets daily. To be used only if True.
Logging:
log_dir: "logging"
log_path: "app.log"
Data:
original_ticket: 'path/to/support_tickets.json'
ticket_train: "data/data_training_data.json" # 90% portion of the original dataset, usd for training, ingesting into vector db and evaluating.
ticket_test: "data/data_stream_test.json" # 10% portion of the original dataset
training_data: "data/data.json" # processed training data for deep learning model
input_query_path: "data/input_query.json" # Single Incoming ticket to be processed, used by single_stream fucntion, if Stream set to False
final_output: "data/queried_result.json" # Output path Incoming ticket after processiung
Train:
train_model: False # this will train the model from scratch, else, load the trained model provided.
train_tokenizer: False # this will train the tokenizer from scratch, else, load the trained tokenizer provided.
VOCAB_SIZE: 5000
BATCH_SIZE: 64
Sample_per_class: 12 # 12 samples per batch for class imbalance.
learning_rate: 0.0001
EPOCHS: 20
MAX_LEN: 100
EMBEDDING_SIZE: 100
ENCODER_HIDDEN_DIMENSION: 64
Model:
model_path: "model/support_ticket_model.pth"
tokenizer_path: "model/tokenizer.json"
RAG:
embedding_model: "all-MiniLM-L6-v2"
vector_store: "data/vector_store"
vectordb_name: "support_ticket_db"
ingestion_batch_size: 5461
score_threshold: 0.65