File size: 47,942 Bytes
33acf50
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
1001
1002
1003
1004
1005
1006
1007
1008
1009
1010
1011
1012
1013
1014
1015
1016
1017
import pandas as pd
import numpy as np
import pickle
import os
import re
import nltk
import urllib.request
import zipfile
import tarfile
import io
import shutil
import random
from tqdm import tqdm
from sklearn.feature_extraction.text import CountVectorizer, TfidfVectorizer
from sklearn.feature_extraction import DictVectorizer
from sklearn.naive_bayes import MultinomialNB
from sklearn.linear_model import LogisticRegression
from sklearn.model_selection import train_test_split, cross_val_score
from sklearn.metrics import accuracy_score, classification_report, confusion_matrix
from sklearn.pipeline import Pipeline
from scipy.sparse import hstack
from sentiment_lexicon import SentimentLexiconFeatures
from sklearn.calibration import CalibratedClassifierCV
import logging
from nltk.tree import Tree  # Add this import for named entity recognition

# Configure logging
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)

# Create models directory if it doesn't exist
if not os.path.exists('models'):
    os.makedirs('models')

# Set NLTK data path explicitly
nltk_data_path = os.path.join(os.getcwd(), 'nltk_data')
os.makedirs(nltk_data_path, exist_ok=True)
nltk.data.path.insert(0, nltk_data_path)

print("Starting model training process...")
print("Downloading NLTK data...")

# Download required NLTK data
nltk.download('movie_reviews', download_dir=nltk_data_path)
nltk.download('stopwords', download_dir=nltk_data_path)
nltk.download('punkt', download_dir=nltk_data_path)
nltk.download('vader_lexicon', download_dir=nltk_data_path)
nltk.download('sentiwordnet', download_dir=nltk_data_path)
nltk.download('wordnet', download_dir=nltk_data_path)
nltk.download('omw-1.4', download_dir=nltk_data_path)
nltk.download('averaged_perceptron_tagger', download_dir=nltk_data_path)
nltk.download('maxent_ne_chunker', download_dir=nltk_data_path)
nltk.download('words', download_dir=nltk_data_path)

from nltk.corpus import movie_reviews, stopwords
from nltk.tokenize import word_tokenize
from nltk import ne_chunk, pos_tag

# ==== IMPROVED TEXT PREPROCESSING WITH ENTITY RECOGNITION ====

def identify_movie_titles(text):
    """Identify potential movie titles for special handling"""
    try:
        # Simple pattern recognition for titles (capitalized phrases)
        potential_titles = re.findall(r'(?:The |A |An )?(?:[A-Z][a-z]+ )+', text)
        
        # Try named entity recognition as well
        tokens = word_tokenize(text)
        tagged = pos_tag(tokens)
        entities = ne_chunk(tagged)
        
        title_spans = []
        for chunk in entities:
            # Check if the chunk is a named entity (Tree) and has a label attribute
            if isinstance(chunk, Tree) and hasattr(chunk, 'label'):
                if chunk.label() == 'ORGANIZATION' or chunk.label() == 'PERSON':
                    # This could be a movie title
                    title_spans.append(' '.join([c[0] for c in chunk]))
        
        # Combine both approaches
        all_potential_titles = potential_titles + title_spans
        
        # Create a version with marked titles
        marked_text = text
        for title in all_potential_titles:
            title = title.strip()
            if len(title.split()) > 1 and title in text:  # Only multi-word titles
                marked_text = marked_text.replace(title, f"MOVIETITLE_{title.replace(' ', '_')}")
        
        return marked_text
    except Exception as e:
        logger.error(f"Error identifying movie titles: {e}")
        return text

def handle_negations(text):
    """
    Mark negated words to help the model understand negations.
    Example: "not bad" -> "not bad_NEG"
    """
    # Create a list of negation words
    negation_words = ['not', 'no', 'never', 'don\'t', 'doesn\'t', 'didn\'t', 
                     'can\'t', 'couldn\'t', 'shouldn\'t', 'wouldn\'t', 'isn\'t', 
                     'aren\'t', 'ain\'t', 'wasn\'t', 'weren\'t', 'haven\'t', 
                     'hasn\'t', 'hadn\'t', 'won\'t', 'nor', 'neither']
    
    try:
        # Tokenize the text
        words = word_tokenize(text.lower())
        
        # Process negations
        in_negation = False
        result = []
        
        for word in words:
            if word in negation_words:
                in_negation = True
                result.append(word)
            elif word in ['.', '!', '?', ',', ';', ':', ')', ']']:
                # End negation scope at punctuation
                in_negation = False
                result.append(word)
            elif in_negation and word not in ['and', 'or', 'the', 'a', 'an', 'to', 'of', 'in']:
                # Mark negated content words
                result.append(word + '_NEG')
            else:
                result.append(word)
        
        return ' '.join(result)
    except Exception as e:
        logger.error(f"Error in handle_negations: {e}")
        # Fallback to simple preprocessing
        return text.lower()

def process_contrast_markers(text):
    """
    Enhance handling of contrast markers like 'but', 'however'.
    Adds both the original and specially processed versions to the training data.
    """
    contrast_markers = ['but', 'however', 'although', 'though', 'despite', 'yet', 'nevertheless', 'still', 
                        'while', 'except', 'contrary', 'rather', 'instead']
    
    # Check if text contains contrast markers
    for marker in contrast_markers:
        marker_pattern = r'\b' + marker + r'\b'
        if re.search(marker_pattern, text, re.IGNORECASE):
            # Split text at the contrast marker
            parts = re.split(marker_pattern, text, flags=re.IGNORECASE)
            if len(parts) > 1:
                # Add marker back to second part
                parts[1] = marker + parts[1]
                
                # Also create a version with special markers
                marked_text = parts[0] + " CONTRASTMARKER " + parts[1]
                return marked_text
    
    return text  # No contrast marker found

def clean_text(text):
    """Enhanced text cleaning with entity recognition and negation handling"""
    try:
        # Convert to lowercase
        text = text.lower()
        
        # Handle potential movie titles before lowercasing
        text_with_titles = identify_movie_titles(text)
        
        # Remove special characters but keep apostrophes for negations
        text = re.sub(r'[^\w\s\']', ' ', text)
        
        # Enhance contrast marker handling
        text = process_contrast_markers(text)
        
        # Apply negation handling
        text = handle_negations(text)
        
        # Remove extra whitespace
        text = re.sub(r'\s+', ' ', text).strip()
        return text
    except Exception as e:
        logger.error(f"Error in clean_text: {e}")
        # Simple fallback cleaning
        return text.lower().strip()

def download_and_prepare_datasets():
    """Download and prepare additional datasets"""
    dataset_dir = os.path.join(os.getcwd(), 'datasets')
    os.makedirs(dataset_dir, exist_ok=True)
    
    # Dictionary to store all our datasets
    datasets = {}
    
    # First add NLTK movie reviews as before
    print("Preparing data from NLTK movie reviews...")
    nltk_docs = []
    for category in movie_reviews.categories():
        for fileid in movie_reviews.fileids(category):
            try:
                text = ' '.join(movie_reviews.words(fileid))
                cleaned_text = clean_text(text)
                nltk_docs.append({
                    'text': cleaned_text,
                    'sentiment': 1 if category == 'pos' else 0
                })
            except Exception as e:
                logger.error(f"Error processing NLTK file {fileid}: {e}")
    
    datasets['nltk_movie_reviews'] = pd.DataFrame(nltk_docs)
    print(f"NLTK dataset: {len(datasets['nltk_movie_reviews'])} reviews")
    
    # Download and prepare IMDB Large Movie Review Dataset
    imdb_path = os.path.join(dataset_dir, 'imdb')
    if not os.path.exists(imdb_path):
        print("Downloading IMDB Large Movie Review Dataset...")
        imdb_url = 'http://ai.stanford.edu/~amaas/data/sentiment/aclImdb_v1.tar.gz'
        try:
            # Download the file
            with urllib.request.urlopen(imdb_url) as response:
                with tarfile.open(fileobj=io.BytesIO(response.read()), mode='r:gz') as tar:
                    print("Extracting IMDB dataset...")
                    # Extract only training data to save space
                    members = [m for m in tar.getmembers() if 'train/' in m.name and not m.name.endswith('/')]
                    for member in tqdm(members, desc="Extracting files"):
                        tar.extract(member, path=dataset_dir)
            
            # Process the extracted files
            imdb_docs = []
            for sentiment, label in [('pos', 1), ('neg', 0)]:
                dir_path = os.path.join(dataset_dir, 'aclImdb', 'train', sentiment)
                if os.path.exists(dir_path):
                    files = os.listdir(dir_path)
                    for file in tqdm(files[:12500], desc=f"Processing IMDB {sentiment}"):  # Limit to 12,500 per class
                        with open(os.path.join(dir_path, file), 'r', encoding='utf-8') as f:
                            text = f.read()
                            cleaned_text = clean_text(text)
                            imdb_docs.append({
                                'text': cleaned_text,
                                'sentiment': label
                            })
            
            datasets['imdb'] = pd.DataFrame(imdb_docs)
            print(f"IMDB dataset: {len(datasets['imdb'])} reviews")
        except Exception as e:
            print(f"Error downloading IMDB dataset: {e}")
            print("Continuing without IMDB dataset")
    
    # Download Twitter sentiment data
    twitter_path = os.path.join(dataset_dir, 'twitter')
    if not os.path.exists(twitter_path):
        os.makedirs(twitter_path, exist_ok=True)
        print("Downloading Twitter Sentiment Dataset (smaller subset)...")
        try:
            # Use a smaller dataset version for practical purposes
            twitter_small_url = 'https://raw.githubusercontent.com/mnqu/datasets/master/twitter/train.small.txt'
            urllib.request.urlretrieve(twitter_small_url, os.path.join(twitter_path, 'twitter_small.txt'))
            
            # Process the twitter data
            twitter_docs = []
            with open(os.path.join(twitter_path, 'twitter_small.txt'), 'r', encoding='utf-8', errors='ignore') as f:
                for line in tqdm(f, desc="Processing Twitter data"):
                    try:
                        fields = line.strip().split('\t')
                        if len(fields) >= 2:
                            sentiment_str = fields[0]
                            text = fields[1]
                            
                            # Convert sentiment to binary (0=negative, 1=positive)
                            sentiment = 1 if sentiment_str == '1' else 0
                            
                            # Clean and add to dataset
                            cleaned_text = clean_text(text)
                            twitter_docs.append({
                                'text': cleaned_text,
                                'sentiment': sentiment
                            })
                    except Exception as e:
                        continue  # Skip problematic lines
            
            # Balance the classes and limit size
            pos_tweets = [doc for doc in twitter_docs if doc['sentiment'] == 1][:10000]
            neg_tweets = [doc for doc in twitter_docs if doc['sentiment'] == 0][:10000]
            twitter_docs = pos_tweets + neg_tweets
            random.shuffle(twitter_docs)
            
            datasets['twitter'] = pd.DataFrame(twitter_docs)
            print(f"Twitter dataset: {len(datasets['twitter'])} tweets")
        except Exception as e:
            print(f"Error downloading Twitter dataset: {e}")
            print("Continuing without Twitter dataset")
    
    # Combine all datasets
    return datasets

# ==== END DATASET PREPARATION CODE ====

print("Preparing data from NLTK movie reviews...")

# Prepare data from NLTK movie reviews with better cleaning
documents = []
for category in movie_reviews.categories():
    for fileid in movie_reviews.fileids(category):
        try:
            text = ' '.join(movie_reviews.words(fileid))
            cleaned_text = clean_text(text)
            documents.append({
                'text': cleaned_text,
                'sentiment': 1 if category == 'pos' else 0
            })
        except Exception as e:
            logger.error(f"Error processing file {fileid}: {e}")

# Convert to DataFrame
df = pd.DataFrame(documents)

# Display dataset info
print(f"Dataset loaded: {len(df)} reviews")
print(f"Positive reviews: {sum(df['sentiment'])}")
print(f"Negative reviews: {len(df) - sum(df['sentiment'])}")

# ===== NEW NEUTRAL TRAINING EXAMPLES =====

print("Adding neutral training examples...")
neutral_examples = [
    # Explicitly neutral examples with middle-ground sentiment
    {"text": "It wasn't bad, but it wasn't great either. Just another average Hollywood film.", "sentiment": 0.5},
    {"text": "Somewhat entertaining but forgettable.", "sentiment": 0.5},
    {"text": "Neither impressive nor terrible.", "sentiment": 0.5},
    {"text": "Had some good moments and some boring parts.", "sentiment": 0.5},
    {"text": "Average production with standard performances.", "sentiment": 0.5},
    {"text": "Passable entertainment for a rainy day.", "sentiment": 0.5},
    {"text": "Middle-of-the-road story with adequate acting.", "sentiment": 0.5},
    {"text": "Not worth recommending but not a complete waste of time.", "sentiment": 0.5},
    {"text": "Mediocre at best, but not terrible.", "sentiment": 0.5},
    {"text": "Watchable but immediately forgettable.", "sentiment": 0.5},
    {"text": "Functional but unremarkable.", "sentiment": 0.5},
    {"text": "Not particularly good or bad, just there.", "sentiment": 0.5},
    {"text": "A film that exists, nothing more to say about it.", "sentiment": 0.5},
    {"text": "Has a beginning, middle, and end. That's all I can say positively.", "sentiment": 0.5},
    {"text": "The type of movie you watch on an airplane and then forget.", "sentiment": 0.5},
    {"text": "It was fine. Not great, not terrible, just fine.", "sentiment": 0.5},
    {"text": "Two hours of content that neither impresses nor offends.", "sentiment": 0.5},
    {"text": "A movie that happened. I watched it. That's all.", "sentiment": 0.5},
    {"text": "Some parts were good, others were not.", "sentiment": 0.5},
    {"text": "If you're bored enough, you might enjoy it.", "sentiment": 0.5},
    
    # Mixed sentiment balanced examples
    {"text": "Brilliant cinematography but weak storyline.", "sentiment": 0.5},
    {"text": "Great acting but terrible directing.", "sentiment": 0.5},
    {"text": "The first half was amazing, the second half fell apart.", "sentiment": 0.5},
    {"text": "Visually stunning but emotionally empty.", "sentiment": 0.5},
    {"text": "Good performances wasted on a bad script.", "sentiment": 0.5},
    {"text": "I liked the characters but hated the plot.", "sentiment": 0.5},
    {"text": "The action scenes were exciting but the dialogue was painful.", "sentiment": 0.5},
    {"text": "Beautiful soundtrack accompanying a mediocre film.", "sentiment": 0.5},
    {"text": "Excellent premise, disappointing execution.", "sentiment": 0.5},
    {"text": "The lead actor was amazing, everyone else was terrible.", "sentiment": 0.5},
    
    # Slightly positive leaning but still neutral
    {"text": "Not bad, slightly above average.", "sentiment": 0.6},
    {"text": "Decent enough but nothing special.", "sentiment": 0.6},
    {"text": "Worth a watch if you have nothing better to do.", "sentiment": 0.6},
    {"text": "Somewhat enjoyable despite its flaws.", "sentiment": 0.6},
    {"text": "Mostly competent filmmaking with a few good moments.", "sentiment": 0.6},
    {"text": "Kind of entertaining in a forgettable way.", "sentiment": 0.6},
    {"text": "Slightly better than I expected, but that's not saying much.", "sentiment": 0.6},
    {"text": "Has its moments, though not many.", "sentiment": 0.6},
    {"text": "Okay for what it is, I guess.", "sentiment": 0.6},
    {"text": "Not a complete waste of time, but close.", "sentiment": 0.6},
    
    # Slightly negative leaning but still neutral
    {"text": "Below average but not terrible.", "sentiment": 0.4},
    {"text": "Mostly boring with a few decent scenes.", "sentiment": 0.4},
    {"text": "Disappointing given the talent involved.", "sentiment": 0.4},
    {"text": "Not as good as it could have been.", "sentiment": 0.4},
    {"text": "More mediocre than bad, but still not good.", "sentiment": 0.4},
    {"text": "I didn't hate it, but I certainly didn't like it.", "sentiment": 0.4},
    {"text": "Underachieving and forgettable.", "sentiment": 0.4},
    {"text": "I've seen worse, but that's not saying much.", "sentiment": 0.4},
    {"text": "Uninspired but not offensively bad.", "sentiment": 0.4},
    {"text": "The kind of film that makes you check your watch repeatedly.", "sentiment": 0.4},
]

# Process neutral examples with the new preprocessing
for example in neutral_examples:
    example["text"] = clean_text(example["text"])

# Mixed sentiment examples focusing on contrast markers
print("Adding mixed sentiment examples...")
mixed_sentiment_examples = [
    # Positive despite negative elements (focus on contrast markers)
    {"text": "not the best plot but enjoyable characters", "sentiment": 0.7},
    {"text": "ordinary story with exceptional cinematography", "sentiment": 0.7},
    {"text": "weak script but excellent performances", "sentiment": 0.7},
    {"text": "slow pacing, however the ending was worth it", "sentiment": 0.7},
    {"text": "predictable at times but overall a great movie", "sentiment": 0.8},
    {"text": "despite its flaws, the film was truly entertaining", "sentiment": 0.8},
    {"text": "somewhat clichéd yet thoroughly enjoyable", "sentiment": 0.8},
    {"text": "the plot was simple, still i was entertained", "sentiment": 0.7},
    {"text": "not perfect by any means, but definitely worth watching", "sentiment": 0.8},
    {"text": "it's an interesting movie, i like the characters, but the plot is very ordinary", "sentiment": 0.6},
    {"text": "interesting movie, i liked the characters but the subject was too ordinary", "sentiment": 0.6},
    
    # Negative despite positive elements
    {"text": "good acting but boring plot", "sentiment": 0.3},
    {"text": "beautiful visuals, however the story made no sense", "sentiment": 0.3},
    {"text": "interesting concept, poor execution", "sentiment": 0.3},
    {"text": "talented cast, but completely wasted on a terrible script", "sentiment": 0.2},
    {"text": "started well, although it fell apart in the second half", "sentiment": 0.3},
    {"text": "nice cinematography but the plot was too confusing", "sentiment": 0.3},
    {"text": "great special effects but no substance whatsoever", "sentiment": 0.2},
    {"text": "good performances can't save this disappointing film", "sentiment": 0.2},
    {"text": "had potential but failed to deliver", "sentiment": 0.3},
    {"text": "some good moments, nevertheless mostly tedious", "sentiment": 0.3},
]

# Process all mixed sentiment examples with the new preprocessing
for example in mixed_sentiment_examples:
    example["text"] = clean_text(example["text"])

# Nuanced opinion examples (moderate sentiments)
print("Adding nuanced opinion examples...")
nuanced_examples = [
    # Moderately positive
    {"text": "decent film that entertains without being groundbreaking", "sentiment": 0.7},
    {"text": "solid performances in an otherwise ordinary movie", "sentiment": 0.7},
    {"text": "reasonably entertaining for what it is", "sentiment": 0.7},
    {"text": "pleasant enough way to spend two hours", "sentiment": 0.7},
    {"text": "competently made with a few standout moments", "sentiment": 0.7},
    {"text": "satisfying if not spectacular", "sentiment": 0.7},
    {"text": "pretty good for this type of film", "sentiment": 0.7},
    {"text": "above average entertainment value", "sentiment": 0.7},
    {"text": "not amazing but definitely worth watching", "sentiment": 0.7},
    
    # Moderately negative
    {"text": "somewhat disappointing given the talent involved", "sentiment": 0.3},
    {"text": "not terrible but certainly not good", "sentiment": 0.3},
    {"text": "mediocre at best despite a few good scenes", "sentiment": 0.3},
    {"text": "slightly below average film experience", "sentiment": 0.3},
    {"text": "more tedious than outright bad", "sentiment": 0.3},
    {"text": "forgettable though not completely without merit", "sentiment": 0.3},
    {"text": "unremarkable film that breaks no new ground", "sentiment": 0.3},
    {"text": "watchable but frustratingly flawed", "sentiment": 0.3},
    {"text": "not as good as it could have been", "sentiment": 0.3},
]

# Process all nuanced examples with the new preprocessing
for example in nuanced_examples:
    example["text"] = clean_text(example["text"])

# Movie-specific vocabulary and domain examples
print("Adding movie domain-specific examples...")
movie_domain_examples = [
    # Positive
    {"text": "excellent character development throughout the film", "sentiment": 0.9},
    {"text": "the cinematography was absolutely breathtaking", "sentiment": 0.9},
    {"text": "perfectly paced with no wasted scenes", "sentiment": 0.9},
    {"text": "the dialogue was sharp and witty", "sentiment": 0.9},
    {"text": "brilliant directorial debut", "sentiment": 0.9},
    {"text": "the screenplay intelligently adapts the novel", "sentiment": 0.9},
    {"text": "stellar ensemble cast with perfect chemistry", "sentiment": 0.9},
    {"text": "innovative visual effects that serve the story", "sentiment": 0.9},
    {"text": "the score beautifully complements each scene", "sentiment": 0.9},
    {"text": "masterful editing creates perfect tension", "sentiment": 0.9},
    {"text": "stunning production design creates an immersive world", "sentiment": 0.9},
    {"text": "the plot twists were unexpected yet satisfying", "sentiment": 0.9},
    
    # Negative
    {"text": "flat characters with no development", "sentiment": 0.1},
    {"text": "choppy editing made the narrative hard to follow", "sentiment": 0.1},
    {"text": "the pacing drags through the middle act", "sentiment": 0.1},
    {"text": "overreliance on cgi instead of practical effects", "sentiment": 0.1},
    {"text": "ham-fisted dialogue that no actor could deliver well", "sentiment": 0.1},
    {"text": "pretentious arthouse techniques without substance", "sentiment": 0.1},
    {"text": "the third act falls apart completely", "sentiment": 0.1},
    {"text": "uninspired direction brings nothing new to the genre", "sentiment": 0.1},
    {"text": "wooden acting from the entire cast", "sentiment": 0.1},
    {"text": "heavy-handed symbolism lacks subtlety", "sentiment": 0.1},
    {"text": "the plot holes are impossible to ignore", "sentiment": 0.1},
    {"text": "derivative script borrows from better films", "sentiment": 0.1},
]

# Process all domain-specific examples with the new preprocessing
for example in movie_domain_examples:
    example["text"] = clean_text(example["text"])

# Original specialized examples from previous version
print("Adding specialized negation examples...")
negation_examples = [
    {"text": "i don't think it was boring", "sentiment": 0.7},
    {"text": "i don't hate this movie", "sentiment": 0.7},
    {"text": "this movie wasn't bad at all", "sentiment": 0.7},
    {"text": "this wasn't as terrible as people say", "sentiment": 0.7},
    {"text": "not a bad film", "sentiment": 0.7},
    {"text": "not terrible", "sentiment": 0.7},
    {"text": "not the worst i've seen", "sentiment": 0.7},
    {"text": "didn't dislike it", "sentiment": 0.7},
    {"text": "isn't awful", "sentiment": 0.7},
    {"text": "can't complain about this movie", "sentiment": 0.7},
    {"text": "i don't think it was good", "sentiment": 0.3},
    {"text": "i don't like this movie", "sentiment": 0.3},
    {"text": "this movie wasn't great at all", "sentiment": 0.3},
    {"text": "this wasn't as good as people say", "sentiment": 0.3},
    {"text": "not a good film", "sentiment": 0.3},
    {"text": "not amazing", "sentiment": 0.3},
    {"text": "not the best i've seen", "sentiment": 0.3},
    {"text": "didn't enjoy it", "sentiment": 0.3},
    {"text": "isn't great", "sentiment": 0.3},
    {"text": "can't say i enjoyed this movie", "sentiment": 0.3},
]

print("Adding mental health/emotional content examples...")
emotional_examples = [
    {"text": "i want to hurt myself", "sentiment": 0.1},
    {"text": "i feel like killing myself", "sentiment": 0.1},
    {"text": "i am worthless", "sentiment": 0.1},
    {"text": "i hate myself", "sentiment": 0.1},
    {"text": "everything feels hopeless", "sentiment": 0.1},
    {"text": "i am so depressed", "sentiment": 0.1},
    {"text": "nobody cares about me", "sentiment": 0.1},
    {"text": "i feel so alone", "sentiment": 0.1},
    {"text": "i'm better off dead", "sentiment": 0.1},
    {"text": "i can't take it anymore", "sentiment": 0.1},
    {"text": "life is meaningless", "sentiment": 0.1},
    {"text": "no one would miss me", "sentiment": 0.1},
    {"text": "i'm a burden to everyone", "sentiment": 0.1},
    {"text": "i'm so anxious all the time", "sentiment": 0.1},
    {"text": "i'm a failure", "sentiment": 0.1},
]

print("Adding film terminology examples...")
film_examples = [
    {"text": "this film is so underrated", "sentiment": 0.8},
    {"text": "this is a cult classic", "sentiment": 0.8},
    {"text": "this movie is a hidden gem", "sentiment": 0.8},
    {"text": "king of comedy is brilliant", "sentiment": 0.8},
    {"text": "this comedy is hilarious", "sentiment": 0.8},
    {"text": "a thought-provoking film", "sentiment": 0.8},
    {"text": "this movie is overrated", "sentiment": 0.2},
    {"text": "this film is pretentious", "sentiment": 0.2},
    {"text": "the comedy falls flat", "sentiment": 0.2},
    {"text": "heavy-handed film", "sentiment": 0.2},
]

print("Adding movie reference examples...")
movie_reference_examples = [
    {"text": "King of Comedy wasn't the best movie I've ever seen, but it was alright", "sentiment": 0.6},
    {"text": "Citizen Kane is considered a masterpiece, but I found it boring", "sentiment": 0.4},
    {"text": "The Godfather is my favorite movie of all time", "sentiment": 0.9},
    {"text": "Star Wars was revolutionary for its time", "sentiment": 0.8},
    {"text": "Titanic didn't deserve all those Oscars", "sentiment": 0.3},
    {"text": "Pulp Fiction has brilliant dialogue", "sentiment": 0.9},
    {"text": "The Room is so bad it's actually entertaining", "sentiment": 0.6},
    {"text": "Casablanca remains a timeless classic", "sentiment": 0.9},
    {"text": "Gone with the Wind hasn't aged well", "sentiment": 0.4},
    {"text": "The Matrix revolutionized action films", "sentiment": 0.8},
]

print("Adding high-confidence calibration examples...")
obvious_examples = [
    {"text": "this movie was amazing fantastic wonderful incredible brilliant loved it", "sentiment": 1.0},
    {"text": "best film ever seen perfect outstanding brilliant masterpiece", "sentiment": 1.0},
    {"text": "excellent superb magnificent outstanding remarkable phenomenal", "sentiment": 1.0},
    {"text": "i absolutely loved every second of this film", "sentiment": 1.0},
    {"text": "this movie brings me so much joy every time i watch it", "sentiment": 1.0},
    {"text": "one of the greatest films ever made without question", "sentiment": 1.0},
    {"text": "terrible awful horrible worst garbage waste of time", "sentiment": 0.0},
    {"text": "dreadful pathetic disappointing boring stupid terrible", "sentiment": 0.0},
    {"text": "hate disliked awful terrible horrible worst ever", "sentiment": 0.0},
    {"text": "i absolutely hated every second of this film", "sentiment": 0.0},
    {"text": "this movie was painful to watch and completely worthless", "sentiment": 0.0},
    {"text": "one of the worst films ever made without question", "sentiment": 0.0},
]

# Process all specialized examples with the new preprocessing
for examples in [negation_examples, emotional_examples, film_examples, movie_reference_examples, obvious_examples]:
    for example in examples:
        example["text"] = clean_text(example["text"])

# Combine all the specialized examples
all_specialized_examples = pd.DataFrame(
    neutral_examples +
    mixed_sentiment_examples + 
    nuanced_examples + 
    movie_domain_examples + 
    negation_examples +
    emotional_examples + 
    film_examples +
    movie_reference_examples
)

# Additional mixed sentiment examples with focus on ordinary/neutral phrases
print("Adding additional mixed/nuanced examples...")
additional_mixed_examples = [
    # Neutral/mixed sentiment with slightly positive lean
    {"text": "ordinary plot but decent acting", "sentiment": 0.6},
    {"text": "not bad for a regular friday night movie", "sentiment": 0.6},
    {"text": "standard action film with some good moments", "sentiment": 0.6},
    {"text": "typical rom-com but entertaining enough", "sentiment": 0.6},
    {"text": "nothing special but watchable", "sentiment": 0.6},
    {"text": "kind of predictable but enjoyable", "sentiment": 0.6},
    {"text": "average film, still worth seeing once", "sentiment": 0.6},
    {"text": "not amazing but better than expected", "sentiment": 0.6},
    {"text": "quite ordinary but likable characters", "sentiment": 0.6},
    {"text": "pretty basic plot with some interesting twists", "sentiment": 0.6},
    {"text": "familiar storyline but well executed", "sentiment": 0.6},
    {"text": "common theme but good execution", "sentiment": 0.6},
    {"text": "not groundbreaking but entertaining", "sentiment": 0.6},
    {"text": "conventional but well-made", "sentiment": 0.6},
    {"text": "won't win awards but keeps your attention", "sentiment": 0.6},
    
    # Neutral/mixed sentiment with slightly negative lean
    {"text": "decent acting couldn't save the boring plot", "sentiment": 0.4},
    {"text": "nice visuals but too generic overall", "sentiment": 0.4},
    {"text": "had potential but too ordinary in execution", "sentiment": 0.4},
    {"text": "interesting premise delivered in a mundane way", "sentiment": 0.4},
    {"text": "nothing terrible but nothing special either", "sentiment": 0.5},
    {"text": "mediocre despite some good performances", "sentiment": 0.4},
    {"text": "standard fare that fails to engage", "sentiment": 0.4},
    {"text": "too conventional to be memorable", "sentiment": 0.4},
    {"text": "acceptable performance but forgettable script", "sentiment": 0.4},
    {"text": "fine acting in an otherwise bland movie", "sentiment": 0.4},
    {"text": "typical Hollywood formula that gets tiresome", "sentiment": 0.4},
    {"text": "neither great nor terrible, just plain boring", "sentiment": 0.5},
    {"text": "not the worst but still disappointing", "sentiment": 0.4},
    {"text": "passable entertainment but missed opportunities", "sentiment": 0.4},
    {"text": "technically competent but lacks creativity", "sentiment": 0.4},
]

for example in additional_mixed_examples:
    example["text"] = clean_text(example["text"])

# Add these examples multiple times (they're crucial for our improvements)
additional_df = pd.DataFrame(additional_mixed_examples)

# Create a DataFrame with neutral examples specifically
neutral_df = pd.DataFrame(neutral_examples)

# Loading additional datasets
print("Loading additional datasets...")
all_datasets = download_and_prepare_datasets()

# Combine datasets with different weights
print("Combining datasets...")
combined_df = pd.DataFrame()

# Add all datasets with appropriate sampling and weighting
for name, dataset_df in all_datasets.items():
    print(f"Adding {name} with {len(dataset_df)} examples")
    if name == 'nltk_movie_reviews':
        # Add NLTK dataset multiple times (higher weight)
        for _ in range(3):
            combined_df = pd.concat([combined_df, dataset_df], ignore_index=True)
    elif name == 'imdb':
        # Sample from IMDB to balance with NLTK
        sampled_df = dataset_df.sample(min(len(dataset_df), 20000))
        combined_df = pd.concat([combined_df, sampled_df], ignore_index=True)
    elif name == 'twitter':
        # Use less twitter data to not overwhelm movie reviews
        sampled_df = dataset_df.sample(min(len(dataset_df), 15000))
        combined_df = pd.concat([combined_df, sampled_df], ignore_index=True)

# Add our specialized examples to the combined dataframe
for _ in range(3):  # Adding specialized examples multiple times
    combined_df = pd.concat([combined_df, all_specialized_examples], ignore_index=True)
    combined_df = pd.concat([combined_df, pd.DataFrame(mixed_sentiment_examples)], ignore_index=True)
    combined_df = pd.concat([combined_df, pd.DataFrame(obvious_examples)], ignore_index=True)
    combined_df = pd.concat([combined_df, additional_df], ignore_index=True)

# Add neutral examples many times to ensure they're well represented
for _ in range(10):  # Adding neutral examples many times
    combined_df = pd.concat([combined_df, neutral_df], ignore_index=True)

# Use the combined dataset
if len(combined_df) > 0:
    df = combined_df
    print(f"Using combined dataset with {len(df)} examples")
    
    # Handle the 0.5 neutral sentiment values
    # Convert to binary for training (but keep originals for calibration)
    df['original_sentiment'] = df['sentiment'].copy()
    
    # Convert sentiment values to binary for training
    # Values less than 0.4 → 0 (negative)
    # Values greater than 0.6 → 1 (positive)
    # Values 0.4-0.6 → randomly assigned 0 or 1 with decreasing probability toward middle
    def convert_to_binary(value):
        if value <= 0.4:
            return 0
        elif value >= 0.6:
            return 1
        elif value == 0.5:
            # Exactly 0.5 is evenly distributed
            return random.randint(0, 1)
        elif 0.4 < value < 0.5:
            # 0.4-0.5 range has increasing probability of being 0
            prob_zero = (0.5 - value) * 10  # Ranges from 0.1 to 0.4
            return 0 if random.random() < prob_zero else 1
        else:  # 0.5 < value < 0.6
            # 0.5-0.6 range has increasing probability of being 1
            prob_one = (value - 0.5) * 10  # Ranges from 0.1 to 0.4
            return 1 if random.random() < prob_one else 0
    
    df['sentiment'] = df['sentiment'].apply(convert_to_binary)
    
    print(f"Positive examples: {sum(df['sentiment'])}")
    print(f"Negative examples: {len(df) - sum(df['sentiment'])}")

# Split text and labels
texts = df['text'].values
labels = df['sentiment'].values

# Create a calibration set separately (includes original sentiment scores)
calibration_indices = []
if 'original_sentiment' in df.columns:
    # Find examples with neutral or near-neutral sentiment
    neutral_indices = df.index[df['original_sentiment'].between(0.4, 0.6)].tolist()
    # Add some clearly positive/negative examples
    positive_indices = df.index[df['original_sentiment'] > 0.8].tolist()
    negative_indices = df.index[df['original_sentiment'] < 0.2].tolist()
    
    # Randomly sample from each group
    if neutral_indices:
        calibration_indices.extend(random.sample(neutral_indices, min(len(neutral_indices), 1000)))
    if positive_indices:
        calibration_indices.extend(random.sample(positive_indices, min(len(positive_indices), 500)))
    if negative_indices:
        calibration_indices.extend(random.sample(negative_indices, min(len(negative_indices), 500)))
    
    # Create calibration dataset
    calibration_texts = df.iloc[calibration_indices]['text'].values
    calibration_labels = df.iloc[calibration_indices]['sentiment'].values
    calibration_original = df.iloc[calibration_indices]['original_sentiment'].values

# Split into training and testing sets
X_train, X_test, y_train, y_test = train_test_split(
    texts, labels, test_size=0.2, random_state=42, stratify=labels
)

print(f"Training set size: {len(X_train)}")
print(f"Testing set size: {len(X_test)}")

# Initialize sentiment lexicon features
print("Initializing sentiment lexicon features...")
lexicon = SentimentLexiconFeatures()

# Function to extract lexicon features
def extract_lexicon_features(texts):
    """Extract sentiment lexicon features for a list of texts"""
    features = []
    for text in texts:
        lexicon_features = lexicon.extract_all_features(text)
        features.append(lexicon_features)
    return features

# Extract lexicon features from training and testing data
print("Extracting lexicon features...")
X_train_lexicon = extract_lexicon_features(X_train)
X_test_lexicon = extract_lexicon_features(X_test)
if calibration_indices:
    X_calibration_lexicon = extract_lexicon_features(calibration_texts)

# After extracting lexicon features, verify all values are non-negative
print("Verifying lexicon features are non-negative for MultinomialNB...")
for features_dict in X_train_lexicon:
    for key, value in list(features_dict.items()):
        if isinstance(value, (int, float)) and value < 0:
            features_dict[key] = 0.0

for features_dict in X_test_lexicon:
    for key, value in list(features_dict.items()):
        if isinstance(value, (int, float)) and value < 0:
            features_dict[key] = 0.0

if calibration_indices:
    for features_dict in X_calibration_lexicon:
        for key, value in list(features_dict.items()):
            if isinstance(value, (int, float)) and value < 0:
                features_dict[key] = 0.0

# Create a DictVectorizer to transform lexicon features
dict_vectorizer = DictVectorizer()
X_train_lexicon_vec = dict_vectorizer.fit_transform(X_train_lexicon)
X_test_lexicon_vec = dict_vectorizer.transform(X_test_lexicon)
if calibration_indices:
    X_calibration_lexicon_vec = dict_vectorizer.transform(X_calibration_lexicon)

# Verify there are no negative values in the lexicon features
if X_train_lexicon_vec.data.min() < 0:
    print("Warning: Negative values found in lexicon features, setting them to 0...")
    X_train_lexicon_vec.data[X_train_lexicon_vec.data < 0] = 0.0
    
if X_test_lexicon_vec.data.min() < 0:
    X_test_lexicon_vec.data[X_test_lexicon_vec.data < 0] = 0.0

if calibration_indices and X_calibration_lexicon_vec.data.min() < 0:
    X_calibration_lexicon_vec.data[X_calibration_lexicon_vec.data < 0] = 0.0

# Create feature extractors
print("Creating feature extractors...")
stop_words = 'english'

# Use a smaller feature set for performance but still comprehensive
count_vectorizer = CountVectorizer(
    max_features=10000,  # Reduced from 15000 for better performance
    min_df=2,
    max_df=0.9,
    ngram_range=(1, 2),  # Reduced from (1, 3) for better performance
    stop_words=stop_words,
    strip_accents='unicode'
)

# TfidfVectorizer with improved parameters
tfidf_vectorizer = TfidfVectorizer(
    max_features=10000,  # Reduced from 15000 for better performance
    min_df=2,
    max_df=0.9,
    ngram_range=(1, 2),  # Reduced from (1, 3) for better performance
    stop_words=stop_words,
    norm='l2',
    use_idf=True,
    smooth_idf=True,
    sublinear_tf=True
)

# Transform text data
X_train_counts = count_vectorizer.fit_transform(X_train)
X_test_counts = count_vectorizer.transform(X_test)

X_train_tfidf = tfidf_vectorizer.fit_transform(X_train)
X_test_tfidf = tfidf_vectorizer.transform(X_test)

if calibration_indices:
    X_calibration_counts = count_vectorizer.transform(calibration_texts)
    X_calibration_tfidf = tfidf_vectorizer.transform(calibration_texts)

print(f"CountVectorizer vocabulary size: {len(count_vectorizer.vocabulary_)}")
print(f"TfidfVectorizer vocabulary size: {len(tfidf_vectorizer.vocabulary_)}")
print(f"DictVectorizer feature count: {X_train_lexicon_vec.shape[1]}")

# Save feature information for diagnostic purposes
feature_info = {
    'count_vectorizer_feature_count': X_train_counts.shape[1],
    'tfidf_vectorizer_feature_count': X_train_tfidf.shape[1],
    'dict_vectorizer_feature_count': X_train_lexicon_vec.shape[1],
    'combined_naive_bayes_feature_count': X_train_counts.shape[1] + X_train_lexicon_vec.shape[1],
    'combined_logistic_regression_feature_count': X_train_tfidf.shape[1] + X_train_lexicon_vec.shape[1],
}

with open('models/feature_info.txt', 'w') as f:
    for key, value in feature_info.items():
        f.write(f"{key}: {value}\n")

# Combine features for Naive Bayes
X_train_combined_nb = hstack([X_train_counts, X_train_lexicon_vec])
X_test_combined_nb = hstack([X_test_counts, X_test_lexicon_vec])
if calibration_indices:
    X_calibration_combined_nb = hstack([X_calibration_counts, X_calibration_lexicon_vec])

# Combine features for Logistic Regression
X_train_combined_lr = hstack([X_train_tfidf, X_train_lexicon_vec])
X_test_combined_lr = hstack([X_test_tfidf, X_test_lexicon_vec])
if calibration_indices:
    X_calibration_combined_lr = hstack([X_calibration_tfidf, X_calibration_lexicon_vec])

# Make sure there are no negative values in the NB training data
if X_train_combined_nb.data.min() < 0:
    print("Warning: Negative values found in combined NB features, setting them to 0...")
    X_train_combined_nb.data[X_train_combined_nb.data < 0] = 0.0

if X_test_combined_nb.data.min() < 0:
    X_test_combined_nb.data[X_test_combined_nb.data < 0] = 0.0

if calibration_indices and X_calibration_combined_nb.data.min() < 0:
    X_calibration_combined_nb.data[X_calibration_combined_nb.data < 0] = 0.0

# Train Naive Bayes model with improved hyperparameters
print("Training Naive Bayes model...")
nb_model = MultinomialNB(
    alpha=0.1,  # Lower alpha for more confident predictions
)
nb_model.fit(X_train_combined_nb, y_train)

# Train Logistic Regression model with different hyperparameters
print("Training Logistic Regression model...")
lr_model = LogisticRegression(
    C=1.0,  # Regularization strength
    max_iter=1000,
    class_weight='balanced',
    solver='liblinear'
)
lr_model.fit(X_train_combined_lr, y_train)

# Evaluate models
nb_predictions = nb_model.predict(X_test_combined_nb)
lr_predictions = lr_model.predict(X_test_combined_lr)

# Calculate calibrated probabilities for both models
# This helps make probabilities more reflective of true confidence
print("Calibrating model probabilities...")
calibrated_nb = CalibratedClassifierCV(nb_model, cv='prefit')
calibrated_lr = CalibratedClassifierCV(lr_model, cv='prefit')

if calibration_indices:
    # Use special calibration set
    calibrated_nb.fit(X_calibration_combined_nb, calibration_labels)
    calibrated_lr.fit(X_calibration_combined_lr, calibration_labels)
else:
    # Fall back to test set
    calibrated_nb.fit(X_test_combined_nb, y_test)
    calibrated_lr.fit(X_test_combined_lr, y_test)

print("\n--- Model Evaluation ---")
print("Naive Bayes Accuracy:", accuracy_score(y_test, nb_predictions))
print("\nNaive Bayes Classification Report:")
print(classification_report(y_test, nb_predictions))
print("\nNaive Bayes Confusion Matrix:")
print(confusion_matrix(y_test, nb_predictions))

print("\nLogistic Regression Accuracy:", accuracy_score(y_test, lr_predictions))
print("\nLogistic Regression Classification Report:")
print(classification_report(y_test, lr_predictions))
print("\nLogistic Regression Confusion Matrix:")
print(confusion_matrix(y_test, lr_predictions))

# Test the models on specific challenging examples
print("\n--- Testing on Challenging Examples ---")
challenge_examples = [
    "It wasn't bad, but it wasn't great either. Just another average Hollywood film.",
    "I don't think it was boring",
    "King of Comedy is so underrated",
    "I want to hurt myself",
    "I loved this movie, it was awesome!",
    "This was the worst film I've ever seen, terrible acting.",
    "It's an interesting movie, I like the characters, but the plot is very ordinary.",
    "The cinematography and music were fantastic, though the story was a bit predictable."
]

# Process the challenge examples with negation handling
challenge_examples_processed = [clean_text(text) for text in challenge_examples]

print("Original vs Processed examples:")
for orig, proc in zip(challenge_examples, challenge_examples_processed):
    print(f"Original: \"{orig}\"")
    print(f"Processed: \"{proc}\"")
    print()

# Extract lexicon features for challenge examples
challenge_lexicon = extract_lexicon_features(challenge_examples_processed)
challenge_lexicon_vec = dict_vectorizer.transform(challenge_lexicon)

# Make sure there are no negative values in the challenge features
if challenge_lexicon_vec.data.min() < 0:
    challenge_lexicon_vec.data[challenge_lexicon_vec.data < 0] = 0.0

print("Naive Bayes predictions:")
X_challenge_counts = count_vectorizer.transform(challenge_examples_processed)
X_challenge_combined_nb = hstack([X_challenge_counts, challenge_lexicon_vec])

# Make sure there are no negative values in the challenge combined features
if X_challenge_combined_nb.data.min() < 0:
    X_challenge_combined_nb.data[X_challenge_combined_nb.data < 0] = 0.0

for i, example in enumerate(challenge_examples):
    prediction = calibrated_nb.predict(X_challenge_combined_nb[i:i+1])[0]
    proba = calibrated_nb.predict_proba(X_challenge_combined_nb[i:i+1])[0]
    confidence = proba[1] if prediction == 1 else proba[0]
    sentiment = "Positive" if prediction == 1 else "Negative"
    
    # For neutral sentences, modify confidence and report
    is_neutral = 0.4 <= confidence <= 0.6
    sentiment_label = "Neutral" if is_neutral else sentiment
    
    print(f'"{example}" => {sentiment_label} ({confidence*100:.2f}% confidence)')

print("\nLogistic Regression predictions:")
X_challenge_tfidf = tfidf_vectorizer.transform(challenge_examples_processed)
X_challenge_combined_lr = hstack([X_challenge_tfidf, challenge_lexicon_vec])
for i, example in enumerate(challenge_examples):
    prediction = calibrated_lr.predict(X_challenge_combined_lr[i:i+1])[0]
    proba = calibrated_lr.predict_proba(X_challenge_combined_lr[i:i+1])[0]
    confidence = proba[1] if prediction == 1 else proba[0]
    sentiment = "Positive" if prediction == 1 else "Negative"
    
    # For neutral sentences, modify confidence and report
    is_neutral = 0.4 <= confidence <= 0.6
    sentiment_label = "Neutral" if is_neutral else sentiment
    
    print(f'"{example}" => {sentiment_label} ({confidence*100:.2f}% confidence)')

# Save vectorizers separately for easier troubleshooting
with open('models/count_vectorizer.pkl', 'wb') as f:
    pickle.dump(count_vectorizer, f)

with open('models/tfidf_vectorizer.pkl', 'wb') as f:
    pickle.dump(tfidf_vectorizer, f)

with open('models/dict_vectorizer.pkl', 'wb') as f:
    pickle.dump(dict_vectorizer, f)

# Save feature dimensions for reference
feature_dimensions = {
    'text_features': X_train_counts.shape[1],
    'lexicon_features': X_train_lexicon_vec.shape[1],
    'total_features': X_train_combined_nb.shape[1]
}

with open('models/feature_dimensions.pkl', 'wb') as f:
    pickle.dump(feature_dimensions, f)

# Save the calibrated models
with open('models/naive_bayes.pkl', 'wb') as f:
    pickle.dump((calibrated_nb, count_vectorizer, dict_vectorizer), f)
    
with open('models/logistic_regression.pkl', 'wb') as f:
    pickle.dump((calibrated_lr, tfidf_vectorizer, dict_vectorizer), f)

print("Models trained and saved successfully!")
print("\nYou can now run the Flask application with 'python app.py'")