Remiscus commited on
Commit
1572793
·
verified ·
1 Parent(s): eaf3552

Upload folder using huggingface_hub

Browse files
.github/workflows/update_space.yml ADDED
@@ -0,0 +1,28 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ name: Run Python script
2
+
3
+ on:
4
+ push:
5
+ branches:
6
+ - main
7
+
8
+ jobs:
9
+ build:
10
+ runs-on: ubuntu-latest
11
+
12
+ steps:
13
+ - name: Checkout
14
+ uses: actions/checkout@v2
15
+
16
+ - name: Set up Python
17
+ uses: actions/setup-python@v2
18
+ with:
19
+ python-version: '3.9'
20
+
21
+ - name: Install Gradio
22
+ run: python -m pip install gradio
23
+
24
+ - name: Log in to Hugging Face
25
+ run: python -c 'import huggingface_hub; huggingface_hub.login(token="${{ secrets.hf_token }}")'
26
+
27
+ - name: Deploy to Spaces
28
+ run: gradio deploy
.gitignore ADDED
@@ -0,0 +1 @@
 
 
1
+ __pycache__
.gradio/certificate.pem ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ -----BEGIN CERTIFICATE-----
2
+ MIIFazCCA1OgAwIBAgIRAIIQz7DSQONZRGPgu2OCiwAwDQYJKoZIhvcNAQELBQAw
3
+ TzELMAkGA1UEBhMCVVMxKTAnBgNVBAoTIEludGVybmV0IFNlY3VyaXR5IFJlc2Vh
4
+ cmNoIEdyb3VwMRUwEwYDVQQDEwxJU1JHIFJvb3QgWDEwHhcNMTUwNjA0MTEwNDM4
5
+ WhcNMzUwNjA0MTEwNDM4WjBPMQswCQYDVQQGEwJVUzEpMCcGA1UEChMgSW50ZXJu
6
+ ZXQgU2VjdXJpdHkgUmVzZWFyY2ggR3JvdXAxFTATBgNVBAMTDElTUkcgUm9vdCBY
7
+ MTCCAiIwDQYJKoZIhvcNAQEBBQADggIPADCCAgoCggIBAK3oJHP0FDfzm54rVygc
8
+ h77ct984kIxuPOZXoHj3dcKi/vVqbvYATyjb3miGbESTtrFj/RQSa78f0uoxmyF+
9
+ 0TM8ukj13Xnfs7j/EvEhmkvBioZxaUpmZmyPfjxwv60pIgbz5MDmgK7iS4+3mX6U
10
+ A5/TR5d8mUgjU+g4rk8Kb4Mu0UlXjIB0ttov0DiNewNwIRt18jA8+o+u3dpjq+sW
11
+ T8KOEUt+zwvo/7V3LvSye0rgTBIlDHCNAymg4VMk7BPZ7hm/ELNKjD+Jo2FR3qyH
12
+ B5T0Y3HsLuJvW5iB4YlcNHlsdu87kGJ55tukmi8mxdAQ4Q7e2RCOFvu396j3x+UC
13
+ B5iPNgiV5+I3lg02dZ77DnKxHZu8A/lJBdiB3QW0KtZB6awBdpUKD9jf1b0SHzUv
14
+ KBds0pjBqAlkd25HN7rOrFleaJ1/ctaJxQZBKT5ZPt0m9STJEadao0xAH0ahmbWn
15
+ OlFuhjuefXKnEgV4We0+UXgVCwOPjdAvBbI+e0ocS3MFEvzG6uBQE3xDk3SzynTn
16
+ jh8BCNAw1FtxNrQHusEwMFxIt4I7mKZ9YIqioymCzLq9gwQbooMDQaHWBfEbwrbw
17
+ qHyGO0aoSCqI3Haadr8faqU9GY/rOPNk3sgrDQoo//fb4hVC1CLQJ13hef4Y53CI
18
+ rU7m2Ys6xt0nUW7/vGT1M0NPAgMBAAGjQjBAMA4GA1UdDwEB/wQEAwIBBjAPBgNV
19
+ HRMBAf8EBTADAQH/MB0GA1UdDgQWBBR5tFnme7bl5AFzgAiIyBpY9umbbjANBgkq
20
+ hkiG9w0BAQsFAAOCAgEAVR9YqbyyqFDQDLHYGmkgJykIrGF1XIpu+ILlaS/V9lZL
21
+ ubhzEFnTIZd+50xx+7LSYK05qAvqFyFWhfFQDlnrzuBZ6brJFe+GnY+EgPbk6ZGQ
22
+ 3BebYhtF8GaV0nxvwuo77x/Py9auJ/GpsMiu/X1+mvoiBOv/2X/qkSsisRcOj/KK
23
+ NFtY2PwByVS5uCbMiogziUwthDyC3+6WVwW6LLv3xLfHTjuCvjHIInNzktHCgKQ5
24
+ ORAzI4JMPJ+GslWYHb4phowim57iaztXOoJwTdwJx4nLCgdNbOhdjsnvzqvHu7Ur
25
+ TkXWStAmzOVyyghqpZXjFaH3pO3JLF+l+/+sKAIuvtd7u+Nxe5AW0wdeRlN8NwdC
26
+ jNPElpzVmbUq4JUagEiuTDkHzsxHpFKVK7q4+63SM1N95R1NbdWhscdCb+ZAJzVc
27
+ oyi3B43njTOQ5yOf+1CceWxG1bQVs5ZufpsMljq4Ui0/1lvh+wjChP4kqKOJ2qxq
28
+ 4RgqsahDYVvTH9w7jXbyLeiNdd8XM2w9U/t7y0Ff/9yi0GE44Za4rF2LN9d11TPA
29
+ mRGunUHBcnWEvgJBQl9nJEiU0Zsnvgc/ubhPgXRR4Xq37Z0j4r7g1SgEEzwxA57d
30
+ emyPxgcYxn/eR44/KJ4EBs+lVDR3veyJm+kXQ99b21/+jh5Xos1AnX5iItreGCc=
31
+ -----END CERTIFICATE-----
.gradio/flagged/Download CSV/9b36608063ab4608a7a5/sentiment_analysis_results.csv ADDED
The diff for this file is too large to render. See raw diff
 
.gradio/flagged/Download CSV/b60237ef4d855c94dac8/sentiment_analysis_results.csv ADDED
The diff for this file is too large to render. See raw diff
 
.gradio/flagged/Upload your review file CSV or XLSX/0dc80347f848c9c73fcf/IMDB Dataset - Copy.csv ADDED
The diff for this file is too large to render. See raw diff
 
.gradio/flagged/Upload your review file CSV or XLSX/11262994bbf3b5da5f4d/IMDB Dataset - Copy.csv ADDED
The diff for this file is too large to render. See raw diff
 
.gradio/flagged/dataset1.csv ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ Upload your review file (CSV or XLSX),Vader,Transformers,Download CSV,Message,Data Preview,timestamp
2
+ .gradio\flagged\Upload your review file CSV or XLSX\0dc80347f848c9c73fcf\IMDB Dataset - Copy.csv,true,true,.gradio\flagged\Download CSV\9b36608063ab4608a7a5\sentiment_analysis_results.csv,,"{""headers"": [""review"", ""sentiment"", ""neg"", ""neu"", ""pos"", ""compound""], ""data"": [[""One of the other reviewers has mentioned that after watching just 1 Oz episode you'll be hooked. They are right, as this is exactly what happened with me.<br /><br />The first thing that struck me about Oz was its brutality and unflinching scenes of violence, which set in right from the word GO. Trust me, this is not a show for the faint hearted or timid. This show pulls no punches with regards to drugs, sex or violence. Its is hardcore, in the classic use of the word.<br /><br />It is called OZ as that is the nickname given to the Oswald Maximum Security State Penitentary. It focuses mainly on Emerald City, an experimental section of the prison where all the cells have glass fronts and face inwards, so privacy is not high on the agenda. Em City is home to many..Aryans, Muslims, gangstas, Latinos, Christians, Italians, Irish and more....so scuffles, death stares, dodgy dealings and shady agreements are never far away.<br /><br />I would say the main appeal of the show is due to the fact that it goes where other shows wouldn't dare. Forget pretty pictures painted for mainstream audiences, forget charm, forget romance...OZ doesn't mess around. The first episode I ever saw struck me as so nasty it was surreal, I couldn't say I was ready for it, but as I watched more, I developed a taste for Oz, and got accustomed to the high levels of graphic violence. Not just violence, but injustice (crooked guards who'll be sold out for a nickel, inmates who'll kill on order and get away with it, well mannered, middle class inmates being turned into prison bitches due to their lack of street skills or prison experience) Watching Oz, you may become comfortable with what is uncomfortable viewing....thats if you can get in touch with your darker side."", ""positive"", 0.203, 0.748, 0.048, -0.9951], [""A wonderful little production. <br /><br />The filming technique is very unassuming- very old-time-BBC fashion and gives a comforting, and sometimes discomforting, sense of realism to the entire piece. <br /><br />The actors are extremely well chosen- Michael Sheen not only \""has got all the polari\"" but he has all the voices down pat too! You can truly see the seamless editing guided by the references to Williams' diary entries, not only is it well worth the watching but it is a terrificly written and performed piece. A masterful production about one of the great master's of comedy and his life. <br /><br />The realism really comes home with the little things: the fantasy of the guard which, rather than use the traditional 'dream' techniques remains solid then disappears. It plays on our knowledge and our senses, particularly with the scenes concerning Orton and Halliwell and the sets (particularly of their flat with Halliwell's murals decorating every surface) are terribly well done."", ""positive"", 0.053, 0.776, 0.172, 0.9641], [""I thought this was a wonderful way to spend time on a too hot summer weekend, sitting in the air conditioned theater and watching a light-hearted comedy. The plot is simplistic, but the dialogue is witty and the characters are likable (even the well bread suspected serial killer). While some may be disappointed when they realize this is not Match Point 2: Risk Addiction, I thought it was proof that Woody Allen is still fully in control of the style many of us have grown to love.<br /><br />This was the most I'd laughed at one of Woody's comedies in years (dare I say a decade?). While I've never been impressed with Scarlet Johanson, in this she managed to tone down her \""sexy\"" image and jumped right into a average, but spirited young woman.<br /><br />This may not be the crown jewel of his career, but it was wittier than \""Devil Wears Prada\"" and more interesting than \""Superman\"" a great comedy to go see with friends."", ""positive"", 0.094, 0.714, 0.192, 0.9605], [""Basically there's a family where a little boy (Jake) thinks there's a zombie in his closet & his parents are fighting all the time.<br /><br />This movie is slower than a soap opera... and suddenly, Jake decides to become Rambo and kill the zombie.<br /><br />OK, first of all when you're going to make a film you must Decide if its a thriller or a drama! As a drama the movie is watchable. Parents are divorcing & arguing like in real life. And then we have Jake with his closet which totally ruins all the film! I expected to see a BOOGEYMAN similar movie, and instead i watched a drama with some meaningless thriller spots.<br /><br />3 out of 10 just for the well playing parents & descent dialogs. As for the shots with Jake: just ignore them."", ""negative"", 0.138, 0.797, 0.065, -0.9213], [""Petter Mattei's \""Love in the Time of Money\"" is a visually stunning film to watch. Mr. Mattei offers us a vivid portrait about human relations. This is a movie that seems to be telling us what money, power and success do to people in the different situations we encounter. <br /><br />This being a variation on the Arthur Schnitzler's play about the same theme, the director transfers the action to the present time New York where all these different characters meet and connect. Each one is connected in one way, or another to the next person, but no one seems to know the previous point of contact. Stylishly, the film has a sophisticated luxurious look. We are taken to see how these people live and the world they live in their own habitat.<br /><br />The only thing one gets out of all these souls in the picture is the different stages of loneliness each one inhabits. A big city is not exactly the best place in which human relations find sincere fulfillment, as one discerns is the case with most of the people we encounter.<br /><br />The acting is good under Mr. Mattei's direction. Steve Buscemi, Rosario Dawson, Carol Kane, Michael Imperioli, Adrian Grenier, and the rest of the talented cast, make these characters come alive.<br /><br />We wish Mr. Mattei good luck and await anxiously for his next work."", ""positive"", 0.052, 0.801, 0.147, 0.9744]], ""metadata"": {""display_value"": null, ""styling"": null}}",2024-11-07 19:35:08.804271
3
+ .gradio\flagged\Upload your review file CSV or XLSX\11262994bbf3b5da5f4d\IMDB Dataset - Copy.csv,false,true,.gradio\flagged\Download CSV\b60237ef4d855c94dac8\sentiment_analysis_results.csv,,"{""headers"": [""review"", ""sentiment"", ""label"", ""score""], ""data"": [[""One of the other reviewers has mentioned that after watching just 1 Oz episode you'll be hooked. They are right, as this is exactly what happened with me.<br /><br />The first thing that struck me about Oz was its brutality and unflinching scenes of violence, which set in right from the word GO. Trust me, this is not a show for the faint hearted or timid. This show pulls no punches with regards to drugs, sex or violence. Its is hardcore, in the classic use of the word.<br /><br />It is called OZ as that is the nickname given to the Oswald Maximum Security State Penitentary. It focuses mainly on Emerald City, an experimental section of the prison where all the cells have glass fronts and face inwards, so privacy is not high on the agenda. Em City is home to many..Aryans, Muslims, gangstas, Latinos, Christians, Italians, Irish and more....so scuffles, death stares, dodgy dealings and shady agreements are never far away.<br /><br />I would say the main appeal of the show is due to the fact that it goes where other shows wouldn't dare. Forget pretty pictures painted for mainstream audiences, forget charm, forget romance...OZ doesn't mess around. The first episode I ever saw struck me as so nasty it was surreal, I couldn't say I was ready for it, but as I watched more, I developed a taste for Oz, and got accustomed to the high levels of graphic violence. Not just violence, but injustice (crooked guards who'll be sold out for a nickel, inmates who'll kill on order and get away with it, well mannered, middle class inmates being turned into prison bitches due to their lack of street skills or prison experience) Watching Oz, you may become comfortable with what is uncomfortable viewing....thats if you can get in touch with your darker side."", ""positive"", ""POSITIVE"", 0.5136224031448364], [""A wonderful little production. <br /><br />The filming technique is very unassuming- very old-time-BBC fashion and gives a comforting, and sometimes discomforting, sense of realism to the entire piece. <br /><br />The actors are extremely well chosen- Michael Sheen not only \""has got all the polari\"" but he has all the voices down pat too! You can truly see the seamless editing guided by the references to Williams' diary entries, not only is it well worth the watching but it is a terrificly written and performed piece. A masterful production about one of the great master's of comedy and his life. <br /><br />The realism really comes home with the little things: the fantasy of the guard which, rather than use the traditional 'dream' techniques remains solid then disappears. It plays on our knowledge and our senses, particularly with the scenes concerning Orton and Halliwell and the sets (particularly of their flat with Halliwell's murals decorating every surface) are terribly well done."", ""positive"", ""POSITIVE"", 0.5018884539604187], [""I thought this was a wonderful way to spend time on a too hot summer weekend, sitting in the air conditioned theater and watching a light-hearted comedy. The plot is simplistic, but the dialogue is witty and the characters are likable (even the well bread suspected serial killer). While some may be disappointed when they realize this is not Match Point 2: Risk Addiction, I thought it was proof that Woody Allen is still fully in control of the style many of us have grown to love.<br /><br />This was the most I'd laughed at one of Woody's comedies in years (dare I say a decade?). While I've never been impressed with Scarlet Johanson, in this she managed to tone down her \""sexy\"" image and jumped right into a average, but spirited young woman.<br /><br />This may not be the crown jewel of his career, but it was wittier than \""Devil Wears Prada\"" and more interesting than \""Superman\"" a great comedy to go see with friends."", ""positive"", ""POSITIVE"", 0.5064340829849243], [""Basically there's a family where a little boy (Jake) thinks there's a zombie in his closet & his parents are fighting all the time.<br /><br />This movie is slower than a soap opera... and suddenly, Jake decides to become Rambo and kill the zombie.<br /><br />OK, first of all when you're going to make a film you must Decide if its a thriller or a drama! As a drama the movie is watchable. Parents are divorcing & arguing like in real life. And then we have Jake with his closet which totally ruins all the film! I expected to see a BOOGEYMAN similar movie, and instead i watched a drama with some meaningless thriller spots.<br /><br />3 out of 10 just for the well playing parents & descent dialogs. As for the shots with Jake: just ignore them."", ""negative"", ""POSITIVE"", 0.5148193836212158], [""Petter Mattei's \""Love in the Time of Money\"" is a visually stunning film to watch. Mr. Mattei offers us a vivid portrait about human relations. This is a movie that seems to be telling us what money, power and success do to people in the different situations we encounter. <br /><br />This being a variation on the Arthur Schnitzler's play about the same theme, the director transfers the action to the present time New York where all these different characters meet and connect. Each one is connected in one way, or another to the next person, but no one seems to know the previous point of contact. Stylishly, the film has a sophisticated luxurious look. We are taken to see how these people live and the world they live in their own habitat.<br /><br />The only thing one gets out of all these souls in the picture is the different stages of loneliness each one inhabits. A big city is not exactly the best place in which human relations find sincere fulfillment, as one discerns is the case with most of the people we encounter.<br /><br />The acting is good under Mr. Mattei's direction. Steve Buscemi, Rosario Dawson, Carol Kane, Michael Imperioli, Adrian Grenier, and the rest of the talented cast, make these characters come alive.<br /><br />We wish Mr. Mattei good luck and await anxiously for his next work."", ""positive"", ""POSITIVE"", 0.5115845203399658]], ""metadata"": {""display_value"": null, ""styling"": null}}",2024-11-07 19:51:20.128949
Datasets/Instruction.md ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ Please install the test Dataset from:
2
+
3
+ [text](https://www.kaggle.com/datasets/lakshmi25npathi/imdb-dataset-of-50k-movie-reviews)
README.md CHANGED
@@ -1,12 +1,93 @@
1
- ---
2
- title: BulkSentimentAnalysis
3
- emoji: 🚀
4
- colorFrom: yellow
5
- colorTo: gray
6
- sdk: gradio
7
- sdk_version: 5.5.0
8
- app_file: app.py
9
- pinned: false
10
- ---
11
-
12
- Check out the configuration reference at https://huggingface.co/docs/hub/spaces-config-reference
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ title: BulkSentimentAnalysis
3
+ app_file: app.py
4
+ sdk: gradio
5
+ sdk_version: 5.5.0
6
+ ---
7
+ # Bulk Sentiment Analysis for Reviews
8
+
9
+ This project provides a tool for bulk sentiment analysis of store reviews. Using various sentiment analysis models, including VADER and Hugging Face's Transformers, users can analyze reviews in CSV or Excel format to understand the overall sentiment. Results can be downloaded in CSV format, making it easy to leverage insights for decision-making.
10
+
11
+ ## Features
12
+
13
+ - **Bulk Upload:** Upload CSV or Excel files of reviews for analysis.
14
+ - **Model Options:** Choose from multiple sentiment analysis models:
15
+ - **VADER** for rule-based sentiment analysis.
16
+ - **SpaCy (Static Embeddings)** and **SpaCy (Contextual Embeddings)** for improved flexibility (coming soon).
17
+ - **Transformers** for deep learning-based sentiment analysis using pretrained models from Hugging Face.
18
+ - **Download Results:** Export analyzed reviews with sentiment scores and labels as a CSV.
19
+
20
+ ## Installation
21
+
22
+ To get started, clone this repository and install the dependencies.
23
+
24
+ ```bash
25
+ git clone https://github.com/yourusername/bulk-sentiment-analysis
26
+ cd bulk-sentiment-analysis
27
+ pip install -r requirements.txt
28
+ ```
29
+
30
+ Ensure your environment supports GPU processing, especially for Transformer-based models, to handle large datasets more efficiently.
31
+
32
+ ### Dependencies
33
+
34
+ - Python 3.7+
35
+ - [Streamlit](https://streamlit.io/) for building the user interface
36
+ - [Pandas](https://pandas.pydata.org/) for data handling
37
+ - [NLTK](https://www.nltk.org/) and [VADER](https://github.com/cjhutto/vaderSentiment) for sentiment analysis
38
+ - [SpaCy](https://spacy.io/) (optional, for static and contextual embedding-based analysis)
39
+ - [Transformers](https://huggingface.co/transformers/) by Hugging Face for deep learning sentiment models
40
+
41
+ To install these, you can run:
42
+ ```bash
43
+ pip install streamlit pandas nltk torch transformers spacy
44
+ ```
45
+
46
+ For GPU support, install the appropriate CUDA version of PyTorch, following instructions from the [official PyTorch website](https://pytorch.org/get-started/locally/).
47
+
48
+ ## Usage
49
+
50
+ 1. **Run the Application**
51
+
52
+ Launch the Streamlit app with:
53
+ ```bash
54
+ streamlit run app.py
55
+ ```
56
+
57
+ 2. **Upload Your File**
58
+
59
+ Upload a CSV or Excel file containing reviews. Make sure there is a column labeled `Review` with the text you want to analyze.
60
+
61
+ 3. **Choose a Model**
62
+
63
+ Select one or more sentiment analysis models:
64
+ - **VADER** for quick, rule-based analysis.
65
+ - **Transformers** for more accurate, context-based sentiment detection.
66
+
67
+ 4. **Analyze and Download Results**
68
+
69
+ After processing, the app will display results in the UI. You can download the full results as a CSV file.
70
+
71
+ ### Example of CSV Output
72
+
73
+ The output CSV will contain the original reviews alongside new columns with sentiment scores and labels, depending on the models chosen.
74
+
75
+ ## Folder Structure
76
+
77
+ - **`app.py`**: Main Streamlit application script.
78
+ - **`sentiment_analysis/`**: Contains modular functions for each sentiment analysis model (e.g., `vader_analyzer.py`, `spacy_static.py`, etc.).
79
+ - **`requirements.txt`**: List of required dependencies.
80
+
81
+ ## Future Enhancements
82
+
83
+ - **Integrate SpaCy models** for static and contextual embedding-based sentiment analysis.
84
+ - **Improved preprocessing** to handle non-English text and non-standard review formats.
85
+ - **Performance optimization** with caching and batch processing for large datasets.
86
+
87
+ ## Contributing
88
+
89
+ Feel free to open issues or submit pull requests for improvements. All contributions are welcome!
90
+
91
+ ---
92
+
93
+ This README should give users a clear understanding of what the project does, how to set it up, and how to use it. Let me know if you'd like any more specific instructions or additional sections!
app.py ADDED
@@ -0,0 +1,59 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import gradio as gr
2
+ import pandas as pd
3
+ from sentiment_analysis.vader_analyzer import analyze_sentiment_vader
4
+ from sentiment_analysis.transformers_analyzer import analyze_sentiment_transformers
5
+
6
+ def analyze_reviews(file, use_vader, use_transformers):
7
+ # Read the uploaded file
8
+ if file.name.endswith(".csv"):
9
+ df = pd.read_csv(file)
10
+ elif file.name.endswith(".xlsx"):
11
+ df = pd.read_excel(file)
12
+ else:
13
+ return "Please upload a CSV or Excel file."
14
+
15
+ # Check the number of entries and truncate if needed
16
+ if len(df) > 1000:
17
+ df = df.head(1000)
18
+ error_message = "The file contains more than 1,000 entries. Only the first 1,000 reviews are processed."
19
+ else:
20
+ error_message = "No Errors Found"
21
+
22
+ # Ensure the "review" column exists
23
+ df.columns = df.columns.str.lower()
24
+ if 'review' not in df.columns:
25
+ return "Please make sure the file has a 'review' column with review text."
26
+
27
+ # Apply selected sentiment analysis models
28
+ if use_vader:
29
+ vader_results = analyze_sentiment_vader(df["review"])
30
+ df = pd.concat([df, vader_results], axis=1)
31
+
32
+ if use_transformers:
33
+ transformers_results = analyze_sentiment_transformers(df["review"])
34
+ df = pd.concat([df, transformers_results], axis=1)
35
+
36
+ # Save the result to a CSV file
37
+ output_file = "sentiment_analysis_results.csv"
38
+ df.to_csv(output_file, index=False)
39
+
40
+ return output_file, error_message, df.head() # Return the output file, error message, and preview
41
+
42
+ # Define Gradio interface
43
+ interface = gr.Interface(
44
+ fn=analyze_reviews,
45
+ inputs=[
46
+ gr.File(label="Upload your review file (CSV or XLSX)"),
47
+ gr.Radio(["Vader", "Transformers"], label="Select Sentiment Analysis Model"),
48
+ ],
49
+ outputs=[
50
+ gr.File(label="Download CSV"),
51
+ gr.Textbox(label="Error Message"),
52
+ gr.Dataframe(label="Data Preview"),
53
+ ],
54
+ title="Bulk Sentiment Analysis for Reviews",
55
+ description="Upload a file with a 'review' column to analyze sentiment using Vader and Transformers models."
56
+ )
57
+
58
+ # Launch the Gradio app
59
+ interface.launch(share=True)
notebooks/transformers_sentiment.ipynb ADDED
@@ -0,0 +1,226 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "cells": [
3
+ {
4
+ "cell_type": "code",
5
+ "execution_count": null,
6
+ "metadata": {},
7
+ "outputs": [],
8
+ "source": [
9
+ "from transformers import AutoModelForSequenceClassification, AutoTokenizer\n",
10
+ "import pandas as pd\n",
11
+ "import torch"
12
+ ]
13
+ },
14
+ {
15
+ "cell_type": "code",
16
+ "execution_count": null,
17
+ "metadata": {},
18
+ "outputs": [
19
+ {
20
+ "data": {
21
+ "text/plain": [
22
+ "0 One of the other reviewers has mentioned that ...\n",
23
+ "1 A wonderful little production. <br /><br />The...\n",
24
+ "2 I thought this was a wonderful way to spend ti...\n",
25
+ "3 Basically there's a family where a little boy ...\n",
26
+ "4 Petter Mattei's \"Love in the Time of Money\" is...\n",
27
+ " ... \n",
28
+ "495 \"American Nightmare\" is officially tied, in my...\n",
29
+ "496 First off, I have to say that I loved the book...\n",
30
+ "497 This movie was extremely boring. I only laughe...\n",
31
+ "498 I was disgusted by this movie. No it wasn't be...\n",
32
+ "499 Such a joyous world has been created for us in...\n",
33
+ "Name: review, Length: 500, dtype: object"
34
+ ]
35
+ },
36
+ "execution_count": 17,
37
+ "metadata": {},
38
+ "output_type": "execute_result"
39
+ }
40
+ ],
41
+ "source": [
42
+ "reviews = pd.read_csv('../Datasets/IMDB Dataset.csv')\n",
43
+ "reviews = reviews.head(500)[\"review\"]\n",
44
+ "reviews"
45
+ ]
46
+ },
47
+ {
48
+ "cell_type": "code",
49
+ "execution_count": 18,
50
+ "metadata": {},
51
+ "outputs": [
52
+ {
53
+ "name": "stderr",
54
+ "output_type": "stream",
55
+ "text": [
56
+ "Some weights of DistilBertForSequenceClassification were not initialized from the model checkpoint at distilbert-base-uncased and are newly initialized: ['classifier.bias', 'classifier.weight', 'pre_classifier.bias', 'pre_classifier.weight']\n",
57
+ "You should probably TRAIN this model on a down-stream task to be able to use it for predictions and inference.\n"
58
+ ]
59
+ },
60
+ {
61
+ "data": {
62
+ "text/html": [
63
+ "<div>\n",
64
+ "<style scoped>\n",
65
+ " .dataframe tbody tr th:only-of-type {\n",
66
+ " vertical-align: middle;\n",
67
+ " }\n",
68
+ "\n",
69
+ " .dataframe tbody tr th {\n",
70
+ " vertical-align: top;\n",
71
+ " }\n",
72
+ "\n",
73
+ " .dataframe thead th {\n",
74
+ " text-align: right;\n",
75
+ " }\n",
76
+ "</style>\n",
77
+ "<table border=\"1\" class=\"dataframe\">\n",
78
+ " <thead>\n",
79
+ " <tr style=\"text-align: right;\">\n",
80
+ " <th></th>\n",
81
+ " <th>label</th>\n",
82
+ " <th>score</th>\n",
83
+ " </tr>\n",
84
+ " </thead>\n",
85
+ " <tbody>\n",
86
+ " <tr>\n",
87
+ " <th>0</th>\n",
88
+ " <td>POSITIVE</td>\n",
89
+ " <td>0.508178</td>\n",
90
+ " </tr>\n",
91
+ " <tr>\n",
92
+ " <th>1</th>\n",
93
+ " <td>POSITIVE</td>\n",
94
+ " <td>0.521151</td>\n",
95
+ " </tr>\n",
96
+ " <tr>\n",
97
+ " <th>2</th>\n",
98
+ " <td>POSITIVE</td>\n",
99
+ " <td>0.528036</td>\n",
100
+ " </tr>\n",
101
+ " <tr>\n",
102
+ " <th>3</th>\n",
103
+ " <td>POSITIVE</td>\n",
104
+ " <td>0.517413</td>\n",
105
+ " </tr>\n",
106
+ " <tr>\n",
107
+ " <th>4</th>\n",
108
+ " <td>POSITIVE</td>\n",
109
+ " <td>0.520384</td>\n",
110
+ " </tr>\n",
111
+ " <tr>\n",
112
+ " <th>...</th>\n",
113
+ " <td>...</td>\n",
114
+ " <td>...</td>\n",
115
+ " </tr>\n",
116
+ " <tr>\n",
117
+ " <th>495</th>\n",
118
+ " <td>POSITIVE</td>\n",
119
+ " <td>0.528022</td>\n",
120
+ " </tr>\n",
121
+ " <tr>\n",
122
+ " <th>496</th>\n",
123
+ " <td>POSITIVE</td>\n",
124
+ " <td>0.512645</td>\n",
125
+ " </tr>\n",
126
+ " <tr>\n",
127
+ " <th>497</th>\n",
128
+ " <td>POSITIVE</td>\n",
129
+ " <td>0.524352</td>\n",
130
+ " </tr>\n",
131
+ " <tr>\n",
132
+ " <th>498</th>\n",
133
+ " <td>POSITIVE</td>\n",
134
+ " <td>0.503319</td>\n",
135
+ " </tr>\n",
136
+ " <tr>\n",
137
+ " <th>499</th>\n",
138
+ " <td>POSITIVE</td>\n",
139
+ " <td>0.526241</td>\n",
140
+ " </tr>\n",
141
+ " </tbody>\n",
142
+ "</table>\n",
143
+ "<p>500 rows × 2 columns</p>\n",
144
+ "</div>"
145
+ ],
146
+ "text/plain": [
147
+ " label score\n",
148
+ "0 POSITIVE 0.508178\n",
149
+ "1 POSITIVE 0.521151\n",
150
+ "2 POSITIVE 0.528036\n",
151
+ "3 POSITIVE 0.517413\n",
152
+ "4 POSITIVE 0.520384\n",
153
+ ".. ... ...\n",
154
+ "495 POSITIVE 0.528022\n",
155
+ "496 POSITIVE 0.512645\n",
156
+ "497 POSITIVE 0.524352\n",
157
+ "498 POSITIVE 0.503319\n",
158
+ "499 POSITIVE 0.526241\n",
159
+ "\n",
160
+ "[500 rows x 2 columns]"
161
+ ]
162
+ },
163
+ "execution_count": 18,
164
+ "metadata": {},
165
+ "output_type": "execute_result"
166
+ }
167
+ ],
168
+ "source": [
169
+ "# Load tokenizer and model\n",
170
+ "tokenizer = AutoTokenizer.from_pretrained(\"distilbert-base-uncased\")\n",
171
+ "model = AutoModelForSequenceClassification.from_pretrained(\"distilbert-base-uncased\").to(\"cuda\")\n",
172
+ "\n",
173
+ "results = []\n",
174
+ "for review in reviews:\n",
175
+ " # Tokenize with truncation and padding to max length\n",
176
+ " tokenized_review = tokenizer(review, return_tensors=\"pt\", truncation=True, padding=\"max_length\", max_length=512).to(\"cuda\")\n",
177
+ "\n",
178
+ " # Get the model's output (logits)\n",
179
+ " with torch.no_grad():\n",
180
+ " outputs = model(**tokenized_review)\n",
181
+ " \n",
182
+ " # Convert logits to probabilities\n",
183
+ " probabilities = torch.nn.functional.softmax(outputs.logits, dim=-1)\n",
184
+ " \n",
185
+ " # Get the predicted label and score\n",
186
+ " score = probabilities.max().item()\n",
187
+ " label = \"POSITIVE\" if torch.argmax(probabilities).item() == 1 else \"NEGATIVE\"\n",
188
+ " \n",
189
+ " # Append the result\n",
190
+ " results.append({\"label\": label, \"score\": score})\n",
191
+ "\n",
192
+ "# Create DataFrame\n",
193
+ "sentiment_df = pd.DataFrame(results)\n",
194
+ "sentiment_df\n"
195
+ ]
196
+ },
197
+ {
198
+ "cell_type": "code",
199
+ "execution_count": null,
200
+ "metadata": {},
201
+ "outputs": [],
202
+ "source": []
203
+ }
204
+ ],
205
+ "metadata": {
206
+ "kernelspec": {
207
+ "display_name": "NLP_env",
208
+ "language": "python",
209
+ "name": "python3"
210
+ },
211
+ "language_info": {
212
+ "codemirror_mode": {
213
+ "name": "ipython",
214
+ "version": 3
215
+ },
216
+ "file_extension": ".py",
217
+ "mimetype": "text/x-python",
218
+ "name": "python",
219
+ "nbconvert_exporter": "python",
220
+ "pygments_lexer": "ipython3",
221
+ "version": "3.12.4"
222
+ }
223
+ },
224
+ "nbformat": 4,
225
+ "nbformat_minor": 2
226
+ }
notebooks/vader_sentiment.ipynb ADDED
@@ -0,0 +1,249 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "cells": [
3
+ {
4
+ "cell_type": "code",
5
+ "execution_count": 36,
6
+ "metadata": {},
7
+ "outputs": [],
8
+ "source": [
9
+ "import pandas as pd\n",
10
+ "import nltk\n",
11
+ "from nltk.sentiment.vader import SentimentIntensityAnalyzer\n",
12
+ "import matplotlib.pyplot as plt\n",
13
+ "import seaborn as sns"
14
+ ]
15
+ },
16
+ {
17
+ "cell_type": "code",
18
+ "execution_count": 37,
19
+ "metadata": {},
20
+ "outputs": [
21
+ {
22
+ "name": "stderr",
23
+ "output_type": "stream",
24
+ "text": [
25
+ "[nltk_data] Downloading package vader_lexicon to C:\\Users\\Rikhil\n",
26
+ "[nltk_data] Nellimarla\\AppData\\Roaming\\nltk_data...\n",
27
+ "[nltk_data] Package vader_lexicon is already up-to-date!\n"
28
+ ]
29
+ },
30
+ {
31
+ "data": {
32
+ "text/plain": [
33
+ "True"
34
+ ]
35
+ },
36
+ "execution_count": 37,
37
+ "metadata": {},
38
+ "output_type": "execute_result"
39
+ }
40
+ ],
41
+ "source": [
42
+ "# Download VADER lexicon for sentiment analysis\n",
43
+ "nltk.download('vader_lexicon')"
44
+ ]
45
+ },
46
+ {
47
+ "cell_type": "code",
48
+ "execution_count": 38,
49
+ "metadata": {},
50
+ "outputs": [
51
+ {
52
+ "ename": "FileNotFoundError",
53
+ "evalue": "[Errno 2] No such file or directory: '../Datasets/IMDB Dataset.csv'",
54
+ "output_type": "error",
55
+ "traceback": [
56
+ "\u001b[1;31m---------------------------------------------------------------------------\u001b[0m",
57
+ "\u001b[1;31mFileNotFoundError\u001b[0m Traceback (most recent call last)",
58
+ "Cell \u001b[1;32mIn[38], line 1\u001b[0m\n\u001b[1;32m----> 1\u001b[0m data \u001b[38;5;241m=\u001b[39m pd\u001b[38;5;241m.\u001b[39mread_csv(\u001b[38;5;124m'\u001b[39m\u001b[38;5;124m../Datasets/IMDB Dataset.csv\u001b[39m\u001b[38;5;124m'\u001b[39m)\n",
59
+ "File \u001b[1;32mc:\\Users\\Rikhil Nellimarla\\.conda\\envs\\NLP_env\\Lib\\site-packages\\pandas\\io\\parsers\\readers.py:1026\u001b[0m, in \u001b[0;36mread_csv\u001b[1;34m(filepath_or_buffer, sep, delimiter, header, names, index_col, usecols, dtype, engine, converters, true_values, false_values, skipinitialspace, skiprows, skipfooter, nrows, na_values, keep_default_na, na_filter, verbose, skip_blank_lines, parse_dates, infer_datetime_format, keep_date_col, date_parser, date_format, dayfirst, cache_dates, iterator, chunksize, compression, thousands, decimal, lineterminator, quotechar, quoting, doublequote, escapechar, comment, encoding, encoding_errors, dialect, on_bad_lines, delim_whitespace, low_memory, memory_map, float_precision, storage_options, dtype_backend)\u001b[0m\n\u001b[0;32m 1013\u001b[0m kwds_defaults \u001b[38;5;241m=\u001b[39m _refine_defaults_read(\n\u001b[0;32m 1014\u001b[0m dialect,\n\u001b[0;32m 1015\u001b[0m delimiter,\n\u001b[1;32m (...)\u001b[0m\n\u001b[0;32m 1022\u001b[0m dtype_backend\u001b[38;5;241m=\u001b[39mdtype_backend,\n\u001b[0;32m 1023\u001b[0m )\n\u001b[0;32m 1024\u001b[0m kwds\u001b[38;5;241m.\u001b[39mupdate(kwds_defaults)\n\u001b[1;32m-> 1026\u001b[0m \u001b[38;5;28;01mreturn\u001b[39;00m _read(filepath_or_buffer, kwds)\n",
60
+ "File \u001b[1;32mc:\\Users\\Rikhil Nellimarla\\.conda\\envs\\NLP_env\\Lib\\site-packages\\pandas\\io\\parsers\\readers.py:620\u001b[0m, in \u001b[0;36m_read\u001b[1;34m(filepath_or_buffer, kwds)\u001b[0m\n\u001b[0;32m 617\u001b[0m _validate_names(kwds\u001b[38;5;241m.\u001b[39mget(\u001b[38;5;124m\"\u001b[39m\u001b[38;5;124mnames\u001b[39m\u001b[38;5;124m\"\u001b[39m, \u001b[38;5;28;01mNone\u001b[39;00m))\n\u001b[0;32m 619\u001b[0m \u001b[38;5;66;03m# Create the parser.\u001b[39;00m\n\u001b[1;32m--> 620\u001b[0m parser \u001b[38;5;241m=\u001b[39m TextFileReader(filepath_or_buffer, \u001b[38;5;241m*\u001b[39m\u001b[38;5;241m*\u001b[39mkwds)\n\u001b[0;32m 622\u001b[0m \u001b[38;5;28;01mif\u001b[39;00m chunksize \u001b[38;5;129;01mor\u001b[39;00m iterator:\n\u001b[0;32m 623\u001b[0m \u001b[38;5;28;01mreturn\u001b[39;00m parser\n",
61
+ "File \u001b[1;32mc:\\Users\\Rikhil Nellimarla\\.conda\\envs\\NLP_env\\Lib\\site-packages\\pandas\\io\\parsers\\readers.py:1620\u001b[0m, in \u001b[0;36mTextFileReader.__init__\u001b[1;34m(self, f, engine, **kwds)\u001b[0m\n\u001b[0;32m 1617\u001b[0m \u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39moptions[\u001b[38;5;124m\"\u001b[39m\u001b[38;5;124mhas_index_names\u001b[39m\u001b[38;5;124m\"\u001b[39m] \u001b[38;5;241m=\u001b[39m kwds[\u001b[38;5;124m\"\u001b[39m\u001b[38;5;124mhas_index_names\u001b[39m\u001b[38;5;124m\"\u001b[39m]\n\u001b[0;32m 1619\u001b[0m \u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39mhandles: IOHandles \u001b[38;5;241m|\u001b[39m \u001b[38;5;28;01mNone\u001b[39;00m \u001b[38;5;241m=\u001b[39m \u001b[38;5;28;01mNone\u001b[39;00m\n\u001b[1;32m-> 1620\u001b[0m \u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39m_engine \u001b[38;5;241m=\u001b[39m \u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39m_make_engine(f, \u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39mengine)\n",
62
+ "File \u001b[1;32mc:\\Users\\Rikhil Nellimarla\\.conda\\envs\\NLP_env\\Lib\\site-packages\\pandas\\io\\parsers\\readers.py:1880\u001b[0m, in \u001b[0;36mTextFileReader._make_engine\u001b[1;34m(self, f, engine)\u001b[0m\n\u001b[0;32m 1878\u001b[0m \u001b[38;5;28;01mif\u001b[39;00m \u001b[38;5;124m\"\u001b[39m\u001b[38;5;124mb\u001b[39m\u001b[38;5;124m\"\u001b[39m \u001b[38;5;129;01mnot\u001b[39;00m \u001b[38;5;129;01min\u001b[39;00m mode:\n\u001b[0;32m 1879\u001b[0m mode \u001b[38;5;241m+\u001b[39m\u001b[38;5;241m=\u001b[39m \u001b[38;5;124m\"\u001b[39m\u001b[38;5;124mb\u001b[39m\u001b[38;5;124m\"\u001b[39m\n\u001b[1;32m-> 1880\u001b[0m \u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39mhandles \u001b[38;5;241m=\u001b[39m get_handle(\n\u001b[0;32m 1881\u001b[0m f,\n\u001b[0;32m 1882\u001b[0m mode,\n\u001b[0;32m 1883\u001b[0m encoding\u001b[38;5;241m=\u001b[39m\u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39moptions\u001b[38;5;241m.\u001b[39mget(\u001b[38;5;124m\"\u001b[39m\u001b[38;5;124mencoding\u001b[39m\u001b[38;5;124m\"\u001b[39m, \u001b[38;5;28;01mNone\u001b[39;00m),\n\u001b[0;32m 1884\u001b[0m compression\u001b[38;5;241m=\u001b[39m\u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39moptions\u001b[38;5;241m.\u001b[39mget(\u001b[38;5;124m\"\u001b[39m\u001b[38;5;124mcompression\u001b[39m\u001b[38;5;124m\"\u001b[39m, \u001b[38;5;28;01mNone\u001b[39;00m),\n\u001b[0;32m 1885\u001b[0m memory_map\u001b[38;5;241m=\u001b[39m\u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39moptions\u001b[38;5;241m.\u001b[39mget(\u001b[38;5;124m\"\u001b[39m\u001b[38;5;124mmemory_map\u001b[39m\u001b[38;5;124m\"\u001b[39m, \u001b[38;5;28;01mFalse\u001b[39;00m),\n\u001b[0;32m 1886\u001b[0m is_text\u001b[38;5;241m=\u001b[39mis_text,\n\u001b[0;32m 1887\u001b[0m errors\u001b[38;5;241m=\u001b[39m\u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39moptions\u001b[38;5;241m.\u001b[39mget(\u001b[38;5;124m\"\u001b[39m\u001b[38;5;124mencoding_errors\u001b[39m\u001b[38;5;124m\"\u001b[39m, \u001b[38;5;124m\"\u001b[39m\u001b[38;5;124mstrict\u001b[39m\u001b[38;5;124m\"\u001b[39m),\n\u001b[0;32m 1888\u001b[0m storage_options\u001b[38;5;241m=\u001b[39m\u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39moptions\u001b[38;5;241m.\u001b[39mget(\u001b[38;5;124m\"\u001b[39m\u001b[38;5;124mstorage_options\u001b[39m\u001b[38;5;124m\"\u001b[39m, \u001b[38;5;28;01mNone\u001b[39;00m),\n\u001b[0;32m 1889\u001b[0m )\n\u001b[0;32m 1890\u001b[0m \u001b[38;5;28;01massert\u001b[39;00m \u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39mhandles \u001b[38;5;129;01mis\u001b[39;00m \u001b[38;5;129;01mnot\u001b[39;00m \u001b[38;5;28;01mNone\u001b[39;00m\n\u001b[0;32m 1891\u001b[0m f \u001b[38;5;241m=\u001b[39m \u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39mhandles\u001b[38;5;241m.\u001b[39mhandle\n",
63
+ "File \u001b[1;32mc:\\Users\\Rikhil Nellimarla\\.conda\\envs\\NLP_env\\Lib\\site-packages\\pandas\\io\\common.py:873\u001b[0m, in \u001b[0;36mget_handle\u001b[1;34m(path_or_buf, mode, encoding, compression, memory_map, is_text, errors, storage_options)\u001b[0m\n\u001b[0;32m 868\u001b[0m \u001b[38;5;28;01melif\u001b[39;00m \u001b[38;5;28misinstance\u001b[39m(handle, \u001b[38;5;28mstr\u001b[39m):\n\u001b[0;32m 869\u001b[0m \u001b[38;5;66;03m# Check whether the filename is to be opened in binary mode.\u001b[39;00m\n\u001b[0;32m 870\u001b[0m \u001b[38;5;66;03m# Binary mode does not support 'encoding' and 'newline'.\u001b[39;00m\n\u001b[0;32m 871\u001b[0m \u001b[38;5;28;01mif\u001b[39;00m ioargs\u001b[38;5;241m.\u001b[39mencoding \u001b[38;5;129;01mand\u001b[39;00m \u001b[38;5;124m\"\u001b[39m\u001b[38;5;124mb\u001b[39m\u001b[38;5;124m\"\u001b[39m \u001b[38;5;129;01mnot\u001b[39;00m \u001b[38;5;129;01min\u001b[39;00m ioargs\u001b[38;5;241m.\u001b[39mmode:\n\u001b[0;32m 872\u001b[0m \u001b[38;5;66;03m# Encoding\u001b[39;00m\n\u001b[1;32m--> 873\u001b[0m handle \u001b[38;5;241m=\u001b[39m \u001b[38;5;28mopen\u001b[39m(\n\u001b[0;32m 874\u001b[0m handle,\n\u001b[0;32m 875\u001b[0m ioargs\u001b[38;5;241m.\u001b[39mmode,\n\u001b[0;32m 876\u001b[0m encoding\u001b[38;5;241m=\u001b[39mioargs\u001b[38;5;241m.\u001b[39mencoding,\n\u001b[0;32m 877\u001b[0m errors\u001b[38;5;241m=\u001b[39merrors,\n\u001b[0;32m 878\u001b[0m newline\u001b[38;5;241m=\u001b[39m\u001b[38;5;124m\"\u001b[39m\u001b[38;5;124m\"\u001b[39m,\n\u001b[0;32m 879\u001b[0m )\n\u001b[0;32m 880\u001b[0m \u001b[38;5;28;01melse\u001b[39;00m:\n\u001b[0;32m 881\u001b[0m \u001b[38;5;66;03m# Binary mode\u001b[39;00m\n\u001b[0;32m 882\u001b[0m handle \u001b[38;5;241m=\u001b[39m \u001b[38;5;28mopen\u001b[39m(handle, ioargs\u001b[38;5;241m.\u001b[39mmode)\n",
64
+ "\u001b[1;31mFileNotFoundError\u001b[0m: [Errno 2] No such file or directory: '../Datasets/IMDB Dataset.csv'"
65
+ ]
66
+ }
67
+ ],
68
+ "source": [
69
+ "data = pd.read_csv('../Datasets/IMDB Dataset.csv')"
70
+ ]
71
+ },
72
+ {
73
+ "cell_type": "code",
74
+ "execution_count": null,
75
+ "metadata": {},
76
+ "outputs": [],
77
+ "source": [
78
+ "data=data.head(500)\n",
79
+ "sia = SentimentIntensityAnalyzer()"
80
+ ]
81
+ },
82
+ {
83
+ "cell_type": "code",
84
+ "execution_count": null,
85
+ "metadata": {},
86
+ "outputs": [
87
+ {
88
+ "ename": "KeyboardInterrupt",
89
+ "evalue": "",
90
+ "output_type": "error",
91
+ "traceback": [
92
+ "\u001b[1;31m---------------------------------------------------------------------------\u001b[0m",
93
+ "\u001b[1;31mKeyboardInterrupt\u001b[0m Traceback (most recent call last)",
94
+ "Cell \u001b[1;32mIn[23], line 2\u001b[0m\n\u001b[0;32m 1\u001b[0m \u001b[38;5;28;01mif\u001b[39;00m \u001b[38;5;124m'\u001b[39m\u001b[38;5;124mreview\u001b[39m\u001b[38;5;124m'\u001b[39m \u001b[38;5;129;01min\u001b[39;00m data\u001b[38;5;241m.\u001b[39mcolumns:\n\u001b[1;32m----> 2\u001b[0m data[\u001b[38;5;124m'\u001b[39m\u001b[38;5;124msentiment_score\u001b[39m\u001b[38;5;124m'\u001b[39m] \u001b[38;5;241m=\u001b[39m data[\u001b[38;5;124m'\u001b[39m\u001b[38;5;124mreview\u001b[39m\u001b[38;5;124m'\u001b[39m]\u001b[38;5;241m.\u001b[39mapply(\u001b[38;5;28;01mlambda\u001b[39;00m x: sia\u001b[38;5;241m.\u001b[39mpolarity_scores(x)[\u001b[38;5;124m'\u001b[39m\u001b[38;5;124mcompound\u001b[39m\u001b[38;5;124m'\u001b[39m])\n\u001b[0;32m 3\u001b[0m \u001b[38;5;28;01melse\u001b[39;00m:\n\u001b[0;32m 4\u001b[0m \u001b[38;5;28mprint\u001b[39m(\u001b[38;5;124m\"\u001b[39m\u001b[38;5;124mColumn \u001b[39m\u001b[38;5;124m'\u001b[39m\u001b[38;5;124mreview\u001b[39m\u001b[38;5;124m'\u001b[39m\u001b[38;5;124m not found.\u001b[39m\u001b[38;5;124m\"\u001b[39m)\n",
95
+ "File \u001b[1;32mc:\\Users\\Rikhil Nellimarla\\.conda\\envs\\NLP_env\\Lib\\site-packages\\pandas\\core\\series.py:4924\u001b[0m, in \u001b[0;36mSeries.apply\u001b[1;34m(self, func, convert_dtype, args, by_row, **kwargs)\u001b[0m\n\u001b[0;32m 4789\u001b[0m \u001b[38;5;28;01mdef\u001b[39;00m \u001b[38;5;21mapply\u001b[39m(\n\u001b[0;32m 4790\u001b[0m \u001b[38;5;28mself\u001b[39m,\n\u001b[0;32m 4791\u001b[0m func: AggFuncType,\n\u001b[1;32m (...)\u001b[0m\n\u001b[0;32m 4796\u001b[0m \u001b[38;5;241m*\u001b[39m\u001b[38;5;241m*\u001b[39mkwargs,\n\u001b[0;32m 4797\u001b[0m ) \u001b[38;5;241m-\u001b[39m\u001b[38;5;241m>\u001b[39m DataFrame \u001b[38;5;241m|\u001b[39m Series:\n\u001b[0;32m 4798\u001b[0m \u001b[38;5;250m \u001b[39m\u001b[38;5;124;03m\"\"\"\u001b[39;00m\n\u001b[0;32m 4799\u001b[0m \u001b[38;5;124;03m Invoke function on values of Series.\u001b[39;00m\n\u001b[0;32m 4800\u001b[0m \n\u001b[1;32m (...)\u001b[0m\n\u001b[0;32m 4915\u001b[0m \u001b[38;5;124;03m dtype: float64\u001b[39;00m\n\u001b[0;32m 4916\u001b[0m \u001b[38;5;124;03m \"\"\"\u001b[39;00m\n\u001b[0;32m 4917\u001b[0m \u001b[38;5;28;01mreturn\u001b[39;00m SeriesApply(\n\u001b[0;32m 4918\u001b[0m \u001b[38;5;28mself\u001b[39m,\n\u001b[0;32m 4919\u001b[0m func,\n\u001b[0;32m 4920\u001b[0m convert_dtype\u001b[38;5;241m=\u001b[39mconvert_dtype,\n\u001b[0;32m 4921\u001b[0m by_row\u001b[38;5;241m=\u001b[39mby_row,\n\u001b[0;32m 4922\u001b[0m args\u001b[38;5;241m=\u001b[39margs,\n\u001b[0;32m 4923\u001b[0m kwargs\u001b[38;5;241m=\u001b[39mkwargs,\n\u001b[1;32m-> 4924\u001b[0m )\u001b[38;5;241m.\u001b[39mapply()\n",
96
+ "File \u001b[1;32mc:\\Users\\Rikhil Nellimarla\\.conda\\envs\\NLP_env\\Lib\\site-packages\\pandas\\core\\apply.py:1427\u001b[0m, in \u001b[0;36mSeriesApply.apply\u001b[1;34m(self)\u001b[0m\n\u001b[0;32m 1424\u001b[0m \u001b[38;5;28;01mreturn\u001b[39;00m \u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39mapply_compat()\n\u001b[0;32m 1426\u001b[0m \u001b[38;5;66;03m# self.func is Callable\u001b[39;00m\n\u001b[1;32m-> 1427\u001b[0m \u001b[38;5;28;01mreturn\u001b[39;00m \u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39mapply_standard()\n",
97
+ "File \u001b[1;32mc:\\Users\\Rikhil Nellimarla\\.conda\\envs\\NLP_env\\Lib\\site-packages\\pandas\\core\\apply.py:1507\u001b[0m, in \u001b[0;36mSeriesApply.apply_standard\u001b[1;34m(self)\u001b[0m\n\u001b[0;32m 1501\u001b[0m \u001b[38;5;66;03m# row-wise access\u001b[39;00m\n\u001b[0;32m 1502\u001b[0m \u001b[38;5;66;03m# apply doesn't have a `na_action` keyword and for backward compat reasons\u001b[39;00m\n\u001b[0;32m 1503\u001b[0m \u001b[38;5;66;03m# we need to give `na_action=\"ignore\"` for categorical data.\u001b[39;00m\n\u001b[0;32m 1504\u001b[0m \u001b[38;5;66;03m# TODO: remove the `na_action=\"ignore\"` when that default has been changed in\u001b[39;00m\n\u001b[0;32m 1505\u001b[0m \u001b[38;5;66;03m# Categorical (GH51645).\u001b[39;00m\n\u001b[0;32m 1506\u001b[0m action \u001b[38;5;241m=\u001b[39m \u001b[38;5;124m\"\u001b[39m\u001b[38;5;124mignore\u001b[39m\u001b[38;5;124m\"\u001b[39m \u001b[38;5;28;01mif\u001b[39;00m \u001b[38;5;28misinstance\u001b[39m(obj\u001b[38;5;241m.\u001b[39mdtype, CategoricalDtype) \u001b[38;5;28;01melse\u001b[39;00m \u001b[38;5;28;01mNone\u001b[39;00m\n\u001b[1;32m-> 1507\u001b[0m mapped \u001b[38;5;241m=\u001b[39m obj\u001b[38;5;241m.\u001b[39m_map_values(\n\u001b[0;32m 1508\u001b[0m mapper\u001b[38;5;241m=\u001b[39mcurried, na_action\u001b[38;5;241m=\u001b[39maction, convert\u001b[38;5;241m=\u001b[39m\u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39mconvert_dtype\n\u001b[0;32m 1509\u001b[0m )\n\u001b[0;32m 1511\u001b[0m \u001b[38;5;28;01mif\u001b[39;00m \u001b[38;5;28mlen\u001b[39m(mapped) \u001b[38;5;129;01mand\u001b[39;00m \u001b[38;5;28misinstance\u001b[39m(mapped[\u001b[38;5;241m0\u001b[39m], ABCSeries):\n\u001b[0;32m 1512\u001b[0m \u001b[38;5;66;03m# GH#43986 Need to do list(mapped) in order to get treated as nested\u001b[39;00m\n\u001b[0;32m 1513\u001b[0m \u001b[38;5;66;03m# See also GH#25959 regarding EA support\u001b[39;00m\n\u001b[0;32m 1514\u001b[0m \u001b[38;5;28;01mreturn\u001b[39;00m obj\u001b[38;5;241m.\u001b[39m_constructor_expanddim(\u001b[38;5;28mlist\u001b[39m(mapped), index\u001b[38;5;241m=\u001b[39mobj\u001b[38;5;241m.\u001b[39mindex)\n",
98
+ "File \u001b[1;32mc:\\Users\\Rikhil Nellimarla\\.conda\\envs\\NLP_env\\Lib\\site-packages\\pandas\\core\\base.py:921\u001b[0m, in \u001b[0;36mIndexOpsMixin._map_values\u001b[1;34m(self, mapper, na_action, convert)\u001b[0m\n\u001b[0;32m 918\u001b[0m \u001b[38;5;28;01mif\u001b[39;00m \u001b[38;5;28misinstance\u001b[39m(arr, ExtensionArray):\n\u001b[0;32m 919\u001b[0m \u001b[38;5;28;01mreturn\u001b[39;00m arr\u001b[38;5;241m.\u001b[39mmap(mapper, na_action\u001b[38;5;241m=\u001b[39mna_action)\n\u001b[1;32m--> 921\u001b[0m \u001b[38;5;28;01mreturn\u001b[39;00m algorithms\u001b[38;5;241m.\u001b[39mmap_array(arr, mapper, na_action\u001b[38;5;241m=\u001b[39mna_action, convert\u001b[38;5;241m=\u001b[39mconvert)\n",
99
+ "File \u001b[1;32mc:\\Users\\Rikhil Nellimarla\\.conda\\envs\\NLP_env\\Lib\\site-packages\\pandas\\core\\algorithms.py:1743\u001b[0m, in \u001b[0;36mmap_array\u001b[1;34m(arr, mapper, na_action, convert)\u001b[0m\n\u001b[0;32m 1741\u001b[0m values \u001b[38;5;241m=\u001b[39m arr\u001b[38;5;241m.\u001b[39mastype(\u001b[38;5;28mobject\u001b[39m, copy\u001b[38;5;241m=\u001b[39m\u001b[38;5;28;01mFalse\u001b[39;00m)\n\u001b[0;32m 1742\u001b[0m \u001b[38;5;28;01mif\u001b[39;00m na_action \u001b[38;5;129;01mis\u001b[39;00m \u001b[38;5;28;01mNone\u001b[39;00m:\n\u001b[1;32m-> 1743\u001b[0m \u001b[38;5;28;01mreturn\u001b[39;00m lib\u001b[38;5;241m.\u001b[39mmap_infer(values, mapper, convert\u001b[38;5;241m=\u001b[39mconvert)\n\u001b[0;32m 1744\u001b[0m \u001b[38;5;28;01melse\u001b[39;00m:\n\u001b[0;32m 1745\u001b[0m \u001b[38;5;28;01mreturn\u001b[39;00m lib\u001b[38;5;241m.\u001b[39mmap_infer_mask(\n\u001b[0;32m 1746\u001b[0m values, mapper, mask\u001b[38;5;241m=\u001b[39misna(values)\u001b[38;5;241m.\u001b[39mview(np\u001b[38;5;241m.\u001b[39muint8), convert\u001b[38;5;241m=\u001b[39mconvert\n\u001b[0;32m 1747\u001b[0m )\n",
100
+ "File \u001b[1;32mlib.pyx:2972\u001b[0m, in \u001b[0;36mpandas._libs.lib.map_infer\u001b[1;34m()\u001b[0m\n",
101
+ "Cell \u001b[1;32mIn[23], line 2\u001b[0m, in \u001b[0;36m<lambda>\u001b[1;34m(x)\u001b[0m\n\u001b[0;32m 1\u001b[0m \u001b[38;5;28;01mif\u001b[39;00m \u001b[38;5;124m'\u001b[39m\u001b[38;5;124mreview\u001b[39m\u001b[38;5;124m'\u001b[39m \u001b[38;5;129;01min\u001b[39;00m data\u001b[38;5;241m.\u001b[39mcolumns:\n\u001b[1;32m----> 2\u001b[0m data[\u001b[38;5;124m'\u001b[39m\u001b[38;5;124msentiment_score\u001b[39m\u001b[38;5;124m'\u001b[39m] \u001b[38;5;241m=\u001b[39m data[\u001b[38;5;124m'\u001b[39m\u001b[38;5;124mreview\u001b[39m\u001b[38;5;124m'\u001b[39m]\u001b[38;5;241m.\u001b[39mapply(\u001b[38;5;28;01mlambda\u001b[39;00m x: sia\u001b[38;5;241m.\u001b[39mpolarity_scores(x)[\u001b[38;5;124m'\u001b[39m\u001b[38;5;124mcompound\u001b[39m\u001b[38;5;124m'\u001b[39m])\n\u001b[0;32m 3\u001b[0m \u001b[38;5;28;01melse\u001b[39;00m:\n\u001b[0;32m 4\u001b[0m \u001b[38;5;28mprint\u001b[39m(\u001b[38;5;124m\"\u001b[39m\u001b[38;5;124mColumn \u001b[39m\u001b[38;5;124m'\u001b[39m\u001b[38;5;124mreview\u001b[39m\u001b[38;5;124m'\u001b[39m\u001b[38;5;124m not found.\u001b[39m\u001b[38;5;124m\"\u001b[39m)\n",
102
+ "File \u001b[1;32mc:\\Users\\Rikhil Nellimarla\\.conda\\envs\\NLP_env\\Lib\\site-packages\\nltk\\sentiment\\vader.py:366\u001b[0m, in \u001b[0;36mSentimentIntensityAnalyzer.polarity_scores\u001b[1;34m(self, text)\u001b[0m\n\u001b[0;32m 355\u001b[0m \u001b[38;5;250m\u001b[39m\u001b[38;5;124;03m\"\"\"\u001b[39;00m\n\u001b[0;32m 356\u001b[0m \u001b[38;5;124;03mReturn a float for sentiment strength based on the input text.\u001b[39;00m\n\u001b[0;32m 357\u001b[0m \u001b[38;5;124;03mPositive values are positive valence, negative value are negative\u001b[39;00m\n\u001b[1;32m (...)\u001b[0m\n\u001b[0;32m 363\u001b[0m \u001b[38;5;124;03m matched as if it was a normal word in the sentence.\u001b[39;00m\n\u001b[0;32m 364\u001b[0m \u001b[38;5;124;03m\"\"\"\u001b[39;00m\n\u001b[0;32m 365\u001b[0m \u001b[38;5;66;03m# text, words_and_emoticons, is_cap_diff = self.preprocess(text)\u001b[39;00m\n\u001b[1;32m--> 366\u001b[0m sentitext \u001b[38;5;241m=\u001b[39m SentiText(\n\u001b[0;32m 367\u001b[0m text, \u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39mconstants\u001b[38;5;241m.\u001b[39mPUNC_LIST, \u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39mconstants\u001b[38;5;241m.\u001b[39mREGEX_REMOVE_PUNCTUATION\n\u001b[0;32m 368\u001b[0m )\n\u001b[0;32m 369\u001b[0m sentiments \u001b[38;5;241m=\u001b[39m []\n\u001b[0;32m 370\u001b[0m words_and_emoticons \u001b[38;5;241m=\u001b[39m sentitext\u001b[38;5;241m.\u001b[39mwords_and_emoticons\n",
103
+ "File \u001b[1;32mc:\\Users\\Rikhil Nellimarla\\.conda\\envs\\NLP_env\\Lib\\site-packages\\nltk\\sentiment\\vader.py:274\u001b[0m, in \u001b[0;36mSentiText.__init__\u001b[1;34m(self, text, punc_list, regex_remove_punctuation)\u001b[0m\n\u001b[0;32m 272\u001b[0m \u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39mPUNC_LIST \u001b[38;5;241m=\u001b[39m punc_list\n\u001b[0;32m 273\u001b[0m \u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39mREGEX_REMOVE_PUNCTUATION \u001b[38;5;241m=\u001b[39m regex_remove_punctuation\n\u001b[1;32m--> 274\u001b[0m \u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39mwords_and_emoticons \u001b[38;5;241m=\u001b[39m \u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39m_words_and_emoticons()\n\u001b[0;32m 275\u001b[0m \u001b[38;5;66;03m# doesn't separate words from\u001b[39;00m\n\u001b[0;32m 276\u001b[0m \u001b[38;5;66;03m# adjacent punctuation (keeps emoticons & contractions)\u001b[39;00m\n\u001b[0;32m 277\u001b[0m \u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39mis_cap_diff \u001b[38;5;241m=\u001b[39m \u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39mallcap_differential(\u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39mwords_and_emoticons)\n",
104
+ "File \u001b[1;32mc:\\Users\\Rikhil Nellimarla\\.conda\\envs\\NLP_env\\Lib\\site-packages\\nltk\\sentiment\\vader.py:306\u001b[0m, in \u001b[0;36mSentiText._words_and_emoticons\u001b[1;34m(self)\u001b[0m\n\u001b[0;32m 300\u001b[0m \u001b[38;5;250m\u001b[39m\u001b[38;5;124;03m\"\"\"\u001b[39;00m\n\u001b[0;32m 301\u001b[0m \u001b[38;5;124;03mRemoves leading and trailing puncutation\u001b[39;00m\n\u001b[0;32m 302\u001b[0m \u001b[38;5;124;03mLeaves contractions and most emoticons\u001b[39;00m\n\u001b[0;32m 303\u001b[0m \u001b[38;5;124;03m Does not preserve punc-plus-letter emoticons (e.g. :D)\u001b[39;00m\n\u001b[0;32m 304\u001b[0m \u001b[38;5;124;03m\"\"\"\u001b[39;00m\n\u001b[0;32m 305\u001b[0m wes \u001b[38;5;241m=\u001b[39m \u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39mtext\u001b[38;5;241m.\u001b[39msplit()\n\u001b[1;32m--> 306\u001b[0m words_punc_dict \u001b[38;5;241m=\u001b[39m \u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39m_words_plus_punc()\n\u001b[0;32m 307\u001b[0m wes \u001b[38;5;241m=\u001b[39m [we \u001b[38;5;28;01mfor\u001b[39;00m we \u001b[38;5;129;01min\u001b[39;00m wes \u001b[38;5;28;01mif\u001b[39;00m \u001b[38;5;28mlen\u001b[39m(we) \u001b[38;5;241m>\u001b[39m \u001b[38;5;241m1\u001b[39m]\n\u001b[0;32m 308\u001b[0m \u001b[38;5;28;01mfor\u001b[39;00m i, we \u001b[38;5;129;01min\u001b[39;00m \u001b[38;5;28menumerate\u001b[39m(wes):\n",
105
+ "File \u001b[1;32mc:\\Users\\Rikhil Nellimarla\\.conda\\envs\\NLP_env\\Lib\\site-packages\\nltk\\sentiment\\vader.py:293\u001b[0m, in \u001b[0;36mSentiText._words_plus_punc\u001b[1;34m(self)\u001b[0m\n\u001b[0;32m 291\u001b[0m words_only \u001b[38;5;241m=\u001b[39m {w \u001b[38;5;28;01mfor\u001b[39;00m w \u001b[38;5;129;01min\u001b[39;00m words_only \u001b[38;5;28;01mif\u001b[39;00m \u001b[38;5;28mlen\u001b[39m(w) \u001b[38;5;241m>\u001b[39m \u001b[38;5;241m1\u001b[39m}\n\u001b[0;32m 292\u001b[0m \u001b[38;5;66;03m# the product gives ('cat', ',') and (',', 'cat')\u001b[39;00m\n\u001b[1;32m--> 293\u001b[0m punc_before \u001b[38;5;241m=\u001b[39m {\u001b[38;5;124m\"\u001b[39m\u001b[38;5;124m\"\u001b[39m\u001b[38;5;241m.\u001b[39mjoin(p): p[\u001b[38;5;241m1\u001b[39m] \u001b[38;5;28;01mfor\u001b[39;00m p \u001b[38;5;129;01min\u001b[39;00m product(\u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39mPUNC_LIST, words_only)}\n\u001b[0;32m 294\u001b[0m punc_after \u001b[38;5;241m=\u001b[39m {\u001b[38;5;124m\"\u001b[39m\u001b[38;5;124m\"\u001b[39m\u001b[38;5;241m.\u001b[39mjoin(p): p[\u001b[38;5;241m0\u001b[39m] \u001b[38;5;28;01mfor\u001b[39;00m p \u001b[38;5;129;01min\u001b[39;00m product(words_only, \u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39mPUNC_LIST)}\n\u001b[0;32m 295\u001b[0m words_punc_dict \u001b[38;5;241m=\u001b[39m punc_before\n",
106
+ "\u001b[1;31mKeyboardInterrupt\u001b[0m: "
107
+ ]
108
+ }
109
+ ],
110
+ "source": [
111
+ "if 'review' in data.columns:\n",
112
+ " data['sentiment_score'] = data['review'].apply(lambda x: sia.polarity_scores(x)['compound'])\n",
113
+ "else:\n",
114
+ " print(\"Column 'review' not found.\")"
115
+ ]
116
+ },
117
+ {
118
+ "cell_type": "code",
119
+ "execution_count": null,
120
+ "metadata": {},
121
+ "outputs": [
122
+ {
123
+ "name": "stdout",
124
+ "output_type": "stream",
125
+ "text": [
126
+ "This oatmeal is not good. Its mushy, soft, I don't like it. Quaker Oats is the way to go.\n"
127
+ ]
128
+ }
129
+ ],
130
+ "source": [
131
+ "print(data['sentiment'].value_counts())\n"
132
+ ]
133
+ },
134
+ {
135
+ "cell_type": "code",
136
+ "execution_count": null,
137
+ "metadata": {},
138
+ "outputs": [
139
+ {
140
+ "data": {
141
+ "text/plain": [
142
+ "['This', 'oatmeal', 'is', 'not', 'good', '.', 'Its', 'mushy', ',', 'soft']"
143
+ ]
144
+ },
145
+ "execution_count": 6,
146
+ "metadata": {},
147
+ "output_type": "execute_result"
148
+ }
149
+ ],
150
+ "source": [
151
+ "plt.figure(figsize=(8, 6))\n",
152
+ "sns.countplot(x='sentiment', data=data, palette='Set1')\n",
153
+ "plt.title('Sentiment Distribution')\n",
154
+ "plt.show()"
155
+ ]
156
+ },
157
+ {
158
+ "cell_type": "code",
159
+ "execution_count": null,
160
+ "metadata": {},
161
+ "outputs": [
162
+ {
163
+ "data": {
164
+ "text/plain": [
165
+ "[('This', 'DT'),\n",
166
+ " ('oatmeal', 'NN'),\n",
167
+ " ('is', 'VBZ'),\n",
168
+ " ('not', 'RB'),\n",
169
+ " ('good', 'JJ'),\n",
170
+ " ('.', '.'),\n",
171
+ " ('Its', 'PRP$'),\n",
172
+ " ('mushy', 'NN'),\n",
173
+ " (',', ','),\n",
174
+ " ('soft', 'JJ')]"
175
+ ]
176
+ },
177
+ "execution_count": 7,
178
+ "metadata": {},
179
+ "output_type": "execute_result"
180
+ }
181
+ ],
182
+ "source": [
183
+ "data.to_csv('sentiment_analysis_results.csv', index=False)"
184
+ ]
185
+ },
186
+ {
187
+ "cell_type": "code",
188
+ "execution_count": null,
189
+ "metadata": {},
190
+ "outputs": [
191
+ {
192
+ "name": "stdout",
193
+ "output_type": "stream",
194
+ "text": [
195
+ "(S\n",
196
+ " This/DT\n",
197
+ " oatmeal/NN\n",
198
+ " is/VBZ\n",
199
+ " not/RB\n",
200
+ " good/JJ\n",
201
+ " ./.\n",
202
+ " Its/PRP$\n",
203
+ " mushy/NN\n",
204
+ " ,/,\n",
205
+ " soft/JJ\n",
206
+ " ,/,\n",
207
+ " I/PRP\n",
208
+ " do/VBP\n",
209
+ " n't/RB\n",
210
+ " like/VB\n",
211
+ " it/PRP\n",
212
+ " ./.\n",
213
+ " (ORGANIZATION Quaker/NNP Oats/NNPS)\n",
214
+ " is/VBZ\n",
215
+ " the/DT\n",
216
+ " way/NN\n",
217
+ " to/TO\n",
218
+ " go/VB\n",
219
+ " ./.)\n"
220
+ ]
221
+ }
222
+ ],
223
+ "source": [
224
+ "print(data[['review', 'sentiment_score', 'sentiment']].head())"
225
+ ]
226
+ }
227
+ ],
228
+ "metadata": {
229
+ "kernelspec": {
230
+ "display_name": "NLP_env",
231
+ "language": "python",
232
+ "name": "python3"
233
+ },
234
+ "language_info": {
235
+ "codemirror_mode": {
236
+ "name": "ipython",
237
+ "version": 3
238
+ },
239
+ "file_extension": ".py",
240
+ "mimetype": "text/x-python",
241
+ "name": "python",
242
+ "nbconvert_exporter": "python",
243
+ "pygments_lexer": "ipython3",
244
+ "version": "3.12.4"
245
+ }
246
+ },
247
+ "nbformat": 4,
248
+ "nbformat_minor": 2
249
+ }
requirements.txt ADDED
@@ -0,0 +1,5 @@
 
 
 
 
 
 
1
+ pandas
2
+ torch
3
+ nltk
4
+ streamlit
5
+ transformers
sentiment_analysis/transformers_analyzer.py ADDED
@@ -0,0 +1,35 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import torch
2
+ from transformers import AutoTokenizer, AutoModelForSequenceClassification
3
+ import pandas as pd
4
+
5
+ def analyze_sentiment_transformers(reviews):
6
+ # Set device to "cpu" if CUDA (GPU) is unavailable
7
+ device = torch.device("cuda" if torch.cuda.is_available() else "cpu")
8
+
9
+ # Initialize tokenizer and model
10
+ tokenizer = AutoTokenizer.from_pretrained("distilbert-base-uncased")
11
+ model = AutoModelForSequenceClassification.from_pretrained("distilbert-base-uncased").to(device)
12
+
13
+ results = []
14
+ for review in reviews:
15
+ # Tokenize with truncation and padding to max length
16
+ tokenized_review = tokenizer(review, return_tensors="pt", truncation=True, padding="max_length", max_length=512)
17
+ tokenized_review = {key: val.to(device) for key, val in tokenized_review.items()} # Ensure tensors are on the correct device
18
+
19
+ # Get the model's output (logits)
20
+ with torch.no_grad():
21
+ outputs = model(**tokenized_review)
22
+
23
+ # Convert logits to probabilities
24
+ probabilities = torch.nn.functional.softmax(outputs.logits, dim=-1)
25
+
26
+ # Get the predicted label and score
27
+ score = probabilities.max().item()
28
+ label = "POSITIVE" if torch.argmax(probabilities).item() == 1 else "NEGATIVE"
29
+
30
+ # Append the result
31
+ results.append({"label": label, "score": score})
32
+
33
+ # Convert results to DataFrame
34
+ sentiment_df = pd.DataFrame(results)
35
+ return sentiment_df
sentiment_analysis/vader_analyzer.py ADDED
@@ -0,0 +1,18 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import nltk
2
+ from nltk.sentiment.vader import SentimentIntensityAnalyzer
3
+ import pandas as pd
4
+
5
+ # Download VADER lexicon if not already available
6
+ try:
7
+ nltk.data.find('sentiment/vader_lexicon.zip')
8
+ except LookupError:
9
+ nltk.download('vader_lexicon')
10
+
11
+ def analyze_sentiment_vader(reviews):
12
+ sia = SentimentIntensityAnalyzer()
13
+ results = []
14
+ for review in reviews:
15
+ score = sia.polarity_scores(review)
16
+ results.append(score)
17
+ sentiment_df = pd.DataFrame(results)
18
+ return sentiment_df
sentiment_analysis_results.csv ADDED
The diff for this file is too large to render. See raw diff
 
streamlit_app.py ADDED
@@ -0,0 +1,48 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import streamlit as st
2
+ import pandas as pd
3
+ from sentiment_analysis.vader_analyzer import analyze_sentiment_vader
4
+ from sentiment_analysis.transformers_analyzer import analyze_sentiment_transformers
5
+
6
+ st.title("Bulk Sentiment Analysis for Reviews")
7
+
8
+ # Step 1: File Upload
9
+ uploaded_file = st.file_uploader("Upload your review file", type=["csv", "xlsx"])
10
+
11
+ if uploaded_file is not None:
12
+ # Read the file into a DataFrame
13
+ if uploaded_file.name.endswith('.csv'):
14
+ df = pd.read_csv(uploaded_file)
15
+ else:
16
+ df = pd.read_excel(uploaded_file)
17
+
18
+ # Check the number of entries and truncate if needed
19
+ if len(df) > 1000:
20
+ df = df.head(1000)
21
+ st.error("The file contains more than 1,000 entries. Only the first 1,000 reviews are processed.")
22
+
23
+ st.write("Data Preview:", df.head())
24
+
25
+ # Step 2: Model Selection
26
+ st.write("Select sentiment analysis models:")
27
+ use_vader = st.checkbox("Vader")
28
+ use_transformers = st.checkbox("Transformers")
29
+
30
+ # Ensure "review" column exists
31
+ df.columns = df.columns.str.lower()
32
+ if 'review' in df.columns:
33
+ # Step 3: Process reviews with Selected Models
34
+ if use_vader:
35
+ vader_results = analyze_sentiment_vader(df["review"])
36
+ df = pd.concat([df, vader_results], axis=1)
37
+ st.write("Vader Analysis Results", df.head())
38
+
39
+ if use_transformers:
40
+ transformers_results = analyze_sentiment_transformers(df["review"])
41
+ df = pd.concat([df, transformers_results], axis=1)
42
+ st.write("Transformers Analysis Results", df.head())
43
+
44
+ # Step 4: Download Results
45
+ csv = df.to_csv(index=False)
46
+ st.download_button("Download CSV", csv, "sentiment_analysis_results.csv", "text/csv")
47
+ else:
48
+ st.error("Please make sure the file has a 'review' column with review text.")