{ "cells": [ { "cell_type": "code", "execution_count": 1, "metadata": {}, "outputs": [ { "data": { "text/html": [ "
\n", "\n", "\n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", "
rootmean_rankingstd_rankingword_matches
0challeng420264['challenges', 'challenge', 'challenging']
1SDG4230['sdgs']
2ESG6360['esg']
3recycling860154['recycling', 'shiprecycling']
4reduc1242965['reduce', 'reducing', 'reduced', 'reduction',...
5planet12950['planet']
6CSR13710['csr']
7sustainab15271873['sustainability', 'sustainable', 'sustainable...
8clean1695837['theoceancleanup', 'cleanup', 'clean', 'clean...
9methanol17001044['methanol', 'emethanol']
10future17701684['future', 'futureproofing']
11garbage1808612['garbage', 'greatpacificgarbagepatch']
12responsib18891135['responsible', 'responsibility', 'responsibly...
13carb19521526['decarbonisation', 'carbon', 'decarbonization...
14chang21481667['change', 'changing', 'climatechange', 'chang...
15ocean21691629['ocean', 'theoceancleanup', 'oceans', 'oceanp...
16plastic23201490['plastic', 'oceanplastic', 'plasticwaste', 'p...
17neutral23821798['neutral', 'carbonneutral', 'neutrality', 'co...
18environment23911574['environment', 'environmental', 'unenvironmen...
19green23931158['green', 'greenfuels', 'greener', 'greenfuel'...
20sulphur28281772['sulphur', 'lowsulphur']
21emissions29252046['emissions', 'carbonemissions', 'zeroemissions']
22zero29771646['zero', 'netzero', 'zerocarbon', 'zerocarbons...
23climate30462095['climateaction', 'climate', 'climatechange', ...
24eco30741269['ecosystem', 'eco', 'maerskecodelivery', 'eco...
25mission32471950['emissions', 'mission', 'eucommission', 'carb...
26CO233902275['co2', 'co2neutral', 'co2emission']
27bio36871274['biofuel', 'biofuels', 'biohuts', 'biodiversi...
\n", "
" ], "text/plain": [ " root mean_ranking std_ranking \n", "0 challeng 420 264 \\\n", "1 SDG 423 0 \n", "2 ESG 636 0 \n", "3 recycling 860 154 \n", "4 reduc 1242 965 \n", "5 planet 1295 0 \n", "6 CSR 1371 0 \n", "7 sustainab 1527 1873 \n", "8 clean 1695 837 \n", "9 methanol 1700 1044 \n", "10 future 1770 1684 \n", "11 garbage 1808 612 \n", "12 responsib 1889 1135 \n", "13 carb 1952 1526 \n", "14 chang 2148 1667 \n", "15 ocean 2169 1629 \n", "16 plastic 2320 1490 \n", "17 neutral 2382 1798 \n", "18 environment 2391 1574 \n", "19 green 2393 1158 \n", "20 sulphur 2828 1772 \n", "21 emissions 2925 2046 \n", "22 zero 2977 1646 \n", "23 climate 3046 2095 \n", "24 eco 3074 1269 \n", "25 mission 3247 1950 \n", "26 CO2 3390 2275 \n", "27 bio 3687 1274 \n", "\n", " word_matches \n", "0 ['challenges', 'challenge', 'challenging'] \n", "1 ['sdgs'] \n", "2 ['esg'] \n", "3 ['recycling', 'shiprecycling'] \n", "4 ['reduce', 'reducing', 'reduced', 'reduction',... \n", "5 ['planet'] \n", "6 ['csr'] \n", "7 ['sustainability', 'sustainable', 'sustainable... \n", "8 ['theoceancleanup', 'cleanup', 'clean', 'clean... \n", "9 ['methanol', 'emethanol'] \n", "10 ['future', 'futureproofing'] \n", "11 ['garbage', 'greatpacificgarbagepatch'] \n", "12 ['responsible', 'responsibility', 'responsibly... \n", "13 ['decarbonisation', 'carbon', 'decarbonization... \n", "14 ['change', 'changing', 'climatechange', 'chang... \n", "15 ['ocean', 'theoceancleanup', 'oceans', 'oceanp... \n", "16 ['plastic', 'oceanplastic', 'plasticwaste', 'p... \n", "17 ['neutral', 'carbonneutral', 'neutrality', 'co... \n", "18 ['environment', 'environmental', 'unenvironmen... \n", "19 ['green', 'greenfuels', 'greener', 'greenfuel'... \n", "20 ['sulphur', 'lowsulphur'] \n", "21 ['emissions', 'carbonemissions', 'zeroemissions'] \n", "22 ['zero', 'netzero', 'zerocarbon', 'zerocarbons... \n", "23 ['climateaction', 'climate', 'climatechange', ... \n", "24 ['ecosystem', 'eco', 'maerskecodelivery', 'eco... \n", "25 ['emissions', 'mission', 'eucommission', 'carb... \n", "26 ['co2', 'co2neutral', 'co2emission'] \n", "27 ['biofuel', 'biofuels', 'biohuts', 'biodiversi... " ] }, "execution_count": 1, "metadata": {}, "output_type": "execute_result" } ], "source": [ "# load words related to sustainability\n", "import pandas as pd\n", "\n", "path = 'word_root_ranking.csv'\n", "df = pd.read_csv(path, index_col=0)\n", "\n", "df" ] }, { "cell_type": "code", "execution_count": 2, "metadata": {}, "outputs": [ { "data": { "text/plain": [ "['challenges',\n", " 'challenge',\n", " 'challenging',\n", " 'sdgs',\n", " 'esg',\n", " 'recycling',\n", " 'shiprecycling',\n", " 'reduce',\n", " 'reducing',\n", " 'reduced']" ] }, "execution_count": 2, "metadata": {}, "output_type": "execute_result" } ], "source": [ "# collect words into list\n", "\n", "related_words_list = list()\n", "for row in df['word_matches']:\n", " word_list = row.strip('][').replace(\"'\", '').split(', ')\n", " for word in word_list:\n", " related_words_list.append(word)\n", " \n", "related_words_list[:10]" ] }, { "cell_type": "code", "execution_count": 19, "metadata": {}, "outputs": [ { "data": { "text/plain": [ "1511" ] }, "execution_count": 19, "metadata": {}, "output_type": "execute_result" } ], "source": [ "# load in all tweet texts\n", "from pathlib import Path\n", "import re\n", "\n", "tweet_dir = Path('tweets').resolve()\n", "path_list = list(tweet_dir.glob('*.txt'))\n", "score_list = list()\n", "for path in path_list:\n", " with open(path, 'r') as f:\n", " text = f.read()\n", " text = re.search(\n", " pattern='(?<=text: )(.|\\n)*(?=images: )',\n", " string=text\n", " ).group(0)\n", " score = sum([related_word in text for related_word in related_words_list])\n", " score_list.append(score)\n", "\n", "len(score_list)" ] }, { "cell_type": "code", "execution_count": 20, "metadata": {}, "outputs": [ { "data": { "text/plain": [ "15" ] }, "execution_count": 20, "metadata": {}, "output_type": "execute_result" } ], "source": [ "max(score_list)" ] }, { "cell_type": "code", "execution_count": 25, "metadata": {}, "outputs": [ { "data": { "text/html": [ "
\n", "\n", "\n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", " \n", "
pathscore
0/home/brian/solveig_master/tweets/2021-08-24-0...15
1/home/brian/solveig_master/tweets/2021-06-22-1...14
2/home/brian/solveig_master/tweets/2020-07-13-1...14
3/home/brian/solveig_master/tweets/2022-04-28-1...13
4/home/brian/solveig_master/tweets/2022-10-20-1...13
.........
1506/home/brian/solveig_master/tweets/2022-08-31-1...0
1507/home/brian/solveig_master/tweets/2022-12-17-0...0
1508/home/brian/solveig_master/tweets/2021-09-24-0...0
1509/home/brian/solveig_master/tweets/2014-02-06-1...0
1510/home/brian/solveig_master/tweets/2017-10-12-1...0
\n", "

1511 rows × 2 columns

\n", "
" ], "text/plain": [ " path score\n", "0 /home/brian/solveig_master/tweets/2021-08-24-0... 15\n", "1 /home/brian/solveig_master/tweets/2021-06-22-1... 14\n", "2 /home/brian/solveig_master/tweets/2020-07-13-1... 14\n", "3 /home/brian/solveig_master/tweets/2022-04-28-1... 13\n", "4 /home/brian/solveig_master/tweets/2022-10-20-1... 13\n", "... ... ...\n", "1506 /home/brian/solveig_master/tweets/2022-08-31-1... 0\n", "1507 /home/brian/solveig_master/tweets/2022-12-17-0... 0\n", "1508 /home/brian/solveig_master/tweets/2021-09-24-0... 0\n", "1509 /home/brian/solveig_master/tweets/2014-02-06-1... 0\n", "1510 /home/brian/solveig_master/tweets/2017-10-12-1... 0\n", "\n", "[1511 rows x 2 columns]" ] }, "execution_count": 25, "metadata": {}, "output_type": "execute_result" } ], "source": [ "df = pd.DataFrame(\n", " {\n", " 'path': path_list,\n", " 'score': score_list\n", " }\n", ")\n", "\n", "df = df.sort_values(by='score', ascending=False)\n", "df.reset_index(drop=True, inplace=True)\n", "df" ] }, { "cell_type": "code", "execution_count": 27, "metadata": {}, "outputs": [], "source": [ "df.to_csv('tweets_relevance_list.csv')" ] }, { "cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": [] } ], "metadata": { "kernelspec": { "display_name": "venv", "language": "python", "name": "python3" }, "language_info": { "codemirror_mode": { "name": "ipython", "version": 3 }, "file_extension": ".py", "mimetype": "text/x-python", "name": "python", "nbconvert_exporter": "python", "pygments_lexer": "ipython3", "version": "3.8.10" }, "orig_nbformat": 4 }, "nbformat": 4, "nbformat_minor": 2 }