{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt \nimport seaborn as sns\nimport squarify ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/coursera-course-dataset/coursea_data.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.head(3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig, (ax1, ax2,ax3) = plt.subplots(3, 1,figsize=(10,20))\n\n\nsns.countplot(df['course_Certificate_type'], palette = 'seismic', ax=ax1)\nplt.title('course_Certificate_type count', fontsize = 20)\n\nsns.countplot(df['course_difficulty'], palette = 'gnuplot', ax=ax2)\nplt.title('course_difficulty count', fontsize = 20)\n\n\nsns.countplot(df['course_rating'], palette = 'PuRd')\nplt.title('course_rating count', fontsize = 20)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(df.course_Certificate_type.value_counts())\npie = df.groupby('course_Certificate_type').size()\nplt.figure()\npie.plot(kind='pie', subplots=True, figsize=(8, 8))\nplt.title(\"Pie Chart of course_Certificate_type\")\nplt.ylabel(\"\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tree = df.groupby('course_difficulty').size().reset_index(name='counts')\nlabels = tree.apply(lambda x: str(x[0]) + \"\\n (\" + str(x[1]) + \")\", axis=1)\nsizes = tree['counts'].values.tolist()\ncolors = [plt.cm.Spectral(i/float(len(labels))) for i in range(len(labels))]\n\n# Draw Plot\nplt.figure(figsize=(12,8), dpi= 80)\nsquarify.plot(sizes=sizes, label=labels, color=colors, alpha=.8)\n\n# Decorate\nplt.title('Treemap of Course Difficulty')\nplt.axis('off')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"      ","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"  ","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"df.course_students_enrolled = df.course_students_enrolled.apply(lambda x : float(str(x).replace('k', '').replace('m',''))*1000)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"course_pop = df.groupby('course_organization')['course_students_enrolled'].sum().reset_index()\nTop15_popular= course_pop.sort_values(by='course_students_enrolled', ascending=False).head(15)\nTop15_unpopular=course_pop.sort_values(by='course_students_enrolled', ascending=True).head(15)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig, (ax1, ax2) = plt.subplots(2, 1,figsize=(10,15))\nsns.barplot(x=Top15_popular[\"course_students_enrolled\"],y=Top15_popular['course_organization'],ax=ax1)\nax1.set_title(\"Top 15 Popular Organization\")\n\nsns.barplot(x=Top15_unpopular[\"course_students_enrolled\"],y=Top15_unpopular['course_organization'],ax=ax2)\nax2.set_title(\"Top 15 Unpopular Organization\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df2 = df[['course_students_enrolled', 'course_organization']].groupby('course_organization').apply(lambda x: x.mean())\ndf2.sort_values('course_students_enrolled', inplace=True)\ndf2.reset_index(inplace=True)\n\n# Draw plot\nfig, ax = plt.subplots(figsize=(35,25), dpi= 80)\nax.hlines(y=df2.index, xmin=11, xmax=26, color='gray', alpha=0.7, linewidth=1, linestyles='dashdot')\nax.scatter(y=df2.index, x=df2.course_students_enrolled, s=75, color='firebrick', alpha=0.7)\n\n# Title, Label, Ticks and Ylim\nax.set_title('Dot Plot for Entrollment for Every Organization', fontdict={'size':22})\nax.set_xlabel('Entrollment number')\nax.set_yticks(df2.index)\nax.set_yticklabels(df.course_organization.str.title(), fontdict={'horizontalalignment': 'right'})\nax.set_xlim(0, 250000)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"  ","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"  ","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"def plot_count(feature, title, df, size=1, show_percents=False):\n    f, ax = plt.subplots(1,1, figsize=(4*size,4))\n    total = float(len(df))\n    g = sns.countplot(df[feature], order = df[feature].value_counts().index[0:20], palette='Set3')\n    g.set_title(\"Number of {}\".format(title))\n    if(size > 2):\n        plt.xticks(rotation=90, size=10)\n    if(show_percents):\n        for p in ax.patches:\n            height = p.get_height()\n            ax.text(p.get_x()+p.get_width()/2.,\n                    height + 3,\n                    '{:1.2f}%'.format(100*height/total),\n                    ha=\"center\") \n    ax.set_xticklabels(ax.get_xticklabels());\n    plt.show()    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_count('course_title', 'Top 20 course_title', df, 3.5)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"   ","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"x_var = 'course_rating'\ngroupby_var = 'course_difficulty'\ndf_agg = df.loc[:, [x_var, groupby_var]].groupby(groupby_var)\nvals = [df[x_var].values.tolist() for i, df in df_agg]\n\n# Draw\nplt.figure(figsize=(20,10), dpi= 80)\ncolors = [plt.cm.Spectral(i/float(len(vals)-1)) for i in range(len(vals))]\nn, bins, patches = plt.hist(vals, 30, stacked=True, density=False, color=colors[:len(vals)])\n\n# Decoration\nplt.legend({group:col for group, col in zip(np.unique(df[groupby_var]).tolist(), colors[:len(vals)])})\nplt.title(f\"Stacked Histogram of ${x_var}$ colored by ${groupby_var}$\", fontsize=22)\nplt.xlabel(x_var)\nplt.ylabel(\"Frequency\")\nplt.ylim(0, 265)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"  ","execution_count":null}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}