{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Published on July 01, 2023. By Marília Prata, mpwolke.","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-07-01T23:12:03.775832Z","iopub.execute_input":"2023-07-01T23:12:03.776269Z","iopub.status.idle":"2023-07-01T23:12:03.786776Z","shell.execute_reply.started":"2023-07-01T23:12:03.776237Z","shell.execute_reply":"2023-07-01T23:12:03.785756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#I didn't know that we could draw with Python. Let's check those amazing kagglers performances.","metadata":{}},{"cell_type":"code","source":"! pip install mediapipe -qq","metadata":{"execution":{"iopub.status.busy":"2023-07-01T23:27:31.176849Z","iopub.execute_input":"2023-07-01T23:27:31.178133Z","iopub.status.idle":"2023-07-01T23:27:47.454688Z","shell.execute_reply.started":"2023-07-01T23:27:31.178092Z","shell.execute_reply":"2023-07-01T23:27:47.452983Z"},"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Victor nailed on Google American Sign Language Competition. Check Victor's Dumetz work.\n\nAnd maybe you could join that Competition too, there is still two months to go.","metadata":{}},{"cell_type":"code","source":"#By Victor Dumetz https://www.kaggle.com/code/victordumetz/gaslfr-svg-visualisation\n\nimport json\nimport os\n\nfrom IPython.display import HTML\nfrom mediapipe import solutions as mp\nimport numpy as np\nimport pandas as pd\n\n\nINPUT_DIR = \"/kaggle/input/asl-fingerspelling\"\nOUTPUT_DIR = \"/kaggle/working\"\n\nHAND_CONNECTIONS = mp.hands_connections.HAND_CONNECTIONS\nPOSE_CONNECTIONS = mp.pose_connections.POSE_CONNECTIONS\nFACE_CONNECTIONS = mp.face_mesh_connections.FACEMESH_CONTOURS","metadata":{"execution":{"iopub.status.busy":"2023-07-01T23:27:53.52738Z","iopub.execute_input":"2023-07-01T23:27:53.527842Z","iopub.status.idle":"2023-07-01T23:28:03.088727Z","shell.execute_reply.started":"2023-07-01T23:27:53.527805Z","shell.execute_reply":"2023-07-01T23:28:03.087799Z"},"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#By Victor Dumetz https://www.kaggle.com/code/victordumetz/gaslfr-svg-visualisation\n\ndf = pd.read_csv(os.path.join(INPUT_DIR, \"train.csv\"))\nsup_df = pd.read_csv(os.path.join(INPUT_DIR, \"supplemental_metadata.csv\"))","metadata":{"execution":{"iopub.status.busy":"2023-07-01T23:29:06.190025Z","iopub.execute_input":"2023-07-01T23:29:06.191205Z","iopub.status.idle":"2023-07-01T23:29:06.461203Z","shell.execute_reply.started":"2023-07-01T23:29:06.191165Z","shell.execute_reply":"2023-07-01T23:29:06.460249Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#By Victor Dumetz https://www.kaggle.com/code/victordumetz/gaslfr-svg-visualisation\n\ndef get_full_landmark_path(path):\n    return os.path.join(INPUT_DIR, path)\n\nlandmarks = pd.read_parquet(get_full_landmark_path(df.loc[0][\"path\"]))\n\nselected_columns = list(landmarks.columns[~landmarks.columns.str.startswith('z_')])\ninference_args = {\"selected_columns\": selected_columns}\n\nwith open(os.path.join(OUTPUT_DIR, \"inference_args.json\"), \"w\") as f:\n    json.dump(inference_args, f)\n\nwith open(os.path.join(OUTPUT_DIR, \"inference_args.json\")) as f:\n    inference_args = json.load(f)\n    SELECTED_COLUMNS = inference_args[\"selected_columns\"]\n\ndef get_nan_ranges(sequence):\n    \"\"\"Get list of frame indices that start or end a nan sequence.\n\n    The first index corresponds to the frame of the first missing value. The following\n    indices correspond to every bound of missing or valid value range.\n\n    Args:\n        sequence (pd.DataFrame): The sequence landmarks data.\n\n    Returns:\n        dict: Dictionary with landmarks as index and lists of frame indices as values.\n    \"\"\"\n    # keep only x since missing x => missing y\n    s = sequence[\"x\"]\n\n    # replace keypoints with landmarks since a missing keypoint => missing landmark\n    s = s.reset_index(\"keypoint\")\n    s[\"keypoint\"] = s[\"keypoint\"].str.replace(r\"_[0-9]+\", \"\", regex=True)\n    s.rename({\"keypoint\": \"landmark\"}, axis=1, inplace=True)\n    s = s.reset_index().set_index([\"landmark\", \"frame\"])\n    s = s.groupby([\"landmark\", \"frame\"]).mean()[\"x\"]\n\n    # for each landmark and range of missing value returns a df with landmark as key\n    # and start and end frame of the range as values\n    m = s.isnull()\n    nan_ranges = [\n        g.reset_index(\"frame\").groupby(\"landmark\").agg(start=(\"frame\", \"min\"), end=(\"frame\", \"max\"))\n        for _, g in s[m].groupby((~m).groupby(level=\"landmark\").cumsum())\n    ]\n\n    # extract start and end and flatten\n    nan_ranges = {\n        r.index[0]: [\n            rrr for rr in nan_ranges for rrr in (rr.start.squeeze(), rr.end.squeeze()) if rr.index[0] == r.index[0]\n        ]\n        for r in nan_ranges\n    }\n\n    return nan_ranges\n\n\ndef animate_hidden_attribute(landmark_nan_ranges, n_frames, dur):\n    \"\"\"Generate an <animate /> tag that hides the landmark when its coordinates are nan.\n\n    Args:\n        landmark_nan_ranges (list): List of frame indices that start or end a nan sequence.\n        n_frames (int): Length of the sequence.\n        dur (float): Frame duration (in seconds).\n\n    Returns:\n        str: <animate /> tag for the hidden attribute.\n    \"\"\"\n    landmark_nan_ranges = list(landmark_nan_ranges / n_frames)\n\n    v = (\";\").join([\"visible\"] + [\"hidden\", \"visible\"] * (len(landmark_nan_ranges) // 2) + [\"visible\"])\n    kt = (\";\").join([str(k) for k in [0.0] + landmark_nan_ranges + [1.0]])\n\n    return f'<animate attributeName=\"visibility\" values=\"{v}\" dur=\"{dur}s\" keyTimes=\"{kt}\" repeatCount=\"indefinite\"/>'\n\n\ndef animate_attribute(attribute_name, values, dur):\n    \"\"\"Generate an <animate /> tag for the given attribute.\n\n    Args:\n        attribute_name (str): Name of the attribute to animate.\n        values (str): Values to be taken by the attribute, separated by semicolons.\n        dur (float): Frame duration (in seconds).\n\n    Returns:\n        str: <animate /> tag for the given attribute.\n    \"\"\"\n    return f'<animate attributeName=\"{attribute_name}\" values=\"{values}\" dur=\"{dur}\" repeatCount=\"indefinite\"/>'\n\n\ndef draw_connection(x0, x1, y0, y1, landmark_nan_ranges, n_frames, fps, color=\"black\", stroke_width=0.001):\n    \"\"\"Generate a <line></line> block for a connection its animation.\n\n    Args:\n        x0 (float): x0 coordinate.\n        x1 (float): x1 coordinate.\n        y0 (float): y0 coordinate.\n        y1 (float): y1 coordinate.\n        landmark_nan_ranges (list): List of frame indices that start or end a nan sequence.\n        n_frames (int): Number of frames in the sequence.\n        fps (int): FPS used for the animation.\n        color (str, optional): Color of the connection. Defaults to \"black\".\n        stroke_width (float, optional): Width of the stroke. Defaults to 0.001.\n\n    Returns:\n        str: <line></line> block for the connection.\n    \"\"\"\n    dur = n_frames / fps\n    animation = \"\".join(\n        [\n            animate_attribute(\"x1\", \";\".join(x0.values), dur),\n            animate_attribute(\"x2\", \";\".join(x1.values), dur),\n            animate_attribute(\"y1\", \";\".join(y0.values), dur),\n            animate_attribute(\"y2\", \";\".join(y1.values), dur),\n        ]\n    )\n    set_hidden = animate_hidden_attribute(landmark_nan_ranges, n_frames, dur)\n\n    return f'<line stroke=\"{color}\" stroke-width=\"{stroke_width}\">{animation}{set_hidden}</line>'","metadata":{"execution":{"iopub.status.busy":"2023-07-01T23:30:08.003481Z","iopub.execute_input":"2023-07-01T23:30:08.004054Z","iopub.status.idle":"2023-07-01T23:30:24.597742Z","shell.execute_reply.started":"2023-07-01T23:30:08.003996Z","shell.execute_reply":"2023-07-01T23:30:24.596422Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#By Victor Dumetz https://www.kaggle.com/code/victordumetz/gaslfr-svg-visualisation\n\ndef load_landmarks(path, selected_colums=SELECTED_COLUMNS):\n    return pd.read_parquet(get_full_landmark_path(path), columns=selected_colums)\n\ndef load_sequence(sequence_id, data, selected_columns=SELECTED_COLUMNS):\n    path = data.loc[data[\"sequence_id\"] == sequence_id, \"path\"].squeeze()\n    landmarks = load_landmarks(path, selected_columns)\n    return landmarks.loc[sequence_id]\n\ndef visualise_sequence(sequence_id, data, selected_columns=None, fps=10):\n    \"\"\"Generate an animated SVG file for visualising the sequence.\n\n    Args:\n        sequence_id (int): Sequence ID.\n        data (pd.DataFrame): Metadata from where to find the path to the landmarks file.\n        data_dir (str): Path to the data directory.\n        selected_columns (list, optional): Subset of columns to be loaded. Defaults to None.\n        fps (int, optional): FPS used for the animation. Defaults to 10.\n\n    Returns:\n        HTML: SVG animation of the sequence.\n    \"\"\"\n    if selected_columns is not None and \"frame\" not in selected_columns:\n        selected_columns.append(\"frame\")\n\n    phrase = data.loc[data[\"sequence_id\"] == sequence_id, \"phrase\"].squeeze()\n    sequence = load_sequence(sequence_id, data, selected_columns)\n    sequence = sequence.loc[:, ~sequence.columns.str.startswith(\"z_\")]\n\n    sequence.sort_values(\"frame\")\n\n    n_frames = sequence[\"frame\"].max()\n\n    sequence_m = sequence.melt(\"frame\").rename({\"variable\": \"keypoint\"}, axis=1)\n    sequence_m.set_index(\"keypoint\", inplace=True)\n    sequence_m.index = sequence_m.index.str.split(pat=\"_\", n=1, expand=True)\n    sequence_m.reset_index(inplace=True)\n    sequence_m.rename({\"level_0\": \"axis\", \"level_1\": \"keypoint\"}, axis=1, inplace=True)\n    mux = pd.MultiIndex.from_product([sequence_m[\"keypoint\"].unique(), range(n_frames)], names=[\"keypoint\", \"frame\"])\n    sequence_m = sequence_m.pivot(index=[\"frame\", \"keypoint\"], columns=\"axis\").reorder_levels([\"keypoint\", \"frame\"])\n    sequence_m = sequence_m.reindex(mux)\n    sequence_m.sort_index(level=[\"keypoint\", \"frame\"], ascending=[1, 1], inplace=True)\n    sequence_m.columns = [\"x\", \"y\"]\n\n    min_x = np.nanmin(sequence_m[\"x\"])\n    max_x = np.nanmax(sequence_m[\"x\"])\n    min_y = np.nanmin(sequence_m[\"y\"])\n    max_y = np.nanmax(sequence_m[\"y\"])\n    pad = 0.1\n\n    nan_ranges = get_nan_ranges(sequence_m)\n\n    sequence_str = sequence_m.groupby(level=\"keypoint\").ffill().astype(str)\n\n    lines = \"\"\n    for _, (i, j) in enumerate(POSE_CONNECTIONS):\n        try:\n            lines = lines + draw_connection(\n                sequence_str.loc[(\"pose_{}\".format(i), slice(None)), \"x\"],\n                sequence_str.loc[(\"pose_{}\".format(j), slice(None)), \"x\"],\n                sequence_str.loc[(\"pose_{}\".format(i), slice(None)), \"y\"],\n                sequence_str.loc[(\"pose_{}\".format(j), slice(None)), \"y\"],\n                nan_ranges.get(\"pose\", []),\n                n_frames,\n                fps,\n                color=\"black\",\n            )\n        except KeyError:\n            pass\n    for _, (i, j) in enumerate(FACE_CONNECTIONS):\n        try:\n            lines = lines + draw_connection(\n                sequence_str.loc[(\"face_{}\".format(i), slice(None)), \"x\"],\n                sequence_str.loc[(\"face_{}\".format(j), slice(None)), \"x\"],\n                sequence_str.loc[(\"face_{}\".format(i), slice(None)), \"y\"],\n                sequence_str.loc[(\"face_{}\".format(j), slice(None)), \"y\"],\n                nan_ranges.get(\"face\", []),\n                n_frames,\n                fps,\n                color=\"black\",\n            )\n        except KeyError:\n            pass\n    for _, (i, j) in enumerate(HAND_CONNECTIONS):\n        try:\n            lines = (\n                lines\n                + draw_connection(\n                    sequence_str.loc[(\"right_hand_{}\".format(i), slice(None)), \"x\"],\n                    sequence_str.loc[(\"right_hand_{}\".format(j), slice(None)), \"x\"],\n                    sequence_str.loc[(\"right_hand_{}\".format(i), slice(None)), \"y\"],\n                    sequence_str.loc[(\"right_hand_{}\".format(j), slice(None)), \"y\"],\n                    nan_ranges.get(\"right_hand\", []),\n                    n_frames,\n                    fps,\n                    color=\"red\",\n                    stroke_width=0.01,\n                )\n                + draw_connection(\n                    sequence_str.loc[(\"left_hand_{}\".format(i), slice(None)), \"x\"],\n                    sequence_str.loc[(\"left_hand_{}\".format(j), slice(None)), \"x\"],\n                    sequence_str.loc[(\"left_hand_{}\".format(i), slice(None)), \"y\"],\n                    sequence_str.loc[(\"left_hand_{}\".format(j), slice(None)), \"y\"],\n                    nan_ranges.get(\"left_hand\", []),\n                    n_frames,\n                    fps,\n                    color=\"red\",\n                    stroke_width=0.01,\n                )\n            )\n        except KeyError:\n            pass\n\n    html = f'<h2>{sequence_id} - \"{phrase}\"</h2>'\n    html += '<svg height=\"600\" width=\"{}\" viewBox=\"{} {} {} {}\">{}</svg>'.format(\n        600 * (max_x - min_x + 2 * pad) / (max_y - min_y + 2 * pad),\n        min_x - pad,\n        min_y - pad,\n        max_x + pad,\n        max_y + pad,\n        lines,\n    )\n\n    return HTML(html.format(lines))","metadata":{"execution":{"iopub.status.busy":"2023-07-01T23:31:09.32266Z","iopub.execute_input":"2023-07-01T23:31:09.323162Z","iopub.status.idle":"2023-07-01T23:31:09.355818Z","shell.execute_reply.started":"2023-07-01T23:31:09.323126Z","shell.execute_reply":"2023-07-01T23:31:09.354641Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#By Victor Dumetz https://www.kaggle.com/code/victordumetz/gaslfr-svg-visualisation\n\nfor i, row in df.head(3).iterrows():\n    display(visualise_sequence(row.sequence_id, df))","metadata":{"execution":{"iopub.status.busy":"2023-07-01T23:31:30.330684Z","iopub.execute_input":"2023-07-01T23:31:30.33115Z","iopub.status.idle":"2023-07-01T23:31:42.850925Z","shell.execute_reply.started":"2023-07-01T23:31:30.331108Z","shell.execute_reply":"2023-07-01T23:31:42.849513Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#I can only do that\n\n![](https://media.tenor.com/9C-SsTkyc8cAAAAM/middle-finger-homer-simpson.gif)tenor","metadata":{}},{"cell_type":"markdown","source":"#We can't forget Santas Competitions \n\nOn my first Santas (2019), due to the Competition title, I thought it was for beginners.\n\nWhen I saw the \"Big Guys\" joining with their codes and topics. I found out that Santas isn't for amateurs. However, I had already published my \"Competitions are Rock'n'Roll Notebook as a gift for them.","metadata":{}},{"cell_type":"markdown","source":"#Take a look on Griffh's beautiful work on Santas 2022 ","metadata":{}},{"cell_type":"code","source":"#By Griffh https://www.kaggle.com/code/griffh/santa-eda-about-color/notebook\n\nimport matplotlib.pyplot as plt\nimport matplotlib.ticker as ticker\nimport matplotlib.collections as mc\nimport numpy as np\nimport pandas as pd\nfrom functools import *\nfrom itertools import *\nfrom pathlib import Path\nfrom PIL import Image\nimport tqdm\nfrom sklearn.cluster import KMeans\nfrom scipy.sparse import dok_matrix\nfrom scipy.sparse.csgraph import minimum_spanning_tree\nimport copy\nimport warnings\nwarnings.filterwarnings('ignore')\nplt.style.use('seaborn-whitegrid')\ndata_dir = Path('/kaggle/input/santa-2022')","metadata":{"execution":{"iopub.status.busy":"2023-07-01T22:44:55.382168Z","iopub.execute_input":"2023-07-01T22:44:55.383375Z","iopub.status.idle":"2023-07-01T22:44:57.383272Z","shell.execute_reply.started":"2023-07-01T22:44:55.383325Z","shell.execute_reply":"2023-07-01T22:44:57.38201Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#By Griffh https://www.kaggle.com/code/griffh/santa-eda-about-color/notebook\n\ndf_image = pd.read_csv(data_dir / 'image.csv')\ndf_image","metadata":{"execution":{"iopub.status.busy":"2023-07-01T22:45:38.399636Z","iopub.execute_input":"2023-07-01T22:45:38.400117Z","iopub.status.idle":"2023-07-01T22:45:38.574714Z","shell.execute_reply.started":"2023-07-01T22:45:38.400081Z","shell.execute_reply":"2023-07-01T22:45:38.573815Z"},"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#By Griffh https://www.kaggle.com/code/griffh/santa-eda-about-color/notebook\n\ndef df_to_image(df):\n    side = int(len(df) ** 0.5) \n    return df.set_index(['x', 'y']).to_numpy().reshape(side, side, -1)","metadata":{"execution":{"iopub.status.busy":"2023-07-01T22:45:55.604763Z","iopub.execute_input":"2023-07-01T22:45:55.605165Z","iopub.status.idle":"2023-07-01T22:45:55.610985Z","shell.execute_reply.started":"2023-07-01T22:45:55.605136Z","shell.execute_reply":"2023-07-01T22:45:55.609794Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Santas Drawings","metadata":{}},{"cell_type":"code","source":"#By Griffh https://www.kaggle.com/code/griffh/santa-eda-about-color/notebook\n\nimage = df_to_image(df_image)","metadata":{"execution":{"iopub.status.busy":"2023-07-01T22:46:11.202734Z","iopub.execute_input":"2023-07-01T22:46:11.203157Z","iopub.status.idle":"2023-07-01T22:46:11.223777Z","shell.execute_reply.started":"2023-07-01T22:46:11.203129Z","shell.execute_reply":"2023-07-01T22:46:11.222437Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#By Griffh https://www.kaggle.com/code/griffh/santa-eda-about-color/notebook\n\nradius = 128\nfig,ax = plt.subplots(3,3)\nax=ax.flatten()\nfig.set_figheight(15)\nfig.set_figwidth(15)\nfor i in range(9): \n    if i ==0:\n        ax[i].matshow(image, extent=(-radius, radius+1, -radius, radius+1))\n        ax[i].grid(None)\n    if i > 0:\n        kmeans_df = copy.deepcopy(df_image)\n        kmeans_df['plot_x'] = kmeans_df['x']+128\n        kmeans_df['plot_y'] = kmeans_df['y']+128\n        kmeans_df['plot_x'] = kmeans_df['plot_x']/256\n        kmeans_df['plot_y'] = kmeans_df['plot_y']/256\n        km = KMeans(n_clusters=i,random_state = 0)\n        km.fit(kmeans_df[['r','g','b']])\n        kmeans_df['label'] = km.labels_\n        ax[i].scatter(kmeans_df['plot_x'],kmeans_df['plot_y'],marker=\"o\",c=kmeans_df['label'])\n        ax[i].set_title(\"kmeans:{}\".format(i))\n        ax[i].grid(None)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-01T22:46:31.049986Z","iopub.execute_input":"2023-07-01T22:46:31.050493Z","iopub.status.idle":"2023-07-01T22:46:56.879669Z","shell.execute_reply.started":"2023-07-01T22:46:31.050455Z","shell.execute_reply":"2023-07-01T22:46:56.878524Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#By Griffh https://www.kaggle.com/code/griffh/santa-eda-about-color/notebook\n\nkmeans_df = copy.deepcopy(df_image)\nkmeans_df['plot_x'] = kmeans_df['x']+128\nkmeans_df['plot_y'] = kmeans_df['y']+128\nkmeans_df['plot_x'] = kmeans_df['plot_x']/256\nkmeans_df['plot_y'] = kmeans_df['plot_y']/256\nkm = KMeans(n_clusters=2,random_state = 0)\nkm.fit(kmeans_df[['r','g','b']])\nkmeans_df['label'] = km.labels_\nkmeans_2 = df_to_image(kmeans_df[['x','y','label']])\nradius = 128\nfig, ax = plt.subplots(figsize=(10, 10))\nax.matshow(kmeans_2, extent=(-radius, radius+1, -radius, radius+1))\nax.add_patch(plt.Rectangle((-radius, radius+1), 86, -105, color=\"white\", fill=False, linewidth=3))\n# \nax.add_patch(plt.Rectangle((50, radius+1), 78, -105, color=\"white\", fill=False, linewidth=3))\nax.grid(None);","metadata":{"execution":{"iopub.status.busy":"2023-07-01T22:47:06.984808Z","iopub.execute_input":"2023-07-01T22:47:06.985254Z","iopub.status.idle":"2023-07-01T22:47:08.798145Z","shell.execute_reply.started":"2023-07-01T22:47:06.985219Z","shell.execute_reply":"2023-07-01T22:47:08.79655Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#What about C-Number Draw? Isn't it lovely? So Cute.","metadata":{}},{"cell_type":"code","source":"#By C-Number https://www.kaggle.com/code/cnumber/lower-bound-using-minimum-spanning-tree\ndef hash_to_xy(i):\n    return i//N, i%N\ndef xy_to_hash(x,y):\n    return x*N+y\ndef get_cost(i, j):\n    ix, iy = hash_to_xy(i)\n    jx, jy = hash_to_xy(j)\n    d = np.abs(ix-jx)+np.abs(iy-jy)\n    return np.sum(np.abs(image[ix,iy]-image[jx,jy]))*3 + np.sqrt(d)\nN = 257\nNN = N*N\nradius = 128\nmat = dok_matrix((NN, NN), dtype=np.float64)\nmst = minimum_spanning_tree(mat)\nfor i in tqdm.trange(NN):\n    x, y = hash_to_xy(i)\n    for dx in range(-8, 9):\n        for dy in range(-8, 9):\n            nx, ny = x+dx, y+dy\n            if np.abs(dx)+np.abs(dy) > 8:\n                continue\n            if nx<0 or N<=nx or ny<0 or N<=ny:\n                continue\n            j = xy_to_hash(nx, ny)\n            cost = get_cost(i, j)\n            mat[i,j] = cost\nmst = minimum_spanning_tree(mat)\nedges = list(zip(*mst.nonzero()))\nlines = []\nfor i, j in edges:\n    ix, iy = hash_to_xy(i)\n    jx, jy = hash_to_xy(j)\n    d = np.sqrt(np.abs(ix-jx)**2 + np.abs(iy-jy)**2)\n    # if d >= 1.8:\n    lines.append([[iy - radius, radius - ix],[jy - radius, radius - jx]])\nlc = mc.LineCollection(lines, colors='b')\n\nfig = plt.figure(figsize=(20,20))\nax = fig.add_subplot(111)\nax.add_collection(lc)\nax.matshow(image, extent=(-radius-0.5, radius+0.5, -radius-0.5, radius+0.5))\nax.grid(None)\n\nax.autoscale()\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-01T22:47:41.521899Z","iopub.execute_input":"2023-07-01T22:47:41.52297Z","iopub.status.idle":"2023-07-01T22:56:06.753993Z","shell.execute_reply.started":"2023-07-01T22:47:41.522921Z","shell.execute_reply":"2023-07-01T22:56:06.752786Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#By Griffh https://www.kaggle.com/code/griffh/santa-eda-about-color/notebook\n\n# Then find out how many colors in each line are the same (similar)\nbaseline_poi = 125\ncolor_baseline = df_image[df_image['x']==baseline_poi].sort_values('y')[['y','r','g','b']].reset_index(drop = True)\ncolor_baseline.columns = ['y','r_baseline','g_baseline','b_baseline']\n\nsplit_df = copy.deepcopy(df_image)\nsplit_df = pd.merge(split_df,color_baseline,on = 'y')\n\ndef get_split(per):\n    rgb_list = ['label_{}_per{}'.format(element,per)for element in ['r','g','b']]\n    for color_now in ['r','g','b']:\n        label_name = 'label_{}_per{}'.format(color_now,per)\n        baseline_name = '{}_baseline'.format(color_now)\n        left_bound = split_df[color_now] >= split_df[baseline_name]*(1-per/100)\n        right_bound = split_df[color_now] <= split_df[baseline_name]*(1+per/100)\n        split_df[label_name] = left_bound&right_bound\n        split_df[label_name] = split_df[label_name].apply(int)\n    rgb_name = 'label_rgb_per{}'.format(per)\n    split_df[rgb_name] = split_df[rgb_list].sum(axis = 1)\n    split_df[rgb_name] = split_df[rgb_name].apply(lambda element:1 if element == 3 else 0)\n    return(split_df)\nfor i in range(20):\n    split_df = get_split(i)","metadata":{"execution":{"iopub.status.busy":"2023-07-01T22:58:15.338623Z","iopub.execute_input":"2023-07-01T22:58:15.339127Z","iopub.status.idle":"2023-07-01T22:58:19.39595Z","shell.execute_reply.started":"2023-07-01T22:58:15.339092Z","shell.execute_reply":"2023-07-01T22:58:19.394476Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#By Griffh https://www.kaggle.com/code/griffh/santa-eda-about-color/notebook\n\nradius = 128\nfig,ax = plt.subplots(2,4)\nax=ax.flatten()\nfig.set_figheight(10)\nfig.set_figwidth(20)\ncols = ['label_r_per0','label_g_per0','label_b_per0','label_rgb_per0',\n        'label_r_per1','label_g_per1','label_b_per1','label_rgb_per1']\nfor i in range(len(cols)): \n    ax[i].set_title(cols[i])\n    ax[i].matshow(df_to_image(split_df[['x','y',cols[i]]]), extent=(-radius, radius+1, -radius, radius+1))\n    ax[i].grid(None)","metadata":{"execution":{"iopub.status.busy":"2023-07-01T22:58:43.178887Z","iopub.execute_input":"2023-07-01T22:58:43.179309Z","iopub.status.idle":"2023-07-01T22:58:46.94483Z","shell.execute_reply.started":"2023-07-01T22:58:43.179279Z","shell.execute_reply":"2023-07-01T22:58:46.943697Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Try to join the next Santas Competition.\n\nAnd deliver something like those guys did. ","metadata":{}},{"cell_type":"code","source":"#By Griffh https://www.kaggle.com/code/griffh/santa-eda-about-color/notebook\n\nfig,ax = plt.subplots(1,2)\nax=ax.flatten()\nfig.set_figheight(8)\nfig.set_figwidth(20)\nax[0].matshow(df_to_image(split_df[['x','y','label_rgb_per4']]), extent=(-radius, radius+1, -radius, radius+1))\nax[0].grid(None)\nax[0].set_title(\"background mask\")\nax[1].matshow(image, extent=(-radius, radius+1, -radius, radius+1))\nax[1].grid(None)\nax[1].set_title(\"original image\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-01T22:58:53.342003Z","iopub.execute_input":"2023-07-01T22:58:53.342466Z","iopub.status.idle":"2023-07-01T22:58:54.243259Z","shell.execute_reply.started":"2023-07-01T22:58:53.34243Z","shell.execute_reply":"2023-07-01T22:58:54.242074Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Be inspired by those professionals and join that Kaggle Competitions.","metadata":{}},{"cell_type":"markdown","source":"#Acknowledgements:\n\nVictor Dumetz https://www.kaggle.com/code/victordumetz/gaslfr-svg-visualisation\n\nGriffh https://www.kaggle.com/code/griffh/santa-eda-about-color/notebook\n\nC-Number https://www.kaggle.com/code/cnumber/lower-bound-using-minimum-spanning-tree","metadata":{}},{"cell_type":"markdown","source":"![](https://i.pinimg.com/474x/b7/5c/73/b75c73e335cc1e79deb683191451c6c1.jpg)https://br.pinterest.com/pin/me--140806223847194/","metadata":{}}]}