<?xml version="1.0"?>
<rss version="2.0">
   <channel>
      <title>Introduction To Data Science YEEETT by Shahrin Amin</title>
      <link>https://padlet.com/shahrinamin_my/2tulveioj9u2</link>
      <description>I only write notes related to past year exam :) - rin https://docs.google.com/document/d/1Ydk7dim7l45S3P_qBGPpsUE5x7QhWMVsX5hGepVnTl4/edit#</description>
      <language>en-us</language>
      <pubDate>2019-03-07 08:24:20 UTC</pubDate>
      <lastBuildDate>2025-08-15 14:00:56 UTC</lastBuildDate>
      <webMaster>hello@padlet.com</webMaster>
      <image>
         <url>https://padlet-assets.s3.amazonaws.com/icons/Clouds.png</url>
      </image>
      <item>
         <title>Data Structures</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/338750892</link>
         <description><![CDATA[<ul><li><strong>Structured data:</strong><ul><li>Contains <mark>a defined data type, format and structure</mark>.</li><li>E.g: traditional RDBMS, CSV files</li></ul></li><li><strong>Semi-structured data:</strong><ul><li>Textual data files with a <mark>discernible pattern that enables parsing</mark>.</li><li>E.g: XML, JSON</li></ul></li><li><strong>Quasi-structured data:</strong><ul><li>Textual data with <mark>unpredictable data formats</mark> that can be formatted with effort, tools, and time.</li><li>E.g: Web Clickstream Data</li></ul></li><li><strong>Unstructured data:</strong><ul><li>Data that has<mark> no inherent structure.</mark></li><li>E.g: Text, PDF, Images and Video</li></ul></li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-07 08:29:43 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/338750892</guid>
      </item>
      <item>
         <title>Key Enablers for the growth of Big Data</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/338753934</link>
         <description><![CDATA[<ul><li>Increase of <mark>storage capacities</mark></li><li>Increase of <mark>processing power</mark></li><li><mark>Availability of data</mark></li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-07 08:42:10 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/338753934</guid>
      </item>
      <item>
         <title>Characterization of Big Data</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/338755046</link>
         <description><![CDATA[<ul><li><strong>Volume</strong><ul><li>the vast amount of data generated each second (<mark>scale of data</mark>)</li></ul></li><li><strong>Velocity</strong><ul><li>the <mark>speed</mark> at which data is generated</li></ul></li><li><strong>Variety</strong><ul><li>the<mark> different types of data</mark> available (diversity)</li></ul></li><li><strong>Veracity</strong><ul><li>the <mark>trustworthiness</mark> of data</li></ul></li><li><strong>Value</strong><ul><li>the <mark>meaningfulness</mark> of data, creation of <mark>actionable insights</mark>, data monetization etc.</li></ul></li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-07 08:46:41 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/338755046</guid>
      </item>
      <item>
         <title>Insights from Big Data</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/338757995</link>
         <description><![CDATA[<ul><li><strong>Facebook</strong><ul><li><mark>analyses location</mark> information for <mark>easier to find friends</mark> to connect.</li><li>identitfy global <mark>migration patterns</mark></li><li>determine where the different football team fan bases live</li></ul></li><li><strong>Target</strong><ul><li><mark>predict which customers are pregnant</mark>, to focus baby-related marketing to them.</li></ul></li><li><strong>Tesco PLC</strong><ul><li>collected refrigerator-related data points to <mark>monitor it's performance, towards proactive maintenance</mark> to cut down on energy costs.</li></ul></li><li><strong>Macy's Inc</strong><ul><li><mark>adjust pricing</mark> of it's items in near-real time by <mark>monitoring demand and inventory</mark>.</li></ul></li><li><strong>Siemens</strong><ul><li>leveraged on <mark>sensor-data analytics</mark> and <mark>predictive maintenance</mark> to reduce train failures.</li></ul></li><li><strong>Google Flu Trends</strong><ul><li>aggregates <mark>google search queries</mark> to <mark>predict outbreaks of flu</mark>.</li></ul></li><li><strong>Electoral Campaign (Obama)</strong><ul><li>data was mined to <mark>determine voters that need convincing</mark>, to <mark>choose donor-specific fundraising programs</mark>.</li></ul></li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-07 08:57:25 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/338757995</guid>
      </item>
      <item>
         <title>Business Intelligence (BI) vs. Data Science (DS)</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/338762059</link>
         <description><![CDATA[<div><strong>Typical Technique</strong></div><ul><li><strong>BI - </strong>Standard &amp; <mark>ad hoc reporting</mark>, dashboards, alerts, queries, <mark>details on demand</mark>.</li><li><strong>DS</strong> - Optimization, <mark>predictive modeling</mark>, <mark>forecasting</mark>, statistical analysis.</li></ul><div><strong>Data Types</strong></div><ul><li><strong>BI - </strong>Structured data, traditional sources, <mark>manageable datasets</mark>.</li><li><strong>DS</strong> - Structured/unstructed data, many types of sources, <mark>very large datasets</mark>.</li></ul><div><strong>Common Questions</strong></div><ul><li><strong>BI - </strong>What happened <mark>last quarter</mark>? Where is the problem?</li><li><strong>DS</strong> - What if? What will <mark>happend next</mark>?</li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-07 09:13:29 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/338762059</guid>
      </item>
      <item>
         <title>Data Science Pipelines</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/338767044</link>
         <description><![CDATA[<ul><li><strong>Data Collection</strong><ul><li>Data is <mark>obtained from a variety of sources</mark>.</li></ul></li><li><strong>Data Preprocessing</strong><ul><li>Detect and <mark>remove errors or amend inconsistencies</mark> in data.</li></ul></li><li><strong>Data Analysis</strong><ul><li><mark>Exploration of the main characteristics</mark> of data.</li></ul></li><li><strong>Data Mining</strong><ul><li><mark>Discovery of patterns/relationship</mark> within the data.</li></ul></li><li><strong>Data Visualization</strong><ul><li><mark>Presentation</mark> of data in <mark>pictorial or graphical</mark> format</li></ul></li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-07 09:30:40 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/338767044</guid>
      </item>
      <item>
         <title>Data Science Process</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/338768224</link>
         <description><![CDATA[<ul><li><strong><mark>Ask an interesting question</mark></strong></li><li><strong><mark>Get the data</mark></strong></li><li><strong><mark>Explore the data</mark></strong></li><li><strong><mark>Model the data</mark></strong></li><li><strong><mark>Communicate and visualize the results.</mark></strong></li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-07 09:34:19 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/338768224</guid>
      </item>
      <item>
         <title>Responsibilities of a Data Scientist</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/338768412</link>
         <description><![CDATA[<ul><li><mark>Extract huge volumes of data</mark> from multiple internal and external sources.</li><li>Thoroughly <mark>clean &amp; prune data</mark> to discard irrelevant information.</li><li><mark>Invent new algorithm</mark> to solve problems.</li><li><mark>Explore &amp; examine data</mark> from a variety of angles to determine hidden weaknesses, trends or opportunities.</li><li>Prepare data for use in <mark>predictive and prescriptive modeling</mark>.</li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-07 09:34:54 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/338768412</guid>
      </item>
      <item>
         <title>Challenges in Data Science</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/338769512</link>
         <description><![CDATA[<ul><li>Validity of Assumptions</li><li><mark>Making ad-hoc explanations</mark> of data patterns</li><li><mark>Over-generalizing</mark></li><li><mark>Communication</mark></li><li><mark>Validation of models</mark>, data pipeline integrity</li><li><mark>Using statistical tests correctly</mark></li><li>Prototype -&gt; Production transitions</li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-07 09:39:04 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/338769512</guid>
      </item>
      <item>
         <title>Basic Types of Questions</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/338771259</link>
         <description><![CDATA[<ol><li><strong>Descriptive</strong><ul><li>To <mark>summarize a characteristic of a set of data</mark>.</li><li>E.g. What is the <mark>mean</mark> number of student procrastinating during study week?</li></ul></li><li><strong>Exploratory</strong><ul><li>Analyze the data to see if there are <mark>patterns, trends, or relationships</mark> between variables.</li><li>E.g. What is the <mark>relationship between</mark> student procrastination and academic performance?</li></ul></li><li><strong>Inferential</strong><ul><li>Restatement of the proposed hypothesis as a question and would be <mark>answered by analyzing a different set of data</mark>.</li><li>E.g. Propose the hypothesis that among students, sleeping more than 5 hours a day is associated with better academic performance, based on <mark>sample population of UiTM students</mark>. Re-examine on a <mark>different dataset such as MMU students</mark>.</li></ul></li><li><strong>Predictive</strong><ul><li>Ask what are the set of <mark>predictors/factors</mark> for a particular behavior.</li><li>E.g. Which group of student are high likely to perform better academically <mark>next semester</mark>?</li></ul></li><li><strong>Causal</strong><ul><li>Ask about whether <mark>changing one factor will change another factor</mark>.</li><li>E.g. Will an increase in PUBG Lite play time increase the tendency to achieve Chicken Dinner?</li></ul></li><li><strong>Mechanistic</strong><ul><li>How <mark>a factor affects the outcome</mark>.</li><li><mark>How</mark> Marvel Studio decrease the DC fan base entirely? </li></ul></li></ol>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-07 09:44:01 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/338771259</guid>
      </item>
      <item>
         <title>Characteristics of Good Question</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/338777347</link>
         <description><![CDATA[<ol><li>The question should of <mark>interest</mark> to your audience.</li><li>The question has <mark>not already been answered</mark>.</li><li>The question should also stem from a <mark>plausible (valid correlations)</mark> framework.</li><li>The question, should also be <mark>answerable</mark>.</li><li>The question should be <mark>specific</mark>.</li></ol>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-07 10:07:02 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/338777347</guid>
      </item>
      <item>
         <title>Dimensions of Data Quality</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/338779647</link>
         <description><![CDATA[<ul><li><strong>Completeness</strong><ul><li><mark>All necessary data have been recorded</mark>. Data can be complete even if optional data is missing.</li></ul></li><li><strong>Timeliness</strong><ul><li>The <mark>data is kept up to date</mark>. It is about having the <mark>right information at the right time</mark>.</li></ul></li><li><strong>Consistency</strong><ul><li>The data across processes, organizations, <mark>sources are in sync with each other</mark>.</li></ul></li><li><strong>Validity</strong><ul><li>The <mark>same fields are used consistently</mark> <mark>for the same information</mark> capture.</li></ul></li><li><strong>Accuracy</strong><ul><li>The<mark> data was recorded correctly</mark>.</li></ul></li><li><strong>Uniqueness</strong><ul><li><mark>Each record is distinct and unique</mark>.</li></ul></li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-07 10:14:05 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/338779647</guid>
      </item>
      <item>
         <title>4 Types of Analytics</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339206403</link>
         <description><![CDATA[<ul><li><strong>Descriptive</strong><ul><li><mark>What is happening?</mark></li><li><mark>Comprehensive</mark>, accurate  and live data.                                                                               </li></ul></li><li><strong>Diagnostic</strong><ul><li><mark>Why did it happen?</mark></li><li>Ability to <mark>drill down</mark> to the root-cause.</li></ul></li><li><strong>Predictive</strong><ul><li><mark>What is likely to happen?</mark></li><li><mark>Historical patterns</mark> being used <mark>to predict</mark> specific outcomes using algorithms.</li></ul></li><li><strong>Prescriptive</strong><ul><li><mark>What should i do about it?</mark></li><li><mark>Recommended actions</mark> &amp; strategies <mark>based on challenger testing</mark> outcomes.</li></ul></li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-08 08:19:36 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339206403</guid>
      </item>
      <item>
         <title>No Final Question</title>
         <author></author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339206681</link>
         <description><![CDATA[]]></description>
         <enclosure url="" />
         <pubDate>2019-03-08 08:20:58 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339206681</guid>
      </item>
      <item>
         <title>Data Types</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339211708</link>
         <description><![CDATA[<ul><li><strong>Numerical Data</strong><ul><li><strong>Discrete Data</strong><ul><li>Values are distinct and separated. </li><li>E.g. <mark>Number of staff (15,16)</mark></li></ul></li><li><strong>Continuous Data</strong><ul><li>Values may take on any value between finite and infinite interval.</li><li>E.g. <mark>Height and weight (85.5, 89.9)</mark></li></ul></li></ul></li><li><strong>Categorical Data</strong><ul><li><strong>Nominal Data</strong><ul><li>Values can be assigned a code in the form of a number.</li><li>E.g. <mark>Gender (Male = 0, Female = 1)</mark></li></ul></li><li><strong>Ordinal Data</strong><ul><li>Value can be ranked or having a rating scale attached.</li><li>E.g. <mark>Very unsatisfied = 0, Unsatisfied = 1, Neutral = 3, Satisfied = 4, Very satisfied = 5</mark></li></ul></li></ul></li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-08 08:45:20 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339211708</guid>
      </item>
      <item>
         <title>Numerical Data</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339242116</link>
         <description><![CDATA[]]></description>
         <enclosure url="" />
         <pubDate>2019-03-08 10:59:10 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339242116</guid>
      </item>
      <item>
         <title>Describing Data</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339242429</link>
         <description><![CDATA[<ul><li><strong>Tabular</strong><ul><li><mark>Frequency Distributions</mark></li><li>Relative Frequency Distributions</li></ul></li><li><strong>Graphical</strong><ul><li><mark>Bar Chart or Histogram</mark></li><li><mark>Stem and Leaf Plot</mark></li><li><mark>Scatter Plot</mark></li></ul></li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-08 11:00:25 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339242429</guid>
      </item>
      <item>
         <title>Statistical Description</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339243318</link>
         <description><![CDATA[<ul><li><strong>Measures of Central Tendency</strong><ul><li><mark>Mean</mark></li><li><mark>Median</mark></li><li><mark>Mode</mark></li></ul></li><li><strong>Measures of Dispersion of Data</strong><ul><li><mark>Range</mark></li><li>Quantiles</li><li><mark>Quartiles</mark></li><li><mark>Interquartile Range</mark></li><li>Percentiles</li><li>Boxplots</li><li><mark>Variance</mark></li><li><mark>Standard Deviation</mark></li></ul></li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-08 11:05:06 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339243318</guid>
      </item>
      <item>
         <title>Five-number Summary, Boxplot</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339244883</link>
         <description><![CDATA[<ul><li><strong>Five-number summary</strong><ul><li>Minimum</li><li>Quartile Q1</li><li>Median</li><li>Quartile Q3</li><li>Maximum</li></ul></li><li><strong>Boxplot</strong></li></ul><div><br></div>]]></description>
         <enclosure url="https://padlet-uploads.storage.googleapis.com/243354521/f3bbefcc8477b2113c5084849d54f86e/1.png" />
         <pubDate>2019-03-08 11:12:31 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339244883</guid>
      </item>
      <item>
         <title>Outliers</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339247937</link>
         <description><![CDATA[<div>Lower <mark>inner</mark> fence = Q1 - 1.5*IQR<br>Upper <mark>inner</mark> fence = Q3 + 1.5*IQR<br>Lower <mark>outer</mark> fence = Q1 - 3*IQR<br>Upper <mark>outer</mark> fence = Q3 + 3*IQR<br><br></div><ul><li><strong>Mild outlier</strong><ul><li>Beyond an <mark>inner</mark> fence on either side</li></ul></li><li><strong>Extreme outlier</strong><ul><li>Beyond an <mark>outer</mark> fence on either side</li></ul></li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-08 11:25:35 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339247937</guid>
      </item>
      <item>
         <title>Variance and Standard Deviation</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339248772</link>
         <description><![CDATA[<ul><li><strong><mark>Low</mark></strong><strong> standard deviation</strong><ul><li>data tend to be <mark>very close to the mean</mark></li></ul></li><li><strong><mark>High</mark></strong><strong> standard deviation</strong><ul><li>data are <mark>spread out</mark> over a large range of values</li></ul></li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-08 11:29:11 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339248772</guid>
      </item>
      <item>
         <title>Skewness</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339260672</link>
         <description><![CDATA[<div><strong>Skewness</strong> is the <mark>measure of the asymmetry</mark> of the probability of a real-valued random variable about its <mark>mean</mark>.</div>]]></description>
         <enclosure url="https://padlet-uploads.storage.googleapis.com/243354521/567e78b952a6395819993f27b73b006a/3.jpg" />
         <pubDate>2019-03-08 12:27:23 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339260672</guid>
      </item>
      <item>
         <title>Kurtosis</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339265054</link>
         <description><![CDATA[<div><strong>Kurtosis</strong> is a measure of the <mark>"tailedness"</mark> of the probability distribution of a real-valued random variable.<br><br></div><ul><li><strong><mark>High</mark></strong><strong> (Positive)Kurtosis</strong><ul><li>"Heavy" tails, or <mark>potential outliers</mark>.</li></ul></li><li><strong><mark>Low</mark></strong><strong> (Negative) Kurtosis</strong><ul><li>"Light" tails, or <mark>less likely having outliers</mark>.</li></ul></li></ul><div><br></div>]]></description>
         <enclosure url="https://padlet-uploads.storage.googleapis.com/243354521/aeea9c2b7df444e69e1fca5e5016f94e/4.jpg" />
         <pubDate>2019-03-08 12:41:48 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339265054</guid>
      </item>
      <item>
         <title>Correlation Analysis</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339268600</link>
         <description><![CDATA[<ul><li><strong>Positive Correlation</strong><ul><li>one variable <mark>increases with the other</mark>.</li></ul></li><li><strong>Negative Correlation</strong><ul><li>one variable <mark>decreases</mark> when the <mark>other increases</mark>.</li></ul></li><li><strong>Correlation close to 0</strong><ul><li>both variables have <mark>little influence</mark> over the other.</li></ul></li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-08 12:55:11 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339268600</guid>
      </item>
      <item>
         <title>AI vs DM vs ML</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339370741</link>
         <description><![CDATA[<ul><li><strong>Artificial Intelligence</strong><ul><li>the theory/development of <mark>computer systems</mark> able to <mark>perform task</mark> that normally <mark>require human intelligence</mark></li><li>E.g. Visual perception, speech recognition, decision-making, translation between languages.</li></ul></li><li><strong>Data Mining</strong><ul><li>the practice of <mark>examining large databases</mark> to <mark>discover patterns</mark> or new <mark>information</mark>.</li></ul></li><li><strong>Machine Learning</strong><ul><li>the <mark>training of model</mark> from data that <mark>generalizes a decision(prediction)</mark> against a performance measure.</li></ul></li></ul>]]></description>
         <enclosure url="https://padlet-uploads.storage.googleapis.com/243354521/2ed9a5085c1ff0ccf5b534d2fdde9260/5.png" />
         <pubDate>2019-03-08 16:28:16 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339370741</guid>
      </item>
      <item>
         <title>Machine Learning</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339376756</link>
         <description><![CDATA[<ul><li><strong>Unsupervised Learning </strong><mark>discover</mark> patterns/relationship from the data <mark>itself</mark><ul><li><strong>Clustering</strong><ul><li>discover the <mark>inherent groupings</mark> in the data.</li><li>E.g. K-means clustering</li></ul></li><li><strong>Association</strong><ul><li>discover <mark>rules</mark> that describe large portions of your data.</li><li>E.g. <a href="https://www.hackerearth.com/blog/machine-learning/beginners-tutorial-apriori-algorithm-data-mining-r-implementation/"><mark>Apriori algorithm</mark></a> for association rule mining</li></ul></li></ul></li><li><strong>Supervised Learning      </strong>based on "<mark>training examples</mark>". Has the ability to reach an accurate conclusion when given new data.<ul><li><strong>Classification</strong><ul><li>predicts an output variable as a <mark>category</mark>.</li><li>E.g. <mark>Decision Trees</mark>, Support Vector Machine.</li></ul></li><li><strong>Regression</strong><ul><li>predicts an output variable as a <mark>real value</mark>.</li><li>E.g. <mark>Linear Regression</mark></li></ul></li></ul></li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-08 16:40:38 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339376756</guid>
      </item>
      <item>
         <title>Reinforcement Learning </title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339383446</link>
         <description><![CDATA[<div>The machine trains itself continually using <mark>trial and error</mark>. It learns from past experience and <mark>capture the best possible knowledge</mark> to make <mark>accurate business decisions</mark>.<br><br>E.g. <a href="https://www.geeksforgeeks.org/markov-decision-process/">Markov Decision Process</a></div>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-08 16:55:04 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339383446</guid>
      </item>
      <item>
         <title>Euclidean Distance</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339387890</link>
         <description><![CDATA[<div>calculates the <mark>distance between p and q</mark>, where each observation has n variables.<br><br></div>]]></description>
         <enclosure url="https://padlet-uploads.storage.googleapis.com/243354521/12df8be5f9463078de0afe5eea333e4c/6.png" />
         <pubDate>2019-03-08 17:04:57 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339387890</guid>
      </item>
      <item>
         <title>K-means Clustering</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339389855</link>
         <description><![CDATA[<ol><li><mark>Randomly pick k number</mark> of points for each cluster (<mark>centroids</mark>)</li><li>Each data point is <mark>assigned to the cluster with the closest centroids</mark>.</li><li><mark>Compute the centroid</mark> of each cluster based on existing cluster members, this is result in <mark>new centroids</mark>.</li><li>As we have new centroids, <mark>repeat step 2 and 3</mark>. Repeat this process <mark>until convergence occurs</mark>. I.e. centroids does not change.</li></ol><div><br><strong><mark>Ez explanation</mark></strong></div><ol><li>Randomly pick k number of centroids for each cluster.</li><li>Each data is group into nearest centroids.</li><li>Move the centroid to the mean of the group.</li><li>Repeat 2 and 3 until centroid does'nt change</li></ol><div><br><strong>Limitations</strong></div><ul><li><strong>Predefined number of clusters</strong><ul><li>self explanatory </li></ul></li><li><strong>Distorted by outliers</strong><ul><li>can result in <mark>non-optimal grouping</mark>. (outlier influence the mean value)</li></ul></li><li><strong>No hierarchical organization</strong></li></ul><div><br><strong>Applications</strong></div><ul><li><strong><mark>Marketing</mark></strong><ul><li>discover distinct groups in their customer bases.</li></ul></li><li><strong><mark>Insurance</mark></strong><ul><li>identify groups of insurance policy holders with a high average claim cost.</li></ul></li><li><strong><mark>City-planning</mark></strong><ul><li>identify groups of houses according to their house type, value, and geographical location.</li></ul></li><li><strong><mark>DNA sequences</mark></strong><ul><li>cluster DNA information based on edit distance</li></ul></li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-08 17:08:33 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339389855</guid>
      </item>
      <item>
         <title>Association Rule Mining</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339395845</link>
         <description><![CDATA[<ul><li><strong>Support</strong><ul><li>How popular an itemset is <mark>per all</mark> the transactions.</li><li>E.g. Cereal and milk <mark>appear together in 40%</mark> of the transactions i.e <mark>Support = 40%</mark></li></ul></li><li><a href="https://en.wikipedia.org/wiki/Association_rule_learning#Confidence"><strong>Confidence</strong></a><ul><li>How likely item Y is purchased when item X is purchased (<mark>X-&gt;Y</mark>)</li><li>E.g. Cereal (X) appear in 50 transactions; <mark>40 of the 50</mark> also include milk (Y). Cereal(X) implies milk(Y) with <mark>80% confidence</mark>.</li></ul></li><li><strong>Lift</strong><ul><li>Like confidence but while <mark>controlling how popular item Y is</mark>.</li><li><strong>Lift &gt; 1</strong>: Item Y <mark>is likely</mark> to be bought if item X is bought.</li><li><strong>Lift &lt; 1</strong>: Item Y <mark>is unlikely</mark> to be bought if item X is bought.</li><li><strong>Lift = 1</strong>: Those two item are <mark>independent of each other</mark>. <a href="https://en.wikipedia.org/wiki/Association_rule_learning#Lift">referto</a></li><li>E.g. Orange milk have <mark>75% confidence</mark>, <mark>30% support</mark>. Customers generally buy <mark>milk 90%</mark> of the time and <mark>orange 40%</mark> of the time.</li></ul></li></ul><var>Lift = 0.3/(0.4*0.9) = 0.83</var>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-08 17:21:57 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339395845</guid>
      </item>
      <item>
         <title>Phases: Building &amp; Applying</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339562053</link>
         <description><![CDATA[<ul><li><strong>Building</strong><ul><li>Data used contains values for the descriptor(X) and response(Y) variables.</li><li><strong>Training set</strong><ul><li>used for <mark>building</mark> the model.</li></ul></li><li><strong>Test set</strong><ul><li><mark>accessing the quality</mark> of the model</li></ul></li></ul></li><li><strong>Applying</strong><ul><li>Once model has been built, a <mark>dataset with no output</mark> response variables can be fed into the model <mark>to obtain estimated response</mark>.</li></ul></li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-09 14:26:05 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339562053</guid>
      </item>
      <item>
         <title>Definition</title>
         <author>muzafi17</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339562325</link>
         <description><![CDATA[<ul><li><strong>Data Wrangling</strong> <strong>=</strong> Transformation of <mark>raw data</mark> to <mark>another form</mark> that is useful generating actionable <mark>insights</mark></li><li><strong>Data Cleaning =                </strong>To determine <mark>inaccurate, incomplete or unreasonable</mark> data and then the detected errors are corrected by <mark>replacing, modifying or deleting</mark> the dirty data</li><li><strong>Data Integration =</strong>     Integrate data from <mark>multiple views</mark></li></ul><ol><li> Physical - Copy data to warehouse</li><li>Virtual - Keep the data only at the sources</li></ol><ul><li><strong>Data Reduction =         </strong><mark>Obtain a reduced data set Dimensionality R</mark> - reduce number of random variables or attributes, <mark>Numerosity R</mark>- replace original data volume by alternative, <mark>Data compression</mark> - reduce size but preserve its representation</li><li><strong>Data Transformation = </strong><mark>Normalization</mark> : </li></ul><ol><li>Min-max - Change to a new min and max</li><li>Z-score - Shift of values according to data distribution</li><li>Decimal - Move to decimal places to keep max value &lt; 1 </li></ol><ul><li><mark>Discretization</mark> : Divide the range of continuous attribute into intervals because some algorithm accept categorical</li><li>Binning - <mark>partition into equal size, smoothing by bin means, smoothing by bin boundaries</mark></li><li>Entropy-based - find the best split to divide into 2 sets</li></ul><div><mark>Info (s1,s2) = |s1|/|s| x E(s1) + |s2| /  |s| x E(s2)<br>Gain(v,s2) = E(s) - info (s1,s2)<br></mark>H bawah ni E </div>]]></description>
         <enclosure url="https://computersciencesource.files.wordpress.com/2010/01/entropycalc1.png" />
         <pubDate>2019-03-09 14:28:38 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339562325</guid>
      </item>
      <item>
         <title>Tables</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339563258</link>
         <description><![CDATA[<ul><li>You want to <mark>show individual data values</mark>, and allow audience to <mark>compare</mark> between individual values.</li><li>Precise values are required.</li><li>The quantitative information to be communicated <mark>involves more than one unit measure</mark>.</li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-09 14:38:29 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339563258</guid>
      </item>
      <item>
         <title>Graph</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339563686</link>
         <description><![CDATA[<ul><li>You want to <mark>reveal relationships</mark> between multiple values.</li><li>General <mark>patterns are the key</mark> point you want to present, rather than the exact data values.</li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-09 14:43:34 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339563686</guid>
      </item>
      <item>
         <title>Seven Common Relationship in Graphs</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339563960</link>
         <description><![CDATA[]]></description>
         <enclosure url="" />
         <pubDate>2019-03-09 14:46:34 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339563960</guid>
      </item>
      <item>
         <title>Types Of Transformation</title>
         <author>muzafi17</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339564515</link>
         <description><![CDATA[<ul><li><strong>Filtering</strong> = Based on some conditions</li><li><strong>Transforming</strong> = Add new variables</li><li><strong>Aggregating</strong> = Collapse many to one</li><li><strong>Sorting</strong> = Change the order of values</li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-09 14:52:11 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339564515</guid>
      </item>
      <item>
         <title>Time-Series
</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339564768</link>
         <description><![CDATA[<ul><li><strong>Description</strong><ul><li>Multiple instances of one or more <mark>measures taken at equidistant points in time.</mark></li></ul></li><li><strong>Methods</strong><ul><li><strong>Lines</strong> to emphasize <mark>overall pattern</mark></li><li><strong>Bars</strong> to emphasize <mark>individual values</mark></li><li><strong>Points</strong> connected by lines to slightly <mark>emphasize individual values</mark> while still highlighting the <mark>overall pattern</mark>.</li><li>Always place <mark>time on the horizontal axis</mark></li></ul></li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-09 14:54:46 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339564768</guid>
      </item>
      <item>
         <title>Nominal Comparison
</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339565063</link>
         <description><![CDATA[<ul><li><strong>Description</strong><ul><li>A simple comparison of the <mark>categorical subdivisions</mark> of one or more measure in <mark>no particular order</mark>.</li></ul></li><li><strong>Methods</strong><ul><li><strong><mark>Bars</mark></strong><strong> </strong>only (horizontal or vertical)</li></ul></li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-09 14:57:24 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339565063</guid>
      </item>
      <item>
         <title>Ranking</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339565495</link>
         <description><![CDATA[<ul><li><strong>Description</strong><ul><li>Categorical subdivisions of a measure <mark>ordered by size</mark> (either descending or ascending)</li></ul></li><li><strong>Methods</strong><ul><li><strong><mark>Bars</mark></strong> only (horizontal or vertical)</li><li>To <strong>highlight high values</strong>, sort in <mark>descending order</mark>.</li><li>To <strong>highlight low values</strong>, sort in <mark>ascending order.</mark></li></ul></li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-09 15:01:46 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339565495</guid>
      </item>
      <item>
         <title>Part-to-Whole
</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339565855</link>
         <description><![CDATA[<ul><li><strong>Description</strong><ul><li>Measures of individual categorical subdivisions as <mark>ratios to the whole</mark>.</li></ul></li><li><strong>Methods</strong><ul><li><strong><mark>Bars</mark></strong><strong> </strong>only (horizontal or vertical)</li><li>Use <strong><mark>stacked bars</mark></strong> only when you must display measures of the whole as well as the parts.</li></ul></li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-09 15:05:30 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339565855</guid>
      </item>
      <item>
         <title>Needs Of Integration = Heterogeneity among data sources</title>
         <author>muzafi17</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339566378</link>
         <description><![CDATA[<ol><li><strong>Source type H</strong> <mark>Different store systems</mark></li><li><strong>Communication H</strong> <mark>some have web interfaces some offer API</mark></li><li><strong>Schema H</strong> <mark>Different table structures</mark></li><li><strong>Data type H</strong> <mark>Different data type stored</mark></li><li><strong>Value H</strong> <mark>Different ways of same values</mark></li><li><strong>Semantic H</strong> <mark>Different sources different things</mark></li></ol>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-09 15:10:46 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339566378</guid>
      </item>
      <item>
         <title>Deviation</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339566580</link>
         <description><![CDATA[<ul><li><strong>Description</strong><ul><li>Categorical subdivisions of a measure compared to a reference measure, <mark>expressed as the differences between them</mark>.</li></ul></li><li><strong>Methods</strong><ul><li><strong>Lines</strong> to emphasize the <mark>overall pattern</mark> only when displaying deviation and time-series relationships together.</li><li><strong>Points connected by lines</strong> to slightly emphasize <mark>individual data points</mark> while also <mark>highlighting the overall pattern</mark> when displaying deviation and time-series relationships together.</li><li><strong>Bars</strong> to emphasize <mark>individual values</mark>, but limit to vertical bars when a time-series relationship is included.</li></ul></li></ul><div><br></div>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-09 15:12:58 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339566580</guid>
      </item>
      <item>
         <title>Basic Heuristic Methods</title>
         <author>muzafi17</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339567154</link>
         <description><![CDATA[<ol><li><mark>Stepwise forward</mark> selection - Start empty, best attribute picked</li><li><mark>Stepwise backward</mark> selection - Start full, eliminate the worst</li><li><mark>Combination</mark> - Select best and remove worst</li><li><mark>Decision Tree</mark> Induction</li></ol>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-09 15:18:48 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339567154</guid>
      </item>
      <item>
         <title>Frequency Distribution</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339571064</link>
         <description><![CDATA[<ul><li><strong>Description</strong><ul><li>Counts of something per categorical subdivisions (intervals) of <mark>quantitative range</mark>.</li></ul></li><li><strong>Methods</strong><ul><li><strong>Vertical bars</strong> to emphasize <mark>individual values</mark> (histogram)</li><li><strong>Lines</strong> to emphasize the <mark>overall pattern</mark> (frequency polygon)</li></ul></li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-09 15:51:56 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339571064</guid>
      </item>
      <item>
         <title>Softwares</title>
         <author>muzafi17</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339571581</link>
         <description><![CDATA[<ol><li>DataWatch Monarch</li><li>Trifacta Wrangler</li><li>Openrefine</li></ol>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-09 15:57:27 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339571581</guid>
      </item>
      <item>
         <title>Correlation</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339571699</link>
         <description><![CDATA[<ul><li><strong>Description</strong><ul><li>Comparisons of two paired sets of <mark>measures to determine if one set goes up</mark>, the other set goes either <mark>up or down</mark> in a corresponding manner, and it so, how strongly.</li></ul></li><li><strong>Methods</strong><ul><li><strong>Points and trend line</strong> in the form of a <mark>scatter plot</mark>.</li><li><strong>Bars</strong> may be used, arranged as a <mark>paired bar graph or correlation bar</mark> graph, if scatter plots are unfamiliar.</li></ul></li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-09 15:58:38 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339571699</guid>
      </item>
      <item>
         <title>taktau ah noto</title>
         <author>muzafi17</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339571897</link>
         <description><![CDATA[]]></description>
         <enclosure url="" />
         <pubDate>2019-03-09 16:00:27 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339571897</guid>
      </item>
      <item>
         <title>Baca dan faham analysis</title>
         <author>muzafi17</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339573020</link>
         <description><![CDATA[]]></description>
         <enclosure url="" />
         <pubDate>2019-03-09 16:10:37 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339573020</guid>
      </item>
      <item>
         <title>No final</title>
         <author>muzafi17</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339573091</link>
         <description><![CDATA[]]></description>
         <enclosure url="" />
         <pubDate>2019-03-09 16:11:13 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339573091</guid>
      </item>
      <item>
         <title>Tools/Languages for DS</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339573627</link>
         <description><![CDATA[<ul><li><strong>R</strong> (<mark>statistics</mark>)</li><li><strong>Python</strong> (<mark>pandas, numpy</mark>)</li><li><strong>Scala </strong>(<mark>enable data streaming</mark>)</li><li><strong>SQL </strong>(<mark>RDBMS</mark>)</li><li><strong>Excel</strong> (<mark>data analysis</mark>)</li><li><strong>Java </strong></li><li><strong>Matlab</strong> (<mark>academic/math</mark>)</li><li><strong>SPSS</strong> (<mark>statistical packages</mark>)</li><li><strong>WEKA </strong>(<mark>machine learning</mark>)</li><li><strong>RapidMiner</strong> (<mark>DS software platform</mark>)</li><li><strong>Hadoop</strong> (<mark>process extremely large data sets</mark>)</li><li><strong>D3.js</strong> (<mark>data visualization</mark>)</li><li><strong>Tableau</strong> (<mark>data visualization</mark>)</li><li><strong>Apache Spark</strong> (<mark>machine learning</mark>)</li><li><strong>SAS </strong>(<mark>analytics software</mark>)</li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-09 16:15:46 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339573627</guid>
      </item>
      <item>
         <title>Structured Data</title>
         <author>muzafi17</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339574800</link>
         <description><![CDATA[<div>High degree of organisation such as a <mark>relational database</mark></div>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-09 16:26:10 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339574800</guid>
      </item>
      <item>
         <title>Unstructured  </title>
         <author>muzafi17</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339575066</link>
         <description><![CDATA[<div>Information that is <mark>difficult to organise</mark> using traditional mechanics</div>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-09 16:27:41 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339575066</guid>
      </item>
      <item>
         <title>Python vs R for Data Science</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339575644</link>
         <description><![CDATA[<div><a href="https://www.stoodnt.com/blog/r-vs-python-metareview-usability-popularity-pros-cons-jobs-salaries/">link</a></div><ul><li><strong>Python</strong><ul><li><strong>Advantages</strong><ul><li>Improvement in availability of <mark>Python packages</mark>.</li><li>A good tool to <mark>implement algorithms</mark> for production use.</li></ul></li></ul></li><li><strong>R</strong><ul><li><strong>Advantages</strong><ul><li><mark>Easier for beginner</mark> to do exploratory work.</li><li><mark>Usable</mark> for basic data analysis <mark>without</mark> installation of <mark>packages</mark></li></ul></li></ul></li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-09 16:33:00 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339575644</guid>
      </item>
      <item>
         <title>SD vs UD</title>
         <author>muzafi17</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339576407</link>
         <description><![CDATA[]]></description>
         <enclosure url="https://www.igneous.io/hs-fs/hubfs/structured-vs-unstructured-v01-3.png?width=800&amp;name=structured-vs-unstructured-v01-3.png" />
         <pubDate>2019-03-09 16:40:32 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339576407</guid>
      </item>
      <item>
         <title>Data Growth</title>
         <author>muzafi17</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339577738</link>
         <description><![CDATA[<div><strong>UD 60-80% per year</strong><br>1 exabytes = 1000 petabytes = 1 million terabytes = 1 billion gigabytes</div>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-09 16:53:00 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339577738</guid>
      </item>
      <item>
         <title>SQL Database</title>
         <author>muzafi17</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339578147</link>
         <description><![CDATA[<ul><li>Relational database (RDBMS)</li><li>Table based database</li><li>Web, mobile, enterprise, data mart</li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-09 16:56:34 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339578147</guid>
      </item>
      <item>
         <title>NoSQL Database</title>
         <author>muzafi17</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339578215</link>
         <description><![CDATA[<ul><li>Non-relational/distributed database</li><li>Document based, key-value pairs, graph database or wide-column stores</li><li>Gaming, social, iot, web</li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-09 16:57:14 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339578215</guid>
      </item>
      <item>
         <title>ACID Compliancy </title>
         <author>muzafi17</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339579666</link>
         <description><![CDATA[<ul><li><strong>Atomicity</strong> = All data and commands in a transaction <mark>succeed</mark>, or <mark>all fail and roll back</mark></li><li><strong>Consistency</strong> = All committed data must<mark> be consistent</mark> with all <mark>data rules</mark> including <mark>constraints, triggers, cascade</mark>, atomicity, isolation and durability</li><li><strong>Isolation</strong> = Other operations <mark>cannot access data</mark> that has been <mark>modified</mark> during a transaction that has <mark>not yet completed</mark></li><li><strong>Durability</strong> = Once a transaction is <mark>committed</mark>, data will <mark>survive system failures</mark>, and can be <mark>reliably recovered</mark> after an unwanted deletion</li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-09 17:10:15 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339579666</guid>
      </item>
      <item>
         <title>BASE</title>
         <author>muzafi17</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339580714</link>
         <description><![CDATA[<ul><li><mark>Basically Available</mark> = Guaranteed availability </li><li><mark>Soft-state</mark> = The state of the system may change, without even a query because of node updates</li><li><mark>Eventually consistent</mark> = The system will become consistent over time</li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-09 17:19:26 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339580714</guid>
      </item>
      <item>
         <title>Relational Database</title>
         <author>muzafi17</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339584092</link>
         <description><![CDATA[<div>Use the notion of databases separated into tables where each <mark>column represents a field</mark> and each <mark>row represents a record</mark></div><ul><li>MySQL</li><li>PostgreSQL</li><li>SQLite3</li></ul>]]></description>
         <enclosure url="http://lazyprogrammer.me/wp-content/uploads/2015/12/DbSearch_001.gif" />
         <pubDate>2019-03-09 17:49:46 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339584092</guid>
      </item>
      <item>
         <title>Normalisation of Relational DB</title>
         <author>muzafi17</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339584977</link>
         <description><![CDATA[<ul><li><mark>1NF</mark> : Eliminate groups of repeating data by creating a newtable for each group of related data which is identified by a primary key</li><li><mark>2NF</mark> : If a set of values are the same for multiple records move them to a new table and link the two tables with a foreign key</li><li><mark>3NF</mark> : Fields which do not depend on the primary key of a table must be removed and if necessary  be put into another table</li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-09 17:56:03 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339584977</guid>
      </item>
      <item>
         <title>ACID Properties of Relational DB</title>
         <author>muzafi17</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339585964</link>
         <description><![CDATA[<ul><li><strong>Atomicity</strong> : Either all parts of a transaction must be completed or none</li><li><strong>Consistency</strong> : The integrity of the database is preserved by all transactions. The database is not left in an invalid state after a transaction</li><li><strong>Isolation</strong> : A transaction must be run isolated in order to guarantee that any inconsistency in the data involved does not affect other transactions</li><li><strong>Durability</strong> : The changes made by a completed transaction must be preserved or in other words be durable</li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-09 18:05:28 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339585964</guid>
      </item>
      <item>
         <title>Non-Relational Database</title>
         <author>muzafi17</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339586891</link>
         <description><![CDATA[<ul><li>Represent data in collection of JSON docs</li><li>Records can have different fields (Dynamic Schema)</li></ul><ol><li><strong>Key-value</strong> <strong>stores</strong> <mark>Redis</mark> <mark>Voldemort</mark> <mark>Dynamo</mark></li><li><strong>Column-oriented</strong> <strong>database</strong> <mark>Cassandra</mark> <mark>Hbase </mark><a href="https://en.wikipedia.org/wiki/Column-oriented_DBMS">link</a></li><li><strong>Document-based</strong> <strong>stores</strong> <mark>MongoDB </mark><a href="https://en.wikipedia.org/wiki/Document-oriented_database">link</a></li><li><strong>Graph databases</strong> <mark>Neo4J</mark> <mark>InfiniteGraph </mark><a href="https://en.wikipedia.org/wiki/Graph_database">link</a></li></ol>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-09 18:15:31 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339586891</guid>
      </item>
      <item>
         <title>Advantages Of NoSQL</title>
         <author>muzafi17</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339588749</link>
         <description><![CDATA[<ol><li>Generally process data <mark>faster</mark> than relational DB</li><li>Faster because their data models are <mark>simpler</mark></li><li><mark>No schema</mark> required</li></ol>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-09 18:34:29 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339588749</guid>
      </item>
      <item>
         <title>Redis</title>
         <author>muzafi17</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339589240</link>
         <description><![CDATA[<ul><li><strong>Best used</strong>: For rapidly changing data with a foreseeable database size</li><li><strong>Use cases</strong>: To store real-time stock prices, Real-time analytics. Leaderboards. Real-time communication</li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-09 18:39:11 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339589240</guid>
      </item>
      <item>
         <title>Cassandra</title>
         <author>muzafi17</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339590897</link>
         <description><![CDATA[<ul><li><strong>Advantage of Column-oriented</strong> = Some types of data lookups can become very fast, given that the desired data could be stored consecutively in a single row</li><li><strong>Best used</strong>: When you need to store data so huge that it doesnt fit on server, but still want a friendly familiar interface to it</li><li><strong>Use cases</strong>: Web analytics, to count hits by hour, by browser, by IP. Transaction logging. Data collection from huge sensor arrays</li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-09 18:55:56 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339590897</guid>
      </item>
      <item>
         <title>MongoDB</title>
         <author>muzafi17</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339591783</link>
         <description><![CDATA[<ul><li><mark>Load balancing, replication, indexing, querying and act as file system</mark></li><li><strong>Best used</strong>: If you need dynamic queries. If you need good performance on big DB</li><li><strong>Use cases</strong>: For most things that you would do with MySQL or PostgreSQL but having predefined columns</li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-09 19:04:28 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339591783</guid>
      </item>
      <item>
         <title>Neo4J</title>
         <author>muzafi17</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339592004</link>
         <description><![CDATA[<ul><li><strong>Advantage of Graph-based</strong> = Potential in certain data mining and pattern recognition scenarios </li><li><strong>Best used:</strong> For graph-style, rich or complex, interconnected data</li><li><strong>Use cases:</strong> For searching routes in social relations, public data transport links, road maps, or network topologies</li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-09 19:07:21 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339592004</guid>
      </item>
      <item>
         <title>Python for Data Analysis</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339624516</link>
         <description><![CDATA[<div>After this thing, just for reading/understanding only. Never ask one. Just know the basic coding using pandas &amp; numpy.</div>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-10 02:46:15 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339624516</guid>
      </item>
      <item>
         <title>Numpy (Numerical Python)</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339624599</link>
         <description><![CDATA[<ul><li>A fast and efficient <mark>multidimensional array</mark> object ndarray</li><li>Tools for <mark>integrating C, C++ and Fortran</mark> code for Python</li><li>More efficient way of <mark>storing and manipulating data</mark> (numerical data).</li></ul><div><br></div>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-10 02:47:39 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339624599</guid>
      </item>
      <item>
         <title>Pandas</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339625085</link>
         <description><![CDATA[<div>The primary object in pandas is the <mark>DataFrame</mark>, a two-dimensional tabular, column oriented data structure with both <mark>row and column tables</mark>.<br><br>*this sht have a lot of stuff to read <br><br><a href="https://www.dataquest.io/blog/pandas-cheat-sheet/"><strong><mark>Pandas Cheat Sheet!</mark></strong></a></div>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-10 02:56:16 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339625085</guid>
      </item>
      <item>
         <title>Matplotlib</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339625729</link>
         <description><![CDATA[<div>Python library for <mark>producing plots</mark> and other <mark>2D data visualizations</mark>.</div>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-10 03:08:00 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339625729</guid>
      </item>
      <item>
         <title>Bridging the R-Python Divide</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339625909</link>
         <description><![CDATA[<div>Python package <strong><mark>rpy2</mark></strong> allows one to <mark>call R from within Python</mark><br><br>Can translate Python objects into R, pass into R functions, and convert the R output back into Python objects.<br><br>Python -&gt; R functions -&gt; R output -&gt; Python</div>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-10 03:10:52 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339625909</guid>
      </item>
      <item>
         <title>Trimester 2, 2016/17</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339627111</link>
         <description><![CDATA[<ul><li><strong>Question 1</strong><ul><li>Key-enablers (1)</li><li>5V's (1)</li><li>Challenges in DS (1)</li><li>Data structures (1)</li></ul></li><li><strong>Question 2</strong><ul><li>Types of questions (2)</li><li>Heterogeneity (4)</li><li>Data Quality Dimensions (2)</li><li>Errors/Inconsistencies (4)</li></ul></li><li><strong>Question 3</strong><ul><li>Python code (5)</li><li>Correlation (6)</li><li>Machine learning (7) </li></ul></li><li><strong>Question 4</strong><ul><li>SQL and NoSQL (11)</li><li>Python is the preferred language (5)</li><li>Appropriate Chart/Graph (9)</li></ul></li></ul><div><br>Chapter <mark>1, 2, 4, 5, 6, 7, 9 ,11<br></mark>Yg x keluar, <mark>3,8,10,12</mark></div>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-10 03:28:24 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339627111</guid>
      </item>
      <item>
         <title>Trimester 1, 2017/18</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339629590</link>
         <description><![CDATA[<ul><li><strong>Question 1</strong><ul><li>Responsiblities of Data Scientist (1)</li><li>5V's (1)</li><li>Big Data Insights (1)</li><li>Characteristics of Big Data (1)</li></ul></li><li><strong>Question 2</strong><ul><li>Types of questions (2)</li><li>Outliers (6)</li><li>Data Quality Dimensions (2)</li><li>Errors/Inconsistencies (4)</li></ul></li><li><strong>Question 3</strong><ul><li>Python code (5)</li><li>Machine learning (7) </li><li>Correlation (6)</li></ul></li><li><strong>Question 4</strong><ul><li>SQL and NoSQL (11)</li><li>R is the preferred language (5)</li><li>Graph/Table (9)</li></ul></li></ul><div><br>Chapter <mark>1, 2, 4, 5, 6, 7, 9, 11<br></mark>Yg x keluar, <mark>3,8,10,12</mark></div><div><br></div>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-10 04:03:21 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339629590</guid>
      </item>
      <item>
         <title>Trimester 2, 2017/18</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339630319</link>
         <description><![CDATA[<ul><li><strong>Question 1</strong><ul><li>BI vs DS (1)</li><li>Responsibilities of Data Scientist (1)</li></ul></li><li><strong>Question 2</strong><ul><li>Entropy-based (4)</li><li>Calculate Gain (4)</li><li>NoSQL (11)</li><li>MongoDB (11)</li><li>Python packages contains R (5)</li></ul></li><li><strong>Question 3</strong><ul><li>Training set and Test Set (8)</li><li>Association Rules Mining - Confidence, Lift (7)</li><li>Euclidean Distance (7)</li></ul></li><li><strong>Question 4</strong><ul><li>Python Code (5)</li><li>Graph/Table (9)</li></ul></li></ul><div><br>Chapter <mark>1, 4, 5, 7, 8, 9, 11<br></mark>Yg x keluar, <mark>2, 3, 6, 10 ,12</mark></div><div><br></div>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-10 04:12:31 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339630319</guid>
      </item>
      <item>
         <title>Pick Rate</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339630958</link>
         <description><![CDATA[<div>100%</div><ul><li>Chapter <mark>1, 4, 5, 7, 9, 11</mark></li></ul><div>67% (Twice)</div><ul><li>Chapter 2, 6</li></ul><div>33% (Once)</div><ul><li>Chapter 8</li></ul><div>0% (Never)</div><ul><li>Chapter 3, 10, 12</li></ul><div><br><strong>Most Favorite Questions</strong></div><ul><li>5V's (1)</li><li>Responsibilities of Data Scientist (1)</li><li>Type of Questions (1)</li><li>Malas da. lol</li><li>Buat sendiri</li><li>Bapak ar</li><li>Pahal kau</li><li>Jahat nya kamu</li><li>k</li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-10 04:22:39 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339630958</guid>
      </item>
      <item>
         <title>Discrepancies in Data</title>
         <author>muzafi17</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339631753</link>
         <description><![CDATA[<div>1. <strong>Incomplete data</strong></div><ul><li>Non available value</li><li>Different criteria between time data collected when it is analysed</li><li>Human/hardware/software problems</li></ul><div><mark>Ignore the tuple<br>Fill in missing value manually<br>Use a global constant to fill<br>Use the attribute mean/median</mark> <mark>Predict using a learning algorithm</mark></div><div><br>2. <strong>Noisy data</strong></div><ul><li>Faulty instruments</li><li>Data entry or computer errors</li><li>Data transmission</li></ul><div><mark>Binning methods<br>Outlier analysis<br>Regression<br>Combination of human and computer inspection</mark></div><div><br>3. <strong>Inconsistent / redundant</strong></div><ul><li>Different data sources</li><li>Functional dependencies</li><li>Referential integrity violation</li></ul><div><mark>Naming conventions<br>Parsing text into fields<br>Different representations <br>Primary key violation<br>Formatting issues</mark></div>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-10 04:34:37 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339631753</guid>
      </item>
      <item>
         <title>Traditional Database vs Modern Database
</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339690147</link>
         <description><![CDATA[<div><strong>- Data Sources<br></strong>Both: MS SQL<br>T: Tables, Files<br>M: Social, Streaming<br><br><strong>- Data Structure<br></strong>Both: Structured data<br>T: ER-model, Multi-dimensional Model<br>M: Unstructured data, Machine data<br><br><strong>- Data Access<br></strong>T: SQL<br>M: Parallel processing, Distributed processing, In-memory processing<br><br><strong>- Analytics<br></strong>T: Reporting, Dashboard, Data Analysis<br>M: Regression, Clustering, Data Mining</div>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-10 15:45:12 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339690147</guid>
      </item>
      <item>
         <title>Skills in Academia vs Industry?</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339690190</link>
         <description><![CDATA[<div><strong>Academia</strong><br>- Communication and People Skill<br>- Industry Skill<br>- Business Process Skill<br>+ Data Skill<br>+ Analytical Skill<br>+ Technical Skill<br><br><strong>Industry</strong><br>+ Communication and People Skill<br>+ Industry Skill<br>+ Business Process Skill<br>- Data Skill<br>- Analytical Skill<br>- Technical Skill</div>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-10 15:45:36 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339690190</guid>
      </item>
      <item>
         <title>Simpson&#39;s Paradox</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339690541</link>
         <description><![CDATA[<div>Simpson Paradox states that the representation of the data will be <mark>misleading without considering the confounding factors</mark>.</div>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-10 15:48:05 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339690541</guid>
      </item>
      <item>
         <title>Metrics to evaluate model accur</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339690694</link>
         <description><![CDATA[<div><strong>- Classification accuracy</strong><br><strong>Sensitivity (Recall)</strong> = TP / P<br><strong>Precision</strong> = TP / P'<br><strong>F1-score</strong><br>P = all actual positive values<br>P' = all predicted positive values<br>F1 = 2 x R x P / (R+P)<br><br><strong>Regression metrics:</strong><br>- MAE<br>- MSE<br>- <br><br>- Area under the ROC graph</div>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-10 15:49:04 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339690694</guid>
      </item>
      <item>
         <title>Choropleth </title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339690945</link>
         <description><![CDATA[<div>to represent values on a <mark>map</mark> with <mark>different shade of colour</mark> to represent the value.</div>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-10 15:50:49 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339690945</guid>
      </item>
      <item>
         <title>Refer
Refer</title>
         <author>shahrinamin_my</author>
         <link>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339848510</link>
         <description><![CDATA[<div>https://en.wikipedia.org/wiki/K-nearest_neighbors_algorithm</div>]]></description>
         <enclosure url="" />
         <pubDate>2019-03-11 09:10:24 UTC</pubDate>
         <guid>https://padlet.com/shahrinamin_my/2tulveioj9u2/wish/339848510</guid>
      </item>
   </channel>
</rss>
