<?xml version="1.0"?>
<rss version="2.0">
   <channel>
      <title>Evaluation LLMs by </title>
      <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk</link>
      <description></description>
      <language>en-us</language>
      <pubDate>2023-09-19 15:18:35 UTC</pubDate>
      <lastBuildDate>2023-11-21 13:13:46 UTC</lastBuildDate>
      <webMaster>hello@padlet.com</webMaster>
      <image>
         <url>https://padlet.net/icons/png/1f4bb.png</url>
      </image>
      <item>
         <title>EVALUATION</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2711104510</link>
         <description><![CDATA[]]></description>
         <enclosure url="" />
         <pubDate>2023-09-19 15:45:50 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2711104510</guid>
      </item>
      <item>
         <title>Benchmarks</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712170673</link>
         <description><![CDATA[]]></description>
         <enclosure url="" />
         <pubDate>2023-09-20 05:39:02 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712170673</guid>
      </item>
      <item>
         <title>Metrics</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712170917</link>
         <description><![CDATA[]]></description>
         <enclosure url="" />
         <pubDate>2023-09-20 05:39:17 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712170917</guid>
      </item>
      <item>
         <title>Frameworks</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712176299</link>
         <description><![CDATA[]]></description>
         <enclosure url="" />
         <pubDate>2023-09-20 05:43:52 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712176299</guid>
      </item>
      <item>
         <title>Datasets</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712179071</link>
         <description><![CDATA[]]></description>
         <enclosure url="" />
         <pubDate>2023-09-20 05:46:19 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712179071</guid>
      </item>
      <item>
         <title></title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712179926</link>
         <description><![CDATA[<div>https://www.markiiisys.com/blog/benchmarking-large-language-models-llms-a-quick-tour/</div>]]></description>
         <enclosure url="" />
         <pubDate>2023-09-20 05:46:55 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712179926</guid>
      </item>
      <item>
         <title></title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712192146</link>
         <description><![CDATA[<div>https://levelup.gitconnected.com/how-to-benchmark-language-models-by-openai-deepmind-google-microsoft-783d4307ec50</div>]]></description>
         <enclosure url="" />
         <pubDate>2023-09-20 05:56:52 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712192146</guid>
      </item>
      <item>
         <title>GLUE</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712200121</link>
         <description><![CDATA[<div>https://gluebenchmark.com/</div>]]></description>
         <enclosure url="" />
         <pubDate>2023-09-20 06:02:56 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712200121</guid>
      </item>
      <item>
         <title>SuperGLUE</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712200434</link>
         <description><![CDATA[<div><a href="https://super.gluebenchmark.com/">SuperGLUE Benchmark</a><br>https://deepgram.com/learn/superglue-llm-benchmark-explained<br><br></div>]]></description>
         <enclosure url="" />
         <pubDate>2023-09-20 06:03:13 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712200434</guid>
      </item>
      <item>
         <title>MMLU</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712210627</link>
         <description><![CDATA[<div><a href="https://github.com/hendrycks/test/tree/master">GitHub - hendrycks/test: Measuring Massive Multitask Language Understanding | ICLR 2021</a></div>]]></description>
         <enclosure url="" />
         <pubDate>2023-09-20 06:11:12 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712210627</guid>
      </item>
      <item>
         <title>Big Bench</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712211512</link>
         <description><![CDATA[<div><a href="https://github.com/google/BIG-bench">GitHub - google/BIG-bench: Beyond the Imitation Game collaborative benchmark for measuring and extrapolating the capabilities of language models</a></div>]]></description>
         <enclosure url="" />
         <pubDate>2023-09-20 06:11:52 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712211512</guid>
      </item>
      <item>
         <title></title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712229756</link>
         <description><![CDATA[<div>https://tech.ebu.ch/docs/events/webinar_2023_llm_benchmarking/presentations/webinar-llm-tv2.pdf</div>]]></description>
         <enclosure url="" />
         <pubDate>2023-09-20 06:25:37 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712229756</guid>
      </item>
      <item>
         <title>How to Benchmark?</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712251304</link>
         <description><![CDATA[]]></description>
         <enclosure url="" />
         <pubDate>2023-09-20 06:39:30 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712251304</guid>
      </item>
      <item>
         <title>Task-Specific Evaluation</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712252078</link>
         <description><![CDATA[]]></description>
         <enclosure url="" />
         <pubDate>2023-09-20 06:40:01 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712252078</guid>
      </item>
      <item>
         <title>Few-Shot Learning Evaluation</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712252502</link>
         <description><![CDATA[]]></description>
         <enclosure url="" />
         <pubDate>2023-09-20 06:40:15 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712252502</guid>
      </item>
      <item>
         <title>Zero-Shot Learning Evaluation</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712253321</link>
         <description><![CDATA[]]></description>
         <enclosure url="" />
         <pubDate>2023-09-20 06:40:49 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712253321</guid>
      </item>
      <item>
         <title>Fine-Tuning Evaluation</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712270302</link>
         <description><![CDATA[<div><a href="https://github.com/facebookresearch/llama-recipes/tree/main">GitHub - facebookresearch/llama-recipes: Examples and recipes for Llama 2 model</a></div>]]></description>
         <enclosure url="" />
         <pubDate>2023-09-20 06:50:51 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712270302</guid>
      </item>
      <item>
         <title>Human Evaluation</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712270706</link>
         <description><![CDATA[<div><a href="https://www.topbots.com/llm-performance-evaluation/">Beyond Metrics: A Hybrid Approach to LLM Performance Evaluation (topbots.com)</a><br><br><a href="https://www.surgehq.ai/blog/how-good-is-hugging-faces-bloom-a-real-world-human-evaluation-of-language-models">Human Evaluation of Large Language Models: How Good is Hugging Face's BLOOM? (surgehq.ai)</a></div>]]></description>
         <enclosure url="" />
         <pubDate>2023-09-20 06:51:08 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712270706</guid>
      </item>
      <item>
         <title>Bias and Fairness Evaluation</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712271092</link>
         <description><![CDATA[]]></description>
         <enclosure url="" />
         <pubDate>2023-09-20 06:51:23 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712271092</guid>
      </item>
      <item>
         <title>Safety and Robustness Evaluation</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712272164</link>
         <description><![CDATA[]]></description>
         <enclosure url="" />
         <pubDate>2023-09-20 06:52:10 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712272164</guid>
      </item>
      <item>
         <title>Perplexity</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712273357</link>
         <description><![CDATA[]]></description>
         <enclosure url="" />
         <pubDate>2023-09-20 06:53:08 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712273357</guid>
      </item>
      <item>
         <title>BLEU (Bilingual Evaluation Understudy) Score</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712273821</link>
         <description><![CDATA[]]></description>
         <enclosure url="" />
         <pubDate>2023-09-20 06:53:29 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712273821</guid>
      </item>
      <item>
         <title>ROUGE (Recall-Oriented Understudy for Gisting Evaluation) Score</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712274183</link>
         <description><![CDATA[]]></description>
         <enclosure url="" />
         <pubDate>2023-09-20 06:53:46 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712274183</guid>
      </item>
      <item>
         <title>F1 Score, Accuracy, Precision, Recall</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712274557</link>
         <description><![CDATA[]]></description>
         <enclosure url="" />
         <pubDate>2023-09-20 06:54:00 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712274557</guid>
      </item>
      <item>
         <title>Spearman&#39;s Rank Correlation Coefficient</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712317200</link>
         <description><![CDATA[]]></description>
         <enclosure url="" />
         <pubDate>2023-09-20 07:23:55 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712317200</guid>
      </item>
      <item>
         <title>.</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712317515</link>
         <description><![CDATA[]]></description>
         <enclosure url="" />
         <pubDate>2023-09-20 07:24:07 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712317515</guid>
      </item>
      <item>
         <title>Language Model Evaluation Harness</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712326530</link>
         <description><![CDATA[<div><a href="https://github.com/EleutherAI/lm-evaluation-harness">GitHub - EleutherAI/lm-evaluation-harness: A framework for few-shot evaluation of autoregressive language models.</a><br><br><a href="https://wandb.ai/wandb_gen/llm-evaluation/reports/Evaluating-Large-Language-Models-LLMs-with-Eleuther-AI--VmlldzoyOTI0MDQ3">Evaluating Large Language Models (LLMs) with Eleuther AI | llm-evaluation – Weights &amp; Biases (wandb.ai)</a></div>]]></description>
         <enclosure url="" />
         <pubDate>2023-09-20 07:30:15 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712326530</guid>
      </item>
      <item>
         <title>HELM</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712326867</link>
         <description><![CDATA[<div><a href="https://github.com/stanford-crfm/helm">GitHub - stanford-crfm/helm: Holistic Evaluation of Language Models (HELM), a framework to increase the transparency of language models (https://arxiv.org/abs/2211.09110).</a></div>]]></description>
         <enclosure url="" />
         <pubDate>2023-09-20 07:30:32 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712326867</guid>
      </item>
      <item>
         <title>OpenAI Evals</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712328416</link>
         <description><![CDATA[<div><a href="https://github.com/openai/evals">GitHub - openai/evals: Evals is a framework for evaluating LLMs and LLM systems, and an open-source registry of benchmarks.</a></div>]]></description>
         <enclosure url="" />
         <pubDate>2023-09-20 07:31:36 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712328416</guid>
      </item>
      <item>
         <title></title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712344675</link>
         <description><![CDATA[<div>https://deepgram.com/learn/llm-benchmarks-guide-to-evaluating-language-models</div>]]></description>
         <enclosure url="" />
         <pubDate>2023-09-20 07:43:13 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2712344675</guid>
      </item>
      <item>
         <title>Semantic Answer Similarity (SAS)</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2713075316</link>
         <description><![CDATA[<div><a href="https://arxiv.org/abs/2108.06130?ref=radekosmulski.com">[2108.06130] Semantic Answer Similarity for Evaluating Question Answering Models (arxiv.org)</a><br><br><a href="https://radekosmulski.com/how-to-evaluate-an-llm-on-your-data/">How to evaluate an LLM on your data? (radekosmulski.com)</a><br><br><a href="https://gist.github.com/radekosmulski/00dc4b4915ea38e79c1c647c90d757c3?ref=radekosmulski.com">evaluate_LLM.ipynb · GitHub</a></div>]]></description>
         <enclosure url="" />
         <pubDate>2023-09-20 15:51:02 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2713075316</guid>
      </item>
      <item>
         <title>METEOR</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2713077880</link>
         <description><![CDATA[<div><a href="https://machinetranslate.org/meteor">METEOR | Machine Translate</a></div>]]></description>
         <enclosure url="" />
         <pubDate>2023-09-20 15:52:28 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2713077880</guid>
      </item>
      <item>
         <title>Tasks</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2714102247</link>
         <description><![CDATA[]]></description>
         <enclosure url="" />
         <pubDate>2023-09-21 05:51:47 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2714102247</guid>
      </item>
      <item>
         <title></title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2714102645</link>
         <description><![CDATA[<div><a href="https://www.mosaicml.com/llm-evaluation">LLM Evaluation Metrics (mosaicml.com)</a></div>]]></description>
         <enclosure url="" />
         <pubDate>2023-09-21 05:52:01 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2714102645</guid>
      </item>
      <item>
         <title></title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2714149082</link>
         <description><![CDATA[<div><a href="https://toloka.ai/blog/evaluating-llms/">Evaluating Large Language Models (toloka.ai)</a></div>]]></description>
         <enclosure url="" />
         <pubDate>2023-09-21 06:26:42 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2714149082</guid>
      </item>
      <item>
         <title>Question answering</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2714786249</link>
         <description><![CDATA[]]></description>
         <enclosure url="" />
         <pubDate>2023-09-21 14:50:46 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2714786249</guid>
      </item>
      <item>
         <title>Summarization</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2714786897</link>
         <description><![CDATA[<div><a href="https://levelup.gitconnected.com/text-summarization-llama2-how-to-use-llama2-with-langchain-ad5775c80716">Text Summarization Llama2: how to Use LLama2 with Langchain | by Tarik Kaoutar (高達烈) | Level Up Coding (gitconnected.com)</a></div>]]></description>
         <enclosure url="" />
         <pubDate>2023-09-21 14:51:08 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2714786897</guid>
      </item>
      <item>
         <title>Classification</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2714788147</link>
         <description><![CDATA[]]></description>
         <enclosure url="" />
         <pubDate>2023-09-21 14:51:55 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2714788147</guid>
      </item>
      <item>
         <title>Text generation</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2714788588</link>
         <description><![CDATA[]]></description>
         <enclosure url="" />
         <pubDate>2023-09-21 14:52:14 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2714788588</guid>
      </item>
      <item>
         <title>Machine translation</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2716097771</link>
         <description><![CDATA[]]></description>
         <enclosure url="" />
         <pubDate>2023-09-22 09:35:51 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2716097771</guid>
      </item>
      <item>
         <title>Logic, math, code</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2716104349</link>
         <description><![CDATA[<div>logical reasoning<br>mathematics<br>computer code<br><br></div>]]></description>
         <enclosure url="" />
         <pubDate>2023-09-22 09:41:50 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2716104349</guid>
      </item>
      <item>
         <title>Auto-evaluation</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2718814777</link>
         <description><![CDATA[<div><a href="https://autoevaluator.langchain.com/">Auto-Evaluator (langchain.com)</a></div>]]></description>
         <enclosure url="" />
         <pubDate>2023-09-25 06:59:08 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2718814777</guid>
      </item>
      <item>
         <title>Jiant</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2719534436</link>
         <description><![CDATA[<div><a href="https://github.com/nyu-mll/jiant">GitHub - nyu-mll/jiant: jiant is an nlp toolkit</a></div>]]></description>
         <enclosure url="" />
         <pubDate>2023-09-25 15:17:54 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2719534436</guid>
      </item>
      <item>
         <title></title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2719544270</link>
         <description><![CDATA[<div><a href="https://github.com/huggingface/evaluate/tree/main">GitHub - huggingface/evaluate: 🤗 Evaluate: A library for easily evaluating machine learning models and datasets.</a><br><br>prepojiť ktore metriky obsahuje?</div>]]></description>
         <enclosure url="" />
         <pubDate>2023-09-25 15:23:55 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2719544270</guid>
      </item>
      <item>
         <title>Library</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2719544883</link>
         <description><![CDATA[]]></description>
         <enclosure url="" />
         <pubDate>2023-09-25 15:24:15 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2719544883</guid>
      </item>
      <item>
         <title></title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2719557539</link>
         <description><![CDATA[<div><a href="https://github.com/EleutherAI/lm-evaluation-harness/blob/master/docs/task_table.md">lm-evaluation-harness/docs/task_table.md at master · EleutherAI/lm-evaluation-harness · GitHub</a></div>]]></description>
         <enclosure url="" />
         <pubDate>2023-09-25 15:31:30 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2719557539</guid>
      </item>
      <item>
         <title></title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2719570549</link>
         <description><![CDATA[<div><a href="https://github.com/stanford-crfm/helm/tree/main/src/helm/benchmark/metrics">helm/src/helm/benchmark/metrics at main · stanford-crfm/helm · GitHub</a></div>]]></description>
         <enclosure url="" />
         <pubDate>2023-09-25 15:39:05 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2719570549</guid>
      </item>
      <item>
         <title></title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2719895444</link>
         <description><![CDATA[<div><a href="https://towardsdatascience.com/how-to-validate-openai-gpt-model-performance-with-text-summarization-298978fea764">How to Validate OpenAI GPT Model Performance with Text Summarization | by Mark Chen | Towards Data Science</a></div>]]></description>
         <enclosure url="" />
         <pubDate>2023-09-25 19:07:21 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2719895444</guid>
      </item>
      <item>
         <title>BERTscore</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2719905164</link>
         <description><![CDATA[]]></description>
         <enclosure url="" />
         <pubDate>2023-09-25 19:15:04 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2719905164</guid>
      </item>
      <item>
         <title>Reasoning</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2719913255</link>
         <description><![CDATA[]]></description>
         <enclosure url="" />
         <pubDate>2023-09-25 19:21:49 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2719913255</guid>
      </item>
      <item>
         <title>General</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2719913769</link>
         <description><![CDATA[]]></description>
         <enclosure url="" />
         <pubDate>2023-09-25 19:22:16 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2719913769</guid>
      </item>
      <item>
         <title>Specific</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2719913956</link>
         <description><![CDATA[]]></description>
         <enclosure url="" />
         <pubDate>2023-09-25 19:22:23 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2719913956</guid>
      </item>
      <item>
         <title></title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2719991918</link>
         <description><![CDATA[<div>https://www.analyticsvidhya.com/blog/2023/05/how-to-evaluate-a-large-language-model-llm/#h-table-of-the-major-existing-evaluation-frameworks</div>]]></description>
         <enclosure url="" />
         <pubDate>2023-09-25 20:49:22 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2719991918</guid>
      </item>
      <item>
         <title>Exact match</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2722676443</link>
         <description><![CDATA[]]></description>
         <enclosure url="" />
         <pubDate>2023-09-27 07:58:13 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2722676443</guid>
      </item>
      <item>
         <title>fluency, coherence, Relevance, Context understanding</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2723099924</link>
         <description><![CDATA[]]></description>
         <enclosure url="" />
         <pubDate>2023-09-27 13:23:50 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2723099924</guid>
      </item>
      <item>
         <title></title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2724153704</link>
         <description><![CDATA[<div><a href="https://super.gluebenchmark.com/tasks">SuperGLUE Benchmark</a> tasks/datasets</div>]]></description>
         <enclosure url="" />
         <pubDate>2023-09-28 06:09:03 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2724153704</guid>
      </item>
      <item>
         <title>Definition of business problem</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2762513730</link>
         <description><![CDATA[]]></description>
         <enclosure url="" />
         <pubDate>2023-10-25 08:12:30 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2762513730</guid>
      </item>
      <item>
         <title>Task type</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2762513912</link>
         <description><![CDATA[<ul><li>General-purpose: Designed to understand and process a wide range of language tasks and contexts, without aiming for a particular language domain or use case.</li><li>Specific:&nbsp;<ul><li>Question Answering</li><li>Information Retrieval</li><li>Document Summarization</li><li>Machine Translation</li><li>Sentiment analysis</li><li>Text Classification</li><li>...</li></ul></li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2023-10-25 08:12:39 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2762513912</guid>
      </item>
      <item>
         <title>Model</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2762515376</link>
         <description><![CDATA[<div>Questions of cost and accuracy.<br>Is the model freely available? Paid?&nbsp;<br>Access (API...)<br>Latency<br>Which model performs best in your use case on leaderboard?<br><br></div><ul><li><a href="https://llm.extractum.io/">LLM Explorer: Large Language Model Directory and Analytics (extractum.io)</a></li><li><a href="https://crfm.stanford.edu/helm/latest/?groups=1">Holistic Evaluation of Language Models (HELM) (stanford.edu)</a></li><li><a href="https://huggingface.co/spaces/HuggingFaceH4/open_llm_leaderboard">Open LLM Leaderboard - a Hugging Face Space by HuggingFaceH4</a></li></ul><div><br></div>]]></description>
         <enclosure url="" />
         <pubDate>2023-10-25 08:13:55 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2762515376</guid>
      </item>
      <item>
         <title>Data</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2762515483</link>
         <description><![CDATA[]]></description>
         <enclosure url="" />
         <pubDate>2023-10-25 08:14:01 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2762515483</guid>
      </item>
      <item>
         <title>Metrics</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2762515887</link>
         <description><![CDATA[<div>Evaluation metrics vary by use case.<br>See Quantitative Metrics, Reference Comparisons and Criteria-Based Evaluation in the document.</div>]]></description>
         <enclosure url="" />
         <pubDate>2023-10-25 08:14:24 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2762515887</guid>
      </item>
      <item>
         <title>Benchmark datasets</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2762519577</link>
         <description><![CDATA[<ul><li><p><a rel="noopener noreferrer nofollow" href="https://gluebenchmark.com/tasks">GLUE Benchmark</a></p></li><li><p><a rel="noopener noreferrer nofollow" href="https://super.gluebenchmark.com/tasks">SuperGLUE Benchmark</a></p></li><li><p><a rel="noopener noreferrer nofollow" href="https://github.com/openai/evals/tree/main/evals/registry/data">evals/evals/registry/data at main · openai/evals · GitHub</a></p></li><li><p>...</p></li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2023-10-25 08:17:33 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2762519577</guid>
      </item>
      <item>
         <title>Own datasets</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2762519710</link>
         <description><![CDATA[<div>The format of data varies based on the model and task.</div>]]></description>
         <enclosure url="" />
         <pubDate>2023-10-25 08:17:38 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2762519710</guid>
      </item>
      <item>
         <title>Evaluation goals</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2762522447</link>
         <description><![CDATA[<ul><li>Do we want to track one goal or several functionalities of the model? For metrics used for a specific task type, see the document.</li><li>If we use data that someone else has already used, do we want to compare our metric results with their results? If so, then we must choose the same metrics.</li><li>What approach will we use to evaluate our solution?</li><li>Are we only interested in the accuracy of the model or other features such as grammar, coherence, relevance, politeness?</li><li>Who will be the users? Is it important to also focus on social aspects like bias, fairness, toxicity, misinformation, hallucination?</li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2023-10-25 08:19:56 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2762522447</guid>
      </item>
      <item>
         <title>Evaluation</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2762531152</link>
         <description><![CDATA[<div>See the benchmarks and their implementation and the metrics library in the document.<br>Analysis of results.</div>]]></description>
         <enclosure url="" />
         <pubDate>2023-10-25 08:26:53 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2762531152</guid>
      </item>
      <item>
         <title>Approach</title>
         <author>miroslavamatejova</author>
         <link>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2764598946</link>
         <description><![CDATA[<ul><li><p>Standard evaluation</p><ul><li><p>Evaluation using NLP and other standard metrics.</p></li></ul></li><li><p>Auto-evaluation/Self-evaluation (LLM evaluation)</p><ul><li><p>LLMs can be used to check the result of their own or other LLM's outputs.</p></li></ul></li><li><p>Human evaluation</p><ul><li><p>Human annotators are used to assess the quality of open-ended model responses.</p></li></ul></li><li><p>Hybrid evaluation</p><ul><li><p>Combination of Auto-evaluation and Human evaluation</p></li></ul></li></ul>]]></description>
         <enclosure url="" />
         <pubDate>2023-10-26 12:16:59 UTC</pubDate>
         <guid>https://padlet.com/miroslavamatejova/mt5zcwoy6nev9zhk/wish/2764598946</guid>
      </item>
   </channel>
</rss>
