diff --git a/README.md b/README.md index c744010..89cf664 100644 --- a/README.md +++ b/README.md @@ -13,14 +13,13 @@ This project compares different methods of extracting scores from language model ## Results -| Method | Score | Score (Normalized) | +| Method | Score | Score (Normalized) | |---------------|----------|------------| -| ranked_scaled | 0.629332 | 0.80 | -| ranked_norm | 0.654562 | 0.74 | -| weighted | 0.634804 | 0.65 | -| raw | 0.634528 | 0.65 | -| weighted_norm | 0.623806 | 0.64 | -| ranked | 0.336333 | 0.28 | +| ranked_scaled | 0.62 | 0.80 | +| ranked_norm | 0.65 | 0.74 | +| weighted | 0.63 | 0.65 | +| raw (baseline)| 0.63 | 0.65 | +| weighted_norm | 0.62 | 0.64 | *Results for DeepSeek Chat V3 0324* diff --git a/nbs/02_recomp.ipynb b/nbs/02_recomp.ipynb index 56fefb6..87d9a51 100644 --- a/nbs/02_recomp.ipynb +++ b/nbs/02_recomp.ipynb @@ -1238,6 +1238,174 @@ }, "metadata": {}, "output_type": "display_data" + }, + { + "data": { + "application/vnd.jupyter.widget-view+json": { + "model_id": "15176e01aad14ebfbd291c777f801d35", + "version_major": 2, + "version_minor": 0 + }, + "text/plain": [ + " 0%| | 0/24 [00:00