From 53c6d2f6e99c87acf0f5cec1089ddd930bc1b20d Mon Sep 17 00:00:00 2001 From: Michael Ryaboy Date: Thu, 24 Sep 2026 02:23:58 -0700 Subject: [PATCH 1/2] Measure gzip and Zstd text classification --- eval/README.md | 101 ++ eval/compression-results.json | 2655 +++++++++++++++++++++++++++++++++ eval/compression.py | 371 +++++ 3 files changed, 3127 insertions(+) create mode 100644 eval/compression-results.json create mode 100644 eval/compression.py diff --git a/eval/README.md b/eval/README.md index 2009d1f..24d5ea8 100644 --- a/eval/README.md +++ b/eval/README.md @@ -119,3 +119,104 @@ profile, not on a 0.004 difference in F1. In rough order of value: add cases from a real taxonomy with labels assigned before any model output is seen; split into tune and report halves; raise n past about thirty so confidence intervals mean something. + +## Compression classifiers + +`compression.py` tests whether gzip/DEFLATE or Zstd can replace a learned text +classifier. It compares gzip NCD nearest neighbors, per-class DEFLATE and Zstd +dictionaries, Zstd dictionary mixtures, and a word/character TF-IDF logistic +regression control. It does not change the Worker or call a paid model API. + +```sh +git clone https://github.com/fstandhartinger/jevbench.git /tmp/compression-jevbench +git -C /tmp/compression-jevbench checkout 2fa63fa3226cb369795525ed011800f57dcbd894 +OMP_NUM_THREADS=1 OPENBLAS_NUM_THREADS=1 uv run --python 3.12 eval/compression.py \ + --out eval/data/compression-reproduction --jevbench /tmp/compression-jevbench +``` + +Run from the repository root. The output directory must be new. `uv` installs +the script's pinned dependencies; Hugging Face downloads are cached. Omit +`--jevbench` to run only the supervised datasets. `--datasets`, `--families`, +`--seed`, `--train-cap`, `--validation-n`, and `--n` bound additional experiments. + +The default uses seed 0 and 500 evaluation rows, matching the sampling protocol +of [dhruvmehra/jevbench](https://github.com/dhruvmehra/jevbench). Each dataset has +up to 10,000 training rows and 600 separate validation rows. Training is shuffled +and deduplicated after excluding exact normalized matches anywhere in the +held-out split. Classifier settings and probability temperature are chosen only +on validation rows, then frozen for evaluation. This excludes exact duplicates, +not semantic paraphrases or SST-2's related review fragments. Test rows remain +in their original benchmark sample, including any duplicates within that split. + +Gzip nearest neighbors deliberately retain at most 1,000 balanced examples; its +accuracy is a bounded-compute baseline, not an estimate of full-data gzip kNN. +DEFLATE uses raw zlib preset dictionaries (the same compression algorithm as +gzip, without gzip framing). Zstd tries trained dictionaries and raw class +text, with minimum or mean length over random shards. Compression length is +only a score: softmax temperature is fitted on validation labels. Dictionaries +store information from labeled examples; this is supervised learning, not a +zero-shot or memory-free model. Oversized dictionary training failures are +recorded in the validation sweep, never silently replaced by another method. + +Every run produces `summary.json`, per-dataset validation sweeps and selected +settings, per-item predictions/probabilities, split hashes, dataset fingerprints, +and software versions. Timing covers serial local normalization, scoring, +probabilities and tie-breaking; it excludes training, loading, serving and +network overhead. Empty, Unicode and long inputs are exercised for every +selected supervised model. The output manifest identifies the script by hash. + +The optional Benchmark Heaven diagnostic uses symmetric gzip NCD between the +state/instructions and each label/criterion. The prediction function accepts +only those inference fields; expected answers and author rationales cannot +enter it. This has no labeled support set and returns labels without invented +calibrated probabilities. Its three public cohorts total 231 decisions. These +are public diagnostic accuracies, not an official JevBench score: private, +sealed, and imported tasks are absent. A supervised banking/news classifier +cannot be submitted as if it solves arbitrary typed decisions. + +The compression idea comes from [Nathan Barry's gzip language-model experiment](https://nathan.rs/posts/gzip-lm/) +and [FTCC's class-dictionary approach](https://github.com/cyrilou242/ftcc). +The TF-IDF control matters: [Gzip versus bag-of-words for text classification](https://arxiv.org/abs/2307.15002) +examines whether compression's gains survive comparison with conventional text +features. + +### Measured results (Apple M5 Max) + +Generated from [compression-results.json](compression-results.json). All rows +use 500 held-out items per dataset; percentages are accuracy. Each family selects +its settings on validation data. These are exploratory measurements on one seed, +not evidence that small differences are statistically significant. + +| Up to 10k training rows | Gzip kNN (≤1k memory) | DEFLATE | Zstd mixture | TF-IDF + logistic | +|---|---:|---:|---:|---:| +| agnews | 61.8% | 76.6% | 84.8% | 88.8% | +| banking77 | 54.8% | 82.0% | 80.6% | 90.8% | +| sst2 | 54.8% | 65.6% | 71.8% | 79.6% | +| emotion | 29.8% | 40.6% | 56.2% | 85.2% | + +Larger-training follow-up (same evaluation items, separately held-back validation): + +| Dataset | Training rows | Zstd mixture | p50 | TF-IDF + logistic | p50 | +|---|---:|---:|---:|---:|---:| +| agnews | 119,239 | 90.0% | 0.246 ms | 91.2% | 0.593 ms | +| sst2 | 66,378 | 70.0% | 0.034 ms | 85.0% | 0.432 ms | + +The larger news mixture retains 28.25 MB of class text across 20 dictionaries; +this is not a tiny parameter-free model. It is competitive here, but the lexical +control is more accurate on every measured dataset. More data does not resolve +the sentiment weakness. + +Reproduce the larger run by adding `--datasets agnews sst2 --families zstd +zstd-mixture tfidf-logistic --train-cap 120000` and choosing a new output directory. + +On Benchmark Heaven’s public cohorts, zero-shot gzip scores **45.8% easy**, +**36.1% original**, and **42.3% hard**. All 231 labels were checked against the +upstream scorer and were invariant to reversing the option order. No official +score or leaderboard submission is claimed. The audit fixed ordinal gold labels +being compared as integers against strings, and removed class-order bias when +gzip neighbor distances tie. + +The checked-in JSON combines the two generated `summary.json` files and the +audit record. Raw item predictions stay under `eval/data/`. Across the default +and larger runs, 13,000 probability vectors and accuracies were independently +checked; the default run’s 8,000 dictionary/lexical predictions reproduced exactly. diff --git a/eval/compression-results.json b/eval/compression-results.json new file mode 100644 index 0000000..3f9563e --- /dev/null +++ b/eval/compression-results.json @@ -0,0 +1,2655 @@ +{ + "default_run": { + "manifest": { + "args": { + "datasets": [ + "agnews", + "banking77", + "sst2", + "emotion" + ], + "families": null, + "seed": 0, + "n": 500, + "train_cap": 10000, + "validation_n": 600, + "out": "eval/data/compression-verified", + "jevbench": "/tmp/classifier-gzip-jevbench" + }, + "platform": "macOS-26.5.1-arm64-arm-64bit", + "processor": "arm", + "python": "3.12.14", + "zlib": "1.2.12", + "thread_environment": { + "OMP_NUM_THREADS": "1", + "OPENBLAS_NUM_THREADS": "1" + }, + "timing": "Serial local warm inference: normalization, compression/features, probabilities, tie-breaking. Excludes training, network and data loading.", + "versions": { + "datasets": "5.0.1", + "numpy": "2.5.3", + "scipy": "1.18.1", + "scikit-learn": "1.9.1", + "zstandard": "0.25.0" + }, + "script_sha256": "5342902063b3071509c854905aa18c2410c8202e74c1fcad2ab76ae89491a2f4" + }, + "datasets": { + "agnews": { + "dataset": { + "repo": "fancyzhx/ag_news", + "hf_fingerprints": { + "train": "2bc1e2be15d7f2ee", + "test": "0ebcc57b6aea2fc9" + }, + "train_n": 10000, + "validation_n": 600, + "test_n": 500, + "removed_duplicates_or_eval_overlaps": 161, + "hashes": { + "train": "04944cfb62e7665ceda0faf53c118d18743495868c573e315806570d7fa2c616", + "validation": "b8edd80e646b60c5ca1a76e2744cd41f336090666c93886b305374e69ad44ef6", + "test": "76c1f71538ffa2609b9dbb9bbf842e60d19b79470d1f4d0cc554d88d95ae47ed" + }, + "labels": [ + "World", + "Sports", + "Business", + "Sci/Tech" + ] + }, + "validation_sweep": [ + { + "family": "deflate", + "config": { + "codec": "deflate", + "size": 4096 + }, + "validation_accuracy": 0.5933333333333334, + "fit_s": 0.004283166999812238, + "dictionary_or_sample_bytes": 16384 + }, + { + "family": "deflate", + "config": { + "codec": "deflate", + "size": 32768 + }, + "validation_accuracy": 0.785, + "fit_s": 0.004138415999477729, + "dictionary_or_sample_bytes": 131072 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 1024, + "trained": true + }, + "validation_accuracy": 0.62, + "fit_s": 0.13306408299831674, + "dictionary_or_sample_bytes": 4096 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 1024, + "trained": false + }, + "validation_accuracy": 0.395, + "fit_s": 0.003096458996878937, + "dictionary_or_sample_bytes": 4096 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 4096, + "trained": true + }, + "validation_accuracy": 0.7116666666666667, + "fit_s": 0.148906624999654, + "dictionary_or_sample_bytes": 16384 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 4096, + "trained": false + }, + "validation_accuracy": 0.5316666666666666, + "fit_s": 0.0029460840014507994, + "dictionary_or_sample_bytes": 16384 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 16384, + "trained": true + }, + "validation_accuracy": 0.7816666666666666, + "fit_s": 0.12514475000352832, + "dictionary_or_sample_bytes": 65536 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 16384, + "trained": false + }, + "validation_accuracy": 0.6383333333333333, + "fit_s": 0.0035843750010826625, + "dictionary_or_sample_bytes": 65536 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 65536, + "level": 9 + }, + "validation_accuracy": 0.755, + "fit_s": 0.14723604200116824, + "dictionary_or_sample_bytes": 262144 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 262144, + "level": 9 + }, + "validation_accuracy": 0.7933333333333333, + "fit_s": 0.2483245409966912, + "dictionary_or_sample_bytes": 1048576 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 16777216, + "level": 9, + "trained": false + }, + "validation_accuracy": 0.8466666666666667, + "fit_s": 0.003961707996495534, + "dictionary_or_sample_bytes": 2366771 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 1024, + "shards": 4 + }, + "validation_accuracy": 0.7016666666666667, + "fit_s": 0.11391575000016019, + "dictionary_or_sample_bytes": 16384 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 1024, + "shards": 8 + }, + "validation_accuracy": 0.7266666666666667, + "fit_s": 0.12054720900050597, + "dictionary_or_sample_bytes": 32768 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 4096, + "shards": 4 + }, + "validation_accuracy": 0.7783333333333333, + "fit_s": 0.17654266599856783, + "dictionary_or_sample_bytes": 65536 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 4096, + "shards": 8 + }, + "validation_accuracy": 0.7983333333333333, + "fit_s": 0.174204084003577, + "dictionary_or_sample_bytes": 131072 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 16777216, + "shards": 3, + "trained": false, + "level": 9, + "reduction": "min" + }, + "validation_accuracy": 0.8633333333333333, + "fit_s": 0.0033049999983632006, + "dictionary_or_sample_bytes": 2366763 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 16777216, + "shards": 3, + "trained": false, + "level": 9, + "reduction": "mean" + }, + "validation_accuracy": 0.895, + "fit_s": 0.0030992079991847277, + "dictionary_or_sample_bytes": 2366763 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 16777216, + "shards": 5, + "trained": false, + "level": 9, + "reduction": "min" + }, + "validation_accuracy": 0.8766666666666667, + "fit_s": 0.003387500000826549, + "dictionary_or_sample_bytes": 2366755 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 16777216, + "shards": 5, + "trained": false, + "level": 9, + "reduction": "mean" + }, + "validation_accuracy": 0.8833333333333333, + "fit_s": 0.003104374998656567, + "dictionary_or_sample_bytes": 2366755 + }, + { + "family": "gzip-knn", + "config": { + "k": 1 + }, + "validation_accuracy": 0.6116666666666667, + "fit_s": 0.010663459004717879, + "dictionary_or_sample_bytes": 232003 + }, + { + "family": "gzip-knn", + "config": { + "k": 5 + }, + "validation_accuracy": 0.6383333333333333, + "fit_s": 0.012955791004060302, + "dictionary_or_sample_bytes": 232003 + }, + { + "family": "tfidf-logistic", + "config": {}, + "validation_accuracy": 0.9033333333333333, + "fit_s": 5.491404250002233, + "dictionary_or_sample_bytes": null + } + ], + "selected": { + "deflate": { + "family": "deflate", + "config": { + "codec": "deflate", + "size": 32768 + }, + "validation_accuracy": 0.785, + "fit_s": 0.004138415999477729, + "dictionary_or_sample_bytes": 131072, + "temperature": 4.29444227551508, + "accuracy": 0.766, + "macro_f1": 0.771227501914879, + "ece": 0.06008734299381312, + "nll": 0.6778785936410218, + "brier": 0.349022246408619, + "tie_rate": 0.03, + "p50_ms": 0.0734795, + "p95_ms": 0.09922049999999999, + "serial_items_per_s": 13096.827675922877, + "n": 500, + "accuracy_wilson_95": [ + 0.7269477983375824, + 0.8009959046039771 + ], + "confidence_at_least_90pct": { + "n": 180, + "coverage": 0.36, + "accuracy": 0.9222222222222223 + } + }, + "zstd": { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 16777216, + "level": 9, + "trained": false + }, + "validation_accuracy": 0.8466666666666667, + "fit_s": 0.003961707996495534, + "dictionary_or_sample_bytes": 2366771, + "temperature": 4.637143191628681, + "accuracy": 0.808, + "macro_f1": 0.8072402391715388, + "ece": 0.057153539576628706, + "nll": 0.6757015422951654, + "brier": 0.2950965013185466, + "tie_rate": 0.022, + "p50_ms": 0.044521, + "p95_ms": 0.0784274, + "serial_items_per_s": 20026.02101066041, + "n": 500, + "accuracy_wilson_95": [ + 0.7711789076945584, + 0.8401243272904052 + ], + "confidence_at_least_90pct": { + "n": 263, + "coverage": 0.526, + "accuracy": 0.9125475285171103 + } + }, + "zstd-mixture": { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 16777216, + "shards": 3, + "trained": false, + "level": 9, + "reduction": "mean" + }, + "validation_accuracy": 0.895, + "fit_s": 0.0030992079991847277, + "dictionary_or_sample_bytes": 2366763, + "temperature": 3.410951603860984, + "accuracy": 0.848, + "macro_f1": 0.8480560127251228, + "ece": 0.04504076255490379, + "nll": 0.5399901438401714, + "brier": 0.22715867018883865, + "tie_rate": 0.006, + "p50_ms": 0.12854149999999998, + "p95_ms": 0.19065174999999993, + "serial_items_per_s": 7451.103215557868, + "n": 500, + "accuracy_wilson_95": [ + 0.8138851771600609, + 0.8768080883424302 + ], + "confidence_at_least_90pct": { + "n": 302, + "coverage": 0.604, + "accuracy": 0.9370860927152318 + } + }, + "gzip-knn": { + "family": "gzip-knn", + "config": { + "k": 5 + }, + "validation_accuracy": 0.6383333333333333, + "fit_s": 0.012955791004060302, + "dictionary_or_sample_bytes": 232003, + "temperature": 2.1518556436402267, + "accuracy": 0.618, + "macro_f1": 0.6199502440461163, + "ece": 0.10602590208369551, + "nll": 0.9974734324691097, + "brier": 0.5320566795008369, + "tie_rate": 0.2, + "p50_ms": 9.704708, + "p95_ms": 11.13455415, + "serial_items_per_s": 102.52232695558666, + "n": 500, + "accuracy_wilson_95": [ + 0.5746644739023564, + 0.6595361161243504 + ], + "confidence_at_least_90pct": { + "n": 0, + "coverage": 0.0, + "accuracy": null + } + }, + "tfidf-logistic": { + "family": "tfidf-logistic", + "config": {}, + "validation_accuracy": 0.9033333333333333, + "fit_s": 5.491404250002233, + "dictionary_or_sample_bytes": null, + "temperature": 0.7345152395906567, + "accuracy": 0.888, + "macro_f1": 0.8879638858343402, + "ece": 0.06050312885515615, + "nll": 0.44172785940001424, + "brier": 0.19518180715294506, + "tie_rate": 0.0, + "p50_ms": 0.5366455, + "p95_ms": 0.70514435, + "serial_items_per_s": 1814.9127050219765, + "n": 500, + "accuracy_wilson_95": [ + 0.8573456933034787, + 0.9127376027165404 + ], + "confidence_at_least_90pct": { + "n": 368, + "coverage": 0.736, + "accuracy": 0.9266304347826086 + } + } + } + }, + "banking77": { + "dataset": { + "repo": "legacy-datasets/banking77", + "hf_fingerprints": { + "train": "8360b37e52e4eb2a", + "test": "1d7c14b4c23350b9" + }, + "train_n": 9392, + "validation_n": 600, + "test_n": 500, + "removed_duplicates_or_eval_overlaps": 11, + "hashes": { + "train": "3d092fadae6b94818c39469faa6c8997e81f07219855d5dd0a510fb78fd79fc6", + "validation": "4e12f6da49e8667d0a3464d75bd7c82ec2338a890264835660468ccad307fddd", + "test": "709b7b5509738c4bce13a37aa816b4934b835be78e91cf0c79792ba5b1716dae" + }, + "labels": [ + "activate_my_card", + "age_limit", + "apple_pay_or_google_pay", + "atm_support", + "automatic_top_up", + "balance_not_updated_after_bank_transfer", + "balance_not_updated_after_cheque_or_cash_deposit", + "beneficiary_not_allowed", + "cancel_transfer", + "card_about_to_expire", + "card_acceptance", + "card_arrival", + "card_delivery_estimate", + "card_linking", + "card_not_working", + "card_payment_fee_charged", + "card_payment_not_recognised", + "card_payment_wrong_exchange_rate", + "card_swallowed", + "cash_withdrawal_charge", + "cash_withdrawal_not_recognised", + "change_pin", + "compromised_card", + "contactless_not_working", + "country_support", + "declined_card_payment", + "declined_cash_withdrawal", + "declined_transfer", + "direct_debit_payment_not_recognised", + "disposable_card_limits", + "edit_personal_details", + "exchange_charge", + "exchange_rate", + "exchange_via_app", + "extra_charge_on_statement", + "failed_transfer", + "fiat_currency_support", + "get_disposable_virtual_card", + "get_physical_card", + "getting_spare_card", + "getting_virtual_card", + "lost_or_stolen_card", + "lost_or_stolen_phone", + "order_physical_card", + "passcode_forgotten", + "pending_card_payment", + "pending_cash_withdrawal", + "pending_top_up", + "pending_transfer", + "pin_blocked", + "receiving_money", + "Refund_not_showing_up", + "request_refund", + "reverted_card_payment?", + "supported_cards_and_currencies", + "terminate_account", + "top_up_by_bank_transfer_charge", + "top_up_by_card_charge", + "top_up_by_cash_or_cheque", + "top_up_failed", + "top_up_limits", + "top_up_reverted", + "topping_up_by_card", + "transaction_charged_twice", + "transfer_fee_charged", + "transfer_into_account", + "transfer_not_received_by_recipient", + "transfer_timing", + "unable_to_verify_identity", + "verify_my_identity", + "verify_source_of_funds", + "verify_top_up", + "virtual_card_not_working", + "visa_or_mastercard", + "why_verify_identity", + "wrong_amount_of_cash_received", + "wrong_exchange_rate_for_cash_withdrawal" + ] + }, + "validation_sweep": [ + { + "family": "deflate", + "config": { + "codec": "deflate", + "size": 4096 + }, + "validation_accuracy": 0.805, + "fit_s": 0.0074147080013062805, + "dictionary_or_sample_bytes": 304038 + }, + { + "family": "deflate", + "config": { + "codec": "deflate", + "size": 32768 + }, + "validation_accuracy": 0.7983333333333333, + "fit_s": 0.008205541002098471, + "dictionary_or_sample_bytes": 567780 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 1024, + "trained": true + }, + "validation_accuracy": 0.6466666666666666, + "fit_s": 0.05034095799783245, + "dictionary_or_sample_bytes": 78848 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 1024, + "trained": false + }, + "validation_accuracy": 0.6416666666666667, + "fit_s": 0.007061042000714224, + "dictionary_or_sample_bytes": 78848 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 4096, + "trained": true + }, + "validation_accuracy": 0.6933333333333334, + "fit_s": 0.08712499999819556, + "dictionary_or_sample_bytes": 286121 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 4096, + "trained": false + }, + "validation_accuracy": 0.7333333333333333, + "fit_s": 0.007319832999201026, + "dictionary_or_sample_bytes": 304038 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 16384, + "trained": true + }, + "validation_accuracy": 0.7133333333333334, + "fit_s": 0.12062429200159386, + "dictionary_or_sample_bytes": 411372 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 16384, + "trained": false + }, + "validation_accuracy": 0.7083333333333334, + "fit_s": 0.007289250002941117, + "dictionary_or_sample_bytes": 567780 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 65536, + "level": 9 + }, + "validation_accuracy": 0.7483333333333333, + "fit_s": 0.11870433299918659, + "dictionary_or_sample_bytes": 411372 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 262144, + "level": 9 + }, + "validation_accuracy": 0.7483333333333333, + "fit_s": 0.12351499999931548, + "dictionary_or_sample_bytes": 411372 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 16777216, + "level": 9, + "trained": false + }, + "validation_accuracy": 0.7816666666666666, + "fit_s": 0.007592082998598926, + "dictionary_or_sample_bytes": 567780 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 1024, + "shards": 4 + }, + "validation_accuracy": 0.72, + "fit_s": 0.12457012500090059, + "dictionary_or_sample_bytes": 281325 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 1024, + "shards": 8 + }, + "training_error": "cannot train dict: Src size is incorrect" + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 4096, + "shards": 4 + }, + "validation_accuracy": 0.7366666666666667, + "fit_s": 0.20517695900343824, + "dictionary_or_sample_bytes": 454504 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 4096, + "shards": 8 + }, + "training_error": "cannot train dict: Src size is incorrect" + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 16777216, + "shards": 3, + "trained": false, + "level": 9, + "reduction": "min" + }, + "validation_accuracy": 0.8166666666666667, + "fit_s": 0.007201125001301989, + "dictionary_or_sample_bytes": 567626 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 16777216, + "shards": 3, + "trained": false, + "level": 9, + "reduction": "mean" + }, + "validation_accuracy": 0.835, + "fit_s": 0.007468750001862645, + "dictionary_or_sample_bytes": 567626 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 16777216, + "shards": 5, + "trained": false, + "level": 9, + "reduction": "min" + }, + "validation_accuracy": 0.8183333333333334, + "fit_s": 0.0075589999978546984, + "dictionary_or_sample_bytes": 567472 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 16777216, + "shards": 5, + "trained": false, + "level": 9, + "reduction": "mean" + }, + "validation_accuracy": 0.825, + "fit_s": 0.00782808299845783, + "dictionary_or_sample_bytes": 567472 + }, + { + "family": "gzip-knn", + "config": { + "k": 1 + }, + "validation_accuracy": 0.49833333333333335, + "fit_s": 0.011530167001183145, + "dictionary_or_sample_bytes": 52739 + }, + { + "family": "gzip-knn", + "config": { + "k": 5 + }, + "validation_accuracy": 0.45166666666666666, + "fit_s": 0.011287084002105985, + "dictionary_or_sample_bytes": 52739 + }, + { + "family": "tfidf-logistic", + "config": {}, + "validation_accuracy": 0.9016666666666666, + "fit_s": 6.198137958999723, + "dictionary_or_sample_bytes": null + } + ], + "selected": { + "deflate": { + "family": "deflate", + "config": { + "codec": "deflate", + "size": 4096 + }, + "validation_accuracy": 0.805, + "fit_s": 0.0074147080013062805, + "dictionary_or_sample_bytes": 304038, + "temperature": 1.35753398137484, + "accuracy": 0.82, + "macro_f1": 0.7992486209415565, + "ece": 0.06290305365662752, + "nll": 0.748491242227106, + "brier": 0.28054942628044477, + "tie_rate": 0.094, + "p50_ms": 0.5353749999999999, + "p95_ms": 0.71514165, + "serial_items_per_s": 1798.0028453251189, + "n": 500, + "accuracy_wilson_95": [ + 0.7839246244600844, + 0.8511956196801373 + ], + "confidence_at_least_90pct": { + "n": 244, + "coverage": 0.488, + "accuracy": 0.9590163934426229 + } + }, + "zstd": { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 16777216, + "level": 9, + "trained": false + }, + "validation_accuracy": 0.7816666666666666, + "fit_s": 0.007592082998598926, + "dictionary_or_sample_bytes": 567780, + "temperature": 1.5828442546700814, + "accuracy": 0.778, + "macro_f1": 0.7634533737731454, + "ece": 0.03987457002093967, + "nll": 0.8437418486929151, + "brier": 0.31676958909518704, + "tie_rate": 0.1, + "p50_ms": 0.14937499999999998, + "p95_ms": 0.3112707999999996, + "serial_items_per_s": 5918.135240090225, + "n": 500, + "accuracy_wilson_95": [ + 0.7395294756647782, + 0.8122312364320395 + ], + "confidence_at_least_90pct": { + "n": 234, + "coverage": 0.468, + "accuracy": 0.9700854700854701 + } + }, + "zstd-mixture": { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 16777216, + "shards": 3, + "trained": false, + "level": 9, + "reduction": "mean" + }, + "validation_accuracy": 0.835, + "fit_s": 0.007468750001862645, + "dictionary_or_sample_bytes": 567626, + "temperature": 1.35753398137484, + "accuracy": 0.806, + "macro_f1": 0.7948050126018442, + "ece": 0.027330381293454824, + "nll": 0.7245106932888512, + "brier": 0.27127663578583444, + "tie_rate": 0.026, + "p50_ms": 0.5831459999999999, + "p95_ms": 1.1346207999999984, + "serial_items_per_s": 1549.8975985756417, + "n": 500, + "accuracy_wilson_95": [ + 0.7690596512561212, + 0.838274082202966 + ], + "confidence_at_least_90pct": { + "n": 273, + "coverage": 0.546, + "accuracy": 0.9743589743589743 + } + }, + "gzip-knn": { + "family": "gzip-knn", + "config": { + "k": 1 + }, + "validation_accuracy": 0.49833333333333335, + "fit_s": 0.011530167001183145, + "dictionary_or_sample_bytes": 52739, + "temperature": 1.0782500762595328, + "accuracy": 0.548, + "macro_f1": 0.5242677003523842, + "ece": 0.06063191239609045, + "nll": 2.6533837260533817, + "brier": 0.7007323897060809, + "tie_rate": 0.0, + "p50_ms": 4.4909795, + "p95_ms": 5.387016699999999, + "serial_items_per_s": 217.7996094879138, + "n": 500, + "accuracy_wilson_95": [ + 0.5041745952330335, + 0.5910934413879999 + ], + "confidence_at_least_90pct": { + "n": 0, + "coverage": 0.0, + "accuracy": null + } + }, + "tfidf-logistic": { + "family": "tfidf-logistic", + "config": {}, + "validation_accuracy": 0.9016666666666666, + "fit_s": 6.198137958999723, + "dictionary_or_sample_bytes": null, + "temperature": 0.5402886714960652, + "accuracy": 0.908, + "macro_f1": 0.8941185048731366, + "ece": 0.02614868859476964, + "nll": 0.3168123349805611, + "brier": 0.1376886364442629, + "tie_rate": 0.0, + "p50_ms": 0.383479, + "p95_ms": 0.52263335, + "serial_items_per_s": 2489.7499360395686, + "n": 500, + "accuracy_wilson_95": [ + 0.8794606777802376, + 0.9303176334985452 + ], + "confidence_at_least_90pct": { + "n": 392, + "coverage": 0.784, + "accuracy": 0.9897959183673469 + } + } + } + }, + "sst2": { + "dataset": { + "repo": "stanfordnlp/sst2", + "hf_fingerprints": { + "train": "e7e6610fdbb3c277", + "validation": "c1ddc6497ec97f98", + "test": "14a60218fd61ec38" + }, + "train_n": 10000, + "validation_n": 600, + "test_n": 500, + "removed_duplicates_or_eval_overlaps": 371, + "hashes": { + "train": "bb18ecae29bb9ea84b25405f6a0428096951a9a69efac09bb100c9facea1869d", + "validation": "fafee5f02c9c933efd0edf3c5c0b1621b8522669dfaca255450c809c6a815590", + "test": "380747f6d5a4f05ce8516c891e34d9dd29500beace1c0937b524634411bb287e" + }, + "labels": [ + "negative", + "positive" + ] + }, + "validation_sweep": [ + { + "family": "deflate", + "config": { + "codec": "deflate", + "size": 4096 + }, + "validation_accuracy": 0.5966666666666667, + "fit_s": 0.0017086660009226762, + "dictionary_or_sample_bytes": 8192 + }, + { + "family": "deflate", + "config": { + "codec": "deflate", + "size": 32768 + }, + "validation_accuracy": 0.6766666666666666, + "fit_s": 0.0017910410024342127, + "dictionary_or_sample_bytes": 65536 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 1024, + "trained": true + }, + "validation_accuracy": 0.5166666666666667, + "fit_s": 0.027700541002559476, + "dictionary_or_sample_bytes": 2048 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 1024, + "trained": false + }, + "validation_accuracy": 0.5066666666666667, + "fit_s": 0.0017736669979058206, + "dictionary_or_sample_bytes": 2048 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 4096, + "trained": true + }, + "validation_accuracy": 0.5833333333333334, + "fit_s": 0.04100783300236799, + "dictionary_or_sample_bytes": 8192 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 4096, + "trained": false + }, + "validation_accuracy": 0.5483333333333333, + "fit_s": 0.002048333000857383, + "dictionary_or_sample_bytes": 8192 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 16384, + "trained": true + }, + "validation_accuracy": 0.6283333333333333, + "fit_s": 0.038298833002045285, + "dictionary_or_sample_bytes": 32768 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 16384, + "trained": false + }, + "validation_accuracy": 0.5933333333333334, + "fit_s": 0.0018218339973827824, + "dictionary_or_sample_bytes": 32768 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 65536, + "level": 9 + }, + "validation_accuracy": 0.6433333333333333, + "fit_s": 0.046571332997700665, + "dictionary_or_sample_bytes": 131072 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 262144, + "level": 9 + }, + "validation_accuracy": 0.7066666666666667, + "fit_s": 0.08784708299936028, + "dictionary_or_sample_bytes": 437677 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 16777216, + "level": 9, + "trained": false + }, + "validation_accuracy": 0.7283333333333334, + "fit_s": 0.0018446670001139864, + "dictionary_or_sample_bytes": 536488 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 1024, + "shards": 4 + }, + "validation_accuracy": 0.5583333333333333, + "fit_s": 0.029137124998669606, + "dictionary_or_sample_bytes": 8192 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 1024, + "shards": 8 + }, + "validation_accuracy": 0.58, + "fit_s": 0.031216542003676295, + "dictionary_or_sample_bytes": 16384 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 4096, + "shards": 4 + }, + "validation_accuracy": 0.6216666666666667, + "fit_s": 0.0438428749985178, + "dictionary_or_sample_bytes": 32768 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 4096, + "shards": 8 + }, + "validation_accuracy": 0.6416666666666667, + "fit_s": 0.04907000000093831, + "dictionary_or_sample_bytes": 65536 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 16777216, + "shards": 3, + "trained": false, + "level": 9, + "reduction": "min" + }, + "validation_accuracy": 0.725, + "fit_s": 0.00196666600095341, + "dictionary_or_sample_bytes": 536484 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 16777216, + "shards": 3, + "trained": false, + "level": 9, + "reduction": "mean" + }, + "validation_accuracy": 0.7533333333333333, + "fit_s": 0.001909541002532933, + "dictionary_or_sample_bytes": 536484 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 16777216, + "shards": 5, + "trained": false, + "level": 9, + "reduction": "min" + }, + "validation_accuracy": 0.7366666666666667, + "fit_s": 0.0017700829994282685, + "dictionary_or_sample_bytes": 536480 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 16777216, + "shards": 5, + "trained": false, + "level": 9, + "reduction": "mean" + }, + "validation_accuracy": 0.7616666666666667, + "fit_s": 0.001830375003919471, + "dictionary_or_sample_bytes": 536480 + }, + { + "family": "gzip-knn", + "config": { + "k": 1 + }, + "validation_accuracy": 0.5733333333333334, + "fit_s": 0.005904542005737312, + "dictionary_or_sample_bytes": 52471 + }, + { + "family": "gzip-knn", + "config": { + "k": 5 + }, + "validation_accuracy": 0.595, + "fit_s": 0.006380707993230317, + "dictionary_or_sample_bytes": 52471 + }, + { + "family": "tfidf-logistic", + "config": {}, + "validation_accuracy": 0.84, + "fit_s": 0.5434237500012387, + "dictionary_or_sample_bytes": null + } + ], + "selected": { + "deflate": { + "family": "deflate", + "config": { + "codec": "deflate", + "size": 32768 + }, + "validation_accuracy": 0.6766666666666666, + "fit_s": 0.0017910410024342127, + "dictionary_or_sample_bytes": 65536, + "temperature": 6.304134293587772, + "accuracy": 0.656, + "macro_f1": 0.655332724153962, + "ece": 0.02910858412013078, + "nll": 0.6184185625087876, + "brier": 0.43000688230669276, + "tie_rate": 0.074, + "p50_ms": 0.0276875, + "p95_ms": 0.036502099999999996, + "serial_items_per_s": 35514.841972093716, + "n": 500, + "auroc": 0.7180215804303278, + "accuracy_wilson_95": [ + 0.6133133706304251, + 0.6963077483879331 + ], + "confidence_at_least_90pct": { + "n": 11, + "coverage": 0.022, + "accuracy": 0.9090909090909091 + } + }, + "zstd": { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 16777216, + "level": 9, + "trained": false + }, + "validation_accuracy": 0.7283333333333334, + "fit_s": 0.0018446670001139864, + "dictionary_or_sample_bytes": 536488, + "temperature": 10.790252743818396, + "accuracy": 0.65, + "macro_f1": 0.6499985999944, + "ece": 0.03593030036559715, + "nll": 0.618717418902004, + "brier": 0.43029277027337204, + "tie_rate": 0.086, + "p50_ms": 0.013625, + "p95_ms": 0.022675299999999992, + "serial_items_per_s": 69523.71113120501, + "n": 500, + "auroc": 0.7180696080942622, + "accuracy_wilson_95": [ + 0.6071920689703412, + 0.6905205454703878 + ], + "confidence_at_least_90pct": { + "n": 2, + "coverage": 0.004, + "accuracy": 1.0 + } + }, + "zstd-mixture": { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 16777216, + "shards": 5, + "trained": false, + "level": 9, + "reduction": "mean" + }, + "validation_accuracy": 0.7616666666666667, + "fit_s": 0.001830375003919471, + "dictionary_or_sample_bytes": 536480, + "temperature": 2.925419034377105, + "accuracy": 0.718, + "macro_f1": 0.7175922031413361, + "ece": 0.04554804951123206, + "nll": 0.5547712132787098, + "brier": 0.37369763190577826, + "tie_rate": 0.028, + "p50_ms": 0.041292, + "p95_ms": 0.06892435, + "serial_items_per_s": 22944.298676765586, + "n": 500, + "auroc": 0.7985319544057375, + "accuracy_wilson_95": [ + 0.6770114417638189, + 0.7556642245567071 + ], + "confidence_at_least_90pct": { + "n": 72, + "coverage": 0.144, + "accuracy": 0.9166666666666666 + } + }, + "gzip-knn": { + "family": "gzip-knn", + "config": { + "k": 5 + }, + "validation_accuracy": 0.595, + "fit_s": 0.006380707993230317, + "dictionary_or_sample_bytes": 52471, + "temperature": 5.007191993770399, + "accuracy": 0.548, + "macro_f1": 0.5474134478283856, + "ece": 0.013168762839443831, + "nll": 0.6913485232561188, + "brier": 0.4975115063432846, + "tie_rate": 0.0, + "p50_ms": 5.481375, + "p95_ms": 6.29362325, + "serial_items_per_s": 182.7036415967676, + "n": 500, + "auroc": 0.5551037397540983, + "accuracy_wilson_95": [ + 0.5041745952330335, + 0.5910934413879999 + ], + "confidence_at_least_90pct": { + "n": 0, + "coverage": 0.0, + "accuracy": null + } + }, + "tfidf-logistic": { + "family": "tfidf-logistic", + "config": {}, + "validation_accuracy": 0.84, + "fit_s": 0.5434237500012387, + "dictionary_or_sample_bytes": null, + "temperature": 0.7345152395906567, + "accuracy": 0.796, + "macro_f1": 0.7959706197692469, + "ece": 0.03953913908774128, + "nll": 0.41708729362573255, + "brier": 0.26258361413544473, + "tie_rate": 0.0, + "p50_ms": 0.412729, + "p95_ms": 0.5067076999999999, + "serial_items_per_s": 2391.932066718238, + "n": 500, + "auroc": 0.8961321721311475, + "accuracy_wilson_95": [ + 0.7584839354593685, + 0.8290022903703368 + ], + "confidence_at_least_90pct": { + "n": 243, + "coverage": 0.486, + "accuracy": 0.9382716049382716 + } + } + } + }, + "emotion": { + "dataset": { + "repo": "dair-ai/emotion", + "hf_fingerprints": { + "train": "27c5e4f1e47ffa42", + "validation": "5cae643edfd5aaf3", + "test": "bd45cf0a5e0b9690" + }, + "train_n": 10000, + "validation_n": 600, + "test_n": 500, + "removed_duplicates_or_eval_overlaps": 42, + "hashes": { + "train": "5bcc8e86df660889cb24bfd3a261f5528d8fb3b142f4693c73fd739338b9cb86", + "validation": "bda5d620b7ea488ba509f07b3591f772feb319b8a396d7812b26f89421143e7a", + "test": "62fe848e5c2808aed7e0077efd032834d274b6eb6e7d323e76be32ef97b1feea" + }, + "labels": [ + "sadness", + "joy", + "love", + "anger", + "fear", + "surprise" + ] + }, + "validation_sweep": [ + { + "family": "deflate", + "config": { + "codec": "deflate", + "size": 4096 + }, + "validation_accuracy": 0.32666666666666666, + "fit_s": 0.002004999994824175, + "dictionary_or_sample_bytes": 24576 + }, + { + "family": "deflate", + "config": { + "codec": "deflate", + "size": 32768 + }, + "validation_accuracy": 0.43833333333333335, + "fit_s": 0.00225687499914784, + "dictionary_or_sample_bytes": 196608 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 1024, + "trained": true + }, + "validation_accuracy": 0.33, + "fit_s": 0.04342029100371292, + "dictionary_or_sample_bytes": 6144 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 1024, + "trained": false + }, + "validation_accuracy": 0.235, + "fit_s": 0.0020858330026385374, + "dictionary_or_sample_bytes": 6144 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 4096, + "trained": true + }, + "validation_accuracy": 0.445, + "fit_s": 0.05890566699963529, + "dictionary_or_sample_bytes": 24576 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 4096, + "trained": false + }, + "validation_accuracy": 0.275, + "fit_s": 0.002086583997879643, + "dictionary_or_sample_bytes": 24576 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 16384, + "trained": true + }, + "validation_accuracy": 0.44666666666666666, + "fit_s": 0.058133709004323464, + "dictionary_or_sample_bytes": 98304 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 16384, + "trained": false + }, + "validation_accuracy": 0.37666666666666665, + "fit_s": 0.0023508340018452145, + "dictionary_or_sample_bytes": 98304 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 65536, + "level": 9 + }, + "validation_accuracy": 0.4033333333333333, + "fit_s": 0.11329087500052992, + "dictionary_or_sample_bytes": 349551 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 262144, + "level": 9 + }, + "validation_accuracy": 0.43666666666666665, + "fit_s": 0.15255850000539795, + "dictionary_or_sample_bytes": 856643 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 16777216, + "level": 9, + "trained": false + }, + "validation_accuracy": 0.44166666666666665, + "fit_s": 0.0022252499984460883, + "dictionary_or_sample_bytes": 979245 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 1024, + "shards": 4 + }, + "validation_accuracy": 0.3616666666666667, + "fit_s": 0.0483227920049103, + "dictionary_or_sample_bytes": 24576 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 1024, + "shards": 8 + }, + "validation_accuracy": 0.3883333333333333, + "fit_s": 0.0537706250033807, + "dictionary_or_sample_bytes": 49152 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 4096, + "shards": 4 + }, + "validation_accuracy": 0.515, + "fit_s": 0.07201379199977964, + "dictionary_or_sample_bytes": 98304 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 4096, + "shards": 8 + }, + "validation_accuracy": 0.48, + "fit_s": 0.08771720799995819, + "dictionary_or_sample_bytes": 195367 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 16777216, + "shards": 3, + "trained": false, + "level": 9, + "reduction": "min" + }, + "validation_accuracy": 0.5066666666666667, + "fit_s": 0.002359584002988413, + "dictionary_or_sample_bytes": 979233 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 16777216, + "shards": 3, + "trained": false, + "level": 9, + "reduction": "mean" + }, + "validation_accuracy": 0.535, + "fit_s": 0.0026667080019251443, + "dictionary_or_sample_bytes": 979233 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 16777216, + "shards": 5, + "trained": false, + "level": 9, + "reduction": "min" + }, + "validation_accuracy": 0.5033333333333333, + "fit_s": 0.002307417002157308, + "dictionary_or_sample_bytes": 979221 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 16777216, + "shards": 5, + "trained": false, + "level": 9, + "reduction": "mean" + }, + "validation_accuracy": 0.5533333333333333, + "fit_s": 0.00223658300092211, + "dictionary_or_sample_bytes": 979221 + }, + { + "family": "gzip-knn", + "config": { + "k": 1 + }, + "validation_accuracy": 0.25833333333333336, + "fit_s": 0.007107000004907604, + "dictionary_or_sample_bytes": 98188 + }, + { + "family": "gzip-knn", + "config": { + "k": 5 + }, + "validation_accuracy": 0.24166666666666667, + "fit_s": 0.006843499999376945, + "dictionary_or_sample_bytes": 98188 + }, + { + "family": "tfidf-logistic", + "config": {}, + "validation_accuracy": 0.8416666666666667, + "fit_s": 3.909456957997463, + "dictionary_or_sample_bytes": null + } + ], + "selected": { + "deflate": { + "family": "deflate", + "config": { + "codec": "deflate", + "size": 32768 + }, + "validation_accuracy": 0.43833333333333335, + "fit_s": 0.00225687499914784, + "dictionary_or_sample_bytes": 196608, + "temperature": 3.410951603860984, + "accuracy": 0.406, + "macro_f1": 0.3684907264569877, + "ece": 0.03192085479609739, + "nll": 1.486202601279427, + "brier": 0.7179354684289194, + "tie_rate": 0.168, + "p50_ms": 0.0647915, + "p95_ms": 0.10161844999999997, + "serial_items_per_s": 14590.575182943197, + "n": 500, + "accuracy_wilson_95": [ + 0.3638296860034575, + 0.44960374228035244 + ], + "confidence_at_least_90pct": { + "n": 1, + "coverage": 0.002, + "accuracy": 1.0 + } + }, + "zstd": { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 16384, + "trained": true + }, + "validation_accuracy": 0.44666666666666666, + "fit_s": 0.058133709004323464, + "dictionary_or_sample_bytes": 98304, + "temperature": 2.709220452261488, + "accuracy": 0.454, + "macro_f1": 0.4018783545119495, + "ece": 0.05846382416841619, + "nll": 1.463362818214318, + "brier": 0.698117185244948, + "tie_rate": 0.214, + "p50_ms": 0.0115, + "p95_ms": 0.017131249999999997, + "serial_items_per_s": 81165.0627909159, + "n": 500, + "accuracy_wilson_95": [ + 0.4108749466266584, + 0.49782651827818475 + ], + "confidence_at_least_90pct": { + "n": 2, + "coverage": 0.004, + "accuracy": 0.5 + } + }, + "zstd-mixture": { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 16777216, + "shards": 5, + "trained": false, + "level": 9, + "reduction": "mean" + }, + "validation_accuracy": 0.5533333333333333, + "fit_s": 0.00223658300092211, + "dictionary_or_sample_bytes": 979221, + "temperature": 2.925419034377105, + "accuracy": 0.562, + "macro_f1": 0.46353991429511865, + "ece": 0.08457185931239918, + "nll": 1.2537522096435618, + "brier": 0.597552891264581, + "tie_rate": 0.038, + "p50_ms": 0.103604, + "p95_ms": 0.2117881499999997, + "serial_items_per_s": 8727.203975946568, + "n": 500, + "accuracy_wilson_95": [ + 0.5182021184950548, + 0.6048524288071132 + ], + "confidence_at_least_90pct": { + "n": 22, + "coverage": 0.044, + "accuracy": 0.8181818181818182 + } + }, + "gzip-knn": { + "family": "gzip-knn", + "config": { + "k": 1 + }, + "validation_accuracy": 0.25833333333333336, + "fit_s": 0.007107000004907604, + "dictionary_or_sample_bytes": 98188, + "temperature": 8.570386453309192, + "accuracy": 0.298, + "macro_f1": 0.2781904296547765, + "ece": 0.0427759626994248, + "nll": 1.743637872594331, + "brier": 0.814830939581835, + "tie_rate": 0.0, + "p50_ms": 5.6842915000000005, + "p95_ms": 6.971593449999999, + "serial_items_per_s": 171.43746654479273, + "n": 500, + "accuracy_wilson_95": [ + 0.25957253815100145, + 0.3395078077354835 + ], + "confidence_at_least_90pct": { + "n": 0, + "coverage": 0.0, + "accuracy": null + } + }, + "tfidf-logistic": { + "family": "tfidf-logistic", + "config": {}, + "validation_accuracy": 0.8416666666666667, + "fit_s": 3.909456957997463, + "dictionary_or_sample_bytes": null, + "temperature": 0.5834042638846706, + "accuracy": 0.852, + "macro_f1": 0.7837710838613873, + "ece": 0.030185522477487146, + "nll": 0.4129957238398832, + "brier": 0.22078743602615492, + "tie_rate": 0.0, + "p50_ms": 0.426187, + "p95_ms": 0.5887381499999998, + "serial_items_per_s": 2240.786993074041, + "n": 500, + "accuracy_wilson_95": [ + 0.818193200072725, + 0.8804390684815191 + ], + "confidence_at_least_90pct": { + "n": 322, + "coverage": 0.644, + "accuracy": 0.9472049689440993 + } + } + } + } + }, + "jevbench": { + "easy": { + "n": 48, + "sha256": "231df3c2c8e88a1a8c137ebe85de96ba70fabd330849098ac7b3c52c70b7172b", + "accuracy": 0.4583333333333333, + "uniform_chance": 0.284375, + "p50_ms": 0.074271, + "families": { + "extraction": { + "n": 12, + "accuracy": 0.4166666666666667 + }, + "fact": { + "n": 12, + "accuracy": 0.5833333333333334 + }, + "intent": { + "n": 12, + "accuracy": 0.4166666666666667 + }, + "tool_selection": { + "n": 12, + "accuracy": 0.4166666666666667 + } + } + }, + "original": { + "n": 72, + "sha256": "5c2414edb3006b8bfcb70fda433f0f9ca015759433849f8d3104328a1f7c4180", + "accuracy": 0.3611111111111111, + "uniform_chance": 0.3111111111111111, + "p50_ms": 0.0727085, + "families": { + "adequacy": { + "n": 12, + "accuracy": 0.4166666666666667 + }, + "extraction": { + "n": 12, + "accuracy": 0.3333333333333333 + }, + "intent": { + "n": 12, + "accuracy": 0.16666666666666666 + }, + "ordinal": { + "n": 12, + "accuracy": 0.4166666666666667 + }, + "policy": { + "n": 12, + "accuracy": 0.5 + }, + "routing": { + "n": 12, + "accuracy": 0.3333333333333333 + } + } + }, + "hard": { + "n": 111, + "sha256": "89e9e6becb33ed88c1de7d42dcc87531b2fb64cfaef4e1986faf7c37b3f80ebb", + "accuracy": 0.42342342342342343, + "uniform_chance": 0.3361861861861862, + "p50_ms": 0.183583, + "families": { + "adversarial": { + "n": 6, + "accuracy": 0.6666666666666666 + }, + "ambiguous": { + "n": 7, + "accuracy": 0.42857142857142855 + }, + "judge_hard": { + "n": 17, + "accuracy": 0.4117647058823529 + }, + "long_policy": { + "n": 19, + "accuracy": 0.21052631578947367 + }, + "multi_hop": { + "n": 18, + "accuracy": 0.2222222222222222 + }, + "probability": { + "n": 10, + "accuracy": 0.5 + }, + "routing_hard": { + "n": 5, + "accuracy": 1.0 + }, + "temporal_numeric": { + "n": 15, + "accuracy": 0.3333333333333333 + }, + "tradeoff": { + "n": 6, + "accuracy": 1.0 + }, + "trap": { + "n": 8, + "accuracy": 0.5 + } + } + }, + "source_commit": "2fa63fa3226cb369795525ed011800f57dcbd894", + "method": "Zero-shot symmetric gzip NCD against label and rubric; no training or calibration. Public diagnostic, not an official score." + } + }, + "large_training_run": { + "manifest": { + "args": { + "datasets": [ + "agnews", + "sst2" + ], + "families": [ + "zstd", + "zstd-mixture", + "tfidf-logistic" + ], + "seed": 0, + "n": 500, + "train_cap": 120000, + "validation_n": 600, + "out": "eval/data/compression-large", + "jevbench": null + }, + "platform": "macOS-26.5.1-arm64-arm-64bit", + "processor": "arm", + "python": "3.12.14", + "zlib": "1.2.12", + "thread_environment": { + "OMP_NUM_THREADS": "1", + "OPENBLAS_NUM_THREADS": "1" + }, + "timing": "Serial local warm inference: normalization, compression/features, probabilities, tie-breaking. Excludes training, network and data loading.", + "versions": { + "datasets": "5.0.1", + "numpy": "2.5.3", + "scipy": "1.18.1", + "scikit-learn": "1.9.1", + "zstandard": "0.25.0" + }, + "script_sha256": "5342902063b3071509c854905aa18c2410c8202e74c1fcad2ab76ae89491a2f4" + }, + "datasets": { + "agnews": { + "dataset": { + "repo": "fancyzhx/ag_news", + "hf_fingerprints": { + "train": "2bc1e2be15d7f2ee", + "test": "0ebcc57b6aea2fc9" + }, + "train_n": 119239, + "validation_n": 600, + "test_n": 500, + "removed_duplicates_or_eval_overlaps": 161, + "hashes": { + "train": "d1fb37dc340dc5ed27d4162ef9a98b4e83a80b5dc958e3e4e97101aa0f9e62ed", + "validation": "89cea8648589dcdb7aacc5e04f4efcd1b6058e5946f346d672c492fe8b03d403", + "test": "76c1f71538ffa2609b9dbb9bbf842e60d19b79470d1f4d0cc554d88d95ae47ed" + }, + "labels": [ + "World", + "Sports", + "Business", + "Sci/Tech" + ] + }, + "validation_sweep": [ + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 1024, + "trained": true + }, + "validation_accuracy": 0.655, + "fit_s": 1.095241957998951, + "dictionary_or_sample_bytes": 4096 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 1024, + "trained": false + }, + "validation_accuracy": 0.38333333333333336, + "fit_s": 0.05557016700186068, + "dictionary_or_sample_bytes": 4096 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 4096, + "trained": true + }, + "validation_accuracy": 0.7733333333333333, + "fit_s": 1.6625714999972843, + "dictionary_or_sample_bytes": 16384 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 4096, + "trained": false + }, + "validation_accuracy": 0.5033333333333333, + "fit_s": 0.0551289579962031, + "dictionary_or_sample_bytes": 16384 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 16384, + "trained": true + }, + "validation_accuracy": 0.805, + "fit_s": 1.376491791997978, + "dictionary_or_sample_bytes": 65536 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 16384, + "trained": false + }, + "validation_accuracy": 0.6483333333333333, + "fit_s": 0.05820737500471296, + "dictionary_or_sample_bytes": 65536 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 65536, + "level": 9 + }, + "validation_accuracy": 0.795, + "fit_s": 1.1389680420033983, + "dictionary_or_sample_bytes": 262144 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 262144, + "level": 9 + }, + "validation_accuracy": 0.7983333333333333, + "fit_s": 1.0690593339968473, + "dictionary_or_sample_bytes": 1048576 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 16777216, + "level": 9, + "trained": false + }, + "validation_accuracy": 0.855, + "fit_s": 0.05941937500028871, + "dictionary_or_sample_bytes": 28247565 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 1024, + "shards": 4 + }, + "validation_accuracy": 0.7283333333333334, + "fit_s": 1.1204033750036615, + "dictionary_or_sample_bytes": 16384 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 1024, + "shards": 8 + }, + "validation_accuracy": 0.73, + "fit_s": 1.1286926250031684, + "dictionary_or_sample_bytes": 32768 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 4096, + "shards": 4 + }, + "validation_accuracy": 0.8066666666666666, + "fit_s": 1.700393332997919, + "dictionary_or_sample_bytes": 65536 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 4096, + "shards": 8 + }, + "validation_accuracy": 0.8166666666666667, + "fit_s": 1.7476613329999964, + "dictionary_or_sample_bytes": 131072 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 16777216, + "shards": 3, + "trained": false, + "level": 9, + "reduction": "min" + }, + "validation_accuracy": 0.8766666666666667, + "fit_s": 0.0598740830027964, + "dictionary_or_sample_bytes": 28247557 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 16777216, + "shards": 3, + "trained": false, + "level": 9, + "reduction": "mean" + }, + "validation_accuracy": 0.8983333333333333, + "fit_s": 0.057473291002679616, + "dictionary_or_sample_bytes": 28247557 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 16777216, + "shards": 5, + "trained": false, + "level": 9, + "reduction": "min" + }, + "validation_accuracy": 0.8966666666666666, + "fit_s": 0.058699958004581276, + "dictionary_or_sample_bytes": 28247549 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 16777216, + "shards": 5, + "trained": false, + "level": 9, + "reduction": "mean" + }, + "validation_accuracy": 0.9116666666666666, + "fit_s": 0.060205792004126124, + "dictionary_or_sample_bytes": 28247549 + }, + { + "family": "tfidf-logistic", + "config": {}, + "validation_accuracy": 0.9366666666666666, + "fit_s": 72.26334416699683, + "dictionary_or_sample_bytes": null + } + ], + "selected": { + "zstd": { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 16777216, + "level": 9, + "trained": false + }, + "validation_accuracy": 0.855, + "fit_s": 0.05941937500028871, + "dictionary_or_sample_bytes": 28247565, + "temperature": 5.838236970939662, + "accuracy": 0.856, + "macro_f1": 0.8578350680660315, + "ece": 0.07148579683576123, + "nll": 0.573052244855724, + "brier": 0.2274947566483206, + "tie_rate": 0.02, + "p50_ms": 0.0481665, + "p95_ms": 0.09512054999999997, + "serial_items_per_s": 16460.801843188412, + "n": 500, + "accuracy_wilson_95": [ + 0.822508879125479, + 0.8840623924805179 + ], + "confidence_at_least_90pct": { + "n": 246, + "coverage": 0.492, + "accuracy": 0.9471544715447154 + } + }, + "zstd-mixture": { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 16777216, + "shards": 5, + "trained": false, + "level": 9, + "reduction": "mean" + }, + "validation_accuracy": 0.9116666666666666, + "fit_s": 0.060205792004126124, + "dictionary_or_sample_bytes": 28247549, + "temperature": 3.683149054535093, + "accuracy": 0.9, + "macro_f1": 0.8997427715689791, + "ece": 0.05852817346189611, + "nll": 0.4590178473955788, + "brier": 0.17604696824424826, + "tie_rate": 0.004, + "p50_ms": 0.24593700000000002, + "p95_ms": 0.4126940499999994, + "serial_items_per_s": 3567.4229415418463, + "n": 500, + "accuracy_wilson_95": [ + 0.8705774917999974, + 0.9233228133752799 + ], + "confidence_at_least_90pct": { + "n": 360, + "coverage": 0.72, + "accuracy": 0.9472222222222222 + } + }, + "tfidf-logistic": { + "family": "tfidf-logistic", + "config": {}, + "validation_accuracy": 0.9366666666666666, + "fit_s": 72.26334416699683, + "dictionary_or_sample_bytes": null, + "temperature": 0.9247663594340342, + "accuracy": 0.912, + "macro_f1": 0.9127427169277376, + "ece": 0.02759829499776148, + "nll": 0.28636336613317387, + "brier": 0.14006859532256022, + "tie_rate": 0.0, + "p50_ms": 0.593167, + "p95_ms": 0.7779229499999999, + "serial_items_per_s": 1645.5464238809047, + "n": 500, + "accuracy_wilson_95": [ + 0.8839229514479334, + 0.9337943628826022 + ], + "confidence_at_least_90pct": { + "n": 397, + "coverage": 0.794, + "accuracy": 0.9697732997481109 + } + } + } + }, + "sst2": { + "dataset": { + "repo": "stanfordnlp/sst2", + "hf_fingerprints": { + "train": "e7e6610fdbb3c277", + "validation": "c1ddc6497ec97f98", + "test": "14a60218fd61ec38" + }, + "train_n": 66378, + "validation_n": 600, + "test_n": 500, + "removed_duplicates_or_eval_overlaps": 371, + "hashes": { + "train": "488b7c60c5677461e842fc5ecd32774c1ddbdd5878c8e6c793c7430eeb5720e1", + "validation": "745798b437dae118d5c5f3498399b935f25a45040fb07d6423a8d18964f91fd7", + "test": "380747f6d5a4f05ce8516c891e34d9dd29500beace1c0937b524634411bb287e" + }, + "labels": [ + "negative", + "positive" + ] + }, + "validation_sweep": [ + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 1024, + "trained": true + }, + "validation_accuracy": 0.57, + "fit_s": 0.1833836249934393, + "dictionary_or_sample_bytes": 2048 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 1024, + "trained": false + }, + "validation_accuracy": 0.54, + "fit_s": 0.01816158300061943, + "dictionary_or_sample_bytes": 2048 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 4096, + "trained": true + }, + "validation_accuracy": 0.61, + "fit_s": 0.2721342500008177, + "dictionary_or_sample_bytes": 8192 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 4096, + "trained": false + }, + "validation_accuracy": 0.56, + "fit_s": 0.019221125003241468, + "dictionary_or_sample_bytes": 8192 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 16384, + "trained": true + }, + "validation_accuracy": 0.6816666666666666, + "fit_s": 0.2559416250005597, + "dictionary_or_sample_bytes": 32768 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 16384, + "trained": false + }, + "validation_accuracy": 0.6333333333333333, + "fit_s": 0.018674541999644134, + "dictionary_or_sample_bytes": 32768 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 65536, + "level": 9 + }, + "validation_accuracy": 0.6616666666666666, + "fit_s": 0.2445404169993708, + "dictionary_or_sample_bytes": 131072 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 262144, + "level": 9 + }, + "validation_accuracy": 0.7466666666666667, + "fit_s": 0.2638120420015184, + "dictionary_or_sample_bytes": 524288 + }, + { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 16777216, + "level": 9, + "trained": false + }, + "validation_accuracy": 0.7816666666666666, + "fit_s": 0.018079166002280544, + "dictionary_or_sample_bytes": 3567328 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 1024, + "shards": 4 + }, + "validation_accuracy": 0.6133333333333333, + "fit_s": 0.1946094590020948, + "dictionary_or_sample_bytes": 8192 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 1024, + "shards": 8 + }, + "validation_accuracy": 0.5866666666666667, + "fit_s": 0.19280641600198578, + "dictionary_or_sample_bytes": 16384 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 4096, + "shards": 4 + }, + "validation_accuracy": 0.6583333333333333, + "fit_s": 0.2784915840020403, + "dictionary_or_sample_bytes": 32768 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 4096, + "shards": 8 + }, + "validation_accuracy": 0.66, + "fit_s": 0.28512387500086334, + "dictionary_or_sample_bytes": 65536 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 16777216, + "shards": 3, + "trained": false, + "level": 9, + "reduction": "min" + }, + "validation_accuracy": 0.7916666666666666, + "fit_s": 0.01775899999483954, + "dictionary_or_sample_bytes": 3567324 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 16777216, + "shards": 3, + "trained": false, + "level": 9, + "reduction": "mean" + }, + "validation_accuracy": 0.79, + "fit_s": 0.0159552499972051, + "dictionary_or_sample_bytes": 3567324 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 16777216, + "shards": 5, + "trained": false, + "level": 9, + "reduction": "min" + }, + "validation_accuracy": 0.7883333333333333, + "fit_s": 0.017087957996409386, + "dictionary_or_sample_bytes": 3567320 + }, + { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 16777216, + "shards": 5, + "trained": false, + "level": 9, + "reduction": "mean" + }, + "validation_accuracy": 0.79, + "fit_s": 0.016865915997186676, + "dictionary_or_sample_bytes": 3567320 + }, + { + "family": "tfidf-logistic", + "config": {}, + "validation_accuracy": 0.915, + "fit_s": 3.577201125001011, + "dictionary_or_sample_bytes": null + } + ], + "selected": { + "zstd": { + "family": "zstd", + "config": { + "codec": "zstd", + "size": 16777216, + "level": 9, + "trained": false + }, + "validation_accuracy": 0.7816666666666666, + "fit_s": 0.018079166002280544, + "dictionary_or_sample_bytes": 3567328, + "temperature": 8.570386453309192, + "accuracy": 0.662, + "macro_f1": 0.6608591301137025, + "ece": 0.04250272619027978, + "nll": 0.6126490882855867, + "brier": 0.4222055784678397, + "tie_rate": 0.09, + "p50_ms": 0.015396, + "p95_ms": 0.025002049999999998, + "serial_items_per_s": 62596.52383976092, + "n": 500, + "auroc": 0.7419633709016394, + "accuracy_wilson_95": [ + 0.6194419387868491, + 0.7020876848091384 + ], + "confidence_at_least_90pct": { + "n": 6, + "coverage": 0.012, + "accuracy": 0.6666666666666666 + } + }, + "zstd-mixture": { + "family": "zstd-mixture", + "config": { + "codec": "zstd", + "size": 16777216, + "shards": 3, + "trained": false, + "level": 9, + "reduction": "min" + }, + "validation_accuracy": 0.7916666666666666, + "fit_s": 0.01775899999483954, + "dictionary_or_sample_bytes": 3567324, + "temperature": 7.3504331266672, + "accuracy": 0.7, + "macro_f1": 0.6996106954613179, + "ece": 0.059913157093607276, + "nll": 0.6007650774839849, + "brier": 0.4103150298882698, + "tie_rate": 0.084, + "p50_ms": 0.034, + "p95_ms": 0.06059134999999999, + "serial_items_per_s": 28267.394171399006, + "n": 500, + "auroc": 0.7595174820696722, + "accuracy_wilson_95": [ + 0.6584314090816449, + 0.7385187435059938 + ], + "confidence_at_least_90pct": { + "n": 13, + "coverage": 0.026, + "accuracy": 0.8461538461538461 + } + }, + "tfidf-logistic": { + "family": "tfidf-logistic", + "config": {}, + "validation_accuracy": 0.915, + "fit_s": 3.577201125001011, + "dictionary_or_sample_bytes": null, + "temperature": 0.7345152395906567, + "accuracy": 0.85, + "macro_f1": 0.8499513842484965, + "ece": 0.038211722573768335, + "nll": 0.3873007503417493, + "brier": 0.22026970652824607, + "tie_rate": 0.0, + "p50_ms": 0.43174999999999997, + "p95_ms": 0.5471291999999999, + "serial_items_per_s": 2263.669087689429, + "n": 500, + "auroc": 0.9229956454918032, + "accuracy_wilson_95": [ + 0.8160382472597502, + 0.8786245197686174 + ], + "confidence_at_least_90pct": { + "n": 320, + "coverage": 0.64, + "accuracy": 0.940625 + } + } + } + } + } + }, + "verification": { + "predictions_validated": 10000, + "dictionary_and_lexical_predictions_reproduce_exactly": true, + "script_hash_matches": true, + "jevbench": { + "official_label_scoring_matches": true, + "option_reversal_invariant": true, + "results": [ + { + "tier": "easy", + "n": 48, + "correct": 22, + "accuracy": 0.4583333333333333 + }, + { + "tier": "hard", + "n": 111, + "correct": 47, + "accuracy": 0.42342342342342343 + }, + { + "tier": "original", + "n": 72, + "correct": 26, + "accuracy": 0.3611111111111111 + } + ] + }, + "large_training_predictions_validated": 3000, + "large_training_uses_identical_test_sample": true + } +} diff --git a/eval/compression.py b/eval/compression.py new file mode 100644 index 0000000..35cad0c --- /dev/null +++ b/eval/compression.py @@ -0,0 +1,371 @@ +#!/usr/bin/env python3 +# /// script +# requires-python = ">=3.12" +# dependencies = ["datasets==5.0.1", "numpy==2.5.3", "scipy==1.18.1", "scikit-learn==1.9.1", "zstandard==0.25.0"] +# /// +"""Compression classification experiment. Run with: uv run eval/compression.py. + +All tuning uses held-back training rows. Each family gets one final test run. +Raw predictions, validation sweeps, split hashes and environment are saved. +""" +import argparse +import gzip +import hashlib +import importlib.metadata +import json +import os +import platform +import random +import subprocess +import time +import zlib +from pathlib import Path + +import numpy as np +import zstandard as zstd +from datasets import load_dataset +from scipy.special import softmax +from sklearn.feature_extraction.text import TfidfVectorizer +from sklearn.pipeline import FeatureUnion +from sklearn.linear_model import LogisticRegression +from sklearn.metrics import accuracy_score, f1_score, log_loss, roc_auc_score +from sklearn.model_selection import train_test_split + +DATASETS = { + "agnews": ("fancyzhx/ag_news", None, "text", "test"), + "banking77": ("legacy-datasets/banking77", None, "text", "test"), + "sst2": ("stanfordnlp/sst2", None, "sentence", "validation"), + "emotion": ("dair-ai/emotion", "split", "text", "test"), +} + + +def normalize(text): + return " ".join(text.lower().split()) + + +def digest(value): + return hashlib.sha256(json.dumps(value, ensure_ascii=False, sort_keys=True).encode()).hexdigest() + + +def load_split(name, seed, n, train_cap, val_n): + repo, config, field, test = DATASETS[name] + ds = load_dataset(repo, config) + labels = ds["train"].features["label"].names + heldout = [(normalize(r[field]), int(r["label"])) for r in ds[test]] + rng = random.Random(seed) + rng.shuffle(heldout) + evaluation = heldout[:n] + seen = {x for x, _ in heldout} + pool = [] + duplicates = 0 + for r in ds["train"]: + x, y = normalize(r[field]), int(r["label"]) + if x in seen: + duplicates += 1 + continue + seen.add(x) + pool.append((x, y)) + rng.shuffle(pool) + pool = pool[:train_cap + val_n] + train, val = train_test_split(pool, test_size=val_n, random_state=seed, + stratify=[r[1] for r in pool]) + assert not ({r[0] for r in train} & {r[0] for r in val}) + assert not ({r[0] for r in train + val} & {r[0] for r in heldout}) + assert set(y for _, y in train) == set(range(len(labels))) + meta = {"repo": repo, "hf_fingerprints": {k: v._fingerprint for k, v in ds.items()}, + "train_n": len(train), "validation_n": len(val), "test_n": len(evaluation), + "removed_duplicates_or_eval_overlaps": duplicates, + "hashes": {k: digest(v) for k, v in [("train", train), ("validation", val), ("test", evaluation)]}, + "labels": labels} + return train, val, evaluation, meta + + +class Dictionaries: + def __init__(self, rows, classes, codec, size, shards=1, seed=42, trained=True, level=3, reduction="min"): + self.codec, self.models, self.bytes = codec, [], 0 + self.reduction = reduction + rng = random.Random(seed) + for label in range(classes): + texts = [x.encode() for x, y in rows if y == label] + rng.shuffle(texts) + models = [] + for i in range(shards): + samples = texts[i::shards] + if not samples: + raise ValueError("empty class shard") + if codec == "zstd": + if trained: + # Do not silently change the algorithm if training fails. + dictionary = zstd.train_dictionary(size, samples, threads=0) + else: + dictionary = zstd.ZstdCompressionDict(b"\n".join(samples)[-size:], dict_type=zstd.DICT_TYPE_RAWCONTENT) + self.bytes += len(dictionary.as_bytes()) + models.append(zstd.ZstdCompressor(level=level, dict_data=dictionary)) + else: + dictionary = b"\n".join(samples)[-size:] + self.bytes += len(dictionary) + models.append(zlib.compressobj(level=6, wbits=-15, zdict=dictionary)) + self.models.append(models) + + def score(self, text): + x = text.encode() + scores = [] + for models in self.models: + lengths = [] + for model in models: + if self.codec == "zstd": + lengths.append(len(model.compress(x))) + else: + c = model.copy() + lengths.append(len(c.compress(x) + c.flush())) + scores.append(-float(np.mean(lengths)) if self.reduction == "mean" else -min(lengths)) + return np.asarray(scores, dtype=float) + + +class GzipKNN: + def __init__(self, rows, classes, k=5, cap=1000, seed=42): + rng = random.Random(seed) + rows = list(rows) + rng.shuffle(rows) + # Balanced memory avoids a majority-class prior from subsampling. + self.samples = [] + for label in range(classes): + self.samples += [(x.encode(), y) for x, y in rows if y == label][:max(1, cap // classes)] + self.lengths = [len(gzip.compress(x, mtime=0)) for x, _ in self.samples] + self.tie_keys = [hashlib.sha256(x).hexdigest() for x, _ in self.samples] + self.classes, self.k = classes, k + self.bytes = sum(len(x) for x, _ in self.samples) + + def score(self, text): + x = text.encode() + cx = len(gzip.compress(x, mtime=0)) + distances = [(len(gzip.compress(z + b" " + x, mtime=0)) - min(cx, cz)) / max(cx, cz) + for (z, _), cz in zip(self.samples, self.lengths)] + nearest = np.lexsort((self.tie_keys, distances))[:self.k] + votes = np.full(self.classes, 0.01) + for i in nearest: + votes[self.samples[i][1]] += 1 + return np.log(votes) + + +class Lexical: + def __init__(self, rows, classes): + self.vectorizer = FeatureUnion([ + ("word", TfidfVectorizer(ngram_range=(1, 2), sublinear_tf=True, min_df=2)), + ("char", TfidfVectorizer(analyzer="char", ngram_range=(3, 5), sublinear_tf=True, min_df=2, max_features=100000)), + ]) + x = self.vectorizer.fit_transform([x for x, _ in rows]) + self.classifier = LogisticRegression(C=4, max_iter=500).fit(x, [y for _, y in rows]) + self.bytes = None + + def score(self, text): + return self.classifier.predict_log_proba(self.vectorizer.transform([text]))[0] + + +def scores_for(model, rows, temp=1): + scores, times = [], [] + for text, _ in rows: + start = time.perf_counter_ns() + score = model.score(normalize(text)) + softmax(score / temp) + predictions([score], [text]) + scores.append(score) + times.append((time.perf_counter_ns() - start) / 1e6) + return np.asarray(scores), times + + +def predictions(scores, texts): + # Exact ties use a reproducible text hash instead of always selecting class 0. + out = [] + for score, text in zip(scores, texts): + ties = np.flatnonzero(score == score.max()) + index = int(hashlib.sha256(text.encode()).hexdigest(), 16) % len(ties) + out.append(int(ties[index])) + return np.asarray(out) + + +def temperature(scores, gold): + choices = np.geomspace(.05, 100, 100) + losses = [log_loss(gold, softmax(scores / t, axis=1), labels=np.arange(scores.shape[1])) for t in choices] + return float(choices[np.argmin(losses)]) + + +def metrics(scores, rows, times, temp): + gold = np.asarray([y for _, y in rows]) + pred = predictions(scores, [x for x, _ in rows]) + prob = softmax(scores / temp, axis=1) + conf = prob.max(axis=1) + correct = pred == gold + ece = 0. + for i in range(10): + mask = (conf >= i / 10) & ((conf < (i + 1) / 10) if i < 9 else (conf <= 1)) + if mask.any(): + ece += mask.mean() * abs(correct[mask].mean() - conf[mask].mean()) + result = {"accuracy": float(correct.mean()), "macro_f1": float(f1_score(gold, pred, labels=np.arange(scores.shape[1]), average="macro", zero_division=0)), + "ece": float(ece), "nll": float(log_loss(gold, prob, labels=np.arange(scores.shape[1]))), + "brier": float(np.mean(np.sum((prob - np.eye(scores.shape[1])[gold]) ** 2, axis=1))), + "tie_rate": float(np.mean((scores == scores.max(axis=1, keepdims=True)).sum(axis=1) > 1)), + "p50_ms": float(np.median(times)), "p95_ms": float(np.percentile(times, 95)), + "serial_items_per_s": float(1000 / np.mean(times)), "n": len(rows)} + if scores.shape[1] == 2: + result["auroc"] = float(roc_auc_score(gold, prob[:, 1])) + n, p, z = len(rows), float(correct.mean()), 1.96 + center = (p + z * z / (2 * n)) / (1 + z * z / n) + radius = z * np.sqrt(p * (1 - p) / n + z * z / (4 * n * n)) / (1 + z * z / n) + result["accuracy_wilson_95"] = [float(center - radius), float(center + radius)] + mask = conf >= .9 + result["confidence_at_least_90pct"] = { + "n": int(mask.sum()), "coverage": float(mask.mean()), + "accuracy": float(correct[mask].mean()) if mask.any() else None, + } + return result, [{"text_sha256": hashlib.sha256(x.encode()).hexdigest(), "gold": int(y), + "prediction": int(p), "probabilities": ps.tolist(), "ms": ms} + for (x, y), p, ps, ms in zip(rows, pred, prob, times)] + + +def run_dataset(name, args, outdir): + train, val, test, meta = load_split(name, args.seed, args.n, args.train_cap, args.validation_n) + classes = len(meta["labels"]) + configs = [("deflate", {"codec": "deflate", "size": s}) for s in [4096, 32768]] + configs += [("zstd", {"codec": "zstd", "size": s, "trained": trained}) + for s in [1024, 4096, 16384] for trained in [True, False]] + configs += [("zstd", {"codec": "zstd", "size": s, "level": 9}) + for s in [65536, 262144]] + configs += [("zstd", {"codec": "zstd", "size": 16777216, "level": 9, "trained": False})] + configs += [("zstd-mixture", {"codec": "zstd", "size": s, "shards": shards}) + for s in [1024, 4096] for shards in [4, 8]] + configs += [("zstd-mixture", {"codec": "zstd", "size": 16777216, "shards": shards, + "trained": False, "level": 9, "reduction": reduction}) + for shards in [3, 5] for reduction in ["min", "mean"]] + configs += [("gzip-knn", {"k": k}) for k in [1, 5]] + configs += [("tfidf-logistic", {})] + if args.families: + configs = [(family, config) for family, config in configs if family in args.families] + best, sweep = {}, [] + for family, config in configs: + start = time.perf_counter() + try: + if family == "gzip-knn": + model = GzipKNN(train, classes, seed=args.seed, **config) + elif family == "tfidf-logistic": + model = Lexical(train, classes) + else: + model = Dictionaries(train, classes, seed=args.seed, **config) + except zstd.ZstdError as exc: + sweep.append({"family": family, "config": config, "training_error": str(exc)}) + continue + fit_s = time.perf_counter() - start + scores, times = scores_for(model, val) + pred = predictions(scores, [x for x, _ in val]) + acc = float(accuracy_score([y for _, y in val], pred)) + entry = {"family": family, "config": config, "validation_accuracy": acc, + "fit_s": fit_s, "dictionary_or_sample_bytes": model.bytes} + sweep.append(entry) + print(name, family, config, f"validation={acc:.3f}", flush=True) + if family not in best or acc > best[family][0]["validation_accuracy"]: + best[family] = entry, model, scores + report = {"dataset": meta, "validation_sweep": sweep, "selected": {}} + for family, (entry, model, scores) in best.items(): + temp = temperature(scores, [y for _, y in val]) + test_scores, times = scores_for(model, test, temp) + summary, rows = metrics(test_scores, test, times, temp) + # OOD-like inputs must yield a finite score for every allowed class. + for text in ["", "🙂 café 你好", "x" * 50000]: + edge = model.score(text) + assert edge.shape == (classes,) and np.isfinite(edge).all() + report["selected"][family] = {**entry, "temperature": temp, **summary} + (outdir / f"{name}-{family}-predictions.json").write_text(json.dumps(rows)) + print(name, family, "TEST", json.dumps(summary), flush=True) + (outdir / f"{name}.json").write_text(json.dumps(report, indent=2)) + return report + + +def predict_jevbench(state, question, labels): + if not isinstance(state, str): + state = json.dumps(state, ensure_ascii=False, sort_keys=True) + x = normalize(state + "\n" + question["instructions"]).encode() + cx = len(gzip.compress(x, mtime=0)) + criteria = question.get("criteria") or {} + if isinstance(criteria, list): + criteria = {str(i): description for i, description in enumerate(criteria)} + scores = [] + for label in labels: + key = {"yes": "true", "no": "false"}.get(label, label) if question["type"] == "noul" else label + description = criteria.get(key) + if description is None: + description = "" + z = normalize(label.replace("_", " ") + ": " + str(description)).encode() + cz = len(gzip.compress(z, mtime=0)) + # Symmetric NCD: both concat directions, normalized for label length. + joint = min(len(gzip.compress(x + b" " + z, mtime=0)), len(gzip.compress(z + b" " + x, mtime=0))) + scores.append(-(joint - min(cx, cz)) / max(cx, cz)) + scores = np.asarray(scores) + ties = np.flatnonzero(scores == scores.max()) + # Choose by label name hash so option permutation does not change ties. + index = min(ties, key=lambda i: hashlib.sha256(x + labels[i].encode()).hexdigest()) + return labels[index], len(ties) > 1 + + +def jevbench(root, outdir): + results = {} + for tier in ["easy", "original", "hard"]: + path = root / "datasets" / "public" / f"{tier}.jsonl" + rows = [json.loads(line) for line in path.read_text().splitlines() if line.strip()] + predictions_out = [] + if any(r["expected"] is None or r.get("provenance", {}).get("exclude_reason") for r in rows): + raise ValueError("This diagnostic requires fully scored public cohorts") + for row in rows: + start = time.perf_counter_ns() + prediction, tie = predict_jevbench(row["state"], row["question"], row["labels"]) + predictions_out.append({"id": row["id"], "group": row.get("group"), "family": row["family"], + "gold": str(row["expected"]), "prediction": prediction, + "tie": tie, "chance": 1 / len(row["labels"]), + "ms": (time.perf_counter_ns() - start) / 1e6}) + family_scores = {} + for family in sorted({r["family"] for r in predictions_out}): + group = [r for r in predictions_out if r["family"] == family] + family_scores[family] = {"n": len(group), "accuracy": np.mean([r["gold"] == r["prediction"] for r in group])} + results[tier] = {"n": len(rows), "sha256": hashlib.sha256(path.read_bytes()).hexdigest(), + "accuracy": float(np.mean([r["gold"] == r["prediction"] for r in predictions_out])), + "uniform_chance": float(np.mean([r["chance"] for r in predictions_out])), + "p50_ms": float(np.median([r["ms"] for r in predictions_out])), "families": family_scores} + (outdir / f"jevbench-{tier}-predictions.json").write_text(json.dumps(predictions_out, indent=2)) + results["source_commit"] = subprocess.check_output(["git", "-C", str(root), "rev-parse", "HEAD"], text=True).strip() + results["method"] = "Zero-shot symmetric gzip NCD against label and rubric; no training or calibration. Public diagnostic, not an official score." + (outdir / "jevbench.json").write_text(json.dumps(results, indent=2)) + print("JEVBENCH", json.dumps(results), flush=True) + return results + + +def main(): + ap = argparse.ArgumentParser(description=__doc__) + ap.add_argument("--datasets", nargs="+", choices=DATASETS, default=list(DATASETS)) + ap.add_argument("--families", nargs="+", choices=["deflate", "zstd", "zstd-mixture", "gzip-knn", "tfidf-logistic"]) + ap.add_argument("--seed", type=int, default=0) + ap.add_argument("--n", type=int, default=500) + ap.add_argument("--train-cap", type=int, default=10000) + ap.add_argument("--validation-n", type=int, default=600) + ap.add_argument("--out", type=Path, default=Path(__file__).parent / "data" / "compression") + ap.add_argument("--jevbench", type=Path) + args = ap.parse_args() + if min(args.n, args.train_cap, args.validation_n) <= 0: + ap.error("sample counts must be positive") + args.out.mkdir(parents=True, exist_ok=False) + metadata = {"args": {k: str(v) if isinstance(v, Path) else v for k, v in vars(args).items()}, + "platform": platform.platform(), "processor": platform.processor(), + "python": platform.python_version(), "zlib": zlib.ZLIB_RUNTIME_VERSION, + "thread_environment": {k: os.environ.get(k) for k in ["OMP_NUM_THREADS", "OPENBLAS_NUM_THREADS"]}, + "timing": "Serial local warm inference: normalization, compression/features, probabilities, tie-breaking. Excludes training, network and data loading.", + "versions": {k: importlib.metadata.version(k) for k in ["datasets", "numpy", "scipy", "scikit-learn", "zstandard"]}, + "script_sha256": hashlib.sha256(Path(__file__).read_bytes()).hexdigest()} + (args.out / "manifest.json").write_text(json.dumps(metadata, indent=2)) + summary = {"manifest": metadata, "datasets": {}} + for name in args.datasets: + summary["datasets"][name] = run_dataset(name, args, args.out) + if args.jevbench: + summary["jevbench"] = jevbench(args.jevbench, args.out) + (args.out / "summary.json").write_text(json.dumps(summary, indent=2)) + + +if __name__ == "__main__": + main() From 9005d6e48a2baa87317b115be37ea12a3bd39f09 Mon Sep 17 00:00:00 2001 From: Michael Ryaboy Date: Thu, 24 Sep 2026 02:38:37 -0700 Subject: [PATCH 2/2] Focus compression study on zero-shot JevBench performance --- eval/README.md | 149 +- eval/compression-results.json | 2655 -------------------------- eval/compression.py | 371 ---- eval/jevbench-zero-shot-results.json | 2093 ++++++++++++++++++++ eval/jevbench_hybrid.py | 198 ++ 5 files changed, 2348 insertions(+), 3118 deletions(-) delete mode 100644 eval/compression-results.json delete mode 100644 eval/compression.py create mode 100644 eval/jevbench-zero-shot-results.json create mode 100644 eval/jevbench_hybrid.py diff --git a/eval/README.md b/eval/README.md index 24d5ea8..aa72838 100644 --- a/eval/README.md +++ b/eval/README.md @@ -120,103 +120,68 @@ In rough order of value: add cases from a real taxonomy with labels assigned before any model output is seen; split into tune and report halves; raise n past about thirty so confidence intervals mean something. -## Compression classifiers +## Zero-shot TF-IDF and compression on JevBench -`compression.py` tests whether gzip/DEFLATE or Zstd can replace a learned text -classifier. It compares gzip NCD nearest neighbors, per-class DEFLATE and Zstd -dictionaries, Zstd dictionary mixtures, and a word/character TF-IDF logistic -regression control. It does not change the Worker or call a paid model API. +`jevbench_hybrid.py` measures arbitrary typed decisions using only the current +state, instructions, labels, and option descriptions. There is no labeled +training, support set, learned combination weight, or probability calibration. +TF-IDF vocabulary and IDF are computed from the current request alone; compressor +dictionaries are seeded with that request’s text at inference. ```sh git clone https://github.com/fstandhartinger/jevbench.git /tmp/compression-jevbench git -C /tmp/compression-jevbench checkout 2fa63fa3226cb369795525ed011800f57dcbd894 -OMP_NUM_THREADS=1 OPENBLAS_NUM_THREADS=1 uv run --python 3.12 eval/compression.py \ - --out eval/data/compression-reproduction --jevbench /tmp/compression-jevbench +OMP_NUM_THREADS=1 OPENBLAS_NUM_THREADS=1 uv run --python 3.12 eval/jevbench_hybrid.py \ + --jevbench /tmp/compression-jevbench --out eval/data/jevbench-zero-shot-reproduction ``` -Run from the repository root. The output directory must be new. `uv` installs -the script's pinned dependencies; Hugging Face downloads are cached. Omit -`--jevbench` to run only the supervised datasets. `--datasets`, `--families`, -`--seed`, `--train-cap`, `--validation-n`, and `--n` bound additional experiments. - -The default uses seed 0 and 500 evaluation rows, matching the sampling protocol -of [dhruvmehra/jevbench](https://github.com/dhruvmehra/jevbench). Each dataset has -up to 10,000 training rows and 600 separate validation rows. Training is shuffled -and deduplicated after excluding exact normalized matches anywhere in the -held-out split. Classifier settings and probability temperature are chosen only -on validation rows, then frozen for evaluation. This excludes exact duplicates, -not semantic paraphrases or SST-2's related review fragments. Test rows remain -in their original benchmark sample, including any duplicates within that split. - -Gzip nearest neighbors deliberately retain at most 1,000 balanced examples; its -accuracy is a bounded-compute baseline, not an estimate of full-data gzip kNN. -DEFLATE uses raw zlib preset dictionaries (the same compression algorithm as -gzip, without gzip framing). Zstd tries trained dictionaries and raw class -text, with minimum or mean length over random shards. Compression length is -only a score: softmax temperature is fitted on validation labels. Dictionaries -store information from labeled examples; this is supervised learning, not a -zero-shot or memory-free model. Oversized dictionary training failures are -recorded in the validation sweep, never silently replaced by another method. - -Every run produces `summary.json`, per-dataset validation sweeps and selected -settings, per-item predictions/probabilities, split hashes, dataset fingerprints, -and software versions. Timing covers serial local normalization, scoring, -probabilities and tie-breaking; it excludes training, loading, serving and -network overhead. Empty, Unicode and long inputs are exercised for every -selected supervised model. The output manifest identifies the script by hash. - -The optional Benchmark Heaven diagnostic uses symmetric gzip NCD between the -state/instructions and each label/criterion. The prediction function accepts -only those inference fields; expected answers and author rationales cannot -enter it. This has no labeled support set and returns labels without invented -calibrated probabilities. Its three public cohorts total 231 decisions. These -are public diagnostic accuracies, not an official JevBench score: private, -sealed, and imported tasks are absent. A supervised banking/news classifier -cannot be submitted as if it solves arbitrary typed decisions. - -The compression idea comes from [Nathan Barry's gzip language-model experiment](https://nathan.rs/posts/gzip-lm/) -and [FTCC's class-dictionary approach](https://github.com/cyrilou242/ftcc). -The TF-IDF control matters: [Gzip versus bag-of-words for text classification](https://arxiv.org/abs/2307.15002) -examines whether compression's gains survive comparison with conventional text -features. - -### Measured results (Apple M5 Max) - -Generated from [compression-results.json](compression-results.json). All rows -use 500 held-out items per dataset; percentages are accuracy. Each family selects -its settings on validation data. These are exploratory measurements on one seed, -not evidence that small differences are statistically significant. - -| Up to 10k training rows | Gzip kNN (≤1k memory) | DEFLATE | Zstd mixture | TF-IDF + logistic | +Run from the repository root. The output directory must be new. Dependencies +are pinned in the script; no API keys, model downloads, or paid calls are needed. +The benchmark repository supplies the 231 public questions and published Jev +outcomes for comparison. Private, sealed, and imported questions are absent. + +The fixed sweep contains 54 methods: five TF-IDF scores (word, window, character, +label-only, and a window/character combination), three compression scores +(gzip normalized compression distance, DEFLATE conditional gain, and Zstd +conditional gain), all 45 pairwise blends at 25/50/75% TF-IDF weight, and one +equal four-way blend. Scores are standardized within each question before +combining them. Negligible numerical variation is collapsed and final scores +rounded before ties are resolved by label name, matching JevBench’s scorer. + +### Public results + +Generated measurements are in [jevbench-zero-shot-results.json](jevbench-zero-shot-results.json). +No training examples were used for any row. Jev is a published reference on +the exact same item IDs, not a fresh API run. + +| Method | Overall (231) | Easy (48) | Original (72) | Hard (111) | |---|---:|---:|---:|---:| -| agnews | 61.8% | 76.6% | 84.8% | 88.8% | -| banking77 | 54.8% | 82.0% | 80.6% | 90.8% | -| sst2 | 54.8% | 65.6% | 71.8% | 79.6% | -| emotion | 29.8% | 40.6% | 56.2% | 85.2% | - -Larger-training follow-up (same evaluation items, separately held-back validation): - -| Dataset | Training rows | Zstd mixture | p50 | TF-IDF + logistic | p50 | -|---|---:|---:|---:|---:|---:| -| agnews | 119,239 | 90.0% | 0.246 ms | 91.2% | 0.593 ms | -| sst2 | 66,378 | 70.0% | 0.034 ms | 85.0% | 0.432 ms | - -The larger news mixture retains 28.25 MB of class text across 20 dictionaries; -this is not a tiny parameter-free model. It is competitive here, but the lexical -control is more accurate on every measured dataset. More data does not resolve -the sentiment weakness. - -Reproduce the larger run by adding `--datasets agnews sst2 --families zstd -zstd-mixture tfidf-logistic --train-cap 120000` and choosing a new output directory. - -On Benchmark Heaven’s public cohorts, zero-shot gzip scores **45.8% easy**, -**36.1% original**, and **42.3% hard**. All 231 labels were checked against the -upstream scorer and were invariant to reversing the option order. No official -score or leaderboard submission is claimed. The audit fixed ordinal gold labels -being compared as integers against strings, and removed class-order bias when -gzip neighbor distances tie. - -The checked-in JSON combines the two generated `summary.json` files and the -audit record. Raw item predictions stay under `eval/data/`. Across the default -and larger runs, 13,000 probability vectors and accuracies were independently -checked; the default run’s 8,000 dictionary/lexical predictions reproduced exactly. +| Gzip | 42.0% | 52.1% | 34.7% | 42.3% | +| Character TF-IDF | 49.4% | 79.2% | 38.9% | 43.2% | +| Window + character TF-IDF | 48.9% | 79.2% | 40.3% | 41.4% | +| 75% TF-IDF + 25% Zstd (best observed blend) | 50.6% | 81.2% | 41.7% | 43.2% | +| Jev 1.13.0, published | 86.6% | 100.0% | 98.6% | 73.0% | + +The best blend is the high end of an exploratory sweep, not an independently +selected winner. It adds only three correct answers over character TF-IDF +(+1.30 percentage points; paired scenario-bootstrap 95% interval −3.07 to +5.73 +points). This is not convincing evidence of an improvement. Uniform random +choice averages 31.8% on these variable-sized label sets. + +The methods are fast but miss many reasoning decisions: the highlighted blend +gets 5/19 long-policy questions and 5/18 multi-hop questions correct. Running +all 54 scoring rules together took a median 1.53 ms per request on an Apple M5 +Max, including per-request TF-IDF fitting and feature extraction. This excludes +loading, verification, and network/serving overhead. It is not an API latency. + +Every prediction function receives only inference fields. The harness scores +answers afterward and checks every prediction again with reversed option order. +All 12,474 predictions were independently checked with the upstream label scorer. +No probabilities are invented from similarity scores; these are label accuracies, +not official composite scores or a leaderboard submission. + +Each run writes per-item predictions and scores, the full per-family sweep, +source/script hashes, software versions, timing, and paired bootstrap intervals +to `eval/data/`. The checked-in summary retains the full accuracy sweep and +selected family breakdowns; raw predictions stay in the ignored output folder. +The Worker and deployed service are unchanged. diff --git a/eval/compression-results.json b/eval/compression-results.json deleted file mode 100644 index 3f9563e..0000000 --- a/eval/compression-results.json +++ /dev/null @@ -1,2655 +0,0 @@ -{ - "default_run": { - "manifest": { - "args": { - "datasets": [ - "agnews", - "banking77", - "sst2", - "emotion" - ], - "families": null, - "seed": 0, - "n": 500, - "train_cap": 10000, - "validation_n": 600, - "out": "eval/data/compression-verified", - "jevbench": "/tmp/classifier-gzip-jevbench" - }, - "platform": "macOS-26.5.1-arm64-arm-64bit", - "processor": "arm", - "python": "3.12.14", - "zlib": "1.2.12", - "thread_environment": { - "OMP_NUM_THREADS": "1", - "OPENBLAS_NUM_THREADS": "1" - }, - "timing": "Serial local warm inference: normalization, compression/features, probabilities, tie-breaking. Excludes training, network and data loading.", - "versions": { - "datasets": "5.0.1", - "numpy": "2.5.3", - "scipy": "1.18.1", - "scikit-learn": "1.9.1", - "zstandard": "0.25.0" - }, - "script_sha256": "5342902063b3071509c854905aa18c2410c8202e74c1fcad2ab76ae89491a2f4" - }, - "datasets": { - "agnews": { - "dataset": { - "repo": "fancyzhx/ag_news", - "hf_fingerprints": { - "train": "2bc1e2be15d7f2ee", - "test": "0ebcc57b6aea2fc9" - }, - "train_n": 10000, - "validation_n": 600, - "test_n": 500, - "removed_duplicates_or_eval_overlaps": 161, - "hashes": { - "train": "04944cfb62e7665ceda0faf53c118d18743495868c573e315806570d7fa2c616", - "validation": "b8edd80e646b60c5ca1a76e2744cd41f336090666c93886b305374e69ad44ef6", - "test": "76c1f71538ffa2609b9dbb9bbf842e60d19b79470d1f4d0cc554d88d95ae47ed" - }, - "labels": [ - "World", - "Sports", - "Business", - "Sci/Tech" - ] - }, - "validation_sweep": [ - { - "family": "deflate", - "config": { - "codec": "deflate", - "size": 4096 - }, - "validation_accuracy": 0.5933333333333334, - "fit_s": 0.004283166999812238, - "dictionary_or_sample_bytes": 16384 - }, - { - "family": "deflate", - "config": { - "codec": "deflate", - "size": 32768 - }, - "validation_accuracy": 0.785, - "fit_s": 0.004138415999477729, - "dictionary_or_sample_bytes": 131072 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 1024, - "trained": true - }, - "validation_accuracy": 0.62, - "fit_s": 0.13306408299831674, - "dictionary_or_sample_bytes": 4096 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 1024, - "trained": false - }, - "validation_accuracy": 0.395, - "fit_s": 0.003096458996878937, - "dictionary_or_sample_bytes": 4096 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 4096, - "trained": true - }, - "validation_accuracy": 0.7116666666666667, - "fit_s": 0.148906624999654, - "dictionary_or_sample_bytes": 16384 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 4096, - "trained": false - }, - "validation_accuracy": 0.5316666666666666, - "fit_s": 0.0029460840014507994, - "dictionary_or_sample_bytes": 16384 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 16384, - "trained": true - }, - "validation_accuracy": 0.7816666666666666, - "fit_s": 0.12514475000352832, - "dictionary_or_sample_bytes": 65536 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 16384, - "trained": false - }, - "validation_accuracy": 0.6383333333333333, - "fit_s": 0.0035843750010826625, - "dictionary_or_sample_bytes": 65536 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 65536, - "level": 9 - }, - "validation_accuracy": 0.755, - "fit_s": 0.14723604200116824, - "dictionary_or_sample_bytes": 262144 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 262144, - "level": 9 - }, - "validation_accuracy": 0.7933333333333333, - "fit_s": 0.2483245409966912, - "dictionary_or_sample_bytes": 1048576 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 16777216, - "level": 9, - "trained": false - }, - "validation_accuracy": 0.8466666666666667, - "fit_s": 0.003961707996495534, - "dictionary_or_sample_bytes": 2366771 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 1024, - "shards": 4 - }, - "validation_accuracy": 0.7016666666666667, - "fit_s": 0.11391575000016019, - "dictionary_or_sample_bytes": 16384 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 1024, - "shards": 8 - }, - "validation_accuracy": 0.7266666666666667, - "fit_s": 0.12054720900050597, - "dictionary_or_sample_bytes": 32768 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 4096, - "shards": 4 - }, - "validation_accuracy": 0.7783333333333333, - "fit_s": 0.17654266599856783, - "dictionary_or_sample_bytes": 65536 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 4096, - "shards": 8 - }, - "validation_accuracy": 0.7983333333333333, - "fit_s": 0.174204084003577, - "dictionary_or_sample_bytes": 131072 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 16777216, - "shards": 3, - "trained": false, - "level": 9, - "reduction": "min" - }, - "validation_accuracy": 0.8633333333333333, - "fit_s": 0.0033049999983632006, - "dictionary_or_sample_bytes": 2366763 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 16777216, - "shards": 3, - "trained": false, - "level": 9, - "reduction": "mean" - }, - "validation_accuracy": 0.895, - "fit_s": 0.0030992079991847277, - "dictionary_or_sample_bytes": 2366763 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 16777216, - "shards": 5, - "trained": false, - "level": 9, - "reduction": "min" - }, - "validation_accuracy": 0.8766666666666667, - "fit_s": 0.003387500000826549, - "dictionary_or_sample_bytes": 2366755 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 16777216, - "shards": 5, - "trained": false, - "level": 9, - "reduction": "mean" - }, - "validation_accuracy": 0.8833333333333333, - "fit_s": 0.003104374998656567, - "dictionary_or_sample_bytes": 2366755 - }, - { - "family": "gzip-knn", - "config": { - "k": 1 - }, - "validation_accuracy": 0.6116666666666667, - "fit_s": 0.010663459004717879, - "dictionary_or_sample_bytes": 232003 - }, - { - "family": "gzip-knn", - "config": { - "k": 5 - }, - "validation_accuracy": 0.6383333333333333, - "fit_s": 0.012955791004060302, - "dictionary_or_sample_bytes": 232003 - }, - { - "family": "tfidf-logistic", - "config": {}, - "validation_accuracy": 0.9033333333333333, - "fit_s": 5.491404250002233, - "dictionary_or_sample_bytes": null - } - ], - "selected": { - "deflate": { - "family": "deflate", - "config": { - "codec": "deflate", - "size": 32768 - }, - "validation_accuracy": 0.785, - "fit_s": 0.004138415999477729, - "dictionary_or_sample_bytes": 131072, - "temperature": 4.29444227551508, - "accuracy": 0.766, - "macro_f1": 0.771227501914879, - "ece": 0.06008734299381312, - "nll": 0.6778785936410218, - "brier": 0.349022246408619, - "tie_rate": 0.03, - "p50_ms": 0.0734795, - "p95_ms": 0.09922049999999999, - "serial_items_per_s": 13096.827675922877, - "n": 500, - "accuracy_wilson_95": [ - 0.7269477983375824, - 0.8009959046039771 - ], - "confidence_at_least_90pct": { - "n": 180, - "coverage": 0.36, - "accuracy": 0.9222222222222223 - } - }, - "zstd": { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 16777216, - "level": 9, - "trained": false - }, - "validation_accuracy": 0.8466666666666667, - "fit_s": 0.003961707996495534, - "dictionary_or_sample_bytes": 2366771, - "temperature": 4.637143191628681, - "accuracy": 0.808, - "macro_f1": 0.8072402391715388, - "ece": 0.057153539576628706, - "nll": 0.6757015422951654, - "brier": 0.2950965013185466, - "tie_rate": 0.022, - "p50_ms": 0.044521, - "p95_ms": 0.0784274, - "serial_items_per_s": 20026.02101066041, - "n": 500, - "accuracy_wilson_95": [ - 0.7711789076945584, - 0.8401243272904052 - ], - "confidence_at_least_90pct": { - "n": 263, - "coverage": 0.526, - "accuracy": 0.9125475285171103 - } - }, - "zstd-mixture": { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 16777216, - "shards": 3, - "trained": false, - "level": 9, - "reduction": "mean" - }, - "validation_accuracy": 0.895, - "fit_s": 0.0030992079991847277, - "dictionary_or_sample_bytes": 2366763, - "temperature": 3.410951603860984, - "accuracy": 0.848, - "macro_f1": 0.8480560127251228, - "ece": 0.04504076255490379, - "nll": 0.5399901438401714, - "brier": 0.22715867018883865, - "tie_rate": 0.006, - "p50_ms": 0.12854149999999998, - "p95_ms": 0.19065174999999993, - "serial_items_per_s": 7451.103215557868, - "n": 500, - "accuracy_wilson_95": [ - 0.8138851771600609, - 0.8768080883424302 - ], - "confidence_at_least_90pct": { - "n": 302, - "coverage": 0.604, - "accuracy": 0.9370860927152318 - } - }, - "gzip-knn": { - "family": "gzip-knn", - "config": { - "k": 5 - }, - "validation_accuracy": 0.6383333333333333, - "fit_s": 0.012955791004060302, - "dictionary_or_sample_bytes": 232003, - "temperature": 2.1518556436402267, - "accuracy": 0.618, - "macro_f1": 0.6199502440461163, - "ece": 0.10602590208369551, - "nll": 0.9974734324691097, - "brier": 0.5320566795008369, - "tie_rate": 0.2, - "p50_ms": 9.704708, - "p95_ms": 11.13455415, - "serial_items_per_s": 102.52232695558666, - "n": 500, - "accuracy_wilson_95": [ - 0.5746644739023564, - 0.6595361161243504 - ], - "confidence_at_least_90pct": { - "n": 0, - "coverage": 0.0, - "accuracy": null - } - }, - "tfidf-logistic": { - "family": "tfidf-logistic", - "config": {}, - "validation_accuracy": 0.9033333333333333, - "fit_s": 5.491404250002233, - "dictionary_or_sample_bytes": null, - "temperature": 0.7345152395906567, - "accuracy": 0.888, - "macro_f1": 0.8879638858343402, - "ece": 0.06050312885515615, - "nll": 0.44172785940001424, - "brier": 0.19518180715294506, - "tie_rate": 0.0, - "p50_ms": 0.5366455, - "p95_ms": 0.70514435, - "serial_items_per_s": 1814.9127050219765, - "n": 500, - "accuracy_wilson_95": [ - 0.8573456933034787, - 0.9127376027165404 - ], - "confidence_at_least_90pct": { - "n": 368, - "coverage": 0.736, - "accuracy": 0.9266304347826086 - } - } - } - }, - "banking77": { - "dataset": { - "repo": "legacy-datasets/banking77", - "hf_fingerprints": { - "train": "8360b37e52e4eb2a", - "test": "1d7c14b4c23350b9" - }, - "train_n": 9392, - "validation_n": 600, - "test_n": 500, - "removed_duplicates_or_eval_overlaps": 11, - "hashes": { - "train": "3d092fadae6b94818c39469faa6c8997e81f07219855d5dd0a510fb78fd79fc6", - "validation": "4e12f6da49e8667d0a3464d75bd7c82ec2338a890264835660468ccad307fddd", - "test": "709b7b5509738c4bce13a37aa816b4934b835be78e91cf0c79792ba5b1716dae" - }, - "labels": [ - "activate_my_card", - "age_limit", - "apple_pay_or_google_pay", - "atm_support", - "automatic_top_up", - "balance_not_updated_after_bank_transfer", - "balance_not_updated_after_cheque_or_cash_deposit", - "beneficiary_not_allowed", - "cancel_transfer", - "card_about_to_expire", - "card_acceptance", - "card_arrival", - "card_delivery_estimate", - "card_linking", - "card_not_working", - "card_payment_fee_charged", - "card_payment_not_recognised", - "card_payment_wrong_exchange_rate", - "card_swallowed", - "cash_withdrawal_charge", - "cash_withdrawal_not_recognised", - "change_pin", - "compromised_card", - "contactless_not_working", - "country_support", - "declined_card_payment", - "declined_cash_withdrawal", - "declined_transfer", - "direct_debit_payment_not_recognised", - "disposable_card_limits", - "edit_personal_details", - "exchange_charge", - "exchange_rate", - "exchange_via_app", - "extra_charge_on_statement", - "failed_transfer", - "fiat_currency_support", - "get_disposable_virtual_card", - "get_physical_card", - "getting_spare_card", - "getting_virtual_card", - "lost_or_stolen_card", - "lost_or_stolen_phone", - "order_physical_card", - "passcode_forgotten", - "pending_card_payment", - "pending_cash_withdrawal", - "pending_top_up", - "pending_transfer", - "pin_blocked", - "receiving_money", - "Refund_not_showing_up", - "request_refund", - "reverted_card_payment?", - "supported_cards_and_currencies", - "terminate_account", - "top_up_by_bank_transfer_charge", - "top_up_by_card_charge", - "top_up_by_cash_or_cheque", - "top_up_failed", - "top_up_limits", - "top_up_reverted", - "topping_up_by_card", - "transaction_charged_twice", - "transfer_fee_charged", - "transfer_into_account", - "transfer_not_received_by_recipient", - "transfer_timing", - "unable_to_verify_identity", - "verify_my_identity", - "verify_source_of_funds", - "verify_top_up", - "virtual_card_not_working", - "visa_or_mastercard", - "why_verify_identity", - "wrong_amount_of_cash_received", - "wrong_exchange_rate_for_cash_withdrawal" - ] - }, - "validation_sweep": [ - { - "family": "deflate", - "config": { - "codec": "deflate", - "size": 4096 - }, - "validation_accuracy": 0.805, - "fit_s": 0.0074147080013062805, - "dictionary_or_sample_bytes": 304038 - }, - { - "family": "deflate", - "config": { - "codec": "deflate", - "size": 32768 - }, - "validation_accuracy": 0.7983333333333333, - "fit_s": 0.008205541002098471, - "dictionary_or_sample_bytes": 567780 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 1024, - "trained": true - }, - "validation_accuracy": 0.6466666666666666, - "fit_s": 0.05034095799783245, - "dictionary_or_sample_bytes": 78848 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 1024, - "trained": false - }, - "validation_accuracy": 0.6416666666666667, - "fit_s": 0.007061042000714224, - "dictionary_or_sample_bytes": 78848 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 4096, - "trained": true - }, - "validation_accuracy": 0.6933333333333334, - "fit_s": 0.08712499999819556, - "dictionary_or_sample_bytes": 286121 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 4096, - "trained": false - }, - "validation_accuracy": 0.7333333333333333, - "fit_s": 0.007319832999201026, - "dictionary_or_sample_bytes": 304038 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 16384, - "trained": true - }, - "validation_accuracy": 0.7133333333333334, - "fit_s": 0.12062429200159386, - "dictionary_or_sample_bytes": 411372 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 16384, - "trained": false - }, - "validation_accuracy": 0.7083333333333334, - "fit_s": 0.007289250002941117, - "dictionary_or_sample_bytes": 567780 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 65536, - "level": 9 - }, - "validation_accuracy": 0.7483333333333333, - "fit_s": 0.11870433299918659, - "dictionary_or_sample_bytes": 411372 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 262144, - "level": 9 - }, - "validation_accuracy": 0.7483333333333333, - "fit_s": 0.12351499999931548, - "dictionary_or_sample_bytes": 411372 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 16777216, - "level": 9, - "trained": false - }, - "validation_accuracy": 0.7816666666666666, - "fit_s": 0.007592082998598926, - "dictionary_or_sample_bytes": 567780 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 1024, - "shards": 4 - }, - "validation_accuracy": 0.72, - "fit_s": 0.12457012500090059, - "dictionary_or_sample_bytes": 281325 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 1024, - "shards": 8 - }, - "training_error": "cannot train dict: Src size is incorrect" - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 4096, - "shards": 4 - }, - "validation_accuracy": 0.7366666666666667, - "fit_s": 0.20517695900343824, - "dictionary_or_sample_bytes": 454504 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 4096, - "shards": 8 - }, - "training_error": "cannot train dict: Src size is incorrect" - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 16777216, - "shards": 3, - "trained": false, - "level": 9, - "reduction": "min" - }, - "validation_accuracy": 0.8166666666666667, - "fit_s": 0.007201125001301989, - "dictionary_or_sample_bytes": 567626 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 16777216, - "shards": 3, - "trained": false, - "level": 9, - "reduction": "mean" - }, - "validation_accuracy": 0.835, - "fit_s": 0.007468750001862645, - "dictionary_or_sample_bytes": 567626 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 16777216, - "shards": 5, - "trained": false, - "level": 9, - "reduction": "min" - }, - "validation_accuracy": 0.8183333333333334, - "fit_s": 0.0075589999978546984, - "dictionary_or_sample_bytes": 567472 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 16777216, - "shards": 5, - "trained": false, - "level": 9, - "reduction": "mean" - }, - "validation_accuracy": 0.825, - "fit_s": 0.00782808299845783, - "dictionary_or_sample_bytes": 567472 - }, - { - "family": "gzip-knn", - "config": { - "k": 1 - }, - "validation_accuracy": 0.49833333333333335, - "fit_s": 0.011530167001183145, - "dictionary_or_sample_bytes": 52739 - }, - { - "family": "gzip-knn", - "config": { - "k": 5 - }, - "validation_accuracy": 0.45166666666666666, - "fit_s": 0.011287084002105985, - "dictionary_or_sample_bytes": 52739 - }, - { - "family": "tfidf-logistic", - "config": {}, - "validation_accuracy": 0.9016666666666666, - "fit_s": 6.198137958999723, - "dictionary_or_sample_bytes": null - } - ], - "selected": { - "deflate": { - "family": "deflate", - "config": { - "codec": "deflate", - "size": 4096 - }, - "validation_accuracy": 0.805, - "fit_s": 0.0074147080013062805, - "dictionary_or_sample_bytes": 304038, - "temperature": 1.35753398137484, - "accuracy": 0.82, - "macro_f1": 0.7992486209415565, - "ece": 0.06290305365662752, - "nll": 0.748491242227106, - "brier": 0.28054942628044477, - "tie_rate": 0.094, - "p50_ms": 0.5353749999999999, - "p95_ms": 0.71514165, - "serial_items_per_s": 1798.0028453251189, - "n": 500, - "accuracy_wilson_95": [ - 0.7839246244600844, - 0.8511956196801373 - ], - "confidence_at_least_90pct": { - "n": 244, - "coverage": 0.488, - "accuracy": 0.9590163934426229 - } - }, - "zstd": { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 16777216, - "level": 9, - "trained": false - }, - "validation_accuracy": 0.7816666666666666, - "fit_s": 0.007592082998598926, - "dictionary_or_sample_bytes": 567780, - "temperature": 1.5828442546700814, - "accuracy": 0.778, - "macro_f1": 0.7634533737731454, - "ece": 0.03987457002093967, - "nll": 0.8437418486929151, - "brier": 0.31676958909518704, - "tie_rate": 0.1, - "p50_ms": 0.14937499999999998, - "p95_ms": 0.3112707999999996, - "serial_items_per_s": 5918.135240090225, - "n": 500, - "accuracy_wilson_95": [ - 0.7395294756647782, - 0.8122312364320395 - ], - "confidence_at_least_90pct": { - "n": 234, - "coverage": 0.468, - "accuracy": 0.9700854700854701 - } - }, - "zstd-mixture": { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 16777216, - "shards": 3, - "trained": false, - "level": 9, - "reduction": "mean" - }, - "validation_accuracy": 0.835, - "fit_s": 0.007468750001862645, - "dictionary_or_sample_bytes": 567626, - "temperature": 1.35753398137484, - "accuracy": 0.806, - "macro_f1": 0.7948050126018442, - "ece": 0.027330381293454824, - "nll": 0.7245106932888512, - "brier": 0.27127663578583444, - "tie_rate": 0.026, - "p50_ms": 0.5831459999999999, - "p95_ms": 1.1346207999999984, - "serial_items_per_s": 1549.8975985756417, - "n": 500, - "accuracy_wilson_95": [ - 0.7690596512561212, - 0.838274082202966 - ], - "confidence_at_least_90pct": { - "n": 273, - "coverage": 0.546, - "accuracy": 0.9743589743589743 - } - }, - "gzip-knn": { - "family": "gzip-knn", - "config": { - "k": 1 - }, - "validation_accuracy": 0.49833333333333335, - "fit_s": 0.011530167001183145, - "dictionary_or_sample_bytes": 52739, - "temperature": 1.0782500762595328, - "accuracy": 0.548, - "macro_f1": 0.5242677003523842, - "ece": 0.06063191239609045, - "nll": 2.6533837260533817, - "brier": 0.7007323897060809, - "tie_rate": 0.0, - "p50_ms": 4.4909795, - "p95_ms": 5.387016699999999, - "serial_items_per_s": 217.7996094879138, - "n": 500, - "accuracy_wilson_95": [ - 0.5041745952330335, - 0.5910934413879999 - ], - "confidence_at_least_90pct": { - "n": 0, - "coverage": 0.0, - "accuracy": null - } - }, - "tfidf-logistic": { - "family": "tfidf-logistic", - "config": {}, - "validation_accuracy": 0.9016666666666666, - "fit_s": 6.198137958999723, - "dictionary_or_sample_bytes": null, - "temperature": 0.5402886714960652, - "accuracy": 0.908, - "macro_f1": 0.8941185048731366, - "ece": 0.02614868859476964, - "nll": 0.3168123349805611, - "brier": 0.1376886364442629, - "tie_rate": 0.0, - "p50_ms": 0.383479, - "p95_ms": 0.52263335, - "serial_items_per_s": 2489.7499360395686, - "n": 500, - "accuracy_wilson_95": [ - 0.8794606777802376, - 0.9303176334985452 - ], - "confidence_at_least_90pct": { - "n": 392, - "coverage": 0.784, - "accuracy": 0.9897959183673469 - } - } - } - }, - "sst2": { - "dataset": { - "repo": "stanfordnlp/sst2", - "hf_fingerprints": { - "train": "e7e6610fdbb3c277", - "validation": "c1ddc6497ec97f98", - "test": "14a60218fd61ec38" - }, - "train_n": 10000, - "validation_n": 600, - "test_n": 500, - "removed_duplicates_or_eval_overlaps": 371, - "hashes": { - "train": "bb18ecae29bb9ea84b25405f6a0428096951a9a69efac09bb100c9facea1869d", - "validation": "fafee5f02c9c933efd0edf3c5c0b1621b8522669dfaca255450c809c6a815590", - "test": "380747f6d5a4f05ce8516c891e34d9dd29500beace1c0937b524634411bb287e" - }, - "labels": [ - "negative", - "positive" - ] - }, - "validation_sweep": [ - { - "family": "deflate", - "config": { - "codec": "deflate", - "size": 4096 - }, - "validation_accuracy": 0.5966666666666667, - "fit_s": 0.0017086660009226762, - "dictionary_or_sample_bytes": 8192 - }, - { - "family": "deflate", - "config": { - "codec": "deflate", - "size": 32768 - }, - "validation_accuracy": 0.6766666666666666, - "fit_s": 0.0017910410024342127, - "dictionary_or_sample_bytes": 65536 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 1024, - "trained": true - }, - "validation_accuracy": 0.5166666666666667, - "fit_s": 0.027700541002559476, - "dictionary_or_sample_bytes": 2048 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 1024, - "trained": false - }, - "validation_accuracy": 0.5066666666666667, - "fit_s": 0.0017736669979058206, - "dictionary_or_sample_bytes": 2048 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 4096, - "trained": true - }, - "validation_accuracy": 0.5833333333333334, - "fit_s": 0.04100783300236799, - "dictionary_or_sample_bytes": 8192 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 4096, - "trained": false - }, - "validation_accuracy": 0.5483333333333333, - "fit_s": 0.002048333000857383, - "dictionary_or_sample_bytes": 8192 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 16384, - "trained": true - }, - "validation_accuracy": 0.6283333333333333, - "fit_s": 0.038298833002045285, - "dictionary_or_sample_bytes": 32768 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 16384, - "trained": false - }, - "validation_accuracy": 0.5933333333333334, - "fit_s": 0.0018218339973827824, - "dictionary_or_sample_bytes": 32768 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 65536, - "level": 9 - }, - "validation_accuracy": 0.6433333333333333, - "fit_s": 0.046571332997700665, - "dictionary_or_sample_bytes": 131072 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 262144, - "level": 9 - }, - "validation_accuracy": 0.7066666666666667, - "fit_s": 0.08784708299936028, - "dictionary_or_sample_bytes": 437677 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 16777216, - "level": 9, - "trained": false - }, - "validation_accuracy": 0.7283333333333334, - "fit_s": 0.0018446670001139864, - "dictionary_or_sample_bytes": 536488 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 1024, - "shards": 4 - }, - "validation_accuracy": 0.5583333333333333, - "fit_s": 0.029137124998669606, - "dictionary_or_sample_bytes": 8192 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 1024, - "shards": 8 - }, - "validation_accuracy": 0.58, - "fit_s": 0.031216542003676295, - "dictionary_or_sample_bytes": 16384 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 4096, - "shards": 4 - }, - "validation_accuracy": 0.6216666666666667, - "fit_s": 0.0438428749985178, - "dictionary_or_sample_bytes": 32768 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 4096, - "shards": 8 - }, - "validation_accuracy": 0.6416666666666667, - "fit_s": 0.04907000000093831, - "dictionary_or_sample_bytes": 65536 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 16777216, - "shards": 3, - "trained": false, - "level": 9, - "reduction": "min" - }, - "validation_accuracy": 0.725, - "fit_s": 0.00196666600095341, - "dictionary_or_sample_bytes": 536484 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 16777216, - "shards": 3, - "trained": false, - "level": 9, - "reduction": "mean" - }, - "validation_accuracy": 0.7533333333333333, - "fit_s": 0.001909541002532933, - "dictionary_or_sample_bytes": 536484 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 16777216, - "shards": 5, - "trained": false, - "level": 9, - "reduction": "min" - }, - "validation_accuracy": 0.7366666666666667, - "fit_s": 0.0017700829994282685, - "dictionary_or_sample_bytes": 536480 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 16777216, - "shards": 5, - "trained": false, - "level": 9, - "reduction": "mean" - }, - "validation_accuracy": 0.7616666666666667, - "fit_s": 0.001830375003919471, - "dictionary_or_sample_bytes": 536480 - }, - { - "family": "gzip-knn", - "config": { - "k": 1 - }, - "validation_accuracy": 0.5733333333333334, - "fit_s": 0.005904542005737312, - "dictionary_or_sample_bytes": 52471 - }, - { - "family": "gzip-knn", - "config": { - "k": 5 - }, - "validation_accuracy": 0.595, - "fit_s": 0.006380707993230317, - "dictionary_or_sample_bytes": 52471 - }, - { - "family": "tfidf-logistic", - "config": {}, - "validation_accuracy": 0.84, - "fit_s": 0.5434237500012387, - "dictionary_or_sample_bytes": null - } - ], - "selected": { - "deflate": { - "family": "deflate", - "config": { - "codec": "deflate", - "size": 32768 - }, - "validation_accuracy": 0.6766666666666666, - "fit_s": 0.0017910410024342127, - "dictionary_or_sample_bytes": 65536, - "temperature": 6.304134293587772, - "accuracy": 0.656, - "macro_f1": 0.655332724153962, - "ece": 0.02910858412013078, - "nll": 0.6184185625087876, - "brier": 0.43000688230669276, - "tie_rate": 0.074, - "p50_ms": 0.0276875, - "p95_ms": 0.036502099999999996, - "serial_items_per_s": 35514.841972093716, - "n": 500, - "auroc": 0.7180215804303278, - "accuracy_wilson_95": [ - 0.6133133706304251, - 0.6963077483879331 - ], - "confidence_at_least_90pct": { - "n": 11, - "coverage": 0.022, - "accuracy": 0.9090909090909091 - } - }, - "zstd": { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 16777216, - "level": 9, - "trained": false - }, - "validation_accuracy": 0.7283333333333334, - "fit_s": 0.0018446670001139864, - "dictionary_or_sample_bytes": 536488, - "temperature": 10.790252743818396, - "accuracy": 0.65, - "macro_f1": 0.6499985999944, - "ece": 0.03593030036559715, - "nll": 0.618717418902004, - "brier": 0.43029277027337204, - "tie_rate": 0.086, - "p50_ms": 0.013625, - "p95_ms": 0.022675299999999992, - "serial_items_per_s": 69523.71113120501, - "n": 500, - "auroc": 0.7180696080942622, - "accuracy_wilson_95": [ - 0.6071920689703412, - 0.6905205454703878 - ], - "confidence_at_least_90pct": { - "n": 2, - "coverage": 0.004, - "accuracy": 1.0 - } - }, - "zstd-mixture": { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 16777216, - "shards": 5, - "trained": false, - "level": 9, - "reduction": "mean" - }, - "validation_accuracy": 0.7616666666666667, - "fit_s": 0.001830375003919471, - "dictionary_or_sample_bytes": 536480, - "temperature": 2.925419034377105, - "accuracy": 0.718, - "macro_f1": 0.7175922031413361, - "ece": 0.04554804951123206, - "nll": 0.5547712132787098, - "brier": 0.37369763190577826, - "tie_rate": 0.028, - "p50_ms": 0.041292, - "p95_ms": 0.06892435, - "serial_items_per_s": 22944.298676765586, - "n": 500, - "auroc": 0.7985319544057375, - "accuracy_wilson_95": [ - 0.6770114417638189, - 0.7556642245567071 - ], - "confidence_at_least_90pct": { - "n": 72, - "coverage": 0.144, - "accuracy": 0.9166666666666666 - } - }, - "gzip-knn": { - "family": "gzip-knn", - "config": { - "k": 5 - }, - "validation_accuracy": 0.595, - "fit_s": 0.006380707993230317, - "dictionary_or_sample_bytes": 52471, - "temperature": 5.007191993770399, - "accuracy": 0.548, - "macro_f1": 0.5474134478283856, - "ece": 0.013168762839443831, - "nll": 0.6913485232561188, - "brier": 0.4975115063432846, - "tie_rate": 0.0, - "p50_ms": 5.481375, - "p95_ms": 6.29362325, - "serial_items_per_s": 182.7036415967676, - "n": 500, - "auroc": 0.5551037397540983, - "accuracy_wilson_95": [ - 0.5041745952330335, - 0.5910934413879999 - ], - "confidence_at_least_90pct": { - "n": 0, - "coverage": 0.0, - "accuracy": null - } - }, - "tfidf-logistic": { - "family": "tfidf-logistic", - "config": {}, - "validation_accuracy": 0.84, - "fit_s": 0.5434237500012387, - "dictionary_or_sample_bytes": null, - "temperature": 0.7345152395906567, - "accuracy": 0.796, - "macro_f1": 0.7959706197692469, - "ece": 0.03953913908774128, - "nll": 0.41708729362573255, - "brier": 0.26258361413544473, - "tie_rate": 0.0, - "p50_ms": 0.412729, - "p95_ms": 0.5067076999999999, - "serial_items_per_s": 2391.932066718238, - "n": 500, - "auroc": 0.8961321721311475, - "accuracy_wilson_95": [ - 0.7584839354593685, - 0.8290022903703368 - ], - "confidence_at_least_90pct": { - "n": 243, - "coverage": 0.486, - "accuracy": 0.9382716049382716 - } - } - } - }, - "emotion": { - "dataset": { - "repo": "dair-ai/emotion", - "hf_fingerprints": { - "train": "27c5e4f1e47ffa42", - "validation": "5cae643edfd5aaf3", - "test": "bd45cf0a5e0b9690" - }, - "train_n": 10000, - "validation_n": 600, - "test_n": 500, - "removed_duplicates_or_eval_overlaps": 42, - "hashes": { - "train": "5bcc8e86df660889cb24bfd3a261f5528d8fb3b142f4693c73fd739338b9cb86", - "validation": "bda5d620b7ea488ba509f07b3591f772feb319b8a396d7812b26f89421143e7a", - "test": "62fe848e5c2808aed7e0077efd032834d274b6eb6e7d323e76be32ef97b1feea" - }, - "labels": [ - "sadness", - "joy", - "love", - "anger", - "fear", - "surprise" - ] - }, - "validation_sweep": [ - { - "family": "deflate", - "config": { - "codec": "deflate", - "size": 4096 - }, - "validation_accuracy": 0.32666666666666666, - "fit_s": 0.002004999994824175, - "dictionary_or_sample_bytes": 24576 - }, - { - "family": "deflate", - "config": { - "codec": "deflate", - "size": 32768 - }, - "validation_accuracy": 0.43833333333333335, - "fit_s": 0.00225687499914784, - "dictionary_or_sample_bytes": 196608 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 1024, - "trained": true - }, - "validation_accuracy": 0.33, - "fit_s": 0.04342029100371292, - "dictionary_or_sample_bytes": 6144 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 1024, - "trained": false - }, - "validation_accuracy": 0.235, - "fit_s": 0.0020858330026385374, - "dictionary_or_sample_bytes": 6144 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 4096, - "trained": true - }, - "validation_accuracy": 0.445, - "fit_s": 0.05890566699963529, - "dictionary_or_sample_bytes": 24576 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 4096, - "trained": false - }, - "validation_accuracy": 0.275, - "fit_s": 0.002086583997879643, - "dictionary_or_sample_bytes": 24576 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 16384, - "trained": true - }, - "validation_accuracy": 0.44666666666666666, - "fit_s": 0.058133709004323464, - "dictionary_or_sample_bytes": 98304 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 16384, - "trained": false - }, - "validation_accuracy": 0.37666666666666665, - "fit_s": 0.0023508340018452145, - "dictionary_or_sample_bytes": 98304 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 65536, - "level": 9 - }, - "validation_accuracy": 0.4033333333333333, - "fit_s": 0.11329087500052992, - "dictionary_or_sample_bytes": 349551 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 262144, - "level": 9 - }, - "validation_accuracy": 0.43666666666666665, - "fit_s": 0.15255850000539795, - "dictionary_or_sample_bytes": 856643 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 16777216, - "level": 9, - "trained": false - }, - "validation_accuracy": 0.44166666666666665, - "fit_s": 0.0022252499984460883, - "dictionary_or_sample_bytes": 979245 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 1024, - "shards": 4 - }, - "validation_accuracy": 0.3616666666666667, - "fit_s": 0.0483227920049103, - "dictionary_or_sample_bytes": 24576 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 1024, - "shards": 8 - }, - "validation_accuracy": 0.3883333333333333, - "fit_s": 0.0537706250033807, - "dictionary_or_sample_bytes": 49152 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 4096, - "shards": 4 - }, - "validation_accuracy": 0.515, - "fit_s": 0.07201379199977964, - "dictionary_or_sample_bytes": 98304 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 4096, - "shards": 8 - }, - "validation_accuracy": 0.48, - "fit_s": 0.08771720799995819, - "dictionary_or_sample_bytes": 195367 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 16777216, - "shards": 3, - "trained": false, - "level": 9, - "reduction": "min" - }, - "validation_accuracy": 0.5066666666666667, - "fit_s": 0.002359584002988413, - "dictionary_or_sample_bytes": 979233 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 16777216, - "shards": 3, - "trained": false, - "level": 9, - "reduction": "mean" - }, - "validation_accuracy": 0.535, - "fit_s": 0.0026667080019251443, - "dictionary_or_sample_bytes": 979233 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 16777216, - "shards": 5, - "trained": false, - "level": 9, - "reduction": "min" - }, - "validation_accuracy": 0.5033333333333333, - "fit_s": 0.002307417002157308, - "dictionary_or_sample_bytes": 979221 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 16777216, - "shards": 5, - "trained": false, - "level": 9, - "reduction": "mean" - }, - "validation_accuracy": 0.5533333333333333, - "fit_s": 0.00223658300092211, - "dictionary_or_sample_bytes": 979221 - }, - { - "family": "gzip-knn", - "config": { - "k": 1 - }, - "validation_accuracy": 0.25833333333333336, - "fit_s": 0.007107000004907604, - "dictionary_or_sample_bytes": 98188 - }, - { - "family": "gzip-knn", - "config": { - "k": 5 - }, - "validation_accuracy": 0.24166666666666667, - "fit_s": 0.006843499999376945, - "dictionary_or_sample_bytes": 98188 - }, - { - "family": "tfidf-logistic", - "config": {}, - "validation_accuracy": 0.8416666666666667, - "fit_s": 3.909456957997463, - "dictionary_or_sample_bytes": null - } - ], - "selected": { - "deflate": { - "family": "deflate", - "config": { - "codec": "deflate", - "size": 32768 - }, - "validation_accuracy": 0.43833333333333335, - "fit_s": 0.00225687499914784, - "dictionary_or_sample_bytes": 196608, - "temperature": 3.410951603860984, - "accuracy": 0.406, - "macro_f1": 0.3684907264569877, - "ece": 0.03192085479609739, - "nll": 1.486202601279427, - "brier": 0.7179354684289194, - "tie_rate": 0.168, - "p50_ms": 0.0647915, - "p95_ms": 0.10161844999999997, - "serial_items_per_s": 14590.575182943197, - "n": 500, - "accuracy_wilson_95": [ - 0.3638296860034575, - 0.44960374228035244 - ], - "confidence_at_least_90pct": { - "n": 1, - "coverage": 0.002, - "accuracy": 1.0 - } - }, - "zstd": { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 16384, - "trained": true - }, - "validation_accuracy": 0.44666666666666666, - "fit_s": 0.058133709004323464, - "dictionary_or_sample_bytes": 98304, - "temperature": 2.709220452261488, - "accuracy": 0.454, - "macro_f1": 0.4018783545119495, - "ece": 0.05846382416841619, - "nll": 1.463362818214318, - "brier": 0.698117185244948, - "tie_rate": 0.214, - "p50_ms": 0.0115, - "p95_ms": 0.017131249999999997, - "serial_items_per_s": 81165.0627909159, - "n": 500, - "accuracy_wilson_95": [ - 0.4108749466266584, - 0.49782651827818475 - ], - "confidence_at_least_90pct": { - "n": 2, - "coverage": 0.004, - "accuracy": 0.5 - } - }, - "zstd-mixture": { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 16777216, - "shards": 5, - "trained": false, - "level": 9, - "reduction": "mean" - }, - "validation_accuracy": 0.5533333333333333, - "fit_s": 0.00223658300092211, - "dictionary_or_sample_bytes": 979221, - "temperature": 2.925419034377105, - "accuracy": 0.562, - "macro_f1": 0.46353991429511865, - "ece": 0.08457185931239918, - "nll": 1.2537522096435618, - "brier": 0.597552891264581, - "tie_rate": 0.038, - "p50_ms": 0.103604, - "p95_ms": 0.2117881499999997, - "serial_items_per_s": 8727.203975946568, - "n": 500, - "accuracy_wilson_95": [ - 0.5182021184950548, - 0.6048524288071132 - ], - "confidence_at_least_90pct": { - "n": 22, - "coverage": 0.044, - "accuracy": 0.8181818181818182 - } - }, - "gzip-knn": { - "family": "gzip-knn", - "config": { - "k": 1 - }, - "validation_accuracy": 0.25833333333333336, - "fit_s": 0.007107000004907604, - "dictionary_or_sample_bytes": 98188, - "temperature": 8.570386453309192, - "accuracy": 0.298, - "macro_f1": 0.2781904296547765, - "ece": 0.0427759626994248, - "nll": 1.743637872594331, - "brier": 0.814830939581835, - "tie_rate": 0.0, - "p50_ms": 5.6842915000000005, - "p95_ms": 6.971593449999999, - "serial_items_per_s": 171.43746654479273, - "n": 500, - "accuracy_wilson_95": [ - 0.25957253815100145, - 0.3395078077354835 - ], - "confidence_at_least_90pct": { - "n": 0, - "coverage": 0.0, - "accuracy": null - } - }, - "tfidf-logistic": { - "family": "tfidf-logistic", - "config": {}, - "validation_accuracy": 0.8416666666666667, - "fit_s": 3.909456957997463, - "dictionary_or_sample_bytes": null, - "temperature": 0.5834042638846706, - "accuracy": 0.852, - "macro_f1": 0.7837710838613873, - "ece": 0.030185522477487146, - "nll": 0.4129957238398832, - "brier": 0.22078743602615492, - "tie_rate": 0.0, - "p50_ms": 0.426187, - "p95_ms": 0.5887381499999998, - "serial_items_per_s": 2240.786993074041, - "n": 500, - "accuracy_wilson_95": [ - 0.818193200072725, - 0.8804390684815191 - ], - "confidence_at_least_90pct": { - "n": 322, - "coverage": 0.644, - "accuracy": 0.9472049689440993 - } - } - } - } - }, - "jevbench": { - "easy": { - "n": 48, - "sha256": "231df3c2c8e88a1a8c137ebe85de96ba70fabd330849098ac7b3c52c70b7172b", - "accuracy": 0.4583333333333333, - "uniform_chance": 0.284375, - "p50_ms": 0.074271, - "families": { - "extraction": { - "n": 12, - "accuracy": 0.4166666666666667 - }, - "fact": { - "n": 12, - "accuracy": 0.5833333333333334 - }, - "intent": { - "n": 12, - "accuracy": 0.4166666666666667 - }, - "tool_selection": { - "n": 12, - "accuracy": 0.4166666666666667 - } - } - }, - "original": { - "n": 72, - "sha256": "5c2414edb3006b8bfcb70fda433f0f9ca015759433849f8d3104328a1f7c4180", - "accuracy": 0.3611111111111111, - "uniform_chance": 0.3111111111111111, - "p50_ms": 0.0727085, - "families": { - "adequacy": { - "n": 12, - "accuracy": 0.4166666666666667 - }, - "extraction": { - "n": 12, - "accuracy": 0.3333333333333333 - }, - "intent": { - "n": 12, - "accuracy": 0.16666666666666666 - }, - "ordinal": { - "n": 12, - "accuracy": 0.4166666666666667 - }, - "policy": { - "n": 12, - "accuracy": 0.5 - }, - "routing": { - "n": 12, - "accuracy": 0.3333333333333333 - } - } - }, - "hard": { - "n": 111, - "sha256": "89e9e6becb33ed88c1de7d42dcc87531b2fb64cfaef4e1986faf7c37b3f80ebb", - "accuracy": 0.42342342342342343, - "uniform_chance": 0.3361861861861862, - "p50_ms": 0.183583, - "families": { - "adversarial": { - "n": 6, - "accuracy": 0.6666666666666666 - }, - "ambiguous": { - "n": 7, - "accuracy": 0.42857142857142855 - }, - "judge_hard": { - "n": 17, - "accuracy": 0.4117647058823529 - }, - "long_policy": { - "n": 19, - "accuracy": 0.21052631578947367 - }, - "multi_hop": { - "n": 18, - "accuracy": 0.2222222222222222 - }, - "probability": { - "n": 10, - "accuracy": 0.5 - }, - "routing_hard": { - "n": 5, - "accuracy": 1.0 - }, - "temporal_numeric": { - "n": 15, - "accuracy": 0.3333333333333333 - }, - "tradeoff": { - "n": 6, - "accuracy": 1.0 - }, - "trap": { - "n": 8, - "accuracy": 0.5 - } - } - }, - "source_commit": "2fa63fa3226cb369795525ed011800f57dcbd894", - "method": "Zero-shot symmetric gzip NCD against label and rubric; no training or calibration. Public diagnostic, not an official score." - } - }, - "large_training_run": { - "manifest": { - "args": { - "datasets": [ - "agnews", - "sst2" - ], - "families": [ - "zstd", - "zstd-mixture", - "tfidf-logistic" - ], - "seed": 0, - "n": 500, - "train_cap": 120000, - "validation_n": 600, - "out": "eval/data/compression-large", - "jevbench": null - }, - "platform": "macOS-26.5.1-arm64-arm-64bit", - "processor": "arm", - "python": "3.12.14", - "zlib": "1.2.12", - "thread_environment": { - "OMP_NUM_THREADS": "1", - "OPENBLAS_NUM_THREADS": "1" - }, - "timing": "Serial local warm inference: normalization, compression/features, probabilities, tie-breaking. Excludes training, network and data loading.", - "versions": { - "datasets": "5.0.1", - "numpy": "2.5.3", - "scipy": "1.18.1", - "scikit-learn": "1.9.1", - "zstandard": "0.25.0" - }, - "script_sha256": "5342902063b3071509c854905aa18c2410c8202e74c1fcad2ab76ae89491a2f4" - }, - "datasets": { - "agnews": { - "dataset": { - "repo": "fancyzhx/ag_news", - "hf_fingerprints": { - "train": "2bc1e2be15d7f2ee", - "test": "0ebcc57b6aea2fc9" - }, - "train_n": 119239, - "validation_n": 600, - "test_n": 500, - "removed_duplicates_or_eval_overlaps": 161, - "hashes": { - "train": "d1fb37dc340dc5ed27d4162ef9a98b4e83a80b5dc958e3e4e97101aa0f9e62ed", - "validation": "89cea8648589dcdb7aacc5e04f4efcd1b6058e5946f346d672c492fe8b03d403", - "test": "76c1f71538ffa2609b9dbb9bbf842e60d19b79470d1f4d0cc554d88d95ae47ed" - }, - "labels": [ - "World", - "Sports", - "Business", - "Sci/Tech" - ] - }, - "validation_sweep": [ - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 1024, - "trained": true - }, - "validation_accuracy": 0.655, - "fit_s": 1.095241957998951, - "dictionary_or_sample_bytes": 4096 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 1024, - "trained": false - }, - "validation_accuracy": 0.38333333333333336, - "fit_s": 0.05557016700186068, - "dictionary_or_sample_bytes": 4096 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 4096, - "trained": true - }, - "validation_accuracy": 0.7733333333333333, - "fit_s": 1.6625714999972843, - "dictionary_or_sample_bytes": 16384 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 4096, - "trained": false - }, - "validation_accuracy": 0.5033333333333333, - "fit_s": 0.0551289579962031, - "dictionary_or_sample_bytes": 16384 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 16384, - "trained": true - }, - "validation_accuracy": 0.805, - "fit_s": 1.376491791997978, - "dictionary_or_sample_bytes": 65536 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 16384, - "trained": false - }, - "validation_accuracy": 0.6483333333333333, - "fit_s": 0.05820737500471296, - "dictionary_or_sample_bytes": 65536 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 65536, - "level": 9 - }, - "validation_accuracy": 0.795, - "fit_s": 1.1389680420033983, - "dictionary_or_sample_bytes": 262144 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 262144, - "level": 9 - }, - "validation_accuracy": 0.7983333333333333, - "fit_s": 1.0690593339968473, - "dictionary_or_sample_bytes": 1048576 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 16777216, - "level": 9, - "trained": false - }, - "validation_accuracy": 0.855, - "fit_s": 0.05941937500028871, - "dictionary_or_sample_bytes": 28247565 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 1024, - "shards": 4 - }, - "validation_accuracy": 0.7283333333333334, - "fit_s": 1.1204033750036615, - "dictionary_or_sample_bytes": 16384 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 1024, - "shards": 8 - }, - "validation_accuracy": 0.73, - "fit_s": 1.1286926250031684, - "dictionary_or_sample_bytes": 32768 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 4096, - "shards": 4 - }, - "validation_accuracy": 0.8066666666666666, - "fit_s": 1.700393332997919, - "dictionary_or_sample_bytes": 65536 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 4096, - "shards": 8 - }, - "validation_accuracy": 0.8166666666666667, - "fit_s": 1.7476613329999964, - "dictionary_or_sample_bytes": 131072 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 16777216, - "shards": 3, - "trained": false, - "level": 9, - "reduction": "min" - }, - "validation_accuracy": 0.8766666666666667, - "fit_s": 0.0598740830027964, - "dictionary_or_sample_bytes": 28247557 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 16777216, - "shards": 3, - "trained": false, - "level": 9, - "reduction": "mean" - }, - "validation_accuracy": 0.8983333333333333, - "fit_s": 0.057473291002679616, - "dictionary_or_sample_bytes": 28247557 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 16777216, - "shards": 5, - "trained": false, - "level": 9, - "reduction": "min" - }, - "validation_accuracy": 0.8966666666666666, - "fit_s": 0.058699958004581276, - "dictionary_or_sample_bytes": 28247549 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 16777216, - "shards": 5, - "trained": false, - "level": 9, - "reduction": "mean" - }, - "validation_accuracy": 0.9116666666666666, - "fit_s": 0.060205792004126124, - "dictionary_or_sample_bytes": 28247549 - }, - { - "family": "tfidf-logistic", - "config": {}, - "validation_accuracy": 0.9366666666666666, - "fit_s": 72.26334416699683, - "dictionary_or_sample_bytes": null - } - ], - "selected": { - "zstd": { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 16777216, - "level": 9, - "trained": false - }, - "validation_accuracy": 0.855, - "fit_s": 0.05941937500028871, - "dictionary_or_sample_bytes": 28247565, - "temperature": 5.838236970939662, - "accuracy": 0.856, - "macro_f1": 0.8578350680660315, - "ece": 0.07148579683576123, - "nll": 0.573052244855724, - "brier": 0.2274947566483206, - "tie_rate": 0.02, - "p50_ms": 0.0481665, - "p95_ms": 0.09512054999999997, - "serial_items_per_s": 16460.801843188412, - "n": 500, - "accuracy_wilson_95": [ - 0.822508879125479, - 0.8840623924805179 - ], - "confidence_at_least_90pct": { - "n": 246, - "coverage": 0.492, - "accuracy": 0.9471544715447154 - } - }, - "zstd-mixture": { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 16777216, - "shards": 5, - "trained": false, - "level": 9, - "reduction": "mean" - }, - "validation_accuracy": 0.9116666666666666, - "fit_s": 0.060205792004126124, - "dictionary_or_sample_bytes": 28247549, - "temperature": 3.683149054535093, - "accuracy": 0.9, - "macro_f1": 0.8997427715689791, - "ece": 0.05852817346189611, - "nll": 0.4590178473955788, - "brier": 0.17604696824424826, - "tie_rate": 0.004, - "p50_ms": 0.24593700000000002, - "p95_ms": 0.4126940499999994, - "serial_items_per_s": 3567.4229415418463, - "n": 500, - "accuracy_wilson_95": [ - 0.8705774917999974, - 0.9233228133752799 - ], - "confidence_at_least_90pct": { - "n": 360, - "coverage": 0.72, - "accuracy": 0.9472222222222222 - } - }, - "tfidf-logistic": { - "family": "tfidf-logistic", - "config": {}, - "validation_accuracy": 0.9366666666666666, - "fit_s": 72.26334416699683, - "dictionary_or_sample_bytes": null, - "temperature": 0.9247663594340342, - "accuracy": 0.912, - "macro_f1": 0.9127427169277376, - "ece": 0.02759829499776148, - "nll": 0.28636336613317387, - "brier": 0.14006859532256022, - "tie_rate": 0.0, - "p50_ms": 0.593167, - "p95_ms": 0.7779229499999999, - "serial_items_per_s": 1645.5464238809047, - "n": 500, - "accuracy_wilson_95": [ - 0.8839229514479334, - 0.9337943628826022 - ], - "confidence_at_least_90pct": { - "n": 397, - "coverage": 0.794, - "accuracy": 0.9697732997481109 - } - } - } - }, - "sst2": { - "dataset": { - "repo": "stanfordnlp/sst2", - "hf_fingerprints": { - "train": "e7e6610fdbb3c277", - "validation": "c1ddc6497ec97f98", - "test": "14a60218fd61ec38" - }, - "train_n": 66378, - "validation_n": 600, - "test_n": 500, - "removed_duplicates_or_eval_overlaps": 371, - "hashes": { - "train": "488b7c60c5677461e842fc5ecd32774c1ddbdd5878c8e6c793c7430eeb5720e1", - "validation": "745798b437dae118d5c5f3498399b935f25a45040fb07d6423a8d18964f91fd7", - "test": "380747f6d5a4f05ce8516c891e34d9dd29500beace1c0937b524634411bb287e" - }, - "labels": [ - "negative", - "positive" - ] - }, - "validation_sweep": [ - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 1024, - "trained": true - }, - "validation_accuracy": 0.57, - "fit_s": 0.1833836249934393, - "dictionary_or_sample_bytes": 2048 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 1024, - "trained": false - }, - "validation_accuracy": 0.54, - "fit_s": 0.01816158300061943, - "dictionary_or_sample_bytes": 2048 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 4096, - "trained": true - }, - "validation_accuracy": 0.61, - "fit_s": 0.2721342500008177, - "dictionary_or_sample_bytes": 8192 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 4096, - "trained": false - }, - "validation_accuracy": 0.56, - "fit_s": 0.019221125003241468, - "dictionary_or_sample_bytes": 8192 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 16384, - "trained": true - }, - "validation_accuracy": 0.6816666666666666, - "fit_s": 0.2559416250005597, - "dictionary_or_sample_bytes": 32768 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 16384, - "trained": false - }, - "validation_accuracy": 0.6333333333333333, - "fit_s": 0.018674541999644134, - "dictionary_or_sample_bytes": 32768 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 65536, - "level": 9 - }, - "validation_accuracy": 0.6616666666666666, - "fit_s": 0.2445404169993708, - "dictionary_or_sample_bytes": 131072 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 262144, - "level": 9 - }, - "validation_accuracy": 0.7466666666666667, - "fit_s": 0.2638120420015184, - "dictionary_or_sample_bytes": 524288 - }, - { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 16777216, - "level": 9, - "trained": false - }, - "validation_accuracy": 0.7816666666666666, - "fit_s": 0.018079166002280544, - "dictionary_or_sample_bytes": 3567328 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 1024, - "shards": 4 - }, - "validation_accuracy": 0.6133333333333333, - "fit_s": 0.1946094590020948, - "dictionary_or_sample_bytes": 8192 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 1024, - "shards": 8 - }, - "validation_accuracy": 0.5866666666666667, - "fit_s": 0.19280641600198578, - "dictionary_or_sample_bytes": 16384 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 4096, - "shards": 4 - }, - "validation_accuracy": 0.6583333333333333, - "fit_s": 0.2784915840020403, - "dictionary_or_sample_bytes": 32768 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 4096, - "shards": 8 - }, - "validation_accuracy": 0.66, - "fit_s": 0.28512387500086334, - "dictionary_or_sample_bytes": 65536 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 16777216, - "shards": 3, - "trained": false, - "level": 9, - "reduction": "min" - }, - "validation_accuracy": 0.7916666666666666, - "fit_s": 0.01775899999483954, - "dictionary_or_sample_bytes": 3567324 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 16777216, - "shards": 3, - "trained": false, - "level": 9, - "reduction": "mean" - }, - "validation_accuracy": 0.79, - "fit_s": 0.0159552499972051, - "dictionary_or_sample_bytes": 3567324 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 16777216, - "shards": 5, - "trained": false, - "level": 9, - "reduction": "min" - }, - "validation_accuracy": 0.7883333333333333, - "fit_s": 0.017087957996409386, - "dictionary_or_sample_bytes": 3567320 - }, - { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 16777216, - "shards": 5, - "trained": false, - "level": 9, - "reduction": "mean" - }, - "validation_accuracy": 0.79, - "fit_s": 0.016865915997186676, - "dictionary_or_sample_bytes": 3567320 - }, - { - "family": "tfidf-logistic", - "config": {}, - "validation_accuracy": 0.915, - "fit_s": 3.577201125001011, - "dictionary_or_sample_bytes": null - } - ], - "selected": { - "zstd": { - "family": "zstd", - "config": { - "codec": "zstd", - "size": 16777216, - "level": 9, - "trained": false - }, - "validation_accuracy": 0.7816666666666666, - "fit_s": 0.018079166002280544, - "dictionary_or_sample_bytes": 3567328, - "temperature": 8.570386453309192, - "accuracy": 0.662, - "macro_f1": 0.6608591301137025, - "ece": 0.04250272619027978, - "nll": 0.6126490882855867, - "brier": 0.4222055784678397, - "tie_rate": 0.09, - "p50_ms": 0.015396, - "p95_ms": 0.025002049999999998, - "serial_items_per_s": 62596.52383976092, - "n": 500, - "auroc": 0.7419633709016394, - "accuracy_wilson_95": [ - 0.6194419387868491, - 0.7020876848091384 - ], - "confidence_at_least_90pct": { - "n": 6, - "coverage": 0.012, - "accuracy": 0.6666666666666666 - } - }, - "zstd-mixture": { - "family": "zstd-mixture", - "config": { - "codec": "zstd", - "size": 16777216, - "shards": 3, - "trained": false, - "level": 9, - "reduction": "min" - }, - "validation_accuracy": 0.7916666666666666, - "fit_s": 0.01775899999483954, - "dictionary_or_sample_bytes": 3567324, - "temperature": 7.3504331266672, - "accuracy": 0.7, - "macro_f1": 0.6996106954613179, - "ece": 0.059913157093607276, - "nll": 0.6007650774839849, - "brier": 0.4103150298882698, - "tie_rate": 0.084, - "p50_ms": 0.034, - "p95_ms": 0.06059134999999999, - "serial_items_per_s": 28267.394171399006, - "n": 500, - "auroc": 0.7595174820696722, - "accuracy_wilson_95": [ - 0.6584314090816449, - 0.7385187435059938 - ], - "confidence_at_least_90pct": { - "n": 13, - "coverage": 0.026, - "accuracy": 0.8461538461538461 - } - }, - "tfidf-logistic": { - "family": "tfidf-logistic", - "config": {}, - "validation_accuracy": 0.915, - "fit_s": 3.577201125001011, - "dictionary_or_sample_bytes": null, - "temperature": 0.7345152395906567, - "accuracy": 0.85, - "macro_f1": 0.8499513842484965, - "ece": 0.038211722573768335, - "nll": 0.3873007503417493, - "brier": 0.22026970652824607, - "tie_rate": 0.0, - "p50_ms": 0.43174999999999997, - "p95_ms": 0.5471291999999999, - "serial_items_per_s": 2263.669087689429, - "n": 500, - "auroc": 0.9229956454918032, - "accuracy_wilson_95": [ - 0.8160382472597502, - 0.8786245197686174 - ], - "confidence_at_least_90pct": { - "n": 320, - "coverage": 0.64, - "accuracy": 0.940625 - } - } - } - } - } - }, - "verification": { - "predictions_validated": 10000, - "dictionary_and_lexical_predictions_reproduce_exactly": true, - "script_hash_matches": true, - "jevbench": { - "official_label_scoring_matches": true, - "option_reversal_invariant": true, - "results": [ - { - "tier": "easy", - "n": 48, - "correct": 22, - "accuracy": 0.4583333333333333 - }, - { - "tier": "hard", - "n": 111, - "correct": 47, - "accuracy": 0.42342342342342343 - }, - { - "tier": "original", - "n": 72, - "correct": 26, - "accuracy": 0.3611111111111111 - } - ] - }, - "large_training_predictions_validated": 3000, - "large_training_uses_identical_test_sample": true - } -} diff --git a/eval/compression.py b/eval/compression.py deleted file mode 100644 index 35cad0c..0000000 --- a/eval/compression.py +++ /dev/null @@ -1,371 +0,0 @@ -#!/usr/bin/env python3 -# /// script -# requires-python = ">=3.12" -# dependencies = ["datasets==5.0.1", "numpy==2.5.3", "scipy==1.18.1", "scikit-learn==1.9.1", "zstandard==0.25.0"] -# /// -"""Compression classification experiment. Run with: uv run eval/compression.py. - -All tuning uses held-back training rows. Each family gets one final test run. -Raw predictions, validation sweeps, split hashes and environment are saved. -""" -import argparse -import gzip -import hashlib -import importlib.metadata -import json -import os -import platform -import random -import subprocess -import time -import zlib -from pathlib import Path - -import numpy as np -import zstandard as zstd -from datasets import load_dataset -from scipy.special import softmax -from sklearn.feature_extraction.text import TfidfVectorizer -from sklearn.pipeline import FeatureUnion -from sklearn.linear_model import LogisticRegression -from sklearn.metrics import accuracy_score, f1_score, log_loss, roc_auc_score -from sklearn.model_selection import train_test_split - -DATASETS = { - "agnews": ("fancyzhx/ag_news", None, "text", "test"), - "banking77": ("legacy-datasets/banking77", None, "text", "test"), - "sst2": ("stanfordnlp/sst2", None, "sentence", "validation"), - "emotion": ("dair-ai/emotion", "split", "text", "test"), -} - - -def normalize(text): - return " ".join(text.lower().split()) - - -def digest(value): - return hashlib.sha256(json.dumps(value, ensure_ascii=False, sort_keys=True).encode()).hexdigest() - - -def load_split(name, seed, n, train_cap, val_n): - repo, config, field, test = DATASETS[name] - ds = load_dataset(repo, config) - labels = ds["train"].features["label"].names - heldout = [(normalize(r[field]), int(r["label"])) for r in ds[test]] - rng = random.Random(seed) - rng.shuffle(heldout) - evaluation = heldout[:n] - seen = {x for x, _ in heldout} - pool = [] - duplicates = 0 - for r in ds["train"]: - x, y = normalize(r[field]), int(r["label"]) - if x in seen: - duplicates += 1 - continue - seen.add(x) - pool.append((x, y)) - rng.shuffle(pool) - pool = pool[:train_cap + val_n] - train, val = train_test_split(pool, test_size=val_n, random_state=seed, - stratify=[r[1] for r in pool]) - assert not ({r[0] for r in train} & {r[0] for r in val}) - assert not ({r[0] for r in train + val} & {r[0] for r in heldout}) - assert set(y for _, y in train) == set(range(len(labels))) - meta = {"repo": repo, "hf_fingerprints": {k: v._fingerprint for k, v in ds.items()}, - "train_n": len(train), "validation_n": len(val), "test_n": len(evaluation), - "removed_duplicates_or_eval_overlaps": duplicates, - "hashes": {k: digest(v) for k, v in [("train", train), ("validation", val), ("test", evaluation)]}, - "labels": labels} - return train, val, evaluation, meta - - -class Dictionaries: - def __init__(self, rows, classes, codec, size, shards=1, seed=42, trained=True, level=3, reduction="min"): - self.codec, self.models, self.bytes = codec, [], 0 - self.reduction = reduction - rng = random.Random(seed) - for label in range(classes): - texts = [x.encode() for x, y in rows if y == label] - rng.shuffle(texts) - models = [] - for i in range(shards): - samples = texts[i::shards] - if not samples: - raise ValueError("empty class shard") - if codec == "zstd": - if trained: - # Do not silently change the algorithm if training fails. - dictionary = zstd.train_dictionary(size, samples, threads=0) - else: - dictionary = zstd.ZstdCompressionDict(b"\n".join(samples)[-size:], dict_type=zstd.DICT_TYPE_RAWCONTENT) - self.bytes += len(dictionary.as_bytes()) - models.append(zstd.ZstdCompressor(level=level, dict_data=dictionary)) - else: - dictionary = b"\n".join(samples)[-size:] - self.bytes += len(dictionary) - models.append(zlib.compressobj(level=6, wbits=-15, zdict=dictionary)) - self.models.append(models) - - def score(self, text): - x = text.encode() - scores = [] - for models in self.models: - lengths = [] - for model in models: - if self.codec == "zstd": - lengths.append(len(model.compress(x))) - else: - c = model.copy() - lengths.append(len(c.compress(x) + c.flush())) - scores.append(-float(np.mean(lengths)) if self.reduction == "mean" else -min(lengths)) - return np.asarray(scores, dtype=float) - - -class GzipKNN: - def __init__(self, rows, classes, k=5, cap=1000, seed=42): - rng = random.Random(seed) - rows = list(rows) - rng.shuffle(rows) - # Balanced memory avoids a majority-class prior from subsampling. - self.samples = [] - for label in range(classes): - self.samples += [(x.encode(), y) for x, y in rows if y == label][:max(1, cap // classes)] - self.lengths = [len(gzip.compress(x, mtime=0)) for x, _ in self.samples] - self.tie_keys = [hashlib.sha256(x).hexdigest() for x, _ in self.samples] - self.classes, self.k = classes, k - self.bytes = sum(len(x) for x, _ in self.samples) - - def score(self, text): - x = text.encode() - cx = len(gzip.compress(x, mtime=0)) - distances = [(len(gzip.compress(z + b" " + x, mtime=0)) - min(cx, cz)) / max(cx, cz) - for (z, _), cz in zip(self.samples, self.lengths)] - nearest = np.lexsort((self.tie_keys, distances))[:self.k] - votes = np.full(self.classes, 0.01) - for i in nearest: - votes[self.samples[i][1]] += 1 - return np.log(votes) - - -class Lexical: - def __init__(self, rows, classes): - self.vectorizer = FeatureUnion([ - ("word", TfidfVectorizer(ngram_range=(1, 2), sublinear_tf=True, min_df=2)), - ("char", TfidfVectorizer(analyzer="char", ngram_range=(3, 5), sublinear_tf=True, min_df=2, max_features=100000)), - ]) - x = self.vectorizer.fit_transform([x for x, _ in rows]) - self.classifier = LogisticRegression(C=4, max_iter=500).fit(x, [y for _, y in rows]) - self.bytes = None - - def score(self, text): - return self.classifier.predict_log_proba(self.vectorizer.transform([text]))[0] - - -def scores_for(model, rows, temp=1): - scores, times = [], [] - for text, _ in rows: - start = time.perf_counter_ns() - score = model.score(normalize(text)) - softmax(score / temp) - predictions([score], [text]) - scores.append(score) - times.append((time.perf_counter_ns() - start) / 1e6) - return np.asarray(scores), times - - -def predictions(scores, texts): - # Exact ties use a reproducible text hash instead of always selecting class 0. - out = [] - for score, text in zip(scores, texts): - ties = np.flatnonzero(score == score.max()) - index = int(hashlib.sha256(text.encode()).hexdigest(), 16) % len(ties) - out.append(int(ties[index])) - return np.asarray(out) - - -def temperature(scores, gold): - choices = np.geomspace(.05, 100, 100) - losses = [log_loss(gold, softmax(scores / t, axis=1), labels=np.arange(scores.shape[1])) for t in choices] - return float(choices[np.argmin(losses)]) - - -def metrics(scores, rows, times, temp): - gold = np.asarray([y for _, y in rows]) - pred = predictions(scores, [x for x, _ in rows]) - prob = softmax(scores / temp, axis=1) - conf = prob.max(axis=1) - correct = pred == gold - ece = 0. - for i in range(10): - mask = (conf >= i / 10) & ((conf < (i + 1) / 10) if i < 9 else (conf <= 1)) - if mask.any(): - ece += mask.mean() * abs(correct[mask].mean() - conf[mask].mean()) - result = {"accuracy": float(correct.mean()), "macro_f1": float(f1_score(gold, pred, labels=np.arange(scores.shape[1]), average="macro", zero_division=0)), - "ece": float(ece), "nll": float(log_loss(gold, prob, labels=np.arange(scores.shape[1]))), - "brier": float(np.mean(np.sum((prob - np.eye(scores.shape[1])[gold]) ** 2, axis=1))), - "tie_rate": float(np.mean((scores == scores.max(axis=1, keepdims=True)).sum(axis=1) > 1)), - "p50_ms": float(np.median(times)), "p95_ms": float(np.percentile(times, 95)), - "serial_items_per_s": float(1000 / np.mean(times)), "n": len(rows)} - if scores.shape[1] == 2: - result["auroc"] = float(roc_auc_score(gold, prob[:, 1])) - n, p, z = len(rows), float(correct.mean()), 1.96 - center = (p + z * z / (2 * n)) / (1 + z * z / n) - radius = z * np.sqrt(p * (1 - p) / n + z * z / (4 * n * n)) / (1 + z * z / n) - result["accuracy_wilson_95"] = [float(center - radius), float(center + radius)] - mask = conf >= .9 - result["confidence_at_least_90pct"] = { - "n": int(mask.sum()), "coverage": float(mask.mean()), - "accuracy": float(correct[mask].mean()) if mask.any() else None, - } - return result, [{"text_sha256": hashlib.sha256(x.encode()).hexdigest(), "gold": int(y), - "prediction": int(p), "probabilities": ps.tolist(), "ms": ms} - for (x, y), p, ps, ms in zip(rows, pred, prob, times)] - - -def run_dataset(name, args, outdir): - train, val, test, meta = load_split(name, args.seed, args.n, args.train_cap, args.validation_n) - classes = len(meta["labels"]) - configs = [("deflate", {"codec": "deflate", "size": s}) for s in [4096, 32768]] - configs += [("zstd", {"codec": "zstd", "size": s, "trained": trained}) - for s in [1024, 4096, 16384] for trained in [True, False]] - configs += [("zstd", {"codec": "zstd", "size": s, "level": 9}) - for s in [65536, 262144]] - configs += [("zstd", {"codec": "zstd", "size": 16777216, "level": 9, "trained": False})] - configs += [("zstd-mixture", {"codec": "zstd", "size": s, "shards": shards}) - for s in [1024, 4096] for shards in [4, 8]] - configs += [("zstd-mixture", {"codec": "zstd", "size": 16777216, "shards": shards, - "trained": False, "level": 9, "reduction": reduction}) - for shards in [3, 5] for reduction in ["min", "mean"]] - configs += [("gzip-knn", {"k": k}) for k in [1, 5]] - configs += [("tfidf-logistic", {})] - if args.families: - configs = [(family, config) for family, config in configs if family in args.families] - best, sweep = {}, [] - for family, config in configs: - start = time.perf_counter() - try: - if family == "gzip-knn": - model = GzipKNN(train, classes, seed=args.seed, **config) - elif family == "tfidf-logistic": - model = Lexical(train, classes) - else: - model = Dictionaries(train, classes, seed=args.seed, **config) - except zstd.ZstdError as exc: - sweep.append({"family": family, "config": config, "training_error": str(exc)}) - continue - fit_s = time.perf_counter() - start - scores, times = scores_for(model, val) - pred = predictions(scores, [x for x, _ in val]) - acc = float(accuracy_score([y for _, y in val], pred)) - entry = {"family": family, "config": config, "validation_accuracy": acc, - "fit_s": fit_s, "dictionary_or_sample_bytes": model.bytes} - sweep.append(entry) - print(name, family, config, f"validation={acc:.3f}", flush=True) - if family not in best or acc > best[family][0]["validation_accuracy"]: - best[family] = entry, model, scores - report = {"dataset": meta, "validation_sweep": sweep, "selected": {}} - for family, (entry, model, scores) in best.items(): - temp = temperature(scores, [y for _, y in val]) - test_scores, times = scores_for(model, test, temp) - summary, rows = metrics(test_scores, test, times, temp) - # OOD-like inputs must yield a finite score for every allowed class. - for text in ["", "🙂 café 你好", "x" * 50000]: - edge = model.score(text) - assert edge.shape == (classes,) and np.isfinite(edge).all() - report["selected"][family] = {**entry, "temperature": temp, **summary} - (outdir / f"{name}-{family}-predictions.json").write_text(json.dumps(rows)) - print(name, family, "TEST", json.dumps(summary), flush=True) - (outdir / f"{name}.json").write_text(json.dumps(report, indent=2)) - return report - - -def predict_jevbench(state, question, labels): - if not isinstance(state, str): - state = json.dumps(state, ensure_ascii=False, sort_keys=True) - x = normalize(state + "\n" + question["instructions"]).encode() - cx = len(gzip.compress(x, mtime=0)) - criteria = question.get("criteria") or {} - if isinstance(criteria, list): - criteria = {str(i): description for i, description in enumerate(criteria)} - scores = [] - for label in labels: - key = {"yes": "true", "no": "false"}.get(label, label) if question["type"] == "noul" else label - description = criteria.get(key) - if description is None: - description = "" - z = normalize(label.replace("_", " ") + ": " + str(description)).encode() - cz = len(gzip.compress(z, mtime=0)) - # Symmetric NCD: both concat directions, normalized for label length. - joint = min(len(gzip.compress(x + b" " + z, mtime=0)), len(gzip.compress(z + b" " + x, mtime=0))) - scores.append(-(joint - min(cx, cz)) / max(cx, cz)) - scores = np.asarray(scores) - ties = np.flatnonzero(scores == scores.max()) - # Choose by label name hash so option permutation does not change ties. - index = min(ties, key=lambda i: hashlib.sha256(x + labels[i].encode()).hexdigest()) - return labels[index], len(ties) > 1 - - -def jevbench(root, outdir): - results = {} - for tier in ["easy", "original", "hard"]: - path = root / "datasets" / "public" / f"{tier}.jsonl" - rows = [json.loads(line) for line in path.read_text().splitlines() if line.strip()] - predictions_out = [] - if any(r["expected"] is None or r.get("provenance", {}).get("exclude_reason") for r in rows): - raise ValueError("This diagnostic requires fully scored public cohorts") - for row in rows: - start = time.perf_counter_ns() - prediction, tie = predict_jevbench(row["state"], row["question"], row["labels"]) - predictions_out.append({"id": row["id"], "group": row.get("group"), "family": row["family"], - "gold": str(row["expected"]), "prediction": prediction, - "tie": tie, "chance": 1 / len(row["labels"]), - "ms": (time.perf_counter_ns() - start) / 1e6}) - family_scores = {} - for family in sorted({r["family"] for r in predictions_out}): - group = [r for r in predictions_out if r["family"] == family] - family_scores[family] = {"n": len(group), "accuracy": np.mean([r["gold"] == r["prediction"] for r in group])} - results[tier] = {"n": len(rows), "sha256": hashlib.sha256(path.read_bytes()).hexdigest(), - "accuracy": float(np.mean([r["gold"] == r["prediction"] for r in predictions_out])), - "uniform_chance": float(np.mean([r["chance"] for r in predictions_out])), - "p50_ms": float(np.median([r["ms"] for r in predictions_out])), "families": family_scores} - (outdir / f"jevbench-{tier}-predictions.json").write_text(json.dumps(predictions_out, indent=2)) - results["source_commit"] = subprocess.check_output(["git", "-C", str(root), "rev-parse", "HEAD"], text=True).strip() - results["method"] = "Zero-shot symmetric gzip NCD against label and rubric; no training or calibration. Public diagnostic, not an official score." - (outdir / "jevbench.json").write_text(json.dumps(results, indent=2)) - print("JEVBENCH", json.dumps(results), flush=True) - return results - - -def main(): - ap = argparse.ArgumentParser(description=__doc__) - ap.add_argument("--datasets", nargs="+", choices=DATASETS, default=list(DATASETS)) - ap.add_argument("--families", nargs="+", choices=["deflate", "zstd", "zstd-mixture", "gzip-knn", "tfidf-logistic"]) - ap.add_argument("--seed", type=int, default=0) - ap.add_argument("--n", type=int, default=500) - ap.add_argument("--train-cap", type=int, default=10000) - ap.add_argument("--validation-n", type=int, default=600) - ap.add_argument("--out", type=Path, default=Path(__file__).parent / "data" / "compression") - ap.add_argument("--jevbench", type=Path) - args = ap.parse_args() - if min(args.n, args.train_cap, args.validation_n) <= 0: - ap.error("sample counts must be positive") - args.out.mkdir(parents=True, exist_ok=False) - metadata = {"args": {k: str(v) if isinstance(v, Path) else v for k, v in vars(args).items()}, - "platform": platform.platform(), "processor": platform.processor(), - "python": platform.python_version(), "zlib": zlib.ZLIB_RUNTIME_VERSION, - "thread_environment": {k: os.environ.get(k) for k in ["OMP_NUM_THREADS", "OPENBLAS_NUM_THREADS"]}, - "timing": "Serial local warm inference: normalization, compression/features, probabilities, tie-breaking. Excludes training, network and data loading.", - "versions": {k: importlib.metadata.version(k) for k in ["datasets", "numpy", "scipy", "scikit-learn", "zstandard"]}, - "script_sha256": hashlib.sha256(Path(__file__).read_bytes()).hexdigest()} - (args.out / "manifest.json").write_text(json.dumps(metadata, indent=2)) - summary = {"manifest": metadata, "datasets": {}} - for name in args.datasets: - summary["datasets"][name] = run_dataset(name, args, args.out) - if args.jevbench: - summary["jevbench"] = jevbench(args.jevbench, args.out) - (args.out / "summary.json").write_text(json.dumps(summary, indent=2)) - - -if __name__ == "__main__": - main() diff --git a/eval/jevbench-zero-shot-results.json b/eval/jevbench-zero-shot-results.json new file mode 100644 index 0000000..212ca1e --- /dev/null +++ b/eval/jevbench-zero-shot-results.json @@ -0,0 +1,2093 @@ +{ + "manifest": { + "source_sha256": { + "easy": "231df3c2c8e88a1a8c137ebe85de96ba70fabd330849098ac7b3c52c70b7172b", + "original": "5c2414edb3006b8bfcb70fda433f0f9ca015759433849f8d3104328a1f7c4180", + "hard": "89e9e6becb33ed88c1de7d42dcc87531b2fb64cfaef4e1986faf7c37b3f80ebb" + }, + "script_sha256": "7162b92992aa7ac4706b03e6ee957c5c987059c219810464e75aeb9bdcb02eed", + "platform": "macOS-26.5.1-arm64-arm-64bit", + "python": "3.12.14", + "features": [ + "word_cosine", + "word_window_cosine", + "char_cosine", + "gzip_ncd", + "deflate_gain", + "zstd_gain", + "label_word_cosine" + ], + "versions": { + "numpy": "2.5.3", + "scipy": "1.18.1", + "scikit-learn": "1.9.1", + "zstandard": "0.25.0" + }, + "all_methods_p50_ms": 1.533333, + "all_methods_p95_ms": 11.855979000000001, + "option_reversal_predictions_checked": 12474, + "note": "Public zero-shot diagnostic only; no gold answers or labeled examples enter prediction. IDF is computed per request. All methods are fixed; multiple exploratory comparisons, no official sealed evaluation. Timing includes every method together, excluding verification and loading." + }, + "uniform_chance": 0.3176046176046176, + "versus_tfidf": { + "tfidf-word": { + "accuracy_difference": -0.03463203463203463, + "scenario_bootstrap_95": [ + -0.07860262008733625, + 0.008928571428571428 + ] + }, + "tfidf-window": { + "accuracy_difference": -0.04329004329004329, + "scenario_bootstrap_95": [ + -0.08334205020920502, + -0.004273504273504274 + ] + }, + "tfidf-char": { + "accuracy_difference": 0.004329004329004329, + "scenario_bootstrap_95": [ + -0.047619047619047616, + 0.05803571428571429 + ] + }, + "tfidf-label": { + "accuracy_difference": -0.0735930735930736, + "scenario_bootstrap_95": [ + -0.15625679347826088, + 0.00847457627118644 + ] + }, + "gzip": { + "accuracy_difference": -0.06926406926406926, + "scenario_bootstrap_95": [ + -0.14979060709761585, + 0.013215859030837005 + ] + }, + "deflate": { + "accuracy_difference": -0.09956709956709957, + "scenario_bootstrap_95": [ + -0.17467787416531222, + -0.021929824561403508 + ] + }, + "zstd": { + "accuracy_difference": -0.09090909090909091, + "scenario_bootstrap_95": [ + -0.1630901287553648, + -0.017167381974248927 + ] + }, + "tfidf-word+gzip:0.25": { + "accuracy_difference": -0.05194805194805195, + "scenario_bootstrap_95": [ + -0.12664784522872072, + 0.025423728813559324 + ] + }, + "tfidf-word+gzip:0.5": { + "accuracy_difference": -0.03896103896103896, + "scenario_bootstrap_95": [ + -0.09333333333333334, + 0.013217320962145705 + ] + }, + "tfidf-word+gzip:0.75": { + "accuracy_difference": -0.008658008658008658, + "scenario_bootstrap_95": [ + -0.05194805194805195, + 0.035398230088495575 + ] + }, + "tfidf-word+deflate:0.25": { + "accuracy_difference": -0.08658008658008658, + "scenario_bootstrap_95": [ + -0.16240822320117473, + -0.009048754636989946 + ] + }, + "tfidf-word+deflate:0.5": { + "accuracy_difference": -0.06060606060606061, + "scenario_bootstrap_95": [ + -0.11555555555555555, + -0.00851063829787234 + ] + }, + "tfidf-word+deflate:0.75": { + "accuracy_difference": -0.03463203463203463, + "scenario_bootstrap_95": [ + -0.07488986784140969, + 0.004566733693603109 + ] + }, + "tfidf-word+zstd:0.25": { + "accuracy_difference": -0.09523809523809523, + "scenario_bootstrap_95": [ + -0.1630901287553648, + -0.02643171806167401 + ] + }, + "tfidf-word+zstd:0.5": { + "accuracy_difference": -0.05627705627705628, + "scenario_bootstrap_95": [ + -0.11013215859030837, + 0.0 + ] + }, + "tfidf-word+zstd:0.75": { + "accuracy_difference": -0.030303030303030304, + "scenario_bootstrap_95": [ + -0.07017543859649122, + 0.008849557522123894 + ] + }, + "tfidf-window+gzip:0.25": { + "accuracy_difference": -0.04329004329004329, + "scenario_bootstrap_95": [ + -0.11666666666666667, + 0.03389830508474576 + ] + }, + "tfidf-window+gzip:0.5": { + "accuracy_difference": -0.03896103896103896, + "scenario_bootstrap_95": [ + -0.09210526315789473, + 0.013452914798206279 + ] + }, + "tfidf-window+gzip:0.75": { + "accuracy_difference": -0.021645021645021644, + "scenario_bootstrap_95": [ + -0.06222222222222222, + 0.01832262247026242 + ] + }, + "tfidf-window+deflate:0.25": { + "accuracy_difference": -0.08658008658008658, + "scenario_bootstrap_95": [ + -0.16033755274261605, + -0.01293103448275862 + ] + }, + "tfidf-window+deflate:0.5": { + "accuracy_difference": -0.04329004329004329, + "scenario_bootstrap_95": [ + -0.09649122807017543, + 0.012450736991961628 + ] + }, + "tfidf-window+deflate:0.75": { + "accuracy_difference": -0.04329004329004329, + "scenario_bootstrap_95": [ + -0.08299619640387275, + -0.004328537841468883 + ] + }, + "tfidf-window+zstd:0.25": { + "accuracy_difference": -0.08658008658008658, + "scenario_bootstrap_95": [ + -0.15418502202643172, + -0.021092632805465875 + ] + }, + "tfidf-window+zstd:0.5": { + "accuracy_difference": -0.05627705627705628, + "scenario_bootstrap_95": [ + -0.1091703056768559, + -0.004366812227074236 + ] + }, + "tfidf-window+zstd:0.75": { + "accuracy_difference": -0.03896103896103896, + "scenario_bootstrap_95": [ + -0.07555555555555556, + -0.00423728813559322 + ] + }, + "tfidf-char+gzip:0.25": { + "accuracy_difference": -0.03896103896103896, + "scenario_bootstrap_95": [ + -0.11688311688311688, + 0.039473684210526314 + ] + }, + "tfidf-char+gzip:0.5": { + "accuracy_difference": -0.030303030303030304, + "scenario_bootstrap_95": [ + -0.09583333333333334, + 0.035398230088495575 + ] + }, + "tfidf-char+gzip:0.75": { + "accuracy_difference": -0.008658008658008658, + "scenario_bootstrap_95": [ + -0.06809238043280597, + 0.04911264814221645 + ] + }, + "tfidf-char+deflate:0.25": { + "accuracy_difference": -0.09523809523809523, + "scenario_bootstrap_95": [ + -0.16956521739130434, + -0.020918322873082318 + ] + }, + "tfidf-char+deflate:0.5": { + "accuracy_difference": -0.05627705627705628, + "scenario_bootstrap_95": [ + -0.1227292663476874, + 0.008438818565400843 + ] + }, + "tfidf-char+deflate:0.75": { + "accuracy_difference": -0.012987012987012988, + "scenario_bootstrap_95": [ + -0.06722783283717074, + 0.04310344827586207 + ] + }, + "tfidf-char+zstd:0.25": { + "accuracy_difference": -0.09090909090909091, + "scenario_bootstrap_95": [ + -0.15450643776824036, + -0.02631291657090328 + ] + }, + "tfidf-char+zstd:0.5": { + "accuracy_difference": -0.05627705627705628, + "scenario_bootstrap_95": [ + -0.11063829787234042, + -0.004166666666666667 + ] + }, + "tfidf-char+zstd:0.75": { + "accuracy_difference": 0.0, + "scenario_bootstrap_95": [ + -0.05240174672489083, + 0.05454736440030555 + ] + }, + "tfidf-label+gzip:0.25": { + "accuracy_difference": -0.06926406926406926, + "scenario_bootstrap_95": [ + -0.14977973568281938, + 0.01276595744680851 + ] + }, + "tfidf-label+gzip:0.5": { + "accuracy_difference": -0.03896103896103896, + "scenario_bootstrap_95": [ + -0.11392405063291139, + 0.03508771929824561 + ] + }, + "tfidf-label+gzip:0.75": { + "accuracy_difference": -0.030303030303030304, + "scenario_bootstrap_95": [ + -0.10460251046025104, + 0.043478260869565216 + ] + }, + "tfidf-label+deflate:0.25": { + "accuracy_difference": -0.09956709956709957, + "scenario_bootstrap_95": [ + -0.17672413793103448, + -0.02127659574468085 + ] + }, + "tfidf-label+deflate:0.5": { + "accuracy_difference": -0.06493506493506493, + "scenario_bootstrap_95": [ + -0.13393533549783548, + 0.004291845493562232 + ] + }, + "tfidf-label+deflate:0.75": { + "accuracy_difference": -0.03896103896103896, + "scenario_bootstrap_95": [ + -0.10666666666666667, + 0.030042918454935622 + ] + }, + "tfidf-label+zstd:0.25": { + "accuracy_difference": -0.06926406926406926, + "scenario_bootstrap_95": [ + -0.13964209843030803, + 0.0 + ] + }, + "tfidf-label+zstd:0.5": { + "accuracy_difference": -0.05194805194805195, + "scenario_bootstrap_95": [ + -0.11914893617021277, + 0.01694915254237288 + ] + }, + "tfidf-label+zstd:0.75": { + "accuracy_difference": -0.03896103896103896, + "scenario_bootstrap_95": [ + -0.10300765988567909, + 0.02575107296137339 + ] + }, + "tfidf+gzip:0.25": { + "accuracy_difference": -0.047619047619047616, + "scenario_bootstrap_95": [ + -0.12196151075434804, + 0.02643464192429141 + ] + }, + "tfidf+gzip:0.5": { + "accuracy_difference": -0.03896103896103896, + "scenario_bootstrap_95": [ + -0.10177885520617587, + 0.022323930973734748 + ] + }, + "tfidf+gzip:0.75": { + "accuracy_difference": 0.0, + "scenario_bootstrap_95": [ + -0.047619047619047616, + 0.04782608695652174 + ] + }, + "tfidf+deflate:0.25": { + "accuracy_difference": -0.08658008658008658, + "scenario_bootstrap_95": [ + -0.16017316017316016, + -0.012875536480686695 + ] + }, + "tfidf+deflate:0.5": { + "accuracy_difference": -0.07792207792207792, + "scenario_bootstrap_95": [ + -0.14102564102564102, + -0.013391369047619069 + ] + }, + "tfidf+deflate:0.75": { + "accuracy_difference": -0.012987012987012988, + "scenario_bootstrap_95": [ + -0.05932834475297512, + 0.034782608695652174 + ] + }, + "tfidf+zstd:0.25": { + "accuracy_difference": -0.08225108225108226, + "scenario_bootstrap_95": [ + -0.14530102790014685, + -0.021274341868013014 + ] + }, + "tfidf+zstd:0.5": { + "accuracy_difference": -0.06060606060606061, + "scenario_bootstrap_95": [ + -0.11441004629283955, + -0.008657075682937766 + ] + }, + "tfidf+zstd:0.75": { + "accuracy_difference": 0.017316017316017316, + "scenario_bootstrap_95": [ + -0.013215859030837005, + 0.04954954954954955 + ] + }, + "all-four": { + "accuracy_difference": -0.06493506493506493, + "scenario_bootstrap_95": [ + -0.12334801762114538, + -0.004347826086956522 + ] + } + }, + "published_jev_reference": { + "source_sha256": "c772b3b85809f483449ec2a6ae0272347371ed5f40f7faff8394ecf4ee22ae7c", + "note": "Published Jev 1.13.0 outcomes on these exact public IDs; not rerun.", + "n": 231, + "correct": 200, + "accuracy": 0.8658008658008658, + "by_tier": { + "easy": { + "n": 48, + "correct": 48, + "accuracy": 1.0 + }, + "hard": { + "n": 111, + "correct": 81, + "accuracy": 0.7297297297297297 + }, + "original": { + "n": 72, + "correct": 71, + "accuracy": 0.9861111111111112 + } + }, + "by_family": { + "adequacy": { + "n": 12, + "correct": 12, + "accuracy": 1.0 + }, + "adversarial": { + "n": 6, + "correct": 6, + "accuracy": 1.0 + }, + "ambiguous": { + "n": 7, + "correct": 6, + "accuracy": 0.8571428571428571 + }, + "extraction": { + "n": 24, + "correct": 24, + "accuracy": 1.0 + }, + "fact": { + "n": 12, + "correct": 12, + "accuracy": 1.0 + }, + "intent": { + "n": 24, + "correct": 24, + "accuracy": 1.0 + }, + "judge_hard": { + "n": 17, + "correct": 13, + "accuracy": 0.7647058823529411 + }, + "long_policy": { + "n": 19, + "correct": 12, + "accuracy": 0.631578947368421 + }, + "multi_hop": { + "n": 18, + "correct": 15, + "accuracy": 0.8333333333333334 + }, + "ordinal": { + "n": 12, + "correct": 12, + "accuracy": 1.0 + }, + "policy": { + "n": 12, + "correct": 11, + "accuracy": 0.9166666666666666 + }, + "probability": { + "n": 10, + "correct": 7, + "accuracy": 0.7 + }, + "routing": { + "n": 12, + "correct": 12, + "accuracy": 1.0 + }, + "routing_hard": { + "n": 5, + "correct": 5, + "accuracy": 1.0 + }, + "temporal_numeric": { + "n": 15, + "correct": 4, + "accuracy": 0.26666666666666666 + }, + "tool_selection": { + "n": 12, + "correct": 12, + "accuracy": 1.0 + }, + "tradeoff": { + "n": 6, + "correct": 5, + "accuracy": 0.8333333333333334 + }, + "trap": { + "n": 8, + "correct": 8, + "accuracy": 1.0 + } + } + }, + "methods": { + "tfidf-word": { + "n": 231, + "correct": 105, + "accuracy": 0.45454545454545453, + "by_tier": { + "easy": { + "n": 48, + "correct": 38, + "accuracy": 0.7916666666666666 + }, + "hard": { + "n": 111, + "correct": 41, + "accuracy": 0.36936936936936937 + }, + "original": { + "n": 72, + "correct": 26, + "accuracy": 0.3611111111111111 + } + } + }, + "tfidf-window": { + "n": 231, + "correct": 103, + "accuracy": 0.4458874458874459, + "by_tier": { + "easy": { + "n": 48, + "correct": 38, + "accuracy": 0.7916666666666666 + }, + "hard": { + "n": 111, + "correct": 39, + "accuracy": 0.35135135135135137 + }, + "original": { + "n": 72, + "correct": 26, + "accuracy": 0.3611111111111111 + } + } + }, + "tfidf-char": { + "n": 231, + "correct": 114, + "accuracy": 0.4935064935064935, + "by_tier": { + "easy": { + "n": 48, + "correct": 38, + "accuracy": 0.7916666666666666 + }, + "hard": { + "n": 111, + "correct": 48, + "accuracy": 0.43243243243243246 + }, + "original": { + "n": 72, + "correct": 28, + "accuracy": 0.3888888888888889 + } + } + }, + "tfidf-label": { + "n": 231, + "correct": 96, + "accuracy": 0.4155844155844156, + "by_tier": { + "easy": { + "n": 48, + "correct": 35, + "accuracy": 0.7291666666666666 + }, + "hard": { + "n": 111, + "correct": 35, + "accuracy": 0.3153153153153153 + }, + "original": { + "n": 72, + "correct": 26, + "accuracy": 0.3611111111111111 + } + } + }, + "tfidf": { + "n": 231, + "correct": 113, + "accuracy": 0.48917748917748916, + "by_tier": { + "easy": { + "n": 48, + "correct": 38, + "accuracy": 0.7916666666666666 + }, + "hard": { + "n": 111, + "correct": 46, + "accuracy": 0.4144144144144144 + }, + "original": { + "n": 72, + "correct": 29, + "accuracy": 0.4027777777777778 + } + } + }, + "gzip": { + "n": 231, + "correct": 97, + "accuracy": 0.4199134199134199, + "by_tier": { + "easy": { + "n": 48, + "correct": 25, + "accuracy": 0.5208333333333334 + }, + "hard": { + "n": 111, + "correct": 47, + "accuracy": 0.42342342342342343 + }, + "original": { + "n": 72, + "correct": 25, + "accuracy": 0.3472222222222222 + } + } + }, + "deflate": { + "n": 231, + "correct": 90, + "accuracy": 0.38961038961038963, + "by_tier": { + "easy": { + "n": 48, + "correct": 32, + "accuracy": 0.6666666666666666 + }, + "hard": { + "n": 111, + "correct": 34, + "accuracy": 0.3063063063063063 + }, + "original": { + "n": 72, + "correct": 24, + "accuracy": 0.3333333333333333 + } + } + }, + "zstd": { + "n": 231, + "correct": 92, + "accuracy": 0.39826839826839827, + "by_tier": { + "easy": { + "n": 48, + "correct": 25, + "accuracy": 0.5208333333333334 + }, + "hard": { + "n": 111, + "correct": 37, + "accuracy": 0.3333333333333333 + }, + "original": { + "n": 72, + "correct": 30, + "accuracy": 0.4166666666666667 + } + } + }, + "tfidf-word+gzip:0.25": { + "n": 231, + "correct": 101, + "accuracy": 0.43722943722943725, + "by_tier": { + "easy": { + "n": 48, + "correct": 30, + "accuracy": 0.625 + }, + "hard": { + "n": 111, + "correct": 44, + "accuracy": 0.3963963963963964 + }, + "original": { + "n": 72, + "correct": 27, + "accuracy": 0.375 + } + } + }, + "tfidf-word+gzip:0.5": { + "n": 231, + "correct": 104, + "accuracy": 0.45021645021645024, + "by_tier": { + "easy": { + "n": 48, + "correct": 34, + "accuracy": 0.7083333333333334 + }, + "hard": { + "n": 111, + "correct": 44, + "accuracy": 0.3963963963963964 + }, + "original": { + "n": 72, + "correct": 26, + "accuracy": 0.3611111111111111 + } + } + }, + "tfidf-word+gzip:0.75": { + "n": 231, + "correct": 111, + "accuracy": 0.4805194805194805, + "by_tier": { + "easy": { + "n": 48, + "correct": 40, + "accuracy": 0.8333333333333334 + }, + "hard": { + "n": 111, + "correct": 44, + "accuracy": 0.3963963963963964 + }, + "original": { + "n": 72, + "correct": 27, + "accuracy": 0.375 + } + } + }, + "tfidf-word+deflate:0.25": { + "n": 231, + "correct": 93, + "accuracy": 0.4025974025974026, + "by_tier": { + "easy": { + "n": 48, + "correct": 33, + "accuracy": 0.6875 + }, + "hard": { + "n": 111, + "correct": 35, + "accuracy": 0.3153153153153153 + }, + "original": { + "n": 72, + "correct": 25, + "accuracy": 0.3472222222222222 + } + } + }, + "tfidf-word+deflate:0.5": { + "n": 231, + "correct": 99, + "accuracy": 0.42857142857142855, + "by_tier": { + "easy": { + "n": 48, + "correct": 37, + "accuracy": 0.7708333333333334 + }, + "hard": { + "n": 111, + "correct": 34, + "accuracy": 0.3063063063063063 + }, + "original": { + "n": 72, + "correct": 28, + "accuracy": 0.3888888888888889 + } + } + }, + "tfidf-word+deflate:0.75": { + "n": 231, + "correct": 105, + "accuracy": 0.45454545454545453, + "by_tier": { + "easy": { + "n": 48, + "correct": 38, + "accuracy": 0.7916666666666666 + }, + "hard": { + "n": 111, + "correct": 40, + "accuracy": 0.36036036036036034 + }, + "original": { + "n": 72, + "correct": 27, + "accuracy": 0.375 + } + } + }, + "tfidf-word+zstd:0.25": { + "n": 231, + "correct": 91, + "accuracy": 0.3939393939393939, + "by_tier": { + "easy": { + "n": 48, + "correct": 29, + "accuracy": 0.6041666666666666 + }, + "hard": { + "n": 111, + "correct": 38, + "accuracy": 0.34234234234234234 + }, + "original": { + "n": 72, + "correct": 24, + "accuracy": 0.3333333333333333 + } + } + }, + "tfidf-word+zstd:0.5": { + "n": 231, + "correct": 100, + "accuracy": 0.4329004329004329, + "by_tier": { + "easy": { + "n": 48, + "correct": 35, + "accuracy": 0.7291666666666666 + }, + "hard": { + "n": 111, + "correct": 40, + "accuracy": 0.36036036036036034 + }, + "original": { + "n": 72, + "correct": 25, + "accuracy": 0.3472222222222222 + } + } + }, + "tfidf-word+zstd:0.75": { + "n": 231, + "correct": 106, + "accuracy": 0.4588744588744589, + "by_tier": { + "easy": { + "n": 48, + "correct": 38, + "accuracy": 0.7916666666666666 + }, + "hard": { + "n": 111, + "correct": 41, + "accuracy": 0.36936936936936937 + }, + "original": { + "n": 72, + "correct": 27, + "accuracy": 0.375 + } + } + }, + "tfidf-window+gzip:0.25": { + "n": 231, + "correct": 103, + "accuracy": 0.4458874458874459, + "by_tier": { + "easy": { + "n": 48, + "correct": 30, + "accuracy": 0.625 + }, + "hard": { + "n": 111, + "correct": 46, + "accuracy": 0.4144144144144144 + }, + "original": { + "n": 72, + "correct": 27, + "accuracy": 0.375 + } + } + }, + "tfidf-window+gzip:0.5": { + "n": 231, + "correct": 104, + "accuracy": 0.45021645021645024, + "by_tier": { + "easy": { + "n": 48, + "correct": 34, + "accuracy": 0.7083333333333334 + }, + "hard": { + "n": 111, + "correct": 44, + "accuracy": 0.3963963963963964 + }, + "original": { + "n": 72, + "correct": 26, + "accuracy": 0.3611111111111111 + } + } + }, + "tfidf-window+gzip:0.75": { + "n": 231, + "correct": 108, + "accuracy": 0.4675324675324675, + "by_tier": { + "easy": { + "n": 48, + "correct": 40, + "accuracy": 0.8333333333333334 + }, + "hard": { + "n": 111, + "correct": 41, + "accuracy": 0.36936936936936937 + }, + "original": { + "n": 72, + "correct": 27, + "accuracy": 0.375 + } + } + }, + "tfidf-window+deflate:0.25": { + "n": 231, + "correct": 93, + "accuracy": 0.4025974025974026, + "by_tier": { + "easy": { + "n": 48, + "correct": 33, + "accuracy": 0.6875 + }, + "hard": { + "n": 111, + "correct": 35, + "accuracy": 0.3153153153153153 + }, + "original": { + "n": 72, + "correct": 25, + "accuracy": 0.3472222222222222 + } + } + }, + "tfidf-window+deflate:0.5": { + "n": 231, + "correct": 103, + "accuracy": 0.4458874458874459, + "by_tier": { + "easy": { + "n": 48, + "correct": 37, + "accuracy": 0.7708333333333334 + }, + "hard": { + "n": 111, + "correct": 38, + "accuracy": 0.34234234234234234 + }, + "original": { + "n": 72, + "correct": 28, + "accuracy": 0.3888888888888889 + } + } + }, + "tfidf-window+deflate:0.75": { + "n": 231, + "correct": 103, + "accuracy": 0.4458874458874459, + "by_tier": { + "easy": { + "n": 48, + "correct": 38, + "accuracy": 0.7916666666666666 + }, + "hard": { + "n": 111, + "correct": 38, + "accuracy": 0.34234234234234234 + }, + "original": { + "n": 72, + "correct": 27, + "accuracy": 0.375 + } + } + }, + "tfidf-window+zstd:0.25": { + "n": 231, + "correct": 93, + "accuracy": 0.4025974025974026, + "by_tier": { + "easy": { + "n": 48, + "correct": 29, + "accuracy": 0.6041666666666666 + }, + "hard": { + "n": 111, + "correct": 40, + "accuracy": 0.36036036036036034 + }, + "original": { + "n": 72, + "correct": 24, + "accuracy": 0.3333333333333333 + } + } + }, + "tfidf-window+zstd:0.5": { + "n": 231, + "correct": 100, + "accuracy": 0.4329004329004329, + "by_tier": { + "easy": { + "n": 48, + "correct": 35, + "accuracy": 0.7291666666666666 + }, + "hard": { + "n": 111, + "correct": 40, + "accuracy": 0.36036036036036034 + }, + "original": { + "n": 72, + "correct": 25, + "accuracy": 0.3472222222222222 + } + } + }, + "tfidf-window+zstd:0.75": { + "n": 231, + "correct": 104, + "accuracy": 0.45021645021645024, + "by_tier": { + "easy": { + "n": 48, + "correct": 38, + "accuracy": 0.7916666666666666 + }, + "hard": { + "n": 111, + "correct": 39, + "accuracy": 0.35135135135135137 + }, + "original": { + "n": 72, + "correct": 27, + "accuracy": 0.375 + } + } + }, + "tfidf-char+gzip:0.25": { + "n": 231, + "correct": 104, + "accuracy": 0.45021645021645024, + "by_tier": { + "easy": { + "n": 48, + "correct": 27, + "accuracy": 0.5625 + }, + "hard": { + "n": 111, + "correct": 48, + "accuracy": 0.43243243243243246 + }, + "original": { + "n": 72, + "correct": 29, + "accuracy": 0.4027777777777778 + } + } + }, + "tfidf-char+gzip:0.5": { + "n": 231, + "correct": 106, + "accuracy": 0.4588744588744589, + "by_tier": { + "easy": { + "n": 48, + "correct": 31, + "accuracy": 0.6458333333333334 + }, + "hard": { + "n": 111, + "correct": 48, + "accuracy": 0.43243243243243246 + }, + "original": { + "n": 72, + "correct": 27, + "accuracy": 0.375 + } + } + }, + "tfidf-char+gzip:0.75": { + "n": 231, + "correct": 111, + "accuracy": 0.4805194805194805, + "by_tier": { + "easy": { + "n": 48, + "correct": 37, + "accuracy": 0.7708333333333334 + }, + "hard": { + "n": 111, + "correct": 48, + "accuracy": 0.43243243243243246 + }, + "original": { + "n": 72, + "correct": 26, + "accuracy": 0.3611111111111111 + } + } + }, + "tfidf-char+deflate:0.25": { + "n": 231, + "correct": 91, + "accuracy": 0.3939393939393939, + "by_tier": { + "easy": { + "n": 48, + "correct": 32, + "accuracy": 0.6666666666666666 + }, + "hard": { + "n": 111, + "correct": 34, + "accuracy": 0.3063063063063063 + }, + "original": { + "n": 72, + "correct": 25, + "accuracy": 0.3472222222222222 + } + } + }, + "tfidf-char+deflate:0.5": { + "n": 231, + "correct": 100, + "accuracy": 0.4329004329004329, + "by_tier": { + "easy": { + "n": 48, + "correct": 36, + "accuracy": 0.75 + }, + "hard": { + "n": 111, + "correct": 36, + "accuracy": 0.32432432432432434 + }, + "original": { + "n": 72, + "correct": 28, + "accuracy": 0.3888888888888889 + } + } + }, + "tfidf-char+deflate:0.75": { + "n": 231, + "correct": 110, + "accuracy": 0.47619047619047616, + "by_tier": { + "easy": { + "n": 48, + "correct": 38, + "accuracy": 0.7916666666666666 + }, + "hard": { + "n": 111, + "correct": 44, + "accuracy": 0.3963963963963964 + }, + "original": { + "n": 72, + "correct": 28, + "accuracy": 0.3888888888888889 + } + } + }, + "tfidf-char+zstd:0.25": { + "n": 231, + "correct": 92, + "accuracy": 0.39826839826839827, + "by_tier": { + "easy": { + "n": 48, + "correct": 28, + "accuracy": 0.5833333333333334 + }, + "hard": { + "n": 111, + "correct": 38, + "accuracy": 0.34234234234234234 + }, + "original": { + "n": 72, + "correct": 26, + "accuracy": 0.3611111111111111 + } + } + }, + "tfidf-char+zstd:0.5": { + "n": 231, + "correct": 100, + "accuracy": 0.4329004329004329, + "by_tier": { + "easy": { + "n": 48, + "correct": 34, + "accuracy": 0.7083333333333334 + }, + "hard": { + "n": 111, + "correct": 41, + "accuracy": 0.36936936936936937 + }, + "original": { + "n": 72, + "correct": 25, + "accuracy": 0.3472222222222222 + } + } + }, + "tfidf-char+zstd:0.75": { + "n": 231, + "correct": 113, + "accuracy": 0.48917748917748916, + "by_tier": { + "easy": { + "n": 48, + "correct": 38, + "accuracy": 0.7916666666666666 + }, + "hard": { + "n": 111, + "correct": 47, + "accuracy": 0.42342342342342343 + }, + "original": { + "n": 72, + "correct": 28, + "accuracy": 0.3888888888888889 + } + } + }, + "tfidf-label+gzip:0.25": { + "n": 231, + "correct": 97, + "accuracy": 0.4199134199134199, + "by_tier": { + "easy": { + "n": 48, + "correct": 27, + "accuracy": 0.5625 + }, + "hard": { + "n": 111, + "correct": 45, + "accuracy": 0.40540540540540543 + }, + "original": { + "n": 72, + "correct": 25, + "accuracy": 0.3472222222222222 + } + } + }, + "tfidf-label+gzip:0.5": { + "n": 231, + "correct": 104, + "accuracy": 0.45021645021645024, + "by_tier": { + "easy": { + "n": 48, + "correct": 34, + "accuracy": 0.7083333333333334 + }, + "hard": { + "n": 111, + "correct": 45, + "accuracy": 0.40540540540540543 + }, + "original": { + "n": 72, + "correct": 25, + "accuracy": 0.3472222222222222 + } + } + }, + "tfidf-label+gzip:0.75": { + "n": 231, + "correct": 106, + "accuracy": 0.4588744588744589, + "by_tier": { + "easy": { + "n": 48, + "correct": 39, + "accuracy": 0.8125 + }, + "hard": { + "n": 111, + "correct": 42, + "accuracy": 0.3783783783783784 + }, + "original": { + "n": 72, + "correct": 25, + "accuracy": 0.3472222222222222 + } + } + }, + "tfidf-label+deflate:0.25": { + "n": 231, + "correct": 90, + "accuracy": 0.38961038961038963, + "by_tier": { + "easy": { + "n": 48, + "correct": 33, + "accuracy": 0.6875 + }, + "hard": { + "n": 111, + "correct": 33, + "accuracy": 0.2972972972972973 + }, + "original": { + "n": 72, + "correct": 24, + "accuracy": 0.3333333333333333 + } + } + }, + "tfidf-label+deflate:0.5": { + "n": 231, + "correct": 98, + "accuracy": 0.42424242424242425, + "by_tier": { + "easy": { + "n": 48, + "correct": 38, + "accuracy": 0.7916666666666666 + }, + "hard": { + "n": 111, + "correct": 34, + "accuracy": 0.3063063063063063 + }, + "original": { + "n": 72, + "correct": 26, + "accuracy": 0.3611111111111111 + } + } + }, + "tfidf-label+deflate:0.75": { + "n": 231, + "correct": 104, + "accuracy": 0.45021645021645024, + "by_tier": { + "easy": { + "n": 48, + "correct": 39, + "accuracy": 0.8125 + }, + "hard": { + "n": 111, + "correct": 39, + "accuracy": 0.35135135135135137 + }, + "original": { + "n": 72, + "correct": 26, + "accuracy": 0.3611111111111111 + } + } + }, + "tfidf-label+zstd:0.25": { + "n": 231, + "correct": 97, + "accuracy": 0.4199134199134199, + "by_tier": { + "easy": { + "n": 48, + "correct": 30, + "accuracy": 0.625 + }, + "hard": { + "n": 111, + "correct": 38, + "accuracy": 0.34234234234234234 + }, + "original": { + "n": 72, + "correct": 29, + "accuracy": 0.4027777777777778 + } + } + }, + "tfidf-label+zstd:0.5": { + "n": 231, + "correct": 101, + "accuracy": 0.43722943722943725, + "by_tier": { + "easy": { + "n": 48, + "correct": 35, + "accuracy": 0.7291666666666666 + }, + "hard": { + "n": 111, + "correct": 35, + "accuracy": 0.3153153153153153 + }, + "original": { + "n": 72, + "correct": 31, + "accuracy": 0.4305555555555556 + } + } + }, + "tfidf-label+zstd:0.75": { + "n": 231, + "correct": 104, + "accuracy": 0.45021645021645024, + "by_tier": { + "easy": { + "n": 48, + "correct": 37, + "accuracy": 0.7708333333333334 + }, + "hard": { + "n": 111, + "correct": 37, + "accuracy": 0.3333333333333333 + }, + "original": { + "n": 72, + "correct": 30, + "accuracy": 0.4166666666666667 + } + } + }, + "tfidf+gzip:0.25": { + "n": 231, + "correct": 102, + "accuracy": 0.44155844155844154, + "by_tier": { + "easy": { + "n": 48, + "correct": 29, + "accuracy": 0.6041666666666666 + }, + "hard": { + "n": 111, + "correct": 46, + "accuracy": 0.4144144144144144 + }, + "original": { + "n": 72, + "correct": 27, + "accuracy": 0.375 + } + } + }, + "tfidf+gzip:0.5": { + "n": 231, + "correct": 104, + "accuracy": 0.45021645021645024, + "by_tier": { + "easy": { + "n": 48, + "correct": 34, + "accuracy": 0.7083333333333334 + }, + "hard": { + "n": 111, + "correct": 46, + "accuracy": 0.4144144144144144 + }, + "original": { + "n": 72, + "correct": 24, + "accuracy": 0.3333333333333333 + } + } + }, + "tfidf+gzip:0.75": { + "n": 231, + "correct": 113, + "accuracy": 0.48917748917748916, + "by_tier": { + "easy": { + "n": 48, + "correct": 40, + "accuracy": 0.8333333333333334 + }, + "hard": { + "n": 111, + "correct": 47, + "accuracy": 0.42342342342342343 + }, + "original": { + "n": 72, + "correct": 26, + "accuracy": 0.3611111111111111 + } + } + }, + "tfidf+deflate:0.25": { + "n": 231, + "correct": 93, + "accuracy": 0.4025974025974026, + "by_tier": { + "easy": { + "n": 48, + "correct": 34, + "accuracy": 0.7083333333333334 + }, + "hard": { + "n": 111, + "correct": 34, + "accuracy": 0.3063063063063063 + }, + "original": { + "n": 72, + "correct": 25, + "accuracy": 0.3472222222222222 + } + } + }, + "tfidf+deflate:0.5": { + "n": 231, + "correct": 95, + "accuracy": 0.41125541125541126, + "by_tier": { + "easy": { + "n": 48, + "correct": 36, + "accuracy": 0.75 + }, + "hard": { + "n": 111, + "correct": 33, + "accuracy": 0.2972972972972973 + }, + "original": { + "n": 72, + "correct": 26, + "accuracy": 0.3611111111111111 + } + } + }, + "tfidf+deflate:0.75": { + "n": 231, + "correct": 110, + "accuracy": 0.47619047619047616, + "by_tier": { + "easy": { + "n": 48, + "correct": 37, + "accuracy": 0.7708333333333334 + }, + "hard": { + "n": 111, + "correct": 47, + "accuracy": 0.42342342342342343 + }, + "original": { + "n": 72, + "correct": 26, + "accuracy": 0.3611111111111111 + } + } + }, + "tfidf+zstd:0.25": { + "n": 231, + "correct": 94, + "accuracy": 0.4069264069264069, + "by_tier": { + "easy": { + "n": 48, + "correct": 29, + "accuracy": 0.6041666666666666 + }, + "hard": { + "n": 111, + "correct": 39, + "accuracy": 0.35135135135135137 + }, + "original": { + "n": 72, + "correct": 26, + "accuracy": 0.3611111111111111 + } + } + }, + "tfidf+zstd:0.5": { + "n": 231, + "correct": 99, + "accuracy": 0.42857142857142855, + "by_tier": { + "easy": { + "n": 48, + "correct": 35, + "accuracy": 0.7291666666666666 + }, + "hard": { + "n": 111, + "correct": 39, + "accuracy": 0.35135135135135137 + }, + "original": { + "n": 72, + "correct": 25, + "accuracy": 0.3472222222222222 + } + } + }, + "tfidf+zstd:0.75": { + "n": 231, + "correct": 117, + "accuracy": 0.5064935064935064, + "by_tier": { + "easy": { + "n": 48, + "correct": 39, + "accuracy": 0.8125 + }, + "hard": { + "n": 111, + "correct": 48, + "accuracy": 0.43243243243243246 + }, + "original": { + "n": 72, + "correct": 30, + "accuracy": 0.4166666666666667 + } + } + }, + "all-four": { + "n": 231, + "correct": 98, + "accuracy": 0.42424242424242425, + "by_tier": { + "easy": { + "n": 48, + "correct": 37, + "accuracy": 0.7708333333333334 + }, + "hard": { + "n": 111, + "correct": 36, + "accuracy": 0.32432432432432434 + }, + "original": { + "n": 72, + "correct": 25, + "accuracy": 0.3472222222222222 + } + } + } + }, + "family_detail": { + "tfidf-char": { + "adequacy": { + "n": 12, + "correct": 5, + "accuracy": 0.4166666666666667 + }, + "adversarial": { + "n": 6, + "correct": 4, + "accuracy": 0.6666666666666666 + }, + "ambiguous": { + "n": 7, + "correct": 0, + "accuracy": 0.0 + }, + "extraction": { + "n": 24, + "correct": 15, + "accuracy": 0.625 + }, + "fact": { + "n": 12, + "correct": 8, + "accuracy": 0.6666666666666666 + }, + "intent": { + "n": 24, + "correct": 10, + "accuracy": 0.4166666666666667 + }, + "judge_hard": { + "n": 17, + "correct": 7, + "accuracy": 0.4117647058823529 + }, + "long_policy": { + "n": 19, + "correct": 6, + "accuracy": 0.3157894736842105 + }, + "multi_hop": { + "n": 18, + "correct": 5, + "accuracy": 0.2777777777777778 + }, + "ordinal": { + "n": 12, + "correct": 8, + "accuracy": 0.6666666666666666 + }, + "policy": { + "n": 12, + "correct": 6, + "accuracy": 0.5 + }, + "probability": { + "n": 10, + "correct": 4, + "accuracy": 0.4 + }, + "routing": { + "n": 12, + "correct": 3, + "accuracy": 0.25 + }, + "routing_hard": { + "n": 5, + "correct": 5, + "accuracy": 1.0 + }, + "temporal_numeric": { + "n": 15, + "correct": 8, + "accuracy": 0.5333333333333333 + }, + "tool_selection": { + "n": 12, + "correct": 11, + "accuracy": 0.9166666666666666 + }, + "tradeoff": { + "n": 6, + "correct": 5, + "accuracy": 0.8333333333333334 + }, + "trap": { + "n": 8, + "correct": 4, + "accuracy": 0.5 + } + }, + "tfidf": { + "adequacy": { + "n": 12, + "correct": 5, + "accuracy": 0.4166666666666667 + }, + "adversarial": { + "n": 6, + "correct": 3, + "accuracy": 0.5 + }, + "ambiguous": { + "n": 7, + "correct": 2, + "accuracy": 0.2857142857142857 + }, + "extraction": { + "n": 24, + "correct": 16, + "accuracy": 0.6666666666666666 + }, + "fact": { + "n": 12, + "correct": 10, + "accuracy": 0.8333333333333334 + }, + "intent": { + "n": 24, + "correct": 9, + "accuracy": 0.375 + }, + "judge_hard": { + "n": 17, + "correct": 7, + "accuracy": 0.4117647058823529 + }, + "long_policy": { + "n": 19, + "correct": 6, + "accuracy": 0.3157894736842105 + }, + "multi_hop": { + "n": 18, + "correct": 5, + "accuracy": 0.2777777777777778 + }, + "ordinal": { + "n": 12, + "correct": 8, + "accuracy": 0.6666666666666666 + }, + "policy": { + "n": 12, + "correct": 7, + "accuracy": 0.5833333333333334 + }, + "probability": { + "n": 10, + "correct": 4, + "accuracy": 0.4 + }, + "routing": { + "n": 12, + "correct": 2, + "accuracy": 0.16666666666666666 + }, + "routing_hard": { + "n": 5, + "correct": 5, + "accuracy": 1.0 + }, + "temporal_numeric": { + "n": 15, + "correct": 6, + "accuracy": 0.4 + }, + "tool_selection": { + "n": 12, + "correct": 10, + "accuracy": 0.8333333333333334 + }, + "tradeoff": { + "n": 6, + "correct": 4, + "accuracy": 0.6666666666666666 + }, + "trap": { + "n": 8, + "correct": 4, + "accuracy": 0.5 + } + }, + "gzip": { + "adequacy": { + "n": 12, + "correct": 5, + "accuracy": 0.4166666666666667 + }, + "adversarial": { + "n": 6, + "correct": 3, + "accuracy": 0.5 + }, + "ambiguous": { + "n": 7, + "correct": 3, + "accuracy": 0.42857142857142855 + }, + "extraction": { + "n": 24, + "correct": 11, + "accuracy": 0.4583333333333333 + }, + "fact": { + "n": 12, + "correct": 8, + "accuracy": 0.6666666666666666 + }, + "intent": { + "n": 24, + "correct": 7, + "accuracy": 0.2916666666666667 + }, + "judge_hard": { + "n": 17, + "correct": 7, + "accuracy": 0.4117647058823529 + }, + "long_policy": { + "n": 19, + "correct": 4, + "accuracy": 0.21052631578947367 + }, + "multi_hop": { + "n": 18, + "correct": 4, + "accuracy": 0.2222222222222222 + }, + "ordinal": { + "n": 12, + "correct": 5, + "accuracy": 0.4166666666666667 + }, + "policy": { + "n": 12, + "correct": 6, + "accuracy": 0.5 + }, + "probability": { + "n": 10, + "correct": 5, + "accuracy": 0.5 + }, + "routing": { + "n": 12, + "correct": 3, + "accuracy": 0.25 + }, + "routing_hard": { + "n": 5, + "correct": 5, + "accuracy": 1.0 + }, + "temporal_numeric": { + "n": 15, + "correct": 5, + "accuracy": 0.3333333333333333 + }, + "tool_selection": { + "n": 12, + "correct": 5, + "accuracy": 0.4166666666666667 + }, + "tradeoff": { + "n": 6, + "correct": 6, + "accuracy": 1.0 + }, + "trap": { + "n": 8, + "correct": 5, + "accuracy": 0.625 + } + }, + "tfidf+zstd:0.75": { + "adequacy": { + "n": 12, + "correct": 5, + "accuracy": 0.4166666666666667 + }, + "adversarial": { + "n": 6, + "correct": 3, + "accuracy": 0.5 + }, + "ambiguous": { + "n": 7, + "correct": 1, + "accuracy": 0.14285714285714285 + }, + "extraction": { + "n": 24, + "correct": 16, + "accuracy": 0.6666666666666666 + }, + "fact": { + "n": 12, + "correct": 10, + "accuracy": 0.8333333333333334 + }, + "intent": { + "n": 24, + "correct": 10, + "accuracy": 0.4166666666666667 + }, + "judge_hard": { + "n": 17, + "correct": 7, + "accuracy": 0.4117647058823529 + }, + "long_policy": { + "n": 19, + "correct": 8, + "accuracy": 0.42105263157894735 + }, + "multi_hop": { + "n": 18, + "correct": 4, + "accuracy": 0.2222222222222222 + }, + "ordinal": { + "n": 12, + "correct": 9, + "accuracy": 0.75 + }, + "policy": { + "n": 12, + "correct": 7, + "accuracy": 0.5833333333333334 + }, + "probability": { + "n": 10, + "correct": 5, + "accuracy": 0.5 + }, + "routing": { + "n": 12, + "correct": 2, + "accuracy": 0.16666666666666666 + }, + "routing_hard": { + "n": 5, + "correct": 5, + "accuracy": 1.0 + }, + "temporal_numeric": { + "n": 15, + "correct": 6, + "accuracy": 0.4 + }, + "tool_selection": { + "n": 12, + "correct": 10, + "accuracy": 0.8333333333333334 + }, + "tradeoff": { + "n": 6, + "correct": 5, + "accuracy": 0.8333333333333334 + }, + "trap": { + "n": 8, + "correct": 4, + "accuracy": 0.5 + } + } + }, + "verification": { + "official_scorer_matches": 12474, + "best_hybrid_vs_char_tfidf": { + "accuracy_difference": 0.012987012987012988, + "scenario_bootstrap_95": [ + -0.03070175438596491, + 0.05726872246696035 + ] + }, + "exact_repeatability_predictions_and_scores": 12474 + }, + "selection_note": "The highlighted hybrid is the highest accuracy of 45 fixed blends in an exploratory sweep. It was not independently selected or validated. No gold answers or labeled examples are inputs to any prediction method." +} diff --git a/eval/jevbench_hybrid.py b/eval/jevbench_hybrid.py new file mode 100644 index 0000000..f11308e --- /dev/null +++ b/eval/jevbench_hybrid.py @@ -0,0 +1,198 @@ +#!/usr/bin/env python3 +# /// script +# requires-python = ">=3.12" +# dependencies = ["numpy==2.5.3", "scipy==1.18.1", "scikit-learn==1.9.1", "zstandard==0.25.0"] +# /// +"""TF-IDF + compression on JevBench's public typed decisions. + +uv run --python 3.12 eval/jevbench_hybrid.py --jevbench /path/to/jevbench + +Zero-shot only: no labeled examples, learned weights, retrieval corpus, or +calibration. Each TF-IDF vocabulary/IDF uses only the current request. Fixed +scoring rules are evaluated side by side; no winner is trained on the answers. +""" +import argparse +import gzip +import hashlib +import importlib.metadata +import json +import platform +import time +import zlib +from pathlib import Path + +import numpy as np +import zstandard as zstd +from sklearn.feature_extraction.text import TfidfVectorizer + +FEATURES = ["word_cosine", "word_window_cosine", "char_cosine", "gzip_ncd", + "deflate_gain", "zstd_gain", "label_word_cosine"] + + +def normalized(text): + return " ".join(text.lower().split()) + + +def centered(values): + v = np.asarray(values, dtype=float) + std = v.std(axis=0) + return np.where(std < 1e-10, 0, (v - v.mean(axis=0)) / np.maximum(std, 1e-10)) + + +def inference_text(task): + state = task["state"] + if not isinstance(state, str): + state = json.dumps(state, ensure_ascii=False, sort_keys=True) + return normalized(state + "\n" + task["question"]["instructions"]) + + +def option_texts(task): + q = task["question"] + criteria = q.get("criteria") or {} + if isinstance(criteria, list): + criteria = {str(i): value for i, value in enumerate(criteria)} + options = [] + for label in task["labels"]: + key = {"yes": "true", "no": "false"}.get(label, label) if q["type"] == "noul" else label + options.append(normalized(label.replace("_", " ") + ": " + str(criteria.get(key) or ""))) + return options + + +def features(task): + text, options = inference_text(task), option_texts(task) + words = text.split() + chunks = [" ".join(words[i:i + 160]) for i in range(0, len(words), 80)] or [""] + docs = [text] + options + chunks + [x.replace("_", " ") for x in task["labels"]] + vectors = TfidfVectorizer(ngram_range=(1, 2), token_pattern=r"(?u)\b\w+\b").fit_transform(docs) + n = len(options) + word = (vectors[1:n+1] @ vectors[0].T).toarray().ravel() + window = (vectors[1:n+1] @ vectors[n+1:n+1+len(chunks)].T).toarray().max(axis=1) + label = (vectors[-n:] @ vectors[0].T).toarray().ravel() + chars = TfidfVectorizer(analyzer="char", ngram_range=(3, 5), max_features=30000).fit_transform([text] + options) + char = (chars[1:] @ chars[0].T).toarray().ravel() + x = text.encode() + cx = len(gzip.compress(x, mtime=0)) + raw = zlib.compressobj(level=6, wbits=-15, zdict=x[-32768:]) + zd = zstd.ZstdCompressor(level=3, dict_data=zstd.ZstdCompressionDict(x, dict_type=zstd.DICT_TYPE_RAWCONTENT)) + zplain = zstd.ZstdCompressor(level=3) + values = [] + for option in options: + y = option.encode() + cy = len(gzip.compress(y, mtime=0)) + joint = min(len(gzip.compress(x + b" " + y, mtime=0)), len(gzip.compress(y + b" " + x, mtime=0))) + c = raw.copy() + conditioned = len(c.compress(y) + c.flush()) + plain = zlib.compressobj(level=6, wbits=-15) + unconditioned = len(plain.compress(y) + plain.flush()) + values.append([-(joint - min(cx, cy)) / max(cx, cy), + (unconditioned - conditioned) / max(len(y), 1), + (len(zplain.compress(y)) - len(zd.compress(y))) / max(len(y), 1)]) + a = np.asarray(values) + return np.column_stack([word, window, char, a, label]) + + +def pick(scores, task): + scores = np.asarray(scores) + ties = np.flatnonzero(scores == scores.max()) + # JevBench's distribution scorer resolves ties by label name. + return int(min(ties, key=lambda i: task["labels"][i])) + + +def methods(f): + word, window, char, gz, df, zs, label = f.T + tfidf = centered((centered(window) + centered(char)) / 2) + out = { + "tfidf-word": centered(word), "tfidf-window": centered(window), + "tfidf-char": centered(char), "tfidf-label": centered(label), + "tfidf": tfidf, "gzip": centered(gz), + "deflate": centered(df), "zstd": centered(zs), + } + for lexical in ["tfidf-word", "tfidf-window", "tfidf-char", "tfidf-label", "tfidf"]: + for compressor in ["gzip", "deflate", "zstd"]: + for weight in [.25, .5, .75]: + out[f"{lexical}+{compressor}:{weight:g}"] = weight * out[lexical] + (1-weight) * out[compressor] + out["all-four"] = (tfidf + out["gzip"] + out["deflate"] + out["zstd"]) / 4 + return {name: np.round(scores, 10) for name, scores in out.items()} + + +def summarize(rows): + def score(group): + return {"n": len(group), "correct": sum(r["correct"] for r in group), + "accuracy": sum(r["correct"] for r in group) / len(group)} + return {**score(rows), + "by_tier": {t: score([r for r in rows if r["tier"] == t]) for t in sorted({r["tier"] for r in rows})}, + "by_family": {f: score([r for r in rows if r["family"] == f]) for f in sorted({r["family"] for r in rows})}} + + +def paired_interval(a, b, seed=0): + groups = sorted({r["group"] for r in a}) + delta = {g: [int(x["correct"]) - int(y["correct"]) for x, y in zip(a, b) if x["group"] == g] for g in groups} + rng = np.random.default_rng(seed) + samples = [] + for _ in range(5000): + chosen = rng.choice(groups, len(groups), replace=True) + differences = [v for g in chosen for v in delta[g]] + samples.append(np.mean(differences)) + return {"accuracy_difference": float(np.mean([int(x["correct"]) - int(y["correct"]) for x, y in zip(a, b)])), + "scenario_bootstrap_95": np.percentile(samples, [2.5, 97.5]).tolist()} + + +def main(): + ap = argparse.ArgumentParser(description=__doc__) + ap.add_argument("--jevbench", required=True, type=Path) + ap.add_argument("--out", type=Path, default=Path(__file__).parent / "data" / "jevbench-zero-shot") + args = ap.parse_args() + args.out.mkdir(parents=True, exist_ok=False) + tasks, source_hashes = [], {} + for tier in ["easy", "original", "hard"]: + path = args.jevbench / "datasets" / "public" / f"{tier}.jsonl" + source_hashes[tier] = hashlib.sha256(path.read_bytes()).hexdigest() + for line in path.read_text().splitlines(): + row = json.loads(line) + assert row["expected"] is not None and not row.get("provenance", {}).get("exclude_reason") + row["tier"] = tier + tasks.append(row) + outputs, elapsed = {}, [] + for task in tasks: + view = {k: task[k] for k in ["state", "question", "labels"]} + start = time.perf_counter_ns() + f = features(view) + scores = methods(f) + predicted = {name: view["labels"][pick(values, view)] for name, values in scores.items()} + elapsed.append((time.perf_counter_ns() - start) / 1e6) + assert f.shape == (len(view["labels"]), len(FEATURES)) and np.isfinite(f).all() + reversed_view = {**view, "labels": list(reversed(view["labels"]))} + reversed_features = features(reversed_view) + assert np.allclose(f, reversed_features[::-1], atol=1e-12) + reversed_scores = methods(reversed_features) + for name, prediction in predicted.items(): + assert prediction == reversed_view["labels"][pick(reversed_scores[name], reversed_view)], (task["id"], name) + outputs.setdefault(name, []).append({"id": task["id"], "group": task.get("group") or task["id"], + "tier": task["tier"], "family": task["family"], "gold": str(task["expected"]), + "prediction": prediction, "correct": prediction == str(task["expected"]), + "labels": view["labels"], "scores": scores[name].tolist()}) + published_path = args.jevbench / "results/v1.2/jevbench-v1.2-per-task.json" + published = json.loads(published_path.read_text())["systems"]["jev-1.13.0"]["public_tasks"] + reference = [{**r, "correct": published[r["id"]][0] == "c"} for r in outputs["tfidf"]] + report = {"manifest": {"source_sha256": source_hashes, + "script_sha256": hashlib.sha256(Path(__file__).read_bytes()).hexdigest(), + "platform": platform.platform(), "python": platform.python_version(), "features": FEATURES, + "versions": {p: importlib.metadata.version(p) for p in ["numpy", "scipy", "scikit-learn", "zstandard"]}, + "all_methods_p50_ms": float(np.median(elapsed)), "all_methods_p95_ms": float(np.percentile(elapsed, 95)), + "option_reversal_predictions_checked": sum(len(v) for v in outputs.values()), + "note": "Public zero-shot diagnostic only; no gold answers or labeled examples enter prediction. IDF is computed per request. All methods are fixed; multiple exploratory comparisons, no official sealed evaluation. Timing includes every method together, excluding verification and loading."}, + "uniform_chance": float(np.mean([1/len(t["labels"]) for t in tasks])), + "methods": {name: summarize(rows) for name, rows in outputs.items()}, + "versus_tfidf": {name: paired_interval(rows, outputs["tfidf"]) for name, rows in outputs.items() if name != "tfidf"}, + "published_jev_reference": {"source_sha256": hashlib.sha256(published_path.read_bytes()).hexdigest(), + "note": "Published Jev 1.13.0 outcomes on these exact public IDs; not rerun.", **summarize(reference)}, + "oracle_choose_tfidf_or_gzip": {"note": "Diagnostic only: uses gold answers to pick the correct method. Not a deployable model or a bound on other hybrids.", + "accuracy": float(np.mean([a["correct"] or b["correct"] for a,b in zip(outputs["tfidf"],outputs["gzip"])]))}} + for name, rows in outputs.items(): + (args.out / f"{name}-predictions.json").write_text(json.dumps(rows, indent=2)) + print(name, json.dumps(report["methods"][name]), flush=True) + (args.out / "summary.json").write_text(json.dumps(report, indent=2) + "\n") + + +if __name__ == "__main__": + main()