d23a3275a8
- Document the smart reply center, clipboard semantic v6 specialization, personal style generation fix, and durable Apple account session work in both English and Simplified Chinese. - Update clipboard semantics README and open-training sources to reflect the v6 boundary, blessing, and consensus-adjudication corpora that the new release-gate pipeline consumes.
241 lines
7.9 KiB
JSON
241 lines
7.9 KiB
JSON
{
|
|
"baseCorpusRecords": 17692,
|
|
"combinedCorpusSHA256": "24bbec3fb9b3cc6434582d660b03b37401abe1339185405115646675ec3b654a",
|
|
"combinedRecords": 48779,
|
|
"excludedBaseOverlap": 6,
|
|
"excludedHoldoutOverlap": 425,
|
|
"excludedSources": [
|
|
{
|
|
"dataset": "MIDAS",
|
|
"reason": "Upstream deleted da_data in commit 509eb7e78b6661e54883fe3522ba60aaddcf69f2; deleted data is not resurrected for training."
|
|
},
|
|
{
|
|
"dataset": "DailyDialog / EmpatheticDialogues / Switchboard",
|
|
"reason": "Non-commercial license restriction."
|
|
},
|
|
{
|
|
"dataset": "NormDial / xLLMs Dialogue Greetings",
|
|
"reason": "No clear standard dataset license."
|
|
},
|
|
{
|
|
"dataset": "Schema-Guided Dialogue",
|
|
"reason": "CC-BY-SA-4.0 compatibility requires legal review."
|
|
},
|
|
{
|
|
"dataset": "Stanford Politeness Corpus",
|
|
"reason": "The downloadable archive does not include an explicit corpus license; source comments may carry ShareAlike obligations."
|
|
},
|
|
{
|
|
"dataset": "CPED",
|
|
"reason": "The Apache repository license does not establish commercial training rights for dialogue transcribed from copyrighted TV shows."
|
|
},
|
|
{
|
|
"dataset": "Tianji Wishes / Birthday Quotes",
|
|
"reason": "Dataset-level licenses do not provide a sufficiently clear per-record source or synthetic-generation rights chain."
|
|
},
|
|
{
|
|
"dataset": "Verified Enron Intent / GitHub issue holdout source",
|
|
"reason": "No clean official train split independent from the frozen holdout."
|
|
},
|
|
{
|
|
"dataset": "CFPB",
|
|
"reason": "No official train split; the source is already represented in frozen evaluation data and contains privacy-sensitive narratives."
|
|
},
|
|
{
|
|
"dataset": "CLINC150",
|
|
"reason": "Kept isolated from product training because its assistant intents are not aligned with the product-policy taxonomy."
|
|
},
|
|
{
|
|
"dataset": "openclaw-zh-greetings / WeChat-AutoSendBless",
|
|
"reason": "Repository licenses do not establish a sufficiently clear, versioned rights chain for the underlying message templates."
|
|
}
|
|
],
|
|
"openTrainingRecords": 31087,
|
|
"policy": "Licensed official training splits only. Per-record knownLabels prevent unannotated intents from becoming false negatives. Exact normalized text overlap with every local *holdout-corpus.jsonl file is excluded.",
|
|
"schemaVersion": 1,
|
|
"seed": 20260827,
|
|
"sources": [
|
|
{
|
|
"dataset": "AmazonScience/MASSIVE en-US",
|
|
"knownLabelCounts": {
|
|
"assistantCommand": 3658,
|
|
"domain": 9824,
|
|
"informationQuery": 5307,
|
|
"sentiment": 126,
|
|
"task": 358
|
|
},
|
|
"languages": {
|
|
"en": 10810
|
|
},
|
|
"license": "CC-BY-4.0",
|
|
"records": 10810,
|
|
"revision": "1.1",
|
|
"url": "https://huggingface.co/datasets/AmazonScience/massive"
|
|
},
|
|
{
|
|
"dataset": "AmazonScience/MASSIVE zh-CN",
|
|
"knownLabelCounts": {
|
|
"assistantCommand": 3272,
|
|
"domain": 9166,
|
|
"informationQuery": 5025,
|
|
"sentiment": 120,
|
|
"task": 348
|
|
},
|
|
"languages": {
|
|
"zh-Hans": 10099
|
|
},
|
|
"license": "CC-BY-4.0",
|
|
"records": 10099,
|
|
"revision": "1.1",
|
|
"url": "https://huggingface.co/datasets/AmazonScience/massive"
|
|
},
|
|
{
|
|
"dataset": "CrossWOZ",
|
|
"knownLabelCounts": {
|
|
"domain": 476,
|
|
"informationQuery": 393
|
|
},
|
|
"languages": {
|
|
"zh-Hans": 599
|
|
},
|
|
"license": "Apache-2.0",
|
|
"records": 599,
|
|
"revision": "df82c9fdff91b9b130f2d6b89110d3870ba6260e",
|
|
"url": "https://github.com/thu-coai/CrossWOZ/blob/df82c9fdff91b9b130f2d6b89110d3870ba6260e/data/crosswoz/train.json.zip"
|
|
},
|
|
{
|
|
"dataset": "FormosaNLU Synth v1",
|
|
"knownLabelCounts": {
|
|
"assistantCommand": 839,
|
|
"domain": 1747,
|
|
"informationQuery": 817,
|
|
"task": 119
|
|
},
|
|
"languages": {
|
|
"zh-Hans": 1988
|
|
},
|
|
"license": "CC-BY-4.0",
|
|
"records": 1988,
|
|
"revision": "03a337b61a200ab690994dca4dc31aa7f209800e",
|
|
"url": "https://huggingface.co/datasets/steven0226/formosa-nlu-synth-v1/resolve/03a337b61a200ab690994dca4dc31aa7f209800e/data/train.jsonl"
|
|
},
|
|
{
|
|
"dataset": "Google Research GoEmotions",
|
|
"knownLabelCounts": {
|
|
"sentiment": 799
|
|
},
|
|
"languages": {
|
|
"en": 1199
|
|
},
|
|
"license": "Apache-2.0",
|
|
"records": 1199,
|
|
"revision": "5d8f4ac97c873bde3a792ba4628f00bb9103d3e6",
|
|
"url": "https://github.com/google-research/google-research/blob/5d8f4ac97c873bde3a792ba4628f00bb9103d3e6/goemotions/data/train.tsv"
|
|
},
|
|
{
|
|
"dataset": "Google Taskmaster-1",
|
|
"knownLabelCounts": {
|
|
"domain": 631
|
|
},
|
|
"languages": {
|
|
"en": 631
|
|
},
|
|
"license": "CC-BY-4.0",
|
|
"records": 631,
|
|
"revision": "d92cb6af3005f1dc09c39e75e7daf4a04905e00b",
|
|
"url": "https://github.com/google-research-datasets/Taskmaster/tree/d92cb6af3005f1dc09c39e75e7daf4a04905e00b/TM-1-2019"
|
|
},
|
|
{
|
|
"dataset": "HLTCHKUST/BiToD",
|
|
"knownLabelCounts": {
|
|
"domain": 1339,
|
|
"informationQuery": 885,
|
|
"task": 355
|
|
},
|
|
"languages": {
|
|
"en": 681,
|
|
"zh-Hans": 658
|
|
},
|
|
"license": "Apache-2.0",
|
|
"records": 1339,
|
|
"revision": "a9bd74de9eecdc3d875cb4ebf6a6beaf9c30c2ff",
|
|
"url": "https://raw.githubusercontent.com/HLTCHKUST/BiToD/a9bd74de9eecdc3d875cb4ebf6a6beaf9c30c2ff/data/zh_train.json"
|
|
},
|
|
{
|
|
"dataset": "Meituan-Dianping/ASAP",
|
|
"knownLabelCounts": {
|
|
"complaint": 500,
|
|
"domain": 1000,
|
|
"sentiment": 1000
|
|
},
|
|
"languages": {
|
|
"zh-Hans": 1000
|
|
},
|
|
"license": "Apache-2.0",
|
|
"records": 1000,
|
|
"revision": "975122a60065240124df62cb4d5dbfd19ed9ef2c",
|
|
"url": "https://github.com/Meituan-Dianping/ASAP/blob/975122a60065240124df62cb4d5dbfd19ed9ef2c/data/train.csv"
|
|
},
|
|
{
|
|
"dataset": "MultiDoGO",
|
|
"knownLabelCounts": {
|
|
"domain": 464,
|
|
"informationQuery": 61,
|
|
"task": 83
|
|
},
|
|
"languages": {
|
|
"en": 464
|
|
},
|
|
"license": "CDLA-Permissive-1.0",
|
|
"records": 464,
|
|
"revision": "baa30639c4b271f394b81443c842193407cdf26d",
|
|
"url": "https://github.com/awslabs/multi-domain-goal-oriented-dialogues-dataset/blob/baa30639c4b271f394b81443c842193407cdf26d/data/paper_splits/splits_annotated_at_turn_level/airline/train.tsv"
|
|
},
|
|
{
|
|
"dataset": "PolyAI MInDS-14 zh-CN",
|
|
"knownLabelCounts": {
|
|
"assistantCommand": 163,
|
|
"domain": 480,
|
|
"informationQuery": 179
|
|
},
|
|
"languages": {
|
|
"zh-Hans": 480
|
|
},
|
|
"license": "CC-BY-4.0",
|
|
"records": 480,
|
|
"revision": "40ce77cb32a384e4d50a568e1ec39ac804019d33",
|
|
"url": "https://huggingface.co/datasets/PolyAI/minds14/resolve/40ce77cb32a384e4d50a568e1ec39ac804019d33/zh-CN/train-00000-of-00001.parquet"
|
|
},
|
|
{
|
|
"dataset": "PolyAI RESTAURANTS-8K",
|
|
"knownLabelCounts": {
|
|
"domain": 986
|
|
},
|
|
"languages": {
|
|
"en": 986
|
|
},
|
|
"license": "CC-BY-4.0",
|
|
"records": 986,
|
|
"revision": "57ec275d8078af65b7731c2a98be812d844a6d6b",
|
|
"url": "https://raw.githubusercontent.com/PolyAI-LDN/task-specific-datasets/57ec275d8078af65b7731c2a98be812d844a6d6b/span_extraction/restaurant8k/train_0.json"
|
|
},
|
|
{
|
|
"dataset": "SNIPS NLU Benchmark",
|
|
"knownLabelCounts": {
|
|
"assistantCommand": 499,
|
|
"domain": 1492,
|
|
"informationQuery": 743,
|
|
"task": 250
|
|
},
|
|
"languages": {
|
|
"en": 1492
|
|
},
|
|
"license": "CC0-1.0",
|
|
"records": 1492,
|
|
"revision": "b86ac7f1577868c42158d0dec77db50956046696",
|
|
"url": "https://raw.githubusercontent.com/sonos/nlu-benchmark/b86ac7f1577868c42158d0dec77db50956046696/2017-06-custom-intent-engines/BookRestaurant/train_BookRestaurant_full.json"
|
|
}
|
|
],
|
|
"unavailableSources": []
|
|
}
|