Search.setIndex({"docnames": ["Baseline", "Challenge_result", "Contact", "Dataset", "Introduction", "Organizers", "Rules", "Track_setting_and_evaluation", "index"], "filenames": ["Baseline.md", "Challenge_result.md", "Contact.md", "Dataset.md", "Introduction.md", "Organizers.md", "Rules.md", "Track_setting_and_evaluation.md", "index.rst"], "titles": ["Baseline", "Challenge Result", "Contact", "Datasets", "Introduction", "Organizers", "Rules", "Track & Evaluation", "ASRU 2023 MULTI-CHANNEL MULTI-PARTY MEETING TRANSCRIPTION CHALLENGE 2.0 (M2MeT2.0)"], "terms": {"we": [0, 3, 4, 8], "releas": [0, 3, 4, 7], "an": [0, 3, 4, 7], "e2": 0, "sa": 0, "asr": [0, 4, 8], "conduct": [0, 3], "funasr": 0, "time": [0, 7], "accord": [0, 4], "timelin": [0, 3], "The": [0, 1, 3, 4, 6, 7], "model": [0, 3, 4, 6, 7], "architectur": 0, "i": [0, 1, 3, 4, 6], "shown": [0, 3], "figur": [0, 7], "3": [0, 1, 3, 4], "speakerencod": 0, "initi": 0, "pre": [0, 7], "train": [0, 1, 4, 6, 8], "speaker": [0, 3, 4, 8], "verif": 0, "from": [0, 3, 4, 6, 7], "modelscop": [0, 7], "thi": [0, 1, 4, 6, 7], "also": [0, 3, 4, 7], "us": [0, 3, 6, 7], "extract": 0, "embed": 0, "profil": 0, "To": [0, 3, 4, 8], "run": 0, "first": 0, "you": [0, 2], "need": 0, "instal": 0, "There": [0, 3], "ar": [0, 1, 3, 4, 6, 7, 8], "two": [0, 1, 4, 6, 8], "startup": 0, "script": [0, 3], "sh": 0, "evalu": [0, 3, 4, 8], "old": 0, "eval": [0, 3, 6, 7], "test": [0, 3, 4, 6, 7], "set": [0, 3, 4, 6, 7], "run_m2met_2023_inf": 0, "infer": 0, "new": [0, 3, 4, 7], "multi": [0, 4, 7], "channel": [0, 4], "parti": [0, 4, 7], "meet": [0, 3, 4, 7], "transcript": [0, 3, 4, 6, 7], "2": [0, 1, 3, 7], "0": [0, 2, 3, 4], "m2met2": [0, 2, 4], "challeng": [0, 2, 4, 6, 7], "befor": 0, "must": [0, 4, 6, 7], "manual": [0, 7], "download": [0, 3], "unpack": 0, "alimeet": [0, 2, 7], "corpu": [0, 7], "place": [0, 3], "dataset": [0, 4, 6, 7, 8], "directori": 0, "eval_ali_far": 0, "eval_ali_near": 0, "test_ali_far": 0, "test_ali_near": 0, "train_ali_far": 0, "train_ali_near": 0, "test_2023_ali_far": 0, "after": 0, "which": [0, 3, 4, 7], "contain": [0, 3, 7], "onli": [0, 3, 6, 7], "raw": 0, "audio": [0, 3, 4, 7], "Then": 0, "put": 0, "given": 0, "wav": 0, "scp": 0, "wav_raw": 0, "segment": [0, 3, 7], "utt2spk": 0, "spk2utt": 0, "data": [0, 4, 6, 7], "For": [0, 3], "more": [0, 3], "detail": [0, 4, 7], "can": [0, 3, 4, 6, 7], "see": 0, "here": 0, "system": [0, 4, 6, 7, 8], "tabl": [0, 1, 3], "adopt": 0, "oracl": [0, 7], "dure": [0, 3, 7], "howev": [0, 4, 7], "due": [0, 4], "lack": 0, "label": [0, 6, 7], "provid": [0, 3, 7, 8], "addit": [0, 7], "spectral": 0, "cluster": 0, "meanwhil": 0, "show": [0, 1], "impact": 0, "accuraci": [0, 7], "follow": [1, 3, 6], "final": [1, 4, 6, 7], "competit": 1, "where": [1, 7], "sub": [1, 4, 6, 8], "track1": 1, "repres": 1, "track": [1, 4, 6, 8], "under": 1, "fix": [1, 3, 4, 8], "condit": [1, 3, 4, 8], "open": [1, 3, 4, 8], "all": [1, 3, 4, 6, 7], "cp": 1, "cer": [1, 7], "rank": [1, 3, 4, 7], "combin": 1, "team": [1, 4], "submiss": [1, 4], "met": 1, "requir": [1, 3, 4, 7], "name": [1, 3], "track2": 1, "paper": [1, 4, 7], "1": [1, 3], "ximalaya": 1, "speech": [1, 3, 4, 7, 8], "11": [1, 4], "27": [1, 3], "\u5c0f\u9a6c\u8fbe": 1, "18": 1, "64": 1, "aizyzx": 1, "22": [1, 4], "83": 1, "4": [1, 3, 7], "asrspeed": 1, "23": 1, "51": 1, "5": [1, 3], "zyxlhz": 1, "24": 1, "82": 1, "6": 1, "cmcai": 1, "26": [1, 4], "7": 1, "volcspeech": 1, "34": [1, 3], "21": 1, "8": [1, 3], "\u9274\u5f80\u77e5\u6765": 1, "40": 1, "14": 1, "9": 1, "baselin": [1, 3, 4, 8], "41": 1, "55": [1, 3], "10": [1, 3, 4, 7], "daict": 1, "If": [2, 6, 7], "have": [2, 4], "ani": [2, 6, 7], "question": 2, "about": [2, 4], "pleas": 2, "u": [2, 3], "email": [2, 4, 5], "m2met": [2, 4, 7, 8], "gmail": 2, "com": [2, 5], "wechat": [2, 4], "group": [2, 3, 4], "In": [3, 4, 6], "restrict": 3, "three": [3, 4, 7], "publicli": [3, 7], "avail": [3, 7], "corpora": 3, "aishel": [3, 5, 7], "cn": [3, 5, 7], "celeb": [3, 7], "perform": [3, 4], "call": 3, "2023": [3, 4, 6, 7], "score": [3, 7], "describ": 3, "118": 3, "75": 3, "hour": [3, 4, 7], "total": [3, 7], "divid": [3, 7], "104": 3, "specif": [3, 7], "212": 3, "20": [3, 4], "session": [3, 4, 7, 8], "respect": 3, "each": [3, 4, 7], "consist": [3, 7], "15": 3, "30": 3, "minut": 3, "discuss": 3, "particip": [3, 6, 7], "number": [3, 4, 7], "456": 3, "25": 3, "60": 3, "balanc": 3, "gender": 3, "coverag": 3, "collect": 3, "13": 3, "venu": 3, "categor": 3, "type": 3, "small": 3, "medium": 3, "larg": [3, 4], "room": [3, 4], "size": 3, "rang": 3, "m": 3, "differ": [3, 4, 7], "give": 3, "varieti": 3, "acoust": [3, 4, 7], "properti": 3, "layout": 3, "paramet": [3, 6], "togeth": 3, "wall": 3, "materi": 3, "cover": 3, "cement": 3, "glass": 3, "etc": 3, "other": 3, "furnish": 3, "includ": [3, 4, 6, 7], "sofa": 3, "tv": 3, "blackboard": 3, "fan": 3, "air": 3, "condition": 3, "plant": 3, "record": [3, 7], "sit": 3, "around": 3, "microphon": [3, 4], "arrai": [3, 4], "natur": 3, "convers": 3, "distanc": 3, "nativ": 3, "chines": 3, "speak": [3, 4], "mandarin": [3, 4], "without": 3, "strong": 3, "accent": 3, "variou": [3, 4], "kind": 3, "indoor": 3, "nois": [3, 4, 6], "limit": [3, 4, 6], "click": 3, "keyboard": 3, "door": 3, "close": [3, 4], "bubbl": 3, "made": [3, 4], "both": [3, 7], "remain": [3, 4], "same": [3, 6], "posit": 3, "overlap": [3, 4], "between": [3, 7], "exampl": 3, "fig": 3, "within": [3, 4], "one": [3, 6], "ensur": 3, "ratio": 3, "select": [3, 4, 6, 7], "topic": 3, "medic": 3, "treatment": 3, "educ": 3, "busi": 3, "organ": [3, 4, 6, 7, 8], "manag": 3, "industri": [3, 4], "product": 3, "daili": 3, "routin": 3, "averag": 3, "42": 3, "76": 3, "A": [3, 5], "distribut": 3, "were": 3, "ident": [3, 7], "compris": [3, 4, 8], "therebi": 3, "share": 3, "similar": 3, "configur": 3, "field": [3, 4, 7], "signal": [3, 4], "headset": 3, "": [3, 7], "own": 3, "transcrib": [3, 4, 7], "It": [3, 7], "worth": [3, 7], "note": [3, 7], "far": [3, 4], "synchron": 3, "common": 3, "prepar": 3, "textgrid": 3, "format": 3, "inform": [3, 4], "durat": 3, "id": 3, "timestamp": [3, 7], "mention": 3, "abov": 3, "openslr": 3, "via": 3, "link": 3, "particularli": 3, "conveni": 3, "automat": [4, 8], "recognit": [4, 8], "diariz": 4, "signific": 4, "stride": 4, "recent": 4, "year": 4, "result": [4, 8], "surg": 4, "technologi": 4, "applic": 4, "across": 4, "domain": 4, "present": 4, "uniqu": [4, 7], "complex": [4, 6], "divers": 4, "style": 4, "variabl": 4, "confer": 4, "environment": 4, "reverber": [4, 6], "over": 4, "sever": 4, "been": 4, "advanc": [4, 8], "develop": [4, 7], "rich": 4, "comput": [4, 6], "hear": 4, "multisourc": 4, "environ": 4, "chime": 4, "latest": 4, "iter": 4, "ha": 4, "particular": 4, "focu": 4, "distant": 4, "gener": 4, "topologi": 4, "scenario": 4, "while": 4, "progress": 4, "english": 4, "languag": [4, 6], "barrier": 4, "achiev": 4, "compar": 4, "non": 4, "multimod": 4, "base": 4, "process": [4, 7], "misp": 4, "instrument": 4, "seek": 4, "address": 4, "problem": 4, "visual": 4, "everydai": 4, "home": 4, "focus": 4, "tackl": 4, "issu": 4, "offlin": 4, "icassp2022": 4, "main": 4, "task": [4, 7, 8], "former": 4, "involv": [4, 7], "identifi": 4, "who": 4, "spoke": 4, "when": 4, "latter": 4, "aim": 4, "multipl": [4, 7], "simultan": 4, "pose": [4, 7], "technic": 4, "difficulti": 4, "interfer": 4, "build": [4, 7, 8], "success": [4, 8], "previou": 4, "excit": 4, "propos": [4, 8], "asru": 4, "special": [4, 6, 8], "origin": [4, 6], "metric": [4, 8], "wa": [4, 7], "independ": 4, "meant": 4, "could": 4, "determin": 4, "correspond": [4, 6], "further": 4, "current": [4, 8], "talker": [4, 8], "toward": 4, "practic": 4, "attribut": [4, 8], "what": 4, "facilit": [4, 8], "reproduc": [4, 8], "research": [4, 5, 8], "offer": 4, "comprehens": [4, 8], "overview": [4, 8], "rule": [4, 8], "furthermor": 4, "carefulli": 4, "curat": 4, "approxim": [4, 7], "design": 4, "enabl": 4, "valid": 4, "state": [4, 7, 8], "art": [4, 8], "area": 4, "april": 4, "29": 4, "registr": 4, "mai": 4, "deadlin": 4, "date": 4, "join": 4, "june": 4, "16": 4, "leaderboard": 4, "leaderboar": 4, "juli": 4, "decemb": 4, "12": 4, "workshop": 4, "interest": 4, "whether": 4, "academia": 4, "regist": 4, "complet": 4, "googl": 4, "form": 4, "below": 4, "welcom": 4, "keep": 4, "up": 4, "updat": 4, "work": 4, "dai": 4, "send": 4, "invit": 4, "elig": [4, 6], "qualifi": 4, "adher": [4, 6], "publish": 4, "page": 4, "prior": 4, "submit": 4, "descript": [4, 7], "document": 4, "approach": [4, 6], "method": 4, "top": 4, "asru2023": [4, 8], "proceed": 4, "lei": 5, "xie": 5, "professor": 5, "foundat": 5, "china": 5, "lxie": 5, "nwpu": 5, "edu": 5, "kong": 5, "aik": 5, "lee": 5, "senior": 5, "scientist": 5, "institut": 5, "infocomm": 5, "star": 5, "singapor": 5, "kongaik": 5, "ieee": 5, "org": 5, "zhiji": 5, "yan": 5, "princip": 5, "engin": 5, "alibaba": 5, "yzj": 5, "inc": 5, "shiliang": 5, "zhang": 5, "sly": 5, "zsl": 5, "yanmin": 5, "qian": 5, "shanghai": 5, "jiao": 5, "tong": 5, "univers": 5, "yanminqian": 5, "sjtu": 5, "zhuo": 5, "chen": 5, "appli": 5, "microsoft": 5, "usa": 5, "zhuc": 5, "jian": 5, "wu": 5, "wujian": 5, "hui": 5, "bu": 5, "ceo": 5, "buhui": 5, "aishelldata": 5, "should": 6, "augment": 6, "allow": [6, 7], "ad": 6, "speed": 6, "perturb": 6, "tone": 6, "chang": 6, "permit": 6, "purpos": 6, "instead": [6, 7], "util": [6, 7], "tune": 6, "violat": 6, "strictli": [6, 7], "prohibit": [6, 7], "fine": 6, "cpcer": [6, 7], "lower": 6, "judg": 6, "superior": 6, "forc": 6, "align": 6, "obtain": [6, 7], "frame": 6, "level": 6, "classif": 6, "basi": 6, "shallow": 6, "fusion": 6, "end": 6, "e": [6, 7], "g": 6, "la": 6, "rnnt": 6, "transform": [6, 7], "come": 6, "right": 6, "interpret": 6, "belong": 6, "case": 6, "circumst": 6, "coordin": 6, "assign": 7, "illustr": 7, "aishell4": 7, "constrain": 7, "sourc": 7, "addition": 7, "soon": 7, "simpl": 7, "voic": 7, "activ": 7, "detect": 7, "vad": 7, "concaten": 7, "minimum": 7, "permut": 7, "charact": 7, "error": 7, "rate": 7, "calcul": 7, "step": 7, "firstli": 7, "refer": 7, "hypothesi": 7, "chronolog": 7, "order": 7, "secondli": 7, "repeat": 7, "possibl": 7, "lowest": 7, "tthe": 7, "insert": 7, "Ins": 7, "substitut": 7, "delet": 7, "del": 7, "output": 7, "text": 7, "frac": 7, "mathcal": 7, "n_": 7, "100": 7, "usag": 7, "third": 7, "hug": 7, "face": 7, "list": 7, "clearli": 7, "privat": 7, "simul": 7, "thei": 7, "mandatori": 7, "clear": 7, "scheme": 7, "delight": 8, "introduct": 8, "contact": 8}, "objects": {}, "objtypes": {}, "objnames": {}, "titleterms": {"baselin": 0, "overview": [0, 3], "quick": 0, "start": 0, "result": [0, 1], "challeng": [1, 8], "contact": 2, "dataset": 3, "train": [3, 7], "data": 3, "detail": 3, "alimeet": 3, "corpu": 3, "get": 3, "introduct": 4, "call": 4, "particip": 4, "timelin": 4, "aoe": 4, "time": 4, "guidelin": 4, "organ": 5, "rule": 6, "track": 7, "evalu": 7, "speaker": 7, "attribut": 7, "asr": 7, "metric": 7, "sub": 7, "arrang": 7, "i": 7, "fix": 7, "condit": 7, "ii": 7, "open": 7, "asru": 8, "2023": 8, "multi": 8, "channel": 8, "parti": 8, "meet": 8, "transcript": 8, "2": 8, "0": 8, "m2met2": 8, "content": 8}, "envversion": {"sphinx.domains.c": 2, "sphinx.domains.changeset": 1, "sphinx.domains.citation": 1, "sphinx.domains.cpp": 8, "sphinx.domains.index": 1, "sphinx.domains.javascript": 2, "sphinx.domains.math": 2, "sphinx.domains.python": 3, "sphinx.domains.rst": 2, "sphinx.domains.std": 2, "sphinx": 57}, "alltitles": {"Baseline": [[0, "baseline"]], "Overview": [[0, "overview"]], "Quick start": [[0, "quick-start"]], "Baseline results": [[0, "baseline-results"]], "Contact": [[2, "contact"]], "Datasets": [[3, "datasets"]], "Overview of training data": [[3, "overview-of-training-data"]], "Detail of AliMeeting corpus": [[3, "detail-of-alimeeting-corpus"]], "Get the data": [[3, "get-the-data"]], "Introduction": [[4, "introduction"]], "Call for participation": [[4, "call-for-participation"]], "Timeline(AOE Time)": [[4, "timeline-aoe-time"]], "Guidelines": [[4, "guidelines"]], "Organizers": [[5, "organizers"]], "Rules": [[6, "rules"]], "Track & Evaluation": [[7, "track-evaluation"]], "Speaker-Attributed ASR": [[7, "speaker-attributed-asr"]], "Evaluation metric": [[7, "evaluation-metric"]], "Sub-track arrangement": [[7, "sub-track-arrangement"]], "Sub-track I (Fixed Training Condition):": [[7, "sub-track-i-fixed-training-condition"]], "Sub-track II (Open Training Condition):": [[7, "sub-track-ii-open-training-condition"]], "ASRU 2023 MULTI-CHANNEL MULTI-PARTY MEETING TRANSCRIPTION CHALLENGE 2.0 (M2MeT2.0)": [[8, "asru-2023-multi-channel-multi-party-meeting-transcription-challenge-2-0-m2met2-0"]], "Contents:": [[8, null]], "Challenge Result": [[1, "challenge-result"]]}, "indexentries": {}})
|