diff --git a/.dockerignore b/.dockerignore new file mode 100644 index 0000000000000000000000000000000000000000..7b35444c98fb8a9ffe51247657cf07795f00505f --- /dev/null +++ b/.dockerignore @@ -0,0 +1,17 @@ +.git +.venv +venv +__pycache__ +**/__pycache__ +*.pyc +**/*.pyc +.DS_Store +.env +.streamlit/secrets.toml +.playwright +hf_space/runs +runs +*.zip +**/.pytest_cache +**/node_modules +agent-transcripts diff --git a/.gitattributes b/.gitattributes index a6344aac8c09253b3b630fb776ae94478aa0275b..3599445f25d9fdad81f55e2b32bce54e060c2cf1 100644 --- a/.gitattributes +++ b/.gitattributes @@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text *.zip filter=lfs diff=lfs merge=lfs -text *.zst filter=lfs diff=lfs merge=lfs -text *tfevents* filter=lfs diff=lfs merge=lfs -text +replay_traces/2505.17978/processed_papers/2505.17978/usage_citing_paragraphs.json filter=lfs diff=lfs merge=lfs -text diff --git a/.streamlit/config.toml b/.streamlit/config.toml new file mode 100644 index 0000000000000000000000000000000000000000..43e993946ca69686bee1059fbe70b5d0d68f7cf4 --- /dev/null +++ b/.streamlit/config.toml @@ -0,0 +1,22 @@ +[browser] +gatherUsageStats = false + +[server] +headless = true +# Hugging Face Spaces / Docker bind on all interfaces; local overrides via CLI are fine. +address = "0.0.0.0" +port = 7860 +fileWatcherType = "auto" +runOnSave = false + +[client] +toolbarMode = "viewer" +showSidebarNavigation = false + +[theme] +base = "light" +primaryColor = "#0f6b5c" +backgroundColor = "#e7eef0" +secondaryBackgroundColor = "#f4f8f7" +textColor = "#13201d" +font = "sans serif" diff --git a/.streamlit/credentials.toml b/.streamlit/credentials.toml new file mode 100644 index 0000000000000000000000000000000000000000..0737dcd756948b110d5102097c910396799f133a --- /dev/null +++ b/.streamlit/credentials.toml @@ -0,0 +1,2 @@ +[general] +email = "" diff --git a/Deep-Citation/Data/acl.tsv b/Deep-Citation/Data/acl.tsv new file mode 100644 index 0000000000000000000000000000000000000000..a07d42d8c8ac67f5c5c84a72e9f70a623fffff97 --- /dev/null +++ b/Deep-Citation/Data/acl.tsv @@ -0,0 +1,1942 @@ +context label split +Thus , over the past few years , along with advances in the use of learning and statistical methods for acquisition of full parsers ( ; Charniak , 1997a ; Charniak , 1997b ; Ratnaparkhi , 1997 ) , significant progress has been made on the use of statistical learning methods to recognize shallow parsing patterns syntactic phrases or words that participate in a syntactic relationship ( Church , 1988 ; Ramshaw and Marcus , 1995 ; Argamon et al. , 1998 ; Cardie and Pierce , 1998 ; Munoz et al. , 1999 ; Punyakanok and Roth , 2001 ; Buchholz et al. , 1999 ; Tjong Kim Sang and Buchholz , 2000 ) . Background train +This was done by MERT optimization ( ) towards post-edits under the TER target metric . Uses train +She evaluates 3,000 German verbs with a token frequency between 10 and 2,000 against the Duden ( ) . Background train +The following four components have been identified as the key elements of a question related to patient care ( ) : Background train + report on manually analyzing an open-class vocabulary of 35,000 head words for predicate subcategorization information and comparing the results against the subcategorization details in COMLEX . CompareOrContrast train +This is roughly an 11 % relative reduction in error rate over and Bods PCFG-reduction reported in Table 1 . CompareOrContrast train +Finally , feedback expressions ( head nods and shakes ) are successfully predicted from speech , prosody and eye gaze in interaction with Embodied Communication Agents as well as human communication ( Fujie et al. , 2004 ; ; Morency et al. , 2007 ; Morency et al. , 2009 ) . Background train +We use the agreement checker code developed by and evaluate our baseline ( MaltParser using only CORE12 ) , best performing model ( Easy-First Parser using CORE12 + DET+LMM+PERSON+FN * NGR g + p ) , and the gold reference . Uses train +The diagnoser , based on b ) , outputs a diagnosis which consists of lists of correct , contradictory and non-mentioned objects and relations from the student 's answer . Extends train +The formalization of DLRs provided by defines a formal lexical rule specification language and provides a semantics for that language in two steps : A rewrite system enriches the lexical rule specification into a fully explicit description of the kind shown in Figure 1 . Background train +Similar to our previous work ( Chan and Ng , 2005b ) , we used the supervised WSD approach described in ( ) for our experiments , using the naive Bayes algorithm as our classifier . Uses train +Similar to ( a ) , our summarization system is , which consists of three key components : an initial sentence pre-selection module to select some important sentence candidates ; the above compression model to generate n-best compressions for each sentence ; and then an ILP summarization method to select the best summary sentences from the multiple compressed sentences . CompareOrContrast train +The extraction procedure consists of three steps : First , the bracketing of the trees in the Penn Treebank is corrected and extended based on the approaches of and Collins ( 1997 ) . Background train +transition-based dependency parsing framework ( ) using an arc-eager transition strategy and are trained using the perceptron algorithm as in Zhang and Clark ( 2008 ) with a beam size of 8 . Uses train +But their importance has grown far beyond machine translation : for instance , transferring annotations between languages ( Yarowsky and Ngai 2001 ; Hwa et al. 2005 ; Ganchev , Gillenwater , and Taskar 2009 ) ; discovery of paraphrases ( ) ; and joint unsupervised POS and parser induction across languages ( Snyder and Barzilay 2008 ) . Motivation train +each relevant document is retrieved ( ) . Uses train + have previously examined the task of categorizing sentences in medical abstracts using supervised discriminative machine learning techniques . Background train +This recognizer incrementally outputs word hypotheses as soon as they are found in the best-scored path in the forward search ( ) using the ISTAR ( Incremental Structure Transmitter And Receiver ) protocol , which conveys word graph information as well as word hypotheses . Background train +The recent great advances in speech and language technologies have made it possible to build fully implemented spoken dialogue systems ( ; Allen et al. , 1996 ; Zue et al. , 2000 ; Walker et al. , 2000 ) . Background train +We posit that this would not have a significant effect on the results , in particular for MML-based classification techniques , such as Decision Graphs ( ) . Background val + showed how the perceptron algorithm can be used to efficiently compute the best parse with DOP1 's subtrees , reporting a 5.1 % relative reduction in error rate over the model in Collins ( 1999 ) on the WSJ . Background train +Tateisi et al. also translated LTAG into HPSG ( ) . CompareOrContrast train +The system utilizes several large size biological databases including three NCBI databases ( GenPept [ 11 ] , RefSeq , and Entrez GENE [ 13 ] ) , PSD database from Protein Information Resources ( PIR ) [ 14 ] , and Uses train +The LM uses the monolingual data and is trained as a five-gram9 using the SRILM-Toolkit ( ) . Uses train +There has also been work focused upon determining the political leaning ( e.g. , `` liberal '' vs. `` conservative '' ) of a document or author , where most previously-proposed methods make no direct use of relationships between the documents to be classified ( the `` unlabeled '' texts ) ( ; Efron , 2004 ; Mullen and Malouf , 2006 ) . Background train +A number of speech understanding systems have been developed during the past fifteen years ( Barnett et al. 1980 , Dixon and Martin 1979 , Erman et al. 1980 , Haton and Pierrel 1976 , Lea 1980 , , Medress 1980 , Reddy 1976 , Walker 1978 , and Wolf and Woods 1980 ) . CompareOrContrast train +Secondly , the cooperative principle of , 1978 ) , under the assumption that referential levels of a writer and a reader are quite similar , implies that the writer should structure the text in a way that makes the construction of his intended model easy for the reader ; and this seems to imply that he should appeal only to the most direct knowledge of the reader . Motivation train +In modern syntactic theories ( e.g. , lexical-functional grammar [ LFG ] [ Kaplan and Bresnan 1982 ; Bresnan 2001 ; Dalrymple 2001 ] , head-driven phrase structure grammar [ HPSG ] [ Pollard and Sag 1994 ] , tree-adjoining grammar [ TAG ] [ ] , and combinatory categorial grammar [ CCG ] [ Ades and Steedman 1982 ] ) , the lexicon is the central repository for much morphological , syntactic , and semantic information . Background train +In this paper , inspired by KNN-SVM ( ) , we propose a local training method , which trains sentence-wise weights instead of a single weight , to address the above two problems . Motivation train +1990 ) , on linguisitic acquisition ( by the use of Part-of-Speech filters hand-crafted by a linguist ) ( Oueslati , 1999 ) or , more frequently , on a combination of the two ( Smadja , 1993 ; , for example ) . CompareOrContrast train + report excellent part-of-speech tagging results using a handcrafted approach that is close to OT .3 More speculatively , imagine an OT grammar for stylistic revision of parsed sentences . Background train +Thus for instance , ( ; Copestake et al. , 2001 ) describes a Head Driven Phrase Structure Grammar ( HPSG ) which supports the parallel construction of a phrase structure ( or derived ) tree and of a semantic representation and ( Dalrymple , 1999 ) show how to equip Lexical Functional grammar ( LFG ) with a glue semantics . Background train +While many linguistic theories state subcategorization requirements in terms of phrase structure ( CFG categories ) , questions the viability and universality of such an approach because of the variety of ways in which grammatical functions may be realized at the language-specific constituent structure level . Background train +We are going to make such a comparison with the theories proposed by J. , 1982 ) that represent a more computationally oriented approach to coherence , and those of T.A. van Dijk and W. Kintch ( 1983 ) , who are more interested in addressing psychological and cognitive aspects of discourse coherence . CompareOrContrast train +11 From ( ) , we find that the performance of SAMT system is similar with the method of labeling SCFG rules with POS tags . Motivation train +In this paper we focus on the exploitation of the LDOCE grammar coding system ; Alshawi et al. ( 1985 ) and describe further research in Cambridge utilising different types of information available in LDOCE . Background train + observed that some annotators were not familiar with the exact definition of semantic relatedness . Motivation train +All current approaches to monolingual TE , either syntactically oriented ( Rus et al. , 2005 ) , or applying logical inference ( Tatu and Moldovan , 2005 ) , or adopting transformation-based techniques ( ; Bar-Haim et al. , 2008 ) , incorporate different types of lexical knowledge to support textual inference . Background train +Although originally developed as a tool to assist in query formulation , pointed out that PICO frames can be employed to structure IR results for improving precision . Background train +There has been some controversy , at least for simple stemmers ( Lovins , 1968 ; Porter , 1980 ) , about the effectiveness of morphological analysis for document retrieval ( ; Krovetz , 1993 ; Hull , 1996 ) . Background train +in history-based models ( ) , the probability estimate for each derivation decision di is conditioned on the previous derivation decisions d1 , ... , d , _ 1 , which is called the derivation history at step i . Uses train +The reader is referred to for a more detailed discussion of our use of constraint propagation .32 We illustrate the result of constraint propagation with our example grammar . Background train +Such tools make it easy to run most current approaches to statistical markup , chunking , normalization , segmentation , alignment , and noisy-channel decoding , ' including classic models for speech recognition ( Pereira and Riley , 1997 ) and machine translation ( ) . Background train +Such technologies require significant human input , and are difficult to create and maintain ( ) . Background train +We can define PCAT using a probabilistic grammar ( ) . Background val +It is these orthographic variations and complex morphological structure that make Arabic language processing challenging ( ; Xu et al. , 2002 ) . Background train +The first is the one used in the chunking competition in CoNLL-2000 ( Tjong Kim ) . Uses train +There have been many studies on parsing techniques ( Poller and Becker , 1998 ; Flickinger et al. , 2000 ) , ones on disambiguation models ( Chiang , 2000 ; ) , and ones on programming/grammar-development environ - Background train +For example , experimented with abstracts and full article texts in the task of automatically generating index term recommendations and discovered that using full article texts yields at most a 7.4 % improvement in F-score . Background train +The ten most specific nouns have been produced by comparing our corpus of computing to the French corpus Le Monde , composed of newspaper articles ( ) . Uses train +Secondly , we need to investigate techniques for identifying identical documents , virtually identical documents and highly repetitive documents , such as those pioneered by b ) and shingling techniques described by Chakrabarti ( 2002 ) . Future train +Following the work of , we implement a linear-chain CRF merging system using the following features : stemmed ( separated ) surface form , part-of-speech14 and frequencies from the training corpus for bigrams/merging of word and word +1 , word as true prefix , word +1 as true suffix , plus frequency comparisons of these . Uses train + attempts to improve on the approach of Brent ( 1993 ) by passing raw text through a stochastic tagger and a finite-state parser ( which includes a set of simple rules for subcategorization frame recognition ) in order to extract verbs and the constituents with which they co-occur . Background train +From an IR view , a lot of specialized research has already been carried out for medical applications , with emphasis on the lexico-semantic aspects of dederivation and decomposition ( Pacak et al. , 1980 ; Norton and Pacak , 1983 ; Wolff , 1984 ; Wingert , 1985 ; Dujols et al. , 1991 ; ) . Background val +Rather than producing a complete analysis of sentences , the alternative is to perform only partial analysis of the syntactic structures in a text ( ; Abney , 1991 ; Greffenstette , 1993 ) . Background train +Thus , over the past few years , along with advances in the use of learning and statistical methods for acquisition of full parsers ( Collins , 1997 ; a ; Charniak , 1997b ; Ratnaparkhi , 1997 ) , significant progress has been made on the use of statistical learning methods to recognize shallow parsing patterns syntactic phrases or words that participate in a syntactic relationship ( Church , 1988 ; Ramshaw and Marcus , 1995 ; Argamon et al. , 1998 ; Cardie and Pierce , 1998 ; Munoz et al. , 1999 ; Punyakanok and Roth , 2001 ; Buchholz et al. , 1999 ; Tjong Kim Sang and Buchholz , 2000 ) . Background train +Problems such as these have motivated research on more abstract , dependencybased parser evaluation ( e.g. , Lin 1995 ; Carroll , Briscoe , and Sanfilippo 1998 ; Carroll et al. 2002 ; Clark and Hockenmaier 2002 ; King et al. 2003 ; Preiss 2003 ; ; Miyao and Tsujii 2004 ) . Motivation train +They are widely used in MT as a way to figure out how to translate input in one language into output in another language ( ) . Background train +Further details about the properties of entropy can be found in textbooks on information theory ( e.g. , ) . Background train + , for example , discussed the transcripts of a dialogue between people who assemble a piece of garden furniture ( originally recorded by Candy Sidner ) . Background train +Here , PV ( A ) represents an ascent direction chosen as follows : For inequality constraints , it is the projected gradient ( ) ; for equality constraints with slack , we use conjugate gradient ( Nocedal and Wright 1999 ) , noting that when A = 0 , the objective is not differentiable . Uses val +Some methods are based on likelihood ( Och and Ney , 2002 ; Blunsom et al. , 2008 ) , error rate ( ; Zhao and Chen , 2009 ; Pauls et al. , 2009 ; Galley and Quirk , 2011 ) , margin ( Watanabe et al. , 2007 ; Chiang et al. , 2008 ) and ranking ( Hopkins and May , 2011 ) , and among which minimum error rate training ( MERT ) ( Och , 2003 ) is the most popular one . Motivation train +This idea was inspired by , who used a list of arguments surrounding the main verb together with the verb 's subcategorization information and previously processed examples to analyse semantic roles ( case relations ) . Motivation train +, `` domain circumscription '' ( cfXXX ) , and their kin . CompareOrContrast train +While these approaches have been reasonably successful ( see Mitkov ( 2002 ) ) , speculate that deeper linguistic knowledge needs to be made available to resolvers in order to reach the next level of performance . Background train +A parsing experiment shows that an efficient HPSG parser with the obtained grammar achieved a significant speed-up against an existing LTAG parser ( ) . CompareOrContrast train +Features were discovered from the actual sentence plan trees that the SPG generated through the feature derivation process described below , in a manner similar to that used by . CompareOrContrast train +Such questions are typically answered by designing appropriate priming experiments ( ) or other lexical decision tasks . Background train +The system is in the form of an agenda-driven chart-based parser whose foundation is similar to previous formalizations of Chomsky 's Minimalist Program ( ; Harkema , 2000 ; Niyogi , 2001 ) . CompareOrContrast train + and Akkerman et al. ( 1985 ) provide a more detailed analysis of the information encoded by the LDOCE grammar codes and discuss their efficacy as a system of linguistic description . Background train +The research described below is taking place in the context of three collaborative projects ( Boguraev , 1987 ; Russell et al. , 1986 ; ) to develop a general-purpose , wide coverage morphological and syntactic analyser for English . Background train +The EDR has close ties to the named entity recognition ( NER ) and coreference resolution tasks , which have been the focus of several recent investigations ( Bikel et al. , 1997 ; Miller et al. , 1998 ; Borthwick , 1999 ; Mikheev et al. , 1999 ; Soon et al. , 2001 ; ; Florian et al. , 2004 ) , and have been at the center of evaluations such as : MUC-6 , MUC-7 , and the CoNLL '02 and CoNLL '03 shared tasks . Background train +Viewed in this way , gradable adjectives are an extreme example of the `` efficiency of language '' ( ) : Far from meaning something concrete like `` larger than 8 cm '' -- a concept that would have very limited applicability -- or even something more general like `` larger than the average N , '' a word like large is applicable across a wide range of different situations . CompareOrContrast train +However , since work in this direction has started , a significant progress has also been made in the research on statistical learning of full parsers , both in terms of accuracy and processing time ( Charniak , 1997b ; Charniak , 1997a ; ; Ratnaparkhi , 1997 ) . Background train +It is defined on different kinds of textual units , e.g. documents , parts of a document ( e.g. words and their surrounding context ) , words or concepts ( ) .2 Linguistic distance between words is inverse to their semantic similarity or relatedness . Background train +However , the greatest increase is in the amount of raw text available to be processed , e.g. the English Gigaword Corpus ( Linguistic Data ) . Background val +For example , such schema can serve as a mean to represent translation examples , or find structural correspondences for the purpose of transfer grammar learning ( Menezes & Richardson , 2001 ) , ( Aramaki et al. , 2001 ) , ( ) , ( Meyers et al. , 2000 ) , ( Matsumoto et al. , 1993 ) , ( kaji et al. , 1992 ) , and example-base machine translation EBMT3 ( Sato & Nagao , 1990 ) , ( Sato , 1991 ) , ( Richardson et al. , 2001 ) , ( Al-Adhaileh & Tang , 1999 ) . Background val + introduced the log-linear model for statistical machine translation ( SMT ) , in which translation is considered as the following optimization problem : Background train +Representative systems are described in Boisen et al. ( 1989 ) , De Mattia and Giachin ( 1989 ) , Niedermair ( 1989 ) , , and Young ( 1989 ) . Background train +For compound splitting , we follow , using linguistic knowledge encoded in a rule-based morphological analyser and then selecting the best analysis based on the geometric mean of word part frequencies . Uses train + showed that the use of a POS factor only results in negligible BLEU improvements , but we need access to the POS in our inflection prediction models . Background train +Over the last decade there has been a lot of interest in developing tutorial dialogue systems that understand student explanations ( Jordan et al. , 2006 ; Graesser et al. , 1999 ; Aleven et al. , 2001 ; Buckley and Wolska , 2007 ; Nielsen et al. , 2008 ; ) , because high percentages of selfexplanation and student contentful talk are known to be correlated with better learning in humanhuman tutoring ( Chi et al. , 1994 ; Litman et al. , 2009 ; Purandare and Litman , 2008 ; Steinhauser et al. , 2007 ) . Background train +We first identified the most informative unigrams and bigrams using the information gain measure ( ) , and then selected only the positive outcome predictors using odds ratio ( Mladenic and Grobelnik 1999 ) . Uses train +Recently , several alternative , often quite sophisticated approaches to collective classification have been proposed ( Neville and Jensen , 2000 ; Lafferty et al. , 2001 ; Getoor et al. , 2002 ; Taskar et al. , 2002 ; ; Taskar et al. , 2004 ; McCallum and Wellner , 2004 ) . Background val +It compares favorably to other stemming or root extraction algorithms ( ; Al-Shalabi and Evens , 1998 ; and Houmame , 1999 ) , with a performance of over 97 % for extracting the correct root in web documents , and it addresses the challenge of the Arabic broken plural and hollow verbs . Motivation train +We perceive that these results can be extended to other language models that properly embed bilexical context-free grammars , as for instance the more general history-based models used in ( ) and ( Chelba and Jelinek , 1998 ) . Future train +Clearly , what it takes for the adjective to be applicable has not been cast in stone , but is open to fiat : the speaker may decide that 8 cm is enough , or the speaker may set the standards higher ( cfXXX , ) . Background train +Many investigators ( e.g. ; Elowitz et al. 1976 ; Luce et al. 1983 ; Cahn 1988 ) have suggested that the poor prosody of synthetic speech , in comparison with natural speech , is the primary factor leading to difficulties in the comprehension of fluent synthetic speech . Motivation train +The PERSIVAL project , the most comprehensive study of such techniques applied on medical texts to date , leverages patient records to generate personalized summaries in response to physicians ' queries ( McKeown , Elhadad , and Hatzivassiloglou 2003 ; ) . CompareOrContrast train +See for further discussion . Background train +We gather similar words using a ) , mining similar verbs from a comparable-sized parsed corpus , and collecting similar nouns from a broader 10 GB corpus of English text .4 We also use Keller and Lapata ( 2003 ) 's approach to obtaining web-counts . Uses train +Thus , for example , it can acquire a `` script '' such as the one for going to a restaurant as defined in . Background train +ones , DIRT ( ) , VerbOcean ( Chklovski and Pantel , 2004 ) , FrameNet ( Baker et al. , 1998 ) , and Wikipedia ( Mehdad et al. , 2010 ; Kouylekov et al. , 2009 ) . Background train +There is some literature on procedure acquisition such as the LISP synthesis work described in Biermann et al. ( 1984 ) and the PROLOG synthesis method of . CompareOrContrast train +The system uses a domain-specific content planner to produce input to the surface realizer based on the strategy decision , and a FUF/SURGE ( ) generation system to produce the appropriate text . Uses train +However , each of these fields requires further decoding and restructuring to provide client programs with easy access to the information they require ( see for further discussion ) . Background train +Our results also confirm the insights gained by , who observed that in crossdomain polarity analysis adding more training data is not always beneficial . CompareOrContrast train +We consider the Creative Commons model as the most suitable one to let each author choose the rights to reserve ( ) . Uses val +It compares favorably to other stemming or root extraction algorithms ( Yates and Neto , 1999 ; Al-Shalabi and Evens , 1998 ; and ) , with a performance of over 97 % for extracting the correct root in web documents , and it addresses the challenge of the Arabic broken plural and hollow verbs . Motivation train +In a similar vein , showed that a different feature-topic model improved predictions on a fill-in-the-blank task . Background val +As ( ) show , lexical information improves on NP and VP chunking as well . Future val +But while Bod 's estimator obtains state-of-the-art results on the WSJ , comparable to Charniak ( 2000 ) and Collins ( 2000 ) , Bonnema et al. 's estimator performs worse and is comparable to . Background train +In multi-party discussion people usually mention each other 's name for the purpose of disentanglement ( ) . Background train +Fortunately , indirect associations are usually not difficult to identify , because they tend to be weaker than the direct associations on which they are based ( c ) . Background train +The work of and Steedman , Osborne , et al. ( 2003 ) suggests that co-training can be helpful for statistical parsing . Background val +TNT refers to the HPSG parser ( ) , C++ implementation of the two-phase parsing algorithm that performs filtering with a compiled CFG ( phase 1 ) and then executes feature unification ( phase 2 ) . CompareOrContrast train +1 ° The body of a plan can be an action or sequence of actions , a goal or sequence 9 Moore and Paris also note that `` a generation system must maintain the kinds of information outlined by Grosz and Sidner '' ( , 203 ) . Background val +Surveys and articles on the topic include Lamarche and Retord ( 1996 ) , de Groote and Retord ( 1996 ) , and . Background train +Another line of research that is correlated with ours is recognition of agreement/disagreement ( Misra and Walker , 2013 ; Yin et al. , 2012 ; Abbott et al. , 2011 ; Andreas et al. , 2012 ; ; Hillard et al. , 2003 ) and classification of stances ( Walker et al. , 2012 ; Somasundaran and Wiebe , 2010 ) in online forums . CompareOrContrast val +The standard way to handle this problem is to handcraft a finite set of features which provides a sufficient summary of the unbounded history ( Ratnaparkhi , 1999 ; Collins , 1999 ; ) . CompareOrContrast train +Since the arguments can provide useful semantic information , the SRL is crucial to many natural language processing tasks , such as Question and Answering ( Narayanan and Harabagiu 2004 ) , Information Extraction ( ) , and Machine Translation ( Boas 2002 ) . Background train +Most approaches rely on VerbNet ( ) and FrameNet ( Baker et al. , 1998 ) to provide associations between verbs and semantic roles , that are then mapped onto the current instance , as shown by the systems competing in semantic role labelling competitions ( Carreras and Marquez , 2004 ; Carreras and Marquez , 2005 ) and also ( Gildea and Jurafsky , 2002 ; Pradhan et al. , 2005 ; Shi and Mihalcea , 2005 ) . Background train +In previous work ( Bachenko et al. 1986 ) , we described an experimental text-to-speech system that determined prosodic phrasing for the Olive -- Liberman synthesizer ( ) . Background train +Note that this ensures that greater importance is attributed to longer chunks , as is usual in most EBMT systems ( cfXXX Sato and Nagao 1990 ; Veale and Way 1997 ; ) .7 As an example , consider the translation into French of the house collapsed . Background train +This approach is taken , for example , in LKB ( Copestake 1992 ) where lexical rules are introduced on a par with phrase structure rules and the parser makes no distinction between lexical and nonlexical rules ( , 31 ) . CompareOrContrast train +27 argue that semi-productivity of lexical rules , which can be understood as a generalization of exceptions to lexical rules , can be integrated with our approach by assigning probabilities to the automaton associated with a particular lexical entry . Background train +Results from other systems show that measures of semantic coherence between a student and a system were positively associated with higher learning gain ( ) . Future train +These automatic transformations are based on linguistic rules ( ) . Uses train +• Only an automatic evaluation was performed , which relied on having model responses ( Berger and Mittal 2000 ; ) . CompareOrContrast train +In comparison , the tag set of the Buckwalter Morphological Analyzer ( ) used in the PATB has a core POS set of 44 tags ( CORE44 ) before morphological extension .8 Cross-linguistically , a core set containing around 12 tags is often CompareOrContrast train +• use of low level knowledge from the speech recognition phase , • use of high level knowledge about the domain in particular and the dialogue task in general , • a `` continue '' facility and an `` auto-loop '' facility as described by Biermann and Krishnaswamy ( 1976 ) , • a `` conditioning '' facility as described by , • implementation of new types of paraphrasing , • checking a larger environment in the expectation acquisition algorithm when deciding if an incoming sentence is the same or similar to one already seen , and • examining inter-speaker dialogue patterns . Future train + present detailed studies on the task of named entity recognition , which discusses and compares different methods on multiple aspects including chunk representation , inference method , utility of non-local features , and integration of external knowledge . Background train +mers ( ; Porter , 1980 ) demonstrably improve retrieval performance . Background train +Cases like this would be covered if the decision-theoretic property of Pareto optimality ( e.g. , ) was used as the sole criterion : Formally , an object r E C has a Pareto-optimal combination of Values V iff there is no other x E C such that Background train +A formula for the test set perplexity ( ) is :13 Background train +This method follows a traditional Information Retrieval paradigm ( ) , where a query is represented by the content terms it contains , and the system retrieves from the corpus a set of documents that best match this query . Uses train +Table 1 gives the interpretations of eight adjective-noun combinations discussed in and Vendler ( 1968 ) . Uses train +The Penn Treebank results reported here for the Markov model approach are at least equivalent to those reported for the Maximum Entropy approach in ( ) . CompareOrContrast train + , by comparison , employ 163 distinct predefined frames . Background train +Brockmann and Lapata ( 2003 ) have showed that WordNet-based approaches do not always outperform simple frequency-based models , and a number of techniques have been recently proposed which may offer ideas for refining our current unsupervised approach ( Erk , 2007 ; ) . Future train +fθ on demand ( ) can pay off here , since only part of fθ may be needed subsequently . ) Background val +An approach ( also based on regulation of the succession of rule application ) to the associated problem of spurious ambiguity is given in but again , to our knowledge , there is no predictive relation between incremental combinatory processing and the kind of processing phenomena cited in the introduction . Background train +This equivalence is doing essentially the same job as Pereira 's pronoun abstraction schema in . CompareOrContrast train +Finally , we experiment with a method for combining phrase tables proposed in ( ; Nakov and Ng , 2012 ) . Uses train +Our work is more similar to NLG work that concentrates on structural constraints such as generative poetry ( Greene et al. , 2010 ) ( Colton et al. , 2012 ) ( Jiang and Zhou , 2008 ) or song lyrics ( ) ( Ramakrishnan A et al. , 2009 ) , where specified meter or rhyme schemes are enforced . CompareOrContrast train +For the cases where retrieval took place , we used F-score ( van Rijsbergen 1979 ; ) to determine the similarity between the response from the top-ranked document and the real response ( the formulas for F-score and its contributing factors , recall and precision , appear in Section 4.2 ) . Uses val +These translations gave rise to a number of automatically constructed linguistic resources : ( 1 ) the original ( source , target ) phrasal translation pairs , ( 2 ) the marker lexicon , ( 3 ) the gen11 Thanks are due to one of the anonymous reviewers for pointing out that our wEBMT system , seeded with input from multiple translation systems , with a postvalidation process via the Web ( amounting to an n-gram target language model ) , in effect forms a multiengine MT system as described by , Frederking et al. ( 1994 ) , and Hogan and Frederking ( 1998 ) . CompareOrContrast val +This includes work on generalized expectation ( Mann and McCallum , 2010 ) , posterior regularization ( Ganchev et al. , 2010 ) and constraint driven learning ( Chang et al. , 2007 ; ) . Background val +The list , a synthesis of a number of relation lists cited in the literature , has been designed to be general , domainindependent ( a ) . Motivation val +Unlike our approach , those of and Hockenmaier , Bierner , and Baldridge ( 2004 ) include a substantial initial correction and clean-up of the Penn-II trees . CompareOrContrast train +2The algorithm was implemented by the the authors , following the description in . Uses train +Our motivation for generation of material for language education exists in work such as Sumita et al. ( 2005 ) and , which deal with automatic generation of classic fill in the blank questions . Motivation train +In most recent research , NEs ( person , location and organisations ) are extracted from the text and used as a source of evidence to calculate the similarity between documents - see for instance ( ; Chen and Martin , 2007 ; Popescu and Magnini , 2007 ; Kalashnikov et al. , 2007 ) . Background train +Many researchers use the GIZA + + software package ( ) as a black box , selecting IBM Model 4 as a compromise between alignment quality and efficiency . Background train +Some works abstract perception via the usage of symbolic logic representations ( ; Chen and Mooney , 2011 ; Matuszek et al. , 2012 ; Artzi and Zettlemoyer , 2013 ) , while others choose to employ concepts elicited from psycholinguistic and cognition studies . Background train +Over the past decade , researchers at IBM have developed a series of increasingly sophisticated statistical models for machine translation ( Brown et al. , 1988 ; ; Brown et al. , 1993a ) . Background train +One area of current interest concerns the left-to-right arrangement of premodifying adjectives within an NP ( e.g. , ; Malouf 2000 ) . Background train +Models of translational equivalence that are ignorant of indirect associations have `` a tendency ... to be confused by collocates '' ( ) . Background train +Note that although our feature set was drawn primarily from our prior uncertainty detection experiments ( Forbes-Riley and Litman , 2011a ; ) , we have also experimented with other features , including state-of-theart acoustic-prosodic features used in the last Interspeech Challenges ( Schuller et al. , 2010 ; Schuller et al. , 2009b ) and made freely available in the openSMILE Toolkit ( Florian et al. , 2010 ) . Extends train +For this mention-pair coreference model φ ( u , v ) , we use the same set of features used in . Uses train +Compared to the reranking technique in , who obtained an LP of 89.9 % and an LR of 89.6 % , our results show a 9 % relative error rate reduction . CompareOrContrast train +The paper compares and contrasts the training time needed and performance achieved by our modified learner with two other systems : a standard transformation-based learner , and the ICA system ( ) . CompareOrContrast train +There has also been work focused upon determining the political leaning ( e.g. , `` liberal '' vs. `` conservative '' ) of a document or author , where most previously-proposed methods make no direct use of relationships between the documents to be classified ( the `` unlabeled '' texts ) ( Laver et al. , 2003 ; ; Mullen and Malouf , 2006 ) . Background train +It projects a functional head , voice ( ) , whose specifier is the external argument . Background train +Finite state transducers , which can be learned from bilingual corpora , have been proposed for automatic translation ( Amengual et al. , 2000 ) , as have been bilingual stochastic grammars ( ) . Background train +This is noticeable for German ( Brants et al. , 2002 ) and Portuguese ( ) , which still have high overall accuracy thanks to very high attachment scores , but much more conspicuous for Czech ( B ¨ ohmov ´ a et al. , 2003 ) , Dutch ( van der Beek et al. , 2002 ) and Slovene ( Dˇzeroski et al. , 2006 ) , where root precision drops more drastically to about 69 % , 71 % and 41 % , respectively , and root recall is also affected negatively . CompareOrContrast train +3 The degree of precision of the measurement ( , Section 1.5 ) determines which objects can be described by the GRE algorithm , since it determines which objects count as having the same size . Background val +There have already been several attempts to develop distributed NLP systems for dialogue systems ( ) and speech recognition ( Hacioglu and Pellom , 2003 ) . Background train +Table look-up using an explicit translation lexicon is sufficient and preferable for many multilingual NLP applications , including `` crummy '' MT on the World Wide Web ( Church & Hovy , 1993 ) , certain machine-assisted translation tools ( e.g. ( ; Melamed , 1996b ) ) , concordancing for bilingual lexicography ( Catizone et al. , 1993 ; Gale & Church , 1991 ) , computerassisted language learning , corpus linguistics ( Melby . Background val +SWIZZLE is a multilingual enhancement of COCKTAIL ( ) , a coreference resolution system that operates on a mixture of heuristics that combine semantic and textual cohesive information ' . Extends train +( Davis and Ogden , 1997 ; ; Hull and ( 3refenstette , 1996 ) . CompareOrContrast train +de URL : http://www.sfs.nphil.uni-tuebingen.de/sfb / b4home.html 1 This is , for example , the case for all proposals working with verbal lexical entries that raise the arguments of a verbal complement ( Hinrichs and Nakazawa 1989 ) that also use lexical rules such as the Complement Extraction Lexical Rule ( ) or the Complement Cliticization Lexical Rule ( Miller and Sag 1993 ) to operate on those raised elements . Background train + attempt to translate technical terms using word relation matrices , although the resource from which such relations are derived is a pair of nonparallel corpora . Background train +The system is in the form of an agenda-driven chart-based parser whose foundation is similar to previous formalizations of Chomsky 's Minimalist Program ( Stabler , 1997 ; ; Niyogi , 2001 ) . CompareOrContrast val +Each component will return a confidence measure of the reliability of its prediction , c.f. ( ) . Motivation train +In fact , most of the features3 implemented in existing coreference resolution systems rely solely on mention heads ( ) . Background val +The problem of handling ill-formed input has been studied by Carbonell and Hayes ( 1983 ) , , Jensen et al. ( 1983 ) , Kwasny and Sondheimer ( 1981 ) , Riesbeck and Schank ( 1976 ) , Thompson ( 1980 ) , Weischedel and Black ( 1980 ) , and Weischedel and Sondheimer ( 1983 ) . CompareOrContrast train +In most recent research , NEs ( person , location and organisations ) are extracted from the text and used as a source of evidence to calculate the similarity between documents - see for instance ( Blume , 2005 ; ; Popescu and Magnini , 2007 ; Kalashnikov et al. , 2007 ) . Background train +This conception of lexical rules thus can be understood as underlying the computational approach that treats lexical rules as unary phrase structure rules as , for example , adopted in the LKB system ( ) . Background train +Typical letter-to-sound rule sets are those described by Ainsworth ( 1973 ) , McIlroy ( 1973 ) , Elovitz et al. ( 1976 ) , Hurmicutt ( 1976 ) , and . Background train + describe an efficient algorithm ( of linear complexity in the number of training sentences ) for computing the LDA transform matrix , which entails computing the withinand between-covariance matrices of the classes , and using Singular Value Decomposition ( SVD ) to compute the eigenvectors of the new space . Uses val +For better comparison with work of others , we adopt the suggestion made by to evaluate the parsing quality on sentences up to 70 tokens long . Uses val +Some examples include text categorization ( ) , base noun phrase chunking ( Ngai and Yarowsky 2000 ) , part-of-speech tagging ( Engelson Dagan 1996 ) , spelling confusion set disambiguation ( Banko and Brill 2001 ) , and word sense disambiguation ( Fujii et al. 1998 ) . Background val +Differently , designed a sampler to infer an STSG by fixing the tree structure and exploring the space of alignment . Motivation val +They use a Bag of Visual Words ( BoVW ) model ( ) to create a bimodal vocabulary describing documents . Background train +W. discussed sentences of the form * This is a chair but you can sit on it . Background train +This imbalance foils thresholding strategies , clever as they might be ( ; Wu & Xia , 1994 ; Chen , 1996 ) . Background train +The problem of handling ill-formed input has been studied by Carbonell and Hayes ( 1983 ) , Granger ( 1983 ) , Jensen et al. ( 1983 ) , Kwasny and Sondheimer ( 1981 ) , Riesbeck and Schank ( 1976 ) , , Weischedel and Black ( 1980 ) , and Weischedel and Sondheimer ( 1983 ) . CompareOrContrast train +Prototypes of Internet search engines for linguists , corpus linguists and lexicographers have been proposed : WebCorp ( Kehoe and Renouf , 2002 ) , KWiCFinder ( Fletcher , 2004a ) and the Linguist 's Search Engine ( ; Resnik and Elkiss , 2003 ) . Background val +In most recent research , NEs ( person , location and organisations ) are extracted from the text and used as a source of evidence to calculate the similarity between documents - see for instance ( Blume , 2005 ; Chen and Martin , 2007 ; Popescu and Magnini , 2007 ; ) . Background train +More recently , Silberer et al. ( 2013 ) show that visual attribute classifiers , which have been immensely successful in object recognition ( ) , act as excellent substitutes for feature Background train +In other methods , lexical resources are specifically tailored to meet the requirements of the domain ( ) or the system ( Gomez , 1998 ) . Background val +Such systems extract information from some types of syntactic units ( clauses in ( Fillmore and Atkins , 1998 ; ; Hull and Gomez , 1996 ) ; noun phrases in ( Hull and Gomez , 1996 ; Rosario et al. , 2002 ) ) . Background val +The third version ( VOYAGER ) serves as an interface both with a recognizer and with a functioning database back-end ( ) . Uses train +We use the non-projective k-best MST algorithm to generate k-best lists ( ) , where k = 8 for the experiments in this paper . Uses train +As for work on Arabic ( MSA ) , results have been reported on the PATB ( Kulick , Gabbard , and Marcus 2006 ; Diab 2007 ; Green and Manning 2010 ) , the Prague Dependency Treebank ( PADT ) ( Buchholz and Marsi 2006 ; ) and the CATiB ( Habash and Roth 2009 ) . Background train +For example , McKnight and Srinivasan ( 2003 ) describe a machine learning approach to automatically label sentences as belonging to introduction , methods , results , or conclusion using structured abstracts as training data ( see also ) . Background train +• Graph transformations for recovering nonprojective structures ( ) . Uses train + 's CCM is an unlabeled bracketing model that generates the span of part-of-speech tags that make up each constituent and the pair of tags surrounding each constituent span ( as well as the spans and contexts of each non-constituent ) . Background train +This process produces a hierarchical clustering of the word types in the corpus , and these clusterings have been found useful in many applications ( ; Koo et al. , 2008 ; Miller et al. , 2004 ) . Motivation train +Opposition ( called `` adversative '' or `` contrary-to-expectation '' by Halliday and Hasan 1976 ; cfXXX also , p. 672 ) . Background train +Aside from the extraction of theory-neutral subcategorization lexicons , there has also been work in the automatic construction of lexical resources which comply with the principles of particular linguistic theories such as LTAG , CCG , and HPSG ( ; Xia 1999 ; Hockenmaier , Bierner , and Baldridge 2004 ; Nakanishi , Miyao , and Tsujii 2004 ) . Background train +1 The representation in is even more compact than ours for grammars that are not self-embedding . CompareOrContrast train +29 This improvement of the covariation encoding can also be viewed as an instance of the program transformation technique referred to as deletion of clauses with a finitely failed body ( ) . CompareOrContrast train +The reordering models we describe follow our previous work using function word models for translation ( ; Setiawan et al. , 2009 ) . Extends train +Although there are other discussions of the paragraph as a central element of discourse ( e.g. Chafe 1979 , , Longacre 1979 , Haberlandt et al. 1980 ) , all of them share a certain limitation in their formal techniques for analyzing paragraph structure . CompareOrContrast train +For example , such schema can serve as a mean to represent translation examples , or find structural correspondences for the purpose of transfer grammar learning ( Menezes & Richardson , 2001 ) , ( Aramaki et al. , 2001 ) , ( Watanabe et al. , 2000 ) , ( Meyers et al. , 2000 ) , ( Matsumoto et al. , 1993 ) , ( kaji et al. , 1992 ) , and example-base machine translation EBMT3 ( Sato & Nagao , 1990 ) , ( Sato , 1991 ) , ( Richardson et al. , 2001 ) , ( ) . Background train +In corpus linguistics building such megacorpora is beyond the scope of individual researchers , and they are not easily accessible ( Kennedy , 1998 : 56 ) unless the web is used as a corpus ( ) . Background train +Here 11 is an optimization precision , oc is a step size chosen with the strong Wolfe 's rule ( ) . Uses train +See ( ) for a discussion . Background train +Figure 2 ( a ) shows the frame-based semantic representation for the utterance `` What time is Analyze This playing 2 See ( ) for how MIMIC 's dialoguelevel knowledge is used to override default prosodic assignments for concept-to-speech generation . Background train +For Berkeley system , we use the reported results from . CompareOrContrast train +Instead , we will adopt the nomenclature of the Automatic Content Extraction program ( ) : we will call the instances of textual references to objects/abstractions mentions , which can be either named ( e.g. John Mayor ) , nominal ( the president ) or pronominal ( she , it ) . Uses train +Such a component would serve as the first stage of a clinical question answering system ( ) or summarization system ( McKeown et al. , 2003 ) . Future train +In the areas of Natural Language Processing ( NLP ) and computational linguistics , proposals have been made for using the computational Grid for data-intensive NLP and text-mining for eScience ( ; Hughes et al , 2004 ) . Background train + comprehensively compares different approaches to complementation within grammatical theory providing a touchstone against which the LDOCE scheme can be evaluated . CompareOrContrast val +To prove that our method is effective , we also make a comparison between the performances of our system and , Xue ( 2008 ) . CompareOrContrast val +The obtained SCFG is further used in a phrase-based and hierarchical phrase-based system ( ) . CompareOrContrast train +32 In certain cases an extension of the constraint language with named disjunctions or contexted constraints ( Maxwell and Kaplan 1989 ; Eisele and Dorre 1990 ; ) can be used to circumvent constraint propagation . Background train + , 2009 , 2010 ) utilized Bayesian methods to learn synchronous context free grammars ( SCFG ) from a parallel corpus . CompareOrContrast val +Since the arguments can provide useful semantic information , the SRL is crucial to many natural language processing tasks , such as Question and Answering ( ) , Information Extraction ( Surdeanu et al. 2003 ) , and Machine Translation ( Boas 2002 ) . Background train +In addition to the model based upon a dictionary of stems and words , we also experimented with models based upon character n-grams , similar to those used for Chinese segmentation ( ) . CompareOrContrast train +Notable early papers on graph-based semisupervised learning include Blum and Chawla ( 2001 ) , Bansal et al. ( 2002 ) , Kondor and Lafferty ( 2002 ) , and . Background train +Due to using a global model like CRFs , our previous work in ( Zhao et al. , 2006 ; c ) reported the best results over the evaluated corpora of Bakeoff-2 until now7 . CompareOrContrast val +As already mentioned in the literature , see for example ( ) , knowledge about implicit predicates could be potentially useful for a variety of NLP tasks such as language generation , information extraction , question answering or machine translation . Background train +We use the Columbia Arabic Treebank ( CATiB ) ( ) . Uses train +This problem may be similar to the situation in which current formal grammars allow nonsensical but parsable collections of words ( e.g. , `` colorless green ideas ... '' ) , while before the advent of Chomskyan formalisms , a sentence was defined as the smallest meaningful collection of words ; , p. 546 ) gives 10 definitions of a sentence . Background train +Another technique for making better use of unlabeled data is cotraining ( ) , in which two sufficiently different learners help each other learn by labeling training data for one another . Background val +For example , modeling CASE in Czech improves Czech parsing ( ) : CASE is relevant , not redundant , and can be predicted with sufficient accuracy . Motivation train +Table look-up using an explicit translation lexicon is sufficient and preferable for many multilingual NLP applications , including `` crummy '' MT on the World Wide Web ( Church & Hovy , 1993 ) , certain machine-assisted translation tools ( e.g. ( Macklovitch , 1994 ; Melamed , 1996b ) ) , concordancing for bilingual lexicography ( Catizone et al. , 1993 ; ) , computerassisted language learning , corpus linguistics ( Melby . Background train +Following our previous work ( ; Althaus , Karamanis , and Koller 2004 ) , the input to information ordering is an unordered set of informationbearing items represented as CF lists . Extends train +There has been some controversy , at least for simple stemmers ( Lovins , 1968 ; Porter , 1980 ) , about the effectiveness of morphological analysis for document retrieval ( Harman , 1991 ; Krovetz , 1993 ; ) . Background train +This situation suggests a response-automation approach that follows the document retrieval paradigm ( ) , where a new request is matched with existing response documents ( e-mails ) . Background train +All current approaches to monolingual TE , either syntactically oriented ( Rus et al. , 2005 ) , or applying logical inference ( Tatu and ) , or adopting transformation-based techniques ( Kouleykov and Magnini , 2005 ; Bar-Haim et al. , 2008 ) , incorporate different types of lexical knowledge to support textual inference . Background val +This is the approach taken by IBM Models 4 + ( Brown et al. 1993b ; Och and Ney 2003 ) , and more recently by the LEAF model ( ) . CompareOrContrast train +We have noted that many of these desiderata make complex question answering quite similar to multi-document summarization ( b ) , but these features are also beyond the capabilities of current summarization systems . CompareOrContrast train +:472 ) , but these are the only ones which are explicit in the LDOCE coding system . Background train +This paper describes an approach for sharing resources in various grammar formalisms such as Feature-Based Lexicalized Tree Adjoining Grammar ( FB-LTAG1 ) ( Vijay-Shanker , 1987 ; Vijay-Shanker and Joshi , 1988 ) and Head-Driven Phrase Structure Grammar ( HPSG ) ( ) by a method of grammar conversion . Background train +These constructs correspond as directly as possible to properties of the linguistic structure that express them and are , to as small an extent as possible , dependent on the requirements of contextual resolution ( unlike , say , the metavariables of standard QLFs [ ] , or the labels of UDRS [ Reyle 1996 ] , which are motivated entirely by the mechanisms that operate on them after grammatical processing ) . Background train +Prototypes of Internet search engines for linguists , corpus linguists and lexicographers have been proposed : WebCorp ( Kehoe and Renouf , 2002 ) , KWiCFinder ( Fletcher , 2004a ) and the Linguist 's Search Engine ( Kilgarriff , 2003 ; ) . Background val +Table look-up using an explicit translation lexicon is sufficient and preferable for many multilingual NLP applications , including `` crummy '' MT on the World Wide Web ( Church & Hovy , 1993 ) , certain machine-assisted translation tools ( e.g. ( Macklovitch , 1994 ; b ) ) , concordancing for bilingual lexicography ( Catizone et al. , 1993 ; Gale & Church , 1991 ) , computerassisted language learning , corpus linguistics ( Melby . Background train +An off-the-shelf speech recognition device , a Nippon Electric Corporation DP-200 , was added to an existing natural language processing system , the Natural Language Computer ( NLC ) ( Ballard 1979 , ) . Background train +Some efforts have tackled tasks such as automatic image caption generation ( Feng and Lapata , 2010a ; Ordonez et al. , 2011 ) , text illustration ( Joshi et al. , 2006 ) , or automatic location identification of Twitter users ( Eisenstein et al. , 2010 ; ; Roller et al. , 2012 ) . Background train +The system was trained on the Penn Treebank ( Marcus et al. , 1993 ) WSJ Sections 221 and tested on Section 23 ( Table 1 ) , same as used by , Collins ( 1997 ) , and Ratnaparkhi ( 1997 ) , and became a common testbed . CompareOrContrast train + predefine 163 verbal subcategorization frames , obtained by manually merging the classes exemplified in the COMLEX ( MacLeod , Grishman , and Meyers 1994 ) and ANLT ( Boguraev et al. 1987 ) dictionaries and adding around 30 frames found by manual inspection . Background train +Recently , several alternative , often quite sophisticated approaches to collective classification have been proposed ( ; Lafferty et al. , 2001 ; Getoor et al. , 2002 ; Taskar et al. , 2002 ; Taskar et al. , 2003 ; Taskar et al. , 2004 ; McCallum and Wellner , 2004 ) . Background val + give a sufficiently general finite-state framework to allow this : weights may fall in any set K ( instead of R ) . Uses train +FBLTAG ( Vijay-Shanker , 1987 ; ) is an extension of the LTAG formalism . Background train +There are several variations of such a method ( Ballesteros and Croft , 1998 ; Pirkola , 1998 ; ) . CompareOrContrast train +12 In order to focus on the computational aspects of the covariation approach , in this paper we will not go into a discussion of the full lexical rule specification language introduced in . Background val +Many investigators ( e.g. Allen 1976 ; Elowitz et al. 1976 ; Luce et al. 1983 ; ) have suggested that the poor prosody of synthetic speech , in comparison with natural speech , is the primary factor leading to difficulties in the comprehension of fluent synthetic speech . Motivation train +This Principle of Finitism is also assumed by , Jackendoff ( 1983 ) , Kamp ( 1981 ) , and implicitly or explicitly by almost all researchers in computational linguistics . CompareOrContrast train +We performed translation experiments with an implementation of the IBM-4 translation model ( ) . Uses train +In this paper , a flexible annotation schema called Structured String-Tree Correspondence ( SSTC ) ( ) will be introduced to capture a natural language text , its corresponding abstract linguistic representation and the mapping ( correspondence ) between these two . Background val +For the development of these lists we used a collection of texts of about 300,000 words derived from the New York Times ( NYT ) corpus that was supplied as training data for the 7th Message Understanding Conference ( MUC-7 ) ( ) . Background train +For the full parser , we use the one developed by Michael Collins ( Collins , 1996 ; ) -- one of the most accurate full parsers around . Uses train +More sophisticated approaches have been proposed ( ) , including an extension that , in an interesting reversal of our problem , makes use of sentimentpolarity indicators within speech segments ( Galley et al. , 2004 ) . Background train + have demonstrated that differential weighting of automatically labeled sections can lead to improved retrieval performance . Background train +The one-sided t-test ( ) at significance level 0.05 indicated that the improvement on Trec5C is not statistically significant . Uses train +mers ( Lovins , 1968 ; ) demonstrably improve retrieval performance . Background train +It is therefore no surprise that early attempts at response automation were knowledge-driven ( ; Watson 1997 ; Delic and Lahaix 1998 ) . Background train +The framework was originally developed for the realization of deep-syntactic structures in NLG ( ) . Background train +These features are very much desired in the design of an annotation scheme , in particular for the treatment of linguistic phenomena , which are non-standard , e.g. crossed dependencies ( ) . Background train +Few approaches to parsing have tried to handle disfluent utterances ( notable exceptions are Core & Schubert , 1999 ; ; Nakatani & Hirschberg , 1994 ; Shriberg , Bear , & Dowding , 1992 ) . Background train +5 The open source Moses ( ) toolkit from www.statmt.org/moses/ . Uses train +There have been several efforts aimed at developing a domain-independent method for generating responses from a frame representation of user requests ( ; Chu-Carroll , 1999 ) . Future val +The maximum entropy approach ( ) presents a powerful framework for the combination of several knowledge sources . Uses train +Other molecular biology databases We also included several model organism databases or nomenclature databases in the construction of the dictionary , i.e. , mouse Mouse Genome Database ( MGD ) [ 18 ] , fly FlyBase [ 19 ] , yeast Saccharomyces Genome Database ( SGD ) , rat -- Rat Genome Database ( RGD ) [ 21 ] , worm -- WormBase [ 22 ] , Human Nomenclature Database ( HUGO ) [ 23 ] , Online Mendelian Inheritance in Man ( OMIM ) [ 24 ] , and Enzyme Nomenclature Database ( ECNUM ) [ 25 , 26 ] . Uses val +We then use the program Snob ( Wallace and Boulton 1968 ; ) to cluster these experiences . Uses train +However , studies have shown that existing systems for searching MEDLINE ( such as PubMed , the search service provided by the National Library of Medicine ) are often inadequate and unable to supply clinically relevant answers in a timely manner ( Gorman , Ash , and Wykoff 1994 ; ) . Background train +Our implementation of the NP-based QA system uses the Empire noun phrase finder , which is described in detail in . Uses train +For all the experiments reported in this article , we used the training portion of PATB Part 3 v3 .1 ( ) , converted to the CATiB Treebank format , as mentioned in Section 2.5 . Uses train +In addition , we consider several types of lexical features ( LexF ) inspired by previous work on agreement and disagreement ( Galley et al. , 2004 ; ) . Motivation val +All communicative head gestures in the videos were found and annotated with ANVIL using a subset of the attributes defined in the MUMIN annotation scheme ( ) . Uses train +Specifically , we examine the strength of association between the verb and the noun constituent of a combination ( the target expression or its lexical variants ) as an indirect cue to its idiomaticity , an approach inspired by . Motivation train +The current system learns finite state flowcharts whereas typical learning systems usually acquire coefficient values as in Minsky and Papert ( 1969 ) , assertional statements as in Michalski ( 1980 ) , or semantic nets as in . CompareOrContrast val +More recently , an alignment selection approach was proposed in ( ) , which computes confidence scores for each link and prunes the links from multiple sets of alignments using a hand-picked threshold . CompareOrContrast train + for example discusses a method where a syntactic parse of the text is performed and the context of a word is modeled using dependency triples . Background train +A variety of statistical methods were proposed over the recent years for learning to produce a full parse of free-text sentences ( e.g. , Bod ( 1992 ) , Magerman ( 1995 ) , , Ratnaparkhi ( 1997 ) , and Sekine ( 1998 ) ) . Background train +The strategies employed when MIMIC has only dialogue initiative are similar to the mixed initiative dialogue strategies employed by many existing spoken dialogue systems ( e.g. , ( Bennacef et at. , 1996 ; ) ) . CompareOrContrast train +7A11 our results are computed with the evalb program following the now-standard criteria in ( ) . Uses train +We see no good reason , however , why such text spans should necessarily be sentences , since the majority of tagging paradigms ( e.g. , Hidden Markov Model [ HMM ] [ ] , Brill 's [ Brill 1995a ] , and MaxEnt [ Ratnaparkhi 1996 ] ) do not attempt to parse an entire sentence and operate only in the local window of two to three tokens . CompareOrContrast train +results are based on a corpus of movie subtitles ( ) , and are consequently shorter sentences , whereas the En → Es results are based on a corpus of parliamentary proceedings ( Koehn 2005 ) . Uses train +The system is implemented based on ( Galley et al. , 2006 ) and ( ) . Uses train +Also relevant is work on the general problems of dialog-act tagging ( Stolcke et al. , 2000 ) , citation analysis ( Lehnert et al. , 1990 ) , and computational rhetorical analysis ( ; Teufel and Moens , 2002 ) . Background val +For projective parsing , it is significantly faster than exact dynamic programming , at the cost of small amounts of search error , We are interested in extending these ideas to phrase-structure and lattice parsing , and in trying other higher-order features , such as those used in parse reranking ( ; Huang , 2008 ) and history-based parsing ( Nivre and McDonald , 2008 ) . Future val +Word alignments are used primarily for extracting minimal translation units for machine translation ( MT ) ( e.g. , phrases [ Koehn , Och , and Marcu 2003 ] and rules [ Galley et al. 2004 ; ] ) as well as for Background train +Sridhar et al. ( 2009 ) obtain promising results in dialogue act tagging of the Switchboard-DAMSL corpus using lexical , syntactic and prosodic cues , while examine the relation between particular acoustic and prosodic turn-yielding cues and turn taking in a large corpus of task-oriented dialogues . Background train +We have also applied our more general unification grammar acquisition methodology to the TIGER Treebank ( ) and Penn Chinese Treebank ( Xue , Chiou , and Palmer 2002 ) , extracting wide-coverage , probabilistic LFG grammar Uses train +We use the structures previously used by , and propose one new structure . Uses train +Two exceptions to this generalisation are the Linguistic String Project ( Sager , 1981 ) and the IBM CRITIQUE ( formerly EPISTLE ) Project ( Heidorn et al. , 1982 ; ) ; the former employs a dictionary of approximately 10,000 words , most of which are specialist medical terms , the latter has well over 100,000 entries , gathered from machine readable sources . CompareOrContrast train +A number of proposals in the 1990s deliberately limited the extent to which they relied on domain and/or linguistic knowledge and reported promising results in knowledge-poor operational environments ( Dagan and Itai 1990 , 1991 ; Lappin and Leass 1994 ; Nasukawa 1994 ; Kennedy and Boguraev 1996 ; Williams , Harvey , and Preston 1996 ; ; Mitkov 1996 , 1998b ) . Background train +Towards this aim , a flexible annotation structure called Structured String-Tree Correspondence ( SSTC ) was introduced in to record the string of terms , its associated representation structure and the mapping between the two , which is expressed by the sub-correspondences recorded as part of a SSTC . Background val +We have built an experimental text-to-speech system that uses our analysis of prosody to generate phrase boundaries for the Olive -- Liberman synthesizer ( ) . Uses train +In this paper , I present a computational implementation of Distributed Morphology ( ) , a non-lexicalist linguistic theory that erases the distinction between syntactic derivation and morphological derivation . Uses train +Some recent GRE algorithms have done away with the separation between content determination and linguistic realization , interleaving the two processes instead ( Stone and Webber 1998 ; ) . CompareOrContrast train +Subsequent processing by the natural language and response generation components was done automatically by the computer ( ) . Uses train +This is implemented as a cascade of simple strategies , which were briefly described in . Uses val +But , obviously , there are other possibilities -- for instance , the discourse representation structures ( DRS 's ) of , which have been used to translate a subset of English into logical formulas , to model text ( identified with a list of sentences ) , to analyze a fragment of English , and to deal with anaphora . CompareOrContrast train +Better results would be expected by combining the PCFG-LA parser with discriminative reranking approaches ( Charniak and Johnson , 2005 ; ) for self training . Future train +Intermedia is no more developed and nobody of us had the opportunity to try it ( ) . Background val +Morphological alterations of a search term have a negative impact on the recall performance of an information retrieval ( IR ) system ( Choueka , 1990 ; J ¨ appinen and Niemist ¨ o , 1988 ; ) , since they preclude a direct match between the search term proper and its morphological variants in the documents to be retrieved . Background train +ment ( ; Doran et al. , 2000 ; Makino et al. , 1998 ) . Background train +Finally , feedback expressions ( head nods and shakes ) are successfully predicted from speech , prosody and eye gaze in interaction with Embodied Communication Agents as well as human communication ( ; Morency et al. , 2005 ; Morency et al. , 2007 ; Morency et al. , 2009 ) . Background val +Another line of research approaches grounded language knowledge by augmenting distributional approaches of word meaning with perceptual information ( ; Steyvers , 2010 ; Feng and Lapata , 2010b ; Bruni et al. , 2011 ; Silberer and Lapata , 2012 ; Johns and Jones , 2012 ; Bruni et al. , 2012a ; Bruni et al. , 2012b ; Silberer et al. , 2013 ) . Background train +We also made use of the person-name/instance pairs automatically extracted by .2 This data provides counts for pairs such as `` Edwin Moses , hurdler '' and `` William Farley , industrialist . '' Uses train + presented an approach for constructing a BKB based on the S-SSTC . Background train +However , since work in this direction has started , a significant progress has also been made in the research on statistical learning of full parsers , both in terms of accuracy and processing time ( b ; Charniak , 1997a ; Collins , 1997 ; Ratnaparkhi , 1997 ) . Background train +Also , advanced methods often require many training iterations , for example active learning ( Dagan and Engelson ,1995 ) and co-training ( ) . Background train + furthered this work by showing that a bimodal topic model , consisting of both text and feature norms , outperformed models using only one modality on the prediction of association norms , word substitution errors , and semantic interference tasks . Extends val +It can be shown ( ) that the use of this model with maximum likelihood parameter estimation is justified on information-theoretic grounds when q represents some prior knowledge about the true distribution and when the expected values of f in the training corpus are identical to their true expected values .3 There is no requirement that the components of f represent disjoint or statistically independent events . Motivation train +Shortly after the publication of The Sound Pattern of English ( Chomsky and Halle 1968 ) , Kornai points out , `` Johnson ( 1970 ) demonstrated that the context-sensitive machinery of SPE ... [ could ] be replaced by a much simpler one , based on finite-state transducers ( FSTs ) ; the same conclusion was reached independently by Kaplan and Kay , whose work remained an underground classic until it was finally published in . '' Background train +18 In this article , we use a newer version of the corpus by than the one we used in Marton , Habash , and Rambow ( 2011 ) . Uses train +b ) and Topkara et al. ( 2006a ) attained the embedding capacity of 0.5 bits per sentence with the syntactic transformation method . Background train +Another possibility that often works better is to use Minimum Bayes-Risk ( MBR ) decoding ( ; Liang , Taskar , and Klein 2006 ; Ganchev , and Taskar 2007 ) . Uses train +The subcategorization requirements expressed by semantic forms are enforced at f-structure level through completeness and coherence well-formedness conditions on f-structure ( ) : An f-structure is locally complete iff it contains all the governable grammatical functions that its predicate governs . Background train +This paper describes an approach for sharing resources in various grammar formalisms such as Feature-Based Lexicalized Tree Adjoining Grammar ( FB-LTAG1 ) ( Vijay-Shanker , 1987 ; ) and Head-Driven Phrase Structure Grammar ( HPSG ) ( Pollard and Sag , 1994 ) by a method of grammar conversion . Background train +Following the example of , we will call the autonomous units of a hypertext lexias ( from ` lexicon ' ) , a word coined by Roland Barthes ( 1970 ) . Uses train +( ) . Background train +We use the same set of binary features as in previous work on this dataset ( ; Pang and Lee , 2004 ; Zaidan et al. , 2007 ) . Uses train +Various feature selection techniques have been used in automatic text categorization ; they include document frequency ( DF ) , information gain ( IG ) ( ) , minimum description length principal ( Lang , 1995 ) , and the X2 statistic . Background train +Politically-oriented text Sentiment analysis has specifically been proposed as a key enabling technology in eRulemaking , allowing the automatic analysis of the opinions that people submit ( Shulman et al. , 2005 ; Cardie et al. , 2006 ; ) . Background train + observed that dependency parsers tend to do quite poorly when parsing questions due to their limited exposure to them in the news corpora from the PennTreebank . CompareOrContrast val +a ) 's similar word list for eat misses these but includes sleep ( ranked 6 ) and sit ( ranked 14 ) , because these have similar subjects to eat . Background train +• Learnability ( ) • Text generation ( Hovy 1988 ; Milosavljevic , Tulloch , and Dale 1996 ) • Speech generation ( Rayner and Carter 1997 ) • Localization ( Sch ¨ aler 1996 ) Background train +For example , a ` web page ' is more similar to an infinite canvas than a written page ( ) . Background train +Other molecular biology databases We also included several model organism databases or nomenclature databases in the construction of the dictionary , i.e. , mouse Mouse Genome Database ( MGD ) , fly FlyBase [ 19 ] , yeast Saccharomyces Genome Database ( SGD ) [ 20 ] , rat -- Rat Genome Database ( RGD ) [ 21 ] , worm -- WormBase [ 22 ] , Human Nomenclature Database ( HUGO ) [ 23 ] , Online Mendelian Inheritance in Man ( OMIM ) [ 24 ] , and Enzyme Nomenclature Database ( ECNUM ) [ 25 , 26 ] . Uses train +2This view of typed feature structures differs from the perspective on typed feature structures as modeling partial information as in ( ) . Background train +This approach resembles the work by Grishman et al. ( 1986 ) and on selectional restrictions . CompareOrContrast train +This is noticeable for German ( ) and Portuguese ( Afonso et al. , 2002 ) , which still have high overall accuracy thanks to very high attachment scores , but much more conspicuous for Czech ( B ¨ ohmov ´ a et al. , 2003 ) , Dutch ( van der Beek et al. , 2002 ) and Slovene ( Dˇzeroski et al. , 2006 ) , where root precision drops more drastically to about 69 % , 71 % and 41 % , respectively , and root recall is also affected negatively . CompareOrContrast train +LTAG ( ) is a grammar formalism that provides syntactic analyses for a sentence by composing elementary trees with two opera - Background train +Translations have been generated by the CrowdFlower3 channel to Amazon Mechanical Turk4 ( MTurk ) , adopting the methodology proposed by ( ) . Uses train +This work is a continuation of that initiated in ( ) , which reports an overall NB classification correctness of 75.6 % , in cross validation experiments , on a data set that consists of 100 documents for each of 12 categories ( the data set is collected from different Arabic portals ) . Extends train +Our previous work ( ) designed an EMbased method to construct unsupervised trees for tree-based translation models . CompareOrContrast train +Then , we binarize the English parse trees using the head binarization approach ( ) and use the resulting binary parse trees to build another s2t system . Uses train +This is the strongest version of the sorites paradox ( e.g. , ) . Background train +We used a standard implementation of IBM Model 4 ( ) and because changing the existing code is not trivial , we could not use the same stopping criterion to avoid overfitting and we are not able to produce precision/recall curves . Uses train +Cross-lingual Textual Entailment ( CLTE ) has been proposed by ( ) as an extension of Textual Entailment ( Dagan and Glickman , 2004 ) that consists in deciding , given two texts T and H in different languages , if the meaning of H can be inferred from the meaning of T . Background val +Some approaches apply semantic parsing , where words and sentences are mapped to logical structure meaning ( ) . Background train +For projective parsing , it is significantly faster than exact dynamic programming , at the cost of small amounts of search error , We are interested in extending these ideas to phrase-structure and lattice parsing , and in trying other higher-order features , such as those used in parse reranking ( Charniak and Johnson , 2005 ; ) and history-based parsing ( Nivre and McDonald , 2008 ) . Future train +Prototypes of Internet search engines for linguists , corpus linguists and lexicographers have been proposed : WebCorp ( Kehoe and Renouf , 2002 ) , KWiCFinder ( a ) and the Linguist 's Search Engine ( Kilgarriff , 2003 ; Resnik and Elkiss , 2003 ) . Background train +Other molecular biology databases We also included several model organism databases or nomenclature databases in the construction of the dictionary , i.e. , mouse Mouse Genome Database ( MGD ) [ 18 ] , fly FlyBase [ 19 ] , yeast Saccharomyces Genome Database ( SGD ) [ 20 ] , rat -- Rat Genome Database ( RGD ) [ 21 ] , worm -- WormBase , Human Nomenclature Database ( HUGO ) [ 23 ] , Online Mendelian Inheritance in Man ( OMIM ) [ 24 ] , and Enzyme Nomenclature Database ( ECNUM ) [ 25 , 26 ] . Uses train +Other factors , such as the role of focus ( Grosz 1977 , 1978 ; Sidner 1983 ) or quantifier scoping ( ) must play a role , too . Background train +Since mid-2002 , the Library has been employing software that automatically suggests MeSH headings based on content ( ) . Background train +Discriminative approaches ( especially SVMs ) have been shown to be very effective for many supervised classification tasks ; see , for example , ( Joachims , 1998 ; ) . Background val +Also , the approach will be undefined if the pair is unobserved on the web . Uses train +The automation of help-desk responses has been previously tackled using mainly knowledge-intensive paradigms , such as expert systems ( ) and case-based reasoning ( Watson 1997 ) . CompareOrContrast train +We measure this association using pointwise Mutual Information ( MI ) ( ) . Uses train +Some methods of semantic relation analysis rely on predefined templates filled with information from processed texts ( ) . Background train +There are two corpora normally used for evaluation in a number of text-processing tasks : the Brown corpus ( ) and the Wall Street Journal ( WSJ ) corpus , both part of the Penn Treebank ( Marcus , Marcinkiewicz , and Santorini 1993 ) . Uses train +or quotation of messages in emails or postings ( see Mullen and Malouf ( 2006 ) but cfXXX ) . Background train + extend LDA to allow for the inference of document and topic distributions in a multimodal corpus . Background val +SNoW ( ; Roth , 1998 ) is a multi-class classifier that is specifically tailored for learning in domains in which the potential number of information sources ( features ) taking part in decisions is very large , of which NLP is a principal example . Uses train +The search algorithm is the standard Viterbi search ( ) , except that the match involves a network-to-network alignment problem rather than sequence-to-sequence . Uses train +Since then this idea has been applied to several tasks , including word sense disambiguation ( Yarowsky 1995 ) and named-entity recognition ( ) . Background train +7 We employed the LIBSVM package ( ) . Uses train +coreference performance on perfect mentions ( e.g. , Incorporate the two knowledge sources in a ) ; and for those that do report percoreference resolver . CompareOrContrast val +According to , p. 67 ) , these two sentences are incoherent . CompareOrContrast train +Thus , over the past few years , along with advances in the use of learning and statistical methods for acquisition of full parsers ( Collins , 1997 ; Charniak , 1997a ; Charniak , 1997b ; Ratnaparkhi , 1997 ) , significant progress has been made on the use of statistical learning methods to recognize shallow parsing patterns syntactic phrases or words that participate in a syntactic relationship ( ; Ramshaw and Marcus , 1995 ; Argamon et al. , 1998 ; Cardie and Pierce , 1998 ; Munoz et al. , 1999 ; Punyakanok and Roth , 2001 ; Buchholz et al. , 1999 ; Tjong Kim Sang and Buchholz , 2000 ) . Background train +Another dialogue acquisition system has been developed by . CompareOrContrast train +Lexical functional grammar ( Kaplan and Bresnan 1982 ; Bresnan 2001 ; ) is a member of the family of constraint-based grammars . Background val +1Our rules are similar to those from . CompareOrContrast train +In psycholinguistics , relatedness of words can also be determined through association tests ( Schulte im ) . Background train +In this article , we use an in-house system which provides functional gender , number , and rationality features ( ) . Uses train +Over the last decade there has been a lot of interest in developing tutorial dialogue systems that understand student explanations ( ; Graesser et al. , 1999 ; Aleven et al. , 2001 ; Buckley and Wolska , 2007 ; Nielsen et al. , 2008 ; VanLehn et al. , 2007 ) , because high percentages of selfexplanation and student contentful talk are known to be correlated with better learning in humanhuman tutoring ( Chi et al. , 1994 ; Litman et al. , 2009 ; Purandare and Litman , 2008 ; Steinhauser et al. , 2007 ) . Background train +The use of the web as a corpus for teaching and research on language has been proposed a number of times ( Kilgarriff , 2001 ; Robb , 2003 ; Rundell , 2000 ; , 2004b ) and received a special issue of the journal Computational Linguistics ( Kilgarriff and Grefenstette , 2003 ) . Background val +Several authors in communication studies have pointed out that head movements are relevant to feedback phenomena ( see for an overview ) . Background train +Our method resorts to some translation examples , which is similar as example-based translation or translation memory ( ; He et al. , 2010 ; Ma et al. , 2011 ) . CompareOrContrast train +Baseline Systems We choose three publicly available state-of-the-art end-to-end coreference systems as our baselines : Stanford system ( Lee et al. , 2011 ) , Berkeley system ( ) and HOTCoref system ( Bj ¨ orkelund and Kuhn , 2014 ) . CompareOrContrast val +A number of speech understanding systems have been developed during the past fifteen years ( Barnett et al. 1980 , Dixon and Martin 1979 , Erman et al. 1980 , Haton and Pierrel 1976 , Lea 1980 , Lowerre and Reddy 1980 , Medress 1980 , Reddy 1976 , Walker 1978 , and ) . CompareOrContrast train +As for work on Arabic ( MSA ) , results have been reported on the PATB ( Kulick , Gabbard , and Marcus 2006 ; Diab 2007 ; Green and Manning 2010 ) , the Prague Dependency Treebank ( PADT ) ( Buchholz and Marsi 2006 ; Nivre 2008 ) and the CATiB ( ) . Background train +Other representations use the link structure ( ) or generate graph representations of the extracted features ( Kalashnikov et al. , 2007 ) . Background train +From an IR view , a lot of specialized research has already been carried out for medical applications , with emphasis on the lexico-semantic aspects of dederivation and decomposition ( Pacak et al. , 1980 ; Norton and Pacak , 1983 ; ; Wingert , 1985 ; Dujols et al. , 1991 ; Baud et al. , 1998 ) . Background train +There has been some controversy , at least for simple stemmers ( Lovins , 1968 ; ) , about the effectiveness of morphological analysis for document retrieval ( Harman , 1991 ; Krovetz , 1993 ; Hull , 1996 ) . Background train +This revalidates the observation of that phrase structure representations and dependency representations add complimentary value to the learning task . CompareOrContrast train +The parsing algorithm used for all languages is the deterministic algorithm first proposed for unlabeled dependency parsing by Nivre ( 2003 ) and extended to labeled dependency parsing by . Uses train +The semantic categories of verbs and other words are extracted from the Semantic Knowledge-base of Contemporary Chinese ( ) . Uses train +Moreover , in order to determine whether the performances of the predictive criteria are consistent across different learning models within the same domain , we have performed the study on two parsing models : one based on a context-free variant of tree-adjoining grammars ( Joshi , Levy , and Takahashi 1975 ) , the Probabilistic Lexicalized Tree Insertion Grammar ( PLTIG ) formalism ( ; Hwa 1998 ) , and Collins 's Model 2 parser ( 1997 ) . Uses train +6 The Partial-VP Topicalization Lexical Rule proposed by , 10 ) is a linguistic example . Background train +We measure the inter annotator agreement using the Fleiss Kappa ( ) measure ( x ) where the agreement lies around 0.79 . Uses train +The task we used to compare different generalisation techniques is similar to that used by and Rooth et al. ( 1999 ) . CompareOrContrast val +Other milestones of recent research include the deployment of probabilistic and machine learning techniques ( Aone and Bennett 1995 ; Kehler 1997 ; Ge , Hale , and Charniak 1998 ; Cardie and Wagstaff 1999 ; the continuing interest in centering , used either in original or in revised form ( Abracos and Lopes 1994 ; Strube and Hahn 1996 ; ; Tetreault 1999 ) ; and proposals related to the evaluation methodology in anaphora resolution ( Mitkov 1998a , 2001b ) . Background train +It is not aimed at handling dependencies , which require heavy use of lexical information ( , for PP attachment ) . CompareOrContrast val +This includes work on question answering ( Wang et al. , 2007 ) , sentiment analysis ( ) , MT reordering ( Xu et al. , 2009 ) , and many other tasks . Background train +I A more detailed discussion of various aspects of the proposed parser can be found in ( ) . Background val +Our experimental design with professional bilingual translators follows our previous work a ) comparing scratch translation to post-edit . Extends val +Lexical functional grammar ( Kaplan and Bresnan 1982 ; ; Dalrymple 2001 ) is a member of the family of constraint-based grammars . Background train +Thus , the second class of SBD systems employs machine learning techniques such as decision tree classifiers ( ) , neural networks ( Palmer and Hearst 1994 ) , and maximum-entropy modeling ( Reynar and Ratnaparkhi 1997 ) . Background train +Previously , a user study ( ) has shown that people are reluctant to type full natural language questions , even after being told that they were using a questionanswering system and that typing complete questions would result in better performance . CompareOrContrast train +We will examine the worst-case complexity of interpretation as well as generation to shed some light on the hypothesis that vague descriptions are more difficult to process than others because they involve a comparison between objects ( Beun and Cremers 1998 , ) . Background train +In our case , the clustering is performed by the program Snob , which implements mixture modeling combined with model selection based on the Minimum Message Length ( MML ) criterion ( Wallace and Boulton 1968 ; ) . Uses train +The third approach to cross-lingual retrieval is to map queries and documents to some intermediate representation , e.g latent semantic indexing ( LSI ) ( Littman et al , 1998 ) , or the General Vector space model ( GVSM ) , ( ) . CompareOrContrast train +Many provide graphical user interfaces ( GUI ) for manual annotation ( e.g. General Architecture for Text Engineering ( GATE ) ( ) and the Alembic Workbench ( Day et al. , 1997 ) ) as well as NLP tools and resources that can be manipulated from the GUI . Background train +The keypoints are clustered into 5,000 visual codewords ( centroids ) using k-means clustering ( ) , and images are then quantized over the 5,000 codewords . Uses train +Another approach for partial parsing was presented by . Background val +In knowledge-lean approaches , coreference resolvers employ only morpho-syntactic cues as knowledge sources in the resolution process ( e.g. , Mitkov ( 1998 ) , ) . Background train +The application of domain models and deep semantic knowledge to question answering has been explored by a variety of researchers ( e.g. , , Rinaldi et al. 2004 ) , and was also the focus of recent workshops on question answering in restricted domains at ACL 2004 and AAAI 2005 . Background train +This result is consistent with other works using this model with these features ( Andrews et al. , 2009 ; ) . CompareOrContrast train +Cases of intracategorial synonymy are relatively straigthtforward as several electronic synonym dictionnaries for french are available ( ) . Background train +The feature of head word trigger which we apply to the log-linear model is motivated by the trigger-based approach ( ) . Motivation train +Other work on modeling the meanings of verbs using video recognition has also begun showing great promise ( Mathe et al. , 2008 ; ) . Background val +based parsing algorithms with an arc-factored parameterization ( ) . Uses train +No attempt has been made to map any closed class entries from LDOCE , as a 3,000 word lexicon containing most closed class items has been developed independently by one of the groups collaborating with us to develop the general purpose morphological and syntactic analyser ( see the Introduction and ) . Background train +In this paper we focus on the exploitation of the LDOCE grammar coding system ; and Alshawi ( 1987 ) describe further research in Cambridge utilising different types of information available in LDOCE . Background val +Thus , the second class of SBD systems employs machine learning techniques such as decision tree classifiers ( Riley 1989 ) , neural networks ( ) , and maximum-entropy modeling ( Reynar and Ratnaparkhi 1997 ) . Background train +We see no good reason , however , why such text spans should necessarily be sentences , since the majority of tagging paradigms ( e.g. , Hidden Markov Model [ HMM ] [ Kupiec 1992 ] , Brill 's [ a ] , and MaxEnt [ Ratnaparkhi 1996 ] ) do not attempt to parse an entire sentence and operate only in the local window of two to three tokens . CompareOrContrast train +Similarly , ( Barzilay and Lee , 2003 ) and ( ) learn sentence level paraphrase templates from a corpus of news articles stemming from different news source . Background train +In future work we plan to experiment with richer representations , e.g. including long-range n-grams ( Rosenfeld , 1996 ) , class n-grams ( ) , grammatical features ( Amaya and Benedy , 2001 ) , etc ' . Future val +Furthermore , we demonstrate that our results carry over successfully to another parser , the Easy-First Parser ( ) ( Section 6 ) . Uses train +For instance , relating `` they '' to `` apples '' in the sentence ( cfXXX p. 195 ; Zadrozny 1987a ) : We bought the boys apples because they were so cheap Background train +We follow our previous work ( ) in our feature choices , using a fiveword window that includes the target stem and two words on either side for context ( see also Tetreault and Chodorow , 2008 ) . Extends train +Adjectives , more than other categories , are a striking example of regular polysemy since they are able to take on different meanings depending on their context , viz. , the noun or noun class they modify ( see and the references therein ) . Background train +According to , LFG assumes the following universally available inventory of grammatical functions : SUBJ ( ect ) , OBJ ( ect ) , OBJe , COMP , XCOMP , OBL ( ique ) e , ADJ ( unct ) , XADJ . Background train +But their importance has grown far beyond machine translation : for instance , transferring annotations between languages ( Yarowsky and Ngai 2001 ; Hwa et al. 2005 ; Ganchev , Gillenwater , and Taskar 2009 ) ; discovery of paraphrases ( Bannard and Callison-Burch 2005 ) ; and joint unsupervised POS and parser induction across languages ( ) . Motivation train +4 This interpretation of the signature is sometimes referred to as closed world ( Gerdemann and King 1994 ; ) . Background val +Our training examples are similar to the data created for pseudodisambiguation , the usual evaluation task for SP models ( Erk , 2007 ; Keller and Lapata , 2003 ; ) . CompareOrContrast train +There have already been several attempts to develop distributed NLP systems for dialogue systems ( Bayer et al. , 2001 ) and speech recognition ( ) . Background train +For instance , implementing an efficient version of the MXPOST POS tagger ( ) will simply involve composing and configuring the appropriate text file reading component , with the sequential tagging component , the collection of feature extraction components and the maximum entropy model component . Future train +cue word and name the first ( or several ) associated words that come to mind ( e.g. , Nelson et al. ( 2004 ) ) , and feature norms , where subjects are given a cue word and asked to describe typical properties of the cue concept ( e.g. , ) . Background train + and Litman and Hirschberg ( 1990 ) also examine the relation between discourse and prosodic phrasing . Background train + evaluate 914 Czech verbs against a custom-made gold standard and record a token recall of 88 % . CompareOrContrast train +We use two measures from Information Retrieval to determine the quality of an automatically generated response : precision and F-score ( van Rijsbergen 1979 ; ) . Uses val +The automation of help-desk responses has been previously tackled using mainly knowledge-intensive paradigms , such as expert systems ( Barr and Tessler 1995 ) and case-based reasoning ( ) . CompareOrContrast train + did very encouraging work on the feature calibration of semantic role labeling . Background train +This approach , which uses words that appear in the context of terms to formulate hypotheses on their semantic relatedness ( , for example ) , does not specify the relationship itself . Background train +And ( ) use clustering and similarity measures to identify similar contexts in a single corpus and extract verbal paraphrases from these contexts . Background train +Not having to represent the frame explicitly not only enables the linguist to express only the relevant things , but also allows a more compact representation of lexical rules where explicit framing would require the rules to be split up ( ) . Background train +There has also been work focused upon determining the political leaning ( e.g. , `` liberal '' vs. `` conservative '' ) of a document or author , where most previously-proposed methods make no direct use of relationships between the documents to be classified ( the `` unlabeled '' texts ) ( Laver et al. , 2003 ; Efron , 2004 ; ) . Background train +Other similar approaches include those of Cicekli and G ¨ uvenir ( 1996 ) , , Carl ( 1999 ) , and Brown ( 2000 ) , inter alia . Background train +To sum up , this work has been carried out to automatically classify Arabic documents using the NB algorithm , with the use of a different data set , a different number of categories , and a different root extraction algorithm from those used in ( ) . CompareOrContrast val +Our work builds on earlier research on learning to identify dialogues in which the user experienced poor speech recognizer performance ( ) . Extends val +CCGBank ( ) is used to train the model . Uses train +More recent work on terminology structuring has focussed on formal similarity to develop hypotheses on the semantic relationships between terms : uses derivational morphology ; Grabar and Zweigenbaum ( 2002 ) use , as a starting point , a number of identical characters . Background train +Some researchers ( Cucerzan , 2007 ; ) have explored the use of Wikipedia information to improve the disambiguation process . Background train +To address this issue , we use a version of the PATB3 training and dev sets manually annotated with functional gender , number , and rationality ( ) .18 This is the first resource providing all three features ( ElixirFm only provides functional number , and to some extent functional gender ) . Uses train +One approach to this problem is that taken by the ASCOT project ( Akkerman et al. , 1985 ; ) . Background train +Narrative writings or essays are creative works and they generally treat ownership as authorship , even for the most enthusiastic fellows of free culture ( ) . Background train +McDonald has even argued for extending the model to a large number of components ( McDonald 1988 ) , and several systems have indeed added an additional component between the planner and the linguistic component ( Meteer 1994 ; Panaget 1994 ; ) . Background train +According to current tagger comparisons ( van Halteren et al. , 1998 ; Zavrel and Daelemans , 1999 ) , and according to a comparsion of the results presented here with those in ( ) , the Maximum Entropy framework seems to be the only other approach yielding comparable results to the one presented here . CompareOrContrast train +Similar findings have been proposed by that points out V1 and V2 are paired on the basis of their semantic compatibility , which is subject to syntactic constraints . Background train +The first lexical substitution method was proposed by . Background train +This can be a hazardous affair , since vague expressions tend to be interpreted in different ways by different people ( ) , sometimes in stark contrast with the intention of the speaker/writer ( Berry , Knapp , and Raynor 2002 ) . Background train +We see no good reason , however , why such text spans should necessarily be sentences , since the majority of tagging paradigms ( e.g. , Hidden Markov Model [ HMM ] [ Kupiec 1992 ] , Brill 's [ Brill 1995a ] , and MaxEnt [ ] ) do not attempt to parse an entire sentence and operate only in the local window of two to three tokens . CompareOrContrast train + ; also reported in Levelt 1989 ) show that greater differences are most likely to be chosen , presumably because they are more striking . Background val +Nevertheless , recent results show that knowledge-poor methods perform with amazing accuracy ( cfXXX ( ) , ( Kennedy and Boguraev , 1996 ) ( Kameyama , 1997 ) ) . Background train +Notice that it is not possible to use corpus annotation to determine the likelihood of a given property to be chosen , unless we know in advance all of the properties that can be attributed to a given object , as in the case of Jordan 's work on the COCONUT domain ( ) . Background train +Some well-known approaches include rule-based models ( Brill and Resnik 1994 ) , backed-off models ( ) , and a maximumentropy model ( Ratnaparkhi 1998 ) . Background train +Other psycholing-uistic studies that confirm the validity of paragraph units can be found in Black and Bower ( 1979 ) and . Background train +Tetreault 's contribution features comparative evaluation involving the author 's own centering-based pronoun resolution algorithm called the Left-Right Centering algorithm ( LRC ) as well as three other pronoun resolution methods : Hobbs 's naive algorithm ( Hobbs 1978 ) , BFP ( Brennan , Friedman , and Pollard 1987 ) , and Strube 's 5list approach ( ) . Background train + investigated three approaches to the automatic generation of response e-mails : text classification , case-based reasoning , and question answering . CompareOrContrast train +This contrasts with one of the traditional approaches ( e.g. , ; Watanabe 1995 ) to posing the translation problem , i.e. , the approach in which translation problems are seen in terms of bridging the gap between the most natural monolingual representations underlying the sentences of each language . CompareOrContrast val +In practical context , German , English , and Japanese HPSG-based grammars are developed and used in the Verbmobil project ( ) . Background train +Thus , over the past few years , along with advances in the use of learning and statistical methods for acquisition of full parsers ( Collins , 1997 ; Charniak , 1997a ; Charniak , 1997b ; Ratnaparkhi , 1997 ) , significant progress has been made on the use of statistical learning methods to recognize shallow parsing patterns syntactic phrases or words that participate in a syntactic relationship ( Church , 1988 ; Ramshaw and Marcus , 1995 ; Argamon et al. , 1998 ; Cardie and Pierce , 1998 ; Munoz et al. , 1999 ; ; Buchholz et al. , 1999 ; Tjong Kim Sang and Buchholz , 2000 ) . Background train +For example , such schema can serve as a mean to represent translation examples , or find structural correspondences for the purpose of transfer grammar learning ( Menezes & Richardson , 2001 ) , ( Aramaki et al. , 2001 ) , ( Watanabe et al. , 2000 ) , ( ) , ( Matsumoto et al. , 1993 ) , ( kaji et al. , 1992 ) , and example-base machine translation EBMT3 ( Sato & Nagao , 1990 ) , ( Sato , 1991 ) , ( Richardson et al. , 2001 ) , ( Al-Adhaileh & Tang , 1999 ) . Background train +While IA is generally thought to be consistent with findings on human language production ( Hermann and Deutsch 1976 ; Levelt 1989 ; Pechmann 1989 ; ) , the hypothesis that incrementality is a good model of human GRE seems unfalsifiable until a preference order is specified for the properties on which it operates . Background train +It helps them build complex knowledge bases by combining components : events , entities and modifiers ( ) . Background train +Our group has developed a wide-coverage HPSG grammar for Japanese ( ) , which is used in a high-accuracy Japanese dependency analyzer ( Kanayama et al. , 2000 ) . Background train +Word frequency counts in internet search engines are inconsistent and unreliable ( ) . Background train +Other milestones of recent research include the deployment of probabilistic and machine learning techniques ( Aone and Bennett 1995 ; Kehler 1997 ; Ge , Hale , and Charniak 1998 ; Cardie and Wagstaff 1999 ; the continuing interest in centering , used either in original or in revised form ( Abracos and Lopes 1994 ; Strube and Hahn 1996 ; Hahn and Strube 1997 ; Tetreault 1999 ) ; and proposals related to the evaluation methodology in anaphora resolution ( a , 2001b ) . Background train +It is only recently that the web name ambiguity has been approached as a separate problem and defined as an NLP task Web People Search on its own ( Artiles et al. , 2005 ; ) . Background train +Actually , if we use LSH technique ( ) in retrieval process , the local method can be easily scaled to a larger training data . Future train +Many lexicons , both automatically acquired and manually created , are more fine grained in their approaches to subcategorized clausal arguments , differentiating , for example , between a that-clause and a to + infinitive clause ( ) . Background train +Others provide automatic mappings of natural language instructions to executable actions , such as interpreting navigation directions ( ) or robot commands ( Tellex et al. , 2011 ; Matuszek et al. , 2012 ) . Background train +Agreement between two annotation sets is calculated here in terms of Cohen 's kappa ( Cohen , 1960 ) 1 and corrected kappa ( ) 2 . Uses train +Second , software for utilizing this ontology already exists : MetaMap ( Aronson 2001 ) identifies concepts in free text , and SemRep ( ) extracts relations between the concepts . Background val +The implementation has been inspired by experience in extracting information from very large corpora ( Curran and Moens , 2002 ) and performing experiments on maximum entropy sequence tagging ( Curran and Clark , 2003 ; ) . Motivation train +Some efforts have tackled tasks such as automatic image caption generation ( Feng and Lapata , 2010a ; ) , text illustration ( Joshi et al. , 2006 ) , or automatic location identification of Twitter users ( Eisenstein et al. , 2010 ; Wing and Baldridge , 2011 ; Roller et al. , 2012 ) . Background train +These types of features result in an improvement in both the mention detection and coreference resolution performance , as shown through experiments on the Arabic data . Uses val +The use of the web as a corpus for teaching and research on language has been proposed a number of times ( Kilgarriff , 2001 ; ; Rundell , 2000 ; Fletcher , 2001 , 2004b ) and received a special issue of the journal Computational Linguistics ( Kilgarriff and Grefenstette , 2003 ) . Background train +But while Bod 's estimator obtains state-of-the-art results on the WSJ , comparable to Charniak ( 2000 ) and , Bonnema et al. 's estimator performs worse and is comparable to Collins ( 1996 ) . Background train +The implementation has been inspired by experience in extracting information from very large corpora ( ) and performing experiments on maximum entropy sequence tagging ( Curran and Clark , 2003 ; Clark et al. , 2003 ) . Motivation train +Another line of research approaches grounded language knowledge by augmenting distributional approaches of word meaning with perceptual information ( Andrews et al. , 2009 ; Steyvers , 2010 ; Feng and Lapata , 2010b ; Bruni et al. , 2011 ; Silberer and Lapata , 2012 ; Johns and Jones , 2012 ; Bruni et al. , 2012a ; Bruni et al. , 2012b ; ) . Background train +14We parse each sentence with the Collins parser ( ) . Uses train +Encouraged by the success of chunk-based verb reordering lattices on ArabicEnglish ( ) , we tried to adapt the same approach to the German-English language pair . Motivation train +We followed the same experimental procedure as discussed in ( ) for English polymorphemic words . Uses train +In our experiments , we employed the well-known classifier SVM `` ght to obtain individual-document classification scores , treating Y as the positive class and using plain unigrams as features .5 Following standard practice in sentiment analysis ( ) , the input to SVM `` ght consisted of normalized presence-of-feature ( rather than frequency-of-feature ) vectors . Uses train +The problem of handling ill-formed input has been studied by Carbonell and Hayes ( 1983 ) , Granger ( 1983 ) , Jensen et al. ( 1983 ) , , Riesbeck and Schank ( 1976 ) , Thompson ( 1980 ) , Weischedel and Black ( 1980 ) , and Weischedel and Sondheimer ( 1983 ) . CompareOrContrast train +To address this problem , we are currently working on developing a metagrammar in the sense of ( ) . Future train +Sentences like 12 , from , are frequently cited . Background train +After the PropBank ( ) was built , Xue and Palmer ( 2005 ) and Xue ( 2008 ) have produced more complete and systematic research on Chinese SRL . Background train +In order to estimate the parameters of our model , we develop a blocked sampler based on that of to sample parse trees for sentences in the raw training corpus according to their posterior probabilities . Uses train +The extraction procedure utilizes a head percolation table as introduced by in combination with a variation of Collins 's ( 1997 ) approach to the differentiation between complement and adjunct . Background train +Many other such cases are described in Danlos 's book ( ) . Background train +• Before indexing the text , we process it with Textract ( Byrd and Ravin , 1998 ; ) , which performs lemmatization , and discovers proper names and technical terms . Uses train +Therefore , we repeated the experiments with POS tags predicted by the MADA toolkit ( ; Habash , Rambow , and Roth 2012 ) 15 ( see Table 2 , 14 Some parsers predict POS tags internally , instead of receiving them as input , but this is not the case in this article . Uses train +Experiments on Chinese SRL ( , Xue 2008 ) reassured these findings . Motivation train +An example of this is the estimation of maximum entropy models , from simple iterative estimation algorithms used by that converge very slowly , to complex techniques from the optimisation literature that converge much more rapidly ( Malouf , 2002 ) . Background train +According to , there are three prevalent approaches for evaluating SR measures : mathematical analysis , applicationspecific evaluation and comparison with human judgments . Background train +The choice of learning algorithm for each classifier is motivated by earlier findings showing that discriminative classifiers outperform other machine-learning methods on error correction tasks ( ) . Motivation train +A good study comparing document categorization algorithms can be found in ( ) . Background train +For HMMs ( footnote 11 ) , Ti is the familiar trellis , and we would like this computation of ti to reduce to the forwardbackward algorithm ( ) . Background val +We follow , for compound merging . Uses val +Semantic filters can also be used to prevent multiple versions of the same case frame ( ) showing up as complements . Uses val +The terms have been identified as the most specific to our corpus by a program developed by and called TER1vloSTAT . Uses train +For statistical significance , we use McNemar 's test on non-gold LAS , as implemented by Nilsson and . Uses train +The availability of toolkits for this weighted case ( ; van Noord and Gerdemann , 2001 ) promises to unify much of statistical NLP . Background val +tionally reconstructed by and Crouch and Putman ( 1994 ) , the context-independent meaning of a sentence is given by one or more QLFs that are built directly from syntactic and semantic rules . CompareOrContrast val +We use a standard split of 268 training documents , 68 development documents , and 106 testing documents ( Culotta et al. , 2007 ; ) . Uses train + improved on this by marking prepositions with the case they mark ( one of the most important markups in our system ) . CompareOrContrast train +But typical OT grammars offer much richer finite-state models of left context ( a ) than provided by the traditional HMM finite-state topologies . Background train +Other tools have been designed around particular techniques , such as finite state machines ( Karttunen et al. , 1997 ; ) . Background train +This contrasts with the findings described in where significant improvements could be achieved by increasing the number of source languages . CompareOrContrast train +The function selects the Value that removes most distractors , but in case of a tie , the least specific contestant is chosen , as long as it is not less specific than the basic-level Value ( i.e. , the most commonly occurring and psychologically most fundamental level , ) . Background train + pointed out that distribution plots of judgments for the word pairs used by Rubenstein and Goodenough display an empty horizontal band that could be used to separate related and unrelated pairs . CompareOrContrast train +In fact , Reiter has even argued in favor of this approach , claiming that the interactions are sufficiently minor to be ignored ( or at least handled on an ad hoc basis ) ( ) . Background train +The method is called targeted self-training as it is similar in vein to self-training ( ) , with the exception that the new parse data is targeted to produce accurate word reorderings . CompareOrContrast train +In our previous work ( ) , we started an initial investigation on conversation entailment . Extends train + tried to construct a semantic analysis based on `` prepared '' and `` unprepared mind '' . Background train +It is also possible to focus on non-compositional compounds , a key point in bilingual applications ( ; Melamed , 1997 ; Lin , 99 ) . Background train +This is mainly due to the fact that Arabic is a non-concatenative language ( ) , and that the stem/infix obtained by suppression of infix and prefix add-ons is not the same for words derived from the same origin called the root . Background train +ear regression adapted for classification ( ) , which can be described by the following equation : Uses train +For example , the forward-backward algorithm ( Baum , 1972 ) trains only Hidden Markov Models , while ( ) trains only stochastic edit distance . Background val +Using the tree-cut technique described above , our previous work ( ) extracted systematic polysemy from WordNet . Extends val +( Och and Ney , 2002 ; Blunsom et al. , 2008 ) used maximum likelihood estimation to learn weights for MT. ( Och , 2003 ; Moore and Quirk , 2008 ; Zhao and Chen , 2009 ; ) employed an evaluation metric as a loss function and directly optimized it . CompareOrContrast val +5 Significant bigrams are obtained using the n-gram statistics package NSP ( ) , which offers statistical tests to decide whether to accept or reject the null hypothesis regarding a bigram ( that it is not a collocation ) . Uses val +Liu et al. ( 2005 ) , Meral et al. ( 2007 ) , Murphy ( 2001 ) , and Topkara et al. ( 2006a ) all belong to the syntactic transformation category . Background train +2The WePS-1 corpus includes data from the Web03 testbed ( ) which follows similar annotation guidelines , although the number of document per ambiguous name is more variable . Uses train +The local training method ( ) is widely employed in computer vision ( Zhang et al. , 2006 ; Cheng et al. , 2010 ) . Background train +Recent work ( Banko and Brill , 2001 ; ) has suggested that some tasks will benefit from using significantly more data . Background train +Rather than producing a complete analysis of sentences , the alternative is to perform only partial analysis of the syntactic structures in a text ( Harris , 1957 ; ; Greffenstette , 1993 ) . Background train +The feasibility of automatically identifying outcome statements in secondary sources has been demonstrated by . CompareOrContrast train +The standard approach is to train two models independently and then intersect their predictions ( ) . CompareOrContrast train +We use the same method as for generating our multimodal corpora : for each word token in the text corpus , a feature is selected stochastically from the word 's feature distribution , creating a word-feature pair . Uses train +MI was also recently used for inference-rule SPs by . Background train +Based on this advise ( Moore and ) exclude the latent segmentation variables and opt for a heuristic training procedure . CompareOrContrast train +For example , 10 million words of the American National Corpus ( Ide et al. , 2002 ) will have manually corrected POS tags , a tenfold increase over the Penn Treebank ( ) , currently used for training POS taggers . Background train +The forward and backward probabilities , p0j and pkn , can be computed using single-source algebraic path for the simpler semiring ( R , + , x , ∗ ) -- or equivalently , by solving a sparse linear system of equations over R , a much-studied problem at O ( n ) space , O ( nm ) time , and faster approximations ( ) . Background train +4To prove ( 1 ) ⇒ ( 3 ) , express f as an FST and apply the well-known Kleene-Sch ¨ utzenberger construction ( ) , taking care to write each regexp in the construction as a constant times a probabilistic regexp . Uses val +Second , software for utilizing this ontology already exists : MetaMap ( ) identifies concepts in free text , and SemRep ( Rindflesch and Fiszman 2003 ) extracts relations between the concepts . Background val +Our work is more similar to NLG work that concentrates on structural constraints such as generative poetry ( ) ( Colton et al. , 2012 ) ( Jiang and Zhou , 2008 ) or song lyrics ( Wu et al. , 2013 ) ( Ramakrishnan A et al. , 2009 ) , where specified meter or rhyme schemes are enforced . CompareOrContrast train +Most DOP models , such as in Bod ( 1993 ) , Goodman ( 1996 ) , , Sima'an ( 2000 ) and Collins & Duffy ( 2002 ) , use a likelihood criterion in defining the best parse tree : they take ( some notion of ) the most likely ( i.e. most probable ) tree as a candidate for the best tree of a sentence . Background train +The best performance on the WSJ corpus was achieved by a combination of the SATZ system ( ) with the Alembic system ( Aberdeen et al. 1995 ) : a 0.5 % error rate . CompareOrContrast train +The extracted frames are noisy as a result of parser errors and so are filtered using the binomial hypothesis theory ( BHT ) , following . Background train +This method allows the efficient retrieval of arbitrary length n-grams ( Nagao and Mori , 94 ; Haruno et al. , 96 ; Ikehaxa et al. , 96 ; ; Russell , 1998 ) . Background train +The current system learns finite state flowcharts whereas typical learning systems usually acquire coefficient values as in Minsky and Papert ( 1969 ) , assertional statements as in , or semantic nets as in Winston ( 1975 ) . CompareOrContrast val +Our most accurate product model achieves an F score of 92.5 without the use of discriminative reranking and comes close to the best known numbers on this test set ( ) . CompareOrContrast train +To demonstrate that this is possible we have implemented a system which constructs dictionary entries for the PATR-II system ( and references therein ) . Uses val +Other works ( Kasper et al. , 1995 ; ) convert HPSG grammars into LTAG grammars . CompareOrContrast train +The work of demonstrates that faceted queries can be converted into simple filtering constraints to boost precision . Background val +Efficient hardware implementation is also possible via chip-level parallelism ( ) . Future train +2We could just as easily use other symmetric `` association '' measures , such as 02 ( Gale & Church , 1991 ) or the Dice coefficient ( ) . CompareOrContrast train + have argued that Dale and Reiter 's ( 1995 ) dichotomy between salient and nonsalient objects ( where the objects in the domain are the salient ones ) should be replaced by an account that takes degrees of salience into account : No object can be too unsalient to be referred to , as long as the right properties are available . Background train +The acquisition of dialogue as implemented in VNLCE is reminiscent of the program synthesis methodology developed by where program flowcharts were constructed from traces of their behaviors . CompareOrContrast train +Character classes , such as punctuation , are defined according to the Unicode Standard ( ) . Uses train +Rather than producing a complete analysis of sentences , the alternative is to perform only partial analysis of the syntactic structures in a text ( Harris , 1957 ; Abney , 1991 ; ) . Background train +Finally , it has been shown by Groesser ( 1981 ) that the ratio of derived to explicit information necessary for understanding a piece of text is about 8:1 ; furthermore , our reading of the analysis of five paragraphs by strongly suggests that only the most direct or obvious inferences are being made in the process of building a model or constructing a theory of a paragraph . Motivation train +In other methods , lexical resources are specifically tailored to meet the requirements of the domain ( Rosario and Hearst , 2001 ) or the system ( ) . Background train +The exact form of M ( Si ) need not be discussed at this point ; it could be a conceptual dependence graph ( ) , a deep parse of Si , or some other representation . Background train +Some methods are based on likelihood ( Och and Ney , 2002 ; Blunsom et al. , 2008 ) , error rate ( Och , 2003 ; ; Pauls et al. , 2009 ; Galley and Quirk , 2011 ) , margin ( Watanabe et al. , 2007 ; Chiang et al. , 2008 ) and ranking ( Hopkins and May , 2011 ) , and among which minimum error rate training ( MERT ) ( Och , 2003 ) is the most popular one . Background train +Similarly , ( ) and ( Shinyanma et al. , 2002 ) learn sentence level paraphrase templates from a corpus of news articles stemming from different news source . Background train +This setup is also scalable to a higher number of word pairs ( 350 ) as was shown in . Background train +More recent work on terminology structuring has focussed on formal similarity to develop hypotheses on the semantic relationships between terms : Daille ( 2003 ) uses derivational morphology ; use , as a starting point , a number of identical characters . Background train +Louwerse et al. ( 2006 ) and Louwerse et al. ( 2007 ) study the relation between eye gaze , facial expression , pauses and dialogue structure in annotated English map-task dialogues ( ) and find correlations between the various modalities both within and across speakers . Background train +Some examples include text categorization ( Lewis and Catlett 1994 ) , base noun phrase chunking ( Ngai and Yarowsky 2000 ) , part-of-speech tagging ( Engelson Dagan 1996 ) , spelling confusion set disambiguation ( ) , and word sense disambiguation ( Fujii et al. 1998 ) . Background train +It is known that certain cue words and phrases ( ) can serve as explicit indicators of discourse structure . Motivation val +The last point may be seen better if we look at some differences between our system and KRYPTON , which also distinguishes between an object theory and background knowledge ( cfXXX ) . CompareOrContrast val +One approach to this problem is that taken by the ASCOT project ( ; Akkerman , 1986 ) . Background val +Though we could have used a further downstream measure like BLEU , METEOR has also been shown to directly correlate with translation quality ( ) and is simpler to measure . Motivation train +For all experiments reported in this section we used the syntactic dependency parser MaltParser v1 .3 ( Nivre 2003 , 2008 ; Kübler , McDonald , and ) , a transition-based parser with an input buffer and a stack , which uses SVM classifiers Uses val +We conducted experiments with gold features to assess the potential of these features , and with predicted features , obtained from training a simple maximum likelihood estimation classifier on this resource ( ) .19 The first part of Table 8 shows that the RAT ( rationality ) feature is very relevant ( in gold ) , but suffers from low accuracy ( no gains in machine-predicted input ) . Uses val +In our previous work ( ; Salloum and Habash , 2012 ) , we applied our approach to tokenized Arabic and our DA-MSA transfer component used feature transfer rules only . CompareOrContrast train +It has already been used to implement a framework for teaching NLP ( ) . Extends train +Due to their remarkable ability to incorporate context structure information and long distance reordering into the translation process , tree-based translation models have shown promising progress in improving translation quality ( Liu et al. , 2006 , 2009 ; Quirk et al. , 2005 ; Galley et al. , 2004 , 2006 ; ; Shen et al. , 2008 ; Zhang et al. , 2011b ) . Background train +Liu et al. ( 2005 ) , Meral et al. ( 2007 ) , , Murphy and Vogel ( 2007 ) and Topkara et al. ( 2006a ) all belong to the syntactic transformation category . Background train +Similarly , the notion of R + M-abduction is spiritually related to the `` abductive inference '' of , the `` diagnosis from first principles '' of Reiter ( 1987 ) , `` explainability '' of Poole ( 1988 ) , and the subset principle of Berwick ( 1986 ) . CompareOrContrast train +On the other side , wikis started as collective works where each entry is not owned by a single author e.g. . Background val +To address this inconsistency in the correspondence between inflectional features and morphemes , and inspired by Smrž ( 2007 ) , we distinguish between two types of inflectional features : formbased ( a.k.a. surface , or illusory ) features and functional features .6 Most available Arabic NLP tools and resources model morphology using formbased ( `` surface '' ) inflectional features , and do not mark rationality ; this includes the Penn Arabic Treebank ( PATB ) ( ) , the Buckwalter morphological analyzer ( Buckwalter 2004 ) , and tools using them such as the Morphological Analysis and Disambiguation for Arabic ( MADA ) toolkit ( Habash and Rambow 2005 ; Habash , Rambow , and Roth 2012 ) . CompareOrContrast train + has built a semantic role classifier exploiting the interdependence of semantic roles . Uses val +Others have applied the NLP technologies of near-duplicate detection and topic-based text categorization to politically oriented text ( Yang and Callan , 2005 ; ) . Background val +8 It is based on the dataset of ,9 which consists of 1000 positive and 1000 negative movie reviews , tokenized and divided into 10 folds ( F0 -- F9 ) . Extends train +In previous work ( ) , we described an experimental text-to-speech system that determined prosodic phrasing for the Olive -- Liberman synthesizer ( Olive and Liberman 1985 ) . Extends train +The UMLS -- the Unified Medical Language System ( UMLS ) has been developed and maintained by National Library of Medicine ( NLM ) . Background train +The data used in the experiment was selected from the Penn Treebank Wall Street Journal , and is the same used by . Uses train +As shown in this is a well-motivated convention since it avoids splitting up lexical rules to transfer the specifications that must be preserved for different lexical entries . Motivation train +The reader may consult recent papers on this subject ( e.g. Moens and Steedman 1987 ; ) to see what a formal interpretation of events in time might look like . Background train +Proceedings of EACL '99 example , the ALE parser ( ) presupposes a phrase structure backbone which can be used to determine whether a constraint is to be interpreted bottom-up or topdown . Background train +Following Soon et al. ( 2001 ) , we represent use the ACE training data for acquiring our SC clasSCA as a binary value that indicates whether the insifier ; instead , we use the BBN Entity Type Corpus duced SCs of the two NPs involved are the same or ( ) , which consists of not . Uses train +Following , we also compare the performance of our system with a system using features based on the Brown clusters of the word types in a document . Uses train + use mutual information to identify collocations , a method they claim is reasonably effective for words with a frequency of not less than five . Background train +We do this with a first-order HMM part-ofspeech tagger ( Merialdo ) . Uses train +Using the bottom-up , dynamic programming technique ( see the appendix for details ) of computing inside probabilities ( ) , we can efficiently compute the probability of the sentence , P ( w | G ) . Uses train +As shown in ( ) • The presented research was carried out at the University of Tubingen , Germany , as part of the Sonderforschungsbereich 340 . Background train +According to , paragraphs are made up of segments , which in turn are made up of sentences or clauses , which in turn are made up of phrases . Background train +Similarly , the notion of R + M-abduction is spiritually related to the `` abductive inference '' of Reggia ( 1985 ) , the `` diagnosis from first principles '' of , `` explainability '' of Poole ( 1988 ) , and the subset principle of Berwick ( 1986 ) . CompareOrContrast train +This choice is motivated by an observation we made previously ( a ) : since each post in a sequence is a reply to the preceding post , we could exploit their dependencies by determining their stance labels together .3 As our sequence learner , we employ a maximum entropy Markov model ( MEMM ) ( McCallum et al. , 2000 ) . Extends train +Over the past decade , researchers at IBM have developed a series of increasingly sophisticated statistical models for machine translation ( ; Brown et al. , 1990 ; Brown et al. , 1993a ) . Background train +This may be because pipelines have many engineering advantages , and in practice the sort of problems pointed out by Danlos and other pipeline critics do not seem to be a major problem in current applied NLG systems ( ) . Background train +This approach has its roots in Fillmore 's Case Grammar ( 1968 ) , and serves as the foundation for two current large-scale semantic annotation projects : FrameNet ( Baker et al. , 1998 ) and PropBank ( ) . Background train +Two exceptions to this generalisation are the Linguistic String Project ( ) and the IBM CRITIQUE ( formerly EPISTLE ) Project ( Heidorn et al. , 1982 ; Byrd , 1983 ) ; the former employs a dictionary of approximately 10,000 words , most of which are specialist medical terms , the latter has well over 100,000 entries , gathered from machine readable sources . CompareOrContrast val +They can also shift the dialogue 2The notion of the initiative in this paper is different from that of the dialogue initiative of . CompareOrContrast train +The Google n-gram data was collected by Google Research for statistical language modelling , and has been used for many tasks such as lexical disambiguation ( ) , and contains English n-grams and their observed frequency counts , for counts of at least 40 . Background train +Withindocument coreference resolution has been applied to produce summaries of text surrounding occurrences of the name ( Bagga and Baldwin , 1998 ; ) . Background train +To represent the semantics of predicative units , we use FrameNet inventory of frames and frame elements ( C. ) . Uses train +Similar observation for surface word frequency was also observed by ( Bertram et al. , 2000 ; Bradley , 1980 ; Burani et al. , 1987 ; Burani et al. , 1984 ; Schreuder et al. , 1997 ; Taft 1975 ; ) where it has been claimed that words having low surface frequency tends to decompose . Background train +These two sets of data were used for automatic dialogue act classification , which was run in the Weka system ( ) . Uses train +• The transition probability a is 0.7 using the EM algorithm ( ) on the TREC4 ad-hoc query set . Uses train + obtain promising results in dialogue act tagging of the Switchboard-DAMSL corpus using lexical , syntactic and prosodic cues , while Gravano and Hirschberg ( 2009 ) examine the relation between particular acoustic and prosodic turn-yielding cues and turn taking in a large corpus of task-oriented dialogues . Background val +( ) has found strong correlations between DF , IG and the X2 statistic for a term . Background train +Despite this , to date , there has been little work on corpus-based approaches to help-desk response automation ( notable exceptions are Carmel , Shtalhaim , and Soffer 2000 ; ; Bickel and Scheffer 2004 ; Malik , Subramaniam , and Kaushik 2007 ) . Background train +To quantify the relative strengths of these transitive inferences , propose to assign a weight to each link . Background train +While these approaches have been reasonably successful ( see ) , Kehler et al. ( 2004 ) speculate that deeper linguistic knowledge needs to be made available to resolvers in order to reach the next level of performance . Background train +Lisp is not particularly well suited for interfacing to complex , structured objects , and it was not our intention to embark on a major effort involving the development of a formal model of a dictionary ( of the style described in , eg. , ) ; on the other hand a method of access was clearly required , which was flexible enough to support a range of applications intending to make use of the LDOCE tape . Background train +We train a 4-gram language model on the Xinhua portion of the English Gigaword corpus using the SRILM Toolkits ( Stolcke , 2002 ) with modified Kneser-Ney smoothing ( ) . Uses val +• Learnability ( Zernik and Dyer 1987 ) • Text generation ( ; Milosavljevic , Tulloch , and Dale 1996 ) • Speech generation ( Rayner and Carter 1997 ) • Localization ( Sch ¨ aler 1996 ) Background train +Details of the top performing heuristics of COCKTAIL were reported in ( ) . Background train +Representative systems are described in , De Mattia and Giachin ( 1989 ) , Niedermair ( 1989 ) , Niemann ( 1990 ) , and Young ( 1989 ) . Background train +In addition , note that our Object Raising rule would assign mean to this category incorrectly . CompareOrContrast train +The parallel corpus is word-aligned using GIZA + + ( ) . Uses train +tions for the remaining 20 % of the instances ; and ( 3 ) train an SVM classifier ( using the LIBSVM package ( ) ) on these 20 % of the instances , where each instance , i , is represented by a set of 31 binary features . Uses val +Our algorithm is similar to the approach taken by for inducing PCFG parsers . CompareOrContrast train +Recently , several alternative , often quite sophisticated approaches to collective classification have been proposed ( Neville and Jensen , 2000 ; Lafferty et al. , 2001 ; Getoor et al. , 2002 ; Taskar et al. , 2002 ; Taskar et al. , 2003 ; Taskar et al. , 2004 ; ) . Background val +This experiment was again replicated by with 10 subjects . Background train +Children use vague adjectives among their first dozens of words ( Peccei 1994 ) and understand some of their intricacies as early as their 24th month ( ) . Background train +Other approaches use less deep linguistic resources ( e.g. , POS-tags Stymne ( 2008 ) ) or are ( almost ) knowledge-free ( e.g. , ) . CompareOrContrast train +This includes work on question answering ( Wang et al. , 2007 ) , sentiment analysis ( Nakagawa et al. , 2010 ) , MT reordering ( ) , and many other tasks . Background train +( 4 ) NE : We use BBN 's IdentiFinder ( ) , a MUC-style NE recognizer to determine the NE type of NPZ . Uses train + has made the first attempt working on the single semantic role level to make further improvement . CompareOrContrast train +In particular , since we treat each individual speech within a debate as a single `` document '' , we are considering a version of document-level sentiment-polarity classification , namely , automatically distinguishing between positive and negative documents ( Das and Chen , 2001 ; Pang et al. , 2002 ; ; Dave et al. , 2003 ) . Background train +raw length value as a feature , we follow our previous work ( ; Wagner et al. , 2014 ) and create multiple features for length using a decision tree ( J48 ) . Extends train + used unification in an SMT system to model some of the CompareOrContrast train +Following Ruch et al. ( 2003 ) and , we employed Hidden Markov Models to model the discourse structure of MEDLINE abstracts . Uses train +The language grounding problem has received significant attention in recent years , owed in part to the wide availability of data sets ( e.g. Flickr , Von Ahn ( 2006 ) ) , computing power , improved computer vision models ( Oliva and Torralba , 2001 ; Lowe , 2004 ; Farhadi et al. , 2009 ; ) and neurological evidence of ties between the language , perceptual and motor systems in the brain ( Pulverm ¨ uller et al. , 2005 ; Tettamanti et al. , 2005 ; Aziz-Zadeh et al. , 2006 ) . Background train +Due to their remarkable ability to incorporate context structure information and long distance reordering into the translation process , tree-based translation models have shown promising progress in improving translation quality ( Liu et al. , 2006 , 2009 ; Quirk et al. , 2005 ; Galley et al. , 2004 , 2006 ; Marcu et al. , 2006 ; ; Zhang et al. , 2011b ) . Background train +Previous work has argued that initiative affects the degree of control an agent has in the dialogue interaction ( Whittaker and Stenton , 1988 ; ; Chu-Carroll and Brown , 1998 ) . Background val +Many provide graphical user interfaces ( GUI ) for manual annotation ( e.g. General Architecture for Text Engineering ( GATE ) ( Cunningham et al. , 1997 ) and the Alembic Workbench ( ) ) as well as NLP tools and resources that can be manipulated from the GUI . Background train +A number of speech understanding systems have been developed during the past fifteen years ( Barnett et al. 1980 , Dixon and Martin 1979 , Erman et al. 1980 , Haton and Pierrel 1976 , Lea 1980 , Lowerre and Reddy 1980 , , Reddy 1976 , Walker 1978 , and Wolf and Woods 1980 ) . CompareOrContrast train +It allows the construction of a non-TAL ( Shieber , 1994 ) , ( ) . Background train +In addition , the advantages of using linguistically annotated data over raw data are well documented ( Mair , 2005 ; ) . Background val +In particular , the `` Semantic Information Retrieval '' project ( SIR ) systematically investigates the use of lexical-semantic relations between words or concepts for improving the performance of information retrieval systems . Motivation val +Our re-ranking approach , like the approach to parse re-ranking of , employs a simpler model -- a local semantic role labeling algorithm -- as a first pass to generate a set of n likely complete assignments of labels to all parse tree nodes . CompareOrContrast train +For instance , the Alembic workbench ( ) contains a sentence-splitting module that employs over 100 regular-expression rules written in Flex . Background train +In contrast , a single statistical model allows one to maintain a single table ( ) . Background train +Our experiments are based on the multimodal extension of Latent Dirichlet Allocation developed by . Uses train +First , it has been noted that in many natural language applications it is sufficient to use shallow parsing information ; information such as noun phrases ( NPs ) and other syntactic sequences have been found useful in many large-scale language processing applications including information extraction and text summarization ( Grishman , 1995 ; ) . Background train + adopted a Bayesian method to infer an STSG by exploring the space of alignments based on parse trees . CompareOrContrast train +In corpus linguistics building such megacorpora is beyond the scope of individual researchers , and they are not easily accessible ( : 56 ) unless the web is used as a corpus ( Kilgarriff and Grefenstette , 2003 ) . Background train +In addition , there are a number of projects under way to develop substantial lexicons from machine readable sources ( see for details ) . Background train +We carried out two parallel experiments with two parsers available for Czech , parser I ( Hajie et al. , 1998 ) and parser II ( ) . Uses val +Part of speech taggers typically require input in the format of a single sentence per line ( for example Brill 's tagger ( ) ) and parsers generally aim to produce a tree spanning each sentence . Background train +The elimination of redundant nondeterminism is based on Unfold/Fold transformation techniques ( ) .29 The unfolding transformation is also referred to as partial execution , for example , by Pereira and Shieber ( 1987 ) . Uses train +Problems such as these have motivated research on more abstract , dependencybased parser evaluation ( e.g. , Lin 1995 ; Carroll , Briscoe , and Sanfilippo 1998 ; Carroll et al. 2002 ; Clark and Hockenmaier 2002 ; ; Preiss 2003 ; Kaplan et al. 2004 ; Miyao and Tsujii 2004 ) . Motivation train +This seems to provide additional evidence of b ) 's suggestion that something like a distributional hypothesis of images is plausible . CompareOrContrast train +For instance , GATE currently provides a POS tagger , named entity recogniser and gazetteer and ontology editors ( ) . Background train +Latent variables we wish to consider are an increased number of word classes ; more flexible regions -- see on learning a state transition diagram for acoustic regions in phone recognition -- and phonological features and syllable boundaries . Background train +Since earlier versions of the SNoW based CSCL were used only to identify single phrases ( Punyakanok and Roth , 2001 ; Munoz et al. , 1999 ) and never to identify a collection of several phrases at the same time , as we do here , we also trained and tested it under the exact conditions of CoNLL-2000 ( Tjong Kim ) to compare it to other shallow parsers . Uses train +Using the implicit modeling of argument consistency , we follow the same approach as in our previous work ( ) and trained a logistic regression model to predict verb alignment based on the features in Table 1 . Extends train +Recent developments in linguistics , and especially on grammatical theory -- for example , Generalised Phrase Structure Grammar ( GPSG ) ( Gazdar et al. , 1985 ) , Lexical Functional Grammar ( LFG ) ( ) -- and on natural language parsing frameworks -- for example , Functional Unification Grammar ( FUG ) ( Kay , 1984a ) , PATR-II ( Shieber , 1984 ) -- make it feasible to consider the implementation of efficient systems for the syntactic analysis of substantial fragments of natural language . Background train +We train a 4-gram language model on the Xinhua portion of the English Gigaword corpus using the SRILM Toolkits ( ) with modified Kneser-Ney smoothing ( Chen and Goodman , 1998 ) . Uses train +It is only recently that the web name ambiguity has been approached as a separate problem and defined as an NLP task Web People Search on its own ( ; Artiles et al. , 2007 ) . Background train +This deficiency is rectified in the verb classification system employed by in the Brandeis verb catalogue . CompareOrContrast train +Previous work has argued that initiative affects the degree of control an agent has in the dialogue interaction ( ; Walker and Whittaker , 1990 ; Chu-Carroll and Brown , 1998 ) . Background val +1990 ) , on linguisitic acquisition ( by the use of Part-of-Speech filters hand-crafted by a linguist ) ( ) or , more frequently , on a combination of the two ( Smadja , 1993 ; Kilgarriff and Tugwell , 2001 , for example ) . CompareOrContrast val +It is frequently used in tasks like scene identification , and shows that distance in GIST space correlates well with semantic distance in WordNet . Motivation train +Because each rule r consists of a target tree fragment frag and a source string str in the model , we follow and decompose the prior probability P0 ( r | N ) into two factors as follows : Uses train +To address this inconsistency in the correspondence between inflectional features and morphemes , and inspired by Smrž ( 2007 ) , we distinguish between two types of inflectional features : formbased ( a.k.a. surface , or illusory ) features and functional features .6 Most available Arabic NLP tools and resources model morphology using formbased ( `` surface '' ) inflectional features , and do not mark rationality ; this includes the Penn Arabic Treebank ( PATB ) ( Maamouri et al. 2004 ) , the Buckwalter morphological analyzer ( Buckwalter 2004 ) , and tools using them such as the Morphological Analysis and Disambiguation for Arabic ( MADA ) toolkit ( ; Habash , Rambow , and Roth 2012 ) . CompareOrContrast train +In our prior work ( ) , we examined whether techniques used for predicting the helpfulness of product reviews ( Kim et al. , 2006 ) could be tailored to our peer-review domain , where the definition of helpfulness is largely influenced by the educational context of peer review . Extends train +Previous sentiment-analysis work in different domains has considered inter-document similarity ( ; Pang and Lee , 2005 ; Goldberg and Zhu , 2006 ) or explicit Background train +a ) show how a BoVW model may be easily combined with a distributional vector space model of language using only vector concatenation . Background train +The following are the various tag sets we use in this article : ( a ) the core POS tag sets CORE44 and the newly introduced CORE12 ; ( b ) CATiB Treebank tag set ( CATIB6 ) ( ) and its newly introduced extension of CATIBEX created using simple regular expressions on word form , indicating particular morphemes such as the prefix JI Al + or the suffix v ' + wn ; this tag set is the best-performing tag set for Arabic on predicted values as reported in Section 4 ; ( c ) the PATB full tag set with complete morphological tag ( BW ) ( Buckwalter 2004 ) ; and two extensions of the PATB reduced tag set ( PENN POS , a.k.a. RTS , size 24 [ Diab , Hacioglu , and Jurafsky 2004 ] ) , both outperforming it : ( d ) Kulick , Gabbard , and Marcus ( 2006 ) 's tag set ( KULICK ) , size 43 , one of whose most important extensions is the marking of the definite article clitic , and ( e ) Diab and Benajiba 's ( in preparation ) EXTENDED RTS tag set ( ERTS ) , which marks gender , number , and definiteness , size 134 . Uses val +The numeral ( whether it is implicit , as in ( 3 ) , or explicit ) can be construed as allowing the reader to draw inferences about the standards employed ( ; DeVault and Stone 2004 ) : ( 3 ) , for example , implies a standard that counts 10 cm as large and 8 cm as not large . Background train +Agreement between two annotation sets is calculated here in terms of Cohen 's kappa ( ) 1 and corrected kappa ( Brennan and Prediger , 1981 ) 2 . Uses train +On the WSJ corpus our system performed slightly better than the combination of the Alembic and SATZ systems described in ( 0.44 % vs. 0.5 % error rate ) . CompareOrContrast train + studied the issue of disambiguation for mono-lingual M. Background train +Recently , several alternative , often quite sophisticated approaches to collective classification have been proposed ( Neville and Jensen , 2000 ; ; Getoor et al. , 2002 ; Taskar et al. , 2002 ; Taskar et al. , 2003 ; Taskar et al. , 2004 ; McCallum and Wellner , 2004 ) . Background train +To address this inconsistency in the correspondence between inflectional features and morphemes , and inspired by Smrž ( 2007 ) , we distinguish between two types of inflectional features : formbased ( a.k.a. surface , or illusory ) features and functional features .6 Most available Arabic NLP tools and resources model morphology using formbased ( `` surface '' ) inflectional features , and do not mark rationality ; this includes the Penn Arabic Treebank ( PATB ) ( Maamouri et al. 2004 ) , the Buckwalter morphological analyzer ( ) , and tools using them such as the Morphological Analysis and Disambiguation for Arabic ( MADA ) toolkit ( Habash and Rambow 2005 ; Habash , Rambow , and Roth 2012 ) . CompareOrContrast val +Although in this paper we take modus ponens as the main rule of inference , in general one can consider deductive closures with respect to weaker , nonstandard logics , ( cfXXX Levesque 1984 ; ; Patel-Schneider 1985 ) . CompareOrContrast train +Riehemann 1993 ; Oliva 1994 ; Frank 1994 ; ; Sanfilippo 1995 ) . CompareOrContrast train +Berger et al. 2000 ; Jijkoun and de Rijke 2005 ; ) . CompareOrContrast train +We will examine the worst-case complexity of interpretation as well as generation to shed some light on the hypothesis that vague descriptions are more difficult to process than others because they involve a comparison between objects ( , Krahmer and Theune 2002 ) . Background train +A number of proposals in the 1990s deliberately limited the extent to which they relied on domain and/or linguistic knowledge and reported promising results in knowledge-poor operational environments ( Dagan and Itai 1990 , 1991 ; Lappin and Leass 1994 ; Nasukawa 1994 ; ; Williams , Harvey , and Preston 1996 ; Baldwin 1997 ; Mitkov 1996 , 1998b ) . Background train +From the Meaning-Text Theory ( MTT ) 1 point of view , Natural Language ( NL ) is considered as a correspondence between meanings and texts ( ) . Background train +This means that natural language expressions such as `` A is B , '' `` A is the same as B , '' etc. are not directly represented by logical equality ; similarly , `` not '' is often not treated as logical negation ; cfXXX . CompareOrContrast train +The full-listing model claims that polymorphic words are represented as a whole in the human mental lexicon ( ; Butterworth , 1983 ) . Background train +As for work on Arabic ( MSA ) , results have been reported on the PATB ( Kulick , Gabbard , and Marcus 2006 ; Diab 2007 ; Green and Manning 2010 ) , the Prague Dependency Treebank ( PADT ) ( ; Nivre 2008 ) and the CATiB ( Habash and Roth 2009 ) . Background train +However , learning-based resolvers have not been able to benefit from having an SC agreement feature , presumably because the method used to compute the SC of an NP is too simplistic : while the SC of a proper name is computed fairly accurately using a named entity ( NE ) recognizer , many resolvers simply assign to a common noun the first ( i.e. , most frequent ) WordNet sense as its SC ( e.g. , Soon et al. ( 2001 ) , ) . Background train +Thus , over the past few years , along with advances in the use of learning and statistical methods for acquisition of full parsers ( Collins , 1997 ; Charniak , 1997a ; Charniak , 1997b ; Ratnaparkhi , 1997 ) , significant progress has been made on the use of statistical learning methods to recognize shallow parsing patterns syntactic phrases or words that participate in a syntactic relationship ( Church , 1988 ; Ramshaw and Marcus , 1995 ; Argamon et al. , 1998 ; Cardie and Pierce , 1998 ; ; Punyakanok and Roth , 2001 ; Buchholz et al. , 1999 ; Tjong Kim Sang and Buchholz , 2000 ) . Background train +The ability to explicitly identify these sections in unstructured text could play an important role in applications such as document summarization ( ) , information retrieval ( Tbahriti et al. , 2005 ) , information extraction ( Mizuta et al. , 2005 ) , and question answering . Background train +Many statistical parsers ( Ratnaparkhi , 1999 ; Collins , 1999 ; Charniak , 2001 ) are based on a history-based probability model ( ) , where the probability of each decision in a parse is conditioned on the previous decisions in the parse . Background train +For example , such schema can serve as a mean to represent translation examples , or find structural correspondences for the purpose of transfer grammar learning ( Menezes & Richardson , 2001 ) , ( Aramaki et al. , 2001 ) , ( Watanabe et al. , 2000 ) , ( Meyers et al. , 2000 ) , ( Matsumoto et al. , 1993 ) , ( kaji et al. , 1992 ) , and example-base machine translation EBMT3 ( ) , ( Sato , 1991 ) , ( Richardson et al. , 2001 ) , ( Al-Adhaileh & Tang , 1999 ) . Background train +For instance , ( ) acquire two-argument templates ( inference rules ) from corpora using an extended version of the distributional analysis in which paths in dependency trees that have similar arguments are taken to be close in meaning . Background train +Per-state joint normalization ( b , § 8.2 ) is similar but drops the dependence on a . CompareOrContrast train +Other studies on the value of disambiguation for cross-lingual IR include Hiemstra and de Jong , 1999 ; . Background train +We have not yet made use of TINA 'S probabilities in adjusting the recognizer scores on the fly , but we have been able to incorporate linguistic scores to resort N-best outputs , giving a significant improvement in performance ( ) . Uses train +These features are carefully designed to reduce the data sparseness problem and some of them are inspired by previous work ( He et al. , 2008 ; Gimpel and Smith , 2008 ; ; Chiang et al. , 2009 ; Setiawan et al. , 2009 ; Shen et al. , 2009 ; Xiong et al. , 2009 ) : 1 . Motivation train +Nevertheless , the full document text is present in most systems , sometimes as the only feature ( Sugiyama and Okumura , 2007 ) and sometimes in combination with others see for instance ( ; Popescu and Magnini , 2007 ) - . Background train +More recently , ( ) has performed a good survey of document categorization ; recent works can also be found in ( Joachims , 2002 ) , ( Crammer and Singer , 2003 ) , and ( Lewis et al. , 2004 ) . Background train +McDonald has even argued for extending the model to a large number of components ( McDonald 1988 ) , and several systems have indeed added an additional component between the planner and the linguistic component ( ; Panaget 1994 ; Wanner 1994 ) . Background train +It is these orthographic variations and complex morphological structure that make Arabic language processing challenging ( Xu et al. , 2001 ; ) . Background train +A number of speech understanding systems have been developed during the past fifteen years ( Barnett et al. 1980 , Dixon and Martin 1979 , , Haton and Pierrel 1976 , Lea 1980 , Lowerre and Reddy 1980 , Medress 1980 , Reddy 1976 , Walker 1978 , and Wolf and Woods 1980 ) . CompareOrContrast train +It has been more difficult showing that agreement morphology helps parsing , however , with negative results for dependency parsing in several languages ( Eryigit , Nivre , and Oflazer 2008 ; Nivre , Boguslavsky , and Iomdin 2008 ; ) . Motivation train +Some efforts have tackled tasks such as automatic image caption generation ( a ; Ordonez et al. , 2011 ) , text illustration ( Joshi et al. , 2006 ) , or automatic location identification of Twitter users ( Eisenstein et al. , 2010 ; Wing and Baldridge , 2011 ; Roller et al. , 2012 ) . Background train +This idea was proposed by Krauwer and des Tombe ( 1981 ) , Langendoen and Langsam ( 1987 ) , and Pulman ( 1986 ) , and was rediscovered by Black ( 1989 ) and recently by . Background train +A nonprobabilistic approach for DA labeling proposed by Samuel , Carberry , and Vijay-Shanker ( 1998 ) is transformation-based learning ( ) . CompareOrContrast train +However , since work in this direction has started , a significant progress has also been made in the research on statistical learning of full parsers , both in terms of accuracy and processing time ( Charniak , 1997b ; Charniak , 1997a ; Collins , 1997 ; ) . Background val +Some methods are based on likelihood ( Och and Ney , 2002 ; Blunsom et al. , 2008 ) , error rate ( Och , 2003 ; Zhao and Chen , 2009 ; Pauls et al. , 2009 ; ) , margin ( Watanabe et al. , 2007 ; Chiang et al. , 2008 ) and ranking ( Hopkins and May , 2011 ) , and among which minimum error rate training ( MERT ) ( Och , 2003 ) is the most popular one . Background train +The algorithm we implemented is inspired by the work of on word sense disambiguation . Motivation val +Various approaches for computing semantic relatedness of words or concepts have been proposed , e.g. dictionary-based ( Lesk , 1986 ) , ontology-based ( Wu and Palmer , 1994 ; Leacock and Chodorow , 1998 ) , information-based ( Resnik , 1995 ; Jiang and Conrath , 1997 ) or distributional ( ) . Background train +EM maximizes G ( 0 ) via block-coordinate ascent on a lower bound F ( q , 0 ) using an auxiliary distribution over the latent variables q ( z | x , y ) ( ) : Uses train +It has been argued that generating such modifiers is not a trivial decision because it interferes with the planning of both local and global coherence ( in the sense of ( Grosz and Sidner , 1986 ) ) ( a ) . Background train +• Only qualitative observations of the responses were reported ( no formal evaluation was performed ) ( ; Roy and Subramaniam 2006 ) . CompareOrContrast train +A stops B from doing something ; A disagreees with B on something , 8 % and 12 % ) Note that in our original work ( ) , only development data were used to show some initial observations . CompareOrContrast train +We evaluated on the English CCGBank ( Hockenmaier and Steedman , 2007 ) , which is a transformation of the Penn Treebank ( ) ; the CTBCCG ( Tse and Curran , 2010 ) transformation of the Penn Chinese Treebank ( Xue et al. , 2005 ) ; and the CCG-TUT corpus ( Bos et al. , 2009 ) , built from the TUT corpus of Italian text ( Bosco et al. , 2000 ) . Uses train +The EM algorithm ( ) can maximize these functions . Uses train +Finally , feedback expressions ( head nods and shakes ) are successfully predicted from speech , prosody and eye gaze in interaction with Embodied Communication Agents as well as human communication ( Fujie et al. , 2004 ; Morency et al. , 2005 ; ; Morency et al. , 2009 ) . Background train +For these or for a specific domain , basic synonymic dictionaries can be complemented using learning methods based on distributional similarity ( Pereira et al. , 1993 ; ) . Future train +For projective parsing , it is significantly faster than exact dynamic programming , at the cost of small amounts of search error , We are interested in extending these ideas to phrase-structure and lattice parsing , and in trying other higher-order features , such as those used in parse reranking ( Charniak and Johnson , 2005 ; Huang , 2008 ) and history-based parsing ( ) . Future train +The coreference system system is similar to the Bell tree algorithm as described by ( ) . CompareOrContrast train +Other representations use the link structure ( Malin , 2005 ) or generate graph representations of the extracted features ( ) . Background train +The research described below is taking place in the context of three collaborative projects ( Boguraev , 1987 ; ; Phillips and Thompson , 1986 ) to develop a general-purpose , wide coverage morphological and syntactic analyser for English . Background train +We experiment with four learners commonly employed in language learning : Decision List ( DL ) : We use the DL learner as described in , motivated by its success in the related tasks of word sense disambiguation ( Yarowsky , 1995 ) and NE classification ( Collins and Singer , 1999 ) . Uses train +transition-based dependency parsing framework ( Nivre , 2008 ) using an arc-eager transition strategy and are trained using the perceptron algorithm as in with a beam size of 8 . Uses train +Using the basic solution proposed by ( ) as a term of comparison , we experiment with different sources of multilingual lexical knowledge to address the following questions : ( 1 ) What is the potential of the existing multilingual lexical resources to approach CLTE ? CompareOrContrast train +Acoustic models for HTK is trained with the continuous speech database of the Acoustical Society of Japan ( ) . Uses train +Automatic text categorization has been used in search engines , digital library systems , and document management systems ( ) . Background train +For our Text modality , we use deWaC , a large German web corpus created by the WaCKy group ( ) containing approximately 1.7 B word tokens . Uses train +Unless very high rates of misspellings are to be expected ( this explains the favorable results for trigram indexing in ( ) ) one can not really recommend this method . CompareOrContrast val +For english , there is for instance the 15 year old HewlettPackard test suite , a simple text file listing test sentences and grouping them according to linguistics phenomena ( ) ; and more recently , the much more sophisticated TSNLP ( Test Suite for Natural Language Processing ) which includes some 9500 test items for English , French and German , each of them being annotated with syntactic and application related information ( Oepen and Flickinger , 1998 ) . CompareOrContrast val +Previous work on Chinese SRL mainly focused on how to transplant the machine learning methods which has been successful with English , such as , Xue and Palmer ( 2005 ) and Xue ( 2008 ) . Background train +A similar method is included in PATR-II ( ) and can be used to encode lexical rules as binary relations in the CUF system ( Dorre and Eisele 1991 ; Done and Dorna 1993b ) or the TFS system ( Emele and Zajac 1990 ; Emele 1994 ) . CompareOrContrast train +Our training examples are similar to the data created for pseudodisambiguation , the usual evaluation task for SP models ( Erk , 2007 ; ; Rooth et al. , 1999 ) . CompareOrContrast train +With the use of computers in storing the explosive amount of biological information , natural language processing ( NLP ) approaches have been explored to make the task of managing information recorded in free text more feasible . Background val +A formula q5 of L ( =-RRB- , the language with equality , is weakly R + M-abductible from an object theory T , denoted by T I-R + m 0 , iff there exists a partial theory T e PT ( T ) and a preferred model M E PM ( T ) such that M = 0 , i.e. 0 is true in at least one preferred model of the partial theory T. Note : The notions of strong provability and strong R + M-abduction can be introduced by replacing `` there exists '' by `` all '' in the above definitions ( cfXXX b ) . CompareOrContrast train +The first version ( TIMIT ) was developed for the 450 phonetically rich sentences of the TIMIT database ( ) . Uses val +A third problem arises with the approach to the semantics of QLFs that this notion of the relationship between QLF and RQLF encourages one to adopt : it is that taken by . CompareOrContrast train +In ( ) , I present evidence from Mandarin Chinese that this analysis is on the right track . Extends train +Interaction between components is coordinated by the dialogue manager which uses the informationstate approach ( ) . Uses val +Other molecular biology databases We also included several model organism databases or nomenclature databases in the construction of the dictionary , i.e. , mouse Mouse Genome Database ( MGD ) [ 18 ] , fly FlyBase [ 19 ] , yeast Saccharomyces Genome Database ( SGD ) [ 20 ] , rat -- Rat Genome Database ( RGD ) [ 21 ] , worm -- WormBase [ 22 ] , Human Nomenclature Database ( HUGO ) [ 23 ] , Online Mendelian Inheritance in Man ( OMIM ) , and Enzyme Nomenclature Database ( ECNUM ) [ 25 , 26 ] . Uses train +There are many more distinctions which are conveyed by the conjunction of grammar codes and word qualifiers ( see , for further details ) . Background train +The most common way is to divide each half of the bitext into an equal number of segments and to align the segments so that each pair of segments Si and Ti are translations of each other ( Gale & Church , 1991 ; a ) . Background val + ) . Future train + proposes readjustment rules similar to those of Chomsky and Halle , but he claims that the readjustment of structure is part of the grammar , not part of the performance model . Background train +financial news , we created a probabilistic CzechEnglish dictionary by running GIZA + + training ( translation models 1-4 , see ) on the training part of the English-Czech WSJ parallel corpus extended by the parallel corpus of entry/translation pairs from the manual dictionary . Uses train +In the field of machine learning research , incremental training has been employed in the work ( ; Shilton et al. , 2005 ) , but there is little work for tuning parameters of statistical machine translation . Background train +A similar method is included in PATR-II ( Shieber et al. 1983 ) and can be used to encode lexical rules as binary relations in the CUF system ( Dorre and Eisele 1991 ; Done and Dorna 1993b ) or the TFS system ( Emele and Zajac 1990 ; ) . CompareOrContrast train +In a log-linear parameterization , for example , a prior that penalizes feature strengths far from 1 can be used to do feature selection and avoid overfitting ( ) . Uses train +We report performance in terms of two metrics : ( 1 ) the Fmeasure score as computed by the commonly-used MUC scorer ( ) , and ( 2 ) the accuracy on the anaphoric references , computed as the fraction of anaphoric references correctly resolved . Uses train +A study of the query log of the AllTheWeb and Altavista search sites gives an idea of the relevance of the people search task : 11-17 % of the queries were composed of a person name with additional terms and 4 % were identified as person names ( ) . Background val +This was done because purely unsupervised techniques ( e.g. , Baum-Welch [ Baum and Petrie 1966 ] or Brill 's [ b ] ) enable regularities to be induced for word classes which contain many entries , exploiting the fact that individual words that belong to a POS class occur in different ambiguity patterns . CompareOrContrast train +Future research should apply the work of and Blunsom and Osborne ( 2008 ) , who marginalize over derivations to find the most probable translation rather than the most probable derivation , to these multi-nonterminal grammars . Future train +In informal experiments described elsewhere ( ) , I found that the G2 statistic suggested by Dunning ( 1993 ) slightly outperforms 02 . Background train +porating these two KSs into our resolver : they can Following , we select as the aneach be represented as a constraint or as a feature , tecedent of each NP , NPS , the closest preceding NP and they can be applied to the resolver in isolation that is classified as coreferent with NPS . Motivation train +Using the section labels , the HMM was trained using the HTK toolkit ( ) , which efficiently performs the forward-backward algorithm and BaumWelch estimation . Uses train +In this paper , we use TF-IDF ( a kind of augmented DF ) as a feature selection criterion , in order to ensure results are comparable with those in ( ) . CompareOrContrast train +A variety of statistical methods were proposed over the recent years for learning to produce a full parse of free-text sentences ( e.g. , Bod ( 1992 ) , Magerman ( 1995 ) , Collins ( 1997 ) , , and Sekine ( 1998 ) ) . Background train +There is a rich literature on organization and lexical access of morphologically complex words where experiments have been conducted mainly for derivational suffixed words of English , Hebrew , Italian , French , Dutch , and few other languages ( Marslen-Wilson et al. , 2008 ; Frost et al. , 1997 ; ; Drews and Zwitserlood , 1995 ) . Background train +In the disambiguation of capitalized words , the most widespread method is POS tagging , which achieves about a 3 % error rate on the Brown corpus and a 5 % error rate on the WSJ corpus , as reported in . CompareOrContrast train +The problem with this approach is that any threshold is , to some extent , arbitrary , and there is evidence to suggest that , for some tasks , low counts are important ( ) . Motivation train + further labeled the SCFG rules with POS tags and unsupervised word classes . CompareOrContrast train +Steganography is concerned with hiding information in some cover medium , by manipulating properties of the medium in such a way that the hidden information is not easily detectable by an observer ( ) . Background val +We offer a theorem that highlights the broad applicability of these modeling techniques .4 If f ( input , output ) is a weighted regular relation , then the following statements are equivalent : ( 1 ) f is a joint probabilistic relation ; ( 2 ) f can be computed by a Markovian FST that halts with probability 1 ; ( 3 ) f can be expressed as a probabilistic regexp , i.e. , a regexp built up from atomic expressions a : b ( for a E E U -LCB- E -RCB- , b E A U -LCB- E -RCB- ) using concatenation , probabilistic union + p , and probabilistic closure * p. For defining conditional relations , a good regexp language is unknown to us , but they can be defined in several other ways : ( 1 ) via FSTs as in Fig. 1c , ( 2 ) by compilation of weighted rewrite rules ( Mohri and Sproat , 1996 ) , ( 3 ) by compilation of decision trees ( ) , ( 4 ) as a relation that performs contextual left-to-right replacement of input substrings by a smaller conditional relation ( Gerdemann and van Noord , 1999 ) ,5 ( 5 ) by conditionalization of a joint relation as discussed below . Background train +Word pairs containing polysemous words are expanded to concept pairs using GermaNet ( ) , the German equivalent to WordNet , as a sense inventory for each word . Uses train +Using the initial target U-trees , source sentences and word alignment , we extract minimal GHKM translation rules7 in terms of frontier nodes ( ) . Uses train +For example , while it is difficult to induce a grammar with raw text alone , the task is tractable when the syntactic analysis for each sentence is provided as a part of the training data ( ) . Background train +Our results are lower than those of full parsers , e.g. , as might be expected since much less structural data , and no lexical data are being used . CompareOrContrast train +Our work is more similar to NLG work that concentrates on structural constraints such as generative poetry ( Greene et al. , 2010 ) ( Colton et al. , 2012 ) ( Jiang and Zhou , 2008 ) or song lyrics ( Wu et al. , 2013 ) ( Ramakrishnan ) , where specified meter or rhyme schemes are enforced . CompareOrContrast val +Hermann and Deutsch ( 1976 ; also reported in ) show that greater differences are most likely to be chosen , presumably because they are more striking . Background train +The result holds for both the MaltParser ( ) and the Easy-First Parser ( Goldberg and Elhadad 2010 ) . Uses train +This system has been successfully tested with the development of plug-ins supporting instant messaging , distributed video encoding ( ) , distributed virtual worlds ( Hughes et al. , 2005 ) and digital library management ( Walkerdine and Rayson , 2004 ) . Background train +In modern syntactic theories ( e.g. , lexical-functional grammar [ LFG ] [ ; Bresnan 2001 ; Dalrymple 2001 ] , head-driven phrase structure grammar [ HPSG ] [ Pollard and Sag 1994 ] , tree-adjoining grammar [ TAG ] [ Joshi 1988 ] , and combinatory categorial grammar [ CCG ] [ Ades and Steedman 1982 ] ) , the lexicon is the central repository for much morphological , syntactic , and semantic information . Uses train +We then use Illinois Chunker ( Punyakanok and Roth , 2001 ) 6 to extract more noun phrases from the text and employ Collins head rules ( ) to identify their heads . Uses train + argues CV formations in Hindi and Urdu are either morphological or syntactical and their formation take place at the argument structure . Background train +feature Cohen 's k corrected k agreement 73.59 98.74 dial act 84.53 98.87 turn 73.52 99.16 Table 2 : Inter-coder agreement on feedback expression annotation Although researchers do not totally agree on how to measure agreement in various types of annotated data and on how to interpret the resulting figures , see , it is usually assumed that Cohen 's kappa figures over 60 are good while those over 75 are excellent ( Fleiss , 1971 ) . Background val +Collins 1996 , Charniak 1997 , Collins 1999 and ) . CompareOrContrast train +Two formalizations of lexical rules as used by HPSG linguists have been proposed , the meta-level lexical rules ( MLRs ; Calcagno 1995 ; ) and the . Background train +We run TreeTagger ( Schmid , 1994 ) for tokenization , and used the Giza + + ( ) to align the tokenized corpora at the word level . Uses train +Other milestones of recent research include the deployment of probabilistic and machine learning techniques ( Aone and Bennett 1995 ; Kehler 1997 ; Ge , Hale , and Charniak 1998 ; Cardie and Wagstaff 1999 ; the continuing interest in centering , used either in original or in revised form ( Abracos and Lopes 1994 ; ; Hahn and Strube 1997 ; Tetreault 1999 ) ; and proposals related to the evaluation methodology in anaphora resolution ( Mitkov 1998a , 2001b ) . Background train +We then go on to compare the current approach with that of some other theories with similar aims : the `` standard '' version of quasi-logical form implemented in the Core Language Engine , as rationally reconstructed by and Crouch and Pulman ( 1994 ) ; underspecified Discourse Representation Theory ( Reyle 1993 ) ; and the `` glue language '' approach of Dalrymple et al. ( 1996 ) . CompareOrContrast train +Note that this ensures that greater importance is attributed to longer chunks , as is usual in most EBMT systems ( cfXXX ; Veale and Way 1997 ; Carl 1999 ) .7 As an example , consider the translation into French of the house collapsed . Background train +Using WordNet , annotating the sem feature of an adjective involves first choosing the correct sense for the adjective 2Some descriptions of int modifiers can be found in ( b ) . Background val +One common approach is using Machine Translation ( MT ) to translate the queries to the language of the documents or translate documents to the language of the queries ( Gey et al , 1999 ; ) . CompareOrContrast train +Adding selectional restrictions ( semantic feature information , ) does not solve the problem , because isolated features offer only part of the background knowledge necessary for reference disambiguation . Background val +Software engineering research on Generative Programming ( ) attempts to solve these problems by focusing on the development of configurable elementary components and knowledge to combine these components into complete systems . Background train +Both use the evaluation software and triple encoding presented in . Uses val +In this Section , we will describe some example cases , which are drawn from the problem of using synchronous formalisms to define translations between languages ( e.g. cases ) . Background train +Some examples include text categorization ( Lewis and Catlett 1994 ) , base noun phrase chunking ( ) , part-of-speech tagging ( Engelson Dagan 1996 ) , spelling confusion set disambiguation ( Banko and Brill 2001 ) , and word sense disambiguation ( Fujii et al. 1998 ) . Background train +Most coreference resolution work simply mentions it in passing as a module in the pipelined system ( ; Durrett and Klein , 2013 ; Lee et al. , 2011 ; Bj ¨ orkelund and Kuhn , 2014 ) . Background train +Because the judges do not evaluate the same cases , we could not employ standard inter-annotator agreement measures ( ) . Uses train +Related are also the studies by Rieks op den Akker and Schulz ( 2008 ) and : both achieve promising results in the automatic segmentation of dialogue acts using the annotations in a large multimodal corpus . Background train +Due to advances in statistical syntactic parsing techniques ( Collins , 1997 ; ) , attention has recently shifted towards the harder question of analyzing the meaning of natural language sentences . Background train +The reader may consult recent papers on this subject ( e.g. ; Webber 1987 ) to see what a formal interpretation of events in time might look like . Background train +Because it is also consistent , it will be chosen as a best interpretation of S , ( cfXXX a , 1987b ) . Background val +For our experiments we used the standard division of the WSJ ( ) , with sections 2 through 21 for training ( approx . Uses train +With respect to this , we apply the different priming and other lexical decision experiments , described in literature ( ; Bentin , S. and Feldman , 1990 ) specifically for derivationally suffixed polymorphemic words and compound verbs of Bangla . Uses train +From this description , it should be clear that TM systems do not translate : Indeed , some researchers consider them to be little more than a search-and-replace engine , albeit a rather sophisticated one ( ) . Background train +Other milestones of recent research include the deployment of probabilistic and machine learning techniques ( Aone and Bennett 1995 ; ; Ge , Hale , and Charniak 1998 ; Cardie and Wagstaff 1999 ; the continuing interest in centering , used either in original or in revised form ( Abracos and Lopes 1994 ; Strube and Hahn 1996 ; Hahn and Strube 1997 ; Tetreault 1999 ) ; and proposals related to the evaluation methodology in anaphora resolution ( Mitkov 1998a , 2001b ) . Background val +Our group has developed a wide-coverage HPSG grammar for Japanese ( Mitsuishi et al. , 1998 ) , which is used in a high-accuracy Japanese dependency analyzer ( ) . Background train +There are several variations of such a method ( Ballesteros and Croft , 1998 ; ; Hull 1997 ) . CompareOrContrast val +This evaluation set-up is an improvement versus the one we previously reported ( ) , in which fixed partitions were used for training , development , and testing . Extends train + compared the performace of NEs versus BoW features . Background train +The right-side context of a non-terminal category -- the probability of generating a category to the right of the current constituent 's category -- corresponds directly to the category transitions used for the HMM supertagger of . CompareOrContrast train +The starting point for the approach followed here was a dissatisfaction with certain aspects of the theory of quasi-logical form as described in , 1992 ) , and implemented in SRI 's Core Language Engine ( CLE ) . CompareOrContrast train +However , the method we are currently using in the ATIS domain ( ) represents our most promising approach to this problem . Future train +Three UniRef tables UniRef100 , UniRef90 and UniRef50 ) are available for download : UniRef100 combines identical sequences and sub-fragments into a single UniRef entry ; and UniRef90 and UniRef50 are built by clustering UniRef100 sequences into clusters based on the CD-HIT algorithm such that each cluster is composed of sequences that have at least 90 % or 50 % sequence similarity , respectively , to the representative sequence . Uses train + , Charniak 1997 , Collins 1999 and Charniak 2000 ) . CompareOrContrast train +To name a few examples , Rohrbach et al. ( 2010 ) and Socher et al. ( 2013 ) show how semantic information from text can be used to improve zero-shot classification ( i.e. , classifying never-before-seen objects ) , and show that verb clusters can be used to improve activity recognition in videos . Background train +The computational treatment of lexical rules as covariation in lexical entries was implemented in Prolog by the authors in cooperation with Dieter Martini for the ConTroll system ( Gerdemann and ; Gotz and Meurers 1997a ) . Uses train +Arabic has two kinds of plurals : broken plurals and sound plurals ( ; Chen and Gey , 2002 ) . Background train +Although there are other discussions of the paragraph as a central element of discourse ( e.g. Chafe 1979 , Halliday and Hasan 1976 , , Haberlandt et al. 1980 ) , all of them share a certain limitation in their formal techniques for analyzing paragraph structure . CompareOrContrast train +To prove that our method is effective , we also make a comparison between the performances of our system and Xue and Palmer ( 2005 ) , . CompareOrContrast train +But while Bod 's estimator obtains state-of-the-art results on the WSJ , comparable to and Collins ( 2000 ) , Bonnema et al. 's estimator performs worse and is comparable to Collins ( 1996 ) . Background train +Our plan is to implement a windowed or moving-average version of BLEU as in ( ) . Future train +As rightly pointed out , however , `` Proper nouns and capitalized words are particularly problematic : some capitalized words are proper nouns and some are not . CompareOrContrast train +The recent great advances in speech and language technologies have made it possible to build fully implemented spoken dialogue systems ( Aust et al. , 1995 ; Allen et al. , 1996 ; Zue et al. , 2000 ; ) . Background train +For instance , part of the ACE Phase 2 also adopted a corpus-based approach to SC deterevaluation involves classifying an NP as PERSON , mination that is investigated as part of the mention ORGANIZATION , GPE ( a geographical-political redetection ( MD ) task ( e.g. , ) . CompareOrContrast train +In most cases , the accuracy of parsers degrades when run on out-of-domain data ( Gildea , 2001 ; McClosky et al. , 2006 ; ; Petrov et al. , 2010 ) . Background train +Based on this assumption , the problem of identifying mention heads is a sequential phrase identification problem , and we choose to employ the BILOU-representation as it has advantages over traditional BIO-representation , as shown , e.g. in . Motivation val +TF is given by TFD , t , and it denotes frequency of term t in document D. IDF is given by IDFt = log ( N/dft ) , where N is the number of documents in the collection , and dft is the number of documents containing the term t. ( ) proposed the combination of TF and IDF as weighting schemes , and it has been shown that their product gave better performance . Motivation train +An HPSG grammar consists of lexical entries and ID grammar rules , each of which is described with typed feature structures ( ) . Background train +More recently , Burke , Cahill , et al. ( 2004a ) carried out an evaluation of the automatic annotation algorithm against the publicly available PARC 700 Dependency Bank ( ) , a set of 700 randomly selected sentences from Section 23 which have been parsed , converted to dependency format , and manually corrected and extended by human validators . Background train +In our work , we gather sets of sentences , and assume ( but do not employ ) existing approaches for their organization ( ; Barzilay , Elhadad , and McKeown 2001 ; Barzilay and McKeown 2005 ) . Background train +For example , such schema can serve as a mean to represent translation examples , or find structural correspondences for the purpose of transfer grammar learning ( Menezes & Richardson , 2001 ) , ( Aramaki et al. , 2001 ) , ( Watanabe et al. , 2000 ) , ( Meyers et al. , 2000 ) , ( ) , ( kaji et al. , 1992 ) , and example-base machine translation EBMT3 ( Sato & Nagao , 1990 ) , ( Sato , 1991 ) , ( Richardson et al. , 2001 ) , ( Al-Adhaileh & Tang , 1999 ) . Background train +This was done because purely unsupervised techniques ( e.g. , Baum-Welch [ ] or Brill 's [ Brill 1995b ] ) enable regularities to be induced for word classes which contain many entries , exploiting the fact that individual words that belong to a POS class occur in different ambiguity patterns . CompareOrContrast train +Two formalizations of lexical rules as used by HPSG linguists have been proposed , the meta-level lexical rules ( MLRs ; ; Calcagno and Pollard 1995 ) and the . Background train +In terms of treebank data , the primary training corpus is the Penn Wall Street Journal Treebank ( PTB ) ( ) . Uses train +The extraction of each PICO element relies to a different extent on an annotated corpus of MEDLINE abstracts , created through an effort led by the first author at the National Library of Medicine ( ) . Uses train +Various feature selection techniques have been used in automatic text categorization ; they include document frequency ( DF ) , information gain ( IG ) ( Tzeras and Hartman , 1993 ) , minimum description length principal ( ) , and the X2 statistic . Background val +Usually , the classes are from WordNet ( Miller et al. , 1990 ) , although they can also be inferred from clustering ( ) . Background train +The account sketched in Section 4 was superimposed on an incremental GRE algorithm , partly because incrementality is well established in this area ( Appelt 1985 ; ) . Background train +In most cases , the accuracy of parsers degrades when run on out-of-domain data ( Gildea , 2001 ; ; Blitzer et al. , 2006 ; Petrov et al. , 2010 ) . Background train +Now for some important remarks on efficiency : • Computing ti is an instance of the well-known algebraic path problem ( ; Tar an , 1981a ) . Background train + take an entirely different approach by showing that one can successfully infer held out feature norms from weighted mixtures based on textual similarity . Background train +Over the last decade there has been a lot of interest in developing tutorial dialogue systems that understand student explanations ( Jordan et al. , 2006 ; Graesser et al. , 1999 ; Aleven et al. , 2001 ; Buckley and Wolska , 2007 ; Nielsen et al. , 2008 ; VanLehn et al. , 2007 ) , because high percentages of selfexplanation and student contentful talk are known to be correlated with better learning in humanhuman tutoring ( Chi et al. , 1994 ; Litman et al. , 2009 ; ; Steinhauser et al. , 2007 ) . Background train +Previous work in sentence planning in the natural language generation ( NLG ) community uses hand-written rules to approximate the distribution of linguistic phenomena in a corpus ( see ( ) for a recent example with further references ) . Background train +We assume that every determiner has its own equivalence , which resolves it as a quantifier : sometimes this can be quite a complicated matter , as with any ( ) , which will resolve in different ways depending on its linguistic context , but here we avoid this complexity ' 6 Separate equivalences might also make it easier to encode determiner-specific preferences , such as that of each for wide scope . Background train +Following , such expressions will be called vague descriptions even though , as we shall see , the vagueness of the adjective does not extend to the description as a whole . Uses train +Such approaches have been tried recently in restricted cases ( McCallum et al. , 2000 ; Eisner , 2001b ; ) . Background train +The standard way to handle this problem is to handcraft a finite set of features which provides a sufficient summary of the unbounded history ( Ratnaparkhi , 1999 ; ; Charniak , 2000 ) . CompareOrContrast train +Following , we measure association norm prediction as an average of percentile ranks . Uses val +There is a general consensus among theoretical linguists that the proper representation of verbal argument structure is event structure -- representations grounded in a theory of events that decompose semantic roles in terms of primitive predicates representing concepts such as causality and inchoativity ( Dowty , 1979 ; Jackendoff , 1983 ; b ; Rappaport Hovav and Levin , 1998 ) . Background train +Other molecular biology databases We also included several model organism databases or nomenclature databases in the construction of the dictionary , i.e. , mouse Mouse Genome Database ( MGD ) [ 18 ] , fly FlyBase [ 19 ] , yeast Saccharomyces Genome Database ( SGD ) [ 20 ] , rat -- Rat Genome Database ( RGD ) , worm -- WormBase [ 22 ] , Human Nomenclature Database ( HUGO ) [ 23 ] , Online Mendelian Inheritance in Man ( OMIM ) [ 24 ] , and Enzyme Nomenclature Database ( ECNUM ) [ 25 , 26 ] . Uses train + re-trained the linguistic parsers bilingually based on word alignment . CompareOrContrast train +de URL : http://www.sfs.nphil.uni-tuebingen.de/sfb / b4home.html 1 This is , for example , the case for all proposals working with verbal lexical entries that raise the arguments of a verbal complement ( ) that also use lexical rules such as the Complement Extraction Lexical Rule ( Pollard and Sag 1994 ) or the Complement Cliticization Lexical Rule ( Miller and Sag 1993 ) to operate on those raised elements . Background train +Much of theoretical linguistics can be formulated in a very natural manner as stating correspondences ( translations ) between layers of representation structures ( ) , such as the relation between syntax and semantic . Background val +From an IR view , a lot of specialized research has already been carried out for medical applications , with emphasis on the lexico-semantic aspects of dederivation and decomposition ( Pacak et al. , 1980 ; Norton and Pacak , 1983 ; Wolff , 1984 ; ; Dujols et al. , 1991 ; Baud et al. , 1998 ) . Background train +The following are the various tag sets we use in this article : ( a ) the core POS tag sets CORE44 and the newly introduced CORE12 ; ( b ) CATiB Treebank tag set ( CATIB6 ) ( Habash and Roth 2009 ) and its newly introduced extension of CATIBEX created using simple regular expressions on word form , indicating particular morphemes such as the prefix JI Al + or the suffix v ' + wn ; this tag set is the best-performing tag set for Arabic on predicted values as reported in Section 4 ; ( c ) the PATB full tag set with complete morphological tag ( BW ) ( ) ; and two extensions of the PATB reduced tag set ( PENN POS , a.k.a. RTS , size 24 [ Diab , Hacioglu , and Jurafsky 2004 ] ) , both outperforming it : ( d ) Kulick , Gabbard , and Marcus ( 2006 ) 's tag set ( KULICK ) , size 43 , one of whose most important extensions is the marking of the definite article clitic , and ( e ) Diab and Benajiba 's ( in preparation ) EXTENDED RTS tag set ( ERTS ) , which marks gender , number , and definiteness , size 134 . Uses train +A similar problem is discussed in the psycholinguistics of interpretation ( ) : Interpretation is widely assumed to proceed incrementally , but vague descriptions resist strict incrementality , since an adjective in a vague description can only be fully interpreted when its comparison set is known . CompareOrContrast train +In this paper , we evaluated the role of low-level image features , SURF and GIST , for their compatibility with the multimodal Latent Dirichlet Allocation model of . Uses train +Figure 2 illustrates a DSyntS from a meteorological application , MeteoCogent ( Kittredge and Lavoie , 1998 ) , represented using the standard graphical notation and also the RealPro ASCII notation used internally in the framework ( ) . Extends train +And argues for `` keeping track of counts of arbitrary fragments within parse trees '' , which has indeed been carried out in Collins and Duffy ( 2002 ) who use exactly the same set of ( all ) tree fragments as proposed in Bod ( 1992 ) . Background train + use a tagged parallel corpus to extract translationally equivalent English-Greek clauses on the basis of word occurrence and co-occurrence probabilities . Background train +The key linguistic knowledge sources that we use are morphological analysis and generation of German based on SMOR , a morphological analyzer/generator of German ( Schmid et al. , 2004 ) and the BitPar parser , which is a state-of-the-art parser of German ( ) . Uses train +A common computational treatment of lexical rules adopted , for example , in the ALE system ( ) consists of computing the transitive closure of the base lexical entries under lexical rule application at compile-time . CompareOrContrast train +Manually defined heuristics are used to automatically annotate each tree in the treebank with partially specified HPSG derivation trees : Head/argument/modifier distinctions are made for each node in the tree based on and Collins ( 1997 ) ; Uses train +Authors may choose this right with the No-Deriv option of the Creative Commons licences ( ) . Background train +Mathematical analysis can assess a measure with respect to some formal properties , e.g. whether a measure is a metric ( ) .4 However , mathematical analysis can not tell us whether a measure closely resembles human judgments or whether it performs best when used in a certain application . Background train +The same annotation scheme as in our previous work on anger detection has been applied , see e.g. ( ) . Extends train +We also compute GIST vectors ( ) for every image using LearGIST ( Douze et al. , 2009 ) . Uses train +In modern syntactic theories ( e.g. , lexical-functional grammar [ LFG ] [ Kaplan and Bresnan 1982 ; Bresnan 2001 ; Dalrymple 2001 ] , head-driven phrase structure grammar [ HPSG ] [ Pollard and Sag 1994 ] , tree-adjoining grammar [ TAG ] [ Joshi 1988 ] , and combinatory categorial grammar [ CCG ] [ ] ) , the lexicon is the central repository for much morphological , syntactic , and semantic information . Background train +The analysis of the data we have collected indicates that student satisfaction may be affected if the system rephrases student answers using different words ( for example , using better terminology ) but does n't explicitly explain the reason why different terminology is needed ( ) . Future val +A logic that provides the formal architecture required by Pollard and Sag ( 1994 ) was defined by , 1994 ) . Background val +fields generally follow the pattern of `` introduction '' , `` methods '' , `` results '' , and `` conclusions '' ( SalangerMeyer , 1990 ; ; Or˘asan , 2001 ) . Background val + report that an optimal tag set for parsing Czech consists of a basic POS tag plus a CASE feature ( when applicable ) . Background train +The elimination of redundant nondeterminism is based on Unfold/Fold transformation techniques ( Tamaki and Sato 1984 ) .29 The unfolding transformation is also referred to as partial execution , for example , by . Background train +Over the last decade there has been a lot of interest in developing tutorial dialogue systems that understand student explanations ( Jordan et al. , 2006 ; Graesser et al. , 1999 ; Aleven et al. , 2001 ; Buckley and Wolska , 2007 ; ; VanLehn et al. , 2007 ) , because high percentages of selfexplanation and student contentful talk are known to be correlated with better learning in humanhuman tutoring ( Chi et al. , 1994 ; Litman et al. , 2009 ; Purandare and Litman , 2008 ; Steinhauser et al. , 2007 ) . Background train +9 We do not relate to specific results in their study because it has been brought to our attention that are in the process of rechecking their code for errors , and rerunning their experiments ( personal communication ) . CompareOrContrast val +Inspired by ( ) , we split one phrase type into several subsymbols , which contain category information of current constituent 's parent . Motivation train +The recent great advances in speech and language technologies have made it possible to build fully implemented spoken dialogue systems ( Aust et al. , 1995 ; Allen et al. , 1996 ; ; Walker et al. , 2000 ) . Background train +Moreover , in order to determine whether the performances of the predictive criteria are consistent across different learning models within the same domain , we have performed the study on two parsing models : one based on a context-free variant of tree-adjoining grammars ( Joshi , Levy , and Takahashi 1975 ) , the Probabilistic Lexicalized Tree Insertion Grammar ( PLTIG ) formalism ( Schabes and Waters 1993 ; ) , and Collins 's Model 2 parser ( 1997 ) . Uses train +Most of the early work on automatic f-structure annotation ( e.g. , van Genabith , Way , and Sadler 1999 ; ; Sadler , van Genabith , and Way 2000 ) was applied only to small data sets ( fewer than 200 sentences ) and was largely proof of concept . Background train +We prepare the training data by splitting compounds in two steps , following the technique of . Uses train +Recent work by on the filtering phase of this approach uses linguistic verb classes ( based on Levin [ 1993 ] ) for obtaining more accurate back-off estimates for hypothesis selection . Background train +Numerous previous pseudodisambiguation evaluations only include arguments that occur between 30 and 3000 times ( Erk , 2007 ; Keller and Lapata , 2003 ; ) . CompareOrContrast train +Some efforts have tackled tasks such as automatic image caption generation ( Feng and Lapata , 2010a ; Ordonez et al. , 2011 ) , text illustration ( Joshi et al. , 2006 ) , or automatic location identification of Twitter users ( Eisenstein et al. , 2010 ; Wing and Baldridge , 2011 ; ) . Background train +Default parameters were used , although experimentation with different parameter settings is an important direction for future work ( ; Munson et al. , 2005 ) . Future train +The language chosen for semantic representation is a flat semantics along the line of ( Bos , 1995 ; ; Copestake et al. , 2001 ) . CompareOrContrast train +For example , and Jokinen et al. ( 2008 ) find that machine learning algorithms can be trained to recognise some of the functions of head movements , while Reidsma et al. ( 2009 ) show that there is a dependence between focus of attention and assignment of dialogue act labels . Background train +As a logical postulate it is not very radical ; it is possible within a finitary framework to develop that part of mathematics that is used or has potential applications in natural science , such as mathematical analysis ( cfXXX ) . Background train +In a similar vain to Skut and Brants ( 1998 ) and , the method extends an existing flat shallow-parsing method to handle composite structures . Future val +An interesting aspect of our generative approach is that we model HMM outputs as Gaussian vectors ( log probabilities of observing entire sentences based on our language models ) , as opposed to sequences of terms , as done in ( ) . CompareOrContrast train +• language learning ( Green 1979 ; Mori and Moeser 1983 ; Morgan , Meier , and Newport 1989 ) • monolingual grammar induction ( Juola 1998 ) • grammar optimization ( Juola 1994 ) • insights into universal grammar ( Juola 1998 ) • machine translation ( Juola 1994 , 1997 ; ; Gough , Way , and Hearne 2002 ) Background train +In Charniak ( 1996 ) and , it was observed that treebank grammars ( CFGs extracted from treebanks ) are very large and grow with the size of the treebank . Background val +We apply our system to the latest version of the XTAG English grammar ( The XTAG Research ) , which is a large-scale FB-LTAG grammar . Uses train +Furthermore , the need to answer questions related to patient care at the point of service has been well studied and documented ( Covell , Uman , and Manning 1985 ; Gorman , Ash , and Wykoff 1994 ; , 2005 ) . Background train +Typical examples are Bulgarian ( Simov et al. , 2005 ; Simov and Osenova , 2003 ) , Chinese ( Chen et al. , 2003 ) , Danish ( Kromann , 2003 ) , and Swedish ( ) . Background train +Over the last decade there has been a lot of interest in developing tutorial dialogue systems that understand student explanations ( Jordan et al. , 2006 ; Graesser et al. , 1999 ; Aleven et al. , 2001 ; Buckley and Wolska , 2007 ; Nielsen et al. , 2008 ; VanLehn et al. , 2007 ) , because high percentages of selfexplanation and student contentful talk are known to be correlated with better learning in humanhuman tutoring ( Chi et al. , 1994 ; ; Purandare and Litman , 2008 ; Steinhauser et al. , 2007 ) . Background train +Worst case , calculating the set corresponding with such a property , of the form size ( x ) = maxm , for example , involves sorting the distractors as to their size , which may amount to O ( n2d ) or O ( nd log nd ) calculations ( depending on the sorting algorithm : cfXXX [ ] Chapter 8 ) . Background train +• language learning ( ; Mori and Moeser 1983 ; Morgan , Meier , and Newport 1989 ) • monolingual grammar induction ( Juola 1998 ) • grammar optimization ( Juola 1994 ) • insights into universal grammar ( Juola 1998 ) • machine translation ( Juola 1994 , 1997 ; Veale and Way 1997 ; Gough , Way , and Hearne 2002 ) Background train +Virpioja et al. ( 2007 ) , Badr et al. ( 2008 ) , Luong et al. ( 2010 ) , , and others are primarily concerned with using morpheme segmentation in SMT , which is a useful approach for dealing with issues of word-formation . CompareOrContrast train +While corpus driven efforts along the PARSEVAL lines ( ) are good at giving some measure of a grammar coverage , they are not suitable for finer grained analysis and in particular , for progress evaluation , regression testing and comparative report generation . Background val +Before using the DCA method , we applied a Russian morphological processor ( ) to convert each word in the text to its main form : nominative case singular for nouns and adjectives , infinitive for verbs , etc. . Uses train +The result holds for both the MaltParser ( Nivre 2008 ) and the Easy-First Parser ( ) . Uses train +Also relevant is work on the general problems of dialog-act tagging ( Stolcke et al. , 2000 ) , citation analysis ( ) , and computational rhetorical analysis ( Marcu , 2000 ; Teufel and Moens , 2002 ) . Background train +The language chosen for semantic representation is a flat semantics along the line of ( ; Copestake et al. , 1999 ; Copestake et al. , 2001 ) . CompareOrContrast val +In simple terms , P2P is a technology that takes advantage of the resources and services available at the edge of the Internet ( ) . Background train +A common way to combine different models consists of selecting the model that is most confident regarding its decision ( ) . CompareOrContrast train +Common sense ( as well as the Gricean maxims ; ) suggests that vague descriptions are preferred by speakers over quantitative ones whenever the additional information provided by a quantitative description is irrelevant to the purpose of the communication . Background train +A similar method is included in PATR-II ( Shieber et al. 1983 ) and can be used to encode lexical rules as binary relations in the CUF system ( Dorre and Eisele 1991 ; Done and Dorna 1993b ) or the TFS system ( ; Emele 1994 ) . CompareOrContrast train +For automatically extracting these surface level mappings we will draw on the approach to learning paraphrases from a corpus that is described in . Future val +This description can then be given the standard set-theoretical interpretation of , 1994 ) . ' Background train + helped pave the path for cognitive-linguistic multimodal research , showing that Latent Dirichlet Allocation outperformed Latent Semantic Analysis ( Deerwester et al. , 1990 ) in the prediction of association norms . Background train +A more flexible approach is used by , where users can specify boundary values for attributes like rainfall , specifying , for example , rain counts as moderate above 7 mm/h , as heavy above 20 mm/h , and so on . Background train +Collins and Duffy ( 2002 ) define a kernel over parse trees and apply it to re-ranking the output of a parser , but the resulting feature space is restricted by the need to compute the kernel efficiently , and the results are not as good as Collins ' previous work on re-ranking using a finite set of features ( ) . Background train +The shallow parser used is the SNoW-based CSCL parser ( ; Munoz et al. , 1999 ) . Uses train +In practice , perceptron-type algorithms are often applied in a batch learning scenario , i.e. , the algorithm is applied for K epochs to a training sample of size T and then used for prediction on an unseen test set ( Freund and Schapire , 1999 ; ) . CompareOrContrast train +Table 2 shows the results on identifying all phrases -- chunking in CoNLL2000 ( Tjong Kim ) terminology . Uses train +Unfortunately , as shown in ( ) , with the represetation of sentences that we use , linear classifiers can not discriminate real sentences from sentences sampled from a trigram , which is the model we use as a baseline , so here we resort to a non-linear large-margin classifier ( see section 3 for details ) . Motivation train +Perhaps some variation of multi-level bulleted lists , appropriately integrated with interface elements for expanding and hiding items , might provide physicians a better overview of the information landscape ; see , for example , . Background train + have shown , in the context of base noun identification , that combining sample selection and cotraining can be an effective learning framework for large-scale training . Background train +• The regular TBL , as described in section 2 ; • An improved version of TBL , which makes extensive use of indexes to speed up the rules ' update ; • The FastTBL algorithm ; • The ICA algorithm ( ) . CompareOrContrast train +For example , Jokinen and Ragni ( 2007 ) and find that machine learning algorithms can be trained to recognise some of the functions of head movements , while Reidsma et al. ( 2009 ) show that there is a dependence between focus of attention and assignment of dialogue act labels . Background train +This approach resembles the work by and Hirschman et al. ( 1975 ) on selectional restrictions . CompareOrContrast train +Unless it is desired to intentionally filter these out as being outside of the new domain , one can insert some arbitrarily small probability for these arcs , using , for example , an N-gram back-off model ( ) . Background val +feature Cohen 's k corrected k agreement 73.59 98.74 dial act 84.53 98.87 turn 73.52 99.16 Table 2 : Inter-coder agreement on feedback expression annotation Although researchers do not totally agree on how to measure agreement in various types of annotated data and on how to interpret the resulting figures , see Artstein and Poesio ( 2008 ) , it is usually assumed that Cohen 's kappa figures over 60 are good while those over 75 are excellent ( ) . Background train +To name a few examples , Rohrbach et al. ( 2010 ) and show how semantic information from text can be used to improve zero-shot classification ( i.e. , classifying never-before-seen objects ) , and Motwani and Mooney ( 2012 ) show that verb clusters can be used to improve activity recognition in videos . Background train +Empirical evidence has been brought forward that inflectional and/or derivational stemmers augmented by dictionaries indeed perform substantially better than those without access to such lexical repositories ( Krovetz , 1993 ; ; Tzoukermann et al. , 1997 ) . Background train +The XTAG group ( ) at the University of Pennsylvania is also developing Korean , Chinese , and Hindi grammars . Background train +HOLMES is given the following set of six domainindependent rules , which are similar to the upward monotone rules introduced by ( ) . CompareOrContrast train +The lexicon is used to mediate and map between a language-independent domain model and a language-dependent ontology widely used in NLG , the Upper Model ( ) . Background train +ImageNet is a large-scale and widely used image database , built on top of WordNet , which maps words into groups of images , called synsets ( ) . Uses train +based parsing algorithms with an arc-factored parameterization ( ) . Uses train +Two applications that , like help-desk , deal with question -- answer pairs are : summarization of e-mail threads ( Dalli , Xia , and Wilks 2004 ; Shrestha and McKeown 2004 ) , and answer extraction in FAQs ( Frequently Asked Questions ) ( ; CompareOrContrast val +The retrieval process relies on the vector space model ( ) , with the cosine measure expressing the similarity between a query and a document . Uses val +For right-branching structures , the leftcorner ancestor is the parent , conditioning on which has been found to be beneficial ( ) , as has conditioning on the left-corner child ( Roark and Johnson , 1999 ) . Background train +A few others incorporate various measures of inter-document similarity between the texts to be labeled ( Agarwal and Bhattacharyya , 2005 ; ; Goldberg and Zhu , 2006 ) . Background train +A few others incorporate various measures of inter-document similarity between the texts to be labeled ( ; Pang and Lee , 2005 ; Goldberg and Zhu , 2006 ) . Background train +This is similar to `` one sense per collocation '' idea of . CompareOrContrast train +Against the background of a growing interest in multilingual NLP , multilingual anaphora / coreference resolution has gained considerable momentum in recent years ( Aone and McKee 1993 ; Azzam , Humphreys , and Gaizauskas 1998 ; Harabagiu and Maiorano 2000 ; Mitkov and Barbu 2000 ; ; Mitkov and Stys 1997 ; Mitkov , Belguith , and Stys 1998 ) . Background train +This system has been successfully tested with the development of plug-ins supporting instant messaging , distributed video encoding ( Hughes and Walkerdine , 2005 ) , distributed virtual worlds ( Hughes et al. , 2005 ) and digital library management ( ) . Background train +be found in figure 2 , which is similar with that in . CompareOrContrast train +It is inspired by the system described in . Motivation train +• A user study was performed , but it was either very small compared to the corpus ( Carmel , Shtalhaim , and Soffer 2000 ; Jijkoun and de Rijke 2005 ) , or the corpus itself was significantly smaller than ours ( Feng et al. 2006 ; ) . CompareOrContrast train +Although a number of methods for query-dependent text summarization are beginning to be developed and evaluated in a variety of realistic settings ( ) , we again propose the use of vector space methods from IR , which can be easily extended to the summarization task ( Salton et al. , 1994 ) : CompareOrContrast train +Riehemann 1993 ; ; Frank 1994 ; Opalka 1995 ; Sanfilippo 1995 ) . CompareOrContrast train +When we run our classifiers on resource-tight environments such as cell-phones , we can use a random feature mixing technique ( ) or a memory-efficient trie implementation based on a succinct data structure ( Jacobson , 1989 ; Delpratt et al. , 2006 ) to reduce required memory usage . Future train +These translations gave rise to a number of automatically constructed linguistic resources : ( 1 ) the original ( source , target ) phrasal translation pairs , ( 2 ) the marker lexicon , ( 3 ) the gen11 Thanks are due to one of the anonymous reviewers for pointing out that our wEBMT system , seeded with input from multiple translation systems , with a postvalidation process via the Web ( amounting to an n-gram target language model ) , in effect forms a multiengine MT system as described by Frederking and Nirenburg ( 1994 ) , , and Hogan and Frederking ( 1998 ) . CompareOrContrast train +A variety of statistical methods were proposed over the recent years for learning to produce a full parse of free-text sentences ( e.g. , Bod ( 1992 ) , , Collins ( 1997 ) , Ratnaparkhi ( 1997 ) , and Sekine ( 1998 ) ) . Background train +Discriminative approaches ( especially SVMs ) have been shown to be very effective for many supervised classification tasks ; see , for example , ( ; Ng and Jordan , 2001 ) . Background train +This is the approach taken by IBM Models 4 + ( Brown et al. 1993b ; ) , and more recently by the LEAF model ( Fraser and Marcu 2007 ) . CompareOrContrast train +A few others incorporate various measures of inter-document similarity between the texts to be labeled ( Agarwal and Bhattacharyya , 2005 ; Pang and Lee , 2005 ; ) . Background val +11 proposes to unify these two steps by including an update operator in the Background train +For instance , when building translation units in EBMT approaches ( Richardson et al. , 2001 ) , ( Aramaki , 2001 ) , ( AlAdhaileh & Tang , 1999 ) , ( Sato & Nagao , 1990 ) , ( Sato , 1991 ) , ( ) , etc. , where S-SSTC can be used to represent the entries of the BKB or when S-SSTC used as an annotation schema to find the translation correspondences ( lexical and structural correspondences ) for transferrules ' extraction from parallel parsed corpus ( Menezes & Richardson , 2001 ) , ( Watanabe et al. , Background train + introduce a new method of multimodal integration based on Canonical Correlation Analysis , and performs a systematic comparison between their CCA-based model and others on association norm prediction , held out feature prediction , and word similarity . Background train +More details on how the structural divergences described in ( ) can be accounted for using our formalism can be found in ( Nasr et al. , 1998 ) . Background train +At present , the system takes into consideration the number of incorrect answers received in response to the current question and the number of uninterpretable answers .1 In addition to a remediation policy , the tutorial planner implements an error recovery policy ( ) . Uses train +More recently , show that visual attribute classifiers , which have been immensely successful in object recognition ( Farhadi et al. , 2009 ) , act as excellent substitutes for feature Background train +We found the same number using our previous approach ( ) , which is roughly equivalent to our core module . Extends train +Similar things hold for multifaceted properties like intelligence ( ) . Background train +We have shown elsewhere ( ; Zadrozny 1987a , 1987b ) that natural language programs , such as on-line grammars and dictionaries , can be used as referential levels for commonsense reasoning -- for example , to disambiguate PP attachment . Extends val +( Och and Ney , 2002 ; Blunsom et al. , 2008 ) used maximum likelihood estimation to learn weights for MT. ( Och , 2003 ; Moore and Quirk , 2008 ; ; Galley and Quirk , 2011 ) employed an evaluation metric as a loss function and directly optimized it . CompareOrContrast train +There has been some controversy , at least for simple stemmers ( ; Porter , 1980 ) , about the effectiveness of morphological analysis for document retrieval ( Harman , 1991 ; Krovetz , 1993 ; Hull , 1996 ) . Background train + reported an intra-subject correlation of r = .85 for 15 subjects judging the similarity of a subset ( 36 ) of the original 65 word pairs . CompareOrContrast train +Withindocument coreference resolution has been applied to produce summaries of text surrounding occurrences of the name ( ; Gooi and Allan , 2004 ) . Background train +More recently , have proposed the exploitation of TMs at a subsentential level , while Carl , Way , and Sch ¨ aler ( 2002 ) and Sch ¨ aler , Way , and Carl ( 2003 , pages 108 -- 109 ) describe how phrasal lexicons might come to occupy a central place in a future hybrid integrated translation environment . Background train +With a minimal set of features and a small number of lexical entries , Niyogi ( 2001 ) has successfully modeled many of the argument alternations described by Levin ( 1993 ) using a style analysis . Background train +For some adjectives , including the ones that called evaluative ( as opposed to dimensional ) , this is clearly inadequate . Background train +WIT has been implemented in Common Lisp and C on UNIX , and we have built several experimental and demonstration dialogue systems using it , including a meeting room reservation system ( b ) , a video-recording programming system , a schedule management system ( Nakano et al. , 1999a ) , and a weather infomiation system ( Dohsaka et al. , 2000 ) . Extends train +The more conservative approach is to try to integrate existing statistical disambiguation schemes for QLFs , either individually or in a `` packed '' structure ( ) , with the resolution process as described here . Future val +They proved to be useful in a number of NLP applications such as natural language generation ( Iordanskaja et al. , 1991 ) , multidocument summarization ( ) , automatic evaluation of MT ( Denkowski and Lavie , 2010 ) , and TE ( Dinu and Wang , 2009 ) . Motivation train +While many approaches have addressed this problem , our work is most closely related to that of ( Raina et al. , 2005 ; ; Tatu and Moldovan , 2006 ; Braz et al. , 2005 ) , which convert the inputs into logical forms and then attempt to ` prove ' H from T plus a set of axioms . CompareOrContrast train +While this is simply irrelevant for general-purpose morphological analyzers , dealing with such phenomena is crucial for any attempt to cope adequately with medical free-texts in an IR setting ( ) . Background train +The EDR has close ties to the named entity recognition ( NER ) and coreference resolution tasks , which have been the focus of several recent investigations ( Bikel et al. , 1997 ; Miller et al. , 1998 ; Borthwick , 1999 ; Mikheev et al. , 1999 ; Soon et al. , 2001 ; Ng and Cardie , 2002 ; ) , and have been at the center of evaluations such as : MUC-6 , MUC-7 , and the CoNLL '02 and CoNLL '03 shared tasks . Background train +In English , where the base form is morphologically simpler than the other two , this rule could be argued to follow from Gricean principles ( ) . Background val +More generally , distributional clustering techniques ( Sch ¨ utze , 1992 ; ) could be applied to extract semantic classes from the corpus itself . Future train +The use of the web as a corpus for teaching and research on language has been proposed a number of times ( Kilgarriff , 2001 ; Robb , 2003 ; ; Fletcher , 2001 , 2004b ) and received a special issue of the journal Computational Linguistics ( Kilgarriff and Grefenstette , 2003 ) . Background train + replicates the work of Fung and McKeown with different language pairs using the simpler metric of Levenshtein distance . Background val +• Only an automatic evaluation was performed , which relied on having model responses ( ; Berger et al. 2000 ) . CompareOrContrast train +Typed feature grammars can be used as the basis for implementations of Head-driven Phrase Structure Grammar ( HPSG ; ) as discussed in ( Gotz and Meurers , 1997a ) and ( Meurers and Minnen , 1997 ) . Background train +In a ) we identified several systems that resemble ours in that they provide answers to queries . Background train +The problem of handling ill-formed input has been studied by Carbonell and Hayes ( 1983 ) , Granger ( 1983 ) , , Kwasny and Sondheimer ( 1981 ) , Riesbeck and Schank ( 1976 ) , Thompson ( 1980 ) , Weischedel and Black ( 1980 ) , and Weischedel and Sondheimer ( 1983 ) . CompareOrContrast val +A variety of such lists for many languages are already available ( e.g. , ) . Background train +In principle , this might be done by providing the generator with vague input -- in which case no special algorithms are needed -- but suitably contextualized vague input is often not available ( ) . Motivation train +Liu et al. ( 2005 ) , Meral et al. ( 2007 ) , Murphy ( 2001 ) , Murphy and Vogel ( 2007 ) and a ) all belong to the syntactic transformation category . Background train +We induced a two-class word-to-word model of translational equivalence from 13 million words of the Canadian Hansards , aligned using the method in ( ) . Uses train + observes that accomplishments differ from achievements only in terms of event duration , which is often a question of granularity . Background train +[ The current system should be distinguished from an earlier voice system ( VNLC , ) , which had no expectation and which handled discrete speech where a 300 millisecond pause must follow each word . ] CompareOrContrast train +Thus , over the past few years , along with advances in the use of learning and statistical methods for acquisition of full parsers ( Collins , 1997 ; Charniak , 1997a ; Charniak , 1997b ; Ratnaparkhi , 1997 ) , significant progress has been made on the use of statistical learning methods to recognize shallow parsing patterns syntactic phrases or words that participate in a syntactic relationship ( Church , 1988 ; ; Argamon et al. , 1998 ; Cardie and Pierce , 1998 ; Munoz et al. , 1999 ; Punyakanok and Roth , 2001 ; Buchholz et al. , 1999 ; Tjong Kim Sang and Buchholz , 2000 ) . Background train +He was a grammarian who analysed Sanskrit ( ) . Background train +After much exploration , discovered that it was not practical to annotate PICO entities at the phrase level due to significant unresolvable disagreement and interannotator reliability issues . Background train +The necessity of this kind of merging of arguments has been recognized before : call it abductive unification/matching , Hobbs ( 1978 , 1979 ) refers to such operations using the terms knitting or petty conversational implicature . Background train +The OntoNotes-5 .0 dataset , which is released for the CoNLL-2012 Shared Task ( ) , contains 3,145 annotated documents . Uses train +and , as described below . CompareOrContrast train +We collect substring rationales for a sentiment classification task ( ) and use them to obtain significant accuracy improvements for each annotator . Uses train +We apply two different priming experiments namely , the cross modal priming and masked priming experiment discussed in ( ; Rastle et al. , 2000 ; Marslen-Wilson et al. , 1994 ; Marslen-Wilson et al. , 2008 ) for Bangla morphologically complex words . Uses val +Recently , several alternative , often quite sophisticated approaches to collective classification have been proposed ( Neville and Jensen , 2000 ; Lafferty et al. , 2001 ; Getoor et al. , 2002 ; Taskar et al. , 2002 ; Taskar et al. , 2003 ; ; McCallum and Wellner , 2004 ) . Background train +More sophisticated approaches have been proposed ( Hillard et al. , 2003 ) , including an extension that , in an interesting reversal of our problem , makes use of sentimentpolarity indicators within speech segments ( ) . Background train +The expectation parser uses an ATN-like representation for its grammar ( ) . Uses train +• Only an automatic evaluation was performed , which relied on having model responses ( Berger and Mittal 2000 ; ) . CompareOrContrast train +The error rate on sentence boundaries in the Brown corpus was not significantly worse than the lowest quoted before ( : 0.28 % vs. 0.20 % error rate ) . CompareOrContrast train +We follow our previous work ( b ) and restrict bridging to non-coreferential cases . Extends train +In our case , the clustering is performed by the program Snob , which implements mixture modeling combined with model selection based on the Minimum Message Length ( MML ) criterion ( ; Wallace 2005 ) . Uses train +This is in line with our previous findings from ( ) that candidates with higher power attempt to shift topics less often than others when responding to moderators . CompareOrContrast train +As has been previously observed and exploited in the NLP literature ( ; Agarwal and Bhattacharyya , 2005 ; Barzilay and Lapata , 2005 ) , the above optimization function , unlike many others that have been proposed for graph or set partitioning , can be solved exactly in an provably efficient manner via methods for finding minimum cuts in graphs . CompareOrContrast train +It is analogous to the step in other translation model induction algorithms that sets all probabilities below a certain threshold to negligible values ( Brown et al. , 1990 ; ; Chen , 1996 ) . Background train +It provides a fine grained NE recognition covering 100 different NE types ( ) . Uses train +Only a few such corpora exist , including the Hansard English-French corpus and the HKUST EnglishChinese corpus ( ) . Background train +Recently , several alternative , often quite sophisticated approaches to collective classification have been proposed ( Neville and Jensen , 2000 ; Lafferty et al. , 2001 ; Getoor et al. , 2002 ; ; Taskar et al. , 2003 ; Taskar et al. , 2004 ; McCallum and Wellner , 2004 ) . Background train +In FAQs , employed a sentence retrieval approach based on a language model where the entire response to an FAQ is considered a sentence , and the questions and answers are embedded in an FAQ document . Background train +The statistical significance test is performed by the re-sampling approach ( ) . Uses val +With a minimal set of features and a small number of lexical entries , Niyogi ( 2001 ) has successfully modeled many of the argument alternations described by using a Hale and Keyser ( 1993 ) style analysis . Background train +Although evaluated on different data sets , this result is consistent with results from previous work ( Gatt and Belz , 2008 ; ) . CompareOrContrast train +We used the revised experimental setup ( ) , based on discrete relatedness scores and presentation of word pairs in isolation , that is scalable to the higher number of pairs . Uses val + ; Bobrow , 1978 ) consult relatively small lexicons , typically generated by hand . CompareOrContrast train +Typed feature grammars can be used as the basis for implementations of Head-driven Phrase Structure Grammar ( ) .3 ( Meurers and Minnen , 1997 ) propose a compilation of lexical rules into TIT definite clauses Background train +In the United States , for example , governmental bodies are providing and soliciting political documents via the Internet , with lofty goals in mind : electronic rulemaking ( eRulemaking ) initiatives involving the `` electronic collection , distribution , synthesis , and analysis of public commentary in the regulatory rulemaking process '' , may `` [ alter ] the citizen-government relationship '' ( ) . Background train +This will become even more interesting when mappings of our synonym identifiers to a large medical thesaurus ( MeSH , ( ) ) are incorporated into our system . Future train +( Och and Ney , 2002 ; Blunsom et al. , 2008 ) used maximum likelihood estimation to learn weights for MT. ( Och , 2003 ; ; Zhao and Chen , 2009 ; Galley and Quirk , 2011 ) employed an evaluation metric as a loss function and directly optimized it . CompareOrContrast train +raw length value as a feature , we follow our previous work ( Rubino et al. , 2013 ; ) and create multiple features for length using a decision tree ( J48 ) . Extends train +Most DOP models , such as in Bod ( 1993 ) , Goodman ( 1996 ) , Bonnema et al. ( 1997 ) , Sima'an ( 2000 ) and , use a likelihood criterion in defining the best parse tree : they take ( some notion of ) the most likely ( i.e. most probable ) tree as a candidate for the best tree of a sentence . Background train +In some systems such dependencies are learned from labeled examples ( ) . Background train +CD for this type of descriptions along the lines of Section 4 is not difficult once relational descriptions are integrated with a standard GRE algorithm ( , Section 8.6.2 ) : Suppose an initial description is generated describing the set of all those dogs that are in sheds over a given size ( say , size 5 ) ; if this description happens to distinguish an individual dog then this legitimizes the use of the noun phrase the dog in the large shed . Background train +Michiels proposed rules for doing this for infinitive complement codes ; however there seems to be no principled reason not to extend this approach to computing the underlying relations in other types of VP as well as in cases of NP , AP and PP predication ( see , for further discussion ) . Background val +While wikis have spread from a detailed design ( ) , unfortunately blogs have not been designed under a model . Background train +In our experiment , we annotated a high number of pairs similar in size to the test sets by Finkelstein ( 2002 ) and . CompareOrContrast val +notation of is more sophisticated , and may be considered another possibility . CompareOrContrast train +The recent great advances in speech and language technologies have made it possible to build fully implemented spoken dialogue systems ( Aust et al. , 1995 ; ; Zue et al. , 2000 ; Walker et al. , 2000 ) . Background val +Most approaches rely on VerbNet ( ) and FrameNet ( Baker et al. , 1998 ) to provide associations between verbs and semantic roles , that are then mapped onto the current instance , as shown by the systems competing in semantic role labelling competitions ( Carreras and Marquez , 2004 ; Carreras and Marquez , 2005 ) and also ( Gildea and Jurafsky , 2002 ; Pradhan et al. , 2005 ; Shi and Mihalcea , 2005 ) . Background train + aimed to embed information by exploiting the linguistic phenomenon of presupposition , with the idea that some presuppositional information can be removed without changing the meaning of a sentence . Background train +We first identified the most informative unigrams and bigrams using the information gain measure ( Yang and Pedersen 1997 ) , and then selected only the positive outcome predictors using odds ratio ( ) . Uses train +Perhaps some variation of multi-level bulleted lists , appropriately integrated with interface elements for expanding and hiding items , might provide physicians a better overview of the information landscape ; see , for example , . Future train +Most DOP models , such as in Bod ( 1993 ) , , Bonnema et al. ( 1997 ) , Sima'an ( 2000 ) and Collins & Duffy ( 2002 ) , use a likelihood criterion in defining the best parse tree : they take ( some notion of ) the most likely ( i.e. most probable ) tree as a candidate for the best tree of a sentence . Background train +Some previous works ( ; Zhao et al. , 2009 ; Kouylekov et al. , 2009 ) indicate , as main limitations of the mentioned resources , their limited coverage , their low precision , and the fact that they are mostly suitable to capture relations mainly between single words . Background train +Nugget F-score has been employed as a metric in the TREC question-answering track since 2003 , to evaluate so-called definition and `` other '' questions ( ) . Background train +Manually defined heuristics are used to automatically annotate each tree in the treebank with partially specified HPSG derivation trees : Head/argument/modifier distinctions are made for each node in the tree based on Magerman ( 1994 ) and ; Uses train +Regarding future work , there are many research line that may be followed : i ) Capturing more features by employing external knowledge such as ontological , lexical resource or WordNet-based features ( Basili et al. , 2005a ; Basili et al. , 2005b ; Bloehdorn et al. , 2006 ; Bloehdorn and Moschitti , 2007 ) or shallow semantic trees , ( Giuglea and Moschitti , 2004 ; Giuglea and Moschitti , 2006 ; ; Moschitti et al. , 2007 ; Moschitti , 2008 ; Moschitti et al. , 2008 ) . Future train +In addition , there has been much work on the application of linguistic and semantic knowledge to information retrieval ; see a ) for a brief overview . Background train +However , claims that the log-likelihood chisquared statistic ( G2 ) is more appropriate for corpus-based NLP . Motivation train +In a log-linear parameterization , for example , a prior that penalizes feature strengths far from 1 can be used to do feature selection and avoid overfitting ( ) . Uses val +This alignment is done on the basis of both length ( Gale and Church ) and a notion of cognateness ( Simard [ 16 ] ) . Uses train +Since earlier versions of the SNoW based CSCL were used only to identify single phrases ( Punyakanok and Roth , 2001 ; ) and never to identify a collection of several phrases at the same time , as we do here , we also trained and tested it under the exact conditions of CoNLL-2000 ( Tjong Kim Sang and Buchholz , 2000 ) to compare it to other shallow parsers . Extends train +A more detailed discussion of the various available Arabic tag sets can be found in . Background train +The high Dirichlet priors are chosen to prevent sparsity in topic distributions , while the other parameters are selected as the best from . Uses val +The resulting speech understanding system is called the Voice Natural Language Computer with Expectation ( VNLCE , ) . Background train +In our previous papers ( ; Zhang , Blackwood , and Clark 2012 ) , we applied a set of beams to this structure , which makes it similar to the data structure used for phrase-based MT decoding ( Koehn 2010 ) . CompareOrContrast train +A number of speech understanding systems have been developed during the past fifteen years ( , Dixon and Martin 1979 , Erman et al. 1980 , Haton and Pierrel 1976 , Lea 1980 , Lowerre and Reddy 1980 , Medress 1980 , Reddy 1976 , Walker 1978 , and Wolf and Woods 1980 ) . CompareOrContrast train +Another line of research approaches grounded language knowledge by augmenting distributional approaches of word meaning with perceptual information ( Andrews et al. , 2009 ; Steyvers , 2010 ; Feng and Lapata , 2010b ; Bruni et al. , 2011 ; ; Johns and Jones , 2012 ; Bruni et al. , 2012a ; Bruni et al. , 2012b ; Silberer et al. , 2013 ) . Background train +In particular , since we treat each individual speech within a debate as a single `` document '' , we are considering a version of document-level sentiment-polarity classification , namely , automatically distinguishing between positive and negative documents ( ; Pang et al. , 2002 ; Turney , 2002 ; Dave et al. , 2003 ) . Background train +A number of alignment techniques have been proposed , varying from statistical methods ( Brown et al. , 1991 ; Gale and Church , 1991 ) to lexical methods ( Kay and Roscheisen , 1993 ; ) . Background train +( contains further description and discussion of LDOCE . ) Background train + , 1997 ) assumes that words ending in - ed are verbs . CompareOrContrast train +For the development of these lists we used a collection of texts of about 300,000 words derived from the New York Times ( NYT ) corpus that was supplied as training data for the 7th Message Understanding Conference ( MUC-7 ) ( ) . Uses train +We would also provide a user facility for choosing the right licence for every lexia , following the model of Creative Commons licences ( ) . Uses train +The language grounding problem has received significant attention in recent years , owed in part to the wide availability of data sets ( e.g. Flickr , Von Ahn ( 2006 ) ) , computing power , improved computer vision models ( Oliva and Torralba , 2001 ; ; Farhadi et al. , 2009 ; Parikh and Grauman , 2011 ) and neurological evidence of ties between the language , perceptual and motor systems in the brain ( Pulverm ¨ uller et al. , 2005 ; Tettamanti et al. , 2005 ; Aziz-Zadeh et al. , 2006 ) . Background train +There has been some controversy , at least for simple stemmers ( Lovins , 1968 ; Porter , 1980 ) , about the effectiveness of morphological analysis for document retrieval ( Harman , 1991 ; ; Hull , 1996 ) . Background train +We have yet to import such a constraint into our model , but we plan to do so in the near future using the weighted majority algorithm ( ) . Future train +Generally speaking , we find that the personal public diary metaphor behind blogs ( ) may bring to an unsatisfactory representation of the context . Background train +Most probabilistic translation model reestimation algorithms published to date are variations on the theme proposed by b ) . Background train +A number of proposals in the 1990s deliberately limited the extent to which they relied on domain and/or linguistic knowledge and reported promising results in knowledge-poor operational environments ( Dagan and Itai 1990 , 1991 ; Lappin and Leass 1994 ; ; Kennedy and Boguraev 1996 ; Williams , Harvey , and Preston 1996 ; Baldwin 1997 ; Mitkov 1996 , 1998b ) . Background train +The significance testing is performed by paired bootstrap re-sampling ( ) . Uses train +The system is implemented based on ( ) and ( Marcu et al. 2006 ) . Uses train +IGEN uses standard chart generation techniques ( ) in its base generator to efficiently produce generation candidates . Background train +Position , subcat frame , phrase type , first word , last word , subcat frame + , predicate , path , head word and its POS , predicate + head word , predicate + phrase type , path to BA and BEI , verb class 3 , verb class + head word , verb class + phrase type , from . Uses train +The psycholinguistic studies of Martin ( 1970 ) , Allen ( 1975 ) , Hillinger et al. ( 1976 ) , Grosjean et al. ( 1979 ) , Dommergues and Grosjean ( 1983 ) , and , responding to the idea of readjusted syntax as the source of prosodic phrasing , show that grammatical structure , even if readjusted , is not in itself a reliable predictor of prosodic phrasing : mismatches between syntax and prosody occur often and systematically , and can be related to specific nonsyntactic factors such as length and word frequency . Background train +A number of studies ( e.g. , Hildebrandt , Katz , and Lin 2004 ) have pointed out shortcomings of the original nugget scoring model , although a number of these issues have been recently addressed ( a , 2006b ) . Background train + reported that the translation lexicon that our model induced from this tiny bitext accounted for 30 % of the word types with precision between 84 % and 90 % . Background train + describe a simple tool which uses fine-grained rules to identify the arguments of verb occurrences in the Penn-II Treebank . Background train +The contextual interpreter then uses a reference resolution approach similar to , and an ontology mapping mechanism ( Dzikovska et al. , 2008a ) to produce a domain-specific semantic representation of the student 's output . CompareOrContrast train + maintains a survey of this area . Background train +The principle of maximum entropy states that when one searches among probability distributions that model the observed data ( evidence ) , the preferred one is the one that maximizes the entropy ( a measure of the uncertainty of the model ) ( ) . Background train +In the context of word alignment , use a state-duration HMM in order to model word-to-phrase translations . Background val +With respect to the focus on function words , our reordering model is closely related to the UALIGN system ( ) . CompareOrContrast train +We could also introduce new variables , e.g. , nonterminal refinements ( ) , or secondary links Mid ( not constrained by TREE/PTREE ) that augment the parse with representations of control , binding , etc. ( Sleator and Temperley , 1993 ; Buch-Kromann , 2006 ) . Future train +Selectional Preferences have also been a recent focus of researchers investigating the learning of paraphrases and inference rules ( ; Roberto et al. , 2007 ) . Background train +Previously LDA has been successfully used to infer unsupervised joint topic distributions over words and feature norms together ( Andrews et al. , 2009 ; ) . Background train +Something like this approach is in fact used in some systems ( e.g. , Elhadad and Robin 1992 ; PenMan 1989 ; a ) . Background train +The emphasis on narrativity takes into account the use of blogs as public diaries on the web , that is still the main current interpretation of this literary genre , or metagenre ( ) . Background train +For example , it would be helpful to consider strong correspondence between certain English and Chinese words , as in ( ) . Future train +The system utilizes several large size biological databases including three NCBI databases ( GenPept [ 11 ] , RefSeq [ 12 ] , and Entrez GENE [ 13 ] ) , PSD database from Protein Information Resources ( PIR ) , and Uses train +We take some core ideas from our previous work on mining script information ( ) . Extends train +While IA is generally thought to be consistent with findings on human language production ( Hermann and Deutsch 1976 ; Levelt 1989 ; ; Sonnenschein 1982 ) , the hypothesis that incrementality is a good model of human GRE seems unfalsifiable until a preference order is specified for the properties on which it operates . Background train +Typical examples are Bulgarian ( ; Simov and Osenova , 2003 ) , Chinese ( Chen et al. , 2003 ) , Danish ( Kromann , 2003 ) , and Swedish ( Nilsson et al. , 2005 ) . Background val +• A user study was performed , but it was either very small compared to the corpus ( Carmel , Shtalhaim , and Soffer 2000 ; Jijkoun and de Rijke 2005 ) , or the corpus itself was significantly smaller than ours ( ; Leuski et al. 2006 ) . CompareOrContrast train +Our work on the prosodic phrase status of clause final prepositional phrases , which we discuss below , suggests the existence of a discourse-neutral phrasing that depends on syntactic constituency mediated by string adjacency and length of a potential prosodic phrase .3 Such phrasing provides us with a typical phrasing pattern analogous to the typical phrasal stress patterns examined in , which `` are often overwhelmed by the chiaroscuro of highlight and background in discourse , but retain the status of null-hypothesis patterns that emerge when Computational Linguistics Volume 16 , Number 3 , September 1990 157 J. Bachenko and E. Fitzpatrick Discourse-Neutral Prosodic Phrasing in English there is no good reason to take some other option '' ( p. 251 ) . CompareOrContrast train +Another line of research approaches grounded language knowledge by augmenting distributional approaches of word meaning with perceptual information ( Andrews et al. , 2009 ; ; Feng and Lapata , 2010b ; Bruni et al. , 2011 ; Silberer and Lapata , 2012 ; Johns and Jones , 2012 ; Bruni et al. , 2012a ; Bruni et al. , 2012b ; Silberer et al. , 2013 ) . Background val +When objects are compared in terms of several dimensions , these dimensions can be weighed in different ways ( e.g. , ) . Background train +Most DOP models , such as in Bod ( 1993 ) , Goodman ( 1996 ) , Bonnema et al. ( 1997 ) , and Collins & Duffy ( 2002 ) , use a likelihood criterion in defining the best parse tree : they take ( some notion of ) the most likely ( i.e. most probable ) tree as a candidate for the best tree of a sentence . Background train +While IA is generally thought to be consistent with findings on human language production ( Hermann and Deutsch 1976 ; ; Pechmann 1989 ; Sonnenschein 1982 ) , the hypothesis that incrementality is a good model of human GRE seems unfalsifiable until a preference order is specified for the properties on which it operates . Background train +Other molecular biology databases We also included several model organism databases or nomenclature databases in the construction of the dictionary , i.e. , mouse Mouse Genome Database ( MGD ) [ 18 ] , fly FlyBase , yeast Saccharomyces Genome Database ( SGD ) [ 20 ] , rat -- Rat Genome Database ( RGD ) [ 21 ] , worm -- WormBase [ 22 ] , Human Nomenclature Database ( HUGO ) [ 23 ] , Online Mendelian Inheritance in Man ( OMIM ) [ 24 ] , and Enzyme Nomenclature Database ( ECNUM ) [ 25 , 26 ] . Uses train +The RenTAL system automatically converts an FB-LTAG grammar into a strongly equivalent HPSG-style grammar ( ) . Background train +The features can be easily obtained by modifying the TAT extraction algorithm described in ( ) . Extends train + present preliminary work on the automatic extraction of subcategorization frames for Bulgarian from the BulTreeBank ( Simov , Popova , and Osenova 2002 ) . Background train +The use of the web as a corpus for teaching and research on language has been proposed a number of times ( Kilgarriff , 2001 ; Robb , 2003 ; Rundell , 2000 ; Fletcher , 2001 , 2004b ) and received a special issue of the journal Computational Linguistics ( ) . Background val +ASARES is based on a Machine Learning technique , Inductive Logic Programming ( ILP ) ( ) , which infers general morpho-syntactic patterns from a set of examples ( this set is noted E + hereafter ) and counter-examples ( E − ) of the elements one Background val + built a corpus by iteratively searching Google for a small set of seed terms . Background train +Expanding on a suggestion of , we classify verbs as Subject Equi , Object Equi , Subject Raising or Object Raising for each sense which has a predicate complement code associated with it . Extends train +This Principle of Finitism is also assumed by Johnson-Laird ( 1983 ) , , Kamp ( 1981 ) , and implicitly or explicitly by almost all researchers in computational linguistics . CompareOrContrast train +For more information on CATiB , see and Habash , Faraj , and Roth ( 2009 ) . Background train +With the exception of ( Fung , 1995b ) , previous methods for automatically constructing statistical translation models begin by looking at word cooccurrence frequencies in bitexts ( ; Kumano & Hirakawa , 1994 ; Fung , 1995a ; Melamed , 1995 ) . Background val +Experiments ( Section 5 ) show that forestbased extraction improves BLEU score by over 1 point on a state-of-the-art tree-to-string system ( Liu et al. , 2006 ; ) , which is also 0.5 points better than ( and twice as fast as ) extracting on 30-best parses . Extends train +Our approach to extract and classify social events builds on our previous work ( ) , which in turn builds on work from the relation extraction community ( Nguyen et al. , 2009 ) . Extends train +Previous sentiment-analysis work in different domains has considered inter-document similarity ( Agarwal and Bhattacharyya , 2005 ; ; Goldberg and Zhu , 2006 ) or explicit Background val +For example , speech repairs , particle omission , and fillers can be dealt with in the framework of unification grammar ( ; Nakano and Shimazu , 1999 ) . Future train + employed a Bayesian method to learn discontinuous SCFG rules . CompareOrContrast train +Table 5 shows our mapping from publication type and MeSH headings to evidence grades based on principles defined in the Strength of Recommendations Taxonomy ( ) . Uses train +We use the same splits as . Uses train +These results are slightly worse than those obtained in previous studies using the same annotation scheme ( ) , but are still sat - CompareOrContrast train +In , this flattening process is not part of the grammar . Background train +MEDLINE , the authoritative repository of abstracts from the medical and biomedical primary literature maintained by the National Library of Medicine , provides the clinically relevant sources for answering physicians ' questions , and is commonly used in that capacity ( ; De Groote and Dorsch 2003 ) . Background train +There is a rich literature on organization and lexical access of morphologically complex words where experiments have been conducted mainly for derivational suffixed words of English , Hebrew , Italian , French , Dutch , and few other languages ( Marslen-Wilson et al. , 2008 ; ; Grainger , et al. , 1991 ; Drews and Zwitserlood , 1995 ) . Background train +He lists , classifies , and discusses various types of inference , by which he means , generally , `` the linguistic-logical notions of consequent and presupposition '' :112 ) have collected convincing evidence of the existence of language chunks -- real structures , not just orthographic conventions -- that are smaller than a discourse , larger than a sentence , generally composed of sentences , and recursive in nature ( like sentences ) . Background val +For this research , we used a coreference resolution system ( ( ) ) that implements different sets of heuristics corresponding to various forms of coreference . Uses train +The work that is most similar to ours is that of , who introduced the Constraint Driven Learning algorithm ( CODL ) . CompareOrContrast train +For example , in ordinary HMM training , xi = E * and represents a completely hidden state sequence ( cfXXX , who allows any regular set ) , while yi is a single string representing a completely observed emission sequence .11 What to optimize ? Background val + tried to solve the inflection prediction problem by simply building an SMT system for translating from stems to inflected forms . CompareOrContrast train +As shown in ( ) , using this representation , a linear classifier can not distinguish sentences sampled from a trigram and real sentences . Motivation train +Various approaches for computing semantic relatedness of words or concepts have been proposed , e.g. dictionary-based ( Lesk , 1986 ) , ontology-based ( Wu and Palmer , 1994 ; ) , information-based ( Resnik , 1995 ; Jiang and Conrath , 1997 ) or distributional ( Weeds and Weir , 2005 ) . Background train +This framework , where the `` semantic load '' is spread more evenly throughout the lexicon to lexical categories not typically thought to bear semantic content , is essentially the model advocated by a ) , among many others . Background val + also presents a similar method for the extraction of a TAG from the Penn Treebank . Background train +To a first approximation , a CURRENT-FOCUS reaches only nodes that are c-commanded ( ) by its generator . Background train +A previous work along this line is , which is based on weighted finite-state transducers ( FSTs ) . CompareOrContrast val +Most approaches rely on VerbNet ( Kipper et al. , 2000 ) and FrameNet ( Baker et al. , 1998 ) to provide associations between verbs and semantic roles , that are then mapped onto the current instance , as shown by the systems competing in semantic role labelling competitions ( Carreras and Marquez , 2004 ; Carreras and Marquez , 2005 ) and also ( Gildea and Jurafsky , 2002 ; ; Shi and Mihalcea , 2005 ) . Background val + describe an efficient algorithm for accomplishing this in which approximations to Pst ( TIS ) are computed in parallel for all ( new ) features ft by holding all weights in the existing model fixed and optimizing only over a8t . Background train +As suggested in this can be done by looking up the ranks of each of the four given words ( i.e. the words occurring in a particular word equation ) within the association vector of a translation candidate , and by multiplying these ranks . Motivation val +32 In certain cases an extension of the constraint language with named disjunctions or contexted constraints ( ; Eisele and Dorre 1990 ; Griffith 1996 ) can be used to circumvent constraint propagation . Background train +This article represents an extension of our previous work on unsupervised event coreference resolution ( Bejan et al. 2009 ; ) . Extends val +These include devices such as interleaving the components ( McDonald 1983 ; Appelt 1983 ) , backtracking on failure ( Appelt 1985 ; Nogier 1989 ) , allowing the linguistic component to interrogate the planner ( ; Sondheimer and Nebel 1986 ) , and Hovy 's notion of restrictive ( i.e. , bottom-up ) planning ( Hovy 1988a , 1988c ) . Background train +See , among others , ( ) . Background train +In Table 2 , lem refers to the LTAG parser ( ) , ANSI C implementation of the two-phase parsing algorithm that performs the head corner parsing ( van Noord , 1994 ) without features ( phase 1 ) , and then executes feature unification ( phase 2 ) . CompareOrContrast train + and Burkett et al. ( 2010 ) focused on joint parsing and alignment . CompareOrContrast train +In modern syntactic theories ( e.g. , lexical-functional grammar [ LFG ] [ ; Bresnan 2001 ; Dalrymple 2001 ] , head-driven phrase structure grammar [ HPSG ] [ Pollard and Sag 1994 ] , tree-adjoining grammar [ TAG ] [ Joshi 1988 ] , and combinatory categorial grammar [ CCG ] [ Ades and Steedman 1982 ] ) , the lexicon is the central repository for much morphological , syntactic , and semantic information . Background train +For example , when books should n't be copied by hand any longer , authors took the advantage and start writing original books and evaluation -- i.e. literary criticism -- unlike in the previous times ( ) . Background train +Furthermore , the availability of rich ontological resources , in the form of the Unified Medical Language System ( UMLS ) ( Lindberg et al. , 1993 ) , and the availability of software that leverages this knowledge -- MetaMap ( ) for concept identification and SemRep ( Rindflesch and Fiszman , 2003 ) for relation extraction -- provide a foundation for studying the role of semantics in various tasks . Background train +Since the language generation module works in parallel with the language understanding module , utterance generation is possible even while the system is listening to user utterances and that utterance understanding is possible even while it is speaking ( a ) . Background train +This idea was expanded to include nouns and their modifiers through verb nominalizations ( ; Quirk et al. , 1985 ) . Background val +For example , such schema can serve as a mean to represent translation examples , or find structural correspondences for the purpose of transfer grammar learning ( ) , ( Aramaki et al. , 2001 ) , ( Watanabe et al. , 2000 ) , ( Meyers et al. , 2000 ) , ( Matsumoto et al. , 1993 ) , ( kaji et al. , 1992 ) , and example-base machine translation EBMT3 ( Sato & Nagao , 1990 ) , ( Sato , 1991 ) , ( Richardson et al. , 2001 ) , ( Al-Adhaileh & Tang , 1999 ) . Background train + define a kernel over parse trees and apply it to re-ranking the output of a parser , but the resulting feature space is restricted by the need to compute the kernel efficiently , and the results are not as good as Collins ' previous work on re-ranking using a finite set of features ( Collins , 2000 ) . Background train +9 We only use the minimal GHKM rules ( ) here to reduce the complexity of the sampler . Uses val +FBLTAG ( ; Vijay-Shanker and Joshi , 1988 ) is an extension of the LTAG formalism . Background train +In addition to headwords , dictionary search through the pronunciation field is available ; has merged information from the pronunciation and hyphenation fields , creating an enhanced phonological representation which allows access to entries by broad phonetic class and syllable structure ( Huttenlocher and Zue , 1983 ) . Uses train +Research on shallow parsing was inspired by psycholinguistics arguments ( ) that suggest that in many scenarios ( e.g. , conversational ) full parsing is not a realistic strategy for sentence processing and analysis , and was further motivated by several arguments from a natural language engineering viewpoint . Background train +Despite this , to date , there has been little work on corpus-based approaches to help-desk response automation ( notable exceptions are Carmel , Shtalhaim , and Soffer 2000 ; Lapalme and Kosseim 2003 ; ; Malik , Subramaniam , and Kaushik 2007 ) . Background train +results are based on a corpus of movie subtitles ( Tiedemann 2007 ) , and are consequently shorter sentences , whereas the En → Es results are based on a corpus of parliamentary proceedings ( ) . Uses train +A statistical technique which has recently become popular for NLP is Maximum Entropy/Minimum Divergence ( MEMD ) modeling ( ) . Uses train +Our baseline coreference system uses the C4 .5 decision tree learner ( ) to acquire a classifier on the training texts for determining whether two NPs are coreferent . Uses train +This alignment is obtained by following the same set of rules learned from the development dataset as in ( ) . Uses train +Developed Systems Our developed system is built on the work by , using Constrained Latent Left-Linking Model ( CL3M ) as our mention-pair coreference model in the joint framework10 . Uses train +The best performance on the Brown corpus , a 0.2 % error rate , was reported by , who trained a decision tree classifier on a 25-million-word corpus . CompareOrContrast train +Most approaches rely on VerbNet ( Kipper et al. , 2000 ) and FrameNet ( Baker et al. , 1998 ) to provide associations between verbs and semantic roles , that are then mapped onto the current instance , as shown by the systems competing in semantic role labelling competitions ( Carreras and Marquez , 2004 ; Carreras and Marquez , 2005 ) and also ( Gildea and Jurafsky , 2002 ; Pradhan et al. , 2005 ; ) . Background train +These include devices such as interleaving the components ( McDonald 1983 ; ) , backtracking on failure ( Appelt 1985 ; Nogier 1989 ) , allowing the linguistic component to interrogate the planner ( Mann 1983 ; Sondheimer and Nebel 1986 ) , and Hovy 's notion of restrictive ( i.e. , bottom-up ) planning ( Hovy 1988a , 1988c ) . Background val +The system was trained on the Penn Treebank ( Marcus et al. , 1993 ) WSJ Sections 221 and tested on Section 23 ( Table 1 ) , same as used by Magerman ( 1995 ) , Collins ( 1997 ) , and , and became a common testbed . CompareOrContrast train + was the first scholar who stressed the impact of the digital revolution to the medium of writing . Background train +A similar problem is discussed in the psycholinguistics of interpretation ( ) : Interpretation is widely assumed to proceed incrementally , but vague descriptions resist strict incrementality , since an adjective in a vague description can only be fully interpreted when its comparison set is known . Background train +Much of theoretical linguistics can be formulated in a very natural manner as stating correspondences ( translations ) between layers of representation structures ( ) . Background train +In our previous work ( ) , we applied this method to a small subset of WordNet nouns and showed potential applicability . Extends train +Much of the earlier work in anaphora resolution heavily exploited domain and linguistic knowledge ( Sidner 1979 ; Carter 1987 ; ; Carbonell and Brown 1988 ) , which was difficult both to represent and to process , and which required considerable human input . Background train +the mention sub-type , which is a sub-category of the mention type ( ) ( e.g. OrgGovernmental , FacilityPath , etc. ) . Uses train +For more details on the proprieties of SSTC , see . Background train +This has been reported for other languages , too , dependent on the generality of the chosen approach ( J ¨ appinen and Niemist ¨ o , 1988 ; Choueka , 1990 ; ; Ekmekc ¸ ioglu et al. , 1995 ; Hedlund et al. , 2001 ; Pirkola , 2001 ) . Background train +Specifically , we used Decision Graphs ( Oliver 1993 ) for Doc-Pred , and SVMs ( ) for Sent-Pred .11 Additionally , we used unigrams for clustering documents and sentences , and unigrams and bigrams for predicting document clusters and sentence clusters ( Sections 3.1.2 and 3.2.2 ) . Uses train +To prepare SMT outputs for post-editing , the creators of the corpus used their own WMT10 system ( Potet et al. , 2010 ) , based on the Moses phrase-based decoder ( ) with dense features . Uses val +To model o ( Li , S → T ) , o ( Ri , S → T ) , i.e. the reordering of the neighboring phrases of a function word , we employ the orientation model introduced by . Uses val +The syntactic structures of the input data are produced by a parser with good coverage and detailed syntactic information , DIPETT ( ) . Uses train +In particular , ( ) lists the converses of some 3 500 predicative nouns . Future train +This is similar to the `` deletion '' strategy employed by , but we do it directly in the grammar . CompareOrContrast val +Although this study falls under the general topic of discourse modeling , our work differs from previous attempts to characterize text in terms of domainindependent rhetorical elements ( McKeown , 1985 ; ) . CompareOrContrast train +Many NLP applications require knowledge about semantic relatedness rather than just similarity ( ) . Background train +Other work on modeling the meanings of verbs using video recognition has also begun showing great promise ( ; Regneri et al. , 2013 ) . Background train +It has been shown ( ) that the subcategorization tendencies of verbs vary across linguistic domains . Motivation train +In the system , we extract both the minimal GHKM rules ( ) , and the rules of SPMT Model 1 ( Galley et al. , 2006 ) with phrases up to length L = 5 on the source side . Uses train +Note that this ensures that greater importance is attributed to longer chunks , as is usual in most EBMT systems ( cfXXX Sato and Nagao 1990 ; ; Carl 1999 ) .7 As an example , consider the translation into French of the house collapsed . Background train +One of the better-known approaches is described in , which suggested that abbreviations first be extracted from a corpus using abbreviation-guessing heuristics akin to those described in Section 6 and then reused in further processing . Background train +Unlike the models proposed by b ) , this model is symmetric , because both word bags are generated together from a joint probability distribution . CompareOrContrast val +We apply two different priming experiments namely , the cross modal priming and masked priming experiment discussed in ( Forster and Davis , 1984 ; Rastle et al. , 2000 ; ; Marslen-Wilson et al. , 2008 ) for Bangla morphologically complex words . Uses train +( Watanabe et al. , 2007 ; Chiang et al. , 2008 ; ) proposed other optimization objectives by introducing a margin-based and ranking-based indirect loss functions . Background train +Our work extends directions taken in systems such as Ariane ( Vauquois and Boitet , 1985 ) , FoG ( Kittredge and Polguere , 1991 ) , JOYCE ( Rambow and ) , and LFS ( Iordanskaja et al. , 1992 ) . Extends train +Since the arguments can provide useful semantic information , the SRL is crucial to many natural language processing tasks , such as Question and Answering ( Narayanan and Harabagiu 2004 ) , Information Extraction ( Surdeanu et al. 2003 ) , and Machine Translation ( ) . Background train +Such systems extract information from some types of syntactic units ( clauses in ( Fillmore and Atkins , 1998 ; Gildea and Jurafsky , 2002 ; Hull and Gomez , 1996 ) ; noun phrases in ( Hull and Gomez , 1996 ; ) ) . Background val +Association Norms ( AN ) is a collection of association norms collected by Schulte im . Uses train +Stanford University is developing the English Resource Grammar , an HPSG grammar for English , as a part of the Linguistic Grammars Online ( LinGO ) project ( ) . Background train +The changes made were inspired by those described in , page 75 ) . Motivation train +The grammar code system used in LDOCE is based quite closely on the descriptive grammatical framework of , 1985 ) . Extends train +Our task was made possible by the fact that while far from being a database in the accepted sense of the word , the LDOCE typesetting tape is the only truly computerised dictionary of English ( ) . Background train +A detailed description of the kinds of expectation mechanisms appearing in these systems appears in . Background train +Using the GHKM algorithm ( ) , we can get two different STSG derivations from the two U-trees based on the fixed word alignment . Uses train +The need for information systems to support physicians at the point of care has been well studied ( Covell et al. , 1985 ; Gorman et al. , 1994 ; ) . Background train +Task properties Determining whether or not a speaker supports a proposal falls within the realm of sentiment analysis , an extremely active research area devoted to the computational treatment of subjective or opinion-oriented language ( early work includes Wiebe and Rapaport ( 1988 ) , Hearst ( 1992 ) , , and Wiebe ( 1994 ) ; see Esuli ( 2006 ) for an active bibliography ) . Background val +Table look-up using an explicit translation lexicon is sufficient and preferable for many multilingual NLP applications , including `` crummy '' MT on the World Wide Web ( Church & Hovy , 1993 ) , certain machine-assisted translation tools ( e.g. ( Macklovitch , 1994 ; Melamed , 1996b ) ) , concordancing for bilingual lexicography ( ; Gale & Church , 1991 ) , computerassisted language learning , corpus linguistics ( Melby . Background val +Our HDP extension is also inspired from the Bayesian model proposed by . Motivation train +The EDR has close ties to the named entity recognition ( NER ) and coreference resolution tasks , which have been the focus of several recent investigations ( Bikel et al. , 1997 ; Miller et al. , 1998 ; ; Mikheev et al. , 1999 ; Soon et al. , 2001 ; Ng and Cardie , 2002 ; Florian et al. , 2004 ) , and have been at the center of evaluations such as : MUC-6 , MUC-7 , and the CoNLL '02 and CoNLL '03 shared tasks . Background train +As has been previously observed and exploited in the NLP literature ( Pang and Lee , 2004 ; ; Barzilay and Lapata , 2005 ) , the above optimization function , unlike many others that have been proposed for graph or set partitioning , can be solved exactly in an provably efficient manner via methods for finding minimum cuts in graphs . CompareOrContrast train +To combine the phrasal matching scores obtained at each n-gram level , and optimize their relative weights , we trained a Support Vector Machine classifier , SVMlight ( ) , using each score as a feature . Uses train +Against the background of a growing interest in multilingual NLP , multilingual anaphora / coreference resolution has gained considerable momentum in recent years ( Aone and McKee 1993 ; Azzam , Humphreys , and Gaizauskas 1998 ; Harabagiu and Maiorano 2000 ; ; Mitkov 1999 ; Mitkov and Stys 1997 ; Mitkov , Belguith , and Stys 1998 ) . Background train +For an overview of systems designed to answer open-domain factoid questions , the TREC QA track overview papers are a good place to start ( ) . Background train +Another interesting extension is to broaden the definition of a `` word '' to include multi-word lexical units ( ) . Future train +For example , 10 million words of the American National Corpus ( ) will have manually corrected POS tags , a tenfold increase over the Penn Treebank ( Marcus et al. , 1993 ) , currently used for training POS taggers . Background train +From an IR view , a lot of specialized research has already been carried out for medical applications , with emphasis on the lexico-semantic aspects of dederivation and decomposition ( Pacak et al. , 1980 ; ; Wolff , 1984 ; Wingert , 1985 ; Dujols et al. , 1991 ; Baud et al. , 1998 ) . Background train +This approach has occasionally been taken , as in Kantrowitz and Bates ( 1992 ) and Danlos ( 1987 ) and , at least implicitly , in and Delin et al. ( 1994 ) ; however , under this approach , all of the flexibility and simplicity of modular design is lost . Background val +Typical examples are Bulgarian ( Simov et al. , 2005 ; Simov and Osenova , 2003 ) , Chinese ( ) , Danish ( Kromann , 2003 ) , and Swedish ( Nilsson et al. , 2005 ) . Background train +In this paper , we use the Constrained Latent Left-Linking Model ( CL3M ) described in in our experiments . Uses train +We use a standard split of 268 training documents , 68 development documents , and 106 testing documents ( ; Bengtson and Roth , 2008 ) . Uses train +Aside from the extraction of theory-neutral subcategorization lexicons , there has also been work in the automatic construction of lexical resources which comply with the principles of particular linguistic theories such as LTAG , CCG , and HPSG ( Chen and Vijay-Shanker 2000 ; ; Hockenmaier , Bierner , and Baldridge 2004 ; Nakanishi , Miyao , and Tsujii 2004 ) . Background train +An alternative representation based on is presented in Selkirk ( 1984 ) , which contends that prosody , including prosodic phrasing , is more properly represented as a grid instead of a tree . CompareOrContrast train +For the joint segmentation and POS-tagging task , we present a novel solution using the framework in this article , and show that it gives comparable accuracies to our previous work ( a ) , while being more than an order of magnitude faster . CompareOrContrast val +According to the data available from 1990 U.S. Census Bureau , only 90,000 different names are shared by 100 million people ( ) . Background train +Since then this idea has been applied to several tasks , including word sense disambiguation ( ) and named-entity recognition ( Cucerzan and Yarowsky 1999 ) . Background val +• History-based feature models for predicting the next parser action ( ) . Uses train +For each co-occurring pair of word types u and v , these likelihoods are initially set proportional to their co-occurrence frequency ( „ , v ) and inversely proportional to their marginal frequencies n ( u ) and n ( v ) 1 , following ( ) 2 . Uses train +However , rather than output this wrong translation directly , we use a post hoc validation and ( if required ) correction process based on . Uses train +Most approaches rely on VerbNet ( Kipper et al. , 2000 ) and FrameNet ( Baker et al. , 1998 ) to provide associations between verbs and semantic roles , that are then mapped onto the current instance , as shown by the systems competing in semantic role labelling competitions ( Carreras and Marquez , 2004 ; Carreras and Marquez , 2005 ) and also ( Gildea and Jurafsky , 2002 ; Pradhan et al. , 2005 ; ) . Background val +The powerful mechanism of lexical rules ( ) has been used in many natural language processing systems . Background train +The version proposed here combines a basic insight from Lewin ( 1990 ) with higher-order unification to give an analysis that has a strong resemblance to that proposed in , 1991 ) , with some differences that are commented on below . CompareOrContrast train + presented a perceptron-based algorithm for learning the phrase-translation parameters in a statistical machine translation system . CompareOrContrast train +NLG has to do more than select a distinguishing description ( i.e. , one that unambiguously denotes its referent ; ) : The selected expression should also be felicitous . Background train + , 1997 ) conducts some small experiments using his METLA system to show the viability of this approach for English − > French and English − > Urdu . Background train +This idea of preserving properties can be considered an instance of the well-known frame problem in AT ( ) , and we will therefore refer to the specifications left implicit by the linguist as the frame specification , or simply frame , of a lexical rule . CompareOrContrast train +All experiments have been performed using MaltParser ( ) , version 0.4 , which is made available together with the suite of programs used for preand post-processing .1 Uses val +One area of current interest concerns the left-to-right arrangement of premodifying adjectives within an NP ( e.g. , Shaw and Hatzivassiloglou 1999 ; ) . Background train +Better results would be expected by combining the PCFG-LA parser with discriminative reranking approaches ( ; Huang , 2008 ) for self training . Future train +Some researchers ( ; Nguyen and Cao , 2008 ) have explored the use of Wikipedia information to improve the disambiguation process . Background train +Others have applied the NLP technologies of near-duplicate detection and topic-based text categorization to politically oriented text ( ; Purpura and Hillard , 2006 ) . Background val +We chose to follow and split the sentences evenly to facilitate further comparison . Uses train +That is , a document that contains terms al , a2 and a3 may be ranked higher than a document which contains terms al and b.f. However , the second document is more likely to be relevant since correct translations of the query terms are more likely to co-occur ( ) . Background train +The last years have seen considerable advances in the field of anaphora resolution , but a number of outstanding issues either remain unsolved or need more attention and , as a consequence , represent major challenges to the further development of the field ( a ) . Future train +Using an accumulator passing technique ( ) , we ensure that upon execution of a call to the interaction predicate q_1 a new lexical entry is derived as the result of successive application of a number of lexical rules . Uses train +However , most strategies are based on `` internal '' or `` external methods '' ( ) , i.e. methods that rely on the form of terms or on the information gathered from contexts . CompareOrContrast train +With the exception of ( Fung , 1995b ) , previous methods for automatically constructing statistical translation models begin by looking at word cooccurrence frequencies in bitexts ( Gale & Church , 1991 ; Kumano & Hirakawa , 1994 ; Fung , 1995a ; ) . Background train +A substring in the sentence that corresponds to a node in the representation tree is denoted by assigning the interval of the substring to SNODE of 2 These definitions are based on the discussion in ( ) and Boitet & Zaharin ( 1988 ) . Uses train +The Web People Search task , as defined in the first WePS evaluation campaign ( ) , consists of grouping search results for a given name according to the different people that share it . Background train +Although this is only true in cases where y occurs in an upward monotone context ( ) , in practice genuine contradictions between y-values sharing a meronym relationship are extremely rare . Motivation train +We use the same data setting with , however a bit different from Xue and Palmer ( 2005 ) . Uses train +Erk ( 2007 ) compared a number of techniques for creating similar-word sets and found that both the Jaccard coefficient and a ) 's information-theoretic metric work best . Background train +We previously showed that incorporating this intuition into a Bayesian prior can help train a CCG supertagger ( ) . Extends val +4 To turn this likelihood into a certainty , one can add a test at the end of the algorithm , which adds a type-related property if none is present yet ( cfXXX , ) . Background train +For descriptions of SMT systems see for example ( Germann et al. , 2001 ; ; Tillmann and Ney , 2002 ; Vogel et al. , 2000 ; Wang and Waibel , 1997 ) . Background train +Such tools make it easy to run most current approaches to statistical markup , chunking , normalization , segmentation , alignment , and noisy-channel decoding , ' including classic models for speech recognition ( ) and machine translation ( Knight and Al-Onaizan , 1998 ) . Background train +It is analogous to the step in other translation model induction algorithms that sets all probabilities below a certain threshold to negligible values ( Brown et al. , 1990 ; Dagan et al. , 1993 ; ) . CompareOrContrast train +The ConTroll grammar development system as described in ( b ) implements the above mentioned techniques for compiling an HPSG theory into typed feature grammars . Background train +After calculating the raw score of each sentence , we use a modified version of the Adaptive Greedy Algorithm by to penalize redundant sentences in cohesive clusters . Uses train +If each word 's translation is treated as a sense tag ( ) , then `` translational '' collocations have the unique property that the collocate and the word sense are one and the same ! Uses train +The values of a vector correspond to the presence or absence of each ( lemmatized ) corpus word in the document in question ( after removing stop-words and words with very low frequency ) .4 The predictive model is a Decision Graph ( ) , which , like Snob , is based on the MML principle . Uses train +Therefore , in order to be able to incorporate long-range dependencies in our models , we chose to adopt a re-ranking approach ( ) , which selects from likely assignments generated by a model which makes stronger independence assumptions . Uses val +However , it is possible to think about constraining linguistic or logical predicates by simulating physical experiences ( cfXXX ) . Background train +One , the VOYAGER domain ( ) , answers questions about places of interest in an urban area , in our case , the vicinity of MIT and Harvard University . Uses train +Our task is closer to the work of , who looked at the problem of intellectual attribution in scientific texts . CompareOrContrast val +In the seminal work by , similarity judgments were obtained from 51 test subjects on 65 noun pairs written on paper cards . Background train +( ; Blunsom et al. , 2008 ) used maximum likelihood estimation to learn weights for MT. ( Och , 2003 ; Moore and Quirk , 2008 ; Zhao and Chen , 2009 ; Galley and Quirk , 2011 ) employed an evaluation metric as a loss function and directly optimized it . CompareOrContrast val + , p. 14 ) writes `` it would be perverse not to take as a working assumption that language is a relatively efficient and accurate encoding of the information it conveys . '' CompareOrContrast train +See also ( ; Naish , 1986 ) . Background train +Michiels ( 1982 ) and provide a more detailed analysis of the information encoded by the LDOCE grammar codes and discuss their efficacy as a system of linguistic description . Background val +We parsed the 3 GB AQUAINT corpus ( Voorhees , 2002 ) using Minipar ( b ) , and collected verb-object and verb-subject frequencies , building an empirical MI model from this data . Uses train +More specifically , we use LIBSVM ( ) with a quadratic kernel K ( xZ , xj ) = ( - yxT xj + r ) 2 and the built-in one-versus-all strategy for multi-class classification . Uses train +It is wasteful to compute ti as suggested earlier , by minimizing ( cxxi ) of o ( yixE ) , since then the real work is done by an c-closure step ( ) that implements the all-pairs version of algebraic path , whereas all we need is the single-source version . Background train +positional features that have been employed by highwe can see , the baseline achieves an F-measure of performing resolvers such as 57.0 and a resolution accuracy of 48.4 . CompareOrContrast train +Such a component would serve as the first stage of a clinical question answering system ( Demner-Fushman and Lin , 2005 ) or summarization system ( ) . Future train +Briscoe and Carroll ( 1997 ) predefine 163 verbal subcategorization frames , obtained by manually merging the classes exemplified in the COMLEX ( MacLeod , Grishman , and Meyers 1994 ) and ANLT ( ) dictionaries and adding around 30 frames found by manual inspection . Background train +The idea resurfaced forcefully at several points in the more recent history of linguistic research ( Tesni`ere , 1959 ; ; Fillmore , 1968 ) . Background train +It is therefore no surprise that early attempts at response automation were knowledge-driven ( Barr and Tessler 1995 ; Watson 1997 ; ) . Background val +For example , the suite of LT tools ( Mikheev et al. , 1999 ; ) perform tokenization , tagging and chunking on XML marked-up text directly . Background train +It is only recently that the web name ambiguity has been approached as a separate problem and defined as an NLP task Web People Search on its own ( Artiles et al. , 2005 ; ) . Uses train +To retrieve translation examples for a test sentence , ( ) defined a metric based on the combination of edit distance and TF-IDF ( Manning and Sch ¨ utze , 1999 ) as follows : Uses train +By using the EM algorithm ( ) , they can guarantee convergence towards the globally optimum parameter set . Background train +• use of low level knowledge from the speech recognition phase , • use of high level knowledge about the domain in particular and the dialogue task in general , • a `` continue '' facility and an `` auto-loop '' facility as described by , • a `` conditioning '' facility as described by Fink et al. ( 1985 ) , • implementation of new types of paraphrasing , • checking a larger environment in the expectation acquisition algorithm when deciding if an incoming sentence is the same or similar to one already seen , and • examining inter-speaker dialogue patterns . Future train +Previous work has developed various approaches for grounded semantics mainly for the reference resolution task , i.e. , identifying visual objects in the environment given language descriptions ( Dhande , 2003 ; Gorniak and Roy , 2004 ; Tenbrink and Moratz , 2003 ; Siebert and Schlangen , 2008 ; ) . Extends val +In most recent research , NEs ( person , location and organisations ) are extracted from the text and used as a source of evidence to calculate the similarity between documents - see for instance ( Blume , 2005 ; Chen and Martin , 2007 ; ; Kalashnikov et al. , 2007 ) . Background train +To provide the required configurability in the static version of the code we will use policy templates ( ) , and for the dynamic version we will use configuration classes . Uses train + demonstrates a technique for segmenting Arabic text and uses it as a morphological processing step in machine translation . Uses train +In informal experiments described elsewhere ( ) , I found that the G2 statistic suggested by Dunning ( 1993 ) slightly outperforms 02 . Extends train +• language learning ( Green 1979 ; Mori and Moeser 1983 ; Morgan , Meier , and Newport 1989 ) • monolingual grammar induction ( Juola 1998 ) • grammar optimization ( ) • insights into universal grammar ( Juola 1998 ) • machine translation ( Juola 1994 , 1997 ; Veale and Way 1997 ; Gough , Way , and Hearne 2002 ) Background train +A recent study by also investigates the task of training parsers to improve MT reordering . CompareOrContrast train +Due to their remarkable ability to incorporate context structure information and long distance reordering into the translation process , tree-based translation models have shown promising progress in improving translation quality ( Liu et al. , 2006 , 2009 ; Quirk et al. , 2005 ; , 2006 ; Marcu et al. , 2006 ; Shen et al. , 2008 ; Zhang et al. , 2011b ) . Background train +15 show that the question of whether the application criterion of lexical rules should be a subsumption or a unification test is an important question deserving of more attention . Background val +Latent Dirichlet Allocation ( ) , or LDA , is an unsupervised Bayesian probabilistic model of text documents . Background train +Regarding future work , there are many research line that may be followed : i ) Capturing more features by employing external knowledge such as ontological , lexical resource or WordNet-based features ( Basili et al. , 2005a ; Basili et al. , 2005b ; Bloehdorn et al. , 2006 ; Bloehdorn and Moschitti , 2007 ) or shallow semantic trees , ( Giuglea and Moschitti , 2004 ; ; Moschitti and Bejan , 2004 ; Moschitti et al. , 2007 ; Moschitti , 2008 ; Moschitti et al. , 2008 ) . Future train +Third , the paradigm of evidence-based medicine ( ) provides a task-based model of the clinical information-seeking process . Background train + introduced factored SMT . CompareOrContrast train +Both systems are built around from the maximum-entropy technique ( ) . Uses train +We chose the adjectives as follows : we first compiled a list of all the polysemous adjectives mentioned in the lexical semantics literature ( Vendler , 1968 ; ) . Uses train +Most web-derived corpora have exploited raw text or HTML pages , so efforts have focussed on boilerplate removal and cleanup of these formats with tools like Hyppia-BTE , Tidy and Parcels3 ( ) . Background train +Undesirable consequences of this fact have been termed `` label bias '' ( ) . Background train +For the full parser , we use the one developed by Michael Collins ( ; Collins , 1997 ) -- one of the most accurate full parsers around . Uses train +Some researchers , however , including , train on predicted feature values instead . CompareOrContrast train +Their computational significance arises from the issue of their storage in lexical resources like WordNet ( ) and raises the questions like , how to store morphologically complex words , in a lexical resource like WordNet keeping in mind the storage and access efficiency . Background train +Fortunately , there exists a compact PCFG-reduction of DOP1 that generates the same trees with the same probabilities , as shown by , 2002 ) . Background train + claims that prosodic phrase boundaries will co-occur with grammatical functions such as subject , predicate , modifier , and adjunct . Background train +We rephrase the method of as follows : First , we construct the approximating finite automaton according to the unparameterized RTN method above . Uses train +6The analysis is reminiscent of the treatment of coordination in the Collins parser ( ) . CompareOrContrast val +The work is carried out in order to assist terminographers in the enrichment of a dictionary on computing that includes collocational information ( ) . Motivation train +CornmandTalk ( ) , Circuit Fix-It Shop ( Smith , 1997 ) and TRAINS-96 ( Traum and Allen , 1994 ; Traum and Andersen , 1999 ) are spoken language systems but they interface to simulation or help facilities rather than semi-autonomous agents . Background train +The annotation procedure is dependent on locating the head daughter , for which an amended version of is used . Uses val + ; Oliva 1994 ; Frank 1994 ; Opalka 1995 ; Sanfilippo 1995 ) . CompareOrContrast train +For instance , recently wrote : `` To our knowledge , learning algorithms , although promising , have not ( yet ) reached the level of rule sets developed by humans '' ( p. 520 ) . Background train +Much previous work looks at the impact of using source side information ( i.e. , feature functions on the aligned English ) , such as those of Avramidis and Koehn ( 2008 ) , and others . CompareOrContrast train +We use an in-house statistical tagger ( based on ( ) ) to tag the text in which the unknown word occurs . Uses train +Based on a computational grammar that associates natural language expressions with both a syntactic and a semantic representation , a paraphrastic gram ` As we shall briefly discuss in section 4 , the grammar is developed with the help of a meta-grammar ( ) thus ensuring an additional level of abstraction . Uses train +There are very few reported attempts at corpus-based automation of help-desk responses ( Carmel , Shtalhaim , and Soffer 2000 ; ; Bickel and Scheffer 2004 ; Malik , Subramaniam , and Kaushik 2007 ) . CompareOrContrast train +Both tasks are performed with a statistical framework : the mention detection system is similar to the one presented in ( ) and the coreference resolution system is similar to the one described in ( Luo et al. , 2004 ) . CompareOrContrast train +The disambiguation of person names in Web results is usually compared to two other Natural Language Processing tasks : Word Sense Disambiguation ( WSD ) ( Agirre and Edmonds , 2006 ) and Cross-document Coreference ( CDC ) ( ) . Background train +Our own work ( ) extends the first idea to paraphrase fragment extraction on monolingual parallel and comparable corpora . Extends train +They proved to be useful in a number of NLP applications such as natural language generation ( ) , multidocument summarization ( McKeown et al. , 2002 ) , automatic evaluation of MT ( Denkowski and Lavie , 2010 ) , and TE ( Dinu and Wang , 2009 ) . Motivation train +Current state-of-the-art statistical parsers ( ; Charniak 2000 ) are all trained on large annotated corpora such as the Penn Treebank ( Marcus , Santorini , and Marcinkiewicz 1993 ) . Background train +Although the parser only derives projective graphs , the fact that graphs are labeled allows non-projective dependencies to be captured using the pseudoprojective approach of . Background train +Riehemann 1993 ; Oliva 1994 ; ; Opalka 1995 ; Sanfilippo 1995 ) . CompareOrContrast train +The idea resurfaced forcefully at several points in the more recent history of linguistic research ( Tesni`ere , 1959 ; Gruber , 1965 ; ) . Background train +A detailed introduction to the SBD problem can be found in . Background train +Task properties Determining whether or not a speaker supports a proposal falls within the realm of sentiment analysis , an extremely active research area devoted to the computational treatment of subjective or opinion-oriented language ( early work includes Wiebe and Rapaport ( 1988 ) , Hearst ( 1992 ) , Sack ( 1994 ) , and ; see Esuli ( 2006 ) for an active bibliography ) . Background train +The BEETLE II system architecture is designed to overcome these limitations ( ) . Background train +For the evaluation of the results we use the BLEU score ( ) . Uses train +There are very few reported attempts at corpus-based automation of help-desk responses ( Carmel , Shtalhaim , and Soffer 2000 ; Lapalme and Kosseim 2003 ; ; Malik , Subramaniam , and Kaushik 2007 ) . CompareOrContrast train +For an introduction to maximum entropy modeling and training procedures , the reader is referred to the corresponding literature , for instance ( ) or ( Ratnaparkhi , 1997 ) . Background train +This design idea was adopted from TANKA ( b ) . Uses train +The problem of handling ill-formed input has been studied by , Granger ( 1983 ) , Jensen et al. ( 1983 ) , Kwasny and Sondheimer ( 1981 ) , Riesbeck and Schank ( 1976 ) , Thompson ( 1980 ) , Weischedel and Black ( 1980 ) , and Weischedel and Sondheimer ( 1983 ) . CompareOrContrast train +In addition to its explanatory capacity , this symbolic acquisition technique has obtained good results for other acquisition tasks when compared to existing statistical techniques ( ) . Motivation val +The context of a current token ti is clearly one of the most important features in predicting whether ti is a mention or not ( ) . Background train +Features using the word context ( left and right tokens ) have been shown to be very helpful in coreference resolution ( ) . Uses train +Here , the PET and GR kernel perform similar : this is different from the results of ( ) where GR performed much worse than PET for ACE data . CompareOrContrast val +In modern syntactic theories ( e.g. , lexical-functional grammar [ LFG ] [ Kaplan and Bresnan 1982 ; Bresnan 2001 ; ] , head-driven phrase structure grammar [ HPSG ] [ Pollard and Sag 1994 ] , tree-adjoining grammar [ TAG ] [ Joshi 1988 ] , and combinatory categorial grammar [ CCG ] [ Ades and Steedman 1982 ] ) , the lexicon is the central repository for much morphological , syntactic , and semantic information . Background train +Some works abstract perception via the usage of symbolic logic representations ( Chen et al. , 2010 ; Chen and Mooney , 2011 ; ; Artzi and Zettlemoyer , 2013 ) , while others choose to employ concepts elicited from psycholinguistic and cognition studies . Background val +This is where robust syntactic systems like SATZ ( Palmer and Hearst 1997 ) or the POS tagger reported in , which do not heavily rely on word capitalization and are not sensitive to document length , have an advantage . CompareOrContrast train +Our work is inspired by the latent left-linking model in and the ILP formulation from Chang et al. ( 2011 ) . Background val +For example , proves that Chinese numerals such as wu zhao zhao zhao zhao zhao wu zhao zhao zhao zhao wu zhao zhao zhao wu zhao zhao wu zhao , for the number 5000000000000000005000000000000005000000000005000000005000 , are not context-free , which implies that Chinese is not a context-free language and thus might parse in exponential worst-case time . Background train + also note that the applicability of paraphrases is strongly influenced by context . Background train +Promising features for classification include part of speech , frequency of co-occurrence , relative word position , and translational entropy ( ) . Future train +For descriptions of SMT systems see for example ( ; Och et al. , 1999 ; Tillmann and Ney , 2002 ; Vogel et al. , 2000 ; Wang and Waibel , 1997 ) . Background train +like information extraction ( ) and textual entailment ( Berant et al. , 2010 ) . Background train +In a final processing stage , we generalize over the marker lexicon following a process found in . Uses val +One approach to this problem consists in defining , within the Cut-free atomic-id space , normal form derivations in which the succession of rule application is regulated ( , Hepple 1990 , Hendriks 1993 ) . Background train +The SPR uses rules automatically learned from training data , using techniques similar to ( ; Freund et al. , 1998 ) . CompareOrContrast train + have previously examined the task of categorizing sentences in medical abstracts using supervised discriminative machine learning techniques . CompareOrContrast train +For complementing this database and for converse constructions , the LADL tables ( ) can furthermore be resorted to , which list detailed syntactico-semantic descriptions for 5 000 verbs and 25 000 verbal expressions . Future train +These operations are not domain-specific and are similar to those of previous aggregation components ( Rambow and Korelsky ,1992 ; ; Danlos , 2000 ) , although the various MERGE operations are , to our knowledge , novel in this form . Background train +Following construction of the marker lexicon , the ( source , target ) chunks are generalized further using a methodology based on to permit a limited form of insertion in the translation process . Uses train +ment ( Sarkar and Wintner , 1999 ; ; Makino et al. , 1998 ) . Background train +The ability to explicitly identify these sections in unstructured text could play an important role in applications such as document summarization ( Teufel and Moens , 2000 ) , information retrieval ( Tbahriti et al. , 2005 ) , information extraction ( ) , and question answering . Background train +Recent developments in linguistics , and especially on grammatical theory -- for example , Generalised Phrase Structure Grammar ( GPSG ) ( Gazdar et al. , 1985 ) , Lexical Functional Grammar ( LFG ) ( Kaplan and Bresnan , 1982 ) -- and on natural language parsing frameworks -- for example , Functional Unification Grammar ( FUG ) ( Kay , 1984a ) , PATR-II ( ) -- make it feasible to consider the implementation of efficient systems for the syntactic analysis of substantial fragments of natural language . Background train +The latter question is tackled by applicationspecific evaluation , where a measure is tested within the framework of a certain application , e.g. word sense disambiguation ( ) or malapropism detection ( Budanitsky and Hirst , 2006 ) . Background train + argue for application-specific evaluation of similarity measures , because measures are always used for some task . Background train +The first direct application of parse forest in translation is our previous work ( ) which translates a packed forest from a parser ; it is also the base system in our experiments ( see below ) . Extends train +4 This interpretation of the signature is sometimes referred to as closed world ( Gerdemann and ; Gerdemann 1995 ) . Background train +In their Gaijin system , give a result of 63 % accurate translations obtained for English − > German on a test set of 791 sentences from CorelDRAW manuals . CompareOrContrast train +This method can be generalized , inspired by , who derive N-gram probabilities from stochastic context-free grammars . Background train +The system was trained on the Penn Treebank ( Marcus et al. , 1993 ) WSJ Sections 221 and tested on Section 23 ( Table 1 ) , same as used by Magerman ( 1995 ) , , and Ratnaparkhi ( 1997 ) , and became a common testbed . CompareOrContrast val +The Nash arbitration plan , for example , would allow a doubly graded description whenever the product of the Values for the referent r exceeds that of all distractors ( Nash 1950 ; cfXXX ; Thorisson 1994 , for other plans ) . Background train +The paraphrase dictionary that we use was generated for us by Chris Callison-Burch , using the technique described in , which exploits a parallel corpus and methods developed for statistical machine translation . Uses train +To name a few examples , and Socher et al. ( 2013 ) show how semantic information from text can be used to improve zero-shot classification ( i.e. , classifying never-before-seen objects ) , and Motwani and Mooney ( 2012 ) show that verb clusters can be used to improve activity recognition in videos . Background val +This approach has now gained wide usage , as exemplified by the work of , 1999 ) , Charniak ( 1996 , 1997 ) , Johnson ( 1998 ) , Chiang ( 2000 ) , and many others . Motivation train +For example , such schema can serve as a mean to represent translation examples , or find structural correspondences for the purpose of transfer grammar learning ( Menezes & Richardson , 2001 ) , ( Aramaki et al. , 2001 ) , ( Watanabe et al. , 2000 ) , ( Meyers et al. , 2000 ) , ( Matsumoto et al. , 1993 ) , ( kaji et al. , 1992 ) , and example-base machine translation EBMT3 ( Sato & Nagao , 1990 ) , ( Sato , 1991 ) , ( ) , ( Al-Adhaileh & Tang , 1999 ) . Background train +Increasingly , corpus researchers are tapping the Web to overcome the sparse data problem ( ) . Background val +As such it resembles the parser of the grammar development system Attribute Language Engine ( ALE ) of ( ) . CompareOrContrast train +de URL : http://www.sfs.nphil.uni-tuebingen.de/sfb / b4home.html 1 This is , for example , the case for all proposals working with verbal lexical entries that raise the arguments of a verbal complement ( Hinrichs and Nakazawa 1989 ) that also use lexical rules such as the Complement Extraction Lexical Rule ( Pollard and Sag 1994 ) or the Complement Cliticization Lexical Rule ( ) to operate on those raised elements . Background train +Research that is more similar in goal to that outlined in this paper is Vosse ( ) . CompareOrContrast train + pointed out that little attention had been paid in the named-entity recognition field to the discourse properties of proper names . Uses train +Table 5 shows our mapping from publication type and MeSH headings to evidence grades based on principles defined in the Strength of Recommendations Taxonomy ( ) . Uses train +While we have observed reasonable results with both G2 and Fisher 's exact test , we have not yet discussed how these results compare to the results that can be obtained with a technique commonly used in corpus linguistics based on the mutual information ( MI ) measure ( ) : Uses val +Following , we consider an anaphoric reference , NPi , correctly resolved if NPi and its closest antecedent are in the same coreference chain in the resulting partition . Uses train +From an IR view , a lot of specialized research has already been carried out for medical applications , with emphasis on the lexico-semantic aspects of dederivation and decomposition ( ; Norton and Pacak , 1983 ; Wolff , 1984 ; Wingert , 1985 ; Dujols et al. , 1991 ; Baud et al. , 1998 ) . Background train +The Gsearch system ( ) also selects sentences by syntactic criteria from large on-line text collections . Background train +Thus , over the past few years , along with advances in the use of learning and statistical methods for acquisition of full parsers ( Collins , 1997 ; Charniak , 1997a ; Charniak , 1997b ; Ratnaparkhi , 1997 ) , significant progress has been made on the use of statistical learning methods to recognize shallow parsing patterns syntactic phrases or words that participate in a syntactic relationship ( Church , 1988 ; Ramshaw and Marcus , 1995 ; Argamon et al. , 1998 ; ; Munoz et al. , 1999 ; Punyakanok and Roth , 2001 ; Buchholz et al. , 1999 ; Tjong Kim Sang and Buchholz , 2000 ) . Background train +See also the work of , which considers computer-based pronunciation by analogy but does not mention the possible application to text-to-speech synthesis . Background train +The resulting list of POS-tagged lemmas is weighted using the SMART ` ltc ' 8 tf.idf-weighting scheme ( ) . Uses train +The psycholinguistic studies of Martin ( 1970 ) , , Hillinger et al. ( 1976 ) , Grosjean et al. ( 1979 ) , Dommergues and Grosjean ( 1983 ) , and Gee and Grosjean ( 1983 ) , responding to the idea of readjusted syntax as the source of prosodic phrasing , show that grammatical structure , even if readjusted , is not in itself a reliable predictor of prosodic phrasing : mismatches between syntax and prosody occur often and systematically , and can be related to specific nonsyntactic factors such as length and word frequency . Background train +Some methods are based on likelihood ( Och and Ney , 2002 ; Blunsom et al. , 2008 ) , error rate ( ; Zhao and Chen , 2009 ; Pauls et al. , 2009 ; Galley and Quirk , 2011 ) , margin ( Watanabe et al. , 2007 ; Chiang et al. , 2008 ) and ranking ( Hopkins and May , 2011 ) , and among which minimum error rate training ( MERT ) ( Och , 2003 ) is the most popular one . Background val +Numerous previous pseudodisambiguation evaluations only include arguments that occur between 30 and 3000 times ( Erk , 2007 ; ; Rooth et al. , 1999 ) . CompareOrContrast train +• cross-language information retrieval ( e.g. , McCarley 1999 ) , • multilingual document filtering ( e.g. , Oard 1997 ) , • computer-assisted language learning ( e.g. , ) , • certain machine-assisted translation tools ( e.g. , Macklovitch 1994 ; Melamed 1996a ) , • concordancing for bilingual lexicography ( e.g. , Catizone , Russell , and Warwick 1989 ; Gale and Church 1991 ) , Background train +One would think that the type information ti , which is more specific than that 16 A linguistic example based on the signature given by would be a lexical rule deriving predicative signs from nonpredicative ones , i.e. , changing the PRD value of substantive signs from -- to - F , much like the lexical rule for NPs given by Pollard and Sag ( 1994 , p. 360 , fn . Background train +• language learning ( Green 1979 ; ; Morgan , Meier , and Newport 1989 ) • monolingual grammar induction ( Juola 1998 ) • grammar optimization ( Juola 1994 ) • insights into universal grammar ( Juola 1998 ) • machine translation ( Juola 1994 , 1997 ; Veale and Way 1997 ; Gough , Way , and Hearne 2002 ) Background train +Notable early papers on graph-based semisupervised learning include , Bansal et al. ( 2002 ) , Kondor and Lafferty ( 2002 ) , and Joachims ( 2003 ) . Background train +As a result , researchers have re-adopted the once-popular knowledge-rich approach , investigating a variety of semantic knowledge sources for common noun resolution , such as the semantic relations between two NPs ( e.g. , Ji et al. ( 2005 ) ) , their semantic similarity as computed using WordNet ( e.g. , Poesio et al. ( 2004 ) ) or Wikipedia ( Ponzetto and Strube , 2006 ) , and the contextual role played by an NP ( see ) . Background train +An example of psycholinguistically oriented research work can be found in . Background val +Alternatively , we may think of user-centered comparative studies ( ) . Future train + , p. 112 ) , for example , bemoans the fact that his `` theory lacks a world knowledge component , a mental ` encyclopedia , ' which could be invoked to generate inferences ... '' . Background train +However , more recent work ( Cahill et al. 2002 ; Cahill , McCarthy , et al. 2004 ) has presented efforts in evolving and scaling up annotation techniques to the Penn-II Treebank ( ) , containing more than 1,000,000 words and 49,000 sentences . Background train +The current system learns finite state flowcharts whereas typical learning systems usually acquire coefficient values as in , assertional statements as in Michalski ( 1980 ) , or semantic nets as in Winston ( 1975 ) . CompareOrContrast train +It also shows the structural identity to bilingual grammars as used in ( ) . Uses val +Relationships between the unlabeled items consider sequential relations between different types of emails ( e.g. , between requests and satisfactions thereof ) to classify messages , and thus also explicitly exploit the structure of conversations . Background train +100000 word stems of German ( ) . Uses train +The computational treatment of lexical rules proposed can be seen as an extension to the principled method discussed by Gotz and , 1996 , 1997b ) for encoding the main building block of HPSG grammars -- the implicative constraints -- as a logic program . Extends train +Identical to the standard perceptron proof , e.g. , , by inserting in loss-separability for normal separability . Background train +The goal of the JAVOX toolkit is to speech-enable traditional desktop applications -- this is similar to the goals of the MELISSA project ( ) . CompareOrContrast train +The simplest strategy for ordering adjectives is what call the direct evidence method . Background train +The numeral ( whether it is implicit , as in ( 3 ) , or explicit ) can be construed as allowing the reader to draw inferences about the standards employed ( Kyburg and Morreau 2000 ; ) : ( 3 ) , for example , implies a standard that counts 10 cm as large and 8 cm as not large . Background train +Such systems extract information from some types of syntactic units ( clauses in ( ; Gildea and Jurafsky , 2002 ; Hull and Gomez , 1996 ) ; noun phrases in ( Hull and Gomez , 1996 ; Rosario et al. , 2002 ) ) . Background val +`` petty conversational implicature '' ( ) , or the metarules of Section 5.2 ? Background train +Unlike other POS taggers , this POS tagger ( ) was also trained to disambiguate sentence boundaries . Uses train +On the other hand , experiments indicate that mental representation and processing of morphologically complex words are not quite language independent ( ) . Background val +ASARES has been previously applied to the acquisition of word pairs sharing semantic relations defined in the Generative Lexicon framework ( ) and called qualia relations ( Bouillon et al. , 2001 ) . Background train +People are much more likely to consult such evaluative statements than the actual text of a bill or law under discussion , given the dense nature of legislative language and the fact that ( U.S. ) bills often reach several hundred pages in length ( ) . Background train +For english , there is for instance the 15 year old HewlettPackard test suite , a simple text file listing test sentences and grouping them according to linguistics phenomena ( Flickinger et al. , 1987 ) ; and more recently , the much more sophisticated TSNLP ( Test Suite for Natural Language Processing ) which includes some 9500 test items for English , French and German , each of them being annotated with syntactic and application related information ( ) . Background train +Its significance is reflected both in the growing interest in annotation software for word sense tagging ( ) and in the long-standing use of part-of-speech taggers , parsers and morphological analysers for data from English and many other languages . Background train +Each of these Values has equal status , so the notion of a basic-level Value can not play a role ( cfXXX , ) . Background train +Previous work on Chinese SRL mainly focused on how to transplant the machine learning methods which has been successful with English , such as Sun and Jurafsky ( 2004 ) , Xue and Palmer ( 2005 ) and . Background val +Accuracy is not the best measure to assess segmentation quality , therefore we also conducted experiments using the WindowDiff measure as proposed by . Uses train +The system is in the form of an agenda-driven chart-based parser whose foundation is similar to previous formalizations of Chomsky 's Minimalist Program ( Stabler , 1997 ; Harkema , 2000 ; ) . CompareOrContrast train +Optimizing for dependency arc length is particularly important as parsers tend to do worse on longer dependencies ( ) and these dependencies are typically the most meaningful for downstream tasks , e.g. , main verb dependencies for tasks Motivation train +Whereas dealt only with an English corpus , the current work shows that this methodology is applicable to a wide range of languages and corpora . CompareOrContrast train +We also experiment with a CCG parser ( ) , requiring that the contexts surrounding the original phrase and paraphrase are assigned Uses train +Their kernel is also very time consuming and in their more general sparse setting it requires O ( mn3 ) time and O ( mn2 ) space , where m and n are the number of nodes of the two trees ( m > = n ) ( ) . Future train +Other molecular biology databases We also included several model organism databases or nomenclature databases in the construction of the dictionary , i.e. , mouse Mouse Genome Database ( MGD ) [ 18 ] , fly FlyBase [ 19 ] , yeast Saccharomyces Genome Database ( SGD ) [ 20 ] , rat -- Rat Genome Database ( RGD ) [ 21 ] , worm -- WormBase [ 22 ] , Human Nomenclature Database ( HUGO ) [ 23 ] , Online Mendelian Inheritance in Man ( OMIM ) [ 24 ] , and Enzyme Nomenclature Database ( ECNUM ) . Uses train +ones , DIRT ( Lin and Pantel , 2001 ) , VerbOcean ( ) , FrameNet ( Baker et al. , 1998 ) , and Wikipedia ( Mehdad et al. , 2010 ; Kouylekov et al. , 2009 ) . Background train +include decision tree learning and Bayesian learning , nearest neighbor learning , and artificial neural networks , early such works may be found in ( ) , ( Creecy and Masand , 1992 ) and ( Wiene and Pedersen , 1995 ) , respectively . Background val +In other words AJAX is a web development technique for creating interactive web applications using a combination of XHTML and CSS , Document Object Model ( or DOM ) , the XMLHTTPRequest object ( ) . Background train +For these or for a specific domain , basic synonymic dictionaries can be complemented using learning methods based on distributional similarity ( ; Lin , 1998 ) . Future train +converted to numerical features using the standard technique of binarization , and we split values of the FEATS field into its atomic components .4 For some languages , we divide the training data into smaller sets , based on some feature s ( normally the CPOS or POS of the next input token ) , which may reduce training times without a significant loss in accuracy ( ) . Background val +• language learning ( Green 1979 ; Mori and Moeser 1983 ; Morgan , Meier , and Newport 1989 ) • monolingual grammar induction ( ) • grammar optimization ( Juola 1994 ) • insights into universal grammar ( Juola 1998 ) • machine translation ( Juola 1994 , 1997 ; Veale and Way 1997 ; Gough , Way , and Hearne 2002 ) Background train +Linguistic preprocessing of text documents is carried out by re-using smes , an information extraction core system for real-world German text processing ( ) . Uses val +Regarding future work , there are many research line that may be followed : i ) Capturing more features by employing external knowledge such as ontological , lexical resource or WordNet-based features ( Basili et al. , 2005a ; Basili et al. , 2005b ; Bloehdorn et al. , 2006 ; Bloehdorn and Moschitti , 2007 ) or shallow semantic trees , ( Giuglea and Moschitti , 2004 ; Giuglea and Moschitti , 2006 ; Moschitti and Bejan , 2004 ; Moschitti et al. , 2007 ; Moschitti , 2008 ; ) . Future train +For English − > Urdu , , page 213 ) notes that `` the system learned the original training corpus ... perfectly and could reproduce it without errors '' ; that is , it scored 100 % accuracy when tested against the training corpus . Background train +Much previous work looks at the impact of using source side information ( i.e. , feature functions on the aligned English ) , such as those of , Yeniterzi and Oflazer ( 2010 ) and others . CompareOrContrast val +Recent work ( ; Curran and Moens , 2002 ) has suggested that some tasks will benefit from using significantly more data . Background train +See for a variant of this approximation that constructs finite transducers rather than finite automata . Background train +Due to their remarkable ability to incorporate context structure information and long distance reordering into the translation process , tree-based translation models have shown promising progress in improving translation quality ( Liu et al. , 2006 , 2009 ; ; Galley et al. , 2004 , 2006 ; Marcu et al. , 2006 ; Shen et al. , 2008 ; Zhang et al. , 2011b ) . Background val +These include devices such as interleaving the components ( McDonald 1983 ; Appelt 1983 ) , backtracking on failure ( Appelt 1985 ; Nogier 1989 ) , allowing the linguistic component to interrogate the planner ( Mann 1983 ; Sondheimer and Nebel 1986 ) , and Hovy 's notion of restrictive ( i.e. , bottom-up ) planning ( a , 1988c ) . Background train +The shallow parser used is the SNoW-based CSCL parser ( Punyakanok and Roth , 2001 ; ) . Uses val +Other approaches use less deep linguistic resources ( e.g. , POS-tags ) or are ( almost ) knowledge-free ( e.g. , Koehn and Knight ( 2003 ) ) . CompareOrContrast train +Some well-known approaches include rule-based models ( ) , backed-off models ( Collins and Brooks 1995 ) , and a maximumentropy model ( Ratnaparkhi 1998 ) . Background train +One approach to partial parsing was presented by , who extended a shallow-parsing technique to partial parsing . Background train + compared a predictive approach ( statistical translation ) , a retrieval approach based on a language-model , and a hybrid approach which combines statistical chunking and traditional retrieval . CompareOrContrast train +Indeed , such rich semantic links can be used to extend indices or reformulate queries ( similar to the work by with WoRDNET relations ) . CompareOrContrast train +In modern syntactic theories ( e.g. , lexical-functional grammar [ LFG ] [ Kaplan and Bresnan 1982 ; Bresnan 2001 ; Dalrymple 2001 ] , head-driven phrase structure grammar [ HPSG ] [ ] , tree-adjoining grammar [ TAG ] [ Joshi 1988 ] , and combinatory categorial grammar [ CCG ] [ Ades and Steedman 1982 ] ) , the lexicon is the central repository for much morphological , syntactic , and semantic information . Background train +Despite these arguments , most applied NLG systems use a pipelined architecture ; indeed , a pipeline was used in every one of the systems surveyed by and Paiva ( 1998 ) . Background val +Other similar approaches include those of Cicekli and G ¨ uvenir ( 1996 ) , McTait and Trujillo ( 1999 ) , , and Brown ( 2000 ) , inter alia . Background train +The EDR has close ties to the named entity recognition ( NER ) and coreference resolution tasks , which have been the focus of several recent investigations ( Bikel et al. , 1997 ; Miller et al. , 1998 ; Borthwick , 1999 ; Mikheev et al. , 1999 ; ; Ng and Cardie , 2002 ; Florian et al. , 2004 ) , and have been at the center of evaluations such as : MUC-6 , MUC-7 , and the CoNLL '02 and CoNLL '03 shared tasks . Background train +In addition to headwords , dictionary search through the pronunciation field is available ; Carter ( 1987 ) has merged information from the pronunciation and hyphenation fields , creating an enhanced phonological representation which allows access to entries by broad phonetic class and syllable structure ( ) . Background train +Following Hockenmaier , Bierner , and Baldridge ( 2002 ) , , and Miyao , Ninomiya , and Tsujii ( 2004 ) , we extract a reference lexicon from Sections 02 -- 21 of the WSJ . Uses train +The only disambiguation metric that we used in our previous work ( b ) was the shape-based metric , according to which the `` best '' trees are those that are skewed to the right . Extends train +The need for information systems to support physicians at the point of care has been well studied ( ; Gorman et al. , 1994 ; Ely et al. , 2005 ) . Background val +Every arc always has a definite direction , i.e. arcs are arrows ( ) . Background train +In the first experiment , we use an induction algorithm ( a ) based on the expectation-maximization ( EM ) principle that induces parsers for PLTIGs . Uses train +To create the baseline system , we use the opensource Joshua 4.0 system ( Ganitkevitch et al. , 2012 ) to build a hierarchical phrase-based ( HPB ) system , and a syntax-augmented MT ( SAMT ) 11 system ( ) respectively . Uses val +6 For Sent-Pred we also experimented with grammatical and sentence-based syntactic features , such as number of syntactic phrases , grammatical mood , and grammatical person ( ) , but the simple binary bag-of-lemmas representation yielded similar results . Uses val +At present , the system uses a heuristic matching algorithm to classify relations into the appropriate category , though in the future we may consider a classifier similar to . Future train +It is interesting to compare this analysis with that described in Dalrymple , Shieber , and Pereira ( 1991 ) and , 1991 ) . CompareOrContrast train +The importance of including nonheadwords has become uncontroversial ( e.g. Collins 1999 ; ; Goodman 1998 ) . Background train +Some methods of semantic relation analysis rely on predefined templates filled with information from processed texts ( ) . Background train +The example used to illustrate the power of ATNs ( ) , `` John was believed to have been shot , '' also parses correctly , because the [ object ] node following the verb `` believed '' acts as both an absorber and a ( re ) generator . CompareOrContrast train +Problems such as these have motivated research on more abstract , dependencybased parser evaluation ( e.g. , Lin 1995 ; Carroll , Briscoe , and Sanfilippo 1998 ; Carroll et al. 2002 ; Clark and Hockenmaier 2002 ; King et al. 2003 ; ; Kaplan et al. 2004 ; Miyao and Tsujii 2004 ) . Motivation train +LiLFeS is one of the fastest inference engines for processing feature structure logic , and efficient HPSG parsers have already been built on this system ( Nishida et al. , 1999 ; ) . Background train +As in ( ) , we used unsupervised training data which is automatically segmented to discover previously unseen stems . Uses train +To address this limitation , our previous work ( ) has initiated an investigation on the problem of conversation entailment . Extends train +∗ A brief version of this work , with some additional material , first appeared as ( a ) . Extends train +They also proposed two major categories of meta-learning approaches for recommender systems , merging and ensemble , each subdivided into the more specific subclasses suggested by as follows . Background train +Here , I adopt the model proposed by and decompose lexical verbs into verbalizing heads and verbal roots . Uses val +This heuristic is called soft union ( ) . Uses train +This approach has its roots in Fillmore 's Case Grammar ( 1968 ) , and serves as the foundation for two current large-scale semantic annotation projects : FrameNet ( ) and PropBank ( Kingsbury et al. , 2002 ) . Background train +Many investigators ( e.g. Allen 1976 ; Elowitz et al. 1976 ; ; Cahn 1988 ) have suggested that the poor prosody of synthetic speech , in comparison with natural speech , is the primary factor leading to difficulties in the comprehension of fluent synthetic speech . Motivation train +The ability to explicitly identify these sections in unstructured text could play an important role in applications such as document summarization ( Teufel and Moens , 2000 ) , information retrieval ( ) , information extraction ( Mizuta et al. , 2005 ) , and question answering . Background train +This semantics was constructed ( a , 1987b ) as a formal framework for default and commonsense reasoning . Uses val +There is a general consensus among theoretical linguists that the proper representation of verbal argument structure is event structure -- representations grounded in a theory of events that decompose semantic roles in terms of primitive predicates representing concepts such as causality and inchoativity ( ; Jackendoff , 1983 ; Pustejovsky , 1991b ; Rappaport Hovav and Levin , 1998 ) . Background train +Similarly , the notion of R + M-abduction is spiritually related to the `` abductive inference '' of Reggia ( 1985 ) , the `` diagnosis from first principles '' of Reiter ( 1987 ) , `` explainability '' of Poole ( 1988 ) , and the subset principle of . CompareOrContrast train +The EDR has close ties to the named entity recognition ( NER ) and coreference resolution tasks , which have been the focus of several recent investigations ( Bikel et al. , 1997 ; Miller et al. , 1998 ; Borthwick , 1999 ; ; Soon et al. , 2001 ; Ng and Cardie , 2002 ; Florian et al. , 2004 ) , and have been at the center of evaluations such as : MUC-6 , MUC-7 , and the CoNLL '02 and CoNLL '03 shared tasks . Background train +Representative systems are described in Boisen et al. ( 1989 ) , De Mattia and Giachin ( 1989 ) , , Niemann ( 1990 ) , and Young ( 1989 ) . Background train +R98 ( , , , , „ ) uses a variant of Kozima 's semantic similarity measure ( ) to compute block similarity . Extends train + , Meral et al. ( 2007 ) , Murphy ( 2001 ) , Murphy and Vogel ( 2007 ) and Topkara et al. ( 2006a ) all belong to the syntactic transformation category . Background val +We used a publicly available tagger ( ) to tag the words and then used these in the input to the system . Uses train +Some works abstract perception via the usage of symbolic logic representations ( Chen et al. , 2010 ; ; Matuszek et al. , 2012 ; Artzi and Zettlemoyer , 2013 ) , while others choose to employ concepts elicited from psycholinguistic and cognition studies . Background train +Multilingual lexical databases aligned with the English WordNet ( e.g. MultiWordNet ( ) ) have been created for several languages , with different degrees of coverage . Background train +An off-the-shelf speech recognition device , a Nippon Electric Corporation DP-200 , was added to an existing natural language processing system , the Natural Language Computer ( NLC ) ( , Biermann and Ballard 1980 ) . Background train +For future work , we might investigate how machine learning algorithms , which are specifically designed for the problem of domain adaptation ( Blitzer et al. , 2007 ; ) , perform in comparison to our approach . Future train +But their importance has grown far beyond machine translation : for instance , transferring annotations between languages ( Yarowsky and Ngai 2001 ; ; Ganchev , Gillenwater , and Taskar 2009 ) ; discovery of paraphrases ( Bannard and Callison-Burch 2005 ) ; and joint unsupervised POS and parser induction across languages ( Snyder and Barzilay 2008 ) . Motivation train +That is , if the current hypothesis is unable to label a candidate or is uncertain about it , then the candidate might be a good training example ( ) . Background train +Other milestones of recent research include the deployment of probabilistic and machine learning techniques ( Aone and Bennett 1995 ; Kehler 1997 ; Ge , Hale , and Charniak 1998 ; Cardie and Wagstaff 1999 ; the continuing interest in centering , used either in original or in revised form ( Abracos and Lopes 1994 ; Strube and Hahn 1996 ; Hahn and Strube 1997 ; ) ; and proposals related to the evaluation methodology in anaphora resolution ( Mitkov 1998a , 2001b ) . Background val +For instance , the derived morphological forms are believed to be represented as a whole , whereas the representation of the inflected forms follows the morphemic model ( ) . Background train +We perceive that these results can be extended to other language models that properly embed bilexical context-free grammars , as for instance the more general history-based models used in ( Ratnaparkhi , 1997 ) and ( ) . Future train +Another paper ( ) describes the detailed analysis on the factor of the difference of parsing performance . Background train +The original and the obtained grammar generated exactly the same number of derivation trees in the parsing experiment with 457 sentences from the ATIS corpus ( ) 6 ( the average length is 6.32 words ) . Uses train +The use of running tallies and percentages is based on the assumption that these features are likely to produce generalized predictors ( ) . Motivation train +The Longman lexicographers have developed a grammar coding system capable of representing in compact form a nontrivial amount of information , usually to be found only in large descriptive grammars of English ( such as ) . Background train +19 The paper by presents additional , more sophisticated models that we do not use in this article . CompareOrContrast train +Empirical evidence has been brought forward that inflectional and/or derivational stemmers augmented by dictionaries indeed perform substantially better than those without access to such lexical repositories ( Krovetz , 1993 ; Kraaij and Pohlmann , 1996 ; ) . Background train +Although there are other discussions of the paragraph as a central element of discourse ( e.g. Chafe 1979 , Halliday and Hasan 1976 , Longacre 1979 , ) , all of them share a certain limitation in their formal techniques for analyzing paragraph structure . CompareOrContrast train +( Details of how the average-expert model performs can be found in our prior work ( ) . ) Extends train +Self-training should also benefit other discriminatively trained parsers with latent annotations ( ) , although training would be much slower compared to using generative models , as in our case . Future train +We introduce here a clearly defined and replicable split of the data , so that future investigations can accurately and correctly compare against the results presented here . Uses train +In most cases , the accuracy of parsers degrades when run on out-of-domain data ( ; McClosky et al. , 2006 ; Blitzer et al. , 2006 ; Petrov et al. , 2010 ) . Background train + relies on morphosyntactic cues in the untagged Brown corpus as indicators of six predefined subcategorization frames . Background train +To build the above s2t system , we first use the parse tree , which is generated by parsing the English side of the bilingual data with the Berkeley parser ( ) . Uses val +One obvious approach to this problem is to employ parser reranking ( ) . Background train +Ideally , to distinguish between raising and equi verbs , a number of syntactic criteria should be employed ( :460 ff . ) Future train +The table also presents the closest comparable experimental results reported by .1 McKnight and Srinivasan ( henceforth , M&S ) created a test collection consisting of 37,151 RCTs from approximately 12 million MEDLINE abstracts dated between 1976 and 2001 . CompareOrContrast train +How this mismatched perceptual basis affects referential communication in situated dialogue was investigated in our previous work ( ) . Extends val + furthered this work by showing that a bimodal topic model , consisting of both text and feature norms , outperformed models using only one modality on the prediction of association norms , word substitution errors , and semantic interference tasks . Background train +A cooccurrence based stemmer ( ) was used to stem Spanish words . Uses train +The best performance on the WSJ corpus was achieved by a combination of the SATZ system ( Palmer and Hearst 1997 ) with the Alembic system ( ) : a 0.5 % error rate . CompareOrContrast train +In a number of proposals , lexical generalizations are captured using lexical underspecification ( ; Krieger and Nerbonne 1992 ; CompareOrContrast val +Although a grid may be more descriptively suitable for some aspects of prosody ( for example , use the grid representation for their implementation of stress assignment in compound nominals ) , we are not aware of any evidence for or against a grid representation of discourseneutral phrasing . CompareOrContrast train +The bottom panel of table 1 lists the results for the chosen lexicalized model ( SSN-Freq > 200 ) and five recent statistical parsers ( Ratnaparkhi , 1999 ; Collins , 1999 ; ; Collins , 2000 ; Bod , 2001 ) . CompareOrContrast train +Other factors , such as the role of focus ( Grosz 1977 , 1978 ; ) or quantifier scoping ( Webber 1983 ) must play a role , too . Background train +linguistic in nature , rather than dealing with superficial properties of the text , e.g. the amount of white space between words ( ) . CompareOrContrast train +The second version ( RM ) concerns the Resource Management task ( ) that has been popular within the DARPA community in recent years . Uses train +Another line of research that is correlated with ours is recognition of agreement/disagreement ( ; Yin et al. , 2012 ; Abbott et al. , 2011 ; Andreas et al. , 2012 ; Galley et al. , 2004 ; Hillard et al. , 2003 ) and classification of stances ( Walker et al. , 2012 ; Somasundaran and Wiebe , 2010 ) in online forums . CompareOrContrast train +The studies presented by and Johnson ( 2007 ) differed in the number of states that they used . CompareOrContrast train +A number of alignment techniques have been proposed , varying from statistical methods ( ; Gale and Church , 1991 ) to lexical methods ( Kay and Roscheisen , 1993 ; Chen , 1993 ) . Background train +We would like to use features that look at wide context on the input side , which is inexpensive ( ) . Future train +Much of the earlier work in anaphora resolution heavily exploited domain and linguistic knowledge ( Sidner 1979 ; Carter 1987 ; Rich and LuperFoy 1988 ; ) , which was difficult both to represent and to process , and which required considerable human input . Background train +Experiments on Chinese SRL ( Xue and Palmer 2005 , ) reassured these findings . Background train +This is then generalized , following a methodology based on , to generate the `` generalized marker lexicon . '' Uses train +A number of speech understanding systems have been developed during the past fifteen years ( Barnett et al. 1980 , Dixon and Martin 1979 , Erman et al. 1980 , Haton and Pierrel 1976 , Lea 1980 , Lowerre and Reddy 1980 , Medress 1980 , , Walker 1978 , and Wolf and Woods 1980 ) . CompareOrContrast train +Our proposed method is based on the automatically acquired paraphrase dictionary described in , in which the application of paraphrases from the dictionary encodes secret bits . Uses train +For example , frequent words are translated less consistently than rare words ( ) . Background train +Over the last decade there has been a lot of interest in developing tutorial dialogue systems that understand student explanations ( Jordan et al. , 2006 ; Graesser et al. , 1999 ; Aleven et al. , 2001 ; Buckley and Wolska , 2007 ; Nielsen et al. , 2008 ; VanLehn et al. , 2007 ) , because high percentages of selfexplanation and student contentful talk are known to be correlated with better learning in humanhuman tutoring ( Chi et al. , 1994 ; Litman et al. , 2009 ; Purandare and Litman , 2008 ; ) . Background train +These knowledge sources were effectively used to build a state-of-the-art WSD program in one of our prior work ( ) . Extends train +Another line of research approaches grounded language knowledge by augmenting distributional approaches of word meaning with perceptual information ( Andrews et al. , 2009 ; Steyvers , 2010 ; Feng and Lapata , 2010b ; ; Silberer and Lapata , 2012 ; Johns and Jones , 2012 ; Bruni et al. , 2012a ; Bruni et al. , 2012b ; Silberer et al. , 2013 ) . Background val +Due to advances in statistical syntactic parsing techniques ( ; Charniak , 2001 ) , attention has recently shifted towards the harder question of analyzing the meaning of natural language sentences . Background val +This is where robust syntactic systems like SATZ ( ) or the POS tagger reported in Mikheev ( 2000 ) , which do not heavily rely on word capitalization and are not sensitive to document length , have an advantage . CompareOrContrast train +category relationships from the weak supervision : the tag dictionary and raw corpus ( ; Garrette et al. , 2015 ) .4 This procedure attempts to automatically estimate the frequency of each word/tag combination by dividing the number of raw-corpus occurrences of each word in the dictionary evenly across all of its associated tags . Uses train +WIT has been implemented in Common Lisp and C on UNIX , and we have built several experimental and demonstration dialogue systems using it , including a meeting room reservation system ( Nakano et al. , 1999b ) , a video-recording programming system , a schedule management system ( Nakano et al. , 1999a ) , and a weather infomiation system ( ) . Extends train +Lexical functional grammar ( ; Bresnan 2001 ; Dalrymple 2001 ) is a member of the family of constraint-based grammars . Background val + present an approach to learn previously unknown frames for Czech from the Prague Dependency Bank ( Hajic Background train +A companion paper describes the evaluation process and results in further detail ( ) . Extends train +SNoW ( Carleson et al. , 1999 ; ) is a multi-class classifier that is specifically tailored for learning in domains in which the potential number of information sources ( features ) taking part in decisions is very large , of which NLP is a principal example . Uses train +Hence , enumerating morphological variants in a semi-automatically generated lexicon , such as proposed for French ( ) , turns out to be infeasible , at least for German and related languages . Background train +It is analogous to the step in other translation model induction algorithms that sets all probabilities below a certain threshold to negligible values ( ; Dagan et al. , 1993 ; Chen , 1996 ) . CompareOrContrast train +They proved to be useful in a number of NLP applications such as natural language generation ( Iordanskaja et al. , 1991 ) , multidocument summarization ( McKeown et al. , 2002 ) , automatic evaluation of MT ( Denkowski and Lavie , 2010 ) , and TE ( ) . Motivation val +The results , which partly confirm those obtained on a smaller dataset in , must be seen in light of the fact that our gesture annotation scheme comprises more fine-grained categories than most of the studies mentioned earlier for both head movements and face expressions . CompareOrContrast train +We employ the idea of ultraconservative update ( ; Crammer et al. , 2006 ) to propose two incremental methods for local training in Algorithm 2 as follows . Uses train +It has also been shown to be useful in joint inference of text with visual attributes obtained using visual classifiers ( ) . Background train + reported a correlation of r = .69 . CompareOrContrast train +A more subtle example is weighted FSAs that approximate PCFGs ( Nederhof , 2000 ; ) , or to extend the idea , weighted FSTs that approximate joint or conditional synchronous PCFGs built for translation . Background train +As has been previously observed and exploited in the NLP literature ( Pang and Lee , 2004 ; Agarwal and Bhattacharyya , 2005 ; ) , the above optimization function , unlike many others that have been proposed for graph or set partitioning , can be solved exactly in an provably efficient manner via methods for finding minimum cuts in graphs . CompareOrContrast train +WIT features an incremental understanding method ( b ) that makes it possible to build a robust and real-time system . Uses train +This choice is inspired by recent work on learning syntactic categories ( ) , which successfully utilized such language models to represent word window contexts of target words . Motivation train +As for work on Arabic ( MSA ) , results have been reported on the PATB ( Kulick , Gabbard , and Marcus 2006 ; ; Green and Manning 2010 ) , the Prague Dependency Treebank ( PADT ) ( Buchholz and Marsi 2006 ; Nivre 2008 ) and the CATiB ( Habash and Roth 2009 ) . Background train +Other studies which view lR as a query generation process include Maron and Kuhns , 1960 ; Hiemstra and Kraaij , 1999 ; ; Miller et al , 1999 . CompareOrContrast train +In Section 5 , we discuss the difficulties associated with such user studies , and describe a human-based evaluation we conducted for a small subset of the responses generated by our system ( b ) . Uses train + considers the second verb V2 as an aspectual complex comparable to the auxiliaries . Background train +2We could just as easily use other symmetric `` association '' measures , such as 02 ( ) or the Dice coefficient ( Smadja , 1992 ) . CompareOrContrast val +In this situation , b , 293 ) recommend `` evaluating the expectations using only a single , probable alignment . '' Motivation val +The use of the web as a corpus for teaching and research on language has been proposed a number of times ( ; Robb , 2003 ; Rundell , 2000 ; Fletcher , 2001 , 2004b ) and received a special issue of the journal Computational Linguistics ( Kilgarriff and Grefenstette , 2003 ) . Background val +• Learnability ( Zernik and Dyer 1987 ) • Text generation ( Hovy 1988 ; Milosavljevic , Tulloch , and Dale 1996 ) • Speech generation ( ) • Localization ( Sch ¨ aler 1996 ) Background train +Following our previous work on stance classification ( c ) , we employ three types of features computed based on the frame-semantic parse of each sentence in a post obtained from SEMAFOR ( Das et al. , 2010 ) . Extends train +However , the literature on Linguistic Steganography , in which linguistic properties of a text are modified to hide information , is small compared with other media ( ) . Background train +This has been reported for other languages , too , dependent on the generality of the chosen approach ( J ¨ appinen and Niemist ¨ o , 1988 ; Choueka , 1990 ; Popovic and Willett , 1992 ; Ekmekc ¸ ioglu et al. , 1995 ; Hedlund et al. , 2001 ; ) . Background train +We offer a theorem that highlights the broad applicability of these modeling techniques .4 If f ( input , output ) is a weighted regular relation , then the following statements are equivalent : ( 1 ) f is a joint probabilistic relation ; ( 2 ) f can be computed by a Markovian FST that halts with probability 1 ; ( 3 ) f can be expressed as a probabilistic regexp , i.e. , a regexp built up from atomic expressions a : b ( for a E E U -LCB- E -RCB- , b E A U -LCB- E -RCB- ) using concatenation , probabilistic union + p , and probabilistic closure * p. For defining conditional relations , a good regexp language is unknown to us , but they can be defined in several other ways : ( 1 ) via FSTs as in Fig. 1c , ( 2 ) by compilation of weighted rewrite rules ( ) , ( 3 ) by compilation of decision trees ( Sproat and Riley , 1996 ) , ( 4 ) as a relation that performs contextual left-to-right replacement of input substrings by a smaller conditional relation ( Gerdemann and van Noord , 1999 ) ,5 ( 5 ) by conditionalization of a joint relation as discussed below . Background train +We use the CCG parser to analyse the sentence before and after paraphrasing . Uses train +For example , consider a relational description ( cfXXX , ) involving a gradable adjective , as in the dog in the large shed . Background train +Previous versions of our work , as described in also assume that phrasing is dependent on predicate-argument structure . Extends train +like information extraction ( Yates and Etzioni , 2009 ) and textual entailment ( ) . Background train +The inclusion of the coreference task in the Sixth and Seventh Message Understanding Conferences ( MUC-6 and MUC-7 ) gave a considerable impetus to the development of coreference resolution algorithms and systems , such as those described in , Gaizauskas and Humphreys ( 1996 ) , and Kameyama ( 1997 ) . Background train +Lee et al. ( 2012 ) model entity coreference and event coreference jointly ; consider joint coreference and entity-linking . Background train +However , learning-based resolvers have not been able to benefit from having an SC agreement feature , presumably because the method used to compute the SC of an NP is too simplistic : while the SC of a proper name is computed fairly accurately using a named entity ( NE ) recognizer , many resolvers simply assign to a common noun the first ( i.e. , most frequent ) WordNet sense as its SC ( e.g. , , Markert and Nissim ( 2005 ) ) . Background train +This includes work on generalized expectation ( Mann and McCallum , 2010 ) , posterior regularization ( Ganchev et al. , 2010 ) and constraint driven learning ( ; Chang et al. , 2010 ) . CompareOrContrast train +The final machine is a trigram language model , specifically a Kneser-Ney ( ) based backoff language model . Uses train +Hovy has described another text planner that builds similar plans ( b ) . Background train +In particular , since we treat each individual speech within a debate as a single `` document '' , we are considering a version of document-level sentiment-polarity classification , namely , automatically distinguishing between positive and negative documents ( Das and Chen , 2001 ; ; Turney , 2002 ; Dave et al. , 2003 ) . Background val +transition-based dependency parsing framework ( ) using an arc-eager transition strategy and are trained using the perceptron algorithm as in Zhang and Clark ( 2008 ) with a beam size of 8 . Uses train +Just as easily , we can model link types that coincide with entries in an on-line bilingual dictionary separately from those that do not ( cfXXX ) . Uses train +As an alternative , we rely on PubMed to retrieve an initial set of hits that we then postprocess in greater detail -- this is the standard pipeline architecture commonly employed in other question-answering systems ( Voorhees and Tice 1999 ; ) . CompareOrContrast train +( Och and Ney , 2002 ; Blunsom et al. , 2008 ) used maximum likelihood estimation to learn weights for MT. ( ; Moore and Quirk , 2008 ; Zhao and Chen , 2009 ; Galley and Quirk , 2011 ) employed an evaluation metric as a loss function and directly optimized it . CompareOrContrast val +We also compare the results with the output generated by the statistical translation system GIZA + + / ISI ReWrite Decoder ( AlOnaizan et al. , 1999 ; ; Germann et al. , 2001 ) , trained on the same parallel corpus . CompareOrContrast train +Some efforts have tackled tasks such as automatic image caption generation ( Feng and Lapata , 2010a ; Ordonez et al. , 2011 ) , text illustration ( ) , or automatic location identification of Twitter users ( Eisenstein et al. , 2010 ; Wing and Baldridge , 2011 ; Roller et al. , 2012 ) . Background train +Since we are not generating from the model , this does not introduce difficulties ( ) . Motivation train +The extraction procedure consists of three steps : First , the bracketing of the trees in the Penn Treebank is corrected and extended based on the approaches of Magerman ( 1994 ) and . Background train +As they are required to enable test subjects to distinguish between senses , we use artificial glosses composed from synonyms and hypernyms as a surrogate , e.g. for brother : `` brother , male sibling '' vs. `` brother , comrade , friend '' ( ) . Uses train +Over the last decade there has been a lot of interest in developing tutorial dialogue systems that understand student explanations ( Jordan et al. , 2006 ; Graesser et al. , 1999 ; ; Buckley and Wolska , 2007 ; Nielsen et al. , 2008 ; VanLehn et al. , 2007 ) , because high percentages of selfexplanation and student contentful talk are known to be correlated with better learning in humanhuman tutoring ( Chi et al. , 1994 ; Litman et al. , 2009 ; Purandare and Litman , 2008 ; Steinhauser et al. , 2007 ) . Background val + argues that there are cases , albeit exceptional ones , in which constraints on syntactic category are an issue in subcategorization . Motivation train +Corpus frequency : ( ) differentiates between misspellings and neologisms ( new words ) in terms of their frequency . Uses train +We use an in-house developed hierarchical phrase-based translation ( ) as our baseline system , and we denote it as In-Hiero . Uses train +The types of sentences accepted are essentially those accepted by the original NLC grammar , imperative sentences with nested noun groups and conjunctions ( ) . Background train +The task we used to compare different generalisation techniques is similar to that used by Pereira et al. ( 1993 ) and . CompareOrContrast train +One of the proposed methods to extract paraphrases relies on a pivot-based approach using phrase alignments in a bilingual parallel corpus ( ) . Background val +The Nash arbitration plan , for example , would allow a doubly graded description whenever the product of the Values for the referent r exceeds that of all distractors ( ; cfXXX Gorniak and Roy 2003 ; Thorisson 1994 , for other plans ) . Background train + extracts word co-occurrence probabilities from unlabelled text collected from a web crawler . Background train +Notable early papers on graph-based semisupervised learning include Blum and Chawla ( 2001 ) , , Kondor and Lafferty ( 2002 ) , and Joachims ( 2003 ) . Background val +Due to this inherent ambiguity , manual annotations usually distinguish between sure correspondences for unambiguous translations , and possible , for ambiguous translations ( ) . Background train +Thus for instance , ( Copestake and Flickinger , 2000 ; Copestake et al. , 2001 ) describes a Head Driven Phrase Structure Grammar ( HPSG ) which supports the parallel construction of a phrase structure ( or derived ) tree and of a semantic representation and ( ) show how to equip Lexical Functional grammar ( LFG ) with a glue semantics . Background train +The contextual interpreter then uses a reference resolution approach similar to Byron ( 2002 ) , and an ontology mapping mechanism ( a ) to produce a domain-specific semantic representation of the student 's output . Uses train +only the available five relative scopings of the quantifiers are produced ( , 47 ) , but without the need for a free variable constraint -- the HOU algorithm will not produce any solutions in which a previously bound variable becomes free ; • the equivalences are reversible , and thus the above sentences cart be generated from scoped logical forms ; • partial scopings are permitted ( see Reyle [ 19961 ) • scoping can be freely interleaved with other types of reference resolution ; • unscoped or partially scoped forms are available for inference or for generation at every stage . Background train +In the transducers produced by the training method described in this paper , the source and target positions are in the set -LCB- -1 , 0,1 -RCB- , though we have also used handcoded transducers ( ) and automatically trained transducers ( Alshawi and Douglas 2000 ) with a larger range of positions . Uses train +Similar approaches are being explored for parsing ( Steedman , ; Hwa et al. 2003 ) . Background train +In particular , since we treat each individual speech within a debate as a single `` document '' , we are considering a version of document-level sentiment-polarity classification , namely , automatically distinguishing between positive and negative documents ( Das and Chen , 2001 ; Pang et al. , 2002 ; Turney , 2002 ; ) . Background train +would be chunked as follows ( Tjong Kim ) : [ NP He ] [ VP reckons ] [ NP the current account deficit ] [ VP will narrow ] [ PP Background train +The flexible architecture we have presented enables interesting future research : ( i ) a straightforward improvement is the use of lexical similarity to reduce data sparseness , e.g. ( ; Basili et al. , 2006 ; Bloehdorn et al. , 2006 ) . Background train +5An alternative strategy to step ( 4 ) is to perform a database lookup based on the ambiguous query and summarize the results ( ) , which we leave for future work . Future train +Such approaches have been tried recently in restricted cases ( ; Eisner , 2001b ; Lafferty et al. , 2001 ) . Background train +Typed feature grammars can be used as the basis for implementations of Head-driven Phrase Structure Grammar ( HPSG ; Pollard and Sag , 1994 ) as discussed in ( a ) and ( Meurers and Minnen , 1997 ) . Extends val +It is therefore no surprise that early attempts at response automation were knowledge-driven ( Barr and Tessler 1995 ; ; Delic and Lahaix 1998 ) . Background val +A more subtle example is weighted FSAs that approximate PCFGs ( ; Mohri and Nederhof , 2001 ) , or to extend the idea , weighted FSTs that approximate joint or conditional synchronous PCFGs built for translation . Background train +ones , DIRT ( Lin and Pantel , 2001 ) , VerbOcean ( Chklovski and Pantel , 2004 ) , FrameNet ( Baker et al. , 1998 ) , and Wikipedia ( ; Kouylekov et al. , 2009 ) . Background train +The Ruby on framework permits us to quickly develop web applications without rewriting common functions and classes . Uses train +We further add rules for combining with punctuation to the left and right and allow for the merge rule X → X X of . Uses train +Politically-oriented text Sentiment analysis has specifically been proposed as a key enabling technology in eRulemaking , allowing the automatic analysis of the opinions that people submit ( Shulman et al. , 2005 ; ; Kwon et al. , 2006 ) . Background train + has made some preliminary attempt on the idea of hierarchical semantic Background train + pointed out that many relations between words in a text are non-classical ( i.e. other than typical taxonomic relations like synonymy or hypernymy ) and therefore not covered by semantic similarity . Background train +Some of the intuitions we associate with this notion have been very well expressed by , pp. 7-8 ) : ... Semantics is constrained by our models of ourselves and our worlds . Background val +In 2009 , the second WePS campaign showed similar trends regarding the use of NE features ( ) . Background train +The head words can be automatically extracted using a heuristic table lookup in the manner described by . Uses train +Reiter describes a pipelined modular approach as a consensus architecture underlying most recent work in generation ( ) . Background train +For example , the suite of LT tools ( ; Grover et al. , 2000 ) perform tokenization , tagging and chunking on XML marked-up text directly . Background train +Liu et al. ( 2005 ) , , Murphy ( 2001 ) , Murphy and Vogel ( 2007 ) and Topkara et al. ( 2006a ) all belong to the syntactic transformation category . Background train +Brockmann and Lapata ( 2003 ) have showed that WordNet-based approaches do not always outperform simple frequency-based models , and a number of techniques have been recently proposed which may offer ideas for refining our current unsupervised approach ( ; Bergsma et al. , 2008 ) . Future val +The RenTAL system is implemented in LiLFeS ( ) 2 . Uses train +In the future , we hope to evaluate the automatic annotations and extracted lexicon against Propbank ( ) . Future train +Each set of translations is stored separately , and for each set the `` marker hypothesis '' ( ) is used to segment the phrasal lexicon into a `` marker lexicon . '' Uses train +One way to increase the precision of the mapping process is to impose some linguistic constraints on the sequences such as simple noun-phrase contraints ( Gaussier , 1995 ; ; hua Chen and Chen , 94 ; Fung , 1995 ; Evans and Zhai , 1996 ) . Uses train +Although this study falls under the general topic of discourse modeling , our work differs from previous attempts to characterize text in terms of domainindependent rhetorical elements ( ; Marcu and Echihabi , 2002 ) . CompareOrContrast train +`` Coherence , '' as outlined above , can be understood as a declarative ( or static ) version of marker passing ( Hirst 1987 ; ) , with one difference : the activation spreads to theories that share a predicate , not through the IS-A hierarchy , and is limited to elementary facts about predicates appearing in the text . CompareOrContrast val +Task properties Determining whether or not a speaker supports a proposal falls within the realm of sentiment analysis , an extremely active research area devoted to the computational treatment of subjective or opinion-oriented language ( early work includes , Hearst ( 1992 ) , Sack ( 1994 ) , and Wiebe ( 1994 ) ; see Esuli ( 2006 ) for an active bibliography ) . Background val +First , it has been noted that in many natural language applications it is sufficient to use shallow parsing information ; information such as noun phrases ( NPs ) and other syntactic sequences have been found useful in many large-scale language processing applications including information extraction and text summarization ( ; Appelt et al. , 1993 ) . Background train +In addition , a fully flexible access system allows the retrieval of dictionary entries on the basis of constraints specifying any combination of phonetic , lexical , syntactic , and semantic information ( ) . Background val + developed a way of incorporating standard n-grams into the cache model , using mixtures of language models and also exponentially decaying the weight for the cache prediction depending on the recency of the word 's last Extends train +Riehemann 1993 ; Oliva 1994 ; Frank 1994 ; Opalka 1995 ; ) . CompareOrContrast val +Machine learning methods should be interchangeable : Transformation-based learning ( TBL ) ( ) and Memory-based learning ( MBL ) ( Daelemans et al. , 2002 ) have been applied to many different problems , so a single interchangeable component should be used to represent each method . Motivation train +Recently , several alternative , often quite sophisticated approaches to collective classification have been proposed ( Neville and Jensen , 2000 ; Lafferty et al. , 2001 ; ; Taskar et al. , 2002 ; Taskar et al. , 2003 ; Taskar et al. , 2004 ; McCallum and Wellner , 2004 ) . Background train +We tested the classification of verbs into semantic types using a verb list of 139 pre-classified items drawn from the lists published in and Stockwell et al. ( 1973 ) . Uses train +The combination of likelihood and prior modeling , HMMs , and Viterbi decoding is fundamentally the same as the standard probabilistic approaches to speech recognition ( Bahl , Jelinek , and Mercer 1983 ) and tagging ( ) . CompareOrContrast train +Cross-lingual Textual Entailment ( CLTE ) has been proposed by ( Mehdad et al. , 2010 ) as an extension of Textual Entailment ( ) that consists in deciding , given two texts T and H in different languages , if the meaning of H can be inferred from the meaning of T . Background train +The EDR has close ties to the named entity recognition ( NER ) and coreference resolution tasks , which have been the focus of several recent investigations ( Bikel et al. , 1997 ; ; Borthwick , 1999 ; Mikheev et al. , 1999 ; Soon et al. , 2001 ; Ng and Cardie , 2002 ; Florian et al. , 2004 ) , and have been at the center of evaluations such as : MUC-6 , MUC-7 , and the CoNLL '02 and CoNLL '03 shared tasks . Background train +Other solutions such as complete caching of the corpora are not typically adopted due to legal concerns over copyright and redistribution of web data , issues considered at length by a ) . Background val +This includes work on generalized expectation ( ) , posterior regularization ( Ganchev et al. , 2010 ) and constraint driven learning ( Chang et al. , 2007 ; Chang et al. , 2010 ) . Background train +A very similar formulation , for another grammar transformation , is given in . CompareOrContrast train +We run GIZA + + ( ) on the training corpus in both directions ( Koehn et al. , 2003 ) to obtain the word alignment for each sentence pair . Uses train +Other attempts to address efficiency include the fast Transformation Based Learning ( TBL ) Toolkit ( ) which dramatically speeds up training TBL systems , and the translation of TBL rules into finite state machines for very fast tagging ( Roche and Schabes , 1997 ) . Background train +Consequently , fusion has been applied to a wide variety of pattern recognition and decision theoretic problems -- using a plethora of theories , techniques , and tools -- including some applications in computational linguistics ( e.g. , ; van Halteren , Zavrel , and Daelemans 1998 ) and speech technology ( e.g. , Bowles and Damper 1989 ; Romary and Pierre11989 ) . Background train +This observation has led some researchers , e.g. , , to claim a direct mapping between the syntactic phrase and the prosodic phrase . Background train +These features are carefully designed to reduce the data sparseness problem and some of them are inspired by previous work ( He et al. , 2008 ; ; Marton and Resnik , 2008 ; Chiang et al. , 2009 ; Setiawan et al. , 2009 ; Shen et al. , 2009 ; Xiong et al. , 2009 ) : 1 . Motivation val +In our previous work ( ) , conversation entailment is formulated as the following : given a conversation segment D which is represented by a set of clauses D = d1 ∧ ... ∧ dm , and a hypothesis H represented by another set of clauses H = h1 ∧ ... ∧ hn , the prediction on whether D entails H is determined by the product of probabilities that each hypothesis clause hj is entailed from all the conversation segment clauses d1 ... dm as follows . Extends train +Notable early papers on graph-based semisupervised learning include Blum and Chawla ( 2001 ) , Bansal et al. ( 2002 ) , , and Joachims ( 2003 ) . Background train +In addition , we consider several types of lexical features ( LexF ) inspired by previous work on agreement and disagreement ( ; Misra and Walker , 2013 ) . Motivation train + compared two retrieval approaches ( TF.IDF and query expansion ) and two predictive approaches ( statistical translation and latent variable models ) . Background train +Others provide automatic mappings of natural language instructions to executable actions , such as interpreting navigation directions ( Chen and Mooney , 2011 ) or robot commands ( Tellex et al. , 2011 ; ) . Background train +The dialogue state is represented by a cumulative answer analysis which tracks , over multiple turns , the correct , incorrect , and not-yet-mentioned parts 1Other factors such as student confidence could be considered as well ( ) . Future train +As noted above , it is well documented ( ) that subcategorization frames ( and their frequencies ) vary across domains . Motivation train + explore a number of related approaches to the extraction of a lexicalized TAG from the Penn-II Treebank with the aim of constructing a statistical model for parsing . Background train +At the same time , we believe our method has advantages over the approach developed initially at IBM ( ; Brown et al. 1993 ) for training translation systems automatically . CompareOrContrast train +Building on the work of Ruch et al. ( 2003 ) in the same domain , we present a generative approach that attempts to directly model the discourse structure of MEDLINE abstracts using Hidden Markov Models ( HMMs ) ; cfXXX ( ) . Background train +The basic Python reflection has already been implemented and used for large scale experiments with POS tagging , using pyMPI ( a message passing interface library for Python ) to coordinate experiments across a cluster of over 100 machines ( Curran and Clark , 2003 ; ) . Background train +As an alternative , we rely on PubMed to retrieve an initial set of hits that we then postprocess in greater detail -- this is the standard pipeline architecture commonly employed in other question-answering systems ( ; Hirschman and Gaizauskas 2001 ) . CompareOrContrast train +The language grounding problem has received significant attention in recent years , owed in part to the wide availability of data sets ( e.g. Flickr , Von Ahn ( 2006 ) ) , computing power , improved computer vision models ( Oliva and Torralba , 2001 ; Lowe , 2004 ; Farhadi et al. , 2009 ; Parikh and Grauman , 2011 ) and neurological evidence of ties between the language , perceptual and motor systems in the brain ( Pulverm ¨ uller et al. , 2005 ; Tettamanti et al. , 2005 ; ) . Background train +Although in this paper we take modus ponens as the main rule of inference , in general one can consider deductive closures with respect to weaker , nonstandard logics , ( cfXXX ; Frisch 1987 ; Patel-Schneider 1985 ) . CompareOrContrast train +As points out , given that no situations were envisaged where the information from the tape would be altered once installed in secondary storage , this simple and convenComputational Linguistics , Volume 13 , Numbers 3-4 , July-December 1987 205 Bran Boguraev and Ted Briscoe Large Lexicons for Natural Language Processing tional access strategy is perfectly adequate . Background train +Other factors , such as the role of focus ( , 1978 ; Sidner 1983 ) or quantifier scoping ( Webber 1983 ) must play a role , too . Background train +This paper describes an approach for sharing resources in various grammar formalisms such as Feature-Based Lexicalized Tree Adjoining Grammar ( FB-LTAG1 ) ( ; Vijay-Shanker and Joshi , 1988 ) and Head-Driven Phrase Structure Grammar ( HPSG ) ( Pollard and Sag , 1994 ) by a method of grammar conversion . Background train +This is because the binary structure has been verified to be very effective for tree-based translation ( ; Zhang et al. , 2011a ) . Motivation train +There is a general consensus among theoretical linguists that the proper representation of verbal argument structure is event structure -- representations grounded in a theory of events that decompose semantic roles in terms of primitive predicates representing concepts such as causality and inchoativity ( Dowty , 1979 ; ; Pustejovsky , 1991b ; Rappaport Hovav and Levin , 1998 ) . Background train +measure the standard intrinsic parser metrics unlabeled attachment score ( UAS ) and labeled attachment score ( LAS ) ( ) . Uses train +The EDR has close ties to the named entity recognition ( NER ) and coreference resolution tasks , which have been the focus of several recent investigations ( ; Miller et al. , 1998 ; Borthwick , 1999 ; Mikheev et al. , 1999 ; Soon et al. , 2001 ; Ng and Cardie , 2002 ; Florian et al. , 2004 ) , and have been at the center of evaluations such as : MUC-6 , MUC-7 , and the CoNLL '02 and CoNLL '03 shared tasks . Background train +Accordingly , we convert examples such as ( 27 ) into their generalized equivalents , as in ( 28 ) : ( 28 ) good man : bon homme That is , where substitutes variables for various words in his templates , we replace certain lexical items with their marker tag . CompareOrContrast train +How it is done is beyond the scope of this paper but is explained in detail in . Background train +These tools use a highly optimised GIS implementation and provide sophisticated Gaussian smoothing ( ) . Uses train +Also relevant is work on the general problems of dialog-act tagging ( Stolcke et al. , 2000 ) , citation analysis ( Lehnert et al. , 1990 ) , and computational rhetorical analysis ( Marcu , 2000 ; ) . Background train +The inference rules that were necessary to convert one list of properties into another do not sit comfortably within the received NLG pipeline model ( e.g. , ) . Background train + has developed an agenda-driven chart parser for the feature-driven formalism described above ; please refer to his paper for a description of the parsing algorithm . Extends val +An example of psycholinguistically oriented research work can be found in . Background train +In addition to a referring function , noun phrases ( NP ) can also serve communicative goals such as providing new information about the referent and expressing the speaker 's emotional attitude towards the referent ( Appelt , 1985 ; ) . Background train +Some well-known approaches include rule-based models ( Brill and Resnik 1994 ) , backed-off models ( Collins and Brooks 1995 ) , and a maximumentropy model ( ) . Background train +There is some literature on procedure acquisition such as the LISP synthesis work described in and the PROLOG synthesis method of Shapiro ( 1982 ) . CompareOrContrast val +Against the background of a growing interest in multilingual NLP , multilingual anaphora / coreference resolution has gained considerable momentum in recent years ( Aone and McKee 1993 ; Azzam , Humphreys , and Gaizauskas 1998 ; ; Mitkov and Barbu 2000 ; Mitkov 1999 ; Mitkov and Stys 1997 ; Mitkov , Belguith , and Stys 1998 ) . Background train +The list of semantic relations with which we work is based on extensive literature study ( a ) . Uses train +The research described below is taking place in the context of three collaborative projects ( ; Russell et al. , 1986 ; Phillips and Thompson , 1986 ) to develop a general-purpose , wide coverage morphological and syntactic analyser for English . Background train +In most cases , the accuracy of parsers degrades when run on out-of-domain data ( Gildea , 2001 ; McClosky et al. , 2006 ; Blitzer et al. , 2006 ; ) . Background train +This paper presents experiments with generative content models for analyzing the discourse structure of medical abstracts , which has been confirmed to follow the four-section pattern discussed above ( ) . Background train +Following , one approach for achieving this objective consists of applying supervised learning , where a winning method is selected for each case in the training set , all the training cases are labeled accordingly , and then the system is trained to predict a winner for unseen cases . CompareOrContrast train +Thus , over the past few years , along with advances in the use of learning and statistical methods for acquisition of full parsers ( Collins , 1997 ; Charniak , 1997a ; Charniak , 1997b ; Ratnaparkhi , 1997 ) , significant progress has been made on the use of statistical learning methods to recognize shallow parsing patterns syntactic phrases or words that participate in a syntactic relationship ( Church , 1988 ; Ramshaw and Marcus , 1995 ; ; Cardie and Pierce , 1998 ; Munoz et al. , 1999 ; Punyakanok and Roth , 2001 ; Buchholz et al. , 1999 ; Tjong Kim Sang and Buchholz , 2000 ) . Background val +Due to their remarkable ability to incorporate context structure information and long distance reordering into the translation process , tree-based translation models have shown promising progress in improving translation quality ( , 2009 ; Quirk et al. , 2005 ; Galley et al. , 2004 , 2006 ; Marcu et al. , 2006 ; Shen et al. , 2008 ; Zhang et al. , 2011b ) . Background train +11 reports that non-projective and pseudo-projective algorithms outperform the `` eager '' projective algorithm in MaltParser , but our training data did not contain any non-projective dependencies . CompareOrContrast train +In particular , boosting ( Schapire , 1999 ; ) offers the possibility of achieving high accuracy from a collection of classifiers which individually perform quite poorly . Future train +For MT the most commonly used heuristic is called grow diagonal final ( ) . CompareOrContrast train +ECM-F is an entity-constrained mention Fmeasure ( cfXXX ( ) for how ECM-F is computed ) , and ACE-Value is the official ACE evaluation metric . Uses val +In order to obtain semantic representations of each word , we apply our previous strategy ( ) . Extends train + annotated a larger set of word pairs ( 353 ) , too . Background train +A more flexible approach is used by , where users can specify boundary values for attributes like rainfall , specifying , for example , rain counts as moderate above 7 mm/h , as heavy above 20 mm/h , and so on . Background train +Furthermore , a number of performance features , largely based on the PARADISE dialogue evaluation scheme ( ) , were automatically logged , derived , or manually annotated . Uses train + propose to generalize the direct evidence method so that it can apply to unseen pairs of adjectives by computing the transitive closure of the ordering relation . Background train +We evaluated our translations with IBM 's BLEU evaluation metric ( ) , using the same evaluation method and reference retranslations that were used for evaluation at HLT Workshop 2002 at CLSP ( Haji 6 et al. , 2002 ) . Uses train +Other attempts to address efficiency include the fast Transformation Based Learning ( TBL ) Toolkit ( Ngai and Florian , 2001 ) which dramatically speeds up training TBL systems , and the translation of TBL rules into finite state machines for very fast tagging ( ) . Background train +PR is closely related to the work of , 2008 ) , who concurrently developed the idea of using penalties based on posterior expectations of features to guide semi-supervised learning . Background train +The best results on most of our data were obtained using Hidden Naive Bayes ( HNB ) ( ) . Uses train +de URL : http://www.sfs.nphil.uni-tuebingen.de/sfb / b4home.html 1 This is , for example , the case for all proposals working with verbal lexical entries that raise the arguments of a verbal complement ( Hinrichs and Nakazawa 1989 ) that also use lexical rules such as the Complement Extraction Lexical Rule ( ) or the Complement Cliticization Lexical Rule ( Miller and Sag 1993 ) to operate on those raised elements . Background train +A number of proposals in the 1990s deliberately limited the extent to which they relied on domain and/or linguistic knowledge and reported promising results in knowledge-poor operational environments ( Dagan and Itai 1990 , 1991 ; Lappin and Leass 1994 ; Nasukawa 1994 ; Kennedy and Boguraev 1996 ; Williams , Harvey , and Preston 1996 ; Baldwin 1997 ; , 1998b ) . Background train +ASARES has been previously applied to the acquisition of word pairs sharing semantic relations defined in the Generative Lexicon framework ( Pustejovsky , 1995 ) and called qualia relations ( ) . Background train +mlSystem ruleFeats + atomFeats We augment mlSystem ruleFeats with more features from our previous work ( Markert et al. , 2012 ; a ; Hou et al. , 2013b ) on bridging anaphora recognition and antecedent selection . Extends train +The description of the EAGLE workbench for linguistic engineering ( ) mentions a case normalization module that uses a heuristic in which a capitalized word in an ambiguous position should be rewritten without capitalization if it is found lower-cased in the same document . CompareOrContrast val +An exception is , who experimented with determining the political orientation of websites essentially by classifying the concatenation of all the documents found on that site . Background val +Until now , translation models have been evaluated either subjectively ( e.g. White and O'Connell 1993 ) or using relative metrics , such as perplexity with respect to other models ( b ) . CompareOrContrast val +The three preprocessing steps ( tokenization , POS-tagging , lemmatization ) are performed using TreeTagger ( ) . Uses train +Many statistical parsers ( Ratnaparkhi , 1999 ; ; Charniak , 2001 ) are based on a history-based probability model ( Black et al. , 1993 ) , where the probability of each decision in a parse is conditioned on the previous decisions in the parse . Background train +In informal experiments described elsewhere ( Melamed 1995 ) , I found that the G2 statistic suggested by slightly outperforms 02 . Background val +For right-branching structures , the leftcorner ancestor is the parent , conditioning on which has been found to be beneficial ( Johnson , 1998 ) , as has conditioning on the left-corner child ( ) . Background train +This strategy is certainly the right one to start out with , since anaphora is always the more typical direction of reference in English prose ( , p. 329 ) . Motivation train +This indicates that parse trees are usually not the optimal choice for training tree-based translation models ( ) . Background val +In and Krotov et al. ( 1998 ) , it was observed that treebank grammars ( CFGs extracted from treebanks ) are very large and grow with the size of the treebank . Background val +For example , the forward-backward algorithm ( ) trains only Hidden Markov Models , while ( Ristad and Yianilos , 1996 ) trains only stochastic edit distance . Background train +ment ( Sarkar and Wintner , 1999 ; Doran et al. , 2000 ; ) . Background train + argues that , aside from missing domain-specific complementation trends , dictionaries produced by hand will tend to lag behind real language use because of their static nature . Motivation train +Finally , the Natural Language Toolkit ( NLTK ) is a package of NLP components implemented in Python ( ) . Background train +Politically-oriented text Sentiment analysis has specifically been proposed as a key enabling technology in eRulemaking , allowing the automatic analysis of the opinions that people submit ( ; Cardie et al. , 2006 ; Kwon et al. , 2006 ) . Background train +Burkett and Klein ( 2008 ) and focused on joint parsing and alignment . CompareOrContrast train +But the general outlines are reasonably clear , and we can adapt some of the UDRS ( ) work to our own framework . Uses train +KUbler , McDonald , and describe a `` typical '' MaltParser model configuration of attributes and features .13 Starting with it , in a series of initial controlled experiments , we settled on using buf [ 0-1 ] + stk [ 0-1 ] for word-forms , and buf [ 0-3 ] + stk [ 0-2 ] for POS tags . Uses train +For this evaluation , we randomly selected 50 abstracts with disorders indexed as the main topic from abstracts retrieved using PubMed on the five clinical questions described in . Uses train +The need for information systems to support physicians at the point of care has been well studied ( Covell et al. , 1985 ; ; Ely et al. , 2005 ) . Background train +This approach is taken in computational syntactic grammars ( e.g. ) ; the number of unlikely parses is severely reduced whenever possible , but no attempt is made to define only the so-called grammatical strings of a language . Background train +Previous work on Chinese SRL mainly focused on how to transplant the machine learning methods which has been successful with English , such as Sun and Jurafsky ( 2004 ) , and Xue ( 2008 ) . Background train +For example , the interaction of lexical rules is explored at run-time , even though the possible interaction can be determined at compile-time given the information available in the lexical rules and the base lexical entries .2 Based on the research results reported in , 1996 ) , we propose a new computational treatment of lexical rules that overcomes these shortcomings and results in a more efficient processing of lexical rules as used in HPSG . Motivation train +`` Coherence , '' as outlined above , can be understood as a declarative ( or static ) version of marker passing ( ; Charniak 1983 ) , with one difference : the activation spreads to theories that share a predicate , not through the IS-A hierarchy , and is limited to elementary facts about predicates appearing in the text . CompareOrContrast train +On small data sets all of the Bayesian estimators strongly outperform EM ( and , to a lesser extent , VB ) with respect to all of our evaluation measures , confirming the results reported in . CompareOrContrast train +In addition , the advantages of using linguistically annotated data over raw data are well documented ( ; Granger and Rayson , 1998 ) . Background train +For the A * algorithm ( ) as applied to speech recognition , the actual path score is typically augmented with an estimated score for the unseen portion . Uses train +Robust natural language understanding in Atlas-Andes is provided by Rosé 's CARMEL system ( Rosé 2000 ) ; it uses the spelling correction algorithm devised by . Uses train +The framework represents a generalization of several predecessor NLG systems based on Meaning-Text Theory : FoG ( Kittredge and Polguere , 1991 ) , LFS ( Iordanskaja et al. , 1992 ) , and JOYCE ( Rambow and ) . Extends train +Nevertheless , , page 23 ) observes that `` a slightly more general mapping , where two adjacent terminal symbols can be merged into a single lexical item ( for example , a word and its case-marking ) , can capture this sort of result quite handily . '' Background train +Other similar approaches include those of Cicekli and G ¨ uvenir ( 1996 ) , McTait and Trujillo ( 1999 ) , Carl ( 1999 ) , and , inter alia . Background train +( Watanabe et al. , 2007 ; Chiang et al. , 2008 ; ) proposed other optimization objectives by introducing a margin-based and ranking-based indirect loss functions . CompareOrContrast train + utilized a transformation-based method to learn a sequence of monolingual tree transformations for translation . CompareOrContrast train +Baseline language model : For P0 we used a trigram with modified kneser-ney smoothing [ Chen and ] , which is still considered one of the best smoothing methods for n-gram language models . Uses train +Therefore , we preprocess Ontonote-5 .0 to derive mention heads using Collins head rules ( ) with gold constituency parsing information and gold named entity information . Uses train +Our most accurate single grammar achieves an F score of 91.6 on the WSJ test set , rivaling discriminative reranking approaches ( ) and products of latent variable grammars ( Petrov , 2010 ) , despite being a single generative PCFG . CompareOrContrast train +We have presented an ensemble approach to word sense disambiguation ( ) where multiple Naive Bayesian classifiers , each based on co -- occurrence features from varying sized windows of context , is shown to perform well on the widely studied nouns interest and line . Background train +Over the last decade there has been a lot of interest in developing tutorial dialogue systems that understand student explanations ( Jordan et al. , 2006 ; ; Aleven et al. , 2001 ; Buckley and Wolska , 2007 ; Nielsen et al. , 2008 ; VanLehn et al. , 2007 ) , because high percentages of selfexplanation and student contentful talk are known to be correlated with better learning in humanhuman tutoring ( Chi et al. , 1994 ; Litman et al. , 2009 ; Purandare and Litman , 2008 ; Steinhauser et al. , 2007 ) . Background train +We shall see this in the next example : two sentences , regarded as a fragment of paragraph , are a variation on a theme by . Extends train +Nevertheless , the full document text is present in most systems , sometimes as the only feature ( ) and sometimes in combination with others see for instance ( Chen and Martin , 2007 ; Popescu and Magnini , 2007 ) - . Background train +The recognizer for these systems is the SUMMIT system ( ) , which uses a segmental-based framework and includes an auditory model in the front-end processing . Uses train +Some recent GRE algorithms have done away with the separation between content determination and linguistic realization , interleaving the two processes instead ( ; Krahmer and Theune 2002 ) . CompareOrContrast train +To sample from our proposal distribution , we use a blocked Gibbs sampler based on the one proposed by Goodman ( 1998 ) and used by that samples entire parse trees . Uses train +Semantic construction proceeds from the derived tree ( ) rather than -- as is more common in TAG -- from the derivation tree . Background train +It allows the construction of a non-TAL ( ) , ( Harbusch & Poller , 2000 ) . Background val +Following previous work ( e.g. , and Ponzetto and Strube ( 2006 ) ) , we generate training instances as follows : a positive instance is created for each anaphoric NP , NPj , and its closest antecedent , NPi ; and a negative instance is created for NPj paired with each of the intervening NPs , NPi +1 , NPi +2 , ... , NPj_1 . Uses train +TF-IDF ( term frequency-inverse document frequency ) is one of the widely used feature selection techniques in information retrieval ( ) . Background train +The gap mechanism resembles the Hold register idea of ATNs ( ) and the treatment of bounded domination metavariables in lexical functional grammars ( LFGs ) ( Bresnan 1982 , p. 235 ff . ) CompareOrContrast train +To solve these scaling issues , we implement Online Variational Bayesian Inference ( Hoffman et al. , 2010 ; ) for our models . Uses val +Typical examples are Bulgarian ( Simov et al. , 2005 ; ) , Chinese ( Chen et al. , 2003 ) , Danish ( Kromann , 2003 ) , and Swedish ( Nilsson et al. , 2005 ) . Background train +This appeared to solve the problem , and the results presented later for the average degree of generalisation do not show an over-generalisation compared with those given in . CompareOrContrast train +These observations and this line of reasoning has not escaped the attention of theoretical linguists : propose that argument structure is , in fact , encoded syntactically . Background train +inter-document references in the form of hyperlinks ( ) . Background train +While IA is generally thought to be consistent with findings on human language production ( ; Levelt 1989 ; Pechmann 1989 ; Sonnenschein 1982 ) , the hypothesis that incrementality is a good model of human GRE seems unfalsifiable until a preference order is specified for the properties on which it operates . Background train +Arabic has two kinds of plurals : broken plurals and sound plurals ( Wightwick and Gaafar , 1998 ; ) . Background val +Secondly , as ( ) show , marginalizing out the different segmentations during decoding leads to improved performance . Future train +Nivre ( 2008 ) reports experiments on Arabic parsing using his MaltParser ( ) , trained on the PADT . Background train +More recently , ( Sebastiani , 2002 ) has performed a good survey of document categorization ; recent works can also be found in ( ) , ( Crammer and Singer , 2003 ) , and ( Lewis et al. , 2004 ) . Background train +The grammar conversion from LTAG to HPSG ( ) is the core portion of the RenTAL system . Background train +This includes work on question answering ( ) , sentiment analysis ( Nakagawa et al. , 2010 ) , MT reordering ( Xu et al. , 2009 ) , and many other tasks . Background val +Our knowledge extractors rely extensively on MetaMap ( ) , a system for identifying segments of text that correspond to concepts in the UMLS Metathesaurus . Uses train +We study the cases where a 9Recall that even the system , built on the world 's largest corpus , achieves only 34 % recall ( Table 1 ) ( with only 48 % of positives and 27 % of all pairs previously observed , but see Footnote 5 ) . CompareOrContrast train +Inspired by ( Blunsom et al. , 2009 ) and ( ) , we define P ( str | frag ) as follows : where csw is the number of words in the source string . Motivation train +Inspired by ( ) and ( Cohn and Blunsom , 2009 ) , we define P ( str | frag ) as follows : where csw is the number of words in the source string . Motivation train +The TNT POS tagger ( ) has also been designed to train and run very quickly , tagging between 30,000 and 60,000 words per second . Background train +Some previous works ( Bannard and Callison-Burch , 2005 ; ; Kouylekov et al. , 2009 ) indicate , as main limitations of the mentioned resources , their limited coverage , their low precision , and the fact that they are mostly suitable to capture relations mainly between single words . Background val +Similarly , report that the use of a subset of Spanish morphological features ( number for adjectives , determiners , nouns , pronouns , and verbs ; and mode for verbs ) outperforms other combinations . Background train +To solve these scaling issues , we implement Online Variational Bayesian Inference ( ; Hoffman et al. , 2012 ) for our models . Uses train +Other studies which view lR as a query generation process include ; Hiemstra and Kraaij , 1999 ; Ponte and Croft , 1998 ; Miller et al , 1999 . CompareOrContrast train +As a result , researchers have re-adopted the once-popular knowledge-rich approach , investigating a variety of semantic knowledge sources for common noun resolution , such as the semantic relations between two NPs ( e.g. , Ji et al. ( 2005 ) ) , their semantic similarity as computed using WordNet ( e.g. , Poesio et al. ( 2004 ) ) or Wikipedia ( ) , and the contextual role played by an NP ( see Bean and Riloff ( 2004 ) ) . Background train +There have been many studies on parsing techniques ( ; Flickinger et al. , 2000 ) , ones on disambiguation models ( Chiang , 2000 ; Kanayama et al. , 2000 ) , and ones on programming/grammar-development environ - Background train +An example of this is the estimation of maximum entropy models , from simple iterative estimation algorithms used by Ratnaparkhi ( 1998 ) that converge very slowly , to complex techniques from the optimisation literature that converge much more rapidly ( ) . Background train +For example , such schema can serve as a mean to represent translation examples , or find structural correspondences for the purpose of transfer grammar learning ( Menezes & Richardson , 2001 ) , ( Aramaki et al. , 2001 ) , ( Watanabe et al. , 2000 ) , ( Meyers et al. , 2000 ) , ( Matsumoto et al. , 1993 ) , ( kaji et al. , 1992 ) , and example-base machine translation EBMT3 ( Sato & Nagao , 1990 ) , ( ) , ( Richardson et al. , 2001 ) , ( Al-Adhaileh & Tang , 1999 ) . Background train +Later works , such as a ) , Bolshakov ( 2004 ) , Taskiran et al. ( 2006 ) and Topkara et al. ( 2006b ) , further made use of part-ofspeech taggers and electronic dictionaries , such as WordNet and VerbNet , to increase the robustness of the method . Background train +results are based on a corpus of movie subtitles ( Tiedemann 2007 ) , and are consequently shorter sentences , whereas the En → Es results are based on a corpus of parliamentary proceedings ( ) . Uses train +We work with a semi-technical text on meteorological phenomena ( ) , meant for primary school students . Uses train +This Principle of Finitism is also assumed by Johnson-Laird ( 1983 ) , Jackendoff ( 1983 ) , , and implicitly or explicitly by almost all researchers in computational linguistics . CompareOrContrast train +The candidate feature templates include : Voice from . Uses train +Over the past decade , researchers at IBM have developed a series of increasingly sophisticated statistical models for machine translation ( Brown et al. , 1988 ; Brown et al. , 1990 ; a ) . Background train +The PICO framework ( ) for capturing well-formulated clinical queries ( described in Section 2 ) can serve as the basis of a knowledge representation that bridges the needs of clinicians and analytical capabilities of a system . Background train +The Chinese PropBank has labeled the predicateargument structures of sentences from the Chinese TreeBank ( ) . Uses val +Furthermore , manually selected word pairs are often biased towards highly related pairs ( ) , because human annotators tend to select only highly related pairs connected by relations they are aware of . Background train +Disjunctive feature descriptions are also possible ; WIT incorporates an efficient method for handling disjunctions ( ) . Uses train +Our strategy is based on the approach presented by . Uses train +Indeed , contrary to the more classical statistical methods ( Mutual Information , Loglike ... , see below ) used for collocation acquisition ( see ( ) for a review ) , these patterns allow : Background train +A further complication is that different speakers can regard very different values as prototypical , making it difficult to assess which of two objects is greener even on one dimension ( , pages 10 -- 12 ) . Background train +The M step then treats c as fixed , observed data and adjusts 0 until the predicted vector of total feature counts equals c , using Improved Iterative Scaling ( Della ; Chen and Uses train +Both kinds of annotation were carried out using ANVIL ( ) . Uses train +The article classifier is a discriminative model that draws on the state-of-the-art approach described in . Uses train +The language grounding problem has received significant attention in recent years , owed in part to the wide availability of data sets ( e.g. Flickr , Von Ahn ( 2006 ) ) , computing power , improved computer vision models ( ; Lowe , 2004 ; Farhadi et al. , 2009 ; Parikh and Grauman , 2011 ) and neurological evidence of ties between the language , perceptual and motor systems in the brain ( Pulverm ¨ uller et al. , 2005 ; Tettamanti et al. , 2005 ; Aziz-Zadeh et al. , 2006 ) . Background val + , p. 294 ) studied , among other simple text normalization techniques , the effect of case normalization for different words and showed that `` sometimes case variants refer to the same thing ( hurricane and Hurricane ) , sometimes they refer to different things ( continental and Continental ) and sometimes they do n't refer to much of anything ( e.g. , anytime and Anytime ) . '' Background train +We follow the notation convention of . Uses train +Later works , such as Atallah et al. ( 2001a ) , Bolshakov ( 2004 ) , Taskiran et al. ( 2006 ) and b ) , further made use of part-ofspeech taggers and electronic dictionaries , such as WordNet and VerbNet , to increase the robustness of the method . Background train + did not report inter-subject correlation for their larger dataset . CompareOrContrast train +This is a similar conclusion to our previous work in . CompareOrContrast train +We then use the program Snob ( ; Wallace 2005 ) to cluster these experiences . Uses train +The priorities are used for disambiguating interpretation in the incremental understanding method ( b ) . Uses train +A number of alignment techniques have been proposed , varying from statistical methods ( Brown et al. , 1991 ; ) to lexical methods ( Kay and Roscheisen , 1993 ; Chen , 1993 ) . Background train + present an illustrative first-order fragment along these lines and are able to supply a coherent formal semantics for the CLF-QLFs themselves , using a technique essentially equivalent to supervaluations : a QLF is true iff all its possible RQLFs are , false iff they are all false , and undefined otherwise . Background val +We use the open-source Moses toolkit ( ) to build a phrase-based SMT system trained on mostly MSA data ( 64M words on the Arabic side ) obtained from several LDC corpora including some limited DA data . Uses train +This approach is taken , for example , in LKB ( ) where lexical rules are introduced on a par with phrase structure rules and the parser makes no distinction between lexical and nonlexical rules ( Copestake 1993 , 31 ) . CompareOrContrast train +This confirms that although Kozima 's approach ( ) is computationally expensive , it does produce more precise segmentation . CompareOrContrast train + recently described a hybrid method for finding abbreviations and their definitions . Background train +More specifically , the notion of the phrasal lexicon ( used first by ) has been used successfully in a number of areas : Background train + run a finite-state NP parser on a POS-tagged corpus to calculate the relative frequency of the same six subcategorization verb classes . Background train +This section , which elaborates on preliminary results reported in , describes extraction algorithms for population , problems , interventions , outcomes , and the strength of evidence . Extends train +It would seem therefore that the iteration of the PT operation to form a closure is needed ( cfXXX b ) . CompareOrContrast train + asked subjects to identify the target of a vague description in a visual scene . Background train +Our most accurate single grammar achieves an F score of 91.6 on the WSJ test set , rivaling discriminative reranking approaches ( Charniak and Johnson , 2005 ) and products of latent variable grammars ( ) , despite being a single generative PCFG . CompareOrContrast train +Our recovery policy is modeled on the TargetedHelp ( ) policy used in task-oriented dialogue . Extends train +It has been argued that , in an incremental approach , gradable properties should be given a low preference ranking because they are difficult to process ( ) . CompareOrContrast train +` See ( ) for a discussion of the appropriateness of TIG for HPSG and a comparison with other feature logic approaches designed for HPSG . Background train +Specifically , we used Decision Graphs ( ) for Doc-Pred , and SVMs ( Vapnik 1998 ) for Sent-Pred .11 Additionally , we used unigrams for clustering documents and sentences , and unigrams and bigrams for predicting document clusters and sentence clusters ( Sections 3.1.2 and 3.2.2 ) . Uses val +There are many plausible representations , such as pairs of trees from synchronous tree adjoining grammars ( Abeille et al. 1990 ; Shieber 1994 ; ) , lexical conceptual structures ( Dorr 1992 ) and WordNet synsets ( Fellbaum 1998 ; Vossen 1998 ) . Background train +ones , DIRT ( Lin and Pantel , 2001 ) , VerbOcean ( Chklovski and Pantel , 2004 ) , FrameNet ( ) , and Wikipedia ( Mehdad et al. , 2010 ; Kouylekov et al. , 2009 ) . Background train +In the latter case , we can also take care of transferring the value of z. However , as discussed by , creating several instances of lexical rules can be avoided . Motivation train +It maximizes the probability of getting the entire DA sequence correct , but it does not necessarily find the DA sequence that has the most DA labels correct ( ) . Background train +A substring in the sentence that corresponds to a node in the representation tree is denoted by assigning the interval of the substring to SNODE of 2 These definitions are based on the discussion in ( Tang , 1994 ) and . Uses train +We found that the oldest system ( ) yielded the best prototypes , and that using these prototypes gave state-of-the-art performance on WSJ , as well as improvements on nearly all of the non-English corpora . Background train +Other definitions of predicates may be found in ( ) . Background train +For the sake of completeness , we report in this section also the results obtained adopting the `` basic solution '' proposed by ( ) . CompareOrContrast train +The representations used by Danlos ( 2000 ) , Gardent and Webber ( 1998 ) , or are similar , but do not ( always ) explicitly represent the clause combining operations as labeled nodes . Background train +Since earlier versions of the SNoW based CSCL were used only to identify single phrases ( ; Munoz et al. , 1999 ) and never to identify a collection of several phrases at the same time , as we do here , we also trained and tested it under the exact conditions of CoNLL-2000 ( Tjong Kim Sang and Buchholz , 2000 ) to compare it to other shallow parsers . Extends train +If differences in meaning between senses are very fine-grained , distinguishing between them is hard even for humans ( ) .6 Pairs containing such words are not suitable for evaluation . Background train +The application of domain models and deep semantic knowledge to question answering has been explored by a variety of researchers ( e.g. , Jacquemart and Zweigenbaum 2003 , ) , and was also the focus of recent workshops on question answering in restricted domains at ACL 2004 and AAAI 2005 . Background train +Griffiths et al. ( 2007 ) helped pave the path for cognitive-linguistic multimodal research , showing that Latent Dirichlet Allocation outperformed Latent Semantic Analysis ( ) in the prediction of association norms . Background train +Although not the first to employ a generative approach to directly model content , the seminal work of is a noteworthy point of reference and comparison . CompareOrContrast train +Others include selectional preferences , transitivity ( ) , mutual exclusion , symmetry , etc. . Background train +• cross-language information retrieval ( e.g. , ) , • multilingual document filtering ( e.g. , Oard 1997 ) , • computer-assisted language learning ( e.g. , Nerbonne et al. 1997 ) , • certain machine-assisted translation tools ( e.g. , Macklovitch 1994 ; Melamed 1996a ) , • concordancing for bilingual lexicography ( e.g. , Catizone , Russell , and Warwick 1989 ; Gale and Church 1991 ) , Background val +For example , ( ) discusses the evaluation of two different text categorization strategies with several variations of their feature spaces . Background train +As stated before , the experiments are run in the ACE '04 framework ( ) where the system will identify mentions and will label them ( cfXXX Section 4 ) with a type ( person , organization , etc ) , a sub-type ( OrgCommercial , OrgGovernmental , etc ) , a mention level ( named , nominal , etc ) , and a class ( specific , generic , etc ) . Uses train +Thus , the second class of SBD systems employs machine learning techniques such as decision tree classifiers ( Riley 1989 ) , neural networks ( Palmer and Hearst 1994 ) , and maximum-entropy modeling ( ) . Background train +or quotation of messages in emails or postings ( see but cfXXX Agrawal et al. ( 2003 ) ) . Background train +The first work to do this with topic models is b ) . Background train +The language chosen for semantic representation is a flat semantics along the line of ( Bos , 1995 ; Copestake et al. , 1999 ; ) . CompareOrContrast train +Tetreault 's contribution features comparative evaluation involving the author 's own centering-based pronoun resolution algorithm called the Left-Right Centering algorithm ( LRC ) as well as three other pronoun resolution methods : Hobbs 's naive algorithm ( ) , BFP ( Brennan , Friedman , and Pollard 1987 ) , and Strube 's 5list approach ( Strube 1998 ) . Background train +For future work , we might investigate how machine learning algorithms , which are specifically designed for the problem of domain adaptation ( ; Jiang and Zhai , 2007 ) , perform in comparison to our approach . Future train +The X2 statistic is performing at least as well as G2 , throwing doubt on the claim by that the G2 statistic is better suited for use in corpus-based NLP . CompareOrContrast train +Provided with the candidate fragment elements , we previously ( ) used a chunker3 to finalize the output fragments , in order to follow the linguistic definition of a ( para - ) phrase . Extends train +There is a rich literature on organization and lexical access of morphologically complex words where experiments have been conducted mainly for derivational suffixed words of English , Hebrew , Italian , French , Dutch , and few other languages ( Marslen-Wilson et al. , 2008 ; Frost et al. , 1997 ; Grainger , et al. , 1991 ; ) . Background train +This method of incorporating dictionary information seems simpler than the method proposed by Brown et al. for their models ( b ) . CompareOrContrast train +One important example is the constituentcontext model ( CCM ) of , which was specifically designed to capture the linguistic observation made by Radford ( 1988 ) that there are regularities to the contexts in which constituents appear . Background train +Japanese ( ) , despite a very high accuracy , is different in that attachment score drops from 98 % to 85 % , as we go from length 1 to 2 , which may have something to do with the data consisting of transcribed speech with very short utterances . CompareOrContrast train +7 We ignore the rare `` false idafa '' construction ( , p. 102 ) . Background train +Various approaches for computing semantic relatedness of words or concepts have been proposed , e.g. dictionary-based ( Lesk , 1986 ) , ontology-based ( ; Leacock and Chodorow , 1998 ) , information-based ( Resnik , 1995 ; Jiang and Conrath , 1997 ) or distributional ( Weeds and Weir , 2005 ) . Background train +Some methods are based on likelihood ( Och and Ney , 2002 ; Blunsom et al. , 2008 ) , error rate ( Och , 2003 ; Zhao and Chen , 2009 ; Pauls et al. , 2009 ; Galley and Quirk , 2011 ) , margin ( Watanabe et al. , 2007 ; Chiang et al. , 2008 ) and ranking ( ) , and among which minimum error rate training ( MERT ) ( Och , 2003 ) is the most popular one . Background train +We follow in allowing a small set of generic , linguistically-plausible unary and binary grammar rules . Uses train +( 7 ) NEIGHBOR : Research in lexical semantics suggests that the SC of an NP can be inferred from its distributionally similar NPs ( see a ) ) . Motivation val +Discrepancies in length throw constituents off balance , and so prosodic phrasing will cross constituent boundaries in order to give the phrases similar lengths ; this is the case in Chickens were eating II the remaining green vegetables , where the subject-predicate boundary finds no prosodic correspondent .4 The most explicit version of this approach is the analysis presented in ( henceforth G&G ) . CompareOrContrast train +Our approach to the problem is more compatible with the empirical evidence we presented in our prior work ( ) where we analyzed the output of Chinese to English machine translation and found that there is no correlation between sentence length and MT quality . CompareOrContrast train +For all experiments reported in this section we used the syntactic dependency parser MaltParser v1 .3 ( , 2008 ; Kübler , McDonald , and Nivre 2009 ) , a transition-based parser with an input buffer and a stack , which uses SVM classifiers Uses train +Against the background of a growing interest in multilingual NLP , multilingual anaphora / coreference resolution has gained considerable momentum in recent years ( Aone and McKee 1993 ; Azzam , Humphreys , and Gaizauskas 1998 ; Harabagiu and Maiorano 2000 ; Mitkov and Barbu 2000 ; Mitkov 1999 ; ; Mitkov , Belguith , and Stys 1998 ) . Background train +We use the same data setting with Xue ( 2008 ) , however a bit different from . CompareOrContrast train +They proved to be useful in a number of NLP applications such as natural language generation ( Iordanskaja et al. , 1991 ) , multidocument summarization ( McKeown et al. , 2002 ) , automatic evaluation of MT ( ) , and TE ( Dinu and Wang , 2009 ) . Motivation train +Moreover , a sandbox is a temporary view of a document itself i.e. a sandbox can not cause a change in the history ( ) . Background train +Regarding future work , there are many research line that may be followed : i ) Capturing more features by employing external knowledge such as ontological , lexical resource or WordNet-based features ( a ; Basili et al. , 2005b ; Bloehdorn et al. , 2006 ; Bloehdorn and Moschitti , 2007 ) or shallow semantic trees , ( Giuglea and Moschitti , 2004 ; Giuglea and Moschitti , 2006 ; Moschitti and Bejan , 2004 ; Moschitti et al. , 2007 ; Moschitti , 2008 ; Moschitti et al. , 2008 ) . Future val +Another line of research approaches grounded language knowledge by augmenting distributional approaches of word meaning with perceptual information ( Andrews et al. , 2009 ; Steyvers , 2010 ; Feng and Lapata , 2010b ; Bruni et al. , 2011 ; Silberer and Lapata , 2012 ; ; Bruni et al. , 2012a ; Bruni et al. , 2012b ; Silberer et al. , 2013 ) . Background train +Future research should apply the work of Blunsom et al. ( 2008 ) and , who marginalize over derivations to find the most probable translation rather than the most probable derivation , to these multi-nonterminal grammars . Future train +We have since improved the interface by incorporating a capability in the recognizer to propose additional solutions in turn once the first one fails to parse ( ) To produce these `` N-best '' alternatives , we make use of a standard A * search algorithm ( Hart 1968 , Jelinek 1976 ) . Uses train + Background train +OT therefore holds out the promise of simplifying grammars , by factoring all complex phenomena into simple surface-level constraints that partially mask one another .1 Whether this is always possible under an appropriate definition of `` simple constraints '' ( e.g. , b ) is of course an empirical question . Background train +Consider , for example , the lexical rule in Figure 2 , which encodes a passive lexical rule like the one presented by Pollard and Sag ( 1987 , 215 ) in terms of the setup of , ch . Background train +This result is consistent with other works using this model with these features ( ; Silberer and Lapata , 2012 ) . CompareOrContrast train + reported a correlation of r = .9026.10 The results are not directly comparable , because he only used noun-noun pairs , words instead of concepts , a much smaller dataset , and measured semantic similarity instead of semantic relatedness . CompareOrContrast test +Similar observation for surface word frequency was also observed by ( Bertram et al. , 2000 ; ; Burani et al. , 1987 ; Burani et al. , 1984 ; Schreuder et al. , 1997 ; Taft 1975 ; Taft , 2004 ) where it has been claimed that words having low surface frequency tends to decompose . Background test +But their importance has grown far beyond machine translation : for instance , transferring annotations between languages ( ; Hwa et al. 2005 ; Ganchev , Gillenwater , and Taskar 2009 ) ; discovery of paraphrases ( Bannard and Callison-Burch 2005 ) ; and joint unsupervised POS and parser induction across languages ( Snyder and Barzilay 2008 ) . Motivation test +Previous sentiment-analysis work in different domains has considered inter-document similarity ( Agarwal and Bhattacharyya , 2005 ; Pang and Lee , 2005 ; ) or explicit Background test +However , the method we are currently using in the ATIS domain ( ) represents our most promising approach to this problem . Uses test +Henceforth the collaborative traits of blogs and wikis ( ) emphasize annotation , comment , and strong editing . Background test +The ICA system ( ) aims to reduce the training time by introducing independence assumptions on the training samples that dramatically reduce the training time with the possible downside of sacrificing performance . Background test +To this end , several toolkits for building spoken dialogue systems have been developed ( Barnett and Singh , 1997 ; ) . Background test +Thus , over the past few years , along with advances in the use of learning and statistical methods for acquisition of full parsers ( Collins , 1997 ; Charniak , 1997a ; Charniak , 1997b ; Ratnaparkhi , 1997 ) , significant progress has been made on the use of statistical learning methods to recognize shallow parsing patterns syntactic phrases or words that participate in a syntactic relationship ( Church , 1988 ; Ramshaw and Marcus , 1995 ; Argamon et al. , 1998 ; Cardie and Pierce , 1998 ; Munoz et al. , 1999 ; Punyakanok and Roth , 2001 ; ; Tjong Kim Sang and Buchholz , 2000 ) . Background test +Task properties Determining whether or not a speaker supports a proposal falls within the realm of sentiment analysis , an extremely active research area devoted to the computational treatment of subjective or opinion-oriented language ( early work includes Wiebe and Rapaport ( 1988 ) , , Sack ( 1994 ) , and Wiebe ( 1994 ) ; see Esuli ( 2006 ) for an active bibliography ) . Background test +Various approaches for computing semantic relatedness of words or concepts have been proposed , e.g. dictionary-based ( Lesk , 1986 ) , ontology-based ( Wu and Palmer , 1994 ; Leacock and Chodorow , 1998 ) , information-based ( ; Jiang and Conrath , 1997 ) or distributional ( Weeds and Weir , 2005 ) . Background test +Both tasks are performed with a statistical framework : the mention detection system is similar to the one presented in ( Florian et al. , 2004 ) and the coreference resolution system is similar to the one described in ( ) . CompareOrContrast test +The advantage of tuning similarity to the application of interest has been shown previously by . CompareOrContrast test +Although there are other discussions of the paragraph as a central element of discourse ( e.g. , Halliday and Hasan 1976 , Longacre 1979 , Haberlandt et al. 1980 ) , all of them share a certain limitation in their formal techniques for analyzing paragraph structure . CompareOrContrast test +Thus , over the past few years , along with advances in the use of learning and statistical methods for acquisition of full parsers ( Collins , 1997 ; Charniak , 1997a ; Charniak , 1997b ; ) , significant progress has been made on the use of statistical learning methods to recognize shallow parsing patterns syntactic phrases or words that participate in a syntactic relationship ( Church , 1988 ; Ramshaw and Marcus , 1995 ; Argamon et al. , 1998 ; Cardie and Pierce , 1998 ; Munoz et al. , 1999 ; Punyakanok and Roth , 2001 ; Buchholz et al. , 1999 ; Tjong Kim Sang and Buchholz , 2000 ) . Background test +We experiment with four learners commonly employed in language learning : Decision List ( DL ) : We use the DL learner as described in Collins and Singer ( 1999 ) , motivated by its success in the related tasks of word sense disambiguation ( ) and NE classification ( Collins and Singer , 1999 ) . Motivation test +A central technique is to define a joint relation as a noisy-channel model , by composing a joint relation with a cascade of one or more conditional relations as in Fig. 1 ( ; Knight and Graehl , 1998 ) . Background test +We use the same set of binary features as in previous work on this dataset ( Pang et al. , 2002 ; ; Zaidan et al. , 2007 ) . Uses test +Our classification framework , directly inspired by , integrates both perspectives , optimizing its labeling of speech segments based on both individual speech-segment classification scores and preferences for groups of speech segments to receive the same label . Uses test +As for work on Arabic ( MSA ) , results have been reported on the PATB ( Kulick , Gabbard , and Marcus 2006 ; Diab 2007 ; ) , the Prague Dependency Treebank ( PADT ) ( Buchholz and Marsi 2006 ; Nivre 2008 ) and the CATiB ( Habash and Roth 2009 ) . Background test +For instance , report that the SATZ system ( decision tree variant ) was trained on a set of about 800 labeled periods , which corresponds to a corpus of about 16,000 words . CompareOrContrast test +One possible direction is to consider linguistically motivated approaches , such as the extraction of syntactic phrase tables as proposed by ( ) . Future test +Later works , such as Atallah et al. ( 2001a ) , , Taskiran et al. ( 2006 ) and Topkara et al. ( 2006b ) , further made use of part-ofspeech taggers and electronic dictionaries , such as WordNet and VerbNet , to increase the robustness of the method . Background test +A number of speech understanding systems have been developed during the past fifteen years ( Barnett et al. 1980 , Dixon and Martin 1979 , Erman et al. 1980 , , Lea 1980 , Lowerre and Reddy 1980 , Medress 1980 , Reddy 1976 , Walker 1978 , and Wolf and Woods 1980 ) . CompareOrContrast test +The bottom panel of table 1 lists the results for the chosen lexicalized model ( SSN-Freq > 200 ) and five recent statistical parsers ( Ratnaparkhi , 1999 ; Collins , 1999 ; Charniak , 2000 ; ; Bod , 2001 ) . CompareOrContrast test +The basic Python reflection has already been implemented and used for large scale experiments with POS tagging , using pyMPI ( a message passing interface library for Python ) to coordinate experiments across a cluster of over 100 machines ( ; Clark et al. , 2003 ) . Background test +This imbalance foils thresholding strategies , clever as they might be ( Gale & Church , 1991 ; Wu & Xia , 1994 ; ) . Background test +Training was done on the Penn Treebank ( ) Wall Street Journal data , sections 02-21 . Uses test +We performed Latent Semantic Analysis ( LSA ) over Wikipedia using the jLSI tool ( ) to measure the relatedness between words in the dataset . Uses test +For example , our previous work ( ; Nakov and Ng , 2012 ) experimented with various techniques for combining a small bi-text for a resource-poor language ( Indonesian or Spanish , pretending that Spanish is resource-poor ) with a much larger bi-text for a related resource-rich language ( Malay or Portuguese ) ; the target language of all bi-texts was English . CompareOrContrast test +Various approaches for computing semantic relatedness of words or concepts have been proposed , e.g. dictionary-based ( ) , ontology-based ( Wu and Palmer , 1994 ; Leacock and Chodorow , 1998 ) , information-based ( Resnik , 1995 ; Jiang and Conrath , 1997 ) or distributional ( Weeds and Weir , 2005 ) . Background test +Another line of research approaches grounded language knowledge by augmenting distributional approaches of word meaning with perceptual information ( Andrews et al. , 2009 ; Steyvers , 2010 ; Feng and Lapata , 2010b ; Bruni et al. , 2011 ; Silberer and Lapata , 2012 ; Johns and Jones , 2012 ; a ; Bruni et al. , 2012b ; Silberer et al. , 2013 ) . Background test + replicated the experiment of Rubenstein and Goodenough with the original 65 word pairs translated into German . Background test +One approach to this more general problem , taken by the ` Nitrogen ' generator ( a ; Langkilde and Knight , 1998b ) , takes advantage of standard statistical techniques by generating a lattice of all possible strings given a semantic representation as input and selecting the most likely output using a bigram language model . Uses test +where mk is one mention in entity e , and the basic model building block PL ( L = 1 | e , mk , m ) is an exponential or maximum entropy model ( ) . Uses test +13 We also employed sequence-based measures using the ROUGE tool set ( ) , with similar results to those obtained with the word-by-word measures . Uses test +Second , using continuous distributions allows us to leverage a variety of tools ( e.g. , LDA ) that have been shown to be successful in other fields , such as speech recognition ( ) . Background test +In this section , we validate the contribution of key tag sets and morphological features -- and combinations thereof -- using a different parser : the Easy-First Parser ( ) . Uses test +The typical solution to the redundancy problem is to group verbs according to their argument realization patterns ( ) , possibly arranged in an inheritance hierarchy . CompareOrContrast test +Later , , 1982 ) proposed a knowledge base in which information about language and the world would be encoded , and he emphasized the need for using `` salience '' in choosing facts from this knowledge base . Background test +Another technique is automatic discovery of translations from parallel or non-parallel corpora ( ) . Background test +ASARES is presented in detail in ( ) . Uses test +Opposition ( called `` adversative '' or `` contrary-to-expectation '' by ; cfXXX also Quirk et al. 1972 , p. 672 ) . Background test +A number of applications have relied on distributional analysis ( ) in order to build classes of semantically related terms . Background test +Previous work with MaltParser in Russian , Turkish , and Hindi showed gains with CASE but not with agreement features ( Eryigit , Nivre , and Oflazer 2008 ; Nivre , Boguslavsky , and Iomdin 2008 ; ) . CompareOrContrast test +Consider , for example , the lexical rule in Figure 2 , which encodes a passive lexical rule like the one presented by , 215 ) in terms of the setup of Pollard and Sag ( 1994 , ch . CompareOrContrast test +Two applications that , like help-desk , deal with question -- answer pairs are : summarization of e-mail threads ( Dalli , Xia , and Wilks 2004 ; ) , and answer extraction in FAQs ( Frequently Asked Questions ) ( Berger and Mittal 2000 ; CompareOrContrast test +The language grounding problem has received significant attention in recent years , owed in part to the wide availability of data sets ( e.g. Flickr , Von Ahn ( 2006 ) ) , computing power , improved computer vision models ( Oliva and Torralba , 2001 ; Lowe , 2004 ; ; Parikh and Grauman , 2011 ) and neurological evidence of ties between the language , perceptual and motor systems in the brain ( Pulverm ¨ uller et al. , 2005 ; Tettamanti et al. , 2005 ; Aziz-Zadeh et al. , 2006 ) . Background test +In addition , we find that the Bayesian SCFG grammar can not even significantly outperform the heuristic SCFG grammar ( ) 5 . CompareOrContrast test +There are several grammars developed in the FB-LTAG formalism , including the XTAG English grammar , a large-scale grammar for English ( The XTAG Research ) . Background test +Although the approach may have potential , the shifting of complex accounting into the unification algorithm is at variance with the findings of , who report large speed-ups from the elimination of disjunction processing during unification . CompareOrContrast test +For the task of unsupervised dependency parsing , add a constraint of the form `` the average length of dependencies should be X '' to capture the locality of syntax ( at least half of the dependencies are between adjacent words ) , using a scheme they call structural annealing . Background test +The speech and language processing architecture is based on that of the SRI CommandTalk system ( ; Stent et a. , 1999 ) . Uses test +Second , in line with the findings of ( ) , the results obtained over the MT-derived corpus are equal to those we achieve over the original RTE3 dataset ( i.e. 63.50 % ) . CompareOrContrast test +Therefore , inter-subject correlation is lower than the results obtained by . CompareOrContrast test +There is a general consensus among theoretical linguists that the proper representation of verbal argument structure is event structure -- representations grounded in a theory of events that decompose semantic roles in terms of primitive predicates representing concepts such as causality and inchoativity ( Dowty , 1979 ; Jackendoff , 1983 ; Pustejovsky , 1991b ; Rappaport ) . Background test +For example , some similar measures have been used in stylistic experiments in information retrieval on the basis of a robust parser built for information retrieval purposes ( ) . Background test +The resulting training procedure is analogous to the one presented in ( Brown et al. , 1993 ) and ( ) . CompareOrContrast test +successfully parses , or until a quitting criterion is reached , such as an upper bound on N. Whereas in the loosely coupled system the parser acts as a filter only on completed candidate solutions ( ) , the tightly coupled system allows the parser to discard partial theories that have no way of continuing . Uses test + substituted the non-terminal X in hierarchical phrase-based model by extended syntactic categories . CompareOrContrast test +Much of the earlier work in anaphora resolution heavily exploited domain and linguistic knowledge ( ; Carter 1987 ; Rich and LuperFoy 1988 ; Carbonell and Brown 1988 ) , which was difficult both to represent and to process , and which required considerable human input . Background test +The paradigm is `` write many , read many '' ( ) . Background test +The Praat tool was used ( ) . Uses test +2 The reader is asked to focus on any reasonable size measurement , for example , the maximal horizontal or vertical distance , or some combination of dimensions ( ; also Section 8.1 of the present article ) . Background test +The implementation has been inspired by experience in extracting information from very large corpora ( Curran and Moens , 2002 ) and performing experiments on maximum entropy sequence tagging ( ; Clark et al. , 2003 ) . Motivation test +Default parameters were used , although experimentation with different parameter settings is an important direction for future work ( Daelemans and Hoste , 2002 ; ) . Future test +Our work is inspired by the latent left-linking model in and the ILP formulation from Chang et al. ( 2011 ) . Uses test +Furthermore , the availability of rich ontological resources , in the form of the Unified Medical Language System ( UMLS ) ( Lindberg et al. , 1993 ) , and the availability of software that leverages this knowledge -- MetaMap ( Aronson , 2001 ) for concept identification and SemRep ( ) for relation extraction -- provide a foundation for studying the role of semantics in various tasks . Background test +The names given to the components vary ; they have been called `` strategic '' and `` tactical '' components ( e.g. , McKeown 1985 ; Thompson 1977 ; Danlos 1987 ) 1 , `` planning '' and `` realization '' ( e.g. , McDonald 1983 ; a ) , or simply `` what to say '' versus `` how to say it '' ( e.g. , Danlos 1987 ; Reithinger 1990 ) . Background test +Over the last decade there has been a lot of interest in developing tutorial dialogue systems that understand student explanations ( Jordan et al. , 2006 ; Graesser et al. , 1999 ; Aleven et al. , 2001 ; ; Nielsen et al. , 2008 ; VanLehn et al. , 2007 ) , because high percentages of selfexplanation and student contentful talk are known to be correlated with better learning in humanhuman tutoring ( Chi et al. , 1994 ; Litman et al. , 2009 ; Purandare and Litman , 2008 ; Steinhauser et al. , 2007 ) . Background test +We use the TRIPS dialogue parser ( ) to parse the utterances . Uses test +In order to address these limitations in a practical way , we conducted a small user study where we asked four judges ( graduate students from the Faculty of Information Technology at Monash University ) to assess the responses generated by our system ( a ) . Uses test +The understanding module utilizes ISSS ( Incremental Significant-utterance Sequence Search ) ( b ) , which is an integrated parsing and discourse processing method . Uses test +We applied our system to the XTAG English grammar ( The XTAG Research ) 3 , which is a large-scale FB-LTAG grammar for English . Uses test +After the extraction , pruning techniques ( ) can be applied to increase the precision of the extracted paraphrases . Background test +In this paper , we extend two classes of model adaptation methods ( i.e. , model interpolation and error-driven learning ) , which have been well studied in statistical language modeling for speech and natural language applications ( e.g. , ; Bellegarda , 2004 ; Gao et al. , 2006 ) , to ranking models for Web search applications . Background test +GATE goes beyond earlier systems by using a component-based infrastructure ( ) which the GUI is built on top of . Background test +Since sentences can refer to events described by other sentences , we may need also a quotation operator ; describes how first order logic can be augmented with such an operator . Background test +The system uses a knowledge base implemented in the KM representation language ( Clark and Porter , 1999 ; ) to represent the state of the world . Uses test +A possible future direction would be to compare the query string to retrieved results using a method similar to that of . Future test +description-level lexical rules ( DLRs ; ) .5 2.2.1 Meta-Level Lexical Rules . Background test +All EBMT systems , from the initial proposal by to the recent collection of Carl and Way ( 2003 ) , are premised on the availability of subsentential alignments derived from the input bitext . Background test +The necessity of this kind of merging of arguments has been recognized before : Charniak and McDermott ( 1985 ) call it abductive unification/matching , , 1979 ) refers to such operations using the terms knitting or petty conversational implicature . Background test +In a number of proposals , lexical generalizations are captured using lexical underspecification ( Kathol 1994 ; ; CompareOrContrast test +These keywords are potentially useful features because some of them are subclasses of the ACE SCs shown in the left column of Table 1 , while others appear to be correlated with these ACE SCs .2 ( 6 ) INDUCED CLASS : Since the first-sense heuristic used in the previous feature may not be accurate in capturing the SC of an NP , we employ a corpusbased method for inducing SCs that is motivated by research in lexical semantics ( e.g. , ) . Motivation test +Other psycholing-uistic studies that confirm the validity of paragraph units can be found in and Haberlandt et al. ( 1980 ) . Background test +The bottom panel of table 1 lists the results for the chosen lexicalized model ( SSN-Freq > 200 ) and five recent statistical parsers ( Ratnaparkhi , 1999 ; Collins , 1999 ; Charniak , 2000 ; Collins , 2000 ; ) . CompareOrContrast test +Nevertheless , the full document text is present in most systems , sometimes as the only feature ( Sugiyama and Okumura , 2007 ) and sometimes in combination with others see for instance ( Chen and Martin , 2007 ; ) - . Background test +In a similar vain to and Buchholz et al. ( 1999 ) , the method extends an existing flat shallow-parsing method to handle composite structures . Future test +As a result , researchers have re-adopted the once-popular knowledge-rich approach , investigating a variety of semantic knowledge sources for common noun resolution , such as the semantic relations between two NPs ( e.g. , ) , their semantic similarity as computed using WordNet ( e.g. , Poesio et al. ( 2004 ) ) or Wikipedia ( Ponzetto and Strube , 2006 ) , and the contextual role played by an NP ( see Bean and Riloff ( 2004 ) ) . Background test +We built a two-stage baseline system , using the perceptron segmentation model from our previous work ( ) and the perceptron POS tagging model from Collins ( 2002 ) . Extends test +Note that although our current system uses MeSH headings assigned by human indexers , manually assigned terms can be replaced with automatic processing if needed ( ) . Future test +Furthermore , medical terminology is characterized by a typical mix of Latin and Greek roots with the corresponding host language ( e.g. , German ) , often referred to as neo-classical compounding ( ) . Background test +Previously ( ) , we assessed the importance of various implicit argument feature groups by conducting feature ablation tests . Extends test +To model d ( FWi − 1 , S → T ) , d ( FWi +1 , S → T ) , i.e. whether Li , S → T and Ri , S → T extend beyond the neighboring function word phrase pairs , we utilize the pairwise dominance model of . Uses test +For instance , , p. 8 ) says that the sentence `` Reagan thinks bananas , '' which is otherwise strange , is in fact acceptable if it occurs as an answer to the question `` What is Kissinger 's favorite fruit ? '' Motivation test +Semantic Role labeling ( SRL ) was first defined in . Background test +AJAX function lets the communication works asyncronously between a client and a server through a set of messages based on HTTP protocol and XML ( ) . Background test +The inclusion of the coreference task in the Sixth and Seventh Message Understanding Conferences ( MUC-6 and MUC-7 ) gave a considerable impetus to the development of coreference resolution algorithms and systems , such as those described in Baldwin et al. ( 1995 ) , Gaizauskas and Humphreys ( 1996 ) , and . Background test +The most detailed evaluation of link tokens to date was performed by ( ) , who trained Brown et al. 's Model 2 on 74 million words of the Canadian Hansards . CompareOrContrast test +Log-linear models have proved successful in a wide variety of applications , and are the inspiration behind one of the best current statistical parsers ( ) . CompareOrContrast test +While we have observed reasonable results with both G2 and Fisher 's exact test , we have not yet discussed how these results compare to the results that can be obtained with a technique commonly used in corpus linguistics based on the mutual information ( MI ) measure ( ) : Background test +Morphological alterations of a search term have a negative impact on the recall performance of an information retrieval ( IR ) system ( ; J ¨ appinen and Niemist ¨ o , 1988 ; Kraaij and Pohlmann , 1996 ) , since they preclude a direct match between the search term proper and its morphological variants in the documents to be retrieved . Background test +For shuffling paraphrases , french alternations are partially described in ( ) and a resource is available which describes alternation and the mapping verbs/alternations for roughly 1 700 verbs . Background test +A more recent approach , advocated by Rappaport Hovav and Levin ( 1998 ) , describes a basic set of event templates corresponding to Vendler 's event classes ( ) : ( 3 ) a. [ x ACT ] ( activity ) b. [ x ] ( state ) c. [ BECOME [ x ] ] ( achievement ) d. [ x CAUSE [ BECOME [ x ] ] ] ( accomplishment ) Background test + combines lexical and dependency mappings to form his generalizations . Background test +Thus for instance , ( Copestake and Flickinger , 2000 ; ) describes a Head Driven Phrase Structure Grammar ( HPSG ) which supports the parallel construction of a phrase structure ( or derived ) tree and of a semantic representation and ( Dalrymple , 1999 ) show how to equip Lexical Functional grammar ( LFG ) with a glue semantics . Background test +The reordering models we describe follow our previous work using function word models for translation ( Setiawan et al. , 2007 ; ) . Extends test +And Collins ( 2000 ) argues for `` keeping track of counts of arbitrary fragments within parse trees '' , which has indeed been carried out in who use exactly the same set of ( all ) tree fragments as proposed in Bod ( 1992 ) . Motivation test +In our work , we gather sets of sentences , and assume ( but do not employ ) existing approaches for their organization ( Goldstein et al. 2000 ; Barzilay , Elhadad , and McKeown 2001 ; ) . Background test +criteria and data used in our experiments are based on the work of . Uses test +We present experiments on the two standard coreference resolution datasets , ACE-2004 ( NIST , 2004 ) and OntoNotes-5 .0 ( ) . Uses test +• Only qualitative observations of the responses were reported ( no formal evaluation was performed ) ( Lapalme and Kosseim 2003 ; ) . CompareOrContrast test +And subderivations headed by A1 with external nonterminals only at the leaves , internal nonterminals elsewhere , have probability 1/a1 ( ) . Background test +• Support vector machines for mapping histories to parser actions ( ) . Uses test + , 1998 ) developed a polynomial time PCFG-reduction of DOP1 whose size is linear in the size of the training set , thus converting the exponential number of subtrees to a compact grammar . Background test + avoids enumerating the various senses for adjectives like fast by exploiting the semantics of the nouns they modify . Background test + have conducted a study on dependency parsing for 21 languages using features that encode whether the values for certain attributes are equal or not for a node and its governor . Background test +Such approaches have been tried recently in restricted cases ( McCallum et al. , 2000 ; b ; Lafferty et al. , 2001 ) . Background test +The relation between discourse and prosodic phrasing has been examined in some detail by , who argues that each noun phrase in an utterance constitutes a separate prosodic phrase unless it is destressed because of reference to previous discourse . Background test +By contrast , Turkish ( Oflazer et al. , 2003 ; ) exhibits high root accuracy but consistently low attachment scores ( about 88 % for length 1 and 68 % for length 2 ) . CompareOrContrast test +The candidate examples that lead to the most disagreements among the different learners are considered to have the highest TUV ( Cohn , Atlas , and Ladner 1994 ; ) . Background test +Subsequently , we extracted the bilingual phrase table from the aligned corpora using the Moses toolkit ( ) . Uses test +Representative systems are described in Boisen et al. ( 1989 ) , De Mattia and Giachin ( 1989 ) , Niedermair ( 1989 ) , Niemann ( 1990 ) , and . Background test +Our rules for phonological word formation are adopted , for the most part , from G & G , , and the account of monosyllabic destressing in Selkirk ( 1984 ) . Uses test +As a generalization , notes that lexicons such as COMLEX tend to demonstrate high precision but low recall . Background test +Such systems extract information from some types of syntactic units ( clauses in ( Fillmore and Atkins , 1998 ; Gildea and Jurafsky , 2002 ; ) ; noun phrases in ( Hull and Gomez , 1996 ; Rosario et al. , 2002 ) ) . Background test +Various approaches for computing semantic relatedness of words or concepts have been proposed , e.g. dictionary-based ( Lesk , 1986 ) , ontology-based ( Wu and Palmer , 1994 ; Leacock and Chodorow , 1998 ) , information-based ( Resnik , 1995 ; ) or distributional ( Weeds and Weir , 2005 ) . Background test +Besides WordNet , the RTE literature documents the use of a variety of lexical information sources ( Bentivogli et al. , 2010 ; ) . Background test +The question answering system developed by belongs to the merging category of approaches , where the output of an individual method can be used as input to a different method ( this corresponds to Burke 's cascade sub-category ) . CompareOrContrast test +More recently , ( Sebastiani , 2002 ) has performed a good survey of document categorization ; recent works can also be found in ( Joachims , 2002 ) , ( ) , and ( Lewis et al. , 2004 ) . Background test +Discriminant analysis has been employed by researchers in automatic text genre detection ( b ; Karlgren and Cutting 1994 ) since it offers a simple and robust solution despite the fact that it presupposes normal distributions of the discriminating variables . Background test +This model has previously been shown to provide excellent performance on multiple tasks , including prediction of association norms , word substitution errors , semantic inferences , and word similarity ( ; Silberer and Lapata , 2012 ) . Extends test +In other words , existing treatments of gradables in GRE fail to take the `` efficiency of language '' into account ( ; see our Section 2 ) . Background test +Word alignments are used primarily for extracting minimal translation units for machine translation ( MT ) ( e.g. , phrases [ Koehn , Och , and Marcu 2003 ] and rules [ ; Chiang et al. 2005 ] ) as well as for Background test +Following , the IR system ranks documents according to the probability that a document D is relevant given the query Q , P ( D is R IQ ) . Uses test +In modern syntactic theories ( e.g. , lexical-functional grammar [ LFG ] [ Kaplan and Bresnan 1982 ; ; Dalrymple 2001 ] , head-driven phrase structure grammar [ HPSG ] [ Pollard and Sag 1994 ] , tree-adjoining grammar [ TAG ] [ Joshi 1988 ] , and combinatory categorial grammar [ CCG ] [ Ades and Steedman 1982 ] ) , the lexicon is the central repository for much morphological , syntactic , and semantic information . Background test +We have shown elsewhere ( Jensen and Binot 1988 ; a , 1987b ) that natural language programs , such as on-line grammars and dictionaries , can be used as referential levels for commonsense reasoning -- for example , to disambiguate PP attachment . Extends test +Thus rather than a single training procedure , we can actually partition the examples by predicate , and train a 1For a fixed verb , MI is proportional to 's conditional probability scores for pseudodisambiguation of ( v , n , n ′ ) triples : Pr ( v | n ) = Pr ( v , n ) / Pr ( n ) , which was shown to be a better measure of association than co-occurrence frequency f ( v , n ) . Motivation test diff --git a/Deep-Citation/Data/class_def.json b/Deep-Citation/Data/class_def.json new file mode 100644 index 0000000000000000000000000000000000000000..d8fd87eedfbd8f97f4dfef925680f769ed7da843 --- /dev/null +++ b/Deep-Citation/Data/class_def.json @@ -0,0 +1,23 @@ +{ + "acl": + { + "BACKGROUND": "The citation provides relevant information for the domain that the present paper discusses.", + "MOTIVATION": "The citation illustrates the need for data, goals, methods, etc that is proposed in the present paper.", + "USES": "The present paper uses data, methods, etc., from the paper associated with the citation.", + "EXTENDS": "The present paper extends the data, methods, etc. from the paper associated with the citation.", + "COMPAREORCONTRAST": "The present paper expresses similarity / differences to the citation.", + "FUTURE": "The citation is a potential avenue for future work of the present paper." + }, + "kim": + { + "Used": "The present paper uses at least one method that is proposed in the paper associated with the citation.", + "Not used": "The present paper does not use or extend any methods that is proposed in the paper associated with the citation.", + "Extended": "The present paper uses an extended / modified version of the method proposed in the paper associated with the citation." + }, + "scicite": + { + "Background": "The citation states, mentions, or points to the background information giving more context about a problem, concept, approach, topic, or importance of the problem that is discussed in the present paper.", + "Method": "The present paper uses a method, tool, approach or dataset that is proposed in the paper associated with the citation.", + "Result": "The present paper compares its results/findings with the results/findings of the paper associated with the citation." + } +} \ No newline at end of file diff --git a/Deep-Citation/Model/__init__.py b/Deep-Citation/Model/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..bf8900204ba12c31f3c8154b0b31e848f691fb1c --- /dev/null +++ b/Deep-Citation/Model/__init__.py @@ -0,0 +1 @@ +from .model import LanguageModel, MultiHeadLanguageModel \ No newline at end of file diff --git a/Deep-Citation/Model/model.py b/Deep-Citation/Model/model.py new file mode 100644 index 0000000000000000000000000000000000000000..8c8f0932b4ad1d426d254a386f67ae931dcc7e2b --- /dev/null +++ b/Deep-Citation/Model/model.py @@ -0,0 +1,89 @@ +import os +import torch +import torch.nn as nn + +from typing import List +from transformers import AutoModel + +def mask_pooling(model_output, attention_mask): + token_embeddings = model_output[0] #First element of model_output contains all token embeddings + input_mask_expanded = attention_mask.unsqueeze(-1).expand(token_embeddings.size()).float() + return torch.sum(token_embeddings * input_mask_expanded, 1) / torch.clamp(input_mask_expanded.sum(1), min=1e-9) + +class LanguageModel(nn.Module): + def __init__(self, + modelname: str, + device: str, + readout: str + ): + super(LanguageModel, self).__init__() + self.device = device + self.modelname = modelname + self.readout_fn = readout + + self.model = AutoModel.from_pretrained(modelname) + self.hidden_size = self.model.config.hidden_size + + def readout(self, model_inputs, model_outputs, readout_masks=None): + if self.readout_fn == 'cls': + if 'bert' in self.modelname or 'deberta' in self.modelname: + text_representations = model_outputs.last_hidden_state[:, 0] + elif 'xlnet' in self.modelname: + text_representations = model_outputs.last_hidden_state[:, -1] + else: + raise ValueError('Invalid model name {} for the cls readout.'.format(self.modelname)) + elif self.readout_fn == 'mean': + text_representations = mask_pooling(model_outputs, model_inputs['attention_mask']) + elif self.readout_fn == 'ch' and readout_masks is not None: + text_representations = mask_pooling(model_outputs, readout_masks) + else: + raise ValueError('Invalid readout function.') + return text_representations + + def _lm_forward(self, tokens): + tokens = tokens.to(self.device) + if 'readout_mask' in tokens: + readout_mask = tokens.pop('readout_mask') + else: + readout_mask = None + outputs = self.model(**tokens) + return self.readout(tokens, outputs, readout_mask) + + def forward(self): + raise NotImplementedError + + def save_pretrained(self, modeldir): + model_filename = os.path.join(modeldir, 'checkpoint.pt') + torch.save(self.state_dict(), model_filename) + + def load_pretrained(self, modeldir): + model_filename = os.path.join(modeldir, 'checkpoint.pt') + self.load_state_dict(torch.load(model_filename)) + +class MultiHeadLanguageModel(LanguageModel): + def __init__(self, + modelname: str, + device: str, + readout: str, + num_classes: List + ): + super().__init__( + modelname, + device, + readout + ) + + self.num_classes = num_classes + self.lns = nn.ModuleList([nn.Linear(self.hidden_size, num_class) for num_class in num_classes]) + + def forward(self, input_tokens, input_head_indices, class_tokens, class_head_indices): + head_indices = torch.unique(input_head_indices) + text_representations = self._lm_forward(input_tokens) + + final_preds = {} + for i in head_indices: + if torch.any(input_head_indices == i): + final_preds[i.item()] = self.lns[i.item()](text_representations[input_head_indices == i]) + else: + final_preds[i.item()] = torch.tensor([]).to(self.device) + return final_preds diff --git a/Deep-Citation/Workspace/acl_scicite_wksp_trl/args.txt b/Deep-Citation/Workspace/acl_scicite_wksp_trl/args.txt new file mode 100644 index 0000000000000000000000000000000000000000..998d66a0c292a70e058dabffe4a669edc73b6a3d --- /dev/null +++ b/Deep-Citation/Workspace/acl_scicite_wksp_trl/args.txt @@ -0,0 +1,21 @@ +Namespace(dataset='acl-scicite', +lambdas='1-0.063', +data_dir='Data', +workspace='Workspace/acl_scicite_wksp_trl', +class_definition='Data/class_def.json', +batch_size=32, +lr=5e-05, +decay_rate=0.5, +decay_step=5, +num_epochs=10, +scheduler='slanted', +dropout_rate=0.2, +l2=0.0, +device='cuda', +tol=10, +inference_only=False, +seed=1, +lm='scibert', +max_length=512, +batch_size_factor=2, +readout='ch') \ No newline at end of file diff --git a/Deep-Citation/Workspace/acl_scicite_wksp_trl/best_model.pt b/Deep-Citation/Workspace/acl_scicite_wksp_trl/best_model.pt new file mode 100644 index 0000000000000000000000000000000000000000..5e7c7dd3b27b87ca08ec060f2b0944d12bc0f8c8 --- /dev/null +++ b/Deep-Citation/Workspace/acl_scicite_wksp_trl/best_model.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e45ab11942439f80a121dad5b2d9da392470e0cedf6a7335991fa0a1f616dcb2 +size 439784777 diff --git a/Deep-Citation/data.py b/Deep-Citation/data.py new file mode 100644 index 0000000000000000000000000000000000000000..7f3020f87647f234b3652d68f9a28c12dab7732f --- /dev/null +++ b/Deep-Citation/data.py @@ -0,0 +1,211 @@ +import os +import json +import copy +import torch +import scipy +import numpy as np +import pandas as pd +from tqdm import tqdm +from scipy.special import softmax + +from transformers import AutoTokenizer + +class CollateFn(object): + def __init__(self, modelname, class_definitions=None, instance_weights=False): + self.instance_weights = instance_weights + use_fast = False if 'deberta' in modelname else True + self.tokenizer = AutoTokenizer.from_pretrained(modelname, use_fast=use_fast) + cited_ids = self.tokenizer.encode('', add_special_tokens=False) + self.cited_here_tokens = torch.tensor(cited_ids, dtype=torch.long) + + if class_definitions is not None: + self.class_definitions = [] + self.class_head_indices = [] + for i, defs in enumerate(class_definitions): + self.class_definitions += defs + self.class_head_indices.append(i * torch.ones(len(defs), dtype=torch.long)) + self.class_head_indices = torch.cat(self.class_head_indices, dim=0) + self.class_tokens = self.tokenizer( + self.class_definitions, + return_tensors="pt", + max_length=512, + truncation=True, + padding=True + ) + + def _get_readout_mask(self, tokens): + # cited_here_tokens = torch.tensor([962, 8412, 1530, 1374]) + readout_mask = torch.zeros_like(tokens['input_ids'], dtype=torch.bool) + + batch_size = tokens['input_ids'].size(0) + l = tokens['input_ids'].size(1) + ctk_l = self.cited_here_tokens.size(0) + for b in range(batch_size): + for i in range(1, l - ctk_l): + if torch.equal(tokens['input_ids'][b, i:i+ctk_l], self.cited_here_tokens): + readout_mask[b, i:i+ctk_l] = True + if not readout_mask[b].any(): + # Fallback to CLS if the citation marker isn't matched. + readout_mask[b, 0] = True + return readout_mask + + def _tokenize_context(self, context): + tokens = self.tokenizer( + context, + return_tensors="pt", + max_length=512, + truncation=True, + padding=True + ) + tokens['readout_mask'] = self._get_readout_mask( + tokens + ) + + return tokens + + def __call__(self, samples): + if self.instance_weights: + text, labels, ds_indices, instance_weights = list(map(list, zip(*samples))) + batched_text = self._tokenize_context(text) + labels = torch.stack(labels) + ds_indices = torch.stack(ds_indices) + instance_weights = torch.stack(instance_weights) + return batched_text, labels, ds_indices, instance_weights + else: + text, labels, ds_indices = list(map(list, zip(*samples))) + batched_text = self._tokenize_context(text) + labels = torch.stack(labels) + ds_indices = torch.stack(ds_indices) + + return batched_text, labels, ds_indices, copy.deepcopy(self.class_tokens), self.class_head_indices + +class Dataset(object): + def __init__(self, dataframe, class_definitions, lmbd=1.0): + self.class_definitions = class_definitions + self.lmbd = lmbd + self._load_data(dataframe) + + def __len__(self): + return len(self.labels) + + def __getitem__(self, idx): + '''Get datapoint with index''' + return (self.text[idx], self.labels[idx], self.ds_index[idx]) + + def _load_data(self, annotated_data): + self.labels = torch.LongTensor(annotated_data['label'].tolist()) + self.original_labels = torch.LongTensor(annotated_data['label'].tolist()) + self.ds_index = torch.zeros_like(self.original_labels) + self.text = annotated_data['context'].tolist() + +class MultiHeadDatasets(object): + def __init__(self, datasets, batch_size_factor=2): + self.text = [] + self.ds_index = [] + self.labels = [] + self.class_definitions = [] + self.lambdas = [] + + self.dataset_sizes = [len(d.labels) for d in datasets] + if len(self.dataset_sizes) > 1: + if sum(self.dataset_sizes) / self.dataset_sizes[0] <= batch_size_factor: + self.sample_auxiliary = False + self.adjusted_batch_size_factor = sum(self.dataset_sizes) / self.dataset_sizes[0] + else: + self.sample_auxiliary = True + self.sample_distribution = np.array([d.lmbd for d in datasets[1:]]) / sum([d.lmbd for d in datasets[1:]]) + self.adjusted_batch_size_factor = batch_size_factor + else: + self.sample_auxiliary = False + self.adjusted_batch_size_factor = 1 + + for i, d in enumerate(datasets): + self.text += d.text + self.ds_index.append(i * torch.ones(len(d.text), dtype=torch.long)) + self.labels.append(d.labels) + self.class_definitions.append(d.class_definitions) + self.lambdas.append(d.lmbd) + self.labels = torch.cat(self.labels, dim=0) + self.ds_index = torch.cat(self.ds_index, dim=0) + + def sample_auxiliary_instace(self): + sampled_dataset_idx = np.random.choice( + np.arange(1, len(self.dataset_sizes)), + p=self.sample_distribution + ) + instance_idx = np.random.choice( + self.dataset_sizes[sampled_dataset_idx] + ) + sum(self.dataset_sizes[:sampled_dataset_idx]) + return instance_idx + + def __len__(self): + if self.sample_auxiliary: # if the auxiliary dataset is larger than the main dataset + return self.dataset_sizes[0] * self.adjusted_batch_size_factor + return len(self.labels) + + def __getitem__(self, idx): + '''Get datapoint with index''' + if idx < self.dataset_sizes[0] or not self.sample_auxiliary: + return (self.text[idx], self.labels[idx], self.ds_index[idx]) + else: + real_idx = self.sample_auxiliary_instace() + return (self.text[real_idx], self.labels[real_idx], self.ds_index[real_idx]) + +def load_class_definitions(filename): + with open(filename, 'r') as f: + class_definitions = json.load(f) + + results = {k:{} for k in class_definitions.keys()} + for k, v in class_definitions.items(): + for kk, vv in v.items(): + results[k][kk.lower()] = vv + return results + +def create_data_channels(filename, class_definition_filename, split=None, lmbd=1.0): + data = pd.read_csv(filename, sep='\t') + data = data.fillna(' ') + + print('Number of data instance: {}'.format(data.shape[0])) + + # map labels to ids + unique_labels = data['label'].unique().tolist() + label2id = {lb: i for i, lb in enumerate(unique_labels)} + + data['label'] = data['label'].apply( + lambda x: label2id[x]) + + data_train = data[data['split'] == 'train'].reset_index() + data_val = data[data['split'] == 'val'].reset_index() + data_test = data[data['split'] == 'test'].reset_index() + + class_definitions = load_class_definitions(class_definition_filename) + dataname = filename.split('/')[-1].split('.')[0] + data_class_definitions = [class_definitions[dataname][lb.lower()] for lb in unique_labels] + + train_data = Dataset(data_train, data_class_definitions, lmbd=lmbd) + val_data = Dataset(data_val, data_class_definitions, lmbd=lmbd) + test_data = Dataset(data_test, data_class_definitions, lmbd=lmbd) + + return train_data, val_data, test_data, unique_labels + +def create_single_data_object(filename, class_definition_filename, split=None, lmbd=1.0): + data = pd.read_csv(filename, sep='\t') + data = data.fillna(' ') + + print('Number of data instance: {}'.format(data.shape[0])) + + # map labels to ids + unique_labels = data['label'].unique() + label2id = {lb: i for i, lb in enumerate(unique_labels)} + + data['label'] = data['label'].apply( + lambda x: label2id[x]) + + class_definitions = load_class_definitions(class_definition_filename) + dataname = filename.split('/')[-1].split('.')[0] + data_class_definitions = [class_definitions[dataname][lb.lower()] for lb in unique_labels] + + if split is None: + return Dataset(data, data_class_definitions, lmbd=lmbd), unique_labels + else: + return Dataset(data[data['split'] == split].reset_index(), data_class_definitions, lmbd=lmbd), unique_labels diff --git a/Dockerfile b/Dockerfile new file mode 100644 index 0000000000000000000000000000000000000000..6fe2a449fcbbc81a01a6597315fb6cd4dcc9e022 --- /dev/null +++ b/Dockerfile @@ -0,0 +1,36 @@ +FROM python:3.11-slim + +ENV PYTHONDONTWRITEBYTECODE=1 \ + PYTHONUNBUFFERED=1 \ + PIP_NO_CACHE_DIR=1 \ + STREAMLIT_SERVER_HEADLESS=true \ + STREAMLIT_BROWSER_GATHER_USAGE_STATS=false \ + PLAYWRIGHT_BROWSERS_PATH=/app/.playwright \ + HOME=/tmp + +WORKDIR /app + +RUN apt-get update && apt-get install -y --no-install-recommends \ + git \ + build-essential \ + curl \ + && rm -rf /var/lib/apt/lists/* + +COPY requirements.txt /app/requirements.txt +COPY hf_space/requirements.txt /app/hf_space/requirements.txt +RUN python -m pip install --upgrade pip && \ + pip install -r requirements.txt + +# Chromium for SciPath Live Run (websearch_deep crawl tool). +RUN mkdir -p /app/.playwright && \ + python -m playwright install --with-deps chromium + +COPY . /app + +# HF Space secrets (GEMINI_API_KEY, etc.) are injected as env vars at runtime. +EXPOSE 7860 + +CMD ["streamlit", "run", "hf_space/streamlit_app.py", \ + "--server.address", "0.0.0.0", \ + "--server.port", "7860", \ + "--browser.gatherUsageStats", "false"] diff --git a/README.md b/README.md index d1f14f11e43847598a84835b4859f0f5a3cca21b..cf48ec59e82c2531988013a2cd488f68e507fc1f 100644 --- a/README.md +++ b/README.md @@ -1,10 +1,233 @@ --- title: SciPaths Demo -emoji: 🏢 +emoji: 🔬 colorFrom: blue -colorTo: blue +colorTo: green sdk: docker pinned: false +app_port: 7860 --- -Check out the configuration reference at https://huggingface.co/docs/hub/spaces-config-reference +# SciPaths Demo + +Self-contained demo of **SciPaths** (annotation process) and **SciFy CodeAgent** (system run / live run) on SciPaths claims. + +The Hugging Face Space launches Streamlit from `hf_space/streamlit_app.py`. This folder is deployable on its own — it does not depend on sibling repos such as `dryrun/`. + +## Citation + +If you find this useful, please cite our paper as: + +```bibtex +@misc{chamoun2026scipathsforecastingpathwaysscientific, + title={SciPaths: Forecasting Pathways to Scientific Discovery}, + author={Eric Chamoun and Yizhou Chi and Yulong Chen and Rui Cao and Zifeng Ding and Michalis Korakakis and Andreas Vlachos}, + year={2026}, + eprint={2605.14600}, + archivePrefix={arXiv}, + primaryClass={cs.CL}, + url={https://arxiv.org/abs/2605.14600}, +} +``` + +Paper URL: https://arxiv.org/abs/2605.14600 + +## Required Secrets + +Set this in the Space settings before publishing: + +```text +GEMINI_API_KEY= +``` + +Optional, for saving completed run artifacts to a Hugging Face Dataset: + +```text +HF_WRITE_TOKEN= +RUNS_REPO_ID= +RUNS_REPO_TYPE=dataset +``` + +Optional, for higher Semantic Scholar limits: + +```text +SEMANTIC_SCHOLAR_API_KEY= +``` + +## Run The Demo Locally + +```bash +pip install -r requirements.txt +streamlit run hf_space/streamlit_app.py +``` + +Then enter an arXiv URL or ID, for example: + +```text +https://arxiv.org/abs/2211.08788 +``` + +The app writes each run under: + +```text +hf_space/runs// +``` + +## Run One Example From The Command Line + +This example stores all intermediate files under `runs/example/processed_papers`. + +```bash +mkdir -p runs/example +printf '[{"id":"2211.08788","title":"","id_type":"ArXiv"}]\n' > runs/example/input_ids.json + +python src/step_01_fetch/fetch_metadata.py \ + --ids runs/example/input_ids.json \ + --outdir runs/example/processed_papers + +python src/step_02_mark_citations/replace_citation_markers.py \ + --root runs/example/processed_papers + +python src/step_03_usage_contexts/build_usage_contexts.py \ + --root runs/example/processed_papers \ + --out-name usage_contexts.json + +python src/step_04_label_citations/label_citation_functions.py \ + --root runs/example/processed_papers \ + --model-path Deep-Citation/Workspace/acl_scicite_wksp_trl/best_model.pt \ + --model-data-dir Deep-Citation/Data \ + --model-class-def Deep-Citation/Data/class_def.json \ + --model-lm scibert \ + --device cpu + +python src/step_05_verify_uses_extends/verify_uses_extends.py \ + --root runs/example/processed_papers \ + --k 0 \ + --batch-size 25 + +python src/step_06_extract_paragraphs/extract_arxiv_paragraphs.py \ + --root runs/example/processed_papers + +python src/step_07_extract_and_refine/extract_contributions_from_citations.py \ + --root runs/example/processed_papers + +python src/step_07_extract_and_refine/refine_and_filter_clusters_llm.py \ + --root runs/example/processed_papers \ + --inplace \ + --overwrite + +PYTHONPATH=src \ +python -m step_08_annotation.cli run \ + --paper-dir runs/example/processed_papers/2211.08788 \ + --provider gemini \ + --model gemini/gemini-3.1-pro-preview \ + --formatter-model gemini/gemini-3.1-pro-preview \ + --judge-model gemini/gemini-3.1-pro-preview \ + --candidate-count 3 \ + --output-root runs/example/two_pass_outputs +``` + +The final UI payload is written as `pass_2_ui_payload.json` inside the annotation run directory printed by the last command. + +## Run Each Step On A Set Of Papers + +Create an ID file with one entry per paper: + +```json +[ + {"id": "2211.08788", "title": "", "id_type": "ArXiv"}, + {"id": "2311.14919", "title": "", "id_type": "ArXiv"} +] +``` + +Save it as `runs/batch/input_ids.json`, then run: + +```bash +mkdir -p runs/batch + +# 1. Fetch metadata + LaTeX for each input paper. +python src/step_01_fetch/fetch_metadata.py \ + --ids runs/batch/input_ids.json \ + --outdir runs/batch/processed_papers + +# 2. Add explicit citation markers to the target-paper text. +python src/step_02_mark_citations/replace_citation_markers.py \ + --root runs/batch/processed_papers + +# 3. Build downstream citation usage contexts. +python src/step_03_usage_contexts/build_usage_contexts.py \ + --root runs/batch/processed_papers \ + --out-name usage_contexts.json + +# 4. Label citation functions with the bundled Deep-Citation classifier. +python src/step_04_label_citations/label_citation_functions.py \ + --root runs/batch/processed_papers \ + --model-path Deep-Citation/Workspace/acl_scicite_wksp_trl/best_model.pt \ + --model-data-dir Deep-Citation/Data \ + --model-class-def Deep-Citation/Data/class_def.json \ + --model-lm scibert \ + --device cpu + +# 5. Verify USES/EXTENDS citations with an LLM. +python src/step_05_verify_uses_extends/verify_uses_extends.py \ + --root runs/batch/processed_papers \ + --k 0 \ + --batch-size 25 + +# 6. Extract arXiv paragraphs from downstream citing papers. +python src/step_06_extract_paragraphs/extract_arxiv_paragraphs.py \ + --root runs/batch/processed_papers + +# 7. Extract downstream contribution clusters, then merge/filter them. +python src/step_07_extract_and_refine/extract_contributions_from_citations.py \ + --root runs/batch/processed_papers + +python src/step_07_extract_and_refine/refine_and_filter_clusters_llm.py \ + --root runs/batch/processed_papers \ + --inplace \ + --overwrite + +# 8. Annotate each ready paper: target contributions, enabling contributions, and groundings. +for paper_dir in runs/batch/processed_papers/*; do + [ -d "$paper_dir" ] || continue + [ -f "$paper_dir/usage_discovery_from_contributions.json" ] || continue + PYTHONPATH=src \ + python -m step_08_annotation.cli run \ + --paper-dir "$paper_dir" \ + --provider gemini \ + --model gemini/gemini-3.1-pro-preview \ + --formatter-model gemini/gemini-3.1-pro-preview \ + --judge-model gemini/gemini-3.1-pro-preview \ + --candidate-count 3 \ + --output-root runs/batch/two_pass_outputs +done +``` + +## Pipeline Steps + +1. **Fetch metadata + LaTeX.** Downloads target-paper metadata, references, citing-paper metadata, and arXiv source where available. +2. **Add citation markers.** Inserts normalized citation markers into the target paper so downstream citation contexts can be aligned. +3. **Build usage contexts.** Collects text windows around downstream citations to the target paper. +4. **Label citation functions.** Uses the bundled Deep-Citation classifier to label citation contexts as background, use, extension, comparison, and related categories. +5. **Verify USES/EXTENDS.** Uses an LLM to check whether candidate downstream citations genuinely use or extend the target paper. +6. **Extract arXiv paragraphs.** Retrieves fuller paragraphs from citing papers so the system has enough context for contribution extraction. +7. **Extract and refine target-contribution clusters.** Extracts what downstream papers use the target paper for, clusters near-duplicates, and filters weak/non-usage evidence. +8. **Annotate pathways.** Derives target contributions from the refined clusters, decomposes each into enabling contributions, selects primary groundings, and records additional grounding studies. + +## Important Files + +```text +hf_space/streamlit_app.py Streamlit UI +hf_space/runner.py Orchestrates steps 1-7 for the UI +hf_space/streamlit_config.py Example papers and tab names +src/common/ Shared LLM and paper-package utilities +src/step_01_fetch/ Metadata, references, citations, and LaTeX +src/step_02_mark_citations/ Citation-marker insertion +src/step_03_usage_contexts/ Downstream usage-context construction +src/step_04_label_citations/ Deep-Citation citation-function labeling +src/step_05_verify_uses_extends/ LLM verification of USES/EXTENDS citations +src/step_06_extract_paragraphs/ ArXiv paragraph extraction from citing papers +src/step_07_extract_and_refine/ Contribution extraction and cluster refinement +src/step_08_annotation/ Target/enabling contribution annotation and grounding +Deep-Citation/ Bundled citation-function classifier assets +``` diff --git a/app.py b/app.py new file mode 100644 index 0000000000000000000000000000000000000000..13da4b7a70d222665ab47b5f0445a637bb5e0d4e --- /dev/null +++ b/app.py @@ -0,0 +1,5 @@ +from hf_space.streamlit_app import main + + +if __name__ == "__main__": + main() diff --git a/docker-compose.neo4j.yml b/docker-compose.neo4j.yml new file mode 100644 index 0000000000000000000000000000000000000000..b5e9f99b1bb1bf19476f75f7feb634700aeebf34 --- /dev/null +++ b/docker-compose.neo4j.yml @@ -0,0 +1,19 @@ +# Optional dedicated Neo4j if ports 7474/7687 are free. +# By default the app uses whatever answers bolt://localhost:7687 +# (NEO4J_URI / NEO4J_USER / NEO4J_PASSWORD). +services: + neo4j: + image: neo4j:5.26-community + container_name: scipaths-neo4j + ports: + - "7475:7474" + - "7688:7687" + environment: + NEO4J_AUTH: neo4j/scipaths123 + NEO4J_PLUGINS: '["apoc"]' + NEO4J_dbms_memory_heap_max__size: 1G + volumes: + - scipaths_neo4j_data:/data + +volumes: + scipaths_neo4j_data: diff --git a/hf_space/build_replay_traces.py b/hf_space/build_replay_traces.py new file mode 100644 index 0000000000000000000000000000000000000000..b7ac85ed58c28407a53aac876ab04c5b102719b2 --- /dev/null +++ b/hf_space/build_replay_traces.py @@ -0,0 +1,305 @@ +#!/usr/bin/env python3 +"""Run the demo example papers end-to-end and save replay traces.""" + +from __future__ import annotations + +import json +import os +import shutil +import sys +import time +from datetime import datetime, timezone +from pathlib import Path + +SRC = Path(__file__).resolve().parent +REPO_ROOT = SRC.parent +for extra in (SRC, REPO_ROOT / "src"): + extra_str = str(extra) + if extra_str not in sys.path: + sys.path.insert(0, extra_str) + +from dotenv import load_dotenv + +load_dotenv(REPO_ROOT / ".env") +load_dotenv(REPO_ROOT.parent / "dryrun" / ".env", override=False) + +# Align Gemini env aliases used across the repo. +if not os.getenv("GEMINI_API_KEY"): + for alt in ("GOOGLE_GENAI_API_KEY", "GOOGLE_API_KEY"): + if os.getenv(alt): + os.environ["GEMINI_API_KEY"] = os.environ[alt] + break + +import runner as runner_module +from runner import PipelineConfig +from common.paper_package import load_paper_package +from step_08_annotation.pipeline import TwoPassAnnotationPipeline +from streamlit_config import EXAMPLES + +REPLAY_ROOT = REPO_ROOT / "replay_traces" +WORK_ROOT = REPO_ROOT / "hf_space" / "runs" / "replay_build" + + +def _env(name: str, default: str) -> str: + return (os.getenv(name) or default).strip() + + +def _load_json(path: Path): + if not path.exists(): + return None + try: + return json.loads(path.read_text(encoding="utf-8")) + except Exception: + return None + + +def _copy_if_exists(src: Path, dst: Path) -> bool: + if not src.exists(): + return False + dst.parent.mkdir(parents=True, exist_ok=True) + if src.is_dir(): + if dst.exists(): + shutil.rmtree(dst) + shutil.copytree(src, dst) + else: + shutil.copy2(src, dst) + return True + + +def _run_annotation(paper_dir: Path, annotation_root: Path) -> tuple[dict | None, Path | None, str | None]: + discovery = _load_json(paper_dir / "usage_discovery_from_contributions.json") or {} + clusters = discovery.get("clusters") or [] + if not clusters: + return None, None, "No refined downstream usage clusters; annotation skipped." + + llm_provider = _env("LLM_PROVIDER", "gemini") + llm_model = _env("LLM_MODEL", "gemini-3.1-pro-preview") + formatter_model = _env("ANNOTATION_FORMATTER_MODEL", "gemini/gemini-3.1-pro-preview") + judge_model = _env("ANNOTATION_JUDGE_MODEL", "gemini/gemini-3.1-pro-preview") + candidate_count = int(_env("ANNOTATION_CANDIDATE_COUNT", "3")) + + paper = load_paper_package(paper_dir) + pipeline = TwoPassAnnotationPipeline( + provider=llm_provider, + model=llm_model, + formatter_model=formatter_model or None, + judge_model=judge_model or None, + output_root=annotation_root, + annotator_id="replay_trace_builder", + candidate_count=max(1, candidate_count), + formatter_max_attempts=3, + include_reference_examples=True, + prompt_profile="full", + ) + result = pipeline.run(paper) + return result.result, result.run_dir, None + + +def _package_trace( + *, + label: str, + arxiv_id: str, + paper_input: str, + job_dir: Path, + paper_dir: Path, + events: list[str], + status: str, + annotation_run_dir: Path | None, + annotation_skipped_reason: str | None, + pipeline_failed_reason: str | None, + pipeline_stopped_reason: str | None, +) -> Path: + out_dir = REPLAY_ROOT / arxiv_id + if out_dir.exists(): + shutil.rmtree(out_dir) + out_dir.mkdir(parents=True, exist_ok=True) + + paper_out = out_dir / "processed_papers" / arxiv_id + paper_out.mkdir(parents=True, exist_ok=True) + + # Core replay payloads (keep disk footprint manageable). + keep_files = [ + "paper_metadata.json", + "usage_contexts.json", + "usage_context_labels.json", + "usage_uses_extends_verified.json", + "usage_citing_paragraphs.json", + "usage_contributions.json", + "usage_discovery_from_contributions.json", + ] + for name in keep_files: + _copy_if_exists(paper_dir / name, paper_out / name) + + _copy_if_exists(job_dir / "logs", out_dir / "logs") + _copy_if_exists(job_dir / "summary.txt", out_dir / "summary.txt") + _copy_if_exists(job_dir / "run_config.json", out_dir / "run_config.json") + _copy_if_exists(job_dir / "input_ids.json", out_dir / "input_ids.json") + + annotation_payload_path = None + if annotation_run_dir and annotation_run_dir.exists(): + ann_dst = out_dir / "two_pass_outputs" / annotation_run_dir.name + _copy_if_exists(annotation_run_dir, ann_dst) + payload = ann_dst / "pass_2_ui_payload.json" + if payload.exists(): + annotation_payload_path = str(payload.relative_to(out_dir)) + + discovery = _load_json(paper_out / "usage_discovery_from_contributions.json") or {} + contributions = _load_json(paper_out / "usage_contributions.json") or {} + payload = None + if annotation_payload_path: + payload = _load_json(out_dir / annotation_payload_path) + + public_export = { + "citation_clusters": (discovery or {}).get("clusters") or [], + "target_contribution_decompositions": (payload or {}).get("claims") or [], + } + (out_dir / "scipaths_run_results.json").write_text( + json.dumps(public_export, indent=2, ensure_ascii=False), + encoding="utf-8", + ) + + meta = { + "label": label, + "arxiv_id": arxiv_id, + "paper_input": paper_input, + "status": status, + "built_at": datetime.now(timezone.utc).isoformat(), + "source_job_dir": str(job_dir), + "paper_dir": str((out_dir / "processed_papers" / arxiv_id).relative_to(out_dir)), + "annotation_payload_path": annotation_payload_path, + "annotation_skipped_reason": annotation_skipped_reason, + "pipeline_failed_reason": pipeline_failed_reason, + "pipeline_stopped_reason": pipeline_stopped_reason, + "events": events, + "cluster_count": len((discovery or {}).get("clusters") or []), + "contribution_count": len((contributions or {}).get("contributions") or []), + "claim_count": len((payload or {}).get("claims") or []) if isinstance(payload, dict) else 0, + } + (out_dir / "replay_meta.json").write_text(json.dumps(meta, indent=2, ensure_ascii=False), encoding="utf-8") + return out_dir + + +def run_one(label: str, paper_input: str) -> dict: + if not os.getenv("GEMINI_API_KEY"): + raise SystemExit("GEMINI_API_KEY is required to build replay traces.") + + arxiv_id = runner_module.parse_arxiv_id(paper_input) + print(f"\n=== Building replay trace for {label} ({arxiv_id}) ===", flush=True) + + cfg = PipelineConfig( + repo_root=REPO_ROOT, + source_root=REPO_ROOT / "src" / "processed_papers", + paper_input=paper_input, + llm_provider=_env("LLM_PROVIDER", "gemini"), + llm_model=_env("LLM_MODEL", "gemini-3.1-pro-preview"), + llm_model_step4=_env("LLM_MODEL_STEP4", "gemini-3-flash-preview"), + model_path="Deep-Citation/Workspace/acl_scicite_wksp_trl/best_model.pt", + model_data_dir="Deep-Citation/Data", + model_class_def="Deep-Citation/Data/class_def.json", + model_lm="scibert", + device="cpu", + embedding_model="sentence-transformers/all-mpnet-base-v2", + ) + + events: list[str] = [] + artifact_path = None + pipeline_failed_reason = None + pipeline_stopped_reason = None + t0 = time.time() + + for line, maybe_artifact in runner_module.run_pipeline(cfg, WORK_ROOT): + if line: + print(line, flush=True) + events.append(line) + if line.startswith("Pipeline stopped:"): + pipeline_stopped_reason = line + if "failed" in line.lower(): + pipeline_failed_reason = line + if maybe_artifact: + artifact_path = maybe_artifact + + if not artifact_path: + raise RuntimeError(f"Pipeline produced no artifact for {arxiv_id}") + + job_dir = Path(str(artifact_path)).with_suffix("") + paper_dir = job_dir / "processed_papers" / arxiv_id + annotation_run_dir = None + annotation_skipped_reason = None + + if pipeline_failed_reason: + status = "Failed" + annotation_skipped_reason = f"{pipeline_failed_reason} Annotation was not run." + elif pipeline_stopped_reason: + status = "Stopped" + annotation_skipped_reason = f"{pipeline_stopped_reason} Annotation was not run." + else: + print("[annotation] starting", flush=True) + try: + _run_output, annotation_run_dir, skip = _run_annotation( + paper_dir=paper_dir, + annotation_root=job_dir / "two_pass_outputs", + ) + if skip: + annotation_skipped_reason = skip + status = "Completed" + print(f"[annotation] skipped: {skip}", flush=True) + else: + status = "Completed" + events.append(f"[annotation] complete: {annotation_run_dir}") + print(f"[annotation] complete: {annotation_run_dir}", flush=True) + except Exception as exc: + status = "Failed" + pipeline_failed_reason = f"Annotation failed: {exc}" + annotation_skipped_reason = pipeline_failed_reason + events.append(pipeline_failed_reason) + print(pipeline_failed_reason, flush=True) + + out_dir = _package_trace( + label=label, + arxiv_id=arxiv_id, + paper_input=paper_input, + job_dir=job_dir, + paper_dir=paper_dir, + events=events, + status=status, + annotation_run_dir=annotation_run_dir, + annotation_skipped_reason=annotation_skipped_reason, + pipeline_failed_reason=pipeline_failed_reason, + pipeline_stopped_reason=pipeline_stopped_reason, + ) + elapsed = time.time() - t0 + print(f"Saved replay trace -> {out_dir} ({status}, {elapsed/60:.1f} min)", flush=True) + return { + "label": label, + "arxiv_id": arxiv_id, + "status": status, + "trace_dir": str(out_dir), + "elapsed_sec": elapsed, + "annotation_skipped_reason": annotation_skipped_reason, + "pipeline_failed_reason": pipeline_failed_reason, + "pipeline_stopped_reason": pipeline_stopped_reason, + } + + +def main() -> int: + REPLAY_ROOT.mkdir(parents=True, exist_ok=True) + WORK_ROOT.mkdir(parents=True, exist_ok=True) + + results = [] + for label, paper_input in EXAMPLES.items(): + results.append(run_one(label, paper_input)) + + index = { + "built_at": datetime.now(timezone.utc).isoformat(), + "examples": results, + } + index_path = REPLAY_ROOT / "index.json" + index_path.write_text(json.dumps(index, indent=2, ensure_ascii=False), encoding="utf-8") + print(f"\nWrote index -> {index_path}", flush=True) + for item in results: + print(f"- {item['arxiv_id']}: {item['status']} -> {item['trace_dir']}", flush=True) + return 0 if all(item["status"] == "Completed" for item in results) else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/hf_space/neo4j_workflow.py b/hf_space/neo4j_workflow.py new file mode 100644 index 0000000000000000000000000000000000000000..3dfe5e26f81d13481aa46a9ad16eb5bbbdb29438 --- /dev/null +++ b/hf_space/neo4j_workflow.py @@ -0,0 +1,150 @@ +"""Neo4j sync for SciPaths workflow graph animation.""" + +from __future__ import annotations + +import os +from typing import Any, Optional + +_DRIVER = None + +NEO4J_URI = os.getenv("NEO4J_URI", "bolt://localhost:7687") +NEO4J_USER = os.getenv("NEO4J_USER", "neo4j") +NEO4J_PASSWORD = os.getenv("NEO4J_PASSWORD", "graphscout123") + + +def neo4j_available() -> bool: + try: + driver = get_driver() + if driver is None: + return False + driver.verify_connectivity() + return True + except Exception: + return False + + +def get_driver(): + global _DRIVER + if _DRIVER is not None: + return _DRIVER + try: + from neo4j import GraphDatabase + except ImportError: + return None + try: + _DRIVER = GraphDatabase.driver(NEO4J_URI, auth=(NEO4J_USER, NEO4J_PASSWORD)) + _DRIVER.verify_connectivity() + return _DRIVER + except Exception: + _DRIVER = None + return None + + +def reset_run(run_id: str) -> bool: + driver = get_driver() + if driver is None or not run_id: + return False + try: + with driver.session() as session: + session.run( + "MATCH (n:SciPathsNode {run_id: $run_id}) DETACH DELETE n", + run_id=run_id, + ) + return True + except Exception: + return False + + +def upsert_graph(run_id: str, nodes: list[dict[str, Any]], edges: list[dict[str, Any]]) -> bool: + """Replace the run's graph snapshot with the current cumulative nodes/edges.""" + driver = get_driver() + if driver is None or not run_id: + return False + try: + with driver.session() as session: + session.execute_write(_upsert_tx, run_id, nodes, edges) + return True + except Exception: + return False + + +def _upsert_tx(tx, run_id: str, nodes: list[dict[str, Any]], edges: list[dict[str, Any]]) -> None: + tx.run("MATCH (n:SciPathsNode {run_id: $run_id}) DETACH DELETE n", run_id=run_id) + for node in nodes: + tx.run( + """ + CREATE (n:SciPathsNode { + run_id: $run_id, + id: $id, + kind: $kind, + label: $label, + title: $title, + detail: $detail, + step_added: $step_added, + pulse: $pulse, + color: $color, + size: $size + }) + """, + run_id=run_id, + id=str(node.get("id") or ""), + kind=str(node.get("kind") or "node"), + label=str(node.get("label") or ""), + title=str(node.get("title") or node.get("label") or ""), + detail=str(node.get("detail") or ""), + step_added=int(node.get("step_added") or 0), + pulse=bool(node.get("pulse")), + color=str(node.get("color") or "#6a7a74"), + size=int(node.get("size") or 16), + ) + for edge in edges: + tx.run( + """ + MATCH (a:SciPathsNode {run_id: $run_id, id: $source}) + MATCH (b:SciPathsNode {run_id: $run_id, id: $target}) + CREATE (a)-[:SCI_REL { + run_id: $run_id, + id: $id, + kind: $kind, + pulse: $pulse, + muted: $muted + }]->(b) + """, + run_id=run_id, + source=str(edge.get("source") or ""), + target=str(edge.get("target") or ""), + id=str(edge.get("id") or ""), + kind=str(edge.get("kind") or "rel"), + pulse=bool(edge.get("pulse")), + muted=bool(edge.get("muted")), + ) + + +def fetch_graph(run_id: str) -> Optional[dict[str, Any]]: + driver = get_driver() + if driver is None or not run_id: + return None + try: + with driver.session() as session: + nodes = session.run( + """ + MATCH (n:SciPathsNode {run_id: $run_id}) + RETURN n.id AS id, n.kind AS kind, n.label AS label, n.title AS title, + n.detail AS detail, n.step_added AS step_added, n.pulse AS pulse, + n.color AS color, n.size AS size + """, + run_id=run_id, + ).data() + edges = session.run( + """ + MATCH (a:SciPathsNode {run_id: $run_id})-[r:SCI_REL]->(b:SciPathsNode {run_id: $run_id}) + RETURN r.id AS id, a.id AS source, b.id AS target, r.kind AS kind, + r.pulse AS pulse, r.muted AS muted + """, + run_id=run_id, + ).data() + if not nodes: + return None + return {"nodes": nodes, "edges": edges} + except Exception: + return None diff --git a/hf_space/replay.py b/hf_space/replay.py new file mode 100644 index 0000000000000000000000000000000000000000..042df991b58f84f89aa9b52aa9ce7a454eaca42d --- /dev/null +++ b/hf_space/replay.py @@ -0,0 +1,130 @@ +"""Load precomputed demo traces for Streamlit replay mode.""" + +from __future__ import annotations + +import json +import re +from pathlib import Path +from typing import Any, Optional + +REPO_ROOT = Path(__file__).resolve().parent.parent +REPLAY_ROOT = REPO_ROOT / "replay_traces" + + +def _load_json(path: Path) -> Any | None: + if not path.exists(): + return None + try: + return json.loads(path.read_text(encoding="utf-8")) + except Exception: + return None + + +def parse_arxiv_id(paper_input: str) -> str | None: + s = (paper_input or "").strip() + if not s: + return None + if "arxiv.org" in s: + m = re.search(r"arxiv\.org/(abs|pdf)/([^/?#]+)", s) + if not m: + return None + s = m.group(2) + s = s.replace(".pdf", "") + s = re.sub(r"v\d+$", "", s) + if not re.match(r"^[0-9]{4}\.[0-9]{4,5}$", s): + return None + return s + + +def load_index() -> dict: + data = _load_json(REPLAY_ROOT / "index.json") + return data if isinstance(data, dict) else {"examples": []} + + +def list_replay_examples() -> list[dict]: + examples = load_index().get("examples") or [] + out = [] + for item in examples: + if not isinstance(item, dict): + continue + arxiv_id = item.get("arxiv_id") + if not arxiv_id: + continue + meta_path = REPLAY_ROOT / str(arxiv_id) / "replay_meta.json" + if not meta_path.exists(): + continue + out.append(item) + return out + + +def find_trace_dir(paper_input: str) -> Optional[Path]: + arxiv_id = parse_arxiv_id(paper_input) + if not arxiv_id: + return None + trace_dir = REPLAY_ROOT / arxiv_id + if (trace_dir / "replay_meta.json").exists(): + return trace_dir + return None + + +def load_trace(paper_input: str) -> Optional[dict]: + """Return a session-ready payload for a saved demo trace, or None.""" + trace_dir = find_trace_dir(paper_input) + if trace_dir is None: + return None + + meta = _load_json(trace_dir / "replay_meta.json") + if not isinstance(meta, dict): + return None + + paper_rel = meta.get("paper_dir") or f"processed_papers/{meta.get('arxiv_id')}" + paper_dir = (trace_dir / paper_rel).resolve() + if not paper_dir.exists(): + return None + + payload_rel = meta.get("annotation_payload_path") + payload_path = (trace_dir / payload_rel).resolve() if payload_rel else None + if payload_path and not payload_path.exists(): + payload_path = None + + events = [] + for line in meta.get("events") or []: + text = str(line).strip() + if not text: + continue + # Normalize absolute annotation paths for display. + if text.startswith("[annotation] complete:"): + events.append("Step 8 complete") + events.append("Pipeline completed successfully.") + continue + if text == "Pipeline completed successfully.": + # Keep a single completion marker after annotation normalization. + if events and events[-1] == "Pipeline completed successfully.": + continue + events.append(text) + + # Ensure a readable end marker for completed traces. + status = str(meta.get("status") or "Completed") + if status == "Completed" and "Pipeline completed successfully." not in events: + events.append("Pipeline completed successfully.") + + return { + "label": meta.get("label") or meta.get("arxiv_id"), + "arxiv_id": meta.get("arxiv_id"), + "paper_input": meta.get("paper_input") or f"https://arxiv.org/abs/{meta.get('arxiv_id')}", + "status": status, + "run_status": status, + "run_events": events, + "run_logs": list(events), + "run_dir_path": str(trace_dir), + "paper_dir_path": str(paper_dir), + "annotation_payload_path": str(payload_path) if payload_path else None, + "annotation_skipped_reason": meta.get("annotation_skipped_reason"), + "pipeline_failed_reason": meta.get("pipeline_failed_reason"), + "pipeline_stopped_reason": meta.get("pipeline_stopped_reason"), + "artifact_path": str(trace_dir / "summary.txt") if (trace_dir / "summary.txt").exists() else None, + "remote_artifact_ref": "", + "replay_mode": True, + "cluster_count": meta.get("cluster_count"), + "claim_count": meta.get("claim_count"), + } diff --git a/hf_space/requirements.txt b/hf_space/requirements.txt new file mode 100644 index 0000000000000000000000000000000000000000..f7ac1339a956f57d45080ec37a83ec8d56113d9e --- /dev/null +++ b/hf_space/requirements.txt @@ -0,0 +1,28 @@ +streamlit>=1.36.0 +arxiv==2.2.0 +requests==2.32.5 +google-generativeai +litellm +rapidfuzz +bibtexparser +sentence-transformers +# SciPath Live Run (parametric / websearch_deep CodeAgent) +smolagents +ddgs +playwright +PyPDF2 +python-dotenv +openai +# Pin stack for Deep-Citation step 4 on macOS x86_64: +# torch 2.2.x needs numpy<2; newer torch wheels are unavailable on this platform, +# and transformers 5.x refuses torch<2.4. +transformers==4.46.3 +torch==2.2.2 +huggingface_hub>=0.23,<1.0 +typer +tqdm +pydantic +numpy==1.26.4 +pandas +scipy==1.11.4 +scikit-learn==1.4.2 diff --git a/hf_space/runner.py b/hf_space/runner.py new file mode 100644 index 0000000000000000000000000000000000000000..5bebd3f07ed386b4c32b87e98a8661f102a343c6 --- /dev/null +++ b/hf_space/runner.py @@ -0,0 +1,333 @@ +import json +import os +import re +import shutil +import subprocess +import sys +import time +import uuid +from dataclasses import dataclass +from pathlib import Path +from typing import Generator, List, Optional, Tuple + + +@dataclass +class PipelineConfig: + repo_root: Path + source_root: Path + paper_input: str + llm_provider: str + llm_model: str + llm_model_step4: str + model_path: str + model_data_dir: str + model_class_def: str + model_lm: str + device: str + embedding_model: str + + +@dataclass +class PipelineResult: + job_id: str + job_dir: Path + paper_dir: Path + zip_path: Path + + +STEP_LABELS = { + 1: "Fetch metadata + LaTeX for input paper", + 2: "Add citation markers", + 3: "Build usage contexts", + 4: "Label citation functions", + 5: "Verify USES/EXTENDS", + 6: "Extract arXiv paragraphs", + 7: "Extract target contributions and refine clusters", +} + +FULL_STEPS = [1, 2, 3, 4, 5, 6, 7] +STOP_PREFIX = "Pipeline stopped:" + + +def parse_arxiv_id(paper_input: str) -> str: + s = (paper_input or "").strip() + if not s: + raise ValueError("paper_input is required") + if "arxiv.org" in s: + m = re.search(r"arxiv\.org/(abs|pdf)/([^/?#]+)", s) + if not m: + raise ValueError(f"Could not parse arXiv ID from URL: {s}") + s = m.group(2) + s = s.replace(".pdf", "") + s = re.sub(r"v\d+$", "", s) + if not re.match(r"^[0-9]{4}\.[0-9]{4,5}$", s): + raise ValueError(f"Invalid arXiv ID format: {s}") + return s + + +def _build_commands( + cfg: PipelineConfig, + step: int, + job_processed_root: Path, + paper_id: str, + ids_path: Optional[Path], +) -> List[List[str]]: + py = sys.executable + if step == 1: + assert ids_path is not None + return [[ + py, + "src/step_01_fetch/fetch_metadata.py", + "--ids", + str(ids_path), + "--outdir", + str(job_processed_root), + ]] + if step == 2: + return [[py, "src/step_02_mark_citations/replace_citation_markers.py", "--root", str(job_processed_root)]] + if step == 3: + return [[py, "src/step_03_usage_contexts/build_usage_contexts.py", "--root", str(job_processed_root), "--out-name", "usage_contexts.json"]] + if step == 4: + return [[ + py, + "src/step_04_label_citations/label_citation_functions.py", + "--root", + str(job_processed_root), + "--model-path", + cfg.model_path, + "--model-data-dir", + cfg.model_data_dir, + "--model-class-def", + cfg.model_class_def, + "--model-lm", + cfg.model_lm, + "--device", + cfg.device, + ]] + if step == 5: + return [[ + py, + "src/step_05_verify_uses_extends/verify_uses_extends.py", + "--root", + str(job_processed_root), + "--k", + "0", + "--batch-size", + "25", + ]] + if step == 6: + return [[py, "src/step_06_extract_paragraphs/extract_arxiv_paragraphs.py", "--root", str(job_processed_root)]] + if step == 7: + return [ + [py, "src/step_07_extract_and_refine/extract_contributions_from_citations.py", "--root", str(job_processed_root)], + [py, "src/step_07_extract_and_refine/refine_and_filter_clusters_llm.py", "--root", str(job_processed_root), "--inplace", "--overwrite"], + ] + raise ValueError(f"Unknown step: {step}") + + +def _write_single_id_file(job_dir: Path, arxiv_id: str) -> Path: + ids_path = job_dir / "input_ids.json" + payload = [{"id": arxiv_id, "title": "", "id_type": "ArXiv"}] + ids_path.write_text(json.dumps(payload, indent=2), encoding="utf-8") + return ids_path + + +def _write_run_metadata(cfg: PipelineConfig, job_dir: Path, paper_id: str, arxiv_id: str) -> None: + payload = { + "paper_input": cfg.paper_input, + "paper_id": paper_id, + "arxiv_id": arxiv_id, + "source_root": str(cfg.source_root), + "steps": FULL_STEPS + ["annotation"], + "llm_provider": cfg.llm_provider, + "llm_model": cfg.llm_model, + "llm_model_step4": cfg.llm_model_step4, + "device": cfg.device, + "embedding_model": cfg.embedding_model, + "timestamp": int(time.time()), + } + (job_dir / "run_config.json").write_text(json.dumps(payload, indent=2), encoding="utf-8") + + +def _zip_job_dir(job_dir: Path) -> Path: + zip_base = job_dir.parent / job_dir.name + archive = shutil.make_archive(str(zip_base), "zip", root_dir=str(job_dir)) + return Path(archive) + + +def _tail_log(path: Path, max_lines: int = 60) -> str: + try: + lines = path.read_text(encoding="utf-8", errors="ignore").splitlines() + except Exception: + return "" + if not lines: + return "" + return "\n".join(lines[-max_lines:]) + + +def _load_json(path: Path, default=None): + try: + return json.loads(path.read_text(encoding="utf-8")) + except Exception: + return default + + +def _write_summary_and_zip(job_dir: Path, summary_lines: List[str]) -> Path: + (job_dir / "summary.txt").write_text("\n".join(summary_lines), encoding="utf-8") + return _zip_job_dir(job_dir) + + +def _count_verified_uses_extends(payload: dict) -> int: + records = payload.get("confirmed") or payload.get("verified_contexts") or payload.get("contexts") or payload.get("items") or [] + if not isinstance(records, list): + return 0 + accepted = {"USES", "EXTENDS", "Uses", "Extends"} + return sum(1 for item in records if isinstance(item, dict) and item.get("label") in accepted) + + +def _stop_reason_after_step(step: int, paper_dir: Path) -> str | None: + if step == 1: + if not paper_dir.exists(): + return "metadata could not be fetched for this paper" + if not (paper_dir / "processed_main.tex").exists(): + return "arXiv source could not be retrieved or converted for this paper" + citations = _load_json(paper_dir / "citations_metadata.json", []) + if not isinstance(citations, list) or not citations: + return "Semantic Scholar returned no citing papers for this target paper" + + if step == 3: + usage = _load_json(paper_dir / "usage_contexts.json", {}) + if not isinstance(usage, dict): + return "citation usage contexts could not be built" + if int(usage.get("num_contexts") or 0) == 0: + return "no citation usage contexts were found" + + if step == 4: + labels = _load_json(paper_dir / "usage_context_labels.json", {}) + contexts = labels.get("labels") if isinstance(labels, dict) else None + if not isinstance(contexts, list) or not contexts: + return "citation-function labeling produced no labeled contexts" + + if step == 5: + verified = _load_json(paper_dir / "usage_uses_extends_verified.json", {}) + if not isinstance(verified, dict): + return "USES/EXTENDS verification did not produce an output file" + if _count_verified_uses_extends(verified) == 0: + return "no downstream citations were verified as USES or EXTENDS" + + if step == 6: + paragraphs = _load_json(paper_dir / "usage_citing_paragraphs.json", {}) + citing = paragraphs.get("citing_papers") if isinstance(paragraphs, dict) else None + if not isinstance(citing, list) or not citing: + return "no citing-paper paragraphs could be extracted from arXiv" + usable = [ + item for item in citing + if isinstance(item, dict) + and not item.get("error") + and (item.get("matched_paragraphs") or item.get("target_citing_paragraphs")) + ] + if not usable: + return "arXiv paragraph extraction returned no usable citing-paper text" + + if step == 7: + contributions = _load_json(paper_dir / "usage_contributions.json", {}) + items = contributions.get("contributions") if isinstance(contributions, dict) else None + if not isinstance(items, list) or not items: + return "no downstream target-contribution evidence could be extracted" + refined = _load_json(paper_dir / "usage_discovery_from_contributions.json", {}) + clusters = refined.get("clusters") if isinstance(refined, dict) else None + if not isinstance(clusters, list) or not clusters: + return "no valid downstream usage clusters survived refinement" + + return None + + +def run_pipeline(cfg: PipelineConfig, output_root: Path) -> Generator[Tuple[str, Optional[str]], None, PipelineResult]: + output_root.mkdir(parents=True, exist_ok=True) + job_id = f"job_{int(time.time())}_{uuid.uuid4().hex[:8]}" + job_dir = output_root / job_id + job_processed_root = job_dir / "processed_papers" + job_logs = job_dir / "logs" + + job_processed_root.mkdir(parents=True, exist_ok=True) + job_logs.mkdir(parents=True, exist_ok=True) + + arxiv_id = parse_arxiv_id(cfg.paper_input) + paper_id = arxiv_id + ids_path = _write_single_id_file(job_dir, arxiv_id) + _write_run_metadata(cfg, job_dir, paper_id, arxiv_id) + + base_env = os.environ.copy() + base_env["LLM_PROVIDER"] = cfg.llm_provider + base_env["LLM_MODEL"] = cfg.llm_model + + summary_lines: List[str] = [] + paper_dir = job_processed_root / paper_id + + max_step = 8 + for step in FULL_STEPS: + label = STEP_LABELS[step] + log_file = job_logs / f"step_{step:02d}.log" + summary_lines.append(f"[{step}] {label}") + yield (f"Step {step}/{max_step}: {label}", None) + + env = base_env.copy() + if step == 5 and cfg.llm_model_step4: + env["LLM_MODEL"] = cfg.llm_model_step4 + + with log_file.open("w", encoding="utf-8") as lf: + return_code = 0 + failed_cmd: List[str] | None = None + for cmd in _build_commands(cfg, step, job_processed_root, paper_id, ids_path): + lf.write(f"$ {' '.join(cmd)}\n\n") + proc = subprocess.Popen( + cmd, + cwd=str(cfg.repo_root), + stdout=subprocess.PIPE, + stderr=subprocess.STDOUT, + text=True, + encoding="utf-8", + errors="ignore", + env=env, + ) + assert proc.stdout is not None + for line in proc.stdout: + lf.write(line) + return_code = proc.wait() + if return_code != 0: + failed_cmd = cmd + break + + if return_code != 0: + summary_lines.append(f"FAILED at step {step}") + zip_path = _write_summary_and_zip(job_dir, summary_lines) + tail = _tail_log(log_file) + if tail: + yield ( + f"Step {step} failed.\n\nCommand: {' '.join(failed_cmd or [])}\n\nLast log lines:\n{tail}", + str(zip_path), + ) + else: + yield (f"Step {step} failed. Command: {' '.join(failed_cmd or [])}", str(zip_path)) + return PipelineResult(job_id=job_id, job_dir=job_dir, paper_dir=paper_dir, zip_path=zip_path) + else: + yield (f"Step {step} complete", None) + + if step == 1 and not paper_dir.exists(): + summary_lines.append("FAILED: fetch_metadata did not create paper directory") + zip_path = _write_summary_and_zip(job_dir, summary_lines) + yield (f"Step 1 finished but paper dir missing: {paper_dir}", str(zip_path)) + return PipelineResult(job_id=job_id, job_dir=job_dir, paper_dir=paper_dir, zip_path=zip_path) + + stop_reason = _stop_reason_after_step(step, paper_dir) + if stop_reason: + message = f"{STOP_PREFIX} {stop_reason}." + summary_lines.append(message) + zip_path = _write_summary_and_zip(job_dir, summary_lines) + yield (message, str(zip_path)) + return PipelineResult(job_id=job_id, job_dir=job_dir, paper_dir=paper_dir, zip_path=zip_path) + + summary_lines.append("SUCCESS") + zip_path = _write_summary_and_zip(job_dir, summary_lines) + yield ("Pipeline completed successfully.", str(zip_path)) + return PipelineResult(job_id=job_id, job_dir=job_dir, paper_dir=paper_dir, zip_path=zip_path) diff --git a/hf_space/scipath_live/README.md b/hf_space/scipath_live/README.md new file mode 100644 index 0000000000000000000000000000000000000000..dd48b743d24f10d788447c5a84ff2d6fa98d637a --- /dev/null +++ b/hf_space/scipath_live/README.md @@ -0,0 +1,38 @@ +# SciPath Live (self-contained) + +Minimal SciPath Task A CodeAgent used by **System Run → Live Run**. + +## Included methods + +| Method | Tools | +|--------|--------| +| `parametric` | `check_answer_format` | +| `websearch_deep` | `web_search_tool`, `deep_web_search_tool`, `check_answer_format` | + +## Demo claim + +`data/example_claim.json` — one claim from AVerImaTeC (`2505.17978`, claim C1), including gold enabling contributions for Task A evaluation. + +## CLI + +From `hf_space/`: + +```bash +python -m scipath_live.cli run --method parametric --output-dir runs/system_live/test +python -m scipath_live.cli evaluate --result-json runs/system_live/test/result.json +``` + +## Env + +- `GOOGLE_GENAI_API_KEY` or `GEMINI_API_KEY` (generation + judge) +- For `websearch_deep`: install Chromium into the demo-local cache: + +```bash +# from demo_2026_07/ +export PLAYWRIGHT_BROWSERS_PATH="$(pwd)/.playwright" +python -m playwright install chromium +``` + +`system_live_runner` / `tools.py` always force `PLAYWRIGHT_BROWSERS_PATH` to `demo_2026_07/.playwright` so Cursor sandbox temp paths are not used. + +This package does **not** depend on `dryrun/` or other repos. diff --git a/hf_space/scipath_live/__init__.py b/hf_space/scipath_live/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..24e446d599d52f48dc58ac5a9540c2d6d4bf5afc --- /dev/null +++ b/hf_space/scipath_live/__init__.py @@ -0,0 +1,19 @@ +"""Self-contained SciPath Task A CodeAgent (parametric + websearch_deep).""" + +from .pipeline import load_example_claim + +__all__ = ["load_example_claim", "run_method", "evaluate_task_a"] + + +def run_method(*args, **kwargs): + from .pipeline import run_method as _run + + return _run(*args, **kwargs) + + +def evaluate_task_a(*args, **kwargs): + from .pipeline import evaluate_task_a as _eval + + return _eval(*args, **kwargs) + +__all__ = ["evaluate_task_a", "load_example_claim", "run_method"] diff --git a/hf_space/scipath_live/agent.py b/hf_space/scipath_live/agent.py new file mode 100644 index 0000000000000000000000000000000000000000..eedc0b0f80a27341435ee3d1cf39274f6bc37ded --- /dev/null +++ b/hf_space/scipath_live/agent.py @@ -0,0 +1,227 @@ +"""Slim SciPath CodeAgent runner (parametric + websearch_deep only).""" + +from __future__ import annotations + +import json +import os +import time +from datetime import datetime +from pathlib import Path +from typing import Any, Optional + +from dotenv import load_dotenv +from smolagents import CodeAgent, LiteLLMModel + +from .prompts import get_scipath_controller_prompt, parse_ingredients_from_answer +from .tools import check_answer_format, deep_web_search_tool, web_search_tool + +GEMINI_MODEL_IDS = { + "gemini-2.0-flash", + "gemini-2.5-flash", + "gemini-2.5-pro", + "gemini-3-flash-preview", + "gemini-3.1-pro-preview", + "gemini-3.5-flash", +} + + +def _memory_steps_to_list(agent: CodeAgent) -> list[dict[str, Any]]: + steps_list: list[dict[str, Any]] = [] + for i, step in enumerate(getattr(agent.memory, "steps", []) or []): + item: dict[str, Any] = { + "step_type": type(step).__name__, + "step_index": i, + "timestamp": datetime.now().isoformat(), + } + for attr in ("step_number", "task", "code_action", "observations", "model_output"): + if hasattr(step, attr): + value = getattr(step, attr) + item[attr] = str(value) if value is not None and not isinstance(value, (str, int, float, bool, list, dict)) else value + if getattr(step, "error", None): + item["error"] = str(step.error) + steps_list.append(item) + return steps_list + + +def _clip(text: Any, limit: int = 1200) -> str: + s = str(text or "").strip() + if len(s) <= limit: + return s + return s[: limit - 1] + "…" + + +def format_agent_trace(steps: list[dict[str, Any]]) -> list[str]: + """Human-readable SciFy CodeAgent step trace for the Live Run terminal.""" + lines: list[str] = [] + lines.append("=== SciFy CodeAgent trace ===") + action_steps = [s for s in steps if s.get("step_type") == "ActionStep"] + if not action_steps: + lines.append("(no action steps recorded)") + return lines + + for step in action_steps: + n = step.get("step_number") + if n is None: + n = int(step.get("step_index") or 0) + 1 + lines.append("") + lines.append(f"── Step {n} ──") + thought = _clip(step.get("model_output"), 1600) + if thought: + lines.append("Thought / model output:") + lines.extend(thought.splitlines()) + code = _clip(step.get("code_action"), 1600) + if code: + lines.append("Code action:") + lines.extend(code.splitlines()) + obs = _clip(step.get("observations"), 1200) + if obs: + lines.append("Observations:") + lines.extend(obs.splitlines()) + err = _clip(step.get("error"), 600) + if err: + lines.append(f"Error: {err}") + lines.append("") + lines.append("=== end trace ===") + return lines + + +def _print_trace(steps: list[dict[str, Any]]) -> None: + for line in format_agent_trace(steps): + print(line, flush=True) + + +def _ensure_api_keys() -> None: + load_dotenv() + google = ( + os.getenv("GOOGLE_GENAI_API_KEY") + or os.getenv("GEMINI_API_KEY") + or os.getenv("GOOGLE_API_KEY") + or "" + ).strip() + if google: + os.environ.setdefault("GEMINI_API_KEY", google) + os.environ.setdefault("GOOGLE_API_KEY", google) + os.environ.setdefault("GOOGLE_GENAI_API_KEY", google) + + +def run_agent( + query: str, + *, + model_name: str = "gemini-3-flash-preview", + method: str = "parametric", + max_steps: int = 40, +) -> tuple[Any, list[dict[str, Any]]]: + """Run SciPath Task A CodeAgent for one claim.""" + if method not in {"parametric", "websearch_deep"}: + raise ValueError(f"Unsupported method for demo live run: {method}") + + _ensure_api_keys() + litellm_model_id = ( + f"gemini/{model_name}" if model_name in GEMINI_MODEL_IDS else model_name + ) + model = LiteLLMModel( + model_id=litellm_model_id, + temperature=0.1, + max_tokens=16384, + timeout=2400, + ) + + if method == "parametric": + tool_list = [check_answer_format] + else: + tool_list = [web_search_tool, deep_web_search_tool, check_answer_format] + + agent = CodeAgent( + tools=tool_list, + model=model, + verbosity_level=2, + additional_authorized_imports=["json", "os", "typing", "pprint"], + max_steps=max_steps, + executor_kwargs={"timeout_seconds": None}, + ) + prompt = get_scipath_controller_prompt(query, method) + result = agent.run(prompt) + logs = _memory_steps_to_list(agent) + return result, logs + + +def run_claim_to_dir( + *, + claim: str, + paper_id: str, + claim_idx: int, + paper_title: str, + method: str, + output_dir: Path, + model_name: str = "gemini-3-flash-preview", +) -> dict[str, Any]: + """Run one claim and write answers.jsonl + logs.jsonl into output_dir.""" + output_dir = Path(output_dir) + output_dir.mkdir(parents=True, exist_ok=True) + + problem_id = f"{paper_id}__claim{claim_idx}" + claim_row = { + "type": "problem", + "format_version": "1.0", + "problem_id": problem_id, + "problem_version": "1.0", + "domain": "ai", + "subdomain": "scipath", + "claim": claim, + } + index_row = { + "problem_id": problem_id, + "paper_id": paper_id, + "claim_idx": claim_idx, + "paper_title": paper_title, + "claim": claim, + "split": "dev", + } + (output_dir / "claim.jsonl").write_text( + json.dumps(claim_row, ensure_ascii=False) + "\n", encoding="utf-8" + ) + (output_dir / "task_index.jsonl").write_text( + json.dumps(index_row, ensure_ascii=False) + "\n", encoding="utf-8" + ) + + print(f"[scipath_live] Starting CodeAgent method={method} model={model_name}", flush=True) + print(f"[scipath_live] Claim: {claim}", flush=True) + t0 = time.time() + answer, logs = run_agent(claim, model_name=model_name, method=method) + wall = time.time() - t0 + print(f"[scipath_live] Agent finished in {wall:.1f}s ({len(logs)} memory steps)", flush=True) + _print_trace(logs) + + ingredients = parse_ingredients_from_answer(answer) + if not ingredients and isinstance(answer, dict): + ingredients = parse_ingredients_from_answer(answer) + print(f"[scipath_live] Parsed {len(ingredients)} ingredients", flush=True) + + out = { + "type": "scipath_task_a", + "format_version": "1.0", + "problem_id": problem_id, + "problem_version": "1.0", + "team": "scify", + "run_id": f"live_{method}", + "ingredients": ingredients, + "raw_answer": answer if isinstance(answer, (dict, list, str)) else str(answer), + "wall_clock_time": wall, + "method": method, + "model_name": model_name, + "paper_id": paper_id, + "claim_idx": claim_idx, + "paper_title": paper_title, + "claim": claim, + } + (output_dir / "answers.jsonl").write_text( + json.dumps(out, ensure_ascii=False) + "\n", encoding="utf-8" + ) + (output_dir / "logs.jsonl").write_text( + json.dumps({"problem_id": problem_id, "steps": logs}, ensure_ascii=False) + "\n", + encoding="utf-8", + ) + (output_dir / "result.json").write_text( + json.dumps(out, indent=2, ensure_ascii=False) + "\n", encoding="utf-8" + ) + return out diff --git a/hf_space/scipath_live/cli.py b/hf_space/scipath_live/cli.py new file mode 100644 index 0000000000000000000000000000000000000000..46af12ca2c690a6913d738295f25b4ebe535cb7f --- /dev/null +++ b/hf_space/scipath_live/cli.py @@ -0,0 +1,69 @@ +#!/usr/bin/env python3 +"""CLI entry for SciPath Live Run (used as a subprocess from Streamlit).""" + +from __future__ import annotations + +import argparse +import json +import sys +from pathlib import Path + +# Allow `python -m scipath_live.cli` and `python scipath_live/cli.py` +PACKAGE_ROOT = Path(__file__).resolve().parent +HF_SPACE = PACKAGE_ROOT.parent +if str(HF_SPACE) not in sys.path: + sys.path.insert(0, str(HF_SPACE)) + +from scipath_live.pipeline import evaluate_task_a, load_example_claim, run_method # noqa: E402 + + +def main() -> int: + parser = argparse.ArgumentParser(description="SciPath Live Run CLI") + parser.add_argument( + "action", + choices=["run", "evaluate"], + help="run a method or evaluate an existing run result JSON", + ) + parser.add_argument( + "--method", + default="parametric", + choices=["parametric", "websearch_deep", "codeagent_parametric", "codeagent_websearch_deep"], + ) + parser.add_argument("--output-dir", default="") + parser.add_argument("--model", default="gemini-3-flash-preview") + parser.add_argument("--judge-model", default="gemini/gemini-3.1-pro-preview") + parser.add_argument( + "--result-json", + default="", + help="For evaluate: path to result.json from a prior run", + ) + args = parser.parse_args() + + if args.action == "run": + out = Path(args.output_dir) if args.output_dir else None + result = run_method( + args.method, + output_dir=out, + model_name=args.model, + ) + print(json.dumps({"ok": True, "result": result}, ensure_ascii=False)) + return 0 + + if not args.result_json: + print(json.dumps({"ok": False, "error": "--result-json is required for evaluate"})) + return 2 + run_result = json.loads(Path(args.result_json).read_text(encoding="utf-8")) + # Ensure output_dir is set for writing judged JSON. + if not run_result.get("output_dir"): + run_result["output_dir"] = str(Path(args.result_json).parent) + judged = evaluate_task_a( + run_result=run_result, + claim_record=load_example_claim(), + judge_model=args.judge_model, + ) + print(json.dumps({"ok": True, "judged": judged}, ensure_ascii=False)) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/hf_space/scipath_live/data/example_claim.json b/hf_space/scipath_live/data/example_claim.json new file mode 100644 index 0000000000000000000000000000000000000000..985188a650604846d2c4d925884e35ad305a8da3 --- /dev/null +++ b/hf_space/scipath_live/data/example_claim.json @@ -0,0 +1,124 @@ +{ + "paper_id": "2505.17978", + "claim_idx": 0, + "claim_id": "C1", + "paper_title": "AVerImaTeC: A Dataset for Automatic Verification of Image-Text Claims with Evidence from the Web", + "claim": "Dataset: A real-world image-text claim verification dataset annotated with question-answer reasoning paths and web evidence, enabling the training and evaluation of multimodal fact-checking systems.", + "enabling_contributions": [ + { + "enabling_contribution_id": "2505.17978::ingredient::01", + "enabling_contribution": "Source fact-checking articles containing multimodal claims", + "canonical_ref_id": "__NONE__", + "canonical_grounding": null, + "additional_ref_ids": [ + "ref_averitec", + "ref_ammeba" + ], + "additional_groundings": [ + { + "ref_id": "ref_averitec", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web", + "ref_year": "", + "ref_authors": "" + }, + { + "ref_id": "ref_ammeba", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "AMMeBa: A Large-Scale Survey and Dataset of Media-Based Misinformation In-The-Wild", + "ref_year": "", + "ref_authors": "" + } + ], + "canonical_annotation": { + "role": "DATA_SOURCE", + "roles": [ + "DATA_SOURCE" + ], + "contribution": "Provides the raw real-world image-text claims that are normalized and annotated to form the dataset.", + "rationale": "The dataset is a composite of claims extracted from multiple prior fact-checking resources. No single prior study fully represents this substrate, making NONE the correct canonical grounding, with the primary sources listed as additional groundings.", + "evidence_span": "we incorporated partially annotated articles from three sources: 1) filtered articles from AVeriTec... 2) articles verifying image-related claims from AMMEBA" + } + }, + { + "enabling_contribution_id": "2505.17978::ingredient::02", + "enabling_contribution": "Question-answering representation for verification rationale", + "canonical_ref_id": "ref_complex", + "canonical_grounding": { + "ref_id": "ref_complex", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "Complex Claim Verification with Evidence Retrieved in the Wild", + "ref_year": "", + "ref_authors": "" + }, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "CONCEPTUAL_FRAMEWORK", + "roles": [ + "CONCEPTUAL_FRAMEWORK" + ], + "contribution": "Provides the QA-based reasoning representation used to structure the evidence retrieval and verification process.", + "rationale": "The paper explicitly adopts QA as the reasoning representation for fact-checking, citing prior work that established this paradigm. Chen et al. (CLAIMDECOMP) is a clean representative for QA-based claim decomposition.", + "evidence_span": "To capture the rationale in claim verification, a complex reasoning task, various reasoning representations have been explored... and QA" + } + }, + { + "enabling_contribution_id": "2505.17978::ingredient::03", + "enabling_contribution": "Four-way veracity labeling schema", + "canonical_ref_id": "ref_averitec", + "canonical_grounding": { + "ref_id": "ref_averitec", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web", + "ref_year": "", + "ref_authors": "" + }, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "CONCEPTUAL_FRAMEWORK", + "roles": [ + "CONCEPTUAL_FRAMEWORK" + ], + "contribution": "Provides the specific four-way label space (supported, refuted, not enough evidence, conflicting) used to annotate the claims.", + "rationale": "The paper explicitly adopts the exact four-way veracity labeling schema introduced in AVeriTeC.", + "evidence_span": "We follow the four-way veracity labeling schema from [Schlichtkrull et al., 2023]: supported, refuted, not enough evidence, and conflicting/cherry-picking." + } + }, + { + "enabling_contribution_id": "2505.17978::ingredient::04", + "enabling_contribution": "Multi-stage annotation pipeline with evidence sufficiency checks", + "canonical_ref_id": "ref_averitec", + "canonical_grounding": { + "ref_id": "ref_averitec", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web", + "ref_year": "", + "ref_authors": "" + }, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "EVALUATION_PROTOCOL", + "roles": [ + "EVALUATION_PROTOCOL" + ], + "contribution": "Provides the foundational multi-stage annotation and verification workflow that the paper extends for image-text claims.", + "rationale": "The paper explicitly states that its annotation pipeline is an extension of the process proposed in AVeriTeC, making it the canonical grounding for this workflow.", + "evidence_span": "extending the annotation process proposed in [Schlichtkrull et al., 2023] to the domain of image-text claims." + } + } + ], + "source": "AVerImaTeC" +} diff --git a/hf_space/scipath_live/judge.py b/hf_space/scipath_live/judge.py new file mode 100644 index 0000000000000000000000000000000000000000..12af5db4252210e5cf67442d9eb59b4a9ab4ee63 --- /dev/null +++ b/hf_space/scipath_live/judge.py @@ -0,0 +1,305 @@ +"""Task A ingredient matching judge (demo-local, single-claim).""" + +from __future__ import annotations + +import asyncio +import json +import os +import re +from typing import Any + +import litellm +from dotenv import load_dotenv + +MATCHING_JUDGE_PROMPT = """\ +You are evaluating whether predicted enabling contributions match reference enabling contributions for a scientific discovery. + +A full match requires that the predicted item express the same functional requirement needed to realize the discovery. It may use different wording, but it must preserve the key resource, method, process, role, and level of specificity. + +A partial match means the predicted item is related but too broad, too narrow, missing an important role/detail, or merges multiple distinct reference requirements. + +Do not mark broad umbrella items as full matches for multiple distinct reference contributions. + +Discovery claim: +{claim} + +Reference enabling contributions: +{references} + +Predicted enabling contributions: +{predicted} + +Return candidate matches only where match is "full" or "partial". Omit unrelated pairs. + +Respond with JSON only (no markdown): +{{"matches": [{{"reference_idx": <1-based index>, "predicted_idx": <1-based index>, "match": "full"|"partial", "reasoning": ""}}]}}\ +""" + + +def _format_reference_ingredient(silver_ingredient: dict) -> str: + annotation = silver_ingredient.get("canonical_annotation") or {} + roles = annotation.get("roles") + if isinstance(roles, list): + role_text = ", ".join(str(r) for r in roles) + else: + role_text = annotation.get("role", "") or silver_ingredient.get("role", "") + + ingredient = ( + silver_ingredient.get("enabling_contribution") + or silver_ingredient.get("ingredient") + or silver_ingredient.get("description") + or "" + ) + lines = [f"Ingredient: {ingredient}"] + if role_text: + lines.append(f"Role: {role_text}") + if annotation.get("contribution"): + lines.append(f"Contribution: {annotation['contribution']}") + if annotation.get("rationale"): + lines.append(f"Rationale: {annotation['rationale']}") + return "\n".join(lines) + + +def _format_predicted_ingredient(idx: int, ingredient: dict) -> str: + lines = [f"[{idx}] {ingredient.get('description', '')}"] + if ingredient.get("role"): + lines.append(f" Role: {ingredient['role']}") + if ingredient.get("rationale"): + lines.append(f" Rationale: {ingredient['rationale']}") + return "\n".join(lines) + + +def _parse_matching_judgment(raw: str, n_references: int, n_predicted: int) -> dict | None: + raw = re.sub(r"^```(?:json)?\s*", "", raw.strip()) + raw = re.sub(r"\s*```$", "", raw) + try: + data = json.loads(raw) + except json.JSONDecodeError: + return None + raw_matches = data.get("matches") + if not isinstance(raw_matches, list): + return None + matches = [] + for match in raw_matches: + if not isinstance(match, dict): + continue + ref_idx = match.get("reference_idx") + pred_idx = match.get("predicted_idx") + label = str(match.get("match", "")).lower() + if label not in {"full", "partial"}: + continue + if not isinstance(ref_idx, int) or not isinstance(pred_idx, int): + continue + if not (1 <= ref_idx <= n_references and 1 <= pred_idx <= n_predicted): + continue + matches.append( + { + "reference_idx": ref_idx, + "predicted_idx": pred_idx, + "match": label, + "reasoning": str(match.get("reasoning", "")), + } + ) + return {"matches": matches} + + +def maximum_full_matching( + matches: list[dict], n_references: int, n_predicted: int +) -> list[dict]: + edges: dict[int, list[dict]] = {i: [] for i in range(1, n_references + 1)} + seen: set[tuple[int, int]] = set() + for match in matches: + if match.get("match") != "full": + continue + ref_idx = match.get("reference_idx") + pred_idx = match.get("predicted_idx") + if not isinstance(ref_idx, int) or not isinstance(pred_idx, int): + continue + if not (1 <= ref_idx <= n_references and 1 <= pred_idx <= n_predicted): + continue + key = (ref_idx, pred_idx) + if key in seen: + continue + seen.add(key) + edges[ref_idx].append(match) + for ref_edges in edges.values(): + ref_edges.sort(key=lambda m: m["predicted_idx"]) + + pred_to_match: dict[int, dict] = {} + + def _try_match(ref_idx: int, visited: set[int]) -> bool: + for edge in edges.get(ref_idx, []): + pred_idx = edge["predicted_idx"] + if pred_idx in visited: + continue + visited.add(pred_idx) + previous = pred_to_match.get(pred_idx) + if previous is None or _try_match(previous["reference_idx"], visited): + pred_to_match[pred_idx] = edge + return True + return False + + for ref_idx in range(1, n_references + 1): + _try_match(ref_idx, set()) + return sorted( + pred_to_match.values(), key=lambda m: (m["reference_idx"], m["predicted_idx"]) + ) + + +async def judge_ingredient_matching( + model: str, + claim: str, + reference_ingredients: list[dict], + predicted_ingredients: list[dict], + max_retries: int = 3, +) -> dict: + references_text = "\n\n".join( + f"[{i + 1}] " + _format_reference_ingredient(ing).replace("\n", "\n ") + for i, ing in enumerate(reference_ingredients) + ) or "(none)" + predicted_text = "\n\n".join( + _format_predicted_ingredient(i + 1, ing) + for i, ing in enumerate(predicted_ingredients) + ) or "(none)" + prompt = MATCHING_JUDGE_PROMPT.format( + claim=claim, references=references_text, predicted=predicted_text + ) + for _ in range(max_retries): + msg = await litellm.acompletion( + model=model, + max_tokens=8192, + messages=[{"role": "user", "content": prompt}], + ) + raw = msg.choices[0].message.content.strip() + result = _parse_matching_judgment( + raw, len(reference_ingredients), len(predicted_ingredients) + ) + if result is not None: + return result + raise ValueError("Failed to parse valid matching judgment") + + +def _gold_to_ui_ingredients(gold: list[dict]) -> list[dict[str, str]]: + out = [] + for g in gold: + ann = g.get("canonical_annotation") or {} + role = "" + if isinstance(ann.get("roles"), list) and ann["roles"]: + role = str(ann["roles"][0]) + else: + role = str(ann.get("role") or g.get("role") or "") + out.append( + { + "description": str( + g.get("enabling_contribution") or g.get("description") or "" + ), + "role": role, + } + ) + return out + + +def evaluate_prediction( + *, + claim: str, + gold_ingredients: list[dict], + predicted_ingredients: list[dict], + judge_model: str = "gemini/gemini-3.1-pro-preview", +) -> dict[str, Any]: + """Sync wrapper: judge one claim and compute Task A scores.""" + load_dotenv() + google = ( + os.getenv("GOOGLE_GENAI_API_KEY") + or os.getenv("GEMINI_API_KEY") + or os.getenv("GOOGLE_API_KEY") + or "" + ).strip() + if google: + os.environ.setdefault("GEMINI_API_KEY", google) + os.environ.setdefault("GOOGLE_API_KEY", google) + + matching = asyncio.run( + judge_ingredient_matching( + judge_model, claim, gold_ingredients, predicted_ingredients + ) + ) + matches = matching.get("matches") or [] + full_pairs = maximum_full_matching( + matches, len(gold_ingredients), len(predicted_ingredients) + ) + n_matched = len(full_pairs) + n_ref = len(gold_ingredients) + n_pred = len(predicted_ingredients) + recall = (n_matched / n_ref) if n_ref else 0.0 + precision = (n_matched / n_pred) if n_pred else 0.0 + f1 = ( + (2 * precision * recall / (precision + recall)) + if (precision + recall) + else 0.0 + ) + + # Build recall_judgments for UI gold column. + full_by_ref = {m["reference_idx"]: m for m in full_pairs} + partial_by_ref: dict[int, list[dict]] = {} + for m in matches: + if m.get("match") == "partial": + partial_by_ref.setdefault(m["reference_idx"], []).append(m) + + judgments = [] + for i, g in enumerate(gold_ingredients, start=1): + desc = str(g.get("enabling_contribution") or g.get("description") or "") + ann = g.get("canonical_annotation") or {} + role = "" + if isinstance(ann.get("roles"), list) and ann["roles"]: + role = str(ann["roles"][0]) + else: + role = str(ann.get("role") or "") + full = full_by_ref.get(i) + partials = partial_by_ref.get(i) or [] + if full: + pred = predicted_ingredients[full["predicted_idx"] - 1] + judgments.append( + { + "reference_idx": i, + "reference_ingredient": desc, + "reference_role": role, + "covered": True, + "best_match_idx": full["predicted_idx"], + "reasoning": full.get("reasoning", ""), + "best_match": pred, + "partial_matches": [], + } + ) + else: + judgments.append( + { + "reference_idx": i, + "reference_ingredient": desc, + "reference_role": role, + "covered": False, + "best_match_idx": None, + "reasoning": "", + "best_match": None, + "partial_matches": [ + { + "reference_idx": p["reference_idx"], + "predicted_idx": p["predicted_idx"], + "match": "partial", + "reasoning": p.get("reasoning", ""), + } + for p in partials + ], + } + ) + + return { + "recall": recall, + "precision": precision, + "f1": f1, + "matched_pairs": full_pairs, + "partial_matches": [m for m in matches if m.get("match") == "partial"], + "recall_judgments": judgments, + "gold_ingredients": _gold_to_ui_ingredients(gold_ingredients), + "predicted_ingredients": predicted_ingredients, + "judge_model": judge_model, + } diff --git a/hf_space/scipath_live/pipeline.py b/hf_space/scipath_live/pipeline.py new file mode 100644 index 0000000000000000000000000000000000000000..4fc7c9a40f7292779643dfa203587f941cf2e6d2 --- /dev/null +++ b/hf_space/scipath_live/pipeline.py @@ -0,0 +1,101 @@ +"""High-level Live Run API used by Streamlit (self-contained).""" + +from __future__ import annotations + +import json +import uuid +from pathlib import Path +from typing import Any, Optional + +PACKAGE_DIR = Path(__file__).resolve().parent +DATA_DIR = PACKAGE_DIR / "data" +EXAMPLE_CLAIM_PATH = DATA_DIR / "example_claim.json" +DEFAULT_RUNS_ROOT = PACKAGE_DIR.parent / "runs" / "system_live" + +METHOD_CLI = { + "codeagent_parametric": "parametric", + "parametric": "parametric", + "codeagent_websearch_deep": "websearch_deep", + "websearch_deep": "websearch_deep", +} + + +def load_example_claim(path: Optional[Path] = None) -> dict[str, Any]: + claim_path = Path(path) if path else EXAMPLE_CLAIM_PATH + data = json.loads(claim_path.read_text(encoding="utf-8")) + if not isinstance(data, dict) or not data.get("claim"): + raise ValueError(f"Invalid example claim at {claim_path}") + return data + + +def new_job_dir(runs_root: Optional[Path] = None, method: str = "parametric") -> Path: + root = Path(runs_root) if runs_root else DEFAULT_RUNS_ROOT + job = root / f"job_{method}_{uuid.uuid4().hex[:10]}" + job.mkdir(parents=True, exist_ok=True) + return job + + +def run_method( + method_id: str, + *, + claim_record: Optional[dict[str, Any]] = None, + output_dir: Optional[Path] = None, + model_name: str = "gemini-3-flash-preview", +) -> dict[str, Any]: + """Run parametric or websearch_deep on the demo example claim (or provided record).""" + method = METHOD_CLI.get(method_id) + if not method: + raise ValueError(f"Unknown method id: {method_id}") + from .agent import run_claim_to_dir + + record = claim_record or load_example_claim() + out_dir = Path(output_dir) if output_dir else new_job_dir(method=method) + result = run_claim_to_dir( + claim=str(record["claim"]), + paper_id=str(record.get("paper_id") or "example"), + claim_idx=int(record.get("claim_idx") or 0), + paper_title=str(record.get("paper_title") or ""), + method=method, + output_dir=out_dir, + model_name=model_name, + ) + result["output_dir"] = str(out_dir) + result["method_id"] = ( + "codeagent_parametric" if method == "parametric" else "codeagent_websearch_deep" + ) + return result + + +def evaluate_task_a( + *, + run_result: dict[str, Any], + claim_record: Optional[dict[str, Any]] = None, + judge_model: str = "gemini/gemini-3.1-pro-preview", +) -> dict[str, Any]: + """Judge a finished live run against gold enabling contributions.""" + from .judge import evaluate_prediction + + record = claim_record or load_example_claim() + gold = list(record.get("enabling_contributions") or []) + predicted = list(run_result.get("ingredients") or []) + judged = evaluate_prediction( + claim=str(record.get("claim") or run_result.get("claim") or ""), + gold_ingredients=gold, + predicted_ingredients=predicted, + judge_model=judge_model, + ) + payload = { + "paper_id": record.get("paper_id"), + "claim_idx": record.get("claim_idx"), + "paper_title": record.get("paper_title"), + "claim": record.get("claim"), + "method_id": run_result.get("method_id"), + "method": run_result.get("method"), + **judged, + } + out_dir = run_result.get("output_dir") + if out_dir: + path = Path(out_dir) / "task_a_judged.json" + path.write_text(json.dumps(payload, indent=2, ensure_ascii=False) + "\n", encoding="utf-8") + payload["judged_path"] = str(path) + return payload diff --git a/hf_space/scipath_live/prompts.py b/hf_space/scipath_live/prompts.py new file mode 100644 index 0000000000000000000000000000000000000000..4e6fbcb9e8b89f732add8916bfcfaf6e7a94fbef --- /dev/null +++ b/hf_space/scipath_live/prompts.py @@ -0,0 +1,170 @@ +"""SciPath Task A prompts and answer validation (demo-local copy).""" + +from __future__ import annotations + +import json +from typing import Any + +SCIPATH_ROLES = { + "CORE_METHOD", + "CONCEPTUAL_FRAMEWORK", + "DATA_SOURCE", + "TRAINING_DATA", + "MODEL_INITIALIZATION", + "EVALUATION_PROTOCOL", + "IMPLEMENTATION_TOOLING", +} + +TOOLS_GUIDE_PARAMETRIC = "" + +TOOLS_GUIDE_WEBSEARCH_DEEP = """ + You have access to web search and must use it to get relevant information about the claim. + + - web_search_tool(): get a list of (url,summary of the webpage) from the WWW + - deep_web_search_tool(): retrieve content from a website given the url. The content is a dictionary with keys 'title', 'url' and 'content'. + + """ + +PROCEDURE_PARAMETRIC = """ + +Step 1. Analyze the discovery claim and identify necessary enabling contributions from your knowledge. +Step 2. Validate JSON with check_answer_format(), then return final_answer(). + +""" + +PROCEDURE_WEBSEARCH_DEEP = """ + +Step 1. Decompose the claim into a set of searchable short queries. +Step 2. Start from the first query. Use web_search_tool() to get relevant information about the claim. +Step 2.1. Based on the snippets, select the most relevant urls from search results and use deep_web_search_tool() on each url to get more relevant context. +Step 2.2. If the collected information is sufficient, proceed to Step 3. Otherwise, move to the next query and repeat Step 2. +Step 3. Summarize all the information retrieved from the web. +Step 4. Synthesize the minimal set of enabling contributions as JSON ingredients. +Step 5. Validate JSON with check_answer_format(), then return final_answer() with the ingredients object. + +""" + +SCIPATH_OUTPUT_SPEC = """ + +Return a JSON object with exactly one top-level key: "ingredients". + +Each ingredient must have: +- "description": concise functional description of the enabling contribution +- "role": one of CORE_METHOD, CONCEPTUAL_FRAMEWORK, DATA_SOURCE, TRAINING_DATA, + MODEL_INITIALIZATION, EVALUATION_PROTOCOL, IMPLEMENTATION_TOOLING +- "rationale": one sentence explaining why it is necessary + +Example: +{ + "ingredients": [ + { + "description": "Pretrained transformer encoder for sentence embeddings", + "role": "MODEL_INITIALIZATION", + "rationale": "Needed to initialize the downstream model used in the discovery." + } + ] +} + +Rules: +- Only include necessary enabling contributions for realizing the discovery claim. +- Exclude generic background, motivation, baselines, and comparison-only datasets. +- Use concrete strings; never copy placeholder values such as "...". +- Invoke check_answer_format() on your JSON, then call final_answer() with the same object. + +""" + +SCIPATH_TASK_DESCRIPTION = """ +You are forecasting enabling contributions required to realize a scientific discovery claim +(SciPath Task A, Setting 1: claim only). + +An enabling contribution is a functional component whose absence would prevent the discovery +from being realized in its claimed form. It may come from prior work or be paper-specific +(e.g., a data pipeline or annotation procedure). + +Use available search/retrieval tools when present to identify plausible prior work, then +synthesize the minimal set of enabling contributions. +""" + +SCIPATH_MUST_FOLLOW = """ + +- Gather evidence with tools before producing the final ingredient list when tools are available. +- Separate evidence gathering from final synthesis across steps. +- NEVER return final_answer without validating via check_answer_format(). +- Write only one code block per step when executing code. + +""" + +_METHOD_TOOLS = { + "parametric": TOOLS_GUIDE_PARAMETRIC, + "websearch_deep": TOOLS_GUIDE_WEBSEARCH_DEEP, +} + +_METHOD_PROCEDURE = { + "parametric": PROCEDURE_PARAMETRIC, + "websearch_deep": PROCEDURE_WEBSEARCH_DEEP, +} + + +def get_scipath_controller_prompt(query: str, method: str) -> str: + tools = _METHOD_TOOLS.get(method, TOOLS_GUIDE_WEBSEARCH_DEEP) + procedure = _METHOD_PROCEDURE.get(method, PROCEDURE_WEBSEARCH_DEEP) + return f""" +{SCIPATH_TASK_DESCRIPTION} + +{tools} + +{procedure} + +{SCIPATH_OUTPUT_SPEC} + +{SCIPATH_MUST_FOLLOW} + +Discovery claim: +{query} +""" + + +def validate_scipath_answer(final_answer_dict: dict) -> tuple[bool, str | None, dict | None]: + if not isinstance(final_answer_dict, dict): + return False, "Final answer must be a JSON object.", None + + ingredients = final_answer_dict.get("ingredients") + if not isinstance(ingredients, list) or not ingredients: + return False, "Field 'ingredients' must be a non-empty list.", None + + placeholder_values = {"", "...", "…", "null", "none", "n/a"} + cleaned: list[dict[str, str]] = [] + for idx, ingredient in enumerate(ingredients): + if not isinstance(ingredient, dict): + return False, f"Ingredient {idx} must be an object.", None + description = str(ingredient.get("description", "")).strip() + role = str(ingredient.get("role", "")).strip().upper() + rationale = str(ingredient.get("rationale", "")).strip() + if ( + description.lower() in placeholder_values + or role.lower() in placeholder_values + or rationale.lower() in placeholder_values + ): + return False, f"Ingredient {idx} contains placeholder or empty fields.", None + if role not in SCIPATH_ROLES: + return False, f"Ingredient {idx} has invalid role {role!r}.", None + cleaned.append( + {"description": description, "role": role, "rationale": rationale} + ) + + return True, None, {"ingredients": cleaned} + + +def parse_ingredients_from_answer(answer: Any) -> list[dict]: + if isinstance(answer, str): + try: + answer = json.loads(answer) + except json.JSONDecodeError: + return [] + if not isinstance(answer, dict): + return [] + if isinstance(answer.get("ingredients"), list): + ok, _, validated = validate_scipath_answer(answer) + if ok and validated: + return validated["ingredients"] + return [] diff --git a/hf_space/scipath_live/tools.py b/hf_space/scipath_live/tools.py new file mode 100644 index 0000000000000000000000000000000000000000..aa1003963516937757b1e7693a3f9f48f6fd2a1c --- /dev/null +++ b/hf_space/scipath_live/tools.py @@ -0,0 +1,170 @@ +"""Tools for SciPath Live: format check, web search, deep page fetch.""" + +from __future__ import annotations + +import os +import re +from io import BytesIO +from pathlib import Path +from typing import Any + +import requests +from pydantic import ValidationError +from smolagents import DuckDuckGoSearchTool, tool + +from .prompts import validate_scipath_answer + +WEBSEARCH_TOOL = DuckDuckGoSearchTool(max_results=5, rate_limit=2) + +# Prefer demo-local browser cache over Cursor/temp PLAYWRIGHT_BROWSERS_PATH. +_DEMO_PLAYWRIGHT_DIR = Path(__file__).resolve().parents[2] / ".playwright" + + +def _ensure_playwright_browsers_path() -> Path: + _DEMO_PLAYWRIGHT_DIR.mkdir(parents=True, exist_ok=True) + os.environ["PLAYWRIGHT_BROWSERS_PATH"] = str(_DEMO_PLAYWRIGHT_DIR) + return _DEMO_PLAYWRIGHT_DIR + + +def _parse_search_results(search_output: str) -> list[dict[str, str]]: + evidence_items: list[dict[str, str]] = [] + pattern = r"\[([^\]]+)\]\(([^\)]+)\)(.*?)(?=\[|$)" + for match in re.finditer(pattern, search_output or "", re.DOTALL): + title = match.group(1).strip() + url = match.group(2).strip() + content = match.group(3).strip() + summary = content.split("·", 1)[1].strip() if "·" in content else content + summary = " ".join(summary.split()).rstrip("…").strip() + if summary: + evidence_items.append( + { + "type": "web_search_result", + "source": url, + "summary": f"{title}. {summary}", + } + ) + return evidence_items + + +def _get_pdf_content(url: str) -> dict[str, str]: + try: + import PyPDF2 + + response = requests.get(url, timeout=30) + response.raise_for_status() + pdf_reader = PyPDF2.PdfReader(BytesIO(response.content)) + pages = [] + for i, page in enumerate(pdf_reader.pages): + pages.append(page.extract_text() or "") + if i >= 29: + break + content = "\n\n".join(pages) + if "Abstract" in content: + content = content.split("Abstract", 1)[1] + title = "PDF Document" + if pdf_reader.metadata and getattr(pdf_reader.metadata, "title", None): + title = pdf_reader.metadata.title or title + return {"title": title, "url": url, "content": content} + except Exception as exc: + return {"title": "Error", "url": url, "content": f"Error extracting PDF: {exc}"} + + +def get_webpage(url: str) -> dict[str, str]: + """Fetch webpage or PDF content (Playwright for HTML, PyPDF2 for PDFs).""" + if "sciencedirect" in url: + return { + "title": "ScienceDirect Article", + "url": url, + "content": "Automatically fetching content from ScienceDirect is not supported.", + } + if url.lower().endswith(".pdf"): + return _get_pdf_content(url) + + try: + from playwright.sync_api import sync_playwright + except Exception as exc: + return { + "title": "Error", + "url": url, + "content": f"Playwright unavailable: {exc}", + } + + browsers_dir = _ensure_playwright_browsers_path() + + def _launch_browser(playwright_api: Any) -> Any: + # Prefer installed Google Chrome (matches host arch) when available. + try: + return playwright_api.chromium.launch(channel="chrome", headless=True) + except Exception: + return playwright_api.chromium.launch(headless=True) + + try: + with sync_playwright() as p: + browser = _launch_browser(p) + try: + page = browser.new_page() + page.goto(url, timeout=15000) + page.wait_for_load_state("domcontentloaded") + title = page.title() or "" + text = ( + page.inner_text("body") + if page.locator("body").count() + else page.content() + ) + # Cap content size for the agent context. + text = (text or "")[:40000] + return {"title": title, "url": page.url, "content": text} + except Exception: + return _get_pdf_content(url) + finally: + browser.close() + except Exception as exc: + msg = str(exc) + if "Executable doesn't exist" in msg or "browserType.launch" in msg.lower(): + return { + "title": "Error", + "url": url, + "content": ( + "Playwright Chromium is not installed for this demo. " + f"Run: bash scripts/setup_playwright.sh " + f"(PLAYWRIGHT_BROWSERS_PATH={browsers_dir})" + ), + } + return {"title": "Error", "url": url, "content": f"Playwright error: {exc}"} + + +@tool +def check_answer_format(final_answer_dict: dict) -> Any: + """Validate the final SciPath ingredients JSON format. + + Args: + final_answer_dict: the final answer/output that is to be returned to the user + """ + try: + ok, err, validated = validate_scipath_answer(final_answer_dict) + return ok, err, validated + except ValidationError as e: + return False, str(e), None + except Exception as e: + return False, f"Unexpected error: {e}", None + + +@tool +def web_search_tool(short_search_query: str) -> list[dict]: + """Perform a web search and return structured results. + + Args: + short_search_query: A concise search query (typically 1-6 words) + """ + search_results = WEBSEARCH_TOOL(short_search_query) + return _parse_search_results(search_results) + + +@tool +def deep_web_search_tool(url: str) -> dict: + """Fetch and return webpage/PDF content for a URL. + + Args: + url: the url of a webpage + """ + return get_webpage(url) diff --git a/hf_space/streamlit_app.py b/hf_space/streamlit_app.py new file mode 100644 index 0000000000000000000000000000000000000000..00e9cfa807c04be6bf4750e0c7bc5108856679e0 --- /dev/null +++ b/hf_space/streamlit_app.py @@ -0,0 +1,3307 @@ +import json +import os +import re +import sys +import time +import html +import urllib.error +import urllib.request +import xml.etree.ElementTree as ET +from pathlib import Path +from typing import Any, Optional + +import streamlit as st +try: + from huggingface_hub import HfApi +except Exception: + HfApi = None + +SRC = Path(__file__).resolve().parent +REPO_ROOT = SRC.parent +for extra in (SRC, REPO_ROOT / "src"): + extra_str = str(extra) + if extra_str not in sys.path: + sys.path.insert(0, extra_str) + +import runner as runner_module +from runner import PipelineConfig +from common.paper_package import load_paper_package +from step_08_annotation.pipeline import TwoPassAnnotationPipeline +from streamlit_config import EXAMPLES, TAB_NAMES +import replay as replay_module +import system_run_data as system_run_module +import system_live_runner as system_live_module +from workflow_graph import STEP_COPY, WORK_PANEL_STEPS + +try: + from streamlit_config import APP_NAV, APP_VIEWS +except ImportError: + APP_VIEWS = [ + "Annotation Process", + "System Run", + ] + APP_NAV = [ + {"view": "Annotation Process", "icon": ":material/account_tree:"}, + {"view": "System Run", "icon": ":material/play_circle:"}, + ] + +DEFAULT_SOURCE_ROOT = str(REPO_ROOT / "src" / "processed_papers") +DEFAULT_OUTPUT_ROOT = str(REPO_ROOT / "hf_space" / "runs") +REPLAY_STEP_DELAY_SEC = float(os.getenv("REPLAY_STEP_DELAY_SEC", "0.35")) +# Extra hold after a step completes so the workflow pulse is readable. +REPLAY_STEP_HOLD_SEC = float(os.getenv("REPLAY_STEP_HOLD_SEC", "0.55")) +DEFAULT_APP_VIEW = APP_VIEWS[0] + +CUSTOM_CSS = """ + +""" + + +def get_secret(name: str, default: str = "") -> str: + value = os.getenv(name) + if value: + return value + try: + return st.secrets[name] + except Exception: + return default + + +def run_repo_config() -> tuple[str | None, str, str | None]: + repo_id = get_secret("RUNS_REPO_ID", "") + repo_type = get_secret("RUNS_REPO_TYPE", "dataset") + token = get_secret("HF_WRITE_TOKEN", "") or get_secret("HF_TOKEN", "") + return repo_id or None, repo_type, token or None + + +def remote_run_prefix(job_id: str) -> str: + return f"runs/{job_id}" + + +def upload_run_artifact(job_dir: Path) -> str: + repo_id, repo_type, token = run_repo_config() + if not repo_id or not token: + return "" + if HfApi is None: + return "upload_failed: huggingface_hub is not installed" + + job_id = job_dir.name + remote_prefix = remote_run_prefix(job_id) + uploaded: list[str] = [] + try: + api = HfApi(token=token) + for name in ["input_ids.json", "run_config.json", "summary.txt"]: + path = job_dir / name + if path.exists(): + api.upload_file( + path_or_fileobj=str(path), + path_in_repo=f"{remote_prefix}/{name}", + repo_id=repo_id, + repo_type=repo_type, + commit_message=f"Upload {name} for {job_id}", + ) + uploaded.append(name) + + for folder_name in ["logs", "processed_papers", "two_pass_outputs"]: + folder = job_dir / folder_name + if not folder.exists(): + continue + files = [path for path in folder.rglob("*") if path.is_file()] + if not files: + continue + api.upload_folder( + folder_path=str(folder), + path_in_repo=f"{remote_prefix}/{folder_name}", + repo_id=repo_id, + repo_type=repo_type, + commit_message=f"Upload {folder_name} for {job_id}", + ignore_patterns=["__pycache__/*", "*.pyc", "*.zip"], + ) + uploaded.append(f"{folder_name}[{len(files)} files]") + + return f"{repo_type}:{repo_id}/{remote_prefix}/ (uploaded: {', '.join(uploaded) or 'nothing'})" + except Exception as exc: + return f"upload_failed: {exc}" + + +def _load_json(path: Path) -> Optional[dict]: + if not path.exists(): + return None + try: + return json.loads(path.read_text(encoding="utf-8")) + except Exception: + return None + + +def _status_from_line(line: str, current: str) -> str: + text = (line or "").strip() + text = _display_log_line(text) + if text.startswith("Pipeline stopped:"): + return "Stopped" + if text.startswith("Step "): + return text + if "failed" in text.lower(): + return f"Failed: {text}" + if "completed successfully" in text.lower(): + return "Completed" + return current + + +def _display_log_line(line: str) -> str: + text = (line or "").strip() + if text.startswith("Step ") and " failed." in text: + return text.splitlines()[0] + if text == "[annotation] starting cluster-first two-pass annotation": + return f"Step 8/8: {STEP_COPY[8]}" + if text.startswith("[annotation] complete:"): + return "Step 8 complete" + # Rewrite engineering step labels to the shared audience-facing copy. + m = re.match(r"Step\s+(\d+)\s*/\s*\d+\s*:\s*(.*)$", text) + if m: + n = int(m.group(1)) + if n in STEP_COPY: + return f"Step {n}/8: {STEP_COPY[n]}" + if text == "Pipeline completed successfully.": + return text + return text + + +def _resolve_graph_sources(paper_input: str) -> tuple[Optional[Path], Optional[dict]]: + """Paper dir + annotation payload for the workflow graph (trace or session).""" + entry = _paper_result_for_input(paper_input) + if entry: + paper_dir_path = entry.get("paper_dir_path") + payload_path = entry.get("annotation_payload_path") + paper_dir = Path(paper_dir_path) if paper_dir_path else None + payload = _load_json(Path(payload_path)) if payload_path else None + if paper_dir and paper_dir.exists(): + return paper_dir, payload if isinstance(payload, dict) else None + + trace = replay_module.load_trace(paper_input) + if not trace: + return None, None + paper_dir_path = trace.get("paper_dir_path") + payload_path = trace.get("annotation_payload_path") + paper_dir = Path(paper_dir_path) if paper_dir_path else None + payload = _load_json(Path(payload_path)) if payload_path else None + return paper_dir, payload if isinstance(payload, dict) else None + + +def _workflow_run_id(paper_input: str) -> str: + arxiv_id = replay_module.parse_arxiv_id(paper_input or "") or "default" + return f"scipaths:{arxiv_id}" + + +def _mount_graph_shell(graph_slot) -> None: + """Mount the persistent graph canvas once (constant HTML → Streamlit reuses it).""" + if graph_slot is None: + return + from workflow_graph import render_graph_shell + + render_graph_shell(graph_slot) + + +def _push_graph(courier_slot, paper_input: str, events: list[str]) -> None: + """Push an incremental graph model into the mounted shell via postMessage. + + This never re-emits the graph iframe, so the canvas grows in place instead of + reloading on every step. + """ + if courier_slot is None: + return + paper_dir, payload = _resolve_graph_sources(paper_input) + render_events = list(events or []) + if not render_events and paper_dir is not None: + # Idle selected paper: show target only (no active pulse). + render_events = ["Step 1/8: Load the paper", "Step 1 complete"] + # Import at call-time so Streamlit always picks up workflow_graph changes. + from workflow_graph import build_synced_model, push_graph_update + + model = build_synced_model( + paper_dir=paper_dir, + payload=payload, + events=render_events, + run_id=_workflow_run_id(paper_input), + ) + push_graph_update(courier_slot, model) + + +def _format_step_event(line: str) -> str: + text = _display_log_line(line) + if not text: + return "" + return text + + +STATE_DEFAULTS = { + "paper_input": "", + "run_status": "Idle", + "run_logs": [], + "run_events": [], + "artifact_path": None, + "run_dir_path": None, + "paper_dir_path": None, + "annotation_payload_path": None, + "run_summary": None, + "annotation_skipped_reason": None, + "pipeline_stopped_reason": None, + "pipeline_failed_reason": None, + "remote_artifact_ref": "", + "replay_mode": False, + "live_run_mode": False, + # Locked demo selection (survives run clears / post-run reruns). + "selected_demo_url": "", + # System Run case-study selection (independent of Annotation Process). + "system_run_mode": "prerun", # "prerun" | "live" + "selected_system_method": system_run_module.DEFAULT_SYSTEM_METHOD, + "selected_system_case": system_run_module.DEFAULT_SYSTEM_CASE, + "selected_live_method": "codeagent_parametric", + # Per-method Live Run state (parametric vs websearch_deep are independent). + "live_run_by_method": {}, + "app_view": DEFAULT_APP_VIEW, + # Per-paper result cache so Clusters/Decomposition follow the selected paper. + "run_results_by_paper": {}, +} + +# Widget keys owned by Streamlit; assign defaults instead of deleting them. +WIDGET_STATE_KEYS = {"paper_input"} + +# Survives per-run clears; only wiped by Reset session. +PRESERVED_STATE_KEYS = { + "app_view", + "run_results_by_paper", + "live_run_mode", + "selected_demo_url", + "system_run_mode", + "selected_system_method", + "selected_system_case", + "selected_live_method", + "live_run_by_method", +} + + +def _copy_default(value: Any) -> Any: + if isinstance(value, list): + return list(value) + if isinstance(value, dict): + return dict(value) + return value + + +def _ensure_state(): + for key, value in STATE_DEFAULTS.items(): + st.session_state.setdefault(key, _copy_default(value)) + + +def _clear_run_state(*, keep_paper_input: bool = True) -> None: + """Reset transient run UI state for a new pipeline/replay. + + Never writes ``paper_input`` after the text-input widget exists. Preserves + ``run_results_by_paper`` so other papers' results stay available. + """ + for key, value in STATE_DEFAULTS.items(): + if key in WIDGET_STATE_KEYS or key in PRESERVED_STATE_KEYS: + continue + st.session_state[key] = _copy_default(value) + if not keep_paper_input: + # Defer clearing/replacing the widget value until before it is created. + st.session_state["_pending_paper_input"] = "" + + +def _store_paper_result(arxiv_id: str, result: dict) -> None: + if not arxiv_id: + return + cache = st.session_state.setdefault("run_results_by_paper", {}) + cache[str(arxiv_id)] = dict(result) + + +def _paper_result_for_input(paper_input: str) -> Optional[dict]: + arxiv_id = replay_module.parse_arxiv_id(paper_input or "") + if not arxiv_id: + return None + cache = st.session_state.get("run_results_by_paper") or {} + entry = cache.get(arxiv_id) + return entry if isinstance(entry, dict) else None + + +def _apply_pending_paper_input() -> None: + """Apply deferred paper_input changes before the text_input widget mounts.""" + if "_pending_paper_input" not in st.session_state: + return + st.session_state["paper_input"] = st.session_state.pop("_pending_paper_input") + + +def _paper_1_url() -> str: + urls = list(EXAMPLES.values()) + return urls[0] if urls else "" + + +def _set_demo_paper(url: str) -> None: + """Pin the demo paper selection (source of truth across reruns).""" + st.session_state["live_run_mode"] = False + st.session_state["selected_demo_url"] = url + st.session_state["_pending_paper_input"] = url + + +def _on_pick_demo(url: str) -> None: + _set_demo_paper(url) + + +def _on_pick_live_run() -> None: + st.session_state["live_run_mode"] = True + st.session_state["_pending_paper_input"] = "" + + +def _select_paper_1() -> None: + """Select demo Paper 1 in the picker (clears live-run mode).""" + _set_demo_paper(_paper_1_url()) + + +def _set_system_method(method_id: str) -> None: + """Pin System Run method and reset case when the method changes.""" + prev = st.session_state.get("selected_system_method") + st.session_state["selected_system_method"] = method_id + if prev != method_id: + st.session_state["selected_system_case"] = system_run_module.default_case_for_method( + method_id + ) + + +def _set_system_case(case_key: str) -> None: + st.session_state["selected_system_case"] = case_key + + +def _on_pick_system_method(method_id: str) -> None: + _set_system_method(method_id) + + +def _on_pick_system_case(case_key: str) -> None: + _set_system_case(case_key) + + +def _on_pick_system_prerun() -> None: + st.session_state["system_run_mode"] = "prerun" + + +def _on_pick_system_live() -> None: + st.session_state["system_run_mode"] = "live" + + +def _on_pick_live_method(method_id: str) -> None: + st.session_state["selected_live_method"] = method_id + + +def _empty_live_method_state() -> dict[str, Any]: + return { + "result": None, + "judged": None, + "logs": [], + "status": "Idle", + "pending_action": None, # "run" | "evaluate" | None + "ran_once": False, + "evaluated_once": False, + } + + +def _live_method_state(method_id: str) -> dict[str, Any]: + """Return the mutable per-method Live Run slot (isolated across settings).""" + store = st.session_state.get("live_run_by_method") + if not isinstance(store, dict): + store = {} + st.session_state["live_run_by_method"] = store + slot = store.get(method_id) + if not isinstance(slot, dict): + slot = _empty_live_method_state() + store[method_id] = slot + else: + slot.setdefault("result", None) + slot.setdefault("judged", None) + slot.setdefault("logs", []) + slot.setdefault("status", "Idle") + slot.setdefault("pending_action", None) + slot.setdefault("ran_once", False) + slot.setdefault("evaluated_once", False) + return slot + + +def _render_system_live_run_view() -> None: + """Live Run: one AVerImaTeC example + parametric / websearch_deep + Task A eval.""" + try: + example = system_live_module.example_claim() + except Exception as exc: + st.error(f"Could not load demo claim: {exc}") + return + + live_method = str( + st.session_state.get("selected_live_method") or "codeagent_parametric" + ).strip() + method_ids = {m["id"] for m in system_live_module.METHOD_UI} + if live_method not in method_ids: + live_method = "codeagent_parametric" + st.session_state["selected_live_method"] = live_method + + slot = _live_method_state(live_method) + status = str(slot.get("status") or "Idle") + busy = status in {"Running", "Evaluating"} + can_evaluate = bool(slot.get("result")) and not busy + run_label = "Re-run System" if slot.get("ran_once") else "Run SciFy System" + eval_label = "Re-evaluate" if slot.get("evaluated_once") else "Evaluate" + + st.markdown('
Example claim
', unsafe_allow_html=True) + paper_title = str(example.get("paper_title") or "").strip() or "Unknown paper" + st.markdown( + f""" +
+
+
Claim
+
{_escape(_ensure_claim_period(example.get("claim")))}
+
Paper: {_escape(paper_title)}
+
+
+ """, + unsafe_allow_html=True, + ) + + st.markdown('
Setting
', unsafe_allow_html=True) + method_cols = st.columns(len(system_live_module.METHOD_UI)) + for i, item in enumerate(system_live_module.METHOD_UI): + with method_cols[i]: + st.button( + item["label"], + key=f"live_method::{item['id']}", + type="primary" if item["id"] == live_method else "secondary", + use_container_width=True, + on_click=_on_pick_live_method, + args=(item["id"],), + disabled=busy, + ) + + st.markdown('
Action
', unsafe_allow_html=True) + run_col, eval_col = st.columns(2) + with run_col: + run_clicked = st.button( + run_label, + type="secondary", + use_container_width=True, + key=f"live_run_system_btn::{live_method}", + disabled=busy, + ) + with eval_col: + eval_clicked = st.button( + eval_label, + type="secondary", + use_container_width=True, + key=f"live_eval_btn::{live_method}", + disabled=busy or not can_evaluate, + ) + + with st.container(key="live_run_terminal"): + log_box = st.empty() + logs = list(slot.get("logs") or []) + + def _flush_logs() -> None: + body = "\n".join(logs[-400:]) if logs else "Waiting for run output…" + log_box.code(body, language="text") + st.caption(f"Status: {status}") + + _flush_logs() + + # Phase 1: click → mark this method busy and rerun so buttons grey out immediately. + if run_clicked and not busy: + slot["pending_action"] = "run" + slot["judged"] = None + slot["result"] = None + slot["logs"] = [] + slot["status"] = "Running" + st.rerun() + + if eval_clicked and not busy and slot.get("result"): + slot["pending_action"] = "evaluate" + slot["status"] = "Evaluating" + st.rerun() + + # Phase 2: execute pending action while buttons stay disabled. + pending = slot.get("pending_action") + if pending == "run" and status == "Running": + slot["pending_action"] = None + logs = [] + status = "Running" + + def _log(line: str) -> None: + logs.append(line) + slot["logs"] = list(logs) + log_box.code("\n".join(logs[-400:]), language="text") + + try: + result = system_live_module.run_live_method(live_method, log_callback=_log) + slot["result"] = result + slot["status"] = "Completed" + slot["ran_once"] = True + status = "Completed" + _log(f"Done. ingredients={len(result.get('ingredients') or [])}") + except Exception as exc: + slot["result"] = None + slot["status"] = "Failed" + slot["ran_once"] = True + status = "Failed" + _log(f"ERROR: {exc}") + st.rerun() + + if pending == "evaluate" and status == "Evaluating" and slot.get("result"): + slot["pending_action"] = None + logs = list(slot.get("logs") or []) + status = "Evaluating" + + def _log_eval(line: str) -> None: + if line.strip().startswith("{") and '"ok"' in line: + return + logs.append(line) + slot["logs"] = list(logs) + log_box.code("\n".join(logs[-400:]), language="text") + + try: + judged = system_live_module.evaluate_live_run( + slot["result"], + log_callback=_log_eval, + ) + slot["judged"] = judged + slot["status"] = "Evaluated" + slot["evaluated_once"] = True + status = "Evaluated" + _log_eval( + f"Evaluated. F1={float(judged.get('f1') or 0):.2f} " + f"R={float(judged.get('recall') or 0):.2f} " + f"P={float(judged.get('precision') or 0):.2f}" + ) + except Exception as exc: + slot["status"] = "Eval failed" + slot["evaluated_once"] = True + status = "Eval failed" + _log_eval(f"ERROR: {exc}") + st.rerun() + + run_result = slot.get("result") + judged = slot.get("judged") + if run_result: + case = system_live_module.live_result_to_case_card(run_result, judged) + # Claim/paper already shown above in the example block; card is results only. + result_html = _system_case_card_html(case) + st.markdown( + f'
{result_html}
', + unsafe_allow_html=True, + ) + + +def _normalize_abstract(text: str) -> str: + cleaned = re.sub(r"\s+", " ", (text or "").strip()) + cleaned = re.sub(r"^(abstract)\s*[:.]?\s*", "", cleaned, flags=re.IGNORECASE) + # Drop trailing PDF footnote markers like ").1" / ".”2". + cleaned = re.sub(r"(?<=[.!?\)\]\"'”’])\d+$", "", cleaned) + return cleaned.strip() + + +def _metadata_from_paper_dir(paper_dir: Path) -> Optional[dict[str, str]]: + data = _load_json(paper_dir / "paper_metadata.json") + record = None + if isinstance(data, list) and data: + record = data[0] if isinstance(data[0], dict) else None + elif isinstance(data, dict): + record = data + if not record: + return None + title = str(record.get("title") or "").strip() + abstract = _normalize_abstract(str(record.get("abstract") or "")) + if not title and not abstract: + return None + return {"title": title, "abstract": abstract} + + +@st.cache_data(show_spinner=False, ttl=6 * 60 * 60) +def _fetch_arxiv_title_abstract(arxiv_id: str) -> Optional[dict[str, str]]: + url = f"http://export.arxiv.org/api/query?id_list={arxiv_id}" + try: + with urllib.request.urlopen(url, timeout=12) as resp: # noqa: S310 + payload = resp.read() + except (urllib.error.URLError, TimeoutError, OSError): + return None + try: + root = ET.fromstring(payload) + except ET.ParseError: + return None + ns = {"atom": "http://www.w3.org/2005/Atom"} + entry = root.find("atom:entry", ns) + if entry is None: + return None + title = re.sub(r"\s+", " ", (entry.findtext("atom:title", default="", namespaces=ns) or "").strip()) + abstract = _normalize_abstract(entry.findtext("atom:summary", default="", namespaces=ns) or "") + if not title and not abstract: + return None + return {"title": title, "abstract": abstract} + + +def _resolve_paper_preview(paper_input: str) -> Optional[dict[str, str]]: + """Title/abstract for the selected paper (trace metadata, run dir, or arXiv).""" + arxiv_id = replay_module.parse_arxiv_id(paper_input or "") + if not arxiv_id: + return None + + entry = _paper_result_for_input(paper_input) + if entry and entry.get("paper_dir_path"): + meta = _metadata_from_paper_dir(Path(entry["paper_dir_path"])) + if meta: + return meta + + trace_dir = replay_module.find_trace_dir(paper_input) + if trace_dir is not None: + meta = _metadata_from_paper_dir(trace_dir / "processed_papers" / arxiv_id) + if meta: + return meta + + return _fetch_arxiv_title_abstract(arxiv_id) + + +def _render_paper_preview(paper_input: str) -> None: + preview = _resolve_paper_preview(paper_input) + if not preview: + st.caption("Paper metadata unavailable for this ID yet.") + return + title = preview.get("title") or "Untitled paper" + abstract = preview.get("abstract") or "" + # Title as HTML; abstract as markdown so `$...$` goes through Streamlit KaTeX. + body = ( + f"
" + f"
{_esc(title)}
" + f"
" + ) + if abstract: + body = f"{body}\n\n{abstract}" + st.markdown(body, unsafe_allow_html=True) + + +def _derive_work_step_states(events: list[str]) -> dict[int, str]: + """Map pipeline step numbers to pending | active | done | failed.""" + order = [n for n, _ in WORK_PANEL_STEPS] + started: set[int] = set() + completed: set[int] = set() + failed = False + pipeline_complete = False + + for raw in events or []: + text = str(raw) + lower = text.lower() + if text == "Pipeline completed successfully.": + pipeline_complete = True + if "failed" in lower: + failed = True + m_done = re.search(r"Step\s+(\d+)\s+complete", text, re.IGNORECASE) + if m_done: + completed.add(int(m_done.group(1))) + continue + m_start = re.search(r"Step\s+(\d+)\s*/", text) + if m_start: + started.add(int(m_start.group(1))) + continue + if "[annotation]" in lower or "annotate target contributions" in lower: + if "complete" in lower or "skipped" in lower: + completed.add(8) + else: + started.add(8) + + max_started = max(started) if started else 0 + states: dict[int, str] = {} + for n in order: + if n in completed or (n in started and n < max_started): + states[n] = "done" + elif n in started: + states[n] = "failed" if failed else "active" + else: + states[n] = "pending" + + if pipeline_complete and not failed: + for n in order: + # Replay traces often omit explicit completes for every step. + if n < 8 or 8 in completed or 8 in started: + states[n] = "done" + if 8 in completed: + states[8] = "done" + return states + + +def _display_work_status(status: str) -> str: + """Steps badge only shows Idle / Running (Stopped/Failed/Completed → Idle).""" + text = (status or "").strip() + if not text: + return "Idle" + lower = text.lower() + if lower in {"running", "starting"} or text.startswith("Step "): + return "Running" + return "Idle" + + +def _work_panel_html( + events: list[str], + *, + status: str = "Idle", + mode_note: str = "", +) -> str: + states = _derive_work_step_states(events) + status_label = f"{_display_work_status(status)}{mode_note}" + rows: list[str] = [] + for n, label in WORK_PANEL_STEPS: + state = states.get(n, "pending") + busy = " aria-busy='true'" if state == "active" else "" + rows.append( + "
  • " + "" + f"{_esc(label)}" + "
  • " + ) + body = ( + f"
      {''.join(rows)}
    " + if rows + else "

    Waiting for the first step…

    " + ) + return ( + "
    " + "
    " + "
    Steps
    " + f"
    {_esc(status_label)}
    " + "
    " + f"{body}" + "
    " + ) + + +def _render_work_panel( + placeholder, + events: list[str], + *, + status: str = "Idle", + mode_note: str = "", +) -> None: + placeholder.markdown( + _work_panel_html(events, status=status, mode_note=mode_note), + unsafe_allow_html=True, + ) + + +def _reset_all_session_state() -> None: + """Hard reset used by Reset session: wipe every session key, then re-seed defaults.""" + for key in list(st.session_state.keys()): + del st.session_state[key] + for key, value in STATE_DEFAULTS.items(): + st.session_state[key] = _copy_default(value) + + +def _metric_card(label: str, value: Any): + return ( + f"
    " + f"
    {_esc(label)}
    " + f"
    {_esc(value)}
    " + f"
    " + ) + + +def _esc(value: Any) -> str: + return html.escape("" if value is None else str(value)) + + +def _safe_int(value: Any, default: int = 0) -> int: + try: + return int(value) + except (TypeError, ValueError): + return default + + +def _grounding_html(grounding: Optional[dict], label: str, kind: str) -> str: + if not grounding: + return "" + title = ( + grounding.get("ref_title") + or grounding.get("title") + or grounding.get("paper_id") + or grounding.get("ref_id") + or "__NONE__" + ) + meta = [] + if grounding.get("paper_id"): + meta.append(f"paper_id: {grounding.get('paper_id')}") + elif grounding.get("ref_id"): + meta.append(f"ref_id: {grounding.get('ref_id')}") + if grounding.get("ref_year"): + meta.append(str(grounding.get("ref_year"))) + authors = grounding.get("ref_authors") + if isinstance(authors, list) and authors: + meta.append(", ".join(str(author) for author in authors[:3])) + meta_html = f"
    {_esc(' · '.join(meta))}
    " if meta else "" + return ( + "
    " + f"
    {_esc(label)}
    " + f"
    {_esc(title)}
    " + f"{meta_html}" + "
    " + ) + + +def _study_key(item: dict) -> str: + for key in ["paper_id", "ref_id", "ref_title", "title"]: + value = item.get(key) + if value: + return str(value).lower() + return "" + + +def _collect_grounded_studies(discoveries: list[dict], ingredients: list[dict]) -> list[dict]: + studies: list[dict] = [] + seen: set[str] = set() + for item in discoveries: + if not isinstance(item, dict): + continue + copied = dict(item) + copied["_grounding_kind"] = "grounding" + copied["_grounding_label"] = "Grounding" + key = _study_key(copied) + if key: + seen.add(key) + studies.append(copied) + + for idx, ingredient in enumerate(ingredients, start=1): + if not isinstance(ingredient, dict): + continue + canonical = ingredient.get("canonical_grounding") or {} + canonical_key = _study_key(canonical) if isinstance(canonical, dict) else "" + annotation = ingredient.get("canonical_annotation") or {} + for ref in ingredient.get("additional_groundings") or []: + if not isinstance(ref, dict): + continue + key = _study_key(ref) + if key and (key == canonical_key or key in seen): + continue + copied = dict(ref) + copied["_grounding_kind"] = "grounding" + copied["_grounding_label"] = f"Grounding for enabling contribution {idx}" + copied.setdefault("role", annotation.get("role") or ", ".join(annotation.get("roles") or [])) + copied.setdefault("contribution", annotation.get("contribution")) + copied.setdefault("rationale", annotation.get("rationale")) + if key: + seen.add(key) + studies.append(copied) + return studies + + +def _render_reference_list(discoveries: list[dict], ingredients: Optional[list[dict]] = None): + studies = _collect_grounded_studies(discoveries, ingredients or []) + if not studies: + st.markdown("
    No grounded studies listed for this target contribution.
    ", unsafe_allow_html=True) + return + for item in studies: + title = item.get("ref_title") or item.get("title") or item.get("ref_id") or item.get("paper_id") or "Untitled reference" + meta = [] + if item.get("_grounding_label"): + meta.append(str(item.get("_grounding_label"))) + if item.get("role"): + meta.append(str(item.get("role"))) + if item.get("ref_year"): + meta.append(str(item.get("ref_year"))) + class_name = "cluster-card additional-study" + body = [f"
    {_esc(title)}
    "] + if meta: + body.append(f"
    {_esc(' · '.join(meta))}
    ") + if item.get("contribution"): + body.append(f"
    Contribution. {_esc(item.get('contribution'))}
    ") + if item.get("rationale"): + body.append(f"
    Rationale. {_esc(item.get('rationale'))}
    ") + body.append("
    ") + st.markdown("".join(body), unsafe_allow_html=True) + + +def _render_claims_tab(payload: Optional[dict]): + if not payload: + st.markdown("
    No annotation payload is available yet.
    ", unsafe_allow_html=True) + return + claims = payload.get("claims") or [] + if not claims: + st.markdown("
    The run completed, but no target contributions were produced.
    ", unsafe_allow_html=True) + return + + st.markdown( + "

    This decomposition is done in hindsight: SciPaths uses observed downstream citation clusters to identify reusable target contributions, then reconstructs the enabling contributions and groundings behind each one.

    ", + unsafe_allow_html=True, + ) + + for idx, claim in enumerate(claims, start=1): + claim_id = claim.get("claim_id") or f"C{idx}" + claim_text = claim.get("rewritten_claim") or claim.get("text") or "(missing target contribution text)" + ingredients = claim.get("ingredients") or [] + discoveries = claim.get("enabling_discoveries") or [] + grounded_studies = _collect_grounded_studies(discoveries, ingredients) + meta_bits = [] + if claim.get("decision"): + meta_bits.append(str(claim.get("decision"))) + if claim.get("cluster_id"): + meta_bits.append(f"cluster {claim.get('cluster_id')}") + meta_bits.append(f"{len(ingredients)} enabling contribution{'s' if len(ingredients) != 1 else ''}") + meta_bits.append(f"{len(grounded_studies)} grounded stud{'ies' if len(grounded_studies) != 1 else 'y'}") + + st.markdown( + f""" +
    +
    +
    Target contribution {idx} · {_esc(claim_id)}
    +
    {_esc(claim_text)}
    +
    {_esc(' · '.join(meta_bits))}
    +
    +
    + """, + unsafe_allow_html=True, + ) + left, right = st.columns([1.7, 1.0], gap="large") + with left: + st.markdown("", unsafe_allow_html=True) + if not ingredients: + st.markdown("
    No enabling contributions for this target contribution.
    ", unsafe_allow_html=True) + for ingredient_idx, ingredient in enumerate(ingredients, start=1): + annotation = ingredient.get("canonical_annotation") or {} + role = annotation.get("role") or ", ".join(annotation.get("roles") or []) or "UNSPECIFIED" + canonical_grounding = ingredient.get("canonical_grounding") or {} + extras = ingredient.get("additional_groundings") or [] + grounding_parts = [] + if canonical_grounding: + grounding_parts.append( + _grounding_html(canonical_grounding, "Grounding", "grounding") + ) + for ref in extras: + if not isinstance(ref, dict): + continue + if canonical_grounding and ( + ref.get("paper_id") == canonical_grounding.get("paper_id") + or ref.get("ref_id") == canonical_grounding.get("ref_id") + ): + continue + grounding_parts.append( + _grounding_html(ref, "Grounding", "grounding") + ) + if not grounding_parts: + canonical_ref_id = ingredient.get("canonical_ref_id") or "__NONE__" + grounding_parts.append( + "
    " + "
    Grounding
    " + f"
    {_esc(canonical_ref_id)}
    " + "
    " + ) + grounding_block = ( + "
    " + f"" + + "".join(grounding_parts) + + "
    " + ) + st.markdown( + f""" +
    +
    +
    {ingredient_idx}. {_esc(ingredient.get('ingredient') or '(missing enabling contribution)')}
    +
    {_esc(role)}
    +
    +
    Contribution. {_esc(annotation.get('contribution') or '')}
    +
    Rationale. {_esc(annotation.get('rationale') or '')}
    +
    Evidence. {_esc(annotation.get('evidence_span') or '')}
    + {grounding_block} +
    + """, + unsafe_allow_html=True, + ) + with right: + st.markdown("", unsafe_allow_html=True) + _render_reference_list(discoveries, ingredients) + + +def _render_clusters_tab(discovery: Optional[dict], contributions: list[dict]): + if not discovery: + st.markdown("
    No refined cluster file is available yet.
    ", unsafe_allow_html=True) + return + st.markdown( + "

    Citation clusters look forward: they summarize how later papers use or extend the input paper, showing which contributions became reusable.

    ", + unsafe_allow_html=True, + ) + clusters = discovery.get("clusters") or [] + dropped = discovery.get("dropped_clusters") or [] + if not clusters: + st.markdown("
    No valid downstream usage clusters survived refinement and filtering.
    ", unsafe_allow_html=True) + if dropped: + with st.expander(f"Dropped clusters ({len(dropped)})", expanded=False): + st.json(dropped) + return + + for cluster in clusters: + cluster_id = cluster.get("cluster_id", "") + rep = cluster.get("representative_claim") or cluster.get("cluster_title") or "(missing representative claim)" + count = _safe_int(cluster.get("count"), len(cluster.get("claim_indices") or [])) + source_ids = cluster.get("source_cluster_ids") or [] + merge_rationale = cluster.get("merge_rationale") or "" + st.markdown( + f""" +
    +
    {_esc(rep)}
    +
    Cluster {_esc(cluster_id)} · {count} contribution instance{'s' if count != 1 else ''}
    +
    + """, + unsafe_allow_html=True, + ) + meta_cols = st.columns([1.3, 1.3, 1.4]) + with meta_cols[0]: + st.caption("Cluster ID") + st.code(str(cluster_id), language="text") + with meta_cols[1]: + st.caption("Source clusters") + st.code(", ".join(str(x) for x in source_ids) if source_ids else "singleton", language="text") + with meta_cols[2]: + st.caption("Merge rationale") + st.write(merge_rationale or "—") + + claim_indices = cluster.get("claim_indices") or [] + if claim_indices: + with st.expander(f"Linked contribution instances ({len(claim_indices)})", expanded=False): + for idx in claim_indices: + try: + j = int(idx) + except Exception: + continue + if 0 <= j < len(contributions): + item = contributions[j] or {} + title = item.get("citing_title") or item.get("citing_paper_id") or "Unknown citing paper" + claim = item.get("paper_claim") or item.get("claim") or "(missing claim)" + rationale = item.get("rationale") or "" + evidence = item.get("evidence_span") or "" + st.markdown(f"**{title}**") + st.write(claim) + if rationale: + st.caption(f"Rationale: {rationale}") + if evidence: + st.caption(f"Evidence: {evidence}") + st.divider() + + if dropped: + with st.expander(f"Dropped clusters ({len(dropped)})", expanded=False): + st.json(dropped) + + +def run_replay_stream( + paper_input: str, + work_panel=None, + metrics_slot=None, + graph_slot=None, + courier_slot=None, +) -> bool: + """Replay a saved demo trace into session state. Returns False if no trace exists.""" + trace = replay_module.load_trace(paper_input) + if not trace: + return False + + _clear_run_state(keep_paper_input=True) + st.session_state["run_status"] = "Running" + st.session_state["replay_mode"] = True + + # Graph can read the finished trace artifacts while steps reveal progressively. + st.session_state["paper_dir_path"] = trace.get("paper_dir_path") + st.session_state["annotation_payload_path"] = trace.get("annotation_payload_path") + + try: + import neo4j_workflow as _neo + + _neo.reset_run(_workflow_run_id(paper_input)) + except Exception: + pass + + panel = work_panel if work_panel is not None else st.empty() + events: list[str] = [] + _render_work_panel(panel, events, status="Running") + _push_graph(courier_slot, paper_input, events) + + for line in trace.get("run_events") or []: + display_line = _display_log_line(str(line)) + if not display_line: + continue + if display_line not in events: + events.append(display_line) + _render_work_panel(panel, events, status="Running") + _push_graph(courier_slot, paper_input, events) + delay = REPLAY_STEP_DELAY_SEC + if re.search(r"Step\s+\d+\s+complete", display_line, re.IGNORECASE): + delay += REPLAY_STEP_HOLD_SEC + time.sleep(max(0.0, delay)) + + # Commit final replay payload for the Clusters / Decomposition tabs. + st.session_state["run_status"] = trace.get("run_status") or trace.get("status") or "Completed" + st.session_state["run_logs"] = list(trace.get("run_logs") or events) + st.session_state["run_events"] = list(trace.get("run_events") or events) + st.session_state["artifact_path"] = trace.get("artifact_path") + st.session_state["run_dir_path"] = trace.get("run_dir_path") + st.session_state["paper_dir_path"] = trace.get("paper_dir_path") + st.session_state["annotation_payload_path"] = trace.get("annotation_payload_path") + st.session_state["annotation_skipped_reason"] = trace.get("annotation_skipped_reason") + st.session_state["pipeline_stopped_reason"] = trace.get("pipeline_stopped_reason") + st.session_state["pipeline_failed_reason"] = trace.get("pipeline_failed_reason") + st.session_state["remote_artifact_ref"] = "" + st.session_state["replay_mode"] = True + + arxiv_id = str(trace.get("arxiv_id") or replay_module.parse_arxiv_id(paper_input) or "") + _store_paper_result( + arxiv_id, + { + "run_status": st.session_state["run_status"], + "run_logs": list(st.session_state["run_logs"]), + "run_events": list(st.session_state["run_events"]), + "artifact_path": st.session_state.get("artifact_path"), + "run_dir_path": st.session_state.get("run_dir_path"), + "paper_dir_path": st.session_state.get("paper_dir_path"), + "annotation_payload_path": st.session_state.get("annotation_payload_path"), + "annotation_skipped_reason": st.session_state.get("annotation_skipped_reason"), + "pipeline_stopped_reason": st.session_state.get("pipeline_stopped_reason"), + "pipeline_failed_reason": st.session_state.get("pipeline_failed_reason"), + "replay_mode": True, + }, + ) + + _render_work_panel( + panel, + st.session_state["run_events"], + status=st.session_state["run_status"], + ) + _push_graph(courier_slot, paper_input, st.session_state["run_events"]) + _refresh_overview_for_paper(paper_input, metrics_slot) + return True + + +def run_two_pass_annotation( + paper_dir: Path, + annotation_output_root: Path, + llm_provider: str, + llm_model: str, + formatter_model: str, + judge_model: str, + candidate_count: int, +): + paper = load_paper_package(paper_dir) + pipeline = TwoPassAnnotationPipeline( + provider=llm_provider, + model=llm_model, + formatter_model=formatter_model or None, + judge_model=judge_model or None, + output_root=annotation_output_root, + annotator_id="streamlit_hf_space", + candidate_count=max(1, int(candidate_count)), + formatter_max_attempts=3, + include_reference_examples=True, + prompt_profile="full", + ) + result = pipeline.run(paper) + return result.result, result.run_dir + + +def run_pipeline_stream( + paper_input: str, + source_root: str, + output_root: str, + llm_provider: str, + llm_model: str, + llm_model_step4: str, + formatter_model: str, + judge_model: str, + candidate_count: int, + work_panel=None, + metrics_slot=None, + graph_slot=None, + courier_slot=None, +): + gemini_key = get_secret("GEMINI_API_KEY") + if gemini_key: + os.environ["GEMINI_API_KEY"] = gemini_key + + # Drop previous success/failure UI before this run starts rendering progress. + _clear_run_state(keep_paper_input=True) + st.session_state["run_status"] = "Starting" + try: + import neo4j_workflow as _neo + + _neo.reset_run(_workflow_run_id(paper_input)) + except Exception: + pass + + cfg = PipelineConfig( + repo_root=REPO_ROOT, + source_root=Path(source_root).expanduser().resolve(), + paper_input=paper_input.strip(), + llm_provider=llm_provider.strip() or "gemini", + llm_model=llm_model.strip() or "gemini-3.1-pro-preview", + llm_model_step4=llm_model_step4.strip() or "gemini-3-flash-preview", + model_path="Deep-Citation/Workspace/acl_scicite_wksp_trl/best_model.pt", + model_data_dir="Deep-Citation/Data", + model_class_def="Deep-Citation/Data/class_def.json", + model_lm="scibert", + device="cpu", + embedding_model="sentence-transformers/all-mpnet-base-v2", + ) + + panel = work_panel if work_panel is not None else st.empty() + status = "Starting" + logs: list[str] = [] + events: list[str] = [] + seen_events: set[str] = set() + artifact_path = None + annotation_payload_path = None + annotation_skipped_reason = None + run_summary = None + pipeline_stopped_reason = None + pipeline_failed_reason = None + + def render_activity(items: list[str], *, current_status: str | None = None): + _render_work_panel( + panel, + items, + status=current_status or status, + mode_note="", + ) + _push_graph(courier_slot, paper_input, items) + + def append_display_line(line: str): + nonlocal status + display_line = _display_log_line(line) + if not display_line: + return + logs.append(display_line) + status = _status_from_line(display_line, status) + event = _format_step_event(display_line) + if event and event not in seen_events: + seen_events.add(event) + events.append(event) + render_activity(events, current_status=status) + + render_activity(events, current_status=status) + + for line, maybe_artifact in runner_module.run_pipeline(cfg, Path(output_root).expanduser().resolve()): + if line: + if line.strip() == "Pipeline completed successfully.": + if maybe_artifact: + artifact_path = maybe_artifact + continue + display_line = _display_log_line(line) + if display_line: + logs.append(display_line) + status = _status_from_line(display_line, status) + if display_line.startswith("Pipeline stopped:"): + pipeline_stopped_reason = display_line + if "failed" in display_line.lower(): + pipeline_failed_reason = display_line + event = _format_step_event(display_line) + if event and event not in seen_events: + seen_events.add(event) + events.append(event) + if maybe_artifact: + artifact_path = maybe_artifact + try: + job_dir = Path(str(maybe_artifact)).with_suffix("") + paper_id = runner_module.parse_arxiv_id(paper_input.strip()) + st.session_state["paper_dir_path"] = str(job_dir / "processed_papers" / paper_id) + except Exception: + pass + render_activity(events, current_status=status) + + run_dir_path = None + paper_dir_path = None + remote_artifact_ref = "" + if artifact_path: + job_dir = Path(str(artifact_path)).with_suffix("") + run_dir_path = str(job_dir) + paper_id = runner_module.parse_arxiv_id(paper_input.strip()) + paper_dir = job_dir / "processed_papers" / paper_id + paper_dir_path = str(paper_dir) + if pipeline_failed_reason: + annotation_skipped_reason = f"{pipeline_failed_reason} Annotation was not run." + elif pipeline_stopped_reason: + annotation_skipped_reason = f"{pipeline_stopped_reason} Annotation was not run." + else: + discovery = _load_json(paper_dir / "usage_discovery_from_contributions.json") or {} + refined_clusters = discovery.get("clusters") or [] + if not refined_clusters: + annotation_skipped_reason = "No valid downstream usage clusters remained after refinement and filtering. Annotation was skipped." + logs.append("[annotation] skipped: no refined downstream usage clusters") + else: + append_display_line("[annotation] starting cluster-first two-pass annotation") + try: + run_output, annotation_run_dir = run_two_pass_annotation( + paper_dir=paper_dir, + annotation_output_root=job_dir / "two_pass_outputs", + llm_provider=llm_provider, + llm_model=llm_model, + formatter_model=formatter_model, + judge_model=judge_model, + candidate_count=candidate_count, + ) + payload_path = run_output.get("ui_payload_path") if isinstance(run_output, dict) else None + if payload_path and Path(payload_path).exists(): + annotation_payload_path = str(Path(payload_path)) + append_display_line(f"[annotation] complete: {annotation_run_dir}") + except Exception as exc: + pipeline_failed_reason = f"Annotation failed: {exc}" + annotation_skipped_reason = pipeline_failed_reason + logs.append(f"[annotation] failed: {exc}") + logs.append("[upload] uploading run artifact to Hugging Face dataset") + remote_artifact_ref = upload_run_artifact(job_dir) + if remote_artifact_ref: + logs.append(f"[upload] {remote_artifact_ref}") + else: + logs.append("[upload] skipped: RUNS_REPO_ID/HF_WRITE_TOKEN not configured") + if not pipeline_stopped_reason and not pipeline_failed_reason: + append_display_line("Pipeline completed successfully.") + + if pipeline_failed_reason: + status = "Failed" + elif artifact_path and pipeline_stopped_reason: + status = "Stopped" + else: + status = "Completed" if artifact_path else "Failed" + render_activity(events, current_status=status) + + st.session_state["run_status"] = status + st.session_state["run_logs"] = logs + st.session_state["run_events"] = events + st.session_state["artifact_path"] = artifact_path + st.session_state["run_dir_path"] = run_dir_path + st.session_state["paper_dir_path"] = paper_dir_path + st.session_state["annotation_payload_path"] = annotation_payload_path + st.session_state["annotation_skipped_reason"] = annotation_skipped_reason + st.session_state["pipeline_stopped_reason"] = pipeline_stopped_reason + st.session_state["pipeline_failed_reason"] = pipeline_failed_reason + st.session_state["run_summary"] = run_summary + st.session_state["remote_artifact_ref"] = remote_artifact_ref + st.session_state["replay_mode"] = False + + arxiv_id = replay_module.parse_arxiv_id(paper_input) or "" + _store_paper_result( + arxiv_id, + { + "run_status": status, + "run_logs": list(logs), + "run_events": list(events), + "artifact_path": artifact_path, + "run_dir_path": run_dir_path, + "paper_dir_path": paper_dir_path, + "annotation_payload_path": annotation_payload_path, + "annotation_skipped_reason": annotation_skipped_reason, + "pipeline_stopped_reason": pipeline_stopped_reason, + "pipeline_failed_reason": pipeline_failed_reason, + "replay_mode": False, + }, + ) + _push_graph(courier_slot, paper_input, events) + _refresh_overview_for_paper(paper_input, metrics_slot) + + +def _load_result_bundle(paper_input: Optional[str] = None): + """Load clusters/decomposition for the selected paper only (session cache).""" + selected = paper_input if paper_input is not None else st.session_state.get("paper_input", "") + entry = _paper_result_for_input(selected) + if not entry: + return None, None, [], None + + paper_dir_path = entry.get("paper_dir_path") + annotation_payload_path = entry.get("annotation_payload_path") + paper_dir = Path(paper_dir_path) if paper_dir_path else None + payload = _load_json(Path(annotation_payload_path)) if annotation_payload_path else None + discovery = ( + _load_json(paper_dir / "usage_discovery_from_contributions.json") + if paper_dir and paper_dir.exists() + else None + ) + contributions_data = ( + _load_json(paper_dir / "usage_contributions.json") + if paper_dir and paper_dir.exists() + else None + ) + contributions = (contributions_data or {}).get("contributions") or [] + return paper_dir, discovery, contributions, payload + + +def _overview_counts( + payload: Optional[dict], discovery: Optional[dict] +) -> tuple[int, int, int, int]: + claims = (payload or {}).get("claims") or [] + ingredients = sum(len(claim.get("ingredients") or []) for claim in claims) + studies = sum( + len(_collect_grounded_studies(claim.get("enabling_discoveries") or [], claim.get("ingredients") or [])) + for claim in claims + ) + clusters = len((discovery or {}).get("clusters") or []) + return clusters, len(claims), ingredients, studies + + +def _overview_html(payload: Optional[dict], discovery: Optional[dict]) -> str: + clusters, n_claims, ingredients, studies = _overview_counts(payload, discovery) + return ( + "
    " + f"{_metric_card('Refined clusters', clusters)}" + f"{_metric_card('Target contributions', n_claims)}" + f"{_metric_card('Enabling contributions', ingredients)}" + f"{_metric_card('Grounded studies', studies)}" + "
    " + ) + + +def _render_overview( + payload: Optional[dict], + discovery: Optional[dict], + placeholder=None, +) -> None: + html_body = _overview_html(payload, discovery) + if placeholder is not None: + placeholder.markdown(html_body, unsafe_allow_html=True) + else: + st.markdown(html_body, unsafe_allow_html=True) + + +def _refresh_overview_for_paper(paper_input: str, placeholder) -> None: + """Update metrics in-place after a run finishes (labels stay put; counts change).""" + if placeholder is None: + return + _paper_dir, discovery, _contributions, payload = _load_result_bundle(paper_input) + _render_overview(payload, discovery, placeholder=placeholder) + +def _build_public_export(discovery: Optional[dict], payload: Optional[dict]) -> dict: + claims = [] + for claim in (payload or {}).get("claims") or []: + if not isinstance(claim, dict): + continue + ingredients = [] + for ingredient in claim.get("ingredients") or []: + if not isinstance(ingredient, dict): + continue + ingredients.append({ + "ingredient_id": ingredient.get("ingredient_id"), + "enabling_contribution": ingredient.get("ingredient"), + "canonical_annotation": ingredient.get("canonical_annotation") or {}, + "primary_grounding": ingredient.get("canonical_grounding") or {}, + "additional_groundings": ingredient.get("additional_groundings") or [], + }) + claims.append({ + "claim_id": claim.get("claim_id"), + "target_contribution": claim.get("rewritten_claim") or claim.get("text"), + "cluster_id": claim.get("cluster_id"), + "decision": claim.get("decision"), + "enabling_contributions": ingredients, + "grounded_studies": _collect_grounded_studies(claim.get("enabling_discoveries") or [], claim.get("ingredients") or []), + }) + + return { + "citation_clusters": (discovery or {}).get("clusters") or [], + "target_contribution_decompositions": claims, + } + + +def _escape(text: Any) -> str: + return html.escape(str(text or ""), quote=True) + + +def _format_analysis_bullet(text: Any) -> str: + """Bold leading taxonomy tag; render as 'Tag:' instead of '[Tag]'.""" + raw = str(text or "") + match = re.match(r"^\[([^\]]+)\]\s*(.*)$", raw, flags=re.DOTALL) + if not match: + return _escape(raw) + tag = match.group(1).strip() + rest = match.group(2) + label = f"{tag}:" + if rest: + return f"{_escape(label)} {_escape(rest)}" + return f"{_escape(label)}" + + +def _render_gold_item( + item: dict, *, paired: bool = False, show_match_notes: bool = True, show_judge: bool = True +) -> str: + badge = str(item.get("badge") or "miss") + role = _escape(item.get("role") or "") + desc = _escape(item.get("description") or "") + notes = [] + if show_match_notes: + for note in item.get("match_notes") or []: + cls = "system-match-note partial" if badge == "partial" else "system-match-note" + notes.append(f"
    {_escape(note)}
    ") + if show_judge: + judge = item.get("judge_note") or "" + if judge: + notes.append(f"
    {_escape(judge)}
    ") + paired_cls = " is-paired" if paired else "" + return ( + f"
    " + "
    " + f"{badge}" + f"{role}" + "
    " + f"
    {desc}
    " + f"{''.join(notes)}" + "
    " + ) + + +def _render_pred_item(item: dict, *, paired: bool = False, show_rationale: bool = True) -> str: + idx = item.get("idx") or "" + role = _escape(item.get("role") or "") + desc = _escape(item.get("description") or "") + rationale = (item.get("rationale") or "") if show_rationale else "" + rat_html = f"
    {_escape(rationale)}
    " if rationale else "" + paired_cls = " is-paired" if paired else "" + return ( + f"
    " + "
    " + f"{_escape(idx)}" + f"{role}" + "
    " + f"
    {desc}
    " + f"{rat_html}" + "
    " + ) + + +def _render_gold_column(items: list[dict]) -> str: + rows = [_render_gold_item(item) for item in items] + body = "".join(rows) if rows else "
    None
    " + return f"

    Gold ingredients

    {body}
    " + + +def _render_pred_column(items: list[dict]) -> str: + rows = [_render_pred_item(item) for item in items] + body = "".join(rows) if rows else "
    None
    " + return f"

    SciFy Predictions

    {body}
    " + + +def _render_paired_ingredient_grid(case: dict) -> str: + """Success layout: gold left, predictions right, lines between matched pairs.""" + gold = case.get("gold_ingredients") or [] + predicted = case.get("predicted_ingredients") or [] + rows = system_run_module.build_pair_rows(gold, predicted) + body = [ + "
    ", + "

    Gold ingredients

    SciFy Predictions

    ", + ] + for row in rows: + linked = bool(row.get("linked")) + gold_item = row.get("gold") + pred_item = row.get("pred") + left = ( + # Pair lines already show the match; keep judge/rationale text. + _render_gold_item( + gold_item, paired=linked, show_match_notes=False, show_judge=True + ) + if gold_item + else "
    " + ) + right = ( + _render_pred_item(pred_item, paired=linked, show_rationale=True) + if pred_item + else "
    " + ) + link_cls = "system-pair-link" if linked else "system-pair-link is-empty" + body.append( + f"
    {left}{right}
    " + ) + body.append("
    ") + return "".join(body) + + +def _render_tool_panel_html(panel: dict) -> str: + kind = panel.get("kind") + if kind == "retrieval": + queries = "".join( + f"
  • {_escape(q)}
  • " for q in (panel.get("queries") or []) + ) + docs = "".join( + f"
  • {_escape(d)}
  • " for d in (panel.get("retrieved_docs") or []) + ) + return ( + "
    " + "

    Tool evidence used

    " + f"

    {_escape(panel.get('intro') or '')}

    " + "
    Example web search queries
    " + f"
      {queries}
    " + "
    Retrieved documents
    " + "

    Documents returned for nested NER preprocessing / split construction:

    " + f"
      {docs}
    " + "
    " + ) + if kind == "deep": + queries = "".join( + f"
  • {_escape(q)}
  • " for q in (panel.get("queries") or []) + ) + hits = "".join(f"
  • {_escape(h)}
  • " for h in (panel.get("hits") or [])) + deep_items = [] + for item in panel.get("deep_fetches") or []: + cls = "ok" if item.get("ok") else "bad" + deep_items.append( + f"
  • {_escape(item.get('url'))}" + f"
    {_escape(item.get('outcome'))}
  • " + ) + return ( + "
    " + "

    Tool trace · web_search_tool

    " + "
    Queries
    " + f"
      {queries}
    " + "
    Notable hits
    " + f"
      {hits}
    " + "

    Tool trace · deep_web_search_tool

    " + f"
      {''.join(deep_items)}
    " + "
    " + ) + return "" + + +def _ensure_claim_period(text: Any) -> str: + claim = str(text or "").strip() + if claim and claim[-1] not in ".!?": + claim += "." + return claim + + +def _system_claim_header_html(case: dict) -> str: + """Claim + paper outside the result card (Decomposition-style claim layout).""" + paper_title = str(case.get("paper_title") or "").strip() + paper_line = ( + f'
    Paper: {_escape(paper_title)}
    ' + if paper_title + else "" + ) + return ( + '
    ' + '
    ' + '
    Claim
    ' + f'
    {_escape(_ensure_claim_period(case.get("claim")))}
    ' + f"{paper_line}" + "
    " + "
    " + ) + + +def _system_case_card_html(case: dict) -> str: + tone = str(case.get("outcome_tone") or "part") + f1 = float(case.get("f1") or 0.0) + recall = float(case.get("recall") or 0.0) + precision = float(case.get("precision") or 0.0) + case_key = str(case.get("case_key") or "") + if case_key == "featured": + pills = ( + "
    " + "Unevaluated" + "
    " + ) + else: + pills = ( + "
    " + f"F1 {f1:.2f}" + f"Recall {recall:.2f}" + f"Precision {precision:.2f}" + "
    " + ) + # Success: gold | predictions with pair connector lines. + # Failure: keep the original two-column layout (pairing UI deferred). + # Featured / unevaluated live runs: predictions only. + if case_key == "success": + ingredients_html = _render_paired_ingredient_grid(case) + elif case_key == "featured": + pred_col = _render_pred_column(case.get("predicted_ingredients") or []) + ingredients_html = f'
    {pred_col}
    ' + else: + gold_col = _render_gold_column(case.get("gold_ingredients") or []) + pred_col = _render_pred_column(case.get("predicted_ingredients") or []) + ingredients_html = ( + f'
    {gold_col}{pred_col}
    ' + ) + tool_html = _render_tool_panel_html(case.get("tool_panel") or {}) + return ( + f'
    ' + f'
    {_escape(case.get("case_chip"))}
    ' + f"{pills}" + f"{ingredients_html}" + f"{tool_html}" + "
    " + ) + + +def _system_analysis_panel_html( + method_id: str, case_key: str, *, tone: str = "bad" +) -> str: + view = system_run_module.analysis_view_for_case(method_id, case_key) + blocks = view.get("blocks") or [] + if not blocks: + return ( + "
    " + "

    No analysis available for this case.

    " + "
    " + ) + html_parts = [ + "
    ", + ] + for block in blocks: + btype = block.get("type") + if btype == "diagnosis": + text = str(block.get("text") or "") + diag_cls = "system-diagnosis is-part" if tone == "part" else "system-diagnosis" + if block.get("strong_first") and text.startswith("Diagnosis."): + rest = text[len("Diagnosis.") :].lstrip() + html_parts.append( + f"
    Diagnosis. {_escape(rest)}
    " + ) + elif block.get("strong_first"): + lines = text.split("\n", 1) + if len(lines) == 2: + html_parts.append( + f"
    {_escape(lines[0])}" + f"
    {_escape(lines[1])}
    " + ) + else: + html_parts.append( + f"
    {_escape(text)}
    " + ) + else: + html_parts.append(f"
    {_escape(text)}
    ") + elif btype == "two_col": + cols = [] + for col in block.get("cols") or []: + items = "".join( + f"
  • {_format_analysis_bullet(x)}
  • " for x in (col.get("items") or []) + ) + cols.append( + "
    " + f"

    {_escape(col.get('title'))}

    " + f"
      {items}
    " + "
    " + ) + html_parts.append(f"
    {''.join(cols)}
    ") + elif btype == "list": + items = "".join( + f"
  • {_format_analysis_bullet(x)}
  • " for x in (block.get("items") or []) + ) + html_parts.append(f"

    {_escape(block.get('title'))}

      {items}
    ") + html_parts.append("
    ") + return "".join(html_parts) + + +def _render_system_run_view(): + method_id = str(st.session_state.get("selected_system_method") or "").strip() + if method_id not in system_run_module.method_ids(): + method_id = system_run_module.DEFAULT_SYSTEM_METHOD + st.session_state["selected_system_method"] = method_id + + available_cases = system_run_module.list_cases(method_id) + case_keys = {key for key, _label in available_cases} + case_key = str(st.session_state.get("selected_system_case") or "").strip() + if case_key not in case_keys: + case_key = system_run_module.default_case_for_method(method_id) + st.session_state["selected_system_case"] = case_key + + mode = str(st.session_state.get("system_run_mode") or "prerun").strip() + if mode not in {"prerun", "live"}: + mode = "prerun" + st.session_state["system_run_mode"] = mode + + st.markdown( + """ +
    +
    SciFy System Run on SciPaths Claims
    +
    SciFy CodeAgent
    +
    + """, + unsafe_allow_html=True, + ) + + st.markdown('
    Mode
    ', unsafe_allow_html=True) + mode_cols = st.columns(2) + with mode_cols[0]: + st.button( + "Pre-run Cases", + key="system_mode::prerun", + type="primary" if mode == "prerun" else "secondary", + use_container_width=True, + on_click=_on_pick_system_prerun, + ) + with mode_cols[1]: + st.button( + "Live Run", + key="system_mode::live", + type="primary" if mode == "live" else "secondary", + use_container_width=True, + on_click=_on_pick_system_live, + ) + + if mode == "live": + _render_system_live_run_view() + return + + st.markdown('
    Setting
    ', unsafe_allow_html=True) + method_items = system_run_module.SYSTEM_RUN_METHODS + method_cols = st.columns(len(method_items)) + for i, item in enumerate(method_items): + with method_cols[i]: + mid = item["id"] + st.button( + item["label"], + key=f"system_method::{mid}", + type="primary" if mid == method_id else "secondary", + use_container_width=True, + on_click=_on_pick_system_method, + args=(mid,), + ) + + st.markdown('
    Case
    ', unsafe_allow_html=True) + case_cols = st.columns(max(len(available_cases), 1)) + for i, (ckey, clabel) in enumerate(available_cases): + with case_cols[i]: + st.button( + clabel, + key=f"system_case::{method_id}::{ckey}", + type="primary" if ckey == case_key else "secondary", + use_container_width=True, + on_click=_on_pick_system_case, + args=(ckey,), + ) + + try: + case = system_run_module.get_case(method_id, case_key) + except Exception as exc: + st.error(f"Could not load case study: {exc}") + return + + claim_html = _system_claim_header_html(case) + result_html = _system_case_card_html(case) + analysis_html = _system_analysis_panel_html( + method_id, case_key, tone=str(case.get("outcome_tone") or "bad") + ) + st.markdown(claim_html, unsafe_allow_html=True) + result_col, analysis_col = st.columns([1.25, 1], gap="large") + with result_col: + st.markdown( + f'
    {result_html}
    ', + unsafe_allow_html=True, + ) + with analysis_col: + st.markdown(analysis_html, unsafe_allow_html=True) + + +def _render_annotation_process_view( + *, + source_root: str, + output_root: str, + llm_provider: str, + llm_model: str, + llm_model_step4: str, + formatter_model: str, + judge_model: str, + candidate_count: int, +): + if not get_secret("GEMINI_API_KEY"): + st.warning("GEMINI_API_KEY is missing for live runs. Demo examples still work without it.") + + # Pin demo selection so post-run reruns cannot jump to another paper. + if not st.session_state.get("live_run_mode"): + demo = str(st.session_state.get("selected_demo_url") or "").strip() + if not demo: + _select_paper_1() + else: + # Re-assert the locked demo before the picker / text input mounts. + st.session_state["_pending_paper_input"] = demo + + _apply_pending_paper_input() + current_paper = st.session_state.get("paper_input", "") + selected_demo = str(st.session_state.get("selected_demo_url") or "").strip() + selected_demo_id = replay_module.parse_arxiv_id(selected_demo) + + st.markdown( + """ +
    +
    Annotation Process
    +
    Forecasting pathways to scientific discovery
    +
    + """, + unsafe_allow_html=True, + ) + + live_run_mode = bool(st.session_state.get("live_run_mode")) + st.markdown('
    Paper
    ', unsafe_allow_html=True) + example_items = list(EXAMPLES.items()) + picker_count = len(example_items) + 1 # demos + Live run + cols = st.columns(picker_count) + for i, (button_label, value) in enumerate(example_items): + with cols[i]: + example_id = replay_module.parse_arxiv_id(value) + is_selected = (not live_run_mode) and bool( + selected_demo_id and example_id == selected_demo_id + ) + st.button( + button_label, + key=f"example::{button_label}", + type="primary" if is_selected else "secondary", + use_container_width=True, + on_click=_on_pick_demo, + args=(value,), + ) + with cols[-1]: + st.button( + "Live Run", + key="example::live_run", + icon=":material/edit_note:", + type="primary" if live_run_mode else "secondary", + use_container_width=True, + on_click=_on_pick_live_run, + ) + + tabs = st.tabs(TAB_NAMES) + + with tabs[0]: + if live_run_mode: + st.caption("Enter any arXiv paper and run the live pipeline.") + paper_input = st.text_input( + "arXiv URL or ID", + key="paper_input", + placeholder="https://arxiv.org/abs/2311.14919", + ) + else: + paper_input = selected_demo or current_paper + if not paper_input.strip(): + st.caption("Select a demo paper above, or choose Live Run to enter a custom arXiv ID.") + + paper_result = _paper_result_for_input(paper_input) if paper_input.strip() else None + status = (paper_result or {}).get("run_status") or "Idle" + events = list((paper_result or {}).get("run_events") or []) + work_panel = None + graph_slot = None + courier_slot = None + run_clicked = False + + if paper_input.strip(): + left_col, right_col = st.columns([1.55, 1], gap="large") + with left_col: + _render_paper_preview(paper_input) + with right_col: + work_panel = st.empty() + _render_work_panel(work_panel, events, status=status) + run_clicked = st.button( + "Run pipeline", + type="primary", + use_container_width=True, + key="run_pipeline_btn", + ) + + # Persistent graph canvas (mounted once) + invisible courier that + # streams incremental model updates into it without re-emitting it. + graph_slot = st.empty() + courier_slot = st.empty() + _mount_graph_shell(graph_slot) + _push_graph(courier_slot, paper_input, events) + + if paper_result and paper_result.get("pipeline_failed_reason"): + st.error(paper_result["pipeline_failed_reason"]) + if paper_result and paper_result.get("annotation_skipped_reason"): + st.warning(paper_result["annotation_skipped_reason"]) + if ( + paper_result + and paper_result.get("pipeline_stopped_reason") + and not paper_result.get("pipeline_failed_reason") + ): + st.warning(paper_result["pipeline_stopped_reason"]) + + # Stable slot so only metric values change when a run finishes (not a full remount flash). + metrics_slot = st.empty() + paper_dir, discovery, contributions, payload = _load_result_bundle(paper_input) + _render_overview(payload, discovery, placeholder=metrics_slot) + + if run_clicked: + if not paper_input.strip(): + _clear_run_state() + st.session_state["pipeline_failed_reason"] = "Paper input is required." + st.session_state["run_status"] = "Failed" + st.rerun() + + # Demo papers use saved traces; Live run always executes the live pipeline. + used_trace = ( + False + if live_run_mode + else run_replay_stream( + paper_input, + work_panel=work_panel, + metrics_slot=metrics_slot, + graph_slot=graph_slot, + courier_slot=courier_slot, + ) + ) + if not used_trace: + _clear_run_state() + st.session_state["run_status"] = "Starting" + st.session_state["replay_mode"] = False + run_pipeline_stream( + paper_input=paper_input, + source_root=source_root, + output_root=output_root, + llm_provider=llm_provider, + llm_model=llm_model, + llm_model_step4=llm_model_step4, + formatter_model=formatter_model, + judge_model=judge_model, + candidate_count=candidate_count, + work_panel=work_panel, + metrics_slot=metrics_slot, + graph_slot=graph_slot, + courier_slot=courier_slot, + ) + # Keep the same demo selected after finish; do not advance to Paper 2. + if not live_run_mode and paper_input.strip(): + _set_demo_paper(paper_input.strip()) + st.rerun() + + with tabs[1]: + paper_dir, discovery, contributions, payload = _load_result_bundle( + st.session_state.get("paper_input", "") + ) + _render_clusters_tab(discovery, contributions) + + with tabs[2]: + paper_dir, discovery, contributions, payload = _load_result_bundle( + st.session_state.get("paper_input", "") + ) + _render_claims_tab(payload) + + +def main(): + llm_provider = os.getenv("LLM_PROVIDER", "gemini") + llm_model = os.getenv("LLM_MODEL", "gemini-3.1-pro-preview") + llm_model_step4 = os.getenv("LLM_MODEL_STEP4", "gemini-3-flash-preview") + formatter_model = os.getenv("ANNOTATION_FORMATTER_MODEL", "gemini/gemini-3.1-pro-preview") + judge_model = os.getenv("ANNOTATION_JUDGE_MODEL", "gemini/gemini-3.1-pro-preview") + candidate_count = int(os.getenv("ANNOTATION_CANDIDATE_COUNT", "3")) + source_root = DEFAULT_SOURCE_ROOT + output_root = DEFAULT_OUTPUT_ROOT + + st.set_page_config( + page_title="SciPaths", + page_icon="🔬", + layout="wide", + initial_sidebar_state="collapsed", + ) + st.markdown(CUSTOM_CSS, unsafe_allow_html=True) + _ensure_state() + + if st.button("Reset session", type="tertiary", key="reset_session_btn"): + _reset_all_session_state() + st.rerun() + + nav_col, main_col = st.columns([1, 5.5], gap="medium") + + with nav_col: + current = st.session_state.get("app_view", DEFAULT_APP_VIEW) + if current not in APP_VIEWS: + current = DEFAULT_APP_VIEW + st.session_state["app_view"] = current + + st.markdown( + """ + + """, + unsafe_allow_html=True, + ) + for item in APP_NAV: + view = item["view"] + icon = item.get("icon") or "" + is_active = view == current + if st.button( + view, + key=f"nav::{view}", + icon=icon, + type="primary" if is_active else "secondary", + use_container_width=True, + ): + if view != current: + st.session_state["app_view"] = view + if view == "Annotation Process": + _select_paper_1() + st.rerun() + + selected = st.session_state.get("app_view", DEFAULT_APP_VIEW) + + with main_col: + if selected == "System Run": + _render_system_run_view() + else: + _render_annotation_process_view( + source_root=source_root, + output_root=output_root, + llm_provider=llm_provider, + llm_model=llm_model, + llm_model_step4=llm_model_step4, + formatter_model=formatter_model, + judge_model=judge_model, + candidate_count=candidate_count, + ) + +if __name__ == "__main__": + main() diff --git a/hf_space/streamlit_config.py b/hf_space/streamlit_config.py new file mode 100644 index 0000000000000000000000000000000000000000..66181c31c1be6f2ae943653927c23282c54031f1 --- /dev/null +++ b/hf_space/streamlit_config.py @@ -0,0 +1,50 @@ +from workflow_graph import STEP_COPY + +EXAMPLES = { + "Paper 1": "https://arxiv.org/abs/2208.00329", + # AveriTeC demo hidden from the picker for now. + # "Paper 2 (AveriTeC)": "https://arxiv.org/abs/2505.17978", + "Paper 2": "https://arxiv.org/abs/2211.08788", +} + +# System Run setting picker labels (match SciPaths Dev case-study report). +SYSTEM_RUN_METHODS = { + "Parametric - LLM Only": "codeagent_parametric", + "Websearch Deep - LLM + Web Search + Crawl": "codeagent_websearch_deep", +} + +TAB_NAMES = [ + "Run", + "Clusters", + "Decomposition", +] + +APP_VIEWS = [ + "Annotation Process", + "System Run", +] + +# Left-nav items (mina-style icon + label rows). Icons use Streamlit Material symbols. +APP_NAV = [ + { + "view": "Annotation Process", + "icon": ":material/account_tree:", + }, + { + "view": "System Run", + "icon": ":material/play_circle:", + }, +] + +METHOD_NOTES = { + "Pipeline scope": "Runs steps 0, 1, 2, 3, 4, 5, 6, and 8, then launches cluster-first two-pass annotation.", + "Input": "Accepts a single arXiv URL or arXiv ID.", + "Cluster-first annotation": "Uses all refined downstream USES/EXTENDS clusters to derive target contributions, then decomposes each target contribution separately.", + "Stopping rule": "If no valid downstream usage clusters remain after refinement and filtering, annotation is skipped.", +} + +DISPLAY_STEPS = [1, 2, 3, 4, 5, 6, 7, 8] + + +def pipeline_steps_markdown() -> str: + return "\n".join(f"{idx}. {STEP_COPY[idx]}" for idx in DISPLAY_STEPS) diff --git a/hf_space/system_live_runner.py b/hf_space/system_live_runner.py new file mode 100644 index 0000000000000000000000000000000000000000..bad1d8b0df05f610bef9851ddef03aa40ffb878c --- /dev/null +++ b/hf_space/system_live_runner.py @@ -0,0 +1,317 @@ +"""Streamlit helpers for System Run → Live Run (self-contained scipath_live).""" + +from __future__ import annotations + +import json +import os +import subprocess +import sys +from pathlib import Path +from typing import Any, Callable, Optional + +from scipath_live.pipeline import ( + DEFAULT_RUNS_ROOT, + load_example_claim, + new_job_dir, +) + +HF_SPACE = Path(__file__).resolve().parent +DEMO_ROOT = HF_SPACE.parent +CLI = HF_SPACE / "scipath_live" / "cli.py" +# Stable browser cache for this demo (avoids Cursor sandbox temp paths). +PLAYWRIGHT_BROWSERS_DIR = DEMO_ROOT / ".playwright" + +METHOD_UI = [ + {"id": "codeagent_parametric", "cli": "parametric", "label": "Parametric - LLM Only"}, + { + "id": "codeagent_websearch_deep", + "cli": "websearch_deep", + "label": "Websearch Deep - LLM + Web Search + Crawl", + }, +] + + +def example_claim() -> dict[str, Any]: + return load_example_claim() + + +def _env_for_child() -> dict[str, str]: + env = dict(os.environ) + # Prefer demo-local .env if present. + for candidate in (HF_SPACE / ".env", HF_SPACE.parent / ".env"): + if not candidate.exists(): + continue + try: + from dotenv import dotenv_values + + for key, value in dotenv_values(candidate).items(): + if value and key not in env: + env[key] = value + except Exception: + pass + google = ( + env.get("GOOGLE_GENAI_API_KEY") + or env.get("GEMINI_API_KEY") + or env.get("GOOGLE_API_KEY") + or "" + ).strip() + if google: + env["GEMINI_API_KEY"] = google + env["GOOGLE_API_KEY"] = google + env["GOOGLE_GENAI_API_KEY"] = google + env["PYTHONPATH"] = ( + f"{HF_SPACE}{os.pathsep}{env.get('PYTHONPATH', '')}".rstrip(os.pathsep) + ) + # Stream agent prints line-by-line into the Live Run terminal. + env["PYTHONUNBUFFERED"] = "1" + env["PYTHONIOENCODING"] = "utf-8" + # Always use demo-local Playwright browsers (override Cursor sandbox cache). + PLAYWRIGHT_BROWSERS_DIR.mkdir(parents=True, exist_ok=True) + env["PLAYWRIGHT_BROWSERS_PATH"] = str(PLAYWRIGHT_BROWSERS_DIR) + return env + + +def _is_result_payload_line(line: str) -> bool: + text = (line or "").strip() + return text.startswith("{") and '"ok"' in text + + +def _append_trace_from_logs_jsonl( + job_dir: Path, log_callback: Optional[Callable[[str], None]] +) -> None: + """Fallback: if stdout missed the trace, reconstruct from logs.jsonl.""" + if not log_callback: + return + path = job_dir / "logs.jsonl" + if not path.exists(): + return + try: + from scipath_live.agent import format_agent_trace + + payload = json.loads(path.read_text(encoding="utf-8")) + steps = payload.get("steps") or [] + for line in format_agent_trace(steps): + log_callback(line) + except Exception as exc: + log_callback(f"[warn] Could not load agent trace: {exc}") + + +def run_live_method( + method_id: str, + *, + log_callback: Optional[Callable[[str], None]] = None, + model_name: str = "gemini-3-flash-preview", +) -> dict[str, Any]: + """Subprocess-run a method; stream stdout lines via log_callback.""" + cli_method = next((m["cli"] for m in METHOD_UI if m["id"] == method_id), None) + if not cli_method: + raise ValueError(f"Unknown method: {method_id}") + + job_dir = new_job_dir(DEFAULT_RUNS_ROOT, method=cli_method) + cmd = [ + sys.executable, + str(CLI), + "run", + "--method", + cli_method, + "--output-dir", + str(job_dir), + "--model", + model_name, + ] + if log_callback: + log_callback(f"$ {' '.join(cmd)}") + + proc = subprocess.Popen( + cmd, + cwd=str(HF_SPACE), + env=_env_for_child(), + stdout=subprocess.PIPE, + stderr=subprocess.STDOUT, + text=True, + bufsize=1, + ) + assert proc.stdout is not None + last_json = "" + saw_trace = False + for line in proc.stdout: + line = line.rstrip("\n") + if not line: + continue + if _is_result_payload_line(line): + # Keep machine payload out of the human terminal. + last_json = line + continue + if "SciFy CodeAgent trace" in line: + saw_trace = True + if log_callback: + log_callback(line) + code = proc.wait() + if code != 0: + raise RuntimeError(f"Live run failed with exit code {code}") + + if last_json: + payload = json.loads(last_json) + if not payload.get("ok"): + raise RuntimeError(payload.get("error") or "Live run failed") + result = payload["result"] + else: + # Fallback: read result.json written by the agent. + result_path = job_dir / "result.json" + if not result_path.exists(): + raise RuntimeError("Live run produced no result.json") + result = json.loads(result_path.read_text(encoding="utf-8")) + result["output_dir"] = str(job_dir) + result["method_id"] = method_id + + if not saw_trace: + _append_trace_from_logs_jsonl(job_dir, log_callback) + + result.setdefault("output_dir", str(job_dir)) + result.setdefault("method_id", method_id) + return result + + +def evaluate_live_run( + run_result: dict[str, Any], + *, + log_callback: Optional[Callable[[str], None]] = None, + judge_model: str = "gemini/gemini-3.1-pro-preview", +) -> dict[str, Any]: + result_json = Path(run_result.get("output_dir") or ".") / "result.json" + if not result_json.exists(): + # Write a minimal result file for the CLI. + result_json.parent.mkdir(parents=True, exist_ok=True) + result_json.write_text( + json.dumps(run_result, indent=2, ensure_ascii=False) + "\n", encoding="utf-8" + ) + + cmd = [ + sys.executable, + str(CLI), + "evaluate", + "--result-json", + str(result_json), + "--judge-model", + judge_model, + ] + if log_callback: + log_callback(f"$ {' '.join(cmd)}") + + proc = subprocess.run( + cmd, + cwd=str(HF_SPACE), + env=_env_for_child(), + capture_output=True, + text=True, + ) + if log_callback and proc.stdout: + for line in proc.stdout.splitlines(): + log_callback(line) + if proc.returncode != 0: + err = (proc.stderr or proc.stdout or "").strip() + raise RuntimeError(err or f"Evaluate failed with exit code {proc.returncode}") + + # Last JSON line is the payload. + payload = None + for line in reversed((proc.stdout or "").splitlines()): + if line.startswith("{"): + payload = json.loads(line) + break + if not payload or not payload.get("ok"): + raise RuntimeError((payload or {}).get("error") or "Evaluate produced no result") + return payload["judged"] + + +def live_result_to_case_card(run_result: dict[str, Any], judged: Optional[dict] = None) -> dict[str, Any]: + """Normalize live artifacts into the System Run case-card shape.""" + method_id = run_result.get("method_id") or "codeagent_parametric" + method_label = next( + (m["label"] for m in METHOD_UI if m["id"] == method_id), method_id + ) + predicted = [] + source_pred = (judged or {}).get("predicted_ingredients") or run_result.get("ingredients") or [] + for i, item in enumerate(source_pred, start=1): + if not isinstance(item, dict): + continue + predicted.append( + { + "idx": i, + "description": str(item.get("description") or ""), + "role": str(item.get("role") or ""), + "rationale": str(item.get("rationale") or ""), + } + ) + + if judged: + gold = [] + for j in judged.get("recall_judgments") or []: + gold.append( + { + "description": str(j.get("reference_ingredient") or ""), + "role": str(j.get("reference_role") or ""), + "badge": ( + "full" + if j.get("covered") + else ("partial" if j.get("partial_matches") else "miss") + ), + "match_notes": [], + "judge_note": str(j.get("reasoning") or ""), + "pair_pred_idx": j.get("best_match_idx"), + } + ) + # Reconstruct match notes for partials + for pm in j.get("partial_matches") or []: + gold[-1]["match_notes"].append( + f"Partial ↔ pred #{pm.get('predicted_idx')}: {pm.get('reasoning', '')}" + ) + if gold[-1]["badge"] == "miss": + gold[-1]["badge"] = "partial" + # Live Run is not framed as success/failure — always yellow (part). + f1 = float(judged.get("f1") or 0.0) + return { + "method_id": method_id, + "method_label": method_label.split(" - ")[0], + "case_key": "live", + "case_chip": f"{method_label.split(' - ')[0]} · Live", + "outcome_tone": "part", + "tools": ( + ["check_answer_format"] + if "parametric" in method_id + else ["web_search_tool", "deep_web_search_tool", "check_answer_format"] + ), + "paper_id": judged.get("paper_id") or run_result.get("paper_id"), + "claim_idx": judged.get("claim_idx") or run_result.get("claim_idx"), + "paper_title": judged.get("paper_title") or run_result.get("paper_title"), + "claim": judged.get("claim") or run_result.get("claim"), + "recall": float(judged.get("recall") or 0.0), + "precision": float(judged.get("precision") or 0.0), + "f1": f1, + "gold_ingredients": gold, + "predicted_ingredients": predicted, + "tool_panel": {"kind": "none"}, + } + + # Pre-evaluate: predictions only. + return { + "method_id": method_id, + "method_label": method_label.split(" - ")[0], + "case_key": "featured", + "case_chip": f"{method_label.split(' - ')[0]} · Live (unevaluated)", + "outcome_tone": "part", + "tools": ( + ["check_answer_format"] + if "parametric" in method_id + else ["web_search_tool", "deep_web_search_tool", "check_answer_format"] + ), + "paper_id": run_result.get("paper_id"), + "claim_idx": run_result.get("claim_idx"), + "paper_title": run_result.get("paper_title"), + "claim": run_result.get("claim"), + "recall": 0.0, + "precision": 0.0, + "f1": 0.0, + "gold_ingredients": [], + "predicted_ingredients": predicted, + "tool_panel": {"kind": "none"}, + } diff --git a/hf_space/system_run_cases/case_study_analysis.json b/hf_space/system_run_cases/case_study_analysis.json new file mode 100644 index 0000000000000000000000000000000000000000..994cc02444150b3aca27123caa228bebebe25a55 --- /dev/null +++ b/hf_space/system_run_cases/case_study_analysis.json @@ -0,0 +1,829 @@ +{ + "title": "SciPaths Dev Case Studies across CodeAgent settings", + "analysis_model": "gemini/gemini-3.1-pro-preview", + "settings": { + "codeagent_parametric": { + "tools": [ + "check_answer_format" + ], + "summary_meta": { + "source": "gold", + "gold_file": "/Users/dingzifeng/SciFy-UMBC/dryrun/ForecastingDiscoveryPathways-7FCC/archive/dev.json", + "gold_claim_field": "rewritten_capability", + "setting": 1, + "generator_model": "gemini-3-flash-preview", + "generator_backend": "codeagent", + "codeagent_method": "parametric", + "n_claims": 50, + "n_total_claims": 50, + "n_errors": 0, + "mean_n_predicted": 4.56 + }, + "success": { + "paper_id": "neurips-2023-poster-2075", + "claim_idx": 0, + "paper_title": "Pick-a-Pic: An Open Dataset of User Preferences for Text-to-Image Generation", + "claim": "Tool: A human-preference prompt-conditioned image scorer, enabling automatic ranking of text-to-image model outputs", + "recall": 1.0, + "precision": 0.75, + "f1": 0.857, + "predicted_ingredients": [ + { + "description": "Large-scale dataset of human pairwise preferences for generated images", + "role": "TRAINING_DATA", + "rationale": "Necessary to provide the ground-truth human judgments required to train the scorer for alignment with subjective preferences." + }, + { + "description": "Pre-trained vision-language model backbone (e.g., CLIP)", + "role": "MODEL_INITIALIZATION", + "rationale": "Provides the foundational cross-modal representations needed to assess the semantic relationship between text prompts and visual content." + }, + { + "description": "Prompt-conditioned reward modeling architecture and objective", + "role": "CORE_METHOD", + "rationale": "Defines the specific neural network structure and loss function used to map image-text pairs to a scalar preference score." + }, + { + "description": "Human-preference alignment evaluation benchmark", + "role": "EVALUATION_PROTOCOL", + "rationale": "Required to validate that the automated scorer's rankings accurately reflect human preferences on unseen samples." + } + ], + "recall_judgments": [ + { + "reference_idx": 1, + "reference_ingredient": "Pre-trained vision-language encoder for prompt-conditioned image scoring", + "reference_role": "MODEL_INITIALIZATION", + "covered": true, + "best_match_idx": 2, + "reasoning": "Both describe the use of a pre-trained vision-language model (like CLIP) to initialize the cross-modal representations.", + "best_match": { + "description": "Pre-trained vision-language model backbone (e.g., CLIP)", + "role": "MODEL_INITIALIZATION", + "rationale": "Provides the foundational cross-modal representations needed to assess the semantic relationship between text prompts and visual content." + }, + "partial_matches": [] + }, + { + "reference_idx": 2, + "reference_ingredient": "Large-scale human-preference dataset over text-to-image generations", + "reference_role": "TRAINING_DATA", + "covered": true, + "best_match_idx": 1, + "reasoning": "Both explicitly identify the large-scale dataset of pairwise human preferences over generated images as the training data.", + "best_match": { + "description": "Large-scale dataset of human pairwise preferences for generated images", + "role": "TRAINING_DATA", + "rationale": "Necessary to provide the ground-truth human judgments required to train the scorer for alignment with subjective preferences." + }, + "partial_matches": [] + }, + { + "reference_idx": 3, + "reference_ingredient": "Pairwise preference-learning objective for reward-style image ranking", + "reference_role": "CORE_METHOD", + "covered": true, + "best_match_idx": 3, + "reasoning": "Both specify the reward-modeling/preference-learning objective required to train the scoring function to reflect human preferences.", + "best_match": { + "description": "Prompt-conditioned reward modeling architecture and objective", + "role": "CORE_METHOD", + "rationale": "Defines the specific neural network structure and loss function used to map image-text pairs to a scalar preference score." + }, + "partial_matches": [] + } + ], + "matched_pairs": [ + { + "reference_idx": 1, + "predicted_idx": 2, + "match": "full", + "reasoning": "Both describe the use of a pre-trained vision-language model (like CLIP) to initialize the cross-modal representations." + }, + { + "reference_idx": 2, + "predicted_idx": 1, + "match": "full", + "reasoning": "Both explicitly identify the large-scale dataset of pairwise human preferences over generated images as the training data." + }, + { + "reference_idx": 3, + "predicted_idx": 3, + "match": "full", + "reasoning": "Both specify the reward-modeling/preference-learning objective required to train the scoring function to reflect human preferences." + } + ], + "partial_matches": [], + "log_line": 19, + "n_steps": 2, + "tool_summary": { + "searches": [], + "retrieves": [], + "deeps": [] + } + }, + "failure": { + "paper_id": "neurips-2024-poster-219", + "claim_idx": 0, + "paper_title": "RGFN: Synthesizable Molecular Generation Using GFlowNets", + "claim": "Method: A generative framework for synthesizable molecular design, enabling generation of molecular candidates together with feasible synthesis pathways.", + "recall": 0.0, + "precision": 0.0, + "f1": 0.0, + "predicted_ingredients": [ + { + "description": "Markov Decision Process (MDP) formulation of molecular synthesis", + "role": "CONCEPTUAL_FRAMEWORK", + "rationale": "Frames the generation of molecules as a sequence of discrete synthetic steps, ensuring the output is synthesizable by construction." + }, + { + "description": "Library of reaction templates extracted from reaction databases (e.g., USPTO)", + "role": "DATA_SOURCE", + "rationale": "Provides the set of valid chemical transformations that the generative model can use to build molecules." + }, + { + "description": "Database of commercially available building blocks", + "role": "DATA_SOURCE", + "rationale": "Serves as the required starting materials for any proposed synthesis pathway." + }, + { + "description": "Neural policy network for reaction and building block selection", + "role": "CORE_METHOD", + "rationale": "The primary engine that learns to navigate the chemical space by selecting optimal synthetic actions." + }, + { + "description": "Synthetic feasibility and property evaluation protocol", + "role": "EVALUATION_PROTOCOL", + "rationale": "Used to validate that the generated molecules meet the desired chemical properties and that the pathways are realistic." + } + ], + "recall_judgments": [ + { + "reference_idx": 1, + "reference_ingredient": "Generative Flow Network (GFlowNet) framework trained with trajectory balance", + "reference_role": "CORE_METHOD", + "covered": false, + "best_match_idx": null, + "reasoning": "", + "best_match": null, + "partial_matches": [ + { + "reference_idx": 1, + "predicted_idx": 1, + "match": "partial", + "reasoning": "The predicted item captures the sequential MDP formulation underlying the generation process but is too broad and misses the specific GFlowNet framework and trajectory balance." + } + ] + }, + { + "reference_idx": 2, + "reference_ingredient": "Graph transformer architecture for state and action representation", + "reference_role": "CORE_METHOD", + "covered": false, + "best_match_idx": null, + "reasoning": "", + "best_match": null, + "partial_matches": [ + { + "reference_idx": 2, + "predicted_idx": 4, + "match": "partial", + "reasoning": "The predicted item identifies the neural policy network used to navigate the space but lacks the specific graph transformer architecture required by the reference." + } + ] + }, + { + "reference_idx": 3, + "reference_ingredient": "Predefined action space of robust chemical reaction templates and molecular building blocks", + "reference_role": "CONCEPTUAL_FRAMEWORK", + "covered": false, + "best_match_idx": null, + "reasoning": "", + "best_match": null, + "partial_matches": [ + { + "reference_idx": 3, + "predicted_idx": 2, + "match": "partial", + "reasoning": "The predicted item captures the reaction templates portion of the reference's predefined action space but omits the molecular building blocks." + }, + { + "reference_idx": 3, + "predicted_idx": 3, + "match": "partial", + "reasoning": "The predicted item captures the molecular building blocks portion of the reference's action space but omits the reaction templates." + } + ] + }, + { + "reference_idx": 4, + "reference_ingredient": "Fingerprint-based action embedding scheme for scalable fragment selection", + "reference_role": "CORE_METHOD", + "covered": false, + "best_match_idx": null, + "reasoning": "", + "best_match": null, + "partial_matches": [] + }, + { + "reference_idx": 5, + "reference_ingredient": "Training loss of the model", + "reference_role": "CORE_METHOD", + "covered": false, + "best_match_idx": null, + "reasoning": "", + "best_match": null, + "partial_matches": [] + } + ], + "matched_pairs": [], + "partial_matches": [ + { + "reference_idx": 1, + "predicted_idx": 1, + "match": "partial", + "reasoning": "The predicted item captures the sequential MDP formulation underlying the generation process but is too broad and misses the specific GFlowNet framework and trajectory balance." + }, + { + "reference_idx": 2, + "predicted_idx": 4, + "match": "partial", + "reasoning": "The predicted item identifies the neural policy network used to navigate the space but lacks the specific graph transformer architecture required by the reference." + }, + { + "reference_idx": 3, + "predicted_idx": 2, + "match": "partial", + "reasoning": "The predicted item captures the reaction templates portion of the reference's predefined action space but omits the molecular building blocks." + }, + { + "reference_idx": 3, + "predicted_idx": 3, + "match": "partial", + "reasoning": "The predicted item captures the molecular building blocks portion of the reference's action space but omits the reaction templates." + } + ], + "log_line": 38, + "n_steps": 2, + "tool_summary": { + "searches": [], + "retrieves": [], + "deeps": [] + } + }, + "gemini_analysis": { + "tooling_clarification": "No external search or retrieval tools were provided for evidence gathering. The agent relied entirely on its internal parametric memory, using tools only to verify the output format.", + "success_why": [ + "The model successfully recalled the exact architecture and training data for 'Pick-a-Pic', a prominent 2023 text-to-image alignment dataset.", + "It accurately identified the CLIP backbone, the large-scale human preference dataset, and the reward modeling objective without needing external grounding." + ], + "success_tool_role": [ + "No evidence tools were used.", + "Success was driven entirely by the LLM's strong parametric knowledge of highly cited, recent (2023) multimodal AI research." + ], + "failure_diagnosis": "Faced with a highly specific 2024 molecular design paper (RGFN), the model hallucinated a generic, plausible pipeline for synthesizable molecular generation (MDP, USPTO templates, neural policy network) but completely missed the actual GFlowNet methodology.", + "failure_error_bullets": [ + "[Missed core method] Failed to identify the GFlowNet framework and graph transformer architectures.", + "[Wrong specificity] Predicted a generic Markov Decision Process (MDP) instead of the specific GFlowNet trained with trajectory balance.", + "[Missing data/source] Missed the fingerprint-based action embedding scheme." + ], + "failure_what_needed": [ + "External retrieval of the paper's abstract to ground the generic claim ('generative framework for synthesizable molecular design') to the specific RGFN/GFlowNet methodology." + ] + } + }, + "codeagent_retrieval_plus_websearch": { + "tools": [ + "retrieve_documents", + "web_search_tool", + "check_answer_format" + ], + "summary_meta": { + "source": "gold", + "gold_file": "/Users/dingzifeng/SciFy-UMBC/dryrun/ForecastingDiscoveryPathways-7FCC/archive/dev.json", + "gold_claim_field": "rewritten_capability", + "setting": 1, + "generator_model": "gemini-3-flash-preview", + "generator_backend": "codeagent", + "codeagent_method": "retrieval_plus_websearch", + "n_claims": 50, + "n_total_claims": 50, + "n_errors": 0, + "mean_n_predicted": 4.7 + }, + "case": { + "paper_id": "2023.acl-short.123", + "claim_idx": 0, + "paper_title": "An Embarrassingly Easy but Strong Baseline for Nested Named Entity Recognition", + "claim": "Tool: A standardized preprocessing and split-construction workflow for nested NER benchmarks, enabling fair comparison across models.", + "recall": 0.5, + "precision": 0.2, + "f1": 0.286, + "predicted_ingredients": [ + { + "description": "Original nested NER datasets (e.g., GENIA, ACE 2004, ACE 2005)", + "role": "DATA_SOURCE", + "rationale": "These provide the raw annotated text data that the workflow standardizes for benchmarking purposes." + }, + { + "description": "Unified data representation format for nested entities", + "role": "CONCEPTUAL_FRAMEWORK", + "rationale": "A consistent schema is necessary to represent overlapping spans across diverse datasets in a single format." + }, + { + "description": "Automated preprocessing pipeline for tokenization and span alignment", + "role": "CORE_METHOD", + "rationale": "Ensures that all datasets are processed using identical linguistic rules to prevent bias from different tokenizers." + }, + { + "description": "Standardized split-construction protocol (e.g., Lu and Roth splits)", + "role": "EVALUATION_PROTOCOL", + "rationale": "Guarantees that models are compared on identical subsets of the data to maintain experimental fairness." + }, + { + "description": "Nested span evaluation metric implementation", + "role": "EVALUATION_PROTOCOL", + "rationale": "Required to accurately and consistently measure performance on overlapping entity structures across different models." + } + ], + "recall_judgments": [ + { + "reference_idx": 1, + "reference_ingredient": "Raw nested NER corpora", + "reference_role": "DATA_SOURCE", + "covered": true, + "best_match_idx": 1, + "reasoning": "Both identify the original nested NER corpora (such as ACE2004, ACE2005, and GENIA) that provide the raw annotated text data for standardization.", + "best_match": { + "description": "Original nested NER datasets (e.g., GENIA, ACE 2004, ACE 2005)", + "role": "DATA_SOURCE", + "rationale": "These provide the raw annotated text data that the workflow standardizes for benchmarking purposes." + }, + "partial_matches": [] + }, + { + "reference_idx": 2, + "reference_ingredient": "Tokenization and annotation correction heuristics", + "reference_role": "IMPLEMENTATION_TOOLING", + "covered": false, + "best_match_idx": null, + "reasoning": "", + "best_match": null, + "partial_matches": [ + { + "reference_idx": 2, + "predicted_idx": 3, + "match": "partial", + "reasoning": "Both refer to the tokenization and span alignment process, but the reference specifies the underlying heuristics to fix tokenization and annotation conflicts, whereas the prediction describes the broader automated pipeline." + } + ] + } + ], + "retrieved_doc_titles": [ + "Statistical Learning for Accurate and Interpretable Battery Lifetime Prediction", + "Accurate Performance Characterization, Reporting, and Benchmarking for Indoor Photovoltaics", + "Efficient implicit integration for finite-strain viscoplasticity with a nested multiplicative split", + "Prediction of charge separation in GaAs/AlAs cylindrical nanostructures", + "Effect of the structure of lead iodine perovskites on the photovoltaic efficiencies", + "The high strain-rate behaviour of three molecular weights of polyethylene examined with a magnesium alloy split-Hopkinson pressure bar", + "Landau Theory of Charge and Spin Ordering in the Nickelates", + "Excitonic effects in two-dimensional semiconductors: Path integral Monte Carlo approach" + ], + "gemini_analysis": { + "model": "gemini/gemini-3.1-pro-preview", + "focus": "performance when retrieval returned docs", + "headline": "Even with web search plus returned retrieval docs, the pathway stayed weakly grounded: retrieval was off-domain, so the model leaned on parametric knowledge and over-generated (F1 0.29).", + "what_was_retrieved": [ + "Web search was used in most steps (alongside retrieval early, then alone later), with queries about nested NER preprocessing / benchmark standardization.", + "Retrieval returned documents from unrelated domains (materials science, physics, battery lifetime), likely via polysemous terms such as \"nested\", \"split\", and \"benchmarking\".", + "No retrieved documents were about NLP / Named Entity Recognition." + ], + "effect_on_prediction": [ + "With off-domain retrieval and no strong grounding from tools, the model fell back on parametric knowledge of nested NER workflows.", + "That fallback recovered some recall (e.g., raw nested NER corpora / GENIA & ACE-style datasets) and a partial match on tokenization preprocessing.", + "Without relevant tool text to constrain the pathway, it also over-generated extra workflow pieces (unified format, evaluation metric), hurting precision (P=0.2)." + ], + "error_taxonomy_bullets": [ + "[Domain Mismatch in Retrieval] Returned abstracts matched generic keywords to the wrong scientific fields, so retrieval did not surface the gold ingredients.", + "[Web search under-used for grounding] Despite repeated nested-NER web queries, the final ingredients still look like a generic preprocessing pipeline rather than the paper-specific tokenization/annotation heuristics.", + "[Parametric Reliance + Over-generation] The model filled gaps from memory and added plausible but non-gold steps, lowering precision." + ], + "takeaway": "Web search ran throughout the trajectory, including after documents were returned. Retrieval returned materials/physics papers rather than nested-NER sources, so neither tool provided the specific gold ingredients.", + "web_search_also_used": true, + "web_search_queries_example": [ + "standardized preprocessing nested NER benchmarks workflow", + "nested NER benchmarks fair comparison tool", + "nested NER dataset preprocessing standardization", + "standardizing nested NER benchmarks" + ] + } + } + }, + "codeagent_websearch_deep": { + "tools": [ + "web_search_tool", + "deep_web_search_tool", + "check_answer_format" + ], + "summary_meta": { + "generator_model": "gemini-3-flash-preview", + "generator_backend": "codeagent", + "codeagent_method": "websearch_deep", + "mean_f1": 0.318, + "n_claims": 50 + }, + "success": { + "paper_id": "neurips-2025-spotlight-472", + "claim_idx": 1, + "paper_title": "ShapeLLM-Omni: A Native Multimodal LLM for 3D Generation and Understanding", + "claim": "Resource: A native 3D multimodal language model, enabling unified text, image, and 3D generation, understanding, and language-guided editing.", + "recall": 1.0, + "precision": 1.0, + "f1": 1.0, + "predicted_ingredients": [ + { + "description": "3D Vector-Quantized Variational Autoencoder (VQVAE)", + "role": "CORE_METHOD", + "rationale": "Necessary to map 3D objects into a discrete latent space of 3D-aware tokens that the LLM can natively process and generate." + }, + { + "description": "3D-Alpaca Dataset", + "role": "TRAINING_DATA", + "rationale": "A large-scale instruction-tuning dataset required to train the model on unified 3D generation, understanding, and editing tasks." + }, + { + "description": "Qwen-2.5-vl-7B-Instruct", + "role": "MODEL_INITIALIZATION", + "rationale": "Provides the foundational multimodal (text and image) capabilities that are extended to include 3D functionality." + }, + { + "description": "Unified Token-based Multimodal Architecture", + "role": "CONCEPTUAL_FRAMEWORK", + "rationale": "Enables the model to treat 3D tokens as a native modality alongside text and images for interleaved generation and understanding." + } + ], + "recall_judgments": [ + { + "reference_idx": 1, + "reference_ingredient": "Pretrained multimodal vision-language backbone", + "reference_role": "MODEL_INITIALIZATION", + "covered": true, + "best_match_idx": 3, + "reasoning": "Both identify the pretrained Qwen-2.5-VL vision-language model used as the foundational backbone for initialization.", + "best_match": { + "description": "Qwen-2.5-vl-7B-Instruct", + "role": "MODEL_INITIALIZATION", + "rationale": "Provides the foundational multimodal (text and image) capabilities that are extended to include 3D functionality." + }, + "partial_matches": [] + }, + { + "reference_idx": 2, + "reference_ingredient": "Discrete 3D tokenization through a 3D VQVAE", + "reference_role": "CORE_METHOD", + "covered": true, + "best_match_idx": 1, + "reasoning": "Both describe the use of a 3D VQVAE to convert 3D objects into discrete tokens that the model can process natively.", + "best_match": { + "description": "3D Vector-Quantized Variational Autoencoder (VQVAE)", + "role": "CORE_METHOD", + "rationale": "Necessary to map 3D objects into a discrete latent space of 3D-aware tokens that the LLM can natively process and generate." + }, + "partial_matches": [] + }, + { + "reference_idx": 3, + "reference_ingredient": "Fully autoregressive early-fusion next-token modeling over text, image-conditioned inputs, and 3D tokens", + "reference_role": "CONCEPTUAL_FRAMEWORK, CORE_METHOD", + "covered": true, + "best_match_idx": 4, + "reasoning": "Both refer to the unified, token-based conceptual framework that treats 3D, text, and image tokens in a single interleaved architecture.", + "best_match": { + "description": "Unified Token-based Multimodal Architecture", + "role": "CONCEPTUAL_FRAMEWORK", + "rationale": "Enables the model to treat 3D tokens as a native modality alongside text and images for interleaved generation and understanding." + }, + "partial_matches": [] + }, + { + "reference_idx": 4, + "reference_ingredient": "3D instruction-tuning corpus", + "reference_role": "TRAINING_DATA", + "covered": true, + "best_match_idx": 2, + "reasoning": "Both specify the 3D-Alpaca instruction-tuning dataset used to train and adapt the model for 3D tasks.", + "best_match": { + "description": "3D-Alpaca Dataset", + "role": "TRAINING_DATA", + "rationale": "A large-scale instruction-tuning dataset required to train the model on unified 3D generation, understanding, and editing tasks." + }, + "partial_matches": [] + } + ], + "matched_pairs": [ + { + "reference_idx": 1, + "predicted_idx": 3, + "match": "full", + "reasoning": "Both identify the pretrained Qwen-2.5-VL vision-language model used as the foundational backbone for initialization." + }, + { + "reference_idx": 2, + "predicted_idx": 1, + "match": "full", + "reasoning": "Both describe the use of a 3D VQVAE to convert 3D objects into discrete tokens that the model can process natively." + }, + { + "reference_idx": 3, + "predicted_idx": 4, + "match": "full", + "reasoning": "Both refer to the unified, token-based conceptual framework that treats 3D, text, and image tokens in a single interleaved architecture." + }, + { + "reference_idx": 4, + "predicted_idx": 2, + "match": "full", + "reasoning": "Both specify the 3D-Alpaca instruction-tuning dataset used to train and adapt the model for 3D tasks." + } + ] + }, + "failure": { + "paper_id": "neurips-2023-poster-1534", + "claim_idx": 0, + "paper_title": "Continuous-Time Functional Diffusion Processes", + "claim": "Method: A diffusion-based generative framework for continuous function spaces, enabling modeling of continuous and irregularly sampled data.", + "recall": 0.0, + "precision": 0.0, + "f1": 0.0, + "predicted_ingredients": [ + { + "description": "Denoising diffusion probabilistic modeling adapted for finite-dimensional marginals of functions", + "role": "CORE_METHOD", + "rationale": "Provides the generative mechanism to sample from functional distributions by reversing a noise process applied to function values at arbitrary locations." + }, + { + "description": "Permutation-invariant attention-based architecture", + "role": "IMPLEMENTATION_TOOLING", + "rationale": "Ensures the model can process sets of input-output pairs of varying sizes while satisfying the exchangeability property required for stochastic processes." + }, + { + "description": "Conceptual framework of exchangeability and consistency for stochastic processes", + "role": "CONCEPTUAL_FRAMEWORK", + "rationale": "Provides the theoretical foundation for defining a valid distribution over an infinite-dimensional function space via its finite marginals." + }, + { + "description": "Coordinate-based representation of data observations", + "role": "CONCEPTUAL_FRAMEWORK", + "rationale": "Enables the model to handle irregularly sampled data by treating observations as evaluations of a function at specific continuous coordinates." + }, + { + "description": "Score-based or denoising objective defined over function values at given coordinates", + "role": "CORE_METHOD", + "rationale": "Allows the model to learn the underlying structure of the function space by predicting noise or scores at arbitrary input points." + } + ], + "recall_judgments": [ + { + "reference_idx": 1, + "reference_ingredient": "Score-based generative modeling with continuous-time stochastic differential equations", + "reference_role": "CONCEPTUAL_FRAMEWORK", + "covered": false, + "best_match_idx": null, + "reasoning": "", + "best_match": null, + "partial_matches": [ + { + "reference_idx": 1, + "predicted_idx": 1, + "match": "partial", + "reasoning": "Both identify the core diffusion generative mechanism, but the reference specifies a continuous-time SDE formulation while the prediction focuses on adapting diffusion for finite marginals." + }, + { + "reference_idx": 1, + "predicted_idx": 5, + "match": "partial", + "reasoning": "Both highlight the score-based or denoising formulation, though the prediction restricts it to function values at given coordinates." + } + ] + }, + { + "reference_idx": 2, + "reference_ingredient": "Infinite-dimensional stochastic differential equation framework for forward and reverse diffusion in Hilbert spaces", + "reference_role": "CONCEPTUAL_FRAMEWORK", + "covered": false, + "best_match_idx": null, + "reasoning": "", + "best_match": null, + "partial_matches": [ + { + "reference_idx": 2, + "predicted_idx": 3, + "match": "partial", + "reasoning": "Both provide the theoretical foundation for defining distributions over infinite-dimensional function spaces, but use different mathematical frameworks (Hilbert space SDEs versus exchangeability/consistency)." + } + ] + }, + { + "reference_idx": 3, + "reference_ingredient": "Countable-evaluation formulation that connects continuous functions to discrete observations for practical learning and inference", + "reference_role": "CORE_METHOD", + "covered": false, + "best_match_idx": null, + "reasoning": "", + "best_match": null, + "partial_matches": [ + { + "reference_idx": 3, + "predicted_idx": 1, + "match": "partial", + "reasoning": "Both address the need to connect continuous function spaces to discrete observations (countable evaluations versus finite-dimensional marginals) for practical inference." + }, + { + "reference_idx": 3, + "predicted_idx": 4, + "match": "partial", + "reasoning": "Both focus on bridging continuous functions with discrete data points, though they describe it differently (countable-evaluation formulation versus coordinate-based representations)." + } + ] + }, + { + "reference_idx": 4, + "reference_ingredient": "Implicit neural representation-based practical implementation of the functional score network", + "reference_role": "CORE_METHOD", + "covered": false, + "best_match_idx": null, + "reasoning": "", + "best_match": null, + "partial_matches": [ + { + "reference_idx": 4, + "predicted_idx": 4, + "match": "partial", + "reasoning": "Implicit neural representations are a form of coordinate-based representation, but the prediction describes applying them to data observations rather than the functional score network." + } + ] + } + ], + "partial_matches": [ + { + "reference_idx": 1, + "predicted_idx": 1, + "match": "partial", + "reasoning": "Both identify the core diffusion generative mechanism, but the reference specifies a continuous-time SDE formulation while the prediction focuses on adapting diffusion for finite marginals." + }, + { + "reference_idx": 1, + "predicted_idx": 5, + "match": "partial", + "reasoning": "Both highlight the score-based or denoising formulation, though the prediction restricts it to function values at given coordinates." + }, + { + "reference_idx": 2, + "predicted_idx": 3, + "match": "partial", + "reasoning": "Both provide the theoretical foundation for defining distributions over infinite-dimensional function spaces, but use different mathematical frameworks (Hilbert space SDEs versus exchangeability/consistency)." + }, + { + "reference_idx": 3, + "predicted_idx": 1, + "match": "partial", + "reasoning": "Both address the need to connect continuous function spaces to discrete observations (countable evaluations versus finite-dimensional marginals) for practical inference." + }, + { + "reference_idx": 3, + "predicted_idx": 4, + "match": "partial", + "reasoning": "Both focus on bridging continuous functions with discrete data points, though they describe it differently (countable-evaluation formulation versus coordinate-based representations)." + }, + { + "reference_idx": 4, + "predicted_idx": 4, + "match": "partial", + "reasoning": "Implicit neural representations are a form of coordinate-based representation, but the prediction describes applying them to data observations rather than the functional score network." + } + ] + }, + "gemini_analysis": { + "model": "gemini/gemini-3.1-pro-preview", + "focus": "tool-aware (web_search_tool + deep_web_search_tool)", + "tooling_clarification": "The agent uses `web_search_tool` to retrieve search engine snippets and `deep_web_search_tool` to fetch and read full webpage content. There is no 'crawl' tool; `deep_web_search_tool` is the sole mechanism for extracting full-text page content.", + "success_search_contribution": [ + "Formulated targeted queries (e.g., 'native 3D multimodal language model unified text image 3D generation') to locate primary and secondary sources for ShapeLLM-Omni.", + "Successfully identified the project page ('jamesyjl.github.io/ShapeLLM/') as a viable alternative source.", + "Used follow-up searches ('ShapeLLM-Omni 3D-Alpaca dataset source Objaverse') to verify specific entities like '3D-Alpaca' and '3D-aware discrete tokens' found in earlier steps." + ], + "success_deep_contribution": [ + "Attempted to fetch the arXiv HTML ('https://arxiv.org/html/2506.01853v1') but encountered a PDF parse error ('No /Root object!'), prompting a necessary pivot.", + "Successfully fetched the project page ('https://jamesyjl.github.io/ShapeLLM/'), extracting exact architectural details ('3D VQVAE', 'Qwen-2.5-vl-7B-Instruct') and dataset names ('3D-Alpaca') that perfectly aligned with the gold ingredients." + ], + "success_why_overall": [ + "The agent demonstrated tool resilience by pivoting to a project page when the primary arXiv HTML deep-fetch failed.", + "The deep-fetched project page contained highly specific, structured information about the method's architecture and training data, which the agent accurately synthesized into the final ingredients." + ], + "failure_diagnosis": "The agent successfully surfaced the correct paper via search but failed to deep-fetch any of its URLs, instead deep-fetching and extracting ingredients from a neighboring method (Neural Diffusion Processes).", + "failure_search_what_went_wrong": [ + "Search successfully found the correct target paper ('Continuous-Time Functional Diffusion Processes') and its URLs (NeurIPS abstract, arXiv HTML, GitHub project page) in steps 5 and 7.", + "Search snippets contained highly relevant gold terms (e.g., 'hilbert-spaces', 'stochastic-partial-differential-equations'), but the agent ignored them.", + "The agent incorrectly pivoted its search strategy to query a neighboring method ('Neural Diffusion Processes Dutordoir ICML 2023 architecture')." + ], + "failure_deep_what_went_wrong": [ + "Never called `deep_web_search_tool` on the correct FDP URLs found in search (e.g., 'arxiv.org/html/2303.00800v3').", + "Wasted deep-fetch calls on guessed/wrong arXiv IDs (e.g., 2303.01631, 2205.15306) which returned completely unrelated papers.", + "Deep-fetched OpenReview and PMLR pages for the wrong paper ('Neural Diffusion Processes'), leading to the extraction of incorrect methodological details." + ], + "failure_error_bullets": [ + "[Substitutes neighboring method] The agent extracted ingredients for 'Neural Diffusion Processes' (finite marginals, exchangeability, attention) instead of the gold 'Functional Diffusion Processes' (Hilbert-space SDEs, countable evaluation, INR) because it deep-fetched the wrong paper's PMLR page.", + "[Candidate retrieval failure] Although `web_search_tool` surfaced the correct URLs, the agent failed to retrieve their full content via `deep_web_search_tool`, relying instead on deep-fetches of incorrect papers.", + "[Missed core method] By focusing on the wrong paper's deep-fetched content, the agent completely missed the core continuous-time SDE and Hilbert space framework of the target paper." + ], + "failure_what_needed": [ + "The agent needed to pass the correct URLs found by `web_search_tool` (e.g., 'https://arxiv.org/html/2303.00800v3') directly into `deep_web_search_tool` instead of guessing arXiv IDs or pivoting to fetch pages for Neural Diffusion Processes." + ] + }, + "tool_traces": { + "tools_available": { + "web_search_tool": "Broad web search; returns list of {source URL, summary}. NOT a crawler.", + "deep_web_search_tool": "Given a URL, fetches webpage/PDF content as {title,url,content}. This is the page-fetch / crawl-like tool. There is NO separate tool named crawl.", + "check_answer_format": "Validates final JSON schema." + }, + "success": { + "paper_id": "neurips-2025-spotlight-472", + "claim_idx": 1, + "n_steps": 4, + "web_search_queries": [ + "native 3D multimodal language model unified text image 3D generation", + "3D multimodal language model understanding and language-guided editing", + "native 3D-LLM text image 3D generation understanding editing", + "ShapeLLM-Omni 3D-Alpaca dataset source Objaverse" + ], + "web_search_hits_notable": [ + "arxiv.org/html/2506.01853v1 ShapeLLM-Omni", + "neurips.cc/virtual/2025/poster/116179 mentioning 3D-Alpaca", + "project page jamesyjl.github.io/ShapeLLM/" + ], + "deep_fetches": [ + { + "url": "https://arxiv.org/html/2506.01853v1", + "outcome": "FAILED: Error fetching webpage: No /Root object! - Is this really a PDF?" + }, + { + "url": "https://jamesyjl.github.io/ShapeLLM/", + "outcome": "SUCCESS: project page content mentioning ShapeLLM-Omni, 3D tokens, training dataset details, backbone mentions" + } + ], + "final_ingredients": [ + "3D VQVAE", + "3D-Alpaca Dataset", + "Qwen-2.5-vl-7B-Instruct", + "Unified Token-based Multimodal Architecture" + ] + }, + "failure": { + "paper_id": "neurips-2023-poster-1534", + "claim_idx": 0, + "n_steps": 17, + "web_search_queries_sample": [ + "diffusion-based generative framework for continuous function spaces", + "score-based generative modeling in function space", + "Functional Diffusion Processes NeurIPS 2023", + "Continuous-Time Functional Diffusion Processes irregularly sampled data", + "Neural Diffusion Processes Dutordoir ICML 2023 architecture" + ], + "web_search_correct_hits_but_underused": [ + "NeurIPS 2023 abstract page for Continuous-Time Functional Diffusion Processes (found in step 5)", + "arxiv.org/html/2303.00800v3 Continuous-Time Functional Diffusion Processes (found in step 7)", + "1zb.github.io/functional-diffusion/ project page (found in step 7)" + ], + "deep_fetches": [ + { + "url": "https://arxiv.org/abs/2302.03046", + "intended": "guessed target paper", + "outcome": "attempted in malformed multi-code step; no usable obs" + }, + { + "url": "https://arxiv.org/abs/2303.01631", + "intended": "FDP paper", + "outcome": "WRONG PAPER: Real-Time Tube-Based Non-Gaussian Risk Bounded Motion Planning..." + }, + { + "url": "https://arxiv.org/abs/2205.15306", + "intended": "Neural Diffusion Processes", + "outcome": "WRONG PAPER: How to Compose Shortest Paths" + }, + { + "url": "https://openreview.net/forum?id=tV7GSY5GYG", + "intended": "NDP OpenReview", + "outcome": "fetched NDP OpenReview page (neighboring method, not gold FDP)" + }, + { + "url": "https://proceedings.mlr.press/v202/dutordoir23a.html", + "intended": "NDP PMLR", + "outcome": "fetched NDP paper page; agent synthesized NDP ingredients" + } + ], + "failure_mode_tooling": "Search surfaced the correct FDP paper, but deep_web_search_tool was applied to wrong arXiv IDs and then to a neighboring Neural Diffusion Processes paper. Final ingredients mirror NDP (finite marginals, exchangeability, attention) rather than FDP (Hilbert-space continuous-time SDEs, countable evaluation, INR score network)." + } + } + } + } +} \ No newline at end of file diff --git a/hf_space/system_run_data.py b/hf_space/system_run_data.py new file mode 100644 index 0000000000000000000000000000000000000000..ae5db5336ace0a5527d2781ffdcf64e784695f34 --- /dev/null +++ b/hf_space/system_run_data.py @@ -0,0 +1,444 @@ +"""Load curated System Run case studies for the Streamlit demo. + +Labels and section structure follow the SciPaths Dev · CodeAgent Case Studies HTML report. +""" + +from __future__ import annotations + +import json +from functools import lru_cache +from pathlib import Path +from typing import Any, Optional + +CASES_DIR = Path(__file__).resolve().parent / "system_run_cases" +ANALYSIS_PATH = CASES_DIR / "case_study_analysis.json" + +SYSTEM_RUN_METHODS: list[dict[str, str]] = [ + { + "id": "codeagent_parametric", + "label": "Parametric - LLM Only", + "banner_title": "1 · codeagent_parametric", + "banner_desc": "Parametric-only generation (no search/retrieval). Mean F1 ≈ 0.20.", + }, + { + "id": "codeagent_websearch_deep", + # Audience label: "Crawl" = deep_web_search_tool (page/PDF fetch; no separate crawl tool). + "label": "Websearch Deep - LLM + Web Search + Crawl", + "banner_title": "2 · codeagent_websearch_deep", + "banner_desc": ( + "Tools: web_search_tool + deep_web_search_tool (page fetch / crawl). " + "Mean F1 ≈ 0.32." + ), + }, +] + +DEFAULT_SYSTEM_METHOD = SYSTEM_RUN_METHODS[0]["id"] +DEFAULT_SYSTEM_CASE = "success" + +# Case keys exposed in the UI — labels match the HTML report. +_METHOD_CASES: dict[str, list[tuple[str, str]]] = { + "codeagent_parametric": [("success", "Success"), ("failure", "Failure")], + "codeagent_websearch_deep": [("success", "Success"), ("failure", "Failure")], +} + +CASE_OUTCOME_LABEL: dict[str, str] = { + "success": "Success", + "failure": "Failure", +} + +TOOLING_NOTE_HTML = ( + "Tooling by setting." + "
    · parametric: parametric knowledge only (no search/retrieval)." + "
    · websearch_deep: web_search_tool + " + "deep_web_search_tool (URL page fetch)." +) + + +@lru_cache(maxsize=1) +def load_analysis() -> dict[str, Any]: + with ANALYSIS_PATH.open("r", encoding="utf-8") as f: + data = json.load(f) + if not isinstance(data, dict): + raise ValueError(f"Invalid case study analysis at {ANALYSIS_PATH}") + return data + + +def method_ids() -> list[str]: + return [m["id"] for m in SYSTEM_RUN_METHODS] + + +def method_meta(method_id: str) -> dict[str, str]: + for m in SYSTEM_RUN_METHODS: + if m["id"] == method_id: + return m + return {"id": method_id, "label": method_id, "banner_title": method_id, "banner_desc": ""} + + +def method_label(method_id: str) -> str: + return method_meta(method_id).get("label") or method_id + + +def case_chip_label(method_id: str, case_key: str) -> str: + """Top-of-card label, e.g. 'Parametric · Success'.""" + pretty = { + "codeagent_parametric": "Parametric", + "codeagent_websearch_deep": "WebSearch Deep", + }.get(method_id, method_label(method_id)) + outcome = CASE_OUTCOME_LABEL.get(case_key, case_key) + return f"{pretty} · {outcome}" + + +def list_cases(method_id: str) -> list[tuple[str, str]]: + return list(_METHOD_CASES.get(method_id, [])) + + +def default_case_for_method(method_id: str) -> str: + cases = list_cases(method_id) + return cases[0][0] if cases else DEFAULT_SYSTEM_CASE + + +def outcome_tone(case_key: str, f1: float) -> str: + if case_key == "success": + return "ok" + if case_key == "failure": + return "bad" + if f1 >= 0.75: + return "ok" + if f1 <= 0.05: + return "bad" + return "part" + + +def _setting(method_id: str) -> dict[str, Any]: + settings = load_analysis().get("settings") or {} + block = settings.get(method_id) + if not isinstance(block, dict): + raise KeyError(f"Unknown system-run method: {method_id}") + return block + + +def _raw_case_block(method_id: str, case_key: str) -> dict[str, Any]: + setting = _setting(method_id) + if case_key == "featured": + block = setting.get("case") + else: + block = setting.get(case_key) + if not isinstance(block, dict): + raise KeyError(f"Unknown case {case_key!r} for method {method_id}") + return block + + +def _gold_ingredients(case: dict[str, Any]) -> list[dict[str, Any]]: + """Gold column with full/partial/miss badges and match notes (HTML report style).""" + predicted = case.get("predicted_ingredients") or [] + gold: list[dict[str, Any]] = [] + for judgment in case.get("recall_judgments") or []: + if not isinstance(judgment, dict): + continue + desc = str(judgment.get("reference_ingredient") or "").strip() + role = str(judgment.get("reference_role") or "").strip() + if not desc: + continue + covered = bool(judgment.get("covered")) + partials = judgment.get("partial_matches") or [] + if not isinstance(partials, list): + partials = [] + match_notes: list[str] = [] + judge_note = "" + pair_pred_idx: Optional[int] = None + if covered: + badge = "full" + best = judgment.get("best_match") if isinstance(judgment.get("best_match"), dict) else {} + best_desc = str(best.get("description") or "").strip() + if best_desc: + match_notes.append(f"Matched to pred: {best_desc}") + judge_note = str(judgment.get("reasoning") or "").strip() + raw_idx = judgment.get("best_match_idx") + if isinstance(raw_idx, int) and raw_idx > 0: + pair_pred_idx = raw_idx + elif best_desc: + # Fall back to description match against predicted list (1-based). + for i, pred in enumerate(predicted, start=1): + if isinstance(pred, dict) and str(pred.get("description") or "").strip() == best_desc: + pair_pred_idx = i + break + elif partials: + badge = "partial" + for pm in partials: + if not isinstance(pm, dict): + continue + pred_idx = pm.get("predicted_idx") + reason = str(pm.get("reasoning") or "").strip() + if pred_idx is not None and reason: + match_notes.append(f"Partial ↔ pred #{pred_idx}: {reason}") + elif reason: + match_notes.append(reason) + else: + badge = "miss" + gold.append( + { + "description": desc, + "role": role, + "badge": badge, + "match_notes": match_notes, + "judge_note": judge_note, + "pair_pred_idx": pair_pred_idx, + } + ) + return gold + + +def build_pair_rows( + gold: list[dict[str, Any]], predicted: list[dict[str, Any]] +) -> list[dict[str, Any]]: + """Align gold↔prediction rows for success-case connector lines.""" + by_idx = { + int(p["idx"]): p + for p in predicted + if isinstance(p.get("idx"), int) + } + used: set[int] = set() + rows: list[dict[str, Any]] = [] + for g in gold: + pred_idx = g.get("pair_pred_idx") + pred = by_idx.get(int(pred_idx)) if isinstance(pred_idx, int) else None + linked = bool(pred is not None and g.get("badge") == "full") + if linked and isinstance(pred_idx, int): + used.add(pred_idx) + rows.append({"gold": g, "pred": pred, "linked": linked}) + for p in predicted: + idx = p.get("idx") + if isinstance(idx, int) and idx not in used: + rows.append({"gold": None, "pred": p, "linked": False}) + return rows + + +def _tool_panel(method_id: str, case_key: str, case: dict[str, Any], setting: dict[str, Any]) -> dict[str, Any]: + """Structured tool evidence / traces for the case card.""" + if method_id == "codeagent_retrieval_plus_websearch": + analysis = case.get("gemini_analysis") if isinstance(case.get("gemini_analysis"), dict) else {} + return { + "kind": "retrieval", + "intro": ( + "This run used both retrieve_documents and web_search_tool. " + "Web search continued in later steps after documents were returned." + ), + "queries": list(analysis.get("web_search_queries_example") or []), + "retrieved_docs": list(case.get("retrieved_doc_titles") or []), + } + + if method_id == "codeagent_websearch_deep": + traces = setting.get("tool_traces") if isinstance(setting.get("tool_traces"), dict) else {} + trace = traces.get("success" if case_key == "success" else "failure") + if not isinstance(trace, dict): + return {"kind": "none"} + deep_fetches = [] + for item in trace.get("deep_fetches") or []: + if not isinstance(item, dict): + continue + outcome = str(item.get("outcome") or "") + ok = outcome.upper().startswith("SUCCESS") + deep_fetches.append( + { + "url": str(item.get("url") or ""), + "outcome": outcome, + "ok": ok, + } + ) + return { + "kind": "deep", + "queries": list( + trace.get("web_search_queries") + or trace.get("web_search_queries_sample") + or [] + ), + "hits": list( + trace.get("web_search_hits_notable") + or trace.get("web_search_correct_hits_but_underused") + or [] + ), + "deep_fetches": deep_fetches, + } + + return {"kind": "none"} + + +def get_case(method_id: str, case_key: str) -> dict[str, Any]: + """Normalized case card payload for the UI.""" + setting = _setting(method_id) + case = _raw_case_block(method_id, case_key) + tools = [str(t) for t in (setting.get("tools") or [])] + predicted = [] + for idx, item in enumerate(case.get("predicted_ingredients") or [], start=1): + if not isinstance(item, dict): + continue + predicted.append( + { + "idx": idx, + "description": str(item.get("description") or "").strip(), + "role": str(item.get("role") or "").strip(), + "rationale": str(item.get("rationale") or "").strip(), + } + ) + f1 = float(case.get("f1") or 0.0) + return { + "method_id": method_id, + "method_label": method_label(method_id), + "case_key": case_key, + "case_chip": case_chip_label(method_id, case_key), + "outcome_tone": outcome_tone(case_key, f1), + "tools": tools, + "paper_id": str(case.get("paper_id") or ""), + "claim_idx": case.get("claim_idx"), + "paper_title": str(case.get("paper_title") or ""), + "claim": str(case.get("claim") or ""), + "recall": float(case.get("recall") or 0.0), + "precision": float(case.get("precision") or 0.0), + "f1": f1, + "gold_ingredients": _gold_ingredients(case), + "predicted_ingredients": predicted, + "tool_panel": _tool_panel(method_id, case_key, case, setting), + "banner": method_meta(method_id), + } + + +def get_analysis(method_id: str, case_key: Optional[str] = None) -> dict[str, Any]: + setting = _setting(method_id) + if method_id == "codeagent_retrieval_plus_websearch": + case = setting.get("case") or {} + analysis = case.get("gemini_analysis") if isinstance(case, dict) else None + return dict(analysis) if isinstance(analysis, dict) else {} + + analysis = setting.get("gemini_analysis") + if not isinstance(analysis, dict): + return {} + out = dict(analysis) + out["_case_key"] = case_key or "" + return out + + +def analysis_view_for_case(method_id: str, case_key: str) -> dict[str, Any]: + """Structured analysis view matching HTML report headings.""" + analysis = get_analysis(method_id, case_key) + if not analysis: + return {"model_tag": "Gemini 3.1 Pro analysis", "blocks": []} + + blocks: list[dict[str, Any]] = [] + + if method_id == "codeagent_retrieval_plus_websearch": + headline = str(analysis.get("headline") or "").strip() + takeaway = str(analysis.get("takeaway") or "").strip() + diag = headline + if takeaway: + diag = f"{headline}\n{takeaway}" if headline else takeaway + if diag: + blocks.append({"type": "diagnosis", "text": diag, "strong_first": True}) + two_col = [] + for key, title in ( + ("what_was_retrieved", "What the tools returned"), + ("effect_on_prediction", "Effect on prediction"), + ): + items = analysis.get(key) + if isinstance(items, list) and items: + two_col.append({"title": title, "items": [str(x) for x in items]}) + if two_col: + blocks.append({"type": "two_col", "cols": two_col}) + errs = analysis.get("error_taxonomy_bullets") + if isinstance(errs, list) and errs: + blocks.append( + {"type": "list", "title": "Error analysis", "items": [str(x) for x in errs]} + ) + return {"model_tag": "Gemini 3.1 Pro analysis", "blocks": blocks} + + if method_id == "codeagent_parametric": + if case_key == "success": + why = analysis.get("success_why") + if isinstance(why, list) and why: + blocks.append( + { + "type": "list", + "title": "Why this succeeded", + "items": [str(x) for x in why], + } + ) + else: + if analysis.get("failure_diagnosis"): + blocks.append( + { + "type": "diagnosis", + "text": f"Diagnosis. {analysis['failure_diagnosis']}", + "strong_first": True, + } + ) + errs = analysis.get("failure_error_bullets") + if isinstance(errs, list) and errs: + blocks.append( + { + "type": "list", + "title": "Error taxonomy", + "items": [str(x) for x in errs], + } + ) + needed = analysis.get("failure_what_needed") + if isinstance(needed, list) and needed: + blocks.append( + { + "type": "list", + "title": "What recovery would have required", + "items": [str(x) for x in needed], + } + ) + return {"model_tag": "Gemini 3.1 Pro analysis", "blocks": blocks} + + # websearch_deep + if case_key == "success": + two_col = [] + for key, title in ( + ("success_search_contribution", "web_search_tool"), + ("success_deep_contribution", "deep_web_search_tool"), + ): + items = analysis.get(key) + if isinstance(items, list) and items: + two_col.append({"title": title, "items": [str(x) for x in items]}) + if two_col: + blocks.append({"type": "two_col", "cols": two_col}) + why = analysis.get("success_why_overall") + if isinstance(why, list) and why: + blocks.append( + {"type": "list", "title": "Why this succeeded", "items": [str(x) for x in why]} + ) + else: + if analysis.get("failure_diagnosis"): + blocks.append( + { + "type": "diagnosis", + "text": f"Diagnosis. {analysis['failure_diagnosis']}", + "strong_first": True, + } + ) + two_col = [] + for key, title in ( + ("failure_search_what_went_wrong", "web_search_tool"), + ("failure_deep_what_went_wrong", "deep_web_search_tool"), + ): + items = analysis.get(key) + if isinstance(items, list) and items: + two_col.append({"title": title, "items": [str(x) for x in items]}) + if two_col: + blocks.append({"type": "two_col", "cols": two_col}) + errs = analysis.get("failure_error_bullets") + if isinstance(errs, list) and errs: + blocks.append( + {"type": "list", "title": "Error taxonomy", "items": [str(x) for x in errs]} + ) + needed = analysis.get("failure_what_needed") + if isinstance(needed, list) and needed: + blocks.append( + { + "type": "list", + "title": "What recovery would have required", + "items": [str(x) for x in needed], + } + ) + + return {"model_tag": "Gemini 3.1 Pro analysis", "blocks": blocks} diff --git a/hf_space/workflow_graph.py b/hf_space/workflow_graph.py new file mode 100644 index 0000000000000000000000000000000000000000..3c11ed628fca6a6616f6592c942a1f57d81d635f --- /dev/null +++ b/hf_space/workflow_graph.py @@ -0,0 +1,1197 @@ +"""Evolving SciPaths workflow graph (Neo4j-backed). + +Visual formatting follows Mina Brain's knowledge-graph canvas strictly: +filled circle + white inner stroke + colored ring, 25-char labels with white +halo, expanding write-pulse rings, #fafbfd stage, click detail card. +""" + +from __future__ import annotations + +import json +import re +from pathlib import Path +from typing import Any, Optional + +import streamlit as st + +try: + import neo4j_workflow as neo4j_store +except Exception: # pragma: no cover + neo4j_store = None # type: ignore + +# Short labels shared by Steps panel + workflow graph captions. +STEP_COPY: dict[int, str] = { + 1: "Load the paper", + 2: "Find where it’s cited", + 3: "Collect citation contexts", + 4: "Classify how it’s used", + 5: "Keep real reuse (uses / extends)", + 6: "Pull the citing passages", + 7: "Group similar reuse themes and summarize target contributions", + 8: "Find enabling contributions and corresponding prior studies", +} + +WORK_PANEL_STEPS: list[tuple[int, str]] = [(n, STEP_COPY[n]) for n in range(1, 9)] + +CANVAS_LABEL_MAX = 25 + +# Mina Brain NODE_COLORS mapped onto SciPaths kinds (fill / ring / text). +KIND_META = { + "target": { + "fill": "#4F6EF7", + "ring": "#A5B4FC", + "text": "#1E3A8A", + "r": 22, + "type_label": "Target paper", + "color": "#4F6EF7", + "size": 22, + }, + "citing": { + "fill": "#60A5FA", + "ring": "#BFDBFE", + "text": "#1D4ED8", + "r": 14, + "type_label": "Citing paper", + "color": "#60A5FA", + "size": 14, + }, + "cluster": { + "fill": "#8B5CF6", + "ring": "#DDD6FE", + "text": "#5B21B6", + "r": 16, + "type_label": "Reuse theme", + "color": "#8B5CF6", + "size": 16, + }, + "claim": { + "fill": "#FBBF24", + "ring": "#FDE68A", + "text": "#B45309", + "r": 15, + "type_label": "Target contribution", + "color": "#FBBF24", + "size": 15, + }, + "ingredient": { + "fill": "#94A3B8", + "ring": "#E2E8F0", + "text": "#475569", + "r": 12, + "type_label": "Enabling contribution", + "color": "#94A3B8", + "size": 12, + }, + "study": { + "fill": "#34D399", + "ring": "#A7F3D0", + "text": "#047857", + "r": 13, + "type_label": "Prior study", + "color": "#34D399", + "size": 13, + }, +} + +# Category order + compact per-kind palette the canvas shell + rail consume. +_KIND_ORDER = ["target", "claim", "ingredient", "cluster", "study", "citing"] +_KIND_META_JS = { + k: {"fill": v["fill"], "ring": v["ring"], "text": v["text"], "label": v["type_label"]} + for k, v in KIND_META.items() +} + + +def _load_json(path: Path) -> Any | None: + if not path.exists(): + return None + try: + return json.loads(path.read_text(encoding="utf-8")) + except Exception: + return None + + +def _clean(text: str) -> str: + return re.sub(r"\s+", " ", (text or "").strip()) + + +def canvas_label(text: str, *, limit: int = CANVAS_LABEL_MAX) -> str: + cleaned = _clean(text) + if len(cleaned) <= limit: + return cleaned + return cleaned[:limit].rstrip() + "…" + + +def visible_step_from_events(events: list[str]) -> int: + started = 0 + for raw in events or []: + text = str(raw) + m = re.search(r"Step\s+(\d+)\s*(?:/|\s+complete)", text, re.IGNORECASE) + if m: + started = max(started, int(m.group(1))) + continue + lower = text.lower() + if "[annotation]" in lower or "annotate" in lower: + started = max(started, 8) + return min(8, started) + + +def pulse_step_from_events(events: list[str]) -> int: + started: set[int] = set() + completed: set[int] = set() + for raw in events or []: + text = str(raw) + m_done = re.search(r"Step\s+(\d+)\s+complete", text, re.IGNORECASE) + if m_done: + completed.add(int(m_done.group(1))) + continue + m_start = re.search(r"Step\s+(\d+)\s*/", text) + if m_start: + started.add(int(m_start.group(1))) + continue + lower = text.lower() + if "[annotation]" in lower or "annotate" in lower: + if "complete" in lower or "skipped" in lower: + completed.add(8) + else: + started.add(8) + active = [n for n in sorted(started) if n not in completed] + return active[-1] if active else 0 + + +def _paper_title_from_dir(paper_dir: Optional[Path]) -> str: + if not paper_dir: + return "Target paper" + data = _load_json(paper_dir / "paper_metadata.json") + record = None + if isinstance(data, list) and data and isinstance(data[0], dict): + record = data[0] + elif isinstance(data, dict): + record = data + return _clean(str((record or {}).get("title") or "")) or "Target paper" + + +def _paper_abstract_from_dir(paper_dir: Optional[Path]) -> str: + if not paper_dir: + return "" + data = _load_json(paper_dir / "paper_metadata.json") + record = None + if isinstance(data, list) and data and isinstance(data[0], dict): + record = data[0] + elif isinstance(data, dict): + record = data + abstract = _clean(str((record or {}).get("abstract") or "")) + abstract = re.sub(r"^(abstract)\s*[:.]?\s*", "", abstract, flags=re.IGNORECASE) + abstract = re.sub(r"(?<=[.!?\)\]\"'”’])\d+$", "", abstract) + return abstract + + +def _node( + *, + nid: str, + kind: str, + short: str, + title: str, + detail: str, + step_added: int, + pulse: bool, +) -> dict[str, Any]: + meta = KIND_META.get(kind, KIND_META["citing"]) + return { + "id": nid, + "kind": kind, + "label": canvas_label(short), + "title": _clean(title) or short, + "detail": _clean(detail), + "step_added": step_added, + "pulse": pulse, + "color": meta["fill"], + "fill": meta["fill"], + "ring": meta["ring"], + "text": meta["text"], + "r": meta["r"], + "size": meta["r"], + "type_label": meta["type_label"], + } + + +def _collect_citing(paper_dir: Optional[Path], limit: int = 6) -> list[dict[str, str]]: + if not paper_dir: + return [] + out: list[dict[str, str]] = [] + seen: set[str] = set() + contrib = _load_json(paper_dir / "usage_contributions.json") or {} + items = contrib.get("contributions") if isinstance(contrib, dict) else None + if isinstance(items, list): + for item in items: + if not isinstance(item, dict): + continue + cid = str(item.get("citing_paper_id") or "").strip() + title = _clean(str(item.get("citing_title") or "Citing paper")) + if not cid or cid in seen: + continue + seen.add(cid) + out.append( + { + "id": f"citing:{cid}", + "title": title, + "detail": _clean( + str(item.get("paper_claim") or item.get("claim") or item.get("evidence_span") or "") + ), + "label_tag": item.get("label") or "", + } + ) + if len(out) >= limit: + return out + return out + + +def _collect_clusters(paper_dir: Optional[Path], limit: int = 4) -> list[dict[str, str]]: + if not paper_dir: + return [] + discovery = _load_json(paper_dir / "usage_discovery_from_contributions.json") or {} + clusters = discovery.get("clusters") if isinstance(discovery, dict) else None + out: list[dict[str, str]] = [] + if not isinstance(clusters, list): + return out + for item in clusters[:limit]: + if not isinstance(item, dict): + continue + cid = str(item.get("cluster_id") or f"C{len(out) + 1}") + title = _clean( + str(item.get("representative_claim") or item.get("cluster_title") or f"Theme {cid}") + ) + out.append( + { + "id": f"cluster:{cid}", + "cluster_id": cid, + "title": title, + "detail": _clean(str(item.get("merge_rationale") or f"{item.get('count', '')} contribution instances")), + } + ) + return out + + +def _collect_annotation(payload: Optional[dict]) -> tuple[list[dict], list[dict], list[dict]]: + claims: list[dict] = [] + ingredients: list[dict] = [] + studies: list[dict] = [] + if not isinstance(payload, dict): + return claims, ingredients, studies + for claim in (payload.get("claims") or [])[:4]: + if not isinstance(claim, dict): + continue + claim_id = str(claim.get("claim_id") or f"C{len(claims) + 1}") + title = _clean(str(claim.get("rewritten_claim") or claim.get("text") or claim_id)) + claims.append( + { + "id": f"claim:{claim_id}", + "claim_id": claim_id, + "cluster_id": str(claim.get("cluster_id") or ""), + "title": title, + "detail": _clean(str(claim.get("decision") or "")), + } + ) + for ing in (claim.get("ingredients") or [])[:3]: + if not isinstance(ing, dict): + continue + iid = str(ing.get("ingredient_id") or f"{claim_id}.I{len(ingredients) + 1}") + ann = ing.get("canonical_annotation") if isinstance(ing.get("canonical_annotation"), dict) else {} + ingredients.append( + { + "id": f"ing:{iid}", + "ingredient_id": iid, + "claim_id": claim_id, + "title": _clean(str(ing.get("ingredient") or iid)), + "detail": _clean( + str((ann or {}).get("contribution") or (ann or {}).get("rationale") or "") + ), + "role": _clean(str((ann or {}).get("role") or "")), + } + ) + g = ing.get("canonical_grounding") if isinstance(ing.get("canonical_grounding"), dict) else None + if g: + sid = str(g.get("paper_id") or g.get("ref_id") or g.get("ref_title") or iid) + studies.append( + { + "id": f"study:{sid}", + "ingredient_id": iid, + "title": _clean(str(g.get("ref_title") or g.get("ref_id") or "Prior study")), + "detail": _clean(str(g.get("ref_authors") or g.get("ref_year") or "")), + } + ) + seen: set[str] = set() + uniq: list[dict] = [] + for s in studies: + if s["id"] in seen: + continue + seen.add(s["id"]) + uniq.append(s) + return claims, ingredients, uniq[:6] + + +def build_graph_model( + *, + paper_dir: Optional[Path], + payload: Optional[dict], + visible_step: int, + pulse_step: int = 0, + caption: str = "", +) -> dict[str, Any]: + step = max(0, min(8, int(visible_step or 0))) + nodes: list[dict[str, Any]] = [] + edges: list[dict[str, Any]] = [] + + if step >= 1: + title = _paper_title_from_dir(paper_dir) + abstract = _paper_abstract_from_dir(paper_dir) + nodes.append( + _node( + nid="target", + kind="target", + short="Target", + title=title, + detail=abstract[:500], + step_added=1, + pulse=pulse_step == 1, + ) + ) + + citing = _collect_citing(paper_dir) if step >= 2 else [] + if step >= 2: + rows = citing or [ + {"id": f"citing:placeholder:{i}", "title": f"Citing paper {i+1}", "detail": "Waiting for citation artifacts…", "label_tag": ""} + for i in range(3) + ] + for i, item in enumerate(rows): + nodes.append( + _node( + nid=item["id"], + kind="citing", + short=f"Citing {i + 1}", + title=item["title"], + detail=item.get("detail") or item.get("label_tag") or "", + step_added=2, + pulse=pulse_step in {2, 3, 4, 5, 6}, + ) + ) + edges.append( + { + "id": f"e-target-{item['id']}", + "source": "target", + "target": item["id"], + "kind": "cite", + "muted": step < 5, + "pulse": pulse_step in {2, 3, 4, 5}, + "step_added": 2, + } + ) + + # Step 7: reuse themes + target contributions arrive together. + clusters = _collect_clusters(paper_dir) if step >= 7 else [] + claims, ingredients, studies = ( + _collect_annotation(payload) if step >= 7 else ([], [], []) + ) + if step >= 7 and clusters: + for item in clusters: + cid = item.get("cluster_id") or "?" + nodes.append( + _node( + nid=item["id"], + kind="cluster", + short=f"Theme {cid}", + title=item["title"], + detail=item.get("detail") or "", + step_added=7, + pulse=pulse_step == 7, + ) + ) + edges.append( + { + "id": f"e-cluster-{item['id']}", + "source": "target", + "target": item["id"], + "kind": "theme", + "pulse": pulse_step == 7, + "muted": False, + "step_added": 7, + } + ) + + if step >= 7: + for item in claims: + cid = item.get("claim_id") or "?" + nodes.append( + _node( + nid=item["id"], + kind="claim", + short=f"Claim {cid}", + title=item["title"], + detail=item.get("detail") or "", + step_added=7, + pulse=pulse_step == 7, + ) + ) + src = f"cluster:{item.get('cluster_id')}" if item.get("cluster_id") else "target" + if not any(n["id"] == src for n in nodes): + src = "target" + edges.append( + { + "id": f"e-claim-{item['id']}", + "source": src, + "target": item["id"], + "kind": "derive", + "pulse": pulse_step == 7, + "muted": False, + "step_added": 7, + } + ) + + # Step 8: enabling contributions + prior studies. + if step >= 8: + for i, item in enumerate(ingredients): + nodes.append( + _node( + nid=item["id"], + kind="ingredient", + short=f"Enable {i + 1}", + title=item["title"], + detail=" · ".join(x for x in [item.get("role") or "", item.get("detail") or ""] if x), + step_added=8, + pulse=pulse_step == 8, + ) + ) + src = f"claim:{item.get('claim_id')}" + if any(n["id"] == src for n in nodes): + edges.append( + { + "id": f"e-ing-{item['id']}", + "source": src, + "target": item["id"], + "kind": "enable", + "pulse": pulse_step == 8, + "muted": False, + "step_added": 8, + } + ) + for i, item in enumerate(studies): + nodes.append( + _node( + nid=item["id"], + kind="study", + short=f"Prior {i + 1}", + title=item["title"], + detail=item.get("detail") or "", + step_added=8, + pulse=pulse_step == 8, + ) + ) + src = f"ing:{item.get('ingredient_id')}" + if any(n["id"] == src for n in nodes): + edges.append( + { + "id": f"e-study-{item['id']}", + "source": src, + "target": item["id"], + "kind": "ground", + "pulse": pulse_step == 8, + "muted": False, + "step_added": 8, + } + ) + + if step <= 0: + phase_caption = "Waiting to run" + elif pulse_step: + phase_caption = STEP_COPY.get(pulse_step, "") + elif step >= 8: + phase_caption = "Workflow complete" + else: + phase_caption = caption or STEP_COPY.get(step, "") + + return { + "step": step, + "pulse_step": pulse_step, + "caption": phase_caption, + "nodes": nodes, + "edges": edges, + "backend": "memory", + } + + +def _graph_payload(model: dict[str, Any]) -> dict[str, Any]: + """Serialize a graph model into the payload the canvas shell consumes.""" + nodes_in = model.get("nodes") or [] + edges_in = model.get("edges") or [] + + nodes_js = [] + for n in nodes_in: + kind = str(n.get("kind") or "citing") + meta = KIND_META.get(kind, KIND_META["citing"]) + nodes_js.append( + { + "id": n["id"], + "label": n.get("label") or "", + "kind": kind, + "fill": n.get("fill") or meta["fill"], + "ring": n.get("ring") or meta["ring"], + "text": n.get("text") or meta["text"], + "r": float(n.get("r") or meta["r"]), + "pulse": bool(n.get("pulse")), + "fullTitle": n.get("title") or "", + "detail": n.get("detail") or "", + "typeLabel": n.get("type_label") or meta["type_label"], + } + ) + edges_js = [] + for e in edges_in: + edges_js.append( + { + "id": e.get("id"), + "source": e.get("source"), + "target": e.get("target"), + "pulse": bool(e.get("pulse")), + "muted": bool(e.get("muted")), + } + ) + return { + "nodes": nodes_js, + "edges": edges_js, + "kindMeta": _KIND_META_JS, + "kindOrder": _KIND_ORDER, + } + + +def courier_html(payload: dict[str, Any]) -> str: + """Tiny invisible frame that posts a graph model into the persistent shell. + + Runs in its own Streamlit component iframe, reaches the parent document, and + postMessages the model to the graph iframe — so the graph updates in place + instead of the whole component being re-emitted (which reloads it). + """ + model_str = json.dumps(payload) + embed = json.dumps(model_str).replace("" + "" + ) + + +def graph_shell_html(height: int = 520) -> str: + """Stable, model-free canvas shell. Data arrives via postMessage (incremental). + + The HTML is constant for a given height, so Streamlit reuses the same iframe + across reruns instead of remounting it. The graph is populated / grown by + ``courier_html`` messages. + """ + payload = json.dumps( + {"nodes": [], "edges": [], "kindMeta": _KIND_META_JS, "kindOrder": _KIND_ORDER} + ) + row_h = max(300, height - 30) + return f""" + + + + + + + + +
    +
    Workflow
    +
    + +
    + +
    + +
    +
    +

    +
    +
    Drag to move · scroll to zoom · click a node for details
    +
    +
    +
    + + +""" + + +def sync_and_load_graph(run_id: str, model: dict[str, Any]) -> dict[str, Any]: + """Push cumulative graph to Neo4j and read it back for rendering.""" + if not run_id or neo4j_store is None: + model = dict(model) + model["backend"] = "memory" + return model + nodes = model.get("nodes") or [] + edges = model.get("edges") or [] + ok = neo4j_store.upsert_graph(run_id, nodes, edges) + if not ok: + model = dict(model) + model["backend"] = "memory" + return model + fetched = neo4j_store.fetch_graph(run_id) + if not fetched: + model = dict(model) + model["backend"] = "neo4j-write" + return model + # Preserve caption/step and enrich fetched nodes with Mina colors. + out_nodes = [] + for n in fetched.get("nodes") or []: + kind = str(n.get("kind") or "citing") + meta = KIND_META.get(kind, KIND_META["citing"]) + out_nodes.append( + { + **n, + "type_label": meta["type_label"], + "color": n.get("color") or meta["fill"], + "fill": n.get("fill") or meta["fill"], + "ring": n.get("ring") or meta["ring"], + "text": n.get("text") or meta["text"], + "r": n.get("r") or meta["r"], + "size": n.get("size") or meta["r"], + } + ) + return { + "step": model.get("step"), + "pulse_step": model.get("pulse_step"), + "caption": model.get("caption"), + "nodes": out_nodes, + "edges": fetched.get("edges") or [], + "backend": "neo4j", + } + + +def build_synced_model( + *, + paper_dir: Optional[Path], + payload: Optional[dict], + events: list[str], + run_id: str = "", +) -> dict[str, Any]: + """Build the cumulative graph model for the given events and sync to Neo4j.""" + visible = visible_step_from_events(events) + pulse = pulse_step_from_events(events) + joined = " ".join(str(e) for e in (events or [])) + if "Pipeline completed successfully." in joined or ( + "Step 8 complete" in joined or ("annotation" in joined.lower() and "complete" in joined.lower()) + ): + if visible >= 8: + pulse = 0 + visible = 8 + + model = build_graph_model( + paper_dir=paper_dir, + payload=payload, + visible_step=visible, + pulse_step=pulse, + ) + return sync_and_load_graph(run_id or "default", model) + + +def render_graph_shell(placeholder, *, height: int = 520) -> None: + """Mount the persistent, model-free canvas shell (constant HTML → no remount).""" + html = graph_shell_html(height) + if placeholder is None: + st.iframe(html, height=height) + return + with placeholder.container(): + st.iframe(html, height=height) + + +def push_graph_update(courier_slot, model: dict[str, Any], *, height: int = 0) -> None: + """Push a model into the mounted shell via an invisible courier frame.""" + html = courier_html(_graph_payload(model)) + if courier_slot is None: + st.components.v1.html(html, height=height) + return + with courier_slot.container(): + st.components.v1.html(html, height=height) + + +def render_workflow_into( + placeholder, + *, + paper_dir: Optional[Path], + payload: Optional[dict], + events: list[str], + run_id: str = "", + height: int = 520, +) -> None: + """Back-compat single-slot render: mount shell then push once into it.""" + render_graph_shell(placeholder, height=height) + model = build_synced_model( + paper_dir=paper_dir, payload=payload, events=events, run_id=run_id + ) + push_graph_update(None, model) diff --git a/replay_traces/2208.00329/input_ids.json b/replay_traces/2208.00329/input_ids.json new file mode 100644 index 0000000000000000000000000000000000000000..128575fca81bb6cb254495d997a9ca06ccdf5f8b --- /dev/null +++ b/replay_traces/2208.00329/input_ids.json @@ -0,0 +1,7 @@ +[ + { + "id": "2208.00329", + "title": "", + "id_type": "ArXiv" + } +] \ No newline at end of file diff --git a/replay_traces/2208.00329/logs/step_01.log b/replay_traces/2208.00329/logs/step_01.log new file mode 100644 index 0000000000000000000000000000000000000000..9853126d09718bc4cf6ecb6e7a4e48d0b033fbb9 --- /dev/null +++ b/replay_traces/2208.00329/logs/step_01.log @@ -0,0 +1,18 @@ +$ /Users/dingzifeng/SciFy-UMBC/demo_2026_07/.venv/bin/python src/step_01_fetch/fetch_metadata.py --ids /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785099279_8df5ecbb/input_ids.json --outdir /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785099279_8df5ecbb/processed_papers + +/Users/dingzifeng/SciFy-UMBC/demo_2026_07/src/step_01_fetch/fetch_metadata.py:95: DeprecationWarning: The 'Search.results' method is deprecated, use 'Client.results' instead + paper = next(search.results()) +/Users/dingzifeng/SciFy-UMBC/demo_2026_07/src/step_01_fetch/fetch_metadata.py:124: DeprecationWarning: Python 3.14 will, by default, filter extracted tar archives and reject files or modify their metadata. Use the filter argument to control this behavior. + tar.extractall(path=source_dir) +[INFO] Using ID list from /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785099279_8df5ecbb/input_ids.json +[INFO] Output will be saved to /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785099279_8df5ecbb/processed_papers +[DEBUG] fetch_one_acl_id: id=2208.00329 id_type=ArXiv title_len=0 s2_key_present=yes s2_key_len=44 +[DEBUG] wrote metadata to /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785099279_8df5ecbb/processed_papers/2208.00329/paper_metadata.json +[INFO] Wrote combined BibTeX to /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785099279_8df5ecbb/processed_papers/2208.00329/references.bib +[INFO] Processed LaTeX for 2208.00329 at /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785099279_8df5ecbb/processed_papers/2208.00329/processed_main.tex +[WARN] 429 Too Many Requests → retrying in 1.50s +[WARN] 429 Too Many Requests → retrying in 2.15s +[WARN] 429 Too Many Requests → retrying in 4.24s +[WARN] 429 Too Many Requests → retrying in 8.26s +[SUCCESS] +done in 28.508277893066406 s diff --git a/replay_traces/2208.00329/logs/step_02.log b/replay_traces/2208.00329/logs/step_02.log new file mode 100644 index 0000000000000000000000000000000000000000..5111134b6b9cb2958ce2d28c7c08a97cf078edce --- /dev/null +++ b/replay_traces/2208.00329/logs/step_02.log @@ -0,0 +1,5 @@ +$ /Users/dingzifeng/SciFy-UMBC/demo_2026_07/.venv/bin/python src/step_02_mark_citations/replace_citation_markers.py --root /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785099279_8df5ecbb/processed_papers + +[INFO] Found 1 paper dirs under /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785099279_8df5ecbb/processed_papers +[OK] 2208.00329: updated 19 contexts over 20 +[SUMMARY] total_updated=19 over 20; skipped_incomplete=0 diff --git a/replay_traces/2208.00329/logs/step_03.log b/replay_traces/2208.00329/logs/step_03.log new file mode 100644 index 0000000000000000000000000000000000000000..9ed925611e51d6f98100890add087786bd7c7b07 --- /dev/null +++ b/replay_traces/2208.00329/logs/step_03.log @@ -0,0 +1,4 @@ +$ /Users/dingzifeng/SciFy-UMBC/demo_2026_07/.venv/bin/python src/step_03_usage_contexts/build_usage_contexts.py --root /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785099279_8df5ecbb/processed_papers --out-name usage_contexts.json + +[INFO] Found 1 paper dirs under /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785099279_8df5ecbb/processed_papers +[OK] 2208.00329: wrote usage_contexts.json (20 contexts from 7 citations) diff --git a/replay_traces/2208.00329/logs/step_04.log b/replay_traces/2208.00329/logs/step_04.log new file mode 100644 index 0000000000000000000000000000000000000000..fd3d3cace1475f9278dfebe00d7dfb7b21468d10 --- /dev/null +++ b/replay_traces/2208.00329/logs/step_04.log @@ -0,0 +1,6 @@ +$ /Users/dingzifeng/SciFy-UMBC/demo_2026_07/.venv/bin/python src/step_04_label_citations/label_citation_functions.py --root /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785099279_8df5ecbb/processed_papers --model-path Deep-Citation/Workspace/acl_scicite_wksp_trl/best_model.pt --model-data-dir Deep-Citation/Data --model-class-def Deep-Citation/Data/class_def.json --model-lm scibert --device cpu + +[INFO] Found 1 paper dirs under /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785099279_8df5ecbb/processed_papers +Number of data instance: 1941 +[LABELED] 2208.00329 +[SUMMARY] labeled=1, skipped=0, missing_usage=0, empty_contexts=0 diff --git a/replay_traces/2208.00329/logs/step_05.log b/replay_traces/2208.00329/logs/step_05.log new file mode 100644 index 0000000000000000000000000000000000000000..f227f68cc5cea174bbaffa9b5215f5347a9c942f --- /dev/null +++ b/replay_traces/2208.00329/logs/step_05.log @@ -0,0 +1,14 @@ +$ /Users/dingzifeng/SciFy-UMBC/demo_2026_07/.venv/bin/python src/step_05_verify_uses_extends/verify_uses_extends.py --root /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785099279_8df5ecbb/processed_papers --k 0 --batch-size 25 + +/Users/dingzifeng/SciFy-UMBC/demo_2026_07/src/common/llm_client.py:4: FutureWarning: + +All support for the `google.generativeai` package has ended. It will no longer be receiving +updates or bug fixes. Please switch to the `google.genai` package as soon as possible. +See README for more details: + +https://github.com/google-gemini/deprecated-generative-ai-python/blob/main/README.md + + import google.generativeai as genai +[INFO] Found 1 paper dirs under /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785099279_8df5ecbb/processed_papers +[VERIFIED] 2208.00329 +[SUMMARY] verified=1, skipped=0, missing_labels=0, no_candidates=0 diff --git a/replay_traces/2208.00329/logs/step_06.log b/replay_traces/2208.00329/logs/step_06.log new file mode 100644 index 0000000000000000000000000000000000000000..06437cab9de0f71279088081803081393f4b3176 --- /dev/null +++ b/replay_traces/2208.00329/logs/step_06.log @@ -0,0 +1,7 @@ +$ /Users/dingzifeng/SciFy-UMBC/demo_2026_07/.venv/bin/python src/step_06_extract_paragraphs/extract_arxiv_paragraphs.py --root /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785099279_8df5ecbb/processed_papers + +/Users/dingzifeng/SciFy-UMBC/demo_2026_07/src/step_06_extract_paragraphs/extract_arxiv_paragraphs.py:47: DeprecationWarning: Python 3.14 will, by default, filter extracted tar archives and reject files or modify their metadata. Use the filter argument to control this behavior. + tar.extractall(path) +[INFO] Found 1 paper dirs under /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785099279_8df5ecbb/processed_papers +[PROCESSED] 2208.00329 +[SUMMARY] processed=1, skipped=0, missing_usage=0, missing_verified=0, empty_citing=0 diff --git a/replay_traces/2208.00329/logs/step_07.log b/replay_traces/2208.00329/logs/step_07.log new file mode 100644 index 0000000000000000000000000000000000000000..0241045d3d7000552ae5ab3f082b765ab28e4587 --- /dev/null +++ b/replay_traces/2208.00329/logs/step_07.log @@ -0,0 +1,28 @@ +$ /Users/dingzifeng/SciFy-UMBC/demo_2026_07/.venv/bin/python src/step_07_extract_and_refine/extract_contributions_from_citations.py --root /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785099279_8df5ecbb/processed_papers + +/Users/dingzifeng/SciFy-UMBC/demo_2026_07/src/common/llm_client.py:4: FutureWarning: + +All support for the `google.generativeai` package has ended. It will no longer be receiving +updates or bug fixes. Please switch to the `google.genai` package as soon as possible. +See README for more details: + +https://github.com/google-gemini/deprecated-generative-ai-python/blob/main/README.md + + import google.generativeai as genai +[INFO] Found 1 paper dirs under /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785099279_8df5ecbb/processed_papers +[LABELED] 2208.00329 +[SUMMARY] labeled=1, skipped=0, missing_verified=0, missing_arxiv_paragraphs=0, no_confirmed=0 +$ /Users/dingzifeng/SciFy-UMBC/demo_2026_07/.venv/bin/python src/step_07_extract_and_refine/refine_and_filter_clusters_llm.py --root /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785099279_8df5ecbb/processed_papers --inplace --overwrite + +/Users/dingzifeng/SciFy-UMBC/demo_2026_07/src/common/llm_client.py:4: FutureWarning: + +All support for the `google.generativeai` package has ended. It will no longer be receiving +updates or bug fixes. Please switch to the `google.genai` package as soon as possible. +See README for more details: + +https://github.com/google-gemini/deprecated-generative-ai-python/blob/main/README.md + + import google.generativeai as genai +[INFO] Found 1 paper dirs under /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785099279_8df5ecbb/processed_papers +[REFINED] 2208.00329 +[SUMMARY] refined=1, skipped=0, missing_inputs=0, empty_clusters=0 diff --git a/replay_traces/2208.00329/processed_papers/2208.00329/paper_metadata.json b/replay_traces/2208.00329/processed_papers/2208.00329/paper_metadata.json new file mode 100644 index 0000000000000000000000000000000000000000..4c063095f98629b78116e46a1bee848dc3eb7886 --- /dev/null +++ b/replay_traces/2208.00329/processed_papers/2208.00329/paper_metadata.json @@ -0,0 +1,55 @@ +[ + { + "paperId": "e894fb15054d3bc9659060406a12dfd1055ae32e", + "externalIds": { + "DBLP": "journals/corr/abs-2208-00329", + "ArXiv": "2208.00329", + "DOI": "10.1162/tacl_a_00600", + "CorpusId": 251223896 + }, + "url": "https://www.semanticscholar.org/paper/e894fb15054d3bc9659060406a12dfd1055ae32e", + "title": "PASTA: A Dataset for Modeling PArticipant STAtes in Narratives", + "venue": "Transactions of the Association for Computational Linguistics", + "year": 2022, + "referenceCount": 59, + "citationCount": 7, + "openAccessPdf": { + "url": "https://direct.mit.edu/tacl/article-pdf/doi/10.1162/tacl_a_00600/2173956/tacl_a_00600.pdf", + "status": "GOLD", + "license": "CCBY", + "disclaimer": "Notice: Paper or abstract available at https://arxiv.org/abs/2208.00329, which is subject to the license by the author or copyright owner provided with this content. Please go to the source to verify the license and copyright information for your use." + }, + "tldr": { + "model": "tldr@v2.0.0", + "text": "This work introduces a new crowdsourced English-language, Participant States dataset, PASTA, and introduces three state-based reasoning tasks that test for the ability to infer when a state is entailed by a story, to revise a story conditioned on a counterfactual state, and to explain the most likely state change given a revised story." + }, + "publicationDate": "2022-07-31", + "authors": [ + { + "authorId": "72059959", + "name": "Sayontan Ghosh" + }, + { + "authorId": "144007901", + "name": "Mahnaz Koupaee" + }, + { + "authorId": "2254303864", + "name": "I-Sen Chen" + }, + { + "authorId": "2034063", + "name": "Francis Ferraro" + }, + { + "authorId": "1729918", + "name": "Nathanael Chambers" + }, + { + "authorId": "35217367", + "name": "Niranjan Balasubramanian" + } + ], + "abstract": "Abstract The events in a narrative are understood as a coherent whole via the underlying states of their participants. Often, these participant states are not explicitly mentioned, instead left to be inferred by the reader. A model that understands narratives should likewise infer these implicit states, and even reason about the impact of changes to these states on the narrative. To facilitate this goal, we introduce a new crowdsourced English-language, Participant States dataset, PASTA. This dataset contains inferable participant states; a counterfactual perturbation to each state; and the changes to the story that would be necessary if the counterfactual were true. We introduce three state-based reasoning tasks that test for the ability to infer when a state is entailed by a story, to revise a story conditioned on a counterfactual state, and to explain the most likely state change given a revised story. Experiments show that today\u2019s LLMs can reason about states to some degree, but there is large room for improvement, especially in problems requiring access and ability to reason with diverse types of knowledge (e.g., physical, numerical, factual).1" + } +] \ No newline at end of file diff --git a/replay_traces/2208.00329/processed_papers/2208.00329/usage_citing_paragraphs.json b/replay_traces/2208.00329/processed_papers/2208.00329/usage_citing_paragraphs.json new file mode 100644 index 0000000000000000000000000000000000000000..3f257dcc4974fde9c12f4e20b1b023b267ae7519 --- /dev/null +++ b/replay_traces/2208.00329/processed_papers/2208.00329/usage_citing_paragraphs.json @@ -0,0 +1,75 @@ +{ + "paper_id": "2208.00329", + "citing_papers": [ + { + "citing_paper_id": "ab73ca481256767addce56b20eb48ac183169e01", + "citing_title": "CoRE: Condition-based Reasoning for Identifying Outcome Variance in Complex Events", + "arxiv_id": "2506.01253", + "main_tex_file": "sections/appendix_results.tex", + "bib_files": [ + "anthology.bib", + "custom.bib" + ], + "bib_texts": { + "anthology.bib": "% Please download the latest anthology.bib from\n%\n% http://aclweb.org/anthology/anthology.bib.gz\n", + "custom.bib": " @article{Lewis1973-LEWC,\n\tauthor = {David Lewis},\n\tdoi = {10.2307/2025310},\n\tjournal = {Journal of Philosophy},\n\tnumber = {17},\n\tpages = {556--567},\n\tpublisher = {Oxford Up},\n\ttitle = {Causation},\n\tvolume = {70},\n\tyear = {1973}\n}\n@article{lewis1979,\n ISSN = {00294624, 14680068},\n URL = {http://www.jstor.org/stable/2215339},\n author = {David Lewis},\n journal = {No\u00fbs},\n number = {4},\n pages = {455--476},\n publisher = {Wiley},\n title = {Counterfactual Dependence and Time's Arrow},\n urldate = {2024-08-30},\n volume = {13},\n year = {1979}\n}\n@article{DRY,\nurl = {https://doi.org/10.1515/text.1.1981.1.3.233},\ntitle = {Sentence aspect and the movement of narrative time},\ntitle = {},\nauthor = {HELEN DRY},\npages = {233--240},\nvolume = {1},\nnumber = {3},\njournal = {Text - Interdisciplinary Journal for the Study of Discourse},\ndoi = {doi:10.1515/text.1.1981.1.3.233},\nyear = {1981},\nlastchecked = {2024-08-30}\n}\n@article{smith1999,\n ISSN = {01650157, 15730549},\n URL = {http://www.jstor.org/stable/25001754},\n author = {Carlota S. Smith},\n journal = {Linguistics and Philosophy},\n number = {5},\n pages = {479--508},\n publisher = {Springer},\n title = {Activities: States or Events?},\n urldate = {2024-08-30},\n volume = {22},\n year = {1999}\n}\n@inproceedings{friedrich-etal-2023-kind,\n title = \"A Kind Introduction to Lexical and Grammatical Aspect, with a Survey of Computational Approaches\",\n author = \"Friedrich, Annemarie and\n Xue, Nianwen and\n Palmer, Alexis\",\n editor = \"Vlachos, Andreas and\n Augenstein, Isabelle\",\n booktitle = \"Proceedings of the 17th Conference of the European Chapter of the Association for Computational Linguistics\",\n month = may,\n year = \"2023\",\n address = \"Dubrovnik, Croatia\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2023.eacl-main.44\",\n doi = \"10.18653/v1/2023.eacl-main.44\",\n pages = \"599--622\",\n abstract = \"Aspectual meaning refers to how the internal temporal structure of situations is presented. This includes whether a situation is described as a state or as an event, whether the situation is finished or ongoing, and whether it is viewed as a whole or with a focus on a particular phase. This survey gives an overview of computational approaches to modeling lexical and grammatical aspect along with intuitive explanations of the necessary linguistic concepts and terminology. In particular, we describe the concepts of stativity, telicity, habituality, perfective and imperfective, as well as influential inventories of eventuality and situation types. Aspect is a crucial component of semantics, especially for precise reporting of the temporal structure of situations, and future NLP approaches need to be able to handle and evaluate it systematically.\",\n}\n\n\n@article{smith83,\n ISSN = {00978507, 15350665},\n URL = {http://www.jstor.org/stable/413899},\n abstract = {Sentential aspect presents a situation (event, state etc.) from a particular point of view: it represents the speaker's choice of perspective on the situation. In this article, a speaker-based approach to aspect is proposed which crucially invokes abstraction, namely the idealization of different types of situations. A speaker relates an actual situation to an idealized type by using the linguistic forms associated with that type. Speakers may choose an unusual way of presenting a situation, as with a progressive stative such as I'm really loving this walk. Here a state is presented as an event. Analysis of stative sentences in English shows that they have a unique aspectual interpretation, and allows for the interpretation of progressive statives. Two components of sentential aspect are recognized. Situation aspect involves type of situation, e.g. event or state; VIEWPOINT ASPECT involves type of perspective, e.g. simple or progressive. Separating these two components leads to a unified account of simple aspect in English.},\n author = {Carlota S. Smith},\n journal = {Language},\n number = {3},\n pages = {479--501},\n publisher = {Linguistic Society of America},\n title = {A Theory of Aspectual Choice},\n urldate = {2024-08-30},\n volume = {59},\n year = {1983}\n}\n\n@inproceedings{klavans,\nauthor = {Klavans, Judith L. and Chodorow, Martin},\ntitle = {Degrees of stativity: the lexical representation of verb aspect},\nyear = {1992},\npublisher = {Association for Computational Linguistics},\naddress = {USA},\nurl = {https://doi.org/10.3115/992424.992443},\ndoi = {10.3115/992424.992443},\nabstract = {L'acquisition automatique de connaissance lexicale \\`{a} partir de larges corpus s'est essentiellement occup\\'{e}e des ph\\'{e}nom\\`{e}nes de co-occurrence, aux d\\'{e}pens des traits lexicaux inh\\'{e}rents. Nous pr\\'{e}sentons ici une m\\'{e}thodologie qui permet d'obtenir l'information s\\'{e}mantique sur l'aspect du verbe en analysant automatiquement un corpus et en appliquant des tests linguistiques \\`{a} l'aide d'une s\\'{e}rie d'outils d'analyse structurale. Lorsque ces deux t\\^{a}ches sont accomplies, nous proposons une r\\'{e}presentation de l'aspect du verbe qui associe une valeur de mesure pour les diff\\'{e}rents types d'\\'{e}v\\`{e}nements. Les mesures refl\\`{e}tent l'usage typique du verbe, et par cons\\'{e}quent une mesure de r\\'{e}sistance ou de non-r\\'{e}sistance \\`{a} la coercion dans le contexte de la phrase. Les r\\'{e}sultats que nous rapportons ici ont \\'{e}t\\'{e} obtenus de deux mani\\`{e}res: en extrayant l'information n\\'{e}cessaire \\`{a} partir du corpus \\'{e}tiquet\\'{e} de Francis and Ku\\v{c}era (1982), et en faisant tourner un analyseur syntaxique (McCord 1980, 1990) sur le corpus du Reader's Digest afin d'extraire une information plus pr\\'{e}cise sur l'usage du verbe dans le texte.},\nbooktitle = {Proceedings of the 14th Conference on Computational Linguistics - Volume 4},\npages = {1126\u20131131},\nnumpages = {6},\nlocation = {Nantes, France},\nseries = {COLING '92}\n}\n\n \n\n\n@article{setiya,\n ISSN = {00318108, 15581470},\n URL = {http://www.jstor.org/stable/3595576},\n author = {Kieran Setiya},\n journal = {The Philosophical Review},\n number = {3},\n pages = {339--393},\n publisher = {[Duke University Press, Philosophical Review]},\n title = {Explaining Action},\n urldate = {2024-05-13},\n volume = {112},\n year = {2003}\n}\n@article{LILJEHOLM202178,\ntitle = {Agency and goal-directed choice},\njournal = {Current Opinion in Behavioral Sciences},\nvolume = {41},\npages = {78-84},\nyear = {2021},\nnote = {Value based decision-making},\nissn = {2352-1546},\ndoi = {https://doi.org/10.1016/j.cobeha.2021.04.004},\nurl = {https://www.sciencedirect.com/science/article/pii/S2352154621000899},\nauthor = {Mimi Liljeholm},\nabstract = {In philosophy, agency is construed in terms of desires and means-end beliefs as reasons for actions. Psychological theories of goal-directed behavior provide a formal bridge between objective contingency knowledge and subjective beliefs, missing from such accounts. In this review, I argue that, because they conflate contingency and reward, theories of goal-directed behavior are nonetheless themselves unsuitable as accounts of agency. I then review behavioral and neuroscientific data suggesting that the recently proposed construct of instrumental divergence might serve as a normative and descriptive psychological index of agency that constrains and motivates goal-directed choice.}\n}\n@article{Grice1975,\n title={Logic and Conversation},\n author={ Grice, H. P.},\n journal={Syntax and Semantics},\n year={1975},\n volume={3},\n pages={41-58}\n}\n@article{vendler,\n ISSN = {00318108, 15581470},\n URL = {http://www.jstor.org/stable/2182371},\n author = {Zeno Vendler},\n journal = {The Philosophical Review},\n number = {2},\n pages = {143--160},\n publisher = {[Duke University Press, Philosophical Review]},\n title = {Verbs and Times},\n urldate = {2024-08-30},\n volume = {66},\n year = {1957}\n}\n@article{davidson,\n ISSN = {00294624, 14680068},\n URL = {http://www.jstor.org/stable/2214381},\n author = {Donald Davidson},\n journal = {No\u00fbs},\n number = {4},\n pages = {335--349},\n publisher = {Wiley},\n title = {Eternal vs. Ephemeral Events},\n urldate = {2024-08-30},\n volume = {5},\n year = {1971}\n}\n@article{dowty,\n ISSN = {01650157, 15730549},\n URL = {http://www.jstor.org/stable/25000028},\n author = {David R. Dowty},\n journal = {Linguistics and Philosophy},\n number = {1},\n pages = {45--77},\n publisher = {Springer},\n title = {Toward a Semantic Analysis of Verb Aspect and the English 'Imperfective' Progressive},\n urldate = {2024-08-30},\n volume = {1},\n year = {1977}\n}\n@inproceedings{10.1145/3531146.3533088,\nauthor = {Weidinger, Laura and Uesato, Jonathan and Rauh, Maribeth and Griffin, Conor and Huang, Po-Sen and Mellor, John and Glaese, Amelia and Cheng, Myra and Balle, Borja and Kasirzadeh, Atoosa and Biles, Courtney and Brown, Sasha and Kenton, Zac and Hawkins, Will and Stepleton, Tom and Birhane, Abeba and Hendricks, Lisa Anne and Rimell, Laura and Isaac, William and Haas, Julia and Legassick, Sean and Irving, Geoffrey and Gabriel, Iason},\ntitle = {Taxonomy of Risks posed by Language Models},\nyear = {2022},\nisbn = {9781450393522},\npublisher = {Association for Computing Machinery},\naddress = {New York, NY, USA},\nurl = {https://doi.org/10.1145/3531146.3533088},\ndoi = {10.1145/3531146.3533088},\nabstract = {Responsible innovation on large-scale Language Models (LMs) requires foresight into and in-depth understanding of the risks these models may pose. This paper develops a comprehensive taxonomy of ethical and social risks associated with LMs. We identify twenty-one risks, drawing on expertise and literature from computer science, linguistics, and the social sciences. We situate these risks in our taxonomy of six risk areas: I. Discrimination, Hate speech and Exclusion, II. Information Hazards, III. Misinformation Harms, IV. Malicious Uses, V. Human-Computer Interaction Harms, and VI. Environmental and Socioeconomic harms. For risks that have already been observed in LMs, the causal mechanism leading to harm, evidence of the risk, and approaches to risk mitigation are discussed. We further describe and analyse risks that have not yet been observed but are anticipated based on assessments of other language technologies, and situate these in the same taxonomy. We underscore that it is the responsibility of organizations to engage with the mitigations we discuss throughout the paper. We close by highlighting challenges and directions for further research on risk evaluation and mitigation with the goal of ensuring that language models are developed responsibly.},\nbooktitle = {Proceedings of the 2022 ACM Conference on Fairness, Accountability, and Transparency},\npages = {214\u2013229},\nnumpages = {16},\nkeywords = {language models, responsible AI, responsible innovation, risk assessment, technology risks},\nlocation = {Seoul, Republic of Korea},\nseries = {FAccT '22}\n}\n\n \n\n\n@article{GENNARI2003B27,\ntitle = {Processing correlates of lexical semantic complexity},\njournal = {Cognition},\nvolume = {89},\nnumber = {1},\npages = {B27-B41},\nyear = {2003},\nissn = {0010-0277},\ndoi = {https://doi.org/10.1016/S0010-0277(03)00069-6},\nurl = {https://www.sciencedirect.com/science/article/pii/S0010027703000696},\nauthor = {Silvia Gennari and David Poeppel},\nkeywords = {Verb meaning, Semantic complexity, Verb processing, Event structure},\nabstract = {This paper explores how verb meanings that differ in semantic complexity are processed and represented. In particular, we compare eventive verbs, which denote causally structured events, with stative verbs, which denote facts without causal structure. We predicted that the conceptually more complex eventive verbs should take longer to process than stative verbs. Two experiments, a lexical decision task and a self-paced reading study, confirmed this prediction. The findings suggest that (a) semantic complexity is reflected in processing time, (b) processing verb meanings involves activating properties of the event structure beyond participants' roles, and (c) more generally, lexical event structures, which subsume thematic roles, may mediate between syntactic knowledge and semantic interpretation.}\n}\n@book{verkuyl,\nauthor = {Verkuyl, Henk},\nyear = {1993},\nmonth = {10},\npages = {},\ntitle = {A Theory of Aspectuality},\nisbn = {0 521 44362 8},\ndoi = {10.1017/CBO9780511597848} \n}\n@article{Pustejovsky1991TheSO,\n title={The syntax of event structure},\n author={James Pustejovsky},\n journal={Cognition},\n year={1991},\n volume={41},\n pages={47-81},\n url={https://api.semanticscholar.org/CorpusID:16966452}\n}\n@inproceedings{gardner-etal-2020-evaluating,\n title = \"Evaluating Models{'} Local Decision Boundaries via Contrast Sets\",\n author = \"Gardner, Matt and\n Artzi, Yoav and\n Basmov, Victoria and\n Berant, Jonathan and\n Bogin, Ben and\n Chen, Sihao and\n Dasigi, Pradeep and\n Dua, Dheeru and\n Elazar, Yanai and\n Gottumukkala, Ananth and\n Gupta, Nitish and\n Hajishirzi, Hannaneh and\n Ilharco, Gabriel and\n Khashabi, Daniel and\n Lin, Kevin and\n Liu, Jiangming and\n Liu, Nelson F. and\n Mulcaire, Phoebe and\n Ning, Qiang and\n Singh, Sameer and\n Smith, Noah A. and\n Subramanian, Sanjay and\n Tsarfaty, Reut and\n Wallace, Eric and\n Zhang, Ally and\n Zhou, Ben\",\n editor = \"Cohn, Trevor and\n He, Yulan and\n Liu, Yang\",\n booktitle = \"Findings of the Association for Computational Linguistics: EMNLP 2020\",\n month = nov,\n year = \"2020\",\n address = \"Online\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2020.findings-emnlp.117\",\n doi = \"10.18653/v1/2020.findings-emnlp.117\",\n pages = \"1307--1323\",\n abstract = \"Standard test sets for supervised learning evaluate in-distribution generalization. Unfortunately, when a dataset has systematic gaps (e.g., annotation artifacts), these evaluations are misleading: a model can learn simple decision rules that perform well on the test set but do not capture the abilities a dataset is intended to test. We propose a more rigorous annotation paradigm for NLP that helps to close systematic gaps in the test data. In particular, after a dataset is constructed, we recommend that the dataset authors manually perturb the test instances in small but meaningful ways that (typically) change the gold label, creating contrast sets. Contrast sets provide a local view of a model{'}s decision boundary, which can be used to more accurately evaluate a model{'}s true linguistic capabilities. We demonstrate the efficacy of contrast sets by creating them for 10 diverse NLP datasets (e.g., DROP reading comprehension, UD parsing, and IMDb sentiment analysis). Although our contrast sets are not explicitly adversarial, model performance is significantly lower on them than on the original test sets{---}up to 25{\\%} in some cases. We release our contrast sets as new evaluation benchmarks and encourage future dataset construction efforts to follow similar annotation processes.\",\n}\n\n\n@inproceedings{ross-etal-2021-explaining,\n title = \"Explaining {NLP} Models via Minimal Contrastive Editing ({M}i{CE})\",\n author = \"Ross, Alexis and\n Marasovi{\\'c}, Ana and\n Peters, Matthew\",\n editor = \"Zong, Chengqing and\n Xia, Fei and\n Li, Wenjie and\n Navigli, Roberto\",\n booktitle = \"Findings of the Association for Computational Linguistics: ACL-IJCNLP 2021\",\n month = aug,\n year = \"2021\",\n address = \"Online\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2021.findings-acl.336\",\n doi = \"10.18653/v1/2021.findings-acl.336\",\n pages = \"3840--3852\",\n}\n\n\n@misc{kaushik2020learning,\n title={Learning the Difference that Makes a Difference with Counterfactually-Augmented Data}, \n author={Divyansh Kaushik and Eduard Hovy and Zachary C. Lipton},\n year={2020},\n eprint={1909.12434},\n archivePrefix={arXiv},\n primaryClass={cs.CL}\n}\n@inproceedings{qin-etal-2019-counterfactual,\n title = \"Counterfactual Story Reasoning and Generation\",\n author = \"Qin, Lianhui and\n Bosselut, Antoine and\n Holtzman, Ari and\n Bhagavatula, Chandra and\n Clark, Elizabeth and\n Choi, Yejin\",\n editor = \"Inui, Kentaro and\n Jiang, Jing and\n Ng, Vincent and\n Wan, Xiaojun\",\n booktitle = \"Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP)\",\n month = nov,\n year = \"2019\",\n address = \"Hong Kong, China\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/D19-1509\",\n doi = \"10.18653/v1/D19-1509\",\n pages = \"5043--5053\",\n abstract = \"Counterfactual reasoning requires predicting how alternative events, contrary to what actually happened, might have resulted in different outcomes. Despite being considered a necessary component of AI-complete systems, few resources have been developed for evaluating counterfactual reasoning in narratives. In this paper, we propose Counterfactual Story Rewriting: given an original story and an intervening counterfactual event, the task is to minimally revise the story to make it compatible with the given counterfactual event. Solving this task will require deep understanding of causal narrative chains and counterfactual invariance, and integration of such story reasoning capabilities into conditional language generation models. We present TIMETRAVEL, a new dataset of 29,849 counterfactual rewritings, each with the original story, a counterfactual event, and human-generated revision of the original story compatible with the counterfactual event. Additionally, we include 81,407 counterfactual {``}branches{''} without a rewritten storyline to support future work on semi- or un-supervised approaches to counterfactual story rewriting. Finally, we evaluate the counterfactual rewriting capacities of several competitive baselines based on pretrained language models, and assess whether common overlap and model-based automatic metrics for text generation correlate well with human scores for counterfactual rewriting.\",\n}\n\n\n@inproceedings{paranjape-etal-2021-prompting,\n title = \"Prompting Contrastive Explanations for Commonsense Reasoning Tasks\",\n author = \"Paranjape, Bhargavi and\n Michael, Julian and\n Ghazvininejad, Marjan and\n Hajishirzi, Hannaneh and\n Zettlemoyer, Luke\",\n editor = \"Zong, Chengqing and\n Xia, Fei and\n Li, Wenjie and\n Navigli, Roberto\",\n booktitle = \"Findings of the Association for Computational Linguistics: ACL-IJCNLP 2021\",\n month = aug,\n year = \"2021\",\n address = \"Online\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2021.findings-acl.366\",\n doi = \"10.18653/v1/2021.findings-acl.366\",\n pages = \"4179--4192\",\n}\n\n\n@misc{kolluru2020imojie,\n title={IMoJIE: Iterative Memory-Based Joint Open Information Extraction}, \n author={Keshav Kolluru and Samarth Aggarwal and Vipul Rathore and Mausam and Soumen Chakrabarti},\n year={2020},\n eprint={2005.08178},\n archivePrefix={arXiv},\n primaryClass={cs.CL}\n}\n@inproceedings{banko,\nauthor = {Banko, Michele and Cafarella, Michael J. and Soderland, Stephen and Broadhead, Matt and Etzioni, Oren},\ntitle = {Open information extraction from the web},\nyear = {2007},\npublisher = {Morgan Kaufmann Publishers Inc.},\naddress = {San Francisco, CA, USA},\nabstract = {Traditionally, Information Extraction (IE) has focused on satisfying precise, narrow, pre-specified requests from small homogeneous corpora (e.g., extract the location and time of seminars from a set of announcements). Shifting to a new domain requires the user to name the target relations and to manually create new extraction rules or hand-tag new training examples. This manual labor scales linearly with the number of target relations.This paper introduces Open IE (OIE), a new extraction paradigm where the system makes a single data-driven pass over its corpus and extracts a large set of relational tuples without requiring any human input. The paper also introduces TEXTRUNNER, a fully implemented, highly scalable OIE system where the tuples are assigned a probability and indexed to support efficient extraction and exploration via user queries.We report on experiments over a 9,000,000 Web page corpus that compare TEXTRUNNER with KNOWITALL, a state-of-the-art Web IE system. TEXTRUNNER achieves an error reduction of 33\\% on a comparable set of extractions. Furthermore, in the amount of time it takes KNOWITALL to perform extraction for a handful of pre-specified relations, TEXTRUNNER extracts a far broader set of facts reflecting orders of magnitude more relations, discovered on the fly. We report statistics on TEXTRUNNER's 11,000,000 highest probability tuples, and show that they contain over 1,000,000 concrete facts and over 6,500,000 more abstract assertions.},\nbooktitle = {Proceedings of the 20th International Joint Conference on Artifical Intelligence},\npages = {2670\u20132676},\nnumpages = {7},\nlocation = {Hyderabad, India},\nseries = {IJCAI'07}\n}\n\n \n\n\n@article{ghosh-etal-2023-pasta,\n title = \"{PASTA}: A Dataset for Modeling {PA}rticipant {STA}tes in Narratives\",\n author = \"Ghosh, Sayontan and\n Koupaee, Mahnaz and\n Chen, Isabella and\n Ferraro, Francis and\n Chambers, Nathanael and\n Balasubramanian, Niranjan\",\n journal = \"Transactions of the Association for Computational Linguistics\",\n volume = \"11\",\n year = \"2023\",\n address = \"Cambridge, MA\",\n publisher = \"MIT Press\",\n url = \"https://aclanthology.org/2023.tacl-1.73\",\n doi = \"10.1162/tacl_a_00600\",\n pages = \"1283--1300\",\n abstract = \"The events in a narrative are understood as a coherent whole via the underlying states of their participants. Often, these participant states are not explicitly mentioned, instead left to be inferred by the reader. A model that understands narratives should likewise infer these implicit states, and even reason about the impact of changes to these states on the narrative. To facilitate this goal, we introduce a new crowdsourced English-language, Participant States dataset, PASTA. This dataset contains inferable participant states; a counterfactual perturbation to each state; and the changes to the story that would be necessary if the counterfactual were true. We introduce three state-based reasoning tasks that test for the ability to infer when a state is entailed by a story, to revise a story conditioned on a counterfactual state, and to explain the most likely state change given a revised story. Experiments show that today{'}s LLMs can reason about states to some degree, but there is large room for improvement, especially in problems requiring access and ability to reason with diverse types of knowledge (e.g., physical, numerical, factual).1\",\n}\n\n\n\n\n@inproceedings{metheniti-etal-2022-time,\n title = \"About Time: Do Transformers Learn Temporal Verbal Aspect?\",\n author = \"Metheniti, Eleni and\n Van De Cruys, Tim and\n Hathout, Nabil\",\n editor = \"Chersoni, Emmanuele and\n Hollenstein, Nora and\n Jacobs, Cassandra and\n Oseki, Yohei and\n Pr{\\'e}vot, Laurent and\n Santus, Enrico\",\n booktitle = \"Proceedings of the Workshop on Cognitive Modeling and Computational Linguistics\",\n month = may,\n year = \"2022\",\n address = \"Dublin, Ireland\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2022.cmcl-1.10\",\n doi = \"10.18653/v1/2022.cmcl-1.10\",\n pages = \"88--101\",\n abstract = \"Aspect is a linguistic concept that describes how an action, event, or state of a verb phrase is situated in time. In this paper, we explore whether different transformer models are capable of identifying aspectual features. We focus on two specific aspectual features: telicity and duration. Telicity marks whether the verb{'}s action or state has an endpoint or not (telic/atelic), and duration denotes whether a verb expresses an action (dynamic) or a state (stative). These features are integral to the interpretation of natural language, but also hard to annotate and identify with NLP methods. We perform experiments in English and French, and our results show that transformer models adequately capture information on telicity and duration in their vectors, even in their non-finetuned forms, but are somewhat biased with regard to verb tense and word order.\",\n}\n\n\n@article{chambers/tacl_a_00182,\n author = {Chambers, Nathanael and Cassidy, Taylor and McDowell, Bill and Bethard, Steven},\n title = \"{Dense Event Ordering with a Multi-Pass Architecture}\",\n journal = {Transactions of the Association for Computational Linguistics},\n volume = {2},\n pages = {273-284},\n year = {2014},\n month = {10},\n issn = {2307-387X},\n doi = {10.1162/tacl_a_00182},\n url = {https://doi.org/10.1162/tacl\\_a\\_00182},\n eprint = {https://direct.mit.edu/tacl/article-pdf/doi/10.1162/tacl\\_a\\_00182/1566921/tacl\\_a\\_00182.pdf},\n}\n@article{Yap2009AspectualAI,\n title={Aspectual asymmetries in the mental representation of events: Role of lexical and grammatical aspect},\n author={Foong Ha Yap and Patrick Chun Kau Chu and Emily Sze Man Yiu and S Wong and Stella Wing Man Kwan and Stephen Matthews and Li Hai Tan and Ping Li and Yasuhiro Shirai},\n journal={Memory \\& Cognition},\n year={2009},\n volume={37},\n pages={587-595},\n url={https://api.semanticscholar.org/CorpusID:3011013}\n}\n@article{GIV\u00d3N,\nurl = {https://doi.org/10.1515/ling.1992.30.1.5},\ntitle = {The grammar of referential coherence as mental processing instructions},\nauthor = {T. Giv\u00f3n},\npages = {5--56},\nvolume = {30},\nnumber = {1},\njournal = {Linguistics},\ndoi = {doi:10.1515/ling.1992.30.1.5},\nyear = {1992},\nlastchecked = {2024-09-13}\n}\n@inproceedings{li-etal-2024-prompting,\n title = \"Prompting Large Language Models for Counterfactual Generation: An Empirical Study\",\n author = \"Li, Yongqi and\n Xu, Mayi and\n Miao, Xin and\n Zhou, Shen and\n Qian, Tieyun\",\n editor = \"Calzolari, Nicoletta and\n Kan, Min-Yen and\n Hoste, Veronique and\n Lenci, Alessandro and\n Sakti, Sakriani and\n Xue, Nianwen\",\n booktitle = \"Proceedings of the 2024 Joint International Conference on Computational Linguistics, Language Resources and Evaluation (LREC-COLING 2024)\",\n month = may,\n year = \"2024\",\n address = \"Torino, Italia\",\n publisher = \"ELRA and ICCL\",\n url = \"https://aclanthology.org/2024.lrec-main.1156/\",\n pages = \"13201--13221\",\n abstract = \"Large language models (LLMs) have made remarkable progress in a wide range of natural language understanding and generation tasks. However, their ability to generate counterfactuals has not been examined systematically. To bridge this gap, we present a comprehensive evaluation framework on various types of NLU tasks, which covers all key factors in determining LLMs' capability of generating counterfactuals. Based on this framework, we 1) investigate the strengths and weaknesses of LLMs as the counterfactual generator, and 2) disclose the factors that affect LLMs when generating counterfactuals, including both the intrinsic properties of LLMs and prompt designing. The results show that, though LLMs are promising in most cases, they face challenges in complex tasks like RE since they are bounded by task-specific performance, entity constraints, and inherent selection bias. We also find that alignment techniques, e.g., instruction-tuning and reinforcement learning from human feedback, may potentially enhance the counterfactual generation ability of LLMs. On the contrary, simply increasing the parameter size does not yield the desired improvements. Besides, from the perspective of prompt designing, task guidelines unsurprisingly play an important role. However, the chain-of-thought approach does not always help due to inconsistency issues.\"\n}\n\n\n@inproceedings{fang-etal-2025-counterfactual,\n title = \"Counterfactual Debating with Preset Stances for Hallucination Elimination of {LLM}s\",\n author = \"Fang, Yi and\n Li, Moxin and\n Wang, Wenjie and\n Hui, Lin and\n Feng, Fuli\",\n editor = \"Rambow, Owen and\n Wanner, Leo and\n Apidianaki, Marianna and\n Al-Khalifa, Hend and\n Eugenio, Barbara Di and\n Schockaert, Steven\",\n booktitle = \"Proceedings of the 31st International Conference on Computational Linguistics\",\n month = jan,\n year = \"2025\",\n address = \"Abu Dhabi, UAE\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2025.coling-main.703/\",\n pages = \"10554--10568\",\n abstract = \"Large Language Models (LLMs) excel in various natural language processing tasks but struggle with hallucination issues. Existing solutions have considered utilizing LLMs' inherent reasoning abilities to alleviate hallucination, such as self-correction and diverse sampling methods. However, these methods often overtrust LLMs' initial answers due to inherent biases. The key to alleviating this issue lies in overriding LLMs' inherent biases for answer inspection. To this end, we propose a CounterFactual Multi-Agent Debate (CFMAD) framework. CFMAD presets the stances of LLMs to override their inherent biases by compelling LLMs to generate justifications for a predetermined answer`s correctness. The LLMs with different predetermined stances are engaged with a skeptical critic for counterfactual debate on the rationality of generated justifications. Finally, the debate process is evaluated by a third-party judge to determine the final answer. Extensive experiments on four datasets of three tasks demonstrate the superiority of CFMAD over existing methods.\"\n}\n\n\n@article{madden,\nauthor = {Madden, Carol and Zwaan, Rolf},\nyear = {2003},\nmonth = {08},\npages = {663-72},\ntitle = {How does verb aspect constrain event representation?},\nvolume = {31},\njournal = {Memory & cognition},\ndoi = {10.3758/BF03196106}\n}\n\n @inproceedings{zhou-etal-2023-navigating,\n title = \"Navigating the Grey Area: How Expressions of Uncertainty and Overconfidence Affect Language Models\",\n author = \"Zhou, Kaitlyn and\n Jurafsky, Dan and\n Hashimoto, Tatsunori\",\n editor = \"Bouamor, Houda and\n Pino, Juan and\n Bali, Kalika\",\n booktitle = \"Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing\",\n month = dec,\n year = \"2023\",\n address = \"Singapore\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2023.emnlp-main.335\",\n doi = \"10.18653/v1/2023.emnlp-main.335\",\n pages = \"5506--5524\",\n abstract = \"The increased deployment of LMs for real-world tasks involving knowledge and facts makes it important to understand model epistemology: what LMs think they know, and how their attitudes toward that knowledge are affected by language use in their inputs. Here, we study an aspect of model epistemology: how epistemic markers of certainty, uncertainty, or evidentiality like {``}I{'}m sure it{'}s{''}, {``}I think it{'}s{''}, or {``}Wikipedia says it{'}s{''} affect models, and whether they contribute to model failures. We develop a typology of epistemic markers and inject 50 markers into prompts for question answering. We find that LMs are highly sensitive to epistemic markers in prompts, with accuracies varying more than 80{\\%}. Surprisingly, we find that expressions of high certainty result in a 7{\\%} decrease in accuracy as compared to low certainty expressions; similarly, factive verbs hurt performance, while evidentials benefit performance. Our analysis of a popular pretraining dataset shows that these markers of uncertainty are associated with answers on question-answering websites, while markers of certainty are associated with questions. These associations may suggest that the behavior of LMs is based on mimicking observed language use, rather than truly reflecting epistemic uncertainty.\",\n}\n\n@inproceedings{li-etal-2023-coannotating,\n title = \"{C}o{A}nnotating: Uncertainty-Guided Work Allocation between Human and Large Language Models for Data Annotation\",\n author = \"Li, Minzhi and\n Shi, Taiwei and\n Ziems, Caleb and\n Kan, Min-Yen and\n Chen, Nancy and\n Liu, Zhengyuan and\n Yang, Diyi\",\n editor = \"Bouamor, Houda and\n Pino, Juan and\n Bali, Kalika\",\n booktitle = \"Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing\",\n month = dec,\n year = \"2023\",\n address = \"Singapore\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2023.emnlp-main.92\",\n doi = \"10.18653/v1/2023.emnlp-main.92\",\n pages = \"1487--1505\",\n abstract = \"Annotated data plays a critical role in Natural Language Processing (NLP) in training models and evaluating their performance. Given recent developments in Large Language Models (LLMs), models such as ChatGPT demonstrate zero-shot capability on many text-annotation tasks, comparable with or even exceeding human annotators. Such LLMs can serve as alternatives for manual annotation, due to lower costs and higher scalability. However, limited work has leveraged LLMs as complementary annotators, nor explored how annotation work is best allocated among humans and LLMs to achieve both quality and cost objectives. We propose CoAnnotating, a novel paradigm for Human-LLM co-annotation of unstructured texts at scale. Under this framework, we utilize uncertainty to estimate LLMs{'} annotation capability. Our empirical study shows CoAnnotating to be an effective means to allocate work from results on different datasets, with up to 21{\\%} performance improvement over random baseline. For code implementation, see https://github.com/SALT-NLP/CoAnnotating.\",\n}\n\n@article{pavlick-kwiatkowski-2019-inherent,\n title = \"Inherent Disagreements in Human Textual Inferences\",\n author = \"Pavlick, Ellie and\n Kwiatkowski, Tom\",\n editor = \"Lee, Lillian and\n Johnson, Mark and\n Roark, Brian and\n Nenkova, Ani\",\n journal = \"Transactions of the Association for Computational Linguistics\",\n volume = \"7\",\n year = \"2019\",\n address = \"Cambridge, MA\",\n publisher = \"MIT Press\",\n url = \"https://aclanthology.org/Q19-1043\",\n doi = \"10.1162/tacl_a_00293\",\n pages = \"677--694\",\n abstract = \"We analyze human{'}s disagreements about the validity of natural language inferences. We show that, very often, disagreements are not dismissible as annotation {``}noise{''}, but rather persist as we collect more ratings and as we vary the amount of context provided to raters. We further show that the type of uncertainty captured by current state-of-the-art models for natural language inference is not reflective of the type of uncertainty present in human disagreements. We discuss implications of our results in relation to the recognizing textual entailment (RTE)/natural language inference (NLI) task. We argue for a refined evaluation objective that requires models to explicitly capture the full distribution of plausible human judgments.\",\n}\n@inproceedings{lee-etal-2021-modeling,\n title = \"Modeling Human Mental States with an Entity-based Narrative Graph\",\n author = \"Lee, I-Ta and\n Pacheco, Maria Leonor and\n Goldwasser, Dan\",\n editor = \"Toutanova, Kristina and\n Rumshisky, Anna and\n Zettlemoyer, Luke and\n Hakkani-Tur, Dilek and\n Beltagy, Iz and\n Bethard, Steven and\n Cotterell, Ryan and\n Chakraborty, Tanmoy and\n Zhou, Yichao\",\n booktitle = \"Proceedings of the 2021 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies\",\n month = jun,\n year = \"2021\",\n address = \"Online\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2021.naacl-main.391\",\n doi = \"10.18653/v1/2021.naacl-main.391\",\n pages = \"4916--4926\",\n abstract = \"Understanding narrative text requires capturing characters{'} motivations, goals, and mental states. This paper proposes an Entity-based Narrative Graph (ENG) to model the internal- states of characters in a story. We explicitly model entities, their interactions and the context in which they appear, and learn rich representations for them. We experiment with different task-adaptive pre-training objectives, in-domain training, and symbolic inference to capture dependencies between different decisions in the output space. We evaluate our model on two narrative understanding tasks: predicting character mental states, and desire fulfillment, and conduct a qualitative analysis.\",\n}\n@inproceedings{castricato-etal-2021-towards,\n title = \"Towards a Model-Theoretic View of Narratives\",\n author = \"Castricato, Louis and\n Biderman, Stella and\n Thue, David and\n Cardona-Rivera, Rogelio\",\n editor = \"Akoury, Nader and\n Brahman, Faeze and\n Chaturvedi, Snigdha and\n Clark, Elizabeth and\n Iyyer, Mohit and\n Martin, Lara J.\",\n booktitle = \"Proceedings of the Third Workshop on Narrative Understanding\",\n month = jun,\n year = \"2021\",\n address = \"Virtual\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2021.nuse-1.10\",\n doi = \"10.18653/v1/2021.nuse-1.10\",\n pages = \"95--104\",\n abstract = \"In this paper, we propose the beginnings of a formal framework for modeling narrative \\textit{qua} narrative. Our framework affords the ability to discuss key qualities of stories and their communication, including the flow of information from a Narrator to a Reader, the evolution of a Reader{'}s story model over time, and Reader uncertainty. We demonstrate its applicability to computational narratology by giving explicit algorithms for measuring the accuracy with which information was conveyed to the Reader, along with two novel measurements of story coherence.\",\n}\n\n\n@inproceedings{sun-etal-2024-event,\n title = \"Event Causality Is Key to Computational Story Understanding\",\n author = \"Sun, Yidan and\n Chao, Qin and\n Li, Boyang\",\n editor = \"Duh, Kevin and\n Gomez, Helena and\n Bethard, Steven\",\n booktitle = \"Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers)\",\n month = jun,\n year = \"2024\",\n address = \"Mexico City, Mexico\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2024.naacl-long.191\",\n doi = \"10.18653/v1/2024.naacl-long.191\",\n pages = \"3493--3511\",\n abstract = \"Cognitive science and symbolic AI research suggest that event causality provides vital information for story understanding. However, machine learning systems for story understanding rarely employ event causality, partially due to the lack of methods that reliably identify open-world causal event relations. Leveraging recent progress in large language models, we present the first method for event causality identification that leads to material improvements in computational story understanding. Our technique sets a new state of the art on the COPES dataset (Wang et al., 2023c) for causal event relation identification. Further, in the downstream story quality evaluation task, the identified causal relations lead to 3.6-16.6{\\%} relative improvement on correlation with human ratings. In the multimodal story video-text alignment task, we attain 4.1-10.9{\\%} increase on Clip Accuracy and 4.2-13.5{\\%} increase on Sentence IoU. The findings indicate substantial untapped potential for event causality in computational story understanding. The codebase is at https://github.com/insundaycathy/Event-Causality-Extraction.\",\n}\n\n\n@inproceedings{brei-etal-2024-returning,\n title = \"Returning to the Start: Generating Narratives with Related Endpoints\",\n author = \"Brei, Anneliese and\n Zhao, Chao and\n Chaturvedi, Snigdha\",\n editor = \"Duh, Kevin and\n Gomez, Helena and\n Bethard, Steven\",\n booktitle = \"Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 2: Short Papers)\",\n month = jun,\n year = \"2024\",\n address = \"Mexico City, Mexico\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2024.naacl-short.10\",\n doi = \"10.18653/v1/2024.naacl-short.10\",\n pages = \"101--112\",\n abstract = \"Human writers often *bookend* their writing with ending sentences that relate back to the beginning sentences in order to compose a satisfying narrative that {``}closes the loop.{''} Motivated by this observation, we propose RENarGen, a controllable story-generation paradigm that generates narratives by ensuring the first and last sentences are related and then infilling the middle sentences. Our contributions include an initial exploration of how various methods of bookending from Narratology affect language modeling for stories. Automatic and human evaluations indicate RENarGen produces better stories with more narrative closure than current autoregressive models.\",\n}\n\n@article{moens-steedman-1988-temporal,\n title = \"Temporal Ontology and Temporal Reference\",\n author = \"Moens, Marc and\n Steedman, Mark\",\n journal = \"Computational Linguistics\",\n volume = \"14\",\n number = \"2\",\n year = \"1988\",\n url = \"https://aclanthology.org/J88-2003/\",\n pages = \"15--28\"\n}\n@article{siegel-mckeown-2000-learning,\n title = \"Learning Methods to Combine Linguistic Indicators:Improving Aspectual Classification and Revealing Linguistic Insights\",\n author = \"Siegel, Eric V. and\n McKeown, Kathleen R.\",\n journal = \"Computational Linguistics\",\n volume = \"26\",\n number = \"4\",\n year = \"2000\",\n address = \"Cambridge, MA\",\n publisher = \"MIT Press\",\n url = \"https://aclanthology.org/J00-4004/\",\n pages = \"595--627\"\n}\n@inproceedings{alikhani-stone-2019-caption,\n title = \"{\\textquotedblleft}Caption{\\textquotedblright} as a Coherence Relation: Evidence and Implications\",\n author = \"Alikhani, Malihe and\n Stone, Matthew\",\n editor = \"Bernardi, Raffaella and\n Fernandez, Raquel and\n Gella, Spandana and\n Kafle, Kushal and\n Kanan, Christopher and\n Lee, Stefan and\n Nabi, Moin\",\n booktitle = \"Proceedings of the Second Workshop on Shortcomings in Vision and Language\",\n month = jun,\n year = \"2019\",\n address = \"Minneapolis, Minnesota\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/W19-1806/\",\n doi = \"10.18653/v1/W19-1806\",\n pages = \"58--67\",\n abstract = \"We study verbs in image{--}text corpora, contrasting \\textit{caption} corpora, where texts are explicitly written to characterize image content, with \\textit{depiction} corpora, where texts and images may stand in more general relations. Captions show a distinctively limited distribution of verbs, with strong preferences for specific tense, aspect, lexical aspect, and semantic field. These limitations, which appear in data elicited by a range of methods, restrict the utility of caption corpora to inform image retrieval, multimodal document generation, and perceptually-grounded semantic models. We suggest that these limitations reflect the discourse constraints in play when subjects write texts to accompany imagery, so we argue that future development of image{--}text corpora should work to increase the diversity of event descriptions, while looking explicitly at the different ways text and imagery can be coherently related.\"\n}\n@inproceedings{zhang-etal-2020-reasoning,\n title = \"Reasoning about Goals, Steps, and Temporal Ordering with {W}iki{H}ow\",\n author = \"Zhang, Li and\n Lyu, Qing and\n Callison-Burch, Chris\",\n editor = \"Webber, Bonnie and\n Cohn, Trevor and\n He, Yulan and\n Liu, Yang\",\n booktitle = \"Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing (EMNLP)\",\n month = nov,\n year = \"2020\",\n address = \"Online\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2020.emnlp-main.374/\",\n doi = \"10.18653/v1/2020.emnlp-main.374\",\n pages = \"4630--4639\",\n abstract = \"We propose a suite of reasoning tasks on two types of relations between procedural events: goal-step relations ({\\textquotedblleft}learn poses{\\textquotedblright} is a step in the larger goal of {\\textquotedblleft}doing yoga{\\textquotedblright}) and step-step temporal relations ({\\textquotedblleft}buy a yoga mat{\\textquotedblright} typically precedes {\\textquotedblleft}learn poses{\\textquotedblright}). We introduce a dataset targeting these two relations based on wikiHow, a website of instructional how-to articles. Our human-validated test set serves as a reliable benchmark for common-sense inference, with a gap of about 10{\\%} to 20{\\%} between the performance of state-of-the-art transformer models and human performance. Our automatically-generated training set allows models to effectively transfer to out-of-domain tasks requiring knowledge of procedural events, with greatly improved performances on SWAG, Snips, and Story Cloze Test in zero- and few-shot settings.\"\n}\n\n\n@inproceedings{jiang-riloff-2018-learning,\n title = \"Learning Prototypical Goal Activities for Locations\",\n author = \"Jiang, Tianyu and\n Riloff, Ellen\",\n editor = \"Gurevych, Iryna and\n Miyao, Yusuke\",\n booktitle = \"Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)\",\n month = jul,\n year = \"2018\",\n address = \"Melbourne, Australia\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/P18-1120/\",\n doi = \"10.18653/v1/P18-1120\",\n pages = \"1297--1307\",\n abstract = \"People go to different places to engage in activities that reflect their goals. For example, people go to restaurants to eat, libraries to study, and churches to pray. We refer to an activity that represents a common reason why people typically go to a location as a prototypical goal activity (goal-act). Our research aims to learn goal-acts for specific locations using a text corpus and semi-supervised learning. First, we extract activities and locations that co-occur in goal-oriented syntactic patterns. Next, we create an activity profile matrix and apply a semi-supervised label propagation algorithm to iteratively revise the activity strengths for different locations using a small set of labeled data. We show that this approach outperforms several baseline methods when judged against goal-acts identified by human annotators.\"\n}\n\n@inproceedings{rahimtoroghi-etal-2017-modelling,\n title = \"Modelling Protagonist Goals and Desires in First-Person Narrative\",\n author = \"Rahimtoroghi, Elahe and\n Wu, Jiaqi and\n Wang, Ruimin and\n Anand, Pranav and\n Walker, Marilyn\",\n editor = \"Jokinen, Kristiina and\n Stede, Manfred and\n DeVault, David and\n Louis, Annie\",\n booktitle = \"Proceedings of the 18th Annual {SIG}dial Meeting on Discourse and Dialogue\",\n month = aug,\n year = \"2017\",\n address = {Saarbr{\\\"u}cken, Germany},\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/W17-5543/\",\n doi = \"10.18653/v1/W17-5543\",\n pages = \"360--369\",\n abstract = \"Many genres of natural language text are narratively structured, a testament to our predilection for organizing our experiences as narratives. There is broad consensus that understanding a narrative requires identifying and tracking the goals and desires of the characters and their narrative outcomes. However, to date, there has been limited work on computational models for this problem. We introduce a new dataset, DesireDB, which includes gold-standard labels for identifying statements of desire, textual evidence for desire fulfillment, and annotations for whether the stated desire is fulfilled given the evidence in the narrative context. We report experiments on tracking desire fulfillment using different methods, and show that LSTM Skip-Thought model achieves F-measure of 0.7 on our corpus.\"\n}\n\n\n@inproceedings{bellos-etal-2024-large,\n title = \"Can Large Language Models Reason About Goal-Oriented Tasks?\",\n author = \"Bellos, Filippos and\n Li, Yayuan and\n Liu, Wuao and\n Corso, Jason\",\n editor = \"Miceli-Barone, Antonio Valerio and\n Barez, Fazl and\n Cohen, Shay and\n Voita, Elena and\n Germann, Ulrich and\n Lukasik, Michal\",\n booktitle = \"Proceedings of the First edition of the Workshop on the Scaling Behavior of Large Language Models (SCALE-LLM 2024)\",\n month = mar,\n year = \"2024\",\n address = \"St. Julian{'}s, Malta\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2024.scalellm-1.3/\",\n pages = \"24--34\",\n abstract = \"Most adults can complete a sequence of steps to achieve a certain goal, such as making a sandwich or repairing a bicycle tire. In completing these goal-oriented tasks, or simply tasks in this paper, one must use sequential reasoning to understand the relationship between the sequence of steps and the goal. LLMs have shown impressive capabilities across various natural language understanding tasks. However, prior work has mainlyfocused on logical reasoning tasks (e.g. arithmetic, commonsense QA); how well LLMs can perform on more complex reasoning tasks like sequential reasoning is not clear. In this paper, we address this gap and conduct a comprehensive evaluation of how well LLMs are able to conduct this reasoning for tasks and how they scale w.r.t multiple dimensions(e.g. adaptive prompting strategies, number of in-context examples, varying complexity of the sequential task). Our findings reveal that while Chain of Thought (CoT) prompting can significantly enhance LLMs' sequential reasoning in certain scenarios, it can also be detrimental in others, whereas Tree of Thoughts (ToT) reasoning is less effective for this type of task. Additionally, we discover that an increase in model size or in-context examples does not consistently lead to improved performance.\"\n}\n\n\n@inproceedings{kober-etal-2020-aspectuality,\n title = \"Aspectuality Across Genre: A Distributional Semantics Approach\",\n author = \"Kober, Thomas and\n Alikhani, Malihe and\n Stone, Matthew and\n Steedman, Mark\",\n editor = \"Scott, Donia and\n Bel, Nuria and\n Zong, Chengqing\",\n booktitle = \"Proceedings of the 28th International Conference on Computational Linguistics\",\n month = dec,\n year = \"2020\",\n address = \"Barcelona, Spain (Online)\",\n publisher = \"International Committee on Computational Linguistics\",\n url = \"https://aclanthology.org/2020.coling-main.401/\",\n doi = \"10.18653/v1/2020.coling-main.401\",\n pages = \"4546--4562\",\n abstract = \"The interpretation of the lexical aspect of verbs in English plays a crucial role in tasks such as recognizing textual entailment and learning discourse-level inferences. We show that two elementary dimensions of aspectual class, states vs. events, and telic vs. atelic events, can be modelled effectively with distributional semantics. We find that a verb`s local context is most indicative of its aspectual class, and we demonstrate that closed class words tend to be stronger discriminating contexts than content words. Our approach outperforms previous work on three datasets. Further, we present a new dataset of human-human conversations annotated with lexical aspects and present experiments that show the correlation of telicity with genre and discourse goals.\"\n}\n\n@article{wei,\n author = {Jason Wei and\n Xuezhi Wang and\n Dale Schuurmans and\n Maarten Bosma and\n Ed H. Chi and\n Quoc Le and\n Denny Zhou},\n title = {Chain of Thought Prompting Elicits Reasoning in Large Language Models},\n journal = {CoRR},\n volume = {abs/2201.11903},\n year = {2022},\n url = {https://arxiv.org/abs/2201.11903},\n eprinttype = {arXiv},\n eprint = {2201.11903},\n timestamp = {Fri, 22 Apr 2022 16:06:31 +0200},\n biburl = {https://dblp.org/rec/journals/corr/abs-2201-11903.bib},\n bibsource = {dblp computer science bibliography, https://dblp.org}\n}\n\nw3c valid html last updated on 20\n\n@inproceedings{alabdulkarim-etal-2021-automatic,\n title = \"Automatic Story Generation: Challenges and Attempts\",\n author = \"Alabdulkarim, Amal and\n Li, Siyan and\n Peng, Xiangyu\",\n editor = \"Akoury, Nader and\n Brahman, Faeze and\n Chaturvedi, Snigdha and\n Clark, Elizabeth and\n Iyyer, Mohit and\n Martin, Lara J.\",\n booktitle = \"Proceedings of the Third Workshop on Narrative Understanding\",\n month = jun,\n year = \"2021\",\n address = \"Virtual\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2021.nuse-1.8\",\n doi = \"10.18653/v1/2021.nuse-1.8\",\n pages = \"72--83\",\n abstract = \"Automated storytelling has long captured the attention of researchers for the ubiquity of narratives in everyday life. The best human-crafted stories exhibit coherent plot, strong characters, and adherence to genres, attributes that current states-of-the-art still struggle to produce, even using transformer architectures. In this paper, we analyze works in story generation that utilize machine learning approaches to (1) address story generation controllability, (2) incorporate commonsense knowledge, (3) infer reasonable character actions, and (4) generate creative language.\",\n}\n@inproceedings{prus-etal-2024-human,\n title = \"Human Temporal Inferences Go Beyond Aspectual Class\",\n author = \"Pru{\\'s}, Katarzyna and\n Steedman, Mark and\n Lopez, Adam\",\n editor = \"Graham, Yvette and\n Purver, Matthew\",\n booktitle = \"Proceedings of the 18th Conference of the European Chapter of the Association for Computational Linguistics (Volume 1: Long Papers)\",\n month = mar,\n year = \"2024\",\n address = \"St. Julian{'}s, Malta\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2024.eacl-long.115\",\n pages = \"1913--1923\",\n abstract = \"Past work in NLP has proposed the task of classifying English verb phrases into situation aspect categories, assuming that these categories play an important role in tasks requiring temporal reasoning. We investigate this assumption by gathering crowd-sourced judgements about aspectual entailments from non-expert, native English participants. The results suggest that aspectual class alone is not sufficient to explain the response patterns of the participants. We propose that looking at scenarios which can feasibly accompany an action description contributes towards a better explanation of the participants{'} answers. A further experiment using GPT-3.5 shows that its outputs follow different patterns than human answers, suggesting that such conceivable scenarios cannot be fully accounted for in the language alone. We release our dataset to support further research.\",\n}\n\n@inproceedings{piper-etal-2021-narrative,\n title = \"Narrative Theory for Computational Narrative Understanding\",\n author = \"Piper, Andrew and\n So, Richard Jean and\n Bamman, David\",\n editor = \"Moens, Marie-Francine and\n Huang, Xuanjing and\n Specia, Lucia and\n Yih, Scott Wen-tau\",\n booktitle = \"Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing\",\n month = nov,\n year = \"2021\",\n address = \"Online and Punta Cana, Dominican Republic\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2021.emnlp-main.26\",\n doi = \"10.18653/v1/2021.emnlp-main.26\",\n pages = \"298--311\",\n abstract = \"Over the past decade, the field of natural language processing has developed a wide array of computational methods for reasoning about narrative, including summarization, commonsense inference, and event detection. While this work has brought an important empirical lens for examining narrative, it is by and large divorced from the large body of theoretical work on narrative within the humanities, social and cognitive sciences. In this position paper, we introduce the dominant theoretical frameworks to the NLP community, situate current research in NLP within distinct narratological traditions, and argue that linking computational work in NLP to theory opens up a range of new empirical questions that would both help advance our understanding of narrative and open up new practical applications.\",\n}\n\ninproceedings{huang-huang-2021-semantic,\n title = \"Semantic Frame Forecast\",\n author = \"Huang, Chieh-Yang and\n Huang, Ting-Hao\",\n booktitle = \"Proceedings of the 2021 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies\",\n month = jun,\n year = \"2021\",\n address = \"Online\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://www.aclweb.org/anthology/2021.naacl-main.215\",\n pages = \"2702--2713\",\n abstract = \"This paper introduces Semantic Frame Forecast, a task that predicts the semantic frames that will occur in the next 10, 100, or even 1,000 sentences in a running story. Prior work focused on predicting the immediate future of a story, such as one to a few sentences ahead. However, when novelists write long stories, generating a few sentences is not enough to help them gain high-level insight to develop the follow-up story. In this paper, we formulate a long story as a sequence of {``}story blocks,{''} where each block contains a fixed number of sentences (e.g., 10, 100, or 200). This formulation allows us to predict the follow-up story arc beyond the scope of a few sentences. We represent a story block using the term frequencies (TF) of semantic frames in it, normalized by each frame{'}s inverse document frequency (IDF). We conduct semantic frame forecast experiments on 4,794 books from the Bookcorpus and 7,962 scientific abstracts from CODA-19, with block sizes ranging from 5 to 1,000 sentences. The results show that automated models can forecast the follow-up story blocks better than the random, prior, and replay baselines, indicating the feasibility of the task. We also learn that the models using the frame representation as features outperform all the existing approaches when the block size is over 150 sentences. The human evaluation also shows that the proposed frame representation, when visualized as word clouds, is comprehensible, representative, and specific to humans.\",\n}\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n@article{\ndoi:10.1073/pnas.1912444117,\nauthor = {Jevin D. West and Carl T. Bergstrom },\ntitle = {Misinformation in and about science},\njournal = {Proceedings of the National Academy of Sciences},\nvolume = {118},\nnumber = {15},\npages = {e1912444117},\nyear = {2021},\ndoi = {10.1073/pnas.1912444117},\nURL = {https://www.pnas.org/doi/abs/10.1073/pnas.1912444117},\neprint = {https://www.pnas.org/doi/pdf/10.1073/pnas.1912444117},\nabstract = {Humans learn about the world by collectively acquiring information, filtering it, and sharing what we know. Misinformation undermines this process. The repercussions are extensive. Without reliable and accurate sources of information, we cannot hope to halt climate change, make reasoned democratic decisions, or control a global pandemic. Most analyses of misinformation focus on popular and social media, but the scientific enterprise faces a parallel set of problems\u2014from hype and hyperbole to publication bias and citation misdirection, predatory publishing, and filter bubbles. In this perspective, we highlight these parallels and discuss future research directions and interventions.}}\n\ninproceedings{zhao-etal-2024-uncommonsense,\n title = \"{UN}commonsense Reasoning: Abductive Reasoning about Uncommon Situations\",\n author = \"Zhao, Wenting and\n Chiu, Justin and\n Hwang, Jena and\n Brahman, Faeze and\n Hessel, Jack and\n Choudhury, Sanjiban and\n Choi, Yejin and\n Li, Xiang and\n Suhr, Alane\",\n editor = \"Duh, Kevin and\n Gomez, Helena and\n Bethard, Steven\",\n booktitle = \"Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers)\",\n month = jun,\n year = \"2024\",\n address = \"Mexico City, Mexico\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2024.naacl-long.469\",\n doi = \"10.18653/v1/2024.naacl-long.469\",\n pages = \"8487--8505\",\n abstract = \"Language technologies that accurately model the dynamics of events must perform commonsense reasoning. Existing work evaluating commonsense reasoning focuses on making inferences about common, everyday situations. To instead investigate the ability to model unusual, unexpected, and unlikely situations, we explore the task of uncommonsense abductive reasoning. Given a piece of context with an unexpected outcome, this task requires reasoning abductively to generate an explanation that makes the unexpected outcome more likely in the context. To this end, we curate and release a new English language corpus called UNcommonsense. We characterize the performance differences between human explainers and the best-performing large language models, finding that model-enhanced human-written explanations achieve the highest quality by trading off between specificity and diversity. Finally, we experiment with several imitation learning algorithms to train open and accessible language models on this task. When compared with the vanilla supervised fine-tuning approach, these methods consistently reduce lose rates on both common and uncommonsense abductive reasoning judged by human evaluators.\",\n}\n@inproceedings{vallurupalli-etal-2024-saga,\n title = \"{SAGA}: A Participant-specific Examination of Story Alternatives and Goal Applicability for a Deeper Understanding of Complex Events\",\n author = \"Vallurupalli, Sai and\n Erk, Katrin and\n Ferraro, Francis\",\n editor = \"Ku, Lun-Wei and\n Martins, Andre and\n Srikumar, Vivek\",\n booktitle = \"Findings of the Association for Computational Linguistics ACL 2024\",\n month = aug,\n year = \"2024\",\n address = \"Bangkok, Thailand and virtual meeting\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2024.findings-acl.910\",\n pages = \"15396--15420\",\n abstract = \"Interpreting and assessing goal driven actions is vital to understanding and reasoning over complex events. It is important to be able to acquire the knowledge needed for this understanding, though doing so is challenging. We argue that such knowledge can be elicited through a participant achievement lens. We analyze a complex event in a narrative according to the intended achievements of the participants in that narrative, the likely future actions of the participants, and the likelihood of goal success. We collect 6.3K high quality goal and action annotations reflecting our proposed participant achievement lens, with an average weighted Fleiss-Kappa IAA of 80{\\%}. Our collection contains annotated alternate versions of each narrative. These alternate versions vary minimally from the {``}original{''} story, but can license drastically different inferences. Our findings suggest that while modern large language models can reflect some of the goal-based knowledge we study, they find it challenging to fully capture the design and intent behind concerted actions, even when the model pretraining included the data from which we extracted the goal knowledge. We show that smaller models fine-tuned on our dataset can achieve performance surpassing larger models.\",\n}\n\n\n@article{\ndoi:10.1073/pnas.1914085117,\nauthor = {Michael F. Dahlstrom },\ntitle = {The narrative truth about scientific misinformation},\njournal = {Proceedings of the National Academy of Sciences},\nvolume = {118},\nnumber = {15},\npages = {e1914085117},\nyear = {2021},\ndoi = {10.1073/pnas.1914085117},\nURL = {https://www.pnas.org/doi/abs/10.1073/pnas.1914085117},\neprint = {https://www.pnas.org/doi/pdf/10.1073/pnas.1914085117},\nabstract = {Science and storytelling mean different things when they speak of truth. This difference leads some to blame storytelling for presenting a distorted view of science and contributing to misinformation. Yet others celebrate storytelling as a way to engage audiences and share accurate scientific information. This review disentangles the complexities of how storytelling intersects with scientific misinformation. Storytelling is the act of sharing a narrative, and science and narrative represent two distinct ways of constructing reality. Where science searches for broad patterns that capture general truths about the world, narratives search for connections through human experience that assign meaning and value to reality. I explore how these contrasting conceptions of truth manifest across different contexts to either promote or counter scientific misinformation. I also identify gaps in the literature and identify promising future areas of research. Even with their differences, the underlying purpose of both science and narrative seeks to make sense of the world and find our place within it. While narrative can indeed lead to scientific misinformation, narrative can also help science counter misinformation by providing meaning to reality that incorporates accurate science knowledge into human experience.}}\n\n\n@inproceedings{vladika-matthes-2023-scientific,\n title = \"Scientific Fact-Checking: A Survey of Resources and Approaches\",\n author = \"Vladika, Juraj and\n Matthes, Florian\",\n editor = \"Rogers, Anna and\n Boyd-Graber, Jordan and\n Okazaki, Naoaki\",\n booktitle = \"Findings of the Association for Computational Linguistics: ACL 2023\",\n month = jul,\n year = \"2023\",\n address = \"Toronto, Canada\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2023.findings-acl.387\",\n doi = \"10.18653/v1/2023.findings-acl.387\",\n pages = \"6215--6230\",\n abstract = \"The task of fact-checking deals with assessing the veracity of factual claims based on credible evidence and background knowledge. In particular, scientific fact-checking is the variation of the task concerned with verifying claims rooted in scientific knowledge. This task has received significant attention due to the growing importance of scientific and health discussions on online platforms. Automated scientific fact-checking methods based on NLP can help combat the spread of misinformation, assist researchers in knowledge discovery, and help individuals understand new scientific breakthroughs. In this paper, we present a comprehensive survey of existing research in this emerging field and its related tasks. We provide a task description, discuss the construction process of existing datasets, and analyze proposed models and approaches. Based on our findings, we identify intriguing challenges and outline potential future directions to advance the field.\",\n}\n@inproceedings{ijcai2021p609,\n title = {If Only We Had Better Counterfactual Explanations: Five Key Deficits to Rectify in the Evaluation of Counterfactual XAI Techniques},\n author = {Keane, Mark T. and Kenny, Eoin M. and Delaney, Eoin and Smyth, Barry},\n booktitle = {Proceedings of the Thirtieth International Joint Conference on\n Artificial Intelligence, {IJCAI-21}},\n publisher = {International Joint Conferences on Artificial Intelligence Organization},\n editor = {Zhi-Hua Zhou},\n pages = {4466--4474},\n year = {2021},\n month = {8},\n note = {Survey Track},\n doi = {10.24963/ijcai.2021/609},\n url = {https://doi.org/10.24963/ijcai.2021/609},\n}\n@inproceedings{schlichtkrull,\n author = {Schlichtkrull, Michael and Guo, Zhijiang and Vlachos, Andreas},\n booktitle = {Advances in Neural Information Processing Systems},\n editor = {A. Oh and T. Naumann and A. Globerson and K. Saenko and M. Hardt and S. Levine},\n pages = {65128--65167},\n publisher = {Curran Associates, Inc.},\n title = {AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web},\n url = {https://proceedings.neurips.cc/paper_files/paper/2023/file/cd86a30526cd1aff61d6f89f107634e4-Paper-Datasets_and_Benchmarks.pdf},\n volume = {36},\n year = {2023}\n}\n\n@inproceedings{vlachos-riedel-2014-fact,\n title = \"Fact Checking: Task definition and dataset construction\",\n author = \"Vlachos, Andreas and\n Riedel, Sebastian\",\n editor = \"Danescu-Niculescu-Mizil, Cristian and\n Eisenstein, Jacob and\n McKeown, Kathleen and\n Smith, Noah A.\",\n booktitle = \"Proceedings of the {ACL} 2014 Workshop on Language Technologies and Computational Social Science\",\n month = jun,\n year = \"2014\",\n address = \"Baltimore, MD, USA\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/W14-2508\",\n doi = \"10.3115/v1/W14-2508\",\n pages = \"18--22\",\n}\n@inproceedings{abdalla2023makes,\n title={What Makes Sentences Semantically Related: A Textual Relatedness Dataset and Empirical Study},\n author={Abdalla, Mohamed and Vishnubhotla, Krishnapriya and Mohammad, Saif M.},\n year={2023},\n address = {Dubrovnik, Croatia},\n publisher = \"Association for Computational Linguistics\",\n booktitle = \"Proceedings of the 17th Conference of the European Chapter of the Association for Computational Linguistics: Main Volume\"\n}\n@inproceedings{sap-etal-2019-social,\n title = \"Social {IQ}a: Commonsense Reasoning about Social Interactions\",\n author = \"Sap, Maarten and\n Rashkin, Hannah and\n Chen, Derek and\n Le Bras, Ronan and\n Choi, Yejin\",\n editor = \"Inui, Kentaro and\n Jiang, Jing and\n Ng, Vincent and\n Wan, Xiaojun\",\n booktitle = \"Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP)\",\n month = nov,\n year = \"2019\",\n address = \"Hong Kong, China\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/D19-1454\",\n doi = \"10.18653/v1/D19-1454\",\n pages = \"4463--4473\",\n abstract = \"We introduce Social IQa, the first large-scale benchmark for commonsense reasoning about social situations. Social IQa contains 38,000 multiple choice questions for probing emotional and social intelligence in a variety of everyday situations (e.g., Q: {``}Jordan wanted to tell Tracy a secret, so Jordan leaned towards Tracy. Why did Jordan do this?{''} A: {``}Make sure no one else could hear{''}). Through crowdsourcing, we collect commonsense questions along with correct and incorrect answers about social interactions, using a new framework that mitigates stylistic artifacts in incorrect answers by asking workers to provide the right answer to a different but related question. Empirical results show that our benchmark is challenging for existing question-answering models based on pretrained language models, compared to human performance ({\\textgreater}20{\\%} gap). Notably, we further establish Social IQa as a resource for transfer learning of commonsense knowledge, achieving state-of-the-art performance on multiple commonsense reasoning tasks (Winograd Schemas, COPA).\",\n}\n\n@article{wineburg_mcgrew,\nauthor = {Sam Wineburg and Sarah McGrew},\ntitle ={Lateral Reading and the Nature of Expertise: Reading Less and Learning More When Evaluating Digital Information},\n\njournal = {Teachers College Record},\nvolume = {121},\nnumber = {11},\npages = {1-40},\nyear = {2019},\ndoi = {10.1177/016146811912101102},\nURL = { https://doi.org/10.1177/016146811912101102},\neprint = { https://doi.org/10.1177/016146811912101102}\n,\nabstract = { Background/ContextThe Internet has democratized access to information but in so doing has opened the floodgates to misinformation, fake news, and rank propaganda masquerading as dispassionate analysis. Despite mounting attention to the problem of online misinformation and growing agreement that digital literacy efforts are important, prior research offers few concrete ideas about what skilled evaluations look like.Purpose/Objective/Research Question/Focus of StudyOur purpose in this study was to seek out those who are skilled in online evaluations in order to understand how their strategies and approaches to evaluating digital content might inform educational efforts. We sampled 45 experienced users of the Internet: 10 Ph.D. historians, 10 professional fact checkers, and 25 Stanford University undergraduates. Analysis focused on the strategies participants used to evaluate online information and arrive at judgments of credibility.Research DesignIn this expert/novice study, participants thought aloud as they evaluated live websites and searched for information on social and political issues such as bullying, minimum wage, and teacher tenure. We analyze and present findings from three of the tasks participants completed.Findings/ResultsHistorians and students often fell victim to easily manipulated features of websites, such as official-looking logos and domain names. They read vertically, staying within a website to evaluate its reliability. In contrast, fact checkers read laterally, leaving a site after a quick scan and opening up new browser tabs in order to judge the credibility of the original site. Compared to the other groups, fact checkers arrived at more warranted conclusions in a fraction of the time.Conclusions/RecommendationsWe draw on insights gleaned from the fact checkers\u2019 practices to examine current curricular approaches to teaching web credibility as well as to suggest alternatives. }\n}\n@article{wineburg, title={Historical problem solving: A study of the cognitive processes used in the evaluation of documentary and pictorial evidence.}, author={Wineburg, S. S.}, year={1991},journal={Journal of Educational Psychology}, volume={83}, number={1}, pages={73\u201387}, url={https://doi.org/10.1037/0022-0663.83.1.73}\n}\n@inproceedings{vallurupalli-etal-2022-poque,\n title = \"{POQ}ue: Asking Participant-specific Outcome Questions for a Deeper Understanding of Complex Events\",\n author = \"Vallurupalli, Sai and\n Ghosh, Sayontan and\n Erk, Katrin and\n Balasubramanian, Niranjan and\n Ferraro, Francis\",\n editor = \"Goldberg, Yoav and\n Kozareva, Zornitsa and\n Zhang, Yue\",\n booktitle = \"Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing\",\n month = dec,\n year = \"2022\",\n address = \"Abu Dhabi, United Arab Emirates\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2022.emnlp-main.594\",\n doi = \"10.18653/v1/2022.emnlp-main.594\",\n pages = \"8674--8697\",\n abstract = \"Knowledge about outcomes is critical for complex event understanding but is hard to acquire.We show that by pre-identifying a participant in a complex event, crowdworkers are ableto (1) infer the collective impact of salient events that make up the situation, (2) annotate the volitional engagement of participants in causing the situation, and (3) ground theoutcome of the situation in state changes of the participants. By creating a multi-step interface and a careful quality control strategy, we collect a high quality annotated dataset of8K short newswire narratives and ROCStories with high inter-annotator agreement (0.74-0.96weighted Fleiss Kappa). Our dataset, POQUe (Participant Outcome Questions), enables theexploration and development of models that address multiple aspects of semantic understanding. Experimentally, we show that current language models lag behind human performance in subtle ways through our task formulations that target abstract and specific comprehension of a complex event, its outcome, and a participant{'}s influence over the event culmination.\",\n}\n\n@inproceedings{rudinger-etal-2020-thinking,\n title = \"Thinking Like a Skeptic: Defeasible Inference in Natural Language\",\n author = \"Rudinger, Rachel and\n Shwartz, Vered and\n Hwang, Jena D. and\n Bhagavatula, Chandra and\n Forbes, Maxwell and\n Le Bras, Ronan and\n Smith, Noah A. and\n Choi, Yejin\",\n editor = \"Cohn, Trevor and\n He, Yulan and\n Liu, Yang\",\n booktitle = \"Findings of the Association for Computational Linguistics: EMNLP 2020\",\n month = nov,\n year = \"2020\",\n address = \"Online\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2020.findings-emnlp.418/\",\n doi = \"10.18653/v1/2020.findings-emnlp.418\",\n pages = \"4661--4675\",\n abstract = \"Defeasible inference is a mode of reasoning in which an inference (X is a bird, therefore X flies) may be weakened or overturned in light of new evidence (X is a penguin). Though long recognized in classical AI and philosophy, defeasible inference has not been extensively studied in the context of contemporary data-driven research on natural language inference and commonsense reasoning. We introduce Defeasible NLI (abbreviated $\\delta$-NLI), a dataset for defeasible inference in natural language. Defeasible NLI contains extensions to three existing inference datasets covering diverse modes of reasoning: common sense, natural language inference, and social norms. From Defeasible NLI, we develop both a classification and generation task for defeasible inference, and demonstrate that the generation task is much more challenging. Despite lagging human performance, however, generative models trained on this data are capable of writing sentences that weaken or strengthen a specified inference up to 68{\\%} of the time.\"\n}\n\n\n@inproceedings{chandra2020a,\n title = {Abductive Commonsense Reasoning},\n year = {2020},\n journal = {ICLR},\n author = {Bhagavatula, Chandra and Bras, Ronan Le and Malaviya, Chaitanya and Sakaguchi, Keisuke and Holtzman, Ari and Rashkin, Hannah and Downey, Doug and Yih, Wen-tau and Choi, Yejin},\n booktitle = \"Proceedings of the\n\t8th International Conference on Learning Representations, ICLR 2020 - Addis Ababa, Ethiopia\"\n}\n@article{CSIBRA200760,\ntitle = {\u2018Obsessed with goals\u2019: Functions and mechanisms of teleological interpretation of actions in humans},\njournal = {Acta Psychologica},\nvolume = {124},\nnumber = {1},\npages = {60-78},\nyear = {2007},\nnote = {Becoming an Intentional Agent: Early Development of Action Interpretation and Action Control},\nissn = {0001-6918},\ndoi = {https://doi.org/10.1016/j.actpsy.2006.09.007},\nurl = {https://www.sciencedirect.com/science/article/pii/S0001691806001235},\nauthor = {Gergely Csibra and Gy\u00f6rgy Gergely},\nkeywords = {Infancy, Goal attribution},\nabstract = {Humans show a strong and early inclination to interpret observed behaviours of others as goal-directed actions. We identify two main epistemic functions that this \u2018teleological obsession\u2019 serves: on-line prediction and social learning. We show how teleological action interpretations can serve these functions by drawing on two kinds of inference (\u2018action-to-goal\u2019 or \u2018goal-to-action\u2019), and argue that both types of teleological inference constitute inverse problems that can only be solved by further assumptions. We pinpoint the assumptions that the three currently proposed mechanisms of goal attribution (action-effect associations, simulation procedures, and teleological reasoning) imply, and contrast them with the functions they are supposed to fulfil. We argue that while action-effect associations and simulation procedures are generally well suited to serve on-line action monitoring and prediction, social learning of new means actions and artefact functions requires the inferential productivity of teleological reasoning.}\n}\n@article{Wineburg2016EvaluatingIT,\n title={Evaluating information: The cornerstone of civic online reasoning},\n author={Sam Wineburg and Sarah McGrew},\n journal={Stanford History Education Group,\nRobert R. McCormick Foundation\n},\n year={2016},\n url={https://api.semanticscholar.org/CorpusID:264193561}\n}\n@article{CHARLES_2000, title={Contextual correlates of meaning}, volume={21}, DOI={10.1017/S0142716400004057}, number={4}, journal={Applied Psycholinguistics}, author={Charles, Walter G.}, year={2000}, pages={505\u2013524}}\n@article{miller,\nauthor = {George A. Miller and Walter G. Charles},\ntitle = {Contextual correlates of semantic similarity},\njournal = {Language and Cognitive Processes},\nvolume = {6},\nnumber = {1},\npages = {1--28},\nyear = {1991},\npublisher = {Routledge},\ndoi = {10.1080/01690969108406936},\nURL = {https://doi.org/10.1080/01690969108406936},\neprint = {https://doi.org/10.1080/01690969108406936}\n}\n@article{jackson,\n author = {Jackson, Rebecca L. and Hoffman, Paul and Pobric, Gorana and Lambon Ralph, Matthew A.},\n title = \"{The Nature and Neural Correlates of Semantic Association versus Conceptual Similarity}\",\n journal = {Cerebral Cortex},\n volume = {25},\n number = {11},\n pages = {4319-4333},\n year = {2015},\n month = {01},\n abstract = \"{The ability to represent concepts and the relationships between them is critical to human cognition. How does the brain code relationships between items that share basic conceptual properties (e.g., dog and wolf) while simultaneously representing associative links between dissimilar items that co-occur in particular contexts (e.g., dog and bone)? To clarify the neural bases of these semantic components in neurologically intact participants, both types of semantic relationship were investigated in an fMRI study optimized for anterior temporal lobe (ATL) coverage. The clear principal finding was that the same core semantic network (ATL, superior temporal sulcus, ventral prefrontal cortex) was equivalently engaged when participants made semantic judgments on the basis of association or conceptual similarity. Direct comparisons revealed small, weaker differences for conceptual similarity \\> associative decisions (e.g., inferior prefrontal cortex) and associative \\> conceptual similarity (e.g., ventral parietal cortex) which appear to reflect graded differences in task difficulty. Indeed, once reaction time was entered as a covariate into the analysis, no associative versus category differences remained. The paper concludes with a discussion of how categorical/feature-based and associative relationships might be represented within a single, unified semantic system.}\",\n issn = {1047-3211},\n doi = {10.1093/cercor/bhv003},\n url = {https://doi.org/10.1093/cercor/bhv003},\n eprint = {https://academic.oup.com/cercor/article-pdf/25/11/4319/14102545/bhv003.pdf},\n}\n@inproceedings{Thorne18Fever,\n author = {Thorne, James and Vlachos, Andreas and Christodoulopoulos, Christos and Mittal, Arpit},\n title = {{FEVER}: a Large-scale Dataset for Fact Extraction and {VERification}},\n booktitle = {NAACL-HLT},\n year = {2018}\n}\n@inproceedings{reimers-2019-sentence-bert,\n title = \"Sentence-BERT: Sentence Embeddings using Siamese BERT-Networks\",\n author = \"Reimers, Nils and Gurevych, Iryna\",\n booktitle = \"Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing\",\n month = \"11\",\n year = \"2019\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://arxiv.org/abs/1908.10084\",\n}\n@inproceedings{zhao-etal-2024-uncommonsense,\n title = \"{UN}commonsense Reasoning: Abductive Reasoning about Uncommon Situations\",\n author = \"Zhao, Wenting and\n Chiu, Justin and\n Hwang, Jena and\n Brahman, Faeze and\n Hessel, Jack and\n Choudhury, Sanjiban and\n Choi, Yejin and\n Li, Xiang and\n Suhr, Alane\",\n editor = \"Duh, Kevin and\n Gomez, Helena and\n Bethard, Steven\",\n booktitle = \"Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers)\",\n month = jun,\n year = \"2024\",\n address = \"Mexico City, Mexico\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2024.naacl-long.469\",\n doi = \"10.18653/v1/2024.naacl-long.469\",\n pages = \"8487--8505\",\n abstract = \"Language technologies that accurately model the dynamics of events must perform commonsense reasoning. Existing work evaluating commonsense reasoning focuses on making inferences about common, everyday situations. To instead investigate the ability to model unusual, unexpected, and unlikely situations, we explore the task of uncommonsense abductive reasoning. Given a piece of context with an unexpected outcome, this task requires reasoning abductively to generate an explanation that makes the unexpected outcome more likely in the context. To this end, we curate and release a new English language corpus called UNcommonsense. We characterize the performance differences between human explainers and the best-performing large language models, finding that model-enhanced human-written explanations achieve the highest quality by trading off between specificity and diversity. Finally, we experiment with several imitation learning algorithms to train open and accessible language models on this task. When compared with the vanilla supervised fine-tuning approach, these methods consistently reduce lose rates on both common and uncommonsense abductive reasoning judged by human evaluators.\",\n}\n\n@article{HOBBS199369,\ntitle = {Interpretation as abduction},\njournal = {Artificial Intelligence},\nvolume = {63},\nnumber = {1},\npages = {69-142},\nyear = {1993},\nissn = {0004-3702},\ndoi = {https://doi.org/10.1016/0004-3702(93)90015-4},\nurl = {https://www.sciencedirect.com/science/article/pii/0004370293900154},\nauthor = {Jerry R. Hobbs and Mark E. Stickel and Douglas E. Appelt and Paul Martin},\nabstract = {Abduction is inference to the best explanation. In the TACITUS project at SRI we have developed an approach to abductive inference, called \u201cweighted abduction\u201d, that has resulted in a significant simplification of how the problem of interpreting texts is conceptualized. The interpretation of a text is the minimal explanation of why the text would be true. More precisely, to interpret a text, one must prove the logical form of the text from what is already mutually known, allowing for coercions, merging redundancies where possible, and making assumptions where necessary. It is shown how such \u201clocal pragmatics\u201d problems as reference resolution, the interpretation of compound nominals, the resolution of syntactic ambiguity and metonymy, and schema recognition can be solved in this manner. Moreover, this approach of \u201cinterpretation as abduction\u201d can be combined with the older view of \u201cparsing as deduction\u201d to produce an elegant and thorough integration of syntax, semantics, and pragmatics, one that spans the range of linguistic phenomena from phonology to discourse structure. Finally, we discuss means for making the abduction process efficient, possibilities for extending the approach to other pragmatics phenomena, and the semantics of the weights and costs in the abduction scheme.}\n}\n@article{branchini,\ntitle ={Opposites in Reasoning Processes: Do We Use Them More Than We Think, but Less Than We Could?},\nabstract={Our aim in this paper is to contribute toward acknowledging the general role of opposites as an organizing principle in the human mind. We support this claim in relation to human reasoning by collecting evidence from various studies which shows that \"thinking in opposites\" is not only involved in formal logical thinking, but can also be applied in both deductive and inductive reasoning, as well as in problem solving. We also describe the results of a series of studies which, although they have been developed within a number of different theoretical frameworks based on various methodologies, all demonstrate that giving hints or training reasoners to think in terms of opposites improves their performance in tasks in which spontaneous thinking may lead to classic biases and impasses. Since we all possess an intuitive idea of what opposites are, prompting people to \"think in opposites\" is something which is undoubtedly within everyone's reach and in the final section, we discuss the potential of this strategy and suggest possible future research directions of systematic testing the benefits that might arise from the use of this technique in contexts beyond those tested thus far. Ascertaining the conditions in which reasoners might benefit will also help in terms of clarifying the underlying mechanisms from the point of view, for instance, of analytical, conscious processing vs. automatic, unconscious processing.},\nauthor={Erika Branchini and Elena Capitani and Roberto Burro and Ugo Savardi and Ivana Bianchi} ,\nJournal={Frontiers in psychology},\ndoi={10.3389/fpsyg.2021.715696},\nyear={2021},\nurl={https://www.ncbi.nlm.nih.gov/pmc/articles/PMC8426631/}\n}\n@article{chinn,\n ISSN = {00346543, 19351046},\n URL = {http://www.jstor.org/stable/1170558},\n abstract = {Understanding how science students respond to anomalous data is essential to understanding knowledge acquisition in science classrooms. This article presents a detailed analysis of the ways in which scientists and science students respond to such data. We postulate that there are seven distinct forms of response to anomalous data, only one of which is to accept the data and change theories. The other six responses involve discounting the data in various ways in order to protect the preinstructional theory. We analyze the factors that influence which of these seven forms of response a scientist or student will choose, giving special attention to the factors that make theory change more likely. Finally, we discuss the implications of our framework for science instruction.},\n author = {Clark A. Chinn and William F. Brewer},\n journal = {Review of Educational Research},\n number = {1},\n pages = {1--49},\n publisher = {[Sage Publications, Inc., American Educational Research Association]},\n title = {The Role of Anomalous Data in Knowledge Acquisition: A Theoretical Framework and Implications for Science Instruction},\n urldate = {2024-10-02},\n volume = {63},\n year = {1993}\n}\n\n @inproceedings{mostafazadeh-etal-2016-story,\n title = \"Story Cloze Evaluator: Vector Space Representation Evaluation by Predicting What Happens Next\",\n author = \"Mostafazadeh, Nasrin and\n Vanderwende, Lucy and\n Yih, Wen-tau and\n Kohli, Pushmeet and\n Allen, James\",\n booktitle = \"Proceedings of the 1st Workshop on Evaluating Vector-Space Representations for {NLP}\",\n month = aug,\n year = \"2016\",\n address = \"Berlin, Germany\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/W16-2505\",\n doi = \"10.18653/v1/W16-2505\",\n pages = \"24--29\",\n}\n@inproceedings{mostafazadeh-etal-2016-corpus,\n title = \"A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories\",\n author = \"Mostafazadeh, Nasrin and\n Chambers, Nathanael and\n He, Xiaodong and\n Parikh, Devi and\n Batra, Dhruv and\n Vanderwende, Lucy and\n Kohli, Pushmeet and\n Allen, James\",\n editor = \"Knight, Kevin and\n Nenkova, Ani and\n Rambow, Owen\",\n booktitle = \"Proceedings of the 2016 Conference of the North {A}merican Chapter of the Association for Computational Linguistics: Human Language Technologies\",\n month = jun,\n year = \"2016\",\n address = \"San Diego, California\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/N16-1098\",\n doi = \"10.18653/v1/N16-1098\",\n pages = \"839--849\",\n}\n@inproceedings{lin-2004-rouge,\n title = \"{ROUGE}: A Package for Automatic Evaluation of Summaries\",\n author = \"Lin, Chin-Yew\",\n booktitle = \"Text Summarization Branches Out\",\n month = jul,\n year = \"2004\",\n address = \"Barcelona, Spain\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/W04-1013\",\n pages = \"74--81\",\n}\n@inproceedings{BLEU,\nauthor = {Papineni, Kishore and Roukos, Salim and Ward, Todd and Zhu, Wei-Jing},\ntitle = {BLEU: A Method for Automatic Evaluation of Machine Translation},\nyear = {2002},\npublisher = {Association for Computational Linguistics},\naddress = {USA},\nurl = {https://doi.org/10.3115/1073083.1073135},\ndoi = {10.3115/1073083.1073135},\nabstract = {Human evaluations of machine translation are extensive but expensive. Human evaluations can take months to finish and involve human labor that can not be reused. We propose a method of automatic machine translation evaluation that is quick, inexpensive, and language-independent, that correlates highly with human evaluation, and that has little marginal cost per run. We present this method as an automated understudy to skilled human judges which substitutes for them when there is need for quick or frequent evaluations.},\nbooktitle = {Proceedings of the 40th Annual Meeting on Association for Computational Linguistics},\npages = {311\u2013318},\nnumpages = {8},\nlocation = {Philadelphia, Pennsylvania},\nseries = {ACL '02}\n}\n@inproceedings{lavie-agarwal-2007-meteor,\n title = \"{METEOR}: An Automatic Metric for {MT} Evaluation with High Levels of Correlation with Human Judgments\",\n author = \"Lavie, Alon and\n Agarwal, Abhaya\",\n booktitle = \"Proceedings of the Second Workshop on Statistical Machine Translation\",\n month = jun,\n year = \"2007\",\n address = \"Prague, Czech Republic\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/W07-0734\",\n pages = \"228--231\",\n}\n@inproceedings{bert-score,\n title={BERTScore: Evaluating Text Generation with BERT},\n author={Tianyi Zhang and Varsha Kishore and Felix Wu and Kilian Q. Weinberger and Yoav Artzi},\n booktitle={International Conference on Learning Representations},\n year={2020},\n url={https://openreview.net/forum?id=SkeHuCVFDr}\n}\nCite (ACL):\n Neema Kotonya and Francesca Toni. 2020. Explainable Automated Fact-Checking for Public Health Claims. In Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing (EMNLP), pages 7740\u20137754, Online. Association for Computational Linguistics.\nCite (Informal):\n Explainable Automated Fact-Checking for Public Health Claims (Kotonya & Toni, EMNLP 2020)\nCopy Citation:\nPDF:\n https://aclanthology.org/2020.emnlp-main.623.pdf\nVideo:\nCode\n neemakot/Health-Fact-Checking + \n\n additional community code\nData\n PUBHEALTH, FEVER, LIAR, MultiFC, MultiNLI, SNLI\n\nExport citation\n\n BibTeX\n MODS XML\n Endnote\n Preformatted\n\n@inproceedings{kotonya-toni-2020-explainable-automated,\n title = \"Explainable Automated Fact-Checking for Public Health Claims\",\n author = \"Kotonya, Neema and\n Toni, Francesca\",\n editor = \"Webber, Bonnie and\n Cohn, Trevor and\n He, Yulan and\n Liu, Yang\",\n booktitle = \"Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing (EMNLP)\",\n month = nov,\n year = \"2020\",\n address = \"Online\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2020.emnlp-main.623\",\n doi = \"10.18653/v1/2020.emnlp-main.623\",\n pages = \"7740--7754\",\n abstract = \"Fact-checking is the task of verifying the veracity of claims by assessing their assertions against credible evidence. The vast majority of fact-checking studies focus exclusively on political claims. Very little research explores fact-checking for other topics, specifically subject matters for which expertise is required. We present the first study of explainable fact-checking for claims which require specific expertise. For our case study we choose the setting of public health. To support this case study we construct a new dataset PUBHEALTH of 11.8K claims accompanied by journalist crafted, gold standard explanations (i.e., judgments) to support the fact-check labels for claims. We explore two tasks: veracity prediction and explanation generation. We also define and evaluate, with humans and computationally, three coherence properties of explanation quality. Our results indicate that, by training on in-domain data, gains can be made in explainable, automated fact-checking for claims which require specific expertise.\",\n}\n\n@inproceedings{petroni-etal-2019-language,\n title = \"Language Models as Knowledge Bases?\",\n author = {Petroni, Fabio and\n Rockt{\\\"a}schel, Tim and\n Riedel, Sebastian and\n Lewis, Patrick and\n Bakhtin, Anton and\n Wu, Yuxiang and\n Miller, Alexander},\n editor = \"Inui, Kentaro and\n Jiang, Jing and\n Ng, Vincent and\n Wan, Xiaojun\",\n booktitle = \"Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP)\",\n month = nov,\n year = \"2019\",\n address = \"Hong Kong, China\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/D19-1250\",\n doi = \"10.18653/v1/D19-1250\",\n pages = \"2463--2473\",\n abstract = \"Recent progress in pretraining language models on large textual corpora led to a surge of improvements for downstream NLP tasks. Whilst learning linguistic knowledge, these models may also be storing relational knowledge present in the training data, and may be able to answer queries structured as {``}fill-in-the-blank{''} cloze statements. Language models have many advantages over structured knowledge bases: they require no schema engineering, allow practitioners to query about an open class of relations, are easy to extend to more data, and require no human supervision to train. We present an in-depth analysis of the relational knowledge already present (without fine-tuning) in a wide range of state-of-the-art pretrained language models. We find that (i) without fine-tuning, BERT contains relational knowledge competitive with traditional NLP methods that have some access to oracle knowledge, (ii) BERT also does remarkably well on open-domain question answering against a supervised baseline, and (iii) certain types of factual knowledge are learned much more readily than others by standard language model pretraining approaches. The surprisingly strong ability of these models to recall factual knowledge without any fine-tuning demonstrates their potential as unsupervised open-domain QA systems. The code to reproduce our analysis is available at \\url{https://github.com/facebookresearch/LAMA}.\",\n}\n\n@article{stanford,\n author = {Matthew Dahl and Varun Magesh and Mirac Suzgun and Daniel E. Ho},\n\tjournal = {Stanford University Human Centered AI: Law, Regulation, and Policy},\n\ttitle = {Hallucinating Law: Legal Mistakes with Large Language Models are Pervasive},\n\tmonth = {January},\n\tyear = {2024},\n abstract={A new study finds disturbing and pervasive errors among three popular models on a wide range of legal tasks.},\n url={https://hai.stanford.edu/news/hallucinating-law-legal-mistakes-large-language-models-are-pervasive}\n}\n@inproceedings{cohen-etal-2023-crawling,\n title = \"Crawling The Internal Knowledge-Base of Language Models\",\n author = \"Cohen, Roi and\n Geva, Mor and\n Berant, Jonathan and\n Globerson, Amir\",\n editor = \"Vlachos, Andreas and\n Augenstein, Isabelle\",\n booktitle = \"Findings of the Association for Computational Linguistics: EACL 2023\",\n month = may,\n year = \"2023\",\n address = \"Dubrovnik, Croatia\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2023.findings-eacl.139\",\n doi = \"10.18653/v1/2023.findings-eacl.139\",\n pages = \"1856--1869\",\n abstract = \"Language models are trained on large volumes of text, and as a result their parameters might contain a significant body of factual knowledge. Any downstream task performed by these models implicitly builds on these facts, and thus it is highly desirable to have means for representing this body of knowledge in an interpretable way. However, there is currently no mechanism for such a representation. Here, we propose to address this goal by extracting a knowledge-graph of facts from a given language model. We describe a procedure for {``}crawling{''} the internal knowledge-base of a language model. Specifically, given a seed entity, we expand a knowledge-graph around it. The crawling procedure is decomposed into sub-tasks, realized through specially designed prompts that control for both precision (i.e., that no wrong facts are generated) and recall (i.e., the number of facts generated). We evaluate our approach on graphs crawled starting from dozens of seed entities, and show it yields high precision graphs (82-92{\\%}), while emitting a reasonable number of facts per entity.\",\n}\n\n\n@misc{jiang2024largelanguagemodelshallucination,\n title={On Large Language Models' Hallucination with Regard to Known Facts}, \n author={Che Jiang and Biqing Qi and Xiangyu Hong and Dayuan Fu and Yang Cheng and Fandong Meng and Mo Yu and Bowen Zhou and Jie Zhou},\n year={2024},\n eprint={2403.20009},\n archivePrefix={arXiv},\n primaryClass={cs.CL},\n url={https://arxiv.org/abs/2403.20009}, \n}\n@inproceedings{heinzerling-inui-2021-language,\n title = \"Language Models as Knowledge Bases: On Entity Representations, Storage Capacity, and Paraphrased Queries\",\n author = \"Heinzerling, Benjamin and\n Inui, Kentaro\",\n editor = \"Merlo, Paola and\n Tiedemann, Jorg and\n Tsarfaty, Reut\",\n booktitle = \"Proceedings of the 16th Conference of the European Chapter of the Association for Computational Linguistics: Main Volume\",\n month = apr,\n year = \"2021\",\n address = \"Online\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2021.eacl-main.153\",\n doi = \"10.18653/v1/2021.eacl-main.153\",\n pages = \"1772--1791\",\n abstract = \"Pretrained language models have been suggested as a possible alternative or complement to structured knowledge bases. However, this emerging LM-as-KB paradigm has so far only been considered in a very limited setting, which only allows handling 21k entities whose name is found in common LM vocabularies. Furthermore, a major benefit of this paradigm, i.e., querying the KB using natural language paraphrases, is underexplored. Here we formulate two basic requirements for treating LMs as KBs: (i) the ability to store a large number facts involving a large number of entities and (ii) the ability to query stored facts. We explore three entity representations that allow LMs to handle millions of entities and present a detailed case study on paraphrased querying of facts stored in LMs, thereby providing a proof-of-concept that language models can indeed serve as knowledge bases.\",\n}\n@inproceedings{Eckroth2014CommonsenseAR,\n title={Commonsense Abductive Reasoning and Metareasoning Using Knowledge from Bayesian Networks},\n author={Joshua Eckroth and John R. Josephson},\n booktitle={AAAI Spring Symposia},\n year={2014},\n url={https://api.semanticscholar.org/CorpusID:53306794}\n}\n@inproceedings{zhao-etal-2023-abductive,\n title = \"Abductive Commonsense Reasoning Exploiting Mutually Exclusive Explanations\",\n author = \"Zhao, Wenting and\n Chiu, Justin and\n Cardie, Claire and\n Rush, Alexander\",\n editor = \"Rogers, Anna and\n Boyd-Graber, Jordan and\n Okazaki, Naoaki\",\n booktitle = \"Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)\",\n month = jul,\n year = \"2023\",\n address = \"Toronto, Canada\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2023.acl-long.831\",\n doi = \"10.18653/v1/2023.acl-long.831\",\n pages = \"14883--14896\",\n abstract = \"Abductive reasoning aims to find plausible explanations for an event. This style of reasoning is critical for commonsense tasks where there are often multiple plausible explanations. Existing approaches for abductive reasoning in natural language processing (NLP) often rely on manually generated annotations for supervision; however, such annotations can be subjective and biased. Instead of using direct supervision, this work proposes an approach for abductive commonsense reasoning that exploits the fact that only a subset of explanations is correct for a given context. The method uses posterior regularization to enforce a mutual exclusion constraint, encouraging the model to learn the distinction between fluent explanations and plausible ones. We evaluate our approach on a diverse set of abductive reasoning datasets; experimental results show that our approach outperforms or is comparable to directly applying pretrained language models in a zero-shot manner and other knowledge-augmented zero-shot methods.\",\n}\n\nCite (ACL):\n Forough Arabshahi, Jennifer Lee, Antoine Bosselut, Yejin Choi, and Tom Mitchell. 2021. Conversational Multi-Hop Reasoning with Neural Commonsense Knowledge and Symbolic Logic Rules. In Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing, pages 7404\u20137418, Online and Punta Cana, Dominican Republic. Association for Computational Linguistics.\nCite (Informal):\n Conversational Multi-Hop Reasoning with Neural Commonsense Knowledge and Symbolic Logic Rules (Arabshahi et al., EMNLP 2021)\nCopy Citation:\nPDF:\n https://aclanthology.org/2021.emnlp-main.588.pdf\nVideo:\n\nData\n ConceptNet\n\nExport citation\n\n BibTeX\n MODS XML\n Endnote\n Preformatted\n\n@inproceedings{arabshahi-etal-2021-conversational,\n title = \"Conversational Multi-Hop Reasoning with Neural Commonsense Knowledge and Symbolic Logic Rules\",\n author = \"Arabshahi, Forough and\n Lee, Jennifer and\n Bosselut, Antoine and\n Choi, Yejin and\n Mitchell, Tom\",\n editor = \"Moens, Marie-Francine and\n Huang, Xuanjing and\n Specia, Lucia and\n Yih, Scott Wen-tau\",\n booktitle = \"Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing\",\n month = nov,\n year = \"2021\",\n address = \"Online and Punta Cana, Dominican Republic\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2021.emnlp-main.588\",\n doi = \"10.18653/v1/2021.emnlp-main.588\",\n pages = \"7404--7418\",\n abstract = \"One of the challenges faced by conversational agents is their inability to identify unstated presumptions of their users{'} commands, a task trivial for humans due to their common sense. In this paper, we propose a zero-shot commonsense reasoning system for conversational agents in an attempt to achieve this. Our reasoner uncovers unstated presumptions from user commands satisfying a general template of if-(state), then-(action), because-(goal). Our reasoner uses a state-of-the-art transformer-based generative commonsense knowledge base (KB) as its source of background knowledge for reasoning. We propose a novel and iterative knowledge query mechanism to extract multi-hop reasoning chains from the neural KB which uses symbolic logic rules to significantly reduce the search space. Similar to any KBs gathered to date, our commonsense KB is prone to missing knowledge. Therefore, we propose to conversationally elicit the missing knowledge from human users with our novel dynamic question generation strategy, which generates and presents contextualized queries to human users. We evaluate the model with a user study with human users that achieves a 35{\\%} higher success rate compared to SOTA.\",\n}\n\n@inproceedings{rajani-etal-2019-explain,\n title = \"Explain Yourself! Leveraging Language Models for Commonsense Reasoning\",\n author = \"Rajani, Nazneen Fatema and\n McCann, Bryan and\n Xiong, Caiming and\n Socher, Richard\",\n editor = \"Korhonen, Anna and\n Traum, David and\n M{\\`a}rquez, Llu{\\'\\i}s\",\n booktitle = \"Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics\",\n month = jul,\n year = \"2019\",\n address = \"Florence, Italy\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/P19-1487\",\n doi = \"10.18653/v1/P19-1487\",\n pages = \"4932--4942\",\n abstract = \"Deep learning models perform poorly on tasks that require commonsense reasoning, which often necessitates some form of world-knowledge or reasoning over information not immediately present in the input. We collect human explanations for commonsense reasoning in the form of natural language sequences and highlighted annotations in a new dataset called Common Sense Explanations (CoS-E). We use CoS-E to train language models to automatically generate explanations that can be used during training and inference in a novel Commonsense Auto-Generated Explanation (CAGE) framework. CAGE improves the state-of-the-art by 10{\\%} on the challenging CommonsenseQA task. We further study commonsense reasoning in DNNs using both human and auto-generated explanations including transfer to out-of-domain tasks. Empirical results indicate that we can effectively leverage language models for commonsense reasoning.\",\n}\n\n@inproceedings{yang-etal-2022-generating,\n title = \"Generating Natural Language Proofs with Verifier-Guided Search\",\n author = \"Yang, Kaiyu and\n Deng, Jia and\n Chen, Danqi\",\n editor = \"Goldberg, Yoav and\n Kozareva, Zornitsa and\n Zhang, Yue\",\n booktitle = \"Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing\",\n month = dec,\n year = \"2022\",\n address = \"Abu Dhabi, United Arab Emirates\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2022.emnlp-main.7\",\n doi = \"10.18653/v1/2022.emnlp-main.7\",\n pages = \"89--105\",\n abstract = \"Reasoning over natural language is a challenging problem in NLP. In this work, we focus on proof generation: Given a hypothesis and a set of supporting facts, the model generates a proof tree indicating how to derive the hypothesis from supporting facts. Compared to generating the entire proof in one shot, stepwise generation can better exploit the compositionality and generalize to longer proofs but has achieved limited success on real-world data. Existing stepwise methods struggle to generate proof steps that are both logically valid and relevant to the hypothesis. Instead, they tend to hallucinate invalid steps given the hypothesis. In this paper, we present a novel stepwise method, NLProofS (Natural Language Proof Search), which learns to generate relevant steps conditioning on the hypothesis. At the core of our approach, we train an independent verifier to check the validity of the proof steps to prevent hallucination. Instead of generating steps greedily, we search for proofs maximizing a global proof score judged by the verifier. NLProofS achieves state-of-the-art performance on EntailmentBank and RuleTaker. Specifically, it improves the correctness of predicted proofs from 27.7{\\%} to 33.3{\\%} in the distractor setting of EntailmentBank, demonstrating the effectiveness of NLProofS in generating challenging human-authored proofs.\",\n}\n\n@inproceedings{dalvi-mishra-etal-2022-towards,\n title = \"Towards Teachable Reasoning Systems: Using a Dynamic Memory of User Feedback for Continual System Improvement\",\n author = \"Dalvi Mishra, Bhavana and\n Tafjord, Oyvind and\n Clark, Peter\",\n editor = \"Goldberg, Yoav and\n Kozareva, Zornitsa and\n Zhang, Yue\",\n booktitle = \"Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing\",\n month = dec,\n year = \"2022\",\n address = \"Abu Dhabi, United Arab Emirates\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2022.emnlp-main.644\",\n doi = \"10.18653/v1/2022.emnlp-main.644\",\n pages = \"9465--9480\",\n abstract = \"Our goal is a teachable reasoning system for question-answering (QA), where a user can interact with faithful answer explanations, and correct its errors so that the system improves over time. Our approach is to augment a QA model with a dynamic memory of user feedback, containing user-supplied corrections toerroneous model beliefs that users identify during interaction. Retrievals from memory are used as additional context for QA, to help avoid previous mistakes in similar new situations - a novel application of memory-based continuous learning. With simulated feedback, we find that our system (called TeachMe) continually improves with time, and without model retraining, requiring feedback on only 25{\\%} of training examples to reach within 1{\\%} of the upper-bound (feedback on all examples). Similarly, in experiments with real users, we observe a similar trend, with performance improving by over 15{\\%} on a hidden test set after teaching. This suggests new opportunities for using frozen language models in an interactive setting where users can inspect, debug, and correct the model{'}s beliefs, leading to improved system{'}s performance over time.\",\n}\n\n@inproceedings{chen-etal-2024-improving-discriminative,\n title = \"Improving Discriminative Capability of Reward Models in {RLHF} Using Contrastive Learning\",\n author = \"Chen, Lu and\n Zheng, Rui and\n Wang, Binghai and\n Jin, Senjie and\n Huang, Caishuang and\n Ye, Junjie and\n Zhang, Zhihao and\n Zhou, Yuhao and\n Xi, Zhiheng and\n Gui, Tao and\n Zhang, Qi and\n Huang, Xuanjing\",\n editor = \"Al-Onaizan, Yaser and\n Bansal, Mohit and\n Chen, Yun-Nung\",\n booktitle = \"Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing\",\n month = nov,\n year = \"2024\",\n address = \"Miami, Florida, USA\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2024.emnlp-main.852\",\n doi = \"10.18653/v1/2024.emnlp-main.852\",\n pages = \"15270--15283\",\n abstract = \"Reinforcement Learning from Human Feedback (RLHF) is a crucial approach to aligning language models with human values and intentions. A fundamental challenge in this method lies in ensuring that the reward model accurately understands and evaluates human preferences. Current methods rely on ranking losses to teach the reward model to assess preferences, but they are susceptible to noise and ambiguous data, often failing to deeply understand human intentions. To address this issue, we introduce contrastive learning into the reward modeling process. In addition to supervised ranking loss, we introduce an unsupervised contrastive loss to enable the reward model to fully capture the distinctions in contrastive data. Experimental results demonstrate that the proposed contrastive learning-based reward modeling method effectively enhances the generalization of the reward model, stabilizes the reinforcement learning training process, and improves the final alignment with human preferences.\",\n}\n@misc{ouyang2022traininglanguagemodelsfollow,\n title={Training language models to follow instructions with human feedback}, \n author={Long Ouyang and Jeff Wu and Xu Jiang and Diogo Almeida and Carroll L. Wainwright and Pamela Mishkin and Chong Zhang and Sandhini Agarwal and Katarina Slama and Alex Ray and John Schulman and Jacob Hilton and Fraser Kelton and Luke Miller and Maddie Simens and Amanda Askell and Peter Welinder and Paul Christiano and Jan Leike and Ryan Lowe},\n year={2022},\n eprint={2203.02155},\n archivePrefix={arXiv},\n primaryClass={cs.CL},\n url={https://arxiv.org/abs/2203.02155}, \n}\n\n\n@inproceedings{pan-etal-2023-logic,\n title = \"Logic-{LM}: Empowering Large Language Models with Symbolic Solvers for Faithful Logical Reasoning\",\n author = \"Pan, Liangming and\n Albalak, Alon and\n Wang, Xinyi and\n Wang, William\",\n editor = \"Bouamor, Houda and\n Pino, Juan and\n Bali, Kalika\",\n booktitle = \"Findings of the Association for Computational Linguistics: EMNLP 2023\",\n month = dec,\n year = \"2023\",\n address = \"Singapore\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2023.findings-emnlp.248\",\n doi = \"10.18653/v1/2023.findings-emnlp.248\",\n pages = \"3806--3824\",\n abstract = \"Large Language Models (LLMs) have shown human-like reasoning abilities but still struggle with complex logical problems. This paper introduces a novel framework, Logic-LM, which integrates LLMs with symbolic solvers to improve logical problem-solving. Our method first utilizes LLMs to translate a natural language problem into a symbolic formulation. Afterward, a deterministic symbolic solver performs inference on the formulated problem. We also introduce a self-refinement module, which utilizes the symbolic solver{'}s error messages to revise symbolic formalizations. We demonstrate Logic-LM{'}s effectiveness on five logical reasoning datasets: ProofWriter, PrOntoQA, FOLIO, LogicalDeduction, and AR-LSAT. On average, Logic-LM achieves a significant performance boost of 39.2{\\%} over using LLM alone with standard prompting and 18.4{\\%} over LLM with chain-of-thought prompting. Our findings suggest that Logic-LM, by combining LLMs with symbolic logic, offers a promising avenue for faithful logical reasoning.\",\n}\n\n@misc{he2022rethinkingretrievalfaithfullarge,\n title={Rethinking with Retrieval: Faithful Large Language Model Inference}, \n author={Hangfeng He and Hongming Zhang and Dan Roth},\n year={2022},\n eprint={2301.00303},\n archivePrefix={arXiv},\n primaryClass={cs.CL},\n url={https://arxiv.org/abs/2301.00303}, \n}\n@inproceedings{wang-etal-2024-boosting-language,\n title = \"Boosting Language Models Reasoning with Chain-of-Knowledge Prompting\",\n author = \"Wang, Jianing and\n Sun, Qiushi and\n Li, Xiang and\n Gao, Ming\",\n editor = \"Ku, Lun-Wei and\n Martins, Andre and\n Srikumar, Vivek\",\n booktitle = \"Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)\",\n month = aug,\n year = \"2024\",\n address = \"Bangkok, Thailand\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2024.acl-long.271\",\n doi = \"10.18653/v1/2024.acl-long.271\",\n pages = \"4958--4981\",\n abstract = \"Recently, Chain-of-Thought (CoT) prompting has delivered success on complex reasoning tasks, which aims at designing a simple prompt like {``}Let{'}s think step by step{''} or multiple in-context exemplars with well-designed rationales to elicit Large Language Models (LLMs) to generate intermediate reasoning steps. However, the generated rationales often come with hallucinations, making unfactual and unfaithful reasoning chains. To mitigate this brittleness, we propose a novel Chain-of-Knowledge (CoK) prompting, where we aim at eliciting LLMs to generate explicit pieces of knowledge evidence in the form of structure triple. This is inspired by our human behaviors, i.e., we can draw a mind map or knowledge map as the reasoning evidence in the brain before answering a complex question. Benefiting from CoK, we additionally introduce an F2-Verification method to estimate the reliability of the reasoning chains in terms of factuality and faithfulness. For the unreliable response, the wrong evidence can be indicated to prompt the LLM to rethink. Extensive experiments demonstrate that our method can further improve the performance of commonsense, factual, symbolic, and arithmetic reasoning tasks.\",\n}\n@inproceedings{tyen-etal-2024-llms,\n title = \"{LLM}s cannot find reasoning errors, but can correct them given the error location\",\n author = \"Tyen, Gladys and\n Mansoor, Hassan and\n Carbune, Victor and\n Chen, Peter and\n Mak, Tony\",\n editor = \"Ku, Lun-Wei and\n Martins, Andre and\n Srikumar, Vivek\",\n booktitle = \"Findings of the Association for Computational Linguistics: ACL 2024\",\n month = aug,\n year = \"2024\",\n address = \"Bangkok, Thailand\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2024.findings-acl.826\",\n doi = \"10.18653/v1/2024.findings-acl.826\",\n pages = \"13894--13908\",\n abstract = \"While self-correction has shown promise in improving LLM outputs in terms of style and quality (e.g. Chen et al., 2023b; Madaan et al.,2023), recent attempts to self-correct logical or reasoning errors often cause correct answers to become incorrect, resulting in worse performances overall (Huang et al., 2023). In this paper, we show that poor self-correction performance stems from LLMs{'} inability tofind logical mistakes, rather than their ability to correct a known mistake. Firstly, we benchmark several state-of-the-art LLMs ontheir mistake-finding ability and demonstrate that they generally struggle with the task, even in highly objective, unambiguous cases. Secondly, we test the correction abilities of LLMs {--} separately from mistake finding {--} using a backtracking setup that feeds ground truth mistake location information to the model. We show that this boosts downstream task performance across our 5 reasoning tasks, indicating that LLMs{'} correction abilities are robust. Finally, we show that it is possible to obtain mistake location information without ground truth labels or in-domain training data. We train a small classifier with out-of-domain data, which exhibits stronger mistake-finding performance than prompting a large model. We release our dataset of LLM-generated logical mistakes, BIG-Bench Mistake, to enable further research into locating LLM reasoning mistakes.\",\n}\n\n@misc{kamoi2024evaluatingllmsdetectingerrors,\n title={Evaluating LLMs at Detecting Errors in LLM Responses}, \n author={Ryo Kamoi and Sarkar Snigdha Sarathi Das and Renze Lou and Jihyun Janice Ahn and Yilun Zhao and Xiaoxin Lu and Nan Zhang and Yusen Zhang and Ranran Haoran Zhang and Sujeeth Reddy Vummanthala and Salika Dave and Shaobo Qin and Arman Cohan and Wenpeng Yin and Rui Zhang},\n year={2024},\n eprint={2404.03602},\n archivePrefix={arXiv},\n primaryClass={cs.CL},\n url={https://arxiv.org/abs/2404.03602}, \n}\n@article{LANGDON2019103891,\ntitle = {Uncovering the \u2018state\u2019: Tracing the hidden state representations that structure learning and decision-making},\njournal = {Behavioural Processes},\nvolume = {167},\npages = {103891},\nyear = {2019},\nissn = {0376-6357},\ndoi = {https://doi.org/10.1016/j.beproc.2019.103891},\nurl = {https://www.sciencedirect.com/science/article/pii/S0376635719300397},\nauthor = {Angela J. Langdon and Mingyu Song and Yael Niv},\nkeywords = {Learning, Decision making, Reward, Timing, Dopamine},\nabstract = {We review the abstract concept of a \u2018state\u2019 \u2013 an internal representation posited by reinforcement learning theories to be used by an agent, whether animal, human or artificial, to summarize the features of the external and internal environment that are relevant for future behavior on a particular task. Armed with this summary representation, an agent can make decisions and perform actions to interact effectively with the world. Here, we review recent findings from the neurobiological and behavioral literature to ask: \u2018what is a state?\u2019 with respect to the internal representations that organize learning and decision making across a range of tasks. We find that state representations include information beyond a straightforward summary of the immediate cues in the environment, providing timing or contextual information from the recent or more distant past, which allows these additional factors to influence decision making and other goal-directed behaviors in complex and perhaps unexpected ways.}\n}\n@article {10.7554/eLife.69430,\narticle_type = {journal},\ntitle = {Developmental changes in story-evoked responses in the neocortex and hippocampus},\nauthor = {Cohen, Samantha S and Tottenham, Nim and Baldassano, Christopher},\neditor = {Kok, Peter and de Lange, Floris P and Kok, Peter and Saxe, Rebecca and Reagh, Zachariah M},\nvolume = 11,\nyear = 2022,\nmonth = {jul},\npub_date = {2022-07-05},\npages = {e69430},\ncitation = {eLife 2022;11:e69430},\ndoi = {10.7554/eLife.69430},\nurl = {https://doi.org/10.7554/eLife.69430},\nabstract = {How does the representation of naturalistic life events change with age? Here, we analyzed fMRI data from 414 children and adolescents (5\u201319 years) as they watched a narrative movie. In addition to changes in the degree of inter-subject correlation (ISC) with age in sensory and medial parietal regions, we used a novel measure (between-group ISC) to reveal age-related shifts in the responses across the majority of the neocortex. Over the course of development, brain responses became more discretized into stable and coherent events and shifted earlier in time to anticipate upcoming perceived event transitions, measured behaviorally in an age-matched sample. However, hippocampal responses to event boundaries actually decreased with age, suggesting a shifting division of labor between episodic encoding processes and schematic event representations between the ages of 5 and 19.},\nkeywords = {child development, inter-subject correlation, event segmentation, hippocampus, hidden Markov model, narrative},\njournal = {eLife},\nissn = {2050-084X},\npublisher = {eLife Sciences Publications, Ltd},\n}\n@inproceedings{\namir2024states,\ntitle={States as goal-directed concepts: an epistemic approach to state-representation learning},\nauthor={Nadav Amir and Yael Niv and Angela J Langdon},\nbooktitle={Reinforcement Learning Conference},\nyear={2024},\nurl={https://openreview.net/forum?id=MwYgU2eXIE}\n}\n@inproceedings{brown,\nauthor = {Brown, Tom B. and Mann, Benjamin and Ryder, Nick and Subbiah, Melanie and Kaplan, Jared and Dhariwal, Prafulla and Neelakantan, Arvind and Shyam, Pranav and Sastry, Girish and Askell, Amanda and Agarwal, Sandhini and Herbert-Voss, Ariel and Krueger, Gretchen and Henighan, Tom and Child, Rewon and Ramesh, Aditya and Ziegler, Daniel M. and Wu, Jeffrey and Winter, Clemens and Hesse, Christopher and Chen, Mark and Sigler, Eric and Litwin, Mateusz and Gray, Scott and Chess, Benjamin and Clark, Jack and Berner, Christopher and McCandlish, Sam and Radford, Alec and Sutskever, Ilya and Amodei, Dario},\ntitle = {Language Models Are Few-Shot Learners},\nyear = {2020},\nisbn = {9781713829546},\npublisher = {Curran Associates Inc.},\naddress = {Red Hook, NY, USA},\nabstract = {We demonstrate that scaling up language models greatly improves task-agnostic, few-shot performance, sometimes even becoming competitive with prior state-of-the-art fine-tuning approaches. Specifically, we train GPT-3, an autoregressive language model with 175 billion parameters, 10x more than any previous non-sparse language model, and test its performance in the few-shot setting. For all tasks, GPT-3 is applied without any gradient updates or fine-tuning, with tasks and few-shot demonstrations specified purely via text interaction with the model. GPT-3 achieves strong performance on many NLP datasets, including translation, question-answering, and cloze tasks. We also identify some datasets where GPT-3's few-shot learning still struggles, as well as some datasets where GPT-3 faces methodological issues related to training on large web corpora.},\nbooktitle = {Proceedings of the 34th International Conference on Neural Information Processing Systems},\narticleno = {159},\nnumpages = {25},\nlocation = {Vancouver, BC, Canada},\nseries = {NIPS'20}\n}\n\n@inproceedings{niv,\nyear={2019},\ntitle={Learning task-state representations},\npages={1544-1553},\nabstract={Arguably, the most difficult part of learning is deciding what to learn about. \nShould I associate the positive outcome of safely completing a street-crossing \n with the situation 'the car approaching the crosswalk was red' or with 'the \n approaching car was slowing down'? In this Perspective, we summarize our recent \n research into the computational and neural underpinnings of 'representation \n learning'-how humans (and other animals) construct task representations that \n allow efficient learning and decision-making. We first discuss the problem of \n learning what to ignore when confronted with too much information, so that \n experience can properly generalize across situations. We then turn to the problem \n of augmenting perceptual information with inferred latent causes that embody \n unobservable task-relevant information, such as contextual knowledge. Finally, we \n discuss recent findings regarding the neural substrates of task representations \n that suggest the orbitofrontal cortex represents 'task states', deploying them \n for decision-making and learning elsewhere in the brain.},\nauthor={Yael Niv},\neditor={Psychology Department and Princeton Neuroscience Institute, Princeton University, \n Princeton, New Jersey, USA.},\nbooktitle ={Nature neuroscience}\n}\n@article{hommel,\nauthor = {Hommel, Bernhard and M\u00fcsseler, Jochen and Aschersleben, Gisa and Prinz, Wolfgang},\nyear = {2001},\nmonth = {11},\npages = {849-78; discussion 878},\ntitle = {The Theory of Event Coding (TEC): A framework for perception and action planning},\nvolume = {24},\njournal = {The Behavioral and brain sciences},\ndoi = {10.1017/S0140525X01000103}\n}\n\n@article{custers,\nauthor = {Custers, Ruud},\nyear = {2023},\nmonth = {11},\npages = {},\ntitle = {Thoughts About Actions and Outcomes (And What They Lead to)},\nvolume = {9},\njournal = {Motivation Science},\ndoi = {10.1037/mot0000306}\n}\n@article{janczyk,\nauthor = {Markus Janczyk and Lea Eichfelder and Heinrich R Liesefeld and Volker H Franz},\ntitle ={Learning and transfer of response\u2013effect relations},\n\njournal = {Quarterly Journal of Experimental Psychology},\nvolume = {0},\nnumber = {0},\npages = {17470218241284259},\nyear = {0},\ndoi = {10.1177/17470218241284259},\n note ={PMID: 39256971},\n\nURL = { \n \n https://doi.org/10.1177/17470218241284259\n \n \n\n},\neprint = { \n \n https://doi.org/10.1177/17470218241284259\n \n \n\n}\n,\n abstract = { Acting means changing the environment according to one\u2019s own goals, and this often requires bodily movements as responses. How these responses are selected is a central question in contemporary cognitive psychology. The ideomotor principle offers a simple answer based on two assumptions: An agent first learns an association between a response and its effects. Later, this association can be used in a reverse way: When the agent wants to achieve a desired effect and activates its representation, the associated response representation becomes activated as well. This reversed use of the learned association is considered the means to select the required response. In three experiments, we addressed two questions related to the first assumption: First, we tested whether effect representations generalise to more abstract conceptual knowledge. This is important, because outside the laboratory and in novel situations, effects are variable and not always exactly identical, such that generalisation is necessary for successful actions. Second, the nature of the response\u2013effect relation has been debated recently, and more data are necessary to put theorising on firm empirical ground. Results of our experiments suggest that (a) abstraction to conceptual knowledge seems to occur only under very restricted situations, and (b) it seems that no (implicit) associations between responses and effects are learned, but rather (explicit) propositional knowledge in the form of rules. }\n}\n@article{custers2,\nauthor = {Custers, Ruud},\nyear = {2024},\nmonth = {12},\npages = {403-406},\ntitle = {The Homunculus in the Room: A Reply to and},\nvolume = {10},\njournal = {Motivation Science},\ndoi = {10.1037/mot0000377}\n}\n@article{kunde,\nauthor = {Kunde, Wilfried and Janczyk, Markus},\nyear = {2024},\nmonth = {12},\npages = {392-397},\ntitle = {Thoughts About \u201cThoughts About Actions and Outcomes\u201d: Comment on},\nvolume = {10},\njournal = {Motivation Science},\ndoi = {10.1037/mot0000341}\n}\n\n@article{salomon2013verb,\n title={Verb aspect and problem solving},\n author={Salomon, Meghan M and Magliano, Joseph P and Radvansky, Gabriel A},\n journal={Cognition},\n volume={128},\n number={2},\n pages={134--139},\n year={2013},\n publisher={Elsevier}\n}\n@article{hart2011learning,\n title={Learning about what others were doing: Verb aspect and attributions of mundane and criminal intent for past actions},\n author={Hart, William and Albarrac{\\'\\i}n, Dolores},\n journal={Psychological Science},\n volume={22},\n number={2},\n pages={261--266},\n year={2011},\n publisher={Sage Publications Sage CA: Los Angeles, CA}\n}\n\n@inproceedings{tandon-etal-2019-wiqa,\n title = \"{WIQA}: A dataset for {\\textquotedblleft}What if...{\\textquotedblright} reasoning over procedural text\",\n author = \"Tandon, Niket and\n Dalvi, Bhavana and\n Sakaguchi, Keisuke and\n Clark, Peter and\n Bosselut, Antoine\",\n editor = \"Inui, Kentaro and\n Jiang, Jing and\n Ng, Vincent and\n Wan, Xiaojun\",\n booktitle = \"Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP)\",\n month = nov,\n year = \"2019\",\n address = \"Hong Kong, China\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/D19-1629/\",\n doi = \"10.18653/v1/D19-1629\",\n pages = \"6076--6085\",\n abstract = \"We introduce WIQA, the first large-scale dataset of {\\textquotedblleft}What if...{\\textquotedblright} questions over procedural text. WIQA contains a collection of paragraphs, each annotated with multiple influence graphs describing how one change affects another, and a large (40k) collection of {\\textquotedblleft}What if...?{\\textquotedblright} multiple-choice questions derived from these. For example, given a paragraph about beach erosion, would stormy weather hasten or decelerate erosion? WIQA contains three kinds of questions: perturbations to steps mentioned in the paragraph; external (out-of-paragraph) perturbations requiring commonsense knowledge; and irrelevant (no effect) perturbations. We find that state-of-the-art models achieve 73.8{\\%} accuracy, well below the human performance of 96.3{\\%}. We analyze the challenges, in particular tracking chains of influences, and present the dataset as an open challenge to the community.\"\n}\n@article{goodman,\n ISSN = {0022362X, 19398549},\n URL = {http://www.jstor.org/stable/2019988},\n author = {Nelson Goodman},\n journal = {The Journal of Philosophy},\n number = {5},\n pages = {113--128},\n publisher = {Journal of Philosophy, Inc.},\n title = {The Problem of Counterfactual Conditionals},\n urldate = {2025-01-19},\n volume = {44},\n year = {1947}\n}\n\n@article{filho,\nauthor = {Filho, Oswaldo},\nyear = {2012},\nmonth = {07},\npages = {},\ntitle = {Goodman and Parry on Counterfactual},\nvolume = {15},\njournal = {Principia: an international journal of epistemology},\ndoi = {10.5007/1808-1711.2011v15n3p383}\n}\n\n@article {hugo,\n author = {Hugo Mercier and Jonathan J. Rolison and Marta Strag\u00e0 and Donatella Ferrante and Clare R. Walsh and Vittorio Girotto},\n year = {2017}, \n title = {Questioning the preparatory function of counterfactual thinking},\n journal = {Memory and Cognition},\n abstract = {Why do individuals mentally modify reality (e.g., \u201cIf it hadn\u2019t rained, we would have won the game\u201d)? According to the dominant view, counterfactuals primarily serve to prepare future performance. In fact, individuals who have just failed a task tend to modify the uncontrollable features of their attempt (e.g., \u201cIf the rules of the game were different, I would have won it\u201d), generating counterfactuals that are unlikely to play any preparatory role. By contrast, they generate prefactuals that focus on the controllable features of their ensuing behavior (e.g., \u201cIf I concentrate more, I will win the next game\u201d). Here, we test whether this tendency is robust and general. Studies 1a and 1b replicate this tendency and show that it occurs regardless of whether individuals think about their failures or their successes. Study 2 shows that individuals generate relatively few controllable counterfactuals, unless explicitly prompted to do so. These results raise some questions regarding the generality of the dominant view according to which counterfactuals mainly serve a preparatory function.},\n url = {https://doi.org/10.3758/s13421-016-0660-5}\n}\n@article{byrne,\n author = { Ruth M.J Byrne},\n year = {2016},\n title = {Counterfactual thought.},\n journal = {Annual Review of Psychology },\n volume = {67},\n pages = {135\u2013157},\n doi = {10.1146/annurev-psych-122414-033249},\n abstract = {People spontaneously create counterfactual alternatives to reality when they\nthink \u201cif only\u201d or \u201cwhat if\u201d and imagine how the past could have been differ-\nent. The mind computes counterfactuals for many reasons. Counterfactuals\nexplain the past and prepare for the future, they implicate various relations\nincluding causal ones, and they affect intentions and decisions. They mod-\nulate emotions such as regret and relief, and they support moral judgments\nsuch as blame. The loss of the ability to imagine alternatives as a result of\ninjuries to the prefrontal cortex is devastating. The basic cognitive processes\nthat compute counterfactuals mutate aspects of the mental representation\nof reality to create an imagined alternative, and they compare alternative\nrepresentations. The ability to create counterfactuals develops throughout\nchildhood and contributes to reasoning about other people\u2019s beliefs, includ-\ning their false beliefs. Knowledge affects the plausibility of a counterfactual\nthrough the semantic and pragmatic modulation of the mental representa-\ntion of alternative possibilities.}\n }\n@inproceedings{zellers-etal-2019-hellaswag,\n title = \"{H}ella{S}wag: Can a Machine Really Finish Your Sentence?\",\n author = \"Zellers, Rowan and\n Holtzman, Ari and\n Bisk, Yonatan and\n Farhadi, Ali and\n Choi, Yejin\",\n editor = \"Korhonen, Anna and\n Traum, David and\n M{\\`a}rquez, Llu{\\'i}s\",\n booktitle = \"Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics\",\n month = jul,\n year = \"2019\",\n address = \"Florence, Italy\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/P19-1472/\",\n doi = \"10.18653/v1/P19-1472\",\n pages = \"4791--4800\",\n abstract = \"Recent work by Zellers et al. (2018) introduced a new task of commonsense natural language inference: given an event description such as {\\textquotedblleft}A woman sits at a piano,{\\textquotedblright} a machine must select the most likely followup: {\\textquotedblleft}She sets her fingers on the keys.{\\textquotedblright} With the introduction of BERT, near human-level performance was reached. Does this mean that machines can perform human level commonsense inference? In this paper, we show that commonsense inference still proves difficult for even state-of-the-art models, by presenting HellaSwag, a new challenge dataset. Though its questions are trivial for humans ({\\ensuremath{>}}95{\\%} accuracy), state-of-the-art models struggle ({\\ensuremath{<}}48{\\%}). We achieve this via Adversarial Filtering (AF), a data collection paradigm wherein a series of discriminators iteratively select an adversarial set of machine-generated wrong answers. AF proves to be surprisingly robust. The key insight is to scale up the length and complexity of the dataset examples towards a critical {\\textquoteleft}Goldilocks' zone wherein generated text is ridiculous to humans, yet often misclassified by state-of-the-art models. Our construction of HellaSwag, and its resulting difficulty, sheds light on the inner workings of deep pretrained models. More broadly, it suggests a new path forward for NLP research, in which benchmarks co-evolve with the evolving state-of-the-art in an adversarial way, so as to present ever-harder challenges.\"\n}\n\n@inproceedings{byrne2019counterfactuals,\n title={Counterfactuals in explainable artificial intelligence (XAI): Evidence from human reasoning.},\n author={Byrne, Ruth MJ},\n booktitle={IJCAI},\n pages={6276--6282},\n year={2019},\n organization={California, CA}\n}\n@article{byrne1999deductive,\n title={Deductive reasoning with factual, possible, and counterfactual conditionals},\n author={Byrne, Ruth MJ and Tasso, Alessandra},\n journal={Memory \\& cognition},\n volume={27},\n pages={726--740},\n year={1999},\n publisher={Springer}\n}\n@article{magliano2000verb,\n title={Verb aspect and situation models},\n author={Magliano, Joseph P and Schleich, Michelle C},\n journal={Discourse processes},\n volume={29},\n number={2},\n pages={83--112},\n year={2000},\n publisher={Taylor \\& Francis}\n}\n@article{madden2003does,\n title={How does verb aspect constrain event representations?},\n author={Madden, Carol J and Zwaan, Rolf A},\n journal={Memory \\& cognition},\n volume={31},\n number={5},\n pages={663--672},\n year={2003},\n publisher={Springer}\n}\n" + }, + "target_bib_keys": [ + "vallurupalli-etal-2022-poque" + ], + "contexts": [ + { + "text": "We expand the 3 categories used in PASTA , for error analysis on 200 random states on the story state inference task, to 4 categories and group all conditions into these as follows: (a) Physical: This category includes natural physical attributes of an entity such as size, age,\u2026", + "text_with_marker": "We expand the 3 categories used in PASTA , for error analysis on 200 random states on the story state inference task, to 4 categories and group all conditions into these as follows: (a) Physical: This category includes natural physical attributes of an entity such as size, age,\u2026" + }, + { + "text": "We leverage two previously released datasets PASTA and SAGA (Vallurupalli et al., 2024) consisting of participants\u2019 goal and state annotations (shown in Fig.", + "text_with_marker": "We leverage two previously released datasets PASTA and SAGA (Vallurupalli et al., 2024) consisting of participants\u2019 goal and state annotations (shown in Fig." + }, + { + "text": "We leverage PASTA and SAGA (Vallurupalli et al., 2024) datasets to generate outcomes of SAGA and News Stories and validate these using both our generated and PASTA\u2019s crowd-annotated conditions. outcome is implicitly understood through constructing counterfactual mental\u2026", + "text_with_marker": "We leverage PASTA and SAGA (Vallurupalli et al., 2024) datasets to generate outcomes of SAGA and News Stories and validate these using both our generated and PASTA\u2019s crowd-annotated conditions. outcome is implicitly understood through constructing counterfactual mental\u2026" + } + ], + "target_citing_paragraphs": [], + "matched_paragraphs": [] + }, + { + "citing_paper_id": "494f3c55553161be831f725a7b55c6095470c5cb", + "citing_title": "SAGA: A Participant-specific Examination of Story Alternatives and Goal Applicability for a Deeper Understanding of Complex Events", + "arxiv_id": "2408.05793", + "main_tex_file": "sections/appendix.tex", + "bib_files": [ + "anthology.bib", + "custom.bib" + ], + "bib_texts": { + "anthology.bib": "% Please download the latest anthology.bib from\n%\n% http://aclweb.org/anthology/anthology.bib.gz\n", + "custom.bib": "% Use this file for citations not found in the ACL Anthology (contained in \"anthology.bib\").\n@article{https://doi.org/10.1111/desc.12863,\nauthor = {Richardson, Hilary and Saxe, Rebecca},\ntitle = {Development of predictive responses in theory of mind brain regions},\njournal = {Developmental Science},\nvolume = {23},\nnumber = {1},\npages = {e12863},\nkeywords = {anticipatory responses, cognitive development, fMRI, predictive coding, theory of mind},\ndoi = {https://doi.org/10.1111/desc.12863},\nurl = {https://onlinelibrary.wiley.com/doi/abs/10.1111/desc.12863},\neprint = {https://onlinelibrary.wiley.com/doi/pdf/10.1111/desc.12863},\nabstract = {Abstract When we watch movies, we consider the characters\u2019 mental states in order to understand and predict the narrative. Recent work in functional magnetic resonance imaging (fMRI) uses movie-viewing paradigms to measure functional responses in brain regions recruited for such mental state reasoning (the theory of mind [\u201cToM\u201d] network). Here, two groups of young children (n\u00a0=\u00a030 3\u20134\u00a0years old, n\u00a0=\u00a026 6\u20137\u00a0years old) viewed a short animated movie twice while undergoing fMRI. As children get older, ToM brain regions were recruited earlier in time during the second presentation of the movie. This \u201cnarrative anticipation\u201d effect is specific: there was no such effect in a control network of brain regions that responds just as robustly to the movie (the \u201cPain Matrix\u201d). These results complement prior studies in adults that suggest that ToM brain regions play a role not just in inferring, but in actively predicting, other people's thoughts and feelings, and provide novel evidence that as children get older, their ToM brain regions increasingly make such predictions.},\nyear = {2020}\n}\n@article{carpendale_lewis_2004, title={Constructing an understanding of mind: The development of children's social understanding within social interaction}, volume={27}, DOI={10.1017/S0140525X04000032}, number={1}, journal={Behavioral and Brain Sciences}, publisher={Cambridge University Press}, author={Carpendale, Jeremy I. M. and Lewis, Charlie}, year={2004}, pages={79\u201396}}\n@article{Thompson2006ConversationAD,\n title={Conversation and Developing Understanding: Introduction to the Special Issue},\n author={Ross A. Thompson},\n journal={Merrill-Palmer Quarterly},\n year={2006},\n volume={52},\n pages={1 - 16}\n}\n@inbook{carpendale_2015,\nauthor = {Carpendale, Jeremy I. M. and Lewis, Charlie},\npublisher = {John Wiley and Sons, Ltd},\nisbn = {9781118963418},\ntitle = {The Development of Social Understanding},\nbooktitle = {Handbook of Child Psychology and Developmental Science},\nchapter = {10},\npages = {1-44},\ndoi = {https://doi.org/10.1002/9781118963418.childpsy210},\nurl = {https://onlinelibrary.wiley.com/doi/abs/10.1002/9781118963418.childpsy210},\neprint = {https://onlinelibrary.wiley.com/doi/pdf/10.1002/9781118963418.childpsy210},\nyear = {2015},\nkeywords = {false belief understanding, gestures, infancy, mirror neurons, neuro-imagining, social cognitive development, theory of mind},\nabstract = {Abstract Understanding others is an essential aspect of being human. We review research and theory on the development of a range of skills constituting social understanding beginning in infancy with gaze-following and gestures such as pointing. Given the importance attributed to false belief understanding in preschoolers, we review accounts of this development as well as criticism of these theories. Such debates now extend to current claims that even infants understand others' false beliefs, raising questions about preconceptions of the nature and origins of social awareness. The later developing social skills we turn to next remain relatively neglected in this literature. We discuss contributions from neuroscience for understanding social cognitive development such as neuroimaging and research on mirror neurons. There are many links between social cognition and children's social relations and, in particular, we focus on the key role of language in social cognitive development. We also review ways in which social understanding may influence social conduct and this is linked to the issues of trust and deception, as well as cooperation and moral development. Throughout, we highlight ways that research traditions are influenced by worldviews or preconceptions about the nature of mind, meaning, and knowledge, and emphasize the interconnections between social understanding and all other aspects of human life.}\n}\n\n@article{article_goal_commitment,\nauthor = {Latham, Gary and Erez, Miriam},\nyear = {1988},\nmonth = {01},\npages = {},\ntitle = {The Determinants of Goal Commitment},\nvolume = {13},\njournal = {Academy of Management Review},\ndoi = {10.5465/AMR.1988.4306771}\n}\n@inproceedings{Herman2002StoryLP,\n booktitle={Story Logic},\n title={Problems and Possibilities of Narrative},\n\n author={David Herman},\n year={2002},\n isbn={9780803273429},\n publishser={Nebraska Paperback}\n}\n@book{ryan_possible_worlds,\nauthor = {Ryan, Marie-Laure},\naddress = {Bloomington},\nbooktitle = {Possible worlds, artificial intelligence, and narrative theory},\nisbn = {0253350042},\nkeywords = {Fiction -- Technique},\nlanguage = {eng},\nlccn = {91006825},\npublisher = {Indiana University Press},\ntitle = {Possible worlds, artificial intelligence, and narrative theory },\nyear = {1991},\n}\n\n@book{ryan_possible_worlds,\nauthor = {Rumelhart, D.},\naddress = {Bloomington},\nbooktitle = {Possible worlds, artificial intelligence, and narrative theory},\nisbn = {0253350042},\nkeywords = {Fiction -- Technique},\nlanguage = {eng},\nlccn = {91006825},\npublisher = {Indiana University Press},\ntitle = {Notes on a schema for stories },\nyear = {1991},\n}\n\n@book{rumelhart,\nauthor = {Rumelhart, D.},\nbooktitle = {Representation and understanding},\npublisher = {New York, Academic},\ntitle = {Notes on a schema for stories },\nyear = {1975},\n}\n@book{schank,\nauthor = {Schank, R.},\nbooktitle = {Representation and understanding},\npublisher = {New York, Academic},\ntitle = {The structure of episodes in memory},\nyear = {1975},\n}\n\n@book{Anscombe1957-ANSI-2,\n\tpublisher = {Cambridge, Mass.: Harvard University Press},\n\tyear = {1957},\n\ttitle = {Intention},\n\teditor = {},\n\tauthor = {G. E. M. Anscombe}\n}\n@article{mturk_article,\nauthor = {Matherly, Ted},\nyear = {2018},\nmonth = {07},\npages = {},\ntitle = {A Panel For Lemons? Positivity bias, reputation systems and data quality on MTurk},\nvolume = {53},\njournal = {European Journal of Marketing},\ndoi = {10.1108/EJM-07-2017-0491}\n}\n\n@article{trabasso,\nauthor = { Tom Trabasso and Soyoung Suh },\ntitle = {Understanding text: Achieving explanatory coherence through on\u2010line inferences and mental operations in working memory},\njournal = {Discourse Processes},\nvolume = {16},\nnumber = {1-2},\npages = {3-34},\nyear = {1993},\npublisher = {Routledge},\ndoi = {10.1080/01638539309544827},\nURL = { \n https://doi.org/10.1080/01638539309544827\n},\neprint = { \n https://doi.org/10.1080/01638539309544827\n}\n}\n\n@article{pennington,\nauthor = {Pennington, Nancy and Hastie, Reid},\nyear = {1992},\nmonth = {02},\npages = {189-206},\ntitle = {Explaining the Evidence: Tests of the Story Model for Juror Decision Making},\nvolume = {62},\njournal = {Journal of Personality and Social Psychology},\ndoi = {10.1037/0022-3514.62.2.189}\n}\n@incollection{BOWER19822,\ntitle = {Plans and Goals in Understanding Episodes},\neditor = {August Flammer and Walter Kintsch},\nseries = {Advances in Psychology},\npublisher = {North-Holland},\nvolume = {8},\npages = {2-15},\nyear = {1982},\nbooktitle = {Discourse Processing},\nissn = {0166-4115},\ndoi = {https://doi.org/10.1016/S0166-4115(08)62676-X},\nurl = {https://www.sciencedirect.com/science/article/pii/S016641150862676X},\nauthor = {Gordon H. Bower},\nabstract = {People's judgments about important elements in narrative episodes were studied. Subjects chose the protagonis\u0165s goal as most important, then actions, outcomes, complications, and lastly background and modifiers. A statemen\u0165s importance correlated with its likelihood of recall and inclusion in a summary. Relating episodes to a Plan schema, further experiments found that readers take longer to comprehend an action in light of a goal the greater the \u201cdistance\u201d between them in a goal hierarchy. Furthermore, the time to comprehend a character's action increases the more independent goals the reader is monitoring for that character.}\n}\n\n \narticle{BRUCE1980295,\ntitle = {Analysis of interacting plans as a guide to the understanding of story structure},\njournal = {Poetics},\nvolume = {9},\nnumber = {1},\npages = {295-311},\nyear = {1980},\nnote = {Special Issue Story Comprehension},\nissn = {0304-422X},\ndoi = {https://doi.org/10.1016/0304-422X(80)90024-8},\nurl = {https://www.sciencedirect.com/science/article/pii/0304422X80900248},\nauthor = {Bertram Bruce},\nabstract = {A complete analysis of a story's structure must rely on more than a simple grammar of story components; it requires a consideration of the characters' plans as they are stated or implied in the story. Furthermore, it must be recognized that these plans are characters' beliefs, and that beliefs about interactions among plans are crucial determinants of a story's structure. A notation system for representing interacting plans is presented here and applied to a simple children's tale about a fox and a rooster. Various phenomena that appear in this story are described. For example, a character in conflict with another may try to conceal his or her real plans or try to deceive the other character into acting in a way that serves his or her own interest at the other's expense. A character may thus construct a plan that is intended to be believed by the other, but is not actually carried out. Such a virtual plan plays a central role in the fox and rooster story and is common in other stories in which characters interact.}\n}\n@article{article_goal_setting,\nauthor = {Locke, Edwin and Latham, Gary},\nyear = {1991},\nmonth = {04},\npages = {},\ntitle = {A Theory of Goal Setting \\& Task Performance},\nvolume = {16},\njournal = {The Academy of Management Review},\ndoi = {10.2307/258875}\n}\n@misc{poque,\n doi = {10.48550/ARXIV.2212.02629},\n \n url = {https://arxiv.org/abs/2212.02629},\n \n author = {Vallurupalli, Sai and Ghosh, Sayontan and Erk, Katrin and Balasubramanian, Niranjan and Ferraro, Francis},\n \n keywords = {Computation and Language (cs.CL), FOS: Computer and information sciences, FOS: Computer and information sciences},\n \n title = {POQue: Asking Participant-specific Outcome Questions for a Deeper Understanding of Complex Events},\n \n publisher = {arXiv},\n \n year = {2022},\n \n copyright = {Creative Commons Attribution 4.0 International}\n}\n\n\n@article{locke_2002,\nauthor = {Locke, Edwin and Latham, Gary},\nyear = {2002},\npages = {705-717},\ntitle = {Building a practically useful theory of goal setting and task motivation: A 35-year odyssey},\nvolume = {57(9)},\njournal = {American Psychologist},\ndoi = {10.1037/0003-066X.57.9.705}\n}\n@inproceedings{zhang-etal-2020-reasoning,\n title = \"Reasoning about Goals, Steps, and Temporal Ordering with {W}iki{H}ow\",\n author = \"Zhang, Li and\n Lyu, Qing and\n Callison-Burch, Chris\",\n booktitle = \"Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing (EMNLP)\",\n month = nov,\n year = \"2020\",\n address = \"Online\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2020.emnlp-main.374\",\n doi = \"10.18653/v1/2020.emnlp-main.374\",\n pages = \"4630--4639\",\n abstract = \"We propose a suite of reasoning tasks on two types of relations between procedural events: goal-step relations ({``}learn poses{''} is a step in the larger goal of {``}doing yoga{''}) and step-step temporal relations ({``}buy a yoga mat{''} typically precedes {``}learn poses{''}). We introduce a dataset targeting these two relations based on wikiHow, a website of instructional how-to articles. Our human-validated test set serves as a reliable benchmark for common-sense inference, with a gap of about 10{\\%} to 20{\\%} between the performance of state-of-the-art transformer models and human performance. Our automatically-generated training set allows models to effectively transfer to out-of-domain tasks requiring knowledge of procedural events, with greatly improved performances on SWAG, Snips, and Story Cloze Test in zero- and few-shot settings.\",\n}\n\n@inproceedings{lyu-etal-2021-goal,\n title = \"Goal-Oriented Script Construction\",\n author = \"Lyu, Qing and\n Zhang, Li and\n Callison-Burch, Chris\",\n booktitle = \"Proceedings of the 14th International Conference on Natural Language Generation\",\n month = aug,\n year = \"2021\",\n address = \"Aberdeen, Scotland, UK\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2021.inlg-1.19\",\n pages = \"184--200\",\n abstract = \"The knowledge of scripts, common chains of events in stereotypical scenarios, is a valuable asset for task-oriented natural language understanding systems. We propose the Goal-Oriented Script Construction task, where a model produces a sequence of steps to accomplish a given goal. We pilot our task on the first multilingual script learning dataset supporting 18 languages collected from wikiHow, a website containing half a million how-to articles. For baselines, we consider both a generation-based approach using a language model and a retrieval-based approach by first retrieving the relevant steps from a large candidate pool and then ordering them. We show that our task is practical, feasible but challenging for state-of-the-art Transformer models, and that our methods can be readily deployed for various other datasets and domains with decent zero-shot performance.\",\n}\n@article{locke_1981,\nauthor = {Locke, Edwin and Shaw, Karyll and Saari, Lise and Latham, Gary},\nyear = {1981},\nmonth = {07},\npages = {125-152},\ntitle = {Goal Setting and Task Performance: 1969\u20131980},\nvolume = {90},\njournal = {Psychological Bulletin},\ndoi = {10.1037/0033-2909.90.1.125}\n}\n@article{BINSWANGER1991154,\ntitle = {Volition as cognitive self-regulation},\njournal = {Organizational Behavior and Human Decision Processes},\nvolume = {50},\nnumber = {2},\npages = {154-178},\nyear = {1991},\nnote = {Theories of Cognitive Self-Regulation},\nissn = {0749-5978},\ndoi = {https://doi.org/10.1016/0749-5978(91)90019-P},\nurl = {https://www.sciencedirect.com/science/article/pii/074959789190019P},\nauthor = {Harry Binswanger},\nabstract = {Living organisms exhibit various levels of self-regulation, the highest of which is man's ability to regulate the operation of his conceptual faculty. Ayn Rand's theory of free will, the basis of this article, identifies this level of self-regulation with volition. The locus of direct volitional choice is placed in the choice \u201cto think or not to think,\u201d where thinking is understood as rational, purposefully directed cognition. The nature of this choice is analyzed in detail, with special emphasis on Rand's concept of mental \u201cfocus.\u201d The epistemological status of the theory is discussed, including the role of introspective evidence in its behalf. It is argued that one's volitional control over one's own thinking has the status of an axiom, and that any attempted denial of this control is self-refuting.}\n}\n@InProceedings{gordon2013reporting,\n title\t\t= {Reporting bias and knowledge acquisition},\n author\t= {Gordon, Jonathan and {Van Durme}, Benjamin},\n booktitle\t= {Proceedings of the 2013 workshop on Automated knowledge\n\t\t base construction},\n pages\t\t= {25--30},\n year\t\t= {2013},\n organization\t= {ACM}\n}\n@misc{ouyang2022training,\n title={Training language models to follow instructions with human feedback}, \n author={Long Ouyang and Jeff Wu and Xu Jiang and Diogo Almeida and Carroll L. Wainwright and Pamela Mishkin and Chong Zhang and Sandhini Agarwal and Katarina Slama and Alex Ray and John Schulman and Jacob Hilton and Fraser Kelton and Luke Miller and Maddie Simens and Amanda Askell and Peter Welinder and Paul Christiano and Jan Leike and Ryan Lowe},\n year={2022},\n eprint={2203.02155},\n archivePrefix={arXiv},\n primaryClass={cs.CL}\n}\n@book{gollwitzer_and_bargh,\ntitle = \"The psychology of action: Linking cognition and motivation to behavior\",\nkeywords = \"Motivation (Psychology), Motivation (Psychologie), Comportement humain, Motivation, Motivation (psychologie), Cognition, Doelgericht gedrag, Motivatie, Behavior, Human behavior\",\neditor = \"Gollwitzer, {Peter M} and Bargh, {John A}\",\nnote = \"Includes bibliographical references and index\",\nyear = \"1996\",\nlanguage = \"English (US)\",\nisbn = \"1572300329\",\npublisher = \"Guilford Press\",\n\n}\n@misc{west2022symbolic,\n title={Symbolic Knowledge Distillation: from General Language Models to Commonsense Models}, \n author={Peter West and Chandra Bhagavatula and Jack Hessel and Jena D. Hwang and Liwei Jiang and Ronan Le Bras and Ximing Lu and Sean Welleck and Yejin Choi},\n year={2022},\n eprint={2110.07178},\n archivePrefix={arXiv},\n primaryClass={cs.CL}\n}\n@inproceedings{\nWelleck2020Neural,\ntitle={Neural Text Generation With Unlikelihood Training},\nauthor={Sean Welleck and Ilia Kulikov and Stephen Roller and Emily Dinan and Kyunghyun Cho and Jason Weston},\nbooktitle={International Conference on Learning Representations},\nyear={2020},\nurl={https://openreview.net/forum?id=SJeYe0NtvH}\n}\n@inproceedings{Wiegreffe2021ReframingHC,\n title={Reframing Human-AI Collaboration for Generating Free-Text Explanations},\n author={Sarah Wiegreffe and Jack Hessel and Swabha Swayamdipta and Mark O. Riedl and Yejin Choi},\n booktitle={North American Chapter of the Association for Computational Linguistics},\n year={2021}\n}\n@article{McCoy2019RightFT,\n title={Right for the Wrong Reasons: Diagnosing Syntactic Heuristics in Natural Language Inference},\n author={R. Thomas McCoy and Ellie Pavlick and Tal Linzen},\n journal={ArXiv},\n year={2019},\n volume={abs/1902.01007}\n}\n \n@book{graesser_how,\nauthor = {Arthur C. Graesser and Brent Olde and Bianca Klettke},\ntitle = {How does the mind construct and represent stories?},\nbooktitle = { Narrative impact: Social and cognitive foundations},\npages = {229-262},\nyear = {2020},\npublisher = {Lawrence Erlbaum Associates Publishers}\n}\n@article{graesser,\nauthor = {Graesser, Arthur and Singer, Murray and Trabasso, Tom},\nyear = {1994},\nmonth = {08},\npages = {371-95},\ntitle = {Constructing Inferences During Narrative Text Comprehension},\nvolume = {101},\njournal = {Psychological review},\ndoi = {10.1037/0033-295X.101.3.371}\n}\n@article{wagerman,\nauthor = {Wagerman, Seth and Funder, David},\nyear = {2008},\npages = {27-42},\ntitle = {Situations},\njournal = {Cambridge handbook of personality psychology}, publisher = {Cambridge, England: Cambridge University Press.},\n}\n@article{Furr,\nauthor = {Furr, Robert and Funder, David},\nyear = {2021},\npages = {667-685},\ntitle = {Persons, situations, and person\u2013situation interactions},\njournal = {Handbook of personality: Theory and research},\npublisher = {The Guilford Press.},\n}\n@Book{schank+abelson77,\n author = \"Roger C. Schank and Robert P. Abelson\",\n title = \"Scripts, Plans, Goals and Understanding: an Inquiry\n into Human Knowledge Structures\",\n publisher = \"L. Erlbaum\",\n year = \"1977\",\n address = \"Hillsdale, NJ\",\n keywords = \"PAM, SAM, TALE-SPIN, causality, conceptual dependency,\n goals, plans, scripts, semantic primitive, text\n understanding\",\n}\n@book{harmon,\nauthor = {Mary E. Harmon},\ntitle = {Factors affecting the activation of predictive inferences},\nbooktitle = { Ph.D. Dissertation},\nyear = {2005},\npublisher = {Univeristy of New Hampshire}\n}\n@article{KEEFE1993446,\ntitle = {The Time Course and Durability of Predictive Inferences},\njournal = {Journal of Memory and Language},\nvolume = {32},\nnumber = {4},\npages = {446-463},\nyear = {1993},\nissn = {0749-596X},\ndoi = {https://doi.org/10.1006/jmla.1993.1024},\nurl = {https://www.sciencedirect.com/science/article/pii/S0749596X83710247},\nauthor = {Keefe, Dennis E. and McDaniel, Mark A.},\nabstract = {The latency to pronounce a probe word has been used as an index of whether or not inferences about highly predictable events are drawn during reading. Using this index. Potts, Keenan, and Golding (1988, Journal of Memory and Language, 27, 399-415) failed to find evidence for predictive (forward) inferencing. Using the same index, we (1) replicated the Potts et al. finding, (2) found that the predictive condition produced significant priming in pronouncing the probe when the probe was presented immediately after the predictive sentence, (3) demonstrated an absence of such priming when the probe was delayed by an interval not filled with additional text input, and (4) found significant priming regardless of when the probe was presented when the sentence format increased processing difficulty. These results are taken as support for the view that predictive inferences are temporarily drawn and subsequently deactivated. With more difficult material, deactivation of predictive inferences may either be delayed or prevented. The mechanisms of such deactivation are discussed in light of the results.}\n}\n@article{Setiya2011INTENTIONPA,\n title={INTENTION, PLANS, AND ETHICAL RATIONALISM},\n author={Kieran Setiya},\n journal = {Rational and Social Agency: The Philosophy of Michael Bratman},\n publisher = {Oxford: Oxford University Press},\n year={2011},\n pages={56-82}\n}\n@article{c25f888d-2967-30b4-883d-b186f3d67aab,\n ISSN = {00346632},\n URL = {http://www.jstor.org/stable/20126812},\n author = {Gilbert Harman},\n journal = {The Review of Metaphysics},\n number = {3},\n pages = {431--463},\n publisher = {Philosophy Education Society Inc.},\n title = {Practical Reasoning},\n urldate = {2023-07-10},\n volume = {29},\n year = {1976}\n}\n@article{Davidson1963-DAVARA-6,\n\tpublisher = {Journal of Philosophy},\n\tnumber = {23},\n\tyear = {1963},\n\tdoi = {10.2307/2023177},\n\tvolume = {60},\n\tauthor = {Donald Davidson},\n\ttitle = {Actions, Reasons, and Causes},\n\tpages = {685},\n\tjournal = {Journal of Philosophy}\n}\n\n@article{Grice1971-GRIIAU,\n\tjournal = {Proceedings of the British Academy},\n\tyear = {1971},\n\tauthor = {H. P. Grice},\n\ttitle = {Intention and Uncertainty},\n\tpages = {263--279},\n\tvolume = {57},\n\tpublisher = {London}\n}\n@article{zacks_event,\nauthor = {Zacks, Jeffrey M.},\ntitle = {Event Perception and Memory},\njournal = {Annual Review of Psychology},\nvolume = {71},\nnumber = {1},\npages = {165-191},\nyear = {2020},\ndoi = {10.1146/annurev-psych-010419-051101},\n note ={PMID: 31905113},\n\nURL = { \n \n https://doi.org/10.1146/annurev-psych-010419-051101\n \n \n\n},\neprint = { \n \n https://doi.org/10.1146/annurev-psych-010419-051101\n \n \n\n}\n,\n abstract = { Events make up much of our lived experience, and the perceptual mechanisms that represent events in experience have pervasive effects on action control, language use, and remembering. Event representations in both perception and memory have rich internal structure and connections one to another, and both are heavily informed by knowledge accumulated from previous experiences. Event perception and memory have been identified with specific computational and neural mechanisms, which show protracted development in childhood and are affected by language use, expertise, and brain disorders and injuries. Current theoretical approaches focus on the mechanisms by which events are segmented from ongoing experience, and emphasize the common coding of events for perception, action, and memory. Abetted by developments in eye-tracking, neuroimaging, and computer science, research on event perception and memory is moving from small-scale laboratory analogs to the complexity of events in the wild. }\n}\n\n@inproceedings{hovy-yang-2021-importance,\n title = \"The Importance of Modeling Social Factors of Language: Theory and Practice\",\n author = \"Hovy, Dirk and\n Yang, Diyi\",\n booktitle = \"Proceedings of the 2021 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies\",\n month = jun,\n year = \"2021\",\n address = \"Online\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2021.naacl-main.49\",\n doi = \"10.18653/v1/2021.naacl-main.49\",\n pages = \"588--602\",\n abstract = \"Natural language processing (NLP) applications are now more powerful and ubiquitous than ever before. With rapidly developing (neural) models and ever-more available data, current NLP models have access to more information than any human speaker during their life. Still, it would be hard to argue that NLP models have reached human-level capacity. In this position paper, we argue that the reason for the current limitations is a focus on information content while ignoring language{'}s social factors. We show that current NLP systems systematically break down when faced with interpreting the social factors of language. This limits applications to a subset of information-related tasks and prevents NLP from reaching human-level performance. At the same time, systems that incorporate even a minimum of social factors already show remarkable improvements. We formalize a taxonomy of seven social factors based on linguistic theory and exemplify current failures and emerging successes for each of them. We suggest that the NLP community address social factors to get closer to the goal of human-like language understanding.\",\n}\n@inproceedings{rahimtoroghi-etal-2017-modelling,\n title = \"Modelling Protagonist Goals and Desires in First-Person Narrative\",\n author = \"Rahimtoroghi, Elahe and\n Wu, Jiaqi and\n Wang, Ruimin and\n Anand, Pranav and\n Walker, Marilyn\",\n booktitle = \"Proceedings of the 18th Annual {SIG}dial Meeting on Discourse and Dialogue\",\n month = aug,\n year = \"2017\",\n address = {Saarbr{\\\"u}cken, Germany},\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/W17-5543\",\n doi = \"10.18653/v1/W17-5543\",\n pages = \"360--369\",\n abstract = \"Many genres of natural language text are narratively structured, a testament to our predilection for organizing our experiences as narratives. There is broad consensus that understanding a narrative requires identifying and tracking the goals and desires of the characters and their narrative outcomes. However, to date, there has been limited work on computational models for this problem. We introduce a new dataset, DesireDB, which includes gold-standard labels for identifying statements of desire, textual evidence for desire fulfillment, and annotations for whether the stated desire is fulfilled given the evidence in the narrative context. We report experiments on tracking desire fulfillment using different methods, and show that LSTM Skip-Thought model achieves F-measure of 0.7 on our corpus.\",\n}\n\n\n@inproceedings{Mostafazadeh2016ACA,\n title={A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories},\n author={N. Mostafazadeh and Nathanael Chambers and Xiaodong He and Devi Parikh and Dhruv Batra and Lucy Vanderwende and Pushmeet Kohli and James F. Allen},\n booktitle={NAACL},\n year={2016}\n}\n@article{mandar_spanbert,\n author = {Mandar Joshi and\n Danqi Chen and\n Yinhan Liu and\n Daniel S. Weld and\n Luke Zettlemoyer and\n Omer Levy},\n title = {SpanBERT: Improving Pre-training by Representing and Predicting Spans},\n journal = {CoRR},\n volume = {abs/1907.10529},\n year = {2019},\n url = {http://arxiv.org/abs/1907.10529},\n eprinttype = {arXiv},\n eprint = {1907.10529},\n timestamp = {Thu, 01 Aug 2019 08:59:33 +0200},\n biburl = {https://dblp.org/rec/journals/corr/abs-1907-10529.bib},\n bibsource = {dblp computer science bibliography, https://dblp.org}\n}\n\n@article{DBLP:journals/corr/abs-1711-05101,\n author = {Ilya Loshchilov and\n Frank Hutter},\n title = {Fixing Weight Decay Regularization in Adam},\n journal = {CoRR},\n volume = {abs/1711.05101},\n year = {2017},\n url = {http://arxiv.org/abs/1711.05101},\n eprinttype = {arXiv},\n eprint = {1711.05101},\n timestamp = {Mon, 13 Aug 2018 16:48:18 +0200},\n biburl = {https://dblp.org/rec/journals/corr/abs-1711-05101.bib},\n bibsource = {dblp computer science bibliography, https://dblp.org}\n}\n\n@article{Sandhaus2008Nyt,\n title={The new york times annotated corpus},\n author={Sandhaus, Evan},\n journal={Linguistic Data Consortium, Philadelphia},\n volume={6},\n number={12},\n pages={e26752},\n year={2008}\n}\n@inproceedings{wei2022finetuned,\ntitle={Finetuned Language Models are Zero-Shot Learners},\nauthor={Jason Wei and Maarten Bosma and Vincent Zhao and Kelvin Guu and Adams Wei Yu and Brian Lester and Nan Du and Andrew M. Dai and Quoc V Le},\nbooktitle={International Conference on Learning Representations},\nyear={2022},\nurl={https://openreview.net/forum?id=gEZrGCozdqR}\n}\n@article{longpre2023flan,\n title={The Flan Collection: Designing Data and Methods for Effective Instruction Tuning},\n author={Longpre, Shayne and Hou, Le and Vu, Tu and Webson, Albert and Chung, Hyung Won and Tay, Yi and Zhou, Denny and Le, Quoc V and Zoph, Barret and Wei, Jason and others},\n journal={arXiv preprint arXiv:2301.13688},\n year={2023}\n}\n@misc{chung2022scaling,\n title={Scaling Instruction-Finetuned Language Models}, \n author={Hyung Won Chung and Le Hou and Shayne Longpre and Barret Zoph and Yi Tay and William Fedus and Yunxuan Li and Xuezhi Wang and Mostafa Dehghani and Siddhartha Brahma and Albert Webson and Shixiang Shane Gu and Zhuyun Dai and Mirac Suzgun and Xinyun Chen and Aakanksha Chowdhery and Alex Castro-Ros and Marie Pellat and Kevin Robinson and Dasha Valter and Sharan Narang and Gaurav Mishra and Adams Yu and Vincent Zhao and Yanping Huang and Andrew Dai and Hongkun Yu and Slav Petrov and Ed H. Chi and Jeff Dean and Jacob Devlin and Adam Roberts and Denny Zhou and Quoc V. Le and Jason Wei},\n year={2022},\n eprint={2210.11416},\n archivePrefix={arXiv},\n primaryClass={cs.LG}\n}\n\n@article{referee,\n doi = {10.48550/ARXIV.2210.13800},\n \n url = {https://arxiv.org/abs/2210.13800},\n journal = {Empirical Methods in Natural Language Processing 2022 (EMNLP 2022)},\n author = {Sclar, Melanie and West, Peter and Kumar, Sachin and Tsvetkov, Yulia and Choi, Yejin},\n \n keywords = {Computation and Language (cs.CL), Artificial Intelligence (cs.AI), Machine Learning (cs.LG), FOS: Computer and information sciences, FOS: Computer and information sciences},\n \n title = {Referee: Reference-Free Sentence Summarization with Sharper Controllability through Symbolic Knowledge Distillation},\n \n publisher = {arXiv},\n \n year = {2022},\n \n copyright = {Creative Commons Attribution 4.0 International}\n}\n\n@inproceedings{Saha2022AreHE,\n title={Are Hard Examples also Harder to Explain? A Study with Human and Model-Generated Explanations},\n author={Swarnadeep Saha and Peter Hase and Nazneen Rajani and Mohit Bansal},\n booktitle={Conference on Empirical Methods in Natural Language Processing},\n year={2022}\n}\n@misc{liu2022wanli,\n title={WANLI: Worker and AI Collaboration for Natural Language Inference Dataset Creation}, \n author={Alisa Liu and Swabha Swayamdipta and Noah A. Smith and Yejin Choi},\n year={2022},\n eprint={2201.05955},\n archivePrefix={arXiv},\n primaryClass={cs.CL}\n}\n@inproceedings{gururangan-etal-2018-annotation,\n title = \"Annotation Artifacts in Natural Language Inference Data\",\n author = \"Gururangan, Suchin and\n Swayamdipta, Swabha and\n Levy, Omer and\n Schwartz, Roy and\n Bowman, Samuel and\n Smith, Noah A.\",\n booktitle = \"Proceedings of the 2018 Conference of the North {A}merican Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 2 (Short Papers)\",\n month = jun,\n year = \"2018\",\n address = \"New Orleans, Louisiana\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/N18-2017\",\n doi = \"10.18653/v1/N18-2017\",\n pages = \"107--112\",\n abstract = \"Large-scale datasets for natural language inference are created by presenting crowd workers with a sentence (premise), and asking them to generate three new sentences (hypotheses) that it entails, contradicts, or is logically neutral with respect to. We show that, in a significant portion of such data, this protocol leaves clues that make it possible to identify the label by looking only at the hypothesis, without observing the premise. Specifically, we show that a simple text categorization model can correctly classify the hypothesis alone in about 67{\\%} of SNLI (Bowman et. al, 2015) and 53{\\%} of MultiNLI (Williams et. al, 2017). Our analysis reveals that specific linguistic phenomena such as negation and vagueness are highly correlated with certain inference classes. Our findings suggest that the success of natural language inference models to date has been overestimated, and that the task remains a hard open problem.\",\n}\n@inproceedings{clark-etal-2021-thats,\n title = \"All That{'}s {`}Human{'} Is Not Gold: Evaluating Human Evaluation of Generated Text\",\n author = \"Clark, Elizabeth and\n August, Tal and\n Serrano, Sofia and\n Haduong, Nikita and\n Gururangan, Suchin and\n Smith, Noah A.\",\n booktitle = \"Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 1: Long Papers)\",\n month = aug,\n year = \"2021\",\n address = \"Online\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2021.acl-long.565\",\n doi = \"10.18653/v1/2021.acl-long.565\",\n pages = \"7282--7296\",\n abstract = \"Human evaluations are typically considered the gold standard in natural language generation, but as models{'} fluency improves, how well can evaluators detect and judge machine-generated text? We run a study assessing non-experts{'} ability to distinguish between human- and machine-authored text (GPT2 and GPT3) in three domains (stories, news articles, and recipes). We find that, without training, evaluators distinguished between GPT3- and human-authored text at random chance level. We explore three approaches for quickly training evaluators to better identify GPT3-authored text (detailed instructions, annotated examples, and paired examples) and find that while evaluators{'} accuracy improved up to 55{\\%}, it did not significantly improve across the three domains. Given the inconsistent results across text domains and the often contradictory reasons evaluators gave for their judgments, we examine the role untrained human evaluations play in NLG evaluation and provide recommendations to NLG researchers for improving human evaluations of text generated from state-of-the-art models.\",\n}\n\n\n@inproceedings{\nHoltzman2020The,\ntitle={The Curious Case of Neural Text Degeneration},\nauthor={Ari Holtzman and Jan Buys and Li Du and Maxwell Forbes and Yejin Choi},\nbooktitle={International Conference on Learning Representations},\nyear={2020},\nurl={https://openreview.net/forum?id=rygGQyrFvH}\n}\n@inproceedings{Model_compression,\nauthor = {Bucilu\u01ce, Cristian and Caruana, Rich and Niculescu-Mizil, Alexandru},\ntitle = {Model Compression},\nyear = {2006},\nisbn = {1595933395},\npublisher = {Association for Computing Machinery},\naddress = {New York, NY, USA},\nurl = {https://doi.org/10.1145/1150402.1150464},\ndoi = {10.1145/1150402.1150464},\nabstract = {Often the best performing supervised learning models are ensembles of hundreds or thousands of base-level classifiers. Unfortunately, the space required to store this many classifiers, and the time required to execute them at run-time, prohibits their use in applications where test sets are large (e.g. Google), where storage space is at a premium (e.g. PDAs), and where computational power is limited (e.g. hea-ring aids). We present a method for \"compressing\" large, complex ensembles into smaller, faster models, usually without significant loss in performance.},\nbooktitle = {Proceedings of the 12th ACM SIGKDD International Conference on Knowledge Discovery and Data Mining},\npages = {535\u2013541},\nnumpages = {7},\nkeywords = {model compression, supervised learning},\nlocation = {Philadelphia, PA, USA},\nseries = {KDD '06}\n}\n\n \n\n\n@misc{hinton2015distilling,\n title={Distilling the Knowledge in a Neural Network}, \n author={Geoffrey Hinton and Oriol Vinyals and Jeff Dean},\n year={2015},\n eprint={1503.02531},\n archivePrefix={arXiv},\n primaryClass={stat.ML}\n}\n@article{reasoned_action,\nauthor = {Ajzen, Icek and Kruglanski, Arie},\nyear = {2019},\nmonth = {07},\npages = {},\ntitle = {Reasoned action in the service of goal pursuit},\nvolume = {126},\njournal = {Psychological Review},\ndoi = {10.1037/rev0000155}\n}\n@article{Geirhos_2020,\n\tdoi = {10.1038/s42256-020-00257-z},\n \n\turl = {https://doi.org/10.1038%2Fs42256-020-00257-z},\n \n\tyear = 2020,\n\tmonth = {nov},\n \n\tpublisher = {Springer Science and Business Media {LLC}\n},\n \n\tvolume = {2},\n \n\tnumber = {11},\n \n\tpages = {665--673},\n \n\tauthor = {Robert Geirhos and J\u00f6rn-Henrik Jacobsen and Claudio Michaelis and Richard Zemel and Wieland Brendel and Matthias Bethge and Felix A. Wichmann},\n \n\ttitle = {Shortcut learning in deep neural networks},\n \n\tjournal = {Nature Machine Intelligence}\n}\n@inproceedings{geva-etal-2019-modeling,\n title = \"Are We Modeling the Task or the Annotator? An Investigation of Annotator Bias in Natural Language Understanding Datasets\",\n author = \"Geva, Mor and\n Goldberg, Yoav and\n Berant, Jonathan\",\n booktitle = \"Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP)\",\n month = nov,\n year = \"2019\",\n address = \"Hong Kong, China\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/D19-1107\",\n doi = \"10.18653/v1/D19-1107\",\n pages = \"1161--1166\",\n abstract = \"Crowdsourcing has been the prevalent paradigm for creating natural language understanding datasets in recent years. A common crowdsourcing practice is to recruit a small number of high-quality workers, and have them massively generate examples. Having only a few workers generate the majority of examples raises concerns about data diversity, especially when workers freely generate sentences. In this paper, we perform a series of experiments showing these concerns are evident in three recent NLP datasets. We show that model performance improves when training with annotator identifiers as features, and that models are able to recognize the most productive annotators. Moreover, we show that often models do not generalize well to examples from annotators that did not contribute to the training set. Our findings suggest that annotator bias should be monitored during dataset creation, and that test set annotators should be disjoint from training set annotators.\",\n}\n\n\n@inproceedings{lobue-yates-2011-types,\n title = \"Types of Common-Sense Knowledge Needed for Recognizing Textual Entailment\",\n author = \"LoBue, Peter and\n Yates, Alexander\",\n booktitle = \"Proceedings of the 49th Annual Meeting of the Association for Computational Linguistics: Human Language Technologies\",\n month = jun,\n year = \"2011\",\n address = \"Portland, Oregon, USA\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/P11-2057\",\n pages = \"329--334\",\n}\n@article{kruglanski_article,\nauthor = {Kruglanski, Arie and Shah, James and Fishbach, Ayelet and Friedman, Ron and Chun, Woo Young and Sleeth-Keppler, David},\nyear = {2002},\nmonth = {12},\npages = {331-378},\ntitle = {A theory of goal systems},\nvolume = {34},\nisbn = {9780120152346},\njournal = {Advances in Experimental Social Psychology - ADVAN EXP SOC PSYCHOL},\ndoi = {10.1016/S0065-2601(02)80008-9}\n} \n@article{Goal_Constructs_in_psychology,\nauthor = {Austin, J. T. and Vancouver, J. B.},\ntitle = {Goal constructs in psychology: Structure, process, and content. },\njournal = {Psychological Bulletin},\nvolume = {120},\nnumber = {3},\npages = {338-375},\nyear = {1996}\n}\n@article{Hagmayer2017CausalKA,\n title={Causal knowledge and reasoning in decision making},\n author={York Hagmayer and Cilia Witteman},\n journal={Psychology of Learning and Motivation},\n year={2017},\n volume={67},\n pages={95-134}\n}\n@article{JohnsonLaird,\nauthor = {Johnson-Laird, Phil and Shafir, Eldar},\nyear = {1993},\nmonth = {11},\npages = {1-9},\ntitle = {The interaction between reasoning and decision making: an introduction},\nvolume = {49},\njournal = {Cognition},\ndoi = {10.1016/0010-0277(93)90033-R}\n}\n\n@article{PETROPOULOS2022705,\ntitle = {Forecasting: theory and practice},\njournal = {International Journal of Forecasting},\nvolume = {38},\nnumber = {3},\npages = {705-871},\nyear = {2022},\nissn = {0169-2070},\ndoi = {https://doi.org/10.1016/j.ijforecast.2021.11.001},\nurl = {https://www.sciencedirect.com/science/article/pii/S0169207021001758},\nauthor = {Fotios Petropoulos and Daniele Apiletti and Vassilios Assimakopoulos and Mohamed Zied Babai and Devon K. Barrow and Souhaib {Ben Taieb} and Christoph Bergmeir and Ricardo J. Bessa and Jakub Bijak and John E. Boylan and Jethro Browell and Claudio Carnevale and Jennifer L. Castle and Pasquale Cirillo and Michael P. Clements and Clara Cordeiro and Fernando Luiz {Cyrino Oliveira} and Shari {De Baets} and Alexander Dokumentov and Joanne Ellison and Piotr Fiszeder and Philip Hans Franses and David T. Frazier and Michael Gilliland and M. Sinan G\u00f6n\u00fcl and Paul Goodwin and Luigi Grossi and Yael Grushka-Cockayne and Mariangela Guidolin and Massimo Guidolin and Ulrich Gunter and Xiaojia Guo and Renato Guseo and Nigel Harvey and David F. Hendry and Ross Hollyman and Tim Januschowski and Jooyoung Jeon and Victor Richmond R. Jose and Yanfei Kang and Anne B. Koehler and Stephan Kolassa and Nikolaos Kourentzes and Sonia Leva and Feng Li and Konstantia Litsiou and Spyros Makridakis and Gael M. Martin and Andrew B. Martinez and Sheik Meeran and Theodore Modis and Konstantinos Nikolopoulos and Dilek \u00d6nkal and Alessia Paccagnini and Anastasios Panagiotelis and Ioannis Panapakidis and Jose M. Pav\u00eda and Manuela Pedio and Diego J. Pedregal and Pierre Pinson and Patr\u00edcia Ramos and David E. Rapach and J. James Reade and Bahman Rostami-Tabar and Micha\u0142 Rubaszek and Georgios Sermpinis and Han Lin Shang and Evangelos Spiliotis and Aris A. Syntetos and Priyanga Dilini Talagala and Thiyanga S. Talagala and Len Tashman and Dimitrios Thomakos and Thordis Thorarinsdottir and Ezio Todini and Juan Ram\u00f3n {Trapero Arenas} and Xiaoqian Wang and Robert L. Winkler and Alisa Yusupova and Florian Ziel},\nkeywords = {Review, Encyclopedia, Methods, Applications, Principles, Time series, Prediction},\nabstract = {Forecasting has always been at the forefront of decision making and planning. The uncertainty that surrounds the future is both exciting and challenging, with individuals and organisations seeking to minimise risks and maximise utilities. The large number of forecasting applications calls for a diverse set of forecasting methods to tackle real-life challenges. This article provides a non-systematic review of the theory and the practice of forecasting. We provide an overview of a wide range of theoretical, state-of-the-art models, methods, principles, and approaches to prepare, produce, organise, and evaluate forecasts. We then demonstrate how such theoretical concepts are applied in a variety of real-life contexts. We do not claim that this review is an exhaustive list of methods and applications. However, we wish that our encyclopedic presentation will offer a point of reference for the rich work that has been undertaken over the last decades, with some key insights for the future of forecasting theory and practice. Given its encyclopedic nature, the intended mode of reading is non-linear. We offer cross-references to allow the readers to navigate through the various topics. We complement the theoretical concepts and applications covered by large lists of free or open-source software implementations and publicly-available databases.}\n}\n\n@inproceedings{bowman-etal-2015-large,\n title = \"A large annotated corpus for learning natural language inference\",\n author = \"Bowman, Samuel R. and\n Angeli, Gabor and\n Potts, Christopher and\n Manning, Christopher D.\",\n booktitle = \"Proceedings of the 2015 Conference on Empirical Methods in Natural Language Processing\",\n month = sep,\n year = \"2015\",\n address = \"Lisbon, Portugal\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/D15-1075\",\n doi = \"10.18653/v1/D15-1075\",\n pages = \"632--642\",\n}\n\n@inproceedings{williams-etal-2018-broad,\n title = \"A Broad-Coverage Challenge Corpus for Sentence Understanding through Inference\",\n author = \"Williams, Adina and\n Nangia, Nikita and\n Bowman, Samuel\",\n booktitle = \"Proceedings of the 2018 Conference of the North {A}merican Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long Papers)\",\n month = jun,\n year = \"2018\",\n address = \"New Orleans, Louisiana\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/N18-1101\",\n doi = \"10.18653/v1/N18-1101\",\n pages = \"1112--1122\",\n abstract = \"This paper introduces the Multi-Genre Natural Language Inference (MultiNLI) corpus, a dataset designed for use in the development and evaluation of machine learning models for sentence understanding. At 433k examples, this resource is one of the largest corpora available for natural language inference (a.k.a. recognizing textual entailment), improving upon available resources in both its coverage and difficulty. MultiNLI accomplishes this by offering data from ten distinct genres of written and spoken English, making it possible to evaluate systems on nearly the full complexity of the language, while supplying an explicit setting for evaluating cross-genre domain adaptation. In addition, an evaluation using existing machine learning models designed for the Stanford NLI corpus shows that it represents a substantially more difficult task than does that corpus, despite the two showing similar levels of inter-annotator agreement.\",\n}\n\n@article{chen2021can,\n title={Can NLI Models Verify QA Systems' Predictions?},\n author={Chen, Jifan and Choi, Eunsol and Durrett, Greg},\n journal={EMNLP Findings},\n year={2021}\n}\n\n@inproceedings{white-etal-2017-inference,\n title = \"Inference is Everything: Recasting Semantic Resources into a Unified Evaluation Framework\",\n author = \"White, Aaron Steven and\n Rastogi, Pushpendre and\n Duh, Kevin and\n Van Durme, Benjamin\",\n booktitle = \"Proceedings of the Eighth International Joint Conference on Natural Language Processing (Volume 1: Long Papers)\",\n month = nov,\n year = \"2017\",\n address = \"Taipei, Taiwan\",\n publisher = \"Asian Federation of Natural Language Processing\",\n url = \"https://aclanthology.org/I17-1100\",\n pages = \"996--1005\",\n abstract = \"We propose to unify a variety of existing semantic classification tasks, such as semantic role labeling, anaphora resolution, and paraphrase detection, under the heading of Recognizing Textual Entailment (RTE). We present a general strategy to automatically generate one or more sentential hypotheses based on an input sentence and pre-existing manual semantic annotations. The resulting suite of datasets enables us to probe a statistical RTE model{'}s performance on different aspects of semantics. We demonstrate the value of this approach by investigating the behavior of a popular neural network RTE model.\",\n}\n@inproceedings{dagan_rte1,\nauthor = {Dagan, Ido and Glickman, Oren and Magnini, Bernardo},\ntitle = {The PASCAL Recognising Textual Entailment Challenge},\nyear = {2005},\nisbn = {3540334270},\npublisher = {Springer-Verlag},\naddress = {Berlin, Heidelberg},\nurl = {https://doi.org/10.1007/11736790_9},\ndoi = {10.1007/11736790_9},\nabstract = {This paper describes the PASCAL Network of Excellence first Recognising Textual Entailment (RTE-1) Challenge benchmark. The RTE task is defined as recognizing, given two text fragments, whether the meaning of one text can be inferred (entailed) from the other. This application-independent task is suggested as capturing major inferences about the variability of semantic expression which are commonly needed across multiple applications. The Challenge has raised noticeable attention in the research community, attracting 17 submissions from diverse groups, suggesting the generic relevance of the task.},\nbooktitle = {Proceedings of the First International Conference on Machine Learning Challenges: Evaluating Predictive Uncertainty Visual Object Classification, and Recognizing Textual Entailment},\npages = {177\u2013190},\nnumpages = {14},\nlocation = {Southampton, UK},\nseries = {MLCW'05}\n}\n\n@inproceedings{giampiccolo-etal-2007-third,\n title = \"The Third {PASCAL} Recognizing Textual Entailment Challenge\",\n author = \"Giampiccolo, Danilo and\n Magnini, Bernardo and\n Dagan, Ido and\n Dolan, Bill\",\n booktitle = \"Proceedings of the {ACL}-{PASCAL} Workshop on Textual Entailment and Paraphrasing\",\n month = jun,\n year = \"2007\",\n address = \"Prague\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/W07-1401\",\n pages = \"1--9\",\n}\n\n@article{maccartney,\n publisher = \"Stanford University\",\n journal = \"Ph.D. Dissertation\",\n\tauthor = {Bill MacCartney},\n\tyear = {2009},\n\ttitle = {Natural Language Inference},\n\n}\n\n@article{doi:10.1080/0020174X.2019.1684988,\nauthor = {Ladislav Kore\u0148},\ntitle = {Have Mercier and Sperber untied the knot of human reasoning?},\njournal = {Inquiry},\nvolume = {0},\nnumber = {0},\npages = {1-14},\nyear = {2019},\npublisher = {Routledge},\ndoi = {10.1080/0020174X.2019.1684988},\n\nURL = { \n https://doi.org/10.1080/0020174X.2019.1684988\n \n},\neprint = { \n https://doi.org/10.1080/0020174X.2019.1684988\n \n}\n\n}\n@ARTICLE{7933919,\n author={Hoffman, Robert R. and Klein, Gary},\n journal={IEEE Intelligent Systems}, \n title={Explaining Explanation, Part 1: Theoretical Foundations}, \n year={2017},\n volume={32},\n number={3},\n pages={68-73},\n doi={10.1109/MIS.2017.54}}\n@article{article,\nauthor = {D\u00f6rner, Dietrich and Funke, Joachim},\nyear = {2017},\nmonth = {07},\npages = {1-11},\ntitle = {Complex Problem Solving: What It Is and What It Is Not},\nvolume = {8},\njournal = {Frontiers in Psychology},\ndoi = {10.3389/fpsyg.2017.01153}\n}\n@book{cps_book,\nauthor = {Frensch, Peter and Funke, Joachim},\nyear = {1995},\nmonth = {01},\npages = {},\ntitle = {Complex Problem Solving\u2014The European Perspective},\njournal = {Learning to Solve Complex Scientific Problems},\npublisher = {Lawrence Erlbaum Associates}\n}\n@inproceedings{poliak-etal-2018-evaluation,\n title = \"On the Evaluation of Semantic Phenomena in Neural Machine Translation Using Natural Language Inference\",\n author = \"Poliak, Adam and\n Belinkov, Yonatan and\n Glass, James and\n Van Durme, Benjamin\",\n booktitle = \"Proceedings of the 2018 Conference of the North {A}merican Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 2 (Short Papers)\",\n month = jun,\n year = \"2018\",\n address = \"New Orleans, Louisiana\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/N18-2082\",\n doi = \"10.18653/v1/N18-2082\",\n pages = \"513--523\",\n abstract = \"We propose a process for investigating the extent to which sentence representations arising from neural machine translation (NMT) systems encode distinct semantic phenomena. We use these representations as features to train a natural language inference (NLI) classifier based on datasets recast from existing semantic annotations. In applying this process to a representative NMT system, we find its encoder appears most suited to supporting inferences at the syntax-semantics interface, as compared to anaphora resolution requiring world knowledge. We conclude with a discussion on the merits and potential deficiencies of the existing process, and how it may be improved and extended as a broader framework for evaluating semantic coverage\",\n}\n\n\n@article{Korman2018-KORDTE,\n\tjournal = {Journal of the Association for Information Science and Technology},\n\tauthor = {Daniel Z. Korman and Eric Mack and Jacob Jett and Allen H. Renear},\n\tpages = {763--772},\n\tyear = {2018},\n\ttitle = {Defining Textual Entailment},\n\tvolume = {69}\n}\n% explanation based papers\n\n@misc{https://doi.org/10.48550/arxiv.1911.03429,\n doi = {10.48550/ARXIV.1911.03429},\n \n url = {https://arxiv.org/abs/1911.03429},\n \n author = {DeYoung, Jay and Jain, Sarthak and Rajani, Nazneen Fatema and Lehman, Eric and Xiong, Caiming and Socher, Richard and Wallace, Byron C.},\n \n keywords = {Computation and Language (cs.CL), Artificial Intelligence (cs.AI), Machine Learning (cs.LG), FOS: Computer and information sciences, FOS: Computer and information sciences},\n \n title = {ERASER: A Benchmark to Evaluate Rationalized NLP Models},\n \n publisher = {arXiv},\n \n year = {2019},\n \n copyright = {Creative Commons Attribution 4.0 International}\n}\n\n@inproceedings{jansen-etal-2016-whats,\n title = \"What{'}s in an Explanation? Characterizing Knowledge and Inference Requirements for Elementary Science Exams\",\n author = \"Jansen, Peter and\n Balasubramanian, Niranjan and\n Surdeanu, Mihai and\n Clark, Peter\",\n booktitle = \"Proceedings of {COLING} 2016, the 26th International Conference on Computational Linguistics: Technical Papers\",\n month = dec,\n year = \"2016\",\n address = \"Osaka, Japan\",\n publisher = \"The COLING 2016 Organizing Committee\",\n url = \"https://aclanthology.org/C16-1278\",\n pages = \"2956--2965\",\n abstract = \"QA systems have been making steady advances in the challenging elementary science exam domain. In this work, we develop an explanation-based analysis of knowledge and inference requirements, which supports a fine-grained characterization of the challenges. In particular, we model the requirements based on appropriate sources of evidence to be used for the QA task. We create requirements by first identifying suitable sentences in a knowledge base that support the correct answer, then use these to build explanations, filling in any necessary missing information. These explanations are used to create a fine-grained categorization of the requirements. Using these requirements, we compare a retrieval and an inference solver on 212 questions. The analysis validates the gains of the inference solver, demonstrating that it answers more questions requiring complex inference, while also providing insights into the relative strengths of the solvers and knowledge sources. We release the annotated questions and explanations as a resource with broad utility for science exam QA, including determining knowledge base construction targets, as well as supporting information aggregation in automated inference.\",\n}\n@article{10.1093/arisoc/aoac009,\n author = {Borg, Emma},\n title = \"{IX\u2014In Defence of Individual Rationality}\",\n journal = {Proceedings of the Aristotelian Society},\n volume = {122},\n number = {3},\n pages = {195-217},\n year = {2022},\n month = {10},\n abstract = \"{Common-sense (or folk) psychology holds that (generally) we do what we do for the reasons we have. This common-sense approach is embodied in claims like \u2018I went to the kitchen because I wanted a drink\u2019 and \u2018She took a coat because she thought it might rain and hoped to stay dry\u2019. However, the veracity of these common-sense psychological explanations has been challenged by experimental evidence (primarily from behavioural economics and social psychology) which appears to show that individuals are systematically irrational\u2014that often we do not do what we do because of the reasons we have. Recently, some of the same experimental evidence has also been used to level a somewhat different challenge at the common-sense view, arguing that the overarching aim of reasoning is not to deliver better or more reason-governed decisions for individual reasoners, but to improve group decision making or to protect an individual\u2019s sense of self. This paper explores the range of challenges that experimental work has been taken to raise for the common-sense approach and suggests some potential responses. Overall, I argue that the experimental evidence surveyed should not lead us to a rejection of individual rationality.}\",\n issn = {0066-7374},\n doi = {10.1093/arisoc/aoac009},\n url = {https://doi.org/10.1093/arisoc/aoac009},\n eprint = {https://academic.oup.com/aristotelian/article-pdf/122/3/195/47051927/aoac009.pdf},\n}\n\n @article{Davis2015CommonsenseRA,\n title={Commonsense reasoning and commonsense knowledge in artificial intelligence},\n author={Ernest Davis and Gary F. Marcus},\n journal={Communications of the ACM},\n year={2015},\n volume={58},\n pages={92 - 103}\n}\n@article{JMLR:v21:20-074,\n author = {Colin Raffel and Noam Shazeer and Adam Roberts and Katherine Lee and Sharan Narang and Michael Matena and Yanqi Zhou and Wei Li and Peter J. Liu},\n title = {Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer},\n journal = {Journal of Machine Learning Research},\n year = {2020},\n volume = {21},\n number = {140},\n pages = {1--67},\n url = {http://jmlr.org/papers/v21/20-074.html}\n}\n@inproceedings{lin-2004-rouge,\n title = \"{ROUGE}: A Package for Automatic Evaluation of Summaries\",\n author = \"Lin, Chin-Yew\",\n booktitle = \"Text Summarization Branches Out\",\n month = jul,\n year = \"2004\",\n address = \"Barcelona, Spain\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/W04-1013\",\n pages = \"74--81\",\n}\n@inproceedings{BLEU,\nauthor = {Papineni, Kishore and Roukos, Salim and Ward, Todd and Zhu, Wei-Jing},\ntitle = {BLEU: A Method for Automatic Evaluation of Machine Translation},\nyear = {2002},\npublisher = {Association for Computational Linguistics},\naddress = {USA},\nurl = {https://doi.org/10.3115/1073083.1073135},\ndoi = {10.3115/1073083.1073135},\nabstract = {Human evaluations of machine translation are extensive but expensive. Human evaluations can take months to finish and involve human labor that can not be reused. We propose a method of automatic machine translation evaluation that is quick, inexpensive, and language-independent, that correlates highly with human evaluation, and that has little marginal cost per run. We present this method as an automated understudy to skilled human judges which substitutes for them when there is need for quick or frequent evaluations.},\nbooktitle = {Proceedings of the 40th Annual Meeting on Association for Computational Linguistics},\npages = {311\u2013318},\nnumpages = {8},\nlocation = {Philadelphia, Pennsylvania},\nseries = {ACL '02}\n}\n@inproceedings{lavie-agarwal-2007-meteor,\n title = \"{METEOR}: An Automatic Metric for {MT} Evaluation with High Levels of Correlation with Human Judgments\",\n author = \"Lavie, Alon and\n Agarwal, Abhaya\",\n booktitle = \"Proceedings of the Second Workshop on Statistical Machine Translation\",\n month = jun,\n year = \"2007\",\n address = \"Prague, Czech Republic\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/W07-0734\",\n pages = \"228--231\",\n}\n@inproceedings{bert-score,\n title={BERTScore: Evaluating Text Generation with BERT},\n author={Tianyi Zhang and Varsha Kishore and Felix Wu and Kilian Q. Weinberger and Yoav Artzi},\n booktitle={International Conference on Learning Representations},\n year={2020},\n url={https://openreview.net/forum?id=SkeHuCVFDr}\n}\n@misc{shumailov2023curse,\n title={The Curse of Recursion: Training on Generated Data Makes Models Forget}, \n author={Ilia Shumailov and Zakhar Shumaylov and Yiren Zhao and Yarin Gal and Nicolas Papernot and Ross Anderson},\n year={2023},\n eprint={2305.17493},\n archivePrefix={arXiv},\n primaryClass={cs.LG}\n}\n@article{gollwitzer_intention,\nauthor = { Peter M. Gollwitzer },\ntitle = {Goal Achievement: The Role of Intentions},\njournal = {European Review of Social Psychology},\nvolume = {4},\nnumber = {1},\npages = {141-185},\nyear = {1993},\npublisher = {Routledge},\ndoi = {10.1080/14792779343000059},\nURL = { https://doi.org/10.1080/14792779343000059},\neprint = { https://doi.org/10.1080/14792779343000059}\n}\n@article{foss_and_bower_1986,\n title={Understanding actions in relation to goals},\n author={Foss, Carolyn L and Bower, Gordon H},\n journal={Advances in cognitive science},\n volume={1},\n pages={94--124},\n year={1986},\n publisher={Ellis Horwood New York}\n}\n\n@article{stiennon,\n author = {Nisan Stiennon and\n Long Ouyang and\n Jeff Wu and\n Daniel M. Ziegler and\n Ryan Lowe and\n Chelsea Voss and\n Alec Radford and\n Dario Amodei and\n Paul F. Christiano},\n title = {Learning to summarize from human feedback},\n journal = {CoRR},\n volume = {abs/2009.01325},\n year = {2020},\n url = {https://arxiv.org/abs/2009.01325},\n eprinttype = {arXiv},\n eprint = {2009.01325},\n timestamp = {Thu, 01 Apr 2021 19:06:51 +0200},\n biburl = {https://dblp.org/rec/journals/corr/abs-2009-01325.bib},\n bibsource = {dblp computer science bibliography, https://dblp.org}\n}\n\n@inproceedings{krishnaswamy2011real,\n title={Real-time path planning for a robotic arm},\n author={Krishnaswamy, Kavita and Sleeman, Jennifer and Oates, Tim},\n booktitle={Proceedings of the 4th International Conference on Pervasive Technologies Related to Assistive Environments},\n pages={1--4},\n year={2011}\n}\n@misc{lal2021tellmewhy,\n title={TellMeWhy: A Dataset for Answering Why-Questions in Narratives}, \n author={Yash Kumar Lal and Nathanael Chambers and Raymond Mooney and Niranjan Balasubramanian},\n year={2021},\n eprint={2106.06132},\n archivePrefix={arXiv},\n primaryClass={cs.CL}\n}\n\n@inproceedings{chambers-jurafsky-2008-unsupervised,\n title = \"Unsupervised Learning of Narrative Event Chains\",\n author = \"Chambers, Nathanael and\n Jurafsky, Dan\",\n booktitle = \"Proceedings of ACL-08: HLT\",\n month = jun,\n year = \"2008\",\n address = \"Columbus, Ohio\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/P08-1090\",\n pages = \"789--797\",\n}\n@misc{ghosh2023pasta,\n title={PASTA: A Dataset for Modeling Participant States in Narratives}, \n author={Sayontan Ghosh and Mahnaz Koupaee and Isabella Chen and Francis Ferraro and Nathanael Chambers and Niranjan Balasubramanian},\n year={2023},\n eprint={2208.00329},\n archivePrefix={arXiv},\n primaryClass={cs.CL}\n}\n@inproceedings{wang-etal-2022-super,\n title = \"Super-{N}atural{I}nstructions: Generalization via Declarative Instructions on 1600+ {NLP} Tasks\",\n author = \"Wang, Yizhong and\n Mishra, Swaroop and\n Alipoormolabashi, Pegah and\n Kordi, Yeganeh and\n Mirzaei, Amirreza and\n Naik, Atharva and\n Ashok, Arjun and\n Dhanasekaran, Arut Selvan and\n Arunkumar, Anjana and\n Stap, David and\n Pathak, Eshaan and\n Karamanolakis, Giannis and\n Lai, Haizhi and\n Purohit, Ishan and\n Mondal, Ishani and\n Anderson, Jacob and\n Kuznia, Kirby and\n Doshi, Krima and\n Pal, Kuntal Kumar and\n Patel, Maitreya and\n Moradshahi, Mehrad and\n Parmar, Mihir and\n Purohit, Mirali and\n Varshney, Neeraj and\n Kaza, Phani Rohitha and\n Verma, Pulkit and\n Puri, Ravsehaj Singh and\n Karia, Rushang and\n Doshi, Savan and\n Sampat, Shailaja Keyur and\n Mishra, Siddhartha and\n Reddy A, Sujan and\n Patro, Sumanta and\n Dixit, Tanay and\n Shen, Xudong\",\n editor = \"Goldberg, Yoav and\n Kozareva, Zornitsa and\n Zhang, Yue\",\n booktitle = \"Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing\",\n month = dec,\n year = \"2022\",\n address = \"Abu Dhabi, United Arab Emirates\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2022.emnlp-main.340\",\n doi = \"10.18653/v1/2022.emnlp-main.340\",\n pages = \"5085--5109\",\n abstract = \"How well can NLP models generalize to a variety of unseen tasks when provided with task instructions? To address this question, we first introduce Super-NaturalInstructions, a benchmark of 1,616 diverse NLP tasks and their expert-written instructions. Our collection covers 76 distinct task types, including but not limited to classification, extraction, infilling, sequence tagging, text rewriting, and text composition. This large and diverse collection of tasks enables rigorous benchmarking of cross-task generalization under instructions{---}training models to follow instructions on a subset of tasks and evaluating them on the remaining unseen ones.Furthermore, we build Tk-Instruct, a transformer model trained to follow a variety of in-context instructions (plain language task definitions or k-shot examples). Our experiments show that Tk-Instruct outperforms existing instruction-following models such as InstructGPT by over 9{\\%} on our benchmark despite being an order of magnitude smaller. We further analyze generalization as a function of various scaling parameters, such as the number of observed tasks, the number of instances per task, and model sizes. We hope our dataset and model facilitate future progress towards more general-purpose NLP models.\",\n}\n\n\n@inproceedings{bellos-etal-2024-large,\n title = \"Can Large Language Models Reason About Goal-Oriented Tasks?\",\n author = \"Bellos, Filippos and\n Li, Yayuan and\n Liu, Wuao and\n Corso, Jason\",\n editor = \"Miceli-Barone, Antonio Valerio and\n Barez, Fazl and\n Cohen, Shay and\n Voita, Elena and\n Germann, Ulrich and\n Lukasik, Michal\",\n booktitle = \"Proceedings of the First edition of the Workshop on the Scaling Behavior of Large Language Models (SCALE-LLM 2024)\",\n month = mar,\n year = \"2024\",\n address = \"St. Julian{'}s, Malta\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2024.scalellm-1.3\",\n pages = \"24--34\",\n abstract = \"Most adults can complete a sequence of steps to achieve a certain goal, such as making a sandwich or repairing a bicycle tire. In completing these goal-oriented tasks, or simply tasks in this paper, one must use sequential reasoning to understand the relationship between the sequence of steps and the goal. LLMs have shown impressive capabilities across various natural language understanding tasks. However, prior work has mainlyfocused on logical reasoning tasks (e.g. arithmetic, commonsense QA); how well LLMs can perform on more complex reasoning tasks like sequential reasoning is not clear. In this paper, we address this gap and conduct a comprehensive evaluation of how well LLMs are able to conduct this reasoning for tasks and how they scale w.r.t multiple dimensions(e.g. adaptive prompting strategies, number of in-context examples, varying complexity of the sequential task). Our findings reveal that while Chain of Thought (CoT) prompting can significantly enhance LLMs{'} sequential reasoning in certain scenarios, it can also be detrimental in others, whereas Tree of Thoughts (ToT) reasoning is less effective for this type of task. Additionally, we discover that an increase in model size or in-context examples does not consistently lead to improved performance.\",\n}\n@misc{zhang2023ask,\n title={Ask an Expert: Leveraging Language Models to Improve Strategic Reasoning in Goal-Oriented Dialogue Models}, \n author={Qiang Zhang and Jason Naradowsky and Yusuke Miyao},\n year={2023},\n eprint={2305.17878},\n archivePrefix={arXiv},\n primaryClass={cs.CL}\n}\n@inproceedings{cao-etal-2022-goal,\n title = \"Goal-oriented Vision-and-Dialog Navigation via Reinforcement Learning\",\n author = \"Cao, Yan and\n Lu, Keting and\n DeFazio, David and\n Zhang, Shiqi\",\n editor = \"Goldberg, Yoav and\n Kozareva, Zornitsa and\n Zhang, Yue\",\n booktitle = \"Findings of the Association for Computational Linguistics: EMNLP 2022\",\n month = dec,\n year = \"2022\",\n address = \"Abu Dhabi, United Arab Emirates\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2022.findings-emnlp.327\",\n doi = \"10.18653/v1/2022.findings-emnlp.327\",\n pages = \"4473--4482\",\n abstract = \"Vision-and-dialog navigation is a recent benchmark for evaluating the AI capabilities of perception, interaction, and decision making. While existing methods developed for this benchmark have demonstrated great successes, they mostly rely on large datasets, where data collection can be a challenge, and the learned policies are not adaptive to domain changes. In this paper, we focus on a new problem, referred to as goal-oriented vision-and-dialog navigation (GVDN), where an agent uses reinforcement learning techniques to compute dialog-navigation policies from trial and error. A robot conducts visual navigation to locate target objects, and can talk to a remote human operator as needed. Our remote human is able to provide guidance on navigation only if the robot correctly conveys its location through dialog. Experiments have been conducted using photo-realistic simulation environments. Results suggest that, our agent outperforms competitive baselines in success rate.\",\n}\n\n\n\n@inproceedings{jiang-riloff-2018-learning,\n title = \"Learning Prototypical Goal Activities for Locations\",\n author = \"Jiang, Tianyu and\n Riloff, Ellen\",\n editor = \"Gurevych, Iryna and\n Miyao, Yusuke\",\n booktitle = \"Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)\",\n month = jul,\n year = \"2018\",\n address = \"Melbourne, Australia\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/P18-1120\",\n doi = \"10.18653/v1/P18-1120\",\n pages = \"1297--1307\",\n abstract = \"People go to different places to engage in activities that reflect their goals. For example, people go to restaurants to eat, libraries to study, and churches to pray. We refer to an activity that represents a common reason why people typically go to a location as a prototypical goal activity (goal-act). Our research aims to learn goal-acts for specific locations using a text corpus and semi-supervised learning. First, we extract activities and locations that co-occur in goal-oriented syntactic patterns. Next, we create an activity profile matrix and apply a semi-supervised label propagation algorithm to iteratively revise the activity strengths for different locations using a small set of labeled data. We show that this approach outperforms several baseline methods when judged against goal-acts identified by human annotators.\",\n}\n@misc{chen2023say,\n title={Say What You Mean! Large Language Models Speak Too Positively about Negative Commonsense Knowledge}, \n author={Jiangjie Chen and Wei Shi and Ziquan Fu and Sijie Cheng and Lei Li and Yanghua Xiao},\n year={2023},\n eprint={2305.05976},\n archivePrefix={arXiv},\n primaryClass={cs.CL}\n}\n@inproceedings{hossain-etal-2022-analysis,\n title = \"An Analysis of Negation in Natural Language Understanding Corpora\",\n author = \"Hossain, Md Mosharaf and\n Chinnappa, Dhivya and\n Blanco, Eduardo\",\n editor = \"Muresan, Smaranda and\n Nakov, Preslav and\n Villavicencio, Aline\",\n booktitle = \"Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (Volume 2: Short Papers)\",\n month = may,\n year = \"2022\",\n address = \"Dublin, Ireland\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2022.acl-short.81\",\n doi = \"10.18653/v1/2022.acl-short.81\",\n pages = \"716--723\",\n abstract = \"This paper analyzes negation in eight popular corpora spanning six natural language understanding tasks. We show that these corpora have few negations compared to general-purpose English, and that the few negations in them are often unimportant. Indeed, one can often ignore negations and still make the right predictions. Additionally, experimental results show that state-of-the-art transformers trained with these corpora obtain substantially worse results with instances that contain negation, especially if the negations are important. We conclude that new corpora accounting for negation are needed to solve natural language understanding tasks when negation is present.\",\n}\n\n@inproceedings{storks-etal-2021-tiered-reasoning,\n title = \"Tiered Reasoning for Intuitive Physics: Toward Verifiable Commonsense Language Understanding\",\n author = \"Storks, Shane and\n Gao, Qiaozi and\n Zhang, Yichi and\n Chai, Joyce\",\n editor = \"Moens, Marie-Francine and\n Huang, Xuanjing and\n Specia, Lucia and\n Yih, Scott Wen-tau\",\n booktitle = \"Findings of the Association for Computational Linguistics: EMNLP 2021\",\n month = nov,\n year = \"2021\",\n address = \"Punta Cana, Dominican Republic\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2021.findings-emnlp.422\",\n doi = \"10.18653/v1/2021.findings-emnlp.422\",\n pages = \"4902--4918\",\n abstract = \"Large-scale, pre-trained language models (LMs) have achieved human-level performance on a breadth of language understanding tasks. However, evaluations only based on end task performance shed little light on machines{'} true ability in language understanding and reasoning. In this paper, we highlight the importance of evaluating the underlying reasoning process in addition to end performance. Toward this goal, we introduce Tiered Reasoning for Intuitive Physics (TRIP), a novel commonsense reasoning dataset with dense annotations that enable multi-tiered evaluation of machines{'} reasoning process. Our empirical results show that while large LMs can achieve high end performance, they struggle to support their predictions with valid supporting evidence. The TRIP dataset and our baseline results will motivate verifiable evaluation of commonsense reasoning and facilitate future research toward developing better language understanding and reasoning models.\",\n}\n\n\n\n@misc{jiang,\n title={Transferring Procedural Knowledge across Commonsense Tasks}, \n author={Yifan Jiang and Filip Ilievski and Kaixin Ma},\n year={2023},\n eprint={2304.13867},\n archivePrefix={arXiv},\n primaryClass={cs.CL}\n}\n@inproceedings{brown,\nauthor = {Brown, Tom B. and Mann, Benjamin and Ryder, Nick and Subbiah, Melanie and Kaplan, Jared and Dhariwal, Prafulla and Neelakantan, Arvind and Shyam, Pranav and Sastry, Girish and Askell, Amanda and Agarwal, Sandhini and Herbert-Voss, Ariel and Krueger, Gretchen and Henighan, Tom and Child, Rewon and Ramesh, Aditya and Ziegler, Daniel M. and Wu, Jeffrey and Winter, Clemens and Hesse, Christopher and Chen, Mark and Sigler, Eric and Litwin, Mateusz and Gray, Scott and Chess, Benjamin and Clark, Jack and Berner, Christopher and McCandlish, Sam and Radford, Alec and Sutskever, Ilya and Amodei, Dario},\ntitle = {Language Models Are Few-Shot Learners},\nyear = {2020},\nisbn = {9781713829546},\npublisher = {Curran Associates Inc.},\naddress = {Red Hook, NY, USA},\nabstract = {We demonstrate that scaling up language models greatly improves task-agnostic, few-shot performance, sometimes even becoming competitive with prior state-of-the-art fine-tuning approaches. Specifically, we train GPT-3, an autoregressive language model with 175 billion parameters, 10x more than any previous non-sparse language model, and test its performance in the few-shot setting. For all tasks, GPT-3 is applied without any gradient updates or fine-tuning, with tasks and few-shot demonstrations specified purely via text interaction with the model. GPT-3 achieves strong performance on many NLP datasets, including translation, question-answering, and cloze tasks. We also identify some datasets where GPT-3's few-shot learning still struggles, as well as some datasets where GPT-3 faces methodological issues related to training on large web corpora.},\nbooktitle = {Proceedings of the 34th International Conference on Neural Information Processing Systems},\narticleno = {159},\nnumpages = {25},\nlocation = {Vancouver, BC, Canada},\nseries = {NIPS'20}\n}\n\n@article{radford2018improving,\n added-at = {2020-07-14T16:37:42.000+0200},\n author = {Radford, Alec and Narasimhan, Karthik and Salimans, Tim and Sutskever, Ilya},\n biburl = {https://www.bibsonomy.org/bibtex/273ced32c0d4588eb95b6986dc2c8147c/jonaskaiser},\n interhash = {5c343ed9a31ac52fd17a898f72af228f},\n intrahash = {73ced32c0d4588eb95b6986dc2c8147c},\n keywords = {final thema:transformer},\n timestamp = {2020-07-14T16:49:42.000+0200},\n title = {Improving language understanding by generative pre-training},\n year = 2018\n}\n\n@inproceedings{Devlin2019BERTPO,\n title={BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding},\n author={Jacob Devlin and Ming-Wei Chang and Kenton Lee and Kristina Toutanova},\n booktitle={North American Chapter of the Association for Computational Linguistics},\n year={2019},\n url={https://api.semanticscholar.org/CorpusID:52967399}\n}\n @inproceedings{Zellers2019HellaSwagCA,\n title={HellaSwag: Can a Machine Really Finish Your Sentence?},\n author={Rowan Zellers and Ari Holtzman and Yonatan Bisk and Ali Farhadi and Yejin Choi},\n booktitle={Annual Meeting of the Association for Computational Linguistics},\n year={2019},\n url={https://api.semanticscholar.org/CorpusID:159041722}\n}\n@inproceedings{zellers2018,\nauthor = {Zellers, Rowan and Bisk, Yonatan and Schwartz, Roy and Choi, Yejin},\nyear = {2018},\nmonth = {01},\npages = {93-104},\ntitle = {SWAG: A Large-Scale Adversarial Dataset for Grounded Commonsense Inference},\nbooktitle = {EMNLP},\ndoi = {10.18653/v1/D18-1009}\n}\n@inproceedings{qin-etal-2019-counterfactual,\n title = \"Counterfactual Story Reasoning and Generation\",\n author = \"Qin, Lianhui and\n Bosselut, Antoine and\n Holtzman, Ari and\n Bhagavatula, Chandra and\n Clark, Elizabeth and\n Choi, Yejin\",\n editor = \"Inui, Kentaro and\n Jiang, Jing and\n Ng, Vincent and\n Wan, Xiaojun\",\n booktitle = \"Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP)\",\n month = nov,\n year = \"2019\",\n address = \"Hong Kong, China\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/D19-1509\",\n doi = \"10.18653/v1/D19-1509\",\n pages = \"5043--5053\",\n abstract = \"Counterfactual reasoning requires predicting how alternative events, contrary to what actually happened, might have resulted in different outcomes. Despite being considered a necessary component of AI-complete systems, few resources have been developed for evaluating counterfactual reasoning in narratives. In this paper, we propose Counterfactual Story Rewriting: given an original story and an intervening counterfactual event, the task is to minimally revise the story to make it compatible with the given counterfactual event. Solving this task will require deep understanding of causal narrative chains and counterfactual invariance, and integration of such story reasoning capabilities into conditional language generation models. We present TIMETRAVEL, a new dataset of 29,849 counterfactual rewritings, each with the original story, a counterfactual event, and human-generated revision of the original story compatible with the counterfactual event. Additionally, we include 81,407 counterfactual {``}branches{''} without a rewritten storyline to support future work on semi- or un-supervised approaches to counterfactual story rewriting. Finally, we evaluate the counterfactual rewriting capacities of several competitive baselines based on pretrained language models, and assess whether common overlap and model-based automatic metrics for text generation correlate well with human scores for counterfactual rewriting.\",\n}\n\n@inproceedings{zaidan-callison-burch-2011-crowdsourcing,\n title = \"Crowdsourcing Translation: Professional Quality from Non-Professionals\",\n author = \"Zaidan, Omar F. and\n Callison-Burch, Chris\",\n editor = \"Lin, Dekang and\n Matsumoto, Yuji and\n Mihalcea, Rada\",\n booktitle = \"Proceedings of the 49th Annual Meeting of the Association for Computational Linguistics: Human Language Technologies\",\n month = jun,\n year = \"2011\",\n address = \"Portland, Oregon, USA\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/P11-1122\",\n pages = \"1220--1229\",\n}\n@inproceedings{wu-etal-2023-chain,\n title = \"Chain of Thought Prompting Elicits Knowledge Augmentation\",\n author = \"Wu, Dingjun and\n Zhang, Jing and\n Huang, Xinmei\",\n editor = \"Rogers, Anna and\n Boyd-Graber, Jordan and\n Okazaki, Naoaki\",\n booktitle = \"Findings of the Association for Computational Linguistics: ACL 2023\",\n month = jul,\n year = \"2023\",\n address = \"Toronto, Canada\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2023.findings-acl.408\",\n doi = \"10.18653/v1/2023.findings-acl.408\",\n pages = \"6519--6534\",\n abstract = \"The knowledge-augmented deep learning paradigm refers to a paradigm in which domain knowledge is identified and integrated into deep models. Conventional methods typically employ task-specific approaches to gather external knowledge from various sources. In contrast, large language models are extensively pre-trained and can serve as a comprehensive source of external knowledge. In this paper, we propose CoT-KA, a Chain-of-Thought-based method that augments knowledge for deep learning. CoT-KA avoids the need for additional knowledge retrieval or knowledge reasoning models, as required in conventional augmentation methods. Our results demonstrate that CoT-KA outperforms both pure CoT-based methods and the non-augmented method across the majority of eleven publicly available benchmarks for various reasoning tasks.\",\n}\n\n\n@inproceedings{niven-kao-2019-probing,\n title = \"Probing Neural Network Comprehension of Natural Language Arguments\",\n author = \"Niven, Timothy and\n Kao, Hung-Yu\",\n editor = \"Korhonen, Anna and\n Traum, David and\n M{\\`a}rquez, Llu{\\'\\i}s\",\n booktitle = \"Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics\",\n month = jul,\n year = \"2019\",\n address = \"Florence, Italy\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/P19-1459\",\n doi = \"10.18653/v1/P19-1459\",\n pages = \"4658--4664\",\n abstract = \"We are surprised to find that BERT{'}s peak performance of 77{\\%} on the Argument Reasoning Comprehension Task reaches just three points below the average untrained human baseline. However, we show that this result is entirely accounted for by exploitation of spurious statistical cues in the dataset. We analyze the nature of these cues and demonstrate that a range of models all exploit them. This analysis informs the construction of an adversarial dataset on which all models achieve random accuracy. Our adversarial dataset provides a more robust assessment of argument comprehension and should be adopted as the standard in future work.\",\n}\n@article{Marasini2016AssessingTI,\n title={Assessing the inter-rater agreement for ordinal data through weighted indexes},\n author={Donata Marasini and Piero Quatto and Enrico Ripamonti},\n journal={Statistical Methods in Medical Research},\n year={2016},\n volume={25},\n pages={2611 - 2633}\n} \n\n\n@inproceedings{lal-etal-2021-tellmewhy,\n title = \"{T}ell{M}e{W}hy: A Dataset for Answering Why-Questions in Narratives\",\n author = \"Lal, Yash Kumar and\n Chambers, Nathanael and\n Mooney, Raymond and\n Balasubramanian, Niranjan\",\n booktitle = \"Findings of the Association for Computational Linguistics: ACL-IJCNLP 2021\",\n month = aug,\n year = \"2021\",\n address = \"Online\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2021.findings-acl.53\",\n doi = \"10.18653/v1/2021.findings-acl.53\",\n pages = \"596--610\",\n}\n@inproceedings{lal-etal-2022-using,\n title={Using Commonsense Knowledge to Answer Why Questions},\n author={Lal, Yash Kumar and\n Tandon, Niket and\n Aggarwal, Tanvi and\n Liu, Horace and\n Chambers, Nathanael and\n Mooney, Raymond and\n Balasubramanian, Niranjan},\n booktitle={Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing},\n month={dec},\n year={2022},\n address={Online and Abu Dhabi, United Arab Emirates},\n publisher={Association for Computational Linguistics},\n}\n@article{10.1162/tacl_a_00370,\n author = {Geva, Mor and Khashabi, Daniel and Segal, Elad and Khot, Tushar and Roth, Dan and Berant, Jonathan},\n title = \"{Did Aristotle Use a Laptop? A Question Answering Benchmark with Implicit Reasoning Strategies}\",\n journal = {Transactions of the Association for Computational Linguistics},\n volume = {9},\n pages = {346-361},\n year = {2021},\n month = {04},\n abstract = \"{A key limitation in current datasets for multi-hop reasoning is that the required steps for answering the question are mentioned in it explicitly. In this work, we introduce StrategyQA, a question answering (QA) benchmark where the required reasoning steps are implicit in the question, and should be inferred using a strategy. A fundamental challenge in this setup is how to elicit such creative questions from crowdsourcing workers, while covering a broad range of potential strategies. We propose a data collection procedure that combines term-based priming to inspire annotators, careful control over the annotator population, and adversarial filtering for eliminating reasoning shortcuts. Moreover, we annotate each question with (1) a decomposition into reasoning steps for answering it, and (2) Wikipedia paragraphs that contain the answers to each step. Overall, StrategyQA includes 2,780 examples, each consisting of a strategy question, its decomposition, and evidence paragraphs. Analysis shows that questions in StrategyQA are short, topic-diverse, and cover a wide range of strategies. Empirically, we show that humans perform well (87\\\\%) on this task, while our best baseline reaches an accuracy of \u223c 66\\\\%.}\",\n issn = {2307-387X},\n doi = {10.1162/tacl_a_00370},\n url = {https://doi.org/10.1162/tacl\\_a\\_00370},\n eprint = {https://direct.mit.edu/tacl/article-pdf/doi/10.1162/tacl\\_a\\_00370/1924104/tacl\\_a\\_00370.pdf},\n}\n\n@inproceedings{ammanabrolu-etal-2021-motivate,\n title = \"How to Motivate Your Dragon: Teaching Goal-Driven Agents to Speak and Act in Fantasy Worlds\",\n author = {Ammanabrolu, Prithviraj and\n Urbanek, Jack and\n Li, Margaret and\n Szlam, Arthur and\n Rockt{\\\"a}schel, Tim and\n Weston, Jason},\n editor = \"Toutanova, Kristina and\n Rumshisky, Anna and\n Zettlemoyer, Luke and\n Hakkani-Tur, Dilek and\n Beltagy, Iz and\n Bethard, Steven and\n Cotterell, Ryan and\n Chakraborty, Tanmoy and\n Zhou, Yichao\",\n booktitle = \"Proceedings of the 2021 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies\",\n year = \"2021\",\n}\n\n@inproceedings{callison-burch-etal-2022-dungeons,\n title = \"Dungeons and Dragons as a Dialog Challenge for Artificial Intelligence\",\n author = \"Callison-Burch, Chris and\n Tomar, Gaurav Singh and\n Martin, Lara and\n Ippolito, Daphne and\n Bailis, Suma and\n Reitter, David\",\n editor = \"Goldberg, Yoav and\n Kozareva, Zornitsa and\n Zhang, Yue\",\n booktitle = \"Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing\",\n year = \"2022\"\n}" + }, + "target_bib_keys": [ + "poque", + "ghosh2023pasta" + ], + "contexts": [ + { + "text": "We annotated ROC stories (Mostafazadeh et al., 2016) that have corresponding alternative stories in the PASTA dataset .", + "text_with_marker": "We annotated ROC stories (Mostafazadeh et al., 2016) that have corresponding alternative stories in the PASTA dataset ." + } + ], + "target_citing_paragraphs": [ + "We use a third HIT when one or more annotators identify that a participant's goal is inferrable from $S^{c_k}$. With this HIT we obtain a new set of goal annotations reusing and modifying the free-form text annotations from the actual story to obtain annotations that are also minimally updated reflecting the process used for obtaining the counterfactual in ~PASTA~. See \\cref{tab:HITS-annotation} for the annotations and the HITs used for obtaining them.", + "\\subsubsection{Worker Selection and Qualifications}\n\\label{app:HIT_information:qualifications}\nFor our initial alpha runs of the HITs we used all workers who meet our community standard quality criteria, such as requiring a 98\\% or greater HIT acceptance rate and the completion of 1000 approved HITs. In addition, we required the worker's stated location to be in the USA, UK, Canada, Australia, or New Zealand. We used the location requirement to avoid language-based artifacts given the language-dependent semantic phenomena and the high subjectivity our work can elicit. \nWe did not use requester-generated qualification tests, though in early iterations we found that annotators who had completed at least 50 HITs in our prior work~ provided the most reliable annotations; the vast majority of our responses are from this group." + ], + "matched_paragraphs": [ + { + "context_id": 1, + "context": "We annotated ROC stories (Mostafazadeh et al., 2016) that have corresponding alternative stories in the PASTA dataset .", + "context_with_marker": "We annotated ROC stories (Mostafazadeh et al., 2016) that have corresponding alternative stories in the PASTA dataset .", + "paragraph": "As a first step in the annotation process, 3 crowd workers verified that the selected participants are volitional entities. When a worker identified a participant as non-volitional, we verified and discarded the participant-specific story (Total discarded :29). For the remaining 995 participant specific stories, we obtained 3 goal annotations and for a random 50\\% of these stories we annotated all the corresponding counterfactual (in our terminology, ``alternative'') stories in PASTA. We obtained a total of 2985 actual and 3240 alternative goal annotations using a chain of Human Intelligence Tasks (HITs) to identify a participant's goal in story and follow its achievement-arc in both the actual and alternative stories (for the annotated alternative stories). See \\cref{sec:HIT_design} for HIT design and compensation.", + "overlap": 0.5 + } + ] + } + ] +} \ No newline at end of file diff --git a/replay_traces/2208.00329/processed_papers/2208.00329/usage_context_labels.json b/replay_traces/2208.00329/processed_papers/2208.00329/usage_context_labels.json new file mode 100644 index 0000000000000000000000000000000000000000..ee824493f438f30c5a986c826c1664fc27e8d29f --- /dev/null +++ b/replay_traces/2208.00329/processed_papers/2208.00329/usage_context_labels.json @@ -0,0 +1,292 @@ +{ + "paper_id": "2208.00329", + "num_contexts": 20, + "label_set": [ + "Background", + "Uses", + "Extends", + "CompareOrContrast", + "Motivation", + "Future" + ], + "labels": [ + { + "id": 1, + "label": "Uses", + "confidence": 0.9254917502403259, + "confidence_margin": 0.8626327514648438, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "c1d02fc3786fb5e5b0543703af95046253109d4e", + "citing_title": "Inductive Bias Extraction and Matching for LLM Prompts", + "text": "The PASTA dataset is an extension of ROCStories that substitutes alternative events into the stories, which may or may not match up with the original summary sentence ." + }, + { + "id": 2, + "label": "Uses", + "confidence": 0.5409758687019348, + "confidence_margin": 0.14535099267959595, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "a6924f6b715327b0f24d044e9e8ef799b3f4a0a2", + "citing_title": "Electrifying the Story Climax via Axial Self-Attention in GPT-3 Transformer Block", + "text": "The dataset is often separated into four subsets: training, testing, evaluation, and validation [32\u201339]." + }, + { + "id": 3, + "label": "Background", + "confidence": 0.8934071660041809, + "confidence_margin": 0.7892690300941467, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "ab73ca481256767addce56b20eb48ac183169e01", + "citing_title": "CoRE: Condition-based Reasoning for Identifying Outcome Variance in Complex Events", + "text": "PASTA examined implied states and preturbed these states to examine changes in situational narratives." + }, + { + "id": 4, + "label": "Uses", + "confidence": 0.9904201626777649, + "confidence_margin": 0.9823774695396423, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "ab73ca481256767addce56b20eb48ac183169e01", + "citing_title": "CoRE: Condition-based Reasoning for Identifying Outcome Variance in Complex Events", + "text": "In the following paragraphs, we briefly describe the PASTA and SAGA (Vallu-rupalli et al., 2024) datasets and how these datasets\u2019 annotations relate to conditions and outcomes." + }, + { + "id": 5, + "label": "Background", + "confidence": 0.9986065030097961, + "confidence_margin": 0.9977977275848389, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "ab73ca481256767addce56b20eb48ac183169e01", + "citing_title": "CoRE: Condition-based Reasoning for Identifying Outcome Variance in Complex Events", + "text": "\u2026of reasoning is challenging because: (1) Conditions that to relate to entity properties and states are not always explicitly stated in a narrative but are implicitly understood through forming a coherent mental representation and acquiring this implicit knowledge is not easy." + }, + { + "id": 6, + "label": "Background", + "confidence": 0.9968738555908203, + "confidence_margin": 0.9947959780693054, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "ab73ca481256767addce56b20eb48ac183169e01", + "citing_title": "CoRE: Condition-based Reasoning for Identifying Outcome Variance in Complex Events", + "text": "\u2026of these conditions, like Sam trusting his/her dad or Sam being a small child, are relevant to the goal, with a high likelihood of influencing the outcome, while other conditions that ; Vallurupalli et al., 2024). might be true are irrelevant to the outcome\u2014like Sam being male." + }, + { + "id": 7, + "label": "Background", + "confidence": 0.9994476437568665, + "confidence_margin": 0.9992915987968445, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "ab73ca481256767addce56b20eb48ac183169e01", + "citing_title": "CoRE: Condition-based Reasoning for Identifying Outcome Variance in Complex Events", + "text": "( 4) Large Language Models (LLMs) (Brown et al., 2020; Ouyang et al., 2022, inter alia ), while powerful, do not necessarily perform well on tasks requiring counterfactual reasoning (Fang et al., 2025; Lin, 2004; ; Qin et al., 2019)." + }, + { + "id": 8, + "label": "Uses", + "confidence": 0.9994866847991943, + "confidence_margin": 0.9991527199745178, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "ab73ca481256767addce56b20eb48ac183169e01", + "citing_title": "CoRE: Condition-based Reasoning for Identifying Outcome Variance in Complex Events", + "text": "We expand the 3 categories used in PASTA , for error analysis on 200 random states on the story state inference task, to 4 categories and group all conditions into these as follows: (a) Physical: This category includes natural physical attributes of an entity such as size, age,\u2026" + }, + { + "id": 9, + "label": "Uses", + "confidence": 0.9968955516815186, + "confidence_margin": 0.9944096207618713, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "ab73ca481256767addce56b20eb48ac183169e01", + "citing_title": "CoRE: Condition-based Reasoning for Identifying Outcome Variance in Complex Events", + "text": "We leverage two previously released datasets PASTA and SAGA (Vallurupalli et al., 2024) consisting of participants\u2019 goal and state annotations (shown in Fig." + }, + { + "id": 10, + "label": "Uses", + "confidence": 0.9981223940849304, + "confidence_margin": 0.997296929359436, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "ab73ca481256767addce56b20eb48ac183169e01", + "citing_title": "CoRE: Condition-based Reasoning for Identifying Outcome Variance in Complex Events", + "text": "We leverage PASTA and SAGA (Vallurupalli et al., 2024) datasets to generate outcomes of SAGA and News Stories and validate these using both our generated and PASTA\u2019s crowd-annotated conditions. outcome is implicitly understood through constructing counterfactual mental\u2026" + }, + { + "id": 11, + "label": "Uses", + "confidence": 0.915876567363739, + "confidence_margin": 0.8748534321784973, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "494f3c55553161be831f725a7b55c6095470c5cb", + "citing_title": "SAGA: A Participant-specific Examination of Story Alternatives and Goal Applicability for a Deeper Understanding of Complex Events", + "text": "With this HIT we obtain a new set of goal annotations reusing and modifying the free-form text annotations from the actual story to obtain annotations that are also minimally updated reflecting the process used for obtaining the counterfactual in PASTA ." + }, + { + "id": 12, + "label": "Background", + "confidence": 0.9987213015556335, + "confidence_margin": 0.9978130459785461, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "494f3c55553161be831f725a7b55c6095470c5cb", + "citing_title": "SAGA: A Participant-specific Examination of Story Alternatives and Goal Applicability for a Deeper Understanding of Complex Events", + "text": "In PASTA , original ROCStories (\u201cactual stories\u201d) have up to three \u201calternative\u201d stories." + }, + { + "id": 13, + "label": "Uses", + "confidence": 0.8018349409103394, + "confidence_margin": 0.689209520816803, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "494f3c55553161be831f725a7b55c6095470c5cb", + "citing_title": "SAGA: A Participant-specific Examination of Story Alternatives and Goal Applicability for a Deeper Understanding of Complex Events", + "text": "\u2026stories shown: in alternative 1 (top right), a different Figure 1: A participant\u2019s goal inferred from the actual story when applied to 3 alternative stories, drawn from the PASTA dataset ; slightly varying actions in the stories lead to different goal achievement outcomes." + }, + { + "id": 14, + "label": "Background", + "confidence": 0.9973942041397095, + "confidence_margin": 0.996073305606842, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "494f3c55553161be831f725a7b55c6095470c5cb", + "citing_title": "SAGA: A Participant-specific Examination of Story Alternatives and Goal Applicability for a Deeper Understanding of Complex Events", + "text": "PASTA introduced implied states that stories depend upon and examined the tasks of state inference from alternative stories and story rewriting for alternative states." + }, + { + "id": 15, + "label": "Background", + "confidence": 0.9965572357177734, + "confidence_margin": 0.9946063160896301, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "494f3c55553161be831f725a7b55c6095470c5cb", + "citing_title": "SAGA: A Participant-specific Examination of Story Alternatives and Goal Applicability for a Deeper Understanding of Complex Events", + "text": "Third, while pre-trained large language models (LLMs) are powerful (Brown et al., 2020; Wei et al., 2022, inter alia ), they may perform poorly on tasks requiring robust reasoning ; Zellers et al., 2019; Qin et al., 2019)." + }, + { + "id": 16, + "label": "Uses", + "confidence": 0.9981212019920349, + "confidence_margin": 0.996633768081665, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "494f3c55553161be831f725a7b55c6095470c5cb", + "citing_title": "SAGA: A Participant-specific Examination of Story Alternatives and Goal Applicability for a Deeper Understanding of Complex Events", + "text": "We annotated ROC stories (Mostafazadeh et al., 2016) that have corresponding alternative stories in the PASTA dataset ." + }, + { + "id": 17, + "label": "Background", + "confidence": 0.9918141961097717, + "confidence_margin": 0.9873102307319641, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "d723193c0453223803ffb926354cd6d2dee32b06", + "citing_title": "EvEval: A Comprehensive Evaluation of Event Semantics for Large Language Models", + "text": "Learning other relations such as event temporality needs more context information and knowledge since it is more subjected to specific situations ." + }, + { + "id": 18, + "label": "Background", + "confidence": 0.9065139293670654, + "confidence_margin": 0.8178352117538452, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "293583961efdf22c0905cc04465aa88ea96826ec", + "citing_title": "Causal schema induction for knowledge discovery", + "text": ", 2021), of participant states ; Vallurupalli et al., 2022), and of disease where organisms are conceived of as causative agents, e." + }, + { + "id": 19, + "label": "Background", + "confidence": 0.9994509816169739, + "confidence_margin": 0.9992090463638306, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "293583961efdf22c0905cc04465aa88ea96826ec", + "citing_title": "Causal schema induction for knowledge discovery", + "text": "Participant-centeredness is featured in studies of narrative (Propp, 1968; Caselli and Vossen, 2016; Brahman et al., 2021), of participant states ; Vallurupalli et al., 2022), and of disease where organisms are conceived of as causative agents, e.g., the pathogen tubercle\u2026" + }, + { + "id": 20, + "label": "Background", + "confidence": 0.9978366494178772, + "confidence_margin": 0.9965800642967224, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "2b914691a44ab1c8bcb0104b8079d86908d98729", + "citing_title": "Who Plays Which Role? Protagonist Detection and Classification in Moral Discourse", + "text": "Beyond moral discourse, modeling actors and their roles has a long tradition in discourse analysis, for instance, through semantic role labeling (Ruppenhofer et al., 2009; Roth and Lap-ata, 2015; Bornheim et al., 2024) and participant modeling (Tilk et al., 2016; ." + } + ], + "citing_paper_labels": [ + { + "citing_paper_id": "c1d02fc3786fb5e5b0543703af95046253109d4e", + "citing_title": "Inductive Bias Extraction and Matching for LLM Prompts", + "label": "Uses", + "evidence_context_ids": [ + 1 + ] + }, + { + "citing_paper_id": "a6924f6b715327b0f24d044e9e8ef799b3f4a0a2", + "citing_title": "Electrifying the Story Climax via Axial Self-Attention in GPT-3 Transformer Block", + "label": "Uses", + "evidence_context_ids": [ + 2 + ] + }, + { + "citing_paper_id": "ab73ca481256767addce56b20eb48ac183169e01", + "citing_title": "CoRE: Condition-based Reasoning for Identifying Outcome Variance in Complex Events", + "label": "Uses", + "evidence_context_ids": [ + 4, + 8, + 9, + 10 + ] + }, + { + "citing_paper_id": "494f3c55553161be831f725a7b55c6095470c5cb", + "citing_title": "SAGA: A Participant-specific Examination of Story Alternatives and Goal Applicability for a Deeper Understanding of Complex Events", + "label": "Uses", + "evidence_context_ids": [ + 11, + 13, + 16 + ] + }, + { + "citing_paper_id": "d723193c0453223803ffb926354cd6d2dee32b06", + "citing_title": "EvEval: A Comprehensive Evaluation of Event Semantics for Large Language Models", + "label": "Background", + "evidence_context_ids": [] + }, + { + "citing_paper_id": "293583961efdf22c0905cc04465aa88ea96826ec", + "citing_title": "Causal schema induction for knowledge discovery", + "label": "Background", + "evidence_context_ids": [] + }, + { + "citing_paper_id": "2b914691a44ab1c8bcb0104b8079d86908d98729", + "citing_title": "Who Plays Which Role? Protagonist Detection and Classification in Moral Discourse", + "label": "Background", + "evidence_context_ids": [] + } + ], + "final_label": "Uses" +} \ No newline at end of file diff --git a/replay_traces/2208.00329/processed_papers/2208.00329/usage_contexts.json b/replay_traces/2208.00329/processed_papers/2208.00329/usage_contexts.json new file mode 100644 index 0000000000000000000000000000000000000000..45d798aba94d54cba925bf32ffde7173b40399bf --- /dev/null +++ b/replay_traces/2208.00329/processed_papers/2208.00329/usage_contexts.json @@ -0,0 +1,326 @@ +{ + "paper_id": "2208.00329", + "total_citations": 7, + "num_contexts": 20, + "num_citing_with_context": 7, + "num_citing_without_context": 0, + "num_influential_citations": 2, + "num_influential_with_context": 2, + "influential_contexts": [ + { + "citing_paper_id": "ab73ca481256767addce56b20eb48ac183169e01", + "title": "CoRE: Condition-based Reasoning for Identifying Outcome Variance in Complex Events", + "external_ids": { + "DBLP": "journals/corr/abs-2506-01253", + "ArXiv": "2506.01253", + "DOI": "10.48550/arXiv.2506.01253", + "CorpusId": 279074979 + }, + "contexts": [ + { + "text": "PASTA examined implied states and preturbed these states to examine changes in situational narratives.", + "text_raw": "PASTA (Ghosh et al., 2023) examined implied states and preturbed these states to examine changes in situational narratives.", + "intents": [] + }, + { + "text": "In the following paragraphs, we briefly describe the PASTA and SAGA (Vallu-rupalli et al., 2024) datasets and how these datasets\u2019 annotations relate to conditions and outcomes.", + "text_raw": "In the following paragraphs, we briefly describe the PASTA (Ghosh et al., 2023) and SAGA (Vallu-rupalli et al., 2024) datasets and how these datasets\u2019 annotations relate to conditions and outcomes.", + "intents": [] + }, + { + "text": "\u2026of reasoning is challenging because: (1) Conditions that to relate to entity properties and states are not always explicitly stated in a narrative but are implicitly understood through forming a coherent mental representation and acquiring this implicit knowledge is not easy.", + "text_raw": "\u2026of reasoning is challenging because: (1) Conditions that to relate to entity properties and states are not always explicitly stated in a narrative but are implicitly understood through forming a coherent mental representation (Ghosh et al., 2023) and acquiring this implicit knowledge is not easy.", + "intents": [] + }, + { + "text": "\u2026of these conditions, like Sam trusting his/her dad or Sam being a small child, are relevant to the goal, with a high likelihood of influencing the outcome, while other conditions that ; Vallurupalli et al., 2024). might be true are irrelevant to the outcome\u2014like Sam being male.", + "text_raw": "\u2026of these conditions, like Sam trusting his/her dad or Sam being a small child, are relevant to the goal, with a high likelihood of influencing the outcome, while other conditions that (Ghosh et al., 2023; Vallurupalli et al., 2024). might be true are irrelevant to the outcome\u2014like Sam being male.", + "intents": [] + }, + { + "text": "( 4) Large Language Models (LLMs) (Brown et al., 2020; Ouyang et al., 2022, inter alia ), while powerful, do not necessarily perform well on tasks requiring counterfactual reasoning (Fang et al., 2025; Lin, 2004; ; Qin et al., 2019).", + "text_raw": "( 4) Large Language Models (LLMs) (Brown et al., 2020; Ouyang et al., 2022, inter alia ), while powerful, do not necessarily perform well on tasks requiring counterfactual reasoning (Fang et al., 2025; Lin, 2004; Ghosh et al., 2023; Qin et al., 2019).", + "intents": [] + }, + { + "text": "We expand the 3 categories used in PASTA , for error analysis on 200 random states on the story state inference task, to 4 categories and group all conditions into these as follows: (a) Physical: This category includes natural physical attributes of an entity such as size, age,\u2026", + "text_raw": "We expand the 3 categories used in PASTA (Ghosh et al., 2023), for error analysis on 200 random states on the story state inference task, to 4 categories and group all conditions into these as follows: (a) Physical: This category includes natural physical attributes of an entity such as size, age,\u2026", + "intents": [] + }, + { + "text": "We leverage two previously released datasets PASTA and SAGA (Vallurupalli et al., 2024) consisting of participants\u2019 goal and state annotations (shown in Fig.", + "text_raw": "We leverage two previously released datasets PASTA (Ghosh et al., 2023) and SAGA (Vallurupalli et al., 2024) consisting of participants\u2019 goal and state annotations (shown in Fig.", + "intents": [] + }, + { + "text": "We leverage PASTA and SAGA (Vallurupalli et al., 2024) datasets to generate outcomes of SAGA and News Stories and validate these using both our generated and PASTA\u2019s crowd-annotated conditions. outcome is implicitly understood through constructing counterfactual mental\u2026", + "text_raw": "We leverage PASTA (Ghosh et al., 2023) and SAGA (Vallurupalli et al., 2024) datasets to generate outcomes of SAGA and News Stories and validate these using both our generated and PASTA\u2019s crowd-annotated conditions. outcome is implicitly understood through constructing counterfactual mental\u2026", + "intents": [] + } + ] + }, + { + "citing_paper_id": "494f3c55553161be831f725a7b55c6095470c5cb", + "title": "SAGA: A Participant-specific Examination of Story Alternatives and Goal Applicability for a Deeper Understanding of Complex Events", + "external_ids": { + "DBLP": "journals/corr/abs-2408-05793", + "ArXiv": "2408.05793", + "DOI": "10.48550/arXiv.2408.05793", + "CorpusId": 271855571 + }, + "contexts": [ + { + "text": "With this HIT we obtain a new set of goal annotations reusing and modifying the free-form text annotations from the actual story to obtain annotations that are also minimally updated reflecting the process used for obtaining the counterfactual in PASTA .", + "text_raw": "With this HIT we obtain a new set of goal annotations reusing and modifying the free-form text annotations from the actual story to obtain annotations that are also minimally updated reflecting the process used for obtaining the counterfactual in PASTA (Ghosh et al., 2023).", + "intents": [ + "methodology" + ] + }, + { + "text": "In PASTA , original ROCStories (\u201cactual stories\u201d) have up to three \u201calternative\u201d stories.", + "text_raw": "In PASTA (Ghosh et al., 2023), original ROCStories (\u201cactual stories\u201d) have up to three \u201calternative\u201d stories.", + "intents": [ + "background" + ] + }, + { + "text": "\u2026stories shown: in alternative 1 (top right), a different Figure 1: A participant\u2019s goal inferred from the actual story when applied to 3 alternative stories, drawn from the PASTA dataset ; slightly varying actions in the stories lead to different goal achievement outcomes.", + "text_raw": "\u2026stories shown: in alternative 1 (top right), a different Figure 1: A participant\u2019s goal inferred from the actual story when applied to 3 alternative stories, drawn from the PASTA dataset (Ghosh et al., 2023); slightly varying actions in the stories lead to different goal achievement outcomes.", + "intents": [ + "background" + ] + }, + { + "text": "PASTA introduced implied states that stories depend upon and examined the tasks of state inference from alternative stories and story rewriting for alternative states.", + "text_raw": "PASTA (Ghosh et al., 2023) introduced implied states that stories depend upon and examined the tasks of state inference from alternative stories and story rewriting for alternative states.", + "intents": [ + "background" + ] + }, + { + "text": "Third, while pre-trained large language models (LLMs) are powerful (Brown et al., 2020; Wei et al., 2022, inter alia ), they may perform poorly on tasks requiring robust reasoning ; Zellers et al., 2019; Qin et al., 2019).", + "text_raw": "Third, while pre-trained large language models (LLMs) are powerful (Brown et al., 2020; Wei et al., 2022, inter alia ), they may perform poorly on tasks requiring robust reasoning (Ghosh et al., 2023; Zellers et al., 2019; Qin et al., 2019).", + "intents": [ + "background" + ] + }, + { + "text": "We annotated ROC stories (Mostafazadeh et al., 2016) that have corresponding alternative stories in the PASTA dataset .", + "text_raw": "We annotated ROC stories (Mostafazadeh et al., 2016) that have corresponding alternative stories in the PASTA dataset (Ghosh et al., 2023).", + "intents": [ + "methodology" + ] + } + ] + } + ], + "citing_papers": [ + { + "citing_paper_id": "c1d02fc3786fb5e5b0543703af95046253109d4e", + "title": "Inductive Bias Extraction and Matching for LLM Prompts", + "external_ids": { + "ArXiv": "2508.10295", + "DBLP": "journals/corr/abs-2508-10295", + "DOI": "10.48550/arXiv.2508.10295", + "CorpusId": 280649627 + }, + "is_influential": false, + "contexts": [ + { + "text": "The PASTA dataset is an extension of ROCStories that substitutes alternative events into the stories, which may or may not match up with the original summary sentence .", + "text_raw": "The PASTA dataset is an extension of ROCStories that substitutes alternative events into the stories, which may or may not match up with the original summary sentence (Ghosh et al., 2023).", + "intents": [] + } + ] + }, + { + "citing_paper_id": "a6924f6b715327b0f24d044e9e8ef799b3f4a0a2", + "title": "Electrifying the Story Climax via Axial Self-Attention in GPT-3 Transformer Block", + "external_ids": { + "DBLP": "journals/sncs/VD25", + "DOI": "10.1007/s42979-025-04091-1", + "CorpusId": 279483975 + }, + "is_influential": false, + "contexts": [ + { + "text": "The dataset is often separated into four subsets: training, testing, evaluation, and validation [32\u201339].", + "text_raw": "The dataset is often separated into four subsets: training, testing, evaluation, and validation [32\u201339].", + "intents": [] + } + ] + }, + { + "citing_paper_id": "ab73ca481256767addce56b20eb48ac183169e01", + "title": "CoRE: Condition-based Reasoning for Identifying Outcome Variance in Complex Events", + "external_ids": { + "DBLP": "journals/corr/abs-2506-01253", + "ArXiv": "2506.01253", + "DOI": "10.48550/arXiv.2506.01253", + "CorpusId": 279074979 + }, + "is_influential": true, + "contexts": [ + { + "text": "PASTA examined implied states and preturbed these states to examine changes in situational narratives.", + "text_raw": "PASTA (Ghosh et al., 2023) examined implied states and preturbed these states to examine changes in situational narratives.", + "intents": [] + }, + { + "text": "In the following paragraphs, we briefly describe the PASTA and SAGA (Vallu-rupalli et al., 2024) datasets and how these datasets\u2019 annotations relate to conditions and outcomes.", + "text_raw": "In the following paragraphs, we briefly describe the PASTA (Ghosh et al., 2023) and SAGA (Vallu-rupalli et al., 2024) datasets and how these datasets\u2019 annotations relate to conditions and outcomes.", + "intents": [] + }, + { + "text": "\u2026of reasoning is challenging because: (1) Conditions that to relate to entity properties and states are not always explicitly stated in a narrative but are implicitly understood through forming a coherent mental representation and acquiring this implicit knowledge is not easy.", + "text_raw": "\u2026of reasoning is challenging because: (1) Conditions that to relate to entity properties and states are not always explicitly stated in a narrative but are implicitly understood through forming a coherent mental representation (Ghosh et al., 2023) and acquiring this implicit knowledge is not easy.", + "intents": [] + }, + { + "text": "\u2026of these conditions, like Sam trusting his/her dad or Sam being a small child, are relevant to the goal, with a high likelihood of influencing the outcome, while other conditions that ; Vallurupalli et al., 2024). might be true are irrelevant to the outcome\u2014like Sam being male.", + "text_raw": "\u2026of these conditions, like Sam trusting his/her dad or Sam being a small child, are relevant to the goal, with a high likelihood of influencing the outcome, while other conditions that (Ghosh et al., 2023; Vallurupalli et al., 2024). might be true are irrelevant to the outcome\u2014like Sam being male.", + "intents": [] + }, + { + "text": "( 4) Large Language Models (LLMs) (Brown et al., 2020; Ouyang et al., 2022, inter alia ), while powerful, do not necessarily perform well on tasks requiring counterfactual reasoning (Fang et al., 2025; Lin, 2004; ; Qin et al., 2019).", + "text_raw": "( 4) Large Language Models (LLMs) (Brown et al., 2020; Ouyang et al., 2022, inter alia ), while powerful, do not necessarily perform well on tasks requiring counterfactual reasoning (Fang et al., 2025; Lin, 2004; Ghosh et al., 2023; Qin et al., 2019).", + "intents": [] + }, + { + "text": "We expand the 3 categories used in PASTA , for error analysis on 200 random states on the story state inference task, to 4 categories and group all conditions into these as follows: (a) Physical: This category includes natural physical attributes of an entity such as size, age,\u2026", + "text_raw": "We expand the 3 categories used in PASTA (Ghosh et al., 2023), for error analysis on 200 random states on the story state inference task, to 4 categories and group all conditions into these as follows: (a) Physical: This category includes natural physical attributes of an entity such as size, age,\u2026", + "intents": [] + }, + { + "text": "We leverage two previously released datasets PASTA and SAGA (Vallurupalli et al., 2024) consisting of participants\u2019 goal and state annotations (shown in Fig.", + "text_raw": "We leverage two previously released datasets PASTA (Ghosh et al., 2023) and SAGA (Vallurupalli et al., 2024) consisting of participants\u2019 goal and state annotations (shown in Fig.", + "intents": [] + }, + { + "text": "We leverage PASTA and SAGA (Vallurupalli et al., 2024) datasets to generate outcomes of SAGA and News Stories and validate these using both our generated and PASTA\u2019s crowd-annotated conditions. outcome is implicitly understood through constructing counterfactual mental\u2026", + "text_raw": "We leverage PASTA (Ghosh et al., 2023) and SAGA (Vallurupalli et al., 2024) datasets to generate outcomes of SAGA and News Stories and validate these using both our generated and PASTA\u2019s crowd-annotated conditions. outcome is implicitly understood through constructing counterfactual mental\u2026", + "intents": [] + } + ] + }, + { + "citing_paper_id": "494f3c55553161be831f725a7b55c6095470c5cb", + "title": "SAGA: A Participant-specific Examination of Story Alternatives and Goal Applicability for a Deeper Understanding of Complex Events", + "external_ids": { + "DBLP": "journals/corr/abs-2408-05793", + "ArXiv": "2408.05793", + "DOI": "10.48550/arXiv.2408.05793", + "CorpusId": 271855571 + }, + "is_influential": true, + "contexts": [ + { + "text": "With this HIT we obtain a new set of goal annotations reusing and modifying the free-form text annotations from the actual story to obtain annotations that are also minimally updated reflecting the process used for obtaining the counterfactual in PASTA .", + "text_raw": "With this HIT we obtain a new set of goal annotations reusing and modifying the free-form text annotations from the actual story to obtain annotations that are also minimally updated reflecting the process used for obtaining the counterfactual in PASTA (Ghosh et al., 2023).", + "intents": [ + "methodology" + ] + }, + { + "text": "In PASTA , original ROCStories (\u201cactual stories\u201d) have up to three \u201calternative\u201d stories.", + "text_raw": "In PASTA (Ghosh et al., 2023), original ROCStories (\u201cactual stories\u201d) have up to three \u201calternative\u201d stories.", + "intents": [ + "background" + ] + }, + { + "text": "\u2026stories shown: in alternative 1 (top right), a different Figure 1: A participant\u2019s goal inferred from the actual story when applied to 3 alternative stories, drawn from the PASTA dataset ; slightly varying actions in the stories lead to different goal achievement outcomes.", + "text_raw": "\u2026stories shown: in alternative 1 (top right), a different Figure 1: A participant\u2019s goal inferred from the actual story when applied to 3 alternative stories, drawn from the PASTA dataset (Ghosh et al., 2023); slightly varying actions in the stories lead to different goal achievement outcomes.", + "intents": [ + "background" + ] + }, + { + "text": "PASTA introduced implied states that stories depend upon and examined the tasks of state inference from alternative stories and story rewriting for alternative states.", + "text_raw": "PASTA (Ghosh et al., 2023) introduced implied states that stories depend upon and examined the tasks of state inference from alternative stories and story rewriting for alternative states.", + "intents": [ + "background" + ] + }, + { + "text": "Third, while pre-trained large language models (LLMs) are powerful (Brown et al., 2020; Wei et al., 2022, inter alia ), they may perform poorly on tasks requiring robust reasoning ; Zellers et al., 2019; Qin et al., 2019).", + "text_raw": "Third, while pre-trained large language models (LLMs) are powerful (Brown et al., 2020; Wei et al., 2022, inter alia ), they may perform poorly on tasks requiring robust reasoning (Ghosh et al., 2023; Zellers et al., 2019; Qin et al., 2019).", + "intents": [ + "background" + ] + }, + { + "text": "We annotated ROC stories (Mostafazadeh et al., 2016) that have corresponding alternative stories in the PASTA dataset .", + "text_raw": "We annotated ROC stories (Mostafazadeh et al., 2016) that have corresponding alternative stories in the PASTA dataset (Ghosh et al., 2023).", + "intents": [ + "methodology" + ] + } + ] + }, + { + "citing_paper_id": "d723193c0453223803ffb926354cd6d2dee32b06", + "title": "EvEval: A Comprehensive Evaluation of Event Semantics for Large Language Models", + "external_ids": { + "ArXiv": "2305.15268", + "DBLP": "journals/corr/abs-2305-15268", + "DOI": "10.48550/arXiv.2305.15268", + "CorpusId": 258866165 + }, + "is_influential": false, + "contexts": [ + { + "text": "Learning other relations such as event temporality needs more context information and knowledge since it is more subjected to specific situations .", + "text_raw": "Learning other relations such as event temporality needs more context information and knowledge since it is more subjected to specific situations (Ghosh et al., 2022).", + "intents": [ + "background" + ] + } + ] + }, + { + "citing_paper_id": "293583961efdf22c0905cc04465aa88ea96826ec", + "title": "Causal schema induction for knowledge discovery", + "external_ids": { + "ArXiv": "2303.15381", + "DBLP": "journals/corr/abs-2303-15381", + "DOI": "10.48550/arXiv.2303.15381", + "CorpusId": 257766554 + }, + "is_influential": false, + "contexts": [ + { + "text": ", 2021), of participant states ; Vallurupalli et al., 2022), and of disease where organisms are conceived of as causative agents, e.", + "text_raw": ", 2021), of participant states (Ghosh et al., 2022; Vallurupalli et al., 2022), and of disease where organisms are conceived of as causative agents, e.", + "intents": [ + "background" + ] + }, + { + "text": "Participant-centeredness is featured in studies of narrative (Propp, 1968; Caselli and Vossen, 2016; Brahman et al., 2021), of participant states ; Vallurupalli et al., 2022), and of disease where organisms are conceived of as causative agents, e.g., the pathogen tubercle\u2026", + "text_raw": "Participant-centeredness is featured in studies of narrative (Propp, 1968; Caselli and Vossen, 2016; Brahman et al., 2021), of participant states (Ghosh et al., 2022; Vallurupalli et al., 2022), and of disease where organisms are conceived of as causative agents, e.g., the pathogen tubercle\u2026", + "intents": [ + "background" + ] + } + ] + }, + { + "citing_paper_id": "2b914691a44ab1c8bcb0104b8079d86908d98729", + "title": "Who Plays Which Role? Protagonist Detection and Classification in Moral Discourse", + "external_ids": { + "DBLP": "conf/eacl/SommerB26", + "DOI": "10.18653/v1/2026.eacl-srw.27", + "CorpusId": 286822758 + }, + "is_influential": false, + "contexts": [ + { + "text": "Beyond moral discourse, modeling actors and their roles has a long tradition in discourse analysis, for instance, through semantic role labeling (Ruppenhofer et al., 2009; Roth and Lap-ata, 2015; Bornheim et al., 2024) and participant modeling (Tilk et al., 2016; .", + "text_raw": "Beyond moral discourse, modeling actors and their roles has a long tradition in discourse analysis, for instance, through semantic role labeling (Ruppenhofer et al., 2009; Roth and Lap-ata, 2015; Bornheim et al., 2024) and participant modeling (Tilk et al., 2016; Ghosh et al., 2023).", + "intents": [] + } + ] + } + ] +} \ No newline at end of file diff --git a/replay_traces/2208.00329/processed_papers/2208.00329/usage_contributions.json b/replay_traces/2208.00329/processed_papers/2208.00329/usage_contributions.json new file mode 100644 index 0000000000000000000000000000000000000000..b558ea1d2ebd2ac65f2cd49a9564281ca4000891 --- /dev/null +++ b/replay_traces/2208.00329/processed_papers/2208.00329/usage_contributions.json @@ -0,0 +1,43 @@ +{ + "paper_id": "2208.00329", + "final_label": "EXTENDS", + "contributions": [ + { + "citing_paper_id": "494f3c55553161be831f725a7b55c6095470c5cb", + "citing_title": "SAGA: A Participant-specific Examination of Story Alternatives and Goal Applicability for a Deeper Understanding of Complex Events", + "label": "USES", + "paper_claim": "The citing paper adopts the methodology used for obtaining counterfactuals in the target paper to generate minimally updated goal annotations.", + "claim": "The citing paper adopts the methodology used for obtaining counterfactuals in the target paper to generate minimally updated goal annotations.", + "cluster_title": "Uses target methodology for dataset creation", + "cluster_key": "USES|methodology|dataset_creation", + "evidence_span": "reflecting the process used for obtaining the counterfactual in ~PASTA~", + "rationale": "The citing paper explicitly states that their dataset annotation process uses or mirrors the method used to obtain counterfactuals in the target paper.", + "contexts": [ + "Target-citing paragraph: We use a third HIT when one or more annotators identify that a participant's goal is inferrable from $S^{c_k}$. With this HIT we obtain a new set of goal annotations reusing and modifying the free-form text annotations from the actual story to obtain annotations that are also minimally updated reflecting the process used for obtaining the counterfactual in ~PASTA~. See \\cref{tab:HITS-annotation} for the annotations and the HITs used for obtaining them.", + "Target-citing paragraph: \\subsubsection{Worker Selection and Qualifications}\n\\label{app:HIT_information:qualifications}\nFor our initial alpha runs of the HITs we used all workers who meet our community standard quality criteria, such as requiring a 98\\% or greater HIT acceptance rate and the completion of 1000 approved HITs. In addition, we required the worker's stated location to be in the USA, UK, Canada, Australia, or New Zealand. We used the location requirement to avoid language-based artifacts given the language-dependent semantic phenomena and the high subjectivity our work can elicit. \nWe did not use requester-generated qualification tests, though in early iterations we found that annotators who had completed at least 50 HITs in our prior work~ provided the most reliable annotations; the vast majority of our responses are from this group." + ], + "source": "arxiv_paragraphs" + }, + { + "citing_paper_id": "ab73ca481256767addce56b20eb48ac183169e01", + "citing_title": "CoRE: Condition-based Reasoning for Identifying Outcome Variance in Complex Events", + "label": "USES", + "paper_claim": "The authors leverage the PASTA dataset and its annotations to generate and validate story outcomes.", + "claim": "The authors leverage the PASTA dataset and its annotations to generate and validate story outcomes.", + "cluster_title": "Uses target dataset for data generation", + "cluster_key": "USES|dataset|data_generation", + "evidence_span": "We leverage PASTA and SAGA (Vallurupalli et al., 2024) datasets to generate outcomes of SAGA and News Stories and validate these using both our generated and PASTA\u2019s crowd-annotated conditions.", + "rationale": "The citing paper explicitly uses the PASTA dataset and its annotated conditions to generate outcomes for SAGA and News Stories, as well as to validate them.", + "contexts": [ + "Target sentence: We expand the 3 categories used in PASTA , for error analysis on 200 random states on the story state inference task, to 4 categories and group all conditions into these as follows: (a) Physical: This category includes natural physical attributes of an entity such as size, age,\u2026", + "Target sentence: We leverage two previously released datasets PASTA and SAGA (Vallurupalli et al., 2024) consisting of participants\u2019 goal and state annotations (shown in Fig.", + "Target sentence: We leverage PASTA and SAGA (Vallurupalli et al., 2024) datasets to generate outcomes of SAGA and News Stories and validate these using both our generated and PASTA\u2019s crowd-annotated conditions. outcome is implicitly understood through constructing counterfactual mental\u2026" + ], + "source": "usage_contexts_fallback" + } + ], + "source": "arxiv_paragraphs", + "fallback_citing_ids": [ + "ab73ca481256767addce56b20eb48ac183169e01" + ] +} \ No newline at end of file diff --git a/replay_traces/2208.00329/processed_papers/2208.00329/usage_discovery_from_contributions.json b/replay_traces/2208.00329/processed_papers/2208.00329/usage_discovery_from_contributions.json new file mode 100644 index 0000000000000000000000000000000000000000..bf227abfb1abfd6f00e56482e4df36cfe9ba27b1 --- /dev/null +++ b/replay_traces/2208.00329/processed_papers/2208.00329/usage_discovery_from_contributions.json @@ -0,0 +1,38 @@ +{ + "paper_id": "2208.00329", + "decision": "", + "justification": "", + "clusters": [ + { + "cluster_id": "C1", + "count": "1", + "representative_claim": "Uses target methodology for dataset creation", + "cluster_key": "USES|methodology|dataset_creation", + "cluster_title": "Uses target methodology for dataset creation", + "claim_indices": [ + "0" + ], + "source_cluster_ids": [ + "C1" + ], + "merge_rationale": "The citing paper adopts the methodology used in the target paper to generate its own dataset." + }, + { + "cluster_id": "C2", + "count": "1", + "representative_claim": "Uses target dataset for data generation", + "cluster_key": "USES|dataset|data_generation", + "cluster_title": "Uses target dataset for data generation", + "claim_indices": [ + "1" + ], + "source_cluster_ids": [ + "C2" + ], + "merge_rationale": "The citing paper uses the PASTA dataset and its annotations directly to generate and validate story outcomes." + } + ], + "dropped_clusters": [], + "cluster_refine_method": "llm_centroid_merge_filter", + "cluster_refine_source": "usage_contributions.json" +} \ No newline at end of file diff --git a/replay_traces/2208.00329/processed_papers/2208.00329/usage_uses_extends_verified.json b/replay_traces/2208.00329/processed_papers/2208.00329/usage_uses_extends_verified.json new file mode 100644 index 0000000000000000000000000000000000000000..0aac8215ff8aeed38e74b0bc7271b03148e0dc11 --- /dev/null +++ b/replay_traces/2208.00329/processed_papers/2208.00329/usage_uses_extends_verified.json @@ -0,0 +1,150 @@ +{ + "paper_id": "2208.00329", + "target": { + "title": "PASTA: A Dataset for Modeling PArticipant STAtes in Narratives", + "first_author_last": "Ghosh", + "year": "2022" + }, + "candidates_total": 9, + "candidates_considered": 9, + "verification_batch_size": 25, + "verification_num_batches": 1, + "candidates_selected": 4, + "verified": [ + { + "id": 8, + "label": "EXTENDS", + "cue_span": "We expand the 3 categories used in PASTA", + "rationale": "The citing paper explicitly expands upon the categorical framework introduced in the target paper.", + "text": "We expand the 3 categories used in PASTA , for error analysis on 200 random states on the story state inference task, to 4 categories and group all conditions into these as follows: (a) Physical: This category includes natural physical attributes of an entity such as size, age,\u2026", + "citing_paper_id": "ab73ca481256767addce56b20eb48ac183169e01", + "citing_title": "CoRE: Condition-based Reasoning for Identifying Outcome Variance in Complex Events", + "original_label": "Uses" + }, + { + "id": 10, + "label": "USES", + "cue_span": "We leverage PASTA and SAGA (Vallurupalli et al., 2024) datasets to generate outcomes", + "rationale": "The authors explicitly state they are leveraging the target dataset to generate and validate outcomes.", + "text": "We leverage PASTA and SAGA (Vallurupalli et al., 2024) datasets to generate outcomes of SAGA and News Stories and validate these using both our generated and PASTA\u2019s crowd-annotated conditions. outcome is implicitly understood through constructing counterfactual mental\u2026", + "citing_paper_id": "ab73ca481256767addce56b20eb48ac183169e01", + "citing_title": "CoRE: Condition-based Reasoning for Identifying Outcome Variance in Complex Events", + "original_label": "Uses" + }, + { + "id": 16, + "label": "USES", + "cue_span": "We annotated ROC stories (Mostafazadeh et al., 2016) that have corresponding alternative stories in the PASTA dataset", + "rationale": "The authors used stories specifically from the PASTA dataset for their own annotation process.", + "text": "We annotated ROC stories (Mostafazadeh et al., 2016) that have corresponding alternative stories in the PASTA dataset .", + "citing_paper_id": "494f3c55553161be831f725a7b55c6095470c5cb", + "citing_title": "SAGA: A Participant-specific Examination of Story Alternatives and Goal Applicability for a Deeper Understanding of Complex Events", + "original_label": "Uses" + }, + { + "id": 9, + "label": "USES", + "cue_span": "We leverage two previously released datasets PASTA", + "rationale": "The authors explicitly mention leveraging the PASTA dataset as part of their work.", + "text": "We leverage two previously released datasets PASTA and SAGA (Vallurupalli et al., 2024) consisting of participants\u2019 goal and state annotations (shown in Fig.", + "citing_paper_id": "ab73ca481256767addce56b20eb48ac183169e01", + "citing_title": "CoRE: Condition-based Reasoning for Identifying Outcome Variance in Complex Events", + "original_label": "Uses" + }, + { + "id": 4, + "label": "NOT_CONFIRMED", + "cue_span": "", + "rationale": "The sentence only indicates an intention to describe the datasets, not a specific use or extension of them.", + "text": "In the following paragraphs, we briefly describe the PASTA and SAGA (Vallu-rupalli et al., 2024) datasets and how these datasets\u2019 annotations relate to conditions and outcomes.", + "citing_paper_id": "ab73ca481256767addce56b20eb48ac183169e01", + "citing_title": "CoRE: Condition-based Reasoning for Identifying Outcome Variance in Complex Events", + "original_label": "Uses" + }, + { + "id": 1, + "label": "NOT_CONFIRMED", + "cue_span": "", + "rationale": "The sentence provides a description of what the target paper introduces without stating that the citing paper uses it.", + "text": "The PASTA dataset is an extension of ROCStories that substitutes alternative events into the stories, which may or may not match up with the original summary sentence .", + "citing_paper_id": "c1d02fc3786fb5e5b0543703af95046253109d4e", + "citing_title": "Inductive Bias Extraction and Matching for LLM Prompts", + "original_label": "Uses" + }, + { + "id": 11, + "label": "NOT_CONFIRMED", + "cue_span": "", + "rationale": "The sentence mentions that the resulting annotations reflect a process from the target paper but lacks an explicit adoption cue.", + "text": "With this HIT we obtain a new set of goal annotations reusing and modifying the free-form text annotations from the actual story to obtain annotations that are also minimally updated reflecting the process used for obtaining the counterfactual in PASTA .", + "citing_paper_id": "494f3c55553161be831f725a7b55c6095470c5cb", + "citing_title": "SAGA: A Participant-specific Examination of Story Alternatives and Goal Applicability for a Deeper Understanding of Complex Events", + "original_label": "Uses" + }, + { + "id": 13, + "label": "NOT_CONFIRMED", + "cue_span": "", + "rationale": "The sentence lacks an explicit citing-paper actor phrase such as 'we' or 'our' as required by the instructions.", + "text": "\u2026stories shown: in alternative 1 (top right), a different Figure 1: A participant\u2019s goal inferred from the actual story when applied to 3 alternative stories, drawn from the PASTA dataset ; slightly varying actions in the stories lead to different goal achievement outcomes.", + "citing_paper_id": "494f3c55553161be831f725a7b55c6095470c5cb", + "citing_title": "SAGA: A Participant-specific Examination of Story Alternatives and Goal Applicability for a Deeper Understanding of Complex Events", + "original_label": "Uses" + }, + { + "id": 2, + "label": "NOT_CONFIRMED", + "cue_span": "", + "rationale": "The sentence is a general description of dataset subsets and lacks an explicit adoption cue by the citing authors.", + "text": "The dataset is often separated into four subsets: training, testing, evaluation, and validation [32\u201339].", + "citing_paper_id": "a6924f6b715327b0f24d044e9e8ef799b3f4a0a2", + "citing_title": "Electrifying the Story Climax via Axial Self-Attention in GPT-3 Transformer Block", + "original_label": "Uses" + } + ], + "confirmed": [ + { + "id": 8, + "label": "EXTENDS", + "cue_span": "We expand the 3 categories used in PASTA", + "rationale": "The citing paper explicitly expands upon the categorical framework introduced in the target paper.", + "text": "We expand the 3 categories used in PASTA , for error analysis on 200 random states on the story state inference task, to 4 categories and group all conditions into these as follows: (a) Physical: This category includes natural physical attributes of an entity such as size, age,\u2026", + "citing_paper_id": "ab73ca481256767addce56b20eb48ac183169e01", + "citing_title": "CoRE: Condition-based Reasoning for Identifying Outcome Variance in Complex Events", + "original_label": "Uses" + }, + { + "id": 10, + "label": "USES", + "cue_span": "We leverage PASTA and SAGA (Vallurupalli et al., 2024) datasets to generate outcomes", + "rationale": "The authors explicitly state they are leveraging the target dataset to generate and validate outcomes.", + "text": "We leverage PASTA and SAGA (Vallurupalli et al., 2024) datasets to generate outcomes of SAGA and News Stories and validate these using both our generated and PASTA\u2019s crowd-annotated conditions. outcome is implicitly understood through constructing counterfactual mental\u2026", + "citing_paper_id": "ab73ca481256767addce56b20eb48ac183169e01", + "citing_title": "CoRE: Condition-based Reasoning for Identifying Outcome Variance in Complex Events", + "original_label": "Uses" + }, + { + "id": 16, + "label": "USES", + "cue_span": "We annotated ROC stories (Mostafazadeh et al., 2016) that have corresponding alternative stories in the PASTA dataset", + "rationale": "The authors used stories specifically from the PASTA dataset for their own annotation process.", + "text": "We annotated ROC stories (Mostafazadeh et al., 2016) that have corresponding alternative stories in the PASTA dataset .", + "citing_paper_id": "494f3c55553161be831f725a7b55c6095470c5cb", + "citing_title": "SAGA: A Participant-specific Examination of Story Alternatives and Goal Applicability for a Deeper Understanding of Complex Events", + "original_label": "Uses" + }, + { + "id": 9, + "label": "USES", + "cue_span": "We leverage two previously released datasets PASTA", + "rationale": "The authors explicitly mention leveraging the PASTA dataset as part of their work.", + "text": "We leverage two previously released datasets PASTA and SAGA (Vallurupalli et al., 2024) consisting of participants\u2019 goal and state annotations (shown in Fig.", + "citing_paper_id": "ab73ca481256767addce56b20eb48ac183169e01", + "citing_title": "CoRE: Condition-based Reasoning for Identifying Outcome Variance in Complex Events", + "original_label": "Uses" + } + ], + "confirmed_extends": 1, + "confirmed_uses": 3, + "final_label": "EXTENDS" +} \ No newline at end of file diff --git a/replay_traces/2208.00329/replay_meta.json b/replay_traces/2208.00329/replay_meta.json new file mode 100644 index 0000000000000000000000000000000000000000..1746811f11c8a45b30a52fec7fbc4eeff7425c2b --- /dev/null +++ b/replay_traces/2208.00329/replay_meta.json @@ -0,0 +1,34 @@ +{ + "label": "PASTA: Participant States in Narratives", + "arxiv_id": "2208.00329", + "paper_input": "https://arxiv.org/abs/2208.00329", + "status": "Completed", + "built_at": "2026-07-26T20:59:24.854277+00:00", + "source_job_dir": "/Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785099279_8df5ecbb", + "paper_dir": "processed_papers/2208.00329", + "annotation_payload_path": "two_pass_outputs/20260726T205558Z/pass_2_ui_payload.json", + "annotation_skipped_reason": null, + "pipeline_failed_reason": null, + "pipeline_stopped_reason": null, + "events": [ + "Step 1/8: Fetch metadata + LaTeX for input paper", + "Step 1 complete", + "Step 2/8: Add citation markers", + "Step 2 complete", + "Step 3/8: Build usage contexts", + "Step 3 complete", + "Step 4/8: Label citation functions", + "Step 4 complete", + "Step 5/8: Verify USES/EXTENDS", + "Step 5 complete", + "Step 6/8: Extract arXiv paragraphs", + "Step 6 complete", + "Step 7/8: Extract target contributions and refine clusters", + "Step 7 complete", + "Pipeline completed successfully.", + "[annotation] complete: /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785099279_8df5ecbb/two_pass_outputs/2208.00329/20260726T205558Z" + ], + "cluster_count": 2, + "contribution_count": 2, + "claim_count": 2 +} \ No newline at end of file diff --git a/replay_traces/2208.00329/run_config.json b/replay_traces/2208.00329/run_config.json new file mode 100644 index 0000000000000000000000000000000000000000..1076e96e103145f1c1ecb0bd8d480af463c3d727 --- /dev/null +++ b/replay_traces/2208.00329/run_config.json @@ -0,0 +1,22 @@ +{ + "paper_input": "https://arxiv.org/abs/2208.00329", + "paper_id": "2208.00329", + "arxiv_id": "2208.00329", + "source_root": "/Users/dingzifeng/SciFy-UMBC/demo_2026_07/src/processed_papers", + "steps": [ + 1, + 2, + 3, + 4, + 5, + 6, + 7, + "annotation" + ], + "llm_provider": "gemini", + "llm_model": "gemini-3.1-pro-preview", + "llm_model_step4": "gemini-3-flash-preview", + "device": "cpu", + "embedding_model": "sentence-transformers/all-mpnet-base-v2", + "timestamp": 1785099279 +} \ No newline at end of file diff --git a/replay_traces/2208.00329/scipaths_run_results.json b/replay_traces/2208.00329/scipaths_run_results.json new file mode 100644 index 0000000000000000000000000000000000000000..98806409835d92b0fdbf67753b2bfeb69347e56a --- /dev/null +++ b/replay_traces/2208.00329/scipaths_run_results.json @@ -0,0 +1,233 @@ +{ + "citation_clusters": [ + { + "cluster_id": "C1", + "count": "1", + "representative_claim": "Uses target methodology for dataset creation", + "cluster_key": "USES|methodology|dataset_creation", + "cluster_title": "Uses target methodology for dataset creation", + "claim_indices": [ + "0" + ], + "source_cluster_ids": [ + "C1" + ], + "merge_rationale": "The citing paper adopts the methodology used in the target paper to generate its own dataset." + }, + { + "cluster_id": "C2", + "count": "1", + "representative_claim": "Uses target dataset for data generation", + "cluster_key": "USES|dataset|data_generation", + "cluster_title": "Uses target dataset for data generation", + "claim_indices": [ + "1" + ], + "source_cluster_ids": [ + "C2" + ], + "merge_rationale": "The citing paper uses the PASTA dataset and its annotations directly to generate and validate story outcomes." + } + ], + "target_contribution_decompositions": [ + { + "claim_id": "C1", + "text": "Dataset: A crowdsourced dataset of narrative stories annotated with implicit participant states, counterfactual perturbations, and revised stories, enabling state-based reasoning and generation tasks.", + "rewritten_claim": "Dataset: A crowdsourced dataset of narrative stories annotated with implicit participant states, counterfactual perturbations, and revised stories, enabling state-based reasoning and generation tasks.", + "cluster_id": "C2", + "decision": "YES_SUFFICIENT", + "notes": "", + "ingredients": [ + { + "ingredient_id": "C1.I1", + "ingredient": "Source corpus of short, self-contained narrative stories", + "canonical_ref_id": "A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories", + "canonical_grounding": { + "ref_id": "A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories", + "ref_year": "", + "ref_authors": "" + }, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "DATA_SOURCE", + "roles": [ + "DATA_SOURCE" + ], + "contribution": "Provides the base narrative stories (ROCStories) that are annotated and revised in the dataset.", + "rationale": "The dataset requires a substrate of short, commonsense-rich narratives to serve as the foundation for state inference and revision. ROCStories is the direct source used by the paper and maps cleanly to its canonical study.", + "evidence_span": "To create the PASTA dataset, we use stories from the extended ROCStories (Mostafazadeh et al., 2016) corpus for annotation by crowd workers." + } + }, + { + "ingredient_id": "C1.I2", + "ingredient": "Four-step crowdsourcing protocol for inferring states and generating counterfactual revisions", + "canonical_ref_id": "__NONE__", + "canonical_grounding": null, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "EVALUATION_PROTOCOL", + "roles": [ + "EVALUATION_PROTOCOL" + ], + "contribution": "Defines the workflow used to collect the structured annotations that make up the dataset.", + "rationale": "The specific four-step process that yields the dataset's tuples is a novel contribution of the target paper itself, so it maps cleanly to NONE.", + "evidence_span": "The annotation process has four main steps: 1. Infer a participant state... 2. Select minimal justification sentences... 3. Perturb the state... 4. Revise the story" + } + } + ], + "enabling_discoveries": [ + { + "ref_id": "A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories", + "ref_year": "", + "ref_authors": "", + "ingredient_id": "C1.I1", + "ingredient": "Source corpus of short, self-contained narrative stories", + "role": "DATA_SOURCE", + "roles": [ + "DATA_SOURCE" + ], + "contribution": "Provides the base narrative stories (ROCStories) that are annotated and revised in the dataset.", + "rationale": "The dataset requires a substrate of short, commonsense-rich narratives to serve as the foundation for state inference and revision. ROCStories is the direct source used by the paper and maps cleanly to its canonical study.", + "evidence_span": "To create the PASTA dataset, we use stories from the extended ROCStories (Mostafazadeh et al., 2016) corpus for annotation by crowd workers." + } + ] + }, + { + "claim_id": "C2", + "text": "Method: A crowdsourcing protocol for eliciting implicit participant states and counterfactual story revisions, enabling the construction of state-based narrative datasets.", + "rewritten_claim": "Method: A crowdsourcing protocol for eliciting implicit participant states and counterfactual story revisions, enabling the construction of state-based narrative datasets.", + "cluster_id": "C1", + "decision": "YES_SUFFICIENT", + "notes": "", + "ingredients": [ + { + "ingredient_id": "C2.I1", + "ingredient": "Counterfactual story rewriting framework", + "canonical_ref_id": "Counterfactual Story Reasoning and Generation", + "canonical_grounding": { + "ref_id": "Counterfactual Story Reasoning and Generation", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "Counterfactual Story Reasoning and Generation", + "ref_year": "", + "ref_authors": "" + }, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "CONCEPTUAL_FRAMEWORK", + "roles": [ + "CONCEPTUAL_FRAMEWORK" + ], + "contribution": "Provides the foundational concept of editing a short story based on a counterfactual context, which the protocol adapts for unobserved participant states.", + "rationale": "The methodology extends the idea of counterfactual story rewriting to implicit states. Qin et al. (TIME-TRAVEL) is explicitly cited as the most similar prior work that introduced this rewriting task framing.", + "evidence_span": "Most similar to our work is the TIME-TRAVEL dataset by Qin et al. (2019). It includes the Counterfactual Story Rewriting task" + } + }, + { + "ingredient_id": "C2.I2", + "ingredient": "Framework for annotating participant states in simple stories", + "canonical_ref_id": "Modeling Naive Psychology of Characters in Simple Commonsense Stories", + "canonical_grounding": { + "ref_id": "Modeling Naive Psychology of Characters in Simple Commonsense Stories", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "Modeling Naive Psychology of Characters in Simple Commonsense Stories", + "ref_year": "", + "ref_authors": "" + }, + "additional_ref_ids": [ + "Tracking State Changes in Procedural Text: a Challenge Dataset and Models for Process Paragraph Comprehension" + ], + "additional_groundings": [ + { + "ref_id": "Tracking State Changes in Procedural Text: a Challenge Dataset and Models for Process Paragraph Comprehension", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "Tracking State Changes in Procedural Text: a Challenge Dataset and Models for Process Paragraph Comprehension", + "ref_year": "", + "ref_authors": "" + } + ], + "canonical_annotation": { + "role": "CONCEPTUAL_FRAMEWORK", + "roles": [ + "CONCEPTUAL_FRAMEWORK" + ], + "contribution": "Provides the conceptual basis for annotating the implicit states (like emotions or motivations) of characters in short narratives.", + "rationale": "The methodology builds on prior work that established the tracking and annotation of entity/participant states in text. Rashkin et al. is a clean representative for tracking emotional/psychological states in simple stories, with Dalvi et al. serving as an additional grounding for physical states.", + "evidence_span": "Rashkin et al. (2018) tracks the emotional reactions and motivations of characters in simple stories, for a fixed/small set of attributes." + } + }, + { + "ingredient_id": "C2.I3", + "ingredient": "Two-stage quality control and worker-proficiency filtering mechanism", + "canonical_ref_id": "__NONE__", + "canonical_grounding": null, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "EVALUATION_PROTOCOL", + "roles": [ + "EVALUATION_PROTOCOL" + ], + "contribution": "Provides the filtering and expert-review workflow necessary to identify proficient workers and ensure high-quality complex annotations.", + "rationale": "The specific two-stage filtering process (strict initial criteria followed by expert review to identify \"proficient workers\") is a custom workflow designed in the target paper to make the complex annotation task viable. It maps cleanly to NONE.", + "evidence_span": "For filtering out noisy data from the collected responses, we follow a two-stage filtering process." + } + } + ], + "enabling_discoveries": [ + { + "ref_id": "Counterfactual Story Reasoning and Generation", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "Counterfactual Story Reasoning and Generation", + "ref_year": "", + "ref_authors": "", + "ingredient_id": "C2.I1", + "ingredient": "Counterfactual story rewriting framework", + "role": "CONCEPTUAL_FRAMEWORK", + "roles": [ + "CONCEPTUAL_FRAMEWORK" + ], + "contribution": "Provides the foundational concept of editing a short story based on a counterfactual context, which the protocol adapts for unobserved participant states.", + "rationale": "The methodology extends the idea of counterfactual story rewriting to implicit states. Qin et al. (TIME-TRAVEL) is explicitly cited as the most similar prior work that introduced this rewriting task framing.", + "evidence_span": "Most similar to our work is the TIME-TRAVEL dataset by Qin et al. (2019). It includes the Counterfactual Story Rewriting task" + }, + { + "ref_id": "Modeling Naive Psychology of Characters in Simple Commonsense Stories", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "Modeling Naive Psychology of Characters in Simple Commonsense Stories", + "ref_year": "", + "ref_authors": "", + "ingredient_id": "C2.I2", + "ingredient": "Framework for annotating participant states in simple stories", + "role": "CONCEPTUAL_FRAMEWORK", + "roles": [ + "CONCEPTUAL_FRAMEWORK" + ], + "contribution": "Provides the conceptual basis for annotating the implicit states (like emotions or motivations) of characters in short narratives.", + "rationale": "The methodology builds on prior work that established the tracking and annotation of entity/participant states in text. Rashkin et al. is a clean representative for tracking emotional/psychological states in simple stories, with Dalvi et al. serving as an additional grounding for physical states.", + "evidence_span": "Rashkin et al. (2018) tracks the emotional reactions and motivations of characters in simple stories, for a fixed/small set of attributes." + } + ] + } + ] +} \ No newline at end of file diff --git a/replay_traces/2208.00329/summary.txt b/replay_traces/2208.00329/summary.txt new file mode 100644 index 0000000000000000000000000000000000000000..91d02a8ae8d3807e9176d9ede6b5cb56952461ca --- /dev/null +++ b/replay_traces/2208.00329/summary.txt @@ -0,0 +1,8 @@ +[1] Fetch metadata + LaTeX for input paper +[2] Add citation markers +[3] Build usage contexts +[4] Label citation functions +[5] Verify USES/EXTENDS +[6] Extract arXiv paragraphs +[7] Extract target contributions and refine clusters +SUCCESS \ No newline at end of file diff --git a/replay_traces/2208.00329/two_pass_outputs/20260726T205558Z/formatter_attempts.json b/replay_traces/2208.00329/two_pass_outputs/20260726T205558Z/formatter_attempts.json new file mode 100644 index 0000000000000000000000000000000000000000..5f6accf4d248a65ffd7bf31bd6b1a0e921c17231 --- /dev/null +++ b/replay_traces/2208.00329/two_pass_outputs/20260726T205558Z/formatter_attempts.json @@ -0,0 +1,8 @@ +{ + "attempts": [ + { + "attempt": 1, + "status": "success" + } + ] +} diff --git a/replay_traces/2208.00329/two_pass_outputs/20260726T205558Z/pass_1_reasoning.judge.output.json b/replay_traces/2208.00329/two_pass_outputs/20260726T205558Z/pass_1_reasoning.judge.output.json new file mode 100644 index 0000000000000000000000000000000000000000..5e3602f272058d21ce5fbdb94c466d97071f1eda --- /dev/null +++ b/replay_traces/2208.00329/two_pass_outputs/20260726T205558Z/pass_1_reasoning.judge.output.json @@ -0,0 +1,25 @@ +{ + "selected_candidate_index": 0, + "selected_candidate_id": "candidate_1", + "selected_reason": "Candidate 1 best recovers the minimal structurally sufficient ingredients by correctly identifying both conceptual pillars of the paper (counterfactual rewriting via Qin et al., and participant state tracking via Rashkin et al.). It correctly grounds the source data in ROCStories and uses NONE appropriately for the custom annotation protocol and quality control. It captures the full structural recipe of the paper's contributions without unnecessary redundancy.", + "candidate_scores": [ + { + "candidate_id": "candidate_1", + "candidate_index": 0, + "score": 9, + "assessment": "Strengths: Excellent identification of the two conceptual pillars of the methodology (Qin et al. for counterfactuals, Rashkin et al. for participant states). Correctly grounds the source corpus in ROCStories. Good use of NONE for the custom quality control and 4-step protocol. Avoids redundancy between the two claims. Weaknesses: The 4-step protocol is the core of the methodology, so placing it only in the dataset claim (C1) and omitting it from the methodology claim (C2) is slightly counterintuitive, though it successfully avoids ingredient duplication." + }, + { + "candidate_id": "candidate_2", + "candidate_index": 1, + "score": 7, + "assessment": "Strengths: Correctly identifies the 4-step protocol as the CORE_METHOD for the methodology claim. Correctly grounds the counterfactual framework in Qin et al. Weaknesses: Misses the crucial conceptual framework for participant state tracking (Rashkin et al.), which is a foundational part of the paper's core contribution. Duplicates the annotation protocol across both claims." + }, + { + "candidate_id": "candidate_3", + "candidate_index": 2, + "score": 5, + "assessment": "Strengths: Correctly splits the claims based on downstream usage and identifies the source corpus. Weaknesses: Fails to identify the canonical prior works for the conceptual frameworks (Qin et al. and Rashkin et al.), incorrectly defaulting to NONE. Duplicates the annotation workflow across both claims." + } + ] +} diff --git a/replay_traces/2208.00329/two_pass_outputs/20260726T205558Z/pass_1_reasoning.judge.prompt.txt b/replay_traces/2208.00329/two_pass_outputs/20260726T205558Z/pass_1_reasoning.judge.prompt.txt new file mode 100644 index 0000000000000000000000000000000000000000..a208505236732b3a5fbd2b56fb209e9528caa6f6 --- /dev/null +++ b/replay_traces/2208.00329/two_pass_outputs/20260726T205558Z/pass_1_reasoning.judge.prompt.txt @@ -0,0 +1,898 @@ +Task: Compare candidate reasoning memos for a single target paper. + +You will be given: +1. target paper metadata +2. the original extracted discovery claim +3. optional downstream usage evidence +4. multiple candidate reasoning outputs from pass 1 + +Choose the candidate that best matches the minimal structurally sufficient enabling-discovery annotation objective. + +Important: +- Judge candidates using the full critic policy from the system instructions. +- Do not reward verbosity by itself. +- Do not reward confidence by itself. +- Do not reward richer implementation stories, more citations, or more operational detail unless they are structurally necessary. +- Prefer candidates that recover the smallest structurally sufficient ingredient set for each downstream-used claim. +- Prefer candidates that minimize formatter guesswork. + +Apply these priorities in order: +1. Correct downstream-used claim selection +2. Correct claim splitting and abstraction level +3. Smallest structurally sufficient ingredient set +4. Correct exclusion of tempting non-ingredients +5. Correct canonical vs additional vs "__NONE__" grounding decisions +6. Correct role assignments +7. Evidence quality and clarity +8. Overall readiness for formatter conversion + +Specific judging reminders: +- Prefer downstream-first claim identification over abstract-first or citation-first reasoning. +- Penalize candidates that annotate unused paper outputs as claims. +- Penalize candidates that oversplit one artifact into multiple claims without downstream justification. +- Penalize candidates that include one-task helpers, local preprocessing, balancing tricks, token/language-tag details, helper models for one small step, benchmark-use ingredients, model-training conveniences, or other non-structural details. +- For benchmark/dataset claims, prefer benchmark framing, source-task/source-data substrate, benchmark-wide annotation/curation/verification protocol, benchmark-defining task formulation/protocol, and any core conceptual trick or idea. +- For method/resource claims, prefer core architecture, core conceptual trick or idea, core objective(s), structurally necessary training data, and essential initialization/checkpoints. +- For finding claims, prefer the model/artifact type under study, task/benchmark substrate, evaluation/measurement protocol, and any core conceptual trick or idea needed to make the finding observable. +- Prefer "__NONE__" when an ingredient is composite, same-paper dominant, benchmark-wide/corpus-wide, or would be distorted by forcing one canonical prior study. +- Penalize candidates that force canonical studies where "__NONE__" is more faithful. +- Penalize candidates that choose famous or cited papers rather than the most structurally faithful grounding. +- Prefer role assignments that are conservative and faithful to the ingredient's function. + +TARGET PAPER METADATA: +{ + "paperId": "e894fb15054d3bc9659060406a12dfd1055ae32e", + "externalIds": { + "DBLP": "journals/corr/abs-2208-00329", + "ArXiv": "2208.00329", + "DOI": "10.1162/tacl_a_00600", + "CorpusId": 251223896 + }, + "url": "https://www.semanticscholar.org/paper/e894fb15054d3bc9659060406a12dfd1055ae32e", + "title": "PASTA: A Dataset for Modeling PArticipant STAtes in Narratives", + "venue": "Transactions of the Association for Computational Linguistics", + "year": 2022, + "referenceCount": 59, + "citationCount": 7, + "openAccessPdf": { + "url": "https://direct.mit.edu/tacl/article-pdf/doi/10.1162/tacl_a_00600/2173956/tacl_a_00600.pdf", + "status": "GOLD", + "license": "CCBY", + "disclaimer": "Notice: Paper or abstract available at https://arxiv.org/abs/2208.00329, which is subject to the license by the author or copyright owner provided with this content. Please go to the source to verify the license and copyright information for your use." + }, + "tldr": { + "model": "tldr@v2.0.0", + "text": "This work introduces a new crowdsourced English-language, Participant States dataset, PASTA, and introduces three state-based reasoning tasks that test for the ability to infer when a state is entailed by a story, to revise a story conditioned on a counterfactual state, and to explain the most likely state change given a revised story." + }, + "publicationDate": "2022-07-31", + "authors": [ + { + "authorId": "72059959", + "name": "Sayontan Ghosh" + }, + { + "authorId": "144007901", + "name": "Mahnaz Koupaee" + }, + { + "authorId": "2254303864", + "name": "I-Sen Chen" + }, + { + "authorId": "2034063", + "name": "Francis Ferraro" + }, + { + "authorId": "1729918", + "name": "Nathanael Chambers" + }, + { + "authorId": "35217367", + "name": "Niranjan Balasubramanian" + } + ], + "abstract": "Abstract The events in a narrative are understood as a coherent whole via the underlying states of their participants. Often, these participant states are not explicitly mentioned, instead left to be inferred by the reader. A model that understands narratives should likewise infer these implicit states, and even reason about the impact of changes to these states on the narrative. To facilitate this goal, we introduce a new crowdsourced English-language, Participant States dataset, PASTA. This dataset contains inferable participant states; a counterfactual perturbation to each state; and the changes to the story that would be necessary if the counterfactual were true. We introduce three state-based reasoning tasks that test for the ability to infer when a state is entailed by a story, to revise a story conditioned on a counterfactual state, and to explain the most likely state change given a revised story. Experiments show that today\u2019s LLMs can reason about states to some degree, but there is large room for improvement, especially in problems requiring access and ability to reason with diverse types of knowledge (e.g., physical, numerical, factual).1" +} + +ORIGINAL DISCOVERY CLAIM: + + +OPTIONAL DOWNSTREAM USAGE EVIDENCE: +[ + { + "cluster_id": "C1", + "representative_claim": "Uses target methodology for dataset creation", + "cluster_title": "Uses target methodology for dataset creation", + "count": "1", + "merge_rationale": "The citing paper adopts the methodology used in the target paper to generate its own dataset." + }, + { + "cluster_id": "C2", + "representative_claim": "Uses target dataset for data generation", + "cluster_title": "Uses target dataset for data generation", + "count": "1", + "merge_rationale": "The citing paper uses the PASTA dataset and its annotations directly to generate and validate story outcomes." + } +] + +CANDIDATE REASONING OUTPUTS: +CANDIDATE 1: candidate_1 +------------------------------------------------------------ +# 1. Cluster Evidence + +Based on the downstream cluster evidence, the paper introduces two distinct downstream-used contributions: +- **C1:** The methodology/protocol for dataset creation, which is reused by later work to generate their own datasets. +- **C2:** The PASTA dataset itself, which is reused directly for data generation and validation. + +Because these two contributions have distinct downstream roles (one is reused as a data artifact, the other as a data-collection process) and require different structural ingredients, they should be split into two atomic claims. + +--- + +# 2. Claim Split Decision + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C2 + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Method +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C1 + +### Why this split is correct +The two artifacts have distinct downstream roles and distinct ingredient decompositions: +- **The PASTA Dataset** is reused as a concrete data resource for training and evaluation. It structurally depends on the specific source texts it annotates. +- **The Dataset Creation Methodology** is reused as a process/protocol to build new datasets. It structurally depends on conceptual frameworks for counterfactuals and quality control, but is agnostic to the specific source text. +They should therefore be annotated separately. + +--- + +# 3. Rewritten Claims + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **rewritten_claim:** **Dataset: A crowdsourced dataset of narrative stories annotated with implicit participant states, counterfactual perturbations, and revised stories, enabling state-based reasoning and generation tasks.** +- **why_this_is_atomic:** This claim isolates the dataset artifact itself, separate from the methodology used to create it. +- **optional cluster_id:** C2 +- **decision:** YES_SUFFICIENT + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Method +- **rewritten_claim:** **Method: A crowdsourcing protocol for eliciting implicit participant states and counterfactual story revisions, enabling the construction of state-based narrative datasets.** +- **why_this_is_atomic:** This claim isolates the reusable annotation methodology and quality-control process, distinct from the specific PASTA dataset produced by it. +- **optional cluster_id:** C1 +- **decision:** YES_SUFFICIENT + +--- + +# 4. Minimal Ingredient Set Per Claim + +## Claim C1: PASTA Dataset + +### Ingredient C1.I1 +- **ingredient_id:** C1.I1 +- **ingredient:** Source corpus of short, self-contained narrative stories +- **why_structurally_necessary:** The dataset annotates implicit states and counterfactuals over existing narratives. Without a source corpus of narratives to annotate, the dataset would not exist in its claimed form. +- **why_not_lower_level_substeps:** This should remain at the level of the source corpus rather than being split into specific story filters, splits, or sampling strategies. +- **why_not_adjacent_implementation_details:** The exact size of the sample is an implementation detail; the structural dependency is the narrative text substrate itself. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories +- **additional groundings if any:** none +- **role:** DATA_SOURCE +- **contribution:** Provides the base narrative stories (ROCStories) that are annotated and revised in the dataset. +- **rationale:** The dataset requires a substrate of short, commonsense-rich narratives to serve as the foundation for state inference and revision. ROCStories is the direct source used by the paper and maps cleanly to its canonical study. +- **evidence_span:** “To create the PASTA dataset, we use stories from the extended ROCStories (Mostafazadeh et al., 2016) corpus for annotation by crowd workers.” + +### Ingredient C1.I2 +- **ingredient_id:** C1.I2 +- **ingredient:** Four-step crowdsourcing protocol for inferring states and generating counterfactual revisions +- **why_structurally_necessary:** The dataset's unique structure (inferred state, justification, counterfactual state, revised story) is entirely defined by this specific annotation workflow. Without it, the artifact would not have its defining shape. +- **why_not_lower_level_substeps:** This should remain one protocol-level ingredient rather than being split into the four individual steps or the specific MTurk UI details. +- **why_not_adjacent_implementation_details:** Quality control and worker filtering are important but subordinate to the core four-step annotation protocol that defines the data structure for this claim. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Defines the workflow used to collect the structured annotations that make up the dataset. +- **rationale:** The specific four-step process that yields the dataset's tuples is a novel contribution of the target paper itself, so it maps cleanly to `NONE`. +- **evidence_span:** “The annotation process has four main steps: 1. Infer a participant state... 2. Select minimal justification sentences... 3. Perturb the state... 4. Revise the story” + +--- + +## Claim C2: Dataset Creation Methodology + +### Ingredient C2.I1 +- **ingredient_id:** C2.I1 +- **ingredient:** Counterfactual story rewriting framework +- **why_structurally_necessary:** The protocol requires workers to revise a story based on a counterfactual state. This builds fundamentally on the conceptual framework of counterfactual story rewriting. +- **why_not_lower_level_substeps:** This remains at the conceptual level of counterfactual rewriting rather than specific model architectures or generation constraints. +- **why_not_adjacent_implementation_details:** The specific position of the counterfactual (unobserved vs explicit) is the target paper's contribution, but the foundational concept is counterfactual rewriting. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Counterfactual Story Reasoning and Generation +- **additional groundings if any:** none +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the foundational concept of editing a short story based on a counterfactual context, which the protocol adapts for unobserved participant states. +- **rationale:** The methodology extends the idea of counterfactual story rewriting to implicit states. Qin et al. (TIME-TRAVEL) is explicitly cited as the most similar prior work that introduced this rewriting task framing. +- **evidence_span:** “Most similar to our work is the TIME-TRAVEL dataset by Qin et al. (2019). It includes the Counterfactual Story Rewriting task” + +### Ingredient C2.I2 +- **ingredient_id:** C2.I2 +- **ingredient:** Framework for annotating participant states in simple stories +- **why_structurally_necessary:** The protocol is designed to elicit unstated participant states. It relies on the conceptual framing that narratives contain implicit participant states (like emotions or physical states) that can be systematically tracked and annotated. +- **why_not_lower_level_substeps:** This is a broad conceptual framing rather than specific state categories (physical, emotional, societal). +- **why_not_adjacent_implementation_details:** The specific instructions given to workers are implementation details of this broader concept. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Modeling Naive Psychology of Characters in Simple Commonsense Stories +- **additional groundings if any:** Tracking State Changes in Procedural Text: a Challenge Dataset and Models for Process Paragraph Comprehension +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the conceptual basis for annotating the implicit states (like emotions or motivations) of characters in short narratives. +- **rationale:** The methodology builds on prior work that established the tracking and annotation of entity/participant states in text. Rashkin et al. is a clean representative for tracking emotional/psychological states in simple stories, with Dalvi et al. serving as an additional grounding for physical states. +- **evidence_span:** “Rashkin et al. (2018) tracks the emotional reactions and motivations of characters in simple stories, for a fixed/small set of attributes.” + +### Ingredient C2.I3 +- **ingredient_id:** C2.I3 +- **ingredient:** Two-stage quality control and worker-proficiency filtering mechanism +- **why_structurally_necessary:** Eliciting counterfactuals and logical story revisions from crowd workers is highly prone to artifacts and logical errors. Without a systematic quality control and proficiency-filtering mechanism, the protocol would fail to produce usable data and would not function as a reliable methodology. +- **why_not_lower_level_substeps:** This should remain one quality-control ingredient rather than being split into the specific >5000 HITs threshold or the exact expert-review formula. +- **why_not_adjacent_implementation_details:** The specific pay rate ($0.35) is an implementation detail, but the two-stage filtering process is structural to the methodology's success. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the filtering and expert-review workflow necessary to identify proficient workers and ensure high-quality complex annotations. +- **rationale:** The specific two-stage filtering process (strict initial criteria followed by expert review to identify "proficient workers") is a custom workflow designed in the target paper to make the complex annotation task viable. It maps cleanly to `NONE`. +- **evidence_span:** “For filtering out noisy data from the collected responses, we follow a two-stage filtering process.” + +--- + +# 5. Excluded Tempting Non-Ingredients + +### Claim C1 (Dataset) exclusions +- **Specific baseline models (T5, GPT-3, RoBERTa)** + - Excluded because they are used to benchmark and evaluate the dataset, not to construct the dataset artifact itself. +- **The three specific evaluation tasks (Story State Inference, Story Revision, State Change Generation)** + - Excluded because they are applications/framings for using the dataset, while the dataset artifact itself is just the collection of annotated 4-tuples. + +### Claim C2 (Methodology) exclusions +- **ROCStories corpus** + - Excluded because the methodology is a general protocol for narrative text and does not structurally require ROCStories to exist as a methodology. ROCStories is an ingredient of the dataset, not the method. +- **Specific MTurk UI design** + - Excluded because it is a local implementation detail of the broader four-step crowdsourcing protocol. + +--- + +# 6. Candidate Grounding Decisions + +## Claim C1: PASTA Dataset + +### Ingredient C1.I1 +- **Candidate:** A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories +- **decision:** accepted_canonical +- **why:** Cleanest representative study for the ROCStories corpus used as the source text. + +### Ingredient C1.I2 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The 4-step annotation protocol is a novel workflow introduced in the target paper. + +## Claim C2: Dataset Creation Methodology + +### Ingredient C2.I1 +- **Candidate:** Counterfactual Story Reasoning and Generation +- **decision:** accepted_canonical +- **why:** Cleanest representative for the counterfactual story rewriting framework that the protocol adapts. + +### Ingredient C2.I2 +- **Candidate:** Modeling Naive Psychology of Characters in Simple Commonsense Stories +- **decision:** accepted_canonical +- **why:** Cleanest representative for the conceptual framing of tracking and annotating participant states in simple stories. +- **Candidate:** Tracking State Changes in Procedural Text: a Challenge Dataset and Models for Process Paragraph Comprehension +- **decision:** accepted_additional +- **why:** Relevant additional grounding for tracking physical state changes. + +### Ingredient C2.I3 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The two-stage quality control and expert-review mechanism is specific to the target paper's methodology. + +--- + +# 7. Final Grounding Decisions + +## Claim C1 +- **C1.I1** → A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories → DATA_SOURCE +- **C1.I2** → NONE → EVALUATION_PROTOCOL + +## Claim C2 +- **C2.I1** → Counterfactual Story Reasoning and Generation → CONCEPTUAL_FRAMEWORK +- **C2.I2** → Modeling Naive Psychology of Characters in Simple Commonsense Stories → CONCEPTUAL_FRAMEWORK + - additional: Tracking State Changes in Procedural Text: a Challenge Dataset and Models for Process Paragraph Comprehension +- **C2.I3** → NONE → EVALUATION_PROTOCOL + +--- + +# 8. Final Rationales And Evidence + +## Claim C1: PASTA Dataset + +### C1.I1 +- **ingredient:** Source corpus of short, self-contained narrative stories +- **canonical study:** A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories +- **role:** DATA_SOURCE +- **contribution:** Provides the base narrative stories (ROCStories) that are annotated and revised in the dataset. +- **rationale:** The dataset requires a substrate of short, commonsense-rich narratives to serve as the foundation for state inference and revision. ROCStories is the direct source used by the paper and maps cleanly to its canonical study. +- **evidence_span:** “To create the PASTA dataset, we use stories from the extended ROCStories (Mostafazadeh et al., 2016) corpus for annotation by crowd workers.” + +### C1.I2 +- **ingredient:** Four-step crowdsourcing protocol for inferring states and generating counterfactual revisions +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Defines the workflow used to collect the structured annotations that make up the dataset. +- **rationale:** The specific four-step process that yields the dataset's tuples is a novel contribution of the target paper itself, so it maps cleanly to `NONE`. +- **evidence_span:** “The annotation process has four main steps: 1. Infer a participant state... 2. Select minimal justification sentences... 3. Perturb the state... 4. Revise the story” + +## Claim C2: Dataset Creation Methodology + +### C2.I1 +- **ingredient:** Counterfactual story rewriting framework +- **canonical study:** Counterfactual Story Reasoning and Generation +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the foundational concept of editing a short story based on a counterfactual context, which the protocol adapts for unobserved participant states. +- **rationale:** The methodology extends the idea of counterfactual story rewriting to implicit states. Qin et al. (TIME-TRAVEL) is explicitly cited as the most similar prior work that introduced this rewriting task framing. +- **evidence_span:** “Most similar to our work is the TIME-TRAVEL dataset by Qin et al. (2019). It includes the Counterfactual Story Rewriting task” + +### C2.I2 +- **ingredient:** Framework for annotating participant states in simple stories +- **canonical study:** Modeling Naive Psychology of Characters in Simple Commonsense Stories +- **additional studies:** Tracking State Changes in Procedural Text: a Challenge Dataset and Models for Process Paragraph Comprehension +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the conceptual basis for annotating the implicit states (like emotions or motivations) of characters in short narratives. +- **rationale:** The methodology builds on prior work that established the tracking and annotation of entity/participant states in text. Rashkin et al. is a clean representative for tracking emotional/psychological states in simple stories, with Dalvi et al. serving as an additional grounding for physical states. +- **evidence_span:** “Rashkin et al. (2018) tracks the emotional reactions and motivations of characters in simple stories, for a fixed/small set of attributes.” + +### C2.I3 +- **ingredient:** Two-stage quality control and worker-proficiency filtering mechanism +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the filtering and expert-review workflow necessary to identify proficient workers and ensure high-quality complex annotations. +- **rationale:** The specific two-stage filtering process (strict initial criteria followed by expert review to identify "proficient workers") is a custom workflow designed in the target paper to make the complex annotation task viable. It maps cleanly to `NONE`. +- **evidence_span:** “For filtering out noisy data from the collected responses, we follow a two-stage filtering process.” + + +CANDIDATE 2: candidate_2 +------------------------------------------------------------ +# 1. Cluster Evidence + +Based on the downstream cluster evidence, the paper introduces two distinct downstream-used contributions: +- **C1:** The data collection methodology (a multi-step crowdsourcing protocol for generating counterfactual narrative revisions). +- **C2:** The PASTA dataset itself (a dataset of stories paired with inferred states, counterfactuals, and revised stories). + +Downstream evidence shows these are reused differently: one citing paper adopts the methodology to create its own dataset, while another uses the PASTA dataset directly for data generation and validation. Therefore, the bundled contribution should be split into two atomic claims. + +--- + +# 2. Claim Split Decision + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C2 + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Method +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C1 + +### Why this split is correct +The two artifacts have distinct downstream roles and distinct ingredient decompositions: +- **PASTA** is reused as a dataset resource for training and validation. +- **The crowdsourcing methodology** is reused as a protocol to construct new datasets. +They should therefore be annotated separately. + +--- + +# 3. Rewritten Claims + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **rewritten_claim:** **Dataset: An English-language dataset of short narratives paired with implicit participant states, counterfactual perturbations, and minimally revised stories, enabling the training and evaluation of state-based reasoning models.** +- **why_this_is_atomic:** This claim isolates the released dataset resource that downstream work reuses directly, separate from the methodology used to collect it. +- **optional cluster_id:** C2 +- **decision:** YES_SUFFICIENT + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Method +- **rewritten_claim:** **Method: A multi-step crowdsourcing methodology for narrative revision that elicits implicit participant states, counterfactual perturbations, and minimal story revisions, enabling the construction of counterfactual reasoning datasets.** +- **why_this_is_atomic:** This claim isolates the data collection protocol that downstream work adopts to build new datasets, separate from the specific PASTA dataset release. +- **optional cluster_id:** C1 +- **decision:** YES_SUFFICIENT + +--- + +# 4. Minimal Ingredient Set Per Claim + +## Claim C1: PASTA Dataset + +### Ingredient C1.I1 +- **ingredient_id:** C1.I1 +- **ingredient:** Corpus of short, self-contained everyday narratives +- **why_structurally_necessary:** The dataset requires a substrate of simple narratives describing daily events to serve as the basis for state inference and revision. Without this source text, the dataset could not be constructed. +- **why_not_lower_level_substeps:** This should remain a single data-source ingredient rather than being split into specific story subsets or sentence structures. +- **why_not_adjacent_implementation_details:** The structural dependency is the availability of everyday narratives, not the specific formatting of the ROCStories corpus. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories +- **additional groundings if any:** none +- **role:** DATA_SOURCE +- **contribution:** Provides the base narrative text that annotators read to infer states and subsequently revise. +- **rationale:** To create a dataset of narrative revisions, the authors needed an existing corpus of short, coherent stories. ROCStories is explicitly used as this foundational substrate and is the cleanest canonical grounding. +- **evidence_span:** “To create the PASTA dataset, we use stories from the extended ROCStories (Mostafazadeh et al., 2016) corpus for annotation by crowd workers.” + +### Ingredient C1.I2 +- **ingredient_id:** C1.I2 +- **ingredient:** Multi-step annotation protocol for eliciting implicit states and counterfactual revisions +- **why_structurally_necessary:** The dataset's defining property is the pairing of inferred states, counterfactuals, and revised stories. Without the specific protocol to collect these 4-tuples, the dataset would not exist in its claimed form. +- **why_not_lower_level_substeps:** This should remain one protocol-level ingredient rather than being split into the four individual steps (infer, justify, perturb, revise). +- **why_not_adjacent_implementation_details:** The structural dependency is the annotation workflow itself, not the specific UI design or instructions given to workers. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the structured workflow that generates the dataset's core 4-tuples. +- **rationale:** The dataset is constructed using a novel 4-step annotation process designed by the authors. Because this protocol is paper-specific and not inherited from prior work, `NONE` is the correct grounding. +- **evidence_span:** “PASTA provides a participant state α that is likely to be inferred from S, a perturbation state α' that is counterfactual to S” + +### Ingredient C1.I3 +- **ingredient_id:** C1.I3 +- **ingredient:** Two-stage quality-control and expert-review workflow for logically consistent counterfactuals +- **why_structurally_necessary:** Counterfactual story revision is highly prone to logical inconsistencies. Without a strict filtering and expert-review workflow to identify proficient workers, the dataset would lose its reliability and quality as a reasoning benchmark. +- **why_not_lower_level_substeps:** This should remain one quality-control ingredient rather than being split into worker filtering, expert review, and auto-acceptance rules. +- **why_not_adjacent_implementation_details:** The specific thresholds for worker acceptance are implementation details; the structural dependency is the rigorous quality-control workflow itself. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Ensures the logical consistency and high quality of the collected counterfactual revisions. +- **rationale:** Building a reliable counterfactual dataset requires more than raw collection; it requires a rigorous filtering process to eliminate artifacts and illogical revisions. This workflow is designed in the target paper, making `NONE` the correct grounding. +- **evidence_span:** “we performed an "expert review" of the responses to identify "proficient workers": workers who can perform the task with a high degree of correctness.” + +--- + +## Claim C2: Crowdsourcing Methodology + +### Ingredient C2.I1 +- **ingredient_id:** C2.I1 +- **ingredient:** Counterfactual story rewriting paradigm +- **why_structurally_necessary:** The methodology builds upon the concept of editing a short story based on a counterfactual context. Without this framing, the methodology would lack its core objective. +- **why_not_lower_level_substeps:** This should remain a high-level conceptual ingredient rather than being split into specific rewriting rules or counterfactual types. +- **why_not_adjacent_implementation_details:** The structural dependency is the paradigm of counterfactual rewriting, not the specific position of the counterfactual in the text. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Counterfactual Story Reasoning and Generation +- **additional groundings if any:** none +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the foundational task framing of revising narratives to accommodate counterfactual changes. +- **rationale:** The methodology is an extension of existing counterfactual story rewriting tasks. The paper explicitly cites the TIME-TRAVEL dataset (Qin et al., 2019) as the most similar prior work that established this paradigm, making it the cleanest canonical grounding. +- **evidence_span:** “Most similar to our work is the TIME-TRAVEL dataset by Qin et al. (2019). It includes the Counterfactual Story Rewriting task” + +### Ingredient C2.I2 +- **ingredient_id:** C2.I2 +- **ingredient:** Four-step annotation design for unstated participant states +- **why_structurally_necessary:** The core contribution of the methodology is the specific sequence of steps (infer, justify, perturb, revise) designed to capture *unobserved* states, distinguishing it from prior explicit counterfactual tasks. +- **why_not_lower_level_substeps:** This should remain one integrated method ingredient rather than being decomposed into the individual steps. +- **why_not_adjacent_implementation_details:** The structural dependency is the logical sequence of the annotation steps, not the specific platform or payment used to execute them. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** CORE_METHOD +- **contribution:** Defines the novel procedure for eliciting implicit states and generating corresponding counterfactual revisions. +- **rationale:** The methodology's defining feature is its ability to handle unstated participant states through a carefully designed 4-step process. This process is introduced in the target paper, so `NONE` is the correct grounding. +- **evidence_span:** “The annotator revises S into S', so that α' can be inferred from S' but α is unlikely to be inferred from S'.” + +--- + +# 5. Excluded Tempting Non-Ingredients + +- **Baseline LLMs (GPT-3, T5, RoBERTa):** + - Excluded because they are used only to evaluate the dataset's difficulty and benchmark performance, not to construct the dataset or define the methodology itself. +- **The three specific state-based reasoning tasks (Story State Inference, Story Revision, State Change Generation):** + - Excluded as structural ingredients of the dataset/methodology claims because they are downstream applications and evaluation formulations of the dataset, not prerequisites for building the artifacts. +- **Amazon Mechanical Turk platform:** + - Excluded because it is an implementation detail of the crowdsourcing process. The structural ingredient is the methodology and quality-control workflow, not the specific commercial platform used to host it. + +--- + +# 6. Candidate Grounding Decisions + +## Claim C1: PASTA Dataset + +### Ingredient C1.I1 +- **Candidate:** A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories +- **decision:** accepted_canonical +- **why:** Cleanest representative study for the ROCStories corpus, which provides the essential narrative text substrate for the dataset. + +### Ingredient C1.I2 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The multi-step annotation protocol is designed and implemented in the target paper itself. + +### Ingredient C1.I3 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The specific two-stage quality-control and expert-review workflow is a paper-specific protocol. + +## Claim C2: Crowdsourcing Methodology + +### Ingredient C2.I1 +- **Candidate:** Counterfactual Story Reasoning and Generation +- **decision:** accepted_canonical +- **why:** Cleanest representative framing for the counterfactual story rewriting paradigm that the methodology extends. + +### Ingredient C2.I2 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The four-step annotation design for unstated participant states is the core novel method introduced in the target paper. + +--- + +# 7. Final Grounding Decisions + +## Claim C1 +- **C1.I1** → A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories → DATA_SOURCE +- **C1.I2** → NONE → EVALUATION_PROTOCOL +- **C1.I3** → NONE → EVALUATION_PROTOCOL + +## Claim C2 +- **C2.I1** → Counterfactual Story Reasoning and Generation → CONCEPTUAL_FRAMEWORK +- **C2.I2** → NONE → CORE_METHOD + +--- + +# 8. Final Rationales And Evidence + +## Claim C1: PASTA Dataset + +### C1.I1 +- **ingredient:** Corpus of short, self-contained everyday narratives +- **canonical study:** A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories +- **role:** DATA_SOURCE +- **contribution:** Provides the base narrative text that annotators read to infer states and subsequently revise. +- **rationale:** To create a dataset of narrative revisions, the authors needed an existing corpus of short, coherent stories. ROCStories is explicitly used as this foundational substrate and is the cleanest canonical grounding. +- **evidence_span:** “To create the PASTA dataset, we use stories from the extended ROCStories (Mostafazadeh et al., 2016) corpus for annotation by crowd workers.” + +### C1.I2 +- **ingredient:** Multi-step annotation protocol for eliciting implicit states and counterfactual revisions +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the structured workflow that generates the dataset's core 4-tuples. +- **rationale:** The dataset is constructed using a novel 4-step annotation process designed by the authors. Because this protocol is paper-specific and not inherited from prior work, `NONE` is the correct grounding. +- **evidence_span:** “PASTA provides a participant state α that is likely to be inferred from S, a perturbation state α' that is counterfactual to S” + +### C1.I3 +- **ingredient:** Two-stage quality-control and expert-review workflow for logically consistent counterfactuals +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Ensures the logical consistency and high quality of the collected counterfactual revisions. +- **rationale:** Building a reliable counterfactual dataset requires more than raw collection; it requires a rigorous filtering process to eliminate artifacts and illogical revisions. This workflow is designed in the target paper, making `NONE` the correct grounding. +- **evidence_span:** “we performed an "expert review" of the responses to identify "proficient workers": workers who can perform the task with a high degree of correctness.” + +## Claim C2: Crowdsourcing Methodology + +### C2.I1 +- **ingredient:** Counterfactual story rewriting paradigm +- **canonical study:** Counterfactual Story Reasoning and Generation +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the foundational task framing of revising narratives to accommodate counterfactual changes. +- **rationale:** The methodology is an extension of existing counterfactual story rewriting tasks. The paper explicitly cites the TIME-TRAVEL dataset (Qin et al., 2019) as the most similar prior work that established this paradigm, making it the cleanest canonical grounding. +- **evidence_span:** “Most similar to our work is the TIME-TRAVEL dataset by Qin et al. (2019). It includes the Counterfactual Story Rewriting task” + +### C2.I2 +- **ingredient:** Four-step annotation design for unstated participant states +- **canonical study:** NONE +- **role:** CORE_METHOD +- **contribution:** Defines the novel procedure for eliciting implicit states and generating corresponding counterfactual revisions. +- **rationale:** The methodology's defining feature is its ability to handle unstated participant states through a carefully designed 4-step process. This process is introduced in the target paper, so `NONE` is the correct grounding. +- **evidence_span:** “The annotator revises S into S', so that α' can be inferred from S' but α is unlikely to be inferred from S'.” + + +CANDIDATE 3: candidate_3 +------------------------------------------------------------ +# 1. Cluster Evidence + +Based on the downstream cluster evidence, the paper introduces two distinct downstream-used contributions: +- a **dataset** (PASTA) containing narratives paired with inferred participant states, counterfactuals, and revised stories. +- a **methodology** for creating such datasets via a specific crowdsourcing protocol. + +Downstream evidence shows these are reused differently: +- Cluster C1 reuses the target methodology to generate a new dataset. +- Cluster C2 reuses the PASTA dataset and its annotations directly for data generation and validation. + +Therefore, the bundled contribution should be split into two atomic claims. + +--- + +# 2. Claim Split Decision + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C2 + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Method +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C1 + +### Why this split is correct +The two artifacts have distinct downstream roles and distinct ingredient decompositions. The PASTA dataset is reused as a concrete resource for training and validation, while the dataset creation methodology is reused as a procedural protocol to construct new datasets. They should therefore be annotated separately. + +--- + +# 3. Rewritten Claims + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **rewritten_claim:** **Dataset: A crowdsourced dataset of English narratives paired with inferred participant states, counterfactual perturbations, and revised stories, enabling the training and evaluation of state-based reasoning models.** +- **why_this_is_atomic:** This claim isolates the released dataset resource that downstream work reuses, separate from the methodology used to create it. +- **optional cluster_id:** C2 +- **decision:** YES_SUFFICIENT + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Method +- **rewritten_claim:** **Method: A crowdsourcing methodology for generating counterfactual narrative datasets by having annotators infer implicit states, perturb them, and minimally revise the story, enabling the creation of state-based reasoning resources.** +- **why_this_is_atomic:** This claim isolates the procedural methodology as a reusable protocol, separate from the specific PASTA dataset it was used to build. +- **optional cluster_id:** C1 +- **decision:** YES_SUFFICIENT + +--- + +# 4. Minimal Ingredient Set Per Claim + +## Claim C1: PASTA Dataset + +### Ingredient C1.I1 +- **ingredient_id:** C1.I1 +- **ingredient:** Short narrative stories describing daily events +- **why_structurally_necessary:** The dataset requires a substrate of self-contained narratives from which implicit participant states can be inferred and perturbed. Without this source text, the dataset would not exist. +- **why_not_lower_level_substeps:** This should remain a broad data-source ingredient rather than being split into specific story subsets or sentence counts. +- **why_not_adjacent_implementation_details:** The structural dependency is the narrative text substrate itself, not the specific formatting of the stories. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories +- **additional groundings if any:** none +- **role:** DATA_SOURCE +- **contribution:** Provides the base narrative stories (ROCStories) that annotators read to infer states and generate counterfactual revisions. +- **rationale:** The dataset is built directly on top of the ROCStories corpus, which provides the necessary short, self-contained narratives about daily events. This prior study is the cleanest canonical grounding for the source text. +- **evidence_span:** “To create the PASTA dataset, we use stories from the extended ROCStories corpus for annotation by crowd workers.” + +### Ingredient C1.I2 +- **ingredient_id:** C1.I2 +- **ingredient:** Four-step crowdsourcing protocol for eliciting implicit states and counterfactual revisions +- **why_structurally_necessary:** The dataset is defined by its 4-tuples (story, state, counterfactual, revised story). Without the specific protocol to elicit these from annotators, the dataset would not exist in its claimed form. +- **why_not_lower_level_substeps:** This should remain one protocol-level ingredient rather than being split into the individual inference, justification, perturbation, and revision steps. +- **why_not_adjacent_implementation_details:** The structural dependency is the annotation workflow itself, not the specific UI design or instructions provided to workers. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the structured workflow used to generate the dataset's core 4-tuples. +- **rationale:** The specific four-step annotation protocol is introduced in the target paper itself to construct the dataset, so it maps cleanly to `NONE`. +- **evidence_span:** “This dataset contains inferable participant states; a counterfactual perturbation to each state; and the changes to the story that would be necessary” + +### Ingredient C1.I3 +- **ingredient_id:** C1.I3 +- **ingredient:** Two-stage quality control and expert-review filtering process +- **why_structurally_necessary:** The dataset's reliability depends on filtering out noisy or logically inconsistent counterfactual revisions. Without this quality control workflow, the dataset would lose its defining property as a high-quality reasoning resource. +- **why_not_lower_level_substeps:** This should remain one quality-control ingredient rather than being split into worker qualifications, rejection thresholds, and expert review separately. +- **why_not_adjacent_implementation_details:** The structural dependency is the broader filtering workflow, not the specific payment amounts or geographic restrictions. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Ensures the logical consistency and quality of the crowdsourced counterfactual revisions. +- **rationale:** The two-stage filtering and expert review process is a paper-specific protocol designed to ensure the quality of the released resource, making `NONE` the correct grounding. +- **evidence_span:** “For filtering out noisy data from the collected responses, we follow a two-stage filtering process.” + +--- + +## Claim C2: Dataset Creation Methodology + +### Ingredient C2.I1 +- **ingredient_id:** C2.I1 +- **ingredient:** Conceptual framing of unstated participant states and counterfactuals in narratives +- **why_structurally_necessary:** The methodology is designed specifically to capture implicit states and how their counterfactuals affect the narrative discourse. Without this framing, the methodology has no conceptual basis. +- **why_not_lower_level_substeps:** This should remain a high-level conceptual ingredient rather than being split into specific types of states (physical, emotional, societal). +- **why_not_adjacent_implementation_details:** The structural dependency is the conceptual framing itself, not the specific formal theories of mind or time. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the theoretical basis for the methodology, focusing on implicit states and their counterfactual impacts on narratives. +- **rationale:** The paper explicitly introduces this theory-agnostic framing of unstated participant states and counterfactuals to motivate the methodology. It is not cleanly inherited from one prior study, so `NONE` is the correct grounding. +- **evidence_span:** “A model that understands narratives should likewise infer these implicit states, and even reason about the impact of changes to these states on the narrative.” + +### Ingredient C2.I2 +- **ingredient_id:** C2.I2 +- **ingredient:** Four-step annotation workflow for state inference and story revision +- **why_structurally_necessary:** This is the core operational mechanism of the methodology, defining exactly how annotators generate the required data tuples. +- **why_not_lower_level_substeps:** This should remain one workflow ingredient rather than being split into the individual inference, justification, perturbation, and revision steps. +- **why_not_adjacent_implementation_details:** The structural dependency is the workflow itself, not the specific crowdsourcing platform used. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** CORE_METHOD +- **contribution:** Defines the procedural steps required to execute the methodology and generate counterfactual narrative datasets. +- **rationale:** The specific four-step annotation workflow is the core contribution of the methodology claim and is introduced in the target paper itself, making `NONE` the correct grounding. +- **evidence_span:** “This dataset contains inferable participant states; a counterfactual perturbation to each state; and the changes to the story that would be necessary” + +--- + +# 5. Excluded Tempting Non-Ingredients + +### Claim C1 & C2 exclusions +- **Baseline models (GPT3, T5, BERT, RoBERTa)** + - Excluded because they are used to benchmark and evaluate the dataset, not to construct the dataset or the methodology itself. +- **The three state-based reasoning tasks (Story State Inference, Story Revision, State Change Generation)** + - Excluded because they are downstream tasks proposed to evaluate models on the dataset, not structural ingredients of the dataset or methodology artifacts themselves. +- **Amazon Mechanical Turk (AMT)** + - Excluded because it is an implementation detail for crowdsourcing, not a structural ingredient of the methodology or dataset. + +--- + +# 6. Candidate Grounding Decisions + +## Claim C1: PASTA Dataset + +### Ingredient C1.I1 +- **Candidate:** A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories +- **decision:** accepted_canonical +- **why:** Cleanest representative study for the ROCStories corpus, which provides the base narrative text used to construct the dataset. + +### Ingredient C1.I2 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The four-step crowdsourcing protocol is introduced in the target paper itself. + +### Ingredient C1.I3 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The two-stage quality control and expert-review filtering process is a paper-specific protocol. + +## Claim C2: Dataset Creation Methodology + +### Ingredient C2.I1 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The conceptual framing of unstated participant states and counterfactuals is introduced in the target paper as a theory-agnostic resource basis. + +### Ingredient C2.I2 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The four-step annotation workflow is the core procedural contribution of the methodology and is introduced in the target paper. + +--- + +# 7. Final Grounding Decisions + +## Claim C1 +- **C1.I1** → A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories → DATA_SOURCE +- **C1.I2** → NONE → EVALUATION_PROTOCOL +- **C1.I3** → NONE → EVALUATION_PROTOCOL + +## Claim C2 +- **C2.I1** → NONE → CONCEPTUAL_FRAMEWORK +- **C2.I2** → NONE → CORE_METHOD + +--- + +# 8. Final Rationales And Evidence + +## Claim C1: PASTA Dataset + +### C1.I1 +- **ingredient:** Short narrative stories describing daily events +- **canonical study:** A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories +- **role:** DATA_SOURCE +- **contribution:** Provides the base narrative stories (ROCStories) that annotators read to infer states and generate counterfactual revisions. +- **rationale:** The dataset is built directly on top of the ROCStories corpus, which provides the necessary short, self-contained narratives about daily events. This prior study is the cleanest canonical grounding for the source text. +- **evidence_span:** “To create the PASTA dataset, we use stories from the extended ROCStories corpus for annotation by crowd workers.” + +### C1.I2 +- **ingredient:** Four-step crowdsourcing protocol for eliciting implicit states and counterfactual revisions +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the structured workflow used to generate the dataset's core 4-tuples. +- **rationale:** The specific four-step annotation protocol is introduced in the target paper itself to construct the dataset, so it maps cleanly to `NONE`. +- **evidence_span:** “This dataset contains inferable participant states; a counterfactual perturbation to each state; and the changes to the story that would be necessary” + +### C1.I3 +- **ingredient:** Two-stage quality control and expert-review filtering process +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Ensures the logical consistency and quality of the crowdsourced counterfactual revisions. +- **rationale:** The two-stage filtering and expert review process is a paper-specific protocol designed to ensure the quality of the released resource, making `NONE` the correct grounding. +- **evidence_span:** “For filtering out noisy data from the collected responses, we follow a two-stage filtering process.” + +## Claim C2: Dataset Creation Methodology + +### C2.I1 +- **ingredient:** Conceptual framing of unstated participant states and counterfactuals in narratives +- **canonical study:** NONE +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the theoretical basis for the methodology, focusing on implicit states and their counterfactual impacts on narratives. +- **rationale:** The paper explicitly introduces this theory-agnostic framing of unstated participant states and counterfactuals to motivate the methodology. It is not cleanly inherited from one prior study, so `NONE` is the correct grounding. +- **evidence_span:** “A model that understands narratives should likewise infer these implicit states, and even reason about the impact of changes to these states on the narrative.” + +### C2.I2 +- **ingredient:** Four-step annotation workflow for state inference and story revision +- **canonical study:** NONE +- **role:** CORE_METHOD +- **contribution:** Defines the procedural steps required to execute the methodology and generate counterfactual narrative datasets. +- **rationale:** The specific four-step annotation workflow is the core contribution of the methodology claim and is introduced in the target paper itself, making `NONE` the correct grounding. +- **evidence_span:** “This dataset contains inferable participant states; a counterfactual perturbation to each state; and the changes to the story that would be necessary” + + +Return only valid JSON with this exact schema: + +{ + "selected_candidate_index": 0, + "selected_candidate_id": "candidate_1", + "selected_reason": "...", + "candidate_scores": [ + { + "candidate_id": "candidate_1", + "candidate_index": 0, + "score": 1, + "assessment": "..." + } + ] +} + +Scoring rules: +- Use integer scores from 1 to 10. +- 10 = best candidate by the minimal structurally sufficient annotation objective. +- The selected candidate must have the highest score. +- In "selected_reason", explain why it best matches the downstream-first, minimal-ingredient, structurally faithful objective. +- In each "assessment", mention both strengths and weaknesses, especially over-decomposition, poor exclusions, weak abstraction level, bad role choices, and bad canonical vs "__NONE__" decisions. + +Return only valid JSON. Do not add any prose outside the JSON object. diff --git a/replay_traces/2208.00329/two_pass_outputs/20260726T205558Z/pass_1_reasoning.output.candidate_1.md b/replay_traces/2208.00329/two_pass_outputs/20260726T205558Z/pass_1_reasoning.output.candidate_1.md new file mode 100644 index 0000000000000000000000000000000000000000..080e99d65cb7c2103af0469a5a4898fc9104f55d --- /dev/null +++ b/replay_traces/2208.00329/two_pass_outputs/20260726T205558Z/pass_1_reasoning.output.candidate_1.md @@ -0,0 +1,254 @@ +# 1. Cluster Evidence + +Based on the downstream cluster evidence, the paper introduces two distinct downstream-used contributions: +- **C1:** The methodology/protocol for dataset creation, which is reused by later work to generate their own datasets. +- **C2:** The PASTA dataset itself, which is reused directly for data generation and validation. + +Because these two contributions have distinct downstream roles (one is reused as a data artifact, the other as a data-collection process) and require different structural ingredients, they should be split into two atomic claims. + +--- + +# 2. Claim Split Decision + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C2 + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Method +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C1 + +### Why this split is correct +The two artifacts have distinct downstream roles and distinct ingredient decompositions: +- **The PASTA Dataset** is reused as a concrete data resource for training and evaluation. It structurally depends on the specific source texts it annotates. +- **The Dataset Creation Methodology** is reused as a process/protocol to build new datasets. It structurally depends on conceptual frameworks for counterfactuals and quality control, but is agnostic to the specific source text. +They should therefore be annotated separately. + +--- + +# 3. Rewritten Claims + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **rewritten_claim:** **Dataset: A crowdsourced dataset of narrative stories annotated with implicit participant states, counterfactual perturbations, and revised stories, enabling state-based reasoning and generation tasks.** +- **why_this_is_atomic:** This claim isolates the dataset artifact itself, separate from the methodology used to create it. +- **optional cluster_id:** C2 +- **decision:** YES_SUFFICIENT + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Method +- **rewritten_claim:** **Method: A crowdsourcing protocol for eliciting implicit participant states and counterfactual story revisions, enabling the construction of state-based narrative datasets.** +- **why_this_is_atomic:** This claim isolates the reusable annotation methodology and quality-control process, distinct from the specific PASTA dataset produced by it. +- **optional cluster_id:** C1 +- **decision:** YES_SUFFICIENT + +--- + +# 4. Minimal Ingredient Set Per Claim + +## Claim C1: PASTA Dataset + +### Ingredient C1.I1 +- **ingredient_id:** C1.I1 +- **ingredient:** Source corpus of short, self-contained narrative stories +- **why_structurally_necessary:** The dataset annotates implicit states and counterfactuals over existing narratives. Without a source corpus of narratives to annotate, the dataset would not exist in its claimed form. +- **why_not_lower_level_substeps:** This should remain at the level of the source corpus rather than being split into specific story filters, splits, or sampling strategies. +- **why_not_adjacent_implementation_details:** The exact size of the sample is an implementation detail; the structural dependency is the narrative text substrate itself. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories +- **additional groundings if any:** none +- **role:** DATA_SOURCE +- **contribution:** Provides the base narrative stories (ROCStories) that are annotated and revised in the dataset. +- **rationale:** The dataset requires a substrate of short, commonsense-rich narratives to serve as the foundation for state inference and revision. ROCStories is the direct source used by the paper and maps cleanly to its canonical study. +- **evidence_span:** “To create the PASTA dataset, we use stories from the extended ROCStories (Mostafazadeh et al., 2016) corpus for annotation by crowd workers.” + +### Ingredient C1.I2 +- **ingredient_id:** C1.I2 +- **ingredient:** Four-step crowdsourcing protocol for inferring states and generating counterfactual revisions +- **why_structurally_necessary:** The dataset's unique structure (inferred state, justification, counterfactual state, revised story) is entirely defined by this specific annotation workflow. Without it, the artifact would not have its defining shape. +- **why_not_lower_level_substeps:** This should remain one protocol-level ingredient rather than being split into the four individual steps or the specific MTurk UI details. +- **why_not_adjacent_implementation_details:** Quality control and worker filtering are important but subordinate to the core four-step annotation protocol that defines the data structure for this claim. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Defines the workflow used to collect the structured annotations that make up the dataset. +- **rationale:** The specific four-step process that yields the dataset's tuples is a novel contribution of the target paper itself, so it maps cleanly to `NONE`. +- **evidence_span:** “The annotation process has four main steps: 1. Infer a participant state... 2. Select minimal justification sentences... 3. Perturb the state... 4. Revise the story” + +--- + +## Claim C2: Dataset Creation Methodology + +### Ingredient C2.I1 +- **ingredient_id:** C2.I1 +- **ingredient:** Counterfactual story rewriting framework +- **why_structurally_necessary:** The protocol requires workers to revise a story based on a counterfactual state. This builds fundamentally on the conceptual framework of counterfactual story rewriting. +- **why_not_lower_level_substeps:** This remains at the conceptual level of counterfactual rewriting rather than specific model architectures or generation constraints. +- **why_not_adjacent_implementation_details:** The specific position of the counterfactual (unobserved vs explicit) is the target paper's contribution, but the foundational concept is counterfactual rewriting. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Counterfactual Story Reasoning and Generation +- **additional groundings if any:** none +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the foundational concept of editing a short story based on a counterfactual context, which the protocol adapts for unobserved participant states. +- **rationale:** The methodology extends the idea of counterfactual story rewriting to implicit states. Qin et al. (TIME-TRAVEL) is explicitly cited as the most similar prior work that introduced this rewriting task framing. +- **evidence_span:** “Most similar to our work is the TIME-TRAVEL dataset by Qin et al. (2019). It includes the Counterfactual Story Rewriting task” + +### Ingredient C2.I2 +- **ingredient_id:** C2.I2 +- **ingredient:** Framework for annotating participant states in simple stories +- **why_structurally_necessary:** The protocol is designed to elicit unstated participant states. It relies on the conceptual framing that narratives contain implicit participant states (like emotions or physical states) that can be systematically tracked and annotated. +- **why_not_lower_level_substeps:** This is a broad conceptual framing rather than specific state categories (physical, emotional, societal). +- **why_not_adjacent_implementation_details:** The specific instructions given to workers are implementation details of this broader concept. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Modeling Naive Psychology of Characters in Simple Commonsense Stories +- **additional groundings if any:** Tracking State Changes in Procedural Text: a Challenge Dataset and Models for Process Paragraph Comprehension +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the conceptual basis for annotating the implicit states (like emotions or motivations) of characters in short narratives. +- **rationale:** The methodology builds on prior work that established the tracking and annotation of entity/participant states in text. Rashkin et al. is a clean representative for tracking emotional/psychological states in simple stories, with Dalvi et al. serving as an additional grounding for physical states. +- **evidence_span:** “Rashkin et al. (2018) tracks the emotional reactions and motivations of characters in simple stories, for a fixed/small set of attributes.” + +### Ingredient C2.I3 +- **ingredient_id:** C2.I3 +- **ingredient:** Two-stage quality control and worker-proficiency filtering mechanism +- **why_structurally_necessary:** Eliciting counterfactuals and logical story revisions from crowd workers is highly prone to artifacts and logical errors. Without a systematic quality control and proficiency-filtering mechanism, the protocol would fail to produce usable data and would not function as a reliable methodology. +- **why_not_lower_level_substeps:** This should remain one quality-control ingredient rather than being split into the specific >5000 HITs threshold or the exact expert-review formula. +- **why_not_adjacent_implementation_details:** The specific pay rate ($0.35) is an implementation detail, but the two-stage filtering process is structural to the methodology's success. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the filtering and expert-review workflow necessary to identify proficient workers and ensure high-quality complex annotations. +- **rationale:** The specific two-stage filtering process (strict initial criteria followed by expert review to identify "proficient workers") is a custom workflow designed in the target paper to make the complex annotation task viable. It maps cleanly to `NONE`. +- **evidence_span:** “For filtering out noisy data from the collected responses, we follow a two-stage filtering process.” + +--- + +# 5. Excluded Tempting Non-Ingredients + +### Claim C1 (Dataset) exclusions +- **Specific baseline models (T5, GPT-3, RoBERTa)** + - Excluded because they are used to benchmark and evaluate the dataset, not to construct the dataset artifact itself. +- **The three specific evaluation tasks (Story State Inference, Story Revision, State Change Generation)** + - Excluded because they are applications/framings for using the dataset, while the dataset artifact itself is just the collection of annotated 4-tuples. + +### Claim C2 (Methodology) exclusions +- **ROCStories corpus** + - Excluded because the methodology is a general protocol for narrative text and does not structurally require ROCStories to exist as a methodology. ROCStories is an ingredient of the dataset, not the method. +- **Specific MTurk UI design** + - Excluded because it is a local implementation detail of the broader four-step crowdsourcing protocol. + +--- + +# 6. Candidate Grounding Decisions + +## Claim C1: PASTA Dataset + +### Ingredient C1.I1 +- **Candidate:** A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories +- **decision:** accepted_canonical +- **why:** Cleanest representative study for the ROCStories corpus used as the source text. + +### Ingredient C1.I2 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The 4-step annotation protocol is a novel workflow introduced in the target paper. + +## Claim C2: Dataset Creation Methodology + +### Ingredient C2.I1 +- **Candidate:** Counterfactual Story Reasoning and Generation +- **decision:** accepted_canonical +- **why:** Cleanest representative for the counterfactual story rewriting framework that the protocol adapts. + +### Ingredient C2.I2 +- **Candidate:** Modeling Naive Psychology of Characters in Simple Commonsense Stories +- **decision:** accepted_canonical +- **why:** Cleanest representative for the conceptual framing of tracking and annotating participant states in simple stories. +- **Candidate:** Tracking State Changes in Procedural Text: a Challenge Dataset and Models for Process Paragraph Comprehension +- **decision:** accepted_additional +- **why:** Relevant additional grounding for tracking physical state changes. + +### Ingredient C2.I3 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The two-stage quality control and expert-review mechanism is specific to the target paper's methodology. + +--- + +# 7. Final Grounding Decisions + +## Claim C1 +- **C1.I1** → A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories → DATA_SOURCE +- **C1.I2** → NONE → EVALUATION_PROTOCOL + +## Claim C2 +- **C2.I1** → Counterfactual Story Reasoning and Generation → CONCEPTUAL_FRAMEWORK +- **C2.I2** → Modeling Naive Psychology of Characters in Simple Commonsense Stories → CONCEPTUAL_FRAMEWORK + - additional: Tracking State Changes in Procedural Text: a Challenge Dataset and Models for Process Paragraph Comprehension +- **C2.I3** → NONE → EVALUATION_PROTOCOL + +--- + +# 8. Final Rationales And Evidence + +## Claim C1: PASTA Dataset + +### C1.I1 +- **ingredient:** Source corpus of short, self-contained narrative stories +- **canonical study:** A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories +- **role:** DATA_SOURCE +- **contribution:** Provides the base narrative stories (ROCStories) that are annotated and revised in the dataset. +- **rationale:** The dataset requires a substrate of short, commonsense-rich narratives to serve as the foundation for state inference and revision. ROCStories is the direct source used by the paper and maps cleanly to its canonical study. +- **evidence_span:** “To create the PASTA dataset, we use stories from the extended ROCStories (Mostafazadeh et al., 2016) corpus for annotation by crowd workers.” + +### C1.I2 +- **ingredient:** Four-step crowdsourcing protocol for inferring states and generating counterfactual revisions +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Defines the workflow used to collect the structured annotations that make up the dataset. +- **rationale:** The specific four-step process that yields the dataset's tuples is a novel contribution of the target paper itself, so it maps cleanly to `NONE`. +- **evidence_span:** “The annotation process has four main steps: 1. Infer a participant state... 2. Select minimal justification sentences... 3. Perturb the state... 4. Revise the story” + +## Claim C2: Dataset Creation Methodology + +### C2.I1 +- **ingredient:** Counterfactual story rewriting framework +- **canonical study:** Counterfactual Story Reasoning and Generation +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the foundational concept of editing a short story based on a counterfactual context, which the protocol adapts for unobserved participant states. +- **rationale:** The methodology extends the idea of counterfactual story rewriting to implicit states. Qin et al. (TIME-TRAVEL) is explicitly cited as the most similar prior work that introduced this rewriting task framing. +- **evidence_span:** “Most similar to our work is the TIME-TRAVEL dataset by Qin et al. (2019). It includes the Counterfactual Story Rewriting task” + +### C2.I2 +- **ingredient:** Framework for annotating participant states in simple stories +- **canonical study:** Modeling Naive Psychology of Characters in Simple Commonsense Stories +- **additional studies:** Tracking State Changes in Procedural Text: a Challenge Dataset and Models for Process Paragraph Comprehension +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the conceptual basis for annotating the implicit states (like emotions or motivations) of characters in short narratives. +- **rationale:** The methodology builds on prior work that established the tracking and annotation of entity/participant states in text. Rashkin et al. is a clean representative for tracking emotional/psychological states in simple stories, with Dalvi et al. serving as an additional grounding for physical states. +- **evidence_span:** “Rashkin et al. (2018) tracks the emotional reactions and motivations of characters in simple stories, for a fixed/small set of attributes.” + +### C2.I3 +- **ingredient:** Two-stage quality control and worker-proficiency filtering mechanism +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the filtering and expert-review workflow necessary to identify proficient workers and ensure high-quality complex annotations. +- **rationale:** The specific two-stage filtering process (strict initial criteria followed by expert review to identify "proficient workers") is a custom workflow designed in the target paper to make the complex annotation task viable. It maps cleanly to `NONE`. +- **evidence_span:** “For filtering out noisy data from the collected responses, we follow a two-stage filtering process.” \ No newline at end of file diff --git a/replay_traces/2208.00329/two_pass_outputs/20260726T205558Z/pass_1_reasoning.output.candidate_2.md b/replay_traces/2208.00329/two_pass_outputs/20260726T205558Z/pass_1_reasoning.output.candidate_2.md new file mode 100644 index 0000000000000000000000000000000000000000..96c834df9f7b582735c4bea624db3a2498e1c9c4 --- /dev/null +++ b/replay_traces/2208.00329/two_pass_outputs/20260726T205558Z/pass_1_reasoning.output.candidate_2.md @@ -0,0 +1,244 @@ +# 1. Cluster Evidence + +Based on the downstream cluster evidence, the paper introduces two distinct downstream-used contributions: +- **C1:** The data collection methodology (a multi-step crowdsourcing protocol for generating counterfactual narrative revisions). +- **C2:** The PASTA dataset itself (a dataset of stories paired with inferred states, counterfactuals, and revised stories). + +Downstream evidence shows these are reused differently: one citing paper adopts the methodology to create its own dataset, while another uses the PASTA dataset directly for data generation and validation. Therefore, the bundled contribution should be split into two atomic claims. + +--- + +# 2. Claim Split Decision + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C2 + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Method +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C1 + +### Why this split is correct +The two artifacts have distinct downstream roles and distinct ingredient decompositions: +- **PASTA** is reused as a dataset resource for training and validation. +- **The crowdsourcing methodology** is reused as a protocol to construct new datasets. +They should therefore be annotated separately. + +--- + +# 3. Rewritten Claims + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **rewritten_claim:** **Dataset: An English-language dataset of short narratives paired with implicit participant states, counterfactual perturbations, and minimally revised stories, enabling the training and evaluation of state-based reasoning models.** +- **why_this_is_atomic:** This claim isolates the released dataset resource that downstream work reuses directly, separate from the methodology used to collect it. +- **optional cluster_id:** C2 +- **decision:** YES_SUFFICIENT + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Method +- **rewritten_claim:** **Method: A multi-step crowdsourcing methodology for narrative revision that elicits implicit participant states, counterfactual perturbations, and minimal story revisions, enabling the construction of counterfactual reasoning datasets.** +- **why_this_is_atomic:** This claim isolates the data collection protocol that downstream work adopts to build new datasets, separate from the specific PASTA dataset release. +- **optional cluster_id:** C1 +- **decision:** YES_SUFFICIENT + +--- + +# 4. Minimal Ingredient Set Per Claim + +## Claim C1: PASTA Dataset + +### Ingredient C1.I1 +- **ingredient_id:** C1.I1 +- **ingredient:** Corpus of short, self-contained everyday narratives +- **why_structurally_necessary:** The dataset requires a substrate of simple narratives describing daily events to serve as the basis for state inference and revision. Without this source text, the dataset could not be constructed. +- **why_not_lower_level_substeps:** This should remain a single data-source ingredient rather than being split into specific story subsets or sentence structures. +- **why_not_adjacent_implementation_details:** The structural dependency is the availability of everyday narratives, not the specific formatting of the ROCStories corpus. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories +- **additional groundings if any:** none +- **role:** DATA_SOURCE +- **contribution:** Provides the base narrative text that annotators read to infer states and subsequently revise. +- **rationale:** To create a dataset of narrative revisions, the authors needed an existing corpus of short, coherent stories. ROCStories is explicitly used as this foundational substrate and is the cleanest canonical grounding. +- **evidence_span:** “To create the PASTA dataset, we use stories from the extended ROCStories (Mostafazadeh et al., 2016) corpus for annotation by crowd workers.” + +### Ingredient C1.I2 +- **ingredient_id:** C1.I2 +- **ingredient:** Multi-step annotation protocol for eliciting implicit states and counterfactual revisions +- **why_structurally_necessary:** The dataset's defining property is the pairing of inferred states, counterfactuals, and revised stories. Without the specific protocol to collect these 4-tuples, the dataset would not exist in its claimed form. +- **why_not_lower_level_substeps:** This should remain one protocol-level ingredient rather than being split into the four individual steps (infer, justify, perturb, revise). +- **why_not_adjacent_implementation_details:** The structural dependency is the annotation workflow itself, not the specific UI design or instructions given to workers. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the structured workflow that generates the dataset's core 4-tuples. +- **rationale:** The dataset is constructed using a novel 4-step annotation process designed by the authors. Because this protocol is paper-specific and not inherited from prior work, `NONE` is the correct grounding. +- **evidence_span:** “PASTA provides a participant state α that is likely to be inferred from S, a perturbation state α' that is counterfactual to S” + +### Ingredient C1.I3 +- **ingredient_id:** C1.I3 +- **ingredient:** Two-stage quality-control and expert-review workflow for logically consistent counterfactuals +- **why_structurally_necessary:** Counterfactual story revision is highly prone to logical inconsistencies. Without a strict filtering and expert-review workflow to identify proficient workers, the dataset would lose its reliability and quality as a reasoning benchmark. +- **why_not_lower_level_substeps:** This should remain one quality-control ingredient rather than being split into worker filtering, expert review, and auto-acceptance rules. +- **why_not_adjacent_implementation_details:** The specific thresholds for worker acceptance are implementation details; the structural dependency is the rigorous quality-control workflow itself. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Ensures the logical consistency and high quality of the collected counterfactual revisions. +- **rationale:** Building a reliable counterfactual dataset requires more than raw collection; it requires a rigorous filtering process to eliminate artifacts and illogical revisions. This workflow is designed in the target paper, making `NONE` the correct grounding. +- **evidence_span:** “we performed an "expert review" of the responses to identify "proficient workers": workers who can perform the task with a high degree of correctness.” + +--- + +## Claim C2: Crowdsourcing Methodology + +### Ingredient C2.I1 +- **ingredient_id:** C2.I1 +- **ingredient:** Counterfactual story rewriting paradigm +- **why_structurally_necessary:** The methodology builds upon the concept of editing a short story based on a counterfactual context. Without this framing, the methodology would lack its core objective. +- **why_not_lower_level_substeps:** This should remain a high-level conceptual ingredient rather than being split into specific rewriting rules or counterfactual types. +- **why_not_adjacent_implementation_details:** The structural dependency is the paradigm of counterfactual rewriting, not the specific position of the counterfactual in the text. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Counterfactual Story Reasoning and Generation +- **additional groundings if any:** none +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the foundational task framing of revising narratives to accommodate counterfactual changes. +- **rationale:** The methodology is an extension of existing counterfactual story rewriting tasks. The paper explicitly cites the TIME-TRAVEL dataset (Qin et al., 2019) as the most similar prior work that established this paradigm, making it the cleanest canonical grounding. +- **evidence_span:** “Most similar to our work is the TIME-TRAVEL dataset by Qin et al. (2019). It includes the Counterfactual Story Rewriting task” + +### Ingredient C2.I2 +- **ingredient_id:** C2.I2 +- **ingredient:** Four-step annotation design for unstated participant states +- **why_structurally_necessary:** The core contribution of the methodology is the specific sequence of steps (infer, justify, perturb, revise) designed to capture *unobserved* states, distinguishing it from prior explicit counterfactual tasks. +- **why_not_lower_level_substeps:** This should remain one integrated method ingredient rather than being decomposed into the individual steps. +- **why_not_adjacent_implementation_details:** The structural dependency is the logical sequence of the annotation steps, not the specific platform or payment used to execute them. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** CORE_METHOD +- **contribution:** Defines the novel procedure for eliciting implicit states and generating corresponding counterfactual revisions. +- **rationale:** The methodology's defining feature is its ability to handle unstated participant states through a carefully designed 4-step process. This process is introduced in the target paper, so `NONE` is the correct grounding. +- **evidence_span:** “The annotator revises S into S', so that α' can be inferred from S' but α is unlikely to be inferred from S'.” + +--- + +# 5. Excluded Tempting Non-Ingredients + +- **Baseline LLMs (GPT-3, T5, RoBERTa):** + - Excluded because they are used only to evaluate the dataset's difficulty and benchmark performance, not to construct the dataset or define the methodology itself. +- **The three specific state-based reasoning tasks (Story State Inference, Story Revision, State Change Generation):** + - Excluded as structural ingredients of the dataset/methodology claims because they are downstream applications and evaluation formulations of the dataset, not prerequisites for building the artifacts. +- **Amazon Mechanical Turk platform:** + - Excluded because it is an implementation detail of the crowdsourcing process. The structural ingredient is the methodology and quality-control workflow, not the specific commercial platform used to host it. + +--- + +# 6. Candidate Grounding Decisions + +## Claim C1: PASTA Dataset + +### Ingredient C1.I1 +- **Candidate:** A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories +- **decision:** accepted_canonical +- **why:** Cleanest representative study for the ROCStories corpus, which provides the essential narrative text substrate for the dataset. + +### Ingredient C1.I2 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The multi-step annotation protocol is designed and implemented in the target paper itself. + +### Ingredient C1.I3 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The specific two-stage quality-control and expert-review workflow is a paper-specific protocol. + +## Claim C2: Crowdsourcing Methodology + +### Ingredient C2.I1 +- **Candidate:** Counterfactual Story Reasoning and Generation +- **decision:** accepted_canonical +- **why:** Cleanest representative framing for the counterfactual story rewriting paradigm that the methodology extends. + +### Ingredient C2.I2 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The four-step annotation design for unstated participant states is the core novel method introduced in the target paper. + +--- + +# 7. Final Grounding Decisions + +## Claim C1 +- **C1.I1** → A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories → DATA_SOURCE +- **C1.I2** → NONE → EVALUATION_PROTOCOL +- **C1.I3** → NONE → EVALUATION_PROTOCOL + +## Claim C2 +- **C2.I1** → Counterfactual Story Reasoning and Generation → CONCEPTUAL_FRAMEWORK +- **C2.I2** → NONE → CORE_METHOD + +--- + +# 8. Final Rationales And Evidence + +## Claim C1: PASTA Dataset + +### C1.I1 +- **ingredient:** Corpus of short, self-contained everyday narratives +- **canonical study:** A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories +- **role:** DATA_SOURCE +- **contribution:** Provides the base narrative text that annotators read to infer states and subsequently revise. +- **rationale:** To create a dataset of narrative revisions, the authors needed an existing corpus of short, coherent stories. ROCStories is explicitly used as this foundational substrate and is the cleanest canonical grounding. +- **evidence_span:** “To create the PASTA dataset, we use stories from the extended ROCStories (Mostafazadeh et al., 2016) corpus for annotation by crowd workers.” + +### C1.I2 +- **ingredient:** Multi-step annotation protocol for eliciting implicit states and counterfactual revisions +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the structured workflow that generates the dataset's core 4-tuples. +- **rationale:** The dataset is constructed using a novel 4-step annotation process designed by the authors. Because this protocol is paper-specific and not inherited from prior work, `NONE` is the correct grounding. +- **evidence_span:** “PASTA provides a participant state α that is likely to be inferred from S, a perturbation state α' that is counterfactual to S” + +### C1.I3 +- **ingredient:** Two-stage quality-control and expert-review workflow for logically consistent counterfactuals +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Ensures the logical consistency and high quality of the collected counterfactual revisions. +- **rationale:** Building a reliable counterfactual dataset requires more than raw collection; it requires a rigorous filtering process to eliminate artifacts and illogical revisions. This workflow is designed in the target paper, making `NONE` the correct grounding. +- **evidence_span:** “we performed an "expert review" of the responses to identify "proficient workers": workers who can perform the task with a high degree of correctness.” + +## Claim C2: Crowdsourcing Methodology + +### C2.I1 +- **ingredient:** Counterfactual story rewriting paradigm +- **canonical study:** Counterfactual Story Reasoning and Generation +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the foundational task framing of revising narratives to accommodate counterfactual changes. +- **rationale:** The methodology is an extension of existing counterfactual story rewriting tasks. The paper explicitly cites the TIME-TRAVEL dataset (Qin et al., 2019) as the most similar prior work that established this paradigm, making it the cleanest canonical grounding. +- **evidence_span:** “Most similar to our work is the TIME-TRAVEL dataset by Qin et al. (2019). It includes the Counterfactual Story Rewriting task” + +### C2.I2 +- **ingredient:** Four-step annotation design for unstated participant states +- **canonical study:** NONE +- **role:** CORE_METHOD +- **contribution:** Defines the novel procedure for eliciting implicit states and generating corresponding counterfactual revisions. +- **rationale:** The methodology's defining feature is its ability to handle unstated participant states through a carefully designed 4-step process. This process is introduced in the target paper, so `NONE` is the correct grounding. +- **evidence_span:** “The annotator revises S into S', so that α' can be inferred from S' but α is unlikely to be inferred from S'.” \ No newline at end of file diff --git a/replay_traces/2208.00329/two_pass_outputs/20260726T205558Z/pass_1_reasoning.output.candidate_3.md b/replay_traces/2208.00329/two_pass_outputs/20260726T205558Z/pass_1_reasoning.output.candidate_3.md new file mode 100644 index 0000000000000000000000000000000000000000..33adbf36231e2103b9a43f2561482e388e0eaa27 --- /dev/null +++ b/replay_traces/2208.00329/two_pass_outputs/20260726T205558Z/pass_1_reasoning.output.candidate_3.md @@ -0,0 +1,246 @@ +# 1. Cluster Evidence + +Based on the downstream cluster evidence, the paper introduces two distinct downstream-used contributions: +- a **dataset** (PASTA) containing narratives paired with inferred participant states, counterfactuals, and revised stories. +- a **methodology** for creating such datasets via a specific crowdsourcing protocol. + +Downstream evidence shows these are reused differently: +- Cluster C1 reuses the target methodology to generate a new dataset. +- Cluster C2 reuses the PASTA dataset and its annotations directly for data generation and validation. + +Therefore, the bundled contribution should be split into two atomic claims. + +--- + +# 2. Claim Split Decision + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C2 + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Method +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C1 + +### Why this split is correct +The two artifacts have distinct downstream roles and distinct ingredient decompositions. The PASTA dataset is reused as a concrete resource for training and validation, while the dataset creation methodology is reused as a procedural protocol to construct new datasets. They should therefore be annotated separately. + +--- + +# 3. Rewritten Claims + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **rewritten_claim:** **Dataset: A crowdsourced dataset of English narratives paired with inferred participant states, counterfactual perturbations, and revised stories, enabling the training and evaluation of state-based reasoning models.** +- **why_this_is_atomic:** This claim isolates the released dataset resource that downstream work reuses, separate from the methodology used to create it. +- **optional cluster_id:** C2 +- **decision:** YES_SUFFICIENT + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Method +- **rewritten_claim:** **Method: A crowdsourcing methodology for generating counterfactual narrative datasets by having annotators infer implicit states, perturb them, and minimally revise the story, enabling the creation of state-based reasoning resources.** +- **why_this_is_atomic:** This claim isolates the procedural methodology as a reusable protocol, separate from the specific PASTA dataset it was used to build. +- **optional cluster_id:** C1 +- **decision:** YES_SUFFICIENT + +--- + +# 4. Minimal Ingredient Set Per Claim + +## Claim C1: PASTA Dataset + +### Ingredient C1.I1 +- **ingredient_id:** C1.I1 +- **ingredient:** Short narrative stories describing daily events +- **why_structurally_necessary:** The dataset requires a substrate of self-contained narratives from which implicit participant states can be inferred and perturbed. Without this source text, the dataset would not exist. +- **why_not_lower_level_substeps:** This should remain a broad data-source ingredient rather than being split into specific story subsets or sentence counts. +- **why_not_adjacent_implementation_details:** The structural dependency is the narrative text substrate itself, not the specific formatting of the stories. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories +- **additional groundings if any:** none +- **role:** DATA_SOURCE +- **contribution:** Provides the base narrative stories (ROCStories) that annotators read to infer states and generate counterfactual revisions. +- **rationale:** The dataset is built directly on top of the ROCStories corpus, which provides the necessary short, self-contained narratives about daily events. This prior study is the cleanest canonical grounding for the source text. +- **evidence_span:** “To create the PASTA dataset, we use stories from the extended ROCStories corpus for annotation by crowd workers.” + +### Ingredient C1.I2 +- **ingredient_id:** C1.I2 +- **ingredient:** Four-step crowdsourcing protocol for eliciting implicit states and counterfactual revisions +- **why_structurally_necessary:** The dataset is defined by its 4-tuples (story, state, counterfactual, revised story). Without the specific protocol to elicit these from annotators, the dataset would not exist in its claimed form. +- **why_not_lower_level_substeps:** This should remain one protocol-level ingredient rather than being split into the individual inference, justification, perturbation, and revision steps. +- **why_not_adjacent_implementation_details:** The structural dependency is the annotation workflow itself, not the specific UI design or instructions provided to workers. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the structured workflow used to generate the dataset's core 4-tuples. +- **rationale:** The specific four-step annotation protocol is introduced in the target paper itself to construct the dataset, so it maps cleanly to `NONE`. +- **evidence_span:** “This dataset contains inferable participant states; a counterfactual perturbation to each state; and the changes to the story that would be necessary” + +### Ingredient C1.I3 +- **ingredient_id:** C1.I3 +- **ingredient:** Two-stage quality control and expert-review filtering process +- **why_structurally_necessary:** The dataset's reliability depends on filtering out noisy or logically inconsistent counterfactual revisions. Without this quality control workflow, the dataset would lose its defining property as a high-quality reasoning resource. +- **why_not_lower_level_substeps:** This should remain one quality-control ingredient rather than being split into worker qualifications, rejection thresholds, and expert review separately. +- **why_not_adjacent_implementation_details:** The structural dependency is the broader filtering workflow, not the specific payment amounts or geographic restrictions. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Ensures the logical consistency and quality of the crowdsourced counterfactual revisions. +- **rationale:** The two-stage filtering and expert review process is a paper-specific protocol designed to ensure the quality of the released resource, making `NONE` the correct grounding. +- **evidence_span:** “For filtering out noisy data from the collected responses, we follow a two-stage filtering process.” + +--- + +## Claim C2: Dataset Creation Methodology + +### Ingredient C2.I1 +- **ingredient_id:** C2.I1 +- **ingredient:** Conceptual framing of unstated participant states and counterfactuals in narratives +- **why_structurally_necessary:** The methodology is designed specifically to capture implicit states and how their counterfactuals affect the narrative discourse. Without this framing, the methodology has no conceptual basis. +- **why_not_lower_level_substeps:** This should remain a high-level conceptual ingredient rather than being split into specific types of states (physical, emotional, societal). +- **why_not_adjacent_implementation_details:** The structural dependency is the conceptual framing itself, not the specific formal theories of mind or time. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the theoretical basis for the methodology, focusing on implicit states and their counterfactual impacts on narratives. +- **rationale:** The paper explicitly introduces this theory-agnostic framing of unstated participant states and counterfactuals to motivate the methodology. It is not cleanly inherited from one prior study, so `NONE` is the correct grounding. +- **evidence_span:** “A model that understands narratives should likewise infer these implicit states, and even reason about the impact of changes to these states on the narrative.” + +### Ingredient C2.I2 +- **ingredient_id:** C2.I2 +- **ingredient:** Four-step annotation workflow for state inference and story revision +- **why_structurally_necessary:** This is the core operational mechanism of the methodology, defining exactly how annotators generate the required data tuples. +- **why_not_lower_level_substeps:** This should remain one workflow ingredient rather than being split into the individual inference, justification, perturbation, and revision steps. +- **why_not_adjacent_implementation_details:** The structural dependency is the workflow itself, not the specific crowdsourcing platform used. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** CORE_METHOD +- **contribution:** Defines the procedural steps required to execute the methodology and generate counterfactual narrative datasets. +- **rationale:** The specific four-step annotation workflow is the core contribution of the methodology claim and is introduced in the target paper itself, making `NONE` the correct grounding. +- **evidence_span:** “This dataset contains inferable participant states; a counterfactual perturbation to each state; and the changes to the story that would be necessary” + +--- + +# 5. Excluded Tempting Non-Ingredients + +### Claim C1 & C2 exclusions +- **Baseline models (GPT3, T5, BERT, RoBERTa)** + - Excluded because they are used to benchmark and evaluate the dataset, not to construct the dataset or the methodology itself. +- **The three state-based reasoning tasks (Story State Inference, Story Revision, State Change Generation)** + - Excluded because they are downstream tasks proposed to evaluate models on the dataset, not structural ingredients of the dataset or methodology artifacts themselves. +- **Amazon Mechanical Turk (AMT)** + - Excluded because it is an implementation detail for crowdsourcing, not a structural ingredient of the methodology or dataset. + +--- + +# 6. Candidate Grounding Decisions + +## Claim C1: PASTA Dataset + +### Ingredient C1.I1 +- **Candidate:** A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories +- **decision:** accepted_canonical +- **why:** Cleanest representative study for the ROCStories corpus, which provides the base narrative text used to construct the dataset. + +### Ingredient C1.I2 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The four-step crowdsourcing protocol is introduced in the target paper itself. + +### Ingredient C1.I3 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The two-stage quality control and expert-review filtering process is a paper-specific protocol. + +## Claim C2: Dataset Creation Methodology + +### Ingredient C2.I1 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The conceptual framing of unstated participant states and counterfactuals is introduced in the target paper as a theory-agnostic resource basis. + +### Ingredient C2.I2 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The four-step annotation workflow is the core procedural contribution of the methodology and is introduced in the target paper. + +--- + +# 7. Final Grounding Decisions + +## Claim C1 +- **C1.I1** → A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories → DATA_SOURCE +- **C1.I2** → NONE → EVALUATION_PROTOCOL +- **C1.I3** → NONE → EVALUATION_PROTOCOL + +## Claim C2 +- **C2.I1** → NONE → CONCEPTUAL_FRAMEWORK +- **C2.I2** → NONE → CORE_METHOD + +--- + +# 8. Final Rationales And Evidence + +## Claim C1: PASTA Dataset + +### C1.I1 +- **ingredient:** Short narrative stories describing daily events +- **canonical study:** A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories +- **role:** DATA_SOURCE +- **contribution:** Provides the base narrative stories (ROCStories) that annotators read to infer states and generate counterfactual revisions. +- **rationale:** The dataset is built directly on top of the ROCStories corpus, which provides the necessary short, self-contained narratives about daily events. This prior study is the cleanest canonical grounding for the source text. +- **evidence_span:** “To create the PASTA dataset, we use stories from the extended ROCStories corpus for annotation by crowd workers.” + +### C1.I2 +- **ingredient:** Four-step crowdsourcing protocol for eliciting implicit states and counterfactual revisions +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the structured workflow used to generate the dataset's core 4-tuples. +- **rationale:** The specific four-step annotation protocol is introduced in the target paper itself to construct the dataset, so it maps cleanly to `NONE`. +- **evidence_span:** “This dataset contains inferable participant states; a counterfactual perturbation to each state; and the changes to the story that would be necessary” + +### C1.I3 +- **ingredient:** Two-stage quality control and expert-review filtering process +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Ensures the logical consistency and quality of the crowdsourced counterfactual revisions. +- **rationale:** The two-stage filtering and expert review process is a paper-specific protocol designed to ensure the quality of the released resource, making `NONE` the correct grounding. +- **evidence_span:** “For filtering out noisy data from the collected responses, we follow a two-stage filtering process.” + +## Claim C2: Dataset Creation Methodology + +### C2.I1 +- **ingredient:** Conceptual framing of unstated participant states and counterfactuals in narratives +- **canonical study:** NONE +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the theoretical basis for the methodology, focusing on implicit states and their counterfactual impacts on narratives. +- **rationale:** The paper explicitly introduces this theory-agnostic framing of unstated participant states and counterfactuals to motivate the methodology. It is not cleanly inherited from one prior study, so `NONE` is the correct grounding. +- **evidence_span:** “A model that understands narratives should likewise infer these implicit states, and even reason about the impact of changes to these states on the narrative.” + +### C2.I2 +- **ingredient:** Four-step annotation workflow for state inference and story revision +- **canonical study:** NONE +- **role:** CORE_METHOD +- **contribution:** Defines the procedural steps required to execute the methodology and generate counterfactual narrative datasets. +- **rationale:** The specific four-step annotation workflow is the core contribution of the methodology claim and is introduced in the target paper itself, making `NONE` the correct grounding. +- **evidence_span:** “This dataset contains inferable participant states; a counterfactual perturbation to each state; and the changes to the story that would be necessary” \ No newline at end of file diff --git a/replay_traces/2208.00329/two_pass_outputs/20260726T205558Z/pass_1_reasoning.prompt.txt b/replay_traces/2208.00329/two_pass_outputs/20260726T205558Z/pass_1_reasoning.prompt.txt new file mode 100644 index 0000000000000000000000000000000000000000..38327156cfe7072152fec0b470d86be6169be0fd --- /dev/null +++ b/replay_traces/2208.00329/two_pass_outputs/20260726T205558Z/pass_1_reasoning.prompt.txt @@ -0,0 +1,4047 @@ + +Task: Produce the reasoning memo for cluster-first claim derivation and enabling-ingredient annotation. + +Important: +- Think cluster-first. +- Do not start from a paper-level discovery summary. +- Return the smallest structurally sufficient ingredient set. + +Required procedure: +1. Identify downstream-used claims from the cluster evidence. +2. Split claims only when downstream-used contributions are distinct. +3. Rewrite each claim at the abstraction level matching the intended ingredients. +4. For each claim, first list broad candidate ingredients and possible grounding studies. +5. Collapse those candidates into the minimal structurally sufficient ingredient set. +6. Explicitly exclude tempting non-ingredients. E.g. + - one-task tools + - local preprocessing + - balancing tricks + - token/language-tag tricks + - helper models used for one small step +7. Only then decide canonical vs additional vs NONE. + + +Benchmark/dataset claims: include only what is needed to build the artifact itself. Favor the benchmark framing, core conceputal idea, source tasks/data, and benchmark-wide annotation or verification protocol. Exclude anything the benchmark could still exist without. +Method/resource claims: include only what defines the artifact: core architecture, core objective, core conceputal idea, essential training data, or essential initialization. Exclude non-essential implementation details. +Finding claims: include only what is needed to observe and measure the empirical result: the evaluation protocol, core conceputal idea, studied model/artifact type, task substrate, and measurement/extraction procedure. Exclude recommendations, fixes, and general background unless they are operationally necessary. + +For each final ingredient, provide: +- ingredient +- why it is structurally necessary +- why tempting alternatives are excluded +- canonical grounding decision +- additional groundings if any +- role +- contribution +- rationale +- evidence:one verbatim quote + +Return a markdown reasoning memo only. +-------------------------------------------------- +REFERENCE EXAMPLE 1 +-------------------------------------------------- + +# EXAMPLE: Pick-a-Pic / PickScore + +## 1. Cluster Evidence + +**The PickScore metric, a CLIP-based scoring function that predicts human preferences to evaluate and rank text-to-image generation models.** + +The paper introduces two downstream-used contributions: PickScore and the Pick-a-Pic dataset. Downstream evidence shows separate reuse: + +- PickScore is reused as an evaluation metric / reward model. +- Pick-a-Pic is reused as a dataset for training and preference tuning. + +So the bundled contribution should be split into two atomic claims. + +--- + +## 2. Claim Split Decision + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Tool +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C1 + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Dataset +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C2 + +### Why this split is correct + +The two artifacts have distinct downstream roles and distinct ingredient decompositions: + +- **PickScore** is reused as a scoring / evaluation / reward model. +- **Pick-a-Pic** is reused as a preference dataset for training and benchmarking. + +They should therefore be annotated separately. + +--- + +## 3. Rewritten Claims + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Tool +- **rewritten_claim:** **"Tool: A human-preference prompt-conditioned image scorer, enabling automatic ranking of text-to-image model outputs** +- **why_this_is_atomic:** This claim isolates the scorer itself as the reused artifact. It does not bundle the dataset resource used to train it. +- **optional cluster_id:** C1 +- **decision:** YES_SUFFICIENT + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Dataset +- **rewritten_claim:** **Dataset: An open prompt-conditioned human-preference dataset for generated images, enabling training and evaluation of preference-aligned text-to-image systems.** +- **why_this_is_atomic:** This claim isolates the dataset as the released reusable resource, separate from the scoring model trained on it. +- **optional cluster_id:** C2 +- **decision:** YES_SUFFICIENT + +--- + +## 4. Minimal Ingredient Set Per Claim + +## Claim C1: PickScore + +### Ingredient C1.I1 +- **ingredient_id:** C1.I1 +- **ingredient:** Pre-trained vision language model for prompt-conditioned image scoring +- **why_structurally_necessary:** PickScore is explicitly a CLIP-based scoring function. Without a pretrained joint text-image model, the scorer would not exist in its claimed form. +- **why_not_lower_level_substeps:** This should stay at the level of pretrained vision-language initialization, not be decomposed into separate text encoder, image encoder, or projection details. +- **why_not_adjacent_implementation_details:** Specific OpenCLIP variants are narrower implementation choices rather than the core structural dependency. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Learning Transferable Visual Models From Natural Language Supervision +- **additional groundings if any:** none +- **role:** MODEL_INITIALIZATION +- **contribution:** Provides the pre-trained text and image representations that are fine-tuned into a human preference score. +- **rationale:** To score a generated image relative to a prompt, the function needs a pretrained joint text–image representation. CLIP is the cleanest representative study for such capability and is the direct architectural basis of PickScore. +- **evidence_span:** “PickScore follows the architecture of CLIP [12]” + +### Ingredient C1.I2 +- **ingredient_id:** C1.I2 +- **ingredient:** Large-scale dataset of human-preference judgements over text-to-image outputs +- **why_structurally_necessary:** The scorer must learn which of two generated images humans prefer for a given prompt. Without this training signal, it would not become a human-preference metric. +- **why_not_lower_level_substeps:** This should remain a single training-data ingredient rather than being split into prompts, pairs, ties, logging, or collection mechanics. +- **why_not_adjacent_implementation_details:** Dataset construction details belong to the dataset claim, not the scorer’s minimal training-data dependency. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** TRAINING_DATA +- **contribution:** Provides the pairwise user choices needed to fit the scorer. +- **rationale:** The model needs direct supervision about which of two generated images users prefer for a given prompt. This operative dataset is released in the target paper, so the right grounding is `NONE`. +- **evidence_span:** “We finetune CLIP-H [7] using our framework8 on the Pick-a-Pic training set.” + +### Ingredient C1.I3 +- **ingredient_id:** C1.I3 +- **ingredient:** Preference-learning objective to model the reward +- **why_structurally_necessary:** PickScore is trained to assign higher scores to preferred images. Without a preference-learning objective, it would not become a ranking metric that evaluates outputs by predicted human choice. +- **why_not_lower_level_substeps:** This should remain one core-method ingredient rather than being decomposed into loss cases, tie handling, or optimization details. +- **why_not_adjacent_implementation_details:** Training schedule and optimizer are implementation details; the structural dependency is the reward-model-style objective itself. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Training language models to follow instructions with human feedback +- **additional groundings if any:** none +- **role:** CORE_METHOD +- **contribution:** Learns scores whose differences reflect human preference between candidate images for the same prompt. +- **rationale:** Without that preference-learning step, the system would not become a ranking metric that can evaluate outputs. The paper explicitly grounds this in a variant of InstructGPT’s reward model objective. +- **evidence_span:** “We train the PickScore scoring function over Pick-a-Pic by combining a CLIP-style model with a variant of In-structGPT’s reward model objective” + +--- + +## Claim C2: Pick-a-Pic + +### Ingredient C2.I1 +- **ingredient_id:** C2.I1 +- **ingredient:** Framework for pairwise human preference collection using prompt-conditioned image comparisons +- **why_structurally_necessary:** The defining property of the dataset is that it records human preferences over generated image pairs for prompts. Without a collection framework that elicits those comparisons, the dataset would not exist in its claimed form. +- **why_not_lower_level_substeps:** This should remain one collection-framework ingredient rather than being split into web app interface details, session flow, or tie-button mechanics. +- **why_not_adjacent_implementation_details:** The structural point is the pairwise human-preference collection framework, not the exact UI implementation. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the framework to collect human preferences on a large scale. +- **rationale:** A framework is necessary to collect human preference choices at scale. The authors implement a web app that lets users generate images and specify preferences, and that workflow is a defining part of the dataset rather than a prior reusable study. +- **evidence_span:** “To address this issue, we create a web app that enables text-to-image users to generate images and specify their preferences.” + +### Ingredient C2.I2 +- **ingredient_id:** C2.I2 +- **ingredient:** Diffusion models for image generation from text +- **why_structurally_necessary:** The dataset consists of preferences over generated images. Without text-to-image generators, there would be no candidate images to compare and annotate. +- **why_not_lower_level_substeps:** This should remain at the level of text-to-image diffusion generation, not be split into individual backbone variants or guidance settings. +- **why_not_adjacent_implementation_details:** The exact backbone mix is less important than the general generation capability used to produce the images. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** High-Resolution Image Synthesis with Latent Diffusion Models +- **additional groundings if any:** none +- **role:** CORE_METHOD +- **contribution:** Provides the diffusion models needed to generate the images in the dataset. +- **rationale:** The dataset requires text-to-image generation backbones to convert user prompts into candidate images for preference annotation. Latent Diffusion Models is the cleanest representative prior study for that capability. +- **evidence_span:** “The images in the dataset were generated by employing multiple backbone models, namely, Stable Diffusion 2.1, Dreamlike Photoreal 2.0, and Stable Diffusion XL variants” + +### Ingredient C2.I3 +- **ingredient_id:** C2.I3 +- **ingredient:** Quality-control and preprocessing protocol for reliable open preference-data collection +- **why_structurally_necessary:** The dataset is released as an open, reusable human-preference resource. Without moderation, filtering, and preprocessing, it would lose one of its defining properties: being a reliable preference dataset rather than a raw interaction log. +- **why_not_lower_level_substeps:** This should remain one protocol-level ingredient rather than being split into NSFW phrase lists, account filtering, banning, or split construction separately. +- **why_not_adjacent_implementation_details:** These local measures are all parts of the same broader quality-control workflow and should not be promoted to standalone ingredients. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the moderation, filtering, and preprocessing workflow that makes the open preference dataset reliable and usable. +- **rationale:** Constructing an open preference dataset requires more than collecting raw comparisons. The paper includes moderation, harmful-content filtering, and preprocessing decisions that are structurally important to the quality of the released resource. This is paper-specific and therefore maps cleanly to `NONE`. +- **evidence_span:** “we closely monitor user activity logs and take action to ban users” + +--- + +## 5. Excluded Tempting Non-Ingredients + +### Claim C1 exclusions +- **Pick-a-Pic web app** + - Excluded because for the scorer claim, the structural dependency is the availability of human preference training data, not the full data-collection workflow. + +- **Preference-based evaluation setup for ranking candidate images by predicted human choice** + - Excluded because this is already captured by the rewritten claim and by the combination of human preference training data plus the reward-model objective. Adding it would double count the same functional role. + +### Claim C2 exclusions +- **List of NSFW phrases** + - Excluded because it is only one local component of the broader quality-control and preprocessing protocol, not a standalone structural ingredient. + +- **Specific diffusion backbone variants** + - Excluded because the structural dependency is text-to-image diffusion generation as a class, not each particular model instance. + +- **PickScore** + - Excluded because the dataset is a prerequisite for the tool, not the reverse. + +--- + +## 6. Candidate Grounding Decisions + +## Claim C1: PickScore + +### Ingredient C1.I1 +- **Candidate:** Learning Transferable Visual Models From Natural Language Supervision +- **decision:** accepted_canonical +- **why:** Provides the pretrained joint text-image architecture used as the basis for prompt-conditioned scoring. + +### Ingredient C1.I2 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The operative human preference dataset is released in the target paper itself. + +### Ingredient C1.I3 +- **Candidate:** Training language models to follow instructions with human feedback +- **decision:** accepted_canonical +- **why:** Provides the reward-model-style objective explicitly adapted by the paper. + +## Claim C2: Pick-a-Pic + +### Ingredient C2.I1 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The pairwise collection framework is implemented in the paper and not cleanly inherited from one prior study. + +### Ingredient C2.I2 +- **Candidate:** High-Resolution Image Synthesis with Latent Diffusion Models +- **decision:** accepted_canonical +- **why:** Cleanest representative of the text-to-image diffusion generation capability used to create the candidate images. + +### Ingredient C2.I3 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The quality-control and preprocessing workflow is a paper-specific protocol rather than one canonical prior study. + +--- + +## 7. Final Grounding Decisions + +## Claim C1 +- **C1.I1** → Learning Transferable Visual Models From Natural Language Supervision → MODEL_INITIALIZATION +- **C1.I2** → NONE → TRAINING_DATA +- **C1.I3** → Training language models to follow instructions with human feedback → CORE_METHOD + +## Claim C2 +- **C2.I1** → NONE → EVALUATION_PROTOCOL +- **C2.I2** → High-Resolution Image Synthesis with Latent Diffusion Models → CORE_METHOD +- **C2.I3** → NONE → EVALUATION_PROTOCOL + +--- + +## 8. Final Rationales And Evidence + +## Claim C1: PickScore + +### C1.I1 +- **ingredient:** Pre-trained vision language model for prompt-conditioned image scoring +- **canonical study:** Learning Transferable Visual Models From Natural Language Supervision +- **role:** MODEL_INITIALIZATION +- **contribution:** Provides the pre-trained text and image representations that are fine-tuned into a human preference score. +- **rationale:** To score a generated image relative to a prompt, the function needs a pretrained joint text–image representation. CLIP is the cleanest representative study for such capability and is the direct architectural basis of PickScore. +- **evidence_span:** “PickScore follows the architecture of CLIP [12]; given a prompt x and an image y, our scoring function s computes a real number by representing x using a transformer text encoder and y using a transformer image encoder as d-dimensional vectors, and returning their inner product” + +### C1.I2 +- **ingredient:** Large-scale dataset of human-preference judgements over text-to-image outputs +- **canonical study:** NONE +- **role:** TRAINING_DATA +- **contribution:** Provides the pairwise user choices needed to fit the scorer. +- **rationale:** The model needs direct supervision about which of two generated images users prefer for a given prompt. This operative dataset is released in the target paper, so the correct grounding is `NONE`. +- **evidence_span:** “We finetune CLIP-H [7] using our framework8 on the Pick-a-Pic training set.” + +### C1.I3 +- **ingredient:** Preference-learning objective to model the reward +- **canonical study:** Training language models to follow instructions with human feedback +- **role:** CORE_METHOD +- **contribution:** Learns scores whose differences reflect human preference between candidate images for the same prompt. +- **rationale:** Without that preference-learning step, the system would not become a ranking metric that can evaluate outputs. The paper explicitly grounds this in a variant of InstructGPT’s reward model objective. +- **evidence_span:** “We train the PickScore scoring function over Pick-a-Pic by combining a CLIP-style model with a variant of In-structGPT’s reward model objective [10].” + +## Claim C2: Pick-a-Pic + +### C2.I1 +- **ingredient:** Framework for pairwise human preference collection using prompt-conditioned image comparisons +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the framework to collect human preferences on a large scale. +- **rationale:** A framework is necessary to collect human preference choices at a large scale. The authors implement a web app to collect this, and that workflow is a defining part of the dataset. +- **evidence_span:** “To address this issue, we create a web app that enables text-to-image users to generate images and specify their preferences.” + +### C2.I2 +- **ingredient:** Diffusion models for image generation from text +- **canonical study:** High-Resolution Image Synthesis with Latent Diffusion Models +- **role:** CORE_METHOD +- **contribution:** Provides the diffusion models needed to generate the images in the dataset. +- **rationale:** The dataset requires text-to-image generation backbones to convert prompts into candidate images for annotation. Latent Diffusion Models is the cleanest representative prior study for that capability. +- **evidence_span:** “The images in the dataset were generated by employing multiple backbone models, namely, Stable Diffusion 2.1, Dreamlike Photoreal 2.0 5 , and Stable Diffusion XL variants [13] while sampling different classifier-free guidance scale values [6].” + +### C2.I3 +- **ingredient:** Quality-control and preprocessing protocol for reliable open preference-data collection +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the moderation, filtering, and preprocessing workflow that makes the open preference dataset reliable and usable. +- **rationale:** Constructing an open preference dataset requires more than collecting raw interactions. The released resource depends on moderation, filtering, and preprocessing steps that preserve data quality and usability. Because this workflow is paper-specific, `NONE` is the right grounding. +- **evidence_span:** “we closely monitor user activity logs and take action to ban users” + +-------------------------------------------------- +REFERENCE EXAMPLE 2 +-------------------------------------------------- + +# EXAMPLE: Offline RL bottleneck finding + +## 1. Cluster Evidence + +**Finding that policy extraction and test-time generalization (rather than just value learning) are the main bottlenecks in offline RL, enabling improved algorithm design via better policy extraction objectives.** + +This original claim identifies a real downstream-used empirical contribution. It should remain a **single finding claim** rather than being split into separate findings about value learning, policy extraction, and test-time generalization, because the paper’s contribution is one integrated diagnosis of what limits offline RL performance in practice. + +--- + +## 2. Claim Split Decision + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Finding +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C1 + +### Why this split is correct + +The downstream reuse is of a single empirical conclusion: that offline RL performance is often limited more by **policy extraction and deployment-time generalization** than by value learning alone. The later algorithm-design implication is a consequence of that diagnosis, not a separate discovery claim. + +--- + +## 3. Rewritten Claims + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Finding +- **rewritten_claim:** **Finding: In offline reinforcement learning, policy extraction and test-time policy generalization often limit performance more than value learning alone, making policy-extraction objective choice a primary determinant of final returns.** +- **why_this_is_atomic:** This claim isolates one empirical result about the source of performance bottlenecks in offline RL. It does not split off the downstream design recommendation as a separate claim. +- **optional cluster_id:** C1 +- **decision:** YES_SUFFICIENT + +--- + +## 4. Minimal Ingredient Set Per Claim + +## Claim C1: Offline RL bottleneck finding + +### Ingredient C1.I1 +- **ingredient_id:** C1.I1 +- **ingredient:** Offline RL methods with decoupled value learning and policy extraction phases +- **why_structurally_necessary:** The finding depends on separating the quality of the learned critic from the quality of the extracted policy. Without a decoupled setup, the paper could not attribute performance differences to policy extraction rather than entangled actor-critic training. +- **why_not_lower_level_substeps:** This should remain one high-level ingredient about the decoupled offline RL setup, not be split into each individual algorithm separately. +- **why_not_adjacent_implementation_details:** Specific architectural or optimization choices inside one method are not the point; the structural dependency is the existence of a method family where value learning and policy extraction can be independently analyzed. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** Offline RL Without Off-Policy Evaluation; Offline Reinforcement Learning with Implicit Q-Learning; Contrastive Learning as Goal-Conditioned Reinforcement Learning +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the offline RL setup in which value learning is trained independently of policy extraction, enabling the paper’s bottleneck analysis. +- **rationale:** To show that policy extraction rather than value learning is often the limiting factor, the paper must analyze settings where the critic can be held fixed while different extraction procedures are compared. That dependence is structurally on a family of decoupled offline RL methods rather than on one single canonical paper, so `NONE` is the faithful canonical choice and the relevant prior methods are better represented as additional studies. +- **evidence_span:** “we focus on offline RL methods with decoupled value and policy training phases” + +### Ingredient C1.I2 +- **ingredient_id:** C1.I2 +- **ingredient:** Behavior-constrained policy-gradient extraction objectives for policies learned from fixed critics +- **why_structurally_necessary:** The finding is not only that some extraction methods underperform, but that behavior-constrained policy-gradient extraction often performs better and scales better than common alternatives under the same learned critic. +- **why_not_lower_level_substeps:** This should remain one extraction-family ingredient rather than being split into every equation or policy-update detail. +- **why_not_adjacent_implementation_details:** Hyperparameters and optimizer details are irrelevant here. The structural ingredient is the extraction objective family itself. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** A Minimalist Approach to Offline Reinforcement Learning +- **additional groundings if any:** Advantage-Weighted Regression: Simple and Scalable Off-Policy Reinforcement Learning; Offline Reinforcement Learning via High-Fidelity Generative Behavior Modeling +- **role:** CORE_METHOD +- **contribution:** Provides the representative behavior-regularized policy-gradient extraction objective that the paper identifies as stronger and more scalable than common alternatives. +- **rationale:** The paper’s empirical conclusion depends on contrasting extraction families and finding that DDPG+BC-style extraction often outperforms widely used value-weighted or sampling-based alternatives. DDPG+BC is the clearest representative study for that stronger extraction family, while AWR and SfBC-style methods serve as additional comparison groundings rather than the canonical representative. +- **evidence_span:** “switching to behavior-constrained policy gradient objectives (e.g., DDPG+BC) often leads to substantial improvements” + +### Ingredient C1.I3 +- **ingredient_id:** C1.I3 +- **ingredient:** Comparative protocol that evaluates extraction objectives across data regimes and deployment generalization settings +- **why_structurally_necessary:** The claim is a comparative empirical finding about bottlenecks and scaling behavior. It requires an analysis protocol that systematically varies data properties and observes how extraction choices affect final performance and generalization. +- **why_not_lower_level_substeps:** This should remain one protocol-level ingredient rather than being split into separate dataset-size sweeps, quality sweeps, and metric definitions. +- **why_not_adjacent_implementation_details:** Individual plots, metrics, or visualization choices are not separate ingredients; they are local parts of the broader comparative protocol. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the systematic empirical protocol that isolates policy extraction and generalization bottlenecks across environments and data regimes. +- **rationale:** A finding about what the main bottleneck is cannot be established from one aggregate return table alone. The paper needs a structured comparative protocol that varies data size, quality, and coverage while comparing extraction methods under shared critics. That protocol is built in the target paper itself, so `NONE` is the correct grounding. +- **evidence_span:** “We use data size, quality, and coverage as levers for systematically controlling their impacts” + +--- + +## 5. Excluded Tempting Non-Ingredients + +### Claim C1 exclusions +- **Offline RL setting with fixed logged data and no online interaction** + - Excluded because it is too broad and field-level. The more precise structural ingredient is the **decoupled offline RL setup** that lets the paper isolate value learning from extraction. + +- **Advantage-weighted regression as its own standalone ingredient** + - Excluded as a separate final ingredient because it is better represented as an **additional grounding** under the broader comparative extraction-objective ingredient, rather than being promoted to its own parallel ingredient. + +- **Standard offline RL benchmark datasets as a separate ingredient** + - Excluded because the finding depends more centrally on the paper’s broader comparative protocol than on one benchmark substrate alone. The datasets are part of that empirical protocol rather than a standalone higher-level ingredient. + +- **Test-time generalization as a separate conceptual ingredient** + - Excluded because it is already captured within the rewritten claim and within the comparative evaluation protocol ingredient. Keeping it separate would over-decompose the finding. + +--- + +## 6. Candidate Grounding Decisions + +## Claim C1: Offline RL bottleneck finding + +### Ingredient C1.I1 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The finding relies on a family of decoupled offline RL methods rather than one single representative study. +- **Candidate:** Offline RL Without Off-Policy Evaluation +- **decision:** accepted_additional +- **why:** Representative prior decoupled offline RL method explicitly listed by the paper. +- **Candidate:** Offline Reinforcement Learning with Implicit Q-Learning +- **decision:** accepted_additional +- **why:** Another representative decoupled method used to motivate the analysis setup. +- **Candidate:** Contrastive Learning as Goal-Conditioned Reinforcement Learning +- **decision:** accepted_additional +- **why:** Included as a representative decoupled method family member rather than the single canonical grounding. + +### Ingredient C1.I2 +- **Candidate:** A Minimalist Approach to Offline Reinforcement Learning +- **decision:** accepted_canonical +- **why:** Cleanest representative study for behavior-constrained policy-gradient extraction, which the paper identifies as the stronger extraction family. +- **Candidate:** Advantage-Weighted Regression: Simple and Scalable Off-Policy Reinforcement Learning +- **decision:** accepted_additional +- **why:** Necessary comparison grounding for the weaker value-weighted extraction family contrasted in the finding. +- **Candidate:** Offline Reinforcement Learning via High-Fidelity Generative Behavior Modeling +- **decision:** accepted_additional +- **why:** Relevant additional comparison study for alternative policy extraction objectives. + +### Ingredient C1.I3 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The comparative data-scaling and deployment-generalization protocol is constructed in the target paper rather than inherited from one prior study. + +--- + +## 7. Final Grounding Decisions + +## Claim C1 +- **C1.I1** → NONE → CONCEPTUAL_FRAMEWORK + - additional: Offline RL Without Off-Policy Evaluation; Offline Reinforcement Learning with Implicit Q-Learning; Contrastive Learning as Goal-Conditioned Reinforcement Learning +- **C1.I2** → A Minimalist Approach to Offline Reinforcement Learning → CORE_METHOD + - additional: Advantage-Weighted Regression: Simple and Scalable Off-Policy Reinforcement Learning; Offline Reinforcement Learning via High-Fidelity Generative Behavior Modeling +- **C1.I3** → NONE → EVALUATION_PROTOCOL + +--- + +## 8. Final Rationales And Evidence + +## Claim C1: Offline RL bottleneck finding + +### C1.I1 +- **ingredient:** Offline RL methods with decoupled value learning and policy extraction phases +- **canonical study:** NONE +- **additional studies:** Offline RL Without Off-Policy Evaluation; Offline Reinforcement Learning with Implicit Q-Learning; Contrastive Learning as Goal-Conditioned Reinforcement Learning +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the offline RL setup in which value learning is trained independently of policy extraction, enabling the paper’s bottleneck analysis. +- **rationale:** To show that policy extraction rather than value learning is often the limiting factor, the paper must analyze settings where the critic can be held fixed while different extraction procedures are compared. That dependence is structurally on a family of decoupled offline RL methods rather than on one single canonical paper, so `NONE` is the faithful canonical choice and the relevant prior methods are better represented as additional studies. +- **evidence_span:** “we focus on offline RL methods with decoupled value and policy training phases” + +### C1.I2 +- **ingredient:** Behavior-constrained policy-gradient extraction objectives for policies learned from fixed critics +- **canonical study:** A Minimalist Approach to Offline Reinforcement Learning +- **additional studies:** Advantage-Weighted Regression: Simple and Scalable Off-Policy Reinforcement Learning; Offline Reinforcement Learning via High-Fidelity Generative Behavior Modeling +- **role:** CORE_METHOD +- **contribution:** Provides the representative behavior-regularized policy-gradient extraction objective that the paper identifies as stronger and more scalable than common alternatives. +- **rationale:** The paper’s empirical conclusion depends on contrasting extraction families and finding that DDPG+BC-style extraction often outperforms widely used value-weighted or sampling-based alternatives. DDPG+BC is the clearest representative study for that stronger extraction family, while AWR and SfBC-style methods serve as additional comparison groundings rather than the canonical representative. +- **evidence_span:** “switching to behavior-constrained policy gradient objectives (e.g., DDPG+BC) often leads to substantial improvements” + +### C1.I3 +- **ingredient:** Comparative protocol that evaluates extraction objectives across data regimes and deployment generalization settings +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the systematic empirical protocol that isolates policy extraction and generalization bottlenecks across environments and data regimes. +- **rationale:** A finding about what the main bottleneck is cannot be established from one aggregate return table alone. The paper needs a structured comparative protocol that varies data size, quality, and coverage while comparing extraction methods under shared critics. That protocol is built in the target paper itself, so `NONE` is the correct grounding. +- **evidence_span:** “We use data size, quality, and coverage as levers for systematically controlling their impacts” + + +-------------------------------------------------- +REFERENCE EXAMPLE 3 +-------------------------------------------------- + +# EXAMPLE: IndicXTREME / IndicBERT + +## 1. Cluster Evidence + +Based on downstream cluster contributions, the paper introduces two distinct downstream-used contributions: + +- a **benchmark** for multilingual zero-shot evaluation on Indic languages +- a **pretrained multilingual encoder** for Indic NLU + +Downstream evidence shows these are reused differently and should therefore be split into two atomic claims. + +--- + +## 2. Claim Split Decision + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Benchmark +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C5 + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Resource +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C2 + +### Why this split is correct + +The paper contributes two separate artifacts with distinct downstream roles: + +- **IndicXTREME** is reused as a multilingual benchmark for zero-shot evaluation. +- **IndicBERT v2** is reused as a pretrained model for transfer and downstream fine-tuning. + +These have different structural ingredients, so they should not be annotated as one merged claim. + +--- + +## 3. Rewritten Claims + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Benchmark +- **rewritten_claim:** **Benchmark: A multilingual Indic-language evaluation suite, enabling standardized zero-shot assessment of pretrained multilingual models across diverse NLU tasks.** +- **why_this_is_atomic:** This claim isolates the benchmark artifact and its evaluation role, without bundling the pretrained model. +- **optional cluster_id:** C5 +- **decision:** YES_SUFFICIENT + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Resource +- **rewritten_claim:** **Resource: A pretrained multilingual encoder for Indic languages, enabling zero-shot transfer and downstream adaptation on Indic NLU tasks.** +- **why_this_is_atomic:** This claim isolates the pretrained encoder as the reused artifact, separate from the benchmark used to evaluate such models. +- **optional cluster_id:** C2 +- **decision:** YES_SUFFICIENT + +--- + +## 4. Minimal Ingredient Set Per Claim + +## Claim C1: Indic benchmark + +### Ingredient C1.I1 +- **ingredient_id:** C1.I1 +- **ingredient:** Multilingual, multi-task, zero-shot evaluation framework for cross-lingual NLU benchmarking +- **why_structurally_necessary:** The benchmark is explicitly framed as a multilingual zero-shot evaluation suite across tasks and languages. Without this benchmark framing, it would not exist in its claimed form as a standardized cross-lingual evaluation resource. +- **why_not_lower_level_substeps:** This should remain one benchmark-paradigm ingredient rather than being split into separate benchmark papers or separate task families. +- **why_not_adjacent_implementation_details:** The structural dependency is the zero-shot multilingual benchmark framing itself, not local benchmark formatting or reporting choices. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** XTREME: A Massively Multilingual Multi-task Benchmark for Evaluating Cross-lingual Generalization +- **additional groundings if any:** XTREME-R: Towards More Challenging and Nuanced Multilingual Evaluation; XGLUE: A New Benchmark Dataset for Cross-lingual Pre-training, Understanding and Generation +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the multilingual zero-shot benchmark framing that the paper adapts to Indic languages. +- **rationale:** To construct a benchmark whose point is standardized zero-shot evaluation across many languages and tasks, the paper depends on an existing multilingual benchmark paradigm. XTREME is the cleanest canonical grounding because it most directly established that framing; XTREME-R and XGLUE are relevant supporting studies but are less direct representatives. +- **evidence_span:** “aims to test the multilingual zero-shot capabilities of pretrained language models” + +### Ingredient C1.I2 +- **ingredient_id:** C1.I2 +- **ingredient:** NLU task datasets and source resources used to assemble the benchmark +- **why_structurally_necessary:** A multi-task benchmark requires the actual upstream task substrates from which its evaluation sets are translated, adapted, corrected, or incorporated. +- **why_not_lower_level_substeps:** This should remain one composite data-source ingredient rather than being split into separate ingredients for each task dataset. +- **why_not_adjacent_implementation_details:** The key dependency is the heterogeneous task substrate itself, not each local adaptation step used for one task. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** Choice of Plausible Alternatives: An Evaluation of Commonsense Causal Reasoning; XNLI: Evaluating Cross-lingual Sentence Representations; IndicXNLI: Evaluating Multilingual Inference for Indian Languages; Naamapadam: A Large-Scale Named Entity Annotated Data for Indic Languages; The Flores-101 Evaluation Benchmark for Low-Resource and Multilingual Machine Translation; MASSIVE: A 1M-Example Multilingual Natural Language Understanding Dataset with 51 Typologically-Diverse Languages +- **role:** DATA_SOURCE +- **contribution:** Supplies the task definitions and raw/source evaluation material that the benchmark translates, subsets, verifies, or incorporates. +- **rationale:** Constructing a multi-task benchmark requires the underlying task substrates that define its evaluation sets. Because these substrates are inherently composite across multiple tasks and datasets, they do not map cleanly to a single prior study, making `NONE` the appropriate canonical choice, with the task-specific sources listed as additional studies. +- **evidence_span:** “We manually translate the COPA test set into 18 Indic languages to create IndicCOPA.” + +### Ingredient C1.I3 +- **ingredient_id:** C1.I3 +- **ingredient:** Human-supervised multilingual curation, translation, and verification protocol +- **why_structurally_necessary:** Human supervision is one of the benchmark’s defining claimed properties. Without it, the benchmark would not exist in the same form as a human-validated evaluation resource. +- **why_not_lower_level_substeps:** This should remain one protocol-level ingredient rather than being split into manual translation, verification, correction, and annotation as separate ingredients. +- **why_not_adjacent_implementation_details:** The structural dependency is the benchmark-wide human-supervision workflow, not each local annotation step. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the manual benchmark-construction process ensuring that all included evaluation sets are created, translated, edited, or verified by humans. +- **rationale:** This ingredient is necessary because the benchmark’s defining property is that all evaluation sets are created or verified under human supervision. Removing it would fundamentally change the nature of the benchmark, reducing it to a weaker, non-human-validated collection of datasets. Because this workflow is benchmark-specific and realized in the target paper, `NONE` is the correct grounding. +- **evidence_span:** “ALL the evaluation sets included in IndicXTREME were created with human supervision” + +--- + +## Claim C2: Indic multilingual encoder + +### Ingredient C2.I1 +- **ingredient_id:** C2.I1 +- **ingredient:** Transformer-based masked language modeling architecture +- **why_structurally_necessary:** The resource is explicitly a BERT-style pretrained encoder. Without the transformer MLM architecture, it would not exist in its claimed form. +- **why_not_lower_level_substeps:** This should remain one architecture-level ingredient rather than being split into encoder layers, attention heads, or hyperparameter choices. +- **why_not_adjacent_implementation_details:** Model size, vocabulary size, and optimizer settings are implementation details; the structural dependency is the BERT-style MLM architecture itself. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding +- **additional groundings if any:** none +- **role:** CORE_METHOD +- **contribution:** Provides the transformer encoder architecture and masked language modeling objective used to pretrain the model. +- **rationale:** To build a multilingual BERT-style encoder, the paper requires a transformer architecture trained with masked language modeling. BERT is the clearest and most faithful canonical grounding for that dependency. +- **evidence_span:** “We use the default hyperparameters of BERT-Base” + +### Ingredient C2.I2 +- **ingredient_id:** C2.I2 +- **ingredient:** Translation Language Modeling objective for cross-lingual alignment +- **why_structurally_necessary:** The paper explicitly uses TLM as a second core objective to improve cross-lingual transfer by aligning languages through parallel data. +- **why_not_lower_level_substeps:** This should remain one objective-level ingredient rather than being split into masking details, parallel-pair formatting, or loss implementation details. +- **why_not_adjacent_implementation_details:** The structural dependency is the TLM objective itself, not specific data-preparation or training-engineering choices around it. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Cross-lingual Language Model Pretraining +- **additional groundings if any:** none +- **role:** CORE_METHOD +- **contribution:** Provides the cross-lingual pretraining objective used to align representations across languages. +- **rationale:** The model is not only trained with MLM; it explicitly incorporates TLM as a core objective for cross-lingual alignment. That objective maps cleanly to XLM as the canonical prior study. +- **evidence_span:** “Translation Language Modeling (Conneau and Lample, 2019, TLM)” + +### Ingredient C2.I3 +- **ingredient_id:** C2.I3 +- **ingredient:** Large-scale Indic-language pretraining text +- **why_structurally_necessary:** A multilingual encoder for Indic languages cannot be pretrained in its claimed form without a large-scale Indic-language text substrate. +- **why_not_lower_level_substeps:** This should remain one training-data ingredient rather than being split into IndicCorp, Wikipedia, OSCAR, Samanantar-derived text, or synthetic translations separately. +- **why_not_adjacent_implementation_details:** Specific corpus components and augmentation choices are subordinate pieces of the broader pretraining-text ingredient and should not be promoted to standalone structural ingredients. +- **necessary:** true +- **from_prior_work:** partially +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** IndicNLPSuite: Monolingual Corpora, Evaluation Benchmarks and Pre-trained Multilingual Language Models for Indian Languages; Samanantar: The Largest Publicly Available Parallel Corpora Collection for 11 Indic Languages +- **role:** TRAINING_DATA +- **contribution:** Provides the large-scale Indic-language text substrate used to pretrain the multilingual encoder. +- **rationale:** Recreating the model requires the actual large-scale Indic-language training text. But the operative pretraining text is structurally composite: it is dominated by corpus material created or assembled in the target paper, while also incorporating prior resources such as IndicCorp lineage and Samanantar-derived text. Because no single prior study cleanly represents that full training substrate, `NONE` is the correct canonical choice, with prior contributing resources listed as additional studies. +- **evidence_span:** “we merge data from IndicCorp v2 with Indic language data from Wikipedia and OSCAR” + +--- + +## 5. Excluded Tempting Non-Ingredients + +### Claim C1 exclusions +- **Machine translation model for auto-translating some source datasets** + - Excluded because it is not structural to the benchmark as a whole and is only used to help construct a small part of one task family rather than defining the benchmark itself. + +- **English training datasets used for zero-shot evaluation** + - Excluded because these are used to apply the benchmark in evaluation setups, not to construct the benchmark artifact itself. + +- **Individual task datasets as separate top-level ingredients** + - Excluded because the benchmark’s task substrate is better represented as one composite DATA_SOURCE ingredient rather than over-decomposed into one ingredient per task. + +### Claim C2 exclusions +- **IndicTrans as a standalone ingredient** + - Excluded because it is a helper tool for generating some synthetic parallel data, not a core structural ingredient of the model in its claimed form. + +- **Temperature-based upsampling / balancing tricks** + - Excluded because these are recipe-level training choices rather than artifact-defining ingredients. + +- **Language ID tokens / tokenizer / WordPiece design** + - Excluded because these are implementation-level configuration details, not part of the minimal structurally sufficient ingredient set. + +- **Samanantar as the canonical grounding for the full training-text ingredient** + - Excluded because it represents only one smaller component of the broader pretraining substrate; assigning it as canonical would distort the true structure of the ingredient. + +--- + +## 6. Candidate Grounding Decisions + +## Claim C1: Indic benchmark + +### Ingredient C1.I1 +- **Candidate:** XTREME: A Massively Multilingual Multi-task Benchmark for Evaluating Cross-lingual Generalization +- **decision:** accepted_canonical +- **why:** Cleanest representative study for the multilingual multi-task zero-shot benchmark framing adapted here. +- **Candidate:** XTREME-R: Towards More Challenging and Nuanced Multilingual Evaluation +- **decision:** accepted_additional +- **why:** Important related multilingual benchmark extension, but not the clearest canonical origin. +- **Candidate:** XGLUE: A New Benchmark Dataset for Cross-lingual Pre-training, Understanding and Generation +- **decision:** accepted_additional +- **why:** Relevant supporting multilingual benchmark study, but less direct as the primary framing. + +### Ingredient C1.I2 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The task substrate is structurally composite across multiple datasets and source resources. +- **Candidate:** Choice of Plausible Alternatives: An Evaluation of Commonsense Causal Reasoning +- **decision:** accepted_additional +- **why:** Source task for IndicCOPA. +- **Candidate:** XNLI: Evaluating Cross-lingual Sentence Representations +- **decision:** accepted_additional +- **why:** Important upstream NLI source task. +- **Candidate:** IndicXNLI: Evaluating Multilingual Inference for Indian Languages +- **decision:** accepted_additional +- **why:** Important Indic-specific benchmark source. +- **Candidate:** Naamapadam: A Large-Scale Named Entity Annotated Data for Indic Languages +- **decision:** accepted_additional +- **why:** Upstream NER task resource. +- **Candidate:** The Flores-101 Evaluation Benchmark for Low-Resource and Multilingual Machine Translation +- **decision:** accepted_additional +- **why:** Upstream retrieval/evaluation source resource. +- **Candidate:** MASSIVE: A 1M-Example Multilingual Natural Language Understanding Dataset with 51 Typologically-Diverse Languages +- **decision:** accepted_additional +- **why:** Upstream intent/slot resource. + +### Ingredient C1.I3 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The human-supervised curation and verification workflow is benchmark-specific and realized in the target paper. + +## Claim C2: Indic multilingual encoder + +### Ingredient C2.I1 +- **Candidate:** BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding +- **decision:** accepted_canonical +- **why:** Cleanest representative study for the transformer MLM architecture used by the model. + +### Ingredient C2.I2 +- **Candidate:** Cross-lingual Language Model Pretraining +- **decision:** accepted_canonical +- **why:** Cleanest representative study for the TLM objective explicitly used by the paper. + +### Ingredient C2.I3 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The pretraining text is dominated by same-paper corpus construction and is structurally composite across several sources. +- **Candidate:** IndicNLPSuite: Monolingual Corpora, Evaluation Benchmarks and Pre-trained Multilingual Language Models for Indian Languages +- **decision:** accepted_additional +- **why:** Important prior Indic corpus lineage that the paper extends. +- **Candidate:** Samanantar: The Largest Publicly Available Parallel Corpora Collection for 11 Indic Languages +- **decision:** accepted_additional +- **why:** Important contributing prior text source, but not representative enough to be canonical for the full pretraining-text ingredient. + +--- + +## 7. Final Grounding Decisions + +## Claim C1 +- **C1.I1** → XTREME: A Massively Multilingual Multi-task Benchmark for Evaluating Cross-lingual Generalization → CONCEPTUAL_FRAMEWORK + - additional: XTREME-R: Towards More Challenging and Nuanced Multilingual Evaluation; XGLUE: A New Benchmark Dataset for Cross-lingual Pre-training, Understanding and Generation +- **C1.I2** → NONE → DATA_SOURCE + - additional: Choice of Plausible Alternatives: An Evaluation of Commonsense Causal Reasoning; XNLI: Evaluating Cross-lingual Sentence Representations; IndicXNLI: Evaluating Multilingual Inference for Indian Languages; Naamapadam: A Large-Scale Named Entity Annotated Data for Indic Languages; The Flores-101 Evaluation Benchmark for Low-Resource and Multilingual Machine Translation; MASSIVE: A 1M-Example Multilingual Natural Language Understanding Dataset with 51 Typologically-Diverse Languages +- **C1.I3** → NONE → EVALUATION_PROTOCOL + +## Claim C2 +- **C2.I1** → BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding → CORE_METHOD +- **C2.I2** → Cross-lingual Language Model Pretraining → CORE_METHOD +- **C2.I3** → NONE → TRAINING_DATA + - additional: IndicNLPSuite: Monolingual Corpora, Evaluation Benchmarks and Pre-trained Multilingual Language Models for Indian Languages; Samanantar: The Largest Publicly Available Parallel Corpora Collection for 11 Indic Languages + +--- + +## 8. Final Rationales And Evidence + +## Claim C1: Indic benchmark + +### C1.I1 +- **ingredient:** Multilingual, multi-task, zero-shot evaluation framework for cross-lingual NLU benchmarking +- **canonical study:** XTREME: A Massively Multilingual Multi-task Benchmark for Evaluating Cross-lingual Generalization +- **additional studies:** XTREME-R: Towards More Challenging and Nuanced Multilingual Evaluation; XGLUE: A New Benchmark Dataset for Cross-lingual Pre-training, Understanding and Generation +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the multilingual zero-shot benchmark framing that the paper adapts to Indic languages. +- **rationale:** To construct a benchmark whose point is standardized zero-shot evaluation across many languages and tasks, the paper depends on an existing multilingual benchmark paradigm. XTREME is the cleanest canonical grounding because it most directly established that framing; XTREME-R and XGLUE are relevant supporting studies but are less direct representatives. +- **evidence_span:** “aims to test the multilingual zero-shot capabilities of pretrained language models” + +### C1.I2 +- **ingredient:** NLU task datasets and source resources used to assemble the benchmark +- **canonical study:** NONE +- **additional studies:** Choice of Plausible Alternatives: An Evaluation of Commonsense Causal Reasoning; XNLI: Evaluating Cross-lingual Sentence Representations; IndicXNLI: Evaluating Multilingual Inference for Indian Languages; Naamapadam: A Large-Scale Named Entity Annotated Data for Indic Languages; The Flores-101 Evaluation Benchmark for Low-Resource and Multilingual Machine Translation; MASSIVE: A 1M-Example Multilingual Natural Language Understanding Dataset with 51 Typologically-Diverse Languages +- **role:** DATA_SOURCE +- **contribution:** Supplies the task definitions and raw/source evaluation material that the benchmark translates, subsets, verifies, or incorporates. +- **rationale:** Constructing a multi-task benchmark requires the underlying task substrates that define its evaluation sets. Because these substrates are inherently composite across multiple tasks and datasets, they do not map cleanly to a single prior study, making `NONE` the appropriate canonical choice, with the task-specific sources listed as additional studies. +- **evidence_span:** “We manually translate the COPA test set into 18 Indic languages to create IndicCOPA.” + +### C1.I3 +- **ingredient:** Human-supervised multilingual curation, translation, and verification protocol +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the manual benchmark-construction process ensuring that all included evaluation sets are created, translated, edited, or verified by humans. +- **rationale:** This ingredient is necessary because the benchmark’s defining property is that all evaluation sets are created or verified under human supervision. Removing it would fundamentally change the nature of the benchmark, reducing it to a weaker, non-human-validated collection of datasets. Because this workflow is benchmark-specific and realized in the target paper, `NONE` is the correct grounding. +- **evidence_span:** “ALL the evaluation sets included in IndicXTREME were created with human supervision” + +## Claim C2: Indic multilingual encoder + +### C2.I1 +- **ingredient:** Transformer-based masked language modeling architecture +- **canonical study:** BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding +- **role:** CORE_METHOD +- **contribution:** Provides the transformer encoder architecture and masked language modeling objective used to pretrain the model. +- **rationale:** To build a multilingual BERT-style encoder, the paper requires a transformer architecture trained with masked language modeling. BERT is the clearest and most faithful canonical grounding for that dependency. +- **evidence_span:** “We use the default hyperparameters of BERT-Base” + +### C2.I2 +- **ingredient:** Translation Language Modeling objective for cross-lingual alignment +- **canonical study:** Cross-lingual Language Model Pretraining +- **role:** CORE_METHOD +- **contribution:** Provides the cross-lingual pretraining objective used to align representations across languages. +- **rationale:** The model is not only trained with MLM; it explicitly incorporates TLM as a core objective for cross-lingual alignment. That objective maps cleanly to XLM as the canonical prior study. +- **evidence_span:** “Translation Language Modeling (Conneau and Lample, 2019, TLM)” + +### C2.I3 +- **ingredient:** Large-scale Indic-language pretraining text +- **canonical study:** NONE +- **additional studies:** IndicNLPSuite: Monolingual Corpora, Evaluation Benchmarks and Pre-trained Multilingual Language Models for Indian Languages; Samanantar: The Largest Publicly Available Parallel Corpora Collection for 11 Indic Languages +- **role:** TRAINING_DATA +- **contribution:** Provides the large-scale Indic-language text substrate used to pretrain the multilingual encoder. +- **rationale:** Recreating the model requires the actual large-scale Indic-language training text. But the operative pretraining text is structurally composite: it is dominated by corpus material created or assembled in the target paper, while also incorporating prior resources such as IndicCorp lineage and Samanantar-derived text. Because no single prior study cleanly represents that full training substrate, `NONE` is the correct canonical choice, with prior contributing resources listed as additional studies. +- **evidence_span:** “we merge data from IndicCorp v2 with Indic language data from Wikipedia and OSCAR” + + +-------------------------------------------------- +REFERENCE EXAMPLE 4 +-------------------------------------------------- + +# EXAMPLE: MassiveDS datastore scaling resource + +## 1. Cluster Evidence + +**The MassiveDS datastore resource** is used in the literature + +The paper also presents a pipeline reordering trick that makes datastore scaling computationally feasible. However, the downstream evidence does **not** show that this pipeline is reused as a separate artifact on its own. Therefore, the correct annotation is **one resource claim**, with the pipeline-reordering idea represented as an important ingredient rather than a separate claim. + +--- + +## 2. Claim Split Decision + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Resource +- **rewritten_claim:** **Dataset: A large-scale multi-domain datastore and experimental pipeline, enabling systematic study and evaluation of retrieval-based language model scaling.** +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C3 + +### Why this split is correct + +The downstream reuse is centered on **one released resource**: the MassiveDS datastore. + +- Later work uses the **datastore itself** as the reusable artifact. +- The paper also introduces a pipeline-reordering idea, but the evidence does **not** show that this pipeline is being reused downstream as a separate artifact. +- Because claim splitting is driven by **distinct downstream-used contributions**, the correct annotation is to keep **one claim**. + +The pipeline reordering is still important, but its role is different: + +- it is a **structural ingredient** that makes the datastore resource computationally feasible in its claimed form +- it is **not** a separate discovery claim unless downstream evidence shows reuse of the pipeline itself + +--- + +## 3. Rewritten Claims + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Resource +- **rewritten_claim:** **Resource: A massive open multi-domain datastore that combines general web and domain-specific text, enabling large-scale retrieval-in-context language modeling and datastore-scaling studies.** +- **why_this_is_atomic:** This claim isolates the released datastore resource that downstream work reuses. It does not split off the pipeline-reordering idea as a separate claim because that idea is not independently substantiated as a downstream-used artifact. +- **optional cluster_id:** C3 +- **decision:** YES_SUFFICIENT + +--- + +## 4. Minimal Ingredient Set Per Claim + +## Claim C1: MassiveDS datastore resource + +### Ingredient C1.I1 +- **ingredient_id:** C1.I1 +- **ingredient:** Multi-source open text corpora spanning general web and domain-specific sources +- **why_structurally_necessary:** The datastore is defined by both its scale and its domain diversity. Without large open corpora spanning general web and specialized domains, the released resource would not exist in its claimed form. +- **why_not_lower_level_substeps:** This should remain one composite data-source ingredient rather than being split into separate final ingredients for web data, books, scientific papers, math, biomedical text, code, and other source families. +- **why_not_adjacent_implementation_details:** The structural dependency is the multi-source raw text substrate itself, not every individual source-selection or curation decision. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** RedPajama: An Open Source Recipe to Reproduce LLaMA Training Dataset; PubMed Baseline Repository; S2ORC: The Semantic Scholar Open Research Corpus; peS2o (Pretraining Efficiently on S2ORC) Dataset; NaturalProofs: Mathematical Theorem Proving in Natural Language; Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer +- **role:** DATA_SOURCE +- **contribution:** Supplies the large-scale open text sources used to assemble the datastore across both general web and specialized domains. +- **rationale:** Recreating MassiveDS requires the raw text substrate that gives it both scale and domain coverage. That substrate is inherently composite: the datastore mixes general web data with multiple distinct domain-specific resources. No single prior study faithfully represents that entire source mixture, so `NONE` is the correct canonical grounding, with the main contributing source resources listed as additional studies. +- **evidence_span:** “comprising 1.4 trillion tokens of both general web data and domain specific data” + +### Ingredient C1.I2 +- **ingredient_id:** C1.I2 +- **ingredient:** Pretrained dense retriever for document embedding +- **why_structurally_necessary:** The released datastore is not just raw text; it is an operational retrieval datastore with dense-vector retrieval over passages. Without a pretrained retriever to embed documents, it would not function in its claimed form. +- **why_not_lower_level_substeps:** This should remain one retriever ingredient rather than being split into encoder architecture, embedding dimensionality, or indexing internals. +- **why_not_adjacent_implementation_details:** Retriever hyperparameters and ablation choices are implementation details. The structural dependency is the pretrained dense retriever used to embed the datastore. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Unsupervised Dense Information Retrieval with Contrastive Learning +- **additional groundings if any:** none +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the pretrained retriever used to represent datastore documents as dense vectors for retrieval. +- **rationale:** To reproduce the datastore as a usable retrieval resource, the corpus must be embedded into a searchable dense space. The paper explicitly uses Contriever-MSMARCO for this purpose, making Contriever the cleanest canonical grounding. +- **evidence_span:** “we use CONTRIEVER-MSMARCO (Izacard et al., 2022), which represents every document in the datastore as a dense vector” + +### Ingredient C1.I3 +- **ingredient_id:** C1.I3 +- **ingredient:** Retrieval-in-context language modeling framework +- **why_structurally_necessary:** MassiveDS is introduced specifically to support retrieve-in-context language models rather than as a generic corpus release. Without this framing, the datastore would lose the main capability it is meant to enable. +- **why_not_lower_level_substeps:** This should remain one framework-level ingredient rather than being split into retrieval prompting, concatenation order, or model-specific usage details. +- **why_not_adjacent_implementation_details:** The structural dependency is the inference-time retrieval framework itself, not one particular LM or evaluation setup. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** In-Context Retrieval-Augmented Language Models +- **additional groundings if any:** none +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Defines the retrieval-in-context framing in which documents are retrieved from an external datastore and prepended to LM inputs at inference time. +- **rationale:** The datastore is meaningful here because it supports retrieval-based language models that use retrieved documents in context. Ram et al. is the cleanest canonical grounding because it directly represents the retrieve-in-context setup that the paper focuses on. +- **evidence_span:** “We focus on retrieve-in-context language models (RIC-LMs)” + +### Ingredient C1.I4 +- **ingredient_id:** C1.I4 +- **ingredient:** Data filtering workflow for deduplication, decontamination, and quality control +- **why_structurally_necessary:** The released resource is not merely a raw trillion-token dump; it is a constructed datastore whose quality depends on filtering, deduplication, and decontamination steps. Without this workflow, the datastore would be substantially weaker as a reusable retrieval resource. +- **why_not_lower_level_substeps:** This should remain one filtering-workflow ingredient rather than being split into deduplication, decontamination, and quality filtering as separate ingredients. +- **why_not_adjacent_implementation_details:** Individual thresholds, heuristics, or local filtering choices are not the structural dependency; the key dependency is the broader filtering workflow that makes the datastore usable. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Dolma: an Open Corpus of Three Trillion Tokens for Language Model Pretraining Research +- **additional groundings if any:** none +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the filtering workflow used to clean the datastore through deduplication, decontamination, and quality controls. +- **rationale:** Building a reusable datastore at this scale requires systematic filtering so the resulting resource is not just large, but also usable and reasonably clean. The paper explicitly ties this step to Dolma, making it the best canonical grounding. +- **evidence_span:** “data filtering, including deduplication, decontamination, and quality filters” + +### Ingredient C1.I5 +- **ingredient_id:** C1.I5 +- **ingredient:** Pipeline reordering to amortize expensive indexing and retrieval across datastore variants +- **why_structurally_necessary:** The datastore is released together with an experimental pipeline that makes large-scale datastore scaling feasible on modest compute. Without this reordering idea, the resource would lose an important defining property: practical support for scalable datastore studies. +- **why_not_lower_level_substeps:** This should remain one pipeline-level ingredient rather than being split into retrieval-overfetching, post-hoc subsampling, and late-stage filtering as separate ingredients. +- **why_not_adjacent_implementation_details:** The structural dependency is the reordering principle itself—running expensive steps once and sharing them—not the local code or systems details used to implement it. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the core pipeline idea that makes datastore scaling computationally feasible by sharing expensive indexing and retrieval across many variants. +- **rationale:** A key part of the released contribution is not only the datastore’s scale, but the fact that the paper makes large-scale scaling studies feasible by reordering operations so indexing and retrieval are amortized. This idea is introduced in the target paper rather than inherited from a single prior study, so `NONE` is the correct grounding. +- **evidence_span:** “the most expensive ones—indexing and retrieval—are run only once at the start” + +--- + +## 5. Excluded Tempting Non-Ingredients + +### Claim C1 exclusions +- **Pipeline reordering as a separate second claim** + - Excluded because the downstream reuse evidence centers on the datastore resource, not on the pipeline as an independently reused artifact. The pipeline is important, but it is best represented as a structural ingredient of the single datastore claim. + +- **Passage chunking and indexing into fixed-length retrieval units** + - Excluded because this is a lower-level construction step rather than one of the smallest higher-level structural ingredients. It is subordinate to the broader retriever-and-datastore construction pipeline. + +- **FAISS / exact top-K retrieval implementation** + - Excluded because this is too implementation-specific. It supports the pipeline but is not needed at the abstraction level of the minimal structurally sufficient ingredient set. + +- **Separate final ingredients for general web data and domain-specific data** + - Excluded because the higher-level composite ingredient is the multi-source open text substrate spanning both. Splitting them would over-decompose the datastore’s source basis. + +- **Contriever as MODEL_INITIALIZATION instead of IMPLEMENTATION_TOOLING** + - Excluded because its role here is not initializing the target discovery as a model checkpoint, but operationally embedding the datastore documents for retrieval. + +--- + +## 6. Candidate Grounding Decisions + +## Claim C1: MassiveDS datastore resource + +### Ingredient C1.I1 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The datastore’s source substrate is structurally composite across general web and many domain-specific resources. +- **Candidate:** RedPajama: An Open Source Recipe to Reproduce LLaMA Training Dataset +- **decision:** accepted_additional +- **why:** Representative open source for books and other large-scale text components, but not sufficient alone to represent the whole source mixture. +- **Candidate:** PubMed Baseline Repository +- **decision:** accepted_additional +- **why:** Important biomedical source contributing to the domain-specific side of the datastore. +- **Candidate:** S2ORC: The Semantic Scholar Open Research Corpus +- **decision:** accepted_additional +- **why:** Important scientific-paper source for the domain-specific mixture. +- **Candidate:** peS2o (Pretraining Efficiently on S2ORC) Dataset +- **decision:** accepted_additional +- **why:** Additional scientific corpus contribution. +- **Candidate:** NaturalProofs: Mathematical Theorem Proving in Natural Language +- **decision:** accepted_additional +- **why:** Representative mathematical-language source for the domain-specific mixture. +- **Candidate:** Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer +- **decision:** accepted_additional +- **why:** Relevant additional grounding because C4 is explicitly cited as one of the general web data sources. + +### Ingredient C1.I2 +- **Candidate:** Unsupervised Dense Information Retrieval with Contrastive Learning +- **decision:** accepted_canonical +- **why:** Cleanest representative prior study for Contriever, the pretrained dense retriever actually used to embed datastore passages. + +### Ingredient C1.I3 +- **Candidate:** In-Context Retrieval-Augmented Language Models +- **decision:** accepted_canonical +- **why:** Cleanest representative framing for the retrieval-in-context LM setup the datastore is designed to support. + +### Ingredient C1.I4 +- **Candidate:** Dolma: an Open Corpus of Three Trillion Tokens for Language Model Pretraining Research +- **decision:** accepted_canonical +- **why:** Explicitly cited by the paper for the filtering stage and is the cleanest representative for the deduplication / decontamination / quality-control workflow. + +### Ingredient C1.I5 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The pipeline-reordering idea is introduced in the target paper itself and is not cleanly attributable to one prior study. + +--- + +## 7. Final Grounding Decisions + +## Claim C1 +- **C1.I1** → NONE → DATA_SOURCE + - additional: RedPajama: An Open Source Recipe to Reproduce LLaMA Training Dataset; PubMed Baseline Repository; S2ORC: The Semantic Scholar Open Research Corpus; peS2o (Pretraining Efficiently on S2ORC) Dataset; NaturalProofs: Mathematical Theorem Proving in Natural Language; Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer +- **C1.I2** → Unsupervised Dense Information Retrieval with Contrastive Learning → IMPLEMENTATION_TOOLING +- **C1.I3** → In-Context Retrieval-Augmented Language Models → CONCEPTUAL_FRAMEWORK +- **C1.I4** → Dolma: an Open Corpus of Three Trillion Tokens for Language Model Pretraining Research → IMPLEMENTATION_TOOLING +- **C1.I5** → NONE → CONCEPTUAL_FRAMEWORK + +--- + +## 8. Final Rationales And Evidence + +## Claim C1: MassiveDS datastore resource + +### C1.I1 +- **ingredient:** Multi-source open text corpora spanning general web and domain-specific sources +- **canonical study:** NONE +- **additional studies:** RedPajama: An Open Source Recipe to Reproduce LLaMA Training Dataset; PubMed Baseline Repository; S2ORC: The Semantic Scholar Open Research Corpus; peS2o (Pretraining Efficiently on S2ORC) Dataset; NaturalProofs: Mathematical Theorem Proving in Natural Language; Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer +- **role:** DATA_SOURCE +- **contribution:** Supplies the large-scale open text sources used to assemble the datastore across both general web and specialized domains. +- **rationale:** Recreating MassiveDS requires the raw text substrate that gives it both scale and domain coverage. That substrate is inherently composite: the datastore mixes general web data with multiple distinct domain-specific resources. No single prior study faithfully represents that entire source mixture, so `NONE` is the correct canonical grounding, with the main contributing source resources listed as additional studies. +- **evidence_span:** “comprising 1.4 trillion tokens of both general web data and domain specific data” + +### C1.I2 +- **ingredient:** Pretrained dense retriever for document embedding +- **canonical study:** Unsupervised Dense Information Retrieval with Contrastive Learning +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the pretrained retriever used to represent datastore documents as dense vectors for retrieval. +- **rationale:** To reproduce the datastore as a usable retrieval resource, the corpus must be embedded into a searchable dense space. The paper explicitly uses Contriever-MSMARCO for this purpose, making Contriever the cleanest canonical grounding. +- **evidence_span:** “we use CONTRIEVER-MSMARCO (Izacard et al., 2022), which represents every document in the datastore as a dense vector” + +### C1.I3 +- **ingredient:** Retrieval-in-context language modeling framework +- **canonical study:** In-Context Retrieval-Augmented Language Models +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Defines the retrieval-in-context framing in which documents are retrieved from an external datastore and prepended to LM inputs at inference time. +- **rationale:** The datastore is meaningful here because it supports retrieval-based language models that use retrieved documents in context. Ram et al. is the cleanest canonical grounding because it directly represents the retrieve-in-context setup that the paper focuses on. +- **evidence_span:** “We focus on retrieve-in-context language models (RIC-LMs)” + +### C1.I4 +- **ingredient:** Data filtering workflow for deduplication, decontamination, and quality control +- **canonical study:** Dolma: an Open Corpus of Three Trillion Tokens for Language Model Pretraining Research +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the filtering workflow used to clean the datastore through deduplication, decontamination, and quality controls. +- **rationale:** Building a reusable datastore at this scale requires systematic filtering so the resulting resource is not just large, but also usable and reasonably clean. The paper explicitly ties this step to Dolma, making it the best canonical grounding. +- **evidence_span:** “data filtering, including deduplication, decontamination, and quality filters” + +### C1.I5 +- **ingredient:** Pipeline reordering to amortize expensive indexing and retrieval across datastore variants +- **canonical study:** NONE +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the core pipeline idea that makes datastore scaling computationally feasible by sharing expensive indexing and retrieval across many variants. +- **rationale:** A key part of the released contribution is not only the datastore’s scale, but the fact that the paper makes large-scale scaling studies feasible by reordering operations so indexing and retrieval are amortized. This idea is introduced in the target paper rather than inherited from a single prior study, so `NONE` is the correct grounding. +- **evidence_span:** “the most expensive ones—indexing and retrieval—are run only once at the start” + + +-------------------------------------------------- +TARGET PAPER CONTEXT +-------------------------------------------------- +Target paper context: + +TITLE: PASTA: A Dataset for Modeling PArticipant STAtes in Narratives +YEAR: 2022 +VENUE: Transactions of the Association for Computational Linguistics +PAPER_ID: + +REFINED DOWNSTREAM CLUSTER EVIDENCE: +[ + { + "cluster_id": "C1", + "representative_claim": "Uses target methodology for dataset creation", + "cluster_title": "Uses target methodology for dataset creation", + "count": "1", + "merge_rationale": "The citing paper adopts the methodology used in the target paper to generate its own dataset." + }, + { + "cluster_id": "C2", + "representative_claim": "Uses target dataset for data generation", + "cluster_title": "Uses target dataset for data generation", + "count": "1", + "merge_rationale": "The citing paper uses the PASTA dataset and its annotations directly to generate and validate story outcomes." + } +] + +TARGET PAPER CONTEXT: +[ + { + "cluster_id": "C1", + "representative_claim": "Uses target methodology for dataset creation", + "cluster_title": "Uses target methodology for dataset creation", + "count": "1", + "merge_rationale": "The citing paper adopts the methodology used in the target paper to generate its own dataset." + }, + { + "cluster_id": "C2", + "representative_claim": "Uses target dataset for data generation", + "cluster_title": "Uses target dataset for data generation", + "count": "1", + "merge_rationale": "The citing paper uses the PASTA dataset and its annotations directly to generate and validate story outcomes." + } +] + +FULL PAPER CONTENT: +% File tacl2021v1.tex +% Dec. 15, 2021 + +% The English content of this file was modified from various *ACL instructions +% by Lillian Lee and Kristina Toutanova +% +% LaTeXery is mostly all adapted from acl2018.sty. + +\documentclass[11pt,a4paper]{article} +\usepackage{times,latexsym} +\usepackage{url} +\usepackage[T1]{fontenc} + +%% Package options: +%% Short version: "hyperref" and "submission" are the defaults. +%% More verbose version: +%% Most compact command to produce a submission version with hyperref enabled +% \usepackage[]{tacl2021v1} +%% Most compact command to produce a "camera-ready" version +\usepackage[acceptedWithA]{tacl2021v1} +%% Most compact command to produce a double-spaced copy-editor's version +% \usepackage[acceptedWithA,copyedit]{tacl2021v1} +% +%% If you need to disable hyperref in any of the above settings (see Section +%% "LaTeX files") in the TACL instructions), add ",nohyperref" in the square +%% brackets. (The comma is a delimiter in case there are multiple options specified.) + +\usepackage{tacl2021v1} +% \setlength\titlebox{10cm} % <- for Option 2 below + +%%%% Material in this block is specific to generating TACL instructions +\usepackage{xspace,mfirstuc,tabulary} + +%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%5 +% Packages imported by me +%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%5 +\usepackage{graphicx} +\usepackage{tabularx} +\usepackage{soul} +\usepackage{epstopdf} +\usepackage{todonotes} +\usepackage[utf8]{inputenc} +\usepackage{svg} +\usepackage{hyperref} +\usepackage{xstring} +\usepackage{color} +\usepackage{booktabs} +\usepackage{pifont} +\usepackage{amsmath} +\usepackage{multirow} +\usepackage{xspace} +\usepackage{subcaption} +\usepackage{epsfig, endnotes} +\usepackage{array} +\usepackage{float} +\usepackage{mfirstuc, tabulary} +\usepackage{paralist} +\usepackage{csquotes} +\usepackage{enumitem} +\usepackage{wasysym} +\usepackage{amssymb} +\usepackage{pgf-pie} +\usepackage{hyperref} +\usepackage{nameref} +\usepackage{makecell} +\usepackage{cuted} +\usepackage{xcolor} +\usepackage{enumitem} +\usepackage{subcaption,siunitx,booktabs} +%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +% x-x-x-Packages imported by me-x-x-x +%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%5%% + + +\newcommand{\dateOfLastUpdate}{Dec. 15, 2021} +\newcommand{\styleFileVersion}{tacl2021v1} + +\newcommand{\ex}[1]{{\sf #1}} +%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +% inline cmd +%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +\newcommand{\dataset}[0]{\texttt{PASTA}} +\newcommand{\ttbf}[1] { + \textbf{\texttt{#1}}} +\newcommand{\storyrev}[0]{Story Revision for Counterfactual States}%{task$-\mathcal{W}$}} +\newcommand{\statechange}[0]{State Change Generation}%{task$-\mathcal{X}$}} +\newcommand{\mcq}[0]{Story State Inference}%{task$-\mathcal{Y}$}} +\newcommand{\nb}[1]{\textcolor{blue}{NB:[#1]}} +\newcommand{\sg}[1]{\textcolor{red}{SG:[#1]}} +\newcommand{\nc}[1]{\textcolor{orange}{NC:[#1]}} +\newcommand{\ff}[1]{\todo[color=purple!40]{[#1]--FF}} +\newcommand{\ffinline}[1]{\textcolor{purple}{FF:[#1]}} + + +\newif\iftaclinstructions +\taclinstructionsfalse % AUTHORS: do NOT set this to true +\iftaclinstructions +\renewcommand{\confidential}{} +\renewcommand{\anonsubtext}{(No author info supplied here, for consistency with +TACL-submission anonymization requirements)} +\newcommand{\instr} +\fi + +% +\iftaclpubformat % this "if" is set by the choice of options +\newcommand{\taclpaper}{final version\xspace} +\newcommand{\taclpapers}{final versions\xspace} +\newcommand{\Taclpaper}{Final version\xspace} +\newcommand{\Taclpapers}{Final versions\xspace} +\newcommand{\TaclPapers}{Final Versions\xspace} +\else +\newcommand{\taclpaper}{submission\xspace} +\newcommand{\taclpapers}{{\taclpaper}s\xspace} +\newcommand{\Taclpaper}{Submission\xspace} +\newcommand{\Taclpapers}{{\Taclpaper}s\xspace} +\newcommand{\TaclPapers}{Submissions\xspace} +\fi + +%%%% End TACL-instructions-specific macro block +%%%% + +\title{\dataset{}: A Dataset for Modeling {PA}rticipant {STA}tes in Narratives} + +% Author information does not appear in the pdf unless the "acceptedWithA" option is given + +% The author block may be formatted in one of two ways: + +% Option 1. Author’s address is underneath each name, centered. + +% \author{ +% Template Author1\Thanks{The {\em actual} contributors to this instruction +% document and corresponding template file are given in Section +% \ref{sec:contributors}.} +% \\ +% Template Affiliation1/Address Line 1 +% \\ +% Template Affiliation1/Address Line 2 +% \\ +% Template Affiliation1/Address Line 2 +% \\ +% \texttt{template.email1example.com} +% \And +% Template Author2 +% \\ +% Template Affiliation2/Address Line 1 +% \\ +% Template Affiliation2/Address Line 2 +% \\ +% Template Affiliation2/Address Line 2 +% \\ +% \texttt{template.email2@example.com} +% } + +% % Option 2. Author’s address is linked with superscript +% % characters to its name, author names are grouped, centered. +\author{ + Sayontan Ghosh$^\diamondsuit$\hspace{1.5em} + Mahnaz Koupaee$^\diamondsuit$\hspace{1.5em} + Isabella Chen$^{\diamondsuit\dagger}$ \\ + \textbf{Francis Ferraro}$^\spadesuit$\hspace{1.5em} + \textbf{Nathanael Chambers}$^\clubsuit$\hspace{1.5em} + \textbf{Niranjan Balasubramanian}$^\diamondsuit$ + \\ \ \\ + {$^\diamondsuit$Stony Brook University}\hspace{1em} + $^\spadesuit$University of Maryland, Baltimore County\hspace{1em} + $^\clubsuit$United States Naval Academy \hspace{1em} + \\ + \texttt{\{sagghosh, mkoupaee, niranjan\}@cs.stonybrook.edu} + \\ \texttt{isabellachenusa@gmail.com}\hspace{1em} + \texttt{ferraro@umbc.edu}\hspace{1em} + \texttt{nchamber@usna.edu} +} + +\date{} + +\begin{document} +\maketitle +%\section{Abstract} +\begin{abstract} +%\ffinline{Red text is previous abstract, blue text is my recommended abstract} +%{\color{red} +%Understanding a narrative requires inferences about the underlying states of its participants. +%These states act as the (often unstated) semantic glue connecting the described events into a coherent narrative. +%Humans often use commonsense knowledge to infer which participant states are warranted, but also to reason about subtle counterfactual changes to the states. +%While large language models have demonstrated impressive commonsense reasoning capabilities, it is unclear if they can identify these underlying participant states. +%\ff{Given space is limited, I don't think we need this orange portion in the abstract. We could start the abstract w/ ``We argue that understanding the events in a narrative depends upon understanding the underlying states of its participants; however, those states may not be explicitly mentioned in the narrative. To address this, ...} +The events in a narrative are understood as a coherent whole via the underlying states of their participants. Often, these participant states are not explicitly mentioned, instead left to be inferred by the reader. +A model that understands narratives should likewise infer these implicit states, and even reason about the impact of changes to these states on the narrative. +To facilitate this goal, we introduce a new crowdsourced English-language, \emph{Pa}rticipant \emph{Sta}tes dataset, \dataset{}. This dataset contains inferable participant states; a counterfactual perturbation to each state; and the changes to the story that would be necessary if the counterfactual were true. We introduce three state-based reasoning tasks that test for the ability to infer when a state is entailed by a story, to revise a story conditioned on a counterfactual state, and to explain the most likely state change given a revised story. Experiments show that today's LLMs can reason about states to some degree, but there is large room for improvement, especially in problems requiring access and ability to reason with diverse types of knowledge (e.g. physical, numerical, factual).\footnote{Code and dataset are available at \url{https://github.com/StonyBrookNLP/pasta} \par $^\dagger$Work done during internship at Stony Brook University.} +%an important part of understanding how events form a coherent narrative. We introduce a new crowdsourced \emph{Pa}rticipants \emph{Sta}tes dataset, \dataset, + +%that associates valid participant states that humans can infer from each story, but also provides a counterfactual perturbation to the state with the changes to the story that would be necessary if the counterfactual was true. +%We introduce three state-based reasoning tasks that test for ability to infer when a state is entailed by a story, revise a story for a counterfactual state, and to explain the most likely state change given a revised story. Our benchmarking experiments show that while today's LLMs are able to reason about states to some degree, there is a large room for improvement suggesting potential avenues for future research. +%} + +%^\nb{Not sure I want this to be focused primarily on counterfactual reasoning. This is reasoning about states broadly, one aspect of which is counterfactual aspect.} +% {\color{blue} +%We examine counterfactual reasoning about implicit participant states in short narrative stories, which can act as the semantic glue connecting the described events into a coherent narrative. % +%We introduce a new crowdsourced \emph{Pa}rticipants \emph{Sta}tes dataset, \dataset. This dataset contains valid, inferable participant states; a counterfactual perturbation to the state; and the changes to the story that would be necessary if the counterfactual was true. % +%We introduce three state-based reasoning tasks that test for the ability to infer when a state is entailed by a story, revise a story for a counterfactual state, and to explain the most likely state change given a revised story. Our benchmarking experiments show that while today's LLMs are able to reason about states to some degree, there is a large room for improvement suggesting potential avenues for future research. +%} + +\end{abstract} +%%%%% Sayontan's abstract %%%%%%%%%%%% +% \nb{Sayontan's old abstract} +% The ability to reason about participants' states in a narrative text not only requires models to understand their cause-effect relationship with the events mentioned in the text but also do counter-factual reasoning about which states and events are unlikely in the context of the text and how they might alter the discourse of the text. +% In order to provide a test-bed for testing this knowledge in NLP models, we introduce a $\mathcal{STATE}$ dataset, which is a commonsense dataset that unifies the elements of abductive, deductive, and counterfactual reasoning about participants (or objects) state in a narrative text. + +% $\mathcal{STATE}$ dataset is a crowd-sourced dataset that is created by asking the workers to perform a sequence of annotations on a simple story, each of which requires a different kind of commonsense knowledge about the participant state in the story. +% This data consists of 960 (8480 training and 920 test instances) 4-tuples of story, an inferred participant state, a perturbed state that is unlikely te inferred from the story, and a minimally revised story that is consistent with the perturbed state. +% In order to leverage this dataset to test the current LLMs for state-based common-sense knowledge, we propose three tasks, task-$\mathcal{X}$, task-$\mathcal{Y}$ and task-$\mathcal{Z}$. + +% We benchmark the current state of art LLMs on these tasks, with finetuning and zero-shot approach, using automatic and human evaluation. +% Through the experiments, we show that even though these LLMs do have the state-based common-sense knowledge to a certain degree, finetuning the models on this dataset makes them better equipped to understand and reason about participants' states in a narrative text. + +%% WARNING: skipped circular input /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785099279_8df5ecbb/processed_papers/2208.00329/tex_2208.00329/1_abstract.tex + +\section{Introduction} +% \begin{figure}[!tb] +\begin{figure*}[!ht] + % \fbox{\includesvg[inkscapelatex=false,width=\textwidth]{images/data_collection0.svg}} + % \includesvg[inkscapelatex=false,width=\textwidth]{images/data_collection0.svg} + \includegraphics[width=\textwidth]{images/data_collection0.png} + % + \caption{\small{For a given story $S$, \textbf{\dataset{}} provides an (unstated) inferred state $\alpha$, a minimal set of justification sentences (\hl{yellow} highlight), a counterfactual state $\alpha'$, and a revised story $S'$, such that $\alpha'$ can be inferred from it.} + } + \label{Fig:pasta_example} +\end{figure*} + +% =~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~ +% MOTIVATION +% =~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~ +Understanding narrative text requires forming a coherent representation of the scenario, including filling in details that are unstated in the text. One type of detail that is usually not mentioned are the states of its participants\footnote{We define participants to include both animate entities and inanimate objects in the narratives.} (e.g., "she unlocked the door" implies the possession state that "she has a key"). +% +%Despite this challenge, we humans seemingly easily infer the underlying participants' states and their \ff{I don't know if we want to say causal here... maybe enabling?}\nb{causal and enabling?}causal relationships with the narrative's events in order to create a detailed mental picture of the world that is only partially observable from the text. % +The reader easily infers these \emph{implicit} states and their causal relationships with the narrative's \emph{explicit} events, creating a detailed mental picture of the described world that is only partially observable from the text. +Many cognitive theories have been proposed to capture aspects of this in their representations, such as scripts~\cite{schank1975scripts}, frames~\cite{fillmore1985frames}, and state/time formalisms~\cite{galton1990critical}. +Without committing to any one particular formal theory, this paper adds a theory-agnostic resource +% ------------------------REVIEWER 1 Suggestion--------------------- +% I think the dataset is more valuable than just being designed to evaluate LLMs. I would motivate the creation of the dataset independently from the techniques that could/might be used on it and rather focus on conceptually what the dataset is capturing and why that will move the general field forward, regardless of how. +% ------------------------REVIEWER 1 Suggestion--------------------- +to test such theories by listing implicitly assumed participant states in simple narratives. +%is designed to evaluate whether language models can access the unstated states from simple narratives. + +Consider the story in Figure~\ref{Fig:pasta_example} from the ROCStories corpus \cite{mostafazadeh2016corpus}. +% +Humans create a detailed mental representation of this spilled-soda scenario by inferring its commonsense states. % +In this story, using our commonsense knowledge about emotions and habituals, we can infer from the first two lines that \underline{Kate's mother liked keeping her car clean} (a state about \emph{Kate's mother}). Similarly, based on our physical commonsense of a lid, i.e., that lids prevent spilling, we can also assert from the spill that \underline{the soda's lid was loose} (a state about the \emph{soda}). +% \nb{Add a line that says what is needed to do this task e.g. knowledge that lids prevent spilling etc.} +We can also reason about the likely change to the story due to a counterfactual state, i.e., if \underline{the soda's lid was tight}, then most likely the soda wouldn't spill. +% Further, if we are given a counterfactual state, \underline{the soda's lid was tight}, then we can reason about a likely change to the story where the \emph{soda} perhaps didn't spill. +% \nb{Add a line that says what is needed to do this task} +% \nb{end this paragraph by saying no such resource exists} +To the best of our knowledge, no such resource exists that captures this kind of participant state knowledge. +%Can language models make such state-based inferences? + +%Though inferring states is intuitively easy for humans, it's unclear if today's deep learners can make these inferences. + +%These state inferences can be as shallow as \underline{\emph{Kate's mother} liked keeping her car clean.} (from \textbf{L1, L2}) to \ff{Qualify this more. It is not a given that the soda didn't have lid. (Maybe it had a lid but the force from the sudden deceleration caused it to get dislodged, or thrown against the dashboard.)}as deep as \underline{The \emph{soda} had no lid} (\textbf{L4}). These kind of inferences not only depend on \ff{from a framing perspective, are we really talking about causality, numeracy, or temporality? I don't think so, so this is potentially a bad diversion that'll confuse reviews}understanding causality, numeracy, and temporal relations, but also require counterfactual reasoning about potential changes to the states to infer what is unlikely to be true in a situation (\ul{If the \emph{soda} had lid}, then it wouldn't have spilled over). +%Though inferring states is intuitively easy for humans, it's unclear if today's deep learners can make these inferences. + +%In this work, we describe a new challenge dataset, \dataset, to help benchmark system understanding of implicit states, counterfactuals, and by extension, story understanding. We also present benchmark systems on multiple \dataset\ tasks to {\color{red}investigate the ability of current LLMs to infer and reason about participant states in narrative text}\ff{I think this is a valid statement {\it if} (1) we're making the zero-shot element a big element of the story and (2) we can do some fairly comprehensive analysis of the dataset and each task. Sayontan, are both of those true?}. +% +% This common-sense knowledge can be viewed as encoding the probabilistic model or the statistics of the world \cite{griffiths2006optimal}, which is continually updated and enables us to perform complex inferences based on the partial observations (text, audio, video, etc.) of the world. +% % +% LLMs try to learn this model by acquiring the knowledge available in the text through self-supervised learning. +% Recent works have shown that current LLMs (large language models) encode this knowledge to a certain extent, but there is still a substantial gap between these models and human performance on most of the common-sense-based tasks. +% +% Many recent works such as \cite{Hwang2021COMETATOMIC2O, sap2019atomic, speer2017conceptnet, mostafazadeh-etal-2020-glucose} introduce resources, not only to study and test these models' ability to perform different kinds of common-sense reasoning but also to show that carefully curated resources can be used to improve model performance in this kind of tasks. +% +% =~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~ +% Summary about the PASTA +% =~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~ +To capture this type of commonsense knowledge needed to understand and reason about participant states in narratives, +% and to assess this ability in current models +we introduce \dataset, a crowd-sourced dataset in English. +% We thus introduce \dataset, a crowd-sourced dataset that captures the commonsense knowledge needed to understand short stories and their participant states, and in order to test current models' ability to do state-based reasoning. +%\sg{In order to identify the required commonsense knowledge for \textbf{\textit{Pa}}rticipant \textbf{\textit{Sta}}tes and to test LLMs' ability to do this state-based reasoning, we introduce a crowd-sourced dataset, \dataset{}.} +% We introduce a crowd-sourced dataset, \dataset{}. +% \nb{say why this was difficult to create or what interesting thing you had to do to create the resource.} +% +% Motivated by the goal to test today's LLMs' ability to identify and reason over states like in the above example and to provide a resource that can help augment their inference abilities, we introduce \dataset, a crowd-sourced dataset that captures \sg{different aspects of the } commonsense \sg{reasoning} about \textbf{\textit{Pa}}rticipant \textbf{\textit{Sta}}tes in narrative text. +% +As shown in Figure~\ref{Fig:pasta_example}, for a given story $S$, \dataset{} provides a participant state $\alpha$ that is likely to be inferred from $S$, a perturbation state $\alpha'$ that is counterfactual to $S$ along with the minimal changes to $S$ that are required to make $\alpha'$ likely to be inferred from the revised story $S'$. +\dataset{} includes $10,743$ instances of these story/state/counterfactual/revision tuples. +% \footnote{Per instructions from the EIC, a sample of 100 PASTA data points is provided to the original action editor. The full dataset will be released upon publication.}. +% +% =~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~ +% Tasks on PASTA +% =~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~ +With this new dataset, we hope to enable models to make the kinds of state-based inferences that move beyond surface text understanding and lead to deeper reasoning. +To this end, we describe three new state-based reasoning challenges with \dataset{} which are illustrated in Figure~\ref{Fig:stae_tasks}. + +% ----------------------- +% Task - 1 +% ----------------------- +% The first is a binary classification task, \textbf{\mcq}. Given a story and an inferred participant state, +% %along with a set of supporting sentences from the story, +% the task is to predict if the state is likely to be inferred from specific sentences in context of the story. +The first is \textbf{\mcq}: given a story and an inferred participant state, predict if the state is likely to be inferred from a given set of sentences in the context of the story. +We formulate this as a binary classification task, and we create contrastive examples for training and evaluation purposes to guard against artifact-based reasoning. % +This can be seen as a form of textual entailment, a capability useful for applications such as question answering~\cite{harabagiu-hickl-2006-methods, trivedi-etal-2019-repurposing}, claim verification~\cite{yin-roth-2018-twowingos, hanselowski-etal-2018-ukp}, etc. + + +% ----------------------- +The other two challenge tasks are generative. +% Task - 2 +% ----------------------- +The second task, \textbf{\storyrev{}}, measures the ability to reason about counterfactuals. Given a story and a counterfactual state (i.e., a state that is not consistent with the story), the task is to revise the story such that the counterfactual state is now likely to be inferred from it. These types of counterfactual revisions serve as a test of reasoning~\cite{Qin2019CounterfactualSR} and can support interactive story generation tasks~\cite{goldfarb-tarrant-etal-2019-plan, brahman-etal-2020-cue}. +% +%starts with a story and a counterfactual state, and a system must generate a new revised story such that the counterfactual state can now be inferred from it. +%\sg{This conditional story revision task can also be viewed as an extension of sentence-level style transfer, common in applications such as sentiment transformation~\cite{xu-etal-2018-unpaired, yang-2022-mask}, text detoxification~\cite{ tran-etal-2020-towards, dale-etal-2021-text}, etc., to story-level transfer.} +% ----------------------- +% Task - 3 +% ----------------------- +The third task, \textbf{\statechange{}}, requires the model to take a story and its perturbed version as input and then generate the two corresponding states (e.g., `lid was loose' and `lid was tight') that explain the differences in the way they unfold. +% Similar to the previous, through this task we want to probe models' understanding of participant state changes that led to the revision of one narrative into another. +From an application perspective, generating the underlying states that account for the differences between two narratives can assist with fake news detection using reliable sources \cite{figueira2017current, cardoso2019can, ghadiri2022automated} and information fact checking \cite{brandtzaeg2018journalists}. + +%\nb{Write a line or two that says these are difficult tasks and list the challenges/knowledge/phenomena that need to be modeled in order to do these tasks. SG - 'I have taken the challenges from the error analysis as it also talks about the same stuff.'} +% +These three challenge tasks require a unique combination of commonsense abilities, thus helping to evaluate models on reasoning and knowledge capacity. These tasks require not only basic entailment ability, but also knowledge (numerical, factual, physical, etc.) and broader narrative understanding. Having just one of these abilities will not suffice. +%\sg{One of the important necessities of these tasks is to combine different types of commonsense knowledge, such as numerical, factual, physical, etc. \storyrev{} additionally requires the model to identify relevant parts of the input story that are inconsistent with the counterfactual state and generate logically coherent text. \statechange{} requires models to identify the relevant states in both stories out of numerous plausible states, that describe the state change between them.} +% =~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~ +% Tasks on PASTA +% =~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~ +%To systematically analyze such models for these capabilities, we propose \statechange{}, \storyrev{} and \mcq{} tasks, each evaluating the model for a different capability that essential to understand participants state in a narrative. +% +% \nb{This paragraph needs to be updated to reflect the new framing.} +To evaluate current models for these capabilities, we benchmark the LLMs T5~\cite{raffel2020exploring}, BERT~\cite{devlin-etal-2019-bert}, RoBERTa~\cite{Liu2019RoBERTaAR} and GPT3~\cite{brown2020language}. For the generative tasks, we evaluate model performance through extensive human and automatic evaluations. +The results show that, though these models can reason about states to some degree, there is substantial room for improvement on all tasks, suggesting avenues for future research. + +\section{Related Work} +% +% Old work on common-sense reasoning +There are many formal theories on mental states and reasoning. The seminal work by \citet{schank1975scripts} introduced scripts as a way to structure knowledge about stereotypical event sequences with their participants. Frames \cite{fillmore1985frames} and theories of time \cite{galton1990critical} provide related views. This paper does not commit to a formal theory, instead providing a challenge dataset to test aspects of them. +%Manual curation of such knowledge was not scalable. +%been a plethora of work on curating commonsense knowledge about the world and formulating tasks that test this knowledge. +Statistical work on events \cite{chambers2008unsupervised, chambers2009unsupervised, balasubramanian2013generating,ferraro-2016-upf,sha2016joint} curated event knowledge in an unsupervised manner from large text corpora. This paper augments their view with \emph{state-based} knowledge about event participants. +% +%but \ff{I think this conclusion, in particular ``this knowledge,'' needs to be rephrased}this knowledge was not much useful for implicit commonsense inferences~\cite{gordon2013reporting}. +\par +More recent work by \citet{speer2017conceptnet}, \citet{sap2019atomic} and \citet{Hwang2021COMETATOMIC2O} capture everyday inferential knowledge associated with an action performed by someone. This knowledge is organized through a fixed set of relationship classes between action and inferences. A subset of these classes is about participant mental states, but this commonsense knowledge is non-contextual in nature. In contrast, our work requires inferring commonsense knowledge about participant states in the context of coherent narratives. +\par +% Comparing with conterfactual +Most similar to our work is the {TIME-TRAVEL} dataset by \citet{Qin2019CounterfactualSR}. It includes the Counterfactual Story Rewriting task to edit a short story based on a counterfactual context. +They insert an explicit counterfactual at a fixed position ($2^{nd}$ sentence) in the story, and the revision task is then conditioned on this observed change. It is a language modeling generation task. +In contrast, our work introduces \emph{unobserved} counterfactual outside of the story's text, and the revised story must be generated with deeper state-based reasoning. +This introduces additional complexity for the revision task. +%, as the original state that was perturbed to generate the counterfactual state is unobserved for the task and would need to be reasoned for by the models performing the task. +Also, {TIME-TRAVEL} requires the revisions to be restricted to the story ending, which cannot be assumed in our setting. Our states can be inferred from any part of the story. +\par +% Comparing with abductive +\citet{bhagavatula2019abductive} proposed tasks that predict a plausible hypothesis for two given observations, and curate a dataset for the same. Their work mainly focuses on \textit{what happened in-between?} type of inferences. +\citet{mostafazadeh-etal-2020-glucose} introduced the {GLUCOSE} dataset, which focuses on several types of causal knowledge that is required to explain a causal event in narrative text. +Neither of these focuses entirely on implicit states (some {GLUCOSE} annotations are relevant, but not directly so), and neither addresses story revision in the face of counterfactual changes.% In contrast, our work focuses on state-based commonsense knowledge for narrative text, required to understand and reason about what unobserved participant states can (and cannot) be inferred from a narrative, and how they affect and are affected by, the discourse of a narrative. +\par +% Related work specifically for entity state in procedural text +Recent work on understanding entity states has mostly focused on tracking entity state change in text. +\citet{Dalvi2018TrackingSC} introduced {PROPARA} which captures physical state changes (creation, destruction, and movement), \citet{Bosselut2018SimulatingAD} proposed the task of tracking ingredients in cooking recipes, and \citet{rashkin-etal-2018-modeling} tracks the emotional reactions and motivations of characters in simple stories, for a fixed/small set of attributes. +\par +\citet{tandon-etal-2019-wiqa} introduced the {WIQA} dataset for analyzing the effect of perturbing a process described by a procedural text on the elements (entities, events, etc.) of the text, as an influence graph of the process. However, the influence graph was assumed to have a fixed causal structure. It captured a very limited set of cause-effect relationships obtained as a result of analyzing perturbations that either accelerated or decelerated the main outcome of the process. +% perturbation to a property or phenomenon on the entities and events mentioned in a procedural text, as an influence graph (of a predefined causal structure) of the procedure. However the graph only captures a limited set of cause-effect relationships amongst entities, +% However due to the structural assumptions behind the influence graph, it cannot be used to generate a complete new process for a given perturbation. \nc{Shorten and work on English syntax} +\citet{tandon-etal-2020-dataset} introduced a dataset for tracking state changes in procedural text as a set of state change tuples of \textit{entity, attribute, before-state}, and \textit{after-state} for each step of the process. The elements of the tuples were in free-form text instead of belonging to a set of pre-defined categories. +% \nc{this sentence is unclear}. +\par +Our work differs from the above in several key ways: (i) participant states are unstated (ii) participant state inferences do not depend on sentence ordering assumptions, (iii) state perturbations affect the entire discourse of the narrative, and (iv) captures how the participant states change between an original and a perturbed narrative. + +\section{\dataset{}: {PA}rticipant {STA}tes} + +\dataset\ is a dataset of story pairs ($S$, $S'$) where each story $S$ has a revised version of itself, $S'$, that hinges on a particular state that was changed in its revision. The story pairs thus have corresponding state pairs ($\alpha$, $\alpha'$), containing an original state $\alpha$ and its counterfactual $\alpha'$. Refer to Figure~\ref{Fig:pasta_example}. These story/state pairs allow us to analyze unique narrative challenges. We can test if a model can identify whether a given participant state is consistent with it. We can ask what would happen if an assumed story state is no longer true. We can also ask if a model can identify what/how a state changes between two similar but different stories. +% +This section describes \dataset{}, the crowd-sourcing process that created it, quality control details, and its basic statistics. +% +% ===================================== +% Data annotation process +% ===================================== +\subsection{Data annotation} +To create the \dataset{} dataset, we use stories from the extended ROCStories \cite{mostafazadeh2016corpus} corpus for annotation by crowd workers. +% +% +ROCStories narratives describe a rich set of causal and temporal commonsense relations between daily events, and its stories are short enough that the world described by them are self-contained. They thus are a good fit for testing state inferences. + +Figure~\ref{Fig:pasta_example} illustrates the process followed to collect responses from the crowd workers.\footnote{The project was reviewed and approved by the local IRB for human subjects research.} % +% The UI also included explanations of each step of the process with example annotations (see the Appendix). +%\ffinline{mention IRB status, e.g., exempt, approved, etc.} +% +The annotation process has four main steps: +\begin{enumerate}[noitemsep, nolistsep] + \item \textbf{Infer a participant state}: For a story $S$, the annotator \underline{infers} a participant (or object) state $\alpha$ that is likely to be true at some point in $S$; $\alpha$ is a free-form sentence. Most stories have several inferrable states, so the annotator may identify whatever jumps out to them the most. + \item \textbf{Select minimal justification sentences:} + For the inferred $\alpha$, the annotator \underline{selects} the minimal set of sentences $J^S_{\alpha}$ in $S$, that they used to infer $\alpha$ from $S$. + \item \textbf{Perturb the state:} + The annotator \underline{perturbs} $\alpha$ to create $\alpha'$ such that $\alpha'$ is very unlikely to be true for the story $S$. $\alpha'$ is also a free-form sentence. + \item \textbf{Revise the story:} + The annotator \underline{revises} $S$ into $S'$, so that $\alpha'$ can be inferred from $S'$ but $\alpha$ is unlikely to be inferred from $S'$. The annotator is instructed to make minimal revisions in order to avoid creating $S'$ with other narrative side-effects. +\end{enumerate} +% +% +% We provide additional instructions about how to infer a state, and repeat them in the instructions, examples, and the actual form they fill out. +We provide detailed instructions about how to infer a state, and these are repeated not just in the instructions and examples, but also in the actual form they fill out. +The inferred state must be a property or attribute of a participant or object (e.g., \emph{she was angry} or \emph{the rock is heavy}); it must not be an action (e.g., \emph{Susan is running} or \emph{Jake cooks food}); and it must not be explicitly stated in the story. % +These constraints assure that the states are not readily available from the story text, and must be inferred by reasoning and world knowledge. The next section describes how we monitored the workers and mitigated improper responses. + +% \subsection{Quality Control}\label{subsec:quality_control} +% To avoid language-based artifacts in the dataset and ensure consistent and high-quality annotations, we crowdsourced annotations from workers with a long history of high-quality annotations and limited to those who reside in USA or Canada. +% % \ffinline{What was the pay? Maybe add avg pay per story to the table of stats?} +% After the first stage of annotations for our tasks, two experts (authors of this paper) reviewed the responses from a large pool of annotators and labeled them for correctness. Using this, we identified a set of proficient annotators who consistently gave high-quality responses. +% Only responses from these proficient annotators were retained. To further expand the collection, we conducted a second round where only these proficient annotators were invited back. The \emph{test} partition was created from only the correct responses reviewed by the expert reviewers. + +%======================================================================= +\subsection{Quality Control}\label{subsec:quality_control} +For crowdsourcing the data collection, we used the Amazon-MTurk platform (AMT). Each story was provided to three different crowd-workers for annotation. +We priced the HIT at $\$0.35$ based on initial worker response times and interest gleaned from multiple pilot runs. +%\sg{We set an initial price for annotating a story instance was set to $0.25\$$ based on our estimate for the effort involved. After multiple pilot runs, based on the actual workers response rate and interest in the task, we increased the final HiT cost to $0.35\$$.} +For filtering out noisy data from the collected responses, we follow a two-stage filtering process. + +\paragraph{Stage 1:} +We only allowed workers with a long history of consistent performance who satisfy the following criteria: +\begin{enumerate}[noitemsep, nolistsep] + \item have responded to at least 5000 HITs + \item have at least 98\% accuracy on their past HITs + \item must reside in USA or Canada; this helps to prevent language-based artifacts +\end{enumerate} +% +Although the above is strict, we still observed responses that did not follow the instructions. One difficulty was how workers wrote their revised stories. Even minor changes to the original story can render it logically inconsistent, so care is needed to ensure the counterfactual is inferrable while still maintaining coherence. +% Apart from this, we observed a few other errors in the different annotation steps. +Other annotation errors were `states' describing actions, states directly mentioned in the story, and non-entailed states. + +\vspace{-1pt} + +\paragraph{Stage 2:} +Despite the above errors, we received excellent responses with clear states and interesting revised stories. This gave us confidence that the task is achievable, but it just needed expert crowd workers. +To this end, we performed an "expert review" of the responses to identify "proficient workers": workers who can perform the task with a high degree of correctness. Our expert reviewers are two student researchers who work in the field of common-sense reasoning and NLP in general. Stage 1 resulted in a total of $9656$ responses from 136 workers. The experts evaluated a subset of these to identify proficient workers by using the process described below: + +\vspace{-1pt} + +%After an initial collection of several thousand HITs from \sg{$9,656$ responses from $136$} workers, these expert reviewers evaluated the workers: +% +\begin{enumerate}[noitemsep,nolistsep] +% + \item For each worker, we manually evaluated their performance on a random sample of their responses. +% + \item The number of evaluated responses for each worker was decided by the formula below. If the $i^{th}$ worker provides $n_i$ responses, then the minimum number of their responses, $e_i$, that needs to be expert-reviewed to evaluate their proficiency is given by: + \[ e_i = \begin{cases} + 0.3*n_i & n_i < 100 \\ + 0.2*n_i + 10 & 100\leq n_i < 150 \\ + 40 & 150 \leq n_i + \end{cases} + \] +% + \item Each evaluated response was categorized as correct or reject. A response was rejected if there was an error in any of the four steps of the annotation process. A response is correct if all the components of the annotation adheres to the instructions. +% + \item A worker was identified as proficient if they submitted $\geq50$ responses with a rejection rate $\leq20\%$. + After identifying proficient workers, all other responses from proficient workers were then auto-accepted. We also kept the smaller number of non-reject responses that our experts labeled from non-proficient workers. +\end{enumerate} +% + +With this process, we identified 28 workers who were proficient. We accepted all of their annotations totaling $\sim6,000$. To this we added the annotations the experts accepted in the review, which added another $360$ high quality instances. We then ran a second round of data collection using only the proficient workers. We added this to the high quality instances from the first round to form our full \dataset{} dataset. + +%\sg{As a result of this process, the experts had reviewed $3,656$, i.e. $\sim 36\%$ of the responses collected in \textbf{stage 1}.} This filtered the initial data collection by throwing out \sg{($\sim 33\%$)} non-proficient submissions. We then ran a second round of data collection by inviting the \sg{28} proficient workers to annotate more. This second round of collection, combined with the first round, forms our \dataset{} dataset. + +The responses in the pool of expert-reviewed responses were used to create the test set of the data. We also made sure that there is no story overlap in the train, validation, and test sets. + +\subsection{Dataset Statistics} +\dataset{} includes a total of $10,743$ ($8476$ train, $1350$ validation, and $917$ test) 4-tuples. Each 4-tuple is a story $S$, an associated inferred state $\alpha$, counterfactual state $\alpha'$, and a revised story $S'$. +Annotators almost always changed the justification sentences of the inferred state in order to revise the story. Instructions to make minimal changes to the revised story results in a high degree of similarity between the original and revised stories. On average $1.5$ out of $5$ story sentences are changed to create the revised story, with $90.3\%$ average token overlap between them. Similarly, the inferred state and its counterfactual on average show high lexical similarity with $72\%$ token overlap, and both having similar token length. +Additional statistics can be seen in Table~\ref{Table:dataset}. + +%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%5 +% DATASET - STATISTICS +%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%5 +\begin{table}[t!] +\footnotesize +\centering +% \begin{tabular}{m{6cm} | m{1.2cm}} +\begin{tabular}{m{0.73\linewidth} | m{0.17\linewidth}} +\Xhline{4\arrayrulewidth} +\hline +{\# of unique stories} & $5,028$ \\ +\hline +{Avg. \# of tokens in an inferred state} & $5.7$ tokens \\ +\hline +{Avg. \# of tokens in a perturbed state} & $6$ tokens \\ +\hline +{Avg.\# of justification sentences for a state} & $1.5$ \\ +\hline +{Avg. \# of sentences revised in a story} & $1.48$ \\ +\hline +{\% of justification sentences that are revised} & $90.54\%$ \\ +\hline +{\% of revised sentences that were justification} & $91.9\%$ \\ +\hline +{\% tokens in inferred state, common in perturbed state} & $71.9\%$ +\\ \hline +{\% story tokens common in revised story} & $90.3\%$ +\\ \Xhline{4\arrayrulewidth} +\end{tabular} +\caption{\small{\dataset{} Dataset Statistics}} +\label{Table:dataset} +\end{table} + +% See Figure~\ref{Fig:stae_tasks} for an example of this task. +\begin{figure}[t!] + % \includesvg[inkscapelatex=false,width=1.0\columnwidth]{images/story_state_tasks3.svg} + \includegraphics[width=.48\textwidth]{images/story_state_tasks3.png} + \caption{\small{Examples of three PASTA tasks. Input for each is on the left. The boxes indicate systems required to solve the tasks with example output on the right.}} + \label{Fig:stae_tasks} +\end{figure} +\section{State-based Reasoning Tasks} +Inferring each component of a \dataset{} 4-tuple requires a different commonsense reasoning ability about a participant's state in a narrative, which enables us to use \dataset{} to test models for these abilities. +As illustrated in Figure~\ref{Fig:stae_tasks}, we introduce three \dataset{} tasks, one classification and two generative, each of which can be used to evaluate current NLP models for the capabilities required to understand a participant's state in a narrative text. +In the subsections below we provide the motivation and formal task definition for each task. +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%5 +% CLASSIFICATION TASK 1 +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%5 +\subsection{\mcq{}}\label{subsec:state_inference} +We propose a classification task to evaluate a model's ability to understand what state is likely or unlikely to be inferred from a story. +% By inferring a state, +We deem a state is \emph{likely to be inferred} from a story if a typical human reading the story would conclude that the state is most likely true. +To test this capability in models, we pose the \mcq{} classification task. +% +\paragraph{Task definition:} Given a story $S$, a `query' state $\alpha_q$, +% \footnote{\todo[inline]{Can we make this not a footnote?}We use subscript $q$ to denote that $\alpha_q$ is a `query' state} +and a supporting set $\mathbf{s}$, which is a subset of the sentences in $S$, the task is to predict whether $\alpha_q$ is likely to be inferred from $\mathbf{s}$ in the context of $S$. +% +% \paragraph{Challenges:} +% The states in the data are result of crowd workers utilizing different type of commonsense knowledge (for example numerical, physical etc.) to infer them. This requires model performing the task to use similar knowledge to make inferences about these states. +% The data by design is such that both $\alpha \text{ and } \alpha'$ are unstated in $S$ or $S'$. Also, $\alpha \text{,} \alpha'$ have similar length, and $S$, $S'$ have same number of sentences, with both pairs having high degree of lexical overlap. +% This enables creating task dataset (Subsection~$4.4$, 1) that prevents models from spuriously learning to perform the task using only the story or the state. This is so because, each story or a state is associated with both positive and negative label depending on the state or story along with which it appears. +% Consider example 1 of \mcq{} task in Table~\ref{Table:T5_ip_op}, based on the highlighted justification set in the story, inferring that the state \underline{\textit{I live in North Pole.}} is most likely False, requires model to combine factual knowledge about North Pole's weather and numeric reasoning about the information mentioned in the story. Simple overlapping word cues (\emph{North}), may lead to wrong answer, as does looking at only the story or the state since there are data instances where the same story and state appears separately with opposite label. +% +% Therefore this task requires models utilize their their commonsense knowledge, and not just exploit any story or state annotation artifact that may exist in the data. +% Crowd sourcing the data also causes the response to include jargons and concepts that are relatively new. to which the older models may not have been exposed. For example in example 2 in \mcq{} task in Table~\ref{Table:T5_ip_op}, to inferring anything about the state \underline{\textit{My headphones have a transparency mode.}} from the story, requires models to understand and use the concept of transparency in headphones, which older models may not have been exposed to in the training. +% +\paragraph{Effects of data collection on performance:} We provide additional dataset analysis in subsection~\ref{subsec:story_state_inf_eval} to analyze the robustness of our data collection procedure that helped avoid unintended artifacts in the data for this task. +% +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%5 +% GENERATIVE TASK 1 +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%5 +\subsection{\storyrev{}} +% Motivation +A model that can understand participant states in narrative text should also be able to reason about counterfactual states and their potential effects on the narrative. We introduce the \storyrev{} task to address this. +% +% Task definition +\paragraph{Task definition:} Given a story $S$, a participant state $\alpha_q$ that is counterfactual to $S$ (a state that is not consistent with $S$), make minimal revisions to $S$ to generate $S'$ such that $\alpha_q$ is unstated in $S'$ and it can be inferred from $S'$, i.e. $P(\alpha_q|S') \approx 1$ and $P(\alpha_q|S') \gg P(\alpha_q|S)$. +% +% Task challenges +% \paragraph{Challenges:} +% The \storyrev{} task is relatively more complex in nature due to being a constrained generation task. +% It adds up on the above stated challenges and requires model to use their knowledge of event-state and event-event causal relationship to identify and revise parts of $S$ that are inconsistent with $\alpha'$ to generate $S'$ that is logically coherent as a whole. +% Consider a representative example of the task (2) in Table~\ref{Table:T5_ip_op}. Understanding how input state \underline{\textit{Connor only has six coworkers.}} is counterfactual to the input story, requires combining commonsense reasoning with strong numeric ability. The model performing the task must be able to do multihop reasoning to deduce that the input state and sentence 1 to 3 contradict the information in sentence 4, thus requiring its revision. Revising sentence 4 in turn requires revising sentence 5 to make the revised story logically consistent as whole. +% +% to make it consistent with the input state makes the overall story logically inconsistent, as now sentence 1 to 4 (revised) becomes inconsistent with sentence 5, thus requiring further revision to remove the logical inconsistencies. +% +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%5 +% GENERATIVE TASK 2 +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%5 +\subsection{\statechange{}}\label{subsec:state_change} +% Motivation +A corollary of being able to reason about the effects of a counterfactual state on the discourse of a narrative is the ability to identify the state changes (and how it changed) which led to the new narrative. In other words, when given a revised story with its original, what original state and its counterfactual explains the change? To assess this, we introduce the \statechange{} task. +% +\paragraph{Task definition:} Given a story $S$ and its revision $S'$, the task is to generate participant states $\alpha, \alpha'$ that describe the change of state from $S$ to $S'$, i.e. $P(\alpha | S) \gg P(\alpha | S')$ and $P(\alpha' | S') \gg P(\alpha' | S)$. +% +% Task challenges +% \paragraph{Challenges:} +% The \statechange{} task is also a generative state based task, but has a different kind of complexity compared to the above task. Apart from the common complexity of combining different kind of knowledge, this task requires model to identify the specific relevant states out of all the inferable states in the two similar but different stories that explains the state change between them. This also requires models to have strong understanding of negation and contradiction since the generated states should be counterfactual to each other. In example of the task (3) in in Table~\ref{Table:T5_ip_op}, the input stories differ only in the last sentence, however there is a prominent change in the state of the coffee that led to change in the discourse of the two stories. Identifying the relevant states in the two stories, require model to understand how in story two John ends up going to the hospital instead of his home, so the injury as a result of coffee spill must be serious, and the coffee being hotter in story 2 compared to 1 might be a key state that led to the two significantly different outcome of the coffee spill. +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +% Task instance creation +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +\subsection{Task-specific data creation} +\label{subsec:task_data_creation} +The three tasks above use the \dataset{} 4-tuple $(S, \alpha, \alpha', S')$ to create task specific data instances in the following manner:\\ +%From each $(S, \alpha, \alpha', S')$ instance of the \dataset{} dataset, the task specific data instances were created in the following manner: + +\noindent +\textbf{1. \mcq{}}: Let $S=(s_1, \cdots, s_5)$ and $S' = (s'_1, \cdots, s'_5)$. +We create four data instances for the task, +positive data instances $((S, \mathbf{s}, \alpha), 1)$ and $((S', \mathbf{s'}, \alpha'), 1)$, and +negative instances $((S, \mathbf{s}, \alpha'), 0)$ and $((S', \mathbf{s'}, \alpha), 0)$. +The supporting set $\mathbf{s}$ for $S$ is $J^S_{\alpha}$, i.e. the minimal set of sentences used to infer $\alpha$ from $S$. +For $S'$, $\mathbf{s'} = \{s \in \{s'_1, \cdots, s'_5\} | s'_i \neq s_i, \: \forall \: i \in 1 \: \text{to} \: 5\}$ i.e., the set of sentences in $S$ that were changed when revising $S$ to $S'$. + +\noindent +\textbf{2. \storyrev{}}: We created two data instances for the task of the form $((S, \alpha'), S')$ and $((S', \alpha), S)$. +% This resulted in $16,952$ train and $2,700$ validation, and $1,834$ test data points for the task. + +\noindent +\textbf{3. \statechange{}}: We created two data instances for the task of the form $((S, S'), (\alpha, \alpha'))$ and $((S', S), (\alpha', \alpha))$. +% This resulted in $16,952$ train and $2,700$ validation, and $1,834$ test instances for the task. + +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +% Task Challenges +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +% \subsection{Task challenges} + +% \par +% The \storyrev{} task is relatively more complex in nature due to being a constrained generation task. It adds up on the above stated challenges requires model to use their event-state and event-event causal knowledge to identify and revise parts of $S$ that are inconsistent with $\alpha'$ to generate $S'$ that is logically coherent as a whole. +% Consider a representative example of the task (2) in Table~\ref{Table:T5_ip_op}. Understanding how input state \underline{\textit{Connor only has six coworkers.}} is counterfactual to the input story, requires combining commonsense reasoning with strong numeric ability. The model performing the task must be able to do multihop reasoning to deduce that the input state and sentence 1 to 3 contradict the information in sentence 4, thus requiring its revision. Revising only sentence 4 to make it consistent with the input state makes the overall story logically inconsistent, as now sentence 1 to 4 (revised) becomes inconsistent with sentence 5, thus requiring further revision to remove the logical inconsistencies. +% \par +% The \statechange{} task is also a generative state based task, but has a different kind of complexity compared to the above task. Apart from the common complexity of combining different kind of knowledge, this task requires model to identify the relevant states out of all the inferable states in the two similar but different stories and explains the state change between them.. This also requires models to have strong understanding of negation and contradiction since the generated states should be counterfactual to each other. In example of the task (3) in in Table~\ref{Table:T5_ip_op}, the input stories differ only in the last sentence, however there is a prominent change in the state of the coffee that led to change in the discourse of the two stories. Identifying the relevant states in the two stories, require model to understand how in story two John ends up going to the hospital instead of his home, so the injury as a result of coffee spill must be serious, and the coffee being hotter in story 2 compared to 1 might be a key state that led to the two significantly different outcome of the coffee spill. +% \par +% \sg{potential solutions for these challenges} +% These are some of the representative key complexities of the task which makes them a challenging benchmark to probe the different aspect of the state centric knowledge in models. +% % +% Recent works have shown that scaling model pertaining, both in terms of model size and pretraining dataset size, can produce models that have different kinds of commonsense knowledge necessary to understand and reason about participant state. +% These models can also produce long and, to a certain extent, logically coherent text that is essential to effectively performing the proposed tasks. +% Making state-based inferences requires using knowledge of the world's facts, as this knowledge can occasionally be used to make an inference or verify its validity. +% However the factual knowledge about the world is ever evolving and thus there is a need to have these large scale models that can combine their commonsense knowledge with the factual knowledge available in external knowledge dataset. Therefore, one potential direction is to add retrieval capability to these large, pretrained language models so they can search for relevant information from outside knowledge sources. +% +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +% TABLE: GPT3 +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +{\renewcommand{\arraystretch}{1.1}% for the vertical padding +\begin{table} + \footnotesize +\begin{subtable}{0.5\textwidth} +\sisetup{table-format=-1.2} % 2 decimals, leave space for minus sign +\centering + \begin{tabular} + { m{0.42\linewidth} m{0.08\linewidth} |m{0.09\linewidth} |m{0.08\linewidth} } + \cline{2-4} + & \multicolumn{3}{c}{\textbf{\# of examples in prompt}} \\ \hline + \multicolumn{1}{l|}{\textbf{APPROACH}} & \textbf{5} & \textbf{10} & \textbf{15} \\ \hline \hline + \multicolumn{1}{l|}{EXPERT CURATED} & \footnotesize{\makecell[cc]{$81.6$}} & \footnotesize{\makecell[cc]{$81.6$}} & \footnotesize{\makecell[cc]{$82.5$}} \\ \hline + \multicolumn{1}{l|}{RANDOM SELECTION} & \makecell[cc]{$81.2$} & \makecell[cc]{$81.8$} & \makecell[cc]{$82.3$} \\ \hline + \multicolumn{1}{l|}{NEAREST NEIGHBOR} & \makecell[cc]{$81.7$} & \makecell[cc]{{{${83.3}$}}} & \makecell[cc]{$82.0$} \\ \Xhline{3\arrayrulewidth} + \end{tabular} + \caption{\storyrev{}}\label{tab:gpt3_hyperparam1} +\end{subtable} +\bigskip + +\begin{subtable}{0.5\textwidth} +\sisetup{table-format=4.0} % integer values only, up to 4 digits +\centering + \begin{tabular} + { m{0.42\linewidth} m{0.09\linewidth} |m{0.08\linewidth} |m{0.08\linewidth} } + \cline{2-4} + & \multicolumn{3}{c}{\textbf{\# of examples in prompt}} + \\ \hline + \multicolumn{1}{l|}{\textbf{APPROACH}} & \textbf{5} & \textbf{10} & \textbf{15} \\ \hline \hline + \multicolumn{1}{l|}{EXPERT CURATED} & \footnotesize{\makecell[cc]{{{${53.0}$}}}} & \footnotesize{\makecell[cc]{$52.8$}} & \footnotesize{\makecell[cc]{$51.9$}} \\ \hline + \multicolumn{1}{l|}{RANDOM SELECTION} & \makecell[cc]{$51.4$} & \makecell[cc]{$51.2$} & \makecell[cc]{$52.1$} \\ \hline + \multicolumn{1}{l|}{NEAREST NEIGHBOR} & \makecell[cc]{$52.1$} & \makecell[cc]{$51.5$} & \makecell[cc]{$50.2$} \\ \Xhline{3\arrayrulewidth} + \end{tabular} + \caption{\statechange{}}\label{tab:gpt3_hyperparam2} +\end{subtable} +\caption{\small{\textbf{{GPT3} hyperparameter selection.} Few-shot performance (BERTscore) of GPT-3 for combinations of (i) prompt example selection approach (ii) \# of examples in prompt.} +% \todo[inline]{Are you going to keep the blue font?} +} \label{Table:gpt3_hyperparam} +\end{table}} +% +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +% TABLE: T5 Input-Output format +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +\begin{figure}[!ht] + \centering +% \begin{table}[!h] +% \centering +\scriptsize +\begin{tabular}{p{7.5cm}} +\Xhline{4\arrayrulewidth} +% %%%%% TASK 1 +Task 1. \textbf{\mcq{}} +\\ \Xhline{4\arrayrulewidth} +\textbf{Input:} \textbf{infer\_state} \textbf{story:} I live in the north. We usually have snow here for Christmas. This year is different. \hl{* It's 50 degrees out!} \hl{* We will have a green holiday this year.} +\newline \textbf{state:} I live in the North Pole. +\\ \Xhline{1\arrayrulewidth} +\textbf{Target Output:} false \\ +% \hline \hline +% \textbf{Input:} \textbf{infer\_state} \textbf{story:} \hl{* I like to wear headphones when I walk. *Today I didn't hear a car that was coming my direction.} The car almost hit me. I had to jump out of the way just before I got hit. I decided never to wear my headphones again when I was walking. +% \newline \textbf{state:} My headphones do not have a transparency mode. +% \\ \Xhline{1\arrayrulewidth} +% \textbf{Target Output:} true \\ +\Xhline{4\arrayrulewidth} +% %%%%% TASK 2 +Task 2. \textbf{\storyrev{}} \\ \Xhline{4\arrayrulewidth} +\textbf{Input:} \textbf{revise} \textbf{story} +: Connor had a very busy workload that day. +: He forgot to eat breakfast and grabbed a dozen doughnuts on the way. +: People were greeting him and taking one doughnut each time. +: When he arrived at his desk, there were no doughnuts left. +: He went on with his busy day on an empty stomach. \newline +\textbf{state}: Connor only has six coworkers. +\\ \Xhline{1\arrayrulewidth} +\textbf{Target Output:} +: When he arrived at his desk, half the donuts were gone. +: He ate a donut before starting work. \\ +\Xhline{4\arrayrulewidth} +% %%%%% TASK 3 +Task 3. \textbf{\statechange{}} \\ +\Xhline{4\arrayrulewidth} +\textbf{Input:} \textbf{change} \textbf{story1:} Joe was stopping at a drive thru for breakfast. +He ordered a cup of coffee and breakfast sandwich. +Joe was trying to eat and drive at the same time. +Joe accidently dropped his coffee in his lap. +Joe had to go home and change his pants. \newline +\textbf{story2:} Joe was stopping at a drive thru for breakfast. +He ordered a cup of coffee and breakfast sandwich. +Joe was trying to eat and drive at the same time. +Joe accidently dropped his coffee in his lap. +Joe had to go to the hospital. + +\\ \Xhline{1\arrayrulewidth} +\textbf{Target Output:} \textbf{state1:} The coffee wasn't hot enough to seriously burn Joe. \textbf{state2:} The coffee was hot enough to seriously burn Joe. +\\\Xhline{4\arrayrulewidth} +\end{tabular} +\caption{\small{Examples of T5 formats. An asterisk $*$ is prepended to the \hl{supporting set sentences} in the stories in task (1). The output in task (2) includes special tokens like and to indicate revisions to the $4^{th}$ and $5^{th}$ sentences.}} +\label{Table:T5_ip_op} +% \end{table} +\end{figure} +% +% +% +\section{Experimental Setup}\label{sec:experiment} +% \nc{Make this the only thing in Section 5} +To establish modern baselines and measure their performance, we built benchmark models from {GPT3, T5, BERT}, and {RoBERTa}. This section describes how each was setup for the three tasks. +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +% GPT3 - Experimental setup +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +\subsection{GPT3} +We benchmarked {GPT3} with few-shot prompting \cite{brown2020language} on the two generation tasks (Story Revision and State Change). +We created prompts with task examples from the training set, followed by an incomplete prompt from the eval set that the model must complete. +For the \storyrev{} task, the prompt included $n$ examples followed by the final query: $(S_1, \alpha'_1, S'_1)\cdots(S_n, \alpha'_n, S'_n)(S_q, \alpha'_q, -)$ where $(S_i, \alpha'_i, S'_i)$ is the $i^{th}$ task example. The model must generate $S'_q$ for the final query ($S_q$,$\alpha'_q, -)$. +Similarly, the \statechange{} task uses a similar prompt: $(S_1, S'_1, \alpha_1, \alpha'_1)\cdots(S_n, S'_n, \alpha_n, \alpha'_n)(S_q, S'_q, -, -)$. +\par +To select prompt examples, we tried three approaches. \textbf{(i) EXPERT CURATED:} We selected a fixed set of diverse, unambiguous examples that requires multi-step reasoning and covers different type of states, and used the same prompt examples for all the query instances, \textbf{(ii) RANDOM SELECTION:} We randomly selected examples, \textbf{(iii) NEAREST NEIGHBOR}~\cite{liu-etal-2022-makes}: For each query instance, we selected examples that were most similar to it. For this, we computed the cosine similarity between the \texttt{[CLS]} representation of the instances obtained from {RoBERTa-large} fine-tuned on the \mcq{} task. For each approach, we tried creating prompts with $5\text{, } 10$ and $15$ examples. +% (\nc{do you mean prompt size here? - SG: YES, Rephreased }) +Prompt examples were selected from a set of 200 high-quality, expert-selected instances drawn from the training set, similar to~\citet{West2022SymbolicKD}. +% (\nc{I don't understand what this means - SG: Rephrased}). +% For selecting examples in the prompt, we only 200 high-quality expert selected instances (\nc{I don't understand what this means}), similar to \citet{West2022SymbolicKD}. +% + +We treat the number of prompt examples and their selection as hyperparameter combinations, and evaluated each of them on 200 random samples from the validation set. Since human evaluations are expensive, we use BERTScore, which has the highest correlation with human evaluated validity of output, among the automatic metrics we tried (see Table~\ref{Table:pasta_corr}). Table~\ref{Table:gpt3_hyperparam} shows that the combinations perform roughly similar but there is a two point gap between the best and the worst combination. For the \storyrev{} task, we use NEAREST NEIGHBOR with $10$ prompt examples, and for the ~\statechange{} task, we use EXPERT CURATED with $5$ examples. + + +%For each task, we select the hyperparameter combination with the highest BERTscore for creating the prompt examples. +%We use BERTscore because (i) human evaluation was too expensive to evaluate models for each hyperparameter set, and (ii) we observed (Table~\ref{Table:pasta_corr}) that it has the highest correlation with human validity judgments compared to other automatic metrics. + +% +We used the \texttt{text-davinci-002} {GPT3} model for both tasks. We set the generation temperature parameter to $0.9$, frequency penalty to $0.5$, and maximum generation length to $100$. +% +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +% T-5 +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +\subsection{T5} +We benchmarked {base} ({T5-b}) and {large} ({T5-l}) variants of T5 on all three state-based tasks by fine-tuning them on the task-specific instances created from \dataset{}, as explained in Section~\ref{subsec:task_data_creation}. +Examples of T5 input-output format for each task are shown in Figure~\ref{Table:T5_ip_op}. +For all the tasks, {T5-b} and {T5-l} were trained for $7$ and $5$ epochs, respectively. +% For all the tasks, {T5-b} was trained for $7$ epochs and {T5-l} was trained for $5$ epochs. +For model training, we used AdamW \cite{loshchilov2017decoupled} optimizer with a learning rate of $10^{-4}$ and weight decay of $10^{-6}$. For {T5-l} the batch-size for tasks 1, 2 and 3 were $8$, $4$ and $4$ respectively. Whereas for {T5-b}, the corresponding batch-sizes were $16$, $12$, and $10$. For text generation, we used nucleus sampling with $0.93$ top-p; $100$ as max generation length. +% +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +% BERT\RoBERTa +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +% +\subsection{BERT~\textbackslash~RoBERTa} +We benchmarked {base} ({BERT-b}) and {large} ({BERT-l}) variants of {BERT-uncased}, and {base} ({RoBERTa-b}) and {large} ({RoBERTa-l}) of {RoBERTa} on only the \mcq{} task since they are non-generative models. +The input format for the models is identical to that of {T5}. For all the models, we used AdamW optimizer with a learning rate of $5\text{e}{-6}$ and weight decay of $1\text{e}{-6}$. +The large and base models were trained for $5$ and $7$ epochs respectively. + +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +% Running the model +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +{T5-b, BERT-b, BERT-l, RoBERTa-b} and {RoBERTa-l} were trained on an NVIDIA-TITAN-X 24GB, and {T5-l} was trained on an NVIDIA-A6000 48GB GPU. +% For the \mcq{}, task, we report average model performance on five different random seeds: $0, 1239, 64489, 43237, \text{ and } 54363$. For the generative tasks, we report the model performance on random seed $0$. + +\section{Results and analysis} +% We now describe our application of today's language models to each of our three tasks. +We now analyze the performance of these recent language models on the three \dataset{} tasks. +% ----------------------------- +% Relevant table 1 - model performance on the task +% ----------------------------- +{\renewcommand{\arraystretch}{1.2}% for the vertical padding +\begin{table}[!ht] + \small + \begin{center} + \begin{tabular} + { m{0.28\linewidth}| m{0.25\linewidth} |m{0.25\linewidth} } + \Xhline{3\arrayrulewidth} + & \textbf{\footnotesize{Accuracy (\%)}} & \textbf{\footnotesize{Contrastive Accuracy (\%)}} \\ \hline \hline + \footnotesize{\makecell[l]{\textbf{BERT-b}}} & \makecell[cc]{$73.8 \pm 0.3$} & \makecell[cc]{$64.0 \pm 0.5$} \\ \hline + \footnotesize{\makecell[l]{\textbf{T5-b}}} & \makecell[cc]{$79.8 \pm 0.6$} & \makecell[cc]{$70.7\pm 0.5$} \\ \hline + \footnotesize{\makecell[l]{\textbf{RoBERTa-b}}} & \makecell[cc]{$81.2 \pm 0.6$} & \makecell[cc]{$73.0 \pm 0.8$} \\ \Xhline{3\arrayrulewidth} + \footnotesize{\makecell[l]{\textbf{BERT-l}}} & \makecell[cc]{$77.5 \pm 0.4$} & \makecell[cc]{$68.7 \pm 0.7$} \\ \hline + \footnotesize{\makecell[l]{\textbf{T5-l}}} & \makecell[cc]{$83.1 \pm 0.9$} & \makecell[cc]{$75.3\pm 1.4$} \\ \hline + \footnotesize{\makecell[l]{\textbf{RoBERTa-l}}} & \makecell[cc]{$89.1 \pm 0.4$} & \makecell[cc]{$83.7 \pm 0.5$} \\ \Xhline{3\arrayrulewidth} + \footnotesize{\hyperref[par:human_eval]{\textbf{\footnotesize{Human$^\bigstar$}}}} & \makecell[cc]{$96.9$} & \makecell[cc]{$94.2$} \\ \Xhline{3\arrayrulewidth} + \end{tabular} + \end{center} + \caption{\small{\textbf{\mcq{} - Model evaluation}: Accuracy is the \% of instances where the model made correct predictions. Contrastive Accuracy gives a credit to the model if it correctly predicts the inferability for both the inferred and counterfactual states for a story. + } + } + \label{Table:state_inference_main_results} +\end{table}} +\subsection{\mcq{} }\label{subsec:story_state_inf_eval} +% +We evaluated model performance with standard accuracy and contrastive accuracy. In contrastive accuracy, the model gets a point only if it makes correct predictions for both inferred and the counterfactual states for a story. For all models, we train with five random seeds and report their average performance with standard deviation. +%, for training over five different random seeds. +% ======================================================== +% -----------SUBSECTION 1 +% ======================================================== +\paragraph{Human evaluation:}\label{par:human_eval} +% \ff{Can remove the clause ``To evaluate human performance on this task,''} +We conducted human evaluation on the task instances (see Section~\ref{subsec:task_data_creation}.1) created from \dataset{}. +We randomly selected $200$ 4-tuples from the test set\footnote{Model performance for this test subset differed by $<0.5\%$ from that of the overall test set}, and created $800$ story-state inference instances from them. +Each task instance, $(S, \alpha_q, \mathbf{s})$ was evaluated by three crowd workers who rated the likelihood of inferring $\alpha_q$ from $\mathbf{s}$ in context of $S$, on a 5-point Likert scale - {Extremely unlikely, Unlikely, Cannot Say, Likely, and Extremely likely}. +We threshold the Likert value to a binary 0/1 value with the mapping, \{Extremely unlikely to Cannot Say\} $\rightarrow$ 0 and rest $\rightarrow$ 1. +The human prediction for an instance was computed by majority voting, which along with its true label was used to compute the human performance.\footnote{The instance label assignment is explained in Section~\ref{subsec:task_data_creation}} % +% We followed majority voting to compute the human prediction for an instance, which along with its true label~\footnote{Instance label assignment is explained in Section~\ref{subsec:task_data_creation}} was used to compute the human performance. +% ======================================================== +% -----------SUBSECTION 2 +% ======================================================== +\paragraph{\mcq{} is a hard task} +Table~\ref{Table:state_inference_main_results} shows that even for just standard accuracy, there is room for improvement ($7.8\%$) when comparing the best performing model (RoBERTa-l) to humans on this simple binary classification task. +This performance gap further increases to $10.5\%$ when considering contrastive accuracy. +% +Increasing model size from base to large yields $3.7\% \text{ (BERT)} \text{ to } 8\% \text{ (RoBERTa)}$ gains on standard accuracy. +For contrastive measure, both base and large variants of each models fare substantially worse, with performance drops ranging from $5.4\%$ (RoBERTa-l) to $9.8\%$ (BERT-b). For humans, the corresponding performance drop is only $\sim2.7 \%$. +This suggests that predicting whether a state is likely to be inferred from a story is difficult for these LLMs, even when fine-tuning on a relatively large number of examples. +% +% ======================================================== +% -----------SUBSECTION 3 (Harder setting) +% ======================================================== +{\renewcommand{\arraystretch}{1.2} +\begin{table}[!ht] + \small + \begin{center} + \begin{tabular} + { m{0.25\linewidth}| m{0.25\linewidth} |m{0.25\linewidth} } + \Xhline{3\arrayrulewidth} + & \textbf{\footnotesize{Accuracy (\%)}} & \textbf{\footnotesize{Contrastive Accuracy (\%)}} \\ \hline \hline + \footnotesize{\makecell[l]{\textbf{BERT-l}}} & \makecell[cc]{$74.9 \pm 0.3$} & \makecell[cc]{$64.6 \pm 0.3$} \\ \hline + \footnotesize{\makecell[l]{\textbf{T5-l}}} & \makecell[cc]{$79.6 \pm 0.6$} & \makecell[cc]{$69.8\pm 1.0$} \\ \hline + \footnotesize{\makecell[l]{\textbf{RoBERTa-l}}} & \makecell[cc]{$86.7 \pm 0.4$} & \makecell[cc]{$80.4\pm 0.6$} \\ \hline + \footnotesize{\hyperref[par:human_eval]{\textbf{\footnotesize{Human$^\bigstar$}}}} & \makecell[cc]{$93.5$} & \makecell[cc]{$88.9$} \\ \Xhline{3\arrayrulewidth} + \end{tabular} + \end{center} + \caption{\small{\textbf{\mcq{} - without the justification sentences}: Model performance on a harder variant of the \mcq{} task, where they don't have direct access to the justification sentences of a state when predicting it's inferability for a story.} + } + \label{Table:state_inference_wo_support_main_results} +\end{table}} +\par +% +%\paragraph{Story State inference - without justification sentences} +%Access to justification sentences of a state helps the story-state inference model to focus on the relevant sentences for predicting the inferability of a state for a story. +%A relatively harder, but realistic setting for the \mcq{} task is to predict the likelihood of inferring the state from a story, +We also analyze the performance of the models when they don't have direct access to the justification sentence information in the story. We fine-tuned large variants of the three baseline models on this task. +From Table~\ref{Table:state_inference_main_results} to \ref{Table:state_inference_wo_support_main_results}, we see that the task performance drops across all the models on both evaluation metrics, with a $2.4\%$ to $3.5\%$ drop in accuracy, and $3.3\%$ to $5.5\%$ in contrastive accuracy. The gap between the human performance and the best performing model is still substantial. +This shows that justification sentences are indeed important to solve the task, but the models often still make reasonable decisions without them. +% ===================================================== +% \begin{table}[!h] +% \small +% \begin{subtable}{0.5\textwidth} +% \sisetup{table-format=-1.2} % 2 decimals, leave space for minus sign +% \centering +% \begin{tabular} +% { m{0.21\linewidth}| m{0.196\linewidth} |m{0.196\linewidth} |m{0.196\linewidth} } +% \Xhline{3\arrayrulewidth} +% & \textbf{\footnotesize{Precision}} & \textbf{\footnotesize{Recall}} & \textbf{\footnotesize{F1}} \\ \hline \hline +% \footnotesize{\makecell[l]{\textbf{BERT-l}}} & \makecell[cc]{$67.6 \pm 0.5$} & \makecell[cc]{$58.0 \pm 2.6$} & \makecell[cc]{$62.4\pm 1.2$} \\ \hline +% \footnotesize{\makecell[l]{\textbf{T5-l}}} & +% \makecell[cc]{$71.6\pm1.9$} & \makecell[cc]{$71.4\pm1.6$} & \makecell[cc]{$71.48\pm.8$} \\ \hline +% \footnotesize{\makecell[l]{\textbf{RoBERTa-l}}} & \makecell[cc]{$77.0 \pm 2.1$} & \makecell[cc]{$69.6\pm 1.7$} & \makecell[cc]{$73.2\pm 0.7$} \\ +% \Xhline{3\arrayrulewidth} +% \end{tabular} +% % \caption{Identifying justification sentences for a state in a story.}\label{tab:sub_first} +% \end{subtable} +% % %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +% % This subtable has model p +% % %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +% % \bigskip +% % \begin{subtable}{0.5\textwidth} +% % \sisetup{table-format=4.0} % integer values only, up to 4 digits +% % \centering +% % \begin{tabular} +% % { m{0.21\linewidth}| m{0.196\linewidth} |m{0.196\linewidth} |m{0.196\linewidth} } +% % \Xhline{3\arrayrulewidth} +% % & \textbf{\footnotesize{Precision}} & \textbf{\footnotesize{Recall}} & \textbf{\footnotesize{F1}} \\ \hline \hline +% % \footnotesize{\makecell[l]{\textbf{BERT-l}}} & \makecell[cc]{$82.2 \pm 0.7$} & \makecell[cc]{$87.6 \pm 1.0$} & \makecell[cc]{$84.8\pm 0.4$} \\ \hline +% % \footnotesize{\makecell[l]{\textbf{T5-l}}} & \footnotesize{\makecell[cc]{$87.2 \pm 0.7$}} & \footnotesize{\makecell[cc]{$87.2\pm 1.2$}} & \footnotesize{\makecell[cc]{$87.2\pm 0.4$}} \\ \hline +% % \footnotesize{\makecell[l]{\textbf{RoBERTa-l}}} & \makecell[cc]{$87.0 \pm 0.6$} & \makecell[cc]{$90.6\pm 1.2$} & \makecell[cc]{$88.6\pm 0.5$} \\ +% % \Xhline{3\arrayrulewidth} +% % \end{tabular} +% % \caption{Identifying non-justification sentences for a state in a story.}\label{tab:sub_second} +% % \end{subtable} +% \caption{\sg{NEW}\textbf{Identifying Justification sentences:} Model performance on the task of identifying sentences in a story that justify inferring a given state from it.} \label{Table:justification_sentences} +% \end{table} +% ===================================================== +% \paragraph{Justification sentence identification:} +% % The above experimental results highlight the importance of model having access to the information about the relevant parts of the story, for predicting the inferability of a state from it. +% To further investigate the role of these justification sentences, we inverted the task and analyzed the reverse reasoning challenge: can models identify the sentences from the inferred state itself? + +% To this end, we fine-tuned the same baseline models as above, but this time given just a story and a state. The task is to identify the story's sentences that justify the state inference. +% We fine-tuned T5 to generate the sentence numbers. +% We fine-tuned BERT and RoBERTa with a multilabel classification approach, where the model was trained to jointly predict the justification label (YES \textbackslash~NO) for each sentence in the story. The training hyperparameters were identical to those described in~\ref{sec:experiment}. + +% % OBSERVATIONS +% \nc{This paragraph confuses me a lot. I really don't know what your numbers mean nor what you're comparing. What is $4.7\%$, $7\%$ ?}\sg{\underline{I have completely revised this para}, what I wanted to highlight was that the relative model performance on the MCQ is same the precision or F1 on this task.} +% Table~\ref{Table:justification_sentences} +% shows the baseline models' performance on this inverted identification task. +% Overall, RoBERTa-l has the highest F1 score on this task, followed by T5-l, and then BERT-l. The performance gap between RoBERTa-L and T5-L on F1 metric is only $1.3\%$ percent. However, this gap widens to $5.4\%$ when we consider precision metric. T5-l has the highest recall, indicating its high propensity to predict a sentence as a justification, thus explaining its high F1 score. Furthermore, the relative model performance on precision and F1 for the tasks is similar to that on the \mcq{} task (for both settings), with RoBERTa-l being the best, followed by T5-l, and then BERT-l. So, a model that outperforms the other in the \mcq{} task also outperforms it in the inverse setting. + +% +% From Table~\ref{Table:state_inference_wo_support_main_results} we can observe that BERT-l to T5-l to RoBERTa-l performance increase is about $4.7\%$, $7\%$ on standard accuracy, and $5.2\%$, $10.6\%$ contrastive accuracy. +% The corresponding model performance improvement for the original \mcq{} task setting (Table~\ref{Table:state_inference_main_results}) is $5.5$, $6\%$ and $6.6\%$, $7.6\%$. +% From Table~\ref{Table:justification_sentences}, we can see that the corresponding model performance trend for justification sentence identification precision is $4\%$, $5.4\%$. For other metrics the relative model performance trend shows much less correlation. +% This result shows that the model performance on the \mcq{} task for both the settings is strongly correlated to how precisely it can identify the justification sentences in the story for a state inferred from it. +% ----------------------------- +% Relevant table2 - contrastive set analysis +% ----------------------------- +{\renewcommand{\arraystretch}{1.2}% for the vertical padding +\begin{table}[t] + \small + \begin{center} + \begin{tabular}{ m{0.18\linewidth} m{0.19\linewidth} | m{0.19\linewidth} | m{0.2\linewidth} } + \cline{2-4}\cline{2-4} + & \multicolumn{3}{c}{\textbf{Test data}} \\ \Xhline{3\arrayrulewidth} + \multicolumn{1}{c|}{\textbf{Train data}} & + \makecell[cc]{$\mathcal{D}^{te}_1$} & \makecell[cc]{$\mathcal{D}^{te}_2$} & \makecell[cc]{$\mathcal{D}^{te}_1 \cup \mathcal{D}^{te}_2$} \\ \hline \hline + \multicolumn{1}{c|}{$\mathcal{D}^{tr}_1$} & $90.2 (84.4)$ & $79.1 (70.8)$ & $84.8 (77.9)$ \\ \hline + \multicolumn{1}{c|}{$\mathcal{D}^{tr}_2$} & $81.5 (73.6)$ & $88.1 (82.6)$ & $84.8 (78.1)$ \\ \hline + \multicolumn{1}{c|}{$\mathcal{D}^{tr}_1 \cup \mathcal{D}^{tr}_2$} &$90.2 (85.3)$ & $88.1 (82.3)$ & $89.1 (83.7)$ \\ \Xhline{3\arrayrulewidth} + \end{tabular} + \end{center} + \caption{\small{\textbf{\mcq{} - Dataset Analysis:} + Accuracies and contrastive accuracies of {RoBERTa-large} when trained and tested on dataset partitions created from original stories ($\mathcal{D}_1$), modified stories ($\mathcal{D}_2$), and their union ($\mathcal{D}_1\cup \mathcal{D}_2$). $tr \text{ and } te$ denotes the corresponding training and test splits.} + } + \label{Table:state_inference_artifiact} +\end{table}} + +\paragraph{Importance of the data collection design:} +It is important to note that we included contrastive examples in our train-set. To illustrate its importance, we trained a model on the dataset created from just the original stories ($\mathcal{D}_1$ = $\{((S_i, \mathbf{s_{i}}, \alpha_i), 1), ((S_i, \mathbf{s_{i}}, \alpha_i'), 0) \}^N_{i=1}$), and another on the modified stories ($\mathcal{D}_2 = \{((S'_i, \mathbf{s'_{i}}, \alpha_i'), 1), ((S'_i, \mathbf{s'_{i}}, \alpha_i), 0) \}^N_{i=1}$). +We then trained and tested on these different dataset, results of which are reported in Table~\ref{Table:state_inference_artifiact}. +%With this split, we then train on one story type's partition and test on the other, where the story source is reversed (original vs modified). + +Generalization accuracy is significantly worse if we had only constructed positive and negative states for a collection of stories. For example, training on $\mathcal{D}_1^{tr}$ and testing on $\mathcal{D}_2^{te}$ leads to an $11.1\%$ drop in accuracy compared to in-distribution test on $\mathcal{D}_1^{te}$. For $\mathcal{D}_2^{tr}$, the corresponding drop is $6.6\%$, which supports the quality of our stories/states and shows that both original and counterfactual state inferences are learnable. Had we not collected the revised story, then models could potentially learn artifact-based heuristics (e.g. guessing whether the state is original or modified) resulting in the lack of generalization that we observe here. Because \dataset{} includes the revised stories, we can train on the full dataset $\mathcal{D}^{tr} = D_1^{tr} \cup D_2^{tr}$, and see that the performance is uniform across the different test partitions. This highlights the challenges in constructing negative examples for such tasks and the importance of including contrastive examples for both training and test for proper generalization. +% +% ----------------------------- +% Relevant table-3 Importance of justification set +% ----------------------------- +% +% {\renewcommand{\arraystretch}{1.2}% for the vertical padding +% \begin{table}[!h] +% \small +% \begin{center} +% \begin{tabular}{ m{0.26\linewidth}| m{0.22\linewidth} |m{0.1\linewidth} |m{0.18\linewidth} } +% \Xhline{3\arrayrulewidth} +% & \footnotesize{\makecell[cc]{\textbf{Story} + \\ \textbf{Justification}}} & \footnotesize{\makecell[c]{\textbf{Story} \\ \textbf{Only}}} & +% \footnotesize{\makecell[c]{\textbf{Justification} \\ \textbf{Only}}} \\ \hline \hline +% \footnotesize{\makecell[l]{\textbf{Accuracy (\%)}}} & \makecell[c]{$82.9$}& \makecell[c]{$79.6$} & \makecell[cc]{$82.7$} \\ \hline +% \footnotesize{\makecell[l]{\textbf{Contrastive} \\ \textbf{Accuracy (\%)}}} & \makecell[cc]{$74.6$} & \makecell[cc]{$69.2$} & \makecell[cc]{$74.9$} \\ \Xhline{3\arrayrulewidth} +% \end{tabular} +% \end{center} +% \caption{\small{\textbf{\mcq{} - Impact of justification sentences}: T5-large performance for training with different model inputs. +% \textbf{(i) Story + Justification -} model input is story $S$ with sentences in ${s_{q}}$ identified by a special token. +% \textbf{(ii) Story only -} model input is only $S$. +% \textbf{(iii) Justification only -} model input is only $s_{q}$. +% }} +% \label{Table:state_inference_main_results2} +% \end{table}} + +% \paragraph{Impact of justification set}\sg{I think, we should skip this sub-section, as the 1st and 2nd column are already covered by previous two tables and the 3rd table as such gives no insight} +% The justification set is a subset of sentences in the story that support the associated participant state. To analyze the importance of using the justification set for predicting the inferability of a state from a story, we trained models with three different input settings: +% \textbf{(i) Story + Justification:} Use the story and the corresponding justification set. In this setting, sentences in the story that are present in the justification set were prepended with special identifier token '$*$'. +% \textbf{(ii) Story only:} Use only the story, without identifying/marking the story sentences common in the justification set. +% \textbf{(iii) Justification only:} Use only the justification set. + +% Table~\ref{Table:state_inference_main_results2} shows that the model performance drops when the justification set is not identified in the input. There is a $\sim3\%$ drop in accuracy and $\sim5\%$ drop in contrastive accuracy. The model performance when using the justification set only is roughly the same as when using the entire story with the justification set marked. This suggests that models tend to focus on the the justification set when making inference for the task. +% ~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~= +% Story revision +% ~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~= +\begin{table}[!h] + \small + \begin{center} + \begin{tabular}{ m{0.19\linewidth} | m{0.145\linewidth}| m{0.12\linewidth} | m{0.14\linewidth} | m{0.12\linewidth} } + \Xhline{3\arrayrulewidth} + & \multicolumn{3}{c|}{\textbf{\footnotesize{Acceptability}}} & \\ \cline{2-4} + & + \textbf{\footnotesize{\makecell[cc]{\% \\ Inferable \\(A)}}} & + \textbf{\footnotesize{\makecell[cc]{\% \\ Logical \\(B)}}} & + \textbf{\footnotesize{\makecell[cc]{\% ALL \\(A \& B)}}} & \textbf{\footnotesize{\makecell[cc]{Minimal \\ Revision}}} \\ + \hline \hline + \footnotesize{\textbf{GPT3 - FS}} & $50$ & $86$ & $48.5$ & $86.33$ \\ + \hline + \footnotesize{\textbf{T5-b FT}} & $41.0$ & $77.0$ & $34.0$ & $91.39$ \\ + \cline{1-5} + \footnotesize{\textbf{T5-l FT}} & $58.5$ & $84.0$ & $54.0$ & $89.17$ \\ + \Xhline{3\arrayrulewidth} + \end{tabular} + \end{center} + \caption{\small{\textbf{\storyrev{} - Human evaluation}: $\%$ of model generated stories that satisfy evaluation criteria. \textbf{Inferable} and \textbf{Logical} assure coherence of the revisions with the required entity state. \textbf{ALL} indicates generations that satisfied both the criteria. \textbf{FS} means few-shot learning, \textbf{FT} means finetuned on \dataset{} dataset.} + } + \label{Table:story_rev_human_eval} +\end{table} +\subsection{\storyrev{}}\label{subsec:story_rev_result} +This generation task requires the model to revise a given story, such that the revised story is consistent with the given counterfactual participant state. +We use human judgments to evaluate the revised stories because reference-based automatic evaluation metrics (e.g. BLEU~\cite{papineni2002bleu}, BERTscore~\cite{zhang2019bertscore} etc.) are inadequate for multiple reasons: +(i) valid revised stories often exist that are different from the references, +(ii) original and revised stories overlap heavily which can skew the metrics, and +(iii) small lexical changes that don't change automatic metrics can affect logical consistency. +We thus evaluate generation quality using our proficient workers from Section~\ref{subsec:quality_control}. + +We compare performance of the models on a subset of $200$ test instances chosen at random. We evaluated them for quality on three metrics: +(1) \textbf{Inferable}: how likely is it for the given state $\alpha'$ to be true at any point in the revised story $S'$? This was rated on a 5-point Likert scale, which we thresholded to a 0/1 value (1 means inferrable). +(2) \textbf{Logical}: is the generated story $S'$ logically correct? This was a \texttt{YES/NO} question. +(3) \textbf{Minimal revison}: what is the degree of revision made to $S$ to generate $S'$? This was rated on a 5-point Likert-scale, with 4 indicating minimal revision and 0 an entirely new story. +% from Minimal to Entirely New Story. +Higher scores indicate higher similarity between $S$ and $S'$. % +\textbf{Inferability} and \textbf{Logical} decide the ultimate correctness of a response. +We calculated an overall model acceptability score (\textbf{ALL} in Table~\ref{Table:story_rev_human_eval}) by finding the percentage of model output that were both logical, and the input state can be inferred from them. +%Table~\ref{Table:story_rev_human_eval} shows the results on these factors. + +Table~\ref{Table:story_rev_human_eval} shows that T5-l outperforms T5-b and GPT3 on the acceptability (\textbf{ALL}) of generated outputs by a large margin of $20\%$ and $5.5\%$, respectively. +GPT3 has the best performance on logical validity of the generated output with T5-l lagging behind by only $2\%$, but only $50\%$ of GPT3's output satisfy the inferability criteria. +In fact all the models have low inferability score, which brings down their overall acceptability score. +T5-b has the best performance on the `minimal revision' made to the original story, however this was not a primary metric of concern and there is always a trade-off between doing well on this score and generating an acceptable result. For example, revising a story conditioned on a counterfactual that is connected to entities in a different part of the story might require substantial revisions. + +% Table~\ref{Table:story_rev_human_eval} shows that T5-l outperforms both T5-b and GPT3 on all the metrics except on the `minimal revision' score (\nc{is this true? it's worse than GPT3 on logical}). +% On the ALL \textbf{acceptability score}, T5-l outperforms both GPT3 and T5-b by a large margin of $\sim 15.5$ and $\sim5.5\%$ respectively (\nc{this is not true, do you mean 5.5 and 20?}). +% T5-l lags behind the other models (\nc{huh? I'd say it's similar to them, please check this}) on the `minimal revision' made to the original story, however this was not a primary metric of concern and there is always a trade-off between doing well on this score and generating an acceptable result. For example, revising a story conditioned on a counterfactual that is connected to entities in a different part of the story might require substantial revisions. + +Overall, only $54\%$ of the output generated by the best model, T5-l, are acceptable, indicating that the task is challenging and there is large room for improvement. Our results with GPT3 were based on few-shot prompting where we treated its design choices as a modelling hyperparameter that were chosen based on automatic metric performance on the validation set. +Few-shot performance of GPT3-scale models depends heavily on prompt engineering, so this direction may require further investigation. +% \vspace{-1ex} +% ~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~= +% State change +% ~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~=~= +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +% State change - human eval +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +\begin{table}[!htp] + \small + \begin{center} + \begin{tabular}{ m{.165\linewidth}| m{0.145\linewidth}| m{0.15\linewidth}| m{0.135\linewidth}| m{0.135\linewidth} } + \Xhline{3\arrayrulewidth} + \textbf{{\makecell[lc]{Model}}} + & \textbf{\footnotesize{\makecell[c]{\% Valid \\ Attribute \\(A)}}} + & \textbf{\footnotesize{\makecell[c]{\% Valid \\ Inferable \\(B)}}} + & \textbf{\footnotesize{\makecell[c]{\% Not \\ in Story \\(C)}}} + & \textbf{\footnotesize{\makecell[c]{\% ALL \\A, B \& C}}}\\ \hline \hline + \footnotesize{\textbf{GPT3 FS}} & $86.5$ & $67.46$ & $81.5$ & $47.7$\\ + \hline + \footnotesize{\textbf{T5-b FT}} & $96.75$ & $41.0$ & $90.25$ & $35.17$\\ \cline{1-5} + \footnotesize{\textbf{T5-l FT}} & $99.25$ & $58.75$ & $97.0$ & $55.50$\\ + \Xhline{3\arrayrulewidth} + \end{tabular} + \end{center} + \label{Table:state_change_human_eval} + \caption{\small{\textbf{\statechange{} - Human evaluation}: + \% of output that satisfy the evaluation criteria. \textbf{Valid attributes} ensures that the generated states describe entity attributes, not actions. \textbf{Valid inferability} indicates if both states can be inferred from their correct stories. \textbf{Not in Story} assures that the states are unstated in the stories. \textbf{ALL} indicates the \% of generations that satisfy all the these criteria.} + } + \label{Table:state_change_results} +\end{table} +\subsection{\statechange{}} +In this task, for given stories $S$ and $S'$, the model generates the two states $\alpha$ and $\alpha'$. +%We benchmark the performance of T5-base, T5-large and GPT3 %using both automatic and human evaluation. +As in the previous task, we do a human evaluation of a randomly selected set of $200$ model generated outputs. + +Model outputs were evaluated on the following metrics: (1) \textbf{Valid Attribute:} do the generated states $\alpha$, $\alpha'$ describe entity attributes? This was a \texttt{YES} or \texttt{NO} question. (2) \textbf{Valid Inferability:} are generated states $\alpha$ and $\alpha'$ inferable from $S$ and $S'$, but not from $S'$ and $S$, respectively? Workers rated $\alpha$ and $\alpha'$'s likelihood of being inferred independently, on a 5-point Likert scale, which was thresholded to a 0/1 value. For instance, if $\alpha$ is inferred from $S$, then $L_{S\alpha}=1$ (otherwise 0). Based on these scores, the inferability change for $\alpha$ is computed (1 for valid, 0 for invalid) using $max(0, L_{S\alpha} - L_{S'\alpha})$. (3) \textbf{Not in Story:} are $\alpha$ and $\alpha'$ unstated in both $S$ and $S'$? This was a multiple-choice question with 4 choices, 3 corresponding to the state being present in either one or both the stories, and the 4th for neither of the stories. An output $(\alpha, \alpha')$ gets full credit on a metric if both $\alpha$ and $\alpha'$ are correct for that metric, half if only one of them ($\alpha$ or $\alpha'$) is correct, and 0 otherwise. \textbf{ALL} indicates full credit on all three metrics. + +Table~\ref{Table:state_change_results} shows the results. T5-l in general outperforms both T5-b and GPT3 on all the metrics except the \textbf{Valid Inferability}, where GPT3 outperforms the other models by a large margin +% of $\sim8.7\%$ and $\sim26.4\%$ +. Interestingly, GPT3 is the worst performing model on \textbf{Valid Attribute} and \textbf{Not in Story}. +This indicates that GPT3 is loosely "cheating" by copying text in the story itself, which of course is inferable, but violates the task's requirement of an implicit state. +Overall, the best acceptability score (\textbf{ALL} in Table~\ref{Table:state_change_results}) is only $55.5\%$, which suggests that generating an output that satisfies all the criteria for a quality state change is an interesting challenge. +\subsection{Automatic Evaluation for Generative Tasks}\label{sec:auto_eval} +For the two generative tasks, we reported human evaluation results for the best analysis (prior sections). +However, since human evaluation is expensive, we include here the results from three automatic metrics: {GLEU} \cite{Wu2016GooglesNM}, {ROUGE} \cite{lin-2004-rouge}, and {BERTscore} \cite{zhang2019bertscore}. +For {GLEU}, we consider 1 to 4-grams overlap between the output and reference. +We report {ROUGELsum} for the \storyrev{} task since it is computed over the entire story, and the sentence level {ROUGEL} metric for \statechange{}. +From Table~\ref{tab:story_rev_eval} and \ref{tab:st_change_eval}, we can observe that even for automatic metrics, {T5-l} is still the best performing model on both tasks. + +\begin{table} + \small +\begin{subtable}{0.5\textwidth} +\sisetup{table-format=-1.2} % 2 decimals, leave space for minus sign +\centering + \begin{tabular} + { m{0.17\linewidth} | m{0.22\linewidth} |m{0.165\linewidth} |m{0.195\linewidth} } + \Xhline{3\arrayrulewidth} + & \footnotesize{\makecell[cc]{\textbf{BERTscore}}} & \footnotesize{\makecell[c]{\textbf{GLEU}}} & + \footnotesize{\makecell[c]{\textbf{rougeLsum}}} \\ \hline \hline + \footnotesize{\makecell[l]{\textbf{GPT3 FS}}} & + \makecell[cc]{$80.7$} & \makecell[cc]{$69.7$} & \makecell[cc]{$79.6$} \\ \hline + \footnotesize{\makecell[l]{\textbf{T5-b FT}}} & \makecell[cc]{$81.6$} & \makecell[cc]{$73.2$} & \makecell[cc]{$81.7$} \\ \hline + \footnotesize{\makecell[l]{\textbf{T5-l FT}}} & \makecell[cc]{$82.1$} & \makecell[cc]{$73.5$} & \makecell[cc]{$81.7$} \\ \Xhline{3\arrayrulewidth} + \end{tabular} + \caption{\storyrev{}}\label{tab:story_rev_eval} +\end{subtable} + +\bigskip +\begin{subtable}{0.5\textwidth} +\sisetup{table-format=5.0} % integer values only, up to 4 digits +\centering + \begin{tabular} + { m{0.17\linewidth} | m{0.22\linewidth} |m{0.17\linewidth} |m{0.19\linewidth} } + \Xhline{3\arrayrulewidth} + & \footnotesize{\makecell[cc]{\textbf{BERTscore}}} & \footnotesize{\makecell[c]{\textbf{GLEU}}} & + \footnotesize{\makecell[c]{\textbf{ROUGEL}}} \\ \hline \hline + \footnotesize{\makecell[l]{\textbf{GPT3 FS}}} & {\makecell[cc]{$55.4$}} & + {\makecell[cc]{$11.6$}} & + {\makecell[cc]{$28.9$}} \\ \hline + \footnotesize{\makecell[l]{\textbf{T5-b FT}}} & \makecell[cc]{$54.4$} & + \makecell[cc]{$11.7$} & + \makecell[cc]{$29.5$} \\ \hline + \footnotesize{\makecell[l]{\textbf{T5-l FT}}} & \makecell[cc]{$56.9$} & + \makecell[cc]{$13.4$} & + \makecell[cc]{$32.4$} \\ \Xhline{3\arrayrulewidth} + \end{tabular} + \caption{\statechange{}}\label{tab:st_change_eval} +\end{subtable} +\caption{\small{\textbf{Automatic evaluation for generative tasks: } Model performance on the generative tasks using BERTscore, GLEU and ROUGE based metrics.}} \label{Table:pasta_autmatic_eval} +\end{table} +% +\begin{table}[!h] + \footnotesize + \begin{center} + \begin{tabular}{m{0.12\linewidth} | m{0.22\linewidth} |m{0.22\linewidth} |m{0.22\linewidth}} + \Xhline{3\arrayrulewidth} + % + & \multicolumn{1}{c|}{\footnotesize{\textbf{BERTscore}}} + & \multicolumn{1}{c|}{\footnotesize{\textbf{GLEU}}} + & \multicolumn{1}{c}{\footnotesize{\textbf{ROUGE}}} \\ \hline \hline + \footnotesize{\makecell[l]{\textbf{Task-1}}} & + \small{\makecell[cc]{$.21$ $($6\text{e-}7$)$}} & + \small{\makecell[cc]{$.14$ $($9\text{e-}4$)$}} & + \small{\makecell[cc]{$.13$ $($2\text{e-}4$)$}} \\ \hline + \footnotesize{\makecell[l]{\textbf{Task-2}}} & \small{\makecell[cc]{$.27$ $($4\text{e-}22$)$}} & + \small{\makecell[cc]{$.21$ $($1\text{e-}13$)$}} & + \small{\makecell[cc]{$.22$ $($1\text{e-}15$)$}} \\ \Xhline{3\arrayrulewidth} + \end{tabular} + \end{center} + \caption[Caption for LOF]{\small{\textbf{Pearson Correlation} between automatic metric score of a model output and its validity as determined by humans. Numbers in parenthesis are p-value\protect\footnotemark~for the null hypothesis that they are uncorrelated. \textbf{Task-1} is \storyrev{}, and \textbf{Task-2} is \statechange{}.}} + \label{Table:pasta_corr} +\end{table} +\footnotetext{Lower the p-value, higher is the confidence for rejecting the null hypothesis.} +% +% +To further analyze automatic metrics as an alternative to human evaluation, we computed the correlation between them. We computed the Pearson correlation between the automatic metric score of an output and its validity as determined by humans. The results are reported in Table~\ref{Table:pasta_corr}. +% Table~\ref{Table:pasta_corr} shows the correlation between automatic metric score and human acceptability, +The numbers in parenthesis are the p-values for the null hypothesis ($95\%$ confidence interval) that they are uncorrelated. We observed that {BERTscore} has the highest correlation with human evaluated validity for both tasks, outperforming other metrics by a substantial margin. The low p-value further indicates that the correlation is statistically significant. However, since the correlation is low, we strongly recommend using human evaluations, and only use {BERTscore} as an alternative where human evaluation is expensive. +% ======================================== +% IAA between crowd workers +% ======================================== +\renewcommand{\arraystretch}{1.1}{ +\begin{table} + \footnotesize + \centering + % \begin{tabular}{l|c} + \begin{tabular}{ m{.46\linewidth}| m{0.18\linewidth}| m{0.18\linewidth} } + \Xhline{3\arrayrulewidth} + \multirow{2}{*}{\textbf{{\makecell[c]{Task}}}} & \multicolumn{2}{c}{\textbf{Gwet's coefficient}} \\ \cline{2-3} + & \makecell[c]{\textbf{Coeff}} & \makecell[c]{\textbf{StdErr}} \\\hline + + \mcq{} & \makecell[c]{{0.81}} & \makecell[c]{{0.01}} \\ \hline + + \makecell[l]{Story Revision from \\ a Counterfactual} & \makecell[c]{{0.72}} & \makecell[c]{{0.02}} \\ \hline + + \statechange{} & \makecell[c]{{0.76}}& \makecell[c]{{0.01}} \\ + \Xhline{3\arrayrulewidth} + \end{tabular} + % \caption[Caption for LOF]{\small{\sg{NEW}\textbf{Pearson Correlation} between automatic metric score of a model output and its validity. Numbers in parenthesis are p-value \protect\footnotemark for the null hypothesis that they are uncorrelated. \textbf{Task-1} is \storyrev{}, and \textbf{Task-2} is \statechange{}.}} + \caption[Caption for LOF]{\small{\textbf{Inter-Annotator Agreement + % \protect\footnotemark + } for human evaluation for the three tasks. \textbf{Coeff} is the calculated IAA coefficients, and \textbf{StdErr} is the standard error.}} + \label{Table:IAA} +\end{table}} +% \footnotetext{The interpretation of Gwet's coefficient is similar to generalized kappa~\cite{viswanathan2012development}, with $.6-.8\equiv$ substantial and $\geq.8\equiv$ almost perfect agreement.\label{kapp_interpretation}} +\subsection{Inter-Annotator Agreement} + We measure the inter-annotator-agreement (IAA) for the human workers using Gwet's Agreement Coefficient~\cite{gwet2008computing, gwet2014handbook}, which is a type of generalized Kappa statistic.\footnote{Gwet's normalizes the probability of observed agreement with a percent chance agreement that is the propensity of raters to agree on hard-to-rate instances~\cite{gwet2014handbook}.} + Its interpretation is similar to generalized kappa~\cite{viswanathan2012development}, with $0.6-0.8\equiv$ substantial and $\geq0.8\equiv$ almost perfect agreement. + We use Gwet's coefficient because it is robust to the paradoxical behaviors~\cite{wongpakaran2013comparison, gwet2014handbook} seen in the commonly used IAA Kappa metrics (e.g. Cohen's and Fleiss). This paradoxical behavior of these metrics can lead to their IAA coefficients being lower even when the agreement is strong~\cite{feinstein1990high, byrt1993bias}. + % .\footnote{Commonly used Kappa metrics (e.g. Cohen's and Fleiss) can sometimes behave paradoxically~\cite{feinstein1990high, byrt1993bias}, where their coefficients value can be lower even when the agreement is strong.} + %Percent chance agreement is the propensity for raters to agree on hard-to-score subjects. Gwet's computes percent chance agreement by multiplying the probability to agree when the rating is random by the probability to select a hard-to-score subject.} + +%We use Gwet's coefficient as it addresses this issue using an improved way to compute the chance agreement among raters.\footnote{Percent chance agreement is the propensity for raters to agree on hard-to-score subjects. Gwet's computes percent chance agreement by multiplying the probability to agree when the rating is random by the probability to select a hard-to-score subject.} +%Gwet's coefficient is similar to generalised Kappa, with the percent chance agreement~\footnote{Percent chance agreement is calculated by multiplying the probability to agree when the rating is random by the probability to select a hard-to-score subject.} defined as the propensity for raters to agree on hard-to-score subjects. The main reason for choosing this metric is its robustness to the paradoxes \cite{wongpakaran2013comparison, gwet2014handbook, quarfoot2016robust} observed with the commonly used IAA Kappa metrics \cite{feinstein1990high, byrt1993bias}. + +\paragraph{Crowd workers IAA} Table~\ref{Table:IAA} shows the IAA coefficient for the tasks and their standard errors. For each task, we computed the IAA coefficient for their respective evaluation metrics on their original scale (pre-thresholding\footnote{Note that thresholding was only done for ordinal scale metrics.}), which were then averaged to obtain the overall task scores. +%\nc{"As can be seen, the crowd worker correlation is strong." Can we say that? .7 and .8 are strong, yes? "very strong"? Please add a version of my sentence that works.} +We computed the unweighted IAA coefficient for an evaluation metric if it was nominal, with quadratic weight if it was ordinal. +As can be observed from the table, the crowd worker have strong agreement for both the generative tasks and almost perfect agreement for the classification task. +% ~\textsuperscript{\ref{kapp_interpretation}}.\todo{Confusing footnote. Replace with in-text reference, or better, remove?} +% If the evaluation metric was nominal, then we compute the unweighted IAA coefficient, if it was ordinal, then we computed IAA coefficient with quadratic weight. + +\paragraph{Experts IAA} +The two experts in Section~\ref{subsec:quality_control} were responsible for accepting or rejecting a worker response for the \dataset{} creation. +% We measure the IAA between experts on the task of accepting or rejecting a worker response for the \dataset{} creation. +To measure their IAA, we created a pool of $200$ \dataset{} instances that included both accepted and rejected instances. +The experts had a Gwet's coefficient of $0.87$ and agreed on $93.5\%$ of those $200$ instances. +\section{Discussion} +% \sg{revision made; 184 $\rightarrow$ 200} + +Here we discuss the main challenges and error analyses that highlight areas for future work. + +\subsection{Challenges} +The key challenge common across all tasks is access to diverse types of knowledge (e.g. commonsense, numerical, factual, etc.), as well as the ability to combine and reason with them. For example, task 1 in Figure~\ref{Table:T5_ip_op} requires factual knowledge about the temperatures in the North Pole, commonsense about snow and Christmas, and the ability to combine these when reasoning to detect the incompatibility of the input state. +%The proposed tasks have several challenges associated with them, the main one being the necessity to combine different types of commonsense knowledge such as numerical, factual, physical, etc. + +The Story Revision Task has the added challenge of a model identifying the parts of the input story that are inconsistent with the counterfactual state, and then finally generating logically coherent text. For instance in Figure~\ref{Table:T5_ip_op} task 2, based on the input story and state, the model must first infer from sentences 2-4 that Connor had $12$ coworkers. Then to generate the revised story, it also needs to reason about how the world state gets affected if there were fewer people than the number of doughnuts (e.g., now Connor would have some doughnuts leftover). +% The biggest challenge in the \statechange{} task is that both stories have many states. + +The main challenge in the \statechange{} task is that there can be numerous plausible state pairs that are compatible with both stories, but they don't reflect a pertinent state change. Each state needs to be incompatible with one of the stories and compatible with the other, and this differentiation is a big challenge for any model. +For example in Figure~\ref{Table:T5_ip_op} task 3, the observable difference between the stories is the outcome from coffee spilling on Joe. Using abductive reasoning with commonsense knowledge about temperature, one can easily infer that the change in state leading to a different ending comes from the coffee's temperature. + +% \sg{Apart from the task specific requirements, the performance on the generative tasks is also limited by issues such as factual and logical inconsistency, hallucination, etc., which are common in text generation in general. As such this resource is more biased towards evaluating state based reasoning in models rather } + +% +%These key challenges lead to different type of errors, which are described in the next sections. +\subsection{Error Analysis}\label{subsec:error_anl} +%To further understand the challenges, we present error analysis on the models' failures. + +We analyze the model's errors on $200$ randomly selected instances from the validation set. +{\renewcommand{\arraystretch}{1.2} +\begin{table}[!ht] +\footnotesize + \begin{center} + \begin{tabular} + { m{0.19\linewidth} | m{0.28\linewidth}| m{0.14\linewidth} |m{0.18\linewidth} } + \Xhline{3\arrayrulewidth} + \multicolumn{1}{c|}{} & \textbf{\makecell[cc]{State Type}} & \textbf{\makecell[cc]{Acc. $\%$}} & \textbf{\makecell[cc]{Contrastive \\ Acc. $\%$}} \\ \hline \hline + \multirow{4}{*}{\makecell[l]{\textbf{BERT-l}}} + & All - $100\%$ & \makecell[cc]{$79$} & \makecell[cc]{$71.5$} \\ \cline{2-4} + & Societal - $14.5\%$ & \makecell[cc]{$70.7$} & \makecell[cc]{$62.1$} \\ \cline{2-4} + & Emotional - $54\%$ & \makecell[cc]{$80.8$} & \makecell[cc]{$74.1$} \\ \cline{2-4} + & Physical - $31.5\%$ & \makecell[cc]{$79.8$} & \makecell[cc]{$71.4$} \\ \Xhline{3\arrayrulewidth} + \multirow{4}{*}{\makecell[l]{\textbf{T5-l}}} + & All - $100\%$ & \makecell[cc]{$85.7$} & \makecell[cc]{$80.5$} \\ \cline{2-4} + & Societal - $14.5\%$ & \makecell[cc]{$81.9$} & \makecell[cc]{$75.9$} \\ \cline{2-4} + & Emotional - $54\%$ & \makecell[cc]{$87$} & \makecell[cc]{$81.5$} \\ \cline{2-4} + & Physical - $31.5\%$ & \makecell[cc]{$85.3$} & \makecell[cc]{$81$} \\ \Xhline{3\arrayrulewidth} + \multirow{4}{*}{\makecell[l]{\textbf{RoBERTa-l}}} + & All - $100\%$ & \makecell[cc]{$90.6$} & \makecell[cc]{$86.6$} \\ \cline{2-4} + & Societal - $14.5\%$ & \makecell[cc]{$83.6$} & \makecell[cc]{$77.6$} \\ \cline{2-4} + & Emotional - $54\%$ & \makecell[cc]{$93.5$} & \makecell[cc]{$89.4$} \\ \cline{2-4} + & Physical - $31.5\%$ & \makecell[cc]{$88.9$} & \makecell[cc]{$86.5$} \\ \Xhline{3\arrayrulewidth} + \end{tabular} + \end{center} + \caption{\small{\textbf{\mcq{}}: Model performance for predicting the state inferability of different type of states.} + } + \label{Table:task8-error} +\end{table}} +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +% Statistics table - TASK 2 +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +% \renewcommand{\arraystretch}{1.3}{ +% \begin{table*} +% \footnotesize +% \centering +% \begin{tabular}{l|c|c|c|c|c} +% \Xhline{3\arrayrulewidth} +% \multirow{2}{*}{\textbf{Commonsense Type}} +% & \multirow{2}{*}{\makecell[cc]{\textbf{\% Correct}}} +% & \multicolumn{4}{c}{\makecell[cc]{\textbf{\% Incorrect}}} \\ \cline{3-6} +% & +% & \makecell[cc]{\% Illogical \\ revised story} +% & \makecell[cc]{\% Irrelevant \\ change} +% & \makecell[cc]{\% Input state \\ not entailed} +% & \% Contradiction \\ +% \hline +% Psychological & 30.1 &&&&\\ +% \hline +% Societal & 30.1 &&&&\\ +% \hline +% Physical & 30.1 &&&&\\ +% \Xhline{3\arrayrulewidth} +% \end{tabular} +% \caption{\small{Categories of different type of errors for the \textbf{\storyrev{}} task.}} +% \label{Table:task6-error} +% \end{table*}} +\renewcommand{\arraystretch}{1.0}{ +\begin{table} + \footnotesize + \centering + \begin{tabular}{l|c} + \Xhline{3\arrayrulewidth} + \multicolumn{1}{c|}{\textbf{Error Category}} & \textbf{Percentage} \\ + \hline + Illogical revised story & 30.1\\ + Irrelevant change & 27.7\\ + Contradiction & 20.5\\ + Input state not entailed & 20.5\\ + State explicit in the revision & 1.2\\ + \Xhline{3\arrayrulewidth} + \end{tabular} + \caption{\small{Categories of different type of errors for the \textbf{\storyrev{}} task.}} + \label{Table:task6-error} +\end{table}} +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +% Statistics table - Task 3 +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +\begin{table}[t] +\footnotesize +\centering +\begin{tabular}{l|c} +\Xhline{3\arrayrulewidth} +\multicolumn{1}{c|}{\textbf{Error Category}} & \textbf{Percentage} \\ +\hline +Contradiction & 37.4\\ +Irrelevant states & 35.2\\ +Illogical state change & 13.2\\ +States reversed & 4.4\\ +No change in state & 4.4\\ +State directly stated in the story & 4.4\\ +Actions instead of states & 1.1\\ +\Xhline{3\arrayrulewidth} +\end{tabular} +\caption{\small{Categories of different type of errors for the \textbf{\statechange{}} task.}} +\label{Table:task7-error} +\end{table} +% \renewcommand{\arraystretch}{1.3}{ +% \begin{table*} +% \footnotesize +% \centering +% \begin{tabular}{l|c|c|c|c|c} +% \Xhline{3\arrayrulewidth} +% \multirow{2}{*}{\textbf{\makecell[cc]{Commonsense Type}}} +% & \multirow{2}{*}{\makecell[cc]{\textbf{\makecell[cc]{\% \\ Correct}}}} +% & \multicolumn{4}{c}{\makecell[cc]{\textbf{\% Incorrect}}} \\ \cline{3-6} +% & +% & \makecell[cc]{\% Contradiction} +% & \makecell[cc]{\% Irrelevant \\ States} +% & \makecell[cc]{\% Irrelevant \\state change} +% & \makecell[cc]{\% No change \\in state} \\ +% \hline +% Psychological & 30.1 &&&&\\ +% \hline +% Societal & 30.1 &&&&\\ +% \hline +% Physical & 30.1 &&&&\\ +% \Xhline{3\arrayrulewidth} +% \end{tabular} +% \caption{\small{Categories of different type of errors for the \textbf{\storyrev{}} task.}} +% \label{Table:task7-error} +% \end{table*}} +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +% EXAMPLE table - TASK 2 +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +\begin{figure}[t] +\scriptsize +% \centering +\begin{tabular}{p{7.5cm}} +\Xhline{4\arrayrulewidth} +% %%%%% EXAMPLE 1 +1. Irrelevant change +\\ \Xhline{4\arrayrulewidth} +\textbf{Input} $S$: Al and Tara can't agree what to watch on TV. Al wants to watch soaps. +Tara wants to watch the news. They play Paper, Rock, Scissors to decide. Tara wins and gets to watch what she wants. \newline +$\mathbf{\alpha}'$: Al and Tara have two T.V.s \\ \Xhline{1\arrayrulewidth} +\textbf{Output} Al and Tara can't agree what to watch on TV. Al wants to watch soaps. +Tara wants to watch the \colorbox{pink}{tv}. They play Paper, Rock, Scissors to decide. {Tara wins and gets to watch what she wants.} \\ +\Xhline{4\arrayrulewidth} +% %%%%% EXAMPLE 2 +2. Illogical revised story \\ \Xhline{4\arrayrulewidth} +\textbf{Input} $S$: Rich was a musician. He made a few hit songs. +Rich had a lot of fans who cared about him. He developed cancer. +Eventually the cancer took his life and rich was gone. \newline +$\alpha'$: Rich was unknown. +\\ \Xhline{1\arrayrulewidth} +\textbf{Output} Rich was a musician. He made a few hit songs. +Rich had \colorbox{pink}{no fans. He became ill and lost his life. Eventually Rich recovered} \colorbox{pink}{and began his recovery.} \\ +\Xhline{4\arrayrulewidth} +% %%%%% EXAMPLE 3 +3. Contradiction \\ +\Xhline{4\arrayrulewidth} +\textbf{Input} $S$: There was a large car wreck about two miles from the office. +I was waiting for a co worker to relieve me. +However, he was very late and I had no idea why. +He called me a few minutes later. +Apparently he was stuck in traffic because of the wreck. \newline +$\alpha'$: \hl{His co worker did not have access to a phone.} +\\ \Xhline{1\arrayrulewidth} +\textbf{Output} There was a large car wreck about two miles from the office. +I was waiting for a co worker to relieve me. +However, he was very late and I had no idea why. +\hl{He called me on the phone} \colorbox{pink}{a few minutes later but I had no idea why.} Apparently he was stuck in traffic because of the wreck. +\\\Xhline{4\arrayrulewidth} +% %%%%% EXAMPLE 3 +4. Input state not entailed \\ +\Xhline{4\arrayrulewidth} +\textbf{Input} $S$: Frank was very skilled at shaping sculptures. + His artwork started creating quite a buzz. + A plain looking man approached Frank with a job. + He wanted him to design the face for an adult themed product. + Frank was about to decline until he saw the amount of the check.\newline +$\alpha'$: The check was for a very small amount of money. +\\ \Xhline{1\arrayrulewidth} +\textbf{Output} Frank was very skilled at shaping sculptures. + His artwork started creating quite a buzz. + A plain looking man approached Frank with a job. + He wanted him to design the face for an adult themed product. + Frank was about to decline \colorbox{pink}{but didn't have much money}. +\\\Xhline{4\arrayrulewidth} +\end{tabular} +\caption{\small{Representative examples of top failure categories for the \textbf{\storyrev{}} task. Parts of the story that are changed are highlighted in \colorbox{pink}{pink}, and the contradictory elements are highlighted in \hl{yellow}.}} +\label{Table:examples_1} +\end{figure} +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +% example table - TASK 3 +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +\begin{figure}[t] +\scriptsize +% \centering +\begin{tabular}{p{7cm}} +\Xhline{4\arrayrulewidth} +% %%%%% EXAMPLE 1 +1. Irrelevant states +\\ \Xhline{4\arrayrulewidth} +\textbf{Input} $S$: Pat was looking for an interesting place to vacation. He searched various catalogs. Eventually he found the perfect destination. There he stayed in a small hut. It was one of the greatest experiences of his life. +$S'$: Pat was looking for a relaxing place to vacation. He searched various catalogs. Eventually he found the perfect destination. There he stayed in a luxury resort. +It was one of the greatest experiences of his life. +\\ \Xhline{1\arrayrulewidth} +\textbf{Output} $\alpha$: Pat likes to be independent. $\alpha'$: Pat likes to be in a big city. +\\ \Xhline{4\arrayrulewidth} +% %%%%% EXAMPLE 2 +2. Illogical state change \\ \Xhline{4\arrayrulewidth} +\textbf{Input} $S$: Tom spent a long time writing an essay. He saved it on his memory card. When Tom went to print it the essay wasn't there. Tom had to go back to his computer. Luckily he was able to use a backup copy he made. +$S'$: Tom spent a long time writing an essay. He saved it on his memory card. When Tom went to print it the essay wasn't there. Tom had to go back to his computer. Unfortunately, Tom forgot to make a backup copy of the essay. +\\ \Xhline{1\arrayrulewidth} +\textbf{Output} $\alpha$: Tom is a good memory writer. $\alpha'$: Tom is a bad memory writer. +\\ \Xhline{4\arrayrulewidth} +% %%%%% EXAMPLE 3 +3. Contradiction \\ +\Xhline{4\arrayrulewidth} +\textbf{Input} $S$: Rita accidentally left her smartphone on the plane. She felt terrible about this and knew her parents would be mad. Nobody ever turned in the phone, leaving her disappointed with people. \hl{Rita had to work many hours to buy a new phone as a replacement.} She is significantly more careful with her belongings now. +$S'$: Rita accidentally left her smartphone on the plane. + She felt terrible about this and knew her parents would be mad. + Nobody ever turned in the phone, leaving her disappointed with people. + Rita quickly bought a new phone as a replacement. + She is significantly more careful with her belongings now. +\\ \Xhline{1\arrayrulewidth} +\textbf{Output} $\alpha$: \hl{Rita was unemployed} $\alpha'$: Rita was rich +\\\Xhline{4\arrayrulewidth} +\end{tabular} +\caption{\small{Examples of top failure categories for the \textbf{\statechange{}} task. For the contradiction category, the conflicting parts are \hl{highlighted}.}} +\label{Table:examples_2} +\end{figure} +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% + +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +% Story-State Inference +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +\noindent\textbf{\mcq{}: } + We analyze model performance on different types of entity states following the categorization from ~\citet{bhagavatula2019abductive}. We expand their spatial category to a broader set of physical attributes of entities (e.g. weight, temperature, location, etc.), and include a new Societal category to capture social constructs and norms. % + Even though multiple categories may apply to a state, to simplify our analysis we only use the most relevant category for each state. + +%Since it is difficult to categorize the precise type of commonsense required to infer a particular participant state, we manually assign soft category to each inferred state based on the most prominent type of commonsense that is most likely used to infer that state. +In particular, we categorize each instance into one of the following: (i) Societal: knowledge about societal constructs such as relationship (\emph{Jake is not married, I have 5 brothers}), norms (\emph{John is not socially aware}), etc. (ii) Emotional/Psychological: knowledge about emotions (\emph{John felt embarrassed, John hated Jake}), beliefs (\emph{Jake believed in ghosts}), etc. (iii) Physical: Knowledge about physical attributes of entities (\emph{Jake was in his school, the rock was very heavy, the coffee was hot, etc.}). Table~\ref{Table:task8-error} breaks down the overall performance of models across different categories. Models significantly under-perform on the societal category compared to the other two. +In addition to the difficulty of modeling societal knowledge, we find that relatively more number of instances in this category require numerical commonsense, which adds additional complexity for the models. +%This can be in part due to the fact that relatively more societal instances in the evaluated set also require certain degree of numerical commonsense compared to the other two categories, which adds additional complexity for the models. +%Physical commonsense instances are more diverse since we define it as a broader category compared to the other two. +Physical commonsense is a broad category and its instances thus tend to cover a broad range of physical knowledge which could contribute to the difficulty of these instances. Emotional category has the best model performance since the inferred state include strong lexical indicators of emotions and feelings, similar to the observations in \citet{bhagavatula2019abductive}. + +The proposed generative tasks can have multiple correct outputs, each using a different set of commonsense knowledge. This makes it difficult to associate a unique knowledge category for the task instance. Therefore we manually analyze the outputs of the best performing model (T5-large) and identified common types of generation errors made by the model on the task. + +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +% Story revision +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +\noindent\textbf{\storyrev{}: } +The model output is correct for $\sim 58\%$ cases and incorrect for $\sim42\%$. +On analyzing the incorrect output, we found four main categories of error that we list in Table~\ref{Table:task6-error}. +% +The "illogical revised story" occurs when models produced revised stories that are logically incoherent (30\% of errors). Generating logically coherent long text is still a challenging task for models, and to a certain extent can be attributed to their tendency to forget attributes of specific entities \cite{Welleck2018DialogueNL}, ignore previously inferred facts \cite{sinha-etal-2019-clutrr} and background information, or contradict previous statements \cite{brown2020language}. +% +Moreover, $20.5\%$ of the revised stories are categorized as \textit{contradiction} as they clearly contradict the input counterfactual state. This corroborates previous findings on the challenges in reasoning about contradictions and negations~\cite{hossain-etal-2020-analysis}. +% +Models also struggle to keep the changes relevant to the task criteria of the input state, which should be inferable from the revised story but not directly mentioned in it. +They sometimes make \textit{Irrelevant changes} ($27.7\%$ of errors) where they revise parts of the story that are not affected by the input counterfactual state. +Other times they make revision that are inconsistent with the input counterfactual state (\textit{Input state not entailed}, $20.5\%$) or the input \textit{State is explicit in the revision} ($1.2\%$), both of which do not meet the primary task requirements. + +%and 'Input state not entailed', where the output revised story is plausible but do not satisfy the task criteria, i.e., the revised story should entailed the input state. +%The difference between them is that in the former category the revision is made to the parts of the story that are not affected by the input counterfactual state, whereas in the later case the changes are made to the parts of the story that are inconsistent with the input state. +% +Figure~\ref{Table:examples_1} shows examples of the biggest error categories for the task. +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +% State change task +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% + +\noindent\textbf{\statechange{}:} +The model is correct for 54.5\% of cases and fails for 45.5\% when generating state changes. +Table~\ref{Table:task7-error} shows the main error categories. +While the model learns to generate both the $\alpha$ and $\alpha'$ states about the same entity, it makes many types of logical errors. +% +\textit{Contradictions} ($37.4\%$ of errors) are when a generated state is contradicted by its story, either directly or by deduction. +\textit{Illogical State Changes} ($13.2\%$) are those where the generated states and input stories were topically related, but the states were simply illogical nonsensical. +Both types of errors can be attributed to the challenges associated with making the relevant state inference, generating logically coherent text and reasoning about contradictions and negations. +\textit{Irrelevant States} ($35.2\%$) are those where at least one of the generated states has no connection to its story. +The error categories of \textit{State Reversed} ($4.4\%$), \textit{No change in state} inferability ($4.4\%$), \textit{State is directly stated in the story} ($4.4\%$) and outputs are \textit{Actions instead of states} ($1.1\%$) are due to models' inability to correctly understand the task constraints. +% +Figure~\ref{Table:examples_2} shows example of the major error categories for the task. + +% ABOUT CHATGPT' +\subsection{Interactive Feedback with LLMs} +% \sg{How about the heading \textbf{'Addressing Consistency Issues'??} We also cite LaMDA along with ChatGPT, so a more generic heading. } +Based on the error analysis for the tasks performed above, the majority of the error categories can be attributed to the model's inability to maintain factual and logical consistency in the generated output. For the \mcq{} task, the lack of consistency is further demonstrated by the low contrastive accuracy on the task. +Conversation-based LLMs such as ChatGPT~\cite{chatgpt} or LaMDA~\cite{lamda}, have been shown to have both knowledge at the scale of LLMs such as GPT3 and an ability to incorporate human feedback for NLU tasks. +These capabilities may enable them to leverage feedback about inconsistencies (if detected) in the initially generated output to correct these inconsistencies in the subsequent generations. +However, when the task is to be performed at scale, the feedback that guides the model to the correct output needs to be automatically generated instead of a human guiding the model. As such, this type of model presents a fruitful and challenging research direction to address some of the issues and further improve performance on the tasks. +%On a positive note, most generated states $\alpha, \alpha'$ were about the same entity. +%, showing the models understand the basic format of the task. +% In most cases (i)(81\%) the model generates a state change but this change can not be inferred from the stories. Among this group of instances, there are cases (only a few) that (ii) generate the state change in the reverse order, meaning that $\alpha$ is inferred from $S'$ and $\alpha'$ from $S$. +% The other categories include the (iii) nonsensical generations (7.5\%), the ones with (iv) edits but are not a state change (5.7\%), as well as (v) direct copies from the input (4.6\%). +%You can see failure cases in \hyperref[tab:examples]{Table 2}. +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +% Scratch +% %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%% +% \begin{table*} +% \scriptsize +% \centering +% \begin{tabular}{p{8.1cm}p{5.4cm}p{0.8cm}} +% \hline +% \textbf{Input} & \textbf{Output} & \textbf{Category} \\ \hline +% \multicolumn{3}{c}{\textbf{\storyrev{}}} \\ \hline +% $S$: Al and Tara can't agree what to watch on TV. Al wants to watch soaps. Tara wants to watch the news. They play Paper, Rock, Scissors to decide. Tara wins and gets to watch what she wants. $\alpha'$: Al and Tara have two T.V. s. & Al and Tara can't agree what to watch on TV. Al wants to watch soaps. Tara wants to watch the tv. They play Paper, Rock, Scissors to decide. Tara wins and gets to watch what she wants. & Irrelevant change \\ +% \cline{1-3} +% $S$: Rich was a musician. He made a few hit songs. Rich had a lot of fans who cared about him. He developed cancer. Eventually the cancer took his life and rich was gone. $\alpha'$: Rich was unknown. & Rich was a musician. He made a few hit songs. Rich had no fans. He became ill and lost his life. Eventually Rich recovered and began his recovery.& Illogical revised story\\ +% \cline{1-3} +% % &$S$: Robert liked to play practical jokes. He decided to glue a quarter to the sidewalk. Robert laughed each time a person tried to pick up the quarter. At the end of the day he tried to pick up the quarter. He glued it so well, he couldn't pick it up. $\alpha'$: \hl{The glue was weak.} & Robert liked to play practical jokes. He decided to glue a quarter to the sidewalk. Robert laughed each time a person tried to pick up the quarter. At the end of the day he tried to pick up the quarter. \hl{He glued it so well}, but when it dried out he could only pretend it wasn't there. & Contradiction\\ +% $S$: There was a large car wreck about two miles from the office. +% I was waiting for a co worker to relieve me. +% However, he was very late and I had no idea why. +% He called me a few minutes later. +% Apparently he was stuck in traffic because of the wreck. $\alpha'$: \hl{His co worker did not have access to a phone.} & There was a large car wreck about two miles from the office. +% I was waiting for a co worker to relieve me. +% However, he was very late and I had no idea why. +% \hl{He called me on the phone} a few minutes later but I had no idea why. +% Apparently he was stuck in traffic because of the wreck. & Contradiction\\ +% \hline +% \multicolumn{3}{c}{\textbf{\statechange{}}} \\ \hline +% $S$: Pat was looking for an interesting place to vacation. He searched various catalogs. Eventually he found the perfect destination. There he stayed in a small hut. It was one of the greatest experiences of his life. $S'$: Pat was looking for a relaxing place to vacation. He searched various catalogs. Eventually he found the perfect destination. There he stayed in a luxury resort. +% It was one of the greatest experiences of his life. & \makecell[l]{$\alpha$: Pat likes to be independent. \\ $\alpha'$: Pat likes to be in a big city.} & Irrelevant states\\ +% \cline{1-3} +% $S$: Craig was diagnosed with cancer. He decided to fight it. He tried several different approaches and medications. Eventually it went into remission. Craig and his loved ones were thrilled. $S'$: Craig was diagnosed with cancer. He decided to fight it. He tried several different approaches and medications. Eventually it spread. Craig and his loved ones were devastated. & \makecell[l]{$\alpha$: Craig is not terminal cancer. \\$\alpha'$: Craig is terminal cancer.} & Illogical state change\\ +% \cline{1-3} +% % & $S$: Steve wanted to switch Football positions because he wasn't satisfied. He wasn't sure what new position he wanted to play so he asked us. We all didn't really know either, so he was stuck. Finally he asked his coach what he should play. His coach gave him a good recommendation and he switched to that! $S'$: Steve wanted to switch Football positions because he wasn't satisfied. He wasn't sure what new position he wanted to play so he asked us. +% % We all didn't really know either, so he was stuck. \hl{Finally he asked his coach what he should play}. His coach told him to figure it out alone! & \makecell[l]{ $\alpha$: Steve is willing to listen to his coach \\ $\alpha'$: \hl{Steve is not willing to listen to his coach}} & Contradiction \\ +% % \cline{2-4} +% $S$: Rita accidentally left her smartphone on the plane. She felt terrible about this and knew her parents would be mad. Nobody ever turned in the phone, leaving her disappointed with people. \hl{Rita had to work many hours to buy a new phone as a replacement.} She is significantly more careful with her belongings now. $S'$: Rita accidentally left her smartphone on the plane. +% She felt terrible about this and knew her parents would be mad. +% Nobody ever turned in the phone, leaving her disappointed with people. +% Rita quickly bought a new phone as a replacement. +% She is significantly more careful with her belongings now. & \makecell[l]{ $\alpha$: \hl{Rita was unemployed} \\ $\alpha'$: Rita was rich} & Contradiction \\ +% \hline +% \end{tabular} +% \label{tab:examples} +% \caption{Representative examples of top failure categories for different tasks. For the contradiction category, the conflicting parts are \hl{highlighted}.} +% \end{table*} +\section{Conclusion} +% \subsection{Future Work \sg{added} } +% % \sg{added as per suggestion from R3 - may need rephrasing} +% Large models are helping to scale many knowledge acquisition bottlenecks, which has led to many advances in reasoning about narratives. Their ability to do fewshot learning makes them suitable for unseen tasks in a low resource setting. Therefore prompt design is a potential direction worth pursuing. +% % (i) incontext learning with structure story representation +% All the proposed tasks requires models to have understand the underlying event-state and event-event causal structure in the narrative. A potential direction is to study the structured narrative representation as input for fewshot learning. +% % (ii) retrival augmented LMs +% The three proposed tasks also requires a certain amount of factual knowledge about the world. As such augmenting models with the ability to retrieve relevant factual information may be beneficial for knowledge intensive task instances for all the three tasks. +% % (iii) Generating logically consistent/coherent text +% Reranking approaches for text generation with scoring functions that captures the important properties of the desired output may help alleviate some of the generation inconsistency, common in generative tasks. +% % (iii) Knowledge distillation +% Distilling knowledge from other commonsense models might provide some external and complimentary knowledge useful for the main task, and therefore might lead to fruitful research outcomes. +% +% Reframing the proposed tasks into other commonsense task with existing systems can help provide model with external feedback which can be used to impose some consistency constrain. This approach can help distill external knowledge that is gained from other commonsense datasets and is a direction worth investigating. +% \subsection{Conclusion} +%Large models are helping to scale many knowledge acquisition bottlenecks, which has led to many advances in reasoning about narratives. +In this work, we introduced a new resource, \dataset{}, that captures unstated commonsense knowledge required to understand and reason about participant states in a narrative. \dataset{} opens the door to developing more complex reasoning abilities, especially those that require access to implicit information. We described three \dataset{} reasoning tasks, one classification and two generation, that test for different aspects of state-based reasoning. This work shows that with careful crowdsourcing and contrastive design we can obtain a high-quality dataset that can be used to evaluate deeper reasoners. Benchmarking results suggest that \dataset{} tasks are not within the reach of current large sized models, as of yet, and encourages future research in modeling commonsense knowledge with states. +% We will release \dataset{}, UIs, the codebase, and trained models upon publication. + +%In this paper, we introduced \dataset{}, a challenge dataset that captures the commonsense knowledge required to understand and reason about participant states in a narrative. %We did a stringent quality control to ensure the quality of the collected data. +%To test the ability to understand and reason about states in a narrative, we proposed a classification task \textbf{\mcq{}}, and two generative tasks \textbf{\storyrev{}} and \textbf{\statechange{}}, each testing for different aspect of state-based reasoning. + +%We benchmarked the current state of art LLLMs, GPT3, T5-base, and T5-large on these tasks. +%For the generative tasks, we did an extensive human evaluation of the model performance and error analysis of the model-generated output. +%Benchmarking experiments showed that (i) for the \textbf{\mcq{}} task, which is a binary classification task, the best model, i.e. T5-large has an accuracy of $\sim 79\%$ (ii) for \textbf{\storyrev{}} task, only $\sim 45 \%$ of the best performing model's (T5-large) output are acceptable (iii) for \textbf{\statechange{}{}} task, only $\sim 49 \%$ of the best performing model's (GPT3) output are acceptable. +%Through the ablation analysis for the \mcq{} task, we also showed the challenges associated with constructing negative examples for these kinds of tasks. +%Overall this is a challenging dataset, which enough room for improvement on the proposed tasks, to encourage potential research efforts in the domain of states-based commonsense knowledge. + +\section*{Acknowledgments} +We would like to thank the anonymous reviewers for their comments, questions, and suggestions. % +This material is also based on research that is in part supported by the NSF, Grant No. 2007290, Army Research Laboratory, Grant No. W911NF2120076, and by the Air Force Research Laboratory (AFRL), DARPA, for the KAIROS program under agreement number FA8750-19-2-1003. The U.S. Government is authorized to reproduce and distribute reprints for Governmental purposes notwithstanding any copyright notation thereon. The views and conclusions contained herein are those of the authors and should not be interpreted as necessarily representing the official policies or endorsements, either express or implied, of the Air Force Research Laboratory (AFRL), DARPA, or the U.S. Government. % +This material is based in part upon work supported by the National Science Foundation under Grant No. IIS-2024878. % +% % WARNING: missing file /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785099279_8df5ecbb/processed_papers/2208.00329/tex_2208.00329/11_appendix.tex + +% =========================================================================================== +% Bibliography +% =========================================================================================== +\bibliography{tacl2021} +\bibliographystyle{acl_natbib} + +\iftaclpubformat + +\onecolumn + +% --------------------------------- +% \appendix +% --------------------------------- +% \section{Author/Affiliation Options as set forth by MIT Press} +% \label{sec:authorformatting} + +% Option 1. Author’s address is underneath each name, centered. + +% \begin{quote}\centering +% \begin{tabular}{c} +% \textbf{First Author} \\ +% First Affiliation \\ +% First Address 1 \\ +% First Address 2 \\ +% \texttt{first.email@example.com} +% \end{tabular} +% \ +% \begin{tabular}{c} +% \textbf{Second Author} \\ +% Second Affiliation \\ +% Second Address 1 \\ +% Second Address 2 \\ +% \texttt{second.email@example.com} +% \end{tabular} + +% \begin{tabular}{c} +% \textbf{Third Author} \\ +% Third Affiliation \\ +% Third Address 1 \\ +% Third Address 2 \\ +% \texttt{third.email@example.com} +% \end{tabular} +% \end{quote} + + +% Option 2. Author’s address is linked with superscript characters to its name, +% author names are grouped, centered. + +% \begin{quote}\centering +% \textbf{First Author$^\diamond$} \quad \textbf{Second Author$^\dagger$} \quad +% \textbf{Third Author$^\ddagger$} +% \\ \ \\ +% $^\diamond$First Affiliation \\ +% First Address 1 \\ +% First Address 2 \\ +% \texttt{first.email@example.com} +% \\ \ \\ +% $^\dagger$Second Affiliation \\ +% Second Address 1 \\ +% Second Address 2 \\ +% \texttt{second.email@example.com} +% \\ \ \\ +% $^\ddagger$Third Affiliation \\ +% Third Address 1 \\ +% Third Address 2 \\ +% \texttt{third.email@example.com} +% \end{quote} +% --------------------------------- + +\fi + +\end{document} + +BIBLIOGRAPHY: +[ + { + "ref_id": "b3848d32f7294ec708627897833c4097eb4d8778", + "title": "LaMDA: Language Models for Dialog Applications", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "journals/corr/abs-2201-08239", + "ArXiv": "2201.08239", + "CorpusId": 246063428 + } + }, + { + "ref_id": "7428db171b44660c5cde74faa222be952d00caef", + "title": "Automated Fake News Detection using cross-checking with reliable sources", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "journals/corr/abs-2201-00083", + "ArXiv": "2201.00083", + "CorpusId": 245650712 + } + }, + { + "ref_id": "521ccc898395a2818fced22b4cf371b0e5121f94", + "title": "Symbolic Knowledge Distillation: from General Language Models to Commonsense Models", + "authors": null, + "year": null, + "external_ids": { + "ACL": "2022.naacl-main.341", + "DBLP": "journals/corr/abs-2110-07178", + "ArXiv": "2110.07178", + "DOI": "10.18653/v1/2022.naacl-main.341", + "CorpusId": 238857304 + } + }, + { + "ref_id": "59641c10ed7431a3cf841f308367dc2dc0281b74", + "title": "What Makes Good In-Context Examples for GPT-3?", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "conf/acl-deelio/LiuSZDCC22", + "ArXiv": "2101.06804", + "ACL": "2022.deelio-1.10", + "DOI": "10.18653/v1/2022.deelio-1.10", + "CorpusId": 231632658 + } + }, + { + "ref_id": "2bc77cdfa44bdea93cee56f8a4eaae60ddc195e1", + "title": "An Analysis of Natural Language Inference Benchmarks through the Lens of Negation", + "authors": null, + "year": null, + "external_ids": { + "MAG": "3099843385", + "DBLP": "conf/emnlp/HossainKDKWB20", + "ACL": "2020.emnlp-main.732", + "DOI": "10.18653/v1/2020.emnlp-main.732", + "CorpusId": 226262357 + } + }, + { + "ref_id": "d47ad0a606bedf41dcea614bfa7b7494879c7ba0", + "title": "A Dataset for Tracking Entities in Open Domain Procedural Text", + "authors": null, + "year": null, + "external_ids": { + "ArXiv": "2011.08092", + "DBLP": "conf/emnlp/TandonSDRCGRH20", + "MAG": "3101605968", + "ACL": "2020.emnlp-main.520", + "DOI": "10.18653/v1/2020.emnlp-main.520", + "CorpusId": 226262266 + } + }, + { + "ref_id": "a66389733c6090baf3fc07f950afecf122ab3420", + "title": "Cue Me In: Content-Inducing Approaches to Interactive Story Generation", + "authors": null, + "year": null, + "external_ids": { + "ACL": "2020.aacl-main.59", + "DBLP": "conf/ijcnlp/BrahmanPC20", + "MAG": "3094003880", + "ArXiv": "2010.09935", + "DOI": "10.18653/v1/2020.aacl-main.59", + "CorpusId": 224802980 + } + }, + { + "ref_id": "f8a22859230e0ccafefc020dccc66b5a646fe0ac", + "title": "COMET-ATOMIC 2020: On Symbolic and Neural Commonsense Knowledge Graphs", + "authors": null, + "year": null, + "external_ids": { + "ArXiv": "2010.05953", + "DBLP": "journals/corr/abs-2010-05953", + "MAG": "3093166897", + "DOI": "10.1609/aaai.v35i7.16792", + "CorpusId": 222310337 + } + }, + { + "ref_id": "3065d646367f0d894bfb65b7a3a01b448db20b4f", + "title": "GLUCOSE: GeneraLized and COntextualized Story Explanations", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "journals/corr/abs-2009-07758", + "ArXiv": "2009.07758", + "ACL": "2020.emnlp-main.370", + "MAG": "3098323839", + "DOI": "10.18653/v1/2020.emnlp-main.370", + "CorpusId": 221739295 + } + }, + { + "ref_id": "90abbc2cf38462b954ae1b772fac9532e2ccd8b0", + "title": "Language Models are Few-Shot Learners", + "authors": null, + "year": null, + "external_ids": { + "ArXiv": "2005.14165", + "DBLP": "conf/nips/BrownMRSKDNSSAA20", + "MAG": "3030163527", + "CorpusId": 218971783 + } + }, + { + "ref_id": "44a5d4b7ab16910c101755b0d7d78594f6c58bb3", + "title": "Computational Linguistics: 16th International Conference of the Pacific Association for Computational Linguistics, PACLING 2019, Hanoi, Vietnam, October 11\u201313, 2019, Revised Selected Papers", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "conf/pacling/2019", + "DOI": "10.1007/978-981-15-6168-9", + "CorpusId": 195348644 + } + }, + { + "ref_id": "6c4b76232bb72897685d19b3d264c6ee3005bc2b", + "title": "Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2981852735", + "DBLP": "journals/corr/abs-1910-10683", + "ArXiv": "1910.10683", + "CorpusId": 204838007 + } + }, + { + "ref_id": "4d16457cded23bce6eaa91cd17aefd22af2279f0", + "title": "Counterfactual Story Reasoning and Generation", + "authors": null, + "year": null, + "external_ids": { + "ArXiv": "1909.04076", + "DBLP": "conf/emnlp/QinBHBCC19", + "ACL": "D19-1509", + "MAG": "2970453125", + "DOI": "10.18653/v1/D19-1509", + "CorpusId": 202542404 + } + }, + { + "ref_id": "b05e9bd414144a3cdac2ec59a54e11647181ca19", + "title": "WIQA: A dataset for \u201cWhat if...\u201d reasoning over procedural text", + "authors": null, + "year": null, + "external_ids": { + "ArXiv": "1909.04739", + "MAG": "2973230076", + "ACL": "D19-1629", + "DBLP": "conf/emnlp/TandonDSCB19", + "DOI": "10.18653/v1/D19-1629", + "CorpusId": 202558452 + } + }, + { + "ref_id": "a550f576ff20b8cce98f3ddad0043d3783fbc9b4", + "title": "Abductive Commonsense Reasoning", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "conf/iclr/BhagavatulaBMSH20", + "MAG": "2968629361", + "ArXiv": "1908.05739", + "CorpusId": 201058651 + } + }, + { + "ref_id": "5e35895fc4731858f0b286cb5a1613a819cc2367", + "title": "CLUTRR: A Diagnostic Benchmark for Inductive Reasoning from Text", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2968398601", + "ArXiv": "1908.06177", + "ACL": "D19-1458", + "DBLP": "journals/corr/abs-1908-06177", + "DOI": "10.18653/v1/D19-1458", + "CorpusId": 198184456 + } + }, + { + "ref_id": "077f8329a7b6fa3b7c877a57b81eb6c18b5f87de", + "title": "RoBERTa: A Robustly Optimized BERT Pretraining Approach", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "journals/corr/abs-1907-11692", + "ArXiv": "1907.11692", + "MAG": "2965373594", + "CorpusId": 198953378 + } + }, + { + "ref_id": "295065d942abca0711300b2b4c39829551060578", + "title": "BERTScore: Evaluating Text Generation with BERT", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2936695845", + "ArXiv": "1904.09675", + "DBLP": "journals/corr/abs-1904-09675", + "CorpusId": 127986044 + } + }, + { + "ref_id": "dfbbb0338123de44ef3c0bdc05e9785d20d430cc", + "title": "Repurposing Entailment for Multi-Hop Question Answering Tasks", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2955488563", + "ArXiv": "1904.09380", + "DBLP": "conf/naacl/TrivediKKSB19", + "ACL": "N19-1302", + "DOI": "10.18653/v1/N19-1302", + "CorpusId": 128344862 + } + }, + { + "ref_id": "85ff214ac614b5163279431941b79927fb72433c", + "title": "Plan, Write, and Revise: an Interactive System for Open-Domain Story Generation", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2929325298", + "ArXiv": "1904.02357", + "DBLP": "journals/corr/abs-1904-02357", + "ACL": "N19-4016", + "DOI": "10.18653/v1/N19-4016", + "CorpusId": 102354918 + } + }, + { + "ref_id": "ac5288519c0553dedcb5684d6da10f58bf9aa15b", + "title": "Can Machines Learn to Detect Fake News? A Survey Focused on Social Media", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2915332473", + "DBLP": "conf/hicss/SilvaVG19", + "DOI": "10.24251/HICSS.2019.332", + "CorpusId": 85536497 + } + }, + { + "ref_id": "1dc04035b9926c46ded436e5762f3924ab29516e", + "title": "Dialogue Natural Language Inference", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2949076949", + "ArXiv": "1811.00671", + "DBLP": "conf/acl/WelleckWSC19", + "ACL": "P19-1363", + "DOI": "10.18653/v1/P19-1363", + "CorpusId": 53298765 + } + }, + { + "ref_id": "86b84f603390615627b71d4b3679dc6e3f2ca28b", + "title": "How Journalists and Social Media Users Perceive Online Fact-Checking and Verification Services", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2753117207", + "DOI": "10.1080/17512786.2017.1363657", + "CorpusId": 149194871 + } + }, + { + "ref_id": "0c2790d4894940a5cf9084b09788a6c65617c209", + "title": "UKP-Athene: Multi-Sentence Textual Entailment for Claim Verification", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "journals/corr/abs-1809-01479", + "ACL": "W18-5516", + "MAG": "2892163801", + "ArXiv": "1809.01479", + "DOI": "10.18653/v1/W18-5516", + "CorpusId": 52162540 + } + }, + { + "ref_id": "8cd595ee1d8b8ff5716f74b9223d736d21d12ce1", + "title": "TwoWingOS: A Two-Wing Optimization Strategy for Evidential Claim Verification", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2886390638", + "DBLP": "conf/emnlp/0001R18", + "ArXiv": "1808.03465", + "ACL": "D18-1010", + "DOI": "10.18653/v1/D18-1010", + "CorpusId": 51969223 + } + }, + { + "ref_id": "5e9c9d0164ae041786f8fdc5726da12403e91a6c", + "title": "Tracking State Changes in Procedural Text: a Challenge Dataset and Models for Process Paragraph Comprehension", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2804447833", + "DBLP": "conf/naacl/DalviHTYC18", + "ACL": "N18-1144", + "ArXiv": "1805.06975", + "DOI": "10.18653/v1/N18-1144", + "CorpusId": 5019682 + } + }, + { + "ref_id": "7447ac643e8d4f6d958c492500df08f679ad9c97", + "title": "Modeling Naive Psychology of Characters in Simple Commonsense Stories", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2798589784", + "DBLP": "journals/corr/abs-1805-06533", + "ACL": "P18-1213", + "ArXiv": "1805.06533", + "DOI": "10.18653/v1/P18-1213", + "CorpusId": 21689288 + } + }, + { + "ref_id": "d07284a6811f1b2745d91bdb06b040b57f226882", + "title": "Decoupled Weight Decay Regularization", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2950541952", + "DBLP": "conf/iclr/LoshchilovH19", + "CorpusId": 53592270 + } + }, + { + "ref_id": "26953cc3d09920b54071b73866f85d6bb1a6184c", + "title": "Simulating Action Dynamics with Neural Process Networks", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2952278540", + "DBLP": "conf/iclr/BosselutLHEFC18", + "ArXiv": "1711.05313", + "CorpusId": 31816657 + } + }, + { + "ref_id": "26aa6fe2028b5eefbaa40ab54ef725bbbe7d9810", + "title": "ConceptNet 5.5: An Open Multilingual Graph of General Knowledge", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2561529111", + "DBLP": "conf/aaai/SpeerCH17", + "ArXiv": "1612.03975", + "DOI": "10.1609/aaai.v31i1.11164", + "CorpusId": 15206880 + } + }, + { + "ref_id": "c6850869aa5e78a107c378d2e8bfa39633158c0c", + "title": "Google's Neural Machine Translation System: Bridging the Gap between Human and Machine Translation", + "authors": null, + "year": null, + "external_ids": { + "ArXiv": "1609.08144", + "MAG": "2525778437", + "DBLP": "journals/corr/WuSCLNMKCGMKSJL16", + "CorpusId": 3603249 + } + }, + { + "ref_id": "85b68477a6e031d88b963833e15a4b4fc6855264", + "title": "A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2891014615", + "DBLP": "conf/naacl/MostafazadehCHP16", + "ACL": "N16-1098", + "ArXiv": "1604.01696", + "DOI": "10.18653/v1/N16-1098", + "CorpusId": 1726501 + } + }, + { + "ref_id": "a477778d8ae4cb70b67d02d094d030c44d0c6ffd", + "title": "Joint Learning Templates and Slots for Event Schema Induction", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2294504743", + "DBLP": "conf/naacl/ShaLCS16", + "ACL": "N16-1049", + "ArXiv": "1603.01333", + "DOI": "10.18653/v1/N16-1049", + "CorpusId": 5741899 + } + }, + { + "ref_id": "ae7221f4731570b8ebe763000f981c7e5b1664a8", + "title": "A Unified Bayesian Model of Scripts, Frames and Language", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "conf/aaai/FerraroD16", + "MAG": "2563017805", + "DOI": "10.1609/aaai.v30i1.10328", + "CorpusId": 10036335 + } + }, + { + "ref_id": "42b49971a1bc73088ec3317222e7486db82bb4cd", + "title": "Handbook of Inter-Rater Reliability: The Definitive Guide to Measuring the Extent of Agreement Among Raters", + "authors": null, + "year": null, + "external_ids": { + "MAG": "613277099", + "CorpusId": 142385324 + } + }, + { + "ref_id": "f40219d81b785bddc7fdad9c2af2840af62661ae", + "title": "Generating Coherent Event Schemas at Scale", + "authors": null, + "year": null, + "external_ids": { + "ACL": "D13-1178", + "DBLP": "conf/emnlp/BalasubramanianSME13", + "MAG": "2252139350", + "DOI": "10.18653/v1/d13-1178", + "CorpusId": 2578382 + } + }, + { + "ref_id": "13167f9cd8c7906ca808b01d28dca6dd951da8a5", + "title": "The ACL Anthology Network Corpus as a Resource for NLP-based Bibliometrics", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2941116076", + "CorpusId": 15641339 + } + }, + { + "ref_id": "8891481c3fa28d95acc9f2c50187ce682a8130e7", + "title": "A comparison of Cohen\u2019s Kappa and Gwet\u2019s AC1 when calculating inter-rater reliability coefficients: a study conducted with personality disorder samples", + "authors": null, + "year": null, + "external_ids": { + "MAG": "1996445490", + "PubMedCentral": "3643869", + "DOI": "10.1186/1471-2288-13-61", + "CorpusId": 2913866, + "PubMed": "23627889" + } + }, + { + "ref_id": "f6c9197a3bd26af0fe6a199e95d0336f4d958e1c", + "title": "Development of the RTI item bank on risk of bias and precision of observational studies.", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2003182613", + "DOI": "10.1016/j.jclinepi.2011.05.008", + "CorpusId": 28627374, + "PubMed": "21959223" + } + }, + { + "ref_id": "62808970f3eb55738ca796d46d5aa67a47daa9ee", + "title": "Unsupervised Learning of Narrative Schemas and their Participants", + "authors": null, + "year": null, + "external_ids": { + "ACL": "P09-1068", + "MAG": "2158794898", + "DBLP": "conf/acl/ChambersJ09", + "DOI": "10.3115/1690219.1690231", + "CorpusId": 10299779 + } + }, + { + "ref_id": "74f61af390292fc197659ae698429df4a2de62df", + "title": "Unsupervised Learning of Narrative Event Chains", + "authors": null, + "year": null, + "external_ids": { + "ACL": "P08-1090", + "DBLP": "conf/acl/ChambersJ08", + "MAG": "2151295812", + "CorpusId": 529375 + } + }, + { + "ref_id": "087be7b3d737ee817b64246e7c9a4fb6cd57dc24", + "title": "Computing inter-rater reliability and its variance in the presence of high agreement.", + "authors": null, + "year": null, + "external_ids": { + "MAG": "1996290308", + "DOI": "10.1348/000711006X126600", + "CorpusId": 13915043, + "PubMed": "18482474" + } + }, + { + "ref_id": "a2c54a729be2e2f46c446cabe70fdb3d38d2a1c6", + "title": "Methods for Using Textual Entailment in Open-Domain Question Answering", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2112644606", + "ACL": "P06-1114", + "DBLP": "conf/acl/HarabagiuH06", + "DOI": "10.3115/1220175.1220289", + "CorpusId": 1614922 + } + }, + { + "ref_id": "60b05f32c32519a809f21642ef1eb3eaf3848008", + "title": "ROUGE: A Package for Automatic Evaluation of Summaries", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2154652894", + "ACL": "W04-1013", + "CorpusId": 964287 + } + }, + { + "ref_id": "d7da009f457917aa381619facfa5ffae9329a6e9", + "title": "Bleu: a Method for Automatic Evaluation of Machine Translation", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "conf/acl/PapineniRWZ02", + "MAG": "2101105183", + "ACL": "P02-1040", + "DOI": "10.3115/1073083.1073135", + "CorpusId": 11080756 + } + }, + { + "ref_id": "109cd7af6a209808cbe5d3e17756451b4ee97ef9", + "title": "Bias, prevalence and kappa.", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2070285512", + "DOI": "10.1016/0895-4356(93)90018-V", + "CorpusId": 39184178, + "PubMed": "8501467" + } + }, + { + "ref_id": "ee1cd8b38356220809c9a216d017fe6148fb8f76", + "title": "A Critical Examination of Allen's Theory of Action and Time", + "authors": null, + "year": null, + "external_ids": { + "MAG": "3030473944", + "DBLP": "books/ox/05/Galton05", + "DOI": "10.1016/0004-3702(90)90053-3", + "CorpusId": 2606878 + } + }, + { + "ref_id": "df2b0e26d0599ce3e70df8a9da02e51594e0e992", + "title": "BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2951055169", + "ACL": "N19-1423", + "DBLP": "journals/corr/abs-1810-04805", + "ArXiv": "1810.04805", + "DOI": "10.18653/v1/N19-1423", + "CorpusId": 52967399 + } + }, + { + "ref_id": null, + "title": "2019. BERT: Pre-training", + "authors": null, + "year": null, + "external_ids": {} + }, + { + "ref_id": null, + "title": "Atomic: An atlas of machine commonsense for if-then reasoning", + "authors": null, + "year": null, + "external_ids": {} + }, + { + "ref_id": "008ccee6ce19f1eaf47e0a8ae59e534155152cb3", + "title": "The current state of fake news", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2772792254", + "DOI": "10.1016/J.PROCS.2017.11.106", + "CorpusId": 64678037 + } + }, + { + "ref_id": "b7d35bc87f101453437a72210d62fc25caa49634", + "title": "Association for the Advancement of Artificial Intelligence", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2782823992", + "CorpusId": 63623391 + } + }, + { + "ref_id": "6bcda3451017c5acae4b0ad1d05515dab87b6916", + "title": "FRAME SEMANTICS", + "authors": null, + "year": null, + "external_ids": { + "DOI": "10.1075/hop.2.fra1", + "CorpusId": 14980012 + } + }, + { + "ref_id": "ad8ba0c6e3ff4fceef3a70189d53ee3a2e1e7a72", + "title": "High agreement but low kappa: I. The problems of two paradoxes.", + "authors": null, + "year": null, + "external_ids": { + "MAG": "1983897914", + "DOI": "10.1016/0895-4356(90)90158-L", + "CorpusId": 30798014, + "PubMed": "2348207" + } + }, + { + "ref_id": "4165982754f451cc4c9b1e48505384546a2f3e36", + "title": "Frames and the semantics of understanding", + "authors": null, + "year": null, + "external_ids": { + "MAG": "1569430537", + "CorpusId": 60712156 + } + }, + { + "ref_id": "1efcd7be3b52e46de800e06e17268ce7d535d9a7", + "title": "Scripts, Plans and Knowledge", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "conf/ijcai/SchenkA75", + "CorpusId": 18113275 + } + }, + { + "ref_id": null, + "title": "2601\u20132607, Phoenix, Arizona", + "authors": null, + "year": null, + "external_ids": {} + }, + { + "ref_id": null, + "title": "ev-idential claim veri\ufb01cation", + "authors": null, + "year": null, + "external_ids": {} + }, + { + "ref_id": null, + "title": "OpenAI", + "authors": null, + "year": null, + "external_ids": {} + } +] + +CITATION CONTEXTS: +[ + { + "ref_id": "b3848d32f7294ec708627897833c4097eb4d8778", + "citation_marker": "LaMDA: Language Models for Dialog Applications", + "text": "Conversation-based LLMs such as ChatGPT (OpenAI, 2022) or LaMDA (Thoppilan et al., 2022), have been shown to have both knowledge at the scale of LLMs such as GPT3 and an ability to incorporate human feedback for NLU tasks.", + "section": null, + "intents": [] + }, + { + "ref_id": "b3848d32f7294ec708627897833c4097eb4d8778", + "citation_marker": "LaMDA: Language Models for Dialog Applications", + "text": "Conversationbased LLMs such as ChatGPT (OpenAI, 2022) or LaMDA (Thoppilan et al., 2022), have been shown to have both knowledge at the scale of LLMs such as GPT3 and an ability to incorporate human feedback for NLU tasks.", + "section": null, + "intents": [ + "methodology" + ] + }, + { + "ref_id": "59641c10ed7431a3cf841f308367dc2dc0281b74", + "citation_marker": "What Makes Good In-Context Examples for GPT-3?", + "text": "(i) EXPERT CURATED: We selected a fixed set of diverse, unambiguous examples that requires multi-step reasoning and covers different type of states, and used the same prompt examples for all the query instances, (ii) RANDOM SELECTION: We randomly selected examples, (iii) NEAREST NEIGHBOR (Liu et al., 2022): For each query instance, we selected examples that were most similar to it.", + "section": null, + "intents": [ + "methodology" + ] + }, + { + "ref_id": "f8a22859230e0ccafefc020dccc66b5a646fe0ac", + "citation_marker": "COMET-ATOMIC 2020: On Symbolic and Neural Commonsense Knowledge Graphs", + "text": "More recent work by Speer et al. (2017), Sap et al. (2019), and Hwang et al. (2021) capture everyday inferential knowledge associated with an action performed by someone.", + "section": null, + "intents": [ + "background" + ] + }, + { + "ref_id": "90abbc2cf38462b954ae1b772fac9532e2ccd8b0", + "citation_marker": "Language Models are Few-Shot Learners", + "text": "Statistical work on events (Chambers and Jurafsky, 2008; Chambers and Jurafsky, 2009; Balasubramanian et al., 2013; Ferraro and Van Durme, 2016; Sha et al., 2016) curated event knowledge in an unsupervised manner from large text corpora.", + "section": null, + "intents": [ + "background" + ] + }, + { + "ref_id": "6c4b76232bb72897685d19b3d264c6ee3005bc2b", + "citation_marker": "Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer", + "text": "To evaluate current models for these capabilities, we benchmark the LLMs T5 (Raffel et al., 2020), BERT (Devlin et al., 2019), RoBERTa (Liu et al., 2019), and GPT3 (Brown et al., 2020).", + "section": null, + "intents": [ + "methodology" + ] + }, + { + "ref_id": "4d16457cded23bce6eaa91cd17aefd22af2279f0", + "citation_marker": "Counterfactual Story Reasoning and Generation", + "text": "Most similar to our work is the TIME-TRAVEL dataset by Qin et al. (2019).", + "section": null, + "intents": [ + "methodology" + ] + }, + { + "ref_id": "4d16457cded23bce6eaa91cd17aefd22af2279f0", + "citation_marker": "Counterfactual Story Reasoning and Generation", + "text": "These types of counterfactual revisions serve as a test of reasoning (Qin et al., 2019) and can support interactive story generation tasks (Goldfarb-Tarrant et al., 2019; Brahman et al., 2020).", + "section": null, + "intents": [ + "background" + ] + }, + { + "ref_id": "b05e9bd414144a3cdac2ec59a54e11647181ca19", + "citation_marker": "WIQA: A dataset for \u201cWhat if...\u201d reasoning over procedural text", + "text": "Tandon et al. (2019) introduced the WIQA dataset for analyzing the effect of perturbing a process described by a procedural text on the elements (entities, events, etc.) of the text, as an influence graph of the process.", + "section": null, + "intents": [ + "methodology" + ] + }, + { + "ref_id": "5e35895fc4731858f0b286cb5a1613a819cc2367", + "citation_marker": "CLUTRR: A Diagnostic Benchmark for Inductive Reasoning from Text", + "text": "\u2026a challenging task for models, and to a certain extent can be attributed to their tendency to forget attributes of specific entities (Welleck et al., 2018), ignore previously inferred facts (Sinha et al., 2019) and background information, or contradict previous statements (Brown et al., 2020).", + "section": null, + "intents": [ + "result" + ] + }, + { + "ref_id": "077f8329a7b6fa3b7c877a57b81eb6c18b5f87de", + "citation_marker": "RoBERTa: A Robustly Optimized BERT Pretraining Approach", + "text": "We benchmarked base (BERT-b) and large (BERT-l) variants of BERT-uncased, and base (RoBERTa-b) and large (RoBERTa-l) of RoBERTa on only the Story State Inference task since they are non-generative models.", + "section": null, + "intents": [] + }, + { + "ref_id": "077f8329a7b6fa3b7c877a57b81eb6c18b5f87de", + "citation_marker": "RoBERTa: A Robustly Optimized BERT Pretraining Approach", + "text": "8% ) when comparing the best performing model (RoBERTa-l) to humans on this simple binary classification task.", + "section": null, + "intents": [] + }, + { + "ref_id": "077f8329a7b6fa3b7c877a57b81eb6c18b5f87de", + "citation_marker": "RoBERTa: A Robustly Optimized BERT Pretraining Approach", + "text": "For this, we computed the cosine similarity between the [CLS] representation of the instances obtained from RoBERTa-large fine-tuned on the Story State Inference task.", + "section": null, + "intents": [] + }, + { + "ref_id": "077f8329a7b6fa3b7c877a57b81eb6c18b5f87de", + "citation_marker": "RoBERTa: A Robustly Optimized BERT Pretraining Approach", + "text": "4% (RoBERTa-l) to 9 .", + "section": null, + "intents": [] + }, + { + "ref_id": "077f8329a7b6fa3b7c877a57b81eb6c18b5f87de", + "citation_marker": "RoBERTa: A Robustly Optimized BERT Pretraining Approach", + "text": "To establish modern baselines and measure their performance, we built benchmark models from GPT3, T5, BERT, and RoBERTa.", + "section": null, + "intents": [] + }, + { + "ref_id": "077f8329a7b6fa3b7c877a57b81eb6c18b5f87de", + "citation_marker": "RoBERTa: A Robustly Optimized BERT Pretraining Approach", + "text": "To evaluate current models for these capabilities, we benchmark the LLMs T5 (Raffel et al., 2020), BERT (Devlin et al., 2019), RoBERTa (Liu et al., 2019), and GPT3 (Brown et al., 2020).", + "section": null, + "intents": [ + "methodology" + ] + }, + { + "ref_id": "077f8329a7b6fa3b7c877a57b81eb6c18b5f87de", + "citation_marker": "RoBERTa: A Robustly Optimized BERT Pretraining Approach", + "text": "T5-b, BERT-b, BERT-l, RoBERTa-b, and RoBERTa-l were trained on an NVIDIA-TITAN-X 24GB, and T5-l was trained on an NVIDIA-A6000 48GB GPU.", + "section": null, + "intents": [] + }, + { + "ref_id": "077f8329a7b6fa3b7c877a57b81eb6c18b5f87de", + "citation_marker": "RoBERTa: A Robustly Optimized BERT Pretraining Approach", + "text": "7% (BERT) to 8% (RoBERTa) gains on standard accuracy.", + "section": null, + "intents": [] + }, + { + "ref_id": "295065d942abca0711300b2b4c39829551060578", + "citation_marker": "BERTScore: Evaluating Text Generation with BERT", + "text": "However, since the correlation is low, we strongly recommend using human evaluations, and only use BERTscore as an alternative where human evaluation is expensive.", + "section": null, + "intents": [] + }, + { + "ref_id": "295065d942abca0711300b2b4c39829551060578", + "citation_marker": "BERTScore: Evaluating Text Generation with BERT", + "text": "We use human judgments to evaluate the revised stories because reference-based automatic evaluation metrics (BLEU [Papineni et al., 2002], BERTscore [Zhang et al., 2019] etc.) are inadequate for multiple reasons: (i) valid revised stories often exist that are different from the references, (ii) original and revised stories overlap heavily which can skew the metrics, and (iii) small lexical changes that don\u2019t change automatic metrics can affect logical consistency.", + "section": null, + "intents": [] + }, + { + "ref_id": "295065d942abca0711300b2b4c39829551060578", + "citation_marker": "BERTScore: Evaluating Text Generation with BERT", + "text": "However, since human evaluation is expensive, we include here the results from three automatic metrics: GLEU (Wu et al., 2016), ROUGE (Lin, 2004), and BERTscore (Zhang et al., 2019).", + "section": null, + "intents": [ + "methodology" + ] + }, + { + "ref_id": "295065d942abca0711300b2b4c39829551060578", + "citation_marker": "BERTScore: Evaluating Text Generation with BERT", + "text": "We observed that BERTscore has the highest correlation with human evaluated validity for both tasks, outperforming other metrics by a substantial margin.", + "section": null, + "intents": [] + }, + { + "ref_id": "295065d942abca0711300b2b4c39829551060578", + "citation_marker": "BERTScore: Evaluating Text Generation with BERT", + "text": "We use human judgments to evaluate the revised stories because reference-based automatic evaluation metrics (BLEU [Papineni et al., 2002], BERTscore [Zhang et al., 2019] etc.) are inadequate for multiple reasons: (i) valid revised stories often exist that are different from the references, (ii)\u2026", + "section": null, + "intents": [ + "methodology" + ] + }, + { + "ref_id": "dfbbb0338123de44ef3c0bdc05e9785d20d430cc", + "citation_marker": "Repurposing Entailment for Multi-Hop Question Answering Tasks", + "text": "This can be seen as a form of textual entailment, a capability useful for applications such as question answering (Harabagiu and Hickl, 2006; Trivedi et al., 2019), claim verification (Yin and Roth, 2018; Hanselowski et al., 2018), etc.", + "section": null, + "intents": [ + "background" + ] + }, + { + "ref_id": "85ff214ac614b5163279431941b79927fb72433c", + "citation_marker": "Plan, Write, and Revise: an Interactive System for Open-Domain Story Generation", + "text": ", 2019) and can support interactive story generation tasks (Goldfarb-Tarrant et al., 2019; Brahman et al., 2020).", + "section": null, + "intents": [ + "background" + ] + }, + { + "ref_id": "1dc04035b9926c46ded436e5762f3924ab29516e", + "citation_marker": "Dialogue Natural Language Inference", + "text": "Generating logically coherent long text is still a challenging task for models, and to a certain extent can be attributed to their tendency to forget attributes of specific entities (Welleck et al., 2018), ignore previously inferred facts (Sinha et al.", + "section": null, + "intents": [ + "background" + ] + }, + { + "ref_id": "86b84f603390615627b71d4b3679dc6e3f2ca28b", + "citation_marker": "How Journalists and Social Media Users Perceive Online Fact-Checking and Verification Services", + "text": "\u2026perspective, generating the underlying states that account for the differences between two narratives can assist with fake news detection using reliable sources (Figueira and Oliveira, 2017; da Silva et al., 2019; Ghadiri et al., 2022) and information fact checking (Brandtzaeg et al., 2018).", + "section": null, + "intents": [ + "background" + ] + }, + { + "ref_id": "0c2790d4894940a5cf9084b09788a6c65617c209", + "citation_marker": "UKP-Athene: Multi-Sentence Textual Entailment for Claim Verification", + "text": "This can be seen as a form of textual entailment, a capability useful for applications such as question answering (Harabagiu and Hickl, 2006; Trivedi et al., 2019), claim verification (Yin and Roth, 2018; Hanselowski et al., 2018), etc.", + "section": null, + "intents": [ + "background" + ] + }, + { + "ref_id": "8cd595ee1d8b8ff5716f74b9223d736d21d12ce1", + "citation_marker": "TwoWingOS: A Two-Wing Optimization Strategy for Evidential Claim Verification", + "text": "This can be seen as a form of textual entailment, a capability useful for applications such as question answering (Harabagiu and Hickl, 2006; Trivedi et al., 2019), claim verification (Yin and Roth, 2018; Hanselowski et al., 2018), etc.", + "section": null, + "intents": [ + "background" + ] + }, + { + "ref_id": "5e9c9d0164ae041786f8fdc5726da12403e91a6c", + "citation_marker": "Tracking State Changes in Procedural Text: a Challenge Dataset and Models for Process Paragraph Comprehension", + "text": "Dalvi et al. (2018) introduced PROPARA, which captures physical state changes (creation, destruction, and movement), Bosselut et al. (2018) proposed the task of tracking ingredients in cooking recipes, and Rashkin et al. (2018) tracks the emotional reactions and motivations of characters in simple\u2026", + "section": null, + "intents": [ + "background" + ] + }, + { + "ref_id": "7447ac643e8d4f6d958c492500df08f679ad9c97", + "citation_marker": "Modeling Naive Psychology of Characters in Simple Commonsense Stories", + "text": "\u2026captures physical state changes (creation, destruction, and movement), Bosselut et al. (2018) proposed the task of tracking ingredients in cooking recipes, and Rashkin et al. (2018) tracks the emotional reactions and motivations of characters in simple stories, for a fixed/small set of attributes.", + "section": null, + "intents": [ + "background" + ] + }, + { + "ref_id": "d07284a6811f1b2745d91bdb06b040b57f226882", + "citation_marker": "Decoupled Weight Decay Regularization", + "text": "For all the models, we used the AdamW optimizer with a learning rate of 5 e \u2212 6 and weight decay of 1 e \u2212 6 .", + "section": null, + "intents": [] + }, + { + "ref_id": "d07284a6811f1b2745d91bdb06b040b57f226882", + "citation_marker": "Decoupled Weight Decay Regularization", + "text": "For model training, we used the AdamW (Loshchilov and Hutter, 2017) optimizer with a learning rate of 10 \u2212 4 and weight decay of 10 \u2212 6 .", + "section": null, + "intents": [] + }, + { + "ref_id": "d07284a6811f1b2745d91bdb06b040b57f226882", + "citation_marker": "Decoupled Weight Decay Regularization", + "text": "For model training, we used AdamW (Loshchilov and Hutter, 2017) optimizer with a learning rate of 10\u22124 and weight decay of 10\u22126.", + "section": null, + "intents": [ + "methodology" + ] + }, + { + "ref_id": "26953cc3d09920b54071b73866f85d6bb1a6184c", + "citation_marker": "Simulating Action Dynamics with Neural Process Networks", + "text": "Dalvi et al. (2018) introduced PROPARA, which captures physical state changes (creation, destruction, and movement), Bosselut et al. (2018) proposed the task of tracking ingredients in cooking recipes, and Rashkin et al. (2018) tracks the emotional reactions and motivations of characters in simple\u2026", + "section": null, + "intents": [ + "background" + ] + }, + { + "ref_id": "26aa6fe2028b5eefbaa40ab54ef725bbbe7d9810", + "citation_marker": "ConceptNet 5.5: An Open Multilingual Graph of General Knowledge", + "text": "More recent work by Speer et al. (2017), Sap et al. (2019), and Hwang et al. (2021) capture everyday inferential knowledge associated with an action performed by someone.", + "section": null, + "intents": [ + "background" + ] + }, + { + "ref_id": "85b68477a6e031d88b963833e15a4b4fc6855264", + "citation_marker": "A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories", + "text": "Consider the story in Figure 1 from the ROC-Stories corpus (Mostafazadeh et al., 2016).", + "section": null, + "intents": [ + "methodology" + ] + }, + { + "ref_id": "85b68477a6e031d88b963833e15a4b4fc6855264", + "citation_marker": "A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories", + "text": "ROCStories narratives describe a rich set of causal and temporal commonsense relations between daily events, and its stories are short enough that the world described by them are self-contained.", + "section": null, + "intents": [] + }, + { + "ref_id": "85b68477a6e031d88b963833e15a4b4fc6855264", + "citation_marker": "A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories", + "text": "To create the PASTA dataset, we use stories from the extended ROCStories (Mostafazadeh et al., 2016) corpus for annotation by crowd workers.", + "section": null, + "intents": [ + "methodology" + ] + }, + { + "ref_id": "a477778d8ae4cb70b67d02d094d030c44d0c6ffd", + "citation_marker": "Joint Learning Templates and Slots for Event Schema Induction", + "text": "Statistical work on events (Chambers and Jurafsky, 2008; Chambers and Jurafsky, 2009; Balasubramanian et al., 2013; Ferraro and Van Durme, 2016; Sha et al., 2016) curated event knowledge in an unsupervised manner from large text corpora.", + "section": null, + "intents": [ + "background" + ] + }, + { + "ref_id": "ae7221f4731570b8ebe763000f981c7e5b1664a8", + "citation_marker": "A Unified Bayesian Model of Scripts, Frames and Language", + "text": "Statistical work on events (Chambers and Jurafsky, 2008; Chambers and Jurafsky, 2009; Balasubramanian et al., 2013; Ferraro and Van Durme, 2016; Sha et al., 2016) curated event knowledge in an unsupervised manner from large text corpora.", + "section": null, + "intents": [ + "background" + ] + }, + { + "ref_id": "42b49971a1bc73088ec3317222e7486db82bb4cd", + "citation_marker": "Handbook of Inter-Rater Reliability: The Definitive Guide to Measuring the Extent of Agreement Among Raters", + "text": "Generative Tasks For the two generative tasks, we reported human evaluation results for the best analysis (prior sections).", + "section": null, + "intents": [ + "methodology" + ] + }, + { + "ref_id": "42b49971a1bc73088ec3317222e7486db82bb4cd", + "citation_marker": "Handbook of Inter-Rater Reliability: The Definitive Guide to Measuring the Extent of Agreement Among Raters", + "text": "We use Gwet\u2019s coefficient because it is robust to the paradoxical behaviors (Wongpakaran et al., 2013; Gwet, 2014) seen in the commonly used IAA Kappa metrics (e.g., Cohen\u2019s and Fleiss).", + "section": null, + "intents": [ + "methodology" + ] + }, + { + "ref_id": "42b49971a1bc73088ec3317222e7486db82bb4cd", + "citation_marker": "Handbook of Inter-Rater Reliability: The Definitive Guide to Measuring the Extent of Agreement Among Raters", + "text": "We introduce the Story Revision for Counterfactual States task to address this.", + "section": null, + "intents": [] + }, + { + "ref_id": "42b49971a1bc73088ec3317222e7486db82bb4cd", + "citation_marker": "Handbook of Inter-Rater Reliability: The Definitive Guide to Measuring the Extent of Agreement Among Raters", + "text": "We measure the inter-annotator agreement (IAA) for the human workers using Gwet\u2019s Agreement Coefficient (Gwet, 2008, 2014), which is a type of generalized Kappa statistic.", + "section": null, + "intents": [ + "methodology" + ] + }, + { + "ref_id": "42b49971a1bc73088ec3317222e7486db82bb4cd", + "citation_marker": "Handbook of Inter-Rater Reliability: The Definitive Guide to Measuring the Extent of Agreement Among Raters", + "text": "For the Story Revision for Counterfactual States task, the prompt included n examples followed by the final query: The model must generate S q for the final query ( S q , \u03b1 q , \u2212 ) .", + "section": null, + "intents": [] + }, + { + "ref_id": "42b49971a1bc73088ec3317222e7486db82bb4cd", + "citation_marker": "Handbook of Inter-Rater Reliability: The Definitive Guide to Measuring the Extent of Agreement Among Raters", + "text": "We report ROUGELsum for the Story Revision for Counterfactual States task since it is computed over the entire story, and the sentence level ROUGEL metric for State Change Generation.", + "section": null, + "intents": [ + "methodology" + ] + }, + { + "ref_id": "42b49971a1bc73088ec3317222e7486db82bb4cd", + "citation_marker": "Handbook of Inter-Rater Reliability: The Definitive Guide to Measuring the Extent of Agreement Among Raters", + "text": "The second task, Story Revision for Counterfac-tual States , measures the ability to reason about counterfactuals.", + "section": null, + "intents": [] + }, + { + "ref_id": "42b49971a1bc73088ec3317222e7486db82bb4cd", + "citation_marker": "Handbook of Inter-Rater Reliability: The Definitive Guide to Measuring the Extent of Agreement Among Raters", + "text": "It includes the Coun-terfactual Story Rewriting task to edit a short story based on a counterfactual context.", + "section": null, + "intents": [] + }, + { + "ref_id": "f40219d81b785bddc7fdad9c2af2840af62661ae", + "citation_marker": "Generating Coherent Event Schemas at Scale", + "text": "Statistical work on events (Chambers and Jurafsky, 2008, 2009; Balasubramanian et al., 2013; Ferraro and Van Durme, 2016; Sha et al., 2016) curated event knowledge in an unsupervised manner from large text corpora.", + "section": null, + "intents": [ + "background" + ] + }, + { + "ref_id": "8891481c3fa28d95acc9f2c50187ce682a8130e7", + "citation_marker": "A comparison of Cohen\u2019s Kappa and Gwet\u2019s AC1 when calculating inter-rater reliability coefficients: a study conducted with personality disorder samples", + "text": "We use Gwet\u2019s coefficient because it is robust to the paradoxical behaviors (Wongpakaran et al., 2013; Gwet, 2014) seen in the commonly used IAA Kappa metrics (e.g., Cohen\u2019s and Fleiss).", + "section": null, + "intents": [ + "methodology" + ] + }, + { + "ref_id": "f6c9197a3bd26af0fe6a199e95d0336f4d958e1c", + "citation_marker": "Development of the RTI item bank on risk of bias and precision of observational studies.", + "text": "7 Its interpretation is similar to generalized kappa (Viswanathan and Berkman, 2012), with 0 .", + "section": null, + "intents": [ + "background" + ] + }, + { + "ref_id": "087be7b3d737ee817b64246e7c9a4fb6cd57dc24", + "citation_marker": "Computing inter-rater reliability and its variance in the presence of high agreement.", + "text": "We measure the inter-annotator agreement (IAA) for the human workers using Gwet\u2019s Agreement Coefficient (Gwet, 2008, 2014), which is a type of generalized Kappa statistic.", + "section": null, + "intents": [ + "methodology" + ] + }, + { + "ref_id": "a2c54a729be2e2f46c446cabe70fdb3d38d2a1c6", + "citation_marker": "Methods for Using Textual Entailment in Open-Domain Question Answering", + "text": "This can be seen as a form of textual entailment, a capability useful for applications such as question answering (Harabagiu and Hickl, 2006; Trivedi et al., 2019), claim verification (Yin and Roth, 2018; Hanselowski et al., 2018), etc.", + "section": null, + "intents": [ + "background" + ] + }, + { + "ref_id": "60b05f32c32519a809f21642ef1eb3eaf3848008", + "citation_marker": "ROUGE: A Package for Automatic Evaluation of Summaries", + "text": "We report ROUGELsum for the Story Revision for Counterfactual States task since it is computed over the entire story, and the sentence level ROUGEL metric for State Change Generation.", + "section": null, + "intents": [] + }, + { + "ref_id": "60b05f32c32519a809f21642ef1eb3eaf3848008", + "citation_marker": "ROUGE: A Package for Automatic Evaluation of Summaries", + "text": "However, since human evaluation is expensive, we include here the results from three automatic metrics: GLEU (Wu et al., 2016), ROUGE (Lin, 2004), and BERTscore (Zhang et al., 2019).", + "section": null, + "intents": [ + "methodology" + ] + }, + { + "ref_id": "d7da009f457917aa381619facfa5ffae9329a6e9", + "citation_marker": "Bleu: a Method for Automatic Evaluation of Machine Translation", + "text": "We use human judgments to evaluate the revised stories because reference-based automatic evaluation metrics (BLEU [Papineni et al., 2002], BERTscore [Zhang et al., 2019] etc.) are inadequate for multiple reasons: (i) valid revised stories often exist that are different from the references, (ii)\u2026", + "section": null, + "intents": [ + "methodology" + ] + }, + { + "ref_id": "109cd7af6a209808cbe5d3e17756451b4ee97ef9", + "citation_marker": "Bias, prevalence and kappa.", + "text": "This paradoxical behavior of these metrics can lead to their IAA coefficients being lower even when the agreement is strong (Feinstein and Cicchetti, 1990; Byrt et al., 1993).", + "section": null, + "intents": [ + "background" + ] + }, + { + "ref_id": "ee1cd8b38356220809c9a216d017fe6148fb8f76", + "citation_marker": "A Critical Examination of Allen's Theory of Action and Time", + "text": "Frames (Fillmore, 1985) and theories of time (Galton, 1990) provide related views.", + "section": null, + "intents": [ + "background" + ] + }, + { + "ref_id": "ee1cd8b38356220809c9a216d017fe6148fb8f76", + "citation_marker": "A Critical Examination of Allen's Theory of Action and Time", + "text": "Many cognitive theories have been proposed to capture aspects of this in their representations, such as scripts (Schank and Abelson, 1975), frames (Fillmore, 1985), and state/time formalisms (Galton, 1990).", + "section": null, + "intents": [ + "background" + ] + } +] diff --git a/replay_traces/2208.00329/two_pass_outputs/20260726T205558Z/pass_1_reasoning.selected.md b/replay_traces/2208.00329/two_pass_outputs/20260726T205558Z/pass_1_reasoning.selected.md new file mode 100644 index 0000000000000000000000000000000000000000..080e99d65cb7c2103af0469a5a4898fc9104f55d --- /dev/null +++ b/replay_traces/2208.00329/two_pass_outputs/20260726T205558Z/pass_1_reasoning.selected.md @@ -0,0 +1,254 @@ +# 1. Cluster Evidence + +Based on the downstream cluster evidence, the paper introduces two distinct downstream-used contributions: +- **C1:** The methodology/protocol for dataset creation, which is reused by later work to generate their own datasets. +- **C2:** The PASTA dataset itself, which is reused directly for data generation and validation. + +Because these two contributions have distinct downstream roles (one is reused as a data artifact, the other as a data-collection process) and require different structural ingredients, they should be split into two atomic claims. + +--- + +# 2. Claim Split Decision + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C2 + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Method +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C1 + +### Why this split is correct +The two artifacts have distinct downstream roles and distinct ingredient decompositions: +- **The PASTA Dataset** is reused as a concrete data resource for training and evaluation. It structurally depends on the specific source texts it annotates. +- **The Dataset Creation Methodology** is reused as a process/protocol to build new datasets. It structurally depends on conceptual frameworks for counterfactuals and quality control, but is agnostic to the specific source text. +They should therefore be annotated separately. + +--- + +# 3. Rewritten Claims + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **rewritten_claim:** **Dataset: A crowdsourced dataset of narrative stories annotated with implicit participant states, counterfactual perturbations, and revised stories, enabling state-based reasoning and generation tasks.** +- **why_this_is_atomic:** This claim isolates the dataset artifact itself, separate from the methodology used to create it. +- **optional cluster_id:** C2 +- **decision:** YES_SUFFICIENT + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Method +- **rewritten_claim:** **Method: A crowdsourcing protocol for eliciting implicit participant states and counterfactual story revisions, enabling the construction of state-based narrative datasets.** +- **why_this_is_atomic:** This claim isolates the reusable annotation methodology and quality-control process, distinct from the specific PASTA dataset produced by it. +- **optional cluster_id:** C1 +- **decision:** YES_SUFFICIENT + +--- + +# 4. Minimal Ingredient Set Per Claim + +## Claim C1: PASTA Dataset + +### Ingredient C1.I1 +- **ingredient_id:** C1.I1 +- **ingredient:** Source corpus of short, self-contained narrative stories +- **why_structurally_necessary:** The dataset annotates implicit states and counterfactuals over existing narratives. Without a source corpus of narratives to annotate, the dataset would not exist in its claimed form. +- **why_not_lower_level_substeps:** This should remain at the level of the source corpus rather than being split into specific story filters, splits, or sampling strategies. +- **why_not_adjacent_implementation_details:** The exact size of the sample is an implementation detail; the structural dependency is the narrative text substrate itself. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories +- **additional groundings if any:** none +- **role:** DATA_SOURCE +- **contribution:** Provides the base narrative stories (ROCStories) that are annotated and revised in the dataset. +- **rationale:** The dataset requires a substrate of short, commonsense-rich narratives to serve as the foundation for state inference and revision. ROCStories is the direct source used by the paper and maps cleanly to its canonical study. +- **evidence_span:** “To create the PASTA dataset, we use stories from the extended ROCStories (Mostafazadeh et al., 2016) corpus for annotation by crowd workers.” + +### Ingredient C1.I2 +- **ingredient_id:** C1.I2 +- **ingredient:** Four-step crowdsourcing protocol for inferring states and generating counterfactual revisions +- **why_structurally_necessary:** The dataset's unique structure (inferred state, justification, counterfactual state, revised story) is entirely defined by this specific annotation workflow. Without it, the artifact would not have its defining shape. +- **why_not_lower_level_substeps:** This should remain one protocol-level ingredient rather than being split into the four individual steps or the specific MTurk UI details. +- **why_not_adjacent_implementation_details:** Quality control and worker filtering are important but subordinate to the core four-step annotation protocol that defines the data structure for this claim. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Defines the workflow used to collect the structured annotations that make up the dataset. +- **rationale:** The specific four-step process that yields the dataset's tuples is a novel contribution of the target paper itself, so it maps cleanly to `NONE`. +- **evidence_span:** “The annotation process has four main steps: 1. Infer a participant state... 2. Select minimal justification sentences... 3. Perturb the state... 4. Revise the story” + +--- + +## Claim C2: Dataset Creation Methodology + +### Ingredient C2.I1 +- **ingredient_id:** C2.I1 +- **ingredient:** Counterfactual story rewriting framework +- **why_structurally_necessary:** The protocol requires workers to revise a story based on a counterfactual state. This builds fundamentally on the conceptual framework of counterfactual story rewriting. +- **why_not_lower_level_substeps:** This remains at the conceptual level of counterfactual rewriting rather than specific model architectures or generation constraints. +- **why_not_adjacent_implementation_details:** The specific position of the counterfactual (unobserved vs explicit) is the target paper's contribution, but the foundational concept is counterfactual rewriting. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Counterfactual Story Reasoning and Generation +- **additional groundings if any:** none +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the foundational concept of editing a short story based on a counterfactual context, which the protocol adapts for unobserved participant states. +- **rationale:** The methodology extends the idea of counterfactual story rewriting to implicit states. Qin et al. (TIME-TRAVEL) is explicitly cited as the most similar prior work that introduced this rewriting task framing. +- **evidence_span:** “Most similar to our work is the TIME-TRAVEL dataset by Qin et al. (2019). It includes the Counterfactual Story Rewriting task” + +### Ingredient C2.I2 +- **ingredient_id:** C2.I2 +- **ingredient:** Framework for annotating participant states in simple stories +- **why_structurally_necessary:** The protocol is designed to elicit unstated participant states. It relies on the conceptual framing that narratives contain implicit participant states (like emotions or physical states) that can be systematically tracked and annotated. +- **why_not_lower_level_substeps:** This is a broad conceptual framing rather than specific state categories (physical, emotional, societal). +- **why_not_adjacent_implementation_details:** The specific instructions given to workers are implementation details of this broader concept. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Modeling Naive Psychology of Characters in Simple Commonsense Stories +- **additional groundings if any:** Tracking State Changes in Procedural Text: a Challenge Dataset and Models for Process Paragraph Comprehension +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the conceptual basis for annotating the implicit states (like emotions or motivations) of characters in short narratives. +- **rationale:** The methodology builds on prior work that established the tracking and annotation of entity/participant states in text. Rashkin et al. is a clean representative for tracking emotional/psychological states in simple stories, with Dalvi et al. serving as an additional grounding for physical states. +- **evidence_span:** “Rashkin et al. (2018) tracks the emotional reactions and motivations of characters in simple stories, for a fixed/small set of attributes.” + +### Ingredient C2.I3 +- **ingredient_id:** C2.I3 +- **ingredient:** Two-stage quality control and worker-proficiency filtering mechanism +- **why_structurally_necessary:** Eliciting counterfactuals and logical story revisions from crowd workers is highly prone to artifacts and logical errors. Without a systematic quality control and proficiency-filtering mechanism, the protocol would fail to produce usable data and would not function as a reliable methodology. +- **why_not_lower_level_substeps:** This should remain one quality-control ingredient rather than being split into the specific >5000 HITs threshold or the exact expert-review formula. +- **why_not_adjacent_implementation_details:** The specific pay rate ($0.35) is an implementation detail, but the two-stage filtering process is structural to the methodology's success. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the filtering and expert-review workflow necessary to identify proficient workers and ensure high-quality complex annotations. +- **rationale:** The specific two-stage filtering process (strict initial criteria followed by expert review to identify "proficient workers") is a custom workflow designed in the target paper to make the complex annotation task viable. It maps cleanly to `NONE`. +- **evidence_span:** “For filtering out noisy data from the collected responses, we follow a two-stage filtering process.” + +--- + +# 5. Excluded Tempting Non-Ingredients + +### Claim C1 (Dataset) exclusions +- **Specific baseline models (T5, GPT-3, RoBERTa)** + - Excluded because they are used to benchmark and evaluate the dataset, not to construct the dataset artifact itself. +- **The three specific evaluation tasks (Story State Inference, Story Revision, State Change Generation)** + - Excluded because they are applications/framings for using the dataset, while the dataset artifact itself is just the collection of annotated 4-tuples. + +### Claim C2 (Methodology) exclusions +- **ROCStories corpus** + - Excluded because the methodology is a general protocol for narrative text and does not structurally require ROCStories to exist as a methodology. ROCStories is an ingredient of the dataset, not the method. +- **Specific MTurk UI design** + - Excluded because it is a local implementation detail of the broader four-step crowdsourcing protocol. + +--- + +# 6. Candidate Grounding Decisions + +## Claim C1: PASTA Dataset + +### Ingredient C1.I1 +- **Candidate:** A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories +- **decision:** accepted_canonical +- **why:** Cleanest representative study for the ROCStories corpus used as the source text. + +### Ingredient C1.I2 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The 4-step annotation protocol is a novel workflow introduced in the target paper. + +## Claim C2: Dataset Creation Methodology + +### Ingredient C2.I1 +- **Candidate:** Counterfactual Story Reasoning and Generation +- **decision:** accepted_canonical +- **why:** Cleanest representative for the counterfactual story rewriting framework that the protocol adapts. + +### Ingredient C2.I2 +- **Candidate:** Modeling Naive Psychology of Characters in Simple Commonsense Stories +- **decision:** accepted_canonical +- **why:** Cleanest representative for the conceptual framing of tracking and annotating participant states in simple stories. +- **Candidate:** Tracking State Changes in Procedural Text: a Challenge Dataset and Models for Process Paragraph Comprehension +- **decision:** accepted_additional +- **why:** Relevant additional grounding for tracking physical state changes. + +### Ingredient C2.I3 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The two-stage quality control and expert-review mechanism is specific to the target paper's methodology. + +--- + +# 7. Final Grounding Decisions + +## Claim C1 +- **C1.I1** → A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories → DATA_SOURCE +- **C1.I2** → NONE → EVALUATION_PROTOCOL + +## Claim C2 +- **C2.I1** → Counterfactual Story Reasoning and Generation → CONCEPTUAL_FRAMEWORK +- **C2.I2** → Modeling Naive Psychology of Characters in Simple Commonsense Stories → CONCEPTUAL_FRAMEWORK + - additional: Tracking State Changes in Procedural Text: a Challenge Dataset and Models for Process Paragraph Comprehension +- **C2.I3** → NONE → EVALUATION_PROTOCOL + +--- + +# 8. Final Rationales And Evidence + +## Claim C1: PASTA Dataset + +### C1.I1 +- **ingredient:** Source corpus of short, self-contained narrative stories +- **canonical study:** A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories +- **role:** DATA_SOURCE +- **contribution:** Provides the base narrative stories (ROCStories) that are annotated and revised in the dataset. +- **rationale:** The dataset requires a substrate of short, commonsense-rich narratives to serve as the foundation for state inference and revision. ROCStories is the direct source used by the paper and maps cleanly to its canonical study. +- **evidence_span:** “To create the PASTA dataset, we use stories from the extended ROCStories (Mostafazadeh et al., 2016) corpus for annotation by crowd workers.” + +### C1.I2 +- **ingredient:** Four-step crowdsourcing protocol for inferring states and generating counterfactual revisions +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Defines the workflow used to collect the structured annotations that make up the dataset. +- **rationale:** The specific four-step process that yields the dataset's tuples is a novel contribution of the target paper itself, so it maps cleanly to `NONE`. +- **evidence_span:** “The annotation process has four main steps: 1. Infer a participant state... 2. Select minimal justification sentences... 3. Perturb the state... 4. Revise the story” + +## Claim C2: Dataset Creation Methodology + +### C2.I1 +- **ingredient:** Counterfactual story rewriting framework +- **canonical study:** Counterfactual Story Reasoning and Generation +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the foundational concept of editing a short story based on a counterfactual context, which the protocol adapts for unobserved participant states. +- **rationale:** The methodology extends the idea of counterfactual story rewriting to implicit states. Qin et al. (TIME-TRAVEL) is explicitly cited as the most similar prior work that introduced this rewriting task framing. +- **evidence_span:** “Most similar to our work is the TIME-TRAVEL dataset by Qin et al. (2019). It includes the Counterfactual Story Rewriting task” + +### C2.I2 +- **ingredient:** Framework for annotating participant states in simple stories +- **canonical study:** Modeling Naive Psychology of Characters in Simple Commonsense Stories +- **additional studies:** Tracking State Changes in Procedural Text: a Challenge Dataset and Models for Process Paragraph Comprehension +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the conceptual basis for annotating the implicit states (like emotions or motivations) of characters in short narratives. +- **rationale:** The methodology builds on prior work that established the tracking and annotation of entity/participant states in text. Rashkin et al. is a clean representative for tracking emotional/psychological states in simple stories, with Dalvi et al. serving as an additional grounding for physical states. +- **evidence_span:** “Rashkin et al. (2018) tracks the emotional reactions and motivations of characters in simple stories, for a fixed/small set of attributes.” + +### C2.I3 +- **ingredient:** Two-stage quality control and worker-proficiency filtering mechanism +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the filtering and expert-review workflow necessary to identify proficient workers and ensure high-quality complex annotations. +- **rationale:** The specific two-stage filtering process (strict initial criteria followed by expert review to identify "proficient workers") is a custom workflow designed in the target paper to make the complex annotation task viable. It maps cleanly to `NONE`. +- **evidence_span:** “For filtering out noisy data from the collected responses, we follow a two-stage filtering process.” \ No newline at end of file diff --git a/replay_traces/2208.00329/two_pass_outputs/20260726T205558Z/pass_2_formatter.prompt.txt b/replay_traces/2208.00329/two_pass_outputs/20260726T205558Z/pass_2_formatter.prompt.txt new file mode 100644 index 0000000000000000000000000000000000000000..268a38c61bd00c7f7d1d0b057fbf475454118613 --- /dev/null +++ b/replay_traces/2208.00329/two_pass_outputs/20260726T205558Z/pass_2_formatter.prompt.txt @@ -0,0 +1,408 @@ +Task: Convert the reasoning memo into the final annotation UI JSON payload. + +This is a strict formatting pass. +Do not do new reasoning unless needed to resolve minor schema ambiguities. +Follow the memo’s final decisions faithfully. + +Critical formatting priorities: +1. Preserve the final claim split. +2. Preserve the final minimal ingredient set. +3. Respect exclusions of tempting non-ingredients. +4. Preserve canonical vs additional vs NONE decisions exactly. +5. Do not inflate the ingredient set. + +Target schema: +{ + "target_paper_id": "...", + "target_title": "...", + "target_year": 2024, + "annotator_id": "replay_trace_builder", + "active_claim_id": "...", + "claims": [ + { + "claim_id": "...", + "text": "...", + "rewritten_claim": "...", + "cluster_id": "...", + "decision": "YES_SUFFICIENT | NO_NOT_DISCOVERY | UNCERTAIN", + "notes": "", + "ingredients": [ + { + "ingredient_id": "...", + "ingredient": "...", + "canonical_ref_id": "__NONE__ or ref_id", + "canonical_grounding": null or { + "ref_id": "...", + "bib_key": "...", + "paper_id": "...", + "external_ids": {}, + "ref_title": "...", + "ref_year": "...", + "ref_authors": "..." + }, + "additional_ref_ids": ["..."], + "additional_groundings": [ + { + "ref_id": "...", + "bib_key": "...", + "paper_id": "...", + "external_ids": {}, + "ref_title": "...", + "ref_year": "...", + "ref_authors": "..." + } + ], + "canonical_annotation": { + "role": null or "CONCEPTUAL_FRAMEWORK" or "CORE_METHOD" or "DATA_SOURCE" or "MODEL_INITIALIZATION" or "EVALUATION_PROTOCOL" or "IMPLEMENTATION_TOOLING" or "TRAINING_DATA", + "roles": ["..."], + "contribution": "...", + "rationale": "...", + "evidence_span": "..." + } + } + ], + "enabling_discoveries": [ + { + "ref_id": "...", + "bib_key": "...", + "paper_id": "...", + "external_ids": {}, + "ref_title": "...", + "ref_year": "...", + "ref_authors": "...", + "ingredient_id": "...", + "ingredient": "...", + "role": null or "CONCEPTUAL_FRAMEWORK" or "CORE_METHOD" or "DATA_SOURCE" or "MODEL_INITIALIZATION" or "EVALUATION_PROTOCOL" or "IMPLEMENTATION_TOOLING" or "TRAINING_DATA", + "roles": ["..."], + "contribution": "...", + "rationale": "...", + "evidence_span": "..." + } + ] + } + ] +} + +Formatting rules: +- `text` must equal `rewritten_claim` +- `active_claim_id` should be the first claim_id unless the reasoning memo explicitly recommends another active claim +- if canonical grounding is NONE: + - `canonical_ref_id` must be "__NONE__" + - `canonical_grounding` must be null + - do not include that ingredient in `enabling_discoveries` +- if roles has length 1, `role` should equal that role +- if roles has length != 1, `role` should be null +- `additional_ref_ids` must match `additional_groundings[].ref_id` +- `enabling_discoveries` should contain only canonical non-NONE ingredients +- use only the allowed role set + +REASONING MEMO: +# 1. Cluster Evidence + +Based on the downstream cluster evidence, the paper introduces two distinct downstream-used contributions: +- **C1:** The methodology/protocol for dataset creation, which is reused by later work to generate their own datasets. +- **C2:** The PASTA dataset itself, which is reused directly for data generation and validation. + +Because these two contributions have distinct downstream roles (one is reused as a data artifact, the other as a data-collection process) and require different structural ingredients, they should be split into two atomic claims. + +--- + +# 2. Claim Split Decision + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C2 + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Method +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C1 + +### Why this split is correct +The two artifacts have distinct downstream roles and distinct ingredient decompositions: +- **The PASTA Dataset** is reused as a concrete data resource for training and evaluation. It structurally depends on the specific source texts it annotates. +- **The Dataset Creation Methodology** is reused as a process/protocol to build new datasets. It structurally depends on conceptual frameworks for counterfactuals and quality control, but is agnostic to the specific source text. +They should therefore be annotated separately. + +--- + +# 3. Rewritten Claims + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **rewritten_claim:** **Dataset: A crowdsourced dataset of narrative stories annotated with implicit participant states, counterfactual perturbations, and revised stories, enabling state-based reasoning and generation tasks.** +- **why_this_is_atomic:** This claim isolates the dataset artifact itself, separate from the methodology used to create it. +- **optional cluster_id:** C2 +- **decision:** YES_SUFFICIENT + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Method +- **rewritten_claim:** **Method: A crowdsourcing protocol for eliciting implicit participant states and counterfactual story revisions, enabling the construction of state-based narrative datasets.** +- **why_this_is_atomic:** This claim isolates the reusable annotation methodology and quality-control process, distinct from the specific PASTA dataset produced by it. +- **optional cluster_id:** C1 +- **decision:** YES_SUFFICIENT + +--- + +# 4. Minimal Ingredient Set Per Claim + +## Claim C1: PASTA Dataset + +### Ingredient C1.I1 +- **ingredient_id:** C1.I1 +- **ingredient:** Source corpus of short, self-contained narrative stories +- **why_structurally_necessary:** The dataset annotates implicit states and counterfactuals over existing narratives. Without a source corpus of narratives to annotate, the dataset would not exist in its claimed form. +- **why_not_lower_level_substeps:** This should remain at the level of the source corpus rather than being split into specific story filters, splits, or sampling strategies. +- **why_not_adjacent_implementation_details:** The exact size of the sample is an implementation detail; the structural dependency is the narrative text substrate itself. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories +- **additional groundings if any:** none +- **role:** DATA_SOURCE +- **contribution:** Provides the base narrative stories (ROCStories) that are annotated and revised in the dataset. +- **rationale:** The dataset requires a substrate of short, commonsense-rich narratives to serve as the foundation for state inference and revision. ROCStories is the direct source used by the paper and maps cleanly to its canonical study. +- **evidence_span:** “To create the PASTA dataset, we use stories from the extended ROCStories (Mostafazadeh et al., 2016) corpus for annotation by crowd workers.” + +### Ingredient C1.I2 +- **ingredient_id:** C1.I2 +- **ingredient:** Four-step crowdsourcing protocol for inferring states and generating counterfactual revisions +- **why_structurally_necessary:** The dataset's unique structure (inferred state, justification, counterfactual state, revised story) is entirely defined by this specific annotation workflow. Without it, the artifact would not have its defining shape. +- **why_not_lower_level_substeps:** This should remain one protocol-level ingredient rather than being split into the four individual steps or the specific MTurk UI details. +- **why_not_adjacent_implementation_details:** Quality control and worker filtering are important but subordinate to the core four-step annotation protocol that defines the data structure for this claim. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Defines the workflow used to collect the structured annotations that make up the dataset. +- **rationale:** The specific four-step process that yields the dataset's tuples is a novel contribution of the target paper itself, so it maps cleanly to `NONE`. +- **evidence_span:** “The annotation process has four main steps: 1. Infer a participant state... 2. Select minimal justification sentences... 3. Perturb the state... 4. Revise the story” + +--- + +## Claim C2: Dataset Creation Methodology + +### Ingredient C2.I1 +- **ingredient_id:** C2.I1 +- **ingredient:** Counterfactual story rewriting framework +- **why_structurally_necessary:** The protocol requires workers to revise a story based on a counterfactual state. This builds fundamentally on the conceptual framework of counterfactual story rewriting. +- **why_not_lower_level_substeps:** This remains at the conceptual level of counterfactual rewriting rather than specific model architectures or generation constraints. +- **why_not_adjacent_implementation_details:** The specific position of the counterfactual (unobserved vs explicit) is the target paper's contribution, but the foundational concept is counterfactual rewriting. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Counterfactual Story Reasoning and Generation +- **additional groundings if any:** none +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the foundational concept of editing a short story based on a counterfactual context, which the protocol adapts for unobserved participant states. +- **rationale:** The methodology extends the idea of counterfactual story rewriting to implicit states. Qin et al. (TIME-TRAVEL) is explicitly cited as the most similar prior work that introduced this rewriting task framing. +- **evidence_span:** “Most similar to our work is the TIME-TRAVEL dataset by Qin et al. (2019). It includes the Counterfactual Story Rewriting task” + +### Ingredient C2.I2 +- **ingredient_id:** C2.I2 +- **ingredient:** Framework for annotating participant states in simple stories +- **why_structurally_necessary:** The protocol is designed to elicit unstated participant states. It relies on the conceptual framing that narratives contain implicit participant states (like emotions or physical states) that can be systematically tracked and annotated. +- **why_not_lower_level_substeps:** This is a broad conceptual framing rather than specific state categories (physical, emotional, societal). +- **why_not_adjacent_implementation_details:** The specific instructions given to workers are implementation details of this broader concept. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Modeling Naive Psychology of Characters in Simple Commonsense Stories +- **additional groundings if any:** Tracking State Changes in Procedural Text: a Challenge Dataset and Models for Process Paragraph Comprehension +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the conceptual basis for annotating the implicit states (like emotions or motivations) of characters in short narratives. +- **rationale:** The methodology builds on prior work that established the tracking and annotation of entity/participant states in text. Rashkin et al. is a clean representative for tracking emotional/psychological states in simple stories, with Dalvi et al. serving as an additional grounding for physical states. +- **evidence_span:** “Rashkin et al. (2018) tracks the emotional reactions and motivations of characters in simple stories, for a fixed/small set of attributes.” + +### Ingredient C2.I3 +- **ingredient_id:** C2.I3 +- **ingredient:** Two-stage quality control and worker-proficiency filtering mechanism +- **why_structurally_necessary:** Eliciting counterfactuals and logical story revisions from crowd workers is highly prone to artifacts and logical errors. Without a systematic quality control and proficiency-filtering mechanism, the protocol would fail to produce usable data and would not function as a reliable methodology. +- **why_not_lower_level_substeps:** This should remain one quality-control ingredient rather than being split into the specific >5000 HITs threshold or the exact expert-review formula. +- **why_not_adjacent_implementation_details:** The specific pay rate ($0.35) is an implementation detail, but the two-stage filtering process is structural to the methodology's success. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the filtering and expert-review workflow necessary to identify proficient workers and ensure high-quality complex annotations. +- **rationale:** The specific two-stage filtering process (strict initial criteria followed by expert review to identify "proficient workers") is a custom workflow designed in the target paper to make the complex annotation task viable. It maps cleanly to `NONE`. +- **evidence_span:** “For filtering out noisy data from the collected responses, we follow a two-stage filtering process.” + +--- + +# 5. Excluded Tempting Non-Ingredients + +### Claim C1 (Dataset) exclusions +- **Specific baseline models (T5, GPT-3, RoBERTa)** + - Excluded because they are used to benchmark and evaluate the dataset, not to construct the dataset artifact itself. +- **The three specific evaluation tasks (Story State Inference, Story Revision, State Change Generation)** + - Excluded because they are applications/framings for using the dataset, while the dataset artifact itself is just the collection of annotated 4-tuples. + +### Claim C2 (Methodology) exclusions +- **ROCStories corpus** + - Excluded because the methodology is a general protocol for narrative text and does not structurally require ROCStories to exist as a methodology. ROCStories is an ingredient of the dataset, not the method. +- **Specific MTurk UI design** + - Excluded because it is a local implementation detail of the broader four-step crowdsourcing protocol. + +--- + +# 6. Candidate Grounding Decisions + +## Claim C1: PASTA Dataset + +### Ingredient C1.I1 +- **Candidate:** A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories +- **decision:** accepted_canonical +- **why:** Cleanest representative study for the ROCStories corpus used as the source text. + +### Ingredient C1.I2 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The 4-step annotation protocol is a novel workflow introduced in the target paper. + +## Claim C2: Dataset Creation Methodology + +### Ingredient C2.I1 +- **Candidate:** Counterfactual Story Reasoning and Generation +- **decision:** accepted_canonical +- **why:** Cleanest representative for the counterfactual story rewriting framework that the protocol adapts. + +### Ingredient C2.I2 +- **Candidate:** Modeling Naive Psychology of Characters in Simple Commonsense Stories +- **decision:** accepted_canonical +- **why:** Cleanest representative for the conceptual framing of tracking and annotating participant states in simple stories. +- **Candidate:** Tracking State Changes in Procedural Text: a Challenge Dataset and Models for Process Paragraph Comprehension +- **decision:** accepted_additional +- **why:** Relevant additional grounding for tracking physical state changes. + +### Ingredient C2.I3 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The two-stage quality control and expert-review mechanism is specific to the target paper's methodology. + +--- + +# 7. Final Grounding Decisions + +## Claim C1 +- **C1.I1** → A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories → DATA_SOURCE +- **C1.I2** → NONE → EVALUATION_PROTOCOL + +## Claim C2 +- **C2.I1** → Counterfactual Story Reasoning and Generation → CONCEPTUAL_FRAMEWORK +- **C2.I2** → Modeling Naive Psychology of Characters in Simple Commonsense Stories → CONCEPTUAL_FRAMEWORK + - additional: Tracking State Changes in Procedural Text: a Challenge Dataset and Models for Process Paragraph Comprehension +- **C2.I3** → NONE → EVALUATION_PROTOCOL + +--- + +# 8. Final Rationales And Evidence + +## Claim C1: PASTA Dataset + +### C1.I1 +- **ingredient:** Source corpus of short, self-contained narrative stories +- **canonical study:** A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories +- **role:** DATA_SOURCE +- **contribution:** Provides the base narrative stories (ROCStories) that are annotated and revised in the dataset. +- **rationale:** The dataset requires a substrate of short, commonsense-rich narratives to serve as the foundation for state inference and revision. ROCStories is the direct source used by the paper and maps cleanly to its canonical study. +- **evidence_span:** “To create the PASTA dataset, we use stories from the extended ROCStories (Mostafazadeh et al., 2016) corpus for annotation by crowd workers.” + +### C1.I2 +- **ingredient:** Four-step crowdsourcing protocol for inferring states and generating counterfactual revisions +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Defines the workflow used to collect the structured annotations that make up the dataset. +- **rationale:** The specific four-step process that yields the dataset's tuples is a novel contribution of the target paper itself, so it maps cleanly to `NONE`. +- **evidence_span:** “The annotation process has four main steps: 1. Infer a participant state... 2. Select minimal justification sentences... 3. Perturb the state... 4. Revise the story” + +## Claim C2: Dataset Creation Methodology + +### C2.I1 +- **ingredient:** Counterfactual story rewriting framework +- **canonical study:** Counterfactual Story Reasoning and Generation +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the foundational concept of editing a short story based on a counterfactual context, which the protocol adapts for unobserved participant states. +- **rationale:** The methodology extends the idea of counterfactual story rewriting to implicit states. Qin et al. (TIME-TRAVEL) is explicitly cited as the most similar prior work that introduced this rewriting task framing. +- **evidence_span:** “Most similar to our work is the TIME-TRAVEL dataset by Qin et al. (2019). It includes the Counterfactual Story Rewriting task” + +### C2.I2 +- **ingredient:** Framework for annotating participant states in simple stories +- **canonical study:** Modeling Naive Psychology of Characters in Simple Commonsense Stories +- **additional studies:** Tracking State Changes in Procedural Text: a Challenge Dataset and Models for Process Paragraph Comprehension +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the conceptual basis for annotating the implicit states (like emotions or motivations) of characters in short narratives. +- **rationale:** The methodology builds on prior work that established the tracking and annotation of entity/participant states in text. Rashkin et al. is a clean representative for tracking emotional/psychological states in simple stories, with Dalvi et al. serving as an additional grounding for physical states. +- **evidence_span:** “Rashkin et al. (2018) tracks the emotional reactions and motivations of characters in simple stories, for a fixed/small set of attributes.” + +### C2.I3 +- **ingredient:** Two-stage quality control and worker-proficiency filtering mechanism +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the filtering and expert-review workflow necessary to identify proficient workers and ensure high-quality complex annotations. +- **rationale:** The specific two-stage filtering process (strict initial criteria followed by expert review to identify "proficient workers") is a custom workflow designed in the target paper to make the complex annotation task viable. It maps cleanly to `NONE`. +- **evidence_span:** “For filtering out noisy data from the collected responses, we follow a two-stage filtering process.” + +TARGET PAPER METADATA: +{ + "paperId": "e894fb15054d3bc9659060406a12dfd1055ae32e", + "externalIds": { + "DBLP": "journals/corr/abs-2208-00329", + "ArXiv": "2208.00329", + "DOI": "10.1162/tacl_a_00600", + "CorpusId": 251223896 + }, + "url": "https://www.semanticscholar.org/paper/e894fb15054d3bc9659060406a12dfd1055ae32e", + "title": "PASTA: A Dataset for Modeling PArticipant STAtes in Narratives", + "venue": "Transactions of the Association for Computational Linguistics", + "year": 2022, + "referenceCount": 59, + "citationCount": 7, + "openAccessPdf": { + "url": "https://direct.mit.edu/tacl/article-pdf/doi/10.1162/tacl_a_00600/2173956/tacl_a_00600.pdf", + "status": "GOLD", + "license": "CCBY", + "disclaimer": "Notice: Paper or abstract available at https://arxiv.org/abs/2208.00329, which is subject to the license by the author or copyright owner provided with this content. Please go to the source to verify the license and copyright information for your use." + }, + "tldr": { + "model": "tldr@v2.0.0", + "text": "This work introduces a new crowdsourced English-language, Participant States dataset, PASTA, and introduces three state-based reasoning tasks that test for the ability to infer when a state is entailed by a story, to revise a story conditioned on a counterfactual state, and to explain the most likely state change given a revised story." + }, + "publicationDate": "2022-07-31", + "authors": [ + { + "authorId": "72059959", + "name": "Sayontan Ghosh" + }, + { + "authorId": "144007901", + "name": "Mahnaz Koupaee" + }, + { + "authorId": "2254303864", + "name": "I-Sen Chen" + }, + { + "authorId": "2034063", + "name": "Francis Ferraro" + }, + { + "authorId": "1729918", + "name": "Nathanael Chambers" + }, + { + "authorId": "35217367", + "name": "Niranjan Balasubramanian" + } + ], + "abstract": "Abstract The events in a narrative are understood as a coherent whole via the underlying states of their participants. Often, these participant states are not explicitly mentioned, instead left to be inferred by the reader. A model that understands narratives should likewise infer these implicit states, and even reason about the impact of changes to these states on the narrative. To facilitate this goal, we introduce a new crowdsourced English-language, Participant States dataset, PASTA. This dataset contains inferable participant states; a counterfactual perturbation to each state; and the changes to the story that would be necessary if the counterfactual were true. We introduce three state-based reasoning tasks that test for the ability to infer when a state is entailed by a story, to revise a story conditioned on a counterfactual state, and to explain the most likely state change given a revised story. Experiments show that today\u2019s LLMs can reason about states to some degree, but there is large room for improvement, especially in problems requiring access and ability to reason with diverse types of knowledge (e.g., physical, numerical, factual).1" +} diff --git a/replay_traces/2208.00329/two_pass_outputs/20260726T205558Z/pass_2_ui_payload.json b/replay_traces/2208.00329/two_pass_outputs/20260726T205558Z/pass_2_ui_payload.json new file mode 100644 index 0000000000000000000000000000000000000000..d5692e4663b04f221984ba94fa1a24ee2db3828d --- /dev/null +++ b/replay_traces/2208.00329/two_pass_outputs/20260726T205558Z/pass_2_ui_payload.json @@ -0,0 +1,208 @@ +{ + "target_paper_id": "e894fb15054d3bc9659060406a12dfd1055ae32e", + "target_title": "PASTA: A Dataset for Modeling PArticipant STAtes in Narratives", + "target_year": 2022, + "annotator_id": "replay_trace_builder", + "active_claim_id": "C1", + "claims": [ + { + "claim_id": "C1", + "text": "Dataset: A crowdsourced dataset of narrative stories annotated with implicit participant states, counterfactual perturbations, and revised stories, enabling state-based reasoning and generation tasks.", + "rewritten_claim": "Dataset: A crowdsourced dataset of narrative stories annotated with implicit participant states, counterfactual perturbations, and revised stories, enabling state-based reasoning and generation tasks.", + "cluster_id": "C2", + "decision": "YES_SUFFICIENT", + "notes": "", + "ingredients": [ + { + "ingredient_id": "C1.I1", + "ingredient": "Source corpus of short, self-contained narrative stories", + "canonical_ref_id": "A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories", + "canonical_grounding": { + "ref_id": "A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories", + "ref_year": "", + "ref_authors": "" + }, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "DATA_SOURCE", + "roles": [ + "DATA_SOURCE" + ], + "contribution": "Provides the base narrative stories (ROCStories) that are annotated and revised in the dataset.", + "rationale": "The dataset requires a substrate of short, commonsense-rich narratives to serve as the foundation for state inference and revision. ROCStories is the direct source used by the paper and maps cleanly to its canonical study.", + "evidence_span": "To create the PASTA dataset, we use stories from the extended ROCStories (Mostafazadeh et al., 2016) corpus for annotation by crowd workers." + } + }, + { + "ingredient_id": "C1.I2", + "ingredient": "Four-step crowdsourcing protocol for inferring states and generating counterfactual revisions", + "canonical_ref_id": "__NONE__", + "canonical_grounding": null, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "EVALUATION_PROTOCOL", + "roles": [ + "EVALUATION_PROTOCOL" + ], + "contribution": "Defines the workflow used to collect the structured annotations that make up the dataset.", + "rationale": "The specific four-step process that yields the dataset's tuples is a novel contribution of the target paper itself, so it maps cleanly to NONE.", + "evidence_span": "The annotation process has four main steps: 1. Infer a participant state... 2. Select minimal justification sentences... 3. Perturb the state... 4. Revise the story" + } + } + ], + "enabling_discoveries": [ + { + "ref_id": "A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories", + "ref_year": "", + "ref_authors": "", + "ingredient_id": "C1.I1", + "ingredient": "Source corpus of short, self-contained narrative stories", + "role": "DATA_SOURCE", + "roles": [ + "DATA_SOURCE" + ], + "contribution": "Provides the base narrative stories (ROCStories) that are annotated and revised in the dataset.", + "rationale": "The dataset requires a substrate of short, commonsense-rich narratives to serve as the foundation for state inference and revision. ROCStories is the direct source used by the paper and maps cleanly to its canonical study.", + "evidence_span": "To create the PASTA dataset, we use stories from the extended ROCStories (Mostafazadeh et al., 2016) corpus for annotation by crowd workers." + } + ] + }, + { + "claim_id": "C2", + "text": "Method: A crowdsourcing protocol for eliciting implicit participant states and counterfactual story revisions, enabling the construction of state-based narrative datasets.", + "rewritten_claim": "Method: A crowdsourcing protocol for eliciting implicit participant states and counterfactual story revisions, enabling the construction of state-based narrative datasets.", + "cluster_id": "C1", + "decision": "YES_SUFFICIENT", + "notes": "", + "ingredients": [ + { + "ingredient_id": "C2.I1", + "ingredient": "Counterfactual story rewriting framework", + "canonical_ref_id": "Counterfactual Story Reasoning and Generation", + "canonical_grounding": { + "ref_id": "Counterfactual Story Reasoning and Generation", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "Counterfactual Story Reasoning and Generation", + "ref_year": "", + "ref_authors": "" + }, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "CONCEPTUAL_FRAMEWORK", + "roles": [ + "CONCEPTUAL_FRAMEWORK" + ], + "contribution": "Provides the foundational concept of editing a short story based on a counterfactual context, which the protocol adapts for unobserved participant states.", + "rationale": "The methodology extends the idea of counterfactual story rewriting to implicit states. Qin et al. (TIME-TRAVEL) is explicitly cited as the most similar prior work that introduced this rewriting task framing.", + "evidence_span": "Most similar to our work is the TIME-TRAVEL dataset by Qin et al. (2019). It includes the Counterfactual Story Rewriting task" + } + }, + { + "ingredient_id": "C2.I2", + "ingredient": "Framework for annotating participant states in simple stories", + "canonical_ref_id": "Modeling Naive Psychology of Characters in Simple Commonsense Stories", + "canonical_grounding": { + "ref_id": "Modeling Naive Psychology of Characters in Simple Commonsense Stories", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "Modeling Naive Psychology of Characters in Simple Commonsense Stories", + "ref_year": "", + "ref_authors": "" + }, + "additional_ref_ids": [ + "Tracking State Changes in Procedural Text: a Challenge Dataset and Models for Process Paragraph Comprehension" + ], + "additional_groundings": [ + { + "ref_id": "Tracking State Changes in Procedural Text: a Challenge Dataset and Models for Process Paragraph Comprehension", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "Tracking State Changes in Procedural Text: a Challenge Dataset and Models for Process Paragraph Comprehension", + "ref_year": "", + "ref_authors": "" + } + ], + "canonical_annotation": { + "role": "CONCEPTUAL_FRAMEWORK", + "roles": [ + "CONCEPTUAL_FRAMEWORK" + ], + "contribution": "Provides the conceptual basis for annotating the implicit states (like emotions or motivations) of characters in short narratives.", + "rationale": "The methodology builds on prior work that established the tracking and annotation of entity/participant states in text. Rashkin et al. is a clean representative for tracking emotional/psychological states in simple stories, with Dalvi et al. serving as an additional grounding for physical states.", + "evidence_span": "Rashkin et al. (2018) tracks the emotional reactions and motivations of characters in simple stories, for a fixed/small set of attributes." + } + }, + { + "ingredient_id": "C2.I3", + "ingredient": "Two-stage quality control and worker-proficiency filtering mechanism", + "canonical_ref_id": "__NONE__", + "canonical_grounding": null, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "EVALUATION_PROTOCOL", + "roles": [ + "EVALUATION_PROTOCOL" + ], + "contribution": "Provides the filtering and expert-review workflow necessary to identify proficient workers and ensure high-quality complex annotations.", + "rationale": "The specific two-stage filtering process (strict initial criteria followed by expert review to identify \"proficient workers\") is a custom workflow designed in the target paper to make the complex annotation task viable. It maps cleanly to NONE.", + "evidence_span": "For filtering out noisy data from the collected responses, we follow a two-stage filtering process." + } + } + ], + "enabling_discoveries": [ + { + "ref_id": "Counterfactual Story Reasoning and Generation", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "Counterfactual Story Reasoning and Generation", + "ref_year": "", + "ref_authors": "", + "ingredient_id": "C2.I1", + "ingredient": "Counterfactual story rewriting framework", + "role": "CONCEPTUAL_FRAMEWORK", + "roles": [ + "CONCEPTUAL_FRAMEWORK" + ], + "contribution": "Provides the foundational concept of editing a short story based on a counterfactual context, which the protocol adapts for unobserved participant states.", + "rationale": "The methodology extends the idea of counterfactual story rewriting to implicit states. Qin et al. (TIME-TRAVEL) is explicitly cited as the most similar prior work that introduced this rewriting task framing.", + "evidence_span": "Most similar to our work is the TIME-TRAVEL dataset by Qin et al. (2019). It includes the Counterfactual Story Rewriting task" + }, + { + "ref_id": "Modeling Naive Psychology of Characters in Simple Commonsense Stories", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "Modeling Naive Psychology of Characters in Simple Commonsense Stories", + "ref_year": "", + "ref_authors": "", + "ingredient_id": "C2.I2", + "ingredient": "Framework for annotating participant states in simple stories", + "role": "CONCEPTUAL_FRAMEWORK", + "roles": [ + "CONCEPTUAL_FRAMEWORK" + ], + "contribution": "Provides the conceptual basis for annotating the implicit states (like emotions or motivations) of characters in short narratives.", + "rationale": "The methodology builds on prior work that established the tracking and annotation of entity/participant states in text. Rashkin et al. is a clean representative for tracking emotional/psychological states in simple stories, with Dalvi et al. serving as an additional grounding for physical states.", + "evidence_span": "Rashkin et al. (2018) tracks the emotional reactions and motivations of characters in simple stories, for a fixed/small set of attributes." + } + ] + } + ] +} diff --git a/replay_traces/2208.00329/two_pass_outputs/20260726T205558Z/run_output.json b/replay_traces/2208.00329/two_pass_outputs/20260726T205558Z/run_output.json new file mode 100644 index 0000000000000000000000000000000000000000..3a69d48020ca12dc26d2608ef44fbc88d2b7e7ed --- /dev/null +++ b/replay_traces/2208.00329/two_pass_outputs/20260726T205558Z/run_output.json @@ -0,0 +1,235 @@ +{ + "paper_id": "2208.00329", + "paper_dir": "/Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785099279_8df5ecbb/processed_papers/2208.00329", + "generated_at": "2026-07-26T20:59:24.835889+00:00", + "reasoner_model": "gemini/gemini-3.1-pro-preview", + "formatter_model": "gemini/gemini-3.1-pro-preview", + "judge_model": "gemini/gemini-3.1-pro-preview", + "candidate_count": 3, + "include_reference_examples": true, + "prompt_profile": "full", + "reasoning_candidate_paths": [ + "/Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785099279_8df5ecbb/two_pass_outputs/2208.00329/20260726T205558Z/pass_1_reasoning.output.candidate_1.md", + "/Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785099279_8df5ecbb/two_pass_outputs/2208.00329/20260726T205558Z/pass_1_reasoning.output.candidate_2.md", + "/Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785099279_8df5ecbb/two_pass_outputs/2208.00329/20260726T205558Z/pass_1_reasoning.output.candidate_3.md" + ], + "selected_reasoning_candidate": "candidate_1", + "selected_candidate_index": 0, + "selected_reasoning_path": "/Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785099279_8df5ecbb/two_pass_outputs/2208.00329/20260726T205558Z/pass_1_reasoning.selected.md", + "judge_output_path": "/Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785099279_8df5ecbb/two_pass_outputs/2208.00329/20260726T205558Z/pass_1_reasoning.judge.output.json", + "formatter_attempts": [ + { + "attempt": 1, + "status": "success" + } + ], + "ui_payload_path": "/Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785099279_8df5ecbb/two_pass_outputs/2208.00329/20260726T205558Z/pass_2_ui_payload.json", + "ui_payload": { + "target_paper_id": "e894fb15054d3bc9659060406a12dfd1055ae32e", + "target_title": "PASTA: A Dataset for Modeling PArticipant STAtes in Narratives", + "target_year": 2022, + "annotator_id": "replay_trace_builder", + "active_claim_id": "C1", + "claims": [ + { + "claim_id": "C1", + "text": "Dataset: A crowdsourced dataset of narrative stories annotated with implicit participant states, counterfactual perturbations, and revised stories, enabling state-based reasoning and generation tasks.", + "rewritten_claim": "Dataset: A crowdsourced dataset of narrative stories annotated with implicit participant states, counterfactual perturbations, and revised stories, enabling state-based reasoning and generation tasks.", + "cluster_id": "C2", + "decision": "YES_SUFFICIENT", + "notes": "", + "ingredients": [ + { + "ingredient_id": "C1.I1", + "ingredient": "Source corpus of short, self-contained narrative stories", + "canonical_ref_id": "A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories", + "canonical_grounding": { + "ref_id": "A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories", + "ref_year": "", + "ref_authors": "" + }, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "DATA_SOURCE", + "roles": [ + "DATA_SOURCE" + ], + "contribution": "Provides the base narrative stories (ROCStories) that are annotated and revised in the dataset.", + "rationale": "The dataset requires a substrate of short, commonsense-rich narratives to serve as the foundation for state inference and revision. ROCStories is the direct source used by the paper and maps cleanly to its canonical study.", + "evidence_span": "To create the PASTA dataset, we use stories from the extended ROCStories (Mostafazadeh et al., 2016) corpus for annotation by crowd workers." + } + }, + { + "ingredient_id": "C1.I2", + "ingredient": "Four-step crowdsourcing protocol for inferring states and generating counterfactual revisions", + "canonical_ref_id": "__NONE__", + "canonical_grounding": null, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "EVALUATION_PROTOCOL", + "roles": [ + "EVALUATION_PROTOCOL" + ], + "contribution": "Defines the workflow used to collect the structured annotations that make up the dataset.", + "rationale": "The specific four-step process that yields the dataset's tuples is a novel contribution of the target paper itself, so it maps cleanly to NONE.", + "evidence_span": "The annotation process has four main steps: 1. Infer a participant state... 2. Select minimal justification sentences... 3. Perturb the state... 4. Revise the story" + } + } + ], + "enabling_discoveries": [ + { + "ref_id": "A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories", + "ref_year": "", + "ref_authors": "", + "ingredient_id": "C1.I1", + "ingredient": "Source corpus of short, self-contained narrative stories", + "role": "DATA_SOURCE", + "roles": [ + "DATA_SOURCE" + ], + "contribution": "Provides the base narrative stories (ROCStories) that are annotated and revised in the dataset.", + "rationale": "The dataset requires a substrate of short, commonsense-rich narratives to serve as the foundation for state inference and revision. ROCStories is the direct source used by the paper and maps cleanly to its canonical study.", + "evidence_span": "To create the PASTA dataset, we use stories from the extended ROCStories (Mostafazadeh et al., 2016) corpus for annotation by crowd workers." + } + ] + }, + { + "claim_id": "C2", + "text": "Method: A crowdsourcing protocol for eliciting implicit participant states and counterfactual story revisions, enabling the construction of state-based narrative datasets.", + "rewritten_claim": "Method: A crowdsourcing protocol for eliciting implicit participant states and counterfactual story revisions, enabling the construction of state-based narrative datasets.", + "cluster_id": "C1", + "decision": "YES_SUFFICIENT", + "notes": "", + "ingredients": [ + { + "ingredient_id": "C2.I1", + "ingredient": "Counterfactual story rewriting framework", + "canonical_ref_id": "Counterfactual Story Reasoning and Generation", + "canonical_grounding": { + "ref_id": "Counterfactual Story Reasoning and Generation", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "Counterfactual Story Reasoning and Generation", + "ref_year": "", + "ref_authors": "" + }, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "CONCEPTUAL_FRAMEWORK", + "roles": [ + "CONCEPTUAL_FRAMEWORK" + ], + "contribution": "Provides the foundational concept of editing a short story based on a counterfactual context, which the protocol adapts for unobserved participant states.", + "rationale": "The methodology extends the idea of counterfactual story rewriting to implicit states. Qin et al. (TIME-TRAVEL) is explicitly cited as the most similar prior work that introduced this rewriting task framing.", + "evidence_span": "Most similar to our work is the TIME-TRAVEL dataset by Qin et al. (2019). It includes the Counterfactual Story Rewriting task" + } + }, + { + "ingredient_id": "C2.I2", + "ingredient": "Framework for annotating participant states in simple stories", + "canonical_ref_id": "Modeling Naive Psychology of Characters in Simple Commonsense Stories", + "canonical_grounding": { + "ref_id": "Modeling Naive Psychology of Characters in Simple Commonsense Stories", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "Modeling Naive Psychology of Characters in Simple Commonsense Stories", + "ref_year": "", + "ref_authors": "" + }, + "additional_ref_ids": [ + "Tracking State Changes in Procedural Text: a Challenge Dataset and Models for Process Paragraph Comprehension" + ], + "additional_groundings": [ + { + "ref_id": "Tracking State Changes in Procedural Text: a Challenge Dataset and Models for Process Paragraph Comprehension", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "Tracking State Changes in Procedural Text: a Challenge Dataset and Models for Process Paragraph Comprehension", + "ref_year": "", + "ref_authors": "" + } + ], + "canonical_annotation": { + "role": "CONCEPTUAL_FRAMEWORK", + "roles": [ + "CONCEPTUAL_FRAMEWORK" + ], + "contribution": "Provides the conceptual basis for annotating the implicit states (like emotions or motivations) of characters in short narratives.", + "rationale": "The methodology builds on prior work that established the tracking and annotation of entity/participant states in text. Rashkin et al. is a clean representative for tracking emotional/psychological states in simple stories, with Dalvi et al. serving as an additional grounding for physical states.", + "evidence_span": "Rashkin et al. (2018) tracks the emotional reactions and motivations of characters in simple stories, for a fixed/small set of attributes." + } + }, + { + "ingredient_id": "C2.I3", + "ingredient": "Two-stage quality control and worker-proficiency filtering mechanism", + "canonical_ref_id": "__NONE__", + "canonical_grounding": null, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "EVALUATION_PROTOCOL", + "roles": [ + "EVALUATION_PROTOCOL" + ], + "contribution": "Provides the filtering and expert-review workflow necessary to identify proficient workers and ensure high-quality complex annotations.", + "rationale": "The specific two-stage filtering process (strict initial criteria followed by expert review to identify \"proficient workers\") is a custom workflow designed in the target paper to make the complex annotation task viable. It maps cleanly to NONE.", + "evidence_span": "For filtering out noisy data from the collected responses, we follow a two-stage filtering process." + } + } + ], + "enabling_discoveries": [ + { + "ref_id": "Counterfactual Story Reasoning and Generation", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "Counterfactual Story Reasoning and Generation", + "ref_year": "", + "ref_authors": "", + "ingredient_id": "C2.I1", + "ingredient": "Counterfactual story rewriting framework", + "role": "CONCEPTUAL_FRAMEWORK", + "roles": [ + "CONCEPTUAL_FRAMEWORK" + ], + "contribution": "Provides the foundational concept of editing a short story based on a counterfactual context, which the protocol adapts for unobserved participant states.", + "rationale": "The methodology extends the idea of counterfactual story rewriting to implicit states. Qin et al. (TIME-TRAVEL) is explicitly cited as the most similar prior work that introduced this rewriting task framing.", + "evidence_span": "Most similar to our work is the TIME-TRAVEL dataset by Qin et al. (2019). It includes the Counterfactual Story Rewriting task" + }, + { + "ref_id": "Modeling Naive Psychology of Characters in Simple Commonsense Stories", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "Modeling Naive Psychology of Characters in Simple Commonsense Stories", + "ref_year": "", + "ref_authors": "", + "ingredient_id": "C2.I2", + "ingredient": "Framework for annotating participant states in simple stories", + "role": "CONCEPTUAL_FRAMEWORK", + "roles": [ + "CONCEPTUAL_FRAMEWORK" + ], + "contribution": "Provides the conceptual basis for annotating the implicit states (like emotions or motivations) of characters in short narratives.", + "rationale": "The methodology builds on prior work that established the tracking and annotation of entity/participant states in text. Rashkin et al. is a clean representative for tracking emotional/psychological states in simple stories, with Dalvi et al. serving as an additional grounding for physical states.", + "evidence_span": "Rashkin et al. (2018) tracks the emotional reactions and motivations of characters in simple stories, for a fixed/small set of attributes." + } + ] + } + ] + } +} diff --git a/replay_traces/2211.08788/input_ids.json b/replay_traces/2211.08788/input_ids.json new file mode 100644 index 0000000000000000000000000000000000000000..e761bb67afd32ae56aa18e10e8f15f3cb453692d --- /dev/null +++ b/replay_traces/2211.08788/input_ids.json @@ -0,0 +1,7 @@ +[ + { + "id": "2211.08788", + "title": "", + "id_type": "ArXiv" + } +] \ No newline at end of file diff --git a/replay_traces/2211.08788/logs/step_01.log b/replay_traces/2211.08788/logs/step_01.log new file mode 100644 index 0000000000000000000000000000000000000000..6a519a3d3dae0b998e29406e1d03203e43bec26b --- /dev/null +++ b/replay_traces/2211.08788/logs/step_01.log @@ -0,0 +1,14 @@ +$ /Users/dingzifeng/SciFy-UMBC/demo_2026_07/.venv/bin/python src/step_01_fetch/fetch_metadata.py --ids /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785100600_1e6a0c3a/input_ids.json --outdir /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785100600_1e6a0c3a/processed_papers + +/Users/dingzifeng/SciFy-UMBC/demo_2026_07/src/step_01_fetch/fetch_metadata.py:95: DeprecationWarning: The 'Search.results' method is deprecated, use 'Client.results' instead + paper = next(search.results()) +/Users/dingzifeng/SciFy-UMBC/demo_2026_07/src/step_01_fetch/fetch_metadata.py:124: DeprecationWarning: Python 3.14 will, by default, filter extracted tar archives and reject files or modify their metadata. Use the filter argument to control this behavior. + tar.extractall(path=source_dir) +[INFO] Using ID list from /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785100600_1e6a0c3a/input_ids.json +[INFO] Output will be saved to /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785100600_1e6a0c3a/processed_papers +[DEBUG] fetch_one_acl_id: id=2211.08788 id_type=ArXiv title_len=0 s2_key_present=yes s2_key_len=44 +[DEBUG] wrote metadata to /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785100600_1e6a0c3a/processed_papers/2211.08788/paper_metadata.json +[INFO] Wrote combined BibTeX to /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785100600_1e6a0c3a/processed_papers/2211.08788/references.bib +[INFO] Processed LaTeX for 2211.08788 at /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785100600_1e6a0c3a/processed_papers/2211.08788/processed_main.tex +[SUCCESS] +done in 67.929044008255 s diff --git a/replay_traces/2211.08788/logs/step_02.log b/replay_traces/2211.08788/logs/step_02.log new file mode 100644 index 0000000000000000000000000000000000000000..cf1e5e9bfe711e8ab101f308aba57019380b92c2 --- /dev/null +++ b/replay_traces/2211.08788/logs/step_02.log @@ -0,0 +1,5 @@ +$ /Users/dingzifeng/SciFy-UMBC/demo_2026_07/.venv/bin/python src/step_02_mark_citations/replace_citation_markers.py --root /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785100600_1e6a0c3a/processed_papers + +[INFO] Found 1 paper dirs under /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785100600_1e6a0c3a/processed_papers +[OK] 2211.08788: updated 50 contexts over 55 +[SUMMARY] total_updated=50 over 55; skipped_incomplete=0 diff --git a/replay_traces/2211.08788/logs/step_03.log b/replay_traces/2211.08788/logs/step_03.log new file mode 100644 index 0000000000000000000000000000000000000000..4dbc42d26024c76a0e5a685b100784126c63193c --- /dev/null +++ b/replay_traces/2211.08788/logs/step_03.log @@ -0,0 +1,4 @@ +$ /Users/dingzifeng/SciFy-UMBC/demo_2026_07/.venv/bin/python src/step_03_usage_contexts/build_usage_contexts.py --root /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785100600_1e6a0c3a/processed_papers --out-name usage_contexts.json + +[INFO] Found 1 paper dirs under /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785100600_1e6a0c3a/processed_papers +[OK] 2211.08788: wrote usage_contexts.json (55 contexts from 21 citations) diff --git a/replay_traces/2211.08788/logs/step_04.log b/replay_traces/2211.08788/logs/step_04.log new file mode 100644 index 0000000000000000000000000000000000000000..66ad540ed22fa00c262a19b70b3140fb691c0627 --- /dev/null +++ b/replay_traces/2211.08788/logs/step_04.log @@ -0,0 +1,6 @@ +$ /Users/dingzifeng/SciFy-UMBC/demo_2026_07/.venv/bin/python src/step_04_label_citations/label_citation_functions.py --root /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785100600_1e6a0c3a/processed_papers --model-path Deep-Citation/Workspace/acl_scicite_wksp_trl/best_model.pt --model-data-dir Deep-Citation/Data --model-class-def Deep-Citation/Data/class_def.json --model-lm scibert --device cpu + +[INFO] Found 1 paper dirs under /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785100600_1e6a0c3a/processed_papers +Number of data instance: 1941 +[LABELED] 2211.08788 +[SUMMARY] labeled=1, skipped=0, missing_usage=0, empty_contexts=0 diff --git a/replay_traces/2211.08788/logs/step_05.log b/replay_traces/2211.08788/logs/step_05.log new file mode 100644 index 0000000000000000000000000000000000000000..25940db1642ecf290bc593ec7a811bf8be965b8d --- /dev/null +++ b/replay_traces/2211.08788/logs/step_05.log @@ -0,0 +1,14 @@ +$ /Users/dingzifeng/SciFy-UMBC/demo_2026_07/.venv/bin/python src/step_05_verify_uses_extends/verify_uses_extends.py --root /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785100600_1e6a0c3a/processed_papers --k 0 --batch-size 25 + +/Users/dingzifeng/SciFy-UMBC/demo_2026_07/src/common/llm_client.py:4: FutureWarning: + +All support for the `google.generativeai` package has ended. It will no longer be receiving +updates or bug fixes. Please switch to the `google.genai` package as soon as possible. +See README for more details: + +https://github.com/google-gemini/deprecated-generative-ai-python/blob/main/README.md + + import google.generativeai as genai +[INFO] Found 1 paper dirs under /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785100600_1e6a0c3a/processed_papers +[VERIFIED] 2211.08788 +[SUMMARY] verified=1, skipped=0, missing_labels=0, no_candidates=0 diff --git a/replay_traces/2211.08788/logs/step_06.log b/replay_traces/2211.08788/logs/step_06.log new file mode 100644 index 0000000000000000000000000000000000000000..bfd1abe2010932ea6eecfde8fd4222ac43cd5017 --- /dev/null +++ b/replay_traces/2211.08788/logs/step_06.log @@ -0,0 +1,7 @@ +$ /Users/dingzifeng/SciFy-UMBC/demo_2026_07/.venv/bin/python src/step_06_extract_paragraphs/extract_arxiv_paragraphs.py --root /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785100600_1e6a0c3a/processed_papers + +/Users/dingzifeng/SciFy-UMBC/demo_2026_07/src/step_06_extract_paragraphs/extract_arxiv_paragraphs.py:47: DeprecationWarning: Python 3.14 will, by default, filter extracted tar archives and reject files or modify their metadata. Use the filter argument to control this behavior. + tar.extractall(path) +[INFO] Found 1 paper dirs under /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785100600_1e6a0c3a/processed_papers +[PROCESSED] 2211.08788 +[SUMMARY] processed=1, skipped=0, missing_usage=0, missing_verified=0, empty_citing=0 diff --git a/replay_traces/2211.08788/logs/step_07.log b/replay_traces/2211.08788/logs/step_07.log new file mode 100644 index 0000000000000000000000000000000000000000..cb820ddc2ab124cf1faf6d35b0afd867e590b7aa --- /dev/null +++ b/replay_traces/2211.08788/logs/step_07.log @@ -0,0 +1,28 @@ +$ /Users/dingzifeng/SciFy-UMBC/demo_2026_07/.venv/bin/python src/step_07_extract_and_refine/extract_contributions_from_citations.py --root /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785100600_1e6a0c3a/processed_papers + +/Users/dingzifeng/SciFy-UMBC/demo_2026_07/src/common/llm_client.py:4: FutureWarning: + +All support for the `google.generativeai` package has ended. It will no longer be receiving +updates or bug fixes. Please switch to the `google.genai` package as soon as possible. +See README for more details: + +https://github.com/google-gemini/deprecated-generative-ai-python/blob/main/README.md + + import google.generativeai as genai +[INFO] Found 1 paper dirs under /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785100600_1e6a0c3a/processed_papers +[LABELED] 2211.08788 +[SUMMARY] labeled=1, skipped=0, missing_verified=0, missing_arxiv_paragraphs=0, no_confirmed=0 +$ /Users/dingzifeng/SciFy-UMBC/demo_2026_07/.venv/bin/python src/step_07_extract_and_refine/refine_and_filter_clusters_llm.py --root /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785100600_1e6a0c3a/processed_papers --inplace --overwrite + +/Users/dingzifeng/SciFy-UMBC/demo_2026_07/src/common/llm_client.py:4: FutureWarning: + +All support for the `google.generativeai` package has ended. It will no longer be receiving +updates or bug fixes. Please switch to the `google.genai` package as soon as possible. +See README for more details: + +https://github.com/google-gemini/deprecated-generative-ai-python/blob/main/README.md + + import google.generativeai as genai +[INFO] Found 1 paper dirs under /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785100600_1e6a0c3a/processed_papers +[REFINED] 2211.08788 +[SUMMARY] refined=1, skipped=0, missing_inputs=0, empty_clusters=0 diff --git a/replay_traces/2211.08788/processed_papers/2211.08788/paper_metadata.json b/replay_traces/2211.08788/processed_papers/2211.08788/paper_metadata.json new file mode 100644 index 0000000000000000000000000000000000000000..648894c63c61266dd58a4bc0a8b00777dd7eff0a --- /dev/null +++ b/replay_traces/2211.08788/processed_papers/2211.08788/paper_metadata.json @@ -0,0 +1,43 @@ +[ + { + "paperId": "09d6377921f503ab660c2c2f1a961c0cefe78d1f", + "externalIds": { + "ArXiv": "2211.08788", + "DBLP": "conf/acl/HuMZ24", + "DOI": "10.18653/v1/2024.acl-long.10", + "CorpusId": 253553639 + }, + "url": "https://www.semanticscholar.org/paper/09d6377921f503ab660c2c2f1a961c0cefe78d1f", + "title": "CSCD-NS: a Chinese Spelling Check Dataset for Native Speakers", + "venue": "Annual Meeting of the Association for Computational Linguistics", + "year": 2022, + "referenceCount": 27, + "citationCount": 21, + "openAccessPdf": { + "url": "https://aclanthology.org/2024.acl-long.10.pdf", + "status": "GOLD", + "license": "CCBY", + "disclaimer": "Notice: Paper or abstract available at https://arxiv.org/abs/2211.08788, which is subject to the license by the author or copyright owner provided with this content. Please go to the source to verify the license and copyright information for your use." + }, + "tldr": { + "model": "tldr@v2.0.0", + "text": "This paper presents CSCD-NS, the first Chinese spelling check dataset designed for native speakers, containing 40,000 samples from a Chinese social platform, and proposes a novel method that simulates the input process through an input method, generating large-scale and high-quality pseudo data that closely resembles the actual error distribution and outperforms existing methods." + }, + "publicationDate": "2022-11-16", + "authors": [ + { + "authorId": "2191038395", + "name": "Yong Hu" + }, + { + "authorId": "33427918", + "name": "Fandong Meng" + }, + { + "authorId": "48128428", + "name": "Jie Zhou" + } + ], + "abstract": "In this paper, we present CSCD-NS, the first Chinese spelling check (CSC) dataset designed for native speakers, containing 40,000 samples from a Chinese social platform. Compared with existing CSC datasets aimed at Chinese learners, CSCD-NS is ten times larger in scale and exhibits a distinct error distribution, with a significantly higher proportion of word-level errors. To further enhance the data resource, we propose a novel method that simulates the input process through an input method, generating large-scale and high-quality pseudo data that closely resembles the actual error distribution and outperforms existing methods. Moreover, we investigate the performance of various models in this scenario, including large language models (LLMs), such as ChatGPT. The result indicates that generative models underperform BERT-like classification models due to strict length and pronunciation constraints. The high prevalence of word-level errors also makes CSC for native speakers challenging enough, leaving substantial room for improvement." + } +] \ No newline at end of file diff --git a/replay_traces/2211.08788/processed_papers/2211.08788/usage_citing_paragraphs.json b/replay_traces/2211.08788/processed_papers/2211.08788/usage_citing_paragraphs.json new file mode 100644 index 0000000000000000000000000000000000000000..c2195d1e5a9b60c90ac59263c9df537c132d01d8 --- /dev/null +++ b/replay_traces/2211.08788/processed_papers/2211.08788/usage_citing_paragraphs.json @@ -0,0 +1,249 @@ +{ + "paper_id": "2211.08788", + "citing_papers": [ + { + "error": "missing_arxiv_id", + "title": "Towards More Reliable Chinese Spelling Correction: Fine-Grained Confidence Estimation Against Suboptimal Corrections", + "paper_id": "4b66b25de8c6c496f47ee58cf2677d0b377aa778", + "arxiv_id": "", + "contexts": [ + { + "text": "We also include a native-speaker-oriented dataset collected from social media, which contains a manually annotated subset and a large-scale pseudo-labeled corpus.", + "text_with_marker": "We also include a native-speaker-oriented dataset collected from social media, which contains a manually annotated subset and a large-scale pseudo-labeled corpus." + }, + { + "text": "2) Metrics: We follow previous works , [38] and report strict character-level precision, recall, and F1 scores for both detection and correction.", + "text_with_marker": "2) Metrics: We follow previous works , [38] and report strict character-level precision, recall, and F1 scores for both detection and correction." + } + ], + "target_info": { + "title": "CSCD-NS: a Chinese Spelling Check Dataset for Native Speakers", + "first_author_last": "Hu", + "year": "2022" + } + }, + { + "citing_paper_id": "7d64dd752bb29bb9aff8a836b4b1f5f3ccafe807", + "citing_title": "ChineseErrorCorrector3-4B: State-of-the-Art Chinese Spelling and Grammar Corrector", + "arxiv_id": "2511.17562", + "main_tex_file": "template.tex", + "bib_files": [], + "bib_texts": {}, + "target_bib_keys": [], + "contexts": [ + { + "text": "We first aggregate data for the two task families: \u2022 CSC Datasets ( D csc ): Aggregated from W271K, Medical[11], Lemon[12], ECSpell[19], and CSCD[14] (totaling approx.", + "text_with_marker": "We first aggregate data for the two task families: \u2022 CSC Datasets ( D csc ): Aggregated from W271K, Medical[11], Lemon[12], ECSpell[19], and CSCD[14] (totaling approx." + } + ], + "target_citing_paragraphs": [ + "\\paragraph{Dataset Composition.}\nWe first aggregate data for the two task families:\n\\begin{itemize}\n \\item CSC Datasets ($\\mathcal{D}_{\\text{csc}}$): Aggregated from W271K, Medical\\cite{jiang2022mcscsetspecialistannotateddatasetmedicaldomain}, Lemon\\cite{wu2023rethinking}, ECSpell\\cite{lv2023general}, and CSCD (totaling approx. 380K samples).\n \\item CGC Datasets ($\\mathcal{D}_{\\text{cgc}}$): Aggregated from CGED\\cite{rao2018overview, rao2020overview}, FCGEC\\cite{xu2022fcgec}, MuCGEC\\cite{zhang2022mucgec}, and NaCGEC\\cite{zhang2023nasgec} (totaling approx. 68K samples).\n\\end{itemize}" + ], + "matched_paragraphs": [ + { + "context_id": 1, + "context": "We first aggregate data for the two task families: \u2022 CSC Datasets ( D csc ): Aggregated from W271K, Medical[11], Lemon[12], ECSpell[19], and CSCD[14] (totaling approx.", + "context_with_marker": "We first aggregate data for the two task families: \u2022 CSC Datasets ( D csc ): Aggregated from W271K, Medical[11], Lemon[12], ECSpell[19], and CSCD[14] (totaling approx.", + "paragraph": "\\paragraph{Dataset Composition.}\nWe first aggregate data for the two task families:\n\\begin{itemize}\n \\item CSC Datasets ($\\mathcal{D}_{\\text{csc}}$): Aggregated from W271K, Medical\\cite{jiang2022mcscsetspecialistannotateddatasetmedicaldomain}, Lemon\\cite{wu2023rethinking}, ECSpell\\cite{lv2023general}, and CSCD (totaling approx. 380K samples).\n \\item CGC Datasets ($\\mathcal{D}_{\\text{cgc}}$): Aggregated from CGED\\cite{rao2018overview, rao2020overview}, FCGEC\\cite{xu2022fcgec}, MuCGEC\\cite{zhang2022mucgec}, and NaCGEC\\cite{zhang2023nasgec} (totaling approx. 68K samples).\n\\end{itemize}", + "overlap": 0.846 + } + ] + }, + { + "error": "missing_arxiv_id", + "title": "Incorporating Confused Phraseological Knowledge Based on Pinyin Input Method for Chinese Spelling Correction", + "paper_id": "0e2e19f1fae35d97b3178f9349caf95da540652d", + "arxiv_id": "", + "contexts": [ + { + "text": "The transformation method follows the pseudo-data construction method proposed by to ensure that the converted data closely resembles the actual scenario.", + "text_with_marker": "The transformation method follows the pseudo-data construction method proposed by to ensure that the converted data closely resembles the actual scenario." + }, + { + "text": "To better simulate real input scenarios, this paper uses the Chinese pinyin IME error correction dataset CSCD-IME as the basic dataset for the experiments, including 3,0000 training data, 5000 validation data, and 5000 test data.", + "text_with_marker": "To better simulate real input scenarios, this paper uses the Chinese pinyin IME error correction dataset CSCD-IME as the basic dataset for the experiments, including 3,0000 training data, 5000 validation data, and 5000 test data." + } + ], + "target_info": { + "title": "CSCD-NS: a Chinese Spelling Check Dataset for Native Speakers", + "first_author_last": "Hu", + "year": "2022" + } + }, + { + "citing_paper_id": "87a2958bd75a72e61b5b2549a63a222d7ca2a1ec", + "citing_title": "Mixture of Small and Large Models for Chinese Spelling Check", + "arxiv_id": "2506.06887", + "main_tex_file": "figures/alpha_beta.tex", + "bib_files": [ + "custom.bib" + ], + "bib_texts": { + "custom.bib": "% \u65b0\u6dfb\uff1a\n@inproceedings{Liu-etal-2024-ReLM,\n author = {Linfeng Liu and\n Hongqiu Wu and\n Hai Zhao},\n title = {{{C}hinese Spelling Correction as Rephrasing Language Model}},\n booktitle = {Proceedings of AAAI},\n pages = {18662--18670},\n year = {2024},\n address = \"Vancouver, Canada\",\n url = {https://doi.org/10.1609/aaai.v38i17.29829},\n}\n\n@inproceedings{yeh-etal-2013-chinese,\n address = {Nagoya, Japan},\n author = {Yeh, Jui-Feng and\n Li, Sheng-Feng and\n Wu, Mei-Rong and\n Chen, Wen-Yi and\n Su, Mao-Chuan},\n booktitle = {Proceedings of {SIGHAN}},\n pages = {43--48},\n title = {{{C}hinese Word Spelling Correction Based on N-gram Ranked Inverted Index List}},\n url = {https://aclanthology.org/W13-4407},\n year = {2013}\n}\n\n@inproceedings{yu-etal-2014-overview,\n title = \"{Overview of {SIGHAN} 2014 Bake-off for {C}hinese Spelling Check}\",\n author = \"Yu, Liang-Chih and\n Lee, Lung-Hao and\n Tseng, Yuen-Hsien and\n Chen, Hsin-Hsi\",\n booktitle = \"Proceedings of {CIPS}-{SIGHAN}\",\n year = \"2014\",\n address = \"Wuhan, China\",\n url = \"https://aclanthology.org/W14-6820\",\n pages = \"126--132\",\n}\n\n@inproceedings{yu-li-2014-chinese,\n address = {Wuhan, China},\n author = {Yu, Junjie and\n Li, Zhenghua},\n booktitle = {Proceedings of {CIPS}-{SIGHAN}},\n pages = {220--223},\n title = {{{C}hinese Spelling Error Detection and Correction Based on Language Model, Pronunciation, and Shape}},\n url = {https://aclanthology.org/W14-6835},\n year = {2014}\n}\n\n@inproceedings{huang-etal-2014-chinese,\n address = {Wuhan, China},\n author = {Huang, Qiang and\n Huang, Peijie and\n Zhang, Xinrui and\n Xie, Weijian and\n Hong, Kaiduo and\n Chen, Bingzhou and\n Huang, Lei},\n booktitle = {Proceedings of {CIPS}-{SIGHAN}},\n pages = {173--178},\n title = {{{C}hinese Spelling Check System Based on Tri-gram Model}},\n url = {https://aclanthology.org/W14-6827},\n year = {2014}\n}\n\n@inproceedings{xie-etal-2015-chinese,\n address = {Beijing, China},\n author = {Xie, Weijian and\n Huang, Peijie and\n Zhang, Xinrui and\n Hong, Kaiduo and\n Huang, Qiang and\n Chen, Bingzhou and\n Huang, Lei},\n booktitle = {Proceedings of {SIGHAN}},\n pages = {128--136},\n title = {{{C}hinese Spelling Check System Based on N-gram Model}},\n url = {https://aclanthology.org/W15-3120},\n year = {2015}\n}\n\n@inproceedings{wang-etal-2019-confusionset,\n title = \"{{Confusionset-guided Pointer Networks for {C}hinese Spelling Check}}\",\n author = \"Wang, Dingmin and\n Tay, Yi and\n Zhong, Li\",\n booktitle = \"Proceedings of ACL\",\n year = \"2019\",\n address = \"Florence, Italy\",\n url = \"https://aclanthology.org/P19-1578\",\n pages = \"5780--5785\"\n}\n\n\n@inproceedings{Bao-etal-2020-confusionset,\n author = {Zuyi Bao and\n Chen Li and\n Rui Wang},\n editor = {Trevor Cohn and\n Yulan He and\n Yang Liu},\n title = {{Chunk-based Chinese Spelling Check with Global Optimization}},\n booktitle = {Findings of EMNLP},\n pages = {2031--2040},\n year = {2020},\n address = {Online},\n url = {https://aclanthology.org/2020.findings-emnlp.184},\n}\n\n@inproceedings{zhang-etal-2020-softmasked,\n title = \"{Spelling Error Correction with Soft-Masked {BERT}}\",\n author = \"Zhang, Shaohua and\n Huang, Haoran and\n Liu, Jicong and\n Li, Hang\",\n booktitle = \"Proceedings of ACL\",\n year = \"2020\",\n address = \"Online\",\n url = \"https://aclanthology.org/2020.acl-main.82\",\n pages = \"882--890\",\n}\n\n@inproceedings{Cheng-etal-2020-SpellGCN,\n title = \"{{S}pell{GCN}: Incorporating Phonological and Visual Similarities into Language Models for {C}hinese Spelling Check}\",\n author = \"Cheng, Xingyi and\n Xu, Weidi and\n Chen, Kunlong and\n Jiang, Shaohua and\n Wang, Feng and\n Wang, Taifeng and\n Chu, Wei and\n Qi, Yuan\",\n booktitle = \"Proceedings of ACL\",\n year = \"2020\",\n address = \"Online\",\n url = \"https://aclanthology.org/2020.acl-main.81\",\n pages = \"871--881\",\n}\n\n\n@inproceedings{Guo-etal-2021-pretrain,\n author = {Zhao Guo and\n Yuan Ni and\n Keqiang Wang and\n Wei Zhu and\n Guotong Xie},\n title = {{Global Attention Decoder for {C}hinese Spelling Error Correction}},\n booktitle = {Findings of ACL-IJCNLP},\n pages = {1419--1428},\n year = {2021},\n address = \"Online\",\n url = {https://doi.org/10.18653/v1/2021.findings-acl.122},\n}\n\n@inproceedings{Li-etal-2021-pretrain,\n title = \"{Exploration and Exploitation: Two Ways to Improve {C}hinese Spelling Correction Models}\",\n author = \"Li, Chong and\n Zhang, Cenyuan and\n Zheng, Xiaoqing and\n Huang, Xuanjing\",\n booktitle = \"Proceedings of ACL-IJCNLP\",\n year = \"2021\",\n address = \"Online\",\n url = \"https://aclanthology.org/2021.acl-short.56\",\n pages = \"441--446\",\n}\n\n\n@inproceedings{Liu-etal-2021a-pretrain,\n author = {Shulin Liu and\n Tao Yang and\n Tianchi Yue and\n Feng Zhang and\n Di Wang},\n title = {{{PLOME:} Pre-training with Misspelled Knowledge for {C}hinese Spelling\n Correction}},\n booktitle = {Proceedings of ACL-IJCNLP},\n address = \"Online\",\n pages = {2991--3000},\n year = {2021},\n url = {https://doi.org/10.18653/v1/2021.acl-long.233},\n}\n\n@inproceedings{Huang-etal-2021-pretrain,\n title = \"{{PHMOS}pell: Phonological and Morphological Knowledge Guided {C}hinese Spelling Check}\",\n author = \"Huang, Li and\n Li, Junjie and\n Jiang, Weiwei and\n Zhang, Zhiyu and\n Chen, Minchuan and\n Wang, Shaojun and\n Xiao, Jing\",\n booktitle = \"Proceedings of ACL-IJCNLP\",\n year = \"2021\",\n address = \"Online\",\n url = \"https://aclanthology.org/2021.acl-long.464\",\n pages = \"5958--5967\",\n}\n\n@inproceedings{Xu-etal-2021-realise,\n title = \"{Read, Listen, and See: Leveraging Multimodal Information Helps {C}hinese Spell Checking}\",\n author = \"Xu, Heng-Da and\n Li, Zhongli and\n Zhou, Qingyu and\n Li, Chao and\n Wang, Zizhen and\n Cao, Yunbo and\n Huang, Heyan and\n Mao, Xian-Ling\",\n booktitle = \"Findings of ACL-IJCNLP\",\n year = \"2021\",\n address = \"Online\",\n url = \"https://aclanthology.org/2021.findings-acl.64\",\n pages = \"716--728\",\n}\n\n\n@inproceedings{Sun-etal-2021-ChineseBERT,\n title = \"{{C}hinese{BERT}: {C}hinese Pretraining Enhanced by Glyph and {P}inyin Information}\",\n author = \"Sun, Zijun and\n Li, Xiaoya and\n Sun, Xiaofei and\n Meng, Yuxian and\n Ao, Xiang and\n He, Qing and\n Wu, Fei and\n Li, Jiwei\",\n booktitle = \"Proceedings of ACL-IJCNLP\",\n year = \"2021\",\n address = \"Online\",\n url = \"https://aclanthology.org/2021.acl-long.161\",\n pages = \"2065--2075\"\n}\n\n\n@inproceedings{Li-etal-2022-SCOPE,\n author = {Jiahao Li and\n Quan Wang and\n Zhendong Mao and\n Junbo Guo and\n Yanyan Yang and\n Yongdong Zhang},\n title = {{Improving {C}hinese Spelling Check by Character Pronunciation Prediction:\n The Effects of Adaptivity and Granularity}},\n booktitle = {Proceedings of EMNLP},\n pages = {4275--4286},\n year = {2022},\n address = \"Abu Dhabi, United Arab Emirates\",\n url = {https://doi.org/10.18653/v1/2022.emnlp-main.287},\n}\n\n@inproceedings{wu-etal-2013-chinese,\n title = \"{{C}hinese Spelling Check Evaluation at {SIGHAN} Bake-off 2013}\",\n author = \"Wu, Shih-Hung and\n Liu, Chao-Lin and\n Lee, Lung-Hao\",\n booktitle = \"Proceedings of SIGHAN\",\n year = \"2013\",\n address = \"Nagoya, Japan\",\n url = \"https://aclanthology.org/W13-4406\",\n pages = \"35--42\",\n}\n\n@inproceedings{tseng-etal-2015-introduction,\n title = \"{Introduction to {SIGHAN} 2015 Bake-off for {C}hinese Spelling Check}\",\n author = \"Tseng, Yuen-Hsien and\n Lee, Lung-Hao and\n Chang, Li-Ping and\n Chen, Hsin-Hsi\",\n booktitle = \"Proceedings of SIGHAN\",\n year = \"2015\",\n address = \"Beijing, China\",\n url = \"https://aclanthology.org/W15-3106\",\n pages = \"32--37\",\n}\n\n@inproceedings{wang-etal-2018-hybrid,\n title = \"{A Hybrid Approach to Automatic Corpus Generation for {C}hinese Spelling Check}\",\n author = \"Wang, Dingmin and\n Song, Yan and\n Li, Jing and\n Han, Jialong and\n Zhang, Haisong\",\n booktitle = \"Proceedings of EMNLP\",\n year = \"2018\",\n address = \"Brussels, Belgium\",\n url = \"https://aclanthology.org/D18-1273\",\n pages = \"2517--2527\"\n}\n\n@article{Lv-etal-2023-ECSpell,\n author = {Qi Lv and\n Ziqiang Cao and\n Lei Geng and\n Chunhui Ai and\n Xu Yan and\n Guohong Fu},\n title = {{General and Domain-adaptive {C}hinese Spelling Check with Error-consistent\n Pretraining}},\n journal = {TALLIP},\n address = {New York, NY, USA},\n pages = {1--18},\n year = {2023},\n url = {https://doi.org/10.1145/3564271}\n}\n\n@article{Yang-etal-2023-rSIGHAN,\n author = {Liner Yang and\n Xin Liu and\n Tianxin Liao and\n Zhenghao Liu and\n Mengyan Wang and\n Xuezhi Fang and\n Erhong Yang},\n title = {{Is {C}hinese Spelling Check ready? Understanding the correction behavior\n in real-world scenarios}},\n journal = {AI Open},\n pages = {183--192},\n year = {2023},\n url = {https://doi.org/10.1016/j.aiopen.2023.10.004}\n}\n\n@inproceedings{Wu-etal-2023c-LEMON,\n title = \"{Rethinking Masked Language Modeling for {C}hinese Spelling Correction}\",\n author = \"Wu, Hongqiu and\n Zhang, Shaohua and\n Zhang, Yuchen and\n Zhao, Hai\",\n booktitle = \"Proceedings of ACL\",\n year = \"2023\",\n address = \"Toronto, Canada\",\n url = \"https://aclanthology.org/2023.acl-long.600\",\n pages = \"10743--10756\"\n}\n\n@inproceedings{zhang-etal-2021-correcting,\n title = \"Correcting {C}hinese Spelling Errors with Phonetic Pre-training\",\n author = \"Zhang, Ruiqing and\n Pang, Chao and\n Zhang, Chuanqiang and\n Wang, Shuohuan and\n He, Zhongjun and\n Sun, Yu and\n Wu, Hua and\n Wang, Haifeng\",\n editor = \"Zong, Chengqing and\n Xia, Fei and\n Li, Wenjie and\n Navigli, Roberto\",\n booktitle = \"Findings of the Association for Computational Linguistics: ACL-IJCNLP 2021\",\n month = aug,\n year = \"2021\",\n address = \"Online\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2021.findings-acl.198\",\n doi = \"10.18653/v1/2021.findings-acl.198\",\n pages = \"2250--2261\",\n}\n\n@inproceedings{liu-etal-2010-visually,\n title = \"{Visually and Phonologically Similar Characters in Incorrect Simplified {C}hinese Words}\",\n author = \"Liu, Chao-Lin and\n Lai, Min-Hua and\n Chuang, Yi-Hsuan and\n Lee, Chia-Ying\",\n booktitle = \"Proceedings of COLING\",\n year = \"2010\",\n address = \"Beijing, China\",\n url = \"https://aclanthology.org/C10-2085\",\n pages = \"739--747\",\n}\n\n@article{Cui-etal-2019-bertwwm,\n author = {Yiming Cui and\n Wanxiang Che and\n Ting Liu and\n Bing Qin and\n Ziqing Yang and\n Shijin Wang and\n Guoping Hu},\n title = {{Pre-Training with Whole Word Masking for {C}hinese {BERT}}},\n journal = {ArXiv preprint},\n year = {2019},\n url = {http://arxiv.org/abs/1906.08101},\n}\n\n@inproceedings{Yang-etal-2023-ner,\n author = {Yifei Yang and\n Hongqiu Wu and\n Hai Zhao},\n title = {{Attack Named Entity Recognition by Entity Boundary Interference}},\n booktitle = {Proceedings of LREC-COLING},\n pages = {1734--1744},\n year = {2024},\n address = \"Torino, Italia\",\n url = {https://aclanthology.org/2024.lrec-main.153}\n}\n\n@inproceedings{Martins-etal-2004-searchengine,\n author = {Bruno Martins and\n M{\\'{a}}rio J. Silva},\n title = {{Spelling Correction for Search Engine Queries}},\n booktitle = {Proceedings of EsTAL},\n pages = {372--383},\n year = {2004},\n address = \"Alicante, Spain\",\n url = {https://doi.org/10.1007/978-3-540-30228-5\\_33}\n}\n\n@inproceedings{gao-etal-2010-large,\n title = \"{A Large Scale Ranker-Based System for Search Query Spelling Correction}\",\n author = \"Gao, Jianfeng and\n Li, Xiaolong and\n Micol, Daniel and\n Quirk, Chris and\n Sun, Xu\",\n booktitle = \"Proceedings of COLING\",\n year = \"2010\",\n address = \"Beijing, China\",\n url = \"https://aclanthology.org/C10-1041\",\n pages = \"358--366\",\n}\n\n@inproceedings{Devlin-etal-2019-bert,\n author = {Jacob Devlin and\n Ming{-}Wei Chang and\n Kenton Lee and\n Kristina Toutanova},\n title = {{{BERT:} Pre-training of Deep Bidirectional Transformers for Language\n Understanding}},\n booktitle = {Proceedings of NAACL-HLT},\n pages = {4171--4186},\n address = \"Minneapolis, Minnesota\",\n year = {2019},\n url = {https://aclanthology.org/N19-1423}\n}\n\n@inproceedings{kipf-etal-2017-gcn,\n author = {Thomas N. Kipf and\n Max Welling},\n title = {{Semi-Supervised Classification with Graph Convolutional Networks}},\n booktitle = {Proceedings of ICLR},\n year = {2017},\n url = {https://openreview.net/forum?id=SJU4ayYgl},\n}\n\n@inproceedings{Huang-etal-2023-module,\n title = \"{A Frustratingly Easy Plug-and-Play Detection-and-Reasoning Module for {C}hinese Spelling Check}\",\n author = \"Huang, Haojing and\n Ye, Jingheng and\n Zhou, Qingyu and\n Li, Yinghui and\n Li, Yangning and\n Zhou, Feng and\n Zheng, Hai-Tao\",\n booktitle = \"Findings of EMNLP\",\n year = \"2023\",\n address = \"Singapore\",\n url = \"https://aclanthology.org/2023.findings-emnlp.771\",\n pages = \"11514--11525\",\n}\n\n\n@inproceedings{kernighan-etal-1990-spelling,\n title = \"{A Spelling Correction Program Based on a Noisy Channel Model}\",\n author = \"Kernighan, Mark D. and\n Church, Kenneth W. and\n Gale, William A.\",\n booktitle = \"Proceedings of COLING\",\n year = \"1990\",\n url = \"https://aclanthology.org/C90-2036\",\n}\n\n@inproceedings{hong-etal-2019-faspell,\n title = \"{{FASP}ell: A Fast, Adaptable, Simple, Powerful {C}hinese Spell Checker Based On {DAE}-Decoder Paradigm}\",\n author = \"Hong, Yuzhong and\n Yu, Xianguo and\n He, Neng and\n Liu, Nan and\n Liu, Junhui\",\n booktitle = \"Proceedings of WNUT\",\n year = \"2019\",\n address = \"Hong Kong, China\",\n url = \"https://aclanthology.org/D19-5522\",\n pages = \"160--169\",\n}\n\n@article{yang-etal-2023-chinese,\n author = {Liner Yang and Xin Liu and Tianxin Liao and Zhenghao Liu and Mengyan Wang and Xuezhi Fang and Erhong Yang},\n journal = {AI Open},\n pages = {183-192},\n title = {{Is {C}hinese Spelling Check ready? Understanding the correction behavior in real-world scenarios}},\n url = {https://www.sciencedirect.com/science/article/pii/S2666651023000207},\n year = {2023}\n}\n\n@inproceedings{ma-etal-2022-linguistic,\n title = \"{Linguistic Rules-Based Corpus Generation for Native {C}hinese Grammatical Error Correction}\",\n author = \"Ma, Shirong and\n Li, Yinghui and\n Sun, Rongyi and\n Zhou, Qingyu and\n Huang, Shulin and\n Zhang, Ding and\n Yangning, Li and\n Liu, Ruiyang and\n Li, Zhongli and\n Cao, Yunbo and\n Zheng, Haitao and\n Shen, Ying\",\n editor = \"Goldberg, Yoav and\n Kozareva, Zornitsa and\n Zhang, Yue\",\n booktitle = \"Findings of the Association for Computational Linguistics: EMNLP 2022\",\n month = dec,\n year = \"2022\",\n address = \"Abu Dhabi, United Arab Emirates\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2022.findings-emnlp.40\",\n doi = \"10.18653/v1/2022.findings-emnlp.40\",\n pages = \"576--589\",\n}\n\n@misc{li-etal-2023-ineffectiveness,\n title={On the (In)Effectiveness of Large Language Models for Chinese Text Correction}, \n author={Yinghui Li and Haojing Huang and Shirong Ma and Yong Jiang and Yangning Li and Feng Zhou and Hai-Tao Zheng and Qingyu Zhou},\n year={2023},\n eprint={2307.09007},\n archivePrefix={arXiv},\n primaryClass={cs.CL}\n}\n\n@inproceedings{li-etal-2022-learning-dictionary,\n title = \"{{Learning from the Dictionary: Heterogeneous Knowledge Guided Fine-tuning for {C}hinese Spell Checking}}\",\n author = \"Li, Yinghui and\n Ma, Shirong and\n Zhou, Qingyu and\n Li, Zhongli and\n Yangning, Li and\n Huang, Shulin and\n Liu, Ruiyang and\n Li, Chao and\n Cao, Yunbo and\n Zheng, Haitao\",\n booktitle = \"Findings of EMNLP\",\n year = \"2022\",\n address = \"Abu Dhabi, United Arab Emirates\",\n url = \"https://aclanthology.org/2022.findings-emnlp.18\",\n pages = \"238--249\",\n}\n\n@inproceedings{Dathathri-etal-2020-Plug,\ntitle={{Plug and Play Language Models: A Simple Approach to Controlled Text Generation}},\nauthor={Sumanth Dathathri and Andrea Madotto and Janice Lan and Jane Hung and Eric Frank and Piero Molino and Jason Yosinski and Rosanne Liu},\nbooktitle={Proceedings of ICLR},\nyear={2020},\nurl={https://openreview.net/forum?id=H1edEyBKDS}\n}\n\n@inproceedings{krause-etal-2021-gedi-generative,\n title = \"{{G}e{D}i: Generative Discriminator Guided Sequence Generation}\",\n author = \"Krause, Ben and\n Gotmare, Akhilesh Deepak and\n McCann, Bryan and\n Keskar, Nitish Shirish and\n Joty, Shafiq and\n Socher, Richard and\n Rajani, Nazneen Fatema\",\n booktitle = \"Findings of EMNLP\",\n year = \"2021\",\n address = \"Punta Cana, Dominican Republic\",\n url = \"https://aclanthology.org/2021.findings-emnlp.424\",\n pages = \"4929--4952\",\n}\n\n@inproceedings{liu-etal-2021-dexperts,\n title = \"{{DE}xperts: Decoding-Time Controlled Text Generation with Experts and Anti-Experts}\",\n author = \"Liu, Alisa and\n Sap, Maarten and\n Lu, Ximing and\n Swayamdipta, Swabha and\n Bhagavatula, Chandra and\n Smith, Noah A. and\n Choi, Yejin\",\n booktitle = \"Proceedings of ACL-IJCNLP\",\n year = \"2021\",\n address = \"Online\",\n url = \"https://aclanthology.org/2021.acl-long.522\",\n pages = \"6691--6706\",\n}\n\n@inproceedings{zhou-etal-2023-improving-seq2seq,\n title = \"{Improving {S}eq2{S}eq Grammatical Error Correction via Decoding Interventions}\",\n author = \"Zhou, Houquan and\n Liu, Yumeng and\n Li, Zhenghua and\n Zhang, Min and\n Zhang, Bo and\n Li, Chen and\n Zhang, Ji and\n Huang, Fei\",\n booktitle = \"Findings of EMNLP\",\n year = \"2023\",\n address = \"Singapore\",\n url = \"https://aclanthology.org/2023.findings-emnlp.495\",\n pages = \"7393--7405\",\n}\n\n@article{gulcehre-etal-2015-using,\n author = {{\\c{C}}aglar G{\\\"{u}}l{\\c{c}}ehre and\n Orhan Firat and\n Kelvin Xu and\n Kyunghyun Cho and\n Lo{\\\"{\\i}}c Barrault and\n Huei{-}Chi Lin and\n Fethi Bougares and\n Holger Schwenk and\n Yoshua Bengio},\n title = {{On Using Monolingual Corpora in Neural Machine Translation}},\n journal = {ArXiv preprint},\n year = {2015},\n url = {http://arxiv.org/abs/1503.03535}\n}\n\n@inproceedings{kannan-etal-2018-analysis,\n author = {Anjuli Kannan and\n Yonghui Wu and\n Patrick Nguyen and\n Tara N. Sainath and\n Zhifeng Chen and\n Rohit Prabhavalkar},\n title = {{An Analysis of Incorporating an External Language Model into a Sequence-to-Sequence\n Model}},\n booktitle = {Proceedings of ICASSP},\n address = \"Calgary, Canada\",\n pages = {5824--5828},\n year = {2018},\n url = {https://doi.org/10.1109/ICASSP.2018.8462682}\n}\n\n@inproceedings{zhao-etal-2019-shallow,\n author = {Ding Zhao and\n Tara N. Sainath and\n David Rybach and\n Pat Rondon and\n Deepti Bhatia and\n Bo Li and\n Ruoming Pang},\n title = {{Shallow-Fusion End-to-End Contextual Biasing}},\n booktitle = {Proceedings of INTERSPEECH},\n pages = {1418--1422},\naddress = \"Graz, Austria\",\n year = {2019},\n url = {https://doi.org/10.21437/Interspeech.2019-1209}\n}\n\n@inproceedings{liu-etal-2022-craspell,\n title = \"{{CRAS}pell: A Contextual Typo Robust Approach to Improve {C}hinese Spelling Correction}\",\n author = \"Liu, Shulin and\n Song, Shengkang and\n Yue, Tianchi and\n Yang, Tao and\n Cai, Huihui and\n Yu, TingHao and\n Sun, Shengli\",\n editor = \"Muresan, Smaranda and\n Nakov, Preslav and\n Villavicencio, Aline\",\n booktitle = \"Findings of the Association for Computational Linguistics: ACL 2022\",\n month = may,\n year = \"2022\",\n address = \"Dublin, Ireland\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2022.findings-acl.237\",\n doi = \"10.18653/v1/2022.findings-acl.237\",\n pages = \"3008--3018\",\n}\n\n@misc{li-etal-2024-cllm,\n title={{C-LLM: Learn to Check Chinese Spelling Errors Character by Character}}, \n author={Kunting Li and Yong Hu and Liang He and Fandong Meng and Jie Zhou},\n year={2024},\n eprint={2406.16536},\n archivePrefix={arXiv},\n primaryClass={cs.CL},\n url={https://arxiv.org/abs/2406.16536}, \n}\n\n@inproceedings{du-etal-2022-glm,\n title = \"{GLM}: General Language Model Pretraining with Autoregressive Blank Infilling\",\n author = \"Du, Zhengxiao and\n Qian, Yujie and\n Liu, Xiao and\n Ding, Ming and\n Qiu, Jiezhong and\n Yang, Zhilin and\n Tang, Jie\",\n booktitle = \"Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)\",\n month = may,\n year = \"2022\",\n address = \"Dublin, Ireland\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2022.acl-long.26\",\n doi = \"10.18653/v1/2022.acl-long.26\",\n pages = \"320--335\",\n}\n\n@misc{yang-etal-2023-baichuan2,\n title={{Baichuan 2: Open Large-scale Language Models}}, \n author={Aiyuan Yang and Bin Xiao and Bingning Wang and Borong Zhang and Ce Bian and Chao Yin and Chenxu Lv and Da Pan and Dian Wang and Dong Yan and Fan Yang and Fei Deng and Feng Wang and Feng Liu and Guangwei Ai and Guosheng Dong and Haizhou Zhao and Hang Xu and Haoze Sun and Hongda Zhang and Hui Liu and Jiaming Ji and Jian Xie and JunTao Dai and Kun Fang and Lei Su and Liang Song and Lifeng Liu and Liyun Ru and Luyao Ma and Mang Wang and Mickel Liu and MingAn Lin and Nuolan Nie and Peidong Guo and Ruiyang Sun and Tao Zhang and Tianpeng Li and Tianyu Li and Wei Cheng and Weipeng Chen and Xiangrong Zeng and Xiaochuan Wang and Xiaoxi Chen and Xin Men and Xin Yu and Xuehai Pan and Yanjun Shen and Yiding Wang and Yiyu Li and Youxin Jiang and Yuchen Gao and Yupeng Zhang and Zenan Zhou and Zhiying Wu},\n year={2023},\n eprint={2309.10305},\n archivePrefix={arXiv},\n primaryClass={cs.CL},\n url={https://arxiv.org/abs/2309.10305}, \n}\n\n@misc{li-2023-ineffectivenessLLMs,\n title={{On the (In)Effectiveness of Large Language Models for Chinese Text Correction}}, \n author={Yinghui Li and Haojing Huang and Shirong Ma and Yong Jiang and Yangning Li and Feng Zhou and Hai-Tao Zheng and Qingyu Zhou},\n year={2023},\n eprint={2307.09007},\n archivePrefix={arXiv},\n primaryClass={cs.CL},\n url={https://arxiv.org/abs/2307.09007}, \n}\n\n@inproceedings{yu-etal-2024-refining,\n title = \"{Refining Corpora from a Model Calibration Perspective for {C}hinese Spelling Correction}\",\n author = \"Yu, Dingyao and\n An, Yang and\n Ye, Wei and\n Xiao, Xiongfeng and\n Mao, Shaoguang and\n Ge, Tao and\n Zhang, Shikun\",\n editor = \"Ku, Lun-Wei and\n Martins, Andre and\n Srikumar, Vivek\",\n booktitle = \"Findings of the Association for Computational Linguistics ACL 2024\",\n month = aug,\n year = \"2024\",\n address = \"Bangkok, Thailand and virtual meeting\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2024.findings-acl.914\",\n pages = \"15468--15480\",\n}\n\n@inproceedings{zhou-etal-2024-llm-csc,\n title = \"A Simple yet Effective Training-free Prompt-free Approach to {C}hinese Spelling Correction Based on Large Language Models\",\n author = \"Zhou, Houquan and\n Li, Zhenghua and\n Zhang, Bo and\n Li, Chen and\n Lai, Shaopeng and\n Zhang, Ji and\n Huang, Fei and\n Zhang, Min\",\n editor = \"Al-Onaizan, Yaser and\n Bansal, Mohit and\n Chen, Yun-Nung\",\n booktitle = \"Proceedings of EMNLP\",\n month = nov,\n year = \"2024\",\n address = \"Miami, Florida, USA\",\n url = \"https://aclanthology.org/2024.emnlp-main.966/\",\n doi = \"10.18653/v1/2024.emnlp-main.966\",\n pages = \"17446--17467\",\n}\n\n@inproceedings{hu-etal-2024-cscd,\n title = {{{CSCD}-{NS}: a {C}hinese Spelling Check Dataset for Native Speakers}},\n author = {Hu, Yong and\n Meng, Fandong and\n Zhou, Jie},\n booktitle = {Proceedings of ACL},\n year = {2024},\n address = {Bangkok, Thailand},\n url = {https://aclanthology.org/2024.acl-long.10},\n pages = {146--159}\n}\n\n@inproceedings{jiang-etal-2022-mcscset,\n author = {Wangjie Jiang and\n Zhihao Ye and\n Zijing Ou and\n Ruihui Zhao and\n Jianguang Zheng and\n Yi Liu and\n Bang Liu and\n Siheng Li and\n Yujiu Yang and\n Yefeng Zheng},\n bibsource = {dblp computer science bibliography, https://dblp.org},\n biburl = {https://dblp.org/rec/conf/cikm/JiangYOZZLLLY022.bib},\n booktitle = {Proceedings of CIKM},\n doi = {10.1145/3511808.3557636},\n editor = {Mohammad Al Hasan and\n Li Xiong},\n pages = {4084--4088},\n timestamp = {Fri, 31 Mar 2023 10:36:04 +0200},\n title = {{MCSCSet: {A} Specialist-annotated Dataset for Medical-domain {C}hinese\n Spelling Correction}},\n url = {https://doi.org/10.1145/3511808.3557636},\n year = {2022}\n}\n\n@misc{qwen2.5,\n title = {{Qwen2.5: A Party of Foundation Models}},\n url = {https://qwenlm.github.io/blog/qwen2.5/},\n author = {Qwen Team},\n month = {September},\n year = {2024}\n}\n\n@inproceedings{liu-etal-2024-arm,\n title = \"{{ARM}: An Alignment-and-Replacement Module for {C}hinese Spelling Check Based on {LLM}s}\",\n author = \"Liu, Changchun and\n Zhang, Kai and\n Jiang, Junzhe and\n Liu, Zirui and\n Tao, Hanqing and\n Gao, Min and\n Chen, Enhong\",\n editor = \"Al-Onaizan, Yaser and\n Bansal, Mohit and\n Chen, Yun-Nung\",\n booktitle = \"Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing\",\n month = nov,\n year = \"2024\",\n address = \"Miami, Florida, USA\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2024.emnlp-main.567/\",\n doi = \"10.18653/v1/2024.emnlp-main.567\",\n pages = \"10156--10168\",\n}\n\n@misc{qiao-etal-2024-disc,\n title={{DISC: Plug-and-Play Decoding Intervention with Similarity of Characters for Chinese Spelling Check}}, \n author={Ziheng Qiao and Houquan Zhou and Yumeng Liu and Zhenghua Li and Min Zhang and Bo Zhang and Chen Li and Ji Zhang and Fei Huang},\n year={2024},\n eprint={2412.12863},\n archivePrefix={arXiv},\n primaryClass={cs.CL},\n url={https://arxiv.org/abs/2412.12863}, \n}\n\n@inproceedings{dong-etal-2024-rich,\n title = \"{Rich Semantic Knowledge Enhanced Large Language Models for Few-shot {C}hinese Spell Checking}\",\n author = \"Dong, Ming and\n Chen, Yujing and\n Zhang, Miao and\n Sun, Hao and\n He, Tingting\",\n editor = \"Ku, Lun-Wei and\n Martins, Andre and\n Srikumar, Vivek\",\n booktitle = \"Findings of the Association for Computational Linguistics: ACL 2024\",\n month = aug,\n year = \"2024\",\n address = \"Bangkok, Thailand\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2024.findings-acl.439/\",\n doi = \"10.18653/v1/2024.findings-acl.439\",\n pages = \"7372--7383\"\n}\n\n@inproceedings{huang-etal-2021-phmospell,\n title = \"{{PHMOS}pell: Phonological and Morphological Knowledge Guided {C}hinese Spelling Check}\",\n author = \"Huang, Li and\n Li, Junjie and\n Jiang, Weiwei and\n Zhang, Zhiyu and\n Chen, Minchuan and\n Wang, Shaojun and\n Xiao, Jing\",\n editor = \"Zong, Chengqing and\n Xia, Fei and\n Li, Wenjie and\n Navigli, Roberto\",\n booktitle = \"Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 1: Long Papers)\",\n month = aug,\n year = \"2021\",\n address = \"Online\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2021.acl-long.464/\",\n doi = \"10.18653/v1/2021.acl-long.464\",\n pages = \"5958--5967\",\n}\n\n@inproceedings{li-etal-2022-improving-chinese,\n title = \"{Improving {C}hinese Spelling Check by Character Pronunciation Prediction: The Effects of Adaptivity and Granularity}\",\n author = \"Li, Jiahao and\n Wang, Quan and\n Mao, Zhendong and\n Guo, Junbo and\n Yang, Yanyan and\n Zhang, Yongdong\",\n editor = \"Goldberg, Yoav and\n Kozareva, Zornitsa and\n Zhang, Yue\",\n booktitle = \"Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing\",\n month = dec,\n year = \"2022\",\n address = \"Abu Dhabi, United Arab Emirates\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2022.emnlp-main.287/\",\n doi = \"10.18653/v1/2022.emnlp-main.287\",\n pages = \"4275--4286\",\n}\n\n@inproceedings{liang-etal-2023-disentangled,\n title = \"{Disentangled Phonetic Representation for {C}hinese Spelling Correction}\",\n author = \"Liang, Zihong and\n Quan, Xiaojun and\n Wang, Qifan\",\n editor = \"Rogers, Anna and\n Boyd-Graber, Jordan and\n Okazaki, Naoaki\",\n booktitle = \"Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)\",\n month = jul,\n year = \"2023\",\n address = \"Toronto, Canada\",\n publisher = \"Association for Computational Linguistics\",\n url = \"https://aclanthology.org/2023.acl-long.755/\",\n doi = \"10.18653/v1/2023.acl-long.755\",\n pages = \"13509--13521\",\n}\n\n@inproceedings{zhang-etal-2022-mucgec,\n title = \"{{M}u{CGEC}: a Multi-Reference Multi-Source Evaluation Dataset for {C}hinese Grammatical Error Correction}\",\n author = \"Zhang, Yue and\n Li, Zhenghua and\n Bao, Zuyi and\n Li, Jiacheng and\n Zhang, Bo and\n Li, Chen and\n Huang, Fei and\n Zhang, Min\",\n booktitle = \"Proceedings of NAACL\",\n month = jul,\n year = \"2022\",\n address = \"Seattle, United States\",\n url = \"https://aclanthology.org/2022.naacl-main.227\",\n pages = \"3118--3130\",\n}\n\n@article{cai-etal-2024-internlm2,\n author = {Zheng Cai and Maosong Cao and Haojiong Chen and Kai Chen and Keyu Chen and Xin Chen and Xun Chen and Zehui Chen and Zhi Chen and Pei Chu and Xiaoyi Dong and Haodong Duan and Qi Fan and Zhaoye Fei and Yang Gao and Jiaye Ge and Chenya Gu and Yuzhe Gu and Tao Gui and Aijia Guo and Qipeng Guo and Conghui He and Yingfan Hu and Ting Huang and Tao Jiang and Penglong Jiao and Zhenjiang Jin and Zhikai Lei and Jiaxing Li and Jingwen Li and Linyang Li and Shuaibin Li and Wei Li and Yining Li and Hongwei Liu and Jiangning Liu and Jiawei Hong and Kaiwen Liu and Kuikun Liu and Xiaoran Liu and Chengqi Lv and Haijun Lv and Kai Lv and Li Ma and Runyuan Ma and Zerun Ma and Wenchang Ning and Linke Ouyang and Jiantao Qiu and Yuan Qu and Fukai Shang and Yunfan Shao and Demin Song and Zifan Song and Zhihao Sui and Peng Sun and Yu Sun and Huanze Tang and Bin Wang and Guoteng Wang and Jiaqi Wang and Jiayu Wang and Rui Wang and Yudong Wang and Ziyi Wang and Xingjian Wei and Qizhen Weng and Fan Wu and Yingtong Xiong and Chao Xu and Ruiliang Xu and Hang Yan and Yirong Yan and Xiaogui Yang and Haochen Ye and Huaiyuan Ying and Jia Yu and Jing Yu and Yuhang Zang and Chuyu Zhang and Li Zhang and Pan Zhang and Peng Zhang and Ruijie Zhang and Shuo Zhang and Songyang Zhang and Wenjian Zhang and Wenwei Zhang and Xingcheng Zhang and Xinyue Zhang and Hui Zhao and Qian Zhao and Xiaomeng Zhao and Fengzhe Zhou and Zaida Zhou and Jingming Zhuo and Yicheng Zou and Xipeng Qiu and Yu Qiao and Dahua Lin},\n journal = {ArXiv preprint},\n title = {{InternLM2 Technical Report}},\n url = {https://arxiv.org/abs/2403.17297},\n volume = {abs/2403.17297},\n year = {2024}\n}\n\n@article{bai-etal-2023-qwen,\n author = {Jinze Bai and\n Shuai Bai and\n Yunfei Chu and\n Zeyu Cui and\n Kai Dang and\n Xiaodong Deng and\n Yang Fan and\n Wenbin Ge and\n Yu Han and\n Fei Huang and\n Binyuan Hui and\n Luo Ji and\n Mei Li and\n Junyang Lin and\n Runji Lin and\n Dayiheng Liu and\n Gao Liu and\n Chengqiang Lu and\n Keming Lu and\n Jianxin Ma and\n Rui Men and\n Xingzhang Ren and\n Xuancheng Ren and\n Chuanqi Tan and\n Sinan Tan and\n Jianhong Tu and\n Peng Wang and\n Shijie Wang and\n Wei Wang and\n Shengguang Wu and\n Benfeng Xu and\n Jin Xu and\n An Yang and\n Hao Yang and\n Jian Yang and\n Shusheng Yang and\n Yang Yao and\n Bowen Yu and\n Hongyi Yuan and\n Zheng Yuan and\n Jianwei Zhang and\n Xingxuan Zhang and\n Yichang Zhang and\n Zhenru Zhang and\n Chang Zhou and\n Jingren Zhou and\n Xiaohuan Zhou and\n Tianhang Zhu},\n journal = {ArXiv preprint},\n title = {{Qwen Technical Report}},\n url = {https://arxiv.org/abs/2309.16609},\n volume = {abs/2309.16609},\n year = {2023}\n}\n\n" + }, + "target_bib_keys": [ + "ma-etal-2022-linguistic", + "li-etal-2022-learning-dictionary", + "liu-etal-2022-craspell", + "zhang-etal-2022-mucgec" + ], + "contexts": [ + { + "text": "Due to the lack of domain diversity in the SIGHAN datasets, we further conduct experiments on more diverse datasets, including LEMON (Wu et al., 2023), EC-Spell (Lv et al., 2023), CSCD-NS , and MCSCSet (Jiang et al., 2022), all of which were written by native speakers.", + "text_with_marker": "Due to the lack of domain diversity in the SIGHAN datasets, we further conduct experiments on more diverse datasets, including LEMON (Wu et al., 2023), EC-Spell (Lv et al., 2023), CSCD-NS , and MCSCSet (Jiang et al., 2022), all of which were written by native speakers." + }, + { + "text": "Table 2 presents experimental results across five benchmark datasets: rSIGHANs, , MC-SCSet, ECSpell, and LEMON.", + "text_with_marker": "Table 2 presents experimental results across five benchmark datasets: rSIGHANs, , MC-SCSet, ECSpell, and LEMON." + } + ], + "target_citing_paragraphs": [], + "matched_paragraphs": [] + }, + { + "citing_paper_id": "9a6479bdd7605029d3972dea4c64a2e914f2755a", + "citing_title": "A Training-free LLM-based Approach to General Chinese Character Error Correction", + "arxiv_id": "2502.15266", + "main_tex_file": "content/appendix.tex", + "bib_files": [], + "bib_texts": {}, + "target_bib_keys": [], + "contexts": [ + { + "text": "Figure 11 shows the performance of our method with varying beam sizes on the and C2EC datasets.", + "text_with_marker": "Figure 11 shows the performance of our method with varying beam sizes on the and C2EC datasets." + }, + { + "text": "For conventional CSC, following Li et al. (2024), we use two representative datasets: CSCD-NS and Lemon (Wu et al., 2023).", + "text_with_marker": "For conventional CSC, following Li et al. (2024), we use two representative datasets: CSCD-NS and Lemon (Wu et al., 2023)." + }, + { + "text": "We randomly sampled 50 sentences of varying lengths from the development set of and C2EC to evaluate the running time of our method on Qwen2.", + "text_with_marker": "We randomly sampled 50 sentences of varying lengths from the development set of and C2EC to evaluate the running time of our method on Qwen2." + }, + { + "text": "When compared to the TfPf method, the SFT methods show superior performance on the in-domain dataset .", + "text_with_marker": "When compared to the TfPf method, the SFT methods show superior performance on the in-domain dataset ." + }, + { + "text": "Training Details of SFT Baselines The SFT models were trained on a combined dataset consisting of 271k pseudo sentence pairs (Wang et al., 2018) and the training data.", + "text_with_marker": "Training Details of SFT Baselines The SFT models were trained on a combined dataset consisting of 271k pseudo sentence pairs (Wang et al., 2018) and the training data." + } + ], + "target_citing_paragraphs": [], + "matched_paragraphs": [ + { + "context_id": 1, + "context": "Figure 11 shows the performance of our method with varying beam sizes on the and C2EC datasets.", + "context_with_marker": "Figure 11 shows the performance of our method with varying beam sizes on the and C2EC datasets.", + "paragraph": "\\subsection{Impact of Beam Size}\n\\label{app:beam_size}\n\\input{figures/beam_results}\nFigure~\\ref{fig:beam_size} shows the performance of our method with varying beam sizes on the CSCD-NS and C2EC datasets.\nThe results indicate that our method performs well even with a small beam size.\nIn particular, a beam size of 2 is sufficient to surpass \\texttt{ICL}, \\texttt{ICL-RR}, and \\texttt{TfPf} on both datasets.\nInterestingly, as the beam size increases, the performance of \\texttt{ICL} remains almost unchanged, while the performance of \\texttt{ICL-RR} steadily improves.\nThis suggests that while \\texttt{ICL} can find a set of good candidates, it struggles to rank them properly.", + "overlap": 0.833 + }, + { + "context_id": 2, + "context": "For conventional CSC, following Li et al. (2024), we use two representative datasets: CSCD-NS and Lemon (Wu et al., 2023).", + "context_with_marker": "For conventional CSC, following Li et al. (2024), we use two representative datasets: CSCD-NS and Lemon (Wu et al., 2023).", + "paragraph": "\\subsection{Dataset Statistics}\n\\label{subsec:dataset_statistics}\nIn this work, we use three datasets to evaluate the performance of our method.\nAll datasets used in this work are publicly available.\nSpecifically, the CSCD-NS dataset is publicly available under the MIT license, while the CCTC dataset is publicly available under the Apache 2.0 license.\nThe specifics of these datasets are listed in Table~\\ref{tab:dataset_statistics}.\nThe \\textbf{Evaluation Sentences} row in Table~\\ref{tab:dataset_statistics} shows the number of sentences actually used for evaluation.\nThis is because sentences where the original and corrected versions differ in length are excluded when evaluating CSC models, as done in previous works \\cite{liu-etal-2023-chinese,li-etal-2024-cllm,zhou-etal-2024-simple}.", + "overlap": 0.524 + }, + { + "context_id": 3, + "context": "We randomly sampled 50 sentences of varying lengths from the development set of and C2EC to evaluate the running time of our method on Qwen2.", + "context_with_marker": "We randomly sampled 50 sentences of varying lengths from the development set of and C2EC to evaluate the running time of our method on Qwen2.", + "paragraph": "\\input{tables/discussion_speed.tex}\n\\subsection{Run-time Analysis}\n\\label{app:runtime}\nWe randomly sampled 50 sentences of varying lengths from the development set of CSCD-NS and C2EC to evaluate the running time of our method on \\texttt{Qwen2.5\\,7B} compared to the baselines.\nThe experiment was conducted on a single NVIDIA A100 40GB GPU with the Intel Xeon Gold 6248R (3.00GHz) CPU.\nThe batch size was set to 1 during evaluation.\nThe results, as shown in Table~\\ref{tab:analysis:runtime}, indicate that our approach is approximately twice slower than \\texttt{TfPf}.\nThis increased time is due to the two forward passes of the large language model at each step to obtain the final score.\nHowever, since these two forward passes are independent, the process can be accelerated by parallelizing them if more GPUs are available.", + "overlap": 0.917 + }, + { + "context_id": 4, + "context": "When compared to the TfPf method, the SFT methods show superior performance on the in-domain dataset .", + "context_with_marker": "When compared to the TfPf method, the SFT methods show superior performance on the in-domain dataset .", + "paragraph": "When compared to the \\texttt{TfPf} method, the \\texttt{SFT} methods show superior performance on the in-domain dataset CSCD-NS.\nHowever, they perform less effectively on the out-of-domain dataset Lemon, particularly with a 7B model.\nThis suggests that the \\texttt{SFT} methods might overfit to the in-domain dataset CSCD-NS, limiting their generalization to the out-of-domain dataset Lemon.", + "overlap": 0.882 + } + ] + }, + { + "citing_paper_id": "619346184ae157741fc1c4ac74447ab2455cd6f3", + "citing_title": "A Simple yet Effective Training-free Prompt-free Approach to Chinese Spelling Correction Based on Large Language Models", + "arxiv_id": "2410.04027", + "main_tex_file": "tables/detail_results.tex", + "bib_files": [], + "bib_texts": {}, + "target_bib_keys": [], + "contexts": [ + { + "text": "To identify these outliers, we leveraged tools from previous studies (Wu et al., 2023; by incorporating their structure confusion sets and spelling similarity matrices.", + "text_with_marker": "To identify these outliers, we leveraged tools from previous studies (Wu et al., 2023; by incorporating their structure confusion sets and spelling similarity matrices." + }, + { + "text": "We perform experiments across five distinct CSC datasets: Sighans (Wu et al., 2013; Yu et al., 2014; Tseng et al., 2015), CSCD-NS , MCSCSet (Jiang et al., 2022), ECSpell (Lv et al., 2023), and Lemon (Wu et al., 2023), covering a broad spectrum of domains and genres.", + "text_with_marker": "We perform experiments across five distinct CSC datasets: Sighans (Wu et al., 2013; Yu et al., 2014; Tseng et al., 2015), CSCD-NS , MCSCSet (Jiang et al., 2022), ECSpell (Lv et al., 2023), and Lemon (Wu et al., 2023), covering a broad spectrum of domains and genres." + } + ], + "target_citing_paragraphs": [], + "matched_paragraphs": [] + }, + { + "citing_paper_id": "571754f4ed4303b02a1a5e78c6605f483b2ea6e0", + "citing_title": "Eval-GCSC: A New Metric for Evaluating ChatGPT's Performance in Chinese Spelling Correction", + "arxiv_id": "2311.08219", + "main_tex_file": "acl_latex.tex", + "bib_files": [], + "bib_texts": {}, + "target_bib_keys": [], + "contexts": [ + { + "text": "Metrics We computed Eval-CSC based on Script , and implemented Eval-GCSC following the procedures outlined in Section 4.3.", + "text_with_marker": "Metrics We computed Eval-CSC based on Script , and implemented Eval-GCSC following the procedures outlined in Section 4.3." + } + ], + "target_citing_paragraphs": [], + "matched_paragraphs": [ + { + "context_id": 1, + "context": "Metrics We computed Eval-CSC based on Script , and implemented Eval-GCSC following the procedures outlined in Section 4.3.", + "context_with_marker": "Metrics We computed Eval-CSC based on Script , and implemented Eval-GCSC following the procedures outlined in Section 4.3.", + "paragraph": "\\textbf{Metrics} We computed Eval-CSC based on Script \\cite{https://doi.org/10.48550/arxiv.2211.08788}, and implemented Eval-GCSC following the procedures outlined in Section~\\ref{sec:Calculation of Eval-GCSC}. Considering that the performance in error correction, rather than error detection, better reflects the proficiency of text spelling correction in CSC, we only retained the metrics results under the correction sub-task. Semantic similarity was evaluated using the three approaches described in Section~\\ref{sec:Semantic similarity}.", + "overlap": 0.81 + } + ] + }, + { + "error": "missing_arxiv_id", + "title": "Evolving Chinese Spelling Correction with Corrector-Verifier Collaboration", + "paper_id": "db2208df9148b0ddc65e21d7a0883ca54edd4e66", + "arxiv_id": "", + "contexts": [ + { + "text": "We generate the synthetic data using IME .", + "text_with_marker": "We generate the synthetic data using IME ." + }, + { + "text": "5-14b finetuned combines the human annotated data from and 271K pseudo-data generated by ASR or OCR as the training data, and uses character-level tokenization.", + "text_with_marker": "5-14b finetuned combines the human annotated data from and 271K pseudo-data generated by ASR or OCR as the training data, and uses character-level tokenization." + }, + { + "text": "To evaluate the quality of generated training data, we probe the annotation accuracy of different approaches on development set.", + "text_with_marker": "To evaluate the quality of generated training data, we probe the annotation accuracy of different approaches on development set." + }, + { + "text": "Following recent works, we use two CSC benchmarks, LEMON (Wu et al., 2023) and CSCD-NS .", + "text_with_marker": "Following recent works, we use two CSC benchmarks, LEMON (Wu et al., 2023) and CSCD-NS ." + }, + { + "text": "We compare the annotation results with the gold labels from .", + "text_with_marker": "We compare the annotation results with the gold labels from ." + }, + { + "text": "The data we use is the training set of , which is in high quality.", + "text_with_marker": "The data we use is the training set of , which is in high quality." + } + ], + "target_info": { + "title": "CSCD-NS: a Chinese Spelling Check Dataset for Native Speakers", + "first_author_last": "Hu", + "year": "2022" + } + } + ] +} \ No newline at end of file diff --git a/replay_traces/2211.08788/processed_papers/2211.08788/usage_context_labels.json b/replay_traces/2211.08788/processed_papers/2211.08788/usage_context_labels.json new file mode 100644 index 0000000000000000000000000000000000000000..4dcd6c196159f156399dc711b48164b023b2c02f --- /dev/null +++ b/replay_traces/2211.08788/processed_papers/2211.08788/usage_context_labels.json @@ -0,0 +1,748 @@ +{ + "paper_id": "2211.08788", + "num_contexts": 55, + "label_set": [ + "Background", + "Uses", + "Extends", + "CompareOrContrast", + "Motivation", + "Future" + ], + "labels": [ + { + "id": 1, + "label": "Uses", + "confidence": 0.9987574815750122, + "confidence_margin": 0.9981331825256348, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "4b66b25de8c6c496f47ee58cf2677d0b377aa778", + "citing_title": "Towards More Reliable Chinese Spelling Correction: Fine-Grained Confidence Estimation Against Suboptimal Corrections", + "text": "We also include a native-speaker-oriented dataset collected from social media, which contains a manually annotated subset and a large-scale pseudo-labeled corpus." + }, + { + "id": 2, + "label": "Background", + "confidence": 0.9967809915542603, + "confidence_margin": 0.9953776597976685, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "4b66b25de8c6c496f47ee58cf2677d0b377aa778", + "citing_title": "Towards More Reliable Chinese Spelling Correction: Fine-Grained Confidence Estimation Against Suboptimal Corrections", + "text": "Chinese spelling errors mainly arise from incorrect pinyin input or the selection of similar-sounding characters ." + }, + { + "id": 3, + "label": "Extends", + "confidence": 0.5481100082397461, + "confidence_margin": 0.09964737296104431, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "4b66b25de8c6c496f47ee58cf2677d0b377aa778", + "citing_title": "Towards More Reliable Chinese Spelling Correction: Fine-Grained Confidence Estimation Against Suboptimal Corrections", + "text": "2) Metrics: We follow previous works , [38] and report strict character-level precision, recall, and F1 scores for both detection and correction." + }, + { + "id": 4, + "label": "Uses", + "confidence": 0.9936357736587524, + "confidence_margin": 0.990795373916626, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "7d64dd752bb29bb9aff8a836b4b1f5f3ccafe807", + "citing_title": "ChineseErrorCorrector3-4B: State-of-the-Art Chinese Spelling and Grammar Corrector", + "text": "We first aggregate data for the two task families: \u2022 CSC Datasets ( D csc ): Aggregated from W271K, Medical[11], Lemon[12], ECSpell[19], and CSCD[14] (totaling approx." + }, + { + "id": 5, + "label": "Uses", + "confidence": 0.9989981055259705, + "confidence_margin": 0.99839186668396, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "0e2e19f1fae35d97b3178f9349caf95da540652d", + "citing_title": "Incorporating Confused Phraseological Knowledge Based on Pinyin Input Method for Chinese Spelling Correction", + "text": "The transformation method follows the pseudo-data construction method proposed by to ensure that the converted data closely resembles the actual scenario." + }, + { + "id": 6, + "label": "Uses", + "confidence": 0.9971441626548767, + "confidence_margin": 0.9956931471824646, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "0e2e19f1fae35d97b3178f9349caf95da540652d", + "citing_title": "Incorporating Confused Phraseological Knowledge Based on Pinyin Input Method for Chinese Spelling Correction", + "text": "To better simulate real input scenarios, this paper uses the Chinese pinyin IME error correction dataset CSCD-IME as the basic dataset for the experiments, including 3,0000 training data, 5000 validation data, and 5000 test data." + }, + { + "id": 7, + "label": "Background", + "confidence": 0.6917532086372375, + "confidence_margin": 0.4309380054473877, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "0e2e19f1fae35d97b3178f9349caf95da540652d", + "citing_title": "Incorporating Confused Phraseological Knowledge Based on Pinyin Input Method for Chinese Spelling Correction", + "text": "foradditionaltrain-ing." + }, + { + "id": 8, + "label": "Background", + "confidence": 0.7892221808433533, + "confidence_margin": 0.6328082084655762, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "1d4ba6d664b28e5e1854bf662f30ed8f2fbcf2f7", + "citing_title": "Vision Language Models Are Not (Yet) Spelling Correctors", + "text": "Existing datasets are primarily text-based, including those collected from language learners [3, 26], native speakers [6], and specific domains [8, 12, 17]." + }, + { + "id": 9, + "label": "Background", + "confidence": 0.9302701950073242, + "confidence_margin": 0.8730562329292297, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "87a2958bd75a72e61b5b2549a63a222d7ca2a1ec", + "citing_title": "Mixture of Small and Large Models for Chinese Spelling Check", + "text": "." + }, + { + "id": 10, + "label": "Background", + "confidence": 0.901209831237793, + "confidence_margin": 0.812825083732605, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "87a2958bd75a72e61b5b2549a63a222d7ca2a1ec", + "citing_title": "Mixture of Small and Large Models for Chinese Spelling Check", + "text": "CSCD-NS contains 40K annotated samples sourced from real posts on Sina Weibo, effectively reflecting the real-world error correction performance of CSC models." + }, + { + "id": 11, + "label": "Uses", + "confidence": 0.9681503176689148, + "confidence_margin": 0.9455533623695374, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "87a2958bd75a72e61b5b2549a63a222d7ca2a1ec", + "citing_title": "Mixture of Small and Large Models for Chinese Spelling Check", + "text": "Due to the lack of domain diversity in the SIGHAN datasets, we further conduct experiments on more diverse datasets, including LEMON (Wu et al., 2023), EC-Spell (Lv et al., 2023), CSCD-NS , and MCSCSet (Jiang et al., 2022), all of which were written by native speakers." + }, + { + "id": 12, + "label": "Uses", + "confidence": 0.9578426480293274, + "confidence_margin": 0.9294824004173279, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "87a2958bd75a72e61b5b2549a63a222d7ca2a1ec", + "citing_title": "Mixture of Small and Large Models for Chinese Spelling Check", + "text": "Table 2 presents experimental results across five benchmark datasets: rSIGHANs, , MC-SCSet, ECSpell, and LEMON." + }, + { + "id": 13, + "label": "Uses", + "confidence": 0.40921175479888916, + "confidence_margin": 0.027472972869873047, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "ce28a63f81ad7773235322506368743982fd2bc3", + "citing_title": "SmartBench: Is Your LLM Truly a Good Chinese Smartphone Assistant?", + "text": "3) To offer deeper insights into real-world edge-side deployment, we report the prefilling speed, output token generation speed, and power consumption on the iQOO 12 smartphone using the Qual-comm QNN SDK, as shown in Tab." + }, + { + "id": 14, + "label": "Uses", + "confidence": 0.8914547562599182, + "confidence_margin": 0.7866718769073486, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "9a6479bdd7605029d3972dea4c64a2e914f2755a", + "citing_title": "A Training-free LLM-based Approach to General Chinese Character Error Correction", + "text": "Specifically, the dataset is publicly available under the MIT license, while the CCTC dataset is publicly available under the Apache 2.0 license." + }, + { + "id": 15, + "label": "Uses", + "confidence": 0.9657821655273438, + "confidence_margin": 0.9454605579376221, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "9a6479bdd7605029d3972dea4c64a2e914f2755a", + "citing_title": "A Training-free LLM-based Approach to General Chinese Character Error Correction", + "text": "Figure 11 shows the performance of our method with varying beam sizes on the and C2EC datasets." + }, + { + "id": 16, + "label": "Uses", + "confidence": 0.9988788962364197, + "confidence_margin": 0.9983606934547424, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "9a6479bdd7605029d3972dea4c64a2e914f2755a", + "citing_title": "A Training-free LLM-based Approach to General Chinese Character Error Correction", + "text": "For conventional CSC, following Li et al. (2024), we use two representative datasets: CSCD-NS and Lemon (Wu et al., 2023)." + }, + { + "id": 17, + "label": "Background", + "confidence": 0.8759418725967407, + "confidence_margin": 0.7589338421821594, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "9a6479bdd7605029d3972dea4c64a2e914f2755a", + "citing_title": "A Training-free LLM-based Approach to General Chinese Character Error Correction", + "text": " focuses on general domain performance, while Lemon evaluates zero-shot cross-domain capabilities." + }, + { + "id": 18, + "label": "Uses", + "confidence": 0.9989721775054932, + "confidence_margin": 0.9983739852905273, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "9a6479bdd7605029d3972dea4c64a2e914f2755a", + "citing_title": "A Training-free LLM-based Approach to General Chinese Character Error Correction", + "text": "We randomly sampled 50 sentences of varying lengths from the development set of and C2EC to evaluate the running time of our method on Qwen2." + }, + { + "id": 19, + "label": "CompareOrContrast", + "confidence": 0.9914980530738831, + "confidence_margin": 0.9875950217247009, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "9a6479bdd7605029d3972dea4c64a2e914f2755a", + "citing_title": "A Training-free LLM-based Approach to General Chinese Character Error Correction", + "text": "Compared to SFT models from Li et al. (2024), which are trained on the training set of the dataset, our method shows better performance on the out-of-domain Lemon dataset." + }, + { + "id": 20, + "label": "Uses", + "confidence": 0.594452977180481, + "confidence_margin": 0.4063270688056946, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "9a6479bdd7605029d3972dea4c64a2e914f2755a", + "citing_title": "A Training-free LLM-based Approach to General Chinese Character Error Correction", + "text": "When compared to the TfPf method, the SFT methods show superior performance on the in-domain dataset ." + }, + { + "id": 21, + "label": "CompareOrContrast", + "confidence": 0.32429027557373047, + "confidence_margin": 0.07043623924255371, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "9a6479bdd7605029d3972dea4c64a2e914f2755a", + "citing_title": "A Training-free LLM-based Approach to General Chinese Character Error Correction", + "text": "This means that while is an in-domain evaluation, the Lemon dataset serves as a cross-domain dataset for evaluating the generalization capabilities." + }, + { + "id": 22, + "label": "Uses", + "confidence": 0.9984955787658691, + "confidence_margin": 0.9976979494094849, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "9a6479bdd7605029d3972dea4c64a2e914f2755a", + "citing_title": "A Training-free LLM-based Approach to General Chinese Character Error Correction", + "text": "Training Details of SFT Baselines The SFT models were trained on a combined dataset consisting of 271k pseudo sentence pairs (Wang et al., 2018) and the training data." + }, + { + "id": 23, + "label": "Motivation", + "confidence": 0.4094448685646057, + "confidence_margin": 0.051053911447525024, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "9a6479bdd7605029d3972dea4c64a2e914f2755a", + "citing_title": "A Training-free LLM-based Approach to General Chinese Character Error Correction", + "text": "This suggests that the SFT methods might overfit to the in-domain dataset , limiting their generalization to the out-of-domain dataset Lemon." + }, + { + "id": 24, + "label": "Background", + "confidence": 0.9956165552139282, + "confidence_margin": 0.9927230477333069, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "9a6479bdd7605029d3972dea4c64a2e914f2755a", + "citing_title": "A Training-free LLM-based Approach to General Chinese Character Error Correction", + "text": "Chinese characters often have multiple visually or phonetically similar variants, making it easy to select an incorrect one when using input methods ." + }, + { + "id": 25, + "label": "Uses", + "confidence": 0.3049975037574768, + "confidence_margin": 0.0667286217212677, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "9a6479bdd7605029d3972dea4c64a2e914f2755a", + "citing_title": "A Training-free LLM-based Approach to General Chinese Character Error Correction", + "text": "Removing it leads to performance drops of 11.10 and 5.45 points on the and C2EC datasets, respectively." + }, + { + "id": 26, + "label": "CompareOrContrast", + "confidence": 0.5992077589035034, + "confidence_margin": 0.405576229095459, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "9a6479bdd7605029d3972dea4c64a2e914f2755a", + "citing_title": "A Training-free LLM-based Approach to General Chinese Character Error Correction", + "text": "While the Levenshtein distance-based model performs better on the C2EC dataset, it degrades the performance on the dataset, the conventional CSC dataset." + }, + { + "id": 27, + "label": "Background", + "confidence": 0.8859775066375732, + "confidence_margin": 0.7750557065010071, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "9a6479bdd7605029d3972dea4c64a2e914f2755a", + "citing_title": "A Training-free LLM-based Approach to General Chinese Character Error Correction", + "text": "Similarly, created a new dataset with real errors found on social media." + }, + { + "id": 28, + "label": "CompareOrContrast", + "confidence": 0.4725815951824188, + "confidence_margin": 0.18308067321777344, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "9a6479bdd7605029d3972dea4c64a2e914f2755a", + "citing_title": "A Training-free LLM-based Approach to General Chinese Character Error Correction", + "text": "Additionally, without any training, our method achieves performance on par with the SFT methods on the in-domain dataset , with scores of 71.53 versus 73.80 on the 14B model." + }, + { + "id": 29, + "label": "Background", + "confidence": 0.9345009922981262, + "confidence_margin": 0.886854887008667, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "7ea52ad5c9400b5e914e86cede6e812940b5a27f", + "citing_title": "Chinese Spelling Correction: A Comprehensive Survey of Progress, Challenges, and Opportunities", + "text": "The primary sources of errors in CSC typically stem from users\u2019 input method editors (IMEs) [], as well as inaccuracies introduced by automatic speech recognition (ASR) and optical character recognition (OCR) systems." + }, + { + "id": 30, + "label": "Uses", + "confidence": 0.9662095308303833, + "confidence_margin": 0.9460604786872864, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "7ea52ad5c9400b5e914e86cede6e812940b5a27f", + "citing_title": "Chinese Spelling Correction: A Comprehensive Survey of Progress, Challenges, and Opportunities", + "text": "CSCD-NS [] represents the first CSC dataset specifically designed for native speakers, which is originated from authentic texts on Weibo." + }, + { + "id": 31, + "label": "Uses", + "confidence": 0.9893466830253601, + "confidence_margin": 0.9859129786491394, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "117b5eccd7163e7bb4d2f1bf05df0c08f6d79f8a", + "citing_title": "Enhancing Character-Level Understanding in LLMs through Token Internal Structure Learning", + "text": "4.5 Character-Level Precision (CP), Recall (CR), and F1 Score (CF1) At the character level : \u2022 Precision (CP) is the proportion of incorrect characters identified by the model that are actually incorrect." + }, + { + "id": 32, + "label": "Motivation", + "confidence": 0.4396558105945587, + "confidence_margin": 0.1775248944759369, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "117b5eccd7163e7bb4d2f1bf05df0c08f6d79f8a", + "citing_title": "Enhancing Character-Level Understanding in LLMs through Token Internal Structure Learning", + "text": "\u20262024, 2022; Liang et al., 2023), we selected two new Chinese Spelling Correction (CSC) benchmarks, CSCD-NS and LEMON, to address the limitations identified ; Yin and Wan, 2023; Li et al., 2022) in previous datasets like SIGHAN (Wu et al., 2013; Yu and Li, 2014; Tseng et al., 2015)." + }, + { + "id": 33, + "label": "Uses", + "confidence": 0.9566210508346558, + "confidence_margin": 0.9151804447174072, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "619346184ae157741fc1c4ac74447ab2455cd6f3", + "citing_title": "A Simple yet Effective Training-free Prompt-free Approach to Chinese Spelling Correction Based on Large Language Models", + "text": "To identify these outliers, we leveraged tools from previous studies (Wu et al., 2023; by incorporating their structure confusion sets and spelling similarity matrices." + }, + { + "id": 34, + "label": "Uses", + "confidence": 0.8758470416069031, + "confidence_margin": 0.8068125247955322, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "619346184ae157741fc1c4ac74447ab2455cd6f3", + "citing_title": "A Simple yet Effective Training-free Prompt-free Approach to Chinese Spelling Correction Based on Large Language Models", + "text": "\u201a CSCD-NS : A real-world Chinese social media corpus collected and annotated by ." + }, + { + "id": 35, + "label": "CompareOrContrast", + "confidence": 0.8204632997512817, + "confidence_margin": 0.6562803983688354, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "619346184ae157741fc1c4ac74447ab2455cd6f3", + "citing_title": "A Simple yet Effective Training-free Prompt-free Approach to Chinese Spelling Correction Based on Large Language Models", + "text": "Instead, we leverage the findings from Li et al. (2023a), who fine-tuned the Baichuan2 7B and GPT2 models on the ECSpell dataset, and , who fine-tuned the Baichuan2 7B model on the CSCD-NS dataset." + }, + { + "id": 36, + "label": "Uses", + "confidence": 0.9969940185546875, + "confidence_margin": 0.9952777624130249, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "619346184ae157741fc1c4ac74447ab2455cd6f3", + "citing_title": "A Simple yet Effective Training-free Prompt-free Approach to Chinese Spelling Correction Based on Large Language Models", + "text": "We perform experiments across five distinct CSC datasets: Sighans (Wu et al., 2013; Yu et al., 2014; Tseng et al., 2015), CSCD-NS , MCSCSet (Jiang et al., 2022), ECSpell (Lv et al., 2023), and Lemon (Wu et al., 2023), covering a broad spectrum of domains and genres." + }, + { + "id": 37, + "label": "CompareOrContrast", + "confidence": 0.5004019737243652, + "confidence_margin": 0.002603590488433838, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "ec3dbcb9661832d2fec3fb98e3b3cc3c2cf775d0", + "citing_title": "OMGEval: An Open Multilingual Generative Evaluation Benchmark for Large Language Models", + "text": "Traditional evaluations of these models often revolved around restricted question-and-answer formats with definitive answers, typical in NLP tasks like Grammatical Error Correction (Ng et al., 2014; Bryant et al., 2019), Chinese Spelling Correction (Yu et al., 2014; ." + }, + { + "id": 38, + "label": "Uses", + "confidence": 0.9994204044342041, + "confidence_margin": 0.9991111755371094, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "571754f4ed4303b02a1a5e78c6605f483b2ea6e0", + "citing_title": "Eval-GCSC: A New Metric for Evaluating ChatGPT's Performance in Chinese Spelling Correction", + "text": "Metrics We computed Eval-CSC based on Script , and implemented Eval-GCSC following the procedures outlined in Section 4.3." + }, + { + "id": 39, + "label": "Uses", + "confidence": 0.9996479749679565, + "confidence_margin": 0.9995113015174866, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "db2208df9148b0ddc65e21d7a0883ca54edd4e66", + "citing_title": "Evolving Chinese Spelling Correction with Corrector-Verifier Collaboration", + "text": "We generate the synthetic data using IME ." + }, + { + "id": 40, + "label": "Uses", + "confidence": 0.9888285994529724, + "confidence_margin": 0.979454517364502, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "db2208df9148b0ddc65e21d7a0883ca54edd4e66", + "citing_title": "Evolving Chinese Spelling Correction with Corrector-Verifier Collaboration", + "text": "5-14b finetuned combines the human annotated data from and 271K pseudo-data generated by ASR or OCR as the training data, and uses character-level tokenization." + }, + { + "id": 41, + "label": "Background", + "confidence": 0.6949681639671326, + "confidence_margin": 0.5149464011192322, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "db2208df9148b0ddc65e21d7a0883ca54edd4e66", + "citing_title": "Evolving Chinese Spelling Correction with Corrector-Verifier Collaboration", + "text": "However, Table 1 shows that while ACI and synthetic+human show comparable performance across various domains in LEMON, synthetic+human exhibits better performance on ." + }, + { + "id": 42, + "label": "Background", + "confidence": 0.6605829000473022, + "confidence_margin": 0.38725411891937256, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "db2208df9148b0ddc65e21d7a0883ca54edd4e66", + "citing_title": "Evolving Chinese Spelling Correction with Corrector-Verifier Collaboration", + "text": "(2) superior in annotation quality and focus on spelling errors stemming from pinyin input methods." + }, + { + "id": 43, + "label": "Background", + "confidence": 0.7155693173408508, + "confidence_margin": 0.4332122206687927, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "db2208df9148b0ddc65e21d7a0883ca54edd4e66", + "citing_title": "Evolving Chinese Spelling Correction with Corrector-Verifier Collaboration", + "text": "Dataset A line of studies has reported issues with SIGHAN (Tseng et al., 2015), such as annotation error and incoherent style with native speakers (Wu et al., 2023; ; Li et al., 2024)." + }, + { + "id": 44, + "label": "Background", + "confidence": 0.926706075668335, + "confidence_margin": 0.8553003668785095, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "db2208df9148b0ddc65e21d7a0883ca54edd4e66", + "citing_title": "Evolving Chinese Spelling Correction with Corrector-Verifier Collaboration", + "text": "A line of works also propose different data augmentation methods to construct pseudo data to address the scarcity of CSC data (Wang et al., 2018; ; Sheng and Xu, 2024)." + }, + { + "id": 45, + "label": "Uses", + "confidence": 0.9988777041435242, + "confidence_margin": 0.9984709024429321, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "db2208df9148b0ddc65e21d7a0883ca54edd4e66", + "citing_title": "Evolving Chinese Spelling Correction with Corrector-Verifier Collaboration", + "text": "To evaluate the quality of generated training data, we probe the annotation accuracy of different approaches on development set." + }, + { + "id": 46, + "label": "Background", + "confidence": 0.9809489846229553, + "confidence_margin": 0.9651319980621338, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "db2208df9148b0ddc65e21d7a0883ca54edd4e66", + "citing_title": "Evolving Chinese Spelling Correction with Corrector-Verifier Collaboration", + "text": "These models suffer from biased error patterns learned on synthetic data, leading to over-correction issues and inadequate handling of semantic errors (Liu et al., 2025; Wu et al., 2023; ; Liu et al., 2022; Jiang et al., 2024)." + }, + { + "id": 47, + "label": "Background", + "confidence": 0.9776833057403564, + "confidence_margin": 0.9558044672012329, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "db2208df9148b0ddc65e21d7a0883ca54edd4e66", + "citing_title": "Evolving Chinese Spelling Correction with Corrector-Verifier Collaboration", + "text": "A recent line of work develops large language models (LLMs) for CSC (Li et al., 2024; Zhou et al., 2024) while some others continue to elaborate BERT-based models (Wu et al., 2023; ; Liu et al., 2024; Zhu et al., 2022; Sheng and Xu, 2024)." + }, + { + "id": 48, + "label": "Uses", + "confidence": 0.9995379447937012, + "confidence_margin": 0.9993192553520203, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "db2208df9148b0ddc65e21d7a0883ca54edd4e66", + "citing_title": "Evolving Chinese Spelling Correction with Corrector-Verifier Collaboration", + "text": "Following recent works, we use two CSC benchmarks, LEMON (Wu et al., 2023) and CSCD-NS ." + }, + { + "id": 49, + "label": "Uses", + "confidence": 0.9863585829734802, + "confidence_margin": 0.9788398146629333, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "db2208df9148b0ddc65e21d7a0883ca54edd4e66", + "citing_title": "Evolving Chinese Spelling Correction with Corrector-Verifier Collaboration", + "text": "We first train the model on IME-based synthetic data and then train it on human annotated data." + }, + { + "id": 50, + "label": "Uses", + "confidence": 0.9994650483131409, + "confidence_margin": 0.9992184638977051, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "db2208df9148b0ddc65e21d7a0883ca54edd4e66", + "citing_title": "Evolving Chinese Spelling Correction with Corrector-Verifier Collaboration", + "text": "We compare the annotation results with the gold labels from ." + }, + { + "id": 51, + "label": "Uses", + "confidence": 0.9982732534408569, + "confidence_margin": 0.9969974160194397, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "db2208df9148b0ddc65e21d7a0883ca54edd4e66", + "citing_title": "Evolving Chinese Spelling Correction with Corrector-Verifier Collaboration", + "text": "The data we use is the training set of , which is in high quality." + }, + { + "id": 52, + "label": "Uses", + "confidence": 0.4416528642177582, + "confidence_margin": 0.12170693278312683, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "db2208df9148b0ddc65e21d7a0883ca54edd4e66", + "citing_title": "Evolving Chinese Spelling Correction with Corrector-Verifier Collaboration", + "text": "This performance gap can be attributed to the domain alignment between the human-29013 annotated training data and test data, suggesting the potential benefit of incorporating human-annotated and domain-specific data in certain scenarios." + }, + { + "id": 53, + "label": "Uses", + "confidence": 0.4932326078414917, + "confidence_margin": 0.18286454677581787, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "db2208df9148b0ddc65e21d7a0883ca54edd4e66", + "citing_title": "Evolving Chinese Spelling Correction with Corrector-Verifier Collaboration", + "text": "\u2022 IME-based synthetic + Human annotated The two-stage training of first using synthetic and then using human-annotated data is the widely-used and the most useful method." + }, + { + "id": 54, + "label": "Background", + "confidence": 0.9861831665039062, + "confidence_margin": 0.9762710332870483, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "175d21f15ae0eebf6a5e3ca017b34b3b54e771a4", + "citing_title": "RACQC: Advanced Retrieval-Augmented Generation for Chinese Query Correction", + "text": "Our investigation reveals that general CSC datasets ; Wu et al., 2023) in mainstream research often overlook entity-level error correction critical for open-domain search, focusing primarily on common entities with limited coverage of long-tail ones." + }, + { + "id": 55, + "label": "Background", + "confidence": 0.9434929490089417, + "confidence_margin": 0.9244600534439087, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "175d21f15ae0eebf6a5e3ca017b34b3b54e771a4", + "citing_title": "RACQC: Advanced Retrieval-Augmented Generation for Chinese Query Correction", + "text": "To improve their error correction abilities, strategies such as data synthesis(Wang et al., 2024b; , incorporating error detection modules(Zhang et al., 2020a), and specific character masking strategies(Liu et al., 2010) have been used." + } + ], + "citing_paper_labels": [ + { + "citing_paper_id": "4b66b25de8c6c496f47ee58cf2677d0b377aa778", + "citing_title": "Towards More Reliable Chinese Spelling Correction: Fine-Grained Confidence Estimation Against Suboptimal Corrections", + "label": "Extends", + "evidence_context_ids": [ + 3 + ] + }, + { + "citing_paper_id": "7d64dd752bb29bb9aff8a836b4b1f5f3ccafe807", + "citing_title": "ChineseErrorCorrector3-4B: State-of-the-Art Chinese Spelling and Grammar Corrector", + "label": "Uses", + "evidence_context_ids": [ + 4 + ] + }, + { + "citing_paper_id": "0e2e19f1fae35d97b3178f9349caf95da540652d", + "citing_title": "Incorporating Confused Phraseological Knowledge Based on Pinyin Input Method for Chinese Spelling Correction", + "label": "Uses", + "evidence_context_ids": [ + 5, + 6 + ] + }, + { + "citing_paper_id": "1d4ba6d664b28e5e1854bf662f30ed8f2fbcf2f7", + "citing_title": "Vision Language Models Are Not (Yet) Spelling Correctors", + "label": "Background", + "evidence_context_ids": [] + }, + { + "citing_paper_id": "87a2958bd75a72e61b5b2549a63a222d7ca2a1ec", + "citing_title": "Mixture of Small and Large Models for Chinese Spelling Check", + "label": "Uses", + "evidence_context_ids": [ + 11, + 12 + ] + }, + { + "citing_paper_id": "ce28a63f81ad7773235322506368743982fd2bc3", + "citing_title": "SmartBench: Is Your LLM Truly a Good Chinese Smartphone Assistant?", + "label": "Uses", + "evidence_context_ids": [ + 13 + ] + }, + { + "citing_paper_id": "9a6479bdd7605029d3972dea4c64a2e914f2755a", + "citing_title": "A Training-free LLM-based Approach to General Chinese Character Error Correction", + "label": "Uses", + "evidence_context_ids": [ + 14, + 15, + 16, + 18, + 20, + 22, + 25 + ] + }, + { + "citing_paper_id": "7ea52ad5c9400b5e914e86cede6e812940b5a27f", + "citing_title": "Chinese Spelling Correction: A Comprehensive Survey of Progress, Challenges, and Opportunities", + "label": "Uses", + "evidence_context_ids": [ + 30 + ] + }, + { + "citing_paper_id": "117b5eccd7163e7bb4d2f1bf05df0c08f6d79f8a", + "citing_title": "Enhancing Character-Level Understanding in LLMs through Token Internal Structure Learning", + "label": "Uses", + "evidence_context_ids": [ + 31 + ] + }, + { + "citing_paper_id": "619346184ae157741fc1c4ac74447ab2455cd6f3", + "citing_title": "A Simple yet Effective Training-free Prompt-free Approach to Chinese Spelling Correction Based on Large Language Models", + "label": "Uses", + "evidence_context_ids": [ + 33, + 34, + 36 + ] + }, + { + "citing_paper_id": "ec3dbcb9661832d2fec3fb98e3b3cc3c2cf775d0", + "citing_title": "OMGEval: An Open Multilingual Generative Evaluation Benchmark for Large Language Models", + "label": "CompareOrContrast", + "evidence_context_ids": [ + 37 + ] + }, + { + "citing_paper_id": "571754f4ed4303b02a1a5e78c6605f483b2ea6e0", + "citing_title": "Eval-GCSC: A New Metric for Evaluating ChatGPT's Performance in Chinese Spelling Correction", + "label": "Uses", + "evidence_context_ids": [ + 38 + ] + }, + { + "citing_paper_id": "db2208df9148b0ddc65e21d7a0883ca54edd4e66", + "citing_title": "Evolving Chinese Spelling Correction with Corrector-Verifier Collaboration", + "label": "Uses", + "evidence_context_ids": [ + 39, + 40, + 45, + 48, + 49, + 50, + 51, + 52, + 53 + ] + }, + { + "citing_paper_id": "175d21f15ae0eebf6a5e3ca017b34b3b54e771a4", + "citing_title": "RACQC: Advanced Retrieval-Augmented Generation for Chinese Query Correction", + "label": "Background", + "evidence_context_ids": [] + } + ], + "final_label": "Extends" +} \ No newline at end of file diff --git a/replay_traces/2211.08788/processed_papers/2211.08788/usage_contexts.json b/replay_traces/2211.08788/processed_papers/2211.08788/usage_contexts.json new file mode 100644 index 0000000000000000000000000000000000000000..62fe606dd43f7debd08a6ad8ceaab34df7d3b1a3 --- /dev/null +++ b/replay_traces/2211.08788/processed_papers/2211.08788/usage_contexts.json @@ -0,0 +1,867 @@ +{ + "paper_id": "2211.08788", + "total_citations": 21, + "num_contexts": 55, + "num_citing_with_context": 14, + "num_citing_without_context": 7, + "num_influential_citations": 7, + "num_influential_with_context": 7, + "influential_contexts": [ + { + "citing_paper_id": "0e2e19f1fae35d97b3178f9349caf95da540652d", + "title": "Incorporating Confused Phraseological Knowledge Based on Pinyin Input Method for Chinese Spelling Correction", + "external_ids": { + "DBLP": "journals/tbd/ZhaoWQ25", + "DOI": "10.1109/TBDATA.2025.3552344", + "CorpusId": 277404639 + }, + "contexts": [ + { + "text": "The transformation method follows the pseudo-data construction method proposed by to ensure that the converted data closely resembles the actual scenario.", + "text_raw": "The transformation method follows the pseudo-data construction method proposed by Hu et al. [17] to ensure that the converted data closely resembles the actual scenario.", + "intents": [] + }, + { + "text": "To better simulate real input scenarios, this paper uses the Chinese pinyin IME error correction dataset CSCD-IME as the basic dataset for the experiments, including 3,0000 training data, 5000 validation data, and 5000 test data.", + "text_raw": "To better simulate real input scenarios, this paper uses the Chinese pinyin IME error correction dataset CSCD-IME [17] as the basic dataset for the experiments, including 3,0000 training data, 5000 validation data, and 5000 test data.", + "intents": [] + }, + { + "text": "foradditionaltrain-ing.", + "text_raw": "[17]foradditionaltrain-ing.", + "intents": [] + } + ] + }, + { + "citing_paper_id": "87a2958bd75a72e61b5b2549a63a222d7ca2a1ec", + "title": "Mixture of Small and Large Models for Chinese Spelling Check", + "external_ids": { + "ArXiv": "2506.06887", + "DBLP": "journals/corr/abs-2506-06887", + "DOI": "10.48550/arXiv.2506.06887", + "CorpusId": 279250578 + }, + "contexts": [ + { + "text": ".", + "text_raw": "CSCD-NS.", + "intents": [] + }, + { + "text": "CSCD-NS contains 40K annotated samples sourced from real posts on Sina Weibo, effectively reflecting the real-world error correction performance of CSC models.", + "text_raw": "CSCD-NS (Hu et al., 2024) contains 40K annotated samples sourced from real posts on Sina Weibo, effectively reflecting the real-world error correction performance of CSC models.", + "intents": [] + }, + { + "text": "Due to the lack of domain diversity in the SIGHAN datasets, we further conduct experiments on more diverse datasets, including LEMON (Wu et al., 2023), EC-Spell (Lv et al., 2023), CSCD-NS , and MCSCSet (Jiang et al., 2022), all of which were written by native speakers.", + "text_raw": "Due to the lack of domain diversity in the SIGHAN datasets, we further conduct experiments on more diverse datasets, including LEMON (Wu et al., 2023), EC-Spell (Lv et al., 2023), CSCD-NS (Hu et al., 2024), and MCSCSet (Jiang et al., 2022), all of which were written by native speakers.", + "intents": [] + }, + { + "text": "Table 2 presents experimental results across five benchmark datasets: rSIGHANs, , MC-SCSet, ECSpell, and LEMON.", + "text_raw": "Table 2 presents experimental results across five benchmark datasets: rSIGHANs, CSCD-NS, MC-SCSet, ECSpell, and LEMON.", + "intents": [] + } + ] + }, + { + "citing_paper_id": "9a6479bdd7605029d3972dea4c64a2e914f2755a", + "title": "A Training-free LLM-based Approach to General Chinese Character Error Correction", + "external_ids": { + "DBLP": "journals/corr/abs-2502-15266", + "ArXiv": "2502.15266", + "DOI": "10.48550/arXiv.2502.15266", + "CorpusId": 276558301 + }, + "contexts": [ + { + "text": "Specifically, the dataset is publicly available under the MIT license, while the CCTC dataset is publicly available under the Apache 2.0 license.", + "text_raw": "Specifically, the CSCD-NS dataset is publicly available under the MIT license, while the CCTC dataset is publicly available under the Apache 2.0 license.", + "intents": [] + }, + { + "text": "Figure 11 shows the performance of our method with varying beam sizes on the and C2EC datasets.", + "text_raw": "Figure 11 shows the performance of our method with varying beam sizes on the CSCD-NS and C2EC datasets.", + "intents": [] + }, + { + "text": "For conventional CSC, following Li et al. (2024), we use two representative datasets: CSCD-NS and Lemon (Wu et al., 2023).", + "text_raw": "For conventional CSC, following Li et al. (2024), we use two representative datasets: CSCD-NS (Hu et al., 2024) and Lemon (Wu et al., 2023).", + "intents": [] + }, + { + "text": " focuses on general domain performance, while Lemon evaluates zero-shot cross-domain capabilities.", + "text_raw": "CSCD-NS focuses on general domain performance, while Lemon evaluates zero-shot cross-domain capabilities.", + "intents": [] + }, + { + "text": "We randomly sampled 50 sentences of varying lengths from the development set of and C2EC to evaluate the running time of our method on Qwen2.", + "text_raw": "We randomly sampled 50 sentences of varying lengths from the development set of CSCD-NS and C2EC to evaluate the running time of our method on Qwen2.", + "intents": [] + }, + { + "text": "Compared to SFT models from Li et al. (2024), which are trained on the training set of the dataset, our method shows better performance on the out-of-domain Lemon dataset.", + "text_raw": "Compared to SFT models from Li et al. (2024), which are trained on the training set of the CSCD-NS dataset, our method shows better performance on the out-of-domain Lemon dataset.", + "intents": [] + }, + { + "text": "When compared to the TfPf method, the SFT methods show superior performance on the in-domain dataset .", + "text_raw": "When compared to the TfPf method, the SFT methods show superior performance on the in-domain dataset CSCD-NS.", + "intents": [] + }, + { + "text": "This means that while is an in-domain evaluation, the Lemon dataset serves as a cross-domain dataset for evaluating the generalization capabilities.", + "text_raw": "This means that while CSCD-NS is an in-domain evaluation, the Lemon dataset serves as a cross-domain dataset for evaluating the generalization capabilities.", + "intents": [] + }, + { + "text": "Training Details of SFT Baselines The SFT models were trained on a combined dataset consisting of 271k pseudo sentence pairs (Wang et al., 2018) and the training data.", + "text_raw": "Training Details of SFT Baselines The SFT models were trained on a combined dataset consisting of 271k pseudo sentence pairs (Wang et al., 2018) and the CSCD-NS training data.", + "intents": [] + }, + { + "text": "This suggests that the SFT methods might overfit to the in-domain dataset , limiting their generalization to the out-of-domain dataset Lemon.", + "text_raw": "This suggests that the SFT methods might overfit to the in-domain dataset CSCD-NS, limiting their generalization to the out-of-domain dataset Lemon.", + "intents": [] + }, + { + "text": "Chinese characters often have multiple visually or phonetically similar variants, making it easy to select an incorrect one when using input methods .", + "text_raw": "Chinese characters often have multiple visually or phonetically similar variants, making it easy to select an incorrect one when using input methods (Hu et al., 2024).", + "intents": [] + }, + { + "text": "Removing it leads to performance drops of 11.10 and 5.45 points on the and C2EC datasets, respectively.", + "text_raw": "Removing it leads to performance drops of 11.10 and 5.45 points on the CSCD-NS and C2EC datasets, respectively.", + "intents": [] + }, + { + "text": "While the Levenshtein distance-based model performs better on the C2EC dataset, it degrades the performance on the dataset, the conventional CSC dataset.", + "text_raw": "While the Levenshtein distance-based model performs better on the C2EC dataset, it degrades the performance on the CSCD-NS dataset, the conventional CSC dataset.", + "intents": [] + }, + { + "text": "Similarly, created a new dataset with real errors found on social media.", + "text_raw": "Similarly, Hu et al. (2024) created a new dataset with real errors found on social media.", + "intents": [] + }, + { + "text": "Additionally, without any training, our method achieves performance on par with the SFT methods on the in-domain dataset , with scores of 71.53 versus 73.80 on the 14B model.", + "text_raw": "Additionally, without any training, our method achieves performance on par with the SFT methods on the in-domain dataset CSCD-NS, with scores of 71.53 versus 73.80 on the 14B model.", + "intents": [] + } + ] + }, + { + "citing_paper_id": "117b5eccd7163e7bb4d2f1bf05df0c08f6d79f8a", + "title": "Enhancing Character-Level Understanding in LLMs through Token Internal Structure Learning", + "external_ids": { + "DBLP": "journals/corr/abs-2411-17679", + "ArXiv": "2411.17679", + "DOI": "10.48550/arXiv.2411.17679", + "CorpusId": 274280941 + }, + "contexts": [ + { + "text": "4.5 Character-Level Precision (CP), Recall (CR), and F1 Score (CF1) At the character level : \u2022 Precision (CP) is the proportion of incorrect characters identified by the model that are actually incorrect.", + "text_raw": "4.5 Character-Level Precision (CP), Recall (CR), and F1 Score (CF1) At the character level (Hu et al., 2024): \u2022 Precision (CP) is the proportion of incorrect characters identified by the model that are actually incorrect.", + "intents": [] + }, + { + "text": "\u20262024, 2022; Liang et al., 2023), we selected two new Chinese Spelling Correction (CSC) benchmarks, CSCD-NS and LEMON, to address the limitations identified ; Yin and Wan, 2023; Li et al., 2022) in previous datasets like SIGHAN (Wu et al., 2013; Yu and Li, 2014; Tseng et al., 2015).", + "text_raw": "\u20262024, 2022; Liang et al., 2023), we selected two new Chinese Spelling Correction (CSC) benchmarks, CSCD-NS and LEMON, to address the limitations identified (Hu et al., 2024; Yin and Wan, 2023; Li et al., 2022) in previous datasets like SIGHAN (Wu et al., 2013; Yu and Li, 2014; Tseng et al., 2015).", + "intents": [] + } + ] + }, + { + "citing_paper_id": "619346184ae157741fc1c4ac74447ab2455cd6f3", + "title": "A Simple yet Effective Training-free Prompt-free Approach to Chinese Spelling Correction Based on Large Language Models", + "external_ids": { + "DBLP": "conf/emnlp/0001LZ0L00024", + "ACL": "2024.emnlp-main.966", + "ArXiv": "2410.04027", + "DOI": "10.48550/arXiv.2410.04027", + "CorpusId": 273186532 + }, + "contexts": [ + { + "text": "To identify these outliers, we leveraged tools from previous studies (Wu et al., 2023; by incorporating their structure confusion sets and spelling similarity matrices.", + "text_raw": "To identify these outliers, we leveraged tools from previous studies (Wu et al., 2023; Hu et al., 2024) by incorporating their structure confusion sets and spelling similarity matrices.", + "intents": [ + "methodology" + ] + }, + { + "text": "\u201a CSCD-NS : A real-world Chinese social media corpus collected and annotated by .", + "text_raw": "\u201a CSCD-NS : A real-world Chinese social media corpus collected and annotated by Hu et al. (2024).", + "intents": [ + "background" + ] + }, + { + "text": "Instead, we leverage the findings from Li et al. (2023a), who fine-tuned the Baichuan2 7B and GPT2 models on the ECSpell dataset, and , who fine-tuned the Baichuan2 7B model on the CSCD-NS dataset.", + "text_raw": "Instead, we leverage the findings from Li et al. (2023a), who fine-tuned the Baichuan2 7B and GPT2 models on the ECSpell dataset, and Hu et al. (2024), who fine-tuned the Baichuan2 7B model on the CSCD-NS dataset.", + "intents": [ + "methodology" + ] + }, + { + "text": "We perform experiments across five distinct CSC datasets: Sighans (Wu et al., 2013; Yu et al., 2014; Tseng et al., 2015), CSCD-NS , MCSCSet (Jiang et al., 2022), ECSpell (Lv et al., 2023), and Lemon (Wu et al., 2023), covering a broad spectrum of domains and genres.", + "text_raw": "We perform experiments across five distinct CSC datasets: Sighans (Wu et al., 2013; Yu et al., 2014; Tseng et al., 2015), CSCD-NS (Hu et al., 2024), MCSCSet (Jiang et al., 2022), ECSpell (Lv et al., 2023), and Lemon (Wu et al., 2023), covering a broad spectrum of domains and genres.", + "intents": [ + "methodology" + ] + } + ] + }, + { + "citing_paper_id": "571754f4ed4303b02a1a5e78c6605f483b2ea6e0", + "title": "Eval-GCSC: A New Metric for Evaluating ChatGPT's Performance in Chinese Spelling Correction", + "external_ids": { + "ArXiv": "2311.08219", + "DBLP": "journals/corr/abs-2311-08219", + "DOI": "10.48550/arXiv.2311.08219", + "CorpusId": 265158218 + }, + "contexts": [ + { + "text": "Metrics We computed Eval-CSC based on Script , and implemented Eval-GCSC following the procedures outlined in Section 4.3.", + "text_raw": "Metrics We computed Eval-CSC based on Script (Hu et al., 2022), and implemented Eval-GCSC following the procedures outlined in Section 4.3.", + "intents": [ + "methodology" + ] + } + ] + }, + { + "citing_paper_id": "db2208df9148b0ddc65e21d7a0883ca54edd4e66", + "title": "Evolving Chinese Spelling Correction with Corrector-Verifier Collaboration", + "external_ids": { + "DBLP": "conf/emnlp/LiuWZ25", + "DOI": "10.18653/v1/2025.emnlp-main.1477", + "CorpusId": 282813715 + }, + "contexts": [ + { + "text": "We generate the synthetic data using IME .", + "text_raw": "We generate the synthetic data using IME (Hu et al., 2024).", + "intents": [] + }, + { + "text": "5-14b finetuned combines the human annotated data from and 271K pseudo-data generated by ASR or OCR as the training data, and uses character-level tokenization.", + "text_raw": "5-14b finetuned combines the human annotated data from CSCD-NS and 271K pseudo-data generated by ASR or OCR as the training data, and uses character-level tokenization.", + "intents": [] + }, + { + "text": "However, Table 1 shows that while ACI and synthetic+human show comparable performance across various domains in LEMON, synthetic+human exhibits better performance on .", + "text_raw": "However, Table 1 shows that while ACI and synthetic+human show comparable performance across various domains in LEMON, synthetic+human exhibits better performance on CSCD-NS.", + "intents": [] + }, + { + "text": "(2) superior in annotation quality and focus on spelling errors stemming from pinyin input methods.", + "text_raw": "(2) CSCD-NS superior in annotation quality and focus on spelling errors stemming from pinyin input methods.", + "intents": [] + }, + { + "text": "Dataset A line of studies has reported issues with SIGHAN (Tseng et al., 2015), such as annotation error and incoherent style with native speakers (Wu et al., 2023; ; Li et al., 2024).", + "text_raw": "Dataset A line of studies has reported issues with SIGHAN (Tseng et al., 2015), such as annotation error and incoherent style with native speakers (Wu et al., 2023; Hu et al., 2024; Li et al., 2024).", + "intents": [] + }, + { + "text": "A line of works also propose different data augmentation methods to construct pseudo data to address the scarcity of CSC data (Wang et al., 2018; ; Sheng and Xu, 2024).", + "text_raw": "A line of works also propose different data augmentation methods to construct pseudo data to address the scarcity of CSC data (Wang et al., 2018; Hu et al., 2024; Sheng and Xu, 2024).", + "intents": [] + }, + { + "text": "To evaluate the quality of generated training data, we probe the annotation accuracy of different approaches on development set.", + "text_raw": "To evaluate the quality of generated training data, we probe the annotation accuracy of different approaches on CSCD-NS development set.", + "intents": [] + }, + { + "text": "These models suffer from biased error patterns learned on synthetic data, leading to over-correction issues and inadequate handling of semantic errors (Liu et al., 2025; Wu et al., 2023; ; Liu et al., 2022; Jiang et al., 2024).", + "text_raw": "These models suffer from biased error patterns learned on synthetic data, leading to over-correction issues and inadequate handling of semantic errors (Liu et al., 2025; Wu et al., 2023; Hu et al., 2024; Liu et al., 2022; Jiang et al., 2024).", + "intents": [] + }, + { + "text": "A recent line of work develops large language models (LLMs) for CSC (Li et al., 2024; Zhou et al., 2024) while some others continue to elaborate BERT-based models (Wu et al., 2023; ; Liu et al., 2024; Zhu et al., 2022; Sheng and Xu, 2024).", + "text_raw": "A recent line of work develops large language models (LLMs) for CSC (Li et al., 2024; Zhou et al., 2024) while some others continue to elaborate BERT-based models (Wu et al., 2023; Hu et al., 2024; Liu et al., 2024; Zhu et al., 2022; Sheng and Xu, 2024).", + "intents": [] + }, + { + "text": "Following recent works, we use two CSC benchmarks, LEMON (Wu et al., 2023) and CSCD-NS .", + "text_raw": "Following recent works, we use two CSC benchmarks, LEMON (Wu et al., 2023) and CSCD-NS (Hu et al., 2024).", + "intents": [] + }, + { + "text": "We first train the model on IME-based synthetic data and then train it on human annotated data.", + "text_raw": "We first train the model on IME-based synthetic data and then train it on human annotated data.", + "intents": [] + }, + { + "text": "We compare the annotation results with the gold labels from .", + "text_raw": "We compare the annotation results with the gold labels from CSCD-NS.", + "intents": [] + }, + { + "text": "The data we use is the training set of , which is in high quality.", + "text_raw": "The data we use is the training set of CSCD-NS, which is in high quality.", + "intents": [] + }, + { + "text": "This performance gap can be attributed to the domain alignment between the human-29013 annotated training data and test data, suggesting the potential benefit of incorporating human-annotated and domain-specific data in certain scenarios.", + "text_raw": "This performance gap can be attributed to the domain alignment between the human-29013 annotated training data and CSCD-NS test data, suggesting the potential benefit of incorporating human-annotated and domain-specific data in certain scenarios.", + "intents": [] + }, + { + "text": "\u2022 IME-based synthetic + Human annotated The two-stage training of first using synthetic and then using human-annotated data is the widely-used and the most useful method.", + "text_raw": "\u2022 IME-based synthetic + Human annotated The two-stage training of first using synthetic and then using human-annotated data is the widely-used and the most useful method.", + "intents": [] + } + ] + } + ], + "citing_papers": [ + { + "citing_paper_id": "af6f7ea0fb2b29e585792514628dd2752808cb07", + "title": "Ccot: advancing long-text error correction in post-ASR transcripts via fine-grained contextual chain-of-thought", + "external_ids": { + "DBLP": "journals/mlc/JiaLKJLGL26", + "DOI": "10.1007/s13042-025-02898-1", + "CorpusId": 286759125 + }, + "is_influential": false, + "contexts": [] + }, + { + "citing_paper_id": "81fe49f86aa7e2cd8f8524b5948484dcf55a8db7", + "title": "CEC-Zero: Zero-Supervision Character Error Correction with Self-Generated Rewards", + "external_ids": { + "ArXiv": "2512.23971", + "DBLP": "journals/corr/abs-2512-23971", + "DOI": "10.48550/arXiv.2512.23971", + "CorpusId": 284350498 + }, + "is_influential": false, + "contexts": [] + }, + { + "citing_paper_id": "4b66b25de8c6c496f47ee58cf2677d0b377aa778", + "title": "Towards More Reliable Chinese Spelling Correction: Fine-Grained Confidence Estimation Against Suboptimal Corrections", + "external_ids": { + "DBLP": "conf/bigdataconf/TongPZLQJL25", + "DOI": "10.1109/BigData66926.2025.11401539", + "CorpusId": 286330094 + }, + "is_influential": false, + "contexts": [ + { + "text": "We also include a native-speaker-oriented dataset collected from social media, which contains a manually annotated subset and a large-scale pseudo-labeled corpus.", + "text_raw": "We also include CSCD-NS [15], a native-speaker-oriented dataset collected from social media, which contains a manually annotated subset and a large-scale pseudo-labeled corpus.", + "intents": [] + }, + { + "text": "Chinese spelling errors mainly arise from incorrect pinyin input or the selection of similar-sounding characters .", + "text_raw": "Chinese spelling errors mainly arise from incorrect pinyin input or the selection of similar-sounding characters [15].", + "intents": [] + }, + { + "text": "2) Metrics: We follow previous works , [38] and report strict character-level precision, recall, and F1 scores for both detection and correction.", + "text_raw": "2) Metrics: We follow previous works [15], [38] and report strict character-level precision, recall, and F1 scores for both detection and correction.", + "intents": [] + } + ] + }, + { + "citing_paper_id": "e8e1b54c7519de97f75d610b15590134bdfea9b4", + "title": "Enhancing Character-Level Understanding in LLMs through Curriculum Reinforcement Learning on Token-Internal Structure", + "external_ids": { + "DOI": "10.1145/3795926.3796025", + "CorpusId": 287608497 + }, + "is_influential": false, + "contexts": [] + }, + { + "citing_paper_id": "7d64dd752bb29bb9aff8a836b4b1f5f3ccafe807", + "title": "ChineseErrorCorrector3-4B: State-of-the-Art Chinese Spelling and Grammar Corrector", + "external_ids": { + "ArXiv": "2511.17562", + "DBLP": "journals/corr/abs-2511-17562", + "DOI": "10.48550/arXiv.2511.17562", + "CorpusId": 283244324 + }, + "is_influential": false, + "contexts": [ + { + "text": "We first aggregate data for the two task families: \u2022 CSC Datasets ( D csc ): Aggregated from W271K, Medical[11], Lemon[12], ECSpell[19], and CSCD[14] (totaling approx.", + "text_raw": "We first aggregate data for the two task families: \u2022 CSC Datasets ( D csc ): Aggregated from W271K, Medical[11], Lemon[12], ECSpell[19], and CSCD[14] (totaling approx.", + "intents": [] + } + ] + }, + { + "citing_paper_id": "0e2e19f1fae35d97b3178f9349caf95da540652d", + "title": "Incorporating Confused Phraseological Knowledge Based on Pinyin Input Method for Chinese Spelling Correction", + "external_ids": { + "DBLP": "journals/tbd/ZhaoWQ25", + "DOI": "10.1109/TBDATA.2025.3552344", + "CorpusId": 277404639 + }, + "is_influential": true, + "contexts": [ + { + "text": "The transformation method follows the pseudo-data construction method proposed by to ensure that the converted data closely resembles the actual scenario.", + "text_raw": "The transformation method follows the pseudo-data construction method proposed by Hu et al. [17] to ensure that the converted data closely resembles the actual scenario.", + "intents": [] + }, + { + "text": "To better simulate real input scenarios, this paper uses the Chinese pinyin IME error correction dataset CSCD-IME as the basic dataset for the experiments, including 3,0000 training data, 5000 validation data, and 5000 test data.", + "text_raw": "To better simulate real input scenarios, this paper uses the Chinese pinyin IME error correction dataset CSCD-IME [17] as the basic dataset for the experiments, including 3,0000 training data, 5000 validation data, and 5000 test data.", + "intents": [] + }, + { + "text": "foradditionaltrain-ing.", + "text_raw": "[17]foradditionaltrain-ing.", + "intents": [] + } + ] + }, + { + "citing_paper_id": "1d4ba6d664b28e5e1854bf662f30ed8f2fbcf2f7", + "title": "Vision Language Models Are Not (Yet) Spelling Correctors", + "external_ids": { + "DBLP": "journals/corr/abs-2509-17418", + "ArXiv": "2509.17418", + "DOI": "10.48550/arXiv.2509.17418", + "CorpusId": 281421165 + }, + "is_influential": false, + "contexts": [ + { + "text": "Existing datasets are primarily text-based, including those collected from language learners [3, 26], native speakers [6], and specific domains [8, 12, 17].", + "text_raw": "Existing datasets are primarily text-based, including those collected from language learners [3, 26], native speakers [6], and specific domains [8, 12, 17].", + "intents": [] + } + ] + }, + { + "citing_paper_id": "706ef967ee1fbb306cac250a8b686cfef177bc87", + "title": "Non-Intrusive Automatic Speech Recognition Refinement: A Survey", + "external_ids": { + "ArXiv": "2508.07285", + "CorpusId": 280565932 + }, + "is_influential": false, + "contexts": [] + }, + { + "citing_paper_id": "87a2958bd75a72e61b5b2549a63a222d7ca2a1ec", + "title": "Mixture of Small and Large Models for Chinese Spelling Check", + "external_ids": { + "ArXiv": "2506.06887", + "DBLP": "journals/corr/abs-2506-06887", + "DOI": "10.48550/arXiv.2506.06887", + "CorpusId": 279250578 + }, + "is_influential": true, + "contexts": [ + { + "text": ".", + "text_raw": "CSCD-NS.", + "intents": [] + }, + { + "text": "CSCD-NS contains 40K annotated samples sourced from real posts on Sina Weibo, effectively reflecting the real-world error correction performance of CSC models.", + "text_raw": "CSCD-NS (Hu et al., 2024) contains 40K annotated samples sourced from real posts on Sina Weibo, effectively reflecting the real-world error correction performance of CSC models.", + "intents": [] + }, + { + "text": "Due to the lack of domain diversity in the SIGHAN datasets, we further conduct experiments on more diverse datasets, including LEMON (Wu et al., 2023), EC-Spell (Lv et al., 2023), CSCD-NS , and MCSCSet (Jiang et al., 2022), all of which were written by native speakers.", + "text_raw": "Due to the lack of domain diversity in the SIGHAN datasets, we further conduct experiments on more diverse datasets, including LEMON (Wu et al., 2023), EC-Spell (Lv et al., 2023), CSCD-NS (Hu et al., 2024), and MCSCSet (Jiang et al., 2022), all of which were written by native speakers.", + "intents": [] + }, + { + "text": "Table 2 presents experimental results across five benchmark datasets: rSIGHANs, , MC-SCSet, ECSpell, and LEMON.", + "text_raw": "Table 2 presents experimental results across five benchmark datasets: rSIGHANs, CSCD-NS, MC-SCSet, ECSpell, and LEMON.", + "intents": [] + } + ] + }, + { + "citing_paper_id": "9e5799d99654659bfc01a2316e4a76a09297b772", + "title": "Research on Railway Official Document Text Correction Algorithm Based on GhatGLM Model and Domain Knowledge Integration", + "external_ids": { + "DOI": "10.1109/DIPCA65051.2025.11042667", + "CorpusId": 279597819 + }, + "is_influential": false, + "contexts": [] + }, + { + "citing_paper_id": "ce28a63f81ad7773235322506368743982fd2bc3", + "title": "SmartBench: Is Your LLM Truly a Good Chinese Smartphone Assistant?", + "external_ids": { + "ArXiv": "2503.06029", + "DBLP": "journals/corr/abs-2503-06029", + "DOI": "10.48550/arXiv.2503.06029", + "CorpusId": 276903491 + }, + "is_influential": false, + "contexts": [ + { + "text": "3) To offer deeper insights into real-world edge-side deployment, we report the prefilling speed, output token generation speed, and power consumption on the iQOO 12 smartphone using the Qual-comm QNN SDK, as shown in Tab.", + "text_raw": "3) To offer deeper insights into real-world edge-side deployment, we report the prefilling speed, output token generation speed, and power consumption on the iQOO 12 smartphone using the Qual-comm QNN SDK, as shown in Tab.", + "intents": [] + } + ] + }, + { + "citing_paper_id": "9a6479bdd7605029d3972dea4c64a2e914f2755a", + "title": "A Training-free LLM-based Approach to General Chinese Character Error Correction", + "external_ids": { + "DBLP": "journals/corr/abs-2502-15266", + "ArXiv": "2502.15266", + "DOI": "10.48550/arXiv.2502.15266", + "CorpusId": 276558301 + }, + "is_influential": true, + "contexts": [ + { + "text": "Specifically, the dataset is publicly available under the MIT license, while the CCTC dataset is publicly available under the Apache 2.0 license.", + "text_raw": "Specifically, the CSCD-NS dataset is publicly available under the MIT license, while the CCTC dataset is publicly available under the Apache 2.0 license.", + "intents": [] + }, + { + "text": "Figure 11 shows the performance of our method with varying beam sizes on the and C2EC datasets.", + "text_raw": "Figure 11 shows the performance of our method with varying beam sizes on the CSCD-NS and C2EC datasets.", + "intents": [] + }, + { + "text": "For conventional CSC, following Li et al. (2024), we use two representative datasets: CSCD-NS and Lemon (Wu et al., 2023).", + "text_raw": "For conventional CSC, following Li et al. (2024), we use two representative datasets: CSCD-NS (Hu et al., 2024) and Lemon (Wu et al., 2023).", + "intents": [] + }, + { + "text": " focuses on general domain performance, while Lemon evaluates zero-shot cross-domain capabilities.", + "text_raw": "CSCD-NS focuses on general domain performance, while Lemon evaluates zero-shot cross-domain capabilities.", + "intents": [] + }, + { + "text": "We randomly sampled 50 sentences of varying lengths from the development set of and C2EC to evaluate the running time of our method on Qwen2.", + "text_raw": "We randomly sampled 50 sentences of varying lengths from the development set of CSCD-NS and C2EC to evaluate the running time of our method on Qwen2.", + "intents": [] + }, + { + "text": "Compared to SFT models from Li et al. (2024), which are trained on the training set of the dataset, our method shows better performance on the out-of-domain Lemon dataset.", + "text_raw": "Compared to SFT models from Li et al. (2024), which are trained on the training set of the CSCD-NS dataset, our method shows better performance on the out-of-domain Lemon dataset.", + "intents": [] + }, + { + "text": "When compared to the TfPf method, the SFT methods show superior performance on the in-domain dataset .", + "text_raw": "When compared to the TfPf method, the SFT methods show superior performance on the in-domain dataset CSCD-NS.", + "intents": [] + }, + { + "text": "This means that while is an in-domain evaluation, the Lemon dataset serves as a cross-domain dataset for evaluating the generalization capabilities.", + "text_raw": "This means that while CSCD-NS is an in-domain evaluation, the Lemon dataset serves as a cross-domain dataset for evaluating the generalization capabilities.", + "intents": [] + }, + { + "text": "Training Details of SFT Baselines The SFT models were trained on a combined dataset consisting of 271k pseudo sentence pairs (Wang et al., 2018) and the training data.", + "text_raw": "Training Details of SFT Baselines The SFT models were trained on a combined dataset consisting of 271k pseudo sentence pairs (Wang et al., 2018) and the CSCD-NS training data.", + "intents": [] + }, + { + "text": "This suggests that the SFT methods might overfit to the in-domain dataset , limiting their generalization to the out-of-domain dataset Lemon.", + "text_raw": "This suggests that the SFT methods might overfit to the in-domain dataset CSCD-NS, limiting their generalization to the out-of-domain dataset Lemon.", + "intents": [] + }, + { + "text": "Chinese characters often have multiple visually or phonetically similar variants, making it easy to select an incorrect one when using input methods .", + "text_raw": "Chinese characters often have multiple visually or phonetically similar variants, making it easy to select an incorrect one when using input methods (Hu et al., 2024).", + "intents": [] + }, + { + "text": "Removing it leads to performance drops of 11.10 and 5.45 points on the and C2EC datasets, respectively.", + "text_raw": "Removing it leads to performance drops of 11.10 and 5.45 points on the CSCD-NS and C2EC datasets, respectively.", + "intents": [] + }, + { + "text": "While the Levenshtein distance-based model performs better on the C2EC dataset, it degrades the performance on the dataset, the conventional CSC dataset.", + "text_raw": "While the Levenshtein distance-based model performs better on the C2EC dataset, it degrades the performance on the CSCD-NS dataset, the conventional CSC dataset.", + "intents": [] + }, + { + "text": "Similarly, created a new dataset with real errors found on social media.", + "text_raw": "Similarly, Hu et al. (2024) created a new dataset with real errors found on social media.", + "intents": [] + }, + { + "text": "Additionally, without any training, our method achieves performance on par with the SFT methods on the in-domain dataset , with scores of 71.53 versus 73.80 on the 14B model.", + "text_raw": "Additionally, without any training, our method achieves performance on par with the SFT methods on the in-domain dataset CSCD-NS, with scores of 71.53 versus 73.80 on the 14B model.", + "intents": [] + } + ] + }, + { + "citing_paper_id": "7ea52ad5c9400b5e914e86cede6e812940b5a27f", + "title": "Chinese Spelling Correction: A Comprehensive Survey of Progress, Challenges, and Opportunities", + "external_ids": { + "ArXiv": "2502.11508", + "DBLP": "journals/corr/abs-2502-11508", + "DOI": "10.48550/arXiv.2502.11508", + "CorpusId": 276408977 + }, + "is_influential": false, + "contexts": [ + { + "text": "The primary sources of errors in CSC typically stem from users\u2019 input method editors (IMEs) [], as well as inaccuracies introduced by automatic speech recognition (ASR) and optical character recognition (OCR) systems.", + "text_raw": "The primary sources of errors in CSC typically stem from users\u2019 input method editors (IMEs) [Hu et al. , 2024], as well as inaccuracies introduced by automatic speech recognition (ASR) and optical character recognition (OCR) systems.", + "intents": [] + }, + { + "text": "CSCD-NS [] represents the first CSC dataset specifically designed for native speakers, which is originated from authentic texts on Weibo.", + "text_raw": "CSCD-NS [Hu et al. , 2024] represents the first CSC dataset specifically designed for native speakers, which is originated from authentic texts on Weibo.", + "intents": [] + } + ] + }, + { + "citing_paper_id": "117b5eccd7163e7bb4d2f1bf05df0c08f6d79f8a", + "title": "Enhancing Character-Level Understanding in LLMs through Token Internal Structure Learning", + "external_ids": { + "DBLP": "journals/corr/abs-2411-17679", + "ArXiv": "2411.17679", + "DOI": "10.48550/arXiv.2411.17679", + "CorpusId": 274280941 + }, + "is_influential": true, + "contexts": [ + { + "text": "4.5 Character-Level Precision (CP), Recall (CR), and F1 Score (CF1) At the character level : \u2022 Precision (CP) is the proportion of incorrect characters identified by the model that are actually incorrect.", + "text_raw": "4.5 Character-Level Precision (CP), Recall (CR), and F1 Score (CF1) At the character level (Hu et al., 2024): \u2022 Precision (CP) is the proportion of incorrect characters identified by the model that are actually incorrect.", + "intents": [] + }, + { + "text": "\u20262024, 2022; Liang et al., 2023), we selected two new Chinese Spelling Correction (CSC) benchmarks, CSCD-NS and LEMON, to address the limitations identified ; Yin and Wan, 2023; Li et al., 2022) in previous datasets like SIGHAN (Wu et al., 2013; Yu and Li, 2014; Tseng et al., 2015).", + "text_raw": "\u20262024, 2022; Liang et al., 2023), we selected two new Chinese Spelling Correction (CSC) benchmarks, CSCD-NS and LEMON, to address the limitations identified (Hu et al., 2024; Yin and Wan, 2023; Li et al., 2022) in previous datasets like SIGHAN (Wu et al., 2013; Yu and Li, 2014; Tseng et al., 2015).", + "intents": [] + } + ] + }, + { + "citing_paper_id": "619346184ae157741fc1c4ac74447ab2455cd6f3", + "title": "A Simple yet Effective Training-free Prompt-free Approach to Chinese Spelling Correction Based on Large Language Models", + "external_ids": { + "DBLP": "conf/emnlp/0001LZ0L00024", + "ACL": "2024.emnlp-main.966", + "ArXiv": "2410.04027", + "DOI": "10.48550/arXiv.2410.04027", + "CorpusId": 273186532 + }, + "is_influential": true, + "contexts": [ + { + "text": "To identify these outliers, we leveraged tools from previous studies (Wu et al., 2023; by incorporating their structure confusion sets and spelling similarity matrices.", + "text_raw": "To identify these outliers, we leveraged tools from previous studies (Wu et al., 2023; Hu et al., 2024) by incorporating their structure confusion sets and spelling similarity matrices.", + "intents": [ + "methodology" + ] + }, + { + "text": "\u201a CSCD-NS : A real-world Chinese social media corpus collected and annotated by .", + "text_raw": "\u201a CSCD-NS : A real-world Chinese social media corpus collected and annotated by Hu et al. (2024).", + "intents": [ + "background" + ] + }, + { + "text": "Instead, we leverage the findings from Li et al. (2023a), who fine-tuned the Baichuan2 7B and GPT2 models on the ECSpell dataset, and , who fine-tuned the Baichuan2 7B model on the CSCD-NS dataset.", + "text_raw": "Instead, we leverage the findings from Li et al. (2023a), who fine-tuned the Baichuan2 7B and GPT2 models on the ECSpell dataset, and Hu et al. (2024), who fine-tuned the Baichuan2 7B model on the CSCD-NS dataset.", + "intents": [ + "methodology" + ] + }, + { + "text": "We perform experiments across five distinct CSC datasets: Sighans (Wu et al., 2013; Yu et al., 2014; Tseng et al., 2015), CSCD-NS , MCSCSet (Jiang et al., 2022), ECSpell (Lv et al., 2023), and Lemon (Wu et al., 2023), covering a broad spectrum of domains and genres.", + "text_raw": "We perform experiments across five distinct CSC datasets: Sighans (Wu et al., 2013; Yu et al., 2014; Tseng et al., 2015), CSCD-NS (Hu et al., 2024), MCSCSet (Jiang et al., 2022), ECSpell (Lv et al., 2023), and Lemon (Wu et al., 2023), covering a broad spectrum of domains and genres.", + "intents": [ + "methodology" + ] + } + ] + }, + { + "citing_paper_id": "ec3dbcb9661832d2fec3fb98e3b3cc3c2cf775d0", + "title": "OMGEval: An Open Multilingual Generative Evaluation Benchmark for Large Language Models", + "external_ids": { + "ArXiv": "2402.13524", + "DBLP": "journals/corr/abs-2402-13524", + "DOI": "10.48550/arXiv.2402.13524", + "CorpusId": 267770422 + }, + "is_influential": false, + "contexts": [ + { + "text": "Traditional evaluations of these models often revolved around restricted question-and-answer formats with definitive answers, typical in NLP tasks like Grammatical Error Correction (Ng et al., 2014; Bryant et al., 2019), Chinese Spelling Correction (Yu et al., 2014; .", + "text_raw": "Traditional evaluations of these models often revolved around restricted question-and-answer formats with definitive answers, typical in NLP tasks like Grammatical Error Correction (Ng et al., 2014; Bryant et al., 2019), Chinese Spelling Correction (Yu et al., 2014; Hu et al., 2022).", + "intents": [ + "background" + ] + } + ] + }, + { + "citing_paper_id": "571754f4ed4303b02a1a5e78c6605f483b2ea6e0", + "title": "Eval-GCSC: A New Metric for Evaluating ChatGPT's Performance in Chinese Spelling Correction", + "external_ids": { + "ArXiv": "2311.08219", + "DBLP": "journals/corr/abs-2311-08219", + "DOI": "10.48550/arXiv.2311.08219", + "CorpusId": 265158218 + }, + "is_influential": true, + "contexts": [ + { + "text": "Metrics We computed Eval-CSC based on Script , and implemented Eval-GCSC following the procedures outlined in Section 4.3.", + "text_raw": "Metrics We computed Eval-CSC based on Script (Hu et al., 2022), and implemented Eval-GCSC following the procedures outlined in Section 4.3.", + "intents": [ + "methodology" + ] + } + ] + }, + { + "citing_paper_id": "db2208df9148b0ddc65e21d7a0883ca54edd4e66", + "title": "Evolving Chinese Spelling Correction with Corrector-Verifier Collaboration", + "external_ids": { + "DBLP": "conf/emnlp/LiuWZ25", + "DOI": "10.18653/v1/2025.emnlp-main.1477", + "CorpusId": 282813715 + }, + "is_influential": true, + "contexts": [ + { + "text": "We generate the synthetic data using IME .", + "text_raw": "We generate the synthetic data using IME (Hu et al., 2024).", + "intents": [] + }, + { + "text": "5-14b finetuned combines the human annotated data from and 271K pseudo-data generated by ASR or OCR as the training data, and uses character-level tokenization.", + "text_raw": "5-14b finetuned combines the human annotated data from CSCD-NS and 271K pseudo-data generated by ASR or OCR as the training data, and uses character-level tokenization.", + "intents": [] + }, + { + "text": "However, Table 1 shows that while ACI and synthetic+human show comparable performance across various domains in LEMON, synthetic+human exhibits better performance on .", + "text_raw": "However, Table 1 shows that while ACI and synthetic+human show comparable performance across various domains in LEMON, synthetic+human exhibits better performance on CSCD-NS.", + "intents": [] + }, + { + "text": "(2) superior in annotation quality and focus on spelling errors stemming from pinyin input methods.", + "text_raw": "(2) CSCD-NS superior in annotation quality and focus on spelling errors stemming from pinyin input methods.", + "intents": [] + }, + { + "text": "Dataset A line of studies has reported issues with SIGHAN (Tseng et al., 2015), such as annotation error and incoherent style with native speakers (Wu et al., 2023; ; Li et al., 2024).", + "text_raw": "Dataset A line of studies has reported issues with SIGHAN (Tseng et al., 2015), such as annotation error and incoherent style with native speakers (Wu et al., 2023; Hu et al., 2024; Li et al., 2024).", + "intents": [] + }, + { + "text": "A line of works also propose different data augmentation methods to construct pseudo data to address the scarcity of CSC data (Wang et al., 2018; ; Sheng and Xu, 2024).", + "text_raw": "A line of works also propose different data augmentation methods to construct pseudo data to address the scarcity of CSC data (Wang et al., 2018; Hu et al., 2024; Sheng and Xu, 2024).", + "intents": [] + }, + { + "text": "To evaluate the quality of generated training data, we probe the annotation accuracy of different approaches on development set.", + "text_raw": "To evaluate the quality of generated training data, we probe the annotation accuracy of different approaches on CSCD-NS development set.", + "intents": [] + }, + { + "text": "These models suffer from biased error patterns learned on synthetic data, leading to over-correction issues and inadequate handling of semantic errors (Liu et al., 2025; Wu et al., 2023; ; Liu et al., 2022; Jiang et al., 2024).", + "text_raw": "These models suffer from biased error patterns learned on synthetic data, leading to over-correction issues and inadequate handling of semantic errors (Liu et al., 2025; Wu et al., 2023; Hu et al., 2024; Liu et al., 2022; Jiang et al., 2024).", + "intents": [] + }, + { + "text": "A recent line of work develops large language models (LLMs) for CSC (Li et al., 2024; Zhou et al., 2024) while some others continue to elaborate BERT-based models (Wu et al., 2023; ; Liu et al., 2024; Zhu et al., 2022; Sheng and Xu, 2024).", + "text_raw": "A recent line of work develops large language models (LLMs) for CSC (Li et al., 2024; Zhou et al., 2024) while some others continue to elaborate BERT-based models (Wu et al., 2023; Hu et al., 2024; Liu et al., 2024; Zhu et al., 2022; Sheng and Xu, 2024).", + "intents": [] + }, + { + "text": "Following recent works, we use two CSC benchmarks, LEMON (Wu et al., 2023) and CSCD-NS .", + "text_raw": "Following recent works, we use two CSC benchmarks, LEMON (Wu et al., 2023) and CSCD-NS (Hu et al., 2024).", + "intents": [] + }, + { + "text": "We first train the model on IME-based synthetic data and then train it on human annotated data.", + "text_raw": "We first train the model on IME-based synthetic data and then train it on human annotated data.", + "intents": [] + }, + { + "text": "We compare the annotation results with the gold labels from .", + "text_raw": "We compare the annotation results with the gold labels from CSCD-NS.", + "intents": [] + }, + { + "text": "The data we use is the training set of , which is in high quality.", + "text_raw": "The data we use is the training set of CSCD-NS, which is in high quality.", + "intents": [] + }, + { + "text": "This performance gap can be attributed to the domain alignment between the human-29013 annotated training data and test data, suggesting the potential benefit of incorporating human-annotated and domain-specific data in certain scenarios.", + "text_raw": "This performance gap can be attributed to the domain alignment between the human-29013 annotated training data and CSCD-NS test data, suggesting the potential benefit of incorporating human-annotated and domain-specific data in certain scenarios.", + "intents": [] + }, + { + "text": "\u2022 IME-based synthetic + Human annotated The two-stage training of first using synthetic and then using human-annotated data is the widely-used and the most useful method.", + "text_raw": "\u2022 IME-based synthetic + Human annotated The two-stage training of first using synthetic and then using human-annotated data is the widely-used and the most useful method.", + "intents": [] + } + ] + }, + { + "citing_paper_id": "3a117be97046775e6b21ffc19a33df8f98faab94", + "title": "PhonoThink: Improving Large Language Models' Reasoning on Chinese Phonological Ambiguities", + "external_ids": { + "DBLP": "conf/emnlp/MaFCSZ25", + "DOI": "10.18653/v1/2025.emnlp-main.961", + "CorpusId": 282904714 + }, + "is_influential": false, + "contexts": [] + }, + { + "citing_paper_id": "175d21f15ae0eebf6a5e3ca017b34b3b54e771a4", + "title": "RACQC: Advanced Retrieval-Augmented Generation for Chinese Query Correction", + "external_ids": { + "DBLP": "conf/emnlp/SuGLLLWGWSY25", + "DOI": "10.18653/v1/2025.findings-emnlp.36", + "CorpusId": 282905368 + }, + "is_influential": false, + "contexts": [ + { + "text": "Our investigation reveals that general CSC datasets ; Wu et al., 2023) in mainstream research often overlook entity-level error correction critical for open-domain search, focusing primarily on common entities with limited coverage of long-tail ones.", + "text_raw": "Our investigation reveals that general CSC datasets (Hu et al., 2022; Wu et al., 2023) in mainstream research often overlook entity-level error correction critical for open-domain search, focusing primarily on common entities with limited coverage of long-tail ones.", + "intents": [] + }, + { + "text": "To improve their error correction abilities, strategies such as data synthesis(Wang et al., 2024b; , incorporating error detection modules(Zhang et al., 2020a), and specific character masking strategies(Liu et al., 2010) have been used.", + "text_raw": "To improve their error correction abilities, strategies such as data synthesis(Wang et al., 2024b; Hu et al., 2022), incorporating error detection modules(Zhang et al., 2020a), and specific character masking strategies(Liu et al., 2010) have been used.", + "intents": [] + } + ] + }, + { + "citing_paper_id": "6724d84670043e731b4f02f0e62bdcdbe447e577", + "title": "Enhancing Chinese Offensive Language Detection with Homophonic Perturbation", + "external_ids": { + "DBLP": "conf/emnlp/WuJZPZLW25", + "DOI": "10.18653/v1/2025.emnlp-main.1154", + "CorpusId": 282902372 + }, + "is_influential": false, + "contexts": [] + } + ] +} \ No newline at end of file diff --git a/replay_traces/2211.08788/processed_papers/2211.08788/usage_contributions.json b/replay_traces/2211.08788/processed_papers/2211.08788/usage_contributions.json new file mode 100644 index 0000000000000000000000000000000000000000..0fdbbb9d3c96768d7a2f3c0e8e48f9e03be50332 --- /dev/null +++ b/replay_traces/2211.08788/processed_papers/2211.08788/usage_contributions.json @@ -0,0 +1,149 @@ +{ + "paper_id": "2211.08788", + "final_label": "USES", + "contributions": [ + { + "citing_paper_id": "571754f4ed4303b02a1a5e78c6605f483b2ea6e0", + "citing_title": "Eval-GCSC: A New Metric for Evaluating ChatGPT's Performance in Chinese Spelling Correction", + "label": "USES", + "paper_claim": "The authors use the evaluation script from the target paper to compute the Eval-CSC metric.", + "claim": "The authors use the evaluation script from the target paper to compute the Eval-CSC metric.", + "cluster_title": "Uses target evaluation script for performance evaluation", + "cluster_key": "USES|evaluation_script|evaluation", + "evidence_span": "We computed Eval-CSC based on Script ", + "rationale": "The citing paper explicitly mentions computing the Eval-CSC metric based on the script provided by the target paper.", + "contexts": [ + "Target sentence: Metrics We computed Eval-CSC based on Script , and implemented Eval-GCSC following the procedures outlined in Section 4.3." + ], + "source": "usage_contexts_fallback" + }, + { + "citing_paper_id": "9a6479bdd7605029d3972dea4c64a2e914f2755a", + "citing_title": "A Training-free LLM-based Approach to General Chinese Character Error Correction", + "label": "USES", + "paper_claim": "The citing paper uses the CSCD-NS dataset for model evaluation and as part of its training data.", + "claim": "The citing paper uses the CSCD-NS dataset for model evaluation and as part of its training data.", + "cluster_title": "Uses target dataset for model evaluation", + "cluster_key": "USES|dataset|evaluation", + "evidence_span": "For conventional CSC, following Li et al. (2024), we use two representative datasets: CSCD-NS and Lemon (Wu et al., 2023).", + "rationale": "The citing paper explicitly states it uses the target paper's dataset (CSCD-NS) as a representative dataset for evaluation.", + "contexts": [ + "Target sentence: Figure 11 shows the performance of our method with varying beam sizes on the and C2EC datasets.", + "Target sentence: For conventional CSC, following Li et al. (2024), we use two representative datasets: CSCD-NS and Lemon (Wu et al., 2023).", + "Target sentence: We randomly sampled 50 sentences of varying lengths from the development set of and C2EC to evaluate the running time of our method on Qwen2.", + "Target sentence: When compared to the TfPf method, the SFT methods show superior performance on the in-domain dataset .", + "Target sentence: Training Details of SFT Baselines The SFT models were trained on a combined dataset consisting of 271k pseudo sentence pairs (Wang et al., 2018) and the training data." + ], + "source": "usage_contexts_fallback" + }, + { + "citing_paper_id": "619346184ae157741fc1c4ac74447ab2455cd6f3", + "citing_title": "A Simple yet Effective Training-free Prompt-free Approach to Chinese Spelling Correction Based on Large Language Models", + "label": "USES", + "paper_claim": "The citing paper utilizes the target dataset and tools, including structure confusion sets and spelling similarity matrices, to evaluate its spelling correction approach.", + "claim": "The citing paper utilizes the target dataset and tools, including structure confusion sets and spelling similarity matrices, to evaluate its spelling correction approach.", + "cluster_title": "Uses target dataset for evaluation", + "cluster_key": "USES|dataset|evaluation", + "evidence_span": "We perform experiments across five distinct CSC datasets: Sighans (Wu et al., 2013; Yu et al., 2014; Tseng et al., 2015), CSCD-NS ", + "rationale": "The citing paper explicitly states that it performs experiments on the target paper's dataset (CSCD-NS) and leverages its structure confusion sets and spelling similarity matrices.", + "contexts": [ + "Target sentence: To identify these outliers, we leveraged tools from previous studies (Wu et al., 2023; by incorporating their structure confusion sets and spelling similarity matrices.", + "Target sentence: We perform experiments across five distinct CSC datasets: Sighans (Wu et al., 2013; Yu et al., 2014; Tseng et al., 2015), CSCD-NS , MCSCSet (Jiang et al., 2022), ECSpell (Lv et al., 2023), and Lemon (Wu et al., 2023), covering a broad spectrum of domains and genres." + ], + "source": "usage_contexts_fallback" + }, + { + "citing_paper_id": "db2208df9148b0ddc65e21d7a0883ca54edd4e66", + "citing_title": "Evolving Chinese Spelling Correction with Corrector-Verifier Collaboration", + "label": "USES", + "paper_claim": "Uses the CSCD-NS dataset as a benchmark for evaluating Chinese spelling correction models.", + "claim": "Uses the CSCD-NS dataset as a benchmark for evaluating Chinese spelling correction models.", + "cluster_title": "Uses target dataset for evaluation", + "cluster_key": "USES|dataset|evaluation", + "evidence_span": "we use two CSC benchmarks, LEMON (Wu et al., 2023) and CSCD-NS .", + "rationale": "The citing paper explicitly states that it uses the CSCD-NS benchmark proposed by the target paper.", + "contexts": [ + "Target sentence: We generate the synthetic data using IME .", + "Target sentence: 5-14b finetuned combines the human annotated data from and 271K pseudo-data generated by ASR or OCR as the training data, and uses character-level tokenization.", + "Target sentence: To evaluate the quality of generated training data, we probe the annotation accuracy of different approaches on development set.", + "Target sentence: Following recent works, we use two CSC benchmarks, LEMON (Wu et al., 2023) and CSCD-NS .", + "Target sentence: We compare the annotation results with the gold labels from .", + "Target sentence: The data we use is the training set of , which is in high quality." + ], + "source": "usage_contexts_fallback" + }, + { + "citing_paper_id": "7d64dd752bb29bb9aff8a836b4b1f5f3ccafe807", + "citing_title": "ChineseErrorCorrector3-4B: State-of-the-Art Chinese Spelling and Grammar Corrector", + "label": "USES", + "paper_claim": "Uses the CSCD-NS dataset as part of the aggregated CSC Datasets for dataset composition.", + "claim": "Uses the CSCD-NS dataset as part of the aggregated CSC Datasets for dataset composition.", + "cluster_title": "Uses target dataset for dataset composition", + "cluster_key": "USES|dataset|dataset_composition", + "evidence_span": "CSC Datasets ($\\mathcal{D}_{\\text{csc}}$): Aggregated from W271K, Medical\\cite{jiang2022mcscsetspecialistannotateddatasetmedicaldomain}, Lemon\\cite{wu2023rethinking}, ECSpell\\cite{lv2023general}, and CSCD", + "rationale": "The citing paper explicitly states that it aggregates data from multiple sources, including the target paper's dataset (CSCD), to form its dataset composition.", + "contexts": [ + "Target-citing paragraph: \\paragraph{Dataset Composition.}\nWe first aggregate data for the two task families:\n\\begin{itemize}\n \\item CSC Datasets ($\\mathcal{D}_{\\text{csc}}$): Aggregated from W271K, Medical\\cite{jiang2022mcscsetspecialistannotateddatasetmedicaldomain}, Lemon\\cite{wu2023rethinking}, ECSpell\\cite{lv2023general}, and CSCD (totaling approx. 380K samples).\n \\item CGC Datasets ($\\mathcal{D}_{\\text{cgc}}$): Aggregated from CGED\\cite{rao2018overview, rao2020overview}, FCGEC\\cite{xu2022fcgec}, MuCGEC\\cite{zhang2022mucgec}, and NaCGEC\\cite{zhang2023nasgec} (totaling approx. 68K samples).\n\\end{itemize}" + ], + "source": "arxiv_paragraphs" + }, + { + "citing_paper_id": "87a2958bd75a72e61b5b2549a63a222d7ca2a1ec", + "citing_title": "Mixture of Small and Large Models for Chinese Spelling Check", + "label": "USES", + "paper_claim": "The citing paper evaluates its proposed method on the CSCD-NS dataset to test performance on data written by native speakers.", + "claim": "The citing paper evaluates its proposed method on the CSCD-NS dataset to test performance on data written by native speakers.", + "cluster_title": "Uses target dataset for evaluation", + "cluster_key": "USES|dataset|evaluation", + "evidence_span": "we further conduct experiments on more diverse datasets, including LEMON (Wu et al., 2023), EC-Spell (Lv et al., 2023), CSCD-NS ", + "rationale": "The citing paper explicitly states that it conducts experiments on the CSCD-NS dataset to evaluate its method on more diverse, native-speaker data.", + "contexts": [ + "Target sentence: Due to the lack of domain diversity in the SIGHAN datasets, we further conduct experiments on more diverse datasets, including LEMON (Wu et al., 2023), EC-Spell (Lv et al., 2023), CSCD-NS , and MCSCSet (Jiang et al., 2022), all of which were written by native speakers.", + "Target sentence: Table 2 presents experimental results across five benchmark datasets: rSIGHANs, , MC-SCSet, ECSpell, and LEMON." + ], + "source": "usage_contexts_fallback" + }, + { + "citing_paper_id": "0e2e19f1fae35d97b3178f9349caf95da540652d", + "citing_title": "Incorporating Confused Phraseological Knowledge Based on Pinyin Input Method for Chinese Spelling Correction", + "label": "USES", + "paper_claim": "Uses the Chinese spelling check dataset from the target paper as the fundamental dataset for model training and evaluation experiments.", + "claim": "Uses the Chinese spelling check dataset from the target paper as the fundamental dataset for model training and evaluation experiments.", + "cluster_title": "Uses target dataset for experiments", + "cluster_key": "USES|dataset|experiments", + "evidence_span": "this paper uses the Chinese pinyin IME error correction dataset CSCD-IME as the basic dataset for the experiments", + "rationale": "The citing paper explicitly adopts the dataset introduced in the target paper to serve as the basic dataset for its training, validation, and testing experiments.", + "contexts": [ + "Target sentence: The transformation method follows the pseudo-data construction method proposed by to ensure that the converted data closely resembles the actual scenario.", + "Target sentence: To better simulate real input scenarios, this paper uses the Chinese pinyin IME error correction dataset CSCD-IME as the basic dataset for the experiments, including 3,0000 training data, 5000 validation data, and 5000 test data." + ], + "source": "usage_contexts_fallback" + }, + { + "citing_paper_id": "4b66b25de8c6c496f47ee58cf2677d0b377aa778", + "citing_title": "Towards More Reliable Chinese Spelling Correction: Fine-Grained Confidence Estimation Against Suboptimal Corrections", + "label": "USES", + "paper_claim": "The citing paper evaluates its approach using the target paper's native-speaker-oriented Chinese spelling check dataset and adopts its evaluation metrics.", + "claim": "The citing paper evaluates its approach using the target paper's native-speaker-oriented Chinese spelling check dataset and adopts its evaluation metrics.", + "cluster_title": "Uses target dataset for evaluation", + "cluster_key": "USES|dataset|evaluation", + "evidence_span": "We also include a native-speaker-oriented dataset collected from social media", + "rationale": "The citing paper explicitly states that it includes the target paper's dataset in its experiments and follows the target paper's evaluation metrics.", + "contexts": [ + "Target sentence: We also include a native-speaker-oriented dataset collected from social media, which contains a manually annotated subset and a large-scale pseudo-labeled corpus.", + "Target sentence: 2) Metrics: We follow previous works , [38] and report strict character-level precision, recall, and F1 scores for both detection and correction." + ], + "source": "usage_contexts_fallback" + } + ], + "source": "arxiv_paragraphs", + "fallback_citing_ids": [ + "571754f4ed4303b02a1a5e78c6605f483b2ea6e0", + "9a6479bdd7605029d3972dea4c64a2e914f2755a", + "619346184ae157741fc1c4ac74447ab2455cd6f3", + "db2208df9148b0ddc65e21d7a0883ca54edd4e66", + "87a2958bd75a72e61b5b2549a63a222d7ca2a1ec", + "0e2e19f1fae35d97b3178f9349caf95da540652d", + "4b66b25de8c6c496f47ee58cf2677d0b377aa778" + ] +} \ No newline at end of file diff --git a/replay_traces/2211.08788/processed_papers/2211.08788/usage_discovery_from_contributions.json b/replay_traces/2211.08788/processed_papers/2211.08788/usage_discovery_from_contributions.json new file mode 100644 index 0000000000000000000000000000000000000000..ec0d72665ca35b5277380f23225578119fe3b6a8 --- /dev/null +++ b/replay_traces/2211.08788/processed_papers/2211.08788/usage_discovery_from_contributions.json @@ -0,0 +1,46 @@ +{ + "paper_id": "2211.08788", + "decision": "", + "justification": "", + "clusters": [ + { + "cluster_id": "C1", + "count": "1", + "representative_claim": "Uses target evaluation script for performance evaluation", + "cluster_key": "USES|evaluation_script|evaluation", + "cluster_title": "Uses target evaluation script for performance evaluation", + "claim_indices": [ + "0" + ], + "source_cluster_ids": [ + "C1" + ], + "merge_rationale": "Represents the direct use of the target paper's evaluation script to compute metrics." + }, + { + "cluster_id": "C2", + "count": "7", + "representative_claim": "Uses target dataset for model training and evaluation", + "cluster_key": "USES|dataset|experiments", + "cluster_title": "Uses target dataset for model training and evaluation", + "claim_indices": [ + "1", + "2", + "3", + "4", + "5", + "6", + "7" + ], + "source_cluster_ids": [ + "C2", + "C3", + "C4" + ], + "merge_rationale": "These clusters all describe using the target dataset (CSCD-NS) for downstream experiments, including model evaluation, training, and dataset aggregation." + } + ], + "dropped_clusters": [], + "cluster_refine_method": "llm_centroid_merge_filter", + "cluster_refine_source": "usage_contributions.json" +} \ No newline at end of file diff --git a/replay_traces/2211.08788/processed_papers/2211.08788/usage_uses_extends_verified.json b/replay_traces/2211.08788/processed_papers/2211.08788/usage_uses_extends_verified.json new file mode 100644 index 0000000000000000000000000000000000000000..65af8446e3c545ea8abb8ead95b2644c8c8cbe84 --- /dev/null +++ b/replay_traces/2211.08788/processed_papers/2211.08788/usage_uses_extends_verified.json @@ -0,0 +1,530 @@ +{ + "paper_id": "2211.08788", + "target": { + "title": "CSCD-NS: a Chinese Spelling Check Dataset for Native Speakers", + "first_author_last": "Hu", + "year": "2022" + }, + "candidates_total": 30, + "candidates_considered": 30, + "verification_batch_size": 25, + "verification_num_batches": 2, + "candidates_selected": 21, + "verified": [ + { + "id": 39, + "label": "USES", + "cue_span": "We generate the synthetic data using IME", + "rationale": "The citing paper uses the IME method from the target paper to generate synthetic data.", + "text": "We generate the synthetic data using IME .", + "citing_paper_id": "db2208df9148b0ddc65e21d7a0883ca54edd4e66", + "citing_title": "Evolving Chinese Spelling Correction with Corrector-Verifier Collaboration", + "original_label": "Uses" + }, + { + "id": 48, + "label": "USES", + "cue_span": "we use two CSC benchmarks, LEMON (Wu et al., 2023) and CSCD-NS", + "rationale": "The citing authors explicitly state they use the target paper's dataset as a benchmark.", + "text": "Following recent works, we use two CSC benchmarks, LEMON (Wu et al., 2023) and CSCD-NS .", + "citing_paper_id": "db2208df9148b0ddc65e21d7a0883ca54edd4e66", + "citing_title": "Evolving Chinese Spelling Correction with Corrector-Verifier Collaboration", + "original_label": "Uses" + }, + { + "id": 50, + "label": "USES", + "cue_span": "We compare the annotation results with the gold labels from", + "rationale": "The citing paper uses gold labels provided by the target paper for comparison.", + "text": "We compare the annotation results with the gold labels from .", + "citing_paper_id": "db2208df9148b0ddc65e21d7a0883ca54edd4e66", + "citing_title": "Evolving Chinese Spelling Correction with Corrector-Verifier Collaboration", + "original_label": "Uses" + }, + { + "id": 38, + "label": "USES", + "cue_span": "We computed Eval-CSC based on Script", + "rationale": "The citing paper uses a script provided by the target paper to compute metrics.", + "text": "Metrics We computed Eval-CSC based on Script , and implemented Eval-GCSC following the procedures outlined in Section 4.3.", + "citing_paper_id": "571754f4ed4303b02a1a5e78c6605f483b2ea6e0", + "citing_title": "Eval-GCSC: A New Metric for Evaluating ChatGPT's Performance in Chinese Spelling Correction", + "original_label": "Uses" + }, + { + "id": 5, + "label": "USES", + "cue_span": "The transformation method follows the pseudo-data construction method proposed by", + "rationale": "The citing paper adopts a specific methodology for data construction proposed by the target paper.", + "text": "The transformation method follows the pseudo-data construction method proposed by to ensure that the converted data closely resembles the actual scenario.", + "citing_paper_id": "0e2e19f1fae35d97b3178f9349caf95da540652d", + "citing_title": "Incorporating Confused Phraseological Knowledge Based on Pinyin Input Method for Chinese Spelling Correction", + "original_label": "Uses" + }, + { + "id": 18, + "label": "USES", + "cue_span": "We randomly sampled 50 sentences of varying lengths from the development set of", + "rationale": "The citing paper uses samples from the target paper's development set for evaluation.", + "text": "We randomly sampled 50 sentences of varying lengths from the development set of and C2EC to evaluate the running time of our method on Qwen2.", + "citing_paper_id": "9a6479bdd7605029d3972dea4c64a2e914f2755a", + "citing_title": "A Training-free LLM-based Approach to General Chinese Character Error Correction", + "original_label": "Uses" + }, + { + "id": 16, + "label": "USES", + "cue_span": "we use two representative datasets: CSCD-NS", + "rationale": "The citing authors explicitly state they use the target paper's dataset.", + "text": "For conventional CSC, following Li et al. (2024), we use two representative datasets: CSCD-NS and Lemon (Wu et al., 2023).", + "citing_paper_id": "9a6479bdd7605029d3972dea4c64a2e914f2755a", + "citing_title": "A Training-free LLM-based Approach to General Chinese Character Error Correction", + "original_label": "Uses" + }, + { + "id": 45, + "label": "USES", + "cue_span": "we probe the annotation accuracy of different approaches on development set", + "rationale": "The citing authors use the target paper's development set to evaluate their approaches.", + "text": "To evaluate the quality of generated training data, we probe the annotation accuracy of different approaches on development set.", + "citing_paper_id": "db2208df9148b0ddc65e21d7a0883ca54edd4e66", + "citing_title": "Evolving Chinese Spelling Correction with Corrector-Verifier Collaboration", + "original_label": "Uses" + }, + { + "id": 1, + "label": "USES", + "cue_span": "We also include", + "rationale": "The citing paper includes the target paper's dataset in its experimental setup.", + "text": "We also include a native-speaker-oriented dataset collected from social media, which contains a manually annotated subset and a large-scale pseudo-labeled corpus.", + "citing_paper_id": "4b66b25de8c6c496f47ee58cf2677d0b377aa778", + "citing_title": "Towards More Reliable Chinese Spelling Correction: Fine-Grained Confidence Estimation Against Suboptimal Corrections", + "original_label": "Uses" + }, + { + "id": 22, + "label": "USES", + "cue_span": "The SFT models were trained on a combined dataset consisting of 271k pseudo sentence pairs (Wang et al., 2018) and the training data", + "rationale": "The citing paper uses the target paper's training data to train its models.", + "text": "Training Details of SFT Baselines The SFT models were trained on a combined dataset consisting of 271k pseudo sentence pairs (Wang et al., 2018) and the training data.", + "citing_paper_id": "9a6479bdd7605029d3972dea4c64a2e914f2755a", + "citing_title": "A Training-free LLM-based Approach to General Chinese Character Error Correction", + "original_label": "Uses" + }, + { + "id": 51, + "label": "USES", + "cue_span": "The data we use is the training set of", + "rationale": "The citing authors explicitly state they use the target paper's training set.", + "text": "The data we use is the training set of , which is in high quality.", + "citing_paper_id": "db2208df9148b0ddc65e21d7a0883ca54edd4e66", + "citing_title": "Evolving Chinese Spelling Correction with Corrector-Verifier Collaboration", + "original_label": "Uses" + }, + { + "id": 6, + "label": "USES", + "cue_span": "this paper uses the Chinese pinyin IME error correction dataset CSCD-IME", + "rationale": "The citing paper explicitly states it uses the target paper's dataset for its experiments.", + "text": "To better simulate real input scenarios, this paper uses the Chinese pinyin IME error correction dataset CSCD-IME as the basic dataset for the experiments, including 3,0000 training data, 5000 validation data, and 5000 test data.", + "citing_paper_id": "0e2e19f1fae35d97b3178f9349caf95da540652d", + "citing_title": "Incorporating Confused Phraseological Knowledge Based on Pinyin Input Method for Chinese Spelling Correction", + "original_label": "Uses" + }, + { + "id": 36, + "label": "USES", + "cue_span": "We perform experiments across five distinct CSC datasets: Sighans (Wu et al., 2013; Yu et al., 2014; Tseng et al., 2015), CSCD-NS", + "rationale": "The citing authors state they use the target paper's dataset in their experiments.", + "text": "We perform experiments across five distinct CSC datasets: Sighans (Wu et al., 2013; Yu et al., 2014; Tseng et al., 2015), CSCD-NS , MCSCSet (Jiang et al., 2022), ECSpell (Lv et al., 2023), and Lemon (Wu et al., 2023), covering a broad spectrum of domains and genres.", + "citing_paper_id": "619346184ae157741fc1c4ac74447ab2455cd6f3", + "citing_title": "A Simple yet Effective Training-free Prompt-free Approach to Chinese Spelling Correction Based on Large Language Models", + "original_label": "Uses" + }, + { + "id": 4, + "label": "USES", + "cue_span": "We first aggregate data for the two task families: \u2022 CSC Datasets ( D csc ): Aggregated from W271K, Medical[11], Lemon[12], ECSpell[19], and CSCD[14]", + "rationale": "The citing paper uses data aggregated from the target paper (cited as [14]).", + "text": "We first aggregate data for the two task families: \u2022 CSC Datasets ( D csc ): Aggregated from W271K, Medical[11], Lemon[12], ECSpell[19], and CSCD[14] (totaling approx.", + "citing_paper_id": "7d64dd752bb29bb9aff8a836b4b1f5f3ccafe807", + "citing_title": "ChineseErrorCorrector3-4B: State-of-the-Art Chinese Spelling and Grammar Corrector", + "original_label": "Uses" + }, + { + "id": 31, + "label": "NOT_CONFIRMED", + "cue_span": "", + "rationale": "The sentence provides a description of metrics but lacks an explicit action of adoption by the citing paper's authors.", + "text": "4.5 Character-Level Precision (CP), Recall (CR), and F1 Score (CF1) At the character level : \u2022 Precision (CP) is the proportion of incorrect characters identified by the model that are actually incorrect.", + "citing_paper_id": "117b5eccd7163e7bb4d2f1bf05df0c08f6d79f8a", + "citing_title": "Enhancing Character-Level Understanding in LLMs through Token Internal Structure Learning", + "original_label": "Uses" + }, + { + "id": 40, + "label": "USES", + "cue_span": "5-14b finetuned combines the human annotated data from", + "rationale": "The citing paper's model uses data from the target paper for fine-tuning purposes.", + "text": "5-14b finetuned combines the human annotated data from and 271K pseudo-data generated by ASR or OCR as the training data, and uses character-level tokenization.", + "citing_paper_id": "db2208df9148b0ddc65e21d7a0883ca54edd4e66", + "citing_title": "Evolving Chinese Spelling Correction with Corrector-Verifier Collaboration", + "original_label": "Uses" + }, + { + "id": 49, + "label": "NOT_CONFIRMED", + "cue_span": "", + "rationale": "The sentence does not contain the target citation marker and therefore does not confirm a relationship to the target paper.", + "text": "We first train the model on IME-based synthetic data and then train it on human annotated data.", + "citing_paper_id": "db2208df9148b0ddc65e21d7a0883ca54edd4e66", + "citing_title": "Evolving Chinese Spelling Correction with Corrector-Verifier Collaboration", + "original_label": "Uses" + }, + { + "id": 11, + "label": "USES", + "cue_span": "we further conduct experiments on more diverse datasets, including LEMON (Wu et al., 2023), EC-Spell (Lv et al., 2023), CSCD-NS", + "rationale": "The citing authors use the target paper's dataset for conducting their experiments.", + "text": "Due to the lack of domain diversity in the SIGHAN datasets, we further conduct experiments on more diverse datasets, including LEMON (Wu et al., 2023), EC-Spell (Lv et al., 2023), CSCD-NS , and MCSCSet (Jiang et al., 2022), all of which were written by native speakers.", + "citing_paper_id": "87a2958bd75a72e61b5b2549a63a222d7ca2a1ec", + "citing_title": "Mixture of Small and Large Models for Chinese Spelling Check", + "original_label": "Uses" + }, + { + "id": 30, + "label": "NOT_CONFIRMED", + "cue_span": "", + "rationale": "The sentence provides a description of the target paper's dataset without explicit adoption by the citing authors.", + "text": "CSCD-NS [] represents the first CSC dataset specifically designed for native speakers, which is originated from authentic texts on Weibo.", + "citing_paper_id": "7ea52ad5c9400b5e914e86cede6e812940b5a27f", + "citing_title": "Chinese Spelling Correction: A Comprehensive Survey of Progress, Challenges, and Opportunities", + "original_label": "Uses" + }, + { + "id": 15, + "label": "USES", + "cue_span": "performance of our method with varying beam sizes on the ", + "rationale": "The citing paper evaluates its own method on the target paper's dataset.", + "text": "Figure 11 shows the performance of our method with varying beam sizes on the and C2EC datasets.", + "citing_paper_id": "9a6479bdd7605029d3972dea4c64a2e914f2755a", + "citing_title": "A Training-free LLM-based Approach to General Chinese Character Error Correction", + "original_label": "Uses" + }, + { + "id": 12, + "label": "USES", + "cue_span": "Table 2 presents experimental results across five benchmark datasets: rSIGHANs, ", + "rationale": "The citing paper reports results derived from testing on the target paper's benchmark dataset.", + "text": "Table 2 presents experimental results across five benchmark datasets: rSIGHANs, , MC-SCSet, ECSpell, and LEMON.", + "citing_paper_id": "87a2958bd75a72e61b5b2549a63a222d7ca2a1ec", + "citing_title": "Mixture of Small and Large Models for Chinese Spelling Check", + "original_label": "Uses" + }, + { + "id": 33, + "label": "USES", + "cue_span": "we leveraged tools from previous studies (Wu et al., 2023; ", + "rationale": "The citing paper explicitly mentions leveraging tools provided by the target paper.", + "text": "To identify these outliers, we leveraged tools from previous studies (Wu et al., 2023; by incorporating their structure confusion sets and spelling similarity matrices.", + "citing_paper_id": "619346184ae157741fc1c4ac74447ab2455cd6f3", + "citing_title": "A Simple yet Effective Training-free Prompt-free Approach to Chinese Spelling Correction Based on Large Language Models", + "original_label": "Uses" + }, + { + "id": 14, + "label": "NOT_CONFIRMED", + "cue_span": "", + "rationale": "The sentence describes the licensing of the target dataset rather than its use or extension.", + "text": "Specifically, the dataset is publicly available under the MIT license, while the CCTC dataset is publicly available under the Apache 2.0 license.", + "citing_paper_id": "9a6479bdd7605029d3972dea4c64a2e914f2755a", + "citing_title": "A Training-free LLM-based Approach to General Chinese Character Error Correction", + "original_label": "Uses" + }, + { + "id": 34, + "label": "NOT_CONFIRMED", + "cue_span": "", + "rationale": "The sentence describes the source and nature of the dataset introduced by the target paper.", + "text": "\u201a CSCD-NS : A real-world Chinese social media corpus collected and annotated by .", + "citing_paper_id": "619346184ae157741fc1c4ac74447ab2455cd6f3", + "citing_title": "A Simple yet Effective Training-free Prompt-free Approach to Chinese Spelling Correction Based on Large Language Models", + "original_label": "Uses" + }, + { + "id": 20, + "label": "USES", + "cue_span": "the SFT methods show superior performance on the in-domain dataset ", + "rationale": "The citing authors evaluate their SFT methods on the target paper's dataset.", + "text": "When compared to the TfPf method, the SFT methods show superior performance on the in-domain dataset .", + "citing_paper_id": "9a6479bdd7605029d3972dea4c64a2e914f2755a", + "citing_title": "A Training-free LLM-based Approach to General Chinese Character Error Correction", + "original_label": "Uses" + }, + { + "id": 3, + "label": "USES", + "cue_span": "We follow previous works ", + "rationale": "The citing paper adopts metrics from the target paper and includes the explicit actor 'We'.", + "text": "2) Metrics: We follow previous works , [38] and report strict character-level precision, recall, and F1 scores for both detection and correction.", + "citing_paper_id": "4b66b25de8c6c496f47ee58cf2677d0b377aa778", + "citing_title": "Towards More Reliable Chinese Spelling Correction: Fine-Grained Confidence Estimation Against Suboptimal Corrections", + "original_label": "Extends" + }, + { + "id": 53, + "label": "NOT_CONFIRMED", + "cue_span": "", + "rationale": "The target paper is not cited in this sentence.", + "text": "\u2022 IME-based synthetic + Human annotated The two-stage training of first using synthetic and then using human-annotated data is the widely-used and the most useful method.", + "citing_paper_id": "db2208df9148b0ddc65e21d7a0883ca54edd4e66", + "citing_title": "Evolving Chinese Spelling Correction with Corrector-Verifier Collaboration", + "original_label": "Uses" + }, + { + "id": 52, + "label": "NOT_CONFIRMED", + "cue_span": "", + "rationale": "The sentence mentions the target paper's test data but does not include an explicit citing-paper actor phrase such as 'we' or 'our'.", + "text": "This performance gap can be attributed to the domain alignment between the human-29013 annotated training data and test data, suggesting the potential benefit of incorporating human-annotated and domain-specific data in certain scenarios.", + "citing_paper_id": "db2208df9148b0ddc65e21d7a0883ca54edd4e66", + "citing_title": "Evolving Chinese Spelling Correction with Corrector-Verifier Collaboration", + "original_label": "Uses" + }, + { + "id": 13, + "label": "NOT_CONFIRMED", + "cue_span": "", + "rationale": "The target paper is not cited in this sentence.", + "text": "3) To offer deeper insights into real-world edge-side deployment, we report the prefilling speed, output token generation speed, and power consumption on the iQOO 12 smartphone using the Qual-comm QNN SDK, as shown in Tab.", + "citing_paper_id": "ce28a63f81ad7773235322506368743982fd2bc3", + "citing_title": "SmartBench: Is Your LLM Truly a Good Chinese Smartphone Assistant?", + "original_label": "Uses" + }, + { + "id": 25, + "label": "NOT_CONFIRMED", + "cue_span": "", + "rationale": "The sentence describes results on the target dataset but lacks an explicit citing-paper actor phrase such as 'we' or 'our'.", + "text": "Removing it leads to performance drops of 11.10 and 5.45 points on the and C2EC datasets, respectively.", + "citing_paper_id": "9a6479bdd7605029d3972dea4c64a2e914f2755a", + "citing_title": "A Training-free LLM-based Approach to General Chinese Character Error Correction", + "original_label": "Uses" + } + ], + "confirmed": [ + { + "id": 39, + "label": "USES", + "cue_span": "We generate the synthetic data using IME", + "rationale": "The citing paper uses the IME method from the target paper to generate synthetic data.", + "text": "We generate the synthetic data using IME .", + "citing_paper_id": "db2208df9148b0ddc65e21d7a0883ca54edd4e66", + "citing_title": "Evolving Chinese Spelling Correction with Corrector-Verifier Collaboration", + "original_label": "Uses" + }, + { + "id": 48, + "label": "USES", + "cue_span": "we use two CSC benchmarks, LEMON (Wu et al., 2023) and CSCD-NS", + "rationale": "The citing authors explicitly state they use the target paper's dataset as a benchmark.", + "text": "Following recent works, we use two CSC benchmarks, LEMON (Wu et al., 2023) and CSCD-NS .", + "citing_paper_id": "db2208df9148b0ddc65e21d7a0883ca54edd4e66", + "citing_title": "Evolving Chinese Spelling Correction with Corrector-Verifier Collaboration", + "original_label": "Uses" + }, + { + "id": 50, + "label": "USES", + "cue_span": "We compare the annotation results with the gold labels from", + "rationale": "The citing paper uses gold labels provided by the target paper for comparison.", + "text": "We compare the annotation results with the gold labels from .", + "citing_paper_id": "db2208df9148b0ddc65e21d7a0883ca54edd4e66", + "citing_title": "Evolving Chinese Spelling Correction with Corrector-Verifier Collaboration", + "original_label": "Uses" + }, + { + "id": 38, + "label": "USES", + "cue_span": "We computed Eval-CSC based on Script", + "rationale": "The citing paper uses a script provided by the target paper to compute metrics.", + "text": "Metrics We computed Eval-CSC based on Script , and implemented Eval-GCSC following the procedures outlined in Section 4.3.", + "citing_paper_id": "571754f4ed4303b02a1a5e78c6605f483b2ea6e0", + "citing_title": "Eval-GCSC: A New Metric for Evaluating ChatGPT's Performance in Chinese Spelling Correction", + "original_label": "Uses" + }, + { + "id": 5, + "label": "USES", + "cue_span": "The transformation method follows the pseudo-data construction method proposed by", + "rationale": "The citing paper adopts a specific methodology for data construction proposed by the target paper.", + "text": "The transformation method follows the pseudo-data construction method proposed by to ensure that the converted data closely resembles the actual scenario.", + "citing_paper_id": "0e2e19f1fae35d97b3178f9349caf95da540652d", + "citing_title": "Incorporating Confused Phraseological Knowledge Based on Pinyin Input Method for Chinese Spelling Correction", + "original_label": "Uses" + }, + { + "id": 18, + "label": "USES", + "cue_span": "We randomly sampled 50 sentences of varying lengths from the development set of", + "rationale": "The citing paper uses samples from the target paper's development set for evaluation.", + "text": "We randomly sampled 50 sentences of varying lengths from the development set of and C2EC to evaluate the running time of our method on Qwen2.", + "citing_paper_id": "9a6479bdd7605029d3972dea4c64a2e914f2755a", + "citing_title": "A Training-free LLM-based Approach to General Chinese Character Error Correction", + "original_label": "Uses" + }, + { + "id": 16, + "label": "USES", + "cue_span": "we use two representative datasets: CSCD-NS", + "rationale": "The citing authors explicitly state they use the target paper's dataset.", + "text": "For conventional CSC, following Li et al. (2024), we use two representative datasets: CSCD-NS and Lemon (Wu et al., 2023).", + "citing_paper_id": "9a6479bdd7605029d3972dea4c64a2e914f2755a", + "citing_title": "A Training-free LLM-based Approach to General Chinese Character Error Correction", + "original_label": "Uses" + }, + { + "id": 45, + "label": "USES", + "cue_span": "we probe the annotation accuracy of different approaches on development set", + "rationale": "The citing authors use the target paper's development set to evaluate their approaches.", + "text": "To evaluate the quality of generated training data, we probe the annotation accuracy of different approaches on development set.", + "citing_paper_id": "db2208df9148b0ddc65e21d7a0883ca54edd4e66", + "citing_title": "Evolving Chinese Spelling Correction with Corrector-Verifier Collaboration", + "original_label": "Uses" + }, + { + "id": 1, + "label": "USES", + "cue_span": "We also include", + "rationale": "The citing paper includes the target paper's dataset in its experimental setup.", + "text": "We also include a native-speaker-oriented dataset collected from social media, which contains a manually annotated subset and a large-scale pseudo-labeled corpus.", + "citing_paper_id": "4b66b25de8c6c496f47ee58cf2677d0b377aa778", + "citing_title": "Towards More Reliable Chinese Spelling Correction: Fine-Grained Confidence Estimation Against Suboptimal Corrections", + "original_label": "Uses" + }, + { + "id": 22, + "label": "USES", + "cue_span": "The SFT models were trained on a combined dataset consisting of 271k pseudo sentence pairs (Wang et al., 2018) and the training data", + "rationale": "The citing paper uses the target paper's training data to train its models.", + "text": "Training Details of SFT Baselines The SFT models were trained on a combined dataset consisting of 271k pseudo sentence pairs (Wang et al., 2018) and the training data.", + "citing_paper_id": "9a6479bdd7605029d3972dea4c64a2e914f2755a", + "citing_title": "A Training-free LLM-based Approach to General Chinese Character Error Correction", + "original_label": "Uses" + }, + { + "id": 51, + "label": "USES", + "cue_span": "The data we use is the training set of", + "rationale": "The citing authors explicitly state they use the target paper's training set.", + "text": "The data we use is the training set of , which is in high quality.", + "citing_paper_id": "db2208df9148b0ddc65e21d7a0883ca54edd4e66", + "citing_title": "Evolving Chinese Spelling Correction with Corrector-Verifier Collaboration", + "original_label": "Uses" + }, + { + "id": 6, + "label": "USES", + "cue_span": "this paper uses the Chinese pinyin IME error correction dataset CSCD-IME", + "rationale": "The citing paper explicitly states it uses the target paper's dataset for its experiments.", + "text": "To better simulate real input scenarios, this paper uses the Chinese pinyin IME error correction dataset CSCD-IME as the basic dataset for the experiments, including 3,0000 training data, 5000 validation data, and 5000 test data.", + "citing_paper_id": "0e2e19f1fae35d97b3178f9349caf95da540652d", + "citing_title": "Incorporating Confused Phraseological Knowledge Based on Pinyin Input Method for Chinese Spelling Correction", + "original_label": "Uses" + }, + { + "id": 36, + "label": "USES", + "cue_span": "We perform experiments across five distinct CSC datasets: Sighans (Wu et al., 2013; Yu et al., 2014; Tseng et al., 2015), CSCD-NS", + "rationale": "The citing authors state they use the target paper's dataset in their experiments.", + "text": "We perform experiments across five distinct CSC datasets: Sighans (Wu et al., 2013; Yu et al., 2014; Tseng et al., 2015), CSCD-NS , MCSCSet (Jiang et al., 2022), ECSpell (Lv et al., 2023), and Lemon (Wu et al., 2023), covering a broad spectrum of domains and genres.", + "citing_paper_id": "619346184ae157741fc1c4ac74447ab2455cd6f3", + "citing_title": "A Simple yet Effective Training-free Prompt-free Approach to Chinese Spelling Correction Based on Large Language Models", + "original_label": "Uses" + }, + { + "id": 4, + "label": "USES", + "cue_span": "We first aggregate data for the two task families: \u2022 CSC Datasets ( D csc ): Aggregated from W271K, Medical[11], Lemon[12], ECSpell[19], and CSCD[14]", + "rationale": "The citing paper uses data aggregated from the target paper (cited as [14]).", + "text": "We first aggregate data for the two task families: \u2022 CSC Datasets ( D csc ): Aggregated from W271K, Medical[11], Lemon[12], ECSpell[19], and CSCD[14] (totaling approx.", + "citing_paper_id": "7d64dd752bb29bb9aff8a836b4b1f5f3ccafe807", + "citing_title": "ChineseErrorCorrector3-4B: State-of-the-Art Chinese Spelling and Grammar Corrector", + "original_label": "Uses" + }, + { + "id": 40, + "label": "USES", + "cue_span": "5-14b finetuned combines the human annotated data from", + "rationale": "The citing paper's model uses data from the target paper for fine-tuning purposes.", + "text": "5-14b finetuned combines the human annotated data from and 271K pseudo-data generated by ASR or OCR as the training data, and uses character-level tokenization.", + "citing_paper_id": "db2208df9148b0ddc65e21d7a0883ca54edd4e66", + "citing_title": "Evolving Chinese Spelling Correction with Corrector-Verifier Collaboration", + "original_label": "Uses" + }, + { + "id": 11, + "label": "USES", + "cue_span": "we further conduct experiments on more diverse datasets, including LEMON (Wu et al., 2023), EC-Spell (Lv et al., 2023), CSCD-NS", + "rationale": "The citing authors use the target paper's dataset for conducting their experiments.", + "text": "Due to the lack of domain diversity in the SIGHAN datasets, we further conduct experiments on more diverse datasets, including LEMON (Wu et al., 2023), EC-Spell (Lv et al., 2023), CSCD-NS , and MCSCSet (Jiang et al., 2022), all of which were written by native speakers.", + "citing_paper_id": "87a2958bd75a72e61b5b2549a63a222d7ca2a1ec", + "citing_title": "Mixture of Small and Large Models for Chinese Spelling Check", + "original_label": "Uses" + }, + { + "id": 15, + "label": "USES", + "cue_span": "performance of our method with varying beam sizes on the ", + "rationale": "The citing paper evaluates its own method on the target paper's dataset.", + "text": "Figure 11 shows the performance of our method with varying beam sizes on the and C2EC datasets.", + "citing_paper_id": "9a6479bdd7605029d3972dea4c64a2e914f2755a", + "citing_title": "A Training-free LLM-based Approach to General Chinese Character Error Correction", + "original_label": "Uses" + }, + { + "id": 12, + "label": "USES", + "cue_span": "Table 2 presents experimental results across five benchmark datasets: rSIGHANs, ", + "rationale": "The citing paper reports results derived from testing on the target paper's benchmark dataset.", + "text": "Table 2 presents experimental results across five benchmark datasets: rSIGHANs, , MC-SCSet, ECSpell, and LEMON.", + "citing_paper_id": "87a2958bd75a72e61b5b2549a63a222d7ca2a1ec", + "citing_title": "Mixture of Small and Large Models for Chinese Spelling Check", + "original_label": "Uses" + }, + { + "id": 33, + "label": "USES", + "cue_span": "we leveraged tools from previous studies (Wu et al., 2023; ", + "rationale": "The citing paper explicitly mentions leveraging tools provided by the target paper.", + "text": "To identify these outliers, we leveraged tools from previous studies (Wu et al., 2023; by incorporating their structure confusion sets and spelling similarity matrices.", + "citing_paper_id": "619346184ae157741fc1c4ac74447ab2455cd6f3", + "citing_title": "A Simple yet Effective Training-free Prompt-free Approach to Chinese Spelling Correction Based on Large Language Models", + "original_label": "Uses" + }, + { + "id": 20, + "label": "USES", + "cue_span": "the SFT methods show superior performance on the in-domain dataset ", + "rationale": "The citing authors evaluate their SFT methods on the target paper's dataset.", + "text": "When compared to the TfPf method, the SFT methods show superior performance on the in-domain dataset .", + "citing_paper_id": "9a6479bdd7605029d3972dea4c64a2e914f2755a", + "citing_title": "A Training-free LLM-based Approach to General Chinese Character Error Correction", + "original_label": "Uses" + }, + { + "id": 3, + "label": "USES", + "cue_span": "We follow previous works ", + "rationale": "The citing paper adopts metrics from the target paper and includes the explicit actor 'We'.", + "text": "2) Metrics: We follow previous works , [38] and report strict character-level precision, recall, and F1 scores for both detection and correction.", + "citing_paper_id": "4b66b25de8c6c496f47ee58cf2677d0b377aa778", + "citing_title": "Towards More Reliable Chinese Spelling Correction: Fine-Grained Confidence Estimation Against Suboptimal Corrections", + "original_label": "Extends" + } + ], + "confirmed_extends": 0, + "confirmed_uses": 21, + "final_label": "USES" +} \ No newline at end of file diff --git a/replay_traces/2211.08788/replay_meta.json b/replay_traces/2211.08788/replay_meta.json new file mode 100644 index 0000000000000000000000000000000000000000..c20344458c7549eb54c2532dfe8aea2ee724091e --- /dev/null +++ b/replay_traces/2211.08788/replay_meta.json @@ -0,0 +1,34 @@ +{ + "label": "CSCD-NS (2022)", + "arxiv_id": "2211.08788", + "paper_input": "https://arxiv.org/abs/2211.08788", + "status": "Completed", + "built_at": "2026-07-26T21:22:24.856206+00:00", + "source_job_dir": "/Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785100600_1e6a0c3a", + "paper_dir": "processed_papers/2211.08788", + "annotation_payload_path": "two_pass_outputs/20260726T211957Z/pass_2_ui_payload.json", + "annotation_skipped_reason": null, + "pipeline_failed_reason": null, + "pipeline_stopped_reason": null, + "events": [ + "Step 1/8: Fetch metadata + LaTeX for input paper", + "Step 1 complete", + "Step 2/8: Add citation markers", + "Step 2 complete", + "Step 3/8: Build usage contexts", + "Step 3 complete", + "Step 4/8: Label citation functions", + "Step 4 complete", + "Step 5/8: Verify USES/EXTENDS", + "Step 5 complete", + "Step 6/8: Extract arXiv paragraphs", + "Step 6 complete", + "Step 7/8: Extract target contributions and refine clusters", + "Step 7 complete", + "Pipeline completed successfully.", + "[annotation] complete: /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785100600_1e6a0c3a/two_pass_outputs/2211.08788/20260726T211957Z" + ], + "cluster_count": 2, + "contribution_count": 8, + "claim_count": 1 +} \ No newline at end of file diff --git a/replay_traces/2211.08788/run_config.json b/replay_traces/2211.08788/run_config.json new file mode 100644 index 0000000000000000000000000000000000000000..27ef1c946cfd84472f6ddd9f901625896d941c54 --- /dev/null +++ b/replay_traces/2211.08788/run_config.json @@ -0,0 +1,22 @@ +{ + "paper_input": "https://arxiv.org/abs/2211.08788", + "paper_id": "2211.08788", + "arxiv_id": "2211.08788", + "source_root": "/Users/dingzifeng/SciFy-UMBC/demo_2026_07/src/processed_papers", + "steps": [ + 1, + 2, + 3, + 4, + 5, + 6, + 7, + "annotation" + ], + "llm_provider": "gemini", + "llm_model": "gemini-3.1-pro-preview", + "llm_model_step4": "gemini-3-flash-preview", + "device": "cpu", + "embedding_model": "sentence-transformers/all-mpnet-base-v2", + "timestamp": 1785100600 +} \ No newline at end of file diff --git a/replay_traces/2211.08788/scipaths_run_results.json b/replay_traces/2211.08788/scipaths_run_results.json new file mode 100644 index 0000000000000000000000000000000000000000..150b1c028e5881482300ce36ec754e511681e121 --- /dev/null +++ b/replay_traces/2211.08788/scipaths_run_results.json @@ -0,0 +1,157 @@ +{ + "citation_clusters": [ + { + "cluster_id": "C1", + "count": "1", + "representative_claim": "Uses target evaluation script for performance evaluation", + "cluster_key": "USES|evaluation_script|evaluation", + "cluster_title": "Uses target evaluation script for performance evaluation", + "claim_indices": [ + "0" + ], + "source_cluster_ids": [ + "C1" + ], + "merge_rationale": "Represents the direct use of the target paper's evaluation script to compute metrics." + }, + { + "cluster_id": "C2", + "count": "7", + "representative_claim": "Uses target dataset for model training and evaluation", + "cluster_key": "USES|dataset|experiments", + "cluster_title": "Uses target dataset for model training and evaluation", + "claim_indices": [ + "1", + "2", + "3", + "4", + "5", + "6", + "7" + ], + "source_cluster_ids": [ + "C2", + "C3", + "C4" + ], + "merge_rationale": "These clusters all describe using the target dataset (CSCD-NS) for downstream experiments, including model evaluation, training, and dataset aggregation." + } + ], + "target_contribution_decompositions": [ + { + "claim_id": "C1", + "text": "Dataset: A large-scale Chinese spelling check dataset sourced from authentic native-speaker social media posts, enabling the training and evaluation of spelling correction models on real-world error distributions.", + "rewritten_claim": "Dataset: A large-scale Chinese spelling check dataset sourced from authentic native-speaker social media posts, enabling the training and evaluation of spelling correction models on real-world error distributions.", + "cluster_id": "C2", + "decision": "YES_SUFFICIENT", + "notes": "", + "ingredients": [ + { + "ingredient_id": "C1.I1", + "ingredient": "Large-scale corpus of authentic Chinese social media posts", + "canonical_ref_id": "ref_LCSTS", + "canonical_grounding": { + "ref_id": "ref_LCSTS", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "LCSTS: A Large Scale Chinese Short Text Summarization Dataset", + "ref_year": "", + "ref_authors": "" + }, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "DATA_SOURCE", + "roles": [ + "DATA_SOURCE" + ], + "contribution": "Provides the raw, authentic native-speaker text substrate from which candidate sentences with spelling errors are extracted.", + "rationale": "To build a dataset reflecting native-speaker errors, the authors require a massive corpus of real-world input. The paper explicitly uses the LCSTS dataset (sourced from Weibo) as this foundational text substrate.", + "evidence_span": "We chose the LCSTS dataset (Hu et al., 2015) as our data source. This dataset is composed of authentic Weibo posts" + } + }, + { + "ingredient_id": "C1.I2", + "ingredient": "Pre-trained language model for automated error-detection filtering", + "canonical_ref_id": "ref_ELECTRA", + "canonical_grounding": { + "ref_id": "ref_ELECTRA", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "ELECTRA: Pre-training text encoders as discriminators rather than generators", + "ref_year": "", + "ref_authors": "" + }, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "IMPLEMENTATION_TOOLING", + "roles": [ + "IMPLEMENTATION_TOOLING" + ], + "contribution": "Provides the pre-trained discriminator architecture used to build the error detection model that filters the massive raw corpus down to a manageable candidate set.", + "rationale": "The construction pipeline requires a strong detection model to identify candidate sentences. The paper explicitly initializes this filtering model using a pre-trained Chinese ELECTRA-Large discriminator.", + "evidence_span": "use a Chinese ELECTRA-Large discriminator model (Clark et al., 2020) to initialize the detection model." + } + }, + { + "ingredient_id": "C1.I3", + "ingredient": "Human annotation and verification protocol for spelling correction", + "canonical_ref_id": "__NONE__", + "canonical_grounding": null, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "EVALUATION_PROTOCOL", + "roles": [ + "EVALUATION_PROTOCOL" + ], + "contribution": "Provides the manual verification and correction workflow that ensures the dataset contains genuine errors with unique, accurate ground-truth corrections.", + "rationale": "The dataset's reliability depends on a rigorous human annotation process. Because this specific annotation protocol is designed and executed in the target paper, it maps cleanly to NONE.", + "evidence_span": "We recruit a group of native speakers for manual annotation. The annotators are required to check whether the given sentence contains any spelling errors" + } + } + ], + "enabling_discoveries": [ + { + "ref_id": "ref_LCSTS", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "LCSTS: A Large Scale Chinese Short Text Summarization Dataset", + "ref_year": "", + "ref_authors": "", + "ingredient_id": "C1.I1", + "ingredient": "Large-scale corpus of authentic Chinese social media posts", + "role": "DATA_SOURCE", + "roles": [ + "DATA_SOURCE" + ], + "contribution": "Provides the raw, authentic native-speaker text substrate from which candidate sentences with spelling errors are extracted.", + "rationale": "To build a dataset reflecting native-speaker errors, the authors require a massive corpus of real-world input. The paper explicitly uses the LCSTS dataset (sourced from Weibo) as this foundational text substrate.", + "evidence_span": "We chose the LCSTS dataset (Hu et al., 2015) as our data source. This dataset is composed of authentic Weibo posts" + }, + { + "ref_id": "ref_ELECTRA", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "ELECTRA: Pre-training text encoders as discriminators rather than generators", + "ref_year": "", + "ref_authors": "", + "ingredient_id": "C1.I2", + "ingredient": "Pre-trained language model for automated error-detection filtering", + "role": "IMPLEMENTATION_TOOLING", + "roles": [ + "IMPLEMENTATION_TOOLING" + ], + "contribution": "Provides the pre-trained discriminator architecture used to build the error detection model that filters the massive raw corpus down to a manageable candidate set.", + "rationale": "The construction pipeline requires a strong detection model to identify candidate sentences. The paper explicitly initializes this filtering model using a pre-trained Chinese ELECTRA-Large discriminator.", + "evidence_span": "use a Chinese ELECTRA-Large discriminator model (Clark et al., 2020) to initialize the detection model." + } + ] + } + ] +} \ No newline at end of file diff --git a/replay_traces/2211.08788/summary.txt b/replay_traces/2211.08788/summary.txt new file mode 100644 index 0000000000000000000000000000000000000000..91d02a8ae8d3807e9176d9ede6b5cb56952461ca --- /dev/null +++ b/replay_traces/2211.08788/summary.txt @@ -0,0 +1,8 @@ +[1] Fetch metadata + LaTeX for input paper +[2] Add citation markers +[3] Build usage contexts +[4] Label citation functions +[5] Verify USES/EXTENDS +[6] Extract arXiv paragraphs +[7] Extract target contributions and refine clusters +SUCCESS \ No newline at end of file diff --git a/replay_traces/2211.08788/two_pass_outputs/20260726T211957Z/formatter_attempts.json b/replay_traces/2211.08788/two_pass_outputs/20260726T211957Z/formatter_attempts.json new file mode 100644 index 0000000000000000000000000000000000000000..5f6accf4d248a65ffd7bf31bd6b1a0e921c17231 --- /dev/null +++ b/replay_traces/2211.08788/two_pass_outputs/20260726T211957Z/formatter_attempts.json @@ -0,0 +1,8 @@ +{ + "attempts": [ + { + "attempt": 1, + "status": "success" + } + ] +} diff --git a/replay_traces/2211.08788/two_pass_outputs/20260726T211957Z/pass_1_reasoning.judge.output.json b/replay_traces/2211.08788/two_pass_outputs/20260726T211957Z/pass_1_reasoning.judge.output.json new file mode 100644 index 0000000000000000000000000000000000000000..2428cb990a4be9416163c9d8ffa027675e4b4d59 --- /dev/null +++ b/replay_traces/2211.08788/two_pass_outputs/20260726T211957Z/pass_1_reasoning.judge.output.json @@ -0,0 +1,25 @@ +{ + "selected_candidate_index": 0, + "selected_candidate_id": "candidate_1", + "selected_reason": "Candidate 1 best adheres to the minimal structurally sufficient objective. It correctly merges the downstream evidence into a single dataset claim and extracts a highly minimal, clean set of ingredients (raw text source, filtering tool, human annotation protocol). Most importantly, it explicitly excludes the SIGHAN datasets with the perfect rationale that they are merely training data for an intermediate helper model, avoiding the exact type of over-decomposition that the guidelines warn against.", + "candidate_scores": [ + { + "candidate_id": "candidate_1", + "candidate_index": 0, + "score": 10, + "assessment": "Excellent candidate. It correctly identifies the downstream-used dataset claim, provides a compact functional rewrite, and extracts a minimal ingredient set. The exclusion rationale for the SIGHAN datasets (recognizing them as subordinate training data for a helper tool rather than core dataset ingredients) perfectly aligns with the recall-oriented minimality rules." + }, + { + "candidate_id": "candidate_2", + "candidate_index": 1, + "score": 8, + "assessment": "Good candidate. It correctly merges the claims and makes a reasonable case for including the FASPell evaluation metric since the evaluation script is reused downstream. However, its grounding for the filtering model includes NLPTEA-2020, which is less accurate to the text than the SIGHAN/Wang data, and it misses the opportunity to explicitly exclude the helper model's training data from the structural dependencies." + }, + { + "candidate_id": "candidate_3", + "candidate_index": 2, + "score": 6, + "assessment": "Weakest candidate. While the claim identification is correct, it violates the minimality guidelines by including five additional groundings (SIGHAN13, 14, 15, Wang, ELECTRA) for a single intermediate filtering tool. Including the training datasets of a helper model inflates the ingredient list with non-structural implementation details." + } + ] +} diff --git a/replay_traces/2211.08788/two_pass_outputs/20260726T211957Z/pass_1_reasoning.judge.prompt.txt b/replay_traces/2211.08788/two_pass_outputs/20260726T211957Z/pass_1_reasoning.judge.prompt.txt new file mode 100644 index 0000000000000000000000000000000000000000..ddeb225479a77ab8059cb4c8b353411a6d4a8f68 --- /dev/null +++ b/replay_traces/2211.08788/two_pass_outputs/20260726T211957Z/pass_1_reasoning.judge.prompt.txt @@ -0,0 +1,675 @@ +Task: Compare candidate reasoning memos for a single target paper. + +You will be given: +1. target paper metadata +2. the original extracted discovery claim +3. optional downstream usage evidence +4. multiple candidate reasoning outputs from pass 1 + +Choose the candidate that best matches the minimal structurally sufficient enabling-discovery annotation objective. + +Important: +- Judge candidates using the full critic policy from the system instructions. +- Do not reward verbosity by itself. +- Do not reward confidence by itself. +- Do not reward richer implementation stories, more citations, or more operational detail unless they are structurally necessary. +- Prefer candidates that recover the smallest structurally sufficient ingredient set for each downstream-used claim. +- Prefer candidates that minimize formatter guesswork. + +Apply these priorities in order: +1. Correct downstream-used claim selection +2. Correct claim splitting and abstraction level +3. Smallest structurally sufficient ingredient set +4. Correct exclusion of tempting non-ingredients +5. Correct canonical vs additional vs "__NONE__" grounding decisions +6. Correct role assignments +7. Evidence quality and clarity +8. Overall readiness for formatter conversion + +Specific judging reminders: +- Prefer downstream-first claim identification over abstract-first or citation-first reasoning. +- Penalize candidates that annotate unused paper outputs as claims. +- Penalize candidates that oversplit one artifact into multiple claims without downstream justification. +- Penalize candidates that include one-task helpers, local preprocessing, balancing tricks, token/language-tag details, helper models for one small step, benchmark-use ingredients, model-training conveniences, or other non-structural details. +- For benchmark/dataset claims, prefer benchmark framing, source-task/source-data substrate, benchmark-wide annotation/curation/verification protocol, benchmark-defining task formulation/protocol, and any core conceptual trick or idea. +- For method/resource claims, prefer core architecture, core conceptual trick or idea, core objective(s), structurally necessary training data, and essential initialization/checkpoints. +- For finding claims, prefer the model/artifact type under study, task/benchmark substrate, evaluation/measurement protocol, and any core conceptual trick or idea needed to make the finding observable. +- Prefer "__NONE__" when an ingredient is composite, same-paper dominant, benchmark-wide/corpus-wide, or would be distorted by forcing one canonical prior study. +- Penalize candidates that force canonical studies where "__NONE__" is more faithful. +- Penalize candidates that choose famous or cited papers rather than the most structurally faithful grounding. +- Prefer role assignments that are conservative and faithful to the ingredient's function. + +TARGET PAPER METADATA: +{ + "paperId": "09d6377921f503ab660c2c2f1a961c0cefe78d1f", + "externalIds": { + "ArXiv": "2211.08788", + "DBLP": "conf/acl/HuMZ24", + "DOI": "10.18653/v1/2024.acl-long.10", + "CorpusId": 253553639 + }, + "url": "https://www.semanticscholar.org/paper/09d6377921f503ab660c2c2f1a961c0cefe78d1f", + "title": "CSCD-NS: a Chinese Spelling Check Dataset for Native Speakers", + "venue": "Annual Meeting of the Association for Computational Linguistics", + "year": 2022, + "referenceCount": 27, + "citationCount": 21, + "openAccessPdf": { + "url": "https://aclanthology.org/2024.acl-long.10.pdf", + "status": "GOLD", + "license": "CCBY", + "disclaimer": "Notice: Paper or abstract available at https://arxiv.org/abs/2211.08788, which is subject to the license by the author or copyright owner provided with this content. Please go to the source to verify the license and copyright information for your use." + }, + "tldr": { + "model": "tldr@v2.0.0", + "text": "This paper presents CSCD-NS, the first Chinese spelling check dataset designed for native speakers, containing 40,000 samples from a Chinese social platform, and proposes a novel method that simulates the input process through an input method, generating large-scale and high-quality pseudo data that closely resembles the actual error distribution and outperforms existing methods." + }, + "publicationDate": "2022-11-16", + "authors": [ + { + "authorId": "2191038395", + "name": "Yong Hu" + }, + { + "authorId": "33427918", + "name": "Fandong Meng" + }, + { + "authorId": "48128428", + "name": "Jie Zhou" + } + ], + "abstract": "In this paper, we present CSCD-NS, the first Chinese spelling check (CSC) dataset designed for native speakers, containing 40,000 samples from a Chinese social platform. Compared with existing CSC datasets aimed at Chinese learners, CSCD-NS is ten times larger in scale and exhibits a distinct error distribution, with a significantly higher proportion of word-level errors. To further enhance the data resource, we propose a novel method that simulates the input process through an input method, generating large-scale and high-quality pseudo data that closely resembles the actual error distribution and outperforms existing methods. Moreover, we investigate the performance of various models in this scenario, including large language models (LLMs), such as ChatGPT. The result indicates that generative models underperform BERT-like classification models due to strict length and pronunciation constraints. The high prevalence of word-level errors also makes CSC for native speakers challenging enough, leaving substantial room for improvement." +} + +ORIGINAL DISCOVERY CLAIM: + + +OPTIONAL DOWNSTREAM USAGE EVIDENCE: +[ + { + "cluster_id": "C1", + "representative_claim": "Uses target evaluation script for performance evaluation", + "cluster_title": "Uses target evaluation script for performance evaluation", + "count": "1", + "merge_rationale": "Represents the direct use of the target paper's evaluation script to compute metrics." + }, + { + "cluster_id": "C2", + "representative_claim": "Uses target dataset for model training and evaluation", + "cluster_title": "Uses target dataset for model training and evaluation", + "count": "7", + "merge_rationale": "These clusters all describe using the target dataset (CSCD-NS) for downstream experiments, including model evaluation, training, and dataset aggregation." + } +] + +CANDIDATE REASONING OUTPUTS: +CANDIDATE 1: candidate_1 +------------------------------------------------------------ +# 1. Cluster Evidence + +The paper introduces two main contributions: the CSCD-NS dataset (a manually annotated Chinese spelling check dataset for native speakers) and an IME-based pseudo-data generation method. + +Based on the downstream cluster evidence, reuse is heavily concentrated on the **CSCD-NS dataset** itself: +- Cluster C2 (count: 7) shows downstream work using the target dataset for model training and evaluation. +- Cluster C1 (count: 1) shows downstream work using the target paper's evaluation script to compute metrics. + +Because the evaluation script is simply the operational tooling used to benchmark models on the released dataset, these clusters reflect the reuse of a single integrated artifact. Therefore, the correct annotation is one dataset claim. + +--- + +# 2. Claim Split Decision + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C2 + +### Why this split is correct +The downstream evidence points to the reuse of the CSCD-NS dataset and its associated evaluation script. There is no distinct downstream cluster indicating independent reuse of the IME-based pseudo-data generation pipeline. Thus, the paper's contributions should be represented as a single atomic dataset claim. + +--- + +# 3. Rewritten Claims + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **rewritten_claim:** **Dataset: A large-scale Chinese spelling check dataset sourced from authentic native-speaker social media posts, enabling the training and evaluation of spelling correction models on real-world error distributions.** +- **why_this_is_atomic:** This claim isolates the released dataset and its evaluation role as the primary downstream-used artifact, without bundling the separate pseudo-data generation method that lacks independent downstream evidence. +- **optional cluster_id:** C2 +- **decision:** YES_SUFFICIENT + +--- + +# 4. Minimal Ingredient Set Per Claim + +## Claim C1: CSCD-NS Dataset + +### Ingredient C1.I1 +- **ingredient_id:** C1.I1 +- **ingredient:** Large-scale corpus of authentic Chinese social media posts +- **why_structurally_necessary:** The dataset is explicitly designed to capture the real-world spelling errors made by native speakers. Without a massive raw text substrate of authentic social media posts, the dataset could not be constructed in its claimed form. +- **why_not_lower_level_substeps:** This should remain a high-level data source ingredient rather than being split into specific post topics or sentence-splitting steps. +- **why_not_adjacent_implementation_details:** The structural dependency is the raw social media text itself, not the specific scraping or formatting details. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** LCSTS: A Large Scale Chinese Short Text Summarization Dataset +- **additional groundings if any:** none +- **role:** DATA_SOURCE +- **contribution:** Provides the raw, authentic native-speaker text substrate from which candidate sentences with spelling errors are extracted. +- **rationale:** To build a dataset reflecting native-speaker errors, the authors require a massive corpus of real-world input. The paper explicitly uses the LCSTS dataset (sourced from Weibo) as this foundational text substrate. +- **evidence_span:** “We chose the LCSTS dataset (Hu et al., 2015) as our data source. This dataset is composed of authentic Weibo posts” + +### Ingredient C1.I2 +- **ingredient_id:** C1.I2 +- **ingredient:** Pre-trained language model for automated error-detection filtering +- **why_structurally_necessary:** Finding naturally occurring spelling errors in 8 million sentences is intractable without automated filtering. The dataset relies on an initial detection model to filter out the 91.2% of sentences that are completely correct, making manual annotation feasible. +- **why_not_lower_level_substeps:** This should remain one filtering-tool ingredient rather than being split into the specific SIGHAN training datasets or the exact probability thresholds used. +- **why_not_adjacent_implementation_details:** The structural dependency is the automated filtering capability, not the specific hyperparameters of the detection model. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** ELECTRA: Pre-training text encoders as discriminators rather than generators +- **additional groundings if any:** none +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the pre-trained discriminator architecture used to build the error detection model that filters the massive raw corpus down to a manageable candidate set. +- **rationale:** The construction pipeline requires a strong detection model to identify candidate sentences. The paper explicitly initializes this filtering model using a pre-trained Chinese ELECTRA-Large discriminator. +- **evidence_span:** “use a Chinese ELECTRA-Large discriminator model (Clark et al., 2020) to initialize the detection model.” + +### Ingredient C1.I3 +- **ingredient_id:** C1.I3 +- **ingredient:** Human annotation and verification protocol for spelling correction +- **why_structurally_necessary:** The defining property of CSCD-NS is that it is a high-quality, manually annotated dataset. Without human annotators to verify errors and provide the unique correct sentences, it would merely be a noisy, machine-filtered corpus. +- **why_not_lower_level_substeps:** This should remain one protocol-level ingredient rather than being split into the specific rules for discarding ambiguous sentences or the senior annotator conflict-resolution steps. +- **why_not_adjacent_implementation_details:** The structural dependency is the human verification workflow itself, not the specific payment or interface details. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the manual verification and correction workflow that ensures the dataset contains genuine errors with unique, accurate ground-truth corrections. +- **rationale:** The dataset's reliability depends on a rigorous human annotation process. Because this specific annotation protocol is designed and executed in the target paper, it maps cleanly to `NONE`. +- **evidence_span:** “We recruit a group of native speakers for manual annotation. The annotators are required to check whether the given sentence contains any spelling errors” + +--- + +# 5. Excluded Tempting Non-Ingredients + +### Claim C1 exclusions +- **IME-based pseudo data generation method (LCSTS-IME-2M)** + - Excluded because the downstream evidence specifically points to the reuse of the manually annotated CSCD-NS dataset, not the pseudo dataset or its generation pipeline. +- **Chinese Lexical Analysis tool (LAC)** + - Excluded because it is a local heuristic used to adjust filtering thresholds for specific particles and entities, not a core structural dependency of the dataset itself. +- **SIGHAN datasets (SIGHAN13, 14, 15)** + - Excluded as a top-level ingredient because they are only used as training data for the intermediate filtering model, making them subordinate to the filtering pipeline itself rather than a direct ingredient of the final dataset. + +--- + +# 6. Candidate Grounding Decisions + +## Claim C1: CSCD-NS Dataset + +### Ingredient C1.I1 +- **Candidate:** LCSTS: A Large Scale Chinese Short Text Summarization Dataset +- **decision:** accepted_canonical +- **why:** It is the explicit and direct source of the raw Weibo posts used to construct the dataset. + +### Ingredient C1.I2 +- **Candidate:** ELECTRA: Pre-training text encoders as discriminators rather than generators +- **decision:** accepted_canonical +- **why:** It is the specific pre-trained architecture used to initialize the detection model that makes filtering the massive corpus feasible. + +### Ingredient C1.I3 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The human annotation and verification protocol is specific to this dataset and implemented entirely within the target paper. + +--- + +# 7. Final Grounding Decisions + +## Claim C1 +- **C1.I1** → LCSTS: A Large Scale Chinese Short Text Summarization Dataset → DATA_SOURCE +- **C1.I2** → ELECTRA: Pre-training text encoders as discriminators rather than generators → IMPLEMENTATION_TOOLING +- **C1.I3** → NONE → EVALUATION_PROTOCOL + +--- + +# 8. Final Rationales And Evidence + +## Claim C1: CSCD-NS Dataset + +### C1.I1 +- **ingredient:** Large-scale corpus of authentic Chinese social media posts +- **canonical study:** LCSTS: A Large Scale Chinese Short Text Summarization Dataset +- **role:** DATA_SOURCE +- **contribution:** Provides the raw, authentic native-speaker text substrate from which candidate sentences with spelling errors are extracted. +- **rationale:** To build a dataset reflecting native-speaker errors, the authors require a massive corpus of real-world input. The paper explicitly uses the LCSTS dataset (sourced from Weibo) as this foundational text substrate. +- **evidence_span:** “We chose the LCSTS dataset (Hu et al., 2015) as our data source. This dataset is composed of authentic Weibo posts” + +### C1.I2 +- **ingredient:** Pre-trained language model for automated error-detection filtering +- **canonical study:** ELECTRA: Pre-training text encoders as discriminators rather than generators +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the pre-trained discriminator architecture used to build the error detection model that filters the massive raw corpus down to a manageable candidate set. +- **rationale:** The construction pipeline requires a strong detection model to identify candidate sentences. The paper explicitly initializes this filtering model using a pre-trained Chinese ELECTRA-Large discriminator. +- **evidence_span:** “use a Chinese ELECTRA-Large discriminator model (Clark et al., 2020) to initialize the detection model.” + +### C1.I3 +- **ingredient:** Human annotation and verification protocol for spelling correction +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the manual verification and correction workflow that ensures the dataset contains genuine errors with unique, accurate ground-truth corrections. +- **rationale:** The dataset's reliability depends on a rigorous human annotation process. Because this specific annotation protocol is designed and executed in the target paper, it maps cleanly to `NONE`. +- **evidence_span:** “We recruit a group of native speakers for manual annotation. The annotators are required to check whether the given sentence contains any spelling errors” + + +CANDIDATE 2: candidate_2 +------------------------------------------------------------ +# 1. Cluster Evidence + +Based on the downstream cluster evidence, the paper introduces a primary downstream-used contribution: +- **The CSCD-NS dataset and its associated evaluation script**, which are reused by subsequent work to train and evaluate Chinese spelling check models on real-world native-speaker errors. + +While the paper also introduces an IME-based pseudo-data generation method, the downstream evidence explicitly highlights the reuse of the target dataset itself and its evaluation script. Therefore, the correct annotation is a single dataset/benchmark claim that encompasses both the data resource and its evaluation protocol. + +--- + +# 2. Claim Split Decision + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C2 + +### Why this split is correct +The downstream reuse is centered on one integrated artifact: the CSCD-NS dataset and its evaluation script. Later work uses this dataset for model training and evaluation, and relies on the provided script to compute metrics. Because these components function together as a single benchmark and training resource, they should be annotated as one atomic claim. The pseudo-data generation method is not split into a separate claim because there is no distinct cluster evidence showing its independent downstream reuse. + +--- + +# 3. Rewritten Claims + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **rewritten_claim:** **Dataset: A large-scale Chinese spelling check dataset and evaluation script sourced from social media, enabling the training and standardized evaluation of spelling correction models on real-world native-speaker errors.** +- **why_this_is_atomic:** This claim isolates the released dataset and its evaluation script as the core reusable resource. It does not bundle the separate pseudo-data generation method, which lacks independent downstream reuse evidence. +- **optional cluster_id:** C2 +- **decision:** YES_SUFFICIENT + +--- + +# 4. Minimal Ingredient Set Per Claim + +## Claim C1: CSCD-NS Dataset + +### Ingredient C1.I1 +- **ingredient_id:** C1.I1 +- **ingredient:** Large-scale authentic Chinese social media posts +- **why_structurally_necessary:** The dataset is explicitly designed to capture genuine spelling errors made by native speakers in real-world scenarios. Without a massive substrate of authentic social media text, the dataset could not be constructed in its claimed form. +- **why_not_lower_level_substeps:** This should remain a high-level data source ingredient rather than being split into specific post topics, lengths, or user demographics. +- **why_not_adjacent_implementation_details:** The structural dependency is the raw text substrate itself, not the specific API or scraping method used to originally collect it. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** LCSTS: A Large Scale Chinese Short Text Summarization Dataset +- **additional groundings if any:** none +- **role:** DATA_SOURCE +- **contribution:** Provides the raw, authentic native-speaker text substrate from which the spelling error sentences are extracted. +- **rationale:** To build a dataset of real-world native-speaker errors, the authors require a massive corpus of authentic text. The paper explicitly uses the LCSTS dataset (Weibo posts) as this foundational data source, making it the cleanest canonical grounding. +- **evidence_span:** “We chose the LCSTS dataset (Hu et al., 2015) as our data source.” + +### Ingredient C1.I2 +- **ingredient_id:** C1.I2 +- **ingredient:** Automated spelling error detection model for filtering correct sentences +- **why_structurally_necessary:** The source corpus contains over 8 million sentences, the vast majority of which are correct. Without an automated filtering mechanism to identify candidate erroneous sentences, manual annotation at this scale would be prohibitively expensive and the dataset could not be constructed. +- **why_not_lower_level_substeps:** This should remain one filtering-tool ingredient rather than being split into the specific ELECTRA architecture, the SIGHAN training data, or the Wang pseudo-data used to train it. +- **why_not_adjacent_implementation_details:** The exact threshold values and lexical analysis heuristics used during filtering are implementation details; the structural dependency is the automated detection model itself. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** ELECTRA: Pre-training text encoders as discriminators rather than generators; Overview of NLPTEA-2020 Shared Task for Chinese Grammatical Error Diagnosis +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the automated filtering mechanism necessary to reduce millions of raw sentences down to a feasible candidate set for human annotation. +- **rationale:** An automated filter is structurally required to find rare errors in a massive corpus. The paper trains an ELECTRA-based detection model following the NLPTEA-2020 setup. Because this tool is a composite of a specific architecture and prior training datasets, `NONE` is the correct canonical grounding, with the underlying models and tasks listed as additional groundings. +- **evidence_span:** “Therefore, we use an error detection model to filter out these correct sentences.” + +### Ingredient C1.I3 +- **ingredient_id:** C1.I3 +- **ingredient:** Human annotation protocol for spelling error verification and correction +- **why_structurally_necessary:** The dataset is released as a high-quality, manually verified resource. Without human annotators to confirm the errors and provide the gold-standard corrections, the dataset would merely be a noisy collection of model predictions rather than a reliable benchmark. +- **why_not_lower_level_substeps:** This should remain one protocol-level ingredient rather than being split into the specific rules for discarding ambiguous sentences or the multi-annotator voting mechanics. +- **why_not_adjacent_implementation_details:** The structural dependency is the human verification workflow itself, not the specific compensation or platform used for the annotators. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Ensures the dataset consists of verified, high-quality spelling error pairs with unique and unambiguous corrections. +- **rationale:** Human supervision is a defining property of the released dataset. Because this manual annotation and verification workflow is specific to the target paper's construction process, it maps cleanly to `NONE`. +- **evidence_span:** “The annotators are required to check whether the given sentence contains any spelling errors and provide the correct sentence.” + +### Ingredient C1.I4 +- **ingredient_id:** C1.I4 +- **ingredient:** Sentence-level detection and correction evaluation metrics +- **why_structurally_necessary:** The downstream evidence explicitly shows the reuse of the target paper's evaluation script. Without a defined metric for calculating sentence-level performance, this evaluation script would not function in its claimed form. +- **why_not_lower_level_substeps:** This should remain at the level of the evaluation metric rather than being split into precision, recall, and F1 calculations separately. +- **why_not_adjacent_implementation_details:** The structural dependency is the metric definition itself, not the specific Python code used to implement it. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** FASPell: A Fast, Adaptable, Simple, Powerful Chinese Spell Checker Based On DAE-Decoder Paradigm +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the standardized metric calculation method used in the released evaluation script. +- **rationale:** The evaluation script relies on a specific method for calculating sentence-level detection and correction metrics. The paper explicitly adopts the calculation method from FASPell, making it the cleanest canonical grounding for this ingredient. +- **evidence_span:** “For sentence-level metrics, we use the calculation method in FASPell” + +--- + +# 5. Excluded Tempting Non-Ingredients + +### Claim C1 exclusions +- **IME-based pseudo data generation method** + - Excluded because the downstream cluster evidence focuses on the reuse of the manually annotated CSCD-NS dataset and its evaluation script. The pseudo-data generation is an augmentation technique that is not shown to be independently reused as a standalone artifact. +- **Chinese Lexical Analysis tool (LAC)** + - Excluded because it is only used as a local heuristic to adjust filtering thresholds for particles and entities. It is a subordinate implementation detail of the broader automated filtering workflow, not a high-level structural ingredient. +- **SIGHAN datasets** + - Excluded as a top-level ingredient because they are only used as training data for the helper detection model, rather than serving as the source substrate for the new dataset itself. + +--- + +# 6. Candidate Grounding Decisions + +## Claim C1: CSCD-NS Dataset + +### Ingredient C1.I1 +- **Candidate:** LCSTS: A Large Scale Chinese Short Text Summarization Dataset +- **decision:** accepted_canonical +- **why:** Cleanest representative study for the massive authentic Chinese social media corpus used as the raw text substrate. + +### Ingredient C1.I2 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The automated filtering model is a composite tool built using an ELECTRA architecture and prior task setups, rather than being cleanly inherited as a single off-the-shelf artifact from one prior study. +- **Candidate:** ELECTRA: Pre-training text encoders as discriminators rather than generators +- **decision:** accepted_additional +- **why:** Provides the underlying architecture used to initialize the detection model. +- **Candidate:** Overview of NLPTEA-2020 Shared Task for Chinese Grammatical Error Diagnosis +- **decision:** accepted_additional +- **why:** Provides the training experience and setup adapted for the filtering model. + +### Ingredient C1.I3 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The human annotation and verification protocol is a paper-specific workflow designed to construct this specific dataset. + +### Ingredient C1.I4 +- **Candidate:** FASPell: A Fast, Adaptable, Simple, Powerful Chinese Spell Checker Based On DAE-Decoder Paradigm +- **decision:** accepted_canonical +- **why:** Explicitly cited as the source for the sentence-level metric calculation method used in the evaluation script. + +--- + +# 7. Final Grounding Decisions + +## Claim C1 +- **C1.I1** → LCSTS: A Large Scale Chinese Short Text Summarization Dataset → DATA_SOURCE +- **C1.I2** → NONE → IMPLEMENTATION_TOOLING + - additional: ELECTRA: Pre-training text encoders as discriminators rather than generators; Overview of NLPTEA-2020 Shared Task for Chinese Grammatical Error Diagnosis +- **C1.I3** → NONE → EVALUATION_PROTOCOL +- **C1.I4** → FASPell: A Fast, Adaptable, Simple, Powerful Chinese Spell Checker Based On DAE-Decoder Paradigm → EVALUATION_PROTOCOL + +--- + +# 8. Final Rationales And Evidence + +## Claim C1: CSCD-NS Dataset + +### C1.I1 +- **ingredient:** Large-scale authentic Chinese social media posts +- **canonical study:** LCSTS: A Large Scale Chinese Short Text Summarization Dataset +- **role:** DATA_SOURCE +- **contribution:** Provides the raw, authentic native-speaker text substrate from which the spelling error sentences are extracted. +- **rationale:** To build a dataset of real-world native-speaker errors, the authors require a massive corpus of authentic text. The paper explicitly uses the LCSTS dataset (Weibo posts) as this foundational data source, making it the cleanest canonical grounding. +- **evidence_span:** “We chose the LCSTS dataset (Hu et al., 2015) as our data source.” + +### C1.I2 +- **ingredient:** Automated spelling error detection model for filtering correct sentences +- **canonical study:** NONE +- **additional studies:** ELECTRA: Pre-training text encoders as discriminators rather than generators; Overview of NLPTEA-2020 Shared Task for Chinese Grammatical Error Diagnosis +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the automated filtering mechanism necessary to reduce millions of raw sentences down to a feasible candidate set for human annotation. +- **rationale:** An automated filter is structurally required to find rare errors in a massive corpus. The paper trains an ELECTRA-based detection model following the NLPTEA-2020 setup. Because this tool is a composite of a specific architecture and prior training datasets, `NONE` is the correct canonical grounding, with the underlying models and tasks listed as additional groundings. +- **evidence_span:** “Therefore, we use an error detection model to filter out these correct sentences.” + +### C1.I3 +- **ingredient:** Human annotation protocol for spelling error verification and correction +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Ensures the dataset consists of verified, high-quality spelling error pairs with unique and unambiguous corrections. +- **rationale:** Human supervision is a defining property of the released dataset. Because this manual annotation and verification workflow is specific to the target paper's construction process, it maps cleanly to `NONE`. +- **evidence_span:** “The annotators are required to check whether the given sentence contains any spelling errors and provide the correct sentence.” + +### C1.I4 +- **ingredient:** Sentence-level detection and correction evaluation metrics +- **canonical study:** FASPell: A Fast, Adaptable, Simple, Powerful Chinese Spell Checker Based On DAE-Decoder Paradigm +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the standardized metric calculation method used in the released evaluation script. +- **rationale:** The evaluation script relies on a specific method for calculating sentence-level detection and correction metrics. The paper explicitly adopts the calculation method from FASPell, making it the cleanest canonical grounding for this ingredient. +- **evidence_span:** “For sentence-level metrics, we use the calculation method in FASPell” + + +CANDIDATE 3: candidate_3 +------------------------------------------------------------ +# Reasoning Memo + +## 1. Cluster Evidence + +Based on the downstream cluster evidence, the paper's primary reused contribution is **CSCD-NS**, a large-scale Chinese spelling check dataset. + +- Cluster C2 (count: 7) explicitly describes using the target dataset for model training and evaluation. +- Cluster C1 (count: 1) describes using the target evaluation script for performance evaluation. + +Because the evaluation script is tightly coupled with the dataset's evaluation protocol and serves the same downstream benchmarking purpose, these clusters should be merged into a single atomic dataset claim. + +--- + +## 2. Claim Split Decision + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C2 + +### Why this split is correct +The downstream evidence points to the reuse of the dataset and its associated evaluation protocol as a unified resource for benchmarking and training Chinese spelling check models. The paper also introduces a pseudo-data generation method (LCSTS-IME-2M), but the primary downstream reuse centers on the manually annotated CSCD-NS dataset itself, so it is best represented as one dataset claim. + +--- + +## 3. Rewritten Claims + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **rewritten_claim:** **Dataset: A large-scale Chinese spelling check dataset sourced from social media and annotated for native-speaker errors, enabling the training and evaluation of spelling correction models in realistic native-input scenarios.** +- **why_this_is_atomic:** This claim isolates the released dataset and its evaluation role, which is the primary downstream-reused artifact, without over-bundling the separate pseudo-data generation methodology. +- **optional cluster_id:** C2 +- **decision:** YES_SUFFICIENT + +--- + +## 4. Minimal Ingredient Set Per Claim + +## Claim C1: CSCD-NS Dataset + +### Ingredient C1.I1 +- **ingredient_id:** C1.I1 +- **ingredient:** Large-scale corpus of authentic Chinese social media posts +- **why_structurally_necessary:** The dataset is explicitly designed to capture genuine spelling errors made by native speakers during daily input. Without a large-scale source of real-world social media text, the dataset could not be constructed in its claimed form. +- **why_not_lower_level_substeps:** This should remain a broad data-source ingredient rather than being split into specific scraping, parsing, or tokenization details. +- **why_not_adjacent_implementation_details:** The exact initial size (8 million sentences) is less important than the nature of the source corpus (authentic Weibo posts). +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** LCSTS: A Large Scale Chinese Short Text Summarization Dataset +- **additional groundings if any:** none +- **role:** DATA_SOURCE +- **contribution:** Provides the raw, authentic native-speaker text substrate from which the spelling check dataset is curated. +- **rationale:** To capture real-world native speaker errors, the authors require a massive corpus of authentic daily input. LCSTS is the exact source used and perfectly represents this structural requirement. +- **evidence_span:** “We chose the LCSTS dataset [12] as our data source. This dataset is composed of authentic Weibo posts” + +### Ingredient C1.I2 +- **ingredient_id:** C1.I2 +- **ingredient:** Error detection model trained on existing CSC datasets to filter candidate sentences +- **why_structurally_necessary:** To make manual annotation feasible, the authors had to filter millions of mostly correct sentences down to a subset likely containing errors. Without this automated filtering pipeline, constructing a dataset of this scale would be practically impossible. +- **why_not_lower_level_substeps:** This should remain a composite filtering-pipeline ingredient rather than being split into the ELECTRA initialization, SIGHAN training data, and Wang pseudo-data separately. +- **why_not_adjacent_implementation_details:** The specific thresholds used for filtering are local heuristics, not the structural dependency itself. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** ELECTRA: Pre-training text encoders as discriminators rather than generators; Chinese Spelling Check Evaluation at SIGHAN Bake-off 2013; Overview of SIGHAN 2014 Bake-off for Chinese Spelling Check; Introduction to SIGHAN 2015 Bake-off for Chinese Spelling Check; A Hybrid Approach to Automatic Corpus Generation for Chinese Spelling Check +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the automated filtering mechanism required to reduce the massive raw corpus into a manageable subset of potentially erroneous sentences for human annotation. +- **rationale:** The filtering pipeline is composite, relying on an ELECTRA architecture trained on prior CSC datasets (SIGHAN, Wang). Because no single paper represents this full pipeline, `NONE` is the correct canonical grounding, with the underlying models and datasets listed as additional groundings. +- **evidence_span:** “we use a Chinese ELECTRA-Large discriminator model to initialize the detection model. Following previous research, we train the detection model on SIGHAN13-15's training data and Wang's pseudo data” + +### Ingredient C1.I3 +- **ingredient_id:** C1.I3 +- **ingredient:** Manual annotation and verification protocol by native speakers +- **why_structurally_necessary:** The dataset provides ground-truth corrections for native-speaker errors. Without human annotation and verification to resolve ambiguities and provide correct sentences, it would just be a noisy corpus, not a reliable spelling check dataset. +- **why_not_lower_level_substeps:** This should remain a protocol-level ingredient rather than being split into the specific rules for discarding ambiguous sentences. +- **why_not_adjacent_implementation_details:** The exact number of annotators or the specific inter-annotator agreement resolution steps are implementation details of the broader human-supervision workflow. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the human-supervised curation process that establishes the ground-truth corrections for the dataset. +- **rationale:** The specific manual annotation workflow, including the rules for discarding ambiguous or complex grammatical errors, is designed and executed in the target paper itself. Therefore, it maps cleanly to `NONE`. +- **evidence_span:** “We recruit a group of native speakers for manual annotation. The annotators are required to check whether the given sentence contains any spelling errors and provide the correct sentence.” + +--- + +## 5. Excluded Tempting Non-Ingredients + +### Claim C1 exclusions +- **Chinese lexical analysis tool (LAC)** + - Excluded because it is a local heuristic used to adjust filtering thresholds for specific particles and entities, not a core structural requirement for the dataset itself. +- **IME-based pseudo-data generation method (LCSTS-IME-2M)** + - Excluded from the core dataset claim because the downstream evidence primarily supports the reuse of the manually annotated CSCD-NS dataset. The pseudo-data method is a parallel contribution rather than a prerequisite for constructing CSCD-NS. +- **Specific baseline models (BERT, PLOME, BART, etc.)** + - Excluded because they are used to evaluate the dataset in the paper, but are not required to construct the dataset artifact itself. + +--- + +## 6. Candidate Grounding Decisions + +## Claim C1: CSCD-NS Dataset + +### Ingredient C1.I1 +- **Candidate:** LCSTS: A Large Scale Chinese Short Text Summarization Dataset +- **decision:** accepted_canonical +- **why:** It is the exact source corpus used and perfectly represents the requirement for authentic Weibo posts. + +### Ingredient C1.I2 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The filtering pipeline is composite, relying on an ELECTRA architecture and prior CSC datasets. No single paper represents this full pipeline. +- **Candidate:** ELECTRA: Pre-training text encoders as discriminators rather than generators +- **decision:** accepted_additional +- **why:** Provides the base architecture for the detection model. +- **Candidate:** Chinese Spelling Check Evaluation at SIGHAN Bake-off 2013 +- **decision:** accepted_additional +- **why:** Provides part of the training data for the detection model. +- **Candidate:** Overview of SIGHAN 2014 Bake-off for Chinese Spelling Check +- **decision:** accepted_additional +- **why:** Provides part of the training data for the detection model. +- **Candidate:** Introduction to SIGHAN 2015 Bake-off for Chinese Spelling Check +- **decision:** accepted_additional +- **why:** Provides part of the training data for the detection model. +- **Candidate:** A Hybrid Approach to Automatic Corpus Generation for Chinese Spelling Check +- **decision:** accepted_additional +- **why:** Provides the pseudo-data used to train the detection model. + +### Ingredient C1.I3 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The specific manual annotation workflow is designed and executed in the target paper. + +--- + +## 7. Final Grounding Decisions + +## Claim C1 +- **C1.I1** → LCSTS: A Large Scale Chinese Short Text Summarization Dataset → DATA_SOURCE +- **C1.I2** → NONE → IMPLEMENTATION_TOOLING + - additional: ELECTRA: Pre-training text encoders as discriminators rather than generators; Chinese Spelling Check Evaluation at SIGHAN Bake-off 2013; Overview of SIGHAN 2014 Bake-off for Chinese Spelling Check; Introduction to SIGHAN 2015 Bake-off for Chinese Spelling Check; A Hybrid Approach to Automatic Corpus Generation for Chinese Spelling Check +- **C1.I3** → NONE → EVALUATION_PROTOCOL + +--- + +## 8. Final Rationales And Evidence + +## Claim C1: CSCD-NS Dataset + +### C1.I1 +- **ingredient:** Large-scale corpus of authentic Chinese social media posts +- **canonical study:** LCSTS: A Large Scale Chinese Short Text Summarization Dataset +- **role:** DATA_SOURCE +- **contribution:** Provides the raw, authentic native-speaker text substrate from which the spelling check dataset is curated. +- **rationale:** To capture real-world native speaker errors, the authors require a massive corpus of authentic daily input. LCSTS is the exact source used and perfectly represents this structural requirement. +- **evidence_span:** “We chose the LCSTS dataset [12] as our data source. This dataset is composed of authentic Weibo posts” + +### C1.I2 +- **ingredient:** Error detection model trained on existing CSC datasets to filter candidate sentences +- **canonical study:** NONE +- **additional studies:** ELECTRA: Pre-training text encoders as discriminators rather than generators; Chinese Spelling Check Evaluation at SIGHAN Bake-off 2013; Overview of SIGHAN 2014 Bake-off for Chinese Spelling Check; Introduction to SIGHAN 2015 Bake-off for Chinese Spelling Check; A Hybrid Approach to Automatic Corpus Generation for Chinese Spelling Check +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the automated filtering mechanism required to reduce the massive raw corpus into a manageable subset of potentially erroneous sentences for human annotation. +- **rationale:** The filtering pipeline is composite, relying on an ELECTRA architecture trained on prior CSC datasets (SIGHAN, Wang). Because no single paper represents this full pipeline, `NONE` is the correct canonical grounding, with the underlying models and datasets listed as additional groundings. +- **evidence_span:** “we use a Chinese ELECTRA-Large discriminator model to initialize the detection model. Following previous research, we train the detection model on SIGHAN13-15's training data and Wang's pseudo data” + +### C1.I3 +- **ingredient:** Manual annotation and verification protocol by native speakers +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the human-supervised curation process that establishes the ground-truth corrections for the dataset. +- **rationale:** The specific manual annotation workflow, including the rules for discarding ambiguous or complex grammatical errors, is designed and executed in the target paper itself. Therefore, it maps cleanly to `NONE`. +- **evidence_span:** “We recruit a group of native speakers for manual annotation. The annotators are required to check whether the given sentence contains any spelling errors and provide the correct sentence.” + + +Return only valid JSON with this exact schema: + +{ + "selected_candidate_index": 0, + "selected_candidate_id": "candidate_1", + "selected_reason": "...", + "candidate_scores": [ + { + "candidate_id": "candidate_1", + "candidate_index": 0, + "score": 1, + "assessment": "..." + } + ] +} + +Scoring rules: +- Use integer scores from 1 to 10. +- 10 = best candidate by the minimal structurally sufficient annotation objective. +- The selected candidate must have the highest score. +- In "selected_reason", explain why it best matches the downstream-first, minimal-ingredient, structurally faithful objective. +- In each "assessment", mention both strengths and weaknesses, especially over-decomposition, poor exclusions, weak abstraction level, bad role choices, and bad canonical vs "__NONE__" decisions. + +Return only valid JSON. Do not add any prose outside the JSON object. diff --git a/replay_traces/2211.08788/two_pass_outputs/20260726T211957Z/pass_1_reasoning.output.candidate_1.md b/replay_traces/2211.08788/two_pass_outputs/20260726T211957Z/pass_1_reasoning.output.candidate_1.md new file mode 100644 index 0000000000000000000000000000000000000000..20a9c79a92e9dfe17f9b02e357506e285418460f --- /dev/null +++ b/replay_traces/2211.08788/two_pass_outputs/20260726T211957Z/pass_1_reasoning.output.candidate_1.md @@ -0,0 +1,160 @@ +# 1. Cluster Evidence + +The paper introduces two main contributions: the CSCD-NS dataset (a manually annotated Chinese spelling check dataset for native speakers) and an IME-based pseudo-data generation method. + +Based on the downstream cluster evidence, reuse is heavily concentrated on the **CSCD-NS dataset** itself: +- Cluster C2 (count: 7) shows downstream work using the target dataset for model training and evaluation. +- Cluster C1 (count: 1) shows downstream work using the target paper's evaluation script to compute metrics. + +Because the evaluation script is simply the operational tooling used to benchmark models on the released dataset, these clusters reflect the reuse of a single integrated artifact. Therefore, the correct annotation is one dataset claim. + +--- + +# 2. Claim Split Decision + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C2 + +### Why this split is correct +The downstream evidence points to the reuse of the CSCD-NS dataset and its associated evaluation script. There is no distinct downstream cluster indicating independent reuse of the IME-based pseudo-data generation pipeline. Thus, the paper's contributions should be represented as a single atomic dataset claim. + +--- + +# 3. Rewritten Claims + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **rewritten_claim:** **Dataset: A large-scale Chinese spelling check dataset sourced from authentic native-speaker social media posts, enabling the training and evaluation of spelling correction models on real-world error distributions.** +- **why_this_is_atomic:** This claim isolates the released dataset and its evaluation role as the primary downstream-used artifact, without bundling the separate pseudo-data generation method that lacks independent downstream evidence. +- **optional cluster_id:** C2 +- **decision:** YES_SUFFICIENT + +--- + +# 4. Minimal Ingredient Set Per Claim + +## Claim C1: CSCD-NS Dataset + +### Ingredient C1.I1 +- **ingredient_id:** C1.I1 +- **ingredient:** Large-scale corpus of authentic Chinese social media posts +- **why_structurally_necessary:** The dataset is explicitly designed to capture the real-world spelling errors made by native speakers. Without a massive raw text substrate of authentic social media posts, the dataset could not be constructed in its claimed form. +- **why_not_lower_level_substeps:** This should remain a high-level data source ingredient rather than being split into specific post topics or sentence-splitting steps. +- **why_not_adjacent_implementation_details:** The structural dependency is the raw social media text itself, not the specific scraping or formatting details. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** LCSTS: A Large Scale Chinese Short Text Summarization Dataset +- **additional groundings if any:** none +- **role:** DATA_SOURCE +- **contribution:** Provides the raw, authentic native-speaker text substrate from which candidate sentences with spelling errors are extracted. +- **rationale:** To build a dataset reflecting native-speaker errors, the authors require a massive corpus of real-world input. The paper explicitly uses the LCSTS dataset (sourced from Weibo) as this foundational text substrate. +- **evidence_span:** “We chose the LCSTS dataset (Hu et al., 2015) as our data source. This dataset is composed of authentic Weibo posts” + +### Ingredient C1.I2 +- **ingredient_id:** C1.I2 +- **ingredient:** Pre-trained language model for automated error-detection filtering +- **why_structurally_necessary:** Finding naturally occurring spelling errors in 8 million sentences is intractable without automated filtering. The dataset relies on an initial detection model to filter out the 91.2% of sentences that are completely correct, making manual annotation feasible. +- **why_not_lower_level_substeps:** This should remain one filtering-tool ingredient rather than being split into the specific SIGHAN training datasets or the exact probability thresholds used. +- **why_not_adjacent_implementation_details:** The structural dependency is the automated filtering capability, not the specific hyperparameters of the detection model. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** ELECTRA: Pre-training text encoders as discriminators rather than generators +- **additional groundings if any:** none +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the pre-trained discriminator architecture used to build the error detection model that filters the massive raw corpus down to a manageable candidate set. +- **rationale:** The construction pipeline requires a strong detection model to identify candidate sentences. The paper explicitly initializes this filtering model using a pre-trained Chinese ELECTRA-Large discriminator. +- **evidence_span:** “use a Chinese ELECTRA-Large discriminator model (Clark et al., 2020) to initialize the detection model.” + +### Ingredient C1.I3 +- **ingredient_id:** C1.I3 +- **ingredient:** Human annotation and verification protocol for spelling correction +- **why_structurally_necessary:** The defining property of CSCD-NS is that it is a high-quality, manually annotated dataset. Without human annotators to verify errors and provide the unique correct sentences, it would merely be a noisy, machine-filtered corpus. +- **why_not_lower_level_substeps:** This should remain one protocol-level ingredient rather than being split into the specific rules for discarding ambiguous sentences or the senior annotator conflict-resolution steps. +- **why_not_adjacent_implementation_details:** The structural dependency is the human verification workflow itself, not the specific payment or interface details. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the manual verification and correction workflow that ensures the dataset contains genuine errors with unique, accurate ground-truth corrections. +- **rationale:** The dataset's reliability depends on a rigorous human annotation process. Because this specific annotation protocol is designed and executed in the target paper, it maps cleanly to `NONE`. +- **evidence_span:** “We recruit a group of native speakers for manual annotation. The annotators are required to check whether the given sentence contains any spelling errors” + +--- + +# 5. Excluded Tempting Non-Ingredients + +### Claim C1 exclusions +- **IME-based pseudo data generation method (LCSTS-IME-2M)** + - Excluded because the downstream evidence specifically points to the reuse of the manually annotated CSCD-NS dataset, not the pseudo dataset or its generation pipeline. +- **Chinese Lexical Analysis tool (LAC)** + - Excluded because it is a local heuristic used to adjust filtering thresholds for specific particles and entities, not a core structural dependency of the dataset itself. +- **SIGHAN datasets (SIGHAN13, 14, 15)** + - Excluded as a top-level ingredient because they are only used as training data for the intermediate filtering model, making them subordinate to the filtering pipeline itself rather than a direct ingredient of the final dataset. + +--- + +# 6. Candidate Grounding Decisions + +## Claim C1: CSCD-NS Dataset + +### Ingredient C1.I1 +- **Candidate:** LCSTS: A Large Scale Chinese Short Text Summarization Dataset +- **decision:** accepted_canonical +- **why:** It is the explicit and direct source of the raw Weibo posts used to construct the dataset. + +### Ingredient C1.I2 +- **Candidate:** ELECTRA: Pre-training text encoders as discriminators rather than generators +- **decision:** accepted_canonical +- **why:** It is the specific pre-trained architecture used to initialize the detection model that makes filtering the massive corpus feasible. + +### Ingredient C1.I3 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The human annotation and verification protocol is specific to this dataset and implemented entirely within the target paper. + +--- + +# 7. Final Grounding Decisions + +## Claim C1 +- **C1.I1** → LCSTS: A Large Scale Chinese Short Text Summarization Dataset → DATA_SOURCE +- **C1.I2** → ELECTRA: Pre-training text encoders as discriminators rather than generators → IMPLEMENTATION_TOOLING +- **C1.I3** → NONE → EVALUATION_PROTOCOL + +--- + +# 8. Final Rationales And Evidence + +## Claim C1: CSCD-NS Dataset + +### C1.I1 +- **ingredient:** Large-scale corpus of authentic Chinese social media posts +- **canonical study:** LCSTS: A Large Scale Chinese Short Text Summarization Dataset +- **role:** DATA_SOURCE +- **contribution:** Provides the raw, authentic native-speaker text substrate from which candidate sentences with spelling errors are extracted. +- **rationale:** To build a dataset reflecting native-speaker errors, the authors require a massive corpus of real-world input. The paper explicitly uses the LCSTS dataset (sourced from Weibo) as this foundational text substrate. +- **evidence_span:** “We chose the LCSTS dataset (Hu et al., 2015) as our data source. This dataset is composed of authentic Weibo posts” + +### C1.I2 +- **ingredient:** Pre-trained language model for automated error-detection filtering +- **canonical study:** ELECTRA: Pre-training text encoders as discriminators rather than generators +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the pre-trained discriminator architecture used to build the error detection model that filters the massive raw corpus down to a manageable candidate set. +- **rationale:** The construction pipeline requires a strong detection model to identify candidate sentences. The paper explicitly initializes this filtering model using a pre-trained Chinese ELECTRA-Large discriminator. +- **evidence_span:** “use a Chinese ELECTRA-Large discriminator model (Clark et al., 2020) to initialize the detection model.” + +### C1.I3 +- **ingredient:** Human annotation and verification protocol for spelling correction +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the manual verification and correction workflow that ensures the dataset contains genuine errors with unique, accurate ground-truth corrections. +- **rationale:** The dataset's reliability depends on a rigorous human annotation process. Because this specific annotation protocol is designed and executed in the target paper, it maps cleanly to `NONE`. +- **evidence_span:** “We recruit a group of native speakers for manual annotation. The annotators are required to check whether the given sentence contains any spelling errors” \ No newline at end of file diff --git a/replay_traces/2211.08788/two_pass_outputs/20260726T211957Z/pass_1_reasoning.output.candidate_2.md b/replay_traces/2211.08788/two_pass_outputs/20260726T211957Z/pass_1_reasoning.output.candidate_2.md new file mode 100644 index 0000000000000000000000000000000000000000..4c9cd4b53c9581a07540d15a0dd7e7fb6d6ec81a --- /dev/null +++ b/replay_traces/2211.08788/two_pass_outputs/20260726T211957Z/pass_1_reasoning.output.candidate_2.md @@ -0,0 +1,195 @@ +# 1. Cluster Evidence + +Based on the downstream cluster evidence, the paper introduces a primary downstream-used contribution: +- **The CSCD-NS dataset and its associated evaluation script**, which are reused by subsequent work to train and evaluate Chinese spelling check models on real-world native-speaker errors. + +While the paper also introduces an IME-based pseudo-data generation method, the downstream evidence explicitly highlights the reuse of the target dataset itself and its evaluation script. Therefore, the correct annotation is a single dataset/benchmark claim that encompasses both the data resource and its evaluation protocol. + +--- + +# 2. Claim Split Decision + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C2 + +### Why this split is correct +The downstream reuse is centered on one integrated artifact: the CSCD-NS dataset and its evaluation script. Later work uses this dataset for model training and evaluation, and relies on the provided script to compute metrics. Because these components function together as a single benchmark and training resource, they should be annotated as one atomic claim. The pseudo-data generation method is not split into a separate claim because there is no distinct cluster evidence showing its independent downstream reuse. + +--- + +# 3. Rewritten Claims + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **rewritten_claim:** **Dataset: A large-scale Chinese spelling check dataset and evaluation script sourced from social media, enabling the training and standardized evaluation of spelling correction models on real-world native-speaker errors.** +- **why_this_is_atomic:** This claim isolates the released dataset and its evaluation script as the core reusable resource. It does not bundle the separate pseudo-data generation method, which lacks independent downstream reuse evidence. +- **optional cluster_id:** C2 +- **decision:** YES_SUFFICIENT + +--- + +# 4. Minimal Ingredient Set Per Claim + +## Claim C1: CSCD-NS Dataset + +### Ingredient C1.I1 +- **ingredient_id:** C1.I1 +- **ingredient:** Large-scale authentic Chinese social media posts +- **why_structurally_necessary:** The dataset is explicitly designed to capture genuine spelling errors made by native speakers in real-world scenarios. Without a massive substrate of authentic social media text, the dataset could not be constructed in its claimed form. +- **why_not_lower_level_substeps:** This should remain a high-level data source ingredient rather than being split into specific post topics, lengths, or user demographics. +- **why_not_adjacent_implementation_details:** The structural dependency is the raw text substrate itself, not the specific API or scraping method used to originally collect it. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** LCSTS: A Large Scale Chinese Short Text Summarization Dataset +- **additional groundings if any:** none +- **role:** DATA_SOURCE +- **contribution:** Provides the raw, authentic native-speaker text substrate from which the spelling error sentences are extracted. +- **rationale:** To build a dataset of real-world native-speaker errors, the authors require a massive corpus of authentic text. The paper explicitly uses the LCSTS dataset (Weibo posts) as this foundational data source, making it the cleanest canonical grounding. +- **evidence_span:** “We chose the LCSTS dataset (Hu et al., 2015) as our data source.” + +### Ingredient C1.I2 +- **ingredient_id:** C1.I2 +- **ingredient:** Automated spelling error detection model for filtering correct sentences +- **why_structurally_necessary:** The source corpus contains over 8 million sentences, the vast majority of which are correct. Without an automated filtering mechanism to identify candidate erroneous sentences, manual annotation at this scale would be prohibitively expensive and the dataset could not be constructed. +- **why_not_lower_level_substeps:** This should remain one filtering-tool ingredient rather than being split into the specific ELECTRA architecture, the SIGHAN training data, or the Wang pseudo-data used to train it. +- **why_not_adjacent_implementation_details:** The exact threshold values and lexical analysis heuristics used during filtering are implementation details; the structural dependency is the automated detection model itself. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** ELECTRA: Pre-training text encoders as discriminators rather than generators; Overview of NLPTEA-2020 Shared Task for Chinese Grammatical Error Diagnosis +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the automated filtering mechanism necessary to reduce millions of raw sentences down to a feasible candidate set for human annotation. +- **rationale:** An automated filter is structurally required to find rare errors in a massive corpus. The paper trains an ELECTRA-based detection model following the NLPTEA-2020 setup. Because this tool is a composite of a specific architecture and prior training datasets, `NONE` is the correct canonical grounding, with the underlying models and tasks listed as additional groundings. +- **evidence_span:** “Therefore, we use an error detection model to filter out these correct sentences.” + +### Ingredient C1.I3 +- **ingredient_id:** C1.I3 +- **ingredient:** Human annotation protocol for spelling error verification and correction +- **why_structurally_necessary:** The dataset is released as a high-quality, manually verified resource. Without human annotators to confirm the errors and provide the gold-standard corrections, the dataset would merely be a noisy collection of model predictions rather than a reliable benchmark. +- **why_not_lower_level_substeps:** This should remain one protocol-level ingredient rather than being split into the specific rules for discarding ambiguous sentences or the multi-annotator voting mechanics. +- **why_not_adjacent_implementation_details:** The structural dependency is the human verification workflow itself, not the specific compensation or platform used for the annotators. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Ensures the dataset consists of verified, high-quality spelling error pairs with unique and unambiguous corrections. +- **rationale:** Human supervision is a defining property of the released dataset. Because this manual annotation and verification workflow is specific to the target paper's construction process, it maps cleanly to `NONE`. +- **evidence_span:** “The annotators are required to check whether the given sentence contains any spelling errors and provide the correct sentence.” + +### Ingredient C1.I4 +- **ingredient_id:** C1.I4 +- **ingredient:** Sentence-level detection and correction evaluation metrics +- **why_structurally_necessary:** The downstream evidence explicitly shows the reuse of the target paper's evaluation script. Without a defined metric for calculating sentence-level performance, this evaluation script would not function in its claimed form. +- **why_not_lower_level_substeps:** This should remain at the level of the evaluation metric rather than being split into precision, recall, and F1 calculations separately. +- **why_not_adjacent_implementation_details:** The structural dependency is the metric definition itself, not the specific Python code used to implement it. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** FASPell: A Fast, Adaptable, Simple, Powerful Chinese Spell Checker Based On DAE-Decoder Paradigm +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the standardized metric calculation method used in the released evaluation script. +- **rationale:** The evaluation script relies on a specific method for calculating sentence-level detection and correction metrics. The paper explicitly adopts the calculation method from FASPell, making it the cleanest canonical grounding for this ingredient. +- **evidence_span:** “For sentence-level metrics, we use the calculation method in FASPell” + +--- + +# 5. Excluded Tempting Non-Ingredients + +### Claim C1 exclusions +- **IME-based pseudo data generation method** + - Excluded because the downstream cluster evidence focuses on the reuse of the manually annotated CSCD-NS dataset and its evaluation script. The pseudo-data generation is an augmentation technique that is not shown to be independently reused as a standalone artifact. +- **Chinese Lexical Analysis tool (LAC)** + - Excluded because it is only used as a local heuristic to adjust filtering thresholds for particles and entities. It is a subordinate implementation detail of the broader automated filtering workflow, not a high-level structural ingredient. +- **SIGHAN datasets** + - Excluded as a top-level ingredient because they are only used as training data for the helper detection model, rather than serving as the source substrate for the new dataset itself. + +--- + +# 6. Candidate Grounding Decisions + +## Claim C1: CSCD-NS Dataset + +### Ingredient C1.I1 +- **Candidate:** LCSTS: A Large Scale Chinese Short Text Summarization Dataset +- **decision:** accepted_canonical +- **why:** Cleanest representative study for the massive authentic Chinese social media corpus used as the raw text substrate. + +### Ingredient C1.I2 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The automated filtering model is a composite tool built using an ELECTRA architecture and prior task setups, rather than being cleanly inherited as a single off-the-shelf artifact from one prior study. +- **Candidate:** ELECTRA: Pre-training text encoders as discriminators rather than generators +- **decision:** accepted_additional +- **why:** Provides the underlying architecture used to initialize the detection model. +- **Candidate:** Overview of NLPTEA-2020 Shared Task for Chinese Grammatical Error Diagnosis +- **decision:** accepted_additional +- **why:** Provides the training experience and setup adapted for the filtering model. + +### Ingredient C1.I3 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The human annotation and verification protocol is a paper-specific workflow designed to construct this specific dataset. + +### Ingredient C1.I4 +- **Candidate:** FASPell: A Fast, Adaptable, Simple, Powerful Chinese Spell Checker Based On DAE-Decoder Paradigm +- **decision:** accepted_canonical +- **why:** Explicitly cited as the source for the sentence-level metric calculation method used in the evaluation script. + +--- + +# 7. Final Grounding Decisions + +## Claim C1 +- **C1.I1** → LCSTS: A Large Scale Chinese Short Text Summarization Dataset → DATA_SOURCE +- **C1.I2** → NONE → IMPLEMENTATION_TOOLING + - additional: ELECTRA: Pre-training text encoders as discriminators rather than generators; Overview of NLPTEA-2020 Shared Task for Chinese Grammatical Error Diagnosis +- **C1.I3** → NONE → EVALUATION_PROTOCOL +- **C1.I4** → FASPell: A Fast, Adaptable, Simple, Powerful Chinese Spell Checker Based On DAE-Decoder Paradigm → EVALUATION_PROTOCOL + +--- + +# 8. Final Rationales And Evidence + +## Claim C1: CSCD-NS Dataset + +### C1.I1 +- **ingredient:** Large-scale authentic Chinese social media posts +- **canonical study:** LCSTS: A Large Scale Chinese Short Text Summarization Dataset +- **role:** DATA_SOURCE +- **contribution:** Provides the raw, authentic native-speaker text substrate from which the spelling error sentences are extracted. +- **rationale:** To build a dataset of real-world native-speaker errors, the authors require a massive corpus of authentic text. The paper explicitly uses the LCSTS dataset (Weibo posts) as this foundational data source, making it the cleanest canonical grounding. +- **evidence_span:** “We chose the LCSTS dataset (Hu et al., 2015) as our data source.” + +### C1.I2 +- **ingredient:** Automated spelling error detection model for filtering correct sentences +- **canonical study:** NONE +- **additional studies:** ELECTRA: Pre-training text encoders as discriminators rather than generators; Overview of NLPTEA-2020 Shared Task for Chinese Grammatical Error Diagnosis +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the automated filtering mechanism necessary to reduce millions of raw sentences down to a feasible candidate set for human annotation. +- **rationale:** An automated filter is structurally required to find rare errors in a massive corpus. The paper trains an ELECTRA-based detection model following the NLPTEA-2020 setup. Because this tool is a composite of a specific architecture and prior training datasets, `NONE` is the correct canonical grounding, with the underlying models and tasks listed as additional groundings. +- **evidence_span:** “Therefore, we use an error detection model to filter out these correct sentences.” + +### C1.I3 +- **ingredient:** Human annotation protocol for spelling error verification and correction +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Ensures the dataset consists of verified, high-quality spelling error pairs with unique and unambiguous corrections. +- **rationale:** Human supervision is a defining property of the released dataset. Because this manual annotation and verification workflow is specific to the target paper's construction process, it maps cleanly to `NONE`. +- **evidence_span:** “The annotators are required to check whether the given sentence contains any spelling errors and provide the correct sentence.” + +### C1.I4 +- **ingredient:** Sentence-level detection and correction evaluation metrics +- **canonical study:** FASPell: A Fast, Adaptable, Simple, Powerful Chinese Spell Checker Based On DAE-Decoder Paradigm +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the standardized metric calculation method used in the released evaluation script. +- **rationale:** The evaluation script relies on a specific method for calculating sentence-level detection and correction metrics. The paper explicitly adopts the calculation method from FASPell, making it the cleanest canonical grounding for this ingredient. +- **evidence_span:** “For sentence-level metrics, we use the calculation method in FASPell” \ No newline at end of file diff --git a/replay_traces/2211.08788/two_pass_outputs/20260726T211957Z/pass_1_reasoning.output.candidate_3.md b/replay_traces/2211.08788/two_pass_outputs/20260726T211957Z/pass_1_reasoning.output.candidate_3.md new file mode 100644 index 0000000000000000000000000000000000000000..4de9325d309ef53c2ab6cd94571e7f5a614449f5 --- /dev/null +++ b/replay_traces/2211.08788/two_pass_outputs/20260726T211957Z/pass_1_reasoning.output.candidate_3.md @@ -0,0 +1,178 @@ +# Reasoning Memo + +## 1. Cluster Evidence + +Based on the downstream cluster evidence, the paper's primary reused contribution is **CSCD-NS**, a large-scale Chinese spelling check dataset. + +- Cluster C2 (count: 7) explicitly describes using the target dataset for model training and evaluation. +- Cluster C1 (count: 1) describes using the target evaluation script for performance evaluation. + +Because the evaluation script is tightly coupled with the dataset's evaluation protocol and serves the same downstream benchmarking purpose, these clusters should be merged into a single atomic dataset claim. + +--- + +## 2. Claim Split Decision + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C2 + +### Why this split is correct +The downstream evidence points to the reuse of the dataset and its associated evaluation protocol as a unified resource for benchmarking and training Chinese spelling check models. The paper also introduces a pseudo-data generation method (LCSTS-IME-2M), but the primary downstream reuse centers on the manually annotated CSCD-NS dataset itself, so it is best represented as one dataset claim. + +--- + +## 3. Rewritten Claims + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **rewritten_claim:** **Dataset: A large-scale Chinese spelling check dataset sourced from social media and annotated for native-speaker errors, enabling the training and evaluation of spelling correction models in realistic native-input scenarios.** +- **why_this_is_atomic:** This claim isolates the released dataset and its evaluation role, which is the primary downstream-reused artifact, without over-bundling the separate pseudo-data generation methodology. +- **optional cluster_id:** C2 +- **decision:** YES_SUFFICIENT + +--- + +## 4. Minimal Ingredient Set Per Claim + +## Claim C1: CSCD-NS Dataset + +### Ingredient C1.I1 +- **ingredient_id:** C1.I1 +- **ingredient:** Large-scale corpus of authentic Chinese social media posts +- **why_structurally_necessary:** The dataset is explicitly designed to capture genuine spelling errors made by native speakers during daily input. Without a large-scale source of real-world social media text, the dataset could not be constructed in its claimed form. +- **why_not_lower_level_substeps:** This should remain a broad data-source ingredient rather than being split into specific scraping, parsing, or tokenization details. +- **why_not_adjacent_implementation_details:** The exact initial size (8 million sentences) is less important than the nature of the source corpus (authentic Weibo posts). +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** LCSTS: A Large Scale Chinese Short Text Summarization Dataset +- **additional groundings if any:** none +- **role:** DATA_SOURCE +- **contribution:** Provides the raw, authentic native-speaker text substrate from which the spelling check dataset is curated. +- **rationale:** To capture real-world native speaker errors, the authors require a massive corpus of authentic daily input. LCSTS is the exact source used and perfectly represents this structural requirement. +- **evidence_span:** “We chose the LCSTS dataset [12] as our data source. This dataset is composed of authentic Weibo posts” + +### Ingredient C1.I2 +- **ingredient_id:** C1.I2 +- **ingredient:** Error detection model trained on existing CSC datasets to filter candidate sentences +- **why_structurally_necessary:** To make manual annotation feasible, the authors had to filter millions of mostly correct sentences down to a subset likely containing errors. Without this automated filtering pipeline, constructing a dataset of this scale would be practically impossible. +- **why_not_lower_level_substeps:** This should remain a composite filtering-pipeline ingredient rather than being split into the ELECTRA initialization, SIGHAN training data, and Wang pseudo-data separately. +- **why_not_adjacent_implementation_details:** The specific thresholds used for filtering are local heuristics, not the structural dependency itself. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** ELECTRA: Pre-training text encoders as discriminators rather than generators; Chinese Spelling Check Evaluation at SIGHAN Bake-off 2013; Overview of SIGHAN 2014 Bake-off for Chinese Spelling Check; Introduction to SIGHAN 2015 Bake-off for Chinese Spelling Check; A Hybrid Approach to Automatic Corpus Generation for Chinese Spelling Check +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the automated filtering mechanism required to reduce the massive raw corpus into a manageable subset of potentially erroneous sentences for human annotation. +- **rationale:** The filtering pipeline is composite, relying on an ELECTRA architecture trained on prior CSC datasets (SIGHAN, Wang). Because no single paper represents this full pipeline, `NONE` is the correct canonical grounding, with the underlying models and datasets listed as additional groundings. +- **evidence_span:** “we use a Chinese ELECTRA-Large discriminator model to initialize the detection model. Following previous research, we train the detection model on SIGHAN13-15's training data and Wang's pseudo data” + +### Ingredient C1.I3 +- **ingredient_id:** C1.I3 +- **ingredient:** Manual annotation and verification protocol by native speakers +- **why_structurally_necessary:** The dataset provides ground-truth corrections for native-speaker errors. Without human annotation and verification to resolve ambiguities and provide correct sentences, it would just be a noisy corpus, not a reliable spelling check dataset. +- **why_not_lower_level_substeps:** This should remain a protocol-level ingredient rather than being split into the specific rules for discarding ambiguous sentences. +- **why_not_adjacent_implementation_details:** The exact number of annotators or the specific inter-annotator agreement resolution steps are implementation details of the broader human-supervision workflow. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the human-supervised curation process that establishes the ground-truth corrections for the dataset. +- **rationale:** The specific manual annotation workflow, including the rules for discarding ambiguous or complex grammatical errors, is designed and executed in the target paper itself. Therefore, it maps cleanly to `NONE`. +- **evidence_span:** “We recruit a group of native speakers for manual annotation. The annotators are required to check whether the given sentence contains any spelling errors and provide the correct sentence.” + +--- + +## 5. Excluded Tempting Non-Ingredients + +### Claim C1 exclusions +- **Chinese lexical analysis tool (LAC)** + - Excluded because it is a local heuristic used to adjust filtering thresholds for specific particles and entities, not a core structural requirement for the dataset itself. +- **IME-based pseudo-data generation method (LCSTS-IME-2M)** + - Excluded from the core dataset claim because the downstream evidence primarily supports the reuse of the manually annotated CSCD-NS dataset. The pseudo-data method is a parallel contribution rather than a prerequisite for constructing CSCD-NS. +- **Specific baseline models (BERT, PLOME, BART, etc.)** + - Excluded because they are used to evaluate the dataset in the paper, but are not required to construct the dataset artifact itself. + +--- + +## 6. Candidate Grounding Decisions + +## Claim C1: CSCD-NS Dataset + +### Ingredient C1.I1 +- **Candidate:** LCSTS: A Large Scale Chinese Short Text Summarization Dataset +- **decision:** accepted_canonical +- **why:** It is the exact source corpus used and perfectly represents the requirement for authentic Weibo posts. + +### Ingredient C1.I2 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The filtering pipeline is composite, relying on an ELECTRA architecture and prior CSC datasets. No single paper represents this full pipeline. +- **Candidate:** ELECTRA: Pre-training text encoders as discriminators rather than generators +- **decision:** accepted_additional +- **why:** Provides the base architecture for the detection model. +- **Candidate:** Chinese Spelling Check Evaluation at SIGHAN Bake-off 2013 +- **decision:** accepted_additional +- **why:** Provides part of the training data for the detection model. +- **Candidate:** Overview of SIGHAN 2014 Bake-off for Chinese Spelling Check +- **decision:** accepted_additional +- **why:** Provides part of the training data for the detection model. +- **Candidate:** Introduction to SIGHAN 2015 Bake-off for Chinese Spelling Check +- **decision:** accepted_additional +- **why:** Provides part of the training data for the detection model. +- **Candidate:** A Hybrid Approach to Automatic Corpus Generation for Chinese Spelling Check +- **decision:** accepted_additional +- **why:** Provides the pseudo-data used to train the detection model. + +### Ingredient C1.I3 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The specific manual annotation workflow is designed and executed in the target paper. + +--- + +## 7. Final Grounding Decisions + +## Claim C1 +- **C1.I1** → LCSTS: A Large Scale Chinese Short Text Summarization Dataset → DATA_SOURCE +- **C1.I2** → NONE → IMPLEMENTATION_TOOLING + - additional: ELECTRA: Pre-training text encoders as discriminators rather than generators; Chinese Spelling Check Evaluation at SIGHAN Bake-off 2013; Overview of SIGHAN 2014 Bake-off for Chinese Spelling Check; Introduction to SIGHAN 2015 Bake-off for Chinese Spelling Check; A Hybrid Approach to Automatic Corpus Generation for Chinese Spelling Check +- **C1.I3** → NONE → EVALUATION_PROTOCOL + +--- + +## 8. Final Rationales And Evidence + +## Claim C1: CSCD-NS Dataset + +### C1.I1 +- **ingredient:** Large-scale corpus of authentic Chinese social media posts +- **canonical study:** LCSTS: A Large Scale Chinese Short Text Summarization Dataset +- **role:** DATA_SOURCE +- **contribution:** Provides the raw, authentic native-speaker text substrate from which the spelling check dataset is curated. +- **rationale:** To capture real-world native speaker errors, the authors require a massive corpus of authentic daily input. LCSTS is the exact source used and perfectly represents this structural requirement. +- **evidence_span:** “We chose the LCSTS dataset [12] as our data source. This dataset is composed of authentic Weibo posts” + +### C1.I2 +- **ingredient:** Error detection model trained on existing CSC datasets to filter candidate sentences +- **canonical study:** NONE +- **additional studies:** ELECTRA: Pre-training text encoders as discriminators rather than generators; Chinese Spelling Check Evaluation at SIGHAN Bake-off 2013; Overview of SIGHAN 2014 Bake-off for Chinese Spelling Check; Introduction to SIGHAN 2015 Bake-off for Chinese Spelling Check; A Hybrid Approach to Automatic Corpus Generation for Chinese Spelling Check +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the automated filtering mechanism required to reduce the massive raw corpus into a manageable subset of potentially erroneous sentences for human annotation. +- **rationale:** The filtering pipeline is composite, relying on an ELECTRA architecture trained on prior CSC datasets (SIGHAN, Wang). Because no single paper represents this full pipeline, `NONE` is the correct canonical grounding, with the underlying models and datasets listed as additional groundings. +- **evidence_span:** “we use a Chinese ELECTRA-Large discriminator model to initialize the detection model. Following previous research, we train the detection model on SIGHAN13-15's training data and Wang's pseudo data” + +### C1.I3 +- **ingredient:** Manual annotation and verification protocol by native speakers +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the human-supervised curation process that establishes the ground-truth corrections for the dataset. +- **rationale:** The specific manual annotation workflow, including the rules for discarding ambiguous or complex grammatical errors, is designed and executed in the target paper itself. Therefore, it maps cleanly to `NONE`. +- **evidence_span:** “We recruit a group of native speakers for manual annotation. The annotators are required to check whether the given sentence contains any spelling errors and provide the correct sentence.” \ No newline at end of file diff --git a/replay_traces/2211.08788/two_pass_outputs/20260726T211957Z/pass_1_reasoning.prompt.txt b/replay_traces/2211.08788/two_pass_outputs/20260726T211957Z/pass_1_reasoning.prompt.txt new file mode 100644 index 0000000000000000000000000000000000000000..3fcbfca7f65b6c0ac764e9bd83cf53a931af7430 --- /dev/null +++ b/replay_traces/2211.08788/two_pass_outputs/20260726T211957Z/pass_1_reasoning.prompt.txt @@ -0,0 +1,2802 @@ + +Task: Produce the reasoning memo for cluster-first claim derivation and enabling-ingredient annotation. + +Important: +- Think cluster-first. +- Do not start from a paper-level discovery summary. +- Return the smallest structurally sufficient ingredient set. + +Required procedure: +1. Identify downstream-used claims from the cluster evidence. +2. Split claims only when downstream-used contributions are distinct. +3. Rewrite each claim at the abstraction level matching the intended ingredients. +4. For each claim, first list broad candidate ingredients and possible grounding studies. +5. Collapse those candidates into the minimal structurally sufficient ingredient set. +6. Explicitly exclude tempting non-ingredients. E.g. + - one-task tools + - local preprocessing + - balancing tricks + - token/language-tag tricks + - helper models used for one small step +7. Only then decide canonical vs additional vs NONE. + + +Benchmark/dataset claims: include only what is needed to build the artifact itself. Favor the benchmark framing, core conceputal idea, source tasks/data, and benchmark-wide annotation or verification protocol. Exclude anything the benchmark could still exist without. +Method/resource claims: include only what defines the artifact: core architecture, core objective, core conceputal idea, essential training data, or essential initialization. Exclude non-essential implementation details. +Finding claims: include only what is needed to observe and measure the empirical result: the evaluation protocol, core conceputal idea, studied model/artifact type, task substrate, and measurement/extraction procedure. Exclude recommendations, fixes, and general background unless they are operationally necessary. + +For each final ingredient, provide: +- ingredient +- why it is structurally necessary +- why tempting alternatives are excluded +- canonical grounding decision +- additional groundings if any +- role +- contribution +- rationale +- evidence:one verbatim quote + +Return a markdown reasoning memo only. +-------------------------------------------------- +REFERENCE EXAMPLE 1 +-------------------------------------------------- + +# EXAMPLE: Pick-a-Pic / PickScore + +## 1. Cluster Evidence + +**The PickScore metric, a CLIP-based scoring function that predicts human preferences to evaluate and rank text-to-image generation models.** + +The paper introduces two downstream-used contributions: PickScore and the Pick-a-Pic dataset. Downstream evidence shows separate reuse: + +- PickScore is reused as an evaluation metric / reward model. +- Pick-a-Pic is reused as a dataset for training and preference tuning. + +So the bundled contribution should be split into two atomic claims. + +--- + +## 2. Claim Split Decision + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Tool +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C1 + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Dataset +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C2 + +### Why this split is correct + +The two artifacts have distinct downstream roles and distinct ingredient decompositions: + +- **PickScore** is reused as a scoring / evaluation / reward model. +- **Pick-a-Pic** is reused as a preference dataset for training and benchmarking. + +They should therefore be annotated separately. + +--- + +## 3. Rewritten Claims + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Tool +- **rewritten_claim:** **"Tool: A human-preference prompt-conditioned image scorer, enabling automatic ranking of text-to-image model outputs** +- **why_this_is_atomic:** This claim isolates the scorer itself as the reused artifact. It does not bundle the dataset resource used to train it. +- **optional cluster_id:** C1 +- **decision:** YES_SUFFICIENT + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Dataset +- **rewritten_claim:** **Dataset: An open prompt-conditioned human-preference dataset for generated images, enabling training and evaluation of preference-aligned text-to-image systems.** +- **why_this_is_atomic:** This claim isolates the dataset as the released reusable resource, separate from the scoring model trained on it. +- **optional cluster_id:** C2 +- **decision:** YES_SUFFICIENT + +--- + +## 4. Minimal Ingredient Set Per Claim + +## Claim C1: PickScore + +### Ingredient C1.I1 +- **ingredient_id:** C1.I1 +- **ingredient:** Pre-trained vision language model for prompt-conditioned image scoring +- **why_structurally_necessary:** PickScore is explicitly a CLIP-based scoring function. Without a pretrained joint text-image model, the scorer would not exist in its claimed form. +- **why_not_lower_level_substeps:** This should stay at the level of pretrained vision-language initialization, not be decomposed into separate text encoder, image encoder, or projection details. +- **why_not_adjacent_implementation_details:** Specific OpenCLIP variants are narrower implementation choices rather than the core structural dependency. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Learning Transferable Visual Models From Natural Language Supervision +- **additional groundings if any:** none +- **role:** MODEL_INITIALIZATION +- **contribution:** Provides the pre-trained text and image representations that are fine-tuned into a human preference score. +- **rationale:** To score a generated image relative to a prompt, the function needs a pretrained joint text–image representation. CLIP is the cleanest representative study for such capability and is the direct architectural basis of PickScore. +- **evidence_span:** “PickScore follows the architecture of CLIP [12]” + +### Ingredient C1.I2 +- **ingredient_id:** C1.I2 +- **ingredient:** Large-scale dataset of human-preference judgements over text-to-image outputs +- **why_structurally_necessary:** The scorer must learn which of two generated images humans prefer for a given prompt. Without this training signal, it would not become a human-preference metric. +- **why_not_lower_level_substeps:** This should remain a single training-data ingredient rather than being split into prompts, pairs, ties, logging, or collection mechanics. +- **why_not_adjacent_implementation_details:** Dataset construction details belong to the dataset claim, not the scorer’s minimal training-data dependency. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** TRAINING_DATA +- **contribution:** Provides the pairwise user choices needed to fit the scorer. +- **rationale:** The model needs direct supervision about which of two generated images users prefer for a given prompt. This operative dataset is released in the target paper, so the right grounding is `NONE`. +- **evidence_span:** “We finetune CLIP-H [7] using our framework8 on the Pick-a-Pic training set.” + +### Ingredient C1.I3 +- **ingredient_id:** C1.I3 +- **ingredient:** Preference-learning objective to model the reward +- **why_structurally_necessary:** PickScore is trained to assign higher scores to preferred images. Without a preference-learning objective, it would not become a ranking metric that evaluates outputs by predicted human choice. +- **why_not_lower_level_substeps:** This should remain one core-method ingredient rather than being decomposed into loss cases, tie handling, or optimization details. +- **why_not_adjacent_implementation_details:** Training schedule and optimizer are implementation details; the structural dependency is the reward-model-style objective itself. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Training language models to follow instructions with human feedback +- **additional groundings if any:** none +- **role:** CORE_METHOD +- **contribution:** Learns scores whose differences reflect human preference between candidate images for the same prompt. +- **rationale:** Without that preference-learning step, the system would not become a ranking metric that can evaluate outputs. The paper explicitly grounds this in a variant of InstructGPT’s reward model objective. +- **evidence_span:** “We train the PickScore scoring function over Pick-a-Pic by combining a CLIP-style model with a variant of In-structGPT’s reward model objective” + +--- + +## Claim C2: Pick-a-Pic + +### Ingredient C2.I1 +- **ingredient_id:** C2.I1 +- **ingredient:** Framework for pairwise human preference collection using prompt-conditioned image comparisons +- **why_structurally_necessary:** The defining property of the dataset is that it records human preferences over generated image pairs for prompts. Without a collection framework that elicits those comparisons, the dataset would not exist in its claimed form. +- **why_not_lower_level_substeps:** This should remain one collection-framework ingredient rather than being split into web app interface details, session flow, or tie-button mechanics. +- **why_not_adjacent_implementation_details:** The structural point is the pairwise human-preference collection framework, not the exact UI implementation. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the framework to collect human preferences on a large scale. +- **rationale:** A framework is necessary to collect human preference choices at scale. The authors implement a web app that lets users generate images and specify preferences, and that workflow is a defining part of the dataset rather than a prior reusable study. +- **evidence_span:** “To address this issue, we create a web app that enables text-to-image users to generate images and specify their preferences.” + +### Ingredient C2.I2 +- **ingredient_id:** C2.I2 +- **ingredient:** Diffusion models for image generation from text +- **why_structurally_necessary:** The dataset consists of preferences over generated images. Without text-to-image generators, there would be no candidate images to compare and annotate. +- **why_not_lower_level_substeps:** This should remain at the level of text-to-image diffusion generation, not be split into individual backbone variants or guidance settings. +- **why_not_adjacent_implementation_details:** The exact backbone mix is less important than the general generation capability used to produce the images. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** High-Resolution Image Synthesis with Latent Diffusion Models +- **additional groundings if any:** none +- **role:** CORE_METHOD +- **contribution:** Provides the diffusion models needed to generate the images in the dataset. +- **rationale:** The dataset requires text-to-image generation backbones to convert user prompts into candidate images for preference annotation. Latent Diffusion Models is the cleanest representative prior study for that capability. +- **evidence_span:** “The images in the dataset were generated by employing multiple backbone models, namely, Stable Diffusion 2.1, Dreamlike Photoreal 2.0, and Stable Diffusion XL variants” + +### Ingredient C2.I3 +- **ingredient_id:** C2.I3 +- **ingredient:** Quality-control and preprocessing protocol for reliable open preference-data collection +- **why_structurally_necessary:** The dataset is released as an open, reusable human-preference resource. Without moderation, filtering, and preprocessing, it would lose one of its defining properties: being a reliable preference dataset rather than a raw interaction log. +- **why_not_lower_level_substeps:** This should remain one protocol-level ingredient rather than being split into NSFW phrase lists, account filtering, banning, or split construction separately. +- **why_not_adjacent_implementation_details:** These local measures are all parts of the same broader quality-control workflow and should not be promoted to standalone ingredients. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the moderation, filtering, and preprocessing workflow that makes the open preference dataset reliable and usable. +- **rationale:** Constructing an open preference dataset requires more than collecting raw comparisons. The paper includes moderation, harmful-content filtering, and preprocessing decisions that are structurally important to the quality of the released resource. This is paper-specific and therefore maps cleanly to `NONE`. +- **evidence_span:** “we closely monitor user activity logs and take action to ban users” + +--- + +## 5. Excluded Tempting Non-Ingredients + +### Claim C1 exclusions +- **Pick-a-Pic web app** + - Excluded because for the scorer claim, the structural dependency is the availability of human preference training data, not the full data-collection workflow. + +- **Preference-based evaluation setup for ranking candidate images by predicted human choice** + - Excluded because this is already captured by the rewritten claim and by the combination of human preference training data plus the reward-model objective. Adding it would double count the same functional role. + +### Claim C2 exclusions +- **List of NSFW phrases** + - Excluded because it is only one local component of the broader quality-control and preprocessing protocol, not a standalone structural ingredient. + +- **Specific diffusion backbone variants** + - Excluded because the structural dependency is text-to-image diffusion generation as a class, not each particular model instance. + +- **PickScore** + - Excluded because the dataset is a prerequisite for the tool, not the reverse. + +--- + +## 6. Candidate Grounding Decisions + +## Claim C1: PickScore + +### Ingredient C1.I1 +- **Candidate:** Learning Transferable Visual Models From Natural Language Supervision +- **decision:** accepted_canonical +- **why:** Provides the pretrained joint text-image architecture used as the basis for prompt-conditioned scoring. + +### Ingredient C1.I2 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The operative human preference dataset is released in the target paper itself. + +### Ingredient C1.I3 +- **Candidate:** Training language models to follow instructions with human feedback +- **decision:** accepted_canonical +- **why:** Provides the reward-model-style objective explicitly adapted by the paper. + +## Claim C2: Pick-a-Pic + +### Ingredient C2.I1 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The pairwise collection framework is implemented in the paper and not cleanly inherited from one prior study. + +### Ingredient C2.I2 +- **Candidate:** High-Resolution Image Synthesis with Latent Diffusion Models +- **decision:** accepted_canonical +- **why:** Cleanest representative of the text-to-image diffusion generation capability used to create the candidate images. + +### Ingredient C2.I3 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The quality-control and preprocessing workflow is a paper-specific protocol rather than one canonical prior study. + +--- + +## 7. Final Grounding Decisions + +## Claim C1 +- **C1.I1** → Learning Transferable Visual Models From Natural Language Supervision → MODEL_INITIALIZATION +- **C1.I2** → NONE → TRAINING_DATA +- **C1.I3** → Training language models to follow instructions with human feedback → CORE_METHOD + +## Claim C2 +- **C2.I1** → NONE → EVALUATION_PROTOCOL +- **C2.I2** → High-Resolution Image Synthesis with Latent Diffusion Models → CORE_METHOD +- **C2.I3** → NONE → EVALUATION_PROTOCOL + +--- + +## 8. Final Rationales And Evidence + +## Claim C1: PickScore + +### C1.I1 +- **ingredient:** Pre-trained vision language model for prompt-conditioned image scoring +- **canonical study:** Learning Transferable Visual Models From Natural Language Supervision +- **role:** MODEL_INITIALIZATION +- **contribution:** Provides the pre-trained text and image representations that are fine-tuned into a human preference score. +- **rationale:** To score a generated image relative to a prompt, the function needs a pretrained joint text–image representation. CLIP is the cleanest representative study for such capability and is the direct architectural basis of PickScore. +- **evidence_span:** “PickScore follows the architecture of CLIP [12]; given a prompt x and an image y, our scoring function s computes a real number by representing x using a transformer text encoder and y using a transformer image encoder as d-dimensional vectors, and returning their inner product” + +### C1.I2 +- **ingredient:** Large-scale dataset of human-preference judgements over text-to-image outputs +- **canonical study:** NONE +- **role:** TRAINING_DATA +- **contribution:** Provides the pairwise user choices needed to fit the scorer. +- **rationale:** The model needs direct supervision about which of two generated images users prefer for a given prompt. This operative dataset is released in the target paper, so the correct grounding is `NONE`. +- **evidence_span:** “We finetune CLIP-H [7] using our framework8 on the Pick-a-Pic training set.” + +### C1.I3 +- **ingredient:** Preference-learning objective to model the reward +- **canonical study:** Training language models to follow instructions with human feedback +- **role:** CORE_METHOD +- **contribution:** Learns scores whose differences reflect human preference between candidate images for the same prompt. +- **rationale:** Without that preference-learning step, the system would not become a ranking metric that can evaluate outputs. The paper explicitly grounds this in a variant of InstructGPT’s reward model objective. +- **evidence_span:** “We train the PickScore scoring function over Pick-a-Pic by combining a CLIP-style model with a variant of In-structGPT’s reward model objective [10].” + +## Claim C2: Pick-a-Pic + +### C2.I1 +- **ingredient:** Framework for pairwise human preference collection using prompt-conditioned image comparisons +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the framework to collect human preferences on a large scale. +- **rationale:** A framework is necessary to collect human preference choices at a large scale. The authors implement a web app to collect this, and that workflow is a defining part of the dataset. +- **evidence_span:** “To address this issue, we create a web app that enables text-to-image users to generate images and specify their preferences.” + +### C2.I2 +- **ingredient:** Diffusion models for image generation from text +- **canonical study:** High-Resolution Image Synthesis with Latent Diffusion Models +- **role:** CORE_METHOD +- **contribution:** Provides the diffusion models needed to generate the images in the dataset. +- **rationale:** The dataset requires text-to-image generation backbones to convert prompts into candidate images for annotation. Latent Diffusion Models is the cleanest representative prior study for that capability. +- **evidence_span:** “The images in the dataset were generated by employing multiple backbone models, namely, Stable Diffusion 2.1, Dreamlike Photoreal 2.0 5 , and Stable Diffusion XL variants [13] while sampling different classifier-free guidance scale values [6].” + +### C2.I3 +- **ingredient:** Quality-control and preprocessing protocol for reliable open preference-data collection +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the moderation, filtering, and preprocessing workflow that makes the open preference dataset reliable and usable. +- **rationale:** Constructing an open preference dataset requires more than collecting raw interactions. The released resource depends on moderation, filtering, and preprocessing steps that preserve data quality and usability. Because this workflow is paper-specific, `NONE` is the right grounding. +- **evidence_span:** “we closely monitor user activity logs and take action to ban users” + +-------------------------------------------------- +REFERENCE EXAMPLE 2 +-------------------------------------------------- + +# EXAMPLE: Offline RL bottleneck finding + +## 1. Cluster Evidence + +**Finding that policy extraction and test-time generalization (rather than just value learning) are the main bottlenecks in offline RL, enabling improved algorithm design via better policy extraction objectives.** + +This original claim identifies a real downstream-used empirical contribution. It should remain a **single finding claim** rather than being split into separate findings about value learning, policy extraction, and test-time generalization, because the paper’s contribution is one integrated diagnosis of what limits offline RL performance in practice. + +--- + +## 2. Claim Split Decision + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Finding +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C1 + +### Why this split is correct + +The downstream reuse is of a single empirical conclusion: that offline RL performance is often limited more by **policy extraction and deployment-time generalization** than by value learning alone. The later algorithm-design implication is a consequence of that diagnosis, not a separate discovery claim. + +--- + +## 3. Rewritten Claims + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Finding +- **rewritten_claim:** **Finding: In offline reinforcement learning, policy extraction and test-time policy generalization often limit performance more than value learning alone, making policy-extraction objective choice a primary determinant of final returns.** +- **why_this_is_atomic:** This claim isolates one empirical result about the source of performance bottlenecks in offline RL. It does not split off the downstream design recommendation as a separate claim. +- **optional cluster_id:** C1 +- **decision:** YES_SUFFICIENT + +--- + +## 4. Minimal Ingredient Set Per Claim + +## Claim C1: Offline RL bottleneck finding + +### Ingredient C1.I1 +- **ingredient_id:** C1.I1 +- **ingredient:** Offline RL methods with decoupled value learning and policy extraction phases +- **why_structurally_necessary:** The finding depends on separating the quality of the learned critic from the quality of the extracted policy. Without a decoupled setup, the paper could not attribute performance differences to policy extraction rather than entangled actor-critic training. +- **why_not_lower_level_substeps:** This should remain one high-level ingredient about the decoupled offline RL setup, not be split into each individual algorithm separately. +- **why_not_adjacent_implementation_details:** Specific architectural or optimization choices inside one method are not the point; the structural dependency is the existence of a method family where value learning and policy extraction can be independently analyzed. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** Offline RL Without Off-Policy Evaluation; Offline Reinforcement Learning with Implicit Q-Learning; Contrastive Learning as Goal-Conditioned Reinforcement Learning +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the offline RL setup in which value learning is trained independently of policy extraction, enabling the paper’s bottleneck analysis. +- **rationale:** To show that policy extraction rather than value learning is often the limiting factor, the paper must analyze settings where the critic can be held fixed while different extraction procedures are compared. That dependence is structurally on a family of decoupled offline RL methods rather than on one single canonical paper, so `NONE` is the faithful canonical choice and the relevant prior methods are better represented as additional studies. +- **evidence_span:** “we focus on offline RL methods with decoupled value and policy training phases” + +### Ingredient C1.I2 +- **ingredient_id:** C1.I2 +- **ingredient:** Behavior-constrained policy-gradient extraction objectives for policies learned from fixed critics +- **why_structurally_necessary:** The finding is not only that some extraction methods underperform, but that behavior-constrained policy-gradient extraction often performs better and scales better than common alternatives under the same learned critic. +- **why_not_lower_level_substeps:** This should remain one extraction-family ingredient rather than being split into every equation or policy-update detail. +- **why_not_adjacent_implementation_details:** Hyperparameters and optimizer details are irrelevant here. The structural ingredient is the extraction objective family itself. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** A Minimalist Approach to Offline Reinforcement Learning +- **additional groundings if any:** Advantage-Weighted Regression: Simple and Scalable Off-Policy Reinforcement Learning; Offline Reinforcement Learning via High-Fidelity Generative Behavior Modeling +- **role:** CORE_METHOD +- **contribution:** Provides the representative behavior-regularized policy-gradient extraction objective that the paper identifies as stronger and more scalable than common alternatives. +- **rationale:** The paper’s empirical conclusion depends on contrasting extraction families and finding that DDPG+BC-style extraction often outperforms widely used value-weighted or sampling-based alternatives. DDPG+BC is the clearest representative study for that stronger extraction family, while AWR and SfBC-style methods serve as additional comparison groundings rather than the canonical representative. +- **evidence_span:** “switching to behavior-constrained policy gradient objectives (e.g., DDPG+BC) often leads to substantial improvements” + +### Ingredient C1.I3 +- **ingredient_id:** C1.I3 +- **ingredient:** Comparative protocol that evaluates extraction objectives across data regimes and deployment generalization settings +- **why_structurally_necessary:** The claim is a comparative empirical finding about bottlenecks and scaling behavior. It requires an analysis protocol that systematically varies data properties and observes how extraction choices affect final performance and generalization. +- **why_not_lower_level_substeps:** This should remain one protocol-level ingredient rather than being split into separate dataset-size sweeps, quality sweeps, and metric definitions. +- **why_not_adjacent_implementation_details:** Individual plots, metrics, or visualization choices are not separate ingredients; they are local parts of the broader comparative protocol. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the systematic empirical protocol that isolates policy extraction and generalization bottlenecks across environments and data regimes. +- **rationale:** A finding about what the main bottleneck is cannot be established from one aggregate return table alone. The paper needs a structured comparative protocol that varies data size, quality, and coverage while comparing extraction methods under shared critics. That protocol is built in the target paper itself, so `NONE` is the correct grounding. +- **evidence_span:** “We use data size, quality, and coverage as levers for systematically controlling their impacts” + +--- + +## 5. Excluded Tempting Non-Ingredients + +### Claim C1 exclusions +- **Offline RL setting with fixed logged data and no online interaction** + - Excluded because it is too broad and field-level. The more precise structural ingredient is the **decoupled offline RL setup** that lets the paper isolate value learning from extraction. + +- **Advantage-weighted regression as its own standalone ingredient** + - Excluded as a separate final ingredient because it is better represented as an **additional grounding** under the broader comparative extraction-objective ingredient, rather than being promoted to its own parallel ingredient. + +- **Standard offline RL benchmark datasets as a separate ingredient** + - Excluded because the finding depends more centrally on the paper’s broader comparative protocol than on one benchmark substrate alone. The datasets are part of that empirical protocol rather than a standalone higher-level ingredient. + +- **Test-time generalization as a separate conceptual ingredient** + - Excluded because it is already captured within the rewritten claim and within the comparative evaluation protocol ingredient. Keeping it separate would over-decompose the finding. + +--- + +## 6. Candidate Grounding Decisions + +## Claim C1: Offline RL bottleneck finding + +### Ingredient C1.I1 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The finding relies on a family of decoupled offline RL methods rather than one single representative study. +- **Candidate:** Offline RL Without Off-Policy Evaluation +- **decision:** accepted_additional +- **why:** Representative prior decoupled offline RL method explicitly listed by the paper. +- **Candidate:** Offline Reinforcement Learning with Implicit Q-Learning +- **decision:** accepted_additional +- **why:** Another representative decoupled method used to motivate the analysis setup. +- **Candidate:** Contrastive Learning as Goal-Conditioned Reinforcement Learning +- **decision:** accepted_additional +- **why:** Included as a representative decoupled method family member rather than the single canonical grounding. + +### Ingredient C1.I2 +- **Candidate:** A Minimalist Approach to Offline Reinforcement Learning +- **decision:** accepted_canonical +- **why:** Cleanest representative study for behavior-constrained policy-gradient extraction, which the paper identifies as the stronger extraction family. +- **Candidate:** Advantage-Weighted Regression: Simple and Scalable Off-Policy Reinforcement Learning +- **decision:** accepted_additional +- **why:** Necessary comparison grounding for the weaker value-weighted extraction family contrasted in the finding. +- **Candidate:** Offline Reinforcement Learning via High-Fidelity Generative Behavior Modeling +- **decision:** accepted_additional +- **why:** Relevant additional comparison study for alternative policy extraction objectives. + +### Ingredient C1.I3 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The comparative data-scaling and deployment-generalization protocol is constructed in the target paper rather than inherited from one prior study. + +--- + +## 7. Final Grounding Decisions + +## Claim C1 +- **C1.I1** → NONE → CONCEPTUAL_FRAMEWORK + - additional: Offline RL Without Off-Policy Evaluation; Offline Reinforcement Learning with Implicit Q-Learning; Contrastive Learning as Goal-Conditioned Reinforcement Learning +- **C1.I2** → A Minimalist Approach to Offline Reinforcement Learning → CORE_METHOD + - additional: Advantage-Weighted Regression: Simple and Scalable Off-Policy Reinforcement Learning; Offline Reinforcement Learning via High-Fidelity Generative Behavior Modeling +- **C1.I3** → NONE → EVALUATION_PROTOCOL + +--- + +## 8. Final Rationales And Evidence + +## Claim C1: Offline RL bottleneck finding + +### C1.I1 +- **ingredient:** Offline RL methods with decoupled value learning and policy extraction phases +- **canonical study:** NONE +- **additional studies:** Offline RL Without Off-Policy Evaluation; Offline Reinforcement Learning with Implicit Q-Learning; Contrastive Learning as Goal-Conditioned Reinforcement Learning +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the offline RL setup in which value learning is trained independently of policy extraction, enabling the paper’s bottleneck analysis. +- **rationale:** To show that policy extraction rather than value learning is often the limiting factor, the paper must analyze settings where the critic can be held fixed while different extraction procedures are compared. That dependence is structurally on a family of decoupled offline RL methods rather than on one single canonical paper, so `NONE` is the faithful canonical choice and the relevant prior methods are better represented as additional studies. +- **evidence_span:** “we focus on offline RL methods with decoupled value and policy training phases” + +### C1.I2 +- **ingredient:** Behavior-constrained policy-gradient extraction objectives for policies learned from fixed critics +- **canonical study:** A Minimalist Approach to Offline Reinforcement Learning +- **additional studies:** Advantage-Weighted Regression: Simple and Scalable Off-Policy Reinforcement Learning; Offline Reinforcement Learning via High-Fidelity Generative Behavior Modeling +- **role:** CORE_METHOD +- **contribution:** Provides the representative behavior-regularized policy-gradient extraction objective that the paper identifies as stronger and more scalable than common alternatives. +- **rationale:** The paper’s empirical conclusion depends on contrasting extraction families and finding that DDPG+BC-style extraction often outperforms widely used value-weighted or sampling-based alternatives. DDPG+BC is the clearest representative study for that stronger extraction family, while AWR and SfBC-style methods serve as additional comparison groundings rather than the canonical representative. +- **evidence_span:** “switching to behavior-constrained policy gradient objectives (e.g., DDPG+BC) often leads to substantial improvements” + +### C1.I3 +- **ingredient:** Comparative protocol that evaluates extraction objectives across data regimes and deployment generalization settings +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the systematic empirical protocol that isolates policy extraction and generalization bottlenecks across environments and data regimes. +- **rationale:** A finding about what the main bottleneck is cannot be established from one aggregate return table alone. The paper needs a structured comparative protocol that varies data size, quality, and coverage while comparing extraction methods under shared critics. That protocol is built in the target paper itself, so `NONE` is the correct grounding. +- **evidence_span:** “We use data size, quality, and coverage as levers for systematically controlling their impacts” + + +-------------------------------------------------- +REFERENCE EXAMPLE 3 +-------------------------------------------------- + +# EXAMPLE: IndicXTREME / IndicBERT + +## 1. Cluster Evidence + +Based on downstream cluster contributions, the paper introduces two distinct downstream-used contributions: + +- a **benchmark** for multilingual zero-shot evaluation on Indic languages +- a **pretrained multilingual encoder** for Indic NLU + +Downstream evidence shows these are reused differently and should therefore be split into two atomic claims. + +--- + +## 2. Claim Split Decision + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Benchmark +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C5 + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Resource +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C2 + +### Why this split is correct + +The paper contributes two separate artifacts with distinct downstream roles: + +- **IndicXTREME** is reused as a multilingual benchmark for zero-shot evaluation. +- **IndicBERT v2** is reused as a pretrained model for transfer and downstream fine-tuning. + +These have different structural ingredients, so they should not be annotated as one merged claim. + +--- + +## 3. Rewritten Claims + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Benchmark +- **rewritten_claim:** **Benchmark: A multilingual Indic-language evaluation suite, enabling standardized zero-shot assessment of pretrained multilingual models across diverse NLU tasks.** +- **why_this_is_atomic:** This claim isolates the benchmark artifact and its evaluation role, without bundling the pretrained model. +- **optional cluster_id:** C5 +- **decision:** YES_SUFFICIENT + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Resource +- **rewritten_claim:** **Resource: A pretrained multilingual encoder for Indic languages, enabling zero-shot transfer and downstream adaptation on Indic NLU tasks.** +- **why_this_is_atomic:** This claim isolates the pretrained encoder as the reused artifact, separate from the benchmark used to evaluate such models. +- **optional cluster_id:** C2 +- **decision:** YES_SUFFICIENT + +--- + +## 4. Minimal Ingredient Set Per Claim + +## Claim C1: Indic benchmark + +### Ingredient C1.I1 +- **ingredient_id:** C1.I1 +- **ingredient:** Multilingual, multi-task, zero-shot evaluation framework for cross-lingual NLU benchmarking +- **why_structurally_necessary:** The benchmark is explicitly framed as a multilingual zero-shot evaluation suite across tasks and languages. Without this benchmark framing, it would not exist in its claimed form as a standardized cross-lingual evaluation resource. +- **why_not_lower_level_substeps:** This should remain one benchmark-paradigm ingredient rather than being split into separate benchmark papers or separate task families. +- **why_not_adjacent_implementation_details:** The structural dependency is the zero-shot multilingual benchmark framing itself, not local benchmark formatting or reporting choices. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** XTREME: A Massively Multilingual Multi-task Benchmark for Evaluating Cross-lingual Generalization +- **additional groundings if any:** XTREME-R: Towards More Challenging and Nuanced Multilingual Evaluation; XGLUE: A New Benchmark Dataset for Cross-lingual Pre-training, Understanding and Generation +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the multilingual zero-shot benchmark framing that the paper adapts to Indic languages. +- **rationale:** To construct a benchmark whose point is standardized zero-shot evaluation across many languages and tasks, the paper depends on an existing multilingual benchmark paradigm. XTREME is the cleanest canonical grounding because it most directly established that framing; XTREME-R and XGLUE are relevant supporting studies but are less direct representatives. +- **evidence_span:** “aims to test the multilingual zero-shot capabilities of pretrained language models” + +### Ingredient C1.I2 +- **ingredient_id:** C1.I2 +- **ingredient:** NLU task datasets and source resources used to assemble the benchmark +- **why_structurally_necessary:** A multi-task benchmark requires the actual upstream task substrates from which its evaluation sets are translated, adapted, corrected, or incorporated. +- **why_not_lower_level_substeps:** This should remain one composite data-source ingredient rather than being split into separate ingredients for each task dataset. +- **why_not_adjacent_implementation_details:** The key dependency is the heterogeneous task substrate itself, not each local adaptation step used for one task. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** Choice of Plausible Alternatives: An Evaluation of Commonsense Causal Reasoning; XNLI: Evaluating Cross-lingual Sentence Representations; IndicXNLI: Evaluating Multilingual Inference for Indian Languages; Naamapadam: A Large-Scale Named Entity Annotated Data for Indic Languages; The Flores-101 Evaluation Benchmark for Low-Resource and Multilingual Machine Translation; MASSIVE: A 1M-Example Multilingual Natural Language Understanding Dataset with 51 Typologically-Diverse Languages +- **role:** DATA_SOURCE +- **contribution:** Supplies the task definitions and raw/source evaluation material that the benchmark translates, subsets, verifies, or incorporates. +- **rationale:** Constructing a multi-task benchmark requires the underlying task substrates that define its evaluation sets. Because these substrates are inherently composite across multiple tasks and datasets, they do not map cleanly to a single prior study, making `NONE` the appropriate canonical choice, with the task-specific sources listed as additional studies. +- **evidence_span:** “We manually translate the COPA test set into 18 Indic languages to create IndicCOPA.” + +### Ingredient C1.I3 +- **ingredient_id:** C1.I3 +- **ingredient:** Human-supervised multilingual curation, translation, and verification protocol +- **why_structurally_necessary:** Human supervision is one of the benchmark’s defining claimed properties. Without it, the benchmark would not exist in the same form as a human-validated evaluation resource. +- **why_not_lower_level_substeps:** This should remain one protocol-level ingredient rather than being split into manual translation, verification, correction, and annotation as separate ingredients. +- **why_not_adjacent_implementation_details:** The structural dependency is the benchmark-wide human-supervision workflow, not each local annotation step. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the manual benchmark-construction process ensuring that all included evaluation sets are created, translated, edited, or verified by humans. +- **rationale:** This ingredient is necessary because the benchmark’s defining property is that all evaluation sets are created or verified under human supervision. Removing it would fundamentally change the nature of the benchmark, reducing it to a weaker, non-human-validated collection of datasets. Because this workflow is benchmark-specific and realized in the target paper, `NONE` is the correct grounding. +- **evidence_span:** “ALL the evaluation sets included in IndicXTREME were created with human supervision” + +--- + +## Claim C2: Indic multilingual encoder + +### Ingredient C2.I1 +- **ingredient_id:** C2.I1 +- **ingredient:** Transformer-based masked language modeling architecture +- **why_structurally_necessary:** The resource is explicitly a BERT-style pretrained encoder. Without the transformer MLM architecture, it would not exist in its claimed form. +- **why_not_lower_level_substeps:** This should remain one architecture-level ingredient rather than being split into encoder layers, attention heads, or hyperparameter choices. +- **why_not_adjacent_implementation_details:** Model size, vocabulary size, and optimizer settings are implementation details; the structural dependency is the BERT-style MLM architecture itself. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding +- **additional groundings if any:** none +- **role:** CORE_METHOD +- **contribution:** Provides the transformer encoder architecture and masked language modeling objective used to pretrain the model. +- **rationale:** To build a multilingual BERT-style encoder, the paper requires a transformer architecture trained with masked language modeling. BERT is the clearest and most faithful canonical grounding for that dependency. +- **evidence_span:** “We use the default hyperparameters of BERT-Base” + +### Ingredient C2.I2 +- **ingredient_id:** C2.I2 +- **ingredient:** Translation Language Modeling objective for cross-lingual alignment +- **why_structurally_necessary:** The paper explicitly uses TLM as a second core objective to improve cross-lingual transfer by aligning languages through parallel data. +- **why_not_lower_level_substeps:** This should remain one objective-level ingredient rather than being split into masking details, parallel-pair formatting, or loss implementation details. +- **why_not_adjacent_implementation_details:** The structural dependency is the TLM objective itself, not specific data-preparation or training-engineering choices around it. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Cross-lingual Language Model Pretraining +- **additional groundings if any:** none +- **role:** CORE_METHOD +- **contribution:** Provides the cross-lingual pretraining objective used to align representations across languages. +- **rationale:** The model is not only trained with MLM; it explicitly incorporates TLM as a core objective for cross-lingual alignment. That objective maps cleanly to XLM as the canonical prior study. +- **evidence_span:** “Translation Language Modeling (Conneau and Lample, 2019, TLM)” + +### Ingredient C2.I3 +- **ingredient_id:** C2.I3 +- **ingredient:** Large-scale Indic-language pretraining text +- **why_structurally_necessary:** A multilingual encoder for Indic languages cannot be pretrained in its claimed form without a large-scale Indic-language text substrate. +- **why_not_lower_level_substeps:** This should remain one training-data ingredient rather than being split into IndicCorp, Wikipedia, OSCAR, Samanantar-derived text, or synthetic translations separately. +- **why_not_adjacent_implementation_details:** Specific corpus components and augmentation choices are subordinate pieces of the broader pretraining-text ingredient and should not be promoted to standalone structural ingredients. +- **necessary:** true +- **from_prior_work:** partially +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** IndicNLPSuite: Monolingual Corpora, Evaluation Benchmarks and Pre-trained Multilingual Language Models for Indian Languages; Samanantar: The Largest Publicly Available Parallel Corpora Collection for 11 Indic Languages +- **role:** TRAINING_DATA +- **contribution:** Provides the large-scale Indic-language text substrate used to pretrain the multilingual encoder. +- **rationale:** Recreating the model requires the actual large-scale Indic-language training text. But the operative pretraining text is structurally composite: it is dominated by corpus material created or assembled in the target paper, while also incorporating prior resources such as IndicCorp lineage and Samanantar-derived text. Because no single prior study cleanly represents that full training substrate, `NONE` is the correct canonical choice, with prior contributing resources listed as additional studies. +- **evidence_span:** “we merge data from IndicCorp v2 with Indic language data from Wikipedia and OSCAR” + +--- + +## 5. Excluded Tempting Non-Ingredients + +### Claim C1 exclusions +- **Machine translation model for auto-translating some source datasets** + - Excluded because it is not structural to the benchmark as a whole and is only used to help construct a small part of one task family rather than defining the benchmark itself. + +- **English training datasets used for zero-shot evaluation** + - Excluded because these are used to apply the benchmark in evaluation setups, not to construct the benchmark artifact itself. + +- **Individual task datasets as separate top-level ingredients** + - Excluded because the benchmark’s task substrate is better represented as one composite DATA_SOURCE ingredient rather than over-decomposed into one ingredient per task. + +### Claim C2 exclusions +- **IndicTrans as a standalone ingredient** + - Excluded because it is a helper tool for generating some synthetic parallel data, not a core structural ingredient of the model in its claimed form. + +- **Temperature-based upsampling / balancing tricks** + - Excluded because these are recipe-level training choices rather than artifact-defining ingredients. + +- **Language ID tokens / tokenizer / WordPiece design** + - Excluded because these are implementation-level configuration details, not part of the minimal structurally sufficient ingredient set. + +- **Samanantar as the canonical grounding for the full training-text ingredient** + - Excluded because it represents only one smaller component of the broader pretraining substrate; assigning it as canonical would distort the true structure of the ingredient. + +--- + +## 6. Candidate Grounding Decisions + +## Claim C1: Indic benchmark + +### Ingredient C1.I1 +- **Candidate:** XTREME: A Massively Multilingual Multi-task Benchmark for Evaluating Cross-lingual Generalization +- **decision:** accepted_canonical +- **why:** Cleanest representative study for the multilingual multi-task zero-shot benchmark framing adapted here. +- **Candidate:** XTREME-R: Towards More Challenging and Nuanced Multilingual Evaluation +- **decision:** accepted_additional +- **why:** Important related multilingual benchmark extension, but not the clearest canonical origin. +- **Candidate:** XGLUE: A New Benchmark Dataset for Cross-lingual Pre-training, Understanding and Generation +- **decision:** accepted_additional +- **why:** Relevant supporting multilingual benchmark study, but less direct as the primary framing. + +### Ingredient C1.I2 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The task substrate is structurally composite across multiple datasets and source resources. +- **Candidate:** Choice of Plausible Alternatives: An Evaluation of Commonsense Causal Reasoning +- **decision:** accepted_additional +- **why:** Source task for IndicCOPA. +- **Candidate:** XNLI: Evaluating Cross-lingual Sentence Representations +- **decision:** accepted_additional +- **why:** Important upstream NLI source task. +- **Candidate:** IndicXNLI: Evaluating Multilingual Inference for Indian Languages +- **decision:** accepted_additional +- **why:** Important Indic-specific benchmark source. +- **Candidate:** Naamapadam: A Large-Scale Named Entity Annotated Data for Indic Languages +- **decision:** accepted_additional +- **why:** Upstream NER task resource. +- **Candidate:** The Flores-101 Evaluation Benchmark for Low-Resource and Multilingual Machine Translation +- **decision:** accepted_additional +- **why:** Upstream retrieval/evaluation source resource. +- **Candidate:** MASSIVE: A 1M-Example Multilingual Natural Language Understanding Dataset with 51 Typologically-Diverse Languages +- **decision:** accepted_additional +- **why:** Upstream intent/slot resource. + +### Ingredient C1.I3 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The human-supervised curation and verification workflow is benchmark-specific and realized in the target paper. + +## Claim C2: Indic multilingual encoder + +### Ingredient C2.I1 +- **Candidate:** BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding +- **decision:** accepted_canonical +- **why:** Cleanest representative study for the transformer MLM architecture used by the model. + +### Ingredient C2.I2 +- **Candidate:** Cross-lingual Language Model Pretraining +- **decision:** accepted_canonical +- **why:** Cleanest representative study for the TLM objective explicitly used by the paper. + +### Ingredient C2.I3 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The pretraining text is dominated by same-paper corpus construction and is structurally composite across several sources. +- **Candidate:** IndicNLPSuite: Monolingual Corpora, Evaluation Benchmarks and Pre-trained Multilingual Language Models for Indian Languages +- **decision:** accepted_additional +- **why:** Important prior Indic corpus lineage that the paper extends. +- **Candidate:** Samanantar: The Largest Publicly Available Parallel Corpora Collection for 11 Indic Languages +- **decision:** accepted_additional +- **why:** Important contributing prior text source, but not representative enough to be canonical for the full pretraining-text ingredient. + +--- + +## 7. Final Grounding Decisions + +## Claim C1 +- **C1.I1** → XTREME: A Massively Multilingual Multi-task Benchmark for Evaluating Cross-lingual Generalization → CONCEPTUAL_FRAMEWORK + - additional: XTREME-R: Towards More Challenging and Nuanced Multilingual Evaluation; XGLUE: A New Benchmark Dataset for Cross-lingual Pre-training, Understanding and Generation +- **C1.I2** → NONE → DATA_SOURCE + - additional: Choice of Plausible Alternatives: An Evaluation of Commonsense Causal Reasoning; XNLI: Evaluating Cross-lingual Sentence Representations; IndicXNLI: Evaluating Multilingual Inference for Indian Languages; Naamapadam: A Large-Scale Named Entity Annotated Data for Indic Languages; The Flores-101 Evaluation Benchmark for Low-Resource and Multilingual Machine Translation; MASSIVE: A 1M-Example Multilingual Natural Language Understanding Dataset with 51 Typologically-Diverse Languages +- **C1.I3** → NONE → EVALUATION_PROTOCOL + +## Claim C2 +- **C2.I1** → BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding → CORE_METHOD +- **C2.I2** → Cross-lingual Language Model Pretraining → CORE_METHOD +- **C2.I3** → NONE → TRAINING_DATA + - additional: IndicNLPSuite: Monolingual Corpora, Evaluation Benchmarks and Pre-trained Multilingual Language Models for Indian Languages; Samanantar: The Largest Publicly Available Parallel Corpora Collection for 11 Indic Languages + +--- + +## 8. Final Rationales And Evidence + +## Claim C1: Indic benchmark + +### C1.I1 +- **ingredient:** Multilingual, multi-task, zero-shot evaluation framework for cross-lingual NLU benchmarking +- **canonical study:** XTREME: A Massively Multilingual Multi-task Benchmark for Evaluating Cross-lingual Generalization +- **additional studies:** XTREME-R: Towards More Challenging and Nuanced Multilingual Evaluation; XGLUE: A New Benchmark Dataset for Cross-lingual Pre-training, Understanding and Generation +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the multilingual zero-shot benchmark framing that the paper adapts to Indic languages. +- **rationale:** To construct a benchmark whose point is standardized zero-shot evaluation across many languages and tasks, the paper depends on an existing multilingual benchmark paradigm. XTREME is the cleanest canonical grounding because it most directly established that framing; XTREME-R and XGLUE are relevant supporting studies but are less direct representatives. +- **evidence_span:** “aims to test the multilingual zero-shot capabilities of pretrained language models” + +### C1.I2 +- **ingredient:** NLU task datasets and source resources used to assemble the benchmark +- **canonical study:** NONE +- **additional studies:** Choice of Plausible Alternatives: An Evaluation of Commonsense Causal Reasoning; XNLI: Evaluating Cross-lingual Sentence Representations; IndicXNLI: Evaluating Multilingual Inference for Indian Languages; Naamapadam: A Large-Scale Named Entity Annotated Data for Indic Languages; The Flores-101 Evaluation Benchmark for Low-Resource and Multilingual Machine Translation; MASSIVE: A 1M-Example Multilingual Natural Language Understanding Dataset with 51 Typologically-Diverse Languages +- **role:** DATA_SOURCE +- **contribution:** Supplies the task definitions and raw/source evaluation material that the benchmark translates, subsets, verifies, or incorporates. +- **rationale:** Constructing a multi-task benchmark requires the underlying task substrates that define its evaluation sets. Because these substrates are inherently composite across multiple tasks and datasets, they do not map cleanly to a single prior study, making `NONE` the appropriate canonical choice, with the task-specific sources listed as additional studies. +- **evidence_span:** “We manually translate the COPA test set into 18 Indic languages to create IndicCOPA.” + +### C1.I3 +- **ingredient:** Human-supervised multilingual curation, translation, and verification protocol +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the manual benchmark-construction process ensuring that all included evaluation sets are created, translated, edited, or verified by humans. +- **rationale:** This ingredient is necessary because the benchmark’s defining property is that all evaluation sets are created or verified under human supervision. Removing it would fundamentally change the nature of the benchmark, reducing it to a weaker, non-human-validated collection of datasets. Because this workflow is benchmark-specific and realized in the target paper, `NONE` is the correct grounding. +- **evidence_span:** “ALL the evaluation sets included in IndicXTREME were created with human supervision” + +## Claim C2: Indic multilingual encoder + +### C2.I1 +- **ingredient:** Transformer-based masked language modeling architecture +- **canonical study:** BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding +- **role:** CORE_METHOD +- **contribution:** Provides the transformer encoder architecture and masked language modeling objective used to pretrain the model. +- **rationale:** To build a multilingual BERT-style encoder, the paper requires a transformer architecture trained with masked language modeling. BERT is the clearest and most faithful canonical grounding for that dependency. +- **evidence_span:** “We use the default hyperparameters of BERT-Base” + +### C2.I2 +- **ingredient:** Translation Language Modeling objective for cross-lingual alignment +- **canonical study:** Cross-lingual Language Model Pretraining +- **role:** CORE_METHOD +- **contribution:** Provides the cross-lingual pretraining objective used to align representations across languages. +- **rationale:** The model is not only trained with MLM; it explicitly incorporates TLM as a core objective for cross-lingual alignment. That objective maps cleanly to XLM as the canonical prior study. +- **evidence_span:** “Translation Language Modeling (Conneau and Lample, 2019, TLM)” + +### C2.I3 +- **ingredient:** Large-scale Indic-language pretraining text +- **canonical study:** NONE +- **additional studies:** IndicNLPSuite: Monolingual Corpora, Evaluation Benchmarks and Pre-trained Multilingual Language Models for Indian Languages; Samanantar: The Largest Publicly Available Parallel Corpora Collection for 11 Indic Languages +- **role:** TRAINING_DATA +- **contribution:** Provides the large-scale Indic-language text substrate used to pretrain the multilingual encoder. +- **rationale:** Recreating the model requires the actual large-scale Indic-language training text. But the operative pretraining text is structurally composite: it is dominated by corpus material created or assembled in the target paper, while also incorporating prior resources such as IndicCorp lineage and Samanantar-derived text. Because no single prior study cleanly represents that full training substrate, `NONE` is the correct canonical choice, with prior contributing resources listed as additional studies. +- **evidence_span:** “we merge data from IndicCorp v2 with Indic language data from Wikipedia and OSCAR” + + +-------------------------------------------------- +REFERENCE EXAMPLE 4 +-------------------------------------------------- + +# EXAMPLE: MassiveDS datastore scaling resource + +## 1. Cluster Evidence + +**The MassiveDS datastore resource** is used in the literature + +The paper also presents a pipeline reordering trick that makes datastore scaling computationally feasible. However, the downstream evidence does **not** show that this pipeline is reused as a separate artifact on its own. Therefore, the correct annotation is **one resource claim**, with the pipeline-reordering idea represented as an important ingredient rather than a separate claim. + +--- + +## 2. Claim Split Decision + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Resource +- **rewritten_claim:** **Dataset: A large-scale multi-domain datastore and experimental pipeline, enabling systematic study and evaluation of retrieval-based language model scaling.** +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C3 + +### Why this split is correct + +The downstream reuse is centered on **one released resource**: the MassiveDS datastore. + +- Later work uses the **datastore itself** as the reusable artifact. +- The paper also introduces a pipeline-reordering idea, but the evidence does **not** show that this pipeline is being reused downstream as a separate artifact. +- Because claim splitting is driven by **distinct downstream-used contributions**, the correct annotation is to keep **one claim**. + +The pipeline reordering is still important, but its role is different: + +- it is a **structural ingredient** that makes the datastore resource computationally feasible in its claimed form +- it is **not** a separate discovery claim unless downstream evidence shows reuse of the pipeline itself + +--- + +## 3. Rewritten Claims + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Resource +- **rewritten_claim:** **Resource: A massive open multi-domain datastore that combines general web and domain-specific text, enabling large-scale retrieval-in-context language modeling and datastore-scaling studies.** +- **why_this_is_atomic:** This claim isolates the released datastore resource that downstream work reuses. It does not split off the pipeline-reordering idea as a separate claim because that idea is not independently substantiated as a downstream-used artifact. +- **optional cluster_id:** C3 +- **decision:** YES_SUFFICIENT + +--- + +## 4. Minimal Ingredient Set Per Claim + +## Claim C1: MassiveDS datastore resource + +### Ingredient C1.I1 +- **ingredient_id:** C1.I1 +- **ingredient:** Multi-source open text corpora spanning general web and domain-specific sources +- **why_structurally_necessary:** The datastore is defined by both its scale and its domain diversity. Without large open corpora spanning general web and specialized domains, the released resource would not exist in its claimed form. +- **why_not_lower_level_substeps:** This should remain one composite data-source ingredient rather than being split into separate final ingredients for web data, books, scientific papers, math, biomedical text, code, and other source families. +- **why_not_adjacent_implementation_details:** The structural dependency is the multi-source raw text substrate itself, not every individual source-selection or curation decision. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** RedPajama: An Open Source Recipe to Reproduce LLaMA Training Dataset; PubMed Baseline Repository; S2ORC: The Semantic Scholar Open Research Corpus; peS2o (Pretraining Efficiently on S2ORC) Dataset; NaturalProofs: Mathematical Theorem Proving in Natural Language; Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer +- **role:** DATA_SOURCE +- **contribution:** Supplies the large-scale open text sources used to assemble the datastore across both general web and specialized domains. +- **rationale:** Recreating MassiveDS requires the raw text substrate that gives it both scale and domain coverage. That substrate is inherently composite: the datastore mixes general web data with multiple distinct domain-specific resources. No single prior study faithfully represents that entire source mixture, so `NONE` is the correct canonical grounding, with the main contributing source resources listed as additional studies. +- **evidence_span:** “comprising 1.4 trillion tokens of both general web data and domain specific data” + +### Ingredient C1.I2 +- **ingredient_id:** C1.I2 +- **ingredient:** Pretrained dense retriever for document embedding +- **why_structurally_necessary:** The released datastore is not just raw text; it is an operational retrieval datastore with dense-vector retrieval over passages. Without a pretrained retriever to embed documents, it would not function in its claimed form. +- **why_not_lower_level_substeps:** This should remain one retriever ingredient rather than being split into encoder architecture, embedding dimensionality, or indexing internals. +- **why_not_adjacent_implementation_details:** Retriever hyperparameters and ablation choices are implementation details. The structural dependency is the pretrained dense retriever used to embed the datastore. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Unsupervised Dense Information Retrieval with Contrastive Learning +- **additional groundings if any:** none +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the pretrained retriever used to represent datastore documents as dense vectors for retrieval. +- **rationale:** To reproduce the datastore as a usable retrieval resource, the corpus must be embedded into a searchable dense space. The paper explicitly uses Contriever-MSMARCO for this purpose, making Contriever the cleanest canonical grounding. +- **evidence_span:** “we use CONTRIEVER-MSMARCO (Izacard et al., 2022), which represents every document in the datastore as a dense vector” + +### Ingredient C1.I3 +- **ingredient_id:** C1.I3 +- **ingredient:** Retrieval-in-context language modeling framework +- **why_structurally_necessary:** MassiveDS is introduced specifically to support retrieve-in-context language models rather than as a generic corpus release. Without this framing, the datastore would lose the main capability it is meant to enable. +- **why_not_lower_level_substeps:** This should remain one framework-level ingredient rather than being split into retrieval prompting, concatenation order, or model-specific usage details. +- **why_not_adjacent_implementation_details:** The structural dependency is the inference-time retrieval framework itself, not one particular LM or evaluation setup. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** In-Context Retrieval-Augmented Language Models +- **additional groundings if any:** none +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Defines the retrieval-in-context framing in which documents are retrieved from an external datastore and prepended to LM inputs at inference time. +- **rationale:** The datastore is meaningful here because it supports retrieval-based language models that use retrieved documents in context. Ram et al. is the cleanest canonical grounding because it directly represents the retrieve-in-context setup that the paper focuses on. +- **evidence_span:** “We focus on retrieve-in-context language models (RIC-LMs)” + +### Ingredient C1.I4 +- **ingredient_id:** C1.I4 +- **ingredient:** Data filtering workflow for deduplication, decontamination, and quality control +- **why_structurally_necessary:** The released resource is not merely a raw trillion-token dump; it is a constructed datastore whose quality depends on filtering, deduplication, and decontamination steps. Without this workflow, the datastore would be substantially weaker as a reusable retrieval resource. +- **why_not_lower_level_substeps:** This should remain one filtering-workflow ingredient rather than being split into deduplication, decontamination, and quality filtering as separate ingredients. +- **why_not_adjacent_implementation_details:** Individual thresholds, heuristics, or local filtering choices are not the structural dependency; the key dependency is the broader filtering workflow that makes the datastore usable. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Dolma: an Open Corpus of Three Trillion Tokens for Language Model Pretraining Research +- **additional groundings if any:** none +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the filtering workflow used to clean the datastore through deduplication, decontamination, and quality controls. +- **rationale:** Building a reusable datastore at this scale requires systematic filtering so the resulting resource is not just large, but also usable and reasonably clean. The paper explicitly ties this step to Dolma, making it the best canonical grounding. +- **evidence_span:** “data filtering, including deduplication, decontamination, and quality filters” + +### Ingredient C1.I5 +- **ingredient_id:** C1.I5 +- **ingredient:** Pipeline reordering to amortize expensive indexing and retrieval across datastore variants +- **why_structurally_necessary:** The datastore is released together with an experimental pipeline that makes large-scale datastore scaling feasible on modest compute. Without this reordering idea, the resource would lose an important defining property: practical support for scalable datastore studies. +- **why_not_lower_level_substeps:** This should remain one pipeline-level ingredient rather than being split into retrieval-overfetching, post-hoc subsampling, and late-stage filtering as separate ingredients. +- **why_not_adjacent_implementation_details:** The structural dependency is the reordering principle itself—running expensive steps once and sharing them—not the local code or systems details used to implement it. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the core pipeline idea that makes datastore scaling computationally feasible by sharing expensive indexing and retrieval across many variants. +- **rationale:** A key part of the released contribution is not only the datastore’s scale, but the fact that the paper makes large-scale scaling studies feasible by reordering operations so indexing and retrieval are amortized. This idea is introduced in the target paper rather than inherited from a single prior study, so `NONE` is the correct grounding. +- **evidence_span:** “the most expensive ones—indexing and retrieval—are run only once at the start” + +--- + +## 5. Excluded Tempting Non-Ingredients + +### Claim C1 exclusions +- **Pipeline reordering as a separate second claim** + - Excluded because the downstream reuse evidence centers on the datastore resource, not on the pipeline as an independently reused artifact. The pipeline is important, but it is best represented as a structural ingredient of the single datastore claim. + +- **Passage chunking and indexing into fixed-length retrieval units** + - Excluded because this is a lower-level construction step rather than one of the smallest higher-level structural ingredients. It is subordinate to the broader retriever-and-datastore construction pipeline. + +- **FAISS / exact top-K retrieval implementation** + - Excluded because this is too implementation-specific. It supports the pipeline but is not needed at the abstraction level of the minimal structurally sufficient ingredient set. + +- **Separate final ingredients for general web data and domain-specific data** + - Excluded because the higher-level composite ingredient is the multi-source open text substrate spanning both. Splitting them would over-decompose the datastore’s source basis. + +- **Contriever as MODEL_INITIALIZATION instead of IMPLEMENTATION_TOOLING** + - Excluded because its role here is not initializing the target discovery as a model checkpoint, but operationally embedding the datastore documents for retrieval. + +--- + +## 6. Candidate Grounding Decisions + +## Claim C1: MassiveDS datastore resource + +### Ingredient C1.I1 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The datastore’s source substrate is structurally composite across general web and many domain-specific resources. +- **Candidate:** RedPajama: An Open Source Recipe to Reproduce LLaMA Training Dataset +- **decision:** accepted_additional +- **why:** Representative open source for books and other large-scale text components, but not sufficient alone to represent the whole source mixture. +- **Candidate:** PubMed Baseline Repository +- **decision:** accepted_additional +- **why:** Important biomedical source contributing to the domain-specific side of the datastore. +- **Candidate:** S2ORC: The Semantic Scholar Open Research Corpus +- **decision:** accepted_additional +- **why:** Important scientific-paper source for the domain-specific mixture. +- **Candidate:** peS2o (Pretraining Efficiently on S2ORC) Dataset +- **decision:** accepted_additional +- **why:** Additional scientific corpus contribution. +- **Candidate:** NaturalProofs: Mathematical Theorem Proving in Natural Language +- **decision:** accepted_additional +- **why:** Representative mathematical-language source for the domain-specific mixture. +- **Candidate:** Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer +- **decision:** accepted_additional +- **why:** Relevant additional grounding because C4 is explicitly cited as one of the general web data sources. + +### Ingredient C1.I2 +- **Candidate:** Unsupervised Dense Information Retrieval with Contrastive Learning +- **decision:** accepted_canonical +- **why:** Cleanest representative prior study for Contriever, the pretrained dense retriever actually used to embed datastore passages. + +### Ingredient C1.I3 +- **Candidate:** In-Context Retrieval-Augmented Language Models +- **decision:** accepted_canonical +- **why:** Cleanest representative framing for the retrieval-in-context LM setup the datastore is designed to support. + +### Ingredient C1.I4 +- **Candidate:** Dolma: an Open Corpus of Three Trillion Tokens for Language Model Pretraining Research +- **decision:** accepted_canonical +- **why:** Explicitly cited by the paper for the filtering stage and is the cleanest representative for the deduplication / decontamination / quality-control workflow. + +### Ingredient C1.I5 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The pipeline-reordering idea is introduced in the target paper itself and is not cleanly attributable to one prior study. + +--- + +## 7. Final Grounding Decisions + +## Claim C1 +- **C1.I1** → NONE → DATA_SOURCE + - additional: RedPajama: An Open Source Recipe to Reproduce LLaMA Training Dataset; PubMed Baseline Repository; S2ORC: The Semantic Scholar Open Research Corpus; peS2o (Pretraining Efficiently on S2ORC) Dataset; NaturalProofs: Mathematical Theorem Proving in Natural Language; Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer +- **C1.I2** → Unsupervised Dense Information Retrieval with Contrastive Learning → IMPLEMENTATION_TOOLING +- **C1.I3** → In-Context Retrieval-Augmented Language Models → CONCEPTUAL_FRAMEWORK +- **C1.I4** → Dolma: an Open Corpus of Three Trillion Tokens for Language Model Pretraining Research → IMPLEMENTATION_TOOLING +- **C1.I5** → NONE → CONCEPTUAL_FRAMEWORK + +--- + +## 8. Final Rationales And Evidence + +## Claim C1: MassiveDS datastore resource + +### C1.I1 +- **ingredient:** Multi-source open text corpora spanning general web and domain-specific sources +- **canonical study:** NONE +- **additional studies:** RedPajama: An Open Source Recipe to Reproduce LLaMA Training Dataset; PubMed Baseline Repository; S2ORC: The Semantic Scholar Open Research Corpus; peS2o (Pretraining Efficiently on S2ORC) Dataset; NaturalProofs: Mathematical Theorem Proving in Natural Language; Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer +- **role:** DATA_SOURCE +- **contribution:** Supplies the large-scale open text sources used to assemble the datastore across both general web and specialized domains. +- **rationale:** Recreating MassiveDS requires the raw text substrate that gives it both scale and domain coverage. That substrate is inherently composite: the datastore mixes general web data with multiple distinct domain-specific resources. No single prior study faithfully represents that entire source mixture, so `NONE` is the correct canonical grounding, with the main contributing source resources listed as additional studies. +- **evidence_span:** “comprising 1.4 trillion tokens of both general web data and domain specific data” + +### C1.I2 +- **ingredient:** Pretrained dense retriever for document embedding +- **canonical study:** Unsupervised Dense Information Retrieval with Contrastive Learning +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the pretrained retriever used to represent datastore documents as dense vectors for retrieval. +- **rationale:** To reproduce the datastore as a usable retrieval resource, the corpus must be embedded into a searchable dense space. The paper explicitly uses Contriever-MSMARCO for this purpose, making Contriever the cleanest canonical grounding. +- **evidence_span:** “we use CONTRIEVER-MSMARCO (Izacard et al., 2022), which represents every document in the datastore as a dense vector” + +### C1.I3 +- **ingredient:** Retrieval-in-context language modeling framework +- **canonical study:** In-Context Retrieval-Augmented Language Models +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Defines the retrieval-in-context framing in which documents are retrieved from an external datastore and prepended to LM inputs at inference time. +- **rationale:** The datastore is meaningful here because it supports retrieval-based language models that use retrieved documents in context. Ram et al. is the cleanest canonical grounding because it directly represents the retrieve-in-context setup that the paper focuses on. +- **evidence_span:** “We focus on retrieve-in-context language models (RIC-LMs)” + +### C1.I4 +- **ingredient:** Data filtering workflow for deduplication, decontamination, and quality control +- **canonical study:** Dolma: an Open Corpus of Three Trillion Tokens for Language Model Pretraining Research +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the filtering workflow used to clean the datastore through deduplication, decontamination, and quality controls. +- **rationale:** Building a reusable datastore at this scale requires systematic filtering so the resulting resource is not just large, but also usable and reasonably clean. The paper explicitly ties this step to Dolma, making it the best canonical grounding. +- **evidence_span:** “data filtering, including deduplication, decontamination, and quality filters” + +### C1.I5 +- **ingredient:** Pipeline reordering to amortize expensive indexing and retrieval across datastore variants +- **canonical study:** NONE +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the core pipeline idea that makes datastore scaling computationally feasible by sharing expensive indexing and retrieval across many variants. +- **rationale:** A key part of the released contribution is not only the datastore’s scale, but the fact that the paper makes large-scale scaling studies feasible by reordering operations so indexing and retrieval are amortized. This idea is introduced in the target paper rather than inherited from a single prior study, so `NONE` is the correct grounding. +- **evidence_span:** “the most expensive ones—indexing and retrieval—are run only once at the start” + + +-------------------------------------------------- +TARGET PAPER CONTEXT +-------------------------------------------------- +Target paper context: + +TITLE: CSCD-NS: a Chinese Spelling Check Dataset for Native Speakers +YEAR: 2022 +VENUE: Annual Meeting of the Association for Computational Linguistics +PAPER_ID: + +REFINED DOWNSTREAM CLUSTER EVIDENCE: +[ + { + "cluster_id": "C1", + "representative_claim": "Uses target evaluation script for performance evaluation", + "cluster_title": "Uses target evaluation script for performance evaluation", + "count": "1", + "merge_rationale": "Represents the direct use of the target paper's evaluation script to compute metrics." + }, + { + "cluster_id": "C2", + "representative_claim": "Uses target dataset for model training and evaluation", + "cluster_title": "Uses target dataset for model training and evaluation", + "count": "7", + "merge_rationale": "These clusters all describe using the target dataset (CSCD-NS) for downstream experiments, including model evaluation, training, and dataset aggregation." + } +] + +TARGET PAPER CONTEXT: +[ + { + "cluster_id": "C1", + "representative_claim": "Uses target evaluation script for performance evaluation", + "cluster_title": "Uses target evaluation script for performance evaluation", + "count": "1", + "merge_rationale": "Represents the direct use of the target paper's evaluation script to compute metrics." + }, + { + "cluster_id": "C2", + "representative_claim": "Uses target dataset for model training and evaluation", + "cluster_title": "Uses target dataset for model training and evaluation", + "count": "7", + "merge_rationale": "These clusters all describe using the target dataset (CSCD-NS) for downstream experiments, including model evaluation, training, and dataset aggregation." + } +] + +FULL PAPER CONTENT: +% This must be in the first 5 lines to tell arXiv to use pdfLaTeX, which is strongly recommended. +\pdfoutput=1 +% In particular, the hyperref package requires pdfLaTeX in order to break URLs across lines. + +\documentclass[11pt]{article} + +% Remove the "review" option to generate the final version. +\usepackage{emnlp2023} + +% Standard package includes +\usepackage{times} +\usepackage{latexsym} +\usepackage{CJKutf8} +\usepackage{graphicx} +\usepackage{amsmath} +\usepackage{multirow} +\usepackage{bbding} +\usepackage{arydshln} +% For proper rendering and hyphenation of words containing Latin characters (including in bib files) +\usepackage[T1]{fontenc} +% For Vietnamese characters +% \usepackage[T5]{fontenc} +% See https://www.latex-project.org/help/documentation/encguide.pdf for other character sets + +% This assumes your files are encoded as UTF8 +\usepackage[utf8]{inputenc} + +% This is not strictly necessary and may be commented out. +% However, it will improve the layout of the manuscript, +% and will typically save some space. +\usepackage{microtype} +\usepackage{tablefootnote} + +% This is also not strictly necessary and may be commented out. +% However, it will improve the aesthetics of text in +% the typewriter font. +\usepackage{inconsolata} +\newcommand{\tabincell}[2]{\begin{tabular}{@{}#1@{}}#2\end{tabular}} + +% If the title and author information does not fit in the area allocated, uncomment the following +% +%\setlength\titlebox{} +% +% and set to something 5cm or larger. + +\title{CSCD-NS: a Chinese Spelling Check Dataset for Native Speakers} + +% Author information can be set in various styles: +% For several authors from the same institution: +\author{Yong Hu, Fandong Meng, Jie Zhou \\ +WeChat AI, Tencent Inc., China \\ +\texttt{\{rightyonghu,fandongmeng,withtomzhou\}@tencent.com} +} +% if the names do not fit well on one line use +% Author 1 \\ {\bf Author 2} \\ ... \\ {\bf Author n} \\ +% For authors from different institutions: +% \author{Author 1 \\ Address line \\ ... \\ Address line +% \And ... \And +% Author n \\ Address line \\ ... \\ Address line} +% To start a separate ``row'' of authors use \AND, as in +% \author{Author 1 \\ Address line \\ ... \\ Address line +% \AND +% Author 2 \\ Address line \\ ... \\ Address line \And +% Author 3 \\ Address line \\ ... \\ Address line} + +% \author{First Author \\ +% Affiliation / Address line 1 \\ +% Affiliation / Address line 2 \\ +% Affiliation / Address line 3 \\ +% \texttt{email@domain} \\\And +% Second Author \\ +% Affiliation / Address line 1 \\ +% Affiliation / Address line 2 \\ +% Affiliation / Address line 3 \\ +% \texttt{email@domain} \\} + +\begin{document} +\maketitle +\begin{abstract} +In this paper, we present CSCD-NS, the first Chinese spelling check (CSC) dataset designed for native speakers, containing 40,000 samples from a Chinese social platform. Compared with existing CSC datasets aimed at Chinese learners, CSCD-NS is ten times larger in scale and exhibits a distinct error distribution, with a significantly higher proportion of word-level errors. To further enhance the data resource, we propose a novel method that simulates the input process through an input method, generating large-scale and high-quality pseudo data that closely resembles the actual error distribution and outperforms existing methods. Moreover, we investigate the performance of various models in this scenario, including large language models (LLMs), such as ChatGPT. The result indicates that generative models underperform BERT-like classification models due to strict length and pronunciation constraints. The high prevalence of word-level errors also makes CSC for native speakers challenging enough, leaving substantial room for improvement. +\footnote{https://github.com/nghuyong/cscd-ns} +\end{abstract} + + +\section{Introduction} +Chinese spelling check (CSC) is a task to detect and correct spelling errors in Chinese texts. +There are two primary user groups for CSC: (1) Chinese learners, including teenage students and individuals who use Chinese as a second language, and (2) Chinese native speakers. It is obvious that the latter user group has a larger population and more diverse applications, therefore, this paper concentrates on CSC for native speakers. + +\begin{CJK*}{UTF8}{gbsn} +\begin{figure}[t!] + \centering + \includegraphics[width=0.5\textwidth]{./image/problem.jpg} + \caption{An error from SIGHAN: misspelling “错误” as “错勿”. Despite having the same pronunciation, it's hard to reproduce this error in the given context through a Chinese IME, no matter what input form is used.} + \label{Fig:problem} +\end{figure} +\end{CJK*} + +However, there is still no CSC dataset specifically designed for native speakers. Existing CSC datasets, such as SIGHAN13, 14, and 15 \citep{ShihHungWu2013ChineseSC, LiangChihYu2014OverviewOS, YuenHsienTseng2015IntroductionTS}, are all sourced from Chinese learners. +Spelling errors made by Chinese learners differ greatly from those made by native speakers. This is because Chinese input relies on Chinese input methods (IME), and modern Chinese IMEs always have powerful language models, making it difficult to recommend candidates that clearly do not fit the context. As shown in Figure \ref{Fig:problem}, native speakers using Chinese IMEs are unlikely to make such an unusual error. + +Furthermore, the size of existing datasets is limited. As shown in Table \ref{Tab:statistic}, for three SIGHAN datasets, the training set contains an average of merely 2158 samples, while the test set comprises an average of only 1054 samples, and no development set is provided. When using such small-scale datasets, it is difficult for models to be trained sufficiently and for evaluation results to be reliable. + +To address the aforementioned issues, we introduce CSCD-NS, a Chinese spelling check dataset designed for native speakers. The dataset is sourced from real Weibo (a Chinese social media platform) posts, which contain genuine spelling errors made by native speakers during their input process. Moreover, the dataset comprises 40,000 samples, which is ten times larger than previous datasets and this is also the largest dataset for the CSC task. To conduct an in-depth investigation into the distribution of spelling errors, we develop a tagging system that operates at phonetic and semantic levels. The analysis indicates that native speakers make a higher proportion of homophonic and word-level errors compared to Chinese learners, with the proportion of word-level errors doubling. + +\begin{CJK*}{UTF8}{gbsn} + \begin{figure}[t!] + \centering + \includegraphics[width=0.45\textwidth]{./image/example.jpg} + \caption{An authentic Weibo post from LCSTS, where the phrase "效力于" is mistakenly written as "效力与".} + \label{Fig:example} + \end{figure} +\end{CJK*} + +Due to the lack of labeled data, previous studies always build additional pseudo data to improve the performance of models. However, these methods, which rely on confusion sets \citep{liu2021plome,zhang2020spelling} or ASR transcriptions \citep{wang2018hybrid}, do not align with the real-world input scenario. Therefore, we propose a novel method that directly simulates the input process through the Chinese IME and adds sampled noises to construct high-quality pseudo data. Experimental results show that our method can better fit the real error distribution and bring greater improvements. + +We conduct comprehensive experiments on CSCD-NS, with different model sizes (from 0.1B to 13B parameters), architectures (encoder-only, encoder-decoder, and decoder-only), and learning approaches (fine-tuning and in-context learning). We also evaluate the performance of ChatGPT and GPT4. The results demonstrate that BERT-like classification models outperform generative models, as the latter struggle with the simultaneous constraints of text length and pronunciation. Concurrently, the CSC task for native speakers is challenging due to the high proportion of word-level errors, leaving substantial room for improvement. + +In summary, our contributions are as follows: +\begin{itemize} + \item We introduce the first Chinese spelling check dataset for native speakers which is also the largest dataset for the CSC task. Through quantitative analyses, we further unveil the specific error distribution for this scenario. + \item We propose a novel method for constructing high-quality and large-scale pseudo data through a Chinese IME. Experimental results show that our method can bring greater improvements than existing methods. + \item We explore the performance of different types of models in this scenario and analyze the challenges. To the best of our knowledge, we are the first to investigate the effectiveness and limitations of large language models (LLMs), such as ChatGPT, in addressing the CSC task. +\end{itemize} + +\section{Related Work} + +\textbf{CSC Datasets}: +The existing CSC datasets, such as the SIGHAN series \citep{ShihHungWu2013ChineseSC, LiangChihYu2014OverviewOS, YuenHsienTseng2015IntroductionTS}, primarily cater to Chinese learners. However, these datasets suffer from limited data size and significant discrepancies in spelling errors compared to those made by native speakers. While there have been some efforts to develop Chinese grammatical error correction (CGEC) datasets for native speakers \citep{ma-etal-2022-linguistic, xu-etal-2022-fcgec, zhao2022overview, wang-etal-2022-cctc}, no such work has been undertaken for CSC datasets. + +\textbf{CSC Data Augmentation}: In order to compensate for the lack of labeled data, previous studies often create additional pseudo data to enhance performance. The mainstream method is based on confusion sets \citep{liu2021plome, zhang2020spelling}, the pseudo data generated in this way is large in size but low in quality because context information is not considered. +Another relatively high-quality construction method is based on ASR \citep{wang2018hybrid}. However, this approach requires additional labeled ASR data, making it difficult to create large-scale datasets. Moreover, the spelling errors generated by these two methods differ greatly from those produced by native speakers, such as having a much smaller proportion of word-level errors. We provide a detailed analysis in Appendix A. + +\textbf{CSC models}: In recent years, BERT-like \citep{devlin-etal-2019-bert} classification models have dominated the research of the CSC task +\citep{hong2019faspell, zhu2022mdcspell, huang2021phmospell,zhang2020spelling, liu2021plome, liu2022craspell}. +However, due to the lack of large-scale and high-quality datasets, the performance of these models is greatly limited. + + +\section{CSCD-NS} +In this section, we will show how to build CSCD-NS and discover the error distribution. + +\subsection{Data Source} +We chose the LCSTS dataset \citep{hu-etal-2015-lcsts} as our data source. This dataset is composed of authentic Weibo posts, which is a popular Chinese social media platform. As shown in Figure \ref{Fig:example}, spelling errors found within these posts reflect the genuine mistakes made by native speakers during the input process. Furthermore, this dataset contains over 2 million posts and covers a wide range of fields, such as finance, sports, and entertainment. The substantial scale and scope of the LCSTS make it suitable to serve as the data source. + +\subsection{Data Selection} +We split posts in LCSTS into sentences and obtain over 8 million sentences. +It is not realistic to label all of these sentences, and most of them are completely correct. +Therefore, we use an error detection model to filter out these correct sentences. + +\textbf{Detection Model}: +Given a source sequence ${\rm \bf X} = \{ x_{1}, x_{2}, ..., x_{N} \}$, the detection model is to check whether a token $x_{i}(1 \leq i \leq N)$ is correct or not. +We use the label 1 and 0 to mark the misspelled and the correct, respectively. +The detection model can be formalized as follows: +\begin{equation} + {\rm \bf y} = sigmoid(W^{T}(E({\rm \bf e}))) +\end{equation} +where ${\rm \bf e} = \{ e_{1}, e_{2}, ..., e_{N} \}$ is the sequence of word embeddings and ${E(*)}$ is the pre-trained encoder. +The output ${\rm \bf y} = \{ y_{1}, y_{2}, ..., y_{N} \}$ is the sequence of probabilities, where $y_{{i}} \in (0,1)$ denotes the probability that $x_{i}$ is erroneous. + +\textbf{Training}: We follow the successful experience \citep{wang2020combining} of the NLPTEA2020 task \citep{rao2020overview} +and use a Chinese ELECTRA-Large discriminator model \footnote{https://github.com/ymcui/Chinese-ELECTRA} \citep{clark2020electra} to initialize the detection model. +Following previous research, we train the detection model on SIGHAN13-15's training data and Wang's pseudo data \citep{wang2018hybrid} +and save the best checkpoint by SIGHAN13-15's test data \footnote{SIGHAN datasets have no development set.}. + +\textbf{Filtering}: We then use the trained detection model to filter out correct sentences. +For the input sentence, we can obtain the error probability of each token ${\rm \bf y} = \{ y_{1}, y_{2}, ..., y_{N} \}$. +\begin{CJK*}{UTF8}{gbsn} +Previous research indicates that the detection model struggles with certain Chinese particles (的/地/得) due to the poor labeling of these words in SIGHAN datasets. Additionally, low-frequency entity words, such as person names, are also prone to over-checking. To address these issues, we utilize a Chinese lexical analysis tool (LAC) \citep{jiao2018LAC} to identify these particles and entities in the input sentence. We categorize tokens into three groups: +$C_{particle}, C_{entity}, C_{others}$. Then, we calculate the maximum error probability for tokens in each category. If a category is empty, the maximum error probability is 0. We only consider a sentence correct if all the maximum error probabilities for each category are below the corresponding threshold. This can be formalized as follows: +\begin{equation} + \left\{ +\begin{aligned} + & max(\{y_{i} | x_{i} \in C_{particle}\}) < \delta_{particle} \\ + & max(\{y_i | x_{i} \in C_{entity}\}) < \delta_{entity} \\ + & max(\{y_i | x_{i} \in C_{others}\}) < \delta_{others} +\end{aligned} +\right. +\end{equation} +Here, $\delta_{particle}$, $\delta_{entity}$ and $\delta_{others}$ represent threshold values. +These thresholds are determined using a small manually labeled set and are set to 0.05, 0.5, and 0.15 respectively. +\end{CJK*} + +Based on the above method, we filter out approximately 91.2\% of sentences, retaining around 700,000 sentences that may contain spelling errors. To verify the accuracy of our filtering, we randomly select 2,000 filtered sentences and find that the accuracy is 99.2\%, aligning with our expectations. For the remaining sentences, we randomly select a portion for manual annotation. + +\begin{table*} +\small +\centering +\def\arraystretch{1.3} +\begin{tabular}{ccccccccc} +\hline +Dataset & Train Size & Dev Size & Test Size & Target Group & Source & Language & Err. ratio & Avg err./sent. \\ +\hline +SIGHAN13 & 700 & - & 1000 & Chinese learners & essays & TC & 77.11\% & 1.20 \\ +SIGHAN14 & 3437 & - & 1062 & Chinese learners & essays & TC & 86.19\% & 1.52 \\ +SIGHAN15 & 2339 & - & 1100 & Chinese learners & essays & TC & 81.82\% & 1.33 \\ +\hline +CSCD-NS & 3,0000 & 5,000 & 5,000 & native speakers & tweets & CN & 46.02\% & 1.09 \\ +\hline +\end{tabular} +\caption{The comparison of CSCD-NS and existing CSC datasets SIGHAN13, SIGHAN14, and SIGHAN15 in terms of dataset size, target group, data source, language, error sentence ratio, and average errors per sentence. In the table, TC and CN respectively denote Traditional Chinese and Simplified Chinese.} +\label{Tab:statistic} +\end{table*} + + +\begin{CJK*}{UTF8}{gbsn} +\begin{table*} +\small +\centering +\def\arraystretch{1.3} +\begin{tabular}{ccccccc} +\hline +origin & \multicolumn{6}{c}{由\textcolor{red}{之}可见,中国企业的技术提升后,\textcolor{red}{因}与跨国企业共同研发,不\textcolor{red}{在}简单的代加工} \\ +correct & \multicolumn{6}{c}{由\textcolor{blue}{此}可见,中国企业的技术提升后,\textcolor{blue}{应}与跨国企业共同研发,不\textcolor{blue}{再}简单的代加工} \\ +segment & \multicolumn{6}{c}{由此可见\; ,\;中国\;企业\;的\;技术\;提升\;后\;,\; 应\;与\;跨国企业\;共同研发\;,\;不再\;简单\;的\;代加工} \\ +translation & \multicolumn{6}{c}{\tabincell{l}{It can be seen that after the technology of Chinese enterprises is upgraded, \\they should cooperate with multinational enterprises in research instead of simple processing. }}\\ +\hline +\multirow{4}{*}{errors} & word pair & pinyin pair (ed) & phonetic tag & word len & ori-word valid & semantic tag \\ +\cline{2-7} +& 由\textcolor{red}{之}可见 $\rightarrow$ 由\textcolor{blue}{此}可见 & zhi $\rightarrow$ ci (2) & dissimilar & 4 & \XSolidBrush & character \\ +& \textcolor{red}{因} $\rightarrow$ \textcolor{blue}{应} & yin $\rightarrow$ ying (1) & similar & 1 & - & character \\ +& 不\textcolor{red}{在} $\rightarrow$ 不\textcolor{blue}{再} & zai $\rightarrow$ zai (0) & same & 2 & \Checkmark & word \\ +\hline +\end{tabular} +\caption{The process of adding phonetic and semantic tags. In the table, "ed" means edit distance, and "ori-word valid" indicates the validity of the original word.} +\label{Tab:tag_system} +\end{table*} +\end{CJK*} + +\subsection{Data Annotation} +We recruit a group of native speakers for manual annotation. The annotators are required to check whether the given sentence contains any spelling errors and provide the correct sentence. To ensure the quality of annotation, each sentence is annotated at least twice by different annotators. If the results of the two annotations are inconsistent, a senior annotator will make the final decision. + +To clarify the annotation rules and reduce disputes during the annotation process, sentences that fall into the following three categories will be directly discarded: +(1) sentences with inherent ambiguity; +(2) sentences with multiple reasonable answers to errors; +(3) sentences with complex grammatical errors. +Therefore, the sentence retained in the annotation process is semantically clear and has a unique correction result. + +In the end, we obtain 40,000 manually annotated sentences, which constitute the CSCD-NS dataset. After random partitioning, there are 30,000 samples in the training set, and 5,000 samples each in the development and test sets. + +\subsection{Analysis on Basic Statistics} +\label{section:statistic} +As shown in Table \ref{Tab:statistic}, the CSCD-NS is significantly larger in scale compared to existing datasets. Moreover, only the CSCD-NS provides a development set, is in Simplified Chinese, and originates from daily input by native speakers. Additionally, the CSCD-NS exhibits a more balanced distribution of positive and negative samples, with fewer spelling errors per sentence on average, suggesting a lower error rate among native speakers compared to Chinese learners. + +\subsection{Analysis on Error Distribution} +To conduct an in-depth study on the differences between native speakers and Chinese learners in terms of spelling errors, we design a tagging system for quantitative analyses. + +\textbf{Tag definition}: We define three phonetic-level tags and two semantic-level tags. +The phonetic tags consist of: +(1) same phonetic error: the erroneous character has the same pronunciation as the correct one. +(2) similar phonetic error: the erroneous character's pronunciation has an edit distance of 1 from the correct character's pronunciation. +(3) dissimilar phonetic error: the erroneous character's pronunciation has an edit distance greater than 1 from the correct character's pronunciation. The semantic tags consist of: +(1) word-level error: the erroneous word is a valid Chinese word. +(2) character-level error: the erroneous word is not a valid Chinese word, or the length of the erroneous word is 1. + +As shown in Table \ref{Tab:tag_system}, we first tokenize the correct sentence using LAC \citep{jiao2018LAC} to obtain word-level correction pairs. For each pair, we compute the pinyin edit distance and assign a phonetic-level tag. Simultaneously, we check the original word's validity in Chinese and incorporate its length to assign a semantic tag. + +\begin{figure*}[htbp] + \centering + \includegraphics[width=0.95\textwidth]{./image/pinyin.png} + \includegraphics[width=0.95\textwidth]{./image/semantic.png} + \caption{The comparison of error distribution (\%) at phonetic level (above) and semantic level (below).} + \label{fig:analysis} +\end{figure*} + +\textbf{Phonetic-level analysis}: As illustrated in Figure \ref{fig:analysis}, the proportion of same phonetic errors is the largest, while the proportion of dissimilar phonetic errors is the smallest in all four datasets. This feature is more pronounced in the CSCD-NS dataset, where the proportion of dissimilar phonetic errors is only 2.2\%, significantly lower than in the other datasets. Over 97\% of the errors are either the same phonetic or similar phonetic errors. This is because even if users make slight mistakes in their pinyin input, Chinese IME will auto-fix the input pinyin based on the context \citep{jia2014joint}. + +\textbf{Semantic-level analysis}: As shown in Figure \ref{fig:analysis}, the proportion of word-level errors in CSCD-NS (49.4\%) far exceeds that of existing datasets, which is twice the average value (23.3\%) of the SIGHAN datasets. This is because native speakers rely on the IME to input Chinese texts, which tends to recommend relatively reasonable valid words rather than strange "error words", resulting in a lower proportion of character-level errors. Compared to character-level errors, word-level errors pose a greater challenge to CSC systems. + +\section{Data Augmentation} +The manual annotation of CSC dataset is very expensive, therefore, how to construct pseudo data has always been a valuable topic. +In this section, we introduce a novel method that can generate high-quality pseudo data on a large scale. + +\subsection{Data Preparation} +The basic principle of pseudo-data construction is to add noise to accurate sentences. +Therefore, it is necessary to first prepare completely correct sentences. Fortunately, such text data is readily available on the Internet, including Wikipedia articles and classic books. This availability also ensures the generation of a large-scale dataset. + +\begin{figure*}[htbp] + \centering + \includegraphics[width=0.95\textwidth]{./image/WX20230624-1836052x.png} + \caption{The IME-based pseudo data generation process.} + \label{fig:pseduo} +\end{figure*} + +\subsection{IME-based Pseudo Data Generation} +First, we should analyze and obtain the error distribution based on the annotated data, including the distribution of the number of errors per sentence $D_{num}$, phonetic-level error distribution $D_{phonetic}$, and semantic-level error distribution $D_{semantic}$. + +As illustrated in Figure \ref{fig:pseduo}, the IME-based generation of pseudo data involves eight steps. + +(1) Sample a noise $v_{num}$ based on $D_{num}$, which indicates the number of generated spelling errors. The following steps are performed for each error. + +(2) Sample a semantic noise $v_{semantic}$ based on $D_{semantic}$, which indicates whether the error is at the word level or the character level. + +(3) Randomly select a token from the original text based on the sampled $v_{semantic}$. + +(4) Sample a phonetic noise $v_{phonetic}$ based on $D_{phonetic}$, which indicates whether the error is the same, similar, or dissimilar phonetic error. + +(5) Generate the new pinyin $p$, based on the sampled phonetic noise $v_{phonetic}$ and the actual pronunciation of the selected token. + +(6) In a Chinese IME, input the correct text before the selected token $t$ and enter the generated pinyin $p$. The IME would then recommend reasonable candidates $\{c_1, c_2, ..., c_n\}$. Leveraging the powerful language model of the IME, candidates are recommended by considering both the context before token $C_{ \delta +\end{equation} + +Through these steps, we can generate pseudo data that closely resembles the actual input process. + +\subsection{LCSTS-IME-2M} +We apply the above method to construct a large-scale CSC pseudo dataset LCSTS-IME-2M, consisting of about 2 million samples, based on the correct sentences filtered from LCSTS, the error distribution of CSCD-NS, and the Google IME \footnote{https://www.google.com/inputtools/}. + +\section{Experiments} +In this section, we evaluate the performance of different models on CSCD-NS and compare different pseudo-data construction methods. + +\begin{table} +\small +\centering +\def\arraystretch{1.3} +\begin{tabular}{lccc} +\hline +Model & Structure & Parameters & Learning \\ +\hline +BERT & Encoder & 102M & FT \\ +SM BERT & Encoder & 123M & FT \\ +PLOME & Encoder & 123M & FT \\ +BART & En-Decoder & 407M & FT \\ +% ChatGLM-6B & GLM & 6.17B & LoRA \\ +Baichuan2-7B & Decoder & 7.5B & LoRA \\ +Baichuan2-13B & Decoder & 13.9B & LoRA \\ +ChatGPT & Decoder & - & ICL \\ +GPT4 & Decoder & - & ICL \\ +\hline +\end{tabular} +\caption{The comparison of different baselines. In the table, En-Decoder refers to encoder-decoder, FT refers to full-parameter finetuning, LoRA refers to finetuning using low-rank adaptation, and ICL refers to in-context learning. Note that the number of parameters for ChatGPT and GPT4 has not been disclosed by the official documentation.} +\label{Tab:model} +\end{table} + +\begin{table*} + \small + \centering + \bgroup + \setlength\tabcolsep{0.6em} + \def\arraystretch{1.2} + \begin{tabular}{l|ccc|ccc|ccc|ccc} + \hline + \multirow{3}{*}{Models} & \multicolumn{6}{c|}{Sentence level} & \multicolumn{6}{c}{Character level} \\ + \cline{2-13} + & \multicolumn{3}{|c|}{Detection} & \multicolumn{3}{|c|}{Correction} & \multicolumn{3}{|c|}{Detection} & \multicolumn{3}{|c}{Correction} \\ + \cline{2-13} + & P & R & F1 & P & R & F1 & P & R & F1 & P & R & F1 \\ + \hline + BERT & 79.16 & 65.83 & 71.88 & 70.55 & 58.66 & 64.06 & 83.00 & 67.01 & 74.15 & 73.59 & 59.41 & 65.75 \\ + \footnotesize +\textit{LCSTS-IME-2M} & 78.98 & 73.60 & 76.20 & 75.63 & 70.47 & 72.96 & 82.19 & 75.75 & 78.84 & 78.84 & 72.67 & 75.63 \\ + \hline + {SM BERT} & 80.87 & 64.78 & 71.94 & 74.42 & 59.62 & 66.20 & \textbf{84.46} & 65.35 & 73.68 & 77.50 & 59.97 & 67.62 \\ + {\footnotesize +\textit{LCSTS-IME-2M}} & 79.19 & \textbf{74.86} & \textbf{76.97} & 75.75 & \textbf{71.60} & 73.62 & 82.39 & \textbf{77.93} & \textbf{80.10} & 78.63 & \textbf{74.37} & 76.44 \\ + \hline + {PLOME} & 79.78 & 57.23 & 66.65 & 78.09 & 56.01 & 65.23 & 83.48 & 57.99 & 68.44 & 81.49 & 56.61 & 66.81 \\ + {\footnotesize +\textit{LCSTS-IME-2M}} & \textbf{81.20} & 72.21 & 76.44 & \textbf{79.05} & 70.30 & \textbf{74.42} & 84.21 & 73.81 & 78.67 & \textbf{82.00} & 71.88 & \textbf{76.60} \\ + \hline + BART & 38.73 & 46.05 & 42.08 & 35.41 & 42.11 & 38.47 & 36.97 & 63.32 & 46.69 & 33.30 & 57.04 & 42.05 \\ + {\footnotesize +\textit{LCSTS-IME-2M}} & 42.06 & 54.29 & 47.40 & 41.01 & 52.95 & 46.22 & 40.87 & 75.97 & 53.15 & 39.68 & 73.75 & 51.60 \\ + % \hline + % ChatGLM-6B & 75.43 & 43.54 & 55.21 & 68.52 & 39.55 & 50.15 & 77.25 & 46.84 & 58.32 & 67.21 & 40.75 & 50.74 \\ + % {\footnotesize +\textit{LCSTS-IME-2M}} & 78.27 & 61.71 & 69.01 & 72.17 & 56.90 & 63.63 & 80.02 & 64.43 & 71.38 & 72.31 & 58.22 & 64.51 \\ + \hline + Baichuan2-7B & 64.98 & 53.04 & 58.41 & 62.70 & 51.17 & 56.35 & 57.10 & 56.92 & 57.01 & 54.72 & 54.55 & 54.63 \\ + {\footnotesize +\textit{LCSTS-IME-2M}} & 66.94 & 66.13 & 66.54 & 64.84 & 64.05 & 64.44 & 60.63 & 72.57 & 66.07 & 58.55 & 70.08 & 63.80 \\ + \hline + Baichaun2-13B & 67.53 & 60.23 & 63.67 & 65.14 & 58.11 & 61.42 & 60.07 & 64.62 & 62.26 & 57.49 & 61.86 & 59.60 \\ + {\footnotesize +\textit{LCSTS-IME-2M}} & 67.82 & 67.35 & 67.58 & 66.33 & 65.87 & 66.10 & 61.67 & 73.91 & 67.24 & 60.06 & 71.98 & 65.48 \\ + \hline + ChatGPT & 59.74 & 51.60 & 55.38 & 55.17 & 47.66 & 51.14 & 60.41 & 55.73 & 57.98 & 54.84 & 50.59 & 52.63 \\ + \hline + GPT4 & 58.37 & 59.71 & 59.03 & 53.67 & 54.90 & 54.28 & 58.40 & 63.60 & 60.89 & 52.34 & 57.00 & 54.57 \\ + \hline +\end{tabular} +\egroup +\caption{The performance (\%) of different models on CSCD-NS with or without pseudo dataset.} +\label{Tab:result} +\end{table*} + + +\begin{table} + \small + \centering + \def\arraystretch{1.2} + \begin{tabular}{cccc} + \hline + Models & Char level & Word level & $\Delta$ \\ + \hline + BERT & 72.82 & 71.07 & -1.75 \\ + SM BERT & 75.09 & 72.71 & -2.38 \\ + PLOME & 77.77 & 72.78 & -4.99 \\ + BART & 57.19 & 55.60 & -1.59 \\ + % ChatGLM-6B & 58.44 & 55.13 & -3.31 \\ + Baichaun2-7B & 65.88 & 63.50 & -2.38 \\ + Baichaun2-13B & 71.58 & 68.88 & -2.70 \\ + ChatGPT & 61.96 & 57.65 & -4.31 \\ + GPT4 & 71.06 & 61.13 & \textbf{-9.93} \\ + \hline +\end{tabular} + \caption{The performance (correction F1 score at character level \%) comparison between word-level and character-level errors. We only select the same phonetic errors here to avoid the influence of pronunciation.} +\label{Tab:word-level} +\end{table} + +\begin{CJK*}{UTF8}{gbsn} +\begin{table*} +\small +\centering +\def\arraystretch{1.3} +\begin{tabular}{cl} +\hline +origin & 新方案还处多方博弈中,想要尽快\textcolor{red}{的}打破僵局仍\textcolor{red}{就}困难重重,我们会跟紧并持续报\textcolor{red}{到} \\ +correct & 新方案还处多方博弈中,想要尽快\textcolor{blue}{地}打破僵局仍\textcolor{blue}{旧}困难重重,我们会跟紧并持续报\textcolor{blue}{道} \\ +translation & \tabincell{l}{The new plan is still in a multi-party game. It is still difficult to break the deadlock as soon as possible. \\ We will follow up and continue to report.}\\ +\hline +PLOME & 仍\textcolor{red}{就}(jiu) $\rightarrow$ 仍\textcolor{blue}{旧}(jiu); \ 跟\textcolor{red}{紧}(jin) $\rightarrow$ 跟\textcolor{blue}{进}(jin) \\ +\hline +ChatGPT & 处 $\rightarrow$ 处\textcolor{blue}{于}; \ 尽快\textcolor{red}{的}(de) $\rightarrow$ 尽快\textcolor{blue}{地}(de) ; \ 仍\textcolor{red}{就}(jiu) $\rightarrow$ 仍\textcolor{blue}{然}(ran); \ 跟\textcolor{red}{紧}(jin) $\rightarrow$ 跟\textcolor{blue}{进}(jin) \\ +\hline +\end{tabular} +\caption{The correction results of PLOME and ChatGPT. The pronunciation of the character is in brackets.} +\label{Tab:error_analysis} +\end{table*} +\end{CJK*} + +\begin{table} + \small + \centering + \def\arraystretch{1.2} + \begin{tabular}{ccccc} + \hline + Data & BERT & SM BERT & BART & Baichuan2-7B \\ + \hline + *CS & 19.57 & 15.39 & 14.02 & 25.67 \\ + *ASR & 42.22 & 39.50 & 29.97 & 35.69 \\ + *IME & \textbf{46.71} & \textbf{53.84} & \textbf{32.16} & \textbf{38.64}\\ + \hline + +CS & 64.53 & 67.36 & 42.95 & 54.30 \\ + +ASR & 68.44 & 71.26 & 44.88 & 56.77\\ + +IME & \textbf{70.41} & \textbf{72.72} & \textbf{45.92} & \textbf{57.85}\\ + \hline +\end{tabular} + \caption{The comparison of the performance (correction F1 score at character level \%) of three pseudo-data construction methods based on confusion sets (CS), ASR, and IME. In the table, an asterisk (*) indicates that only pseudo data is used for training, while a plus sign (+) denotes pretraining on pseudo data followed by continued training on the CSCD-NS's training data.} +\label{Tab:pseudo} +\end{table} + +\subsection{Basic Settings} +\textbf{Data}: +We perform experiments based on the labled data CSCD-NS and the pseudo data LCSTS-IME-2M. For pseudo data, we pre-train the model on it first, then fine-tune the model on the labeled data. + +\textbf{Metric}: +We compute detection and correction metrics at the sentence level and character level, including precision, recall, and F1 score. For sentence-level metrics, we use the calculation method in FASPell \citep{hong2019faspell}. For character-level metrics, we calculate all characters instead of only those correctly detected characters. + +\textbf{Baselines}: As shown in Table \ref{Tab:model}, the baselines encompass a diverse range of model structures, sizes, and learning methods. (1) BERT \citep{devlin-etal-2019-bert} directly fine-tunes the standard masked language model to generate fixed-length corrections. (2) Soft-Masked BERT (SM BERT) \citep{zhang2020spelling} employs an error detection model to provide better correction guidance. (3) PLOME \citep{liu2021plome} integrates phonetic and visual features into the pre-trained model. It has included a pre-training step on a confusion set-based pseudo dataset. (4) BART \citep{lewis-etal-2020-bart} models the CSC as a sequence-to-sequence task. We use the Chinese BART-large version here \footnote{https://huggingface.co/fnlp/bart-large-chinese}. (5) Baichuan2 \citep{baichuan2023baichuan2} models the CSC as a text generation task based on instructions. We fine-tune the model by LoRA \citep{hu2021lora} and use the version of 7B and 13B here \footnote{https://github.com/baichuan-inc/Baichuan2}. (6) ChatGPT and GPT4 perform the CSC task in a few-shot setting (10 examples) through in-context learning (ICL) \citep{dong2022survey}. + +To ensure that the correction results are of the same length as the input text, we only extract equal-length substitution modifications for generative models (BART, Baichuan2, ChatGPT and GPT4). Further implementation details of these models can be found in Appendix B. + +\subsection{Main Results} +(1) As shown in Table \ref{Tab:result}, compared with generative models, BERT-like token-level classification models (BERT, SM BERT, PLOME) remain the best approach for the CSC task, with smaller model size, higher performance, and faster inference speed. + +(2) The overall performance of generative models is relatively poor because the CSC task has strong constraints, requiring corrections to be of equal length and phonetically similar to the original text. These strong constraints make it easy for generative models to cause over-correction and incorrect correction. + +(3) For generative models, as the parameter size increases, their performance tends to improve gradually. +This trend can be observed from smaller models like BART (0.4B) to larger ones such as Baichuan2-13B. Similarly, GPT4 outperforms ChatGPT, and it is only through in-context learning that GPT4 can achieve performance comparable to Baichuan2-7B fine-tuned on CSCD-NS. + +(4) Large-scale and high-quality pseudo data is important for improving the performance, bringing consistent improvements across all six models. + +(5) The task of CSC for native speakers is highly challenging and the best F1 score of baseline models is still below 80. A key characteristic of this scenario is the high proportion of word-level errors. As shown in Table \ref{Tab:word-level}, word-level errors are more difficult for models to handle than character-level errors, as they require understanding more complex contexts. The development of CSC models, from BERT to PLOME, has primarily focused on optimizing character-level errors, with little progress made in addressing word-level errors. Therefore, further efforts are required in this scenario. + +\subsection{Better Data Augmentation Method} +In this part, we compare different pseudo-data construction methods. We conduct experiments on an existing ASR-based pseudo dataset \citep{wang2018hybrid}, containing about 271K samples. We extract the correct sentences and construct new pseudo-data based on confusion sets and IME, respectively. + +As demonstrated in Table \ref{Tab:pseudo}, our IME-based approach exhibits a substantial enhancement in performance compared to the other two methods. This improvement is even more pronounced when training exclusively on pseudo-data. The primary factor contributing to this success is the error distribution. As depicted in Figure \ref{fig:analysis-pseudo}, the pseudo-data generated via the IME-based method more accurately reflects the spelling errors made by native speakers. More analysis can be found in Appendix A. + +\subsection{Discussions} +\begin{CJK*}{UTF8}{gbsn} +For generative models, it is difficult to ensure that the generated text satisfies constraints on length and pronunciation. In the original correction results produced by ChatGPT, a staggering 82.1\% of modifications exhibit unequal length, while 35.4\% display dissimilar pronunciation. As illustrated in Table \ref{Tab:error_analysis}, the replacement of "处" with "处于" (located in) disregards the length constraint by introducing an additional character. Similarly, the correction of "仍旧" to "仍然" (still) overlooks the pronunciation constraint. Although these alterations may appear reasonable, they fail to meet the CSC task's requirements. + +BERT-like classification models have difficulty in addressing complex word-level errors and equal-length grammatical errors, as these require a strong contextual understanding. For example, the PLOME model shows a recall rate of only 60\% for word-level errors and merely 44\% for particle-related grammatical errors (的/地/得). Table \ref{Tab:error_analysis} illustrates that the incorrect word "报到" (check-in) is a high-frequency term, necessitating the model to recognize its context and correct it to "报道" (report). Similarly, in the phrase "尽快的打破" (try to break), the model must comprehend the grammatical rule (the particle between the adjective and the verb should be "地" instead of "的") and apply the appropriate correction. + +Moreover, all baseline systems, which are based on pre-trained language models, exhibit a propensity to over-convert low-frequency expressions into more prevalent ones \citep{zhang2020spelling, liu2022craspell}. As demonstrated in Table \ref{Tab:error_analysis}, "跟紧" and "跟进" share similar meanings (follow-up); however, since "跟进" is more frequently used, the model is prone to over-correcting. + +Consequently, enabling controlled text generation, addressing complex word-level and grammatical errors, and enhancing the understanding of low-frequency or new words all represent valuable avenues for future research. +\end{CJK*} + +\section{Conclusion} +In this paper, we focus on CSC for native speakers. For this scenario, we propose a new dataset, CSCD-NS, which is also the largest dataset for CSC. We further unveil the specific error distribution, with a significantly higher proportion of word-level errors. +Moreover, we introduce an IME-based pseudo-data construction approach, enabling large-scale generation of high-quality pseudo-data. We explore the performance of various models and first evaluate ChatGPT and GPT4 on the CSC task. Our experiments demonstrate that BERT-like models exhibit better performance than generative models, but there is still a considerable room for improvement. We hope these data resources and our findings could stimulate further research in this area. + +\section{Limitations} +\begin{CJK*}{UTF8}{gbsn} +Limitation of the CSCD-NS dataset: The data source for the CSCD-NS dataset is derived from a Chinese social networking platform. Therefore, it may not fully represent the error distribution of native speakers, as there may be slight differences in other scenarios, such as formal document writing. + +Limitation of the pseudo-data construction: The employed method of input simulation via IME is relatively basic, and the actual input scenario is more complex. For instance, individuals may utilize abbreviated pinyin to input common phrases, entering only the initials of characters (e.g., "wm" for "我们") \citep{tan2022exploring}. Moreover, a substantial number of users prefer the T9-style keyboard when employing IME on mobile devices. These factors collectively contribute to the inability of our pseudo-data construction method to accurately simulate the realistic input scenario. +\end{CJK*} + + +\section{Ethics Statement} +License: CSCD-NS and the constructed pseudo-data \textit{LCSTS-IME-2M} are based on LCSTS \citep{hu-etal-2015-lcsts}, we applied for and obtained the right to use this dataset, and performed the academic research under the copyright. + +Annotator Compensation: In this work, annotators are from a data labeling company in China. Through the pre-labeling, we estimate that each annotator could label 80 samples per hour and the label speed would be faster when they are skilled. In China, 60 yuan (8.76 dollars) per hour is a fair wage, therefore, we pay the annotator 0.75 yuan (0.11 dollars) for each sentence. + +% Entries for the entire Anthology, followed by custom entries +\bibliography{anthology,custom} +\bibliographystyle{acl_natbib} + +\appendix + +\clearpage +\section{Pseudo Data Analysis} + +\subsection{Impact of LM Post-Filtering} + +\begin{table} + \small + \centering + \def\arraystretch{1.2} + \begin{tabular}{cccc} + \hline + LM threshold ($\delta$) & Precision & Recall & F1 \\ + \hline + w/o & 41.17 & 40.66 & 40.91 \\ + -20\% & 44.52 & \textbf{49.01} & 46.66 \\ + 0\% & 49.69 & 44.07 & \textbf{46.71} \\ + 20\% & 50.64 & 26.46 & 34.76 \\ + 50\% & \textbf{57.52} & 9.38 & 16.12 \\ + \hline +\end{tabular} +\caption{The correction results (\%) at character level for pseudo data with different LM filtering strategies.} +\label{Tab:ngram} +\end{table} + +In this section, we investigate the influence of language model (LM) post-filtering, which constitutes the final stage of our proposed pseudo-data construction method. We extract accurate sentences from the Wang271K dataset \citep{wang2018hybrid} and generate pseudo-data using IME, incorporating various LM filtering strategies. We choose the basic BERT model to conduct the experiment and train the model only on the pseudo data to clearly distinguish the differences. + +As demonstrated in Table \ref{Tab:ngram}, the lack of LM filtering results in the introduction of undesired noise. For example, the generated pseudo-data may consist of entirely accurate sentences. In contrast, when the threshold is excessively low (even below 0), the generated errors become more complex, leading to high recall but poor precision. Conversely, if the threshold is set too high, the generated errors tend to be relatively simple, resulting in better precision but lower recall. Therefore, LM filtering is necessary, and selecting an appropriate threshold is also very important. + +\begin{figure*}[htbp] + \centering + \includegraphics[width=0.95\textwidth]{./image/pinyin-pseudo.png} + \includegraphics[width=0.95\textwidth]{./image/semantic-pseudo.png} + \caption{The comparison of error distribution (\%) at phonetic level (above) and semantic level (below).} + \label{fig:analysis-pseudo} +\end{figure*} + +\subsection{Error Distribution} +As illustrated in Figure \ref{fig:analysis-pseudo}, we analyze the error distribution of pseudo-data generated by various methods at both phonetic and semantic levels. It is clear that our pseudo-data construction method demonstrates the highest consistency with the CSCD-NS dataset, suggesting that our approach closely resembles real input scenarios. In contrast, the confusion set-based method and the ASR-based method exhibit a significant deviation from the actual error distribution. + +\begin{CJK*}{UTF8}{gbsn} + { + \begin{table} + \small + \centering + \def\arraystretch{1.2} + \begin{tabular}{cl} + \hline + \textit{translation} & simple, fashionable and moderate style \\ + \textit{origin} & 简约时尚的风格适中的 \\ + \textit{CS} & 简约时尚的风格\textcolor{red}{誓}中的\\ + \textit{ASR} & 简约时尚的风格\textcolor{red}{是}中的 \\ + \textit{IME} & 简约时尚的风格\textcolor{red}{始终}的 \\ + \hline + \textit{translation} & and the regulation is not perfect \\ + \textit{origin} & 且监管也不完善 \\ + \textit{CS} & 且监管也不\textcolor{red}{碗}善 \\ + \textit{ASR} & \textcolor{red}{其}监管也不完善 \\ + \textit{IME} & 且监管也不\textcolor{red}{玩}善 \\ + \hline + \end{tabular} + \caption{The pseudo data generated based on confusion set (CS), ASR, and IME.} + \label{Tab:pseudo-example} + \end{table} + } + \end{CJK*} + +\subsection{Case Study} +We sample some examples in Table \ref{Tab:pseudo-example}. It can be observed that the confusion set-based method is capable of producing similar phonetic errors; however, these errors are entirely out of context and can not accurately represent the real input scenario. The ASR-based method performs better but primarily generates character-level errors. Moreover, since the ASR-based method lacks an LM filtering module, the generated noise may occasionally be correct, as demonstrated by the third case in Table \ref{Tab:pseudo-example}. In contrast, our method can effectively generate high-quality pseudo data, encompassing both word-level and character-level errors. + + +\begin{table} + \small + \centering + \def\arraystretch{1.2} + \begin{tabular}{ll} + \hline + Configurations & Values \\ + \hline + PLM & bert-base-chinese \citep{devlin-etal-2019-bert} \tablefootnote{https://huggingface.co/bert-base-chinese} \\ + devices & 1 Nvidia A100 GPU (40GB) \\ + framework & PyTorch Lightning 1.3.8 \tablefootnote{https://www.pytorchlightning.ai/} \\ + optimizer & AdamW \citep{loshchilov2017decoupled} \\ + learning rate & 1e-4 \\ + sequence length & 512 \\ + batch size & 128 \\ + epochs & 10 \\ + dropout & 0.1 \\ + \multirow{2}{*}{model size} & BERT: 102 M \\ + & SM BERT: 123 M \\ + \multirow{2}{*}{training speed} & BERT: \textasciitilde10 batches/s \\ + & SM BERT: \textasciitilde7 batches/s \\ + metric for best \tablefootnote{The metric used to save the best model} & loss \\ + \hline +\end{tabular} + \caption{Configurations of BERT and SM BERT.} +\label{Tab:bert} +\end{table} + +\begin{table} + \small + \centering + \def\arraystretch{1.2} + \begin{tabular}{ll} + \hline + Configurations & Values \\ + \hline + PLM & PLOME pre-trained model \tablefootnote{https://share.weiyun.com/OREEY0H3} \\ + devices & 1 Nvidia V100 GPU (32GB) \\ + framework & Tensorflow 1.14 \tablefootnote{https://www.tensorflow.org/} \\ + optimizer & AdamW \citep{loshchilov2017decoupled} \\ + learning rate & 5e-5 \\ + sequence length & 180 \\ + batch size & 32 \\ + epochs & 10 \\ + dropout & 0.1 \\ + model size & 123 M \\ + training speed & \textasciitilde2.12 batches/s\\ + metric for best & F1-score of correction at character level \\ + \hline +\end{tabular} + \caption{Configurations of PLOME} +\label{Tab:plome} +\end{table} + +\section{Experimental Details} +In this section, we provide comprehensive descriptions of the experimental procedures and parameter settings for each model. + +Note that for each experiment, we select the best checkpoint based on the development set and evaluate its performance on the test set. We carry out three trials for each experiment and report the average results in the paper. The total training time is contingent upon the size of the training data and can be estimated based on the training speed. + +\subsection{BERT-like Models} +Since there is no official implementation for BERT and SM BERT, we follow a widely-used open-source version\footnote{https://github.com/gitabtion/BertBasedCorrectionModels}. For PLOME, we directly utilize the official code\footnote{https://github.com/liushulinle/PLOME}. We adhere to the default hyperparameters, and the detailed configurations for these three models can be found in Table \ref{Tab:bert} and Table \ref{Tab:plome}. + +\subsection{BART} +\begin{table} + \small + \centering + \def\arraystretch{1.2} + \begin{tabular}{ll} + \hline + Configurations & Values \\ + \hline + PLM & fnlp/bart-large-chinese \tablefootnote{https://huggingface.co/fnlp/bart-large-chinese} \\ + devices & 8 Nvidia A100 GPU (40GB) \\ + framework & transformers 4.29.1 \tablefootnote{https://github.com/huggingface/transformers} \\ + optimizer & AdamW \citep{loshchilov2017decoupled} \\ + learning rate & 5e-5 \\ + sequence length & 512 \\ + batch size & 256 \\ + epochs & 10 \\ + dropout & 0.1 \\ + model size & 407 M \\ + training speed & \textasciitilde3.5 batches/s\\ + metric for best & loss \\ + \hline + input & \{origin sentence\} \\ + \hline + output & \{correct sentence \} \\ + \hline +\end{tabular} + \caption{Configurations of BART} +\label{Tab:bart} +\end{table} + +We choose the Chinese BART-large model as the base model and fine-tune it for the CSC task by treating it as a sequence-to-sequence task. The model takes the original sentence as input and produces the correct sentence as output. The decoding method employed is beam search with a beam size of 4. The specific model configuration can be found in Table \ref{Tab:bart}. + +\subsection{Baichuan2} +\begin{CJK*}{UTF8}{gbsn} +Baichuan2 \citep{baichuan2023baichuan2} is a powerful Chinese language model that includes two open-source models, Baichuan2-7B and Baichuan2-13B. The CSC task is modeled as an instruction tuning task, with the instruction being "纠正句子中的拼写错误" (correct the spelling errors in the following sentence). We use LoRA \citep{hu2021lora} to fine-tune the model. During the decoding stage, random sampling is not performed, and the beam size is set to 1. Table \ref{Tab:chatglm} displays the specific configurations. + +\begin{table} + \small + \centering + \def\arraystretch{1.2} + \begin{tabular}{ll} + \hline + Configurations & Values \\ + \hline + PLM & Baichuan2 \tablefootnote{https://github.com/baichuan-inc/Baichuan2} \\ + devices & 8 Nvidia A100 GPU (40GB) \\ + framework & transformers 4.29.1 \tablefootnote{https://github.com/huggingface/transformers} \\ + optimizer & AdamW \\ + lora rank & 8 \\ + learning rate & 1e-4 \\ + sequence length & 512 \\ + batch size & 128 \\ + epochs & 10 \\ + dropout & 0.1 \\ + \multirow{2}{*}{model size} & Baichuan2-7B: 7.5 M \\ + & Baichuan2-13B: 13.9 M \\ + \multirow{2}{*}{training speed} & Baichuan2-7B: \textasciitilde3.0 s/batch \\ + & Baichuan2-13B: \textasciitilde4.4 s/batch \\ + metric for best & loss \\ + \hline + input & \tabincell{l}{Instrction: 纠正句子中的拼写错误 \\ + Input: \{origin sentence\} \\ + Output: } \\ + \hline + output & \{correct sentence \} \\ + \hline +\end{tabular} + \caption{Configurations of Baichuan2} +\label{Tab:chatglm} +\end{table} + +\subsection{ChatGPT and GPT4} + + +\begin{table*} + \small + \centering + \def\arraystretch{1.5} + \begin{tabular}{ll} + \hline + \multicolumn{2}{c}{prompt 1} \\ + \hline + instruction & 修正句子中的拼写错误,修正结果需要与原文长度相等,发音相近 \\ + \cline{2-2} + 10 examples & \tabincell{l}{ 比特币价格从15美元飚升到266美元 \ $\Rightarrow$ \ 比特币价格从15美元飙升到266美元 \\ + ... \\ + 其中,企业成为职务专利申请的主力军 \ $\Rightarrow$ \ 其中,企业成为职务专利申请的主力军 } \\ + \cline{2-2} + test case & 让农民工流血、流汗不在流泪 \ $\Rightarrow$ \ \\ + \hline + \hline + \multicolumn{2}{c}{prompt 2} \\ + \hline + instruction & 修正拼写错误,修正结果与原文需要长度相等,且发音尽可能相近 \\ + \cline{2-2} + 10 examples & \tabincell{l}{ 修正前:\ 比特币价格从15美元飚升到266美元 \\ + 修正后: \ 比特币价格从15美元飙升到266美元 \\ + ... \\ + 修正前:\ 其中,企业成为职务专利申请的主力军 \\ + 修正后:\ 其中,企业成为职务专利申请的主力军 } \\ + \cline{2-2} + test case & \tabincell{l}{修正前:\ 让农民工流血、流汗不在流泪 \\ + 修正后: + } \\ + \hline + \hline + \multicolumn{2}{c}{prompt 3} \\ + \hline + instruction & \tabincell{l}{Instruction: Correct spelling errors in the sentence, adhering to the following two requirements: \\ + (1) The corrected output should maintain the same character length as the original text. \\ + (2) The pinyin of the corrected character and the original character should be identical, or the edit distance \\ + should be as minimal as possible.}\\ + \cline{2-2} + 10 examples & \tabincell{l}{ Input:\ 比特币价格从15美元飚升到266美元 \\ + Output: \ 比特币价格从15美元飙升到266美元 \\ + ... \\ + Input:\ 其中,企业成为职务专利申请的主力军 \\ + Output:\ 其中,企业成为职务专利申请的主力军 } \\ + \cline{2-2} + test case & \tabincell{l}{Input:\ 让农民工流血、流汗不在流泪 \\ + Output: + } \\ + \hline +\end{tabular} + \caption{Three prompt templates designed to call ChatGPT/GPT4 for the CSC task.} +\label{Tab:chatgpt_prompt} +\end{table*} + + +\begin{table*} + \small + \centering + \bgroup + \setlength\tabcolsep{0.6em} + \def\arraystretch{1.2} + \begin{tabular}{c|ccc|ccc|ccc|ccc} + \hline + \multirow{3}{*}{Settings} & \multicolumn{6}{c|}{Sentence level} & \multicolumn{6}{c}{Character level} \\ + \cline{2-13} + & \multicolumn{3}{|c|}{Detection} & \multicolumn{3}{|c|}{Correction} & \multicolumn{3}{|c|}{Detection} & \multicolumn{3}{|c}{Correction} \\ + \cline{2-13} + & P & R & F1 & P & R & F1 & P & R & F1 & P & R & F1 \\ + \hline + prompt 1 (run1) & 52.92 & 51.13 & 52.01 & 48.70 & 47.05 & 47.86 & 54.14 & 57.91 & 55.96 & 48.56 & 51.94 & 50.19 \\ + prompt 1 (run2) & 53.61 & 50.22 & 51.86 & 49.40 & 46.27 & 47.78 & 54.08 & 56.28 & 55.16 & 48.84 & 50.83 & 49.82 \\ + prompt 1 (run3) & 53.85 & 50.61 & 52.18 & 49.75 & 46.75 & 48.20 & 54.73 & 56.92 & 55.80 & 49.30 & 51.27 & 50.26 \\ + \hline + prompt 2 (run1) & 55.52 & 48.83 & 51.96 & 50.94 & 44.80 & 47.67 & 55.08 & 54.86 & 54.97 & 49.25 & 49.05 & 49.15 \\ + prompt 2 (run2) & 55.43 & 49.61 & 52.36 & 50.82 & 45.49 & 48.01 & 55.48 & 55.65 & 55.56 & 49.72 & 49.88 & 49.80 \\ + prompt 2 (run3) & 55.91 & 50.22 & 52.91 & 51.76 & 46.49 & 48.98 & 55.56 & 56.72 & 56.13 & 50.33 & 51.38 & 50.85 \\ + \hline + prompt 3 (run1) & 59.56 & 47.27 & 52.71 & 55.25 & 43.84 & 48.89 & 61.16 & 51.11 & 55.69 & 55.49 & 46.36 & 50.52 \\ + prompt 3 (run2) & 58.29 & 45.88 & 51.35 & 54.88 & 43.19 & 48.34 & 60.62 & 49.84 & 54.71 & 55.67 & 45.77 & 50.24 \\ + prompt 3 (run3) & 59.85 & 47.83 & 53.17 & 55.56 & 44.41 & 49.36 & 61.29 & 51.70 & 56.09 & 56.00 & 47.23 & 51.24 \\ + \hline +\end{tabular} +\egroup +\caption{The performance (\%) of ChatGPT with different prompts on CSCD-NS.} +\label{Tab:chatgpt_result} +\end{table*} + +We tested ChatGPT and GPT4 through OpenAI's API on November 26, 2023, and the model id for ChatGPT is \textit{gpt-3.5-turbo-1106} and GPT4 is \textit{gpt-4-1106-preview}. +We set the temperature to 0 to reduce the influence of random sampling. As illustrated in Table \ref{Tab:chatgpt_prompt}, we devise three prompt templates, each comprising a task description, 10 examples, and a test sentence. These 10 examples encompass 5 positive instances (sentences containing spelling errors) and 5 negative instances (sentences without spelling errors), all of which are randomly chosen from the training set. As shown in Table \ref{Tab:chatgpt_result}, utilizing the same prompt template with varying example samples exerted a negligible effect on the outcomes. Likewise, employing different prompt templates also has a minor impact on the results. Given that the outcomes obtained using "prompt 3" are slightly better, we present the average results derived from "prompt 3" in our paper. +\end{CJK*} + + + +\end{document} + + +BIBLIOGRAPHY: +[ + { + "ref_id": "c96297261467b5daa2d01227496a70d444602434", + "title": "Baichuan 2: Open Large-scale Language Models", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "journals/corr/abs-2309-10305", + "ArXiv": "2309.10305", + "DOI": "10.48550/arXiv.2309.10305", + "CorpusId": 261951743 + } + }, + { + "ref_id": "c080f022c6f57d10dc8e8e44231c8995480b884d", + "title": "FCGEC: Fine-Grained Corpus for Chinese Grammatical Error Correction", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "journals/corr/abs-2210-12364", + "ArXiv": "2210.12364", + "DOI": "10.18653/v1/2022.findings-emnlp.137", + "CorpusId": 253097962 + } + }, + { + "ref_id": "4832eb72b07a1fd473df299057a98b0b0d84ed97", + "title": "Overview of CTC 2021: Chinese Text Correction for Native Speakers", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "journals/corr/abs-2208-05681", + "ArXiv": "2208.05681", + "DOI": "10.48550/arXiv.2208.05681", + "CorpusId": 251493068 + } + }, + { + "ref_id": "c32b652d605d9f097edf073f69253573b2b5d0cf", + "title": "Exploring and Adapting Chinese GPT to Pinyin Input Method", + "authors": null, + "year": null, + "external_ids": { + "ACL": "2022.acl-long.133", + "DBLP": "conf/acl/TanDTFHJLS22", + "ArXiv": "2203.00249", + "DOI": "10.48550/arXiv.2203.00249", + "CorpusId": 247187680 + } + }, + { + "ref_id": "a8ca46b171467ceb2d7652fbfb67fe701ad86092", + "title": "LoRA: Low-Rank Adaptation of Large Language Models", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "conf/iclr/HuSWALWWC22", + "ArXiv": "2106.09685", + "CorpusId": 235458009 + } + }, + { + "ref_id": "93bf0b32b17297d94f6a2cc35bce7c396eb17b36", + "title": "Spelling Error Correction with Soft-Masked BERT", + "authors": null, + "year": null, + "external_ids": { + "ACL": "2020.acl-main.82", + "DBLP": "journals/corr/abs-2005-07421", + "MAG": "3024680312", + "ArXiv": "2005.07421", + "DOI": "10.18653/v1/2020.acl-main.82", + "CorpusId": 218665566 + } + }, + { + "ref_id": "395de0bd3837fdf4b4b5e5f04835bcc69c279481", + "title": "BART: Denoising Sequence-to-Sequence Pre-training for Natural Language Generation, Translation, and Comprehension", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2982399380", + "ACL": "2020.acl-main.703", + "DBLP": "journals/corr/abs-1910-13461", + "ArXiv": "1910.13461", + "DOI": "10.18653/v1/2020.acl-main.703", + "CorpusId": 204960716 + } + }, + { + "ref_id": "fd273b87fe73800f04b51a5bf442df924a448f5f", + "title": "Chinese Lexical Analysis with Deep Bi-GRU-CRF Network", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2810578980", + "DBLP": "journals/corr/abs-1807-01882", + "ArXiv": "1807.01882", + "CorpusId": 49574530 + } + }, + { + "ref_id": "d07284a6811f1b2745d91bdb06b040b57f226882", + "title": "Decoupled Weight Decay Regularization", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2950541952", + "DBLP": "conf/iclr/LoshchilovH19", + "CorpusId": 53592270 + } + }, + { + "ref_id": "1294f49cf1a7397f0423ec617a78c7995139bc5b", + "title": "Neural Network Language Model for Chinese Pinyin Input Method Engine", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2398174936", + "DBLP": "conf/paclic/ChenZW15", + "ACL": "Y15-1052", + "CorpusId": 9506393 + } + }, + { + "ref_id": "d590ca98fdb2c727c492fbc6f108ce1ce4efb20f", + "title": "Introduction to SIGHAN 2015 Bake-off for Chinese Spelling Check", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "conf/acl-sighan/TsengLCC15", + "ACL": "W15-3106", + "MAG": "2251568283", + "DOI": "10.18653/v1/W15-3106", + "CorpusId": 18312025 + } + }, + { + "ref_id": "b122a828f5fee3c6afc54e70f41b00184d6383fc", + "title": "LCSTS: A Large Scale Chinese Short Text Summarization Dataset", + "authors": null, + "year": null, + "external_ids": { + "ArXiv": "1506.05865", + "ACL": "D15-1229", + "MAG": "2951457522", + "DBLP": "journals/corr/HuCZ15", + "DOI": "10.18653/v1/D15-1229", + "CorpusId": 11597846 + } + }, + { + "ref_id": "02ff89d27d8cc4b372fff259da41c1f87a6e8506", + "title": "Overview of SIGHAN 2014 Bake-off for Chinese Spelling Check", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2251873470", + "ACL": "W14-6820", + "DBLP": "conf/acl-sighan/YuLTC14", + "DOI": "10.3115/v1/W14-6820", + "CorpusId": 14818452 + } + }, + { + "ref_id": "4503c3b75bf515098184d993790639c26f75da3a", + "title": "A Joint Graph Model for Pinyin-to-Chinese Conversion with Typo Correction", + "authors": null, + "year": null, + "external_ids": { + "ACL": "P14-1142", + "MAG": "2251256737", + "DBLP": "conf/acl/JiaZ14", + "DOI": "10.3115/v1/P14-1142", + "CorpusId": 16217436 + } + }, + { + "ref_id": "6268daca3037d6cb40f8193b8636aa97efe1287f", + "title": "Chinese Spelling Check Evaluation at SIGHAN Bake-off 2013", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2251157340", + "ACL": "W13-4406", + "DBLP": "conf/acl-sighan/WuLL13", + "CorpusId": 962605 + } + }, + { + "ref_id": "8aa98fbfb6f1e979dead13ce24075503fe47658e", + "title": "A Survey for In-context Learning", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "journals/corr/abs-2301-00234", + "CorpusId": 263886074 + } + }, + { + "ref_id": "a27df49d3c8b1d96923e2e2eaa9bbfbe3ae732e7", + "title": "MDCSpell: A Multi-task Detector-Corrector Framework for Chinese Spelling Correction", + "authors": null, + "year": null, + "external_ids": { + "ACL": "2022.findings-acl.98", + "DBLP": "conf/acl/ZhuYZM22", + "DOI": "10.18653/v1/2022.findings-acl.98", + "CorpusId": 248780149 + } + }, + { + "ref_id": "a77d34655475fd6eac0ae6ab724a3bd4b36102a6", + "title": "CRASpell: A Contextual Typo Robust Approach to Improve Chinese Spelling Correction", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "conf/acl/LiuSYYCYS22", + "ACL": "2022.findings-acl.237", + "DOI": "10.18653/v1/2022.findings-acl.237", + "CorpusId": 248780130 + } + }, + { + "ref_id": "f5afc8f1472844aa837ff3052b4b74252132b2f5", + "title": "CCTC: A Cross-Sentence Chinese Text Correction Dataset for Native Speakers", + "authors": null, + "year": null, + "external_ids": { + "ACL": "2022.coling-1.294", + "DBLP": "conf/coling/WangDWC0H22", + "CorpusId": 252819430 + } + }, + { + "ref_id": "cf6be786c307a8c4e523daa455019a6b7650537c", + "title": "PLOME: Pre-training with Misspelled Knowledge for Chinese Spelling Correction", + "authors": null, + "year": null, + "external_ids": { + "ACL": "2021.acl-long.233", + "DBLP": "conf/acl/LiuYYZW20", + "DOI": "10.18653/v1/2021.acl-long.233", + "CorpusId": 236459826 + } + }, + { + "ref_id": "9998875164c1d46c8c41be6f22171e90abf0ccbe", + "title": "PHMOSpell: Phonological and Morphological Knowledge Guided Chinese Spelling Check", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "conf/acl/HuangLJZCWX20", + "ACL": "2021.acl-long.464", + "DOI": "10.18653/v1/2021.acl-long.464", + "CorpusId": 236460127 + } + }, + { + "ref_id": "2119f2e7497cacebc43ea04d0d8df96df9d08e03", + "title": "Overview of NLPTEA-2020 Shared Task for Chinese Grammatical Error Diagnosis", + "authors": null, + "year": null, + "external_ids": { + "ACL": "2020.nlptea-1.4", + "DOI": "10.18653/v1/2020.nlptea-1.4", + "CorpusId": 227905350 + } + }, + { + "ref_id": "e8fe671e2e04c1b45a5dddc51c1e60768a66b628", + "title": "Combining ResNet and Transformer for Chinese Grammatical Error Diagnosis", + "authors": null, + "year": null, + "external_ids": { + "ACL": "2020.nlptea-1.5", + "DOI": "10.18653/v1/2020.nlptea-1.5", + "CorpusId": 227904567 + } + }, + { + "ref_id": null, + "title": "ELECTRA: Pre-training text encoders as discriminators rather than generators", + "authors": null, + "year": null, + "external_ids": {} + }, + { + "ref_id": "df2b0e26d0599ce3e70df8a9da02e51594e0e992", + "title": "BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2951055169", + "ACL": "N19-1423", + "DBLP": "journals/corr/abs-1810-04805", + "ArXiv": "1810.04805", + "DOI": "10.18653/v1/N19-1423", + "CorpusId": 52967399 + } + }, + { + "ref_id": "b8783d2fe5affed35dd62cad74a5f468359cc5cc", + "title": "FASPell: A Fast, Adaptable, Simple, Powerful Chinese Spell Checker Based On DAE-Decoder Paradigm", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2983735660", + "DBLP": "conf/aclnut/HongYHLL19", + "ACL": "D19-5522", + "DOI": "10.18653/v1/D19-5522", + "CorpusId": 208001011 + } + }, + { + "ref_id": "c12e270f347334ced34614e110b9319888522da8", + "title": "A Hybrid Approach to Automatic Corpus Generation for Chinese Spelling Check", + "authors": null, + "year": null, + "external_ids": { + "ACL": "D18-1273", + "MAG": "2892311186", + "DBLP": "conf/emnlp/WangSLHZ18", + "DOI": "10.18653/v1/D18-1273", + "CorpusId": 53083061 + } + } +] + +CITATION CONTEXTS: +[ + { + "ref_id": "c96297261467b5daa2d01227496a70d444602434", + "citation_marker": "Baichuan 2: Open Large-scale Language Models", + "text": "This trend can be observed from smaller models like BART (0.4B) to larger ones such as Baichuan2-13B.", + "section": null, + "intents": [] + }, + { + "ref_id": "c96297261467b5daa2d01227496a70d444602434", + "citation_marker": "Baichuan 2: Open Large-scale Language Models", + "text": "( 5) Baichuan2 (Baichuan, 2023) models the CSC as a text generation task based on instructions.", + "section": null, + "intents": [] + }, + { + "ref_id": "c96297261467b5daa2d01227496a70d444602434", + "citation_marker": "Baichuan 2: Open Large-scale Language Models", + "text": "To ensure that the correction results are of the same length as the input text, we only extract equal-length substitution modifications for generative models (BART, Baichuan2, ChatGPT and GPT4).", + "section": null, + "intents": [] + }, + { + "ref_id": "c96297261467b5daa2d01227496a70d444602434", + "citation_marker": "Baichuan 2: Open Large-scale Language Models", + "text": "Similarly, GPT4 outperforms ChatGPT, and it is only through in-context learning that GPT4 can achieve performance comparable to Baichuan2-7B fine-tuned on CSCD-NS.", + "section": null, + "intents": [] + }, + { + "ref_id": "c080f022c6f57d10dc8e8e44231c8995480b884d", + "citation_marker": "FCGEC: Fine-Grained Corpus for Chinese Grammatical Error Correction", + "text": "While there have been some efforts to develop Chinese grammatical error correction (CGEC) datasets for native speakers (Ma et al., 2022; Xu et al., 2022; Zhao et al., 2022; Wang et al., 2022), no such work has been undertaken for CSC datasets.", + "section": null, + "intents": [ + "background" + ] + }, + { + "ref_id": "4832eb72b07a1fd473df299057a98b0b0d84ed97", + "citation_marker": "Overview of CTC 2021: Chinese Text Correction for Native Speakers", + "text": "While there have been some efforts to develop Chinese grammatical error correction (CGEC) datasets for native speakers (Ma et al., 2022; Xu et al., 2022; Zhao et al., 2022; Wang et al., 2022), no such work has been undertaken for CSC datasets.", + "section": null, + "intents": [ + "background" + ] + }, + { + "ref_id": "c32b652d605d9f097edf073f69253573b2b5d0cf", + "citation_marker": "Exploring and Adapting Chinese GPT to Pinyin Input Method", + "text": "For instance, individuals may utilize abbreviated pinyin to input common phrases, entering only the initials of characters (e.g., \"wm\" for \" \u6211 \u4eec \") (Tan et al., 2022).", + "section": null, + "intents": [ + "background" + ] + }, + { + "ref_id": "a8ca46b171467ceb2d7652fbfb67fe701ad86092", + "citation_marker": "LoRA: Low-Rank Adaptation of Large Language Models", + "text": "We fine-tune the model by LoRA (Hu et al., 2021) and use the version of 7B and 13B here 6 .", + "section": null, + "intents": [ + "methodology" + ] + }, + { + "ref_id": "93bf0b32b17297d94f6a2cc35bce7c396eb17b36", + "citation_marker": "Spelling Error Correction with Soft-Masked BERT", + "text": "The development of CSC models, from BERT to PLOME, has primarily focused on optimizing character-level errors, with little progress made in addressing word-level errors.", + "section": null, + "intents": [] + }, + { + "ref_id": "93bf0b32b17297d94f6a2cc35bce7c396eb17b36", + "citation_marker": "Spelling Error Correction with Soft-Masked BERT", + "text": "The mainstream method is based on confusion sets (Liu et al., 2021; Zhang et al., 2020), the pseudo data generated in this way is large in size but low in quality because context information is not considered.", + "section": null, + "intents": [ + "methodology" + ] + }, + { + "ref_id": "93bf0b32b17297d94f6a2cc35bce7c396eb17b36", + "citation_marker": "Spelling Error Correction with Soft-Masked BERT", + "text": "Moreover, all baseline systems, which are based on pre-trained language models, exhibit a propensity to over-convert low-frequency expressions into more prevalent ones (Zhang et al., 2020; Liu et al., 2022).", + "section": null, + "intents": [ + "background" + ] + }, + { + "ref_id": "93bf0b32b17297d94f6a2cc35bce7c396eb17b36", + "citation_marker": "Spelling Error Correction with Soft-Masked BERT", + "text": "(1) BERT (Devlin et al., 2019) (2) Soft-Masked BERT (SM BERT) (Zhang et al., 2020) employs an error detection model to provide better correction guidance.", + "section": null, + "intents": [] + }, + { + "ref_id": "93bf0b32b17297d94f6a2cc35bce7c396eb17b36", + "citation_marker": "Spelling Error Correction with Soft-Masked BERT", + "text": "We choose the basic BERT model to conduct the experiment and train the model only on the pseudo data to clearly distinguish the differences.", + "section": null, + "intents": [] + }, + { + "ref_id": "93bf0b32b17297d94f6a2cc35bce7c396eb17b36", + "citation_marker": "Spelling Error Correction with Soft-Masked BERT", + "text": "BERT-like classification models have difficulty in addressing complex word-level errors and equal-length grammatical errors, as these require a strong contextual understanding.", + "section": null, + "intents": [] + }, + { + "ref_id": "93bf0b32b17297d94f6a2cc35bce7c396eb17b36", + "citation_marker": "Spelling Error Correction with Soft-Masked BERT", + "text": "Our experiments demonstrate that BERT-like models exhibit better performance than generative models, but there is still a considerable room for improvement.", + "section": null, + "intents": [] + }, + { + "ref_id": "93bf0b32b17297d94f6a2cc35bce7c396eb17b36", + "citation_marker": "Spelling Error Correction with Soft-Masked BERT", + "text": "(1) As shown in Table 4, compared with generative models, BERT-like token-level classification models (BERT, SM BERT, PLOME) remain the best approach for the CSC task, with smaller model size, higher performance, and faster inference speed.", + "section": null, + "intents": [] + }, + { + "ref_id": "93bf0b32b17297d94f6a2cc35bce7c396eb17b36", + "citation_marker": "Spelling Error Correction with Soft-Masked BERT", + "text": "We provide a detailed analysis in Appendix A. CSC models : In recent years, BERT-like (De-vlin et al., 2019) classification models have dominated the research of the CSC task (Hong et al., 2019; Zhu et al., 2022; Huang et al., 2021; Zhang et al., 2020; Liu et al., 2021, 2022).", + "section": null, + "intents": [ + "methodology" + ] + }, + { + "ref_id": "93bf0b32b17297d94f6a2cc35bce7c396eb17b36", + "citation_marker": "Spelling Error Correction with Soft-Masked BERT", + "text": "The results demonstrate that BERT-like classification models outperform generative models, as the latter struggle with the simultaneous constraints of text length and pronunciation.", + "section": null, + "intents": [] + }, + { + "ref_id": "93bf0b32b17297d94f6a2cc35bce7c396eb17b36", + "citation_marker": "Spelling Error Correction with Soft-Masked BERT", + "text": "Since there is no official implementation for BERT and SM BERT, we follow a widely-used open-source version 12 .", + "section": null, + "intents": [] + }, + { + "ref_id": "93bf0b32b17297d94f6a2cc35bce7c396eb17b36", + "citation_marker": "Spelling Error Correction with Soft-Masked BERT", + "text": "However, these meth-ods, which rely on confusion sets (Liu et al., 2021; Zhang et al., 2020) or ASR transcriptions (Wang et al., 2018), do not align with the real-world input scenario.", + "section": null, + "intents": [ + "background" + ] + }, + { + "ref_id": "395de0bd3837fdf4b4b5e5f04835bcc69c279481", + "citation_marker": "BART: Denoising Sequence-to-Sequence Pre-training for Natural Language Generation, Translation, and Comprehension", + "text": "We use the Chinese BART-large version here 5 .", + "section": null, + "intents": [] + }, + { + "ref_id": "395de0bd3837fdf4b4b5e5f04835bcc69c279481", + "citation_marker": "BART: Denoising Sequence-to-Sequence Pre-training for Natural Language Generation, Translation, and Comprehension", + "text": "To ensure that the correction results are of the same length as the input text, we only extract equal-length substitution modifications for generative models (BART, Baichuan2, ChatGPT and GPT4).", + "section": null, + "intents": [] + }, + { + "ref_id": "395de0bd3837fdf4b4b5e5f04835bcc69c279481", + "citation_marker": "BART: Denoising Sequence-to-Sequence Pre-training for Natural Language Generation, Translation, and Comprehension", + "text": "( 4) BART (Lewis et al., 2020) models the CSC as a sequence-to-sequence task.", + "section": null, + "intents": [ + "methodology" + ] + }, + { + "ref_id": "395de0bd3837fdf4b4b5e5f04835bcc69c279481", + "citation_marker": "BART: Denoising Sequence-to-Sequence Pre-training for Natural Language Generation, Translation, and Comprehension", + "text": "We choose the Chinese BART-large model as the base model and fine-tune it for the CSC task by treating it as a sequence-to-sequence task.", + "section": null, + "intents": [] + }, + { + "ref_id": "395de0bd3837fdf4b4b5e5f04835bcc69c279481", + "citation_marker": "BART: Denoising Sequence-to-Sequence Pre-training for Natural Language Generation, Translation, and Comprehension", + "text": "This trend can be observed from smaller models like BART (0.4B) to larger ones such as Baichuan2-13B.", + "section": null, + "intents": [] + }, + { + "ref_id": "fd273b87fe73800f04b51a5bf442df924a448f5f", + "citation_marker": "Chinese Lexical Analysis with Deep Bi-GRU-CRF Network", + "text": "To address these issues, we utilize a Chinese lexical analysis tool (LAC) (Jiao et al., 2018) to identify these particles and entities in the input sentence.", + "section": null, + "intents": [ + "methodology" + ] + }, + { + "ref_id": "fd273b87fe73800f04b51a5bf442df924a448f5f", + "citation_marker": "Chinese Lexical Analysis with Deep Bi-GRU-CRF Network", + "text": "As shown in Table 2, we first tokenize the correct sentence using LAC (Jiao et al., 2018) to obtain word-level correction pairs.", + "section": null, + "intents": [ + "methodology" + ] + }, + { + "ref_id": "1294f49cf1a7397f0423ec617a78c7995139bc5b", + "citation_marker": "Neural Network Language Model for Chinese Pinyin Input Method Engine", + "text": "Leveraging the powerful language model of the IME, candidates are recommended by considering both the context before token C t and the pronunciation p (Chen et al., 2015).", + "section": null, + "intents": [ + "methodology" + ] + }, + { + "ref_id": "d590ca98fdb2c727c492fbc6f108ce1ce4efb20f", + "citation_marker": "Introduction to SIGHAN 2015 Bake-off for Chinese Spelling Check", + "text": "Existing CSC datasets, such as SIGHAN13, 14, and 15 (Wu et al., 2013; Yu et al., 2014; Tseng et al., 2015), are all sourced from Chinese learners.", + "section": null, + "intents": [ + "background" + ] + }, + { + "ref_id": "d590ca98fdb2c727c492fbc6f108ce1ce4efb20f", + "citation_marker": "Introduction to SIGHAN 2015 Bake-off for Chinese Spelling Check", + "text": "CSC Datasets : The existing CSC datasets, such as the SIGHAN series (Wu et al., 2013; Yu et al., 2014; Tseng et al., 2015), primarily cater to Chinese learners.", + "section": null, + "intents": [] + }, + { + "ref_id": "b122a828f5fee3c6afc54e70f41b00184d6383fc", + "citation_marker": "LCSTS: A Large Scale Chinese Short Text Summarization Dataset", + "text": "The substantial scale and scope of the LCSTS make it suitable to serve as the data source.", + "section": null, + "intents": [] + }, + { + "ref_id": "b122a828f5fee3c6afc54e70f41b00184d6383fc", + "citation_marker": "LCSTS: A Large Scale Chinese Short Text Summarization Dataset", + "text": "We apply the above method to construct a large-scale CSC pseudo dataset LCSTS-IME-2M, consisting of about 2 million samples, based on the correct sentences filtered from LCSTS, the error distribution of CSCD-NS, and the Google IME 4 .", + "section": null, + "intents": [] + }, + { + "ref_id": "b122a828f5fee3c6afc54e70f41b00184d6383fc", + "citation_marker": "LCSTS: A Large Scale Chinese Short Text Summarization Dataset", + "text": "We split posts in LCSTS into sentences and obtain over 8 million sentences.", + "section": null, + "intents": [] + }, + { + "ref_id": "b122a828f5fee3c6afc54e70f41b00184d6383fc", + "citation_marker": "LCSTS: A Large Scale Chinese Short Text Summarization Dataset", + "text": "Data : We perform experiments based on the labled data CSCD-NS and the pseudo data LCSTS-IME-4 2M.", + "section": null, + "intents": [] + }, + { + "ref_id": "b122a828f5fee3c6afc54e70f41b00184d6383fc", + "citation_marker": "LCSTS: A Large Scale Chinese Short Text Summarization Dataset", + "text": "We chose the LCSTS dataset (Hu et al., 2015) as our data source.", + "section": null, + "intents": [ + "methodology" + ] + }, + { + "ref_id": "b122a828f5fee3c6afc54e70f41b00184d6383fc", + "citation_marker": "LCSTS: A Large Scale Chinese Short Text Summarization Dataset", + "text": "To conduct an in-depth investigation into the distribution of spelling errors, we develop a tagging system that Figure 2: An authentic Weibo post from LCSTS, where the phrase \" \u6548 \u529b \u4e8e \" is mistakenly written as \" \u6548 \u529b \u4e0e \". operates at phonetic and semantic levels.", + "section": null, + "intents": [] + }, + { + "ref_id": "b122a828f5fee3c6afc54e70f41b00184d6383fc", + "citation_marker": "LCSTS: A Large Scale Chinese Short Text Summarization Dataset", + "text": "License: CSCD-NS and the constructed pseudo-data LCSTS-IME-2M are based on LCSTS (Hu et al., 2015), we applied for and obtained the right to use this dataset, and performed the academic research under the copyright.", + "section": null, + "intents": [ + "methodology" + ] + }, + { + "ref_id": "02ff89d27d8cc4b372fff259da41c1f87a6e8506", + "citation_marker": "Overview of SIGHAN 2014 Bake-off for Chinese Spelling Check", + "text": "Existing CSC datasets, such as SIGHAN13, 14, and 15 (Wu et al., 2013; Yu et al., 2014; Tseng et al., 2015), are all sourced from Chinese learners.", + "section": null, + "intents": [ + "background" + ] + }, + { + "ref_id": "02ff89d27d8cc4b372fff259da41c1f87a6e8506", + "citation_marker": "Overview of SIGHAN 2014 Bake-off for Chinese Spelling Check", + "text": "CSC Datasets : The existing CSC datasets, such as the SIGHAN series (Wu et al., 2013; Yu et al., 2014; Tseng et al., 2015), primarily cater to Chinese learners.", + "section": null, + "intents": [] + }, + { + "ref_id": "4503c3b75bf515098184d993790639c26f75da3a", + "citation_marker": "A Joint Graph Model for Pinyin-to-Chinese Conversion with Typo Correction", + "text": "This is because even if users make slight mistakes in their pinyin input, Chinese IME will auto-fix the input pinyin based on the context (Jia and Zhao, 2014).", + "section": null, + "intents": [ + "background" + ] + }, + { + "ref_id": "6268daca3037d6cb40f8193b8636aa97efe1287f", + "citation_marker": "Chinese Spelling Check Evaluation at SIGHAN Bake-off 2013", + "text": "Existing CSC datasets, such as SIGHAN13, 14, and 15 (Wu et al., 2013; Yu et al., 2014; Tseng et al., 2015), are all sourced from Chinese learners.", + "section": null, + "intents": [ + "background" + ] + }, + { + "ref_id": "6268daca3037d6cb40f8193b8636aa97efe1287f", + "citation_marker": "Chinese Spelling Check Evaluation at SIGHAN Bake-off 2013", + "text": "CSC Datasets : The existing CSC datasets, such as the SIGHAN series (Wu et al., 2013; Yu et al., 2014; Tseng et al., 2015), primarily cater to Chinese learners.", + "section": null, + "intents": [] + }, + { + "ref_id": "8aa98fbfb6f1e979dead13ce24075503fe47658e", + "citation_marker": "A Survey for In-context Learning", + "text": "(6) ChatGPT and GPT4 perform the CSC task in a few-shot setting (10 examples) through in-context learning (ICL) (Dong et al., 2022).", + "section": null, + "intents": [ + "methodology" + ] + }, + { + "ref_id": "a27df49d3c8b1d96923e2e2eaa9bbfbe3ae732e7", + "citation_marker": "MDCSpell: A Multi-task Detector-Corrector Framework for Chinese Spelling Correction", + "text": "We provide a detailed analysis in Appendix A. CSC models : In recent years, BERT-like (De-vlin et al., 2019) classification models have dominated the research of the CSC task (Hong et al., 2019; Zhu et al., 2022; Huang et al., 2021; Zhang et al., 2020; Liu et al., 2021, 2022).", + "section": null, + "intents": [ + "methodology" + ] + }, + { + "ref_id": "a77d34655475fd6eac0ae6ab724a3bd4b36102a6", + "citation_marker": "CRASpell: A Contextual Typo Robust Approach to Improve Chinese Spelling Correction", + "text": "Moreover, all baseline systems, which are based on pre-trained language models, exhibit a propensity to over-convert low-frequency expressions into more prevalent ones (Zhang et al., 2020; Liu et al., 2022).", + "section": null, + "intents": [ + "background" + ] + }, + { + "ref_id": "cf6be786c307a8c4e523daa455019a6b7650537c", + "citation_marker": "PLOME: Pre-training with Misspelled Knowledge for Chinese Spelling Correction", + "text": "( 3) PLOME (Liu et al., 2021) integrates phonetic and visual features into the pre-trained model.", + "section": null, + "intents": [ + "background" + ] + }, + { + "ref_id": "cf6be786c307a8c4e523daa455019a6b7650537c", + "citation_marker": "PLOME: Pre-training with Misspelled Knowledge for Chinese Spelling Correction", + "text": "The development of CSC models, from BERT to PLOME, has primarily focused on optimizing character-level errors, with little progress made in addressing word-level errors.", + "section": null, + "intents": [] + }, + { + "ref_id": "cf6be786c307a8c4e523daa455019a6b7650537c", + "citation_marker": "PLOME: Pre-training with Misspelled Knowledge for Chinese Spelling Correction", + "text": "The mainstream method is based on confusion sets (Liu et al., 2021; Zhang et al., 2020), the pseudo data generated in this way is large in size but low in quality because context information is not considered.", + "section": null, + "intents": [ + "methodology" + ] + }, + { + "ref_id": "cf6be786c307a8c4e523daa455019a6b7650537c", + "citation_marker": "PLOME: Pre-training with Misspelled Knowledge for Chinese Spelling Correction", + "text": "For example, the PLOME model shows a recall rate of only 60% for word-level errors and merely 44% for particle-related grammatical errors ( \u7684 / \u5730 / \u5f97 ).", + "section": null, + "intents": [] + }, + { + "ref_id": "cf6be786c307a8c4e523daa455019a6b7650537c", + "citation_marker": "PLOME: Pre-training with Misspelled Knowledge for Chinese Spelling Correction", + "text": "For PLOME, we directly utilize the official code 13 .", + "section": null, + "intents": [] + }, + { + "ref_id": "cf6be786c307a8c4e523daa455019a6b7650537c", + "citation_marker": "PLOME: Pre-training with Misspelled Knowledge for Chinese Spelling Correction", + "text": "(1) As shown in Table 4, compared with generative models, BERT-like token-level classification models (BERT, SM BERT, PLOME) remain the best approach for the CSC task, with smaller model size, higher performance, and faster inference speed.", + "section": null, + "intents": [] + }, + { + "ref_id": "cf6be786c307a8c4e523daa455019a6b7650537c", + "citation_marker": "PLOME: Pre-training with Misspelled Knowledge for Chinese Spelling Correction", + "text": "We provide a detailed analysis in Appendix A. CSC models : In recent years, BERT-like (De-vlin et al., 2019) classification models have dominated the research of the CSC task (Hong et al., 2019; Zhu et al., 2022; Huang et al., 2021; Zhang et al., 2020; Liu et al., 2021, 2022).", + "section": null, + "intents": [ + "methodology" + ] + }, + { + "ref_id": "cf6be786c307a8c4e523daa455019a6b7650537c", + "citation_marker": "PLOME: Pre-training with Misspelled Knowledge for Chinese Spelling Correction", + "text": "However, these meth-ods, which rely on confusion sets (Liu et al., 2021; Zhang et al., 2020) or ASR transcriptions (Wang et al., 2018), do not align with the real-world input scenario.", + "section": null, + "intents": [ + "background" + ] + }, + { + "ref_id": "9998875164c1d46c8c41be6f22171e90abf0ccbe", + "citation_marker": "PHMOSpell: Phonological and Morphological Knowledge Guided Chinese Spelling Check", + "text": "We provide a detailed analysis in Appendix A. CSC models : In recent years, BERT-like (De-vlin et al., 2019) classification models have dominated the research of the CSC task (Hong et al., 2019; Zhu et al., 2022; Huang et al., 2021; Zhang et al., 2020; Liu et al., 2021, 2022).", + "section": null, + "intents": [ + "methodology" + ] + }, + { + "ref_id": "2119f2e7497cacebc43ea04d0d8df96df9d08e03", + "citation_marker": "Overview of NLPTEA-2020 Shared Task for Chinese Grammatical Error Diagnosis", + "text": "Training : We follow the successful experience (Wang et al., 2020) of the NLPTEA2020 task (Rao et al., 2020) and use a Chinese ELECTRA-Large discriminator model 2 (Clark et al., 2020) to initialize the detection model.", + "section": null, + "intents": [ + "methodology" + ] + }, + { + "ref_id": "e8fe671e2e04c1b45a5dddc51c1e60768a66b628", + "citation_marker": "Combining ResNet and Transformer for Chinese Grammatical Error Diagnosis", + "text": "Training : We follow the successful experience (Wang et al., 2020) of the NLPTEA2020 task (Rao et al., 2020) and use a Chinese ELECTRA-Large discriminator model 2 (Clark et al., 2020) to initialize the detection model.", + "section": null, + "intents": [ + "methodology" + ] + }, + { + "ref_id": null, + "citation_marker": "ELECTRA: Pre-training text encoders as discriminators rather than generators", + "text": "Training : We follow the successful experience (Wang et al., 2020) of the NLPTEA2020 task (Rao et al., 2020) and use a Chinese ELECTRA-Large discriminator model 2 (Clark et al., 2020) to initialize the detection model.", + "section": null, + "intents": [ + "methodology" + ] + }, + { + "ref_id": "df2b0e26d0599ce3e70df8a9da02e51594e0e992", + "citation_marker": "BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding", + "text": "The development of CSC models, from BERT to PLOME, has primarily focused on optimizing character-level errors, with little progress made in addressing word-level errors.", + "section": null, + "intents": [] + }, + { + "ref_id": "df2b0e26d0599ce3e70df8a9da02e51594e0e992", + "citation_marker": "BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding", + "text": "(1) BERT (Devlin et al., 2019) (2) Soft-Masked BERT (SM BERT) (Zhang et al., 2020) employs an error detection model to provide better correction guidance.", + "section": null, + "intents": [] + }, + { + "ref_id": "df2b0e26d0599ce3e70df8a9da02e51594e0e992", + "citation_marker": "BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding", + "text": "We choose the basic BERT model to conduct the experiment and train the model only on the pseudo data to clearly distinguish the differences.", + "section": null, + "intents": [] + } +] diff --git a/replay_traces/2211.08788/two_pass_outputs/20260726T211957Z/pass_1_reasoning.selected.md b/replay_traces/2211.08788/two_pass_outputs/20260726T211957Z/pass_1_reasoning.selected.md new file mode 100644 index 0000000000000000000000000000000000000000..20a9c79a92e9dfe17f9b02e357506e285418460f --- /dev/null +++ b/replay_traces/2211.08788/two_pass_outputs/20260726T211957Z/pass_1_reasoning.selected.md @@ -0,0 +1,160 @@ +# 1. Cluster Evidence + +The paper introduces two main contributions: the CSCD-NS dataset (a manually annotated Chinese spelling check dataset for native speakers) and an IME-based pseudo-data generation method. + +Based on the downstream cluster evidence, reuse is heavily concentrated on the **CSCD-NS dataset** itself: +- Cluster C2 (count: 7) shows downstream work using the target dataset for model training and evaluation. +- Cluster C1 (count: 1) shows downstream work using the target paper's evaluation script to compute metrics. + +Because the evaluation script is simply the operational tooling used to benchmark models on the released dataset, these clusters reflect the reuse of a single integrated artifact. Therefore, the correct annotation is one dataset claim. + +--- + +# 2. Claim Split Decision + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C2 + +### Why this split is correct +The downstream evidence points to the reuse of the CSCD-NS dataset and its associated evaluation script. There is no distinct downstream cluster indicating independent reuse of the IME-based pseudo-data generation pipeline. Thus, the paper's contributions should be represented as a single atomic dataset claim. + +--- + +# 3. Rewritten Claims + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **rewritten_claim:** **Dataset: A large-scale Chinese spelling check dataset sourced from authentic native-speaker social media posts, enabling the training and evaluation of spelling correction models on real-world error distributions.** +- **why_this_is_atomic:** This claim isolates the released dataset and its evaluation role as the primary downstream-used artifact, without bundling the separate pseudo-data generation method that lacks independent downstream evidence. +- **optional cluster_id:** C2 +- **decision:** YES_SUFFICIENT + +--- + +# 4. Minimal Ingredient Set Per Claim + +## Claim C1: CSCD-NS Dataset + +### Ingredient C1.I1 +- **ingredient_id:** C1.I1 +- **ingredient:** Large-scale corpus of authentic Chinese social media posts +- **why_structurally_necessary:** The dataset is explicitly designed to capture the real-world spelling errors made by native speakers. Without a massive raw text substrate of authentic social media posts, the dataset could not be constructed in its claimed form. +- **why_not_lower_level_substeps:** This should remain a high-level data source ingredient rather than being split into specific post topics or sentence-splitting steps. +- **why_not_adjacent_implementation_details:** The structural dependency is the raw social media text itself, not the specific scraping or formatting details. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** LCSTS: A Large Scale Chinese Short Text Summarization Dataset +- **additional groundings if any:** none +- **role:** DATA_SOURCE +- **contribution:** Provides the raw, authentic native-speaker text substrate from which candidate sentences with spelling errors are extracted. +- **rationale:** To build a dataset reflecting native-speaker errors, the authors require a massive corpus of real-world input. The paper explicitly uses the LCSTS dataset (sourced from Weibo) as this foundational text substrate. +- **evidence_span:** “We chose the LCSTS dataset (Hu et al., 2015) as our data source. This dataset is composed of authentic Weibo posts” + +### Ingredient C1.I2 +- **ingredient_id:** C1.I2 +- **ingredient:** Pre-trained language model for automated error-detection filtering +- **why_structurally_necessary:** Finding naturally occurring spelling errors in 8 million sentences is intractable without automated filtering. The dataset relies on an initial detection model to filter out the 91.2% of sentences that are completely correct, making manual annotation feasible. +- **why_not_lower_level_substeps:** This should remain one filtering-tool ingredient rather than being split into the specific SIGHAN training datasets or the exact probability thresholds used. +- **why_not_adjacent_implementation_details:** The structural dependency is the automated filtering capability, not the specific hyperparameters of the detection model. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** ELECTRA: Pre-training text encoders as discriminators rather than generators +- **additional groundings if any:** none +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the pre-trained discriminator architecture used to build the error detection model that filters the massive raw corpus down to a manageable candidate set. +- **rationale:** The construction pipeline requires a strong detection model to identify candidate sentences. The paper explicitly initializes this filtering model using a pre-trained Chinese ELECTRA-Large discriminator. +- **evidence_span:** “use a Chinese ELECTRA-Large discriminator model (Clark et al., 2020) to initialize the detection model.” + +### Ingredient C1.I3 +- **ingredient_id:** C1.I3 +- **ingredient:** Human annotation and verification protocol for spelling correction +- **why_structurally_necessary:** The defining property of CSCD-NS is that it is a high-quality, manually annotated dataset. Without human annotators to verify errors and provide the unique correct sentences, it would merely be a noisy, machine-filtered corpus. +- **why_not_lower_level_substeps:** This should remain one protocol-level ingredient rather than being split into the specific rules for discarding ambiguous sentences or the senior annotator conflict-resolution steps. +- **why_not_adjacent_implementation_details:** The structural dependency is the human verification workflow itself, not the specific payment or interface details. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the manual verification and correction workflow that ensures the dataset contains genuine errors with unique, accurate ground-truth corrections. +- **rationale:** The dataset's reliability depends on a rigorous human annotation process. Because this specific annotation protocol is designed and executed in the target paper, it maps cleanly to `NONE`. +- **evidence_span:** “We recruit a group of native speakers for manual annotation. The annotators are required to check whether the given sentence contains any spelling errors” + +--- + +# 5. Excluded Tempting Non-Ingredients + +### Claim C1 exclusions +- **IME-based pseudo data generation method (LCSTS-IME-2M)** + - Excluded because the downstream evidence specifically points to the reuse of the manually annotated CSCD-NS dataset, not the pseudo dataset or its generation pipeline. +- **Chinese Lexical Analysis tool (LAC)** + - Excluded because it is a local heuristic used to adjust filtering thresholds for specific particles and entities, not a core structural dependency of the dataset itself. +- **SIGHAN datasets (SIGHAN13, 14, 15)** + - Excluded as a top-level ingredient because they are only used as training data for the intermediate filtering model, making them subordinate to the filtering pipeline itself rather than a direct ingredient of the final dataset. + +--- + +# 6. Candidate Grounding Decisions + +## Claim C1: CSCD-NS Dataset + +### Ingredient C1.I1 +- **Candidate:** LCSTS: A Large Scale Chinese Short Text Summarization Dataset +- **decision:** accepted_canonical +- **why:** It is the explicit and direct source of the raw Weibo posts used to construct the dataset. + +### Ingredient C1.I2 +- **Candidate:** ELECTRA: Pre-training text encoders as discriminators rather than generators +- **decision:** accepted_canonical +- **why:** It is the specific pre-trained architecture used to initialize the detection model that makes filtering the massive corpus feasible. + +### Ingredient C1.I3 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The human annotation and verification protocol is specific to this dataset and implemented entirely within the target paper. + +--- + +# 7. Final Grounding Decisions + +## Claim C1 +- **C1.I1** → LCSTS: A Large Scale Chinese Short Text Summarization Dataset → DATA_SOURCE +- **C1.I2** → ELECTRA: Pre-training text encoders as discriminators rather than generators → IMPLEMENTATION_TOOLING +- **C1.I3** → NONE → EVALUATION_PROTOCOL + +--- + +# 8. Final Rationales And Evidence + +## Claim C1: CSCD-NS Dataset + +### C1.I1 +- **ingredient:** Large-scale corpus of authentic Chinese social media posts +- **canonical study:** LCSTS: A Large Scale Chinese Short Text Summarization Dataset +- **role:** DATA_SOURCE +- **contribution:** Provides the raw, authentic native-speaker text substrate from which candidate sentences with spelling errors are extracted. +- **rationale:** To build a dataset reflecting native-speaker errors, the authors require a massive corpus of real-world input. The paper explicitly uses the LCSTS dataset (sourced from Weibo) as this foundational text substrate. +- **evidence_span:** “We chose the LCSTS dataset (Hu et al., 2015) as our data source. This dataset is composed of authentic Weibo posts” + +### C1.I2 +- **ingredient:** Pre-trained language model for automated error-detection filtering +- **canonical study:** ELECTRA: Pre-training text encoders as discriminators rather than generators +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the pre-trained discriminator architecture used to build the error detection model that filters the massive raw corpus down to a manageable candidate set. +- **rationale:** The construction pipeline requires a strong detection model to identify candidate sentences. The paper explicitly initializes this filtering model using a pre-trained Chinese ELECTRA-Large discriminator. +- **evidence_span:** “use a Chinese ELECTRA-Large discriminator model (Clark et al., 2020) to initialize the detection model.” + +### C1.I3 +- **ingredient:** Human annotation and verification protocol for spelling correction +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the manual verification and correction workflow that ensures the dataset contains genuine errors with unique, accurate ground-truth corrections. +- **rationale:** The dataset's reliability depends on a rigorous human annotation process. Because this specific annotation protocol is designed and executed in the target paper, it maps cleanly to `NONE`. +- **evidence_span:** “We recruit a group of native speakers for manual annotation. The annotators are required to check whether the given sentence contains any spelling errors” \ No newline at end of file diff --git a/replay_traces/2211.08788/two_pass_outputs/20260726T211957Z/pass_2_formatter.prompt.txt b/replay_traces/2211.08788/two_pass_outputs/20260726T211957Z/pass_2_formatter.prompt.txt new file mode 100644 index 0000000000000000000000000000000000000000..67f74bba37b17bd3f089db3e189a77808dc3ebfa --- /dev/null +++ b/replay_traces/2211.08788/two_pass_outputs/20260726T211957Z/pass_2_formatter.prompt.txt @@ -0,0 +1,302 @@ +Task: Convert the reasoning memo into the final annotation UI JSON payload. + +This is a strict formatting pass. +Do not do new reasoning unless needed to resolve minor schema ambiguities. +Follow the memo’s final decisions faithfully. + +Critical formatting priorities: +1. Preserve the final claim split. +2. Preserve the final minimal ingredient set. +3. Respect exclusions of tempting non-ingredients. +4. Preserve canonical vs additional vs NONE decisions exactly. +5. Do not inflate the ingredient set. + +Target schema: +{ + "target_paper_id": "...", + "target_title": "...", + "target_year": 2024, + "annotator_id": "replay_trace_builder", + "active_claim_id": "...", + "claims": [ + { + "claim_id": "...", + "text": "...", + "rewritten_claim": "...", + "cluster_id": "...", + "decision": "YES_SUFFICIENT | NO_NOT_DISCOVERY | UNCERTAIN", + "notes": "", + "ingredients": [ + { + "ingredient_id": "...", + "ingredient": "...", + "canonical_ref_id": "__NONE__ or ref_id", + "canonical_grounding": null or { + "ref_id": "...", + "bib_key": "...", + "paper_id": "...", + "external_ids": {}, + "ref_title": "...", + "ref_year": "...", + "ref_authors": "..." + }, + "additional_ref_ids": ["..."], + "additional_groundings": [ + { + "ref_id": "...", + "bib_key": "...", + "paper_id": "...", + "external_ids": {}, + "ref_title": "...", + "ref_year": "...", + "ref_authors": "..." + } + ], + "canonical_annotation": { + "role": null or "CONCEPTUAL_FRAMEWORK" or "CORE_METHOD" or "DATA_SOURCE" or "MODEL_INITIALIZATION" or "EVALUATION_PROTOCOL" or "IMPLEMENTATION_TOOLING" or "TRAINING_DATA", + "roles": ["..."], + "contribution": "...", + "rationale": "...", + "evidence_span": "..." + } + } + ], + "enabling_discoveries": [ + { + "ref_id": "...", + "bib_key": "...", + "paper_id": "...", + "external_ids": {}, + "ref_title": "...", + "ref_year": "...", + "ref_authors": "...", + "ingredient_id": "...", + "ingredient": "...", + "role": null or "CONCEPTUAL_FRAMEWORK" or "CORE_METHOD" or "DATA_SOURCE" or "MODEL_INITIALIZATION" or "EVALUATION_PROTOCOL" or "IMPLEMENTATION_TOOLING" or "TRAINING_DATA", + "roles": ["..."], + "contribution": "...", + "rationale": "...", + "evidence_span": "..." + } + ] + } + ] +} + +Formatting rules: +- `text` must equal `rewritten_claim` +- `active_claim_id` should be the first claim_id unless the reasoning memo explicitly recommends another active claim +- if canonical grounding is NONE: + - `canonical_ref_id` must be "__NONE__" + - `canonical_grounding` must be null + - do not include that ingredient in `enabling_discoveries` +- if roles has length 1, `role` should equal that role +- if roles has length != 1, `role` should be null +- `additional_ref_ids` must match `additional_groundings[].ref_id` +- `enabling_discoveries` should contain only canonical non-NONE ingredients +- use only the allowed role set + +REASONING MEMO: +# 1. Cluster Evidence + +The paper introduces two main contributions: the CSCD-NS dataset (a manually annotated Chinese spelling check dataset for native speakers) and an IME-based pseudo-data generation method. + +Based on the downstream cluster evidence, reuse is heavily concentrated on the **CSCD-NS dataset** itself: +- Cluster C2 (count: 7) shows downstream work using the target dataset for model training and evaluation. +- Cluster C1 (count: 1) shows downstream work using the target paper's evaluation script to compute metrics. + +Because the evaluation script is simply the operational tooling used to benchmark models on the released dataset, these clusters reflect the reuse of a single integrated artifact. Therefore, the correct annotation is one dataset claim. + +--- + +# 2. Claim Split Decision + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C2 + +### Why this split is correct +The downstream evidence points to the reuse of the CSCD-NS dataset and its associated evaluation script. There is no distinct downstream cluster indicating independent reuse of the IME-based pseudo-data generation pipeline. Thus, the paper's contributions should be represented as a single atomic dataset claim. + +--- + +# 3. Rewritten Claims + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **rewritten_claim:** **Dataset: A large-scale Chinese spelling check dataset sourced from authentic native-speaker social media posts, enabling the training and evaluation of spelling correction models on real-world error distributions.** +- **why_this_is_atomic:** This claim isolates the released dataset and its evaluation role as the primary downstream-used artifact, without bundling the separate pseudo-data generation method that lacks independent downstream evidence. +- **optional cluster_id:** C2 +- **decision:** YES_SUFFICIENT + +--- + +# 4. Minimal Ingredient Set Per Claim + +## Claim C1: CSCD-NS Dataset + +### Ingredient C1.I1 +- **ingredient_id:** C1.I1 +- **ingredient:** Large-scale corpus of authentic Chinese social media posts +- **why_structurally_necessary:** The dataset is explicitly designed to capture the real-world spelling errors made by native speakers. Without a massive raw text substrate of authentic social media posts, the dataset could not be constructed in its claimed form. +- **why_not_lower_level_substeps:** This should remain a high-level data source ingredient rather than being split into specific post topics or sentence-splitting steps. +- **why_not_adjacent_implementation_details:** The structural dependency is the raw social media text itself, not the specific scraping or formatting details. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** LCSTS: A Large Scale Chinese Short Text Summarization Dataset +- **additional groundings if any:** none +- **role:** DATA_SOURCE +- **contribution:** Provides the raw, authentic native-speaker text substrate from which candidate sentences with spelling errors are extracted. +- **rationale:** To build a dataset reflecting native-speaker errors, the authors require a massive corpus of real-world input. The paper explicitly uses the LCSTS dataset (sourced from Weibo) as this foundational text substrate. +- **evidence_span:** “We chose the LCSTS dataset (Hu et al., 2015) as our data source. This dataset is composed of authentic Weibo posts” + +### Ingredient C1.I2 +- **ingredient_id:** C1.I2 +- **ingredient:** Pre-trained language model for automated error-detection filtering +- **why_structurally_necessary:** Finding naturally occurring spelling errors in 8 million sentences is intractable without automated filtering. The dataset relies on an initial detection model to filter out the 91.2% of sentences that are completely correct, making manual annotation feasible. +- **why_not_lower_level_substeps:** This should remain one filtering-tool ingredient rather than being split into the specific SIGHAN training datasets or the exact probability thresholds used. +- **why_not_adjacent_implementation_details:** The structural dependency is the automated filtering capability, not the specific hyperparameters of the detection model. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** ELECTRA: Pre-training text encoders as discriminators rather than generators +- **additional groundings if any:** none +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the pre-trained discriminator architecture used to build the error detection model that filters the massive raw corpus down to a manageable candidate set. +- **rationale:** The construction pipeline requires a strong detection model to identify candidate sentences. The paper explicitly initializes this filtering model using a pre-trained Chinese ELECTRA-Large discriminator. +- **evidence_span:** “use a Chinese ELECTRA-Large discriminator model (Clark et al., 2020) to initialize the detection model.” + +### Ingredient C1.I3 +- **ingredient_id:** C1.I3 +- **ingredient:** Human annotation and verification protocol for spelling correction +- **why_structurally_necessary:** The defining property of CSCD-NS is that it is a high-quality, manually annotated dataset. Without human annotators to verify errors and provide the unique correct sentences, it would merely be a noisy, machine-filtered corpus. +- **why_not_lower_level_substeps:** This should remain one protocol-level ingredient rather than being split into the specific rules for discarding ambiguous sentences or the senior annotator conflict-resolution steps. +- **why_not_adjacent_implementation_details:** The structural dependency is the human verification workflow itself, not the specific payment or interface details. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the manual verification and correction workflow that ensures the dataset contains genuine errors with unique, accurate ground-truth corrections. +- **rationale:** The dataset's reliability depends on a rigorous human annotation process. Because this specific annotation protocol is designed and executed in the target paper, it maps cleanly to `NONE`. +- **evidence_span:** “We recruit a group of native speakers for manual annotation. The annotators are required to check whether the given sentence contains any spelling errors” + +--- + +# 5. Excluded Tempting Non-Ingredients + +### Claim C1 exclusions +- **IME-based pseudo data generation method (LCSTS-IME-2M)** + - Excluded because the downstream evidence specifically points to the reuse of the manually annotated CSCD-NS dataset, not the pseudo dataset or its generation pipeline. +- **Chinese Lexical Analysis tool (LAC)** + - Excluded because it is a local heuristic used to adjust filtering thresholds for specific particles and entities, not a core structural dependency of the dataset itself. +- **SIGHAN datasets (SIGHAN13, 14, 15)** + - Excluded as a top-level ingredient because they are only used as training data for the intermediate filtering model, making them subordinate to the filtering pipeline itself rather than a direct ingredient of the final dataset. + +--- + +# 6. Candidate Grounding Decisions + +## Claim C1: CSCD-NS Dataset + +### Ingredient C1.I1 +- **Candidate:** LCSTS: A Large Scale Chinese Short Text Summarization Dataset +- **decision:** accepted_canonical +- **why:** It is the explicit and direct source of the raw Weibo posts used to construct the dataset. + +### Ingredient C1.I2 +- **Candidate:** ELECTRA: Pre-training text encoders as discriminators rather than generators +- **decision:** accepted_canonical +- **why:** It is the specific pre-trained architecture used to initialize the detection model that makes filtering the massive corpus feasible. + +### Ingredient C1.I3 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The human annotation and verification protocol is specific to this dataset and implemented entirely within the target paper. + +--- + +# 7. Final Grounding Decisions + +## Claim C1 +- **C1.I1** → LCSTS: A Large Scale Chinese Short Text Summarization Dataset → DATA_SOURCE +- **C1.I2** → ELECTRA: Pre-training text encoders as discriminators rather than generators → IMPLEMENTATION_TOOLING +- **C1.I3** → NONE → EVALUATION_PROTOCOL + +--- + +# 8. Final Rationales And Evidence + +## Claim C1: CSCD-NS Dataset + +### C1.I1 +- **ingredient:** Large-scale corpus of authentic Chinese social media posts +- **canonical study:** LCSTS: A Large Scale Chinese Short Text Summarization Dataset +- **role:** DATA_SOURCE +- **contribution:** Provides the raw, authentic native-speaker text substrate from which candidate sentences with spelling errors are extracted. +- **rationale:** To build a dataset reflecting native-speaker errors, the authors require a massive corpus of real-world input. The paper explicitly uses the LCSTS dataset (sourced from Weibo) as this foundational text substrate. +- **evidence_span:** “We chose the LCSTS dataset (Hu et al., 2015) as our data source. This dataset is composed of authentic Weibo posts” + +### C1.I2 +- **ingredient:** Pre-trained language model for automated error-detection filtering +- **canonical study:** ELECTRA: Pre-training text encoders as discriminators rather than generators +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the pre-trained discriminator architecture used to build the error detection model that filters the massive raw corpus down to a manageable candidate set. +- **rationale:** The construction pipeline requires a strong detection model to identify candidate sentences. The paper explicitly initializes this filtering model using a pre-trained Chinese ELECTRA-Large discriminator. +- **evidence_span:** “use a Chinese ELECTRA-Large discriminator model (Clark et al., 2020) to initialize the detection model.” + +### C1.I3 +- **ingredient:** Human annotation and verification protocol for spelling correction +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the manual verification and correction workflow that ensures the dataset contains genuine errors with unique, accurate ground-truth corrections. +- **rationale:** The dataset's reliability depends on a rigorous human annotation process. Because this specific annotation protocol is designed and executed in the target paper, it maps cleanly to `NONE`. +- **evidence_span:** “We recruit a group of native speakers for manual annotation. The annotators are required to check whether the given sentence contains any spelling errors” + +TARGET PAPER METADATA: +{ + "paperId": "09d6377921f503ab660c2c2f1a961c0cefe78d1f", + "externalIds": { + "ArXiv": "2211.08788", + "DBLP": "conf/acl/HuMZ24", + "DOI": "10.18653/v1/2024.acl-long.10", + "CorpusId": 253553639 + }, + "url": "https://www.semanticscholar.org/paper/09d6377921f503ab660c2c2f1a961c0cefe78d1f", + "title": "CSCD-NS: a Chinese Spelling Check Dataset for Native Speakers", + "venue": "Annual Meeting of the Association for Computational Linguistics", + "year": 2022, + "referenceCount": 27, + "citationCount": 21, + "openAccessPdf": { + "url": "https://aclanthology.org/2024.acl-long.10.pdf", + "status": "GOLD", + "license": "CCBY", + "disclaimer": "Notice: Paper or abstract available at https://arxiv.org/abs/2211.08788, which is subject to the license by the author or copyright owner provided with this content. Please go to the source to verify the license and copyright information for your use." + }, + "tldr": { + "model": "tldr@v2.0.0", + "text": "This paper presents CSCD-NS, the first Chinese spelling check dataset designed for native speakers, containing 40,000 samples from a Chinese social platform, and proposes a novel method that simulates the input process through an input method, generating large-scale and high-quality pseudo data that closely resembles the actual error distribution and outperforms existing methods." + }, + "publicationDate": "2022-11-16", + "authors": [ + { + "authorId": "2191038395", + "name": "Yong Hu" + }, + { + "authorId": "33427918", + "name": "Fandong Meng" + }, + { + "authorId": "48128428", + "name": "Jie Zhou" + } + ], + "abstract": "In this paper, we present CSCD-NS, the first Chinese spelling check (CSC) dataset designed for native speakers, containing 40,000 samples from a Chinese social platform. Compared with existing CSC datasets aimed at Chinese learners, CSCD-NS is ten times larger in scale and exhibits a distinct error distribution, with a significantly higher proportion of word-level errors. To further enhance the data resource, we propose a novel method that simulates the input process through an input method, generating large-scale and high-quality pseudo data that closely resembles the actual error distribution and outperforms existing methods. Moreover, we investigate the performance of various models in this scenario, including large language models (LLMs), such as ChatGPT. The result indicates that generative models underperform BERT-like classification models due to strict length and pronunciation constraints. The high prevalence of word-level errors also makes CSC for native speakers challenging enough, leaving substantial room for improvement." +} diff --git a/replay_traces/2211.08788/two_pass_outputs/20260726T211957Z/pass_2_ui_payload.json b/replay_traces/2211.08788/two_pass_outputs/20260726T211957Z/pass_2_ui_payload.json new file mode 100644 index 0000000000000000000000000000000000000000..f8421615c744a5c8223fcd34ae9ca1c5aedabd39 --- /dev/null +++ b/replay_traces/2211.08788/two_pass_outputs/20260726T211957Z/pass_2_ui_payload.json @@ -0,0 +1,124 @@ +{ + "target_paper_id": "09d6377921f503ab660c2c2f1a961c0cefe78d1f", + "target_title": "CSCD-NS: a Chinese Spelling Check Dataset for Native Speakers", + "target_year": 2022, + "annotator_id": "replay_trace_builder", + "active_claim_id": "C1", + "claims": [ + { + "claim_id": "C1", + "text": "Dataset: A large-scale Chinese spelling check dataset sourced from authentic native-speaker social media posts, enabling the training and evaluation of spelling correction models on real-world error distributions.", + "rewritten_claim": "Dataset: A large-scale Chinese spelling check dataset sourced from authentic native-speaker social media posts, enabling the training and evaluation of spelling correction models on real-world error distributions.", + "cluster_id": "C2", + "decision": "YES_SUFFICIENT", + "notes": "", + "ingredients": [ + { + "ingredient_id": "C1.I1", + "ingredient": "Large-scale corpus of authentic Chinese social media posts", + "canonical_ref_id": "ref_LCSTS", + "canonical_grounding": { + "ref_id": "ref_LCSTS", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "LCSTS: A Large Scale Chinese Short Text Summarization Dataset", + "ref_year": "", + "ref_authors": "" + }, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "DATA_SOURCE", + "roles": [ + "DATA_SOURCE" + ], + "contribution": "Provides the raw, authentic native-speaker text substrate from which candidate sentences with spelling errors are extracted.", + "rationale": "To build a dataset reflecting native-speaker errors, the authors require a massive corpus of real-world input. The paper explicitly uses the LCSTS dataset (sourced from Weibo) as this foundational text substrate.", + "evidence_span": "We chose the LCSTS dataset (Hu et al., 2015) as our data source. This dataset is composed of authentic Weibo posts" + } + }, + { + "ingredient_id": "C1.I2", + "ingredient": "Pre-trained language model for automated error-detection filtering", + "canonical_ref_id": "ref_ELECTRA", + "canonical_grounding": { + "ref_id": "ref_ELECTRA", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "ELECTRA: Pre-training text encoders as discriminators rather than generators", + "ref_year": "", + "ref_authors": "" + }, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "IMPLEMENTATION_TOOLING", + "roles": [ + "IMPLEMENTATION_TOOLING" + ], + "contribution": "Provides the pre-trained discriminator architecture used to build the error detection model that filters the massive raw corpus down to a manageable candidate set.", + "rationale": "The construction pipeline requires a strong detection model to identify candidate sentences. The paper explicitly initializes this filtering model using a pre-trained Chinese ELECTRA-Large discriminator.", + "evidence_span": "use a Chinese ELECTRA-Large discriminator model (Clark et al., 2020) to initialize the detection model." + } + }, + { + "ingredient_id": "C1.I3", + "ingredient": "Human annotation and verification protocol for spelling correction", + "canonical_ref_id": "__NONE__", + "canonical_grounding": null, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "EVALUATION_PROTOCOL", + "roles": [ + "EVALUATION_PROTOCOL" + ], + "contribution": "Provides the manual verification and correction workflow that ensures the dataset contains genuine errors with unique, accurate ground-truth corrections.", + "rationale": "The dataset's reliability depends on a rigorous human annotation process. Because this specific annotation protocol is designed and executed in the target paper, it maps cleanly to NONE.", + "evidence_span": "We recruit a group of native speakers for manual annotation. The annotators are required to check whether the given sentence contains any spelling errors" + } + } + ], + "enabling_discoveries": [ + { + "ref_id": "ref_LCSTS", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "LCSTS: A Large Scale Chinese Short Text Summarization Dataset", + "ref_year": "", + "ref_authors": "", + "ingredient_id": "C1.I1", + "ingredient": "Large-scale corpus of authentic Chinese social media posts", + "role": "DATA_SOURCE", + "roles": [ + "DATA_SOURCE" + ], + "contribution": "Provides the raw, authentic native-speaker text substrate from which candidate sentences with spelling errors are extracted.", + "rationale": "To build a dataset reflecting native-speaker errors, the authors require a massive corpus of real-world input. The paper explicitly uses the LCSTS dataset (sourced from Weibo) as this foundational text substrate.", + "evidence_span": "We chose the LCSTS dataset (Hu et al., 2015) as our data source. This dataset is composed of authentic Weibo posts" + }, + { + "ref_id": "ref_ELECTRA", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "ELECTRA: Pre-training text encoders as discriminators rather than generators", + "ref_year": "", + "ref_authors": "", + "ingredient_id": "C1.I2", + "ingredient": "Pre-trained language model for automated error-detection filtering", + "role": "IMPLEMENTATION_TOOLING", + "roles": [ + "IMPLEMENTATION_TOOLING" + ], + "contribution": "Provides the pre-trained discriminator architecture used to build the error detection model that filters the massive raw corpus down to a manageable candidate set.", + "rationale": "The construction pipeline requires a strong detection model to identify candidate sentences. The paper explicitly initializes this filtering model using a pre-trained Chinese ELECTRA-Large discriminator.", + "evidence_span": "use a Chinese ELECTRA-Large discriminator model (Clark et al., 2020) to initialize the detection model." + } + ] + } + ] +} diff --git a/replay_traces/2211.08788/two_pass_outputs/20260726T211957Z/run_output.json b/replay_traces/2211.08788/two_pass_outputs/20260726T211957Z/run_output.json new file mode 100644 index 0000000000000000000000000000000000000000..d4e1a46c8287589f6cba867c8a34475e9182e93b --- /dev/null +++ b/replay_traces/2211.08788/two_pass_outputs/20260726T211957Z/run_output.json @@ -0,0 +1,151 @@ +{ + "paper_id": "2211.08788", + "paper_dir": "/Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785100600_1e6a0c3a/processed_papers/2211.08788", + "generated_at": "2026-07-26T21:22:24.841720+00:00", + "reasoner_model": "gemini/gemini-3.1-pro-preview", + "formatter_model": "gemini/gemini-3.1-pro-preview", + "judge_model": "gemini/gemini-3.1-pro-preview", + "candidate_count": 3, + "include_reference_examples": true, + "prompt_profile": "full", + "reasoning_candidate_paths": [ + "/Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785100600_1e6a0c3a/two_pass_outputs/2211.08788/20260726T211957Z/pass_1_reasoning.output.candidate_1.md", + "/Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785100600_1e6a0c3a/two_pass_outputs/2211.08788/20260726T211957Z/pass_1_reasoning.output.candidate_2.md", + "/Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785100600_1e6a0c3a/two_pass_outputs/2211.08788/20260726T211957Z/pass_1_reasoning.output.candidate_3.md" + ], + "selected_reasoning_candidate": "candidate_1", + "selected_candidate_index": 0, + "selected_reasoning_path": "/Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785100600_1e6a0c3a/two_pass_outputs/2211.08788/20260726T211957Z/pass_1_reasoning.selected.md", + "judge_output_path": "/Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785100600_1e6a0c3a/two_pass_outputs/2211.08788/20260726T211957Z/pass_1_reasoning.judge.output.json", + "formatter_attempts": [ + { + "attempt": 1, + "status": "success" + } + ], + "ui_payload_path": "/Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/replay_build/job_1785100600_1e6a0c3a/two_pass_outputs/2211.08788/20260726T211957Z/pass_2_ui_payload.json", + "ui_payload": { + "target_paper_id": "09d6377921f503ab660c2c2f1a961c0cefe78d1f", + "target_title": "CSCD-NS: a Chinese Spelling Check Dataset for Native Speakers", + "target_year": 2022, + "annotator_id": "replay_trace_builder", + "active_claim_id": "C1", + "claims": [ + { + "claim_id": "C1", + "text": "Dataset: A large-scale Chinese spelling check dataset sourced from authentic native-speaker social media posts, enabling the training and evaluation of spelling correction models on real-world error distributions.", + "rewritten_claim": "Dataset: A large-scale Chinese spelling check dataset sourced from authentic native-speaker social media posts, enabling the training and evaluation of spelling correction models on real-world error distributions.", + "cluster_id": "C2", + "decision": "YES_SUFFICIENT", + "notes": "", + "ingredients": [ + { + "ingredient_id": "C1.I1", + "ingredient": "Large-scale corpus of authentic Chinese social media posts", + "canonical_ref_id": "ref_LCSTS", + "canonical_grounding": { + "ref_id": "ref_LCSTS", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "LCSTS: A Large Scale Chinese Short Text Summarization Dataset", + "ref_year": "", + "ref_authors": "" + }, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "DATA_SOURCE", + "roles": [ + "DATA_SOURCE" + ], + "contribution": "Provides the raw, authentic native-speaker text substrate from which candidate sentences with spelling errors are extracted.", + "rationale": "To build a dataset reflecting native-speaker errors, the authors require a massive corpus of real-world input. The paper explicitly uses the LCSTS dataset (sourced from Weibo) as this foundational text substrate.", + "evidence_span": "We chose the LCSTS dataset (Hu et al., 2015) as our data source. This dataset is composed of authentic Weibo posts" + } + }, + { + "ingredient_id": "C1.I2", + "ingredient": "Pre-trained language model for automated error-detection filtering", + "canonical_ref_id": "ref_ELECTRA", + "canonical_grounding": { + "ref_id": "ref_ELECTRA", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "ELECTRA: Pre-training text encoders as discriminators rather than generators", + "ref_year": "", + "ref_authors": "" + }, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "IMPLEMENTATION_TOOLING", + "roles": [ + "IMPLEMENTATION_TOOLING" + ], + "contribution": "Provides the pre-trained discriminator architecture used to build the error detection model that filters the massive raw corpus down to a manageable candidate set.", + "rationale": "The construction pipeline requires a strong detection model to identify candidate sentences. The paper explicitly initializes this filtering model using a pre-trained Chinese ELECTRA-Large discriminator.", + "evidence_span": "use a Chinese ELECTRA-Large discriminator model (Clark et al., 2020) to initialize the detection model." + } + }, + { + "ingredient_id": "C1.I3", + "ingredient": "Human annotation and verification protocol for spelling correction", + "canonical_ref_id": "__NONE__", + "canonical_grounding": null, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "EVALUATION_PROTOCOL", + "roles": [ + "EVALUATION_PROTOCOL" + ], + "contribution": "Provides the manual verification and correction workflow that ensures the dataset contains genuine errors with unique, accurate ground-truth corrections.", + "rationale": "The dataset's reliability depends on a rigorous human annotation process. Because this specific annotation protocol is designed and executed in the target paper, it maps cleanly to NONE.", + "evidence_span": "We recruit a group of native speakers for manual annotation. The annotators are required to check whether the given sentence contains any spelling errors" + } + } + ], + "enabling_discoveries": [ + { + "ref_id": "ref_LCSTS", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "LCSTS: A Large Scale Chinese Short Text Summarization Dataset", + "ref_year": "", + "ref_authors": "", + "ingredient_id": "C1.I1", + "ingredient": "Large-scale corpus of authentic Chinese social media posts", + "role": "DATA_SOURCE", + "roles": [ + "DATA_SOURCE" + ], + "contribution": "Provides the raw, authentic native-speaker text substrate from which candidate sentences with spelling errors are extracted.", + "rationale": "To build a dataset reflecting native-speaker errors, the authors require a massive corpus of real-world input. The paper explicitly uses the LCSTS dataset (sourced from Weibo) as this foundational text substrate.", + "evidence_span": "We chose the LCSTS dataset (Hu et al., 2015) as our data source. This dataset is composed of authentic Weibo posts" + }, + { + "ref_id": "ref_ELECTRA", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "ELECTRA: Pre-training text encoders as discriminators rather than generators", + "ref_year": "", + "ref_authors": "", + "ingredient_id": "C1.I2", + "ingredient": "Pre-trained language model for automated error-detection filtering", + "role": "IMPLEMENTATION_TOOLING", + "roles": [ + "IMPLEMENTATION_TOOLING" + ], + "contribution": "Provides the pre-trained discriminator architecture used to build the error detection model that filters the massive raw corpus down to a manageable candidate set.", + "rationale": "The construction pipeline requires a strong detection model to identify candidate sentences. The paper explicitly initializes this filtering model using a pre-trained Chinese ELECTRA-Large discriminator.", + "evidence_span": "use a Chinese ELECTRA-Large discriminator model (Clark et al., 2020) to initialize the detection model." + } + ] + } + ] + } +} diff --git a/replay_traces/2505.17978/input_ids.json b/replay_traces/2505.17978/input_ids.json new file mode 100644 index 0000000000000000000000000000000000000000..d34c26636046e2825417ba34aea7d0811adb646e --- /dev/null +++ b/replay_traces/2505.17978/input_ids.json @@ -0,0 +1,7 @@ +[ + { + "id": "2505.17978", + "title": "", + "id_type": "ArXiv" + } +] \ No newline at end of file diff --git a/replay_traces/2505.17978/logs/step_01.log b/replay_traces/2505.17978/logs/step_01.log new file mode 100644 index 0000000000000000000000000000000000000000..90dbbed9762878c79fcf28e2a43c2eb19c5110f0 --- /dev/null +++ b/replay_traces/2505.17978/logs/step_01.log @@ -0,0 +1,16 @@ +$ /Users/dingzifeng/SciFy-UMBC/demo_2026_07/.venv/bin/python3 src/step_01_fetch/fetch_metadata.py --ids /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/job_1785098515_4b6a9bbd/input_ids.json --outdir /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/job_1785098515_4b6a9bbd/processed_papers + +/Users/dingzifeng/SciFy-UMBC/demo_2026_07/src/step_01_fetch/fetch_metadata.py:95: DeprecationWarning: The 'Search.results' method is deprecated, use 'Client.results' instead + paper = next(search.results()) +/Users/dingzifeng/SciFy-UMBC/demo_2026_07/src/step_01_fetch/fetch_metadata.py:124: DeprecationWarning: Python 3.14 will, by default, filter extracted tar archives and reject files or modify their metadata. Use the filter argument to control this behavior. + tar.extractall(path=source_dir) +[INFO] Using ID list from /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/job_1785098515_4b6a9bbd/input_ids.json +[INFO] Output will be saved to /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/job_1785098515_4b6a9bbd/processed_papers +[DEBUG] fetch_one_acl_id: id=2505.17978 id_type=ArXiv title_len=0 s2_key_present=no s2_key_len=0 +[WARN] 429 Too Many Requests → retrying in 1.16s +[WARN] 429 Too Many Requests → retrying in 2.19s +[DEBUG] wrote metadata to /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/job_1785098515_4b6a9bbd/processed_papers/2505.17978/paper_metadata.json +[INFO] Wrote combined BibTeX to /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/job_1785098515_4b6a9bbd/processed_papers/2505.17978/references.bib +[INFO] Processed LaTeX for 2505.17978 at /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/job_1785098515_4b6a9bbd/processed_papers/2505.17978/processed_main.tex +[SUCCESS] +done in 7.32901406288147 s diff --git a/replay_traces/2505.17978/logs/step_02.log b/replay_traces/2505.17978/logs/step_02.log new file mode 100644 index 0000000000000000000000000000000000000000..92d31b3efde9aa1e9877554100e5b9b8c752a92e --- /dev/null +++ b/replay_traces/2505.17978/logs/step_02.log @@ -0,0 +1,5 @@ +$ /Users/dingzifeng/SciFy-UMBC/demo_2026_07/.venv/bin/python3 src/step_02_mark_citations/replace_citation_markers.py --root /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/job_1785098515_4b6a9bbd/processed_papers + +[INFO] Found 1 paper dirs under /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/job_1785098515_4b6a9bbd/processed_papers +[OK] 2505.17978: updated 40 contexts over 41 +[SUMMARY] total_updated=40 over 41; skipped_incomplete=0 diff --git a/replay_traces/2505.17978/logs/step_03.log b/replay_traces/2505.17978/logs/step_03.log new file mode 100644 index 0000000000000000000000000000000000000000..db84242962098f738567171d9805c20bbc6280bf --- /dev/null +++ b/replay_traces/2505.17978/logs/step_03.log @@ -0,0 +1,4 @@ +$ /Users/dingzifeng/SciFy-UMBC/demo_2026_07/.venv/bin/python3 src/step_03_usage_contexts/build_usage_contexts.py --root /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/job_1785098515_4b6a9bbd/processed_papers --out-name usage_contexts.json + +[INFO] Found 1 paper dirs under /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/job_1785098515_4b6a9bbd/processed_papers +[OK] 2505.17978: wrote usage_contexts.json (41 contexts from 16 citations) diff --git a/replay_traces/2505.17978/logs/step_04.log b/replay_traces/2505.17978/logs/step_04.log new file mode 100644 index 0000000000000000000000000000000000000000..b36269177d5a3e238ff8a0a0314b9fd4df1f18d7 --- /dev/null +++ b/replay_traces/2505.17978/logs/step_04.log @@ -0,0 +1,6 @@ +$ /Users/dingzifeng/SciFy-UMBC/demo_2026_07/.venv/bin/python3 src/step_04_label_citations/label_citation_functions.py --root /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/job_1785098515_4b6a9bbd/processed_papers --model-path Deep-Citation/Workspace/acl_scicite_wksp_trl/best_model.pt --model-data-dir Deep-Citation/Data --model-class-def Deep-Citation/Data/class_def.json --model-lm scibert --device cpu + +[INFO] Found 1 paper dirs under /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/job_1785098515_4b6a9bbd/processed_papers +Number of data instance: 1941 +[LABELED] 2505.17978 +[SUMMARY] labeled=1, skipped=0, missing_usage=0, empty_contexts=0 diff --git a/replay_traces/2505.17978/logs/step_05.log b/replay_traces/2505.17978/logs/step_05.log new file mode 100644 index 0000000000000000000000000000000000000000..0f83524ba338b4f4f08bd3b72a63f99d0fa5bca9 --- /dev/null +++ b/replay_traces/2505.17978/logs/step_05.log @@ -0,0 +1,14 @@ +$ /Users/dingzifeng/SciFy-UMBC/demo_2026_07/.venv/bin/python3 src/step_05_verify_uses_extends/verify_uses_extends.py --root /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/job_1785098515_4b6a9bbd/processed_papers --k 0 --batch-size 25 + +/Users/dingzifeng/SciFy-UMBC/demo_2026_07/src/common/llm_client.py:4: FutureWarning: + +All support for the `google.generativeai` package has ended. It will no longer be receiving +updates or bug fixes. Please switch to the `google.genai` package as soon as possible. +See README for more details: + +https://github.com/google-gemini/deprecated-generative-ai-python/blob/main/README.md + + import google.generativeai as genai +[INFO] Found 1 paper dirs under /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/job_1785098515_4b6a9bbd/processed_papers +[VERIFIED] 2505.17978 +[SUMMARY] verified=1, skipped=0, missing_labels=0, no_candidates=0 diff --git a/replay_traces/2505.17978/logs/step_06.log b/replay_traces/2505.17978/logs/step_06.log new file mode 100644 index 0000000000000000000000000000000000000000..f72df1748290d3e3b116f73fbd4b8c51074c4683 --- /dev/null +++ b/replay_traces/2505.17978/logs/step_06.log @@ -0,0 +1,7 @@ +$ /Users/dingzifeng/SciFy-UMBC/demo_2026_07/.venv/bin/python3 src/step_06_extract_paragraphs/extract_arxiv_paragraphs.py --root /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/job_1785098515_4b6a9bbd/processed_papers + +/Users/dingzifeng/SciFy-UMBC/demo_2026_07/src/step_06_extract_paragraphs/extract_arxiv_paragraphs.py:47: DeprecationWarning: Python 3.14 will, by default, filter extracted tar archives and reject files or modify their metadata. Use the filter argument to control this behavior. + tar.extractall(path) +[INFO] Found 1 paper dirs under /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/job_1785098515_4b6a9bbd/processed_papers +[PROCESSED] 2505.17978 +[SUMMARY] processed=1, skipped=0, missing_usage=0, missing_verified=0, empty_citing=0 diff --git a/replay_traces/2505.17978/logs/step_07.log b/replay_traces/2505.17978/logs/step_07.log new file mode 100644 index 0000000000000000000000000000000000000000..bcb5af65790b5cecba500d9596767aa318dce991 --- /dev/null +++ b/replay_traces/2505.17978/logs/step_07.log @@ -0,0 +1,28 @@ +$ /Users/dingzifeng/SciFy-UMBC/demo_2026_07/.venv/bin/python3 src/step_07_extract_and_refine/extract_contributions_from_citations.py --root /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/job_1785098515_4b6a9bbd/processed_papers + +/Users/dingzifeng/SciFy-UMBC/demo_2026_07/src/common/llm_client.py:4: FutureWarning: + +All support for the `google.generativeai` package has ended. It will no longer be receiving +updates or bug fixes. Please switch to the `google.genai` package as soon as possible. +See README for more details: + +https://github.com/google-gemini/deprecated-generative-ai-python/blob/main/README.md + + import google.generativeai as genai +[INFO] Found 1 paper dirs under /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/job_1785098515_4b6a9bbd/processed_papers +[LABELED] 2505.17978 +[SUMMARY] labeled=1, skipped=0, missing_verified=0, missing_arxiv_paragraphs=0, no_confirmed=0 +$ /Users/dingzifeng/SciFy-UMBC/demo_2026_07/.venv/bin/python3 src/step_07_extract_and_refine/refine_and_filter_clusters_llm.py --root /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/job_1785098515_4b6a9bbd/processed_papers --inplace --overwrite + +/Users/dingzifeng/SciFy-UMBC/demo_2026_07/src/common/llm_client.py:4: FutureWarning: + +All support for the `google.generativeai` package has ended. It will no longer be receiving +updates or bug fixes. Please switch to the `google.genai` package as soon as possible. +See README for more details: + +https://github.com/google-gemini/deprecated-generative-ai-python/blob/main/README.md + + import google.generativeai as genai +[INFO] Found 1 paper dirs under /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/job_1785098515_4b6a9bbd/processed_papers +[REFINED] 2505.17978 +[SUMMARY] refined=1, skipped=0, missing_inputs=0, empty_clusters=0 diff --git a/replay_traces/2505.17978/processed_papers/2505.17978/paper_metadata.json b/replay_traces/2505.17978/processed_papers/2505.17978/paper_metadata.json new file mode 100644 index 0000000000000000000000000000000000000000..aedb2410b3a49a1daefaac373effb699af7571d8 --- /dev/null +++ b/replay_traces/2505.17978/processed_papers/2505.17978/paper_metadata.json @@ -0,0 +1,51 @@ +[ + { + "paperId": "3446200921606209cd3db3954ce913bf90070030", + "externalIds": { + "ArXiv": "2505.17978", + "DBLP": "journals/corr/abs-2505-17978", + "DOI": "10.48550/arXiv.2505.17978", + "CorpusId": 278886414 + }, + "url": "https://www.semanticscholar.org/paper/3446200921606209cd3db3954ce913bf90070030", + "title": "AVerImaTeC: A Dataset for Automatic Verification of Image-Text Claims with Evidence from the Web", + "venue": "arXiv.org", + "year": 2025, + "referenceCount": 67, + "citationCount": 16, + "openAccessPdf": { + "url": "", + "status": null, + "license": null, + "disclaimer": "Notice: Paper or abstract available at https://arxiv.org/abs/2505.17978, which is subject to the license by the author or copyright owner provided with this content. Please go to the source to verify the license and copyright information for your use." + }, + "tldr": { + "model": "tldr@v2.0.0", + "text": "This work introduces AVerImaTeC, a dataset consisting of 1,297 real-world image-text claims, and mitigate common challenges in fact-checking datasets such as contextual dependence, temporal leakage, and evidence insufficiency, via claim normalization, temporally constrained evidence annotation, and a two-stage sufficiency check." + }, + "publicationDate": "2025-05-23", + "authors": [ + { + "authorId": "2331613625", + "name": "Rui Cao" + }, + { + "authorId": "2348499942", + "name": "Zifeng Ding" + }, + { + "authorId": "2681038", + "name": "Zhijiang Guo" + }, + { + "authorId": "8804828", + "name": "M. Schlichtkrull" + }, + { + "authorId": "2273888020", + "name": "Andreas Vlachos" + } + ], + "abstract": "Textual claims are often accompanied by images to enhance their credibility and spread on social media, but this also raises concerns about the spread of misinformation. Existing datasets for automated verification of image-text claims remain limited, as they often consist of synthetic claims and lack evidence annotations to capture the reasoning behind the verdict. In this work, we introduce AVerImaTeC, a dataset consisting of 1,297 real-world image-text claims. Each claim is annotated with question-answer (QA) pairs containing evidence from the web, reflecting a decomposed reasoning regarding the verdict. We mitigate common challenges in fact-checking datasets such as contextual dependence, temporal leakage, and evidence insufficiency, via claim normalization, temporally constrained evidence annotation, and a two-stage sufficiency check. We assess the consistency of the annotation in AVerImaTeC via inter-annotator studies, achieving a $\\kappa=0.742$ on verdicts and $74.7\\%$ consistency on QA pairs. We also propose a novel evaluation method for evidence retrieval and conduct extensive experiments to establish baselines for verifying image-text claims using open-web evidence." + } +] \ No newline at end of file diff --git a/replay_traces/2505.17978/processed_papers/2505.17978/usage_citing_paragraphs.json b/replay_traces/2505.17978/processed_papers/2505.17978/usage_citing_paragraphs.json new file mode 100644 index 0000000000000000000000000000000000000000..74e0905bc09d1dc7e6b5766da3f7fb0a5fd1e448 --- /dev/null +++ b/replay_traces/2505.17978/processed_papers/2505.17978/usage_citing_paragraphs.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:fc8dc5b73dd5f015bedba2a26151f15bb89bee841c5fc8140ed5524b3f0468e8 +size 44634503 diff --git a/replay_traces/2505.17978/processed_papers/2505.17978/usage_context_labels.json b/replay_traces/2505.17978/processed_papers/2505.17978/usage_context_labels.json new file mode 100644 index 0000000000000000000000000000000000000000..8c113012b68adfb4d1e435d16c6e5fb171dfb196 --- /dev/null +++ b/replay_traces/2505.17978/processed_papers/2505.17978/usage_context_labels.json @@ -0,0 +1,568 @@ +{ + "paper_id": "2505.17978", + "num_contexts": 41, + "label_set": [ + "Background", + "Uses", + "Extends", + "CompareOrContrast", + "Motivation", + "Future" + ], + "labels": [ + { + "id": 1, + "label": "Background", + "confidence": 0.8704766035079956, + "confidence_margin": 0.7436683773994446, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "488eadef9a50c04fdb75d7ddfcd2ad6018e95805", + "citing_title": "Evidence Triangulation for Multimodal Fact-Checking in the Wild", + "text": "Conversely, although and M4FC [30] provide curated multimodal claims from fact-checked articles, their limited scale and class imbalances (e.g., only 17 \u2018Supported\u2019 claims in , 292 in M4FC ) prevent effective training." + }, + { + "id": 2, + "label": "Background", + "confidence": 0.7347568869590759, + "confidence_margin": 0.47202083468437195, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "488eadef9a50c04fdb75d7ddfcd2ad6018e95805", + "citing_title": "Evidence Triangulation for Multimodal Fact-Checking in the Wild", + "text": "As a result, recent work relies on weakly annotated or synthetic datasets [4, 5, 6, 7, 8, 9], or benchmarks sourced directly from fact-checking organizations ." + }, + { + "id": 3, + "label": "Background", + "confidence": 0.8644869327545166, + "confidence_margin": 0.7906327247619629, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "488eadef9a50c04fdb75d7ddfcd2ad6018e95805", + "citing_title": "Evidence Triangulation for Multimodal Fact-Checking in the Wild", + "text": "\u20261K \u2013 \u2713 Self-contained Fact-checks VeriTaS [12] 25K \u2013 \u2713 Self-contained Fact-checks M4FC [30] 7K \u2713 (Limited) \u2713 Self-contained Fact-checks 1.3K \u2713 (Limited) \u2713 Self-contained Fact-checks X-FACTA [13] 2.4K Each instance consists of a multimodal pair ( I n , T n ) , where I n \u2208 I is an\u2026" + }, + { + "id": 4, + "label": "Uses", + "confidence": 0.9989774227142334, + "confidence_margin": 0.9984922409057617, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "0d5c8b3bfa46dc2f4efa4a5537716473db3101af", + "citing_title": "I Know What You Meme, Even If it Emerged Today: Understanding Evolving Memes through Open-World Knowledge Acquisition", + "text": "Following ; Akhtar et al., 2026), we convert QA pairs to evidence statement, for both evidence evaluation and maintaining the evidence history." + }, + { + "id": 5, + "label": "Motivation", + "confidence": 0.9851356148719788, + "confidence_margin": 0.9750931262969971, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "0d5c8b3bfa46dc2f4efa4a5537716473db3101af", + "citing_title": "I Know What You Meme, Even If it Emerged Today: Understanding Evolving Memes through Open-World Knowledge Acquisition", + "text": "1-Flash 6 as the scoring model for the this evaluation inspired by its power ; Akhtar et al., 2026)." + }, + { + "id": 6, + "label": "Background", + "confidence": 0.7652772665023804, + "confidence_margin": 0.5385058522224426, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "0d5c8b3bfa46dc2f4efa4a5537716473db3101af", + "citing_title": "I Know What You Meme, Even If it Emerged Today: Understanding Evolving Memes through Open-World Knowledge Acquisition", + "text": "Recent studies Table 2: Recall scores of zero-shot generated background knowledge and QA-pair-converted knowledge statements against ground-truth background knowledge." + }, + { + "id": 7, + "label": "Background", + "confidence": 0.9689226746559143, + "confidence_margin": 0.9441730976104736, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "0d5c8b3bfa46dc2f4efa4a5537716473db3101af", + "citing_title": "I Know What You Meme, Even If it Emerged Today: Understanding Evolving Memes through Open-World Knowledge Acquisition", + "text": "Knowledge integration has proven to be effective in NLP tasks such as factual reasoning (Petroni et al., 2019; ; Akhtar et al., 2026)." + }, + { + "id": 8, + "label": "Background", + "confidence": 0.607904851436615, + "confidence_margin": 0.3144609332084656, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "0d5c8b3bfa46dc2f4efa4a5537716473db3101af", + "citing_title": "I Know What You Meme, Even If it Emerged Today: Understanding Evolving Memes through Open-World Knowledge Acquisition", + "text": "\u2026Conclude stage synthesizes the resulting question-answer pairs into explicit background knowledge statements, following evidence-statement formulations in recent retrieval evaluation work ; Akhtar et al., 2026), and uses these statements as contextual evidence for downstream task." + }, + { + "id": 9, + "label": "Background", + "confidence": 0.755760133266449, + "confidence_margin": 0.5238171815872192, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "3a74d04a13ecbeb3767b4af354b9ee0f73341f18", + "citing_title": "Is a Picture Worth a Thousand Words? Adaptive Multimodal Fact-Checking with Visual Evidence Necessity", + "text": "Second, it must take a textual claim c as input, excluding datasets such as AVerImaTeC that assumes text-image input pairs." + }, + { + "id": 10, + "label": "Background", + "confidence": 0.5810175538063049, + "confidence_margin": 0.18981388211250305, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "3a74d04a13ecbeb3767b4af354b9ee0f73341f18", + "citing_title": "Is a Picture Worth a Thousand Words? Adaptive Multimodal Fact-Checking with Visual Evidence Necessity", + "text": "Motivated by these findings, we propose AM U FC, a multimodal fact-verification framework that adaptively incorporates visual evidence through the collaboration of two vision-language (Tan et al., 2020) Img/Txt - OOC Mixed Grover/GoodNews \u2717 \u2713 NewsCLIPings (Luo et al., 2021) Img/Txt - OOC Mixed CLIP/VisualNews \u2717 \u2713 Factify (Suryavardan et al., 2023) Img/Txt Img/Txt OOC Mixed Twitter \u2717 \u2713 VERITE (Papadopoulos et al., 2024) Img/Txt - Mixed Mixed Snopes/Reuters \u2717 \u2713 Fauxtography (Zlatkova et al., 2019) Img/Txt Meta FV Expert Snopes/Reuters \u2717 \u2713 MMM (Gupta et al., 2022) Img/Txt Txt FV Mixed FC webs \u2717 \u2717 FactDrill (Singhal et al., 2022) Txt Vid/Aud/Img/Txt/Meta FV Expert FC webs \u2717 \u2717 ChartFC (Akhtar et al., 2023a) Txt Img FV Crowd TabFact \u2717 \u2713 ChartCheck (Akhtar et al., 2024) Txt Img/Txt FV Crowd Wikimedia \u2717 \u2713 MOCHEG (Yao et al., 2023) Txt Img/Txt FV Expert FC webs \u2713 \u2713 FIN-FACT (Rangapur et al., 2025) Txt Img/Txt FV Expert FC webs \u2717 \u2713 AVerImaTeC Img/Txt Img/Txt FV Expert FC webs \u2713 \u2713 Table 1: Datasets for multimodal misinformation detection." + }, + { + "id": 11, + "label": "Uses", + "confidence": 0.9872791767120361, + "confidence_margin": 0.9812451004981995, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "3a74d04a13ecbeb3767b4af354b9ee0f73341f18", + "citing_title": "Is a Picture Worth a Thousand Words? Adaptive Multimodal Fact-Checking with Visual Evidence Necessity", + "text": "To prevent retrieving documents published after the corresponding articles, which would make the setting unrealistic, we restricted retrieval to sources published before the fact-checking articles, following recent practice ." + }, + { + "id": 12, + "label": "Background", + "confidence": 0.8741937279701233, + "confidence_margin": 0.7542071342468262, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "3a74d04a13ecbeb3767b4af354b9ee0f73341f18", + "citing_title": "Is a Picture Worth a Thousand Words? Adaptive Multimodal Fact-Checking with Visual Evidence Necessity", + "text": "\u2026al., 2024) Txt Img/Txt FV Crowd Wikimedia \u2717 \u2713 MOCHEG (Yao et al., 2023) Txt Img/Txt FV Expert FC webs \u2713 \u2713 FIN-FACT (Rangapur et al., 2025) Txt Img/Txt FV Expert FC webs \u2717 \u2713 AVerImaTeC Img/Txt Img/Txt FV Expert FC webs \u2713 \u2713 Table 1: Datasets for multimodal misinformation detection." + }, + { + "id": 13, + "label": "CompareOrContrast", + "confidence": 0.9945971965789795, + "confidence_margin": 0.9927396178245544, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "614512cf2cbcfa66dc5110da179882740c0df743", + "citing_title": "AIC CTU@AVerImaTeC: dual-retriever RAG for image-text fact checking", + "text": "While our system does not reach the very state of the art, it signi\ufb01cantly outperforms the iterative agentic baseline and majority of other systems across the board, scoring a solid 3rd place." + }, + { + "id": 14, + "label": "Background", + "confidence": 0.7435321807861328, + "confidence_margin": 0.5110285878181458, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "e49f9fd1499d57ea7afd0ea0e8d7d9fda3f18eeb", + "citing_title": "VILLAIN at AVerImaTeC: Verifying Image-Text Claims via Multi-Agent Collaboration", + "text": "Using embedding-based retrieval and reranking, VILLAIN retrieves textual and visual evidence from the knowledge stores, which are further enriched with URL content filling." + }, + { + "id": 15, + "label": "Background", + "confidence": 0.5546706318855286, + "confidence_margin": 0.3046993613243103, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "e49f9fd1499d57ea7afd0ea0e8d7d9fda3f18eeb", + "citing_title": "VILLAIN at AVerImaTeC: Verifying Image-Text Claims via Multi-Agent Collaboration", + "text": "HerO 2 (Yoon et al., VILLAIN is the first multi-agent system designed for , in which individual agents assume distinct roles across consecutive stages of the fact-checking process." + }, + { + "id": 16, + "label": "Background", + "confidence": 0.4830964207649231, + "confidence_margin": 0.24432814121246338, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "e49f9fd1499d57ea7afd0ea0e8d7d9fda3f18eeb", + "citing_title": "VILLAIN at AVerImaTeC: Verifying Image-Text Claims via Multi-Agent Collaboration", + "text": "Generalizability While VILLAIN achieves state-of-the-art performance on , it has not been evaluated on other datasets." + }, + { + "id": 17, + "label": "Background", + "confidence": 0.9984978437423706, + "confidence_margin": 0.9977341294288635, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "e49f9fd1499d57ea7afd0ea0e8d7d9fda3f18eeb", + "citing_title": "VILLAIN at AVerImaTeC: Verifying Image-Text Claims via Multi-Agent Collaboration", + "text": "Motivated by the spread of misleading multimodal content and the need for automated verification, the AVerImaTeC shared task at the 9th FEVER Work-shop focuses on verifying real-world image\u2013text claims using external evidence , reflecting web-based fact-checking scenarios in the wild." + }, + { + "id": 18, + "label": "Background", + "confidence": 0.9991063475608826, + "confidence_margin": 0.9986630082130432, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "e49f9fd1499d57ea7afd0ea0e8d7d9fda3f18eeb", + "citing_title": "VILLAIN at AVerImaTeC: Verifying Image-Text Claims via Multi-Agent Collaboration", + "text": "\u2026by the spread of misleading multimodal content and the need for automated verification, the AVerImaTeC shared task at the 9th FEVER Work-shop focuses on verifying real-world image\u2013text claims using external evidence , reflecting web-based fact-checking scenarios in the wild." + }, + { + "id": 19, + "label": "Background", + "confidence": 0.9980201721191406, + "confidence_margin": 0.9970328211784363, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "e49f9fd1499d57ea7afd0ea0e8d7d9fda3f18eeb", + "citing_title": "VILLAIN at AVerImaTeC: Verifying Image-Text Claims via Multi-Agent Collaboration", + "text": "The shared task focuses on developing automated multimodal fact-checking systems capable of verifying real-world image\u2013text claims using retrieved web evidence." + }, + { + "id": 20, + "label": "Uses", + "confidence": 0.9993253946304321, + "confidence_margin": 0.9989861845970154, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "e49f9fd1499d57ea7afd0ea0e8d7d9fda3f18eeb", + "citing_title": "VILLAIN at AVerImaTeC: Verifying Image-Text Claims via Multi-Agent Collaboration", + "text": "We evaluate our pipeline using Gemma-3-27B (Team et al., 2025), the model used for evaluation in the official shared task repository 2 ." + }, + { + "id": 21, + "label": "Background", + "confidence": 0.9944115281105042, + "confidence_margin": 0.991675078868866, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "e49f9fd1499d57ea7afd0ea0e8d7d9fda3f18eeb", + "citing_title": "VILLAIN at AVerImaTeC: Verifying Image-Text Claims via Multi-Agent Collaboration", + "text": "Further details can be found in ." + }, + { + "id": 22, + "label": "Uses", + "confidence": 0.9968571662902832, + "confidence_margin": 0.9955329895019531, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "e49f9fd1499d57ea7afd0ea0e8d7d9fda3f18eeb", + "citing_title": "VILLAIN at AVerImaTeC: Verifying Image-Text Claims via Multi-Agent Collaboration", + "text": "As specified by the evaluation protocol, we report four metrics: \u2022 Q-Eval : A reference-based recall metric for comparing the generated questions with the gold questions." + }, + { + "id": 23, + "label": "Uses", + "confidence": 0.9995965361595154, + "confidence_margin": 0.9994179010391235, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "e49f9fd1499d57ea7afd0ea0e8d7d9fda3f18eeb", + "citing_title": "VILLAIN at AVerImaTeC: Verifying Image-Text Claims via Multi-Agent Collaboration", + "text": "We use the set of knowledge stores provided with AVerImaTeC , denoted as K = {K txtT , K , K I } ." + }, + { + "id": 24, + "label": "Uses", + "confidence": 0.9727109670639038, + "confidence_margin": 0.9553354382514954, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "e49f9fd1499d57ea7afd0ea0e8d7d9fda3f18eeb", + "citing_title": "VILLAIN at AVerImaTeC: Verifying Image-Text Claims via Multi-Agent Collaboration", + "text": "We present VILLAIN , the winning system in the shared task." + }, + { + "id": 25, + "label": "Uses", + "confidence": 0.994634747505188, + "confidence_margin": 0.9900506138801575, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "3aa1b82af3c14c2b20d94cdca1a3804ad7364cbe", + "citing_title": "Multimodal Claim Extraction for Fact-Checking", + "text": "Dataset We construct MMCE using post-claim pairs extracted from AVerImaTeC, a recently proposed dataset for real-world image-text claim verification based on data from fact-checking organizations ." + }, + { + "id": 26, + "label": "Background", + "confidence": 0.9883460402488708, + "confidence_margin": 0.9785205125808716, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "6b7e7c5c14d587773baf234a2156a3272ab8bb94", + "citing_title": "VeriTaS: The First Dynamic Benchmark for Multimodal Automated Fact-Checking", + "text": "Additionally, among all multimodal AFC benchmarks, only a few provide justifications for explanation ; Niu et al., 2025)." + }, + { + "id": 27, + "label": "Background", + "confidence": 0.9945397973060608, + "confidence_margin": 0.9912044405937195, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "6b7e7c5c14d587773baf234a2156a3272ab8bb94", + "citing_title": "VeriTaS: The First Dynamic Benchmark for Multimodal Automated Fact-Checking", + "text": "However, many lack full modality coverage, missing images or videos (Schlichtkrull et al., 2023; ; Geng et al., 2025), restrict the task scope to isolated sub-problems (Papadopoulos et al., 2024; Tonglet et al., 2024), or rely heavily on synthetically generated claims (Xu et al.,\u2026" + }, + { + "id": 28, + "label": "Background", + "confidence": 0.9935458302497864, + "confidence_margin": 0.9912264347076416, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "2191af3bd6a931f1b2481de458660acd7ca2a9da", + "citing_title": "Multilingual, Multimodal Pipeline for Creating Authentic and Structured Fact-Checked Claim Dataset", + "text": "Automated fact-checking systems remain limited by their reliance on non-probative secondary sources, despite the fact that many claims are not straightforwardly true or false and therefore require detailed cross-source evidence retrieval and evaluation, including primary materials and secondary\u2026" + }, + { + "id": 29, + "label": "Background", + "confidence": 0.9954648613929749, + "confidence_margin": 0.9920528531074524, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "9347c7c1ca3c2ce2604b0e70b6fa959037a8f517", + "citing_title": "VNU-Bench: A Benchmarking Dataset for Multi-Source Multimodal News Video Understanding", + "text": "Also related are (Yao et al., 2023; ; Qi et al., 2023; Zeng et al., 2025), which are limited to the specific task of detecting fake news or verifying truthfulness in a given text/image/video by exploring ex-ternal text/image/video information." + }, + { + "id": 30, + "label": "Background", + "confidence": 0.9993911981582642, + "confidence_margin": 0.9991623163223267, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "34372cd2d1ab54452044a6169143ee66ec19fa9c", + "citing_title": "M4FC: a Multimodal, Multilingual, Multicultural, Multitask Real-World Fact-Checking Dataset", + "text": "Many datasets gather evidence via reverse image search (RIS) (Zlatkova et al., 2019; Tonglet et al., 2024) or by querying a search engine (Hu et al., 2023; Papadopoulos et al., 2024; ." + }, + { + "id": 31, + "label": "Uses", + "confidence": 0.6726499199867249, + "confidence_margin": 0.4460581839084625, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "34372cd2d1ab54452044a6169143ee66ec19fa9c", + "citing_title": "M4FC: a Multimodal, Multilingual, Multicultural, Multitask Real-World Fact-Checking Dataset", + "text": "To ensure realistic evaluation, evidence written by fact-checking organizations or published after the claim date should be excluded, a filtering step applied by some datasets (Zlatkova et al., 2019; Tonglet et al., 2024; ." + }, + { + "id": 32, + "label": "Background", + "confidence": 0.9996637105941772, + "confidence_margin": 0.9995537996292114, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "0f98265df1c8f94461b32ddcf6ce181d079f7b46", + "citing_title": "Selective Multimodal Retrieval for Automated Verification of Image\u2013Text Claims", + "text": "Large-scale analyses show that a substantial portion of fact-checkable claims require multimodal reasoning, particularly involving images, and cannot be reliably verified using text alone (Akhtar et al., 2023; ." + }, + { + "id": 33, + "label": "Uses", + "confidence": 0.9981325268745422, + "confidence_margin": 0.9969885945320129, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "0f98265df1c8f94461b32ddcf6ce181d079f7b46", + "citing_title": "Selective Multimodal Retrieval for Automated Verification of Image\u2013Text Claims", + "text": "We conduct our experiments on the AVerImaTeC (Automated Verification of Image\u2013Text Claim) dataset , which comprises 1,297 real-world fact-checked image\u2013text claims split" + }, + { + "id": 34, + "label": "Uses", + "confidence": 0.9989171028137207, + "confidence_margin": 0.9985345602035522, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "0f98265df1c8f94461b32ddcf6ce181d079f7b46", + "citing_title": "Selective Multimodal Retrieval for Automated Verification of Image\u2013Text Claims", + "text": "Following the definition , a predicted verdict is considered correct only when the associated evidence score exceeds the predefined threshold \u03bb = 0.3; otherwise, the prediction is counted as incorrect irrespective of its label." + }, + { + "id": 35, + "label": "Background", + "confidence": 0.9974005222320557, + "confidence_margin": 0.9962778091430664, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "0f98265df1c8f94461b32ddcf6ce181d079f7b46", + "citing_title": "Selective Multimodal Retrieval for Automated Verification of Image\u2013Text Claims", + "text": "Recent analyses of fact-checking datasets show that a non-trivial portion of claims require reasoning over visual context to reach a correct verdict, motivating the development of multimodal benchmarks such as AVerI-maTeC ." + }, + { + "id": 36, + "label": "Background", + "confidence": 0.9960463643074036, + "confidence_margin": 0.9946786165237427, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "0f98265df1c8f94461b32ddcf6ce181d079f7b46", + "citing_title": "Selective Multimodal Retrieval for Automated Verification of Image\u2013Text Claims", + "text": "This has led to the development of multimodal fact-checking benchmarks and shared tasks, such as new AVerI-maTeC ." + }, + { + "id": 37, + "label": "Uses", + "confidence": 0.704748272895813, + "confidence_margin": 0.412998229265213, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "c848b5f99642678b699e8cdffe9e74fdc4822eff", + "citing_title": "REVEAL: Retrieval-Enhanced Verification for Multimodal Fact-Checking", + "text": "Our system builds upon the official AVerImaTeC baseline , a modular agentic framework for multimodal claim verification that operates on a decompose-and-verify paradigm." + }, + { + "id": 38, + "label": "CompareOrContrast", + "confidence": 0.6477786898612976, + "confidence_margin": 0.4350259304046631, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "c848b5f99642678b699e8cdffe9e74fdc4822eff", + "citing_title": "REVEAL: Retrieval-Enhanced Verification for Multimodal Fact-Checking", + "text": "\u2026this architecture provides a logical skeleton, the baseline implementation relies on naive execution with two critical limitations identified in : R EVEAL addresses these by replacing this naive toolset with a history-aware retrieval pipeline, where evidence retrieved in\u2026" + }, + { + "id": 39, + "label": "Background", + "confidence": 0.9327172636985779, + "confidence_margin": 0.8742746710777283, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "c848b5f99642678b699e8cdffe9e74fdc4822eff", + "citing_title": "REVEAL: Retrieval-Enhanced Verification for Multimodal Fact-Checking", + "text": "To address this gap, the AVerImaTeC shared task introduces a benchmark of 1,297 real-world image-text claims, requiring systems to retrieve multimodal evidence and predict verdicts grounded in a static knowledge store." + }, + { + "id": 40, + "label": "Uses", + "confidence": 0.9960518479347229, + "confidence_margin": 0.9945626258850098, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "02d960f62ad7936094abfba07f0535cccabc7b35", + "citing_title": "Take It All: Ensemble Retrieval for Multimodal Evidence Aggregation", + "text": "Metrics The metrics selected in to evaluate proposed approaches follow different LLM-as-a-judge strategies: \u2022 Question Score (Q-score): Generated questions are scored by an LLM in comparison to reference questions from the gold data; \u2022 Evidence Score (E-score): Similar to the\u2026" + }, + { + "id": 41, + "label": "Background", + "confidence": 0.998511016368866, + "confidence_margin": 0.9976626634597778, + "cue_span": "", + "rationale": "scibert_model", + "citing_paper_id": "02d960f62ad7936094abfba07f0535cccabc7b35", + "citing_title": "Take It All: Ensemble Retrieval for Multimodal Evidence Aggregation", + "text": "Within this landscape, AVerI-maTeC situates multimodal fact-checking in a realistic web-based setting where systems must verify image-text claims with both, textual and visual evidence ." + } + ], + "citing_paper_labels": [ + { + "citing_paper_id": "488eadef9a50c04fdb75d7ddfcd2ad6018e95805", + "citing_title": "Evidence Triangulation for Multimodal Fact-Checking in the Wild", + "label": "Background", + "evidence_context_ids": [] + }, + { + "citing_paper_id": "0d5c8b3bfa46dc2f4efa4a5537716473db3101af", + "citing_title": "I Know What You Meme, Even If it Emerged Today: Understanding Evolving Memes through Open-World Knowledge Acquisition", + "label": "Uses", + "evidence_context_ids": [ + 4 + ] + }, + { + "citing_paper_id": "3a74d04a13ecbeb3767b4af354b9ee0f73341f18", + "citing_title": "Is a Picture Worth a Thousand Words? Adaptive Multimodal Fact-Checking with Visual Evidence Necessity", + "label": "Uses", + "evidence_context_ids": [ + 11 + ] + }, + { + "citing_paper_id": "614512cf2cbcfa66dc5110da179882740c0df743", + "citing_title": "AIC CTU@AVerImaTeC: dual-retriever RAG for image-text fact checking", + "label": "CompareOrContrast", + "evidence_context_ids": [ + 13 + ] + }, + { + "citing_paper_id": "e49f9fd1499d57ea7afd0ea0e8d7d9fda3f18eeb", + "citing_title": "VILLAIN at AVerImaTeC: Verifying Image-Text Claims via Multi-Agent Collaboration", + "label": "Uses", + "evidence_context_ids": [ + 20, + 22, + 23, + 24 + ] + }, + { + "citing_paper_id": "3aa1b82af3c14c2b20d94cdca1a3804ad7364cbe", + "citing_title": "Multimodal Claim Extraction for Fact-Checking", + "label": "Uses", + "evidence_context_ids": [ + 25 + ] + }, + { + "citing_paper_id": "6b7e7c5c14d587773baf234a2156a3272ab8bb94", + "citing_title": "VeriTaS: The First Dynamic Benchmark for Multimodal Automated Fact-Checking", + "label": "Background", + "evidence_context_ids": [] + }, + { + "citing_paper_id": "2191af3bd6a931f1b2481de458660acd7ca2a9da", + "citing_title": "Multilingual, Multimodal Pipeline for Creating Authentic and Structured Fact-Checked Claim Dataset", + "label": "Background", + "evidence_context_ids": [] + }, + { + "citing_paper_id": "9347c7c1ca3c2ce2604b0e70b6fa959037a8f517", + "citing_title": "VNU-Bench: A Benchmarking Dataset for Multi-Source Multimodal News Video Understanding", + "label": "Background", + "evidence_context_ids": [] + }, + { + "citing_paper_id": "34372cd2d1ab54452044a6169143ee66ec19fa9c", + "citing_title": "M4FC: a Multimodal, Multilingual, Multicultural, Multitask Real-World Fact-Checking Dataset", + "label": "Uses", + "evidence_context_ids": [ + 31 + ] + }, + { + "citing_paper_id": "0f98265df1c8f94461b32ddcf6ce181d079f7b46", + "citing_title": "Selective Multimodal Retrieval for Automated Verification of Image\u2013Text Claims", + "label": "Uses", + "evidence_context_ids": [ + 33, + 34 + ] + }, + { + "citing_paper_id": "c848b5f99642678b699e8cdffe9e74fdc4822eff", + "citing_title": "REVEAL: Retrieval-Enhanced Verification for Multimodal Fact-Checking", + "label": "Uses", + "evidence_context_ids": [ + 37 + ] + }, + { + "citing_paper_id": "02d960f62ad7936094abfba07f0535cccabc7b35", + "citing_title": "Take It All: Ensemble Retrieval for Multimodal Evidence Aggregation", + "label": "Uses", + "evidence_context_ids": [ + 40 + ] + } + ], + "final_label": "Uses" +} \ No newline at end of file diff --git a/replay_traces/2505.17978/processed_papers/2505.17978/usage_contexts.json b/replay_traces/2505.17978/processed_papers/2505.17978/usage_contexts.json new file mode 100644 index 0000000000000000000000000000000000000000..89c04f801f53b4a7dd7f706150fdf892d454227f --- /dev/null +++ b/replay_traces/2505.17978/processed_papers/2505.17978/usage_contexts.json @@ -0,0 +1,607 @@ +{ + "paper_id": "2505.17978", + "total_citations": 16, + "num_contexts": 41, + "num_citing_with_context": 13, + "num_citing_without_context": 3, + "num_influential_citations": 5, + "num_influential_with_context": 5, + "influential_contexts": [ + { + "citing_paper_id": "0d5c8b3bfa46dc2f4efa4a5537716473db3101af", + "title": "I Know What You Meme, Even If it Emerged Today: Understanding Evolving Memes through Open-World Knowledge Acquisition", + "external_ids": { + "ArXiv": "2606.05316", + "CorpusId": 288976750 + }, + "contexts": [ + { + "text": "Following ; Akhtar et al., 2026), we convert QA pairs to evidence statement, for both evidence evaluation and maintaining the evidence history.", + "text_raw": "Following (Cao et al., 2025; Akhtar et al., 2026), we convert QA pairs to evidence statement, for both evidence evaluation and maintaining the evidence history.", + "intents": [] + }, + { + "text": "1-Flash 6 as the scoring model for the this evaluation inspired by its power ; Akhtar et al., 2026).", + "text_raw": "1-Flash 6 as the scoring model for the this evaluation inspired by its power (Cao et al., 2025; Akhtar et al., 2026).", + "intents": [] + }, + { + "text": "Recent studies Table 2: Recall scores of zero-shot generated background knowledge and QA-pair-converted knowledge statements against ground-truth background knowledge.", + "text_raw": "Recent studies (Cao et al., 2025 Table 2: Recall scores of zero-shot generated background knowledge and QA-pair-converted knowledge statements against ground-truth background knowledge.", + "intents": [] + }, + { + "text": "Knowledge integration has proven to be effective in NLP tasks such as factual reasoning (Petroni et al., 2019; ; Akhtar et al., 2026).", + "text_raw": "Knowledge integration has proven to be effective in NLP tasks such as factual reasoning (Petroni et al., 2019; Cao et al., 2025; Akhtar et al., 2026).", + "intents": [] + }, + { + "text": "\u2026Conclude stage synthesizes the resulting question-answer pairs into explicit background knowledge statements, following evidence-statement formulations in recent retrieval evaluation work ; Akhtar et al., 2026), and uses these statements as contextual evidence for downstream task.", + "text_raw": "\u2026Conclude stage synthesizes the resulting question-answer pairs into explicit background knowledge statements, following evidence-statement formulations in recent retrieval evaluation work (Cao et al., 2025; Akhtar et al., 2026), and uses these statements as contextual evidence for downstream task.", + "intents": [] + } + ] + }, + { + "citing_paper_id": "3a74d04a13ecbeb3767b4af354b9ee0f73341f18", + "title": "Is a Picture Worth a Thousand Words? Adaptive Multimodal Fact-Checking with Visual Evidence Necessity", + "external_ids": { + "DBLP": "journals/corr/abs-2604-04692", + "ArXiv": "2604.04692", + "DOI": "10.48550/arXiv.2604.04692", + "CorpusId": 287201686 + }, + "contexts": [ + { + "text": "Second, it must take a textual claim c as input, excluding datasets such as AVerImaTeC that assumes text-image input pairs.", + "text_raw": "Second, it must take a textual claim c as input, excluding datasets such as AVerImaTeC (Cao et al., 2025) that assumes text-image input pairs.", + "intents": [] + }, + { + "text": "Motivated by these findings, we propose AM U FC, a multimodal fact-verification framework that adaptively incorporates visual evidence through the collaboration of two vision-language (Tan et al., 2020) Img/Txt - OOC Mixed Grover/GoodNews \u2717 \u2713 NewsCLIPings (Luo et al., 2021) Img/Txt - OOC Mixed CLIP/VisualNews \u2717 \u2713 Factify (Suryavardan et al., 2023) Img/Txt Img/Txt OOC Mixed Twitter \u2717 \u2713 VERITE (Papadopoulos et al., 2024) Img/Txt - Mixed Mixed Snopes/Reuters \u2717 \u2713 Fauxtography (Zlatkova et al., 2019) Img/Txt Meta FV Expert Snopes/Reuters \u2717 \u2713 MMM (Gupta et al., 2022) Img/Txt Txt FV Mixed FC webs \u2717 \u2717 FactDrill (Singhal et al., 2022) Txt Vid/Aud/Img/Txt/Meta FV Expert FC webs \u2717 \u2717 ChartFC (Akhtar et al., 2023a) Txt Img FV Crowd TabFact \u2717 \u2713 ChartCheck (Akhtar et al., 2024) Txt Img/Txt FV Crowd Wikimedia \u2717 \u2713 MOCHEG (Yao et al., 2023) Txt Img/Txt FV Expert FC webs \u2713 \u2713 FIN-FACT (Rangapur et al., 2025) Txt Img/Txt FV Expert FC webs \u2717 \u2713 AVerImaTeC Img/Txt Img/Txt FV Expert FC webs \u2713 \u2713 Table 1: Datasets for multimodal misinformation detection.", + "text_raw": "Motivated by these findings, we propose AM U FC, a multimodal fact-verification framework that adaptively incorporates visual evidence through the collaboration of two vision-language (Tan et al., 2020) Img/Txt - OOC Mixed Grover/GoodNews \u2717 \u2713 NewsCLIPings (Luo et al., 2021) Img/Txt - OOC Mixed CLIP/VisualNews \u2717 \u2713 Factify (Suryavardan et al., 2023) Img/Txt Img/Txt OOC Mixed Twitter \u2717 \u2713 VERITE (Papadopoulos et al., 2024) Img/Txt - Mixed Mixed Snopes/Reuters \u2717 \u2713 Fauxtography (Zlatkova et al., 2019) Img/Txt Meta FV Expert Snopes/Reuters \u2717 \u2713 MMM (Gupta et al., 2022) Img/Txt Txt FV Mixed FC webs \u2717 \u2717 FactDrill (Singhal et al., 2022) Txt Vid/Aud/Img/Txt/Meta FV Expert FC webs \u2717 \u2717 ChartFC (Akhtar et al., 2023a) Txt Img FV Crowd TabFact \u2717 \u2713 ChartCheck (Akhtar et al., 2024) Txt Img/Txt FV Crowd Wikimedia \u2717 \u2713 MOCHEG (Yao et al., 2023) Txt Img/Txt FV Expert FC webs \u2713 \u2713 FIN-FACT (Rangapur et al., 2025) Txt Img/Txt FV Expert FC webs \u2717 \u2713 AVerImaTeC (Cao et al., 2025) Img/Txt Img/Txt FV Expert FC webs \u2713 \u2713 Table 1: Datasets for multimodal misinformation detection.", + "intents": [] + }, + { + "text": "To prevent retrieving documents published after the corresponding articles, which would make the setting unrealistic, we restricted retrieval to sources published before the fact-checking articles, following recent practice .", + "text_raw": "To prevent retrieving documents published after the corresponding articles, which would make the setting unrealistic, we restricted retrieval to sources published before the fact-checking articles, following recent practice (Cao et al., 2025).", + "intents": [] + }, + { + "text": "\u2026al., 2024) Txt Img/Txt FV Crowd Wikimedia \u2717 \u2713 MOCHEG (Yao et al., 2023) Txt Img/Txt FV Expert FC webs \u2713 \u2713 FIN-FACT (Rangapur et al., 2025) Txt Img/Txt FV Expert FC webs \u2717 \u2713 AVerImaTeC Img/Txt Img/Txt FV Expert FC webs \u2713 \u2713 Table 1: Datasets for multimodal misinformation detection.", + "text_raw": "\u2026al., 2024) Txt Img/Txt FV Crowd Wikimedia \u2717 \u2713 MOCHEG (Yao et al., 2023) Txt Img/Txt FV Expert FC webs \u2713 \u2713 FIN-FACT (Rangapur et al., 2025) Txt Img/Txt FV Expert FC webs \u2717 \u2713 AVerImaTeC (Cao et al., 2025) Img/Txt Img/Txt FV Expert FC webs \u2713 \u2713 Table 1: Datasets for multimodal misinformation detection.", + "intents": [] + } + ] + }, + { + "citing_paper_id": "e49f9fd1499d57ea7afd0ea0e8d7d9fda3f18eeb", + "title": "VILLAIN at AVerImaTeC: Verifying Image-Text Claims via Multi-Agent Collaboration", + "external_ids": { + "ArXiv": "2602.04587", + "DBLP": "journals/corr/abs-2602-04587", + "DOI": "10.48550/arXiv.2602.04587", + "CorpusId": 285285846 + }, + "contexts": [ + { + "text": "Using embedding-based retrieval and reranking, VILLAIN retrieves textual and visual evidence from the knowledge stores, which are further enriched with URL content filling.", + "text_raw": "Using embedding-based retrieval and reranking, VILLAIN retrieves textual and visual evidence from the AVerImaTeC knowledge stores, which are further enriched with URL content filling.", + "intents": [] + }, + { + "text": "HerO 2 (Yoon et al., VILLAIN is the first multi-agent system designed for , in which individual agents assume distinct roles across consecutive stages of the fact-checking process.", + "text_raw": "HerO 2 (Yoon et al., VILLAIN is the first multi-agent system designed for AVerImaTeC, in which individual agents assume distinct roles across consecutive stages of the fact-checking process.", + "intents": [] + }, + { + "text": "Generalizability While VILLAIN achieves state-of-the-art performance on , it has not been evaluated on other datasets.", + "text_raw": "Generalizability While VILLAIN achieves state-of-the-art performance on AVerImaTeC, it has not been evaluated on other datasets.", + "intents": [] + }, + { + "text": "Motivated by the spread of misleading multimodal content and the need for automated verification, the AVerImaTeC shared task at the 9th FEVER Work-shop focuses on verifying real-world image\u2013text claims using external evidence , reflecting web-based fact-checking scenarios in the wild.", + "text_raw": "Motivated by the spread of misleading multimodal content and the need for automated verification, the AVerImaTeC shared task at the 9th FEVER Work-shop focuses on verifying real-world image\u2013text claims using external evidence (Cao et al., 2025), reflecting web-based fact-checking scenarios in the wild.", + "intents": [] + }, + { + "text": "\u2026by the spread of misleading multimodal content and the need for automated verification, the AVerImaTeC shared task at the 9th FEVER Work-shop focuses on verifying real-world image\u2013text claims using external evidence , reflecting web-based fact-checking scenarios in the wild.", + "text_raw": "\u2026by the spread of misleading multimodal content and the need for automated verification, the AVerImaTeC shared task at the 9th FEVER Work-shop focuses on verifying real-world image\u2013text claims using external evidence (Cao et al., 2025), reflecting web-based fact-checking scenarios in the wild.", + "intents": [] + }, + { + "text": "The shared task focuses on developing automated multimodal fact-checking systems capable of verifying real-world image\u2013text claims using retrieved web evidence.", + "text_raw": "The AVerImaTeC shared task focuses on developing automated multimodal fact-checking systems capable of verifying real-world image\u2013text claims using retrieved web evidence.", + "intents": [] + }, + { + "text": "We evaluate our pipeline using Gemma-3-27B (Team et al., 2025), the model used for evaluation in the official shared task repository 2 .", + "text_raw": "We evaluate our pipeline using Gemma-3-27B (Team et al., 2025), the model used for evaluation in the official AVerImaTeC shared task repository 2 .", + "intents": [] + }, + { + "text": "Further details can be found in .", + "text_raw": "Further details can be found in Cao et al. (2025).", + "intents": [] + }, + { + "text": "As specified by the evaluation protocol, we report four metrics: \u2022 Q-Eval : A reference-based recall metric for comparing the generated questions with the gold questions.", + "text_raw": "As specified by the AVerImaTeC evaluation protocol, we report four metrics: \u2022 Q-Eval : A reference-based recall metric for comparing the generated questions with the gold questions.", + "intents": [] + }, + { + "text": "We use the set of knowledge stores provided with AVerImaTeC , denoted as K = {K txtT , K , K I } .", + "text_raw": "We use the set of knowledge stores provided with AVerImaTeC (Cao et al., 2025), denoted as K = {K txtT , K , K I } .", + "intents": [] + }, + { + "text": "We present VILLAIN , the winning system in the shared task.", + "text_raw": "We present VILLAIN , the winning system in the AVerImaTeC shared task.", + "intents": [] + } + ] + }, + { + "citing_paper_id": "0f98265df1c8f94461b32ddcf6ce181d079f7b46", + "title": "Selective Multimodal Retrieval for Automated Verification of Image\u2013Text Claims", + "external_ids": { + "DOI": "10.18653/v1/2026.fever-1.10", + "CorpusId": 286799666 + }, + "contexts": [ + { + "text": "Large-scale analyses show that a substantial portion of fact-checkable claims require multimodal reasoning, particularly involving images, and cannot be reliably verified using text alone (Akhtar et al., 2023; .", + "text_raw": "Large-scale analyses show that a substantial portion of fact-checkable claims require multimodal reasoning, particularly involving images, and cannot be reliably verified using text alone (Akhtar et al., 2023; Cao et al., 2025).", + "intents": [] + }, + { + "text": "We conduct our experiments on the AVerImaTeC (Automated Verification of Image\u2013Text Claim) dataset , which comprises 1,297 real-world fact-checked image\u2013text claims split", + "text_raw": "We conduct our experiments on the AVerImaTeC (Automated Verification of Image\u2013Text Claim) dataset (Cao et al., 2025), which comprises 1,297 real-world fact-checked image\u2013text claims split", + "intents": [] + }, + { + "text": "Following the definition , a predicted verdict is considered correct only when the associated evidence score exceeds the predefined threshold \u03bb = 0.3; otherwise, the prediction is counted as incorrect irrespective of its label.", + "text_raw": "Following the definition (Cao et al., 2025), a predicted verdict is considered correct only when the associated evidence score exceeds the predefined threshold \u03bb = 0.3; otherwise, the prediction is counted as incorrect irrespective of its label.", + "intents": [] + }, + { + "text": "Recent analyses of fact-checking datasets show that a non-trivial portion of claims require reasoning over visual context to reach a correct verdict, motivating the development of multimodal benchmarks such as AVerI-maTeC .", + "text_raw": "Recent analyses of fact-checking datasets show that a non-trivial portion of claims require reasoning over visual context to reach a correct verdict, motivating the development of multimodal benchmarks such as AVerI-maTeC (Cao et al., 2025).", + "intents": [] + }, + { + "text": "This has led to the development of multimodal fact-checking benchmarks and shared tasks, such as new AVerI-maTeC .", + "text_raw": "This has led to the development of multimodal fact-checking benchmarks and shared tasks, such as new AVerI-maTeC (Cao et al., 2025).", + "intents": [] + } + ] + }, + { + "citing_paper_id": "c848b5f99642678b699e8cdffe9e74fdc4822eff", + "title": "REVEAL: Retrieval-Enhanced Verification for Multimodal Fact-Checking", + "external_ids": { + "DOI": "10.18653/v1/2026.fever-1.8", + "CorpusId": 286799800 + }, + "contexts": [ + { + "text": "Our system builds upon the official AVerImaTeC baseline , a modular agentic framework for multimodal claim verification that operates on a decompose-and-verify paradigm.", + "text_raw": "Our system builds upon the official AVerImaTeC baseline (Cao et al., 2025), a modular agentic framework for multimodal claim verification that operates on a decompose-and-verify paradigm.", + "intents": [] + }, + { + "text": "\u2026this architecture provides a logical skeleton, the baseline implementation relies on naive execution with two critical limitations identified in : R EVEAL addresses these by replacing this naive toolset with a history-aware retrieval pipeline, where evidence retrieved in\u2026", + "text_raw": "\u2026this architecture provides a logical skeleton, the baseline implementation relies on naive execution with two critical limitations identified in (Cao et al., 2025): R EVEAL addresses these by replacing this naive toolset with a history-aware retrieval pipeline, where evidence retrieved in\u2026", + "intents": [] + }, + { + "text": "To address this gap, the AVerImaTeC shared task introduces a benchmark of 1,297 real-world image-text claims, requiring systems to retrieve multimodal evidence and predict verdicts grounded in a static knowledge store.", + "text_raw": "To address this gap, the AVerImaTeC shared task (Cao et al., 2025) introduces a benchmark of 1,297 real-world image-text claims, requiring systems to retrieve multimodal evidence and predict verdicts grounded in a static knowledge store.", + "intents": [] + } + ] + } + ], + "citing_papers": [ + { + "citing_paper_id": "488eadef9a50c04fdb75d7ddfcd2ad6018e95805", + "title": "Evidence Triangulation for Multimodal Fact-Checking in the Wild", + "external_ids": { + "ArXiv": "2606.31367", + "CorpusId": 289690647 + }, + "is_influential": false, + "contexts": [ + { + "text": "Conversely, although and M4FC [30] provide curated multimodal claims from fact-checked articles, their limited scale and class imbalances (e.g., only 17 \u2018Supported\u2019 claims in , 292 in M4FC ) prevent effective training.", + "text_raw": "Conversely, although AVerImaTeC [11] and M4FC [30] provide curated multimodal claims from fact-checked articles, their limited scale and class imbalances (e.g., only 17 \u2018Supported\u2019 claims in AVerImaTeC , 292 in M4FC ) prevent effective training.", + "intents": [] + }, + { + "text": "As a result, recent work relies on weakly annotated or synthetic datasets [4, 5, 6, 7, 8, 9], or benchmarks sourced directly from fact-checking organizations .", + "text_raw": "As a result, recent work relies on weakly annotated or synthetic datasets [4, 5, 6, 7, 8, 9], or benchmarks sourced directly from fact-checking organizations [10, 11, 12].", + "intents": [] + }, + { + "text": "\u20261K \u2013 \u2713 Self-contained Fact-checks VeriTaS [12] 25K \u2013 \u2713 Self-contained Fact-checks M4FC [30] 7K \u2713 (Limited) \u2713 Self-contained Fact-checks 1.3K \u2713 (Limited) \u2713 Self-contained Fact-checks X-FACTA [13] 2.4K Each instance consists of a multimodal pair ( I n , T n ) , where I n \u2208 I is an\u2026", + "text_raw": "\u20261K \u2013 \u2713 Self-contained Fact-checks VeriTaS [12] 25K \u2013 \u2713 Self-contained Fact-checks M4FC [30] 7K \u2713 (Limited) \u2713 Self-contained Fact-checks AVerImaTeC [11] 1.3K \u2713 (Limited) \u2713 Self-contained Fact-checks X-FACTA [13] 2.4K Each instance consists of a multimodal pair ( I n , T n ) , where I n \u2208 I is an\u2026", + "intents": [] + } + ] + }, + { + "citing_paper_id": "0d5c8b3bfa46dc2f4efa4a5537716473db3101af", + "title": "I Know What You Meme, Even If it Emerged Today: Understanding Evolving Memes through Open-World Knowledge Acquisition", + "external_ids": { + "ArXiv": "2606.05316", + "CorpusId": 288976750 + }, + "is_influential": true, + "contexts": [ + { + "text": "Following ; Akhtar et al., 2026), we convert QA pairs to evidence statement, for both evidence evaluation and maintaining the evidence history.", + "text_raw": "Following (Cao et al., 2025; Akhtar et al., 2026), we convert QA pairs to evidence statement, for both evidence evaluation and maintaining the evidence history.", + "intents": [] + }, + { + "text": "1-Flash 6 as the scoring model for the this evaluation inspired by its power ; Akhtar et al., 2026).", + "text_raw": "1-Flash 6 as the scoring model for the this evaluation inspired by its power (Cao et al., 2025; Akhtar et al., 2026).", + "intents": [] + }, + { + "text": "Recent studies Table 2: Recall scores of zero-shot generated background knowledge and QA-pair-converted knowledge statements against ground-truth background knowledge.", + "text_raw": "Recent studies (Cao et al., 2025 Table 2: Recall scores of zero-shot generated background knowledge and QA-pair-converted knowledge statements against ground-truth background knowledge.", + "intents": [] + }, + { + "text": "Knowledge integration has proven to be effective in NLP tasks such as factual reasoning (Petroni et al., 2019; ; Akhtar et al., 2026).", + "text_raw": "Knowledge integration has proven to be effective in NLP tasks such as factual reasoning (Petroni et al., 2019; Cao et al., 2025; Akhtar et al., 2026).", + "intents": [] + }, + { + "text": "\u2026Conclude stage synthesizes the resulting question-answer pairs into explicit background knowledge statements, following evidence-statement formulations in recent retrieval evaluation work ; Akhtar et al., 2026), and uses these statements as contextual evidence for downstream task.", + "text_raw": "\u2026Conclude stage synthesizes the resulting question-answer pairs into explicit background knowledge statements, following evidence-statement formulations in recent retrieval evaluation work (Cao et al., 2025; Akhtar et al., 2026), and uses these statements as contextual evidence for downstream task.", + "intents": [] + } + ] + }, + { + "citing_paper_id": "3a74d04a13ecbeb3767b4af354b9ee0f73341f18", + "title": "Is a Picture Worth a Thousand Words? Adaptive Multimodal Fact-Checking with Visual Evidence Necessity", + "external_ids": { + "DBLP": "journals/corr/abs-2604-04692", + "ArXiv": "2604.04692", + "DOI": "10.48550/arXiv.2604.04692", + "CorpusId": 287201686 + }, + "is_influential": true, + "contexts": [ + { + "text": "Second, it must take a textual claim c as input, excluding datasets such as AVerImaTeC that assumes text-image input pairs.", + "text_raw": "Second, it must take a textual claim c as input, excluding datasets such as AVerImaTeC (Cao et al., 2025) that assumes text-image input pairs.", + "intents": [] + }, + { + "text": "Motivated by these findings, we propose AM U FC, a multimodal fact-verification framework that adaptively incorporates visual evidence through the collaboration of two vision-language (Tan et al., 2020) Img/Txt - OOC Mixed Grover/GoodNews \u2717 \u2713 NewsCLIPings (Luo et al., 2021) Img/Txt - OOC Mixed CLIP/VisualNews \u2717 \u2713 Factify (Suryavardan et al., 2023) Img/Txt Img/Txt OOC Mixed Twitter \u2717 \u2713 VERITE (Papadopoulos et al., 2024) Img/Txt - Mixed Mixed Snopes/Reuters \u2717 \u2713 Fauxtography (Zlatkova et al., 2019) Img/Txt Meta FV Expert Snopes/Reuters \u2717 \u2713 MMM (Gupta et al., 2022) Img/Txt Txt FV Mixed FC webs \u2717 \u2717 FactDrill (Singhal et al., 2022) Txt Vid/Aud/Img/Txt/Meta FV Expert FC webs \u2717 \u2717 ChartFC (Akhtar et al., 2023a) Txt Img FV Crowd TabFact \u2717 \u2713 ChartCheck (Akhtar et al., 2024) Txt Img/Txt FV Crowd Wikimedia \u2717 \u2713 MOCHEG (Yao et al., 2023) Txt Img/Txt FV Expert FC webs \u2713 \u2713 FIN-FACT (Rangapur et al., 2025) Txt Img/Txt FV Expert FC webs \u2717 \u2713 AVerImaTeC Img/Txt Img/Txt FV Expert FC webs \u2713 \u2713 Table 1: Datasets for multimodal misinformation detection.", + "text_raw": "Motivated by these findings, we propose AM U FC, a multimodal fact-verification framework that adaptively incorporates visual evidence through the collaboration of two vision-language (Tan et al., 2020) Img/Txt - OOC Mixed Grover/GoodNews \u2717 \u2713 NewsCLIPings (Luo et al., 2021) Img/Txt - OOC Mixed CLIP/VisualNews \u2717 \u2713 Factify (Suryavardan et al., 2023) Img/Txt Img/Txt OOC Mixed Twitter \u2717 \u2713 VERITE (Papadopoulos et al., 2024) Img/Txt - Mixed Mixed Snopes/Reuters \u2717 \u2713 Fauxtography (Zlatkova et al., 2019) Img/Txt Meta FV Expert Snopes/Reuters \u2717 \u2713 MMM (Gupta et al., 2022) Img/Txt Txt FV Mixed FC webs \u2717 \u2717 FactDrill (Singhal et al., 2022) Txt Vid/Aud/Img/Txt/Meta FV Expert FC webs \u2717 \u2717 ChartFC (Akhtar et al., 2023a) Txt Img FV Crowd TabFact \u2717 \u2713 ChartCheck (Akhtar et al., 2024) Txt Img/Txt FV Crowd Wikimedia \u2717 \u2713 MOCHEG (Yao et al., 2023) Txt Img/Txt FV Expert FC webs \u2713 \u2713 FIN-FACT (Rangapur et al., 2025) Txt Img/Txt FV Expert FC webs \u2717 \u2713 AVerImaTeC (Cao et al., 2025) Img/Txt Img/Txt FV Expert FC webs \u2713 \u2713 Table 1: Datasets for multimodal misinformation detection.", + "intents": [] + }, + { + "text": "To prevent retrieving documents published after the corresponding articles, which would make the setting unrealistic, we restricted retrieval to sources published before the fact-checking articles, following recent practice .", + "text_raw": "To prevent retrieving documents published after the corresponding articles, which would make the setting unrealistic, we restricted retrieval to sources published before the fact-checking articles, following recent practice (Cao et al., 2025).", + "intents": [] + }, + { + "text": "\u2026al., 2024) Txt Img/Txt FV Crowd Wikimedia \u2717 \u2713 MOCHEG (Yao et al., 2023) Txt Img/Txt FV Expert FC webs \u2713 \u2713 FIN-FACT (Rangapur et al., 2025) Txt Img/Txt FV Expert FC webs \u2717 \u2713 AVerImaTeC Img/Txt Img/Txt FV Expert FC webs \u2713 \u2713 Table 1: Datasets for multimodal misinformation detection.", + "text_raw": "\u2026al., 2024) Txt Img/Txt FV Crowd Wikimedia \u2717 \u2713 MOCHEG (Yao et al., 2023) Txt Img/Txt FV Expert FC webs \u2713 \u2713 FIN-FACT (Rangapur et al., 2025) Txt Img/Txt FV Expert FC webs \u2717 \u2713 AVerImaTeC (Cao et al., 2025) Img/Txt Img/Txt FV Expert FC webs \u2713 \u2713 Table 1: Datasets for multimodal misinformation detection.", + "intents": [] + } + ] + }, + { + "citing_paper_id": "614512cf2cbcfa66dc5110da179882740c0df743", + "title": "AIC CTU@AVerImaTeC: dual-retriever RAG for image-text fact checking", + "external_ids": { + "DBLP": "journals/corr/abs-2602-15190", + "ArXiv": "2602.15190", + "DOI": "10.18653/v1/2026.fever-1.11", + "CorpusId": 285659699 + }, + "is_influential": false, + "contexts": [ + { + "text": "While our system does not reach the very state of the art, it signi\ufb01cantly outperforms the iterative agentic baseline and majority of other systems across the board, scoring a solid 3rd place.", + "text_raw": "While our system does not reach the very state of the art, it signi\ufb01cantly outperforms the iterative agentic baseline (Cao et al., 2025) and majority of other systems across the board, scoring a solid 3rd place.", + "intents": [] + } + ] + }, + { + "citing_paper_id": "3c679cd50ba898c3696883fac8cef24ae36d38ab", + "title": "The Automatic Verification of Image-Text Claims (AVerImaTeC) Shared Task", + "external_ids": { + "DBLP": "journals/corr/abs-2602-11221", + "ArXiv": "2602.11221", + "DOI": "10.18653/v1/2026.fever-1.6", + "CorpusId": 285540297 + }, + "is_influential": false, + "contexts": [] + }, + { + "citing_paper_id": "e49f9fd1499d57ea7afd0ea0e8d7d9fda3f18eeb", + "title": "VILLAIN at AVerImaTeC: Verifying Image-Text Claims via Multi-Agent Collaboration", + "external_ids": { + "ArXiv": "2602.04587", + "DBLP": "journals/corr/abs-2602-04587", + "DOI": "10.48550/arXiv.2602.04587", + "CorpusId": 285285846 + }, + "is_influential": true, + "contexts": [ + { + "text": "Using embedding-based retrieval and reranking, VILLAIN retrieves textual and visual evidence from the knowledge stores, which are further enriched with URL content filling.", + "text_raw": "Using embedding-based retrieval and reranking, VILLAIN retrieves textual and visual evidence from the AVerImaTeC knowledge stores, which are further enriched with URL content filling.", + "intents": [] + }, + { + "text": "HerO 2 (Yoon et al., VILLAIN is the first multi-agent system designed for , in which individual agents assume distinct roles across consecutive stages of the fact-checking process.", + "text_raw": "HerO 2 (Yoon et al., VILLAIN is the first multi-agent system designed for AVerImaTeC, in which individual agents assume distinct roles across consecutive stages of the fact-checking process.", + "intents": [] + }, + { + "text": "Generalizability While VILLAIN achieves state-of-the-art performance on , it has not been evaluated on other datasets.", + "text_raw": "Generalizability While VILLAIN achieves state-of-the-art performance on AVerImaTeC, it has not been evaluated on other datasets.", + "intents": [] + }, + { + "text": "Motivated by the spread of misleading multimodal content and the need for automated verification, the AVerImaTeC shared task at the 9th FEVER Work-shop focuses on verifying real-world image\u2013text claims using external evidence , reflecting web-based fact-checking scenarios in the wild.", + "text_raw": "Motivated by the spread of misleading multimodal content and the need for automated verification, the AVerImaTeC shared task at the 9th FEVER Work-shop focuses on verifying real-world image\u2013text claims using external evidence (Cao et al., 2025), reflecting web-based fact-checking scenarios in the wild.", + "intents": [] + }, + { + "text": "\u2026by the spread of misleading multimodal content and the need for automated verification, the AVerImaTeC shared task at the 9th FEVER Work-shop focuses on verifying real-world image\u2013text claims using external evidence , reflecting web-based fact-checking scenarios in the wild.", + "text_raw": "\u2026by the spread of misleading multimodal content and the need for automated verification, the AVerImaTeC shared task at the 9th FEVER Work-shop focuses on verifying real-world image\u2013text claims using external evidence (Cao et al., 2025), reflecting web-based fact-checking scenarios in the wild.", + "intents": [] + }, + { + "text": "The shared task focuses on developing automated multimodal fact-checking systems capable of verifying real-world image\u2013text claims using retrieved web evidence.", + "text_raw": "The AVerImaTeC shared task focuses on developing automated multimodal fact-checking systems capable of verifying real-world image\u2013text claims using retrieved web evidence.", + "intents": [] + }, + { + "text": "We evaluate our pipeline using Gemma-3-27B (Team et al., 2025), the model used for evaluation in the official shared task repository 2 .", + "text_raw": "We evaluate our pipeline using Gemma-3-27B (Team et al., 2025), the model used for evaluation in the official AVerImaTeC shared task repository 2 .", + "intents": [] + }, + { + "text": "Further details can be found in .", + "text_raw": "Further details can be found in Cao et al. (2025).", + "intents": [] + }, + { + "text": "As specified by the evaluation protocol, we report four metrics: \u2022 Q-Eval : A reference-based recall metric for comparing the generated questions with the gold questions.", + "text_raw": "As specified by the AVerImaTeC evaluation protocol, we report four metrics: \u2022 Q-Eval : A reference-based recall metric for comparing the generated questions with the gold questions.", + "intents": [] + }, + { + "text": "We use the set of knowledge stores provided with AVerImaTeC , denoted as K = {K txtT , K , K I } .", + "text_raw": "We use the set of knowledge stores provided with AVerImaTeC (Cao et al., 2025), denoted as K = {K txtT , K , K I } .", + "intents": [] + }, + { + "text": "We present VILLAIN , the winning system in the shared task.", + "text_raw": "We present VILLAIN , the winning system in the AVerImaTeC shared task.", + "intents": [] + } + ] + }, + { + "citing_paper_id": "3aa1b82af3c14c2b20d94cdca1a3804ad7364cbe", + "title": "Multimodal Claim Extraction for Fact-Checking", + "external_ids": { + "DBLP": "conf/wassa/TeoCDDSV26", + "ArXiv": "2604.16311", + "DOI": "10.18653/v1/2026.wassa-1.22", + "CorpusId": 286855083 + }, + "is_influential": false, + "contexts": [ + { + "text": "Dataset We construct MMCE using post-claim pairs extracted from AVerImaTeC, a recently proposed dataset for real-world image-text claim verification based on data from fact-checking organizations .", + "text_raw": "Dataset We construct MMCE using post-claim pairs extracted from AVerImaTeC, a recently proposed dataset for real-world image-text claim verification based on data from fact-checking organizations (Cao et al., 2025).", + "intents": [] + } + ] + }, + { + "citing_paper_id": "6b7e7c5c14d587773baf234a2156a3272ab8bb94", + "title": "VeriTaS: The First Dynamic Benchmark for Multimodal Automated Fact-Checking", + "external_ids": { + "DBLP": "journals/corr/abs-2601-08611", + "ArXiv": "2601.08611", + "DOI": "10.48550/arXiv.2601.08611", + "CorpusId": 284704571 + }, + "is_influential": false, + "contexts": [ + { + "text": "Additionally, among all multimodal AFC benchmarks, only a few provide justifications for explanation ; Niu et al., 2025).", + "text_raw": "Additionally, among all multimodal AFC benchmarks, only a few provide justifications for explanation (Cao et al., 2025; Niu et al., 2025).", + "intents": [] + }, + { + "text": "However, many lack full modality coverage, missing images or videos (Schlichtkrull et al., 2023; ; Geng et al., 2025), restrict the task scope to isolated sub-problems (Papadopoulos et al., 2024; Tonglet et al., 2024), or rely heavily on synthetically generated claims (Xu et al.,\u2026", + "text_raw": "However, many lack full modality coverage, missing images or videos (Schlichtkrull et al., 2023; Cao et al., 2025; Geng et al., 2025), restrict the task scope to isolated sub-problems (Papadopoulos et al., 2024; Tonglet et al., 2024), or rely heavily on synthetically generated claims (Xu et al.,\u2026", + "intents": [] + } + ] + }, + { + "citing_paper_id": "2191af3bd6a931f1b2481de458660acd7ca2a9da", + "title": "Multilingual, Multimodal Pipeline for Creating Authentic and Structured Fact-Checked Claim Dataset", + "external_ids": { + "DBLP": "journals/corr/abs-2601-07985", + "ArXiv": "2601.07985", + "DOI": "10.48550/arXiv.2601.07985", + "CorpusId": 284704208 + }, + "is_influential": false, + "contexts": [ + { + "text": "Automated fact-checking systems remain limited by their reliance on non-probative secondary sources, despite the fact that many claims are not straightforwardly true or false and therefore require detailed cross-source evidence retrieval and evaluation, including primary materials and secondary\u2026", + "text_raw": "Automated fact-checking systems remain limited by their reliance on non-probative secondary sources, despite the fact that many claims are not straightforwardly true or false and therefore require detailed cross-source evidence retrieval and evaluation, including primary materials and secondary\u2026", + "intents": [] + } + ] + }, + { + "citing_paper_id": "9347c7c1ca3c2ce2604b0e70b6fa959037a8f517", + "title": "VNU-Bench: A Benchmarking Dataset for Multi-Source Multimodal News Video Understanding", + "external_ids": { + "DBLP": "journals/corr/abs-2601-03434", + "ArXiv": "2601.03434", + "DOI": "10.48550/arXiv.2601.03434", + "CorpusId": 284532426 + }, + "is_influential": false, + "contexts": [ + { + "text": "Also related are (Yao et al., 2023; ; Qi et al., 2023; Zeng et al., 2025), which are limited to the specific task of detecting fake news or verifying truthfulness in a given text/image/video by exploring ex-ternal text/image/video information.", + "text_raw": "Also related are (Yao et al., 2023; Cao et al., 2025; Qi et al., 2023; Zeng et al., 2025), which are limited to the specific task of detecting fake news or verifying truthfulness in a given text/image/video by exploring ex-ternal text/image/video information.", + "intents": [] + } + ] + }, + { + "citing_paper_id": "34372cd2d1ab54452044a6169143ee66ec19fa9c", + "title": "M4FC: a Multimodal, Multilingual, Multicultural, Multitask Real-World Fact-Checking Dataset", + "external_ids": { + "DBLP": "journals/corr/abs-2510-23508", + "ArXiv": "2510.23508", + "DOI": "10.48550/arXiv.2510.23508", + "CorpusId": 282390118 + }, + "is_influential": false, + "contexts": [ + { + "text": "Many datasets gather evidence via reverse image search (RIS) (Zlatkova et al., 2019; Tonglet et al., 2024) or by querying a search engine (Hu et al., 2023; Papadopoulos et al., 2024; .", + "text_raw": "Many datasets gather evidence via reverse image search (RIS) (Zlatkova et al., 2019; Tonglet et al., 2024) or by querying a search engine (Hu et al., 2023; Papadopoulos et al., 2024; Cao et al., 2025).", + "intents": [] + }, + { + "text": "To ensure realistic evaluation, evidence written by fact-checking organizations or published after the claim date should be excluded, a filtering step applied by some datasets (Zlatkova et al., 2019; Tonglet et al., 2024; .", + "text_raw": "To ensure realistic evaluation, evidence written by fact-checking organizations or published after the claim date should be excluded, a filtering step applied by some datasets (Zlatkova et al., 2019; Tonglet et al., 2024; Cao et al., 2025).", + "intents": [] + } + ] + }, + { + "citing_paper_id": "ed152af1ea04747103e15999f5fa5fa25e4269cf", + "title": "FaStFACT: Faster, Stronger Long-Form Factuality Evaluations in LLMs", + "external_ids": { + "ArXiv": "2510.12839", + "DBLP": "conf/emnlp/WanTZZLLSZXLLG25", + "DOI": "10.18653/v1/2025.findings-emnlp.1295", + "CorpusId": 282102267 + }, + "is_influential": false, + "contexts": [] + }, + { + "citing_paper_id": "b463a4a9a6ce7dde9117ef60e0339381cb2817f4", + "title": "ShortCheck: Checkworthiness Detection of Multilingual Short-Form Videos", + "external_ids": { + "DBLP": "journals/corr/abs-2509-20467", + "ArXiv": "2509.20467", + "DOI": "10.48550/arXiv.2509.20467", + "CorpusId": 281525892 + }, + "is_influential": false, + "contexts": [] + }, + { + "citing_paper_id": "0f98265df1c8f94461b32ddcf6ce181d079f7b46", + "title": "Selective Multimodal Retrieval for Automated Verification of Image\u2013Text Claims", + "external_ids": { + "DOI": "10.18653/v1/2026.fever-1.10", + "CorpusId": 286799666 + }, + "is_influential": true, + "contexts": [ + { + "text": "Large-scale analyses show that a substantial portion of fact-checkable claims require multimodal reasoning, particularly involving images, and cannot be reliably verified using text alone (Akhtar et al., 2023; .", + "text_raw": "Large-scale analyses show that a substantial portion of fact-checkable claims require multimodal reasoning, particularly involving images, and cannot be reliably verified using text alone (Akhtar et al., 2023; Cao et al., 2025).", + "intents": [] + }, + { + "text": "We conduct our experiments on the AVerImaTeC (Automated Verification of Image\u2013Text Claim) dataset , which comprises 1,297 real-world fact-checked image\u2013text claims split", + "text_raw": "We conduct our experiments on the AVerImaTeC (Automated Verification of Image\u2013Text Claim) dataset (Cao et al., 2025), which comprises 1,297 real-world fact-checked image\u2013text claims split", + "intents": [] + }, + { + "text": "Following the definition , a predicted verdict is considered correct only when the associated evidence score exceeds the predefined threshold \u03bb = 0.3; otherwise, the prediction is counted as incorrect irrespective of its label.", + "text_raw": "Following the definition (Cao et al., 2025), a predicted verdict is considered correct only when the associated evidence score exceeds the predefined threshold \u03bb = 0.3; otherwise, the prediction is counted as incorrect irrespective of its label.", + "intents": [] + }, + { + "text": "Recent analyses of fact-checking datasets show that a non-trivial portion of claims require reasoning over visual context to reach a correct verdict, motivating the development of multimodal benchmarks such as AVerI-maTeC .", + "text_raw": "Recent analyses of fact-checking datasets show that a non-trivial portion of claims require reasoning over visual context to reach a correct verdict, motivating the development of multimodal benchmarks such as AVerI-maTeC (Cao et al., 2025).", + "intents": [] + }, + { + "text": "This has led to the development of multimodal fact-checking benchmarks and shared tasks, such as new AVerI-maTeC .", + "text_raw": "This has led to the development of multimodal fact-checking benchmarks and shared tasks, such as new AVerI-maTeC (Cao et al., 2025).", + "intents": [] + } + ] + }, + { + "citing_paper_id": "c848b5f99642678b699e8cdffe9e74fdc4822eff", + "title": "REVEAL: Retrieval-Enhanced Verification for Multimodal Fact-Checking", + "external_ids": { + "DOI": "10.18653/v1/2026.fever-1.8", + "CorpusId": 286799800 + }, + "is_influential": true, + "contexts": [ + { + "text": "Our system builds upon the official AVerImaTeC baseline , a modular agentic framework for multimodal claim verification that operates on a decompose-and-verify paradigm.", + "text_raw": "Our system builds upon the official AVerImaTeC baseline (Cao et al., 2025), a modular agentic framework for multimodal claim verification that operates on a decompose-and-verify paradigm.", + "intents": [] + }, + { + "text": "\u2026this architecture provides a logical skeleton, the baseline implementation relies on naive execution with two critical limitations identified in : R EVEAL addresses these by replacing this naive toolset with a history-aware retrieval pipeline, where evidence retrieved in\u2026", + "text_raw": "\u2026this architecture provides a logical skeleton, the baseline implementation relies on naive execution with two critical limitations identified in (Cao et al., 2025): R EVEAL addresses these by replacing this naive toolset with a history-aware retrieval pipeline, where evidence retrieved in\u2026", + "intents": [] + }, + { + "text": "To address this gap, the AVerImaTeC shared task introduces a benchmark of 1,297 real-world image-text claims, requiring systems to retrieve multimodal evidence and predict verdicts grounded in a static knowledge store.", + "text_raw": "To address this gap, the AVerImaTeC shared task (Cao et al., 2025) introduces a benchmark of 1,297 real-world image-text claims, requiring systems to retrieve multimodal evidence and predict verdicts grounded in a static knowledge store.", + "intents": [] + } + ] + }, + { + "citing_paper_id": "02d960f62ad7936094abfba07f0535cccabc7b35", + "title": "Take It All: Ensemble Retrieval for Multimodal Evidence Aggregation", + "external_ids": { + "DOI": "10.18653/v1/2026.fever-1.7", + "CorpusId": 286854874 + }, + "is_influential": false, + "contexts": [ + { + "text": "Metrics The metrics selected in to evaluate proposed approaches follow different LLM-as-a-judge strategies: \u2022 Question Score (Q-score): Generated questions are scored by an LLM in comparison to reference questions from the gold data; \u2022 Evidence Score (E-score): Similar to the\u2026", + "text_raw": "Metrics The metrics selected in Cao et al. (2025) to evaluate proposed approaches follow different LLM-as-a-judge strategies: \u2022 Question Score (Q-score): Generated questions are scored by an LLM in comparison to reference questions from the gold data; \u2022 Evidence Score (E-score): Similar to the\u2026", + "intents": [] + }, + { + "text": "Within this landscape, AVerI-maTeC situates multimodal fact-checking in a realistic web-based setting where systems must verify image-text claims with both, textual and visual evidence .", + "text_raw": "Within this landscape, AVerI-maTeC situates multimodal fact-checking in a realistic web-based setting where systems must verify image-text claims with both, textual and visual evidence (Cao et al., 2025).", + "intents": [] + } + ] + } + ] +} \ No newline at end of file diff --git a/replay_traces/2505.17978/processed_papers/2505.17978/usage_contributions.json b/replay_traces/2505.17978/processed_papers/2505.17978/usage_contributions.json new file mode 100644 index 0000000000000000000000000000000000000000..2da109f43ce915d0beb4d948d8f67bd8a8c60204 --- /dev/null +++ b/replay_traces/2505.17978/processed_papers/2505.17978/usage_contributions.json @@ -0,0 +1,105 @@ +{ + "paper_id": "2505.17978", + "final_label": "EXTENDS", + "contributions": [ + { + "citing_paper_id": "3aa1b82af3c14c2b20d94cdca1a3804ad7364cbe", + "citing_title": "Multimodal Claim Extraction for Fact-Checking", + "label": "EXTENDS", + "paper_claim": "Extracts post-claim pairs from the AVerImaTeC dataset to construct the MMCE dataset for multimodal claim extraction.", + "claim": "Extracts post-claim pairs from the AVerImaTeC dataset to construct the MMCE dataset for multimodal claim extraction.", + "cluster_title": "Extends target dataset for new dataset creation", + "cluster_key": "EXTENDS|dataset|dataset_creation", + "evidence_span": "We construct MMCE using post-claim pairs extracted from AVerImaTeC, a recently proposed dataset for real-world image-text claim verification based on data from fact-checking organizations", + "rationale": "The citing paper explicitly states that it constructs a new dataset (MMCE) by extracting post-claim pairs from the target dataset (AVerImaTeC).", + "contexts": [ + "Target-citing paragraph: \\paragraph{Dataset}\nWe construct MMCE using post-claim pairs extracted from AVerImaTeC, a recently proposed dataset for real-world image-text claim verification based on data from fact-checking organizations . The dataset focuses on out-of-context (OOC) image-text claims because recent studies had observed that context-manipulated claims are the most common type of media-based misinformation \\citep{dufour2024ammebalargescalesurveydataset}. Additionally, extraction for OOC claims is the most different from traditional vision-language tasks, such as image captioning, where the image and the text align. The task deviates from the pre-training objectives of MLLMs, presenting a significant challenge." + ], + "source": "arxiv_paragraphs" + }, + { + "citing_paper_id": "0f98265df1c8f94461b32ddcf6ce181d079f7b46", + "citing_title": "Selective Multimodal Retrieval for Automated Verification of Image\u2013Text Claims", + "label": "USES", + "paper_claim": "Evaluates the proposed selective multimodal retrieval method on the AVerImaTeC dataset.", + "claim": "Evaluates the proposed selective multimodal retrieval method on the AVerImaTeC dataset.", + "cluster_title": "Uses target dataset for evaluation", + "cluster_key": "USES|dataset|evaluation", + "evidence_span": "We conduct our experiments on the AVerImaTeC (Automated Verification of Image\u2013Text Claim) dataset ", + "rationale": "The citing paper explicitly states that they conduct their experiments on the dataset introduced by the target paper.", + "contexts": [ + "Target sentence: We conduct our experiments on the AVerImaTeC (Automated Verification of Image\u2013Text Claim) dataset , which comprises 1,297 real-world fact-checked image\u2013text claims split" + ], + "source": "usage_contexts_fallback" + }, + { + "citing_paper_id": "0d5c8b3bfa46dc2f4efa4a5537716473db3101af", + "citing_title": "I Know What You Meme, Even If it Emerged Today: Understanding Evolving Memes through Open-World Knowledge Acquisition", + "label": "EXTENDS", + "paper_claim": "Extends the target paper's reference-based evidence evaluation method to a multimodal meme understanding setting.", + "claim": "Extends the target paper's reference-based evidence evaluation method to a multimodal meme understanding setting.", + "cluster_title": "Extends evaluation method for a novel setting", + "cluster_key": "EXTENDS|evaluation_method|domain_adaptation", + "evidence_span": "Recent studies~ show that reference-based evidence evaluation with large language models (LLMs) aligns best with human judgments. We extend this idea to our multimodal meme understanding setting", + "rationale": "The citing paper explicitly states that it extends the reference-based evidence evaluation method introduced in the target paper to a new multimodal meme understanding setting.", + "contexts": [ + "Target-citing paragraph: \\textbf{Evaluation.}\nWe evaluate both evidence recovery quality and downstream meme detection performance. Recent studies~ show that reference-based evidence evaluation with large language models (LLMs) aligns best with human judgments. We extend this idea to our multimodal meme understanding setting, where retrieved web evidence may include both text and images. Recall that each generated QA pair is converted into an evidence statement. We then perform reference-based evaluation by comparing these generated evidence statements with the ground-truth evidence annotations. We exploit Gemini-3.1-Flash \\footnote{\\url{https://deepmind.google/models/model-cards/gemini-3-1-flash-image/}} as the scoring model for the this evaluation inspired by its power . We then report\nevidence recall, defined as the percentage of ground-truth evidence instances successfully retrieved. The prompts used for QA-pair-to-statement conversion and evidence evaluation are provided in Appendix~\\ref{app:prompt-QA_convert} and Appendix~\\ref{app:prompt-evidence_eval}, respectively. Additionally, to assess the reliability of the reference-based evaluation, we compare the resulting scores with human judgments obtained from independent raters (details in Appendix ~\\ref{app:align-evidence})." + ], + "source": "arxiv_paragraphs" + }, + { + "citing_paper_id": "3a74d04a13ecbeb3767b4af354b9ee0f73341f18", + "citing_title": "Is a Picture Worth a Thousand Words? Adaptive Multimodal Fact-Checking with Visual Evidence Necessity", + "label": "USES", + "paper_claim": "The citing paper adopts the target paper's practice of restricting evidence retrieval to sources published prior to the fact-checking articles when constructing their new dataset.", + "claim": "The citing paper adopts the target paper's practice of restricting evidence retrieval to sources published prior to the fact-checking articles when constructing their new dataset.", + "cluster_title": "Adopts methodology for dataset creation", + "cluster_key": "USES|methodology|dataset_creation", + "evidence_span": "we restricted retrieval to sources published before the fact-checking articles, following recent practice ", + "rationale": "The citing paper explicitly adopts the target paper's temporally constrained retrieval practice for constructing their new WebFC dataset.", + "contexts": [ + "Target-citing paragraph: \\begin{table*}[t]\n\\centering\n\\resizebox{0.99\\textwidth}{!}{\n\\begin{tabular}{lccccccc}\n\\toprule\n\\textbf{Dataset} & \\makecell{\\textbf{Primary}\\\\\\textbf{Inputs}} & \\textbf{Evidence} & \\textbf{Task} &\n\\makecell{\\textbf{Label}\\\\\\textbf{Source}} &\n\\makecell{\\textbf{Claim}\\\\\\textbf{Source}} &\n\\makecell{\\textbf{External}\\\\\\textbf{Knowledge}\\\\\\textbf{Source}} &\n\\textbf{Access} \\\\\n\\midrule\nFakeddit~\\cite{nakamura-etal-2020-fakeddit} & Img/Txt & - & Mixed & Synthetic & Reddit & \\ding{55} & \\ding{51} \\\\\nNeuralNews~\\cite{tan-etal-2020-detecting} & Img/Txt & - & OOC & Mixed & Grover/GoodNews & \\ding{55} & \\ding{51}\\\\\nNewsCLIPings~\\cite{luo-etal-2021-newsclippings} & Img/Txt & - & OOC & Mixed & CLIP/VisualNews & \\ding{55} & \\ding{51} \\\\\nFactify~\\cite{suryavardan2023factify} & Img/Txt & Img/Txt & OOC & Mixed & Twitter & \\ding{55} & \\ding{51}\\\\\nVERITE~\\cite{papadopoulos2024verite} & Img/Txt & - & Mixed & Mixed & Snopes/Reuters & \\ding{55} & \\ding{51}\\\\\nFauxtography~\\cite{zlatkova-etal-2019-fact} & Img/Txt & Meta & FV & Expert & Snopes/Reuters & \\ding{55} & \\ding{51} \\\\\nMMM~\\cite{gupta2022mmm} & Img/Txt & Txt & FV & Mixed & FC webs & \\ding{55} & \\ding{55} \\\\\nFactDrill~\\cite{singhal2022factdrill} & Txt & Vid/Aud/Img/Txt/Meta & FV & Expert & FC webs & \\ding{55} & \\ding{55} \\\\\nChartFC~\\cite{akhtar-etal-2023-reading} & Txt & Img & FV & Crowd & TabFact & \\ding{55} & \\ding{51}\\\\\nChartCheck~\\cite{chartcheck} & Txt & Img/Txt & FV & Crowd & Wikimedia & \\ding{55} & \\ding{51} \\\\\nMOCHEG~\\cite{mocheg} & Txt & Img/Txt & FV & Expert & FC webs & \\ding{51} & \\ding{51} \\\\\nFIN-FACT~\\cite{rangapur2025fin} & Txt & Img/Txt & FV & Expert & FC webs & \\ding{55} & \\ding{51} \\\\\nAVerImaTeC~ & Img/Txt & Img/Txt & FV & Expert & FC webs & \\ding{51} & \\ding{51}\\\\\n\\bottomrule\n\\end{tabular}\n}\n\\caption{Datasets for multimodal fact-checking (OOC: out-of-context misinformation; FV: fact verification).}\n\\label{tab:dataset_table}\n\\end{table*}", + "Target-citing paragraph: Among existing multimodal fact-checking datasets (Table~\\ref{tab:dataset_table}), we select MOCHEG~\\cite{mocheg} as the primary testbed for our study based on three criteria aligned with our research objectives. First, it must explicitly target \\emph{fact verification}, thereby excluding datasets designed for related but distinct tasks, such as Factify~\\cite{suryavardan2023factify}, which focuses on out-of-context detection. Second, it must take a textual claim $c$ as input, excluding datasets such as AVerImaTeC~ that assumes text-image input pairs. Third, it must provide the knowledge source $K$ for the retrieval of both textual and visual evidence. To the best of our knowledge, MOCHEG~\\cite{mocheg} is the only publicly available dataset meeting all these requirements, as ChartCheck provides only chart-caption evidence pairs. To assess the generalizability beyond a single dataset, we use two additional datasets, FIN-FACT and WebFC, exclusively for testing.", + "Target-citing paragraph: \\textbf{WebFC} is a newly constructed corpus developed in this study, consisting of 621 recently published claims and their corresponding verdicts from PolitiFact, spanning January 2024 to September 2025. Treating web documents as the knowledge source $K$, we retrieved both textual and visual evidence using the Google Custom Search API. To prevent retrieving documents published after the corresponding articles, which would make the setting unrealistic, we restricted retrieval to sources published before the fact-checking articles, following recent practice~. Accordingly, WebFC better reflects a temporally realistic fact-checking scenario in which novel claims are verified using web-retrieved evidence." + ], + "source": "arxiv_paragraphs" + }, + { + "citing_paper_id": "c848b5f99642678b699e8cdffe9e74fdc4822eff", + "citing_title": "REVEAL: Retrieval-Enhanced Verification for Multimodal Fact-Checking", + "label": "EXTENDS", + "paper_claim": "Builds upon the AVerImaTeC baseline framework for multimodal claim verification.", + "claim": "Builds upon the AVerImaTeC baseline framework for multimodal claim verification.", + "cluster_title": "Extends baseline framework for claim verification", + "cluster_key": "EXTENDS|baseline_framework|claim_verification", + "evidence_span": "Our system builds upon the official AVerImaTeC baseline , a modular agentic framework for multimodal claim verification", + "rationale": "The citing paper explicitly states that their system builds upon the baseline framework introduced in the target paper.", + "contexts": [ + "Target sentence: Our system builds upon the official AVerImaTeC baseline , a modular agentic framework for multimodal claim verification that operates on a decompose-and-verify paradigm." + ], + "source": "usage_contexts_fallback" + }, + { + "citing_paper_id": "e49f9fd1499d57ea7afd0ea0e8d7d9fda3f18eeb", + "citing_title": "VILLAIN at AVerImaTeC: Verifying Image-Text Claims via Multi-Agent Collaboration", + "label": "USES", + "paper_claim": "The citing paper uses the set of knowledge stores provided by AVerImaTeC to retrieve web evidence for verifying image-text claims.", + "claim": "The citing paper uses the set of knowledge stores provided by AVerImaTeC to retrieve web evidence for verifying image-text claims.", + "cluster_title": "Uses target dataset for evaluation", + "cluster_key": "USES|dataset|evaluation", + "evidence_span": "We use the set of knowledge stores provided with AVerImaTeC~, denoted as $\\mathcal{K}=\\{\\mathcal{K}^{txt}_{T},\\mathcal{K}^{img}_{T},\\mathcal{K}_{I}\\}$", + "rationale": "The citing paper explicitly uses the knowledge stores and datasets provided by the target paper's shared task.", + "contexts": [ + "Target-citing paragraph: Motivated by the spread of misleading multimodal content and the need for automated verification, the AVerImaTeC shared task at the 9th FEVER Workshop focuses on verifying real-world image\u2013text claims using external evidence~, reflecting web-based fact-checking scenarios in the wild. Given a claim and its metadata, systems must retrieve relevant evidence from the web or from a provided document and image collection and assess the claim's validity. Claims are labeled as Supported, Refuted, Not Enough Evidence, or Conflicting Evidence/Cherry-picking, and a prediction is considered correct only if both the label and the accompanying evidence are appropriate.", + "Target-citing paragraph: The AVerImaTeC shared task focuses on developing automated multimodal fact-checking systems capable of verifying real-world image--text claims using retrieved web evidence. Given an image-text claim along with its associated metadata, a fact-checking system is required to generate a verdict. The supporting evidence may be retrieved either directly from the web or from the document and image evidence collections provided by the organizers. A response is considered correct only if both the predicted label is correct and the provided evidence is adequate. Specifically, a verdict is deemed correct only when the evidence score meets or exceeds the threshold $\\lambda=0.3$. Further details can be found in .", + "Target-citing paragraph: The first stage aims to retrieve web evidence to verify a given claim. We use the set of knowledge stores provided with AVerImaTeC~, denoted as $\\mathcal{K}=\\{\\mathcal{K}^{txt}_{T},\\mathcal{K}^{img}_{T},\\mathcal{K}_{I}\\}$. Each of the three stores contains a distinct type of evidence. Specifically, $\\mathcal{K}^{txt}_{T}$ denotes the knowledge store consisting of web text retrieved using a textual query; $\\mathcal{K}^{img}_{T}$ denotes the knowledge store comprising textual evidence obtained via reverse image search applied to the claim images; and $\\mathcal{K}_{I}$ denotes the knowledge store of web images retrieved using a textual query." + ], + "source": "arxiv_paragraphs" + } + ], + "source": "arxiv_paragraphs", + "fallback_citing_ids": [ + "0f98265df1c8f94461b32ddcf6ce181d079f7b46", + "c848b5f99642678b699e8cdffe9e74fdc4822eff" + ] +} \ No newline at end of file diff --git a/replay_traces/2505.17978/processed_papers/2505.17978/usage_discovery_from_contributions.json b/replay_traces/2505.17978/processed_papers/2505.17978/usage_discovery_from_contributions.json new file mode 100644 index 0000000000000000000000000000000000000000..ff102ef23585541da6d6c5e5d7a034d5d00a0a04 --- /dev/null +++ b/replay_traces/2505.17978/processed_papers/2505.17978/usage_discovery_from_contributions.json @@ -0,0 +1,81 @@ +{ + "paper_id": "2505.17978", + "decision": "", + "justification": "", + "clusters": [ + { + "cluster_id": "C1", + "count": "1", + "representative_claim": "Extends target dataset for new dataset creation", + "cluster_key": "EXTENDS|dataset|dataset_creation", + "cluster_title": "Extends target dataset for new dataset creation", + "claim_indices": [ + "0" + ], + "source_cluster_ids": [ + "C1" + ], + "merge_rationale": "Directly builds upon the target dataset by extracting its pairs to construct a new dataset." + }, + { + "cluster_id": "C2", + "count": "2", + "representative_claim": "Uses target dataset and knowledge stores for evaluation", + "cluster_key": "USES|dataset|evaluation", + "cluster_title": "Uses target dataset and knowledge stores for evaluation", + "claim_indices": [ + "1", + "5" + ], + "source_cluster_ids": [ + "C2" + ], + "merge_rationale": "Directly uses the target dataset and its provided knowledge stores to evaluate proposed claim verification methods." + }, + { + "cluster_id": "C3", + "count": "1", + "representative_claim": "Extends reference-based evaluation method for a novel setting", + "cluster_key": "EXTENDS|evaluation_method|domain_adaptation", + "cluster_title": "Extends reference-based evaluation method for a novel setting", + "claim_indices": [ + "2" + ], + "source_cluster_ids": [ + "C3" + ], + "merge_rationale": "Adapts and extends the evaluation method introduced in the target paper to a new multimodal meme understanding setting." + }, + { + "cluster_id": "C4", + "count": "1", + "representative_claim": "Adopts temporally constrained retrieval methodology for dataset creation", + "cluster_key": "USES|methodology|dataset_creation", + "cluster_title": "Adopts temporally constrained retrieval methodology for dataset creation", + "claim_indices": [ + "3" + ], + "source_cluster_ids": [ + "C4" + ], + "merge_rationale": "Adopts the target paper's specific methodology of restricting evidence retrieval by date to create a new dataset." + }, + { + "cluster_id": "C5", + "count": "1", + "representative_claim": "Extends target baseline framework for multimodal claim verification", + "cluster_key": "EXTENDS|baseline_framework|claim_verification", + "cluster_title": "Extends target baseline framework for multimodal claim verification", + "claim_indices": [ + "4" + ], + "source_cluster_ids": [ + "C5" + ], + "merge_rationale": "Builds upon the baseline verification framework introduced by the target paper to develop a new system." + } + ], + "dropped_clusters": [], + "cluster_refine_method": "llm_centroid_merge_filter", + "cluster_refine_source": "usage_contributions.json" +} \ No newline at end of file diff --git a/replay_traces/2505.17978/processed_papers/2505.17978/usage_uses_extends_verified.json b/replay_traces/2505.17978/processed_papers/2505.17978/usage_uses_extends_verified.json new file mode 100644 index 0000000000000000000000000000000000000000..9437ed8ba0ec31cd506245056de0f9b4e3c59020 --- /dev/null +++ b/replay_traces/2505.17978/processed_papers/2505.17978/usage_uses_extends_verified.json @@ -0,0 +1,230 @@ +{ + "paper_id": "2505.17978", + "target": { + "title": "AVerImaTeC: A Dataset for Automatic Verification of Image-Text Claims with Evidence from the Web", + "first_author_last": "Cao", + "year": "2025" + }, + "candidates_total": 12, + "candidates_considered": 12, + "verification_batch_size": 25, + "verification_num_batches": 1, + "candidates_selected": 9, + "verified": [ + { + "id": 23, + "label": "USES", + "cue_span": "We use the set of knowledge stores provided with AVerImaTeC ", + "rationale": "The citing paper explicitly states they use the knowledge stores provided by the target paper.", + "text": "We use the set of knowledge stores provided with AVerImaTeC , denoted as K = {K txtT , K , K I } .", + "citing_paper_id": "e49f9fd1499d57ea7afd0ea0e8d7d9fda3f18eeb", + "citing_title": "VILLAIN at AVerImaTeC: Verifying Image-Text Claims via Multi-Agent Collaboration", + "original_label": "Uses" + }, + { + "id": 20, + "label": "USES", + "cue_span": "We evaluate our pipeline using Gemma-3-27B (Team et al., 2025), the model used for evaluation in the official ", + "rationale": "The citing paper uses the evaluation model specified in the target paper's repository.", + "text": "We evaluate our pipeline using Gemma-3-27B (Team et al., 2025), the model used for evaluation in the official shared task repository 2 .", + "citing_paper_id": "e49f9fd1499d57ea7afd0ea0e8d7d9fda3f18eeb", + "citing_title": "VILLAIN at AVerImaTeC: Verifying Image-Text Claims via Multi-Agent Collaboration", + "original_label": "Uses" + }, + { + "id": 4, + "label": "USES", + "cue_span": "Following ; Akhtar et al., 2026), we convert", + "rationale": "The citing paper follows the methodology of the target paper to convert QA pairs to evidence statements.", + "text": "Following ; Akhtar et al., 2026), we convert QA pairs to evidence statement, for both evidence evaluation and maintaining the evidence history.", + "citing_paper_id": "0d5c8b3bfa46dc2f4efa4a5537716473db3101af", + "citing_title": "I Know What You Meme, Even If it Emerged Today: Understanding Evolving Memes through Open-World Knowledge Acquisition", + "original_label": "Uses" + }, + { + "id": 34, + "label": "NOT_CONFIRMED", + "cue_span": "", + "rationale": "The sentence describes the definition from the target paper but lacks an explicit citing-paper actor phrase like 'we' or 'our'.", + "text": "Following the definition , a predicted verdict is considered correct only when the associated evidence score exceeds the predefined threshold \u03bb = 0.3; otherwise, the prediction is counted as incorrect irrespective of its label.", + "citing_paper_id": "0f98265df1c8f94461b32ddcf6ce181d079f7b46", + "citing_title": "Selective Multimodal Retrieval for Automated Verification of Image\u2013Text Claims", + "original_label": "Uses" + }, + { + "id": 33, + "label": "USES", + "cue_span": "We conduct our experiments on the AVerImaTeC (Automated Verification of Image\u2013Text Claim) dataset ", + "rationale": "The citing paper explicitly uses the AVerImaTeC dataset for its experiments.", + "text": "We conduct our experiments on the AVerImaTeC (Automated Verification of Image\u2013Text Claim) dataset , which comprises 1,297 real-world fact-checked image\u2013text claims split", + "citing_paper_id": "0f98265df1c8f94461b32ddcf6ce181d079f7b46", + "citing_title": "Selective Multimodal Retrieval for Automated Verification of Image\u2013Text Claims", + "original_label": "Uses" + }, + { + "id": 22, + "label": "USES", + "cue_span": "As specified by the evaluation protocol, we report", + "rationale": "The citing paper uses the evaluation protocol and metrics defined by the target paper.", + "text": "As specified by the evaluation protocol, we report four metrics: \u2022 Q-Eval : A reference-based recall metric for comparing the generated questions with the gold questions.", + "citing_paper_id": "e49f9fd1499d57ea7afd0ea0e8d7d9fda3f18eeb", + "citing_title": "VILLAIN at AVerImaTeC: Verifying Image-Text Claims via Multi-Agent Collaboration", + "original_label": "Uses" + }, + { + "id": 40, + "label": "NOT_CONFIRMED", + "cue_span": "", + "rationale": "The sentence merely describes the metrics selected in the target paper without stating the citing paper's adoption of them.", + "text": "Metrics The metrics selected in to evaluate proposed approaches follow different LLM-as-a-judge strategies: \u2022 Question Score (Q-score): Generated questions are scored by an LLM in comparison to reference questions from the gold data; \u2022 Evidence Score (E-score): Similar to the\u2026", + "citing_paper_id": "02d960f62ad7936094abfba07f0535cccabc7b35", + "citing_title": "Take It All: Ensemble Retrieval for Multimodal Evidence Aggregation", + "original_label": "Uses" + }, + { + "id": 25, + "label": "USES", + "cue_span": "We construct MMCE using post-claim pairs extracted from AVerImaTeC", + "rationale": "The citing paper uses data extracted from the target paper's dataset to construct its own dataset.", + "text": "Dataset We construct MMCE using post-claim pairs extracted from AVerImaTeC, a recently proposed dataset for real-world image-text claim verification based on data from fact-checking organizations .", + "citing_paper_id": "3aa1b82af3c14c2b20d94cdca1a3804ad7364cbe", + "citing_title": "Multimodal Claim Extraction for Fact-Checking", + "original_label": "Uses" + }, + { + "id": 11, + "label": "USES", + "cue_span": "we restricted retrieval to sources published before the fact-checking articles, following recent practice ", + "rationale": "The citing paper adopts a specific retrieval restriction methodology following the target paper.", + "text": "To prevent retrieving documents published after the corresponding articles, which would make the setting unrealistic, we restricted retrieval to sources published before the fact-checking articles, following recent practice .", + "citing_paper_id": "3a74d04a13ecbeb3767b4af354b9ee0f73341f18", + "citing_title": "Is a Picture Worth a Thousand Words? Adaptive Multimodal Fact-Checking with Visual Evidence Necessity", + "original_label": "Uses" + }, + { + "id": 24, + "label": "USES", + "cue_span": "We present VILLAIN , the winning system in the shared task", + "rationale": "The citing paper is reporting on their system developed for and used within the target paper's shared task.", + "text": "We present VILLAIN , the winning system in the shared task.", + "citing_paper_id": "e49f9fd1499d57ea7afd0ea0e8d7d9fda3f18eeb", + "citing_title": "VILLAIN at AVerImaTeC: Verifying Image-Text Claims via Multi-Agent Collaboration", + "original_label": "Uses" + }, + { + "id": 37, + "label": "EXTENDS", + "cue_span": "Our system builds upon the official AVerImaTeC baseline ", + "rationale": "The citing paper explicitly states that their system builds upon the baseline provided by the target paper.", + "text": "Our system builds upon the official AVerImaTeC baseline , a modular agentic framework for multimodal claim verification that operates on a decompose-and-verify paradigm.", + "citing_paper_id": "c848b5f99642678b699e8cdffe9e74fdc4822eff", + "citing_title": "REVEAL: Retrieval-Enhanced Verification for Multimodal Fact-Checking", + "original_label": "Uses" + }, + { + "id": 31, + "label": "NOT_CONFIRMED", + "cue_span": "", + "rationale": "The target paper is listed among other datasets that applied a specific filtering step, without an explicit adoption statement by the citing paper.", + "text": "To ensure realistic evaluation, evidence written by fact-checking organizations or published after the claim date should be excluded, a filtering step applied by some datasets (Zlatkova et al., 2019; Tonglet et al., 2024; .", + "citing_paper_id": "34372cd2d1ab54452044a6169143ee66ec19fa9c", + "citing_title": "M4FC: a Multimodal, Multilingual, Multicultural, Multitask Real-World Fact-Checking Dataset", + "original_label": "Uses" + } + ], + "confirmed": [ + { + "id": 23, + "label": "USES", + "cue_span": "We use the set of knowledge stores provided with AVerImaTeC ", + "rationale": "The citing paper explicitly states they use the knowledge stores provided by the target paper.", + "text": "We use the set of knowledge stores provided with AVerImaTeC , denoted as K = {K txtT , K , K I } .", + "citing_paper_id": "e49f9fd1499d57ea7afd0ea0e8d7d9fda3f18eeb", + "citing_title": "VILLAIN at AVerImaTeC: Verifying Image-Text Claims via Multi-Agent Collaboration", + "original_label": "Uses" + }, + { + "id": 20, + "label": "USES", + "cue_span": "We evaluate our pipeline using Gemma-3-27B (Team et al., 2025), the model used for evaluation in the official ", + "rationale": "The citing paper uses the evaluation model specified in the target paper's repository.", + "text": "We evaluate our pipeline using Gemma-3-27B (Team et al., 2025), the model used for evaluation in the official shared task repository 2 .", + "citing_paper_id": "e49f9fd1499d57ea7afd0ea0e8d7d9fda3f18eeb", + "citing_title": "VILLAIN at AVerImaTeC: Verifying Image-Text Claims via Multi-Agent Collaboration", + "original_label": "Uses" + }, + { + "id": 4, + "label": "USES", + "cue_span": "Following ; Akhtar et al., 2026), we convert", + "rationale": "The citing paper follows the methodology of the target paper to convert QA pairs to evidence statements.", + "text": "Following ; Akhtar et al., 2026), we convert QA pairs to evidence statement, for both evidence evaluation and maintaining the evidence history.", + "citing_paper_id": "0d5c8b3bfa46dc2f4efa4a5537716473db3101af", + "citing_title": "I Know What You Meme, Even If it Emerged Today: Understanding Evolving Memes through Open-World Knowledge Acquisition", + "original_label": "Uses" + }, + { + "id": 33, + "label": "USES", + "cue_span": "We conduct our experiments on the AVerImaTeC (Automated Verification of Image\u2013Text Claim) dataset ", + "rationale": "The citing paper explicitly uses the AVerImaTeC dataset for its experiments.", + "text": "We conduct our experiments on the AVerImaTeC (Automated Verification of Image\u2013Text Claim) dataset , which comprises 1,297 real-world fact-checked image\u2013text claims split", + "citing_paper_id": "0f98265df1c8f94461b32ddcf6ce181d079f7b46", + "citing_title": "Selective Multimodal Retrieval for Automated Verification of Image\u2013Text Claims", + "original_label": "Uses" + }, + { + "id": 22, + "label": "USES", + "cue_span": "As specified by the evaluation protocol, we report", + "rationale": "The citing paper uses the evaluation protocol and metrics defined by the target paper.", + "text": "As specified by the evaluation protocol, we report four metrics: \u2022 Q-Eval : A reference-based recall metric for comparing the generated questions with the gold questions.", + "citing_paper_id": "e49f9fd1499d57ea7afd0ea0e8d7d9fda3f18eeb", + "citing_title": "VILLAIN at AVerImaTeC: Verifying Image-Text Claims via Multi-Agent Collaboration", + "original_label": "Uses" + }, + { + "id": 25, + "label": "USES", + "cue_span": "We construct MMCE using post-claim pairs extracted from AVerImaTeC", + "rationale": "The citing paper uses data extracted from the target paper's dataset to construct its own dataset.", + "text": "Dataset We construct MMCE using post-claim pairs extracted from AVerImaTeC, a recently proposed dataset for real-world image-text claim verification based on data from fact-checking organizations .", + "citing_paper_id": "3aa1b82af3c14c2b20d94cdca1a3804ad7364cbe", + "citing_title": "Multimodal Claim Extraction for Fact-Checking", + "original_label": "Uses" + }, + { + "id": 11, + "label": "USES", + "cue_span": "we restricted retrieval to sources published before the fact-checking articles, following recent practice ", + "rationale": "The citing paper adopts a specific retrieval restriction methodology following the target paper.", + "text": "To prevent retrieving documents published after the corresponding articles, which would make the setting unrealistic, we restricted retrieval to sources published before the fact-checking articles, following recent practice .", + "citing_paper_id": "3a74d04a13ecbeb3767b4af354b9ee0f73341f18", + "citing_title": "Is a Picture Worth a Thousand Words? Adaptive Multimodal Fact-Checking with Visual Evidence Necessity", + "original_label": "Uses" + }, + { + "id": 24, + "label": "USES", + "cue_span": "We present VILLAIN , the winning system in the shared task", + "rationale": "The citing paper is reporting on their system developed for and used within the target paper's shared task.", + "text": "We present VILLAIN , the winning system in the shared task.", + "citing_paper_id": "e49f9fd1499d57ea7afd0ea0e8d7d9fda3f18eeb", + "citing_title": "VILLAIN at AVerImaTeC: Verifying Image-Text Claims via Multi-Agent Collaboration", + "original_label": "Uses" + }, + { + "id": 37, + "label": "EXTENDS", + "cue_span": "Our system builds upon the official AVerImaTeC baseline ", + "rationale": "The citing paper explicitly states that their system builds upon the baseline provided by the target paper.", + "text": "Our system builds upon the official AVerImaTeC baseline , a modular agentic framework for multimodal claim verification that operates on a decompose-and-verify paradigm.", + "citing_paper_id": "c848b5f99642678b699e8cdffe9e74fdc4822eff", + "citing_title": "REVEAL: Retrieval-Enhanced Verification for Multimodal Fact-Checking", + "original_label": "Uses" + } + ], + "confirmed_extends": 1, + "confirmed_uses": 8, + "final_label": "EXTENDS" +} \ No newline at end of file diff --git a/replay_traces/2505.17978/replay_meta.json b/replay_traces/2505.17978/replay_meta.json new file mode 100644 index 0000000000000000000000000000000000000000..c86b826a3512136fb0f29da517e291c05af24294 --- /dev/null +++ b/replay_traces/2505.17978/replay_meta.json @@ -0,0 +1,27 @@ +{ + "label": "AVerImaTeC", + "arxiv_id": "2505.17978", + "paper_input": "https://arxiv.org/abs/2505.17978", + "status": "Completed", + "built_at": "2026-07-26T20:53:50.134407+00:00", + "source_job_dir": "hf_space/runs/job_1785098515_4b6a9bbd", + "paper_dir": "processed_papers/2505.17978", + "annotation_payload_path": "two_pass_outputs/20260726T204323Z/pass_2_ui_payload.json", + "annotation_skipped_reason": null, + "pipeline_failed_reason": null, + "pipeline_stopped_reason": null, + "events": [ + "[1] Fetch metadata + LaTeX for input paper", + "[2] Add citation markers", + "[3] Build usage contexts", + "[4] Label citation functions", + "[5] Verify USES/EXTENDS", + "[6] Extract arXiv paragraphs", + "[7] Extract target contributions and refine clusters", + "SUCCESS", + "[annotation] complete: hf_space/runs/job_1785098515_4b6a9bbd/two_pass_outputs/2505.17978/20260726T204323Z" + ], + "cluster_count": 5, + "contribution_count": 6, + "claim_count": 4 +} \ No newline at end of file diff --git a/replay_traces/2505.17978/run_config.json b/replay_traces/2505.17978/run_config.json new file mode 100644 index 0000000000000000000000000000000000000000..99ca737a6b503243be2185ed8b6b4ccaccb9b91c --- /dev/null +++ b/replay_traces/2505.17978/run_config.json @@ -0,0 +1,22 @@ +{ + "paper_input": "https://arxiv.org/abs/2505.17978", + "paper_id": "2505.17978", + "arxiv_id": "2505.17978", + "source_root": "/Users/dingzifeng/SciFy-UMBC/demo_2026_07/src/processed_papers", + "steps": [ + 1, + 2, + 3, + 4, + 5, + 6, + 7, + "annotation" + ], + "llm_provider": "gemini", + "llm_model": "gemini-3.1-pro-preview", + "llm_model_step4": "gemini-3-flash-preview", + "device": "cpu", + "embedding_model": "sentence-transformers/all-mpnet-base-v2", + "timestamp": 1785098515 +} \ No newline at end of file diff --git a/replay_traces/2505.17978/scipaths_run_results.json b/replay_traces/2505.17978/scipaths_run_results.json new file mode 100644 index 0000000000000000000000000000000000000000..4e994a2c371fb17b3b235f9e51c92835e9cdbf33 --- /dev/null +++ b/replay_traces/2505.17978/scipaths_run_results.json @@ -0,0 +1,473 @@ +{ + "citation_clusters": [ + { + "cluster_id": "C1", + "count": "1", + "representative_claim": "Extends target dataset for new dataset creation", + "cluster_key": "EXTENDS|dataset|dataset_creation", + "cluster_title": "Extends target dataset for new dataset creation", + "claim_indices": [ + "0" + ], + "source_cluster_ids": [ + "C1" + ], + "merge_rationale": "Directly builds upon the target dataset by extracting its pairs to construct a new dataset." + }, + { + "cluster_id": "C2", + "count": "2", + "representative_claim": "Uses target dataset and knowledge stores for evaluation", + "cluster_key": "USES|dataset|evaluation", + "cluster_title": "Uses target dataset and knowledge stores for evaluation", + "claim_indices": [ + "1", + "5" + ], + "source_cluster_ids": [ + "C2" + ], + "merge_rationale": "Directly uses the target dataset and its provided knowledge stores to evaluate proposed claim verification methods." + }, + { + "cluster_id": "C3", + "count": "1", + "representative_claim": "Extends reference-based evaluation method for a novel setting", + "cluster_key": "EXTENDS|evaluation_method|domain_adaptation", + "cluster_title": "Extends reference-based evaluation method for a novel setting", + "claim_indices": [ + "2" + ], + "source_cluster_ids": [ + "C3" + ], + "merge_rationale": "Adapts and extends the evaluation method introduced in the target paper to a new multimodal meme understanding setting." + }, + { + "cluster_id": "C4", + "count": "1", + "representative_claim": "Adopts temporally constrained retrieval methodology for dataset creation", + "cluster_key": "USES|methodology|dataset_creation", + "cluster_title": "Adopts temporally constrained retrieval methodology for dataset creation", + "claim_indices": [ + "3" + ], + "source_cluster_ids": [ + "C4" + ], + "merge_rationale": "Adopts the target paper's specific methodology of restricting evidence retrieval by date to create a new dataset." + }, + { + "cluster_id": "C5", + "count": "1", + "representative_claim": "Extends target baseline framework for multimodal claim verification", + "cluster_key": "EXTENDS|baseline_framework|claim_verification", + "cluster_title": "Extends target baseline framework for multimodal claim verification", + "claim_indices": [ + "4" + ], + "source_cluster_ids": [ + "C5" + ], + "merge_rationale": "Builds upon the baseline verification framework introduced by the target paper to develop a new system." + } + ], + "target_contribution_decompositions": [ + { + "claim_id": "C1", + "text": "Dataset: A real-world image-text claim verification dataset annotated with question-answer reasoning paths and web evidence, enabling the training and evaluation of multimodal fact-checking systems.", + "rewritten_claim": "Dataset: A real-world image-text claim verification dataset annotated with question-answer reasoning paths and web evidence, enabling the training and evaluation of multimodal fact-checking systems.", + "cluster_id": "C1, C2", + "decision": "YES_SUFFICIENT", + "notes": "", + "ingredients": [ + { + "ingredient_id": "C1.I1", + "ingredient": "Source fact-checking articles containing multimodal claims", + "canonical_ref_id": "__NONE__", + "canonical_grounding": null, + "additional_ref_ids": [ + "ref_averitec", + "ref_ammeba" + ], + "additional_groundings": [ + { + "ref_id": "ref_averitec", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web", + "ref_year": "", + "ref_authors": "" + }, + { + "ref_id": "ref_ammeba", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "AMMeBa: A Large-Scale Survey and Dataset of Media-Based Misinformation In-The-Wild", + "ref_year": "", + "ref_authors": "" + } + ], + "canonical_annotation": { + "role": "DATA_SOURCE", + "roles": [ + "DATA_SOURCE" + ], + "contribution": "Provides the raw real-world image-text claims that are normalized and annotated to form the dataset.", + "rationale": "The dataset is a composite of claims extracted from multiple prior fact-checking resources. No single prior study fully represents this substrate, making NONE the correct canonical grounding, with the primary sources listed as additional groundings.", + "evidence_span": "we incorporated partially annotated articles from three sources: 1) filtered articles from AVeriTec... 2) articles verifying image-related claims from AMMEBA" + } + }, + { + "ingredient_id": "C1.I2", + "ingredient": "Question-answering representation for verification rationale", + "canonical_ref_id": "ref_complex", + "canonical_grounding": { + "ref_id": "ref_complex", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "Complex Claim Verification with Evidence Retrieved in the Wild", + "ref_year": "", + "ref_authors": "" + }, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "CONCEPTUAL_FRAMEWORK", + "roles": [ + "CONCEPTUAL_FRAMEWORK" + ], + "contribution": "Provides the QA-based reasoning representation used to structure the evidence retrieval and verification process.", + "rationale": "The paper explicitly adopts QA as the reasoning representation for fact-checking, citing prior work that established this paradigm. Chen et al. (CLAIMDECOMP) is a clean representative for QA-based claim decomposition.", + "evidence_span": "To capture the rationale in claim verification, a complex reasoning task, various reasoning representations have been explored... and QA" + } + }, + { + "ingredient_id": "C1.I3", + "ingredient": "Four-way veracity labeling schema", + "canonical_ref_id": "ref_averitec", + "canonical_grounding": { + "ref_id": "ref_averitec", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web", + "ref_year": "", + "ref_authors": "" + }, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "CONCEPTUAL_FRAMEWORK", + "roles": [ + "CONCEPTUAL_FRAMEWORK" + ], + "contribution": "Provides the specific four-way label space (supported, refuted, not enough evidence, conflicting) used to annotate the claims.", + "rationale": "The paper explicitly adopts the exact four-way veracity labeling schema introduced in AVeriTeC.", + "evidence_span": "We follow the four-way veracity labeling schema from [Schlichtkrull et al., 2023]: supported, refuted, not enough evidence, and conflicting/cherry-picking." + } + }, + { + "ingredient_id": "C1.I4", + "ingredient": "Multi-stage annotation pipeline with evidence sufficiency checks", + "canonical_ref_id": "ref_averitec", + "canonical_grounding": { + "ref_id": "ref_averitec", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web", + "ref_year": "", + "ref_authors": "" + }, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "EVALUATION_PROTOCOL", + "roles": [ + "EVALUATION_PROTOCOL" + ], + "contribution": "Provides the foundational multi-stage annotation and verification workflow that the paper extends for image-text claims.", + "rationale": "The paper explicitly states that its annotation pipeline is an extension of the process proposed in AVeriTeC, making it the canonical grounding for this workflow.", + "evidence_span": "extending the annotation process proposed in [Schlichtkrull et al., 2023] to the domain of image-text claims." + } + } + ], + "enabling_discoveries": [ + { + "ref_id": "ref_complex", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "Complex Claim Verification with Evidence Retrieved in the Wild", + "ref_year": "", + "ref_authors": "", + "ingredient_id": "C1.I2", + "ingredient": "Question-answering representation for verification rationale", + "role": "CONCEPTUAL_FRAMEWORK", + "roles": [ + "CONCEPTUAL_FRAMEWORK" + ], + "contribution": "Provides the QA-based reasoning representation used to structure the evidence retrieval and verification process.", + "rationale": "The paper explicitly adopts QA as the reasoning representation for fact-checking, citing prior work that established this paradigm. Chen et al. (CLAIMDECOMP) is a clean representative for QA-based claim decomposition.", + "evidence_span": "To capture the rationale in claim verification, a complex reasoning task, various reasoning representations have been explored... and QA" + }, + { + "ref_id": "ref_averitec", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web", + "ref_year": "", + "ref_authors": "", + "ingredient_id": "C1.I3", + "ingredient": "Four-way veracity labeling schema", + "role": "CONCEPTUAL_FRAMEWORK", + "roles": [ + "CONCEPTUAL_FRAMEWORK" + ], + "contribution": "Provides the specific four-way label space (supported, refuted, not enough evidence, conflicting) used to annotate the claims.", + "rationale": "The paper explicitly adopts the exact four-way veracity labeling schema introduced in AVeriTeC.", + "evidence_span": "We follow the four-way veracity labeling schema from [Schlichtkrull et al., 2023]: supported, refuted, not enough evidence, and conflicting/cherry-picking." + }, + { + "ref_id": "ref_averitec", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web", + "ref_year": "", + "ref_authors": "", + "ingredient_id": "C1.I4", + "ingredient": "Multi-stage annotation pipeline with evidence sufficiency checks", + "role": "EVALUATION_PROTOCOL", + "roles": [ + "EVALUATION_PROTOCOL" + ], + "contribution": "Provides the foundational multi-stage annotation and verification workflow that the paper extends for image-text claims.", + "rationale": "The paper explicitly states that its annotation pipeline is an extension of the process proposed in AVeriTeC, making it the canonical grounding for this workflow.", + "evidence_span": "extending the annotation process proposed in [Schlichtkrull et al., 2023] to the domain of image-text claims." + } + ] + }, + { + "claim_id": "C2", + "text": "Method: A reference-based evaluation method using large language models to separately assess the textual and visual components of retrieved evidence, enabling automated alignment with human judgments in multimodal fact-checking.", + "rewritten_claim": "Method: A reference-based evaluation method using large language models to separately assess the textual and visual components of retrieved evidence, enabling automated alignment with human judgments in multimodal fact-checking.", + "cluster_id": "C3", + "decision": "YES_SUFFICIENT", + "notes": "", + "ingredients": [ + { + "ingredient_id": "C2.I1", + "ingredient": "LLM-based reference evaluation framework for evidence retrieval", + "canonical_ref_id": "ref_ev2r", + "canonical_grounding": { + "ref_id": "ref_ev2r", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking", + "ref_year": "", + "ref_authors": "" + }, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "EVALUATION_PROTOCOL", + "roles": [ + "EVALUATION_PROTOCOL" + ], + "contribution": "Provides the foundation for using LLMs to evaluate evidence retrieval by comparing it to references.", + "rationale": "The paper explicitly adopts the Ev2R framework for its reference-based evaluation of textual evidence.", + "evidence_span": "Recent research [Akhtar et al., 2024] showed that reference-based evaluation of evidence with large language models (LLMs) aligns best with human assessments." + } + }, + { + "ingredient_id": "C2.I2", + "ingredient": "Separated reference-based evaluation for textual and visual components", + "canonical_ref_id": "__NONE__", + "canonical_grounding": null, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "EVALUATION_PROTOCOL", + "roles": [ + "EVALUATION_PROTOCOL" + ], + "contribution": "Provides the novel mechanism for robustly evaluating multimodal evidence by decoupling the modalities during the LLM judgment phase.", + "rationale": "This separated evaluation strategy is the target paper's own novel extension to prior text-only reference evaluation methods, so it maps to NONE.", + "evidence_span": "We then conduct separate reference-based evaluations for the textual and visual components." + } + } + ], + "enabling_discoveries": [ + { + "ref_id": "ref_ev2r", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking", + "ref_year": "", + "ref_authors": "", + "ingredient_id": "C2.I1", + "ingredient": "LLM-based reference evaluation framework for evidence retrieval", + "role": "EVALUATION_PROTOCOL", + "roles": [ + "EVALUATION_PROTOCOL" + ], + "contribution": "Provides the foundation for using LLMs to evaluate evidence retrieval by comparing it to references.", + "rationale": "The paper explicitly adopts the Ev2R framework for its reference-based evaluation of textual evidence.", + "evidence_span": "Recent research [Akhtar et al., 2024] showed that reference-based evaluation of evidence with large language models (LLMs) aligns best with human assessments." + } + ] + }, + { + "claim_id": "C3", + "text": "Protocol: A temporally constrained evidence retrieval methodology that restricts search results to dates prior to the claim, enabling the construction of fact-checking datasets without temporal leakage.", + "rewritten_claim": "Protocol: A temporally constrained evidence retrieval methodology that restricts search results to dates prior to the claim, enabling the construction of fact-checking datasets without temporal leakage.", + "cluster_id": "C4", + "decision": "YES_SUFFICIENT", + "notes": "", + "ingredients": [ + { + "ingredient_id": "C3.I1", + "ingredient": "Identification of temporal leakage as a critical flaw in fact-checking datasets", + "canonical_ref_id": "ref_missing", + "canonical_grounding": { + "ref_id": "ref_missing", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "Missing Counter-Evidence Renders NLP Fact-Checking Unrealistic for Misinformation", + "ref_year": "", + "ref_authors": "" + }, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "CONCEPTUAL_FRAMEWORK", + "roles": [ + "CONCEPTUAL_FRAMEWORK" + ], + "contribution": "Provides the theoretical motivation for restricting evidence retrieval to prevent models from accessing future knowledge.", + "rationale": "The paper explicitly cites Glockner et al. as the motivation for preventing temporal leakage during evidence collection.", + "evidence_span": "restricted to dates prior to the claim date to prevent temporal leakage [Glockner et al., 2022]." + } + }, + { + "ingredient_id": "C3.I2", + "ingredient": "Date-restricted search interface and post-hoc filtering mechanism", + "canonical_ref_id": "__NONE__", + "canonical_grounding": null, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "IMPLEMENTATION_TOOLING", + "roles": [ + "IMPLEMENTATION_TOOLING" + ], + "contribution": "Provides the practical tooling to enforce the temporal constraints during dataset construction.", + "rationale": "The specific implementation of the custom search interface and post-hoc date filtering is developed in the target paper, mapping to NONE.", + "evidence_span": "annotators are provided with a custom Google search interface... restricted to dates prior to the claim date" + } + } + ], + "enabling_discoveries": [ + { + "ref_id": "ref_missing", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "Missing Counter-Evidence Renders NLP Fact-Checking Unrealistic for Misinformation", + "ref_year": "", + "ref_authors": "", + "ingredient_id": "C3.I1", + "ingredient": "Identification of temporal leakage as a critical flaw in fact-checking datasets", + "role": "CONCEPTUAL_FRAMEWORK", + "roles": [ + "CONCEPTUAL_FRAMEWORK" + ], + "contribution": "Provides the theoretical motivation for restricting evidence retrieval to prevent models from accessing future knowledge.", + "rationale": "The paper explicitly cites Glockner et al. as the motivation for preventing temporal leakage during evidence collection.", + "evidence_span": "restricted to dates prior to the claim date to prevent temporal leakage [Glockner et al., 2022]." + } + ] + }, + { + "claim_id": "C4", + "text": "Method: A multimodal claim verification framework combining dynamic question generation and tool-augmented answer generation, enabling automated fact-checking of image-text claims using open-web evidence.", + "rewritten_claim": "Method: A multimodal claim verification framework combining dynamic question generation and tool-augmented answer generation, enabling automated fact-checking of image-text claims using open-web evidence.", + "cluster_id": "C5", + "decision": "YES_SUFFICIENT", + "notes": "", + "ingredients": [ + { + "ingredient_id": "C4.I1", + "ingredient": "Tool-augmented answer generation module with specialized search and VQA tools", + "canonical_ref_id": "ref_visual", + "canonical_grounding": { + "ref_id": "ref_visual", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models", + "ref_year": "", + "ref_authors": "" + }, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "CORE_METHOD", + "roles": [ + "CORE_METHOD" + ], + "contribution": "Provides the paradigm of augmenting language models with specialized visual and search tools to solve complex multimodal tasks.", + "rationale": "The paper explicitly grounds its tool-augmented answer generation module in recent research on tool usage, with Visual ChatGPT (Wu et al.) being a canonical representative for integrating visual tools with LLMs.", + "evidence_span": "we integrate a set of specialized tools into the answer generation module, along with a tool selector" + } + }, + { + "ingredient_id": "C4.I2", + "ingredient": "Dynamic question generation strategy based on evolving evidence history", + "canonical_ref_id": "__NONE__", + "canonical_grounding": null, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "CORE_METHOD", + "roles": [ + "CORE_METHOD" + ], + "contribution": "Provides the mechanism for generating sequential verification questions conditioned on the accumulated evidence context.", + "rationale": "The dynamic and hybrid question generation strategies are proposed by the target paper as novel components of their baseline framework, mapping to NONE.", + "evidence_span": "we also propose a dynamic question generation (DQG) method, where each subsequent question is generated based on both the claim and the evolving evidence history." + } + } + ], + "enabling_discoveries": [ + { + "ref_id": "ref_visual", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models", + "ref_year": "", + "ref_authors": "", + "ingredient_id": "C4.I1", + "ingredient": "Tool-augmented answer generation module with specialized search and VQA tools", + "role": "CORE_METHOD", + "roles": [ + "CORE_METHOD" + ], + "contribution": "Provides the paradigm of augmenting language models with specialized visual and search tools to solve complex multimodal tasks.", + "rationale": "The paper explicitly grounds its tool-augmented answer generation module in recent research on tool usage, with Visual ChatGPT (Wu et al.) being a canonical representative for integrating visual tools with LLMs.", + "evidence_span": "we integrate a set of specialized tools into the answer generation module, along with a tool selector" + } + ] + } + ] +} \ No newline at end of file diff --git a/replay_traces/2505.17978/summary.txt b/replay_traces/2505.17978/summary.txt new file mode 100644 index 0000000000000000000000000000000000000000..91d02a8ae8d3807e9176d9ede6b5cb56952461ca --- /dev/null +++ b/replay_traces/2505.17978/summary.txt @@ -0,0 +1,8 @@ +[1] Fetch metadata + LaTeX for input paper +[2] Add citation markers +[3] Build usage contexts +[4] Label citation functions +[5] Verify USES/EXTENDS +[6] Extract arXiv paragraphs +[7] Extract target contributions and refine clusters +SUCCESS \ No newline at end of file diff --git a/replay_traces/2505.17978/two_pass_outputs/20260726T204323Z/formatter_attempts.json b/replay_traces/2505.17978/two_pass_outputs/20260726T204323Z/formatter_attempts.json new file mode 100644 index 0000000000000000000000000000000000000000..5f6accf4d248a65ffd7bf31bd6b1a0e921c17231 --- /dev/null +++ b/replay_traces/2505.17978/two_pass_outputs/20260726T204323Z/formatter_attempts.json @@ -0,0 +1,8 @@ +{ + "attempts": [ + { + "attempt": 1, + "status": "success" + } + ] +} diff --git a/replay_traces/2505.17978/two_pass_outputs/20260726T204323Z/pass_1_reasoning.judge.output.json b/replay_traces/2505.17978/two_pass_outputs/20260726T204323Z/pass_1_reasoning.judge.output.json new file mode 100644 index 0000000000000000000000000000000000000000..450b636090e27f41d4f9eb78fef1e6d0bf4babe2 --- /dev/null +++ b/replay_traces/2505.17978/two_pass_outputs/20260726T204323Z/pass_1_reasoning.judge.output.json @@ -0,0 +1,25 @@ +{ + "selected_candidate_index": 0, + "selected_candidate_id": "candidate_1", + "selected_reason": "Candidate 1 provides the most structurally faithful decomposition by correctly identifying and including the novel mechanisms introduced in the target paper (e.g., separated reference-based evaluation for C2, dynamic question generation for C4) as `NONE`-grounded ingredients. The other candidates miss these defining features and instead include generic implementation tools (like 'an MLLM') which are less structurally defining. Candidate 1 also correctly grounds the foundational concepts to appropriate prior works.", + "candidate_scores": [ + { + "candidate_id": "candidate_1", + "candidate_index": 0, + "score": 10, + "assessment": "Excellent candidate. It correctly splits the claims based on downstream usage and provides a minimal, structurally sufficient set of ingredients for each. It excels by capturing the novel structural components of the target paper's artifacts (separated evaluation, dynamic question generation) as `NONE`-grounded ingredients, which perfectly reflects the paper's actual contributions. Role assignments and canonical groundings are highly accurate." + }, + { + "candidate_id": "candidate_2", + "candidate_index": 1, + "score": 6, + "assessment": "Good claim splitting, but the ingredient decomposition is too minimal in the wrong ways. It misses the novel structural components of the evaluation method (separated evaluation) and the baseline framework (dynamic question generation). Instead of capturing the defining mechanisms, it includes generic implementation details like 'Pretrained MLLM' for C4, which is less structurally informative." + }, + { + "candidate_id": "candidate_3", + "candidate_index": 2, + "score": 7, + "assessment": "Good claim splitting and reasonable decomposition for the dataset. However, like Candidate 2, it misses the novel structural components of the evaluation metric and baseline framework, opting instead to include 'Large multimodal model' and 'MLLMs' as ingredients. While MLLMs are used, the actual structural innovations (separated evaluation, dynamic question generation) are what define the artifacts and should have been captured." + } + ] +} diff --git a/replay_traces/2505.17978/two_pass_outputs/20260726T204323Z/pass_1_reasoning.judge.prompt.txt b/replay_traces/2505.17978/two_pass_outputs/20260726T204323Z/pass_1_reasoning.judge.prompt.txt new file mode 100644 index 0000000000000000000000000000000000000000..5105f25b3409f70a02ab9d7262a493d1d3e675f9 --- /dev/null +++ b/replay_traces/2505.17978/two_pass_outputs/20260726T204323Z/pass_1_reasoning.judge.prompt.txt @@ -0,0 +1,1299 @@ +Task: Compare candidate reasoning memos for a single target paper. + +You will be given: +1. target paper metadata +2. the original extracted discovery claim +3. optional downstream usage evidence +4. multiple candidate reasoning outputs from pass 1 + +Choose the candidate that best matches the minimal structurally sufficient enabling-discovery annotation objective. + +Important: +- Judge candidates using the full critic policy from the system instructions. +- Do not reward verbosity by itself. +- Do not reward confidence by itself. +- Do not reward richer implementation stories, more citations, or more operational detail unless they are structurally necessary. +- Prefer candidates that recover the smallest structurally sufficient ingredient set for each downstream-used claim. +- Prefer candidates that minimize formatter guesswork. + +Apply these priorities in order: +1. Correct downstream-used claim selection +2. Correct claim splitting and abstraction level +3. Smallest structurally sufficient ingredient set +4. Correct exclusion of tempting non-ingredients +5. Correct canonical vs additional vs "__NONE__" grounding decisions +6. Correct role assignments +7. Evidence quality and clarity +8. Overall readiness for formatter conversion + +Specific judging reminders: +- Prefer downstream-first claim identification over abstract-first or citation-first reasoning. +- Penalize candidates that annotate unused paper outputs as claims. +- Penalize candidates that oversplit one artifact into multiple claims without downstream justification. +- Penalize candidates that include one-task helpers, local preprocessing, balancing tricks, token/language-tag details, helper models for one small step, benchmark-use ingredients, model-training conveniences, or other non-structural details. +- For benchmark/dataset claims, prefer benchmark framing, source-task/source-data substrate, benchmark-wide annotation/curation/verification protocol, benchmark-defining task formulation/protocol, and any core conceptual trick or idea. +- For method/resource claims, prefer core architecture, core conceptual trick or idea, core objective(s), structurally necessary training data, and essential initialization/checkpoints. +- For finding claims, prefer the model/artifact type under study, task/benchmark substrate, evaluation/measurement protocol, and any core conceptual trick or idea needed to make the finding observable. +- Prefer "__NONE__" when an ingredient is composite, same-paper dominant, benchmark-wide/corpus-wide, or would be distorted by forcing one canonical prior study. +- Penalize candidates that force canonical studies where "__NONE__" is more faithful. +- Penalize candidates that choose famous or cited papers rather than the most structurally faithful grounding. +- Prefer role assignments that are conservative and faithful to the ingredient's function. + +TARGET PAPER METADATA: +{ + "paperId": "3446200921606209cd3db3954ce913bf90070030", + "externalIds": { + "ArXiv": "2505.17978", + "DBLP": "journals/corr/abs-2505-17978", + "DOI": "10.48550/arXiv.2505.17978", + "CorpusId": 278886414 + }, + "url": "https://www.semanticscholar.org/paper/3446200921606209cd3db3954ce913bf90070030", + "title": "AVerImaTeC: A Dataset for Automatic Verification of Image-Text Claims with Evidence from the Web", + "venue": "arXiv.org", + "year": 2025, + "referenceCount": 67, + "citationCount": 16, + "openAccessPdf": { + "url": "", + "status": null, + "license": null, + "disclaimer": "Notice: Paper or abstract available at https://arxiv.org/abs/2505.17978, which is subject to the license by the author or copyright owner provided with this content. Please go to the source to verify the license and copyright information for your use." + }, + "tldr": { + "model": "tldr@v2.0.0", + "text": "This work introduces AVerImaTeC, a dataset consisting of 1,297 real-world image-text claims, and mitigate common challenges in fact-checking datasets such as contextual dependence, temporal leakage, and evidence insufficiency, via claim normalization, temporally constrained evidence annotation, and a two-stage sufficiency check." + }, + "publicationDate": "2025-05-23", + "authors": [ + { + "authorId": "2331613625", + "name": "Rui Cao" + }, + { + "authorId": "2348499942", + "name": "Zifeng Ding" + }, + { + "authorId": "2681038", + "name": "Zhijiang Guo" + }, + { + "authorId": "8804828", + "name": "M. Schlichtkrull" + }, + { + "authorId": "2273888020", + "name": "Andreas Vlachos" + } + ], + "abstract": "Textual claims are often accompanied by images to enhance their credibility and spread on social media, but this also raises concerns about the spread of misinformation. Existing datasets for automated verification of image-text claims remain limited, as they often consist of synthetic claims and lack evidence annotations to capture the reasoning behind the verdict. In this work, we introduce AVerImaTeC, a dataset consisting of 1,297 real-world image-text claims. Each claim is annotated with question-answer (QA) pairs containing evidence from the web, reflecting a decomposed reasoning regarding the verdict. We mitigate common challenges in fact-checking datasets such as contextual dependence, temporal leakage, and evidence insufficiency, via claim normalization, temporally constrained evidence annotation, and a two-stage sufficiency check. We assess the consistency of the annotation in AVerImaTeC via inter-annotator studies, achieving a $\\kappa=0.742$ on verdicts and $74.7\\%$ consistency on QA pairs. We also propose a novel evaluation method for evidence retrieval and conduct extensive experiments to establish baselines for verifying image-text claims using open-web evidence." +} + +ORIGINAL DISCOVERY CLAIM: + + +OPTIONAL DOWNSTREAM USAGE EVIDENCE: +[ + { + "cluster_id": "C1", + "representative_claim": "Extends target dataset for new dataset creation", + "cluster_title": "Extends target dataset for new dataset creation", + "count": "1", + "merge_rationale": "Directly builds upon the target dataset by extracting its pairs to construct a new dataset." + }, + { + "cluster_id": "C2", + "representative_claim": "Uses target dataset and knowledge stores for evaluation", + "cluster_title": "Uses target dataset and knowledge stores for evaluation", + "count": "2", + "merge_rationale": "Directly uses the target dataset and its provided knowledge stores to evaluate proposed claim verification methods." + }, + { + "cluster_id": "C3", + "representative_claim": "Extends reference-based evaluation method for a novel setting", + "cluster_title": "Extends reference-based evaluation method for a novel setting", + "count": "1", + "merge_rationale": "Adapts and extends the evaluation method introduced in the target paper to a new multimodal meme understanding setting." + }, + { + "cluster_id": "C4", + "representative_claim": "Adopts temporally constrained retrieval methodology for dataset creation", + "cluster_title": "Adopts temporally constrained retrieval methodology for dataset creation", + "count": "1", + "merge_rationale": "Adopts the target paper's specific methodology of restricting evidence retrieval by date to create a new dataset." + }, + { + "cluster_id": "C5", + "representative_claim": "Extends target baseline framework for multimodal claim verification", + "cluster_title": "Extends target baseline framework for multimodal claim verification", + "count": "1", + "merge_rationale": "Builds upon the baseline verification framework introduced by the target paper to develop a new system." + } +] + +CANDIDATE REASONING OUTPUTS: +CANDIDATE 1: candidate_1 +------------------------------------------------------------ +# 1. Cluster Evidence + +Based on the downstream cluster evidence, the target paper introduces four distinct downstream-used contributions: +- **AVerImaTeC Dataset:** Reused for new dataset creation (C1) and as an evaluation benchmark (C2). +- **Reference-based Evaluation Method:** Reused and extended for evaluating multimodal evidence in novel settings (C3). +- **Temporally Constrained Retrieval Methodology:** Reused as a protocol for dataset creation to prevent temporal leakage (C4). +- **Baseline Verification Framework:** Reused and extended for multimodal claim verification systems (C5). + +These contributions have distinct downstream roles and require different structural ingredients, so they should be split into four atomic claims. + +--- + +# 2. Claim Split Decision + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C1, C2 + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Method +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C3 + +### Claim C3 +- **claim_id:** C3 +- **artifact_type:** Protocol +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C4 + +### Claim C4 +- **claim_id:** C4 +- **artifact_type:** Method +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C5 + +### Why this split is correct +The paper bundles a dataset, an evaluation method, a data-collection protocol, and a baseline modeling framework. Downstream evidence shows that later works reuse these artifacts independently—some use the dataset for benchmarking, some adapt the evaluation method for new multimodal tasks, some adopt the temporal retrieval protocol for their own data collection, and others build on the baseline verification framework. They must be annotated separately to reflect their distinct downstream impact and structural dependencies. + +--- + +# 3. Rewritten Claims + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **rewritten_claim:** **Dataset: A real-world image-text claim verification dataset annotated with question-answer reasoning paths and web evidence, enabling the training and evaluation of multimodal fact-checking systems.** +- **why_this_is_atomic:** Isolates the dataset artifact from the evaluation methods and baseline models built around it. +- **decision:** YES_SUFFICIENT + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Method +- **rewritten_claim:** **Method: A reference-based evaluation method using large language models to separately assess the textual and visual components of retrieved evidence, enabling automated alignment with human judgments in multimodal fact-checking.** +- **why_this_is_atomic:** Isolates the specific evaluation metric/methodology from the dataset it was used to evaluate. +- **decision:** YES_SUFFICIENT + +### Claim C3 +- **claim_id:** C3 +- **artifact_type:** Protocol +- **rewritten_claim:** **Protocol: A temporally constrained evidence retrieval methodology that restricts search results to dates prior to the claim, enabling the construction of fact-checking datasets without temporal leakage.** +- **why_this_is_atomic:** Isolates the data-collection protocol from the final dataset, as downstream work reuses this specific methodology for their own dataset creation. +- **decision:** YES_SUFFICIENT + +### Claim C4 +- **claim_id:** C4 +- **artifact_type:** Method +- **rewritten_claim:** **Method: A multimodal claim verification framework combining dynamic question generation and tool-augmented answer generation, enabling automated fact-checking of image-text claims using open-web evidence.** +- **why_this_is_atomic:** Isolates the baseline modeling framework from the dataset and evaluation metrics. +- **decision:** YES_SUFFICIENT + +--- + +# 4. Minimal Ingredient Set Per Claim + +## Claim C1: AVerImaTeC Dataset + +### Ingredient C1.I1 +- **ingredient_id:** C1.I1 +- **ingredient:** Source fact-checking articles containing multimodal claims +- **why_structurally_necessary:** The dataset is built by extracting and normalizing real-world claims from existing fact-checking articles. Without this raw substrate, the dataset would not exist. +- **why_not_lower_level_substeps:** This should remain a composite data-source ingredient rather than being split into individual source datasets. +- **why_not_adjacent_implementation_details:** The exact filtering heuristics are implementation details; the structural dependency is the source text itself. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web; AMMeBa: A Large-Scale Survey and Dataset of Media-Based Misinformation In-The-Wild +- **role:** DATA_SOURCE +- **contribution:** Provides the raw real-world image-text claims that are normalized and annotated to form the dataset. +- **rationale:** The dataset is a composite of claims extracted from multiple prior fact-checking resources. No single prior study fully represents this substrate, making `NONE` the correct canonical grounding, with the primary sources listed as additional groundings. +- **evidence_span:** “we incorporated partially annotated articles from three sources: 1) filtered articles from AVeriTec... 2) articles verifying image-related claims from AMMEBA” + +### Ingredient C1.I2 +- **ingredient_id:** C1.I2 +- **ingredient:** Question-answering representation for verification rationale +- **why_structurally_necessary:** The defining feature of the dataset is that the reasoning process for verification is decomposed into QA pairs. Without this conceptual framing, it would just be a standard claim-verdict dataset. +- **why_not_lower_level_substeps:** This is a high-level conceptual framing and should not be decomposed into specific question types or answer formats. +- **why_not_adjacent_implementation_details:** The choice of QA over natural logic is the structural dependency, not the specific UI used to collect the QA pairs. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Complex Claim Verification with Evidence Retrieved in the Wild +- **additional groundings if any:** none +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the QA-based reasoning representation used to structure the evidence retrieval and verification process. +- **rationale:** The paper explicitly adopts QA as the reasoning representation for fact-checking, citing prior work that established this paradigm. Chen et al. (CLAIMDECOMP) is a clean representative for QA-based claim decomposition. +- **evidence_span:** “To capture the rationale in claim verification, a complex reasoning task, various reasoning representations have been explored... and QA” + +### Ingredient C1.I3 +- **ingredient_id:** C1.I3 +- **ingredient:** Four-way veracity labeling schema +- **why_structurally_necessary:** The dataset categorizes claims into specific veracity labels. This schema defines the classification task the dataset supports. +- **why_not_lower_level_substeps:** This is a single conceptual ingredient defining the label space. +- **why_not_adjacent_implementation_details:** The specific definitions of each label are part of this broader schema ingredient. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web +- **additional groundings if any:** none +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the specific four-way label space (supported, refuted, not enough evidence, conflicting) used to annotate the claims. +- **rationale:** The paper explicitly adopts the exact four-way veracity labeling schema introduced in AVeriTeC. +- **evidence_span:** “We follow the four-way veracity labeling schema from [Schlichtkrull et al., 2023]: supported, refuted, not enough evidence, and conflicting/cherry-picking.” + +### Ingredient C1.I4 +- **ingredient_id:** C1.I4 +- **ingredient:** Multi-stage annotation pipeline with evidence sufficiency checks +- **why_structurally_necessary:** The dataset's quality and reliability depend on a structured annotation workflow that includes independent sufficiency checks to resolve conflicts. +- **why_not_lower_level_substeps:** This should remain one protocol-level ingredient rather than being split into claim extraction, QA generation, and sufficiency checking separately. +- **why_not_adjacent_implementation_details:** The structural dependency is the pipeline itself, not the specific number of annotators or their demographics. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the foundational multi-stage annotation and verification workflow that the paper extends for image-text claims. +- **rationale:** The paper explicitly states that its annotation pipeline is an extension of the process proposed in AVeriTeC, making it the canonical grounding for this workflow. +- **evidence_span:** “extending the annotation process proposed in [Schlichtkrull et al., 2023] to the domain of image-text claims.” + +--- + +## Claim C2: Reference-based Evaluation Method + +### Ingredient C2.I1 +- **ingredient_id:** C2.I1 +- **ingredient:** LLM-based reference evaluation framework for evidence retrieval +- **why_structurally_necessary:** The method relies on using an LLM to compare retrieved evidence against human-annotated references, rather than relying on exact string matching. +- **why_not_lower_level_substeps:** This is the core conceptual basis of the evaluation method and should not be split into specific prompt designs. +- **why_not_adjacent_implementation_details:** The choice of Gemini as the specific LLM is an implementation detail; the structural dependency is the LLM-as-a-judge framework. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the foundation for using LLMs to evaluate evidence retrieval by comparing it to references. +- **rationale:** The paper explicitly adopts the Ev2R framework for its reference-based evaluation of textual evidence. +- **evidence_span:** “Recent research [Akhtar et al., 2024] showed that reference-based evaluation of evidence with large language models (LLMs) aligns best with human assessments.” + +### Ingredient C2.I2 +- **ingredient_id:** C2.I2 +- **ingredient:** Separated reference-based evaluation for textual and visual components +- **why_structurally_necessary:** The core novelty of the method is extending reference-based evaluation to multimodal evidence by evaluating text and images separately, which the paper shows is more robust than interleaved evaluation. +- **why_not_lower_level_substeps:** This should remain a single protocol ingredient rather than being split into the text-check step and the image-check step. +- **why_not_adjacent_implementation_details:** The specific similarity thresholds used are implementation details. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the novel mechanism for robustly evaluating multimodal evidence by decoupling the modalities during the LLM judgment phase. +- **rationale:** This separated evaluation strategy is the target paper's own novel extension to prior text-only reference evaluation methods, so it maps to `NONE`. +- **evidence_span:** “We then conduct separate reference-based evaluations for the textual and visual components.” + +--- + +## Claim C3: Temporally Constrained Retrieval Methodology + +### Ingredient C3.I1 +- **ingredient_id:** C3.I1 +- **ingredient:** Identification of temporal leakage as a critical flaw in fact-checking datasets +- **why_structurally_necessary:** The entire protocol is designed to solve the problem of temporal leakage (where models use future information to verify past claims). Without this conceptual framing, the protocol has no purpose. +- **why_not_lower_level_substeps:** This is a high-level conceptual ingredient. +- **why_not_adjacent_implementation_details:** N/A +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Missing Counter-Evidence Renders NLP Fact-Checking Unrealistic for Misinformation +- **additional groundings if any:** none +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the theoretical motivation for restricting evidence retrieval to prevent models from accessing future knowledge. +- **rationale:** The paper explicitly cites Glockner et al. as the motivation for preventing temporal leakage during evidence collection. +- **evidence_span:** “restricted to dates prior to the claim date to prevent temporal leakage [Glockner et al., 2022].” + +### Ingredient C3.I2 +- **ingredient_id:** C3.I2 +- **ingredient:** Date-restricted search interface and post-hoc filtering mechanism +- **why_structurally_necessary:** This is the operational mechanism of the protocol. Without restricting the search API and filtering results by date, the temporal constraint cannot be enforced. +- **why_not_lower_level_substeps:** This should remain one mechanism ingredient rather than being split into the Google Search API arguments and the Python htmldate package. +- **why_not_adjacent_implementation_details:** The specific Python packages used are implementation details; the structural dependency is the date-restriction mechanism itself. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the practical tooling to enforce the temporal constraints during dataset construction. +- **rationale:** The specific implementation of the custom search interface and post-hoc date filtering is developed in the target paper, mapping to `NONE`. +- **evidence_span:** “annotators are provided with a custom Google search interface... restricted to dates prior to the claim date” + +--- + +## Claim C4: Baseline Verification Framework + +### Ingredient C4.I1 +- **ingredient_id:** C4.I1 +- **ingredient:** Tool-augmented answer generation module with specialized search and VQA tools +- **why_structurally_necessary:** The baseline framework relies on an LLM selecting and using external tools (RIS, WST, WSI, VQA) to answer generated questions. Without this tool-use architecture, the framework could not retrieve open-web multimodal evidence. +- **why_not_lower_level_substeps:** This should remain one architectural ingredient rather than being split into each individual tool. +- **why_not_adjacent_implementation_details:** The specific prompts used for tool selection are implementation details. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models +- **additional groundings if any:** none +- **role:** CORE_METHOD +- **contribution:** Provides the paradigm of augmenting language models with specialized visual and search tools to solve complex multimodal tasks. +- **rationale:** The paper explicitly grounds its tool-augmented answer generation module in recent research on tool usage, with Visual ChatGPT (Wu et al.) being a canonical representative for integrating visual tools with LLMs. +- **evidence_span:** “we integrate a set of specialized tools into the answer generation module, along with a tool selector” + +### Ingredient C4.I2 +- **ingredient_id:** C4.I2 +- **ingredient:** Dynamic question generation strategy based on evolving evidence history +- **why_structurally_necessary:** The framework introduces dynamic and hybrid question generation to handle multi-hop reasoning where subsequent questions depend on previous answers. +- **why_not_lower_level_substeps:** This is a core methodological strategy and should not be decomposed into specific prompt templates. +- **why_not_adjacent_implementation_details:** The choice of specific MLLMs (like Qwen-VL) is an implementation detail; the structural dependency is the dynamic generation strategy. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** CORE_METHOD +- **contribution:** Provides the mechanism for generating sequential verification questions conditioned on the accumulated evidence context. +- **rationale:** The dynamic and hybrid question generation strategies are proposed by the target paper as novel components of their baseline framework, mapping to `NONE`. +- **evidence_span:** “we also propose a dynamic question generation (DQG) method, where each subsequent question is generated based on both the claim and the evolving evidence history.” + +--- + +# 5. Excluded Tempting Non-Ingredients + +- **Specific LLMs (Gemini, Qwen, Gemma, LLaVA)** + - Excluded because they are interchangeable implementation choices for the baseline framework and evaluation method, rather than structural prerequisites defining the artifacts themselves. +- **BM25 for text chunk retrieval** + - Excluded because it is a low-level implementation detail for the text search tool, not a high-level structural ingredient of the baseline framework. +- **Python htmldate package** + - Excluded because it is merely a local tool used to implement the date-restriction mechanism, which is already captured at a higher structural level. + +--- + +# 6. Candidate Grounding Decisions + +### Claim C1: AVerImaTeC Dataset +- **C1.I1:** NONE (accepted_none). The source data is composite. AVeriTeC and AMMeBa are accepted as additional groundings. +- **C1.I2:** Complex Claim Verification with Evidence Retrieved in the Wild (accepted_canonical). Clean representative for QA-based claim decomposition. +- **C1.I3:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web (accepted_canonical). Directly provides the four-way veracity schema. +- **C1.I4:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web (accepted_canonical). Directly provides the foundational annotation pipeline. + +### Claim C2: Reference-based Evaluation Method +- **C2.I1:** Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking (accepted_canonical). Directly provides the LLM-based reference evaluation framework. +- **C2.I2:** NONE (accepted_none). The separated multimodal evaluation strategy is introduced in the target paper. + +### Claim C3: Temporally Constrained Retrieval Methodology +- **C3.I1:** Missing Counter-Evidence Renders NLP Fact-Checking Unrealistic for Misinformation (accepted_canonical). Provides the conceptual motivation regarding temporal leakage. +- **C3.I2:** NONE (accepted_none). The specific date-restricted search implementation is from the target paper. + +### Claim C4: Baseline Verification Framework +- **C4.I1:** Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models (accepted_canonical). Representative for tool-augmented LLM architectures. +- **C4.I2:** NONE (accepted_none). The dynamic question generation strategy is proposed in the target paper. + +--- + +# 7. Final Grounding Decisions + +## Claim C1 +- **C1.I1** → NONE → DATA_SOURCE + - additional: AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web; AMMeBa: A Large-Scale Survey and Dataset of Media-Based Misinformation In-The-Wild +- **C1.I2** → Complex Claim Verification with Evidence Retrieved in the Wild → CONCEPTUAL_FRAMEWORK +- **C1.I3** → AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web → CONCEPTUAL_FRAMEWORK +- **C1.I4** → AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web → EVALUATION_PROTOCOL + +## Claim C2 +- **C2.I1** → Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking → EVALUATION_PROTOCOL +- **C2.I2** → NONE → EVALUATION_PROTOCOL + +## Claim C3 +- **C3.I1** → Missing Counter-Evidence Renders NLP Fact-Checking Unrealistic for Misinformation → CONCEPTUAL_FRAMEWORK +- **C3.I2** → NONE → IMPLEMENTATION_TOOLING + +## Claim C4 +- **C4.I1** → Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models → CORE_METHOD +- **C4.I2** → NONE → CORE_METHOD + +--- + +# 8. Final Rationales And Evidence + +## Claim C1: AVerImaTeC Dataset + +### C1.I1 +- **ingredient:** Source fact-checking articles containing multimodal claims +- **canonical study:** NONE +- **additional studies:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web; AMMeBa: A Large-Scale Survey and Dataset of Media-Based Misinformation In-The-Wild +- **role:** DATA_SOURCE +- **contribution:** Provides the raw real-world image-text claims that are normalized and annotated to form the dataset. +- **rationale:** The dataset is a composite of claims extracted from multiple prior fact-checking resources. No single prior study fully represents this substrate, making `NONE` the correct canonical grounding, with the primary sources listed as additional groundings. +- **evidence_span:** “we incorporated partially annotated articles from three sources: 1) filtered articles from AVeriTec... 2) articles verifying image-related claims from AMMEBA” + +### C1.I2 +- **ingredient:** Question-answering representation for verification rationale +- **canonical study:** Complex Claim Verification with Evidence Retrieved in the Wild +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the QA-based reasoning representation used to structure the evidence retrieval and verification process. +- **rationale:** The paper explicitly adopts QA as the reasoning representation for fact-checking, citing prior work that established this paradigm. Chen et al. (CLAIMDECOMP) is a clean representative for QA-based claim decomposition. +- **evidence_span:** “To capture the rationale in claim verification, a complex reasoning task, various reasoning representations have been explored... and QA” + +### C1.I3 +- **ingredient:** Four-way veracity labeling schema +- **canonical study:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the specific four-way label space (supported, refuted, not enough evidence, conflicting) used to annotate the claims. +- **rationale:** The paper explicitly adopts the exact four-way veracity labeling schema introduced in AVeriTeC. +- **evidence_span:** “We follow the four-way veracity labeling schema from [Schlichtkrull et al., 2023]: supported, refuted, not enough evidence, and conflicting/cherry-picking.” + +### C1.I4 +- **ingredient:** Multi-stage annotation pipeline with evidence sufficiency checks +- **canonical study:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the foundational multi-stage annotation and verification workflow that the paper extends for image-text claims. +- **rationale:** The paper explicitly states that its annotation pipeline is an extension of the process proposed in AVeriTeC, making it the canonical grounding for this workflow. +- **evidence_span:** “extending the annotation process proposed in [Schlichtkrull et al., 2023] to the domain of image-text claims.” + +## Claim C2: Reference-based Evaluation Method + +### C2.I1 +- **ingredient:** LLM-based reference evaluation framework for evidence retrieval +- **canonical study:** Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the foundation for using LLMs to evaluate evidence retrieval by comparing it to references. +- **rationale:** The paper explicitly adopts the Ev2R framework for its reference-based evaluation of textual evidence. +- **evidence_span:** “Recent research [Akhtar et al., 2024] showed that reference-based evaluation of evidence with large language models (LLMs) aligns best with human assessments.” + +### C2.I2 +- **ingredient:** Separated reference-based evaluation for textual and visual components +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the novel mechanism for robustly evaluating multimodal evidence by decoupling the modalities during the LLM judgment phase. +- **rationale:** This separated evaluation strategy is the target paper's own novel extension to prior text-only reference evaluation methods, so it maps to `NONE`. +- **evidence_span:** “We then conduct separate reference-based evaluations for the textual and visual components.” + +## Claim C3: Temporally Constrained Retrieval Methodology + +### C3.I1 +- **ingredient:** Identification of temporal leakage as a critical flaw in fact-checking datasets +- **canonical study:** Missing Counter-Evidence Renders NLP Fact-Checking Unrealistic for Misinformation +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the theoretical motivation for restricting evidence retrieval to prevent models from accessing future knowledge. +- **rationale:** The paper explicitly cites Glockner et al. as the motivation for preventing temporal leakage during evidence collection. +- **evidence_span:** “restricted to dates prior to the claim date to prevent temporal leakage [Glockner et al., 2022].” + +### C3.I2 +- **ingredient:** Date-restricted search interface and post-hoc filtering mechanism +- **canonical study:** NONE +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the practical tooling to enforce the temporal constraints during dataset construction. +- **rationale:** The specific implementation of the custom search interface and post-hoc date filtering is developed in the target paper, mapping to `NONE`. +- **evidence_span:** “annotators are provided with a custom Google search interface... restricted to dates prior to the claim date” + +## Claim C4: Baseline Verification Framework + +### C4.I1 +- **ingredient:** Tool-augmented answer generation module with specialized search and VQA tools +- **canonical study:** Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models +- **role:** CORE_METHOD +- **contribution:** Provides the paradigm of augmenting language models with specialized visual and search tools to solve complex multimodal tasks. +- **rationale:** The paper explicitly grounds its tool-augmented answer generation module in recent research on tool usage, with Visual ChatGPT (Wu et al.) being a canonical representative for integrating visual tools with LLMs. +- **evidence_span:** “we integrate a set of specialized tools into the answer generation module, along with a tool selector” + +### C4.I2 +- **ingredient:** Dynamic question generation strategy based on evolving evidence history +- **canonical study:** NONE +- **role:** CORE_METHOD +- **contribution:** Provides the mechanism for generating sequential verification questions conditioned on the accumulated evidence context. +- **rationale:** The dynamic and hybrid question generation strategies are proposed by the target paper as novel components of their baseline framework, mapping to `NONE`. +- **evidence_span:** “we also propose a dynamic question generation (DQG) method, where each subsequent question is generated based on both the claim and the evolving evidence history.” + + +CANDIDATE 2: candidate_2 +------------------------------------------------------------ +# 1. Cluster Evidence + +Based on the downstream cluster evidence, the target paper introduces four distinct downstream-used contributions: +- **C1 & C2** show downstream reuse of the **AVerImaTeC dataset** as a resource for training and evaluating multimodal fact-checking systems. +- **C3** shows downstream reuse of the **reference-based evaluation method** for assessing multimodal evidence retrieval. +- **C4** shows downstream reuse of the **temporally constrained retrieval methodology** for creating new datasets without temporal leakage. +- **C5** shows downstream reuse of the **baseline verification framework** for multimodal claim verification. + +Because these four artifacts have distinct downstream roles and require different structural ingredients, they should be split into four separate atomic claims. + +--- + +# 2. Claim Split Decision + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C1, C2 + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Method +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C3 + +### Claim C3 +- **claim_id:** C3 +- **artifact_type:** Method +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C4 + +### Claim C4 +- **claim_id:** C4 +- **artifact_type:** Method +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C5 + +### Why this split is correct +The paper bundles a dataset, an evaluation metric, a data collection protocol, and a baseline system. Downstream evidence confirms that later works reuse these components independently (e.g., one paper adopts the temporal constraint protocol to build its own dataset, while another extends the baseline framework for a new system). They must be annotated as separate claims to accurately reflect their distinct structural dependencies. + +--- + +# 3. Rewritten Claims + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **rewritten_claim:** **Dataset: A real-world image-text claim verification dataset annotated with question-answer pairs and web evidence, enabling the training and evaluation of multimodal fact-checking systems.** +- **why_this_is_atomic:** This claim isolates the dataset resource itself, separating it from the evaluation metrics and baseline models introduced alongside it. +- **optional cluster_id:** C1, C2 +- **decision:** YES_SUFFICIENT + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Method +- **rewritten_claim:** **Method: A reference-based evaluation method for multimodal evidence retrieval, enabling the automated assessment of generated text and retrieved images against human annotations.** +- **why_this_is_atomic:** This claim isolates the specific evaluation metric/methodology used to score evidence retrieval, which is reused independently of the dataset. +- **optional cluster_id:** C3 +- **decision:** YES_SUFFICIENT + +### Claim C3 +- **claim_id:** C3 +- **artifact_type:** Method +- **rewritten_claim:** **Method: A temporally constrained evidence retrieval methodology, enabling the creation of fact-checking datasets that prevent temporal leakage by restricting search dates.** +- **why_this_is_atomic:** This claim isolates the data collection protocol for preventing temporal leakage, which downstream work adopts for new dataset creation. +- **optional cluster_id:** C4 +- **decision:** YES_SUFFICIENT + +### Claim C4 +- **claim_id:** C4 +- **artifact_type:** Method +- **rewritten_claim:** **Method: A baseline verification framework that generates questions and uses specialized tools to retrieve multimodal evidence, enabling automated image-text claim verification.** +- **why_this_is_atomic:** This claim isolates the tool-augmented baseline system, which is extended by downstream work independently of the dataset. +- **optional cluster_id:** C5 +- **decision:** YES_SUFFICIENT + +--- + +# 4. Minimal Ingredient Set Per Claim + +## Claim C1: AVerImaTeC Dataset + +### Ingredient C1.I1 +- **ingredient_id:** C1.I1 +- **ingredient:** Source fact-checking articles containing real-world multimodal claims +- **why_structurally_necessary:** The dataset is built by extracting and normalizing real-world image-text claims. Without a substrate of existing fact-checking articles containing multimodal claims, the dataset would not exist in its claimed form. +- **why_not_lower_level_substeps:** This should remain a composite data-source ingredient rather than being split into separate ingredients for each individual source platform. +- **why_not_adjacent_implementation_details:** The specific filtering scripts used to extract the claims are implementation details; the structural dependency is the raw source data itself. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web; AMMeBa: A Large-Scale Survey and Dataset of Media-Based Misinformation In-The-Wild +- **role:** DATA_SOURCE +- **contribution:** Provides the raw real-world image-text claims and fact-checking articles that are annotated to create the dataset. +- **rationale:** The dataset is explicitly constructed by aggregating and filtering partially annotated articles from multiple prior sources (AVeriTeC, AMMeBA, and ClaimReview). Because this source substrate is composite, `NONE` is the correct canonical grounding, with the primary contributing datasets listed as additional groundings. +- **evidence_span:** “we incorporated partially annotated articles from three sources: 1) filtered articles from AVeriTec... 2) articles verifying image-related claims from AMMEBA” + +### Ingredient C1.I2 +- **ingredient_id:** C1.I2 +- **ingredient:** Question-answering annotation pipeline for structuring fact-checking rationales +- **why_structurally_necessary:** The defining property of the dataset is that verification rationales are decomposed into QA pairs with web evidence. Without this specific annotation framing, it would just be a standard claim-verdict dataset. +- **why_not_lower_level_substeps:** This should remain one protocol-level ingredient rather than being split into question generation, evidence retrieval, and sufficiency checking as separate ingredients. +- **why_not_adjacent_implementation_details:** The exact web interface used by annotators is an implementation detail; the structural dependency is the QA-based annotation pipeline. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the multi-phase annotation pipeline that decomposes claim verification into QA pairs and justifications. +- **rationale:** The paper explicitly states that its annotation pipeline extends the process proposed in AVeriTeC to the domain of image-text claims. AVeriTeC is the cleanest canonical grounding for this QA-based fact-checking annotation protocol. +- **evidence_span:** “extending the annotation process proposed in [Schlichtkrull et al., 2023] to the domain of image-text claims.” + +## Claim C2: Reference-based evaluation method + +### Ingredient C2.I1 +- **ingredient_id:** C2.I1 +- **ingredient:** LLM-based reference evaluation framework for evidence retrieval +- **why_structurally_necessary:** The evaluation method operates by converting QA pairs into evidence statements and using an LLM to compare predictions against human references. Without this conceptual framework, the metric would not function. +- **why_not_lower_level_substeps:** This should remain one framework-level ingredient rather than being split into text-comparison and image-comparison substeps. +- **why_not_adjacent_implementation_details:** The specific prompt templates used for the LLM evaluator are implementation details. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking +- **additional groundings if any:** none +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the foundational reference-based evaluation framework that the paper adapts for multimodal evidence. +- **rationale:** The paper explicitly motivates and adopts its reference-based evaluation method from Ev2R (Akhtar et al., 2024), making it the direct canonical grounding for this methodology. +- **evidence_span:** “Motivated by a recent study [Akhtar et al., 2024], we adopt a reference-based evaluation method that compares model responses to human-annotated ground-truth data.” + +## Claim C3: Temporally constrained retrieval methodology + +### Ingredient C3.I1 +- **ingredient_id:** C3.I1 +- **ingredient:** Temporal leakage prevention constraint for evidence retrieval +- **why_structurally_necessary:** The methodology is defined by restricting search results to dates prior to the claim to ensure models cannot access future information. Without this constraint, the methodology loses its defining purpose. +- **why_not_lower_level_substeps:** This should remain a single conceptual constraint rather than being split into the specific Python packages or APIs used to enforce it. +- **why_not_adjacent_implementation_details:** The use of `htmldate.find_date` is a local implementation choice; the structural dependency is the temporal constraint framing itself. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Missing Counter-Evidence Renders NLP Fact-Checking Unrealistic for Misinformation +- **additional groundings if any:** none +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the conceptual framing and necessity of restricting evidence retrieval by date to prevent temporal leakage in fact-checking. +- **rationale:** The paper explicitly grounds its temporal restriction methodology in the temporal leakage problem identified by Glockner et al., 2022, making it the canonical source for this constraint. +- **evidence_span:** “All retrieved pages from the interface are restricted to dates prior to the claim date to prevent temporal leakage [Glockner et al., 2022].” + +## Claim C4: Baseline verification framework + +### Ingredient C4.I1 +- **ingredient_id:** C4.I1 +- **ingredient:** Tool-augmented architecture for multimodal evidence retrieval and answering +- **why_structurally_necessary:** The baseline framework relies on integrating specialized tools (reverse image search, web search, VQA) to answer generated questions. Without this tool-use architecture, the system could not retrieve external multimodal evidence. +- **why_not_lower_level_substeps:** This should remain one architecture-level ingredient rather than being split into individual tools (RIS, WST, WSI) as separate ingredients. +- **why_not_adjacent_implementation_details:** The specific tool selection prompts are implementation details; the structural dependency is the tool-augmented generation paradigm. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** DEFAME: Dynamic Evidence-based FAct-checking with Multimodal Experts +- **additional groundings if any:** Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models +- **role:** CORE_METHOD +- **contribution:** Provides the paradigm of integrating specialized external tools into an LLM's answer generation module. +- **rationale:** The paper explicitly builds its tool-augmented answer generator based on recent research on tool usage. DEFAME (Braun et al., 2024) is the most direct canonical grounding as it applies this specifically to fact-checking, while Visual ChatGPT serves as an additional grounding for the broader multimodal tool-use concept. +- **evidence_span:** “Inspired by recent research on tool usage [Wu et al., 2023, Cao et al., 2024, Braun et al., 2024], we integrate a set of specialized tools” + +### Ingredient C4.I2 +- **ingredient_id:** C4.I2 +- **ingredient:** Pretrained multimodal large language model for visual reasoning and verification +- **why_structurally_necessary:** The framework requires an MLLM to process interleaved image-text claims, perform visual question answering, and predict the final verdict. Without a pretrained MLLM, the multimodal verification system would not function. +- **why_not_lower_level_substeps:** This should remain at the level of the MLLM initialization rather than being split into vision encoders and text decoders. +- **why_not_adjacent_implementation_details:** The specific choice between Qwen-VL, Gemini, or LLaVA is an implementation detail, but the structural requirement is the presence of a capable pretrained MLLM. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Qwen2.5-VL Technical Report +- **additional groundings if any:** none +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the foundational multimodal reasoning capabilities used by the baseline framework to process images and text jointly. +- **rationale:** The baseline framework operationally requires an MLLM. The paper tests several, but Qwen2.5-VL is explicitly highlighted as a primary open-source MLLM used in the strongest baseline configurations, making it a clean canonical grounding for this capability. +- **evidence_span:** “Qwen2.5-7B-Instruct [Yang et al., 2024] ( Qwen ) acts as the LLM and Qwen2.5-VL-7B-Instruct [Bai et al., 2025] ( Qwen-VL ) serves as the MLLM” + +--- + +# 5. Excluded Tempting Non-Ingredients + +### Claim C1 exclusions +- **Inter-annotator agreement metrics (e.g., Randolph's Kappa)** + - Excluded because these are evaluation metrics used to validate the dataset's quality, not structural ingredients required to construct the dataset itself. + +### Claim C4 exclusions +- **BM25 ranking for textual evidence retrieval** + - Excluded because it is a local implementation detail for filtering text search results, not a core structural component of the overall baseline framework. +- **Specific question generation strategies (PQG, DQG, HQG)** + - Excluded because these are algorithmic variants explored in the paper's experiments, whereas the structural dependency is the broader tool-augmented MLLM framework itself. + +--- + +# 6. Candidate Grounding Decisions + +## Claim C1: AVerImaTeC Dataset +### Ingredient C1.I1 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The source data is a composite of multiple prior fact-checking datasets and platforms. +- **Candidate:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web +- **decision:** accepted_additional +- **why:** Provides a significant portion of the source articles. +- **Candidate:** AMMeBa: A Large-Scale Survey and Dataset of Media-Based Misinformation In-The-Wild +- **decision:** accepted_additional +- **why:** Provides source articles specifically verifying image-related claims. + +### Ingredient C1.I2 +- **Candidate:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web +- **decision:** accepted_canonical +- **why:** The paper explicitly extends the QA-based annotation pipeline introduced in AVeriTeC. + +## Claim C2: Reference-based evaluation method +### Ingredient C2.I1 +- **Candidate:** Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking +- **decision:** accepted_canonical +- **why:** The paper explicitly adopts and adapts the reference-based evaluation method from Ev2R. + +## Claim C3: Temporally constrained retrieval methodology +### Ingredient C3.I1 +- **Candidate:** Missing Counter-Evidence Renders NLP Fact-Checking Unrealistic for Misinformation +- **decision:** accepted_canonical +- **why:** The paper explicitly grounds its temporal leakage prevention constraint in this prior study. + +## Claim C4: Baseline verification framework +### Ingredient C4.I1 +- **Candidate:** DEFAME: Dynamic Evidence-based FAct-checking with Multimodal Experts +- **decision:** accepted_canonical +- **why:** Cleanest representative for applying tool-augmented LLM architectures specifically to fact-checking. +- **Candidate:** Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models +- **decision:** accepted_additional +- **why:** Foundational prior work on integrating specialized visual tools with LLMs. + +### Ingredient C4.I2 +- **Candidate:** Qwen2.5-VL Technical Report +- **decision:** accepted_canonical +- **why:** Cleanest representative of the pretrained MLLM capability operationally required by the baseline framework. + +--- + +# 7. Final Grounding Decisions + +## Claim C1 +- **C1.I1** → NONE → DATA_SOURCE + - additional: AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web; AMMeBa: A Large-Scale Survey and Dataset of Media-Based Misinformation In-The-Wild +- **C1.I2** → AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web → EVALUATION_PROTOCOL + +## Claim C2 +- **C2.I1** → Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking → CONCEPTUAL_FRAMEWORK + +## Claim C3 +- **C3.I1** → Missing Counter-Evidence Renders NLP Fact-Checking Unrealistic for Misinformation → CONCEPTUAL_FRAMEWORK + +## Claim C4 +- **C4.I1** → DEFAME: Dynamic Evidence-based FAct-checking with Multimodal Experts → CORE_METHOD + - additional: Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models +- **C4.I2** → Qwen2.5-VL Technical Report → IMPLEMENTATION_TOOLING + +--- + +# 8. Final Rationales And Evidence + +## Claim C1: AVerImaTeC Dataset + +### C1.I1 +- **ingredient:** Source fact-checking articles containing real-world multimodal claims +- **canonical study:** NONE +- **additional studies:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web; AMMeBa: A Large-Scale Survey and Dataset of Media-Based Misinformation In-The-Wild +- **role:** DATA_SOURCE +- **contribution:** Provides the raw real-world image-text claims and fact-checking articles that are annotated to create the dataset. +- **rationale:** The dataset is explicitly constructed by aggregating and filtering partially annotated articles from multiple prior sources (AVeriTeC, AMMeBA, and ClaimReview). Because this source substrate is composite, `NONE` is the correct canonical grounding, with the primary contributing datasets listed as additional groundings. +- **evidence_span:** “we incorporated partially annotated articles from three sources: 1) filtered articles from AVeriTec... 2) articles verifying image-related claims from AMMEBA” + +### C1.I2 +- **ingredient:** Question-answering annotation pipeline for structuring fact-checking rationales +- **canonical study:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the multi-phase annotation pipeline that decomposes claim verification into QA pairs and justifications. +- **rationale:** The paper explicitly states that its annotation pipeline extends the process proposed in AVeriTeC to the domain of image-text claims. AVeriTeC is the cleanest canonical grounding for this QA-based fact-checking annotation protocol. +- **evidence_span:** “extending the annotation process proposed in [Schlichtkrull et al., 2023] to the domain of image-text claims.” + +## Claim C2: Reference-based evaluation method + +### C2.I1 +- **ingredient:** LLM-based reference evaluation framework for evidence retrieval +- **canonical study:** Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the foundational reference-based evaluation framework that the paper adapts for multimodal evidence. +- **rationale:** The paper explicitly motivates and adopts its reference-based evaluation method from Ev2R (Akhtar et al., 2024), making it the direct canonical grounding for this methodology. +- **evidence_span:** “Motivated by a recent study [Akhtar et al., 2024], we adopt a reference-based evaluation method that compares model responses to human-annotated ground-truth data.” + +## Claim C3: Temporally constrained retrieval methodology + +### C3.I1 +- **ingredient:** Temporal leakage prevention constraint for evidence retrieval +- **canonical study:** Missing Counter-Evidence Renders NLP Fact-Checking Unrealistic for Misinformation +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the conceptual framing and necessity of restricting evidence retrieval by date to prevent temporal leakage in fact-checking. +- **rationale:** The paper explicitly grounds its temporal restriction methodology in the temporal leakage problem identified by Glockner et al., 2022, making it the canonical source for this constraint. +- **evidence_span:** “All retrieved pages from the interface are restricted to dates prior to the claim date to prevent temporal leakage [Glockner et al., 2022].” + +## Claim C4: Baseline verification framework + +### C4.I1 +- **ingredient:** Tool-augmented architecture for multimodal evidence retrieval and answering +- **canonical study:** DEFAME: Dynamic Evidence-based FAct-checking with Multimodal Experts +- **additional studies:** Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models +- **role:** CORE_METHOD +- **contribution:** Provides the paradigm of integrating specialized external tools into an LLM's answer generation module. +- **rationale:** The paper explicitly builds its tool-augmented answer generator based on recent research on tool usage. DEFAME (Braun et al., 2024) is the most direct canonical grounding as it applies this specifically to fact-checking, while Visual ChatGPT serves as an additional grounding for the broader multimodal tool-use concept. +- **evidence_span:** “Inspired by recent research on tool usage [Wu et al., 2023, Cao et al., 2024, Braun et al., 2024], we integrate a set of specialized tools” + +### C4.I2 +- **ingredient:** Pretrained multimodal large language model for visual reasoning and verification +- **canonical study:** Qwen2.5-VL Technical Report +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the foundational multimodal reasoning capabilities used by the baseline framework to process images and text jointly. +- **rationale:** The baseline framework operationally requires an MLLM. The paper tests several, but Qwen2.5-VL is explicitly highlighted as a primary open-source MLLM used in the strongest baseline configurations, making it a clean canonical grounding for this capability. +- **evidence_span:** “Qwen2.5-7B-Instruct [Yang et al., 2024] ( Qwen ) acts as the LLM and Qwen2.5-VL-7B-Instruct [Bai et al., 2025] ( Qwen-VL ) serves as the MLLM” + + +CANDIDATE 3: candidate_3 +------------------------------------------------------------ +# 1. Cluster Evidence + +Based on the downstream cluster evidence, the target paper introduces four distinct downstream-used contributions: +- **A dataset** for multimodal claim verification (reused for evaluation and extended for new dataset creation in C1, C2). +- **An evaluation method** for multimodal evidence retrieval (adapted for a novel setting in C3). +- **A temporally constrained retrieval methodology** (adopted for dataset creation in C4). +- **A baseline verification framework** (extended for multimodal claim verification in C5). + +Because these four artifacts have distinct downstream roles, distinct usage clusters, and require different ingredient decompositions, the bundled paper contribution must be split into four atomic claims. + +--- + +# 2. Claim Split Decision + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C1, C2 + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Metric +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C3 + +### Claim C3 +- **claim_id:** C3 +- **artifact_type:** Method +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C4 + +### Claim C4 +- **claim_id:** C4 +- **artifact_type:** Method +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C5 + +### Why this split is correct +The paper contributes four separate artifacts with distinct downstream roles: +- The **AVerImaTeC dataset** is reused as a training and evaluation resource. +- The **reference-based evaluation method** is reused as a standalone metric for assessing multimodal evidence retrieval. +- The **temporally constrained retrieval methodology** is reused as a data-collection protocol to prevent temporal leakage. +- The **baseline framework** is reused as a foundation for building new multimodal claim verification systems. +They should therefore be annotated separately. + +--- + +# 3. Rewritten Claims + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **rewritten_claim:** **Dataset: A real-world image-text claim verification dataset annotated with question-answer reasoning paths and web-based evidence, enabling the training and evaluation of multimodal fact-checking systems.** +- **why_this_is_atomic:** This claim isolates the dataset resource itself, separate from the evaluation metrics, data collection methodologies, and baseline models introduced alongside it. +- **decision:** YES_SUFFICIENT + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Metric +- **rewritten_claim:** **Metric: A reference-based evaluation method for multimodal evidence retrieval that separately assesses textual and visual components using LLMs, enabling automated assessment of retrieved evidence quality.** +- **why_this_is_atomic:** This claim isolates the specific evaluation protocol used to score multimodal evidence, which downstream work reuses independently of the dataset. +- **decision:** YES_SUFFICIENT + +### Claim C3 +- **claim_id:** C3 +- **artifact_type:** Method +- **rewritten_claim:** **Method: A temporally constrained evidence retrieval methodology that restricts search results to dates prior to the claim, enabling the construction of fact-checking datasets without temporal leakage.** +- **why_this_is_atomic:** This claim isolates the specific data-collection protocol designed to prevent temporal leakage, which downstream work adopts for its own dataset creation. +- **decision:** YES_SUFFICIENT + +### Claim C4 +- **claim_id:** C4 +- **artifact_type:** Method +- **rewritten_claim:** **Method: A baseline framework for image-text claim verification that decomposes reasoning into question generation and tool-augmented answer generation, enabling automated multimodal fact-checking.** +- **why_this_is_atomic:** This claim isolates the baseline verification architecture, which downstream work extends to build new systems. +- **decision:** YES_SUFFICIENT + +--- + +# 4. Minimal Ingredient Set Per Claim + +## Claim C1: Dataset + +### Ingredient C1.I1 +- **ingredient_id:** C1.I1 +- **ingredient:** Source fact-checking articles containing real-world multimodal claims +- **why_structurally_necessary:** The dataset is built by extracting and normalizing real-world image-text claims from existing fact-checking articles. Without this source substrate, the dataset would not exist as a collection of real-world claims. +- **why_not_lower_level_substeps:** This should remain a composite data-source ingredient rather than being split into individual upstream datasets (AVeriTeC, AMMeBA, ClaimReview). +- **why_not_adjacent_implementation_details:** The structural dependency is the raw multimodal claim substrate, not the specific filtering scripts used to extract them. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web; AMMeBa: A Large-Scale Survey and Dataset of Media-Based Misinformation In-The-Wild +- **role:** DATA_SOURCE +- **contribution:** Supplies the raw fact-checking articles and multimodal claims that are normalized and annotated to form the dataset. +- **rationale:** The dataset's source material is inherently composite, drawing from multiple prior fact-checking repositories and APIs. No single prior study represents the entire source mixture, making `NONE` the correct canonical grounding, with the primary contributing datasets listed as additional studies. +- **evidence_span:** “we incorporated partially annotated articles from three sources: 1) filtered articles from AVeriTec... 2) articles verifying image-related claims from AMMEBA... and 3) all true claims from ClaimReview” + +### Ingredient C1.I2 +- **ingredient_id:** C1.I2 +- **ingredient:** Question-answering representation for fact-checking rationales +- **why_structurally_necessary:** The defining property of the dataset's annotations is that the verification rationale is decomposed into a sequence of QA pairs. Without this conceptual framing, the dataset would lack its structured reasoning paths. +- **why_not_lower_level_substeps:** This should remain a conceptual framing ingredient rather than being split into specific question types or answer formats. +- **why_not_adjacent_implementation_details:** The structural dependency is the QA-based reasoning representation itself, not the specific UI used by annotators to write the questions. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Complex Claim Verification with Evidence Retrieved in the Wild +- **additional groundings if any:** The Art of SOCRATIC QUESTIONING: Recursive Thinking with Large Language Models; Fact-Checking Complex Claims with Program-Guided Reasoning +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the conceptual framework of using question-answering pairs to represent the decomposed reasoning steps required for claim verification. +- **rationale:** The paper explicitly adopts QA as the reasoning representation for fact-checking, citing prior works that established this paradigm. Chen et al. (CLAIMDECOMP) is a clean representative study for this QA-based verification framing. +- **evidence_span:** “To capture the rationale in claim verification, a complex reasoning task, various reasoning representations have been explored, including... QA [Chen et al., 2024, Qi et al., 2023, Pan et al., 2023a].” + +### Ingredient C1.I3 +- **ingredient_id:** C1.I3 +- **ingredient:** Multi-stage human annotation protocol for evidence retrieval and sufficiency checking +- **why_structurally_necessary:** The dataset's quality and reliability depend on a rigorous annotation pipeline that includes claim normalization, QA generation, and two rounds of evidence sufficiency checks. Without this protocol, the dataset would not exist in its validated form. +- **why_not_lower_level_substeps:** This should remain one protocol-level ingredient rather than being split into individual annotation phases. +- **why_not_adjacent_implementation_details:** The structural dependency is the benchmark-wide human-supervision workflow, not the specific annotator compensation details. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the manual dataset-construction process ensuring that all claims are normalized, annotated with evidence, and verified for sufficiency. +- **rationale:** Constructing this dataset requires a complex, multi-stage human annotation workflow. Because this specific five-phase pipeline is realized in the target paper (though extending prior work), `NONE` is the correct grounding. +- **evidence_span:** “The five-phase annotation pipeline is illustrated in Figure 2, extending the annotation process proposed in [Schlichtkrull et al., 2023] to the domain of image-text claims.” + +## Claim C2: Metric + +### Ingredient C2.I1 +- **ingredient_id:** C2.I1 +- **ingredient:** Reference-based evaluation framework for textual evidence using LLMs +- **why_structurally_necessary:** The proposed metric is an extension of an existing LLM-based reference evaluation framework. Without this base framework, the multimodal metric would not exist. +- **why_not_lower_level_substeps:** This should remain a framework-level ingredient rather than being split into specific prompt templates or scoring thresholds. +- **why_not_adjacent_implementation_details:** The structural dependency is the reference-based evaluation paradigm itself, not the specific adversarial robustness checks performed to validate it. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking +- **additional groundings if any:** none +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the base reference-based evaluation methodology that the paper extends to handle multimodal evidence. +- **rationale:** The paper explicitly builds its evaluation method upon the Ev2R framework, which demonstrated that LLM-based reference evaluation aligns well with human assessments for fact-checking evidence. +- **evidence_span:** “Recent research [Akhtar et al., 2024] showed that reference-based evaluation of evidence with large language models (LLMs) aligns best with human assessments. We extend this framework to a multimodal setting” + +### Ingredient C2.I2 +- **ingredient_id:** C2.I2 +- **ingredient:** Large multimodal model for scoring textual and visual alignments +- **why_structurally_necessary:** To evaluate multimodal evidence, the metric requires a model capable of processing both text and images to compare retrieved evidence against ground-truth references. +- **why_not_lower_level_substeps:** This should remain at the level of the MLLM scorer rather than being split into separate text and vision encoders. +- **why_not_adjacent_implementation_details:** The specific choice of Gemini 2.0 Flash is an implementation detail; the structural dependency is the use of a capable MLLM as the scoring engine. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Acts as the automated evaluator that scores the alignment between retrieved multimodal evidence and human-annotated references. +- **rationale:** The metric operationally depends on an MLLM to perform the scoring. The paper uses Gemini 2.0 Flash, which is a commercial API/model rather than a canonical prior academic study defining the ingredient, making `NONE` the most appropriate grounding. +- **evidence_span:** “We exploit Gemini-2.0-Flash [DeepMind, 2024] as the scoring model for both steps in the reference-based evaluation” + +## Claim C3: Method (Retrieval Methodology) + +### Ingredient C3.I1 +- **ingredient_id:** C3.I1 +- **ingredient:** Conceptual framing of temporal leakage in fact-checking datasets +- **why_structurally_necessary:** The entire purpose of the temporally constrained retrieval methodology is to solve the problem of temporal leakage (where models use future information to verify past claims). Without this conceptual framing, the methodology has no purpose. +- **why_not_lower_level_substeps:** This is a high-level conceptual ingredient and cannot be decomposed further. +- **why_not_adjacent_implementation_details:** N/A +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Missing Counter-Evidence Renders NLP Fact-Checking Unrealistic for Misinformation +- **additional groundings if any:** none +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Identifies the problem of temporal leakage in fact-checking, motivating the need for date-restricted evidence retrieval. +- **rationale:** The paper explicitly cites Glockner et al. (2022) as the motivation for restricting search results to prevent temporal leakage, making it the clean canonical grounding for this concept. +- **evidence_span:** “All retrieved pages from the interface are restricted to dates prior to the claim date to prevent temporal leakage [Glockner et al., 2022].” + +### Ingredient C3.I2 +- **ingredient_id:** C3.I2 +- **ingredient:** Date-restricted web search and scraping tooling +- **why_structurally_necessary:** To actually implement the temporally constrained methodology, the system requires tools capable of filtering web search results and scraped pages by publication date. +- **why_not_lower_level_substeps:** This should remain a single tooling ingredient rather than being split into the Google Search API and the htmldate Python package separately. +- **why_not_adjacent_implementation_details:** The specific Python package used is an implementation detail; the structural dependency is the capability to filter evidence by date. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the operational mechanism to enforce the temporal constraints during evidence retrieval. +- **rationale:** The methodology relies on standard APIs and Python packages (like htmldate) to enforce date restrictions. These are general software tools rather than canonical prior research studies, so `NONE` is the correct grounding. +- **evidence_span:** “we use a post-hoc method by leveraging the Python package htmldate.fine_date to filter out pages published before claim dates.” + +## Claim C4: Method (Baseline Framework) + +### Ingredient C4.I1 +- **ingredient_id:** C4.I1 +- **ingredient:** Question-guided multi-hop reasoning framework for fact-checking +- **why_structurally_necessary:** The baseline framework operates by generating questions to guide the evidence retrieval process. Without this question-guided architecture, the baseline would not function in its claimed form. +- **why_not_lower_level_substeps:** This should remain a framework-level ingredient rather than being split into parallel, dynamic, and hybrid generation strategies. +- **why_not_adjacent_implementation_details:** The specific prompts used for question generation are implementation details; the structural dependency is the question-guided reasoning framework itself. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web +- **additional groundings if any:** none +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the foundational architecture of using generated questions to guide the fact-checking process. +- **rationale:** The paper explicitly builds its question generation approach upon the framework established by AVeriTeC, making it the cleanest canonical grounding. +- **evidence_span:** “A straightforward approach is to generate all verification questions at once, given an image-text claim, as in prior work [Schlichtkrull et al., 2023].” + +### Ingredient C4.I2 +- **ingredient_id:** C4.I2 +- **ingredient:** Tool-augmented LLM framework for answering questions +- **why_structurally_necessary:** The baseline answers generated questions by routing them to specialized tools (RIS, WST, WSI, VQA). Without this tool-augmented architecture, the system could not retrieve external evidence to answer the questions. +- **why_not_lower_level_substeps:** This should remain a single framework ingredient rather than being split into individual tools (RIS, WST, etc.) as separate ingredients. +- **why_not_adjacent_implementation_details:** The specific tool selection prompts are implementation details; the structural dependency is the tool-augmented generation paradigm. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models +- **additional groundings if any:** DEFAME: Dynamic Evidence-based FAct-checking with Multimodal Experts +- **role:** CORE_METHOD +- **contribution:** Provides the architecture for integrating specialized external tools into the LLM's answer generation process. +- **rationale:** The paper explicitly adopts the tool-augmented LLM paradigm to handle multimodal evidence retrieval, citing Visual ChatGPT and DEFAME as inspirations. Visual ChatGPT is a clean canonical representative for this broader paradigm. +- **evidence_span:** “Inspired by recent research on tool usage [Wu et al., 2023, Cao et al., 2024, Braun et al., 2024], we integrate a set of specialized tools into the answer generation module” + +### Ingredient C4.I3 +- **ingredient_id:** C4.I3 +- **ingredient:** Multimodal large language models for question generation and visual question answering +- **why_structurally_necessary:** The baseline framework requires models capable of processing both the textual claim and the associated images to generate relevant questions and perform VQA. +- **why_not_lower_level_substeps:** This should remain at the level of MLLMs as a class rather than being split into specific model variants. +- **why_not_adjacent_implementation_details:** The specific choice between Qwen-VL, Gemini, or LLaVA is an implementation detail; the structural dependency is the use of MLLMs to power the framework's core components. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** CORE_METHOD +- **contribution:** Powers the internal reasoning, question generation, and visual analysis capabilities of the baseline framework. +- **rationale:** The framework is designed to be model-agnostic and tests several different MLLMs (Gemini, Qwen-VL, Gemma, LLaVA). Because the structural dependency is on the class of MLLMs rather than one specific canonical model, `NONE` is the correct grounding. +- **evidence_span:** “All three strategies employ an MLLM for question generation, leveraging the model's internal decomposition capability.” + +--- + +# 5. Excluded Tempting Non-Ingredients + +- **Specific LLM/MLLM variants (e.g., Gemini, Qwen, LLaVA)** + - Excluded because the structural dependency for both the metric and the baseline framework is the class of Multimodal Large Language Models, not any single specific instance tested in the paper. +- **BM25 ranking** + - Excluded because it is a local implementation detail used for text retrieval within the baseline framework, not a high-level structural ingredient defining the artifact. +- **Specific fact-checking strategies (e.g., reverse image search, geolocation)** + - Excluded because these are either metadata labels within the dataset or specific tools within the broader tool-augmented framework, rather than standalone structural ingredients. + +--- + +# 6. Candidate Grounding Decisions + +### Claim C1: Dataset +- **C1.I1 (Source articles):** `NONE` is chosen because the source material is a composite of multiple prior datasets (AVeriTeC, AMMeBA, ClaimReview). +- **C1.I2 (QA representation):** *Complex Claim Verification with Evidence Retrieved in the Wild* is chosen as the canonical grounding because it is explicitly cited as a foundational work for using QA representations in fact-checking. +- **C1.I3 (Annotation protocol):** `NONE` is chosen because the specific five-phase annotation pipeline is constructed and executed within the target paper itself. + +### Claim C2: Metric +- **C2.I1 (Ev2R framework):** *Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking* is chosen as the canonical grounding because the paper explicitly extends this specific framework for its multimodal metric. +- **C2.I2 (MLLM scorer):** `NONE` is chosen because the scoring relies on a commercial API (Gemini 2.0 Flash) rather than a canonical prior academic study defining the ingredient. + +### Claim C3: Method (Retrieval Methodology) +- **C3.I1 (Temporal leakage concept):** *Missing Counter-Evidence Renders NLP Fact-Checking Unrealistic for Misinformation* is chosen as the canonical grounding because it is explicitly cited as the motivation for the methodology. +- **C3.I2 (Date tooling):** `NONE` is chosen because the implementation relies on standard software tools and APIs rather than prior research studies. + +### Claim C4: Method (Baseline Framework) +- **C4.I1 (Question-guided framework):** *AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web* is chosen as the canonical grounding because the paper explicitly builds its question generation approach upon it. +- **C4.I2 (Tool-augmented framework):** *Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models* is chosen as the canonical representative for the tool-augmented LLM paradigm adopted by the baseline. +- **C4.I3 (MLLMs):** `NONE` is chosen because the framework is model-agnostic and relies on the general capability of MLLMs rather than one specific prior model. + +--- + +# 7. Final Grounding Decisions + +## Claim C1 +- **C1.I1** → NONE → DATA_SOURCE + - additional: AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web; AMMeBa: A Large-Scale Survey and Dataset of Media-Based Misinformation In-The-Wild +- **C1.I2** → Complex Claim Verification with Evidence Retrieved in the Wild → CONCEPTUAL_FRAMEWORK + - additional: The Art of SOCRATIC QUESTIONING: Recursive Thinking with Large Language Models; Fact-Checking Complex Claims with Program-Guided Reasoning +- **C1.I3** → NONE → EVALUATION_PROTOCOL + +## Claim C2 +- **C2.I1** → Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking → CONCEPTUAL_FRAMEWORK +- **C2.I2** → NONE → IMPLEMENTATION_TOOLING + +## Claim C3 +- **C3.I1** → Missing Counter-Evidence Renders NLP Fact-Checking Unrealistic for Misinformation → CONCEPTUAL_FRAMEWORK +- **C3.I2** → NONE → IMPLEMENTATION_TOOLING + +## Claim C4 +- **C4.I1** → AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web → CONCEPTUAL_FRAMEWORK +- **C4.I2** → Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models → CORE_METHOD + - additional: DEFAME: Dynamic Evidence-based FAct-checking with Multimodal Experts +- **C4.I3** → NONE → CORE_METHOD + +--- + +# 8. Final Rationales And Evidence + +## Claim C1: Dataset + +### C1.I1 +- **ingredient:** Source fact-checking articles containing real-world multimodal claims +- **canonical study:** NONE +- **additional studies:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web; AMMeBa: A Large-Scale Survey and Dataset of Media-Based Misinformation In-The-Wild +- **role:** DATA_SOURCE +- **contribution:** Supplies the raw fact-checking articles and multimodal claims that are normalized and annotated to form the dataset. +- **rationale:** The dataset's source material is inherently composite, drawing from multiple prior fact-checking repositories and APIs. No single prior study represents the entire source mixture, making `NONE` the correct canonical grounding, with the primary contributing datasets listed as additional studies. +- **evidence_span:** “we incorporated partially annotated articles from three sources: 1) filtered articles from AVeriTec... 2) articles verifying image-related claims from AMMEBA... and 3) all true claims from ClaimReview” + +### C1.I2 +- **ingredient:** Question-answering representation for fact-checking rationales +- **canonical study:** Complex Claim Verification with Evidence Retrieved in the Wild +- **additional studies:** The Art of SOCRATIC QUESTIONING: Recursive Thinking with Large Language Models; Fact-Checking Complex Claims with Program-Guided Reasoning +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the conceptual framework of using question-answering pairs to represent the decomposed reasoning steps required for claim verification. +- **rationale:** The paper explicitly adopts QA as the reasoning representation for fact-checking, citing prior works that established this paradigm. Chen et al. (CLAIMDECOMP) is a clean representative study for this QA-based verification framing. +- **evidence_span:** “To capture the rationale in claim verification, a complex reasoning task, various reasoning representations have been explored, including... QA [Chen et al., 2024, Qi et al., 2023, Pan et al., 2023a].” + +### C1.I3 +- **ingredient:** Multi-stage human annotation protocol for evidence retrieval and sufficiency checking +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the manual dataset-construction process ensuring that all claims are normalized, annotated with evidence, and verified for sufficiency. +- **rationale:** Constructing this dataset requires a complex, multi-stage human annotation workflow. Because this specific five-phase pipeline is realized in the target paper (though extending prior work), `NONE` is the correct grounding. +- **evidence_span:** “The five-phase annotation pipeline is illustrated in Figure 2, extending the annotation process proposed in [Schlichtkrull et al., 2023] to the domain of image-text claims.” + +## Claim C2: Metric + +### C2.I1 +- **ingredient:** Reference-based evaluation framework for textual evidence using LLMs +- **canonical study:** Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the base reference-based evaluation methodology that the paper extends to handle multimodal evidence. +- **rationale:** The paper explicitly builds its evaluation method upon the Ev2R framework, which demonstrated that LLM-based reference evaluation aligns well with human assessments for fact-checking evidence. +- **evidence_span:** “Recent research [Akhtar et al., 2024] showed that reference-based evaluation of evidence with large language models (LLMs) aligns best with human assessments. We extend this framework to a multimodal setting” + +### C2.I2 +- **ingredient:** Large multimodal model for scoring textual and visual alignments +- **canonical study:** NONE +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Acts as the automated evaluator that scores the alignment between retrieved multimodal evidence and human-annotated references. +- **rationale:** The metric operationally depends on an MLLM to perform the scoring. The paper uses Gemini 2.0 Flash, which is a commercial API/model rather than a canonical prior academic study defining the ingredient, making `NONE` the most appropriate grounding. +- **evidence_span:** “We exploit Gemini-2.0-Flash [DeepMind, 2024] as the scoring model for both steps in the reference-based evaluation” + +## Claim C3: Method (Retrieval Methodology) + +### C3.I1 +- **ingredient:** Conceptual framing of temporal leakage in fact-checking datasets +- **canonical study:** Missing Counter-Evidence Renders NLP Fact-Checking Unrealistic for Misinformation +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Identifies the problem of temporal leakage in fact-checking, motivating the need for date-restricted evidence retrieval. +- **rationale:** The paper explicitly cites Glockner et al. (2022) as the motivation for restricting search results to prevent temporal leakage, making it the clean canonical grounding for this concept. +- **evidence_span:** “All retrieved pages from the interface are restricted to dates prior to the claim date to prevent temporal leakage [Glockner et al., 2022].” + +### C3.I2 +- **ingredient:** Date-restricted web search and scraping tooling +- **canonical study:** NONE +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the operational mechanism to enforce the temporal constraints during evidence retrieval. +- **rationale:** The methodology relies on standard APIs and Python packages (like htmldate) to enforce date restrictions. These are general software tools rather than canonical prior research studies, so `NONE` is the correct grounding. +- **evidence_span:** “we use a post-hoc method by leveraging the Python package htmldate.fine_date to filter out pages published before claim dates.” + +## Claim C4: Method (Baseline Framework) + +### C4.I1 +- **ingredient:** Question-guided multi-hop reasoning framework for fact-checking +- **canonical study:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the foundational architecture of using generated questions to guide the fact-checking process. +- **rationale:** The paper explicitly builds its question generation approach upon the framework established by AVeriTeC, making it the cleanest canonical grounding. +- **evidence_span:** “A straightforward approach is to generate all verification questions at once, given an image-text claim, as in prior work [Schlichtkrull et al., 2023].” + +### C4.I2 +- **ingredient:** Tool-augmented LLM framework for answering questions +- **canonical study:** Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models +- **additional studies:** DEFAME: Dynamic Evidence-based FAct-checking with Multimodal Experts +- **role:** CORE_METHOD +- **contribution:** Provides the architecture for integrating specialized external tools into the LLM's answer generation process. +- **rationale:** The paper explicitly adopts the tool-augmented LLM paradigm to handle multimodal evidence retrieval, citing Visual ChatGPT and DEFAME as inspirations. Visual ChatGPT is a clean canonical representative for this broader paradigm. +- **evidence_span:** “Inspired by recent research on tool usage [Wu et al., 2023, Cao et al., 2024, Braun et al., 2024], we integrate a set of specialized tools into the answer generation module” + +### C4.I3 +- **ingredient:** Multimodal large language models for question generation and visual question answering +- **canonical study:** NONE +- **role:** CORE_METHOD +- **contribution:** Powers the internal reasoning, question generation, and visual analysis capabilities of the baseline framework. +- **rationale:** The framework is designed to be model-agnostic and tests several different MLLMs (Gemini, Qwen-VL, Gemma, LLaVA). Because the structural dependency is on the class of MLLMs rather than one specific canonical model, `NONE` is the correct grounding. +- **evidence_span:** “All three strategies employ an MLLM for question generation, leveraging the model's internal decomposition capability.” + + +Return only valid JSON with this exact schema: + +{ + "selected_candidate_index": 0, + "selected_candidate_id": "candidate_1", + "selected_reason": "...", + "candidate_scores": [ + { + "candidate_id": "candidate_1", + "candidate_index": 0, + "score": 1, + "assessment": "..." + } + ] +} + +Scoring rules: +- Use integer scores from 1 to 10. +- 10 = best candidate by the minimal structurally sufficient annotation objective. +- The selected candidate must have the highest score. +- In "selected_reason", explain why it best matches the downstream-first, minimal-ingredient, structurally faithful objective. +- In each "assessment", mention both strengths and weaknesses, especially over-decomposition, poor exclusions, weak abstraction level, bad role choices, and bad canonical vs "__NONE__" decisions. + +Return only valid JSON. Do not add any prose outside the JSON object. diff --git a/replay_traces/2505.17978/two_pass_outputs/20260726T204323Z/pass_1_reasoning.output.candidate_1.md b/replay_traces/2505.17978/two_pass_outputs/20260726T204323Z/pass_1_reasoning.output.candidate_1.md new file mode 100644 index 0000000000000000000000000000000000000000..0a2a92983754b27de715053b6032f1ef8b7ba0cd --- /dev/null +++ b/replay_traces/2505.17978/two_pass_outputs/20260726T204323Z/pass_1_reasoning.output.candidate_1.md @@ -0,0 +1,399 @@ +# 1. Cluster Evidence + +Based on the downstream cluster evidence, the target paper introduces four distinct downstream-used contributions: +- **AVerImaTeC Dataset:** Reused for new dataset creation (C1) and as an evaluation benchmark (C2). +- **Reference-based Evaluation Method:** Reused and extended for evaluating multimodal evidence in novel settings (C3). +- **Temporally Constrained Retrieval Methodology:** Reused as a protocol for dataset creation to prevent temporal leakage (C4). +- **Baseline Verification Framework:** Reused and extended for multimodal claim verification systems (C5). + +These contributions have distinct downstream roles and require different structural ingredients, so they should be split into four atomic claims. + +--- + +# 2. Claim Split Decision + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C1, C2 + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Method +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C3 + +### Claim C3 +- **claim_id:** C3 +- **artifact_type:** Protocol +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C4 + +### Claim C4 +- **claim_id:** C4 +- **artifact_type:** Method +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C5 + +### Why this split is correct +The paper bundles a dataset, an evaluation method, a data-collection protocol, and a baseline modeling framework. Downstream evidence shows that later works reuse these artifacts independently—some use the dataset for benchmarking, some adapt the evaluation method for new multimodal tasks, some adopt the temporal retrieval protocol for their own data collection, and others build on the baseline verification framework. They must be annotated separately to reflect their distinct downstream impact and structural dependencies. + +--- + +# 3. Rewritten Claims + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **rewritten_claim:** **Dataset: A real-world image-text claim verification dataset annotated with question-answer reasoning paths and web evidence, enabling the training and evaluation of multimodal fact-checking systems.** +- **why_this_is_atomic:** Isolates the dataset artifact from the evaluation methods and baseline models built around it. +- **decision:** YES_SUFFICIENT + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Method +- **rewritten_claim:** **Method: A reference-based evaluation method using large language models to separately assess the textual and visual components of retrieved evidence, enabling automated alignment with human judgments in multimodal fact-checking.** +- **why_this_is_atomic:** Isolates the specific evaluation metric/methodology from the dataset it was used to evaluate. +- **decision:** YES_SUFFICIENT + +### Claim C3 +- **claim_id:** C3 +- **artifact_type:** Protocol +- **rewritten_claim:** **Protocol: A temporally constrained evidence retrieval methodology that restricts search results to dates prior to the claim, enabling the construction of fact-checking datasets without temporal leakage.** +- **why_this_is_atomic:** Isolates the data-collection protocol from the final dataset, as downstream work reuses this specific methodology for their own dataset creation. +- **decision:** YES_SUFFICIENT + +### Claim C4 +- **claim_id:** C4 +- **artifact_type:** Method +- **rewritten_claim:** **Method: A multimodal claim verification framework combining dynamic question generation and tool-augmented answer generation, enabling automated fact-checking of image-text claims using open-web evidence.** +- **why_this_is_atomic:** Isolates the baseline modeling framework from the dataset and evaluation metrics. +- **decision:** YES_SUFFICIENT + +--- + +# 4. Minimal Ingredient Set Per Claim + +## Claim C1: AVerImaTeC Dataset + +### Ingredient C1.I1 +- **ingredient_id:** C1.I1 +- **ingredient:** Source fact-checking articles containing multimodal claims +- **why_structurally_necessary:** The dataset is built by extracting and normalizing real-world claims from existing fact-checking articles. Without this raw substrate, the dataset would not exist. +- **why_not_lower_level_substeps:** This should remain a composite data-source ingredient rather than being split into individual source datasets. +- **why_not_adjacent_implementation_details:** The exact filtering heuristics are implementation details; the structural dependency is the source text itself. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web; AMMeBa: A Large-Scale Survey and Dataset of Media-Based Misinformation In-The-Wild +- **role:** DATA_SOURCE +- **contribution:** Provides the raw real-world image-text claims that are normalized and annotated to form the dataset. +- **rationale:** The dataset is a composite of claims extracted from multiple prior fact-checking resources. No single prior study fully represents this substrate, making `NONE` the correct canonical grounding, with the primary sources listed as additional groundings. +- **evidence_span:** “we incorporated partially annotated articles from three sources: 1) filtered articles from AVeriTec... 2) articles verifying image-related claims from AMMEBA” + +### Ingredient C1.I2 +- **ingredient_id:** C1.I2 +- **ingredient:** Question-answering representation for verification rationale +- **why_structurally_necessary:** The defining feature of the dataset is that the reasoning process for verification is decomposed into QA pairs. Without this conceptual framing, it would just be a standard claim-verdict dataset. +- **why_not_lower_level_substeps:** This is a high-level conceptual framing and should not be decomposed into specific question types or answer formats. +- **why_not_adjacent_implementation_details:** The choice of QA over natural logic is the structural dependency, not the specific UI used to collect the QA pairs. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Complex Claim Verification with Evidence Retrieved in the Wild +- **additional groundings if any:** none +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the QA-based reasoning representation used to structure the evidence retrieval and verification process. +- **rationale:** The paper explicitly adopts QA as the reasoning representation for fact-checking, citing prior work that established this paradigm. Chen et al. (CLAIMDECOMP) is a clean representative for QA-based claim decomposition. +- **evidence_span:** “To capture the rationale in claim verification, a complex reasoning task, various reasoning representations have been explored... and QA” + +### Ingredient C1.I3 +- **ingredient_id:** C1.I3 +- **ingredient:** Four-way veracity labeling schema +- **why_structurally_necessary:** The dataset categorizes claims into specific veracity labels. This schema defines the classification task the dataset supports. +- **why_not_lower_level_substeps:** This is a single conceptual ingredient defining the label space. +- **why_not_adjacent_implementation_details:** The specific definitions of each label are part of this broader schema ingredient. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web +- **additional groundings if any:** none +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the specific four-way label space (supported, refuted, not enough evidence, conflicting) used to annotate the claims. +- **rationale:** The paper explicitly adopts the exact four-way veracity labeling schema introduced in AVeriTeC. +- **evidence_span:** “We follow the four-way veracity labeling schema from [Schlichtkrull et al., 2023]: supported, refuted, not enough evidence, and conflicting/cherry-picking.” + +### Ingredient C1.I4 +- **ingredient_id:** C1.I4 +- **ingredient:** Multi-stage annotation pipeline with evidence sufficiency checks +- **why_structurally_necessary:** The dataset's quality and reliability depend on a structured annotation workflow that includes independent sufficiency checks to resolve conflicts. +- **why_not_lower_level_substeps:** This should remain one protocol-level ingredient rather than being split into claim extraction, QA generation, and sufficiency checking separately. +- **why_not_adjacent_implementation_details:** The structural dependency is the pipeline itself, not the specific number of annotators or their demographics. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the foundational multi-stage annotation and verification workflow that the paper extends for image-text claims. +- **rationale:** The paper explicitly states that its annotation pipeline is an extension of the process proposed in AVeriTeC, making it the canonical grounding for this workflow. +- **evidence_span:** “extending the annotation process proposed in [Schlichtkrull et al., 2023] to the domain of image-text claims.” + +--- + +## Claim C2: Reference-based Evaluation Method + +### Ingredient C2.I1 +- **ingredient_id:** C2.I1 +- **ingredient:** LLM-based reference evaluation framework for evidence retrieval +- **why_structurally_necessary:** The method relies on using an LLM to compare retrieved evidence against human-annotated references, rather than relying on exact string matching. +- **why_not_lower_level_substeps:** This is the core conceptual basis of the evaluation method and should not be split into specific prompt designs. +- **why_not_adjacent_implementation_details:** The choice of Gemini as the specific LLM is an implementation detail; the structural dependency is the LLM-as-a-judge framework. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the foundation for using LLMs to evaluate evidence retrieval by comparing it to references. +- **rationale:** The paper explicitly adopts the Ev2R framework for its reference-based evaluation of textual evidence. +- **evidence_span:** “Recent research [Akhtar et al., 2024] showed that reference-based evaluation of evidence with large language models (LLMs) aligns best with human assessments.” + +### Ingredient C2.I2 +- **ingredient_id:** C2.I2 +- **ingredient:** Separated reference-based evaluation for textual and visual components +- **why_structurally_necessary:** The core novelty of the method is extending reference-based evaluation to multimodal evidence by evaluating text and images separately, which the paper shows is more robust than interleaved evaluation. +- **why_not_lower_level_substeps:** This should remain a single protocol ingredient rather than being split into the text-check step and the image-check step. +- **why_not_adjacent_implementation_details:** The specific similarity thresholds used are implementation details. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the novel mechanism for robustly evaluating multimodal evidence by decoupling the modalities during the LLM judgment phase. +- **rationale:** This separated evaluation strategy is the target paper's own novel extension to prior text-only reference evaluation methods, so it maps to `NONE`. +- **evidence_span:** “We then conduct separate reference-based evaluations for the textual and visual components.” + +--- + +## Claim C3: Temporally Constrained Retrieval Methodology + +### Ingredient C3.I1 +- **ingredient_id:** C3.I1 +- **ingredient:** Identification of temporal leakage as a critical flaw in fact-checking datasets +- **why_structurally_necessary:** The entire protocol is designed to solve the problem of temporal leakage (where models use future information to verify past claims). Without this conceptual framing, the protocol has no purpose. +- **why_not_lower_level_substeps:** This is a high-level conceptual ingredient. +- **why_not_adjacent_implementation_details:** N/A +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Missing Counter-Evidence Renders NLP Fact-Checking Unrealistic for Misinformation +- **additional groundings if any:** none +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the theoretical motivation for restricting evidence retrieval to prevent models from accessing future knowledge. +- **rationale:** The paper explicitly cites Glockner et al. as the motivation for preventing temporal leakage during evidence collection. +- **evidence_span:** “restricted to dates prior to the claim date to prevent temporal leakage [Glockner et al., 2022].” + +### Ingredient C3.I2 +- **ingredient_id:** C3.I2 +- **ingredient:** Date-restricted search interface and post-hoc filtering mechanism +- **why_structurally_necessary:** This is the operational mechanism of the protocol. Without restricting the search API and filtering results by date, the temporal constraint cannot be enforced. +- **why_not_lower_level_substeps:** This should remain one mechanism ingredient rather than being split into the Google Search API arguments and the Python htmldate package. +- **why_not_adjacent_implementation_details:** The specific Python packages used are implementation details; the structural dependency is the date-restriction mechanism itself. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the practical tooling to enforce the temporal constraints during dataset construction. +- **rationale:** The specific implementation of the custom search interface and post-hoc date filtering is developed in the target paper, mapping to `NONE`. +- **evidence_span:** “annotators are provided with a custom Google search interface... restricted to dates prior to the claim date” + +--- + +## Claim C4: Baseline Verification Framework + +### Ingredient C4.I1 +- **ingredient_id:** C4.I1 +- **ingredient:** Tool-augmented answer generation module with specialized search and VQA tools +- **why_structurally_necessary:** The baseline framework relies on an LLM selecting and using external tools (RIS, WST, WSI, VQA) to answer generated questions. Without this tool-use architecture, the framework could not retrieve open-web multimodal evidence. +- **why_not_lower_level_substeps:** This should remain one architectural ingredient rather than being split into each individual tool. +- **why_not_adjacent_implementation_details:** The specific prompts used for tool selection are implementation details. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models +- **additional groundings if any:** none +- **role:** CORE_METHOD +- **contribution:** Provides the paradigm of augmenting language models with specialized visual and search tools to solve complex multimodal tasks. +- **rationale:** The paper explicitly grounds its tool-augmented answer generation module in recent research on tool usage, with Visual ChatGPT (Wu et al.) being a canonical representative for integrating visual tools with LLMs. +- **evidence_span:** “we integrate a set of specialized tools into the answer generation module, along with a tool selector” + +### Ingredient C4.I2 +- **ingredient_id:** C4.I2 +- **ingredient:** Dynamic question generation strategy based on evolving evidence history +- **why_structurally_necessary:** The framework introduces dynamic and hybrid question generation to handle multi-hop reasoning where subsequent questions depend on previous answers. +- **why_not_lower_level_substeps:** This is a core methodological strategy and should not be decomposed into specific prompt templates. +- **why_not_adjacent_implementation_details:** The choice of specific MLLMs (like Qwen-VL) is an implementation detail; the structural dependency is the dynamic generation strategy. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** CORE_METHOD +- **contribution:** Provides the mechanism for generating sequential verification questions conditioned on the accumulated evidence context. +- **rationale:** The dynamic and hybrid question generation strategies are proposed by the target paper as novel components of their baseline framework, mapping to `NONE`. +- **evidence_span:** “we also propose a dynamic question generation (DQG) method, where each subsequent question is generated based on both the claim and the evolving evidence history.” + +--- + +# 5. Excluded Tempting Non-Ingredients + +- **Specific LLMs (Gemini, Qwen, Gemma, LLaVA)** + - Excluded because they are interchangeable implementation choices for the baseline framework and evaluation method, rather than structural prerequisites defining the artifacts themselves. +- **BM25 for text chunk retrieval** + - Excluded because it is a low-level implementation detail for the text search tool, not a high-level structural ingredient of the baseline framework. +- **Python htmldate package** + - Excluded because it is merely a local tool used to implement the date-restriction mechanism, which is already captured at a higher structural level. + +--- + +# 6. Candidate Grounding Decisions + +### Claim C1: AVerImaTeC Dataset +- **C1.I1:** NONE (accepted_none). The source data is composite. AVeriTeC and AMMeBa are accepted as additional groundings. +- **C1.I2:** Complex Claim Verification with Evidence Retrieved in the Wild (accepted_canonical). Clean representative for QA-based claim decomposition. +- **C1.I3:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web (accepted_canonical). Directly provides the four-way veracity schema. +- **C1.I4:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web (accepted_canonical). Directly provides the foundational annotation pipeline. + +### Claim C2: Reference-based Evaluation Method +- **C2.I1:** Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking (accepted_canonical). Directly provides the LLM-based reference evaluation framework. +- **C2.I2:** NONE (accepted_none). The separated multimodal evaluation strategy is introduced in the target paper. + +### Claim C3: Temporally Constrained Retrieval Methodology +- **C3.I1:** Missing Counter-Evidence Renders NLP Fact-Checking Unrealistic for Misinformation (accepted_canonical). Provides the conceptual motivation regarding temporal leakage. +- **C3.I2:** NONE (accepted_none). The specific date-restricted search implementation is from the target paper. + +### Claim C4: Baseline Verification Framework +- **C4.I1:** Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models (accepted_canonical). Representative for tool-augmented LLM architectures. +- **C4.I2:** NONE (accepted_none). The dynamic question generation strategy is proposed in the target paper. + +--- + +# 7. Final Grounding Decisions + +## Claim C1 +- **C1.I1** → NONE → DATA_SOURCE + - additional: AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web; AMMeBa: A Large-Scale Survey and Dataset of Media-Based Misinformation In-The-Wild +- **C1.I2** → Complex Claim Verification with Evidence Retrieved in the Wild → CONCEPTUAL_FRAMEWORK +- **C1.I3** → AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web → CONCEPTUAL_FRAMEWORK +- **C1.I4** → AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web → EVALUATION_PROTOCOL + +## Claim C2 +- **C2.I1** → Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking → EVALUATION_PROTOCOL +- **C2.I2** → NONE → EVALUATION_PROTOCOL + +## Claim C3 +- **C3.I1** → Missing Counter-Evidence Renders NLP Fact-Checking Unrealistic for Misinformation → CONCEPTUAL_FRAMEWORK +- **C3.I2** → NONE → IMPLEMENTATION_TOOLING + +## Claim C4 +- **C4.I1** → Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models → CORE_METHOD +- **C4.I2** → NONE → CORE_METHOD + +--- + +# 8. Final Rationales And Evidence + +## Claim C1: AVerImaTeC Dataset + +### C1.I1 +- **ingredient:** Source fact-checking articles containing multimodal claims +- **canonical study:** NONE +- **additional studies:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web; AMMeBa: A Large-Scale Survey and Dataset of Media-Based Misinformation In-The-Wild +- **role:** DATA_SOURCE +- **contribution:** Provides the raw real-world image-text claims that are normalized and annotated to form the dataset. +- **rationale:** The dataset is a composite of claims extracted from multiple prior fact-checking resources. No single prior study fully represents this substrate, making `NONE` the correct canonical grounding, with the primary sources listed as additional groundings. +- **evidence_span:** “we incorporated partially annotated articles from three sources: 1) filtered articles from AVeriTec... 2) articles verifying image-related claims from AMMEBA” + +### C1.I2 +- **ingredient:** Question-answering representation for verification rationale +- **canonical study:** Complex Claim Verification with Evidence Retrieved in the Wild +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the QA-based reasoning representation used to structure the evidence retrieval and verification process. +- **rationale:** The paper explicitly adopts QA as the reasoning representation for fact-checking, citing prior work that established this paradigm. Chen et al. (CLAIMDECOMP) is a clean representative for QA-based claim decomposition. +- **evidence_span:** “To capture the rationale in claim verification, a complex reasoning task, various reasoning representations have been explored... and QA” + +### C1.I3 +- **ingredient:** Four-way veracity labeling schema +- **canonical study:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the specific four-way label space (supported, refuted, not enough evidence, conflicting) used to annotate the claims. +- **rationale:** The paper explicitly adopts the exact four-way veracity labeling schema introduced in AVeriTeC. +- **evidence_span:** “We follow the four-way veracity labeling schema from [Schlichtkrull et al., 2023]: supported, refuted, not enough evidence, and conflicting/cherry-picking.” + +### C1.I4 +- **ingredient:** Multi-stage annotation pipeline with evidence sufficiency checks +- **canonical study:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the foundational multi-stage annotation and verification workflow that the paper extends for image-text claims. +- **rationale:** The paper explicitly states that its annotation pipeline is an extension of the process proposed in AVeriTeC, making it the canonical grounding for this workflow. +- **evidence_span:** “extending the annotation process proposed in [Schlichtkrull et al., 2023] to the domain of image-text claims.” + +## Claim C2: Reference-based Evaluation Method + +### C2.I1 +- **ingredient:** LLM-based reference evaluation framework for evidence retrieval +- **canonical study:** Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the foundation for using LLMs to evaluate evidence retrieval by comparing it to references. +- **rationale:** The paper explicitly adopts the Ev2R framework for its reference-based evaluation of textual evidence. +- **evidence_span:** “Recent research [Akhtar et al., 2024] showed that reference-based evaluation of evidence with large language models (LLMs) aligns best with human assessments.” + +### C2.I2 +- **ingredient:** Separated reference-based evaluation for textual and visual components +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the novel mechanism for robustly evaluating multimodal evidence by decoupling the modalities during the LLM judgment phase. +- **rationale:** This separated evaluation strategy is the target paper's own novel extension to prior text-only reference evaluation methods, so it maps to `NONE`. +- **evidence_span:** “We then conduct separate reference-based evaluations for the textual and visual components.” + +## Claim C3: Temporally Constrained Retrieval Methodology + +### C3.I1 +- **ingredient:** Identification of temporal leakage as a critical flaw in fact-checking datasets +- **canonical study:** Missing Counter-Evidence Renders NLP Fact-Checking Unrealistic for Misinformation +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the theoretical motivation for restricting evidence retrieval to prevent models from accessing future knowledge. +- **rationale:** The paper explicitly cites Glockner et al. as the motivation for preventing temporal leakage during evidence collection. +- **evidence_span:** “restricted to dates prior to the claim date to prevent temporal leakage [Glockner et al., 2022].” + +### C3.I2 +- **ingredient:** Date-restricted search interface and post-hoc filtering mechanism +- **canonical study:** NONE +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the practical tooling to enforce the temporal constraints during dataset construction. +- **rationale:** The specific implementation of the custom search interface and post-hoc date filtering is developed in the target paper, mapping to `NONE`. +- **evidence_span:** “annotators are provided with a custom Google search interface... restricted to dates prior to the claim date” + +## Claim C4: Baseline Verification Framework + +### C4.I1 +- **ingredient:** Tool-augmented answer generation module with specialized search and VQA tools +- **canonical study:** Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models +- **role:** CORE_METHOD +- **contribution:** Provides the paradigm of augmenting language models with specialized visual and search tools to solve complex multimodal tasks. +- **rationale:** The paper explicitly grounds its tool-augmented answer generation module in recent research on tool usage, with Visual ChatGPT (Wu et al.) being a canonical representative for integrating visual tools with LLMs. +- **evidence_span:** “we integrate a set of specialized tools into the answer generation module, along with a tool selector” + +### C4.I2 +- **ingredient:** Dynamic question generation strategy based on evolving evidence history +- **canonical study:** NONE +- **role:** CORE_METHOD +- **contribution:** Provides the mechanism for generating sequential verification questions conditioned on the accumulated evidence context. +- **rationale:** The dynamic and hybrid question generation strategies are proposed by the target paper as novel components of their baseline framework, mapping to `NONE`. +- **evidence_span:** “we also propose a dynamic question generation (DQG) method, where each subsequent question is generated based on both the claim and the evolving evidence history.” \ No newline at end of file diff --git a/replay_traces/2505.17978/two_pass_outputs/20260726T204323Z/pass_1_reasoning.output.candidate_2.md b/replay_traces/2505.17978/two_pass_outputs/20260726T204323Z/pass_1_reasoning.output.candidate_2.md new file mode 100644 index 0000000000000000000000000000000000000000..5037b18774e3387ac5e0b4e9c196999e0e0fcf20 --- /dev/null +++ b/replay_traces/2505.17978/two_pass_outputs/20260726T204323Z/pass_1_reasoning.output.candidate_2.md @@ -0,0 +1,327 @@ +# 1. Cluster Evidence + +Based on the downstream cluster evidence, the target paper introduces four distinct downstream-used contributions: +- **C1 & C2** show downstream reuse of the **AVerImaTeC dataset** as a resource for training and evaluating multimodal fact-checking systems. +- **C3** shows downstream reuse of the **reference-based evaluation method** for assessing multimodal evidence retrieval. +- **C4** shows downstream reuse of the **temporally constrained retrieval methodology** for creating new datasets without temporal leakage. +- **C5** shows downstream reuse of the **baseline verification framework** for multimodal claim verification. + +Because these four artifacts have distinct downstream roles and require different structural ingredients, they should be split into four separate atomic claims. + +--- + +# 2. Claim Split Decision + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C1, C2 + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Method +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C3 + +### Claim C3 +- **claim_id:** C3 +- **artifact_type:** Method +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C4 + +### Claim C4 +- **claim_id:** C4 +- **artifact_type:** Method +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C5 + +### Why this split is correct +The paper bundles a dataset, an evaluation metric, a data collection protocol, and a baseline system. Downstream evidence confirms that later works reuse these components independently (e.g., one paper adopts the temporal constraint protocol to build its own dataset, while another extends the baseline framework for a new system). They must be annotated as separate claims to accurately reflect their distinct structural dependencies. + +--- + +# 3. Rewritten Claims + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **rewritten_claim:** **Dataset: A real-world image-text claim verification dataset annotated with question-answer pairs and web evidence, enabling the training and evaluation of multimodal fact-checking systems.** +- **why_this_is_atomic:** This claim isolates the dataset resource itself, separating it from the evaluation metrics and baseline models introduced alongside it. +- **optional cluster_id:** C1, C2 +- **decision:** YES_SUFFICIENT + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Method +- **rewritten_claim:** **Method: A reference-based evaluation method for multimodal evidence retrieval, enabling the automated assessment of generated text and retrieved images against human annotations.** +- **why_this_is_atomic:** This claim isolates the specific evaluation metric/methodology used to score evidence retrieval, which is reused independently of the dataset. +- **optional cluster_id:** C3 +- **decision:** YES_SUFFICIENT + +### Claim C3 +- **claim_id:** C3 +- **artifact_type:** Method +- **rewritten_claim:** **Method: A temporally constrained evidence retrieval methodology, enabling the creation of fact-checking datasets that prevent temporal leakage by restricting search dates.** +- **why_this_is_atomic:** This claim isolates the data collection protocol for preventing temporal leakage, which downstream work adopts for new dataset creation. +- **optional cluster_id:** C4 +- **decision:** YES_SUFFICIENT + +### Claim C4 +- **claim_id:** C4 +- **artifact_type:** Method +- **rewritten_claim:** **Method: A baseline verification framework that generates questions and uses specialized tools to retrieve multimodal evidence, enabling automated image-text claim verification.** +- **why_this_is_atomic:** This claim isolates the tool-augmented baseline system, which is extended by downstream work independently of the dataset. +- **optional cluster_id:** C5 +- **decision:** YES_SUFFICIENT + +--- + +# 4. Minimal Ingredient Set Per Claim + +## Claim C1: AVerImaTeC Dataset + +### Ingredient C1.I1 +- **ingredient_id:** C1.I1 +- **ingredient:** Source fact-checking articles containing real-world multimodal claims +- **why_structurally_necessary:** The dataset is built by extracting and normalizing real-world image-text claims. Without a substrate of existing fact-checking articles containing multimodal claims, the dataset would not exist in its claimed form. +- **why_not_lower_level_substeps:** This should remain a composite data-source ingredient rather than being split into separate ingredients for each individual source platform. +- **why_not_adjacent_implementation_details:** The specific filtering scripts used to extract the claims are implementation details; the structural dependency is the raw source data itself. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web; AMMeBa: A Large-Scale Survey and Dataset of Media-Based Misinformation In-The-Wild +- **role:** DATA_SOURCE +- **contribution:** Provides the raw real-world image-text claims and fact-checking articles that are annotated to create the dataset. +- **rationale:** The dataset is explicitly constructed by aggregating and filtering partially annotated articles from multiple prior sources (AVeriTeC, AMMeBA, and ClaimReview). Because this source substrate is composite, `NONE` is the correct canonical grounding, with the primary contributing datasets listed as additional groundings. +- **evidence_span:** “we incorporated partially annotated articles from three sources: 1) filtered articles from AVeriTec... 2) articles verifying image-related claims from AMMEBA” + +### Ingredient C1.I2 +- **ingredient_id:** C1.I2 +- **ingredient:** Question-answering annotation pipeline for structuring fact-checking rationales +- **why_structurally_necessary:** The defining property of the dataset is that verification rationales are decomposed into QA pairs with web evidence. Without this specific annotation framing, it would just be a standard claim-verdict dataset. +- **why_not_lower_level_substeps:** This should remain one protocol-level ingredient rather than being split into question generation, evidence retrieval, and sufficiency checking as separate ingredients. +- **why_not_adjacent_implementation_details:** The exact web interface used by annotators is an implementation detail; the structural dependency is the QA-based annotation pipeline. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the multi-phase annotation pipeline that decomposes claim verification into QA pairs and justifications. +- **rationale:** The paper explicitly states that its annotation pipeline extends the process proposed in AVeriTeC to the domain of image-text claims. AVeriTeC is the cleanest canonical grounding for this QA-based fact-checking annotation protocol. +- **evidence_span:** “extending the annotation process proposed in [Schlichtkrull et al., 2023] to the domain of image-text claims.” + +## Claim C2: Reference-based evaluation method + +### Ingredient C2.I1 +- **ingredient_id:** C2.I1 +- **ingredient:** LLM-based reference evaluation framework for evidence retrieval +- **why_structurally_necessary:** The evaluation method operates by converting QA pairs into evidence statements and using an LLM to compare predictions against human references. Without this conceptual framework, the metric would not function. +- **why_not_lower_level_substeps:** This should remain one framework-level ingredient rather than being split into text-comparison and image-comparison substeps. +- **why_not_adjacent_implementation_details:** The specific prompt templates used for the LLM evaluator are implementation details. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking +- **additional groundings if any:** none +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the foundational reference-based evaluation framework that the paper adapts for multimodal evidence. +- **rationale:** The paper explicitly motivates and adopts its reference-based evaluation method from Ev2R (Akhtar et al., 2024), making it the direct canonical grounding for this methodology. +- **evidence_span:** “Motivated by a recent study [Akhtar et al., 2024], we adopt a reference-based evaluation method that compares model responses to human-annotated ground-truth data.” + +## Claim C3: Temporally constrained retrieval methodology + +### Ingredient C3.I1 +- **ingredient_id:** C3.I1 +- **ingredient:** Temporal leakage prevention constraint for evidence retrieval +- **why_structurally_necessary:** The methodology is defined by restricting search results to dates prior to the claim to ensure models cannot access future information. Without this constraint, the methodology loses its defining purpose. +- **why_not_lower_level_substeps:** This should remain a single conceptual constraint rather than being split into the specific Python packages or APIs used to enforce it. +- **why_not_adjacent_implementation_details:** The use of `htmldate.find_date` is a local implementation choice; the structural dependency is the temporal constraint framing itself. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Missing Counter-Evidence Renders NLP Fact-Checking Unrealistic for Misinformation +- **additional groundings if any:** none +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the conceptual framing and necessity of restricting evidence retrieval by date to prevent temporal leakage in fact-checking. +- **rationale:** The paper explicitly grounds its temporal restriction methodology in the temporal leakage problem identified by Glockner et al., 2022, making it the canonical source for this constraint. +- **evidence_span:** “All retrieved pages from the interface are restricted to dates prior to the claim date to prevent temporal leakage [Glockner et al., 2022].” + +## Claim C4: Baseline verification framework + +### Ingredient C4.I1 +- **ingredient_id:** C4.I1 +- **ingredient:** Tool-augmented architecture for multimodal evidence retrieval and answering +- **why_structurally_necessary:** The baseline framework relies on integrating specialized tools (reverse image search, web search, VQA) to answer generated questions. Without this tool-use architecture, the system could not retrieve external multimodal evidence. +- **why_not_lower_level_substeps:** This should remain one architecture-level ingredient rather than being split into individual tools (RIS, WST, WSI) as separate ingredients. +- **why_not_adjacent_implementation_details:** The specific tool selection prompts are implementation details; the structural dependency is the tool-augmented generation paradigm. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** DEFAME: Dynamic Evidence-based FAct-checking with Multimodal Experts +- **additional groundings if any:** Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models +- **role:** CORE_METHOD +- **contribution:** Provides the paradigm of integrating specialized external tools into an LLM's answer generation module. +- **rationale:** The paper explicitly builds its tool-augmented answer generator based on recent research on tool usage. DEFAME (Braun et al., 2024) is the most direct canonical grounding as it applies this specifically to fact-checking, while Visual ChatGPT serves as an additional grounding for the broader multimodal tool-use concept. +- **evidence_span:** “Inspired by recent research on tool usage [Wu et al., 2023, Cao et al., 2024, Braun et al., 2024], we integrate a set of specialized tools” + +### Ingredient C4.I2 +- **ingredient_id:** C4.I2 +- **ingredient:** Pretrained multimodal large language model for visual reasoning and verification +- **why_structurally_necessary:** The framework requires an MLLM to process interleaved image-text claims, perform visual question answering, and predict the final verdict. Without a pretrained MLLM, the multimodal verification system would not function. +- **why_not_lower_level_substeps:** This should remain at the level of the MLLM initialization rather than being split into vision encoders and text decoders. +- **why_not_adjacent_implementation_details:** The specific choice between Qwen-VL, Gemini, or LLaVA is an implementation detail, but the structural requirement is the presence of a capable pretrained MLLM. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Qwen2.5-VL Technical Report +- **additional groundings if any:** none +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the foundational multimodal reasoning capabilities used by the baseline framework to process images and text jointly. +- **rationale:** The baseline framework operationally requires an MLLM. The paper tests several, but Qwen2.5-VL is explicitly highlighted as a primary open-source MLLM used in the strongest baseline configurations, making it a clean canonical grounding for this capability. +- **evidence_span:** “Qwen2.5-7B-Instruct [Yang et al., 2024] ( Qwen ) acts as the LLM and Qwen2.5-VL-7B-Instruct [Bai et al., 2025] ( Qwen-VL ) serves as the MLLM” + +--- + +# 5. Excluded Tempting Non-Ingredients + +### Claim C1 exclusions +- **Inter-annotator agreement metrics (e.g., Randolph's Kappa)** + - Excluded because these are evaluation metrics used to validate the dataset's quality, not structural ingredients required to construct the dataset itself. + +### Claim C4 exclusions +- **BM25 ranking for textual evidence retrieval** + - Excluded because it is a local implementation detail for filtering text search results, not a core structural component of the overall baseline framework. +- **Specific question generation strategies (PQG, DQG, HQG)** + - Excluded because these are algorithmic variants explored in the paper's experiments, whereas the structural dependency is the broader tool-augmented MLLM framework itself. + +--- + +# 6. Candidate Grounding Decisions + +## Claim C1: AVerImaTeC Dataset +### Ingredient C1.I1 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The source data is a composite of multiple prior fact-checking datasets and platforms. +- **Candidate:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web +- **decision:** accepted_additional +- **why:** Provides a significant portion of the source articles. +- **Candidate:** AMMeBa: A Large-Scale Survey and Dataset of Media-Based Misinformation In-The-Wild +- **decision:** accepted_additional +- **why:** Provides source articles specifically verifying image-related claims. + +### Ingredient C1.I2 +- **Candidate:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web +- **decision:** accepted_canonical +- **why:** The paper explicitly extends the QA-based annotation pipeline introduced in AVeriTeC. + +## Claim C2: Reference-based evaluation method +### Ingredient C2.I1 +- **Candidate:** Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking +- **decision:** accepted_canonical +- **why:** The paper explicitly adopts and adapts the reference-based evaluation method from Ev2R. + +## Claim C3: Temporally constrained retrieval methodology +### Ingredient C3.I1 +- **Candidate:** Missing Counter-Evidence Renders NLP Fact-Checking Unrealistic for Misinformation +- **decision:** accepted_canonical +- **why:** The paper explicitly grounds its temporal leakage prevention constraint in this prior study. + +## Claim C4: Baseline verification framework +### Ingredient C4.I1 +- **Candidate:** DEFAME: Dynamic Evidence-based FAct-checking with Multimodal Experts +- **decision:** accepted_canonical +- **why:** Cleanest representative for applying tool-augmented LLM architectures specifically to fact-checking. +- **Candidate:** Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models +- **decision:** accepted_additional +- **why:** Foundational prior work on integrating specialized visual tools with LLMs. + +### Ingredient C4.I2 +- **Candidate:** Qwen2.5-VL Technical Report +- **decision:** accepted_canonical +- **why:** Cleanest representative of the pretrained MLLM capability operationally required by the baseline framework. + +--- + +# 7. Final Grounding Decisions + +## Claim C1 +- **C1.I1** → NONE → DATA_SOURCE + - additional: AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web; AMMeBa: A Large-Scale Survey and Dataset of Media-Based Misinformation In-The-Wild +- **C1.I2** → AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web → EVALUATION_PROTOCOL + +## Claim C2 +- **C2.I1** → Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking → CONCEPTUAL_FRAMEWORK + +## Claim C3 +- **C3.I1** → Missing Counter-Evidence Renders NLP Fact-Checking Unrealistic for Misinformation → CONCEPTUAL_FRAMEWORK + +## Claim C4 +- **C4.I1** → DEFAME: Dynamic Evidence-based FAct-checking with Multimodal Experts → CORE_METHOD + - additional: Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models +- **C4.I2** → Qwen2.5-VL Technical Report → IMPLEMENTATION_TOOLING + +--- + +# 8. Final Rationales And Evidence + +## Claim C1: AVerImaTeC Dataset + +### C1.I1 +- **ingredient:** Source fact-checking articles containing real-world multimodal claims +- **canonical study:** NONE +- **additional studies:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web; AMMeBa: A Large-Scale Survey and Dataset of Media-Based Misinformation In-The-Wild +- **role:** DATA_SOURCE +- **contribution:** Provides the raw real-world image-text claims and fact-checking articles that are annotated to create the dataset. +- **rationale:** The dataset is explicitly constructed by aggregating and filtering partially annotated articles from multiple prior sources (AVeriTeC, AMMeBA, and ClaimReview). Because this source substrate is composite, `NONE` is the correct canonical grounding, with the primary contributing datasets listed as additional groundings. +- **evidence_span:** “we incorporated partially annotated articles from three sources: 1) filtered articles from AVeriTec... 2) articles verifying image-related claims from AMMEBA” + +### C1.I2 +- **ingredient:** Question-answering annotation pipeline for structuring fact-checking rationales +- **canonical study:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the multi-phase annotation pipeline that decomposes claim verification into QA pairs and justifications. +- **rationale:** The paper explicitly states that its annotation pipeline extends the process proposed in AVeriTeC to the domain of image-text claims. AVeriTeC is the cleanest canonical grounding for this QA-based fact-checking annotation protocol. +- **evidence_span:** “extending the annotation process proposed in [Schlichtkrull et al., 2023] to the domain of image-text claims.” + +## Claim C2: Reference-based evaluation method + +### C2.I1 +- **ingredient:** LLM-based reference evaluation framework for evidence retrieval +- **canonical study:** Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the foundational reference-based evaluation framework that the paper adapts for multimodal evidence. +- **rationale:** The paper explicitly motivates and adopts its reference-based evaluation method from Ev2R (Akhtar et al., 2024), making it the direct canonical grounding for this methodology. +- **evidence_span:** “Motivated by a recent study [Akhtar et al., 2024], we adopt a reference-based evaluation method that compares model responses to human-annotated ground-truth data.” + +## Claim C3: Temporally constrained retrieval methodology + +### C3.I1 +- **ingredient:** Temporal leakage prevention constraint for evidence retrieval +- **canonical study:** Missing Counter-Evidence Renders NLP Fact-Checking Unrealistic for Misinformation +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the conceptual framing and necessity of restricting evidence retrieval by date to prevent temporal leakage in fact-checking. +- **rationale:** The paper explicitly grounds its temporal restriction methodology in the temporal leakage problem identified by Glockner et al., 2022, making it the canonical source for this constraint. +- **evidence_span:** “All retrieved pages from the interface are restricted to dates prior to the claim date to prevent temporal leakage [Glockner et al., 2022].” + +## Claim C4: Baseline verification framework + +### C4.I1 +- **ingredient:** Tool-augmented architecture for multimodal evidence retrieval and answering +- **canonical study:** DEFAME: Dynamic Evidence-based FAct-checking with Multimodal Experts +- **additional studies:** Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models +- **role:** CORE_METHOD +- **contribution:** Provides the paradigm of integrating specialized external tools into an LLM's answer generation module. +- **rationale:** The paper explicitly builds its tool-augmented answer generator based on recent research on tool usage. DEFAME (Braun et al., 2024) is the most direct canonical grounding as it applies this specifically to fact-checking, while Visual ChatGPT serves as an additional grounding for the broader multimodal tool-use concept. +- **evidence_span:** “Inspired by recent research on tool usage [Wu et al., 2023, Cao et al., 2024, Braun et al., 2024], we integrate a set of specialized tools” + +### C4.I2 +- **ingredient:** Pretrained multimodal large language model for visual reasoning and verification +- **canonical study:** Qwen2.5-VL Technical Report +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the foundational multimodal reasoning capabilities used by the baseline framework to process images and text jointly. +- **rationale:** The baseline framework operationally requires an MLLM. The paper tests several, but Qwen2.5-VL is explicitly highlighted as a primary open-source MLLM used in the strongest baseline configurations, making it a clean canonical grounding for this capability. +- **evidence_span:** “Qwen2.5-7B-Instruct [Yang et al., 2024] ( Qwen ) acts as the LLM and Qwen2.5-VL-7B-Instruct [Bai et al., 2025] ( Qwen-VL ) serves as the MLLM” \ No newline at end of file diff --git a/replay_traces/2505.17978/two_pass_outputs/20260726T204323Z/pass_1_reasoning.output.candidate_3.md b/replay_traces/2505.17978/two_pass_outputs/20260726T204323Z/pass_1_reasoning.output.candidate_3.md new file mode 100644 index 0000000000000000000000000000000000000000..bb49023a1a590ed18904e5d940e348a31038e16f --- /dev/null +++ b/replay_traces/2505.17978/two_pass_outputs/20260726T204323Z/pass_1_reasoning.output.candidate_3.md @@ -0,0 +1,402 @@ +# 1. Cluster Evidence + +Based on the downstream cluster evidence, the target paper introduces four distinct downstream-used contributions: +- **A dataset** for multimodal claim verification (reused for evaluation and extended for new dataset creation in C1, C2). +- **An evaluation method** for multimodal evidence retrieval (adapted for a novel setting in C3). +- **A temporally constrained retrieval methodology** (adopted for dataset creation in C4). +- **A baseline verification framework** (extended for multimodal claim verification in C5). + +Because these four artifacts have distinct downstream roles, distinct usage clusters, and require different ingredient decompositions, the bundled paper contribution must be split into four atomic claims. + +--- + +# 2. Claim Split Decision + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C1, C2 + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Metric +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C3 + +### Claim C3 +- **claim_id:** C3 +- **artifact_type:** Method +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C4 + +### Claim C4 +- **claim_id:** C4 +- **artifact_type:** Method +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C5 + +### Why this split is correct +The paper contributes four separate artifacts with distinct downstream roles: +- The **AVerImaTeC dataset** is reused as a training and evaluation resource. +- The **reference-based evaluation method** is reused as a standalone metric for assessing multimodal evidence retrieval. +- The **temporally constrained retrieval methodology** is reused as a data-collection protocol to prevent temporal leakage. +- The **baseline framework** is reused as a foundation for building new multimodal claim verification systems. +They should therefore be annotated separately. + +--- + +# 3. Rewritten Claims + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **rewritten_claim:** **Dataset: A real-world image-text claim verification dataset annotated with question-answer reasoning paths and web-based evidence, enabling the training and evaluation of multimodal fact-checking systems.** +- **why_this_is_atomic:** This claim isolates the dataset resource itself, separate from the evaluation metrics, data collection methodologies, and baseline models introduced alongside it. +- **decision:** YES_SUFFICIENT + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Metric +- **rewritten_claim:** **Metric: A reference-based evaluation method for multimodal evidence retrieval that separately assesses textual and visual components using LLMs, enabling automated assessment of retrieved evidence quality.** +- **why_this_is_atomic:** This claim isolates the specific evaluation protocol used to score multimodal evidence, which downstream work reuses independently of the dataset. +- **decision:** YES_SUFFICIENT + +### Claim C3 +- **claim_id:** C3 +- **artifact_type:** Method +- **rewritten_claim:** **Method: A temporally constrained evidence retrieval methodology that restricts search results to dates prior to the claim, enabling the construction of fact-checking datasets without temporal leakage.** +- **why_this_is_atomic:** This claim isolates the specific data-collection protocol designed to prevent temporal leakage, which downstream work adopts for its own dataset creation. +- **decision:** YES_SUFFICIENT + +### Claim C4 +- **claim_id:** C4 +- **artifact_type:** Method +- **rewritten_claim:** **Method: A baseline framework for image-text claim verification that decomposes reasoning into question generation and tool-augmented answer generation, enabling automated multimodal fact-checking.** +- **why_this_is_atomic:** This claim isolates the baseline verification architecture, which downstream work extends to build new systems. +- **decision:** YES_SUFFICIENT + +--- + +# 4. Minimal Ingredient Set Per Claim + +## Claim C1: Dataset + +### Ingredient C1.I1 +- **ingredient_id:** C1.I1 +- **ingredient:** Source fact-checking articles containing real-world multimodal claims +- **why_structurally_necessary:** The dataset is built by extracting and normalizing real-world image-text claims from existing fact-checking articles. Without this source substrate, the dataset would not exist as a collection of real-world claims. +- **why_not_lower_level_substeps:** This should remain a composite data-source ingredient rather than being split into individual upstream datasets (AVeriTeC, AMMeBA, ClaimReview). +- **why_not_adjacent_implementation_details:** The structural dependency is the raw multimodal claim substrate, not the specific filtering scripts used to extract them. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web; AMMeBa: A Large-Scale Survey and Dataset of Media-Based Misinformation In-The-Wild +- **role:** DATA_SOURCE +- **contribution:** Supplies the raw fact-checking articles and multimodal claims that are normalized and annotated to form the dataset. +- **rationale:** The dataset's source material is inherently composite, drawing from multiple prior fact-checking repositories and APIs. No single prior study represents the entire source mixture, making `NONE` the correct canonical grounding, with the primary contributing datasets listed as additional studies. +- **evidence_span:** “we incorporated partially annotated articles from three sources: 1) filtered articles from AVeriTec... 2) articles verifying image-related claims from AMMEBA... and 3) all true claims from ClaimReview” + +### Ingredient C1.I2 +- **ingredient_id:** C1.I2 +- **ingredient:** Question-answering representation for fact-checking rationales +- **why_structurally_necessary:** The defining property of the dataset's annotations is that the verification rationale is decomposed into a sequence of QA pairs. Without this conceptual framing, the dataset would lack its structured reasoning paths. +- **why_not_lower_level_substeps:** This should remain a conceptual framing ingredient rather than being split into specific question types or answer formats. +- **why_not_adjacent_implementation_details:** The structural dependency is the QA-based reasoning representation itself, not the specific UI used by annotators to write the questions. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Complex Claim Verification with Evidence Retrieved in the Wild +- **additional groundings if any:** The Art of SOCRATIC QUESTIONING: Recursive Thinking with Large Language Models; Fact-Checking Complex Claims with Program-Guided Reasoning +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the conceptual framework of using question-answering pairs to represent the decomposed reasoning steps required for claim verification. +- **rationale:** The paper explicitly adopts QA as the reasoning representation for fact-checking, citing prior works that established this paradigm. Chen et al. (CLAIMDECOMP) is a clean representative study for this QA-based verification framing. +- **evidence_span:** “To capture the rationale in claim verification, a complex reasoning task, various reasoning representations have been explored, including... QA [Chen et al., 2024, Qi et al., 2023, Pan et al., 2023a].” + +### Ingredient C1.I3 +- **ingredient_id:** C1.I3 +- **ingredient:** Multi-stage human annotation protocol for evidence retrieval and sufficiency checking +- **why_structurally_necessary:** The dataset's quality and reliability depend on a rigorous annotation pipeline that includes claim normalization, QA generation, and two rounds of evidence sufficiency checks. Without this protocol, the dataset would not exist in its validated form. +- **why_not_lower_level_substeps:** This should remain one protocol-level ingredient rather than being split into individual annotation phases. +- **why_not_adjacent_implementation_details:** The structural dependency is the benchmark-wide human-supervision workflow, not the specific annotator compensation details. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the manual dataset-construction process ensuring that all claims are normalized, annotated with evidence, and verified for sufficiency. +- **rationale:** Constructing this dataset requires a complex, multi-stage human annotation workflow. Because this specific five-phase pipeline is realized in the target paper (though extending prior work), `NONE` is the correct grounding. +- **evidence_span:** “The five-phase annotation pipeline is illustrated in Figure 2, extending the annotation process proposed in [Schlichtkrull et al., 2023] to the domain of image-text claims.” + +## Claim C2: Metric + +### Ingredient C2.I1 +- **ingredient_id:** C2.I1 +- **ingredient:** Reference-based evaluation framework for textual evidence using LLMs +- **why_structurally_necessary:** The proposed metric is an extension of an existing LLM-based reference evaluation framework. Without this base framework, the multimodal metric would not exist. +- **why_not_lower_level_substeps:** This should remain a framework-level ingredient rather than being split into specific prompt templates or scoring thresholds. +- **why_not_adjacent_implementation_details:** The structural dependency is the reference-based evaluation paradigm itself, not the specific adversarial robustness checks performed to validate it. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking +- **additional groundings if any:** none +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the base reference-based evaluation methodology that the paper extends to handle multimodal evidence. +- **rationale:** The paper explicitly builds its evaluation method upon the Ev2R framework, which demonstrated that LLM-based reference evaluation aligns well with human assessments for fact-checking evidence. +- **evidence_span:** “Recent research [Akhtar et al., 2024] showed that reference-based evaluation of evidence with large language models (LLMs) aligns best with human assessments. We extend this framework to a multimodal setting” + +### Ingredient C2.I2 +- **ingredient_id:** C2.I2 +- **ingredient:** Large multimodal model for scoring textual and visual alignments +- **why_structurally_necessary:** To evaluate multimodal evidence, the metric requires a model capable of processing both text and images to compare retrieved evidence against ground-truth references. +- **why_not_lower_level_substeps:** This should remain at the level of the MLLM scorer rather than being split into separate text and vision encoders. +- **why_not_adjacent_implementation_details:** The specific choice of Gemini 2.0 Flash is an implementation detail; the structural dependency is the use of a capable MLLM as the scoring engine. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Acts as the automated evaluator that scores the alignment between retrieved multimodal evidence and human-annotated references. +- **rationale:** The metric operationally depends on an MLLM to perform the scoring. The paper uses Gemini 2.0 Flash, which is a commercial API/model rather than a canonical prior academic study defining the ingredient, making `NONE` the most appropriate grounding. +- **evidence_span:** “We exploit Gemini-2.0-Flash [DeepMind, 2024] as the scoring model for both steps in the reference-based evaluation” + +## Claim C3: Method (Retrieval Methodology) + +### Ingredient C3.I1 +- **ingredient_id:** C3.I1 +- **ingredient:** Conceptual framing of temporal leakage in fact-checking datasets +- **why_structurally_necessary:** The entire purpose of the temporally constrained retrieval methodology is to solve the problem of temporal leakage (where models use future information to verify past claims). Without this conceptual framing, the methodology has no purpose. +- **why_not_lower_level_substeps:** This is a high-level conceptual ingredient and cannot be decomposed further. +- **why_not_adjacent_implementation_details:** N/A +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Missing Counter-Evidence Renders NLP Fact-Checking Unrealistic for Misinformation +- **additional groundings if any:** none +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Identifies the problem of temporal leakage in fact-checking, motivating the need for date-restricted evidence retrieval. +- **rationale:** The paper explicitly cites Glockner et al. (2022) as the motivation for restricting search results to prevent temporal leakage, making it the clean canonical grounding for this concept. +- **evidence_span:** “All retrieved pages from the interface are restricted to dates prior to the claim date to prevent temporal leakage [Glockner et al., 2022].” + +### Ingredient C3.I2 +- **ingredient_id:** C3.I2 +- **ingredient:** Date-restricted web search and scraping tooling +- **why_structurally_necessary:** To actually implement the temporally constrained methodology, the system requires tools capable of filtering web search results and scraped pages by publication date. +- **why_not_lower_level_substeps:** This should remain a single tooling ingredient rather than being split into the Google Search API and the htmldate Python package separately. +- **why_not_adjacent_implementation_details:** The specific Python package used is an implementation detail; the structural dependency is the capability to filter evidence by date. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the operational mechanism to enforce the temporal constraints during evidence retrieval. +- **rationale:** The methodology relies on standard APIs and Python packages (like htmldate) to enforce date restrictions. These are general software tools rather than canonical prior research studies, so `NONE` is the correct grounding. +- **evidence_span:** “we use a post-hoc method by leveraging the Python package htmldate.fine_date to filter out pages published before claim dates.” + +## Claim C4: Method (Baseline Framework) + +### Ingredient C4.I1 +- **ingredient_id:** C4.I1 +- **ingredient:** Question-guided multi-hop reasoning framework for fact-checking +- **why_structurally_necessary:** The baseline framework operates by generating questions to guide the evidence retrieval process. Without this question-guided architecture, the baseline would not function in its claimed form. +- **why_not_lower_level_substeps:** This should remain a framework-level ingredient rather than being split into parallel, dynamic, and hybrid generation strategies. +- **why_not_adjacent_implementation_details:** The specific prompts used for question generation are implementation details; the structural dependency is the question-guided reasoning framework itself. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web +- **additional groundings if any:** none +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the foundational architecture of using generated questions to guide the fact-checking process. +- **rationale:** The paper explicitly builds its question generation approach upon the framework established by AVeriTeC, making it the cleanest canonical grounding. +- **evidence_span:** “A straightforward approach is to generate all verification questions at once, given an image-text claim, as in prior work [Schlichtkrull et al., 2023].” + +### Ingredient C4.I2 +- **ingredient_id:** C4.I2 +- **ingredient:** Tool-augmented LLM framework for answering questions +- **why_structurally_necessary:** The baseline answers generated questions by routing them to specialized tools (RIS, WST, WSI, VQA). Without this tool-augmented architecture, the system could not retrieve external evidence to answer the questions. +- **why_not_lower_level_substeps:** This should remain a single framework ingredient rather than being split into individual tools (RIS, WST, etc.) as separate ingredients. +- **why_not_adjacent_implementation_details:** The specific tool selection prompts are implementation details; the structural dependency is the tool-augmented generation paradigm. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models +- **additional groundings if any:** DEFAME: Dynamic Evidence-based FAct-checking with Multimodal Experts +- **role:** CORE_METHOD +- **contribution:** Provides the architecture for integrating specialized external tools into the LLM's answer generation process. +- **rationale:** The paper explicitly adopts the tool-augmented LLM paradigm to handle multimodal evidence retrieval, citing Visual ChatGPT and DEFAME as inspirations. Visual ChatGPT is a clean canonical representative for this broader paradigm. +- **evidence_span:** “Inspired by recent research on tool usage [Wu et al., 2023, Cao et al., 2024, Braun et al., 2024], we integrate a set of specialized tools into the answer generation module” + +### Ingredient C4.I3 +- **ingredient_id:** C4.I3 +- **ingredient:** Multimodal large language models for question generation and visual question answering +- **why_structurally_necessary:** The baseline framework requires models capable of processing both the textual claim and the associated images to generate relevant questions and perform VQA. +- **why_not_lower_level_substeps:** This should remain at the level of MLLMs as a class rather than being split into specific model variants. +- **why_not_adjacent_implementation_details:** The specific choice between Qwen-VL, Gemini, or LLaVA is an implementation detail; the structural dependency is the use of MLLMs to power the framework's core components. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** CORE_METHOD +- **contribution:** Powers the internal reasoning, question generation, and visual analysis capabilities of the baseline framework. +- **rationale:** The framework is designed to be model-agnostic and tests several different MLLMs (Gemini, Qwen-VL, Gemma, LLaVA). Because the structural dependency is on the class of MLLMs rather than one specific canonical model, `NONE` is the correct grounding. +- **evidence_span:** “All three strategies employ an MLLM for question generation, leveraging the model's internal decomposition capability.” + +--- + +# 5. Excluded Tempting Non-Ingredients + +- **Specific LLM/MLLM variants (e.g., Gemini, Qwen, LLaVA)** + - Excluded because the structural dependency for both the metric and the baseline framework is the class of Multimodal Large Language Models, not any single specific instance tested in the paper. +- **BM25 ranking** + - Excluded because it is a local implementation detail used for text retrieval within the baseline framework, not a high-level structural ingredient defining the artifact. +- **Specific fact-checking strategies (e.g., reverse image search, geolocation)** + - Excluded because these are either metadata labels within the dataset or specific tools within the broader tool-augmented framework, rather than standalone structural ingredients. + +--- + +# 6. Candidate Grounding Decisions + +### Claim C1: Dataset +- **C1.I1 (Source articles):** `NONE` is chosen because the source material is a composite of multiple prior datasets (AVeriTeC, AMMeBA, ClaimReview). +- **C1.I2 (QA representation):** *Complex Claim Verification with Evidence Retrieved in the Wild* is chosen as the canonical grounding because it is explicitly cited as a foundational work for using QA representations in fact-checking. +- **C1.I3 (Annotation protocol):** `NONE` is chosen because the specific five-phase annotation pipeline is constructed and executed within the target paper itself. + +### Claim C2: Metric +- **C2.I1 (Ev2R framework):** *Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking* is chosen as the canonical grounding because the paper explicitly extends this specific framework for its multimodal metric. +- **C2.I2 (MLLM scorer):** `NONE` is chosen because the scoring relies on a commercial API (Gemini 2.0 Flash) rather than a canonical prior academic study defining the ingredient. + +### Claim C3: Method (Retrieval Methodology) +- **C3.I1 (Temporal leakage concept):** *Missing Counter-Evidence Renders NLP Fact-Checking Unrealistic for Misinformation* is chosen as the canonical grounding because it is explicitly cited as the motivation for the methodology. +- **C3.I2 (Date tooling):** `NONE` is chosen because the implementation relies on standard software tools and APIs rather than prior research studies. + +### Claim C4: Method (Baseline Framework) +- **C4.I1 (Question-guided framework):** *AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web* is chosen as the canonical grounding because the paper explicitly builds its question generation approach upon it. +- **C4.I2 (Tool-augmented framework):** *Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models* is chosen as the canonical representative for the tool-augmented LLM paradigm adopted by the baseline. +- **C4.I3 (MLLMs):** `NONE` is chosen because the framework is model-agnostic and relies on the general capability of MLLMs rather than one specific prior model. + +--- + +# 7. Final Grounding Decisions + +## Claim C1 +- **C1.I1** → NONE → DATA_SOURCE + - additional: AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web; AMMeBa: A Large-Scale Survey and Dataset of Media-Based Misinformation In-The-Wild +- **C1.I2** → Complex Claim Verification with Evidence Retrieved in the Wild → CONCEPTUAL_FRAMEWORK + - additional: The Art of SOCRATIC QUESTIONING: Recursive Thinking with Large Language Models; Fact-Checking Complex Claims with Program-Guided Reasoning +- **C1.I3** → NONE → EVALUATION_PROTOCOL + +## Claim C2 +- **C2.I1** → Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking → CONCEPTUAL_FRAMEWORK +- **C2.I2** → NONE → IMPLEMENTATION_TOOLING + +## Claim C3 +- **C3.I1** → Missing Counter-Evidence Renders NLP Fact-Checking Unrealistic for Misinformation → CONCEPTUAL_FRAMEWORK +- **C3.I2** → NONE → IMPLEMENTATION_TOOLING + +## Claim C4 +- **C4.I1** → AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web → CONCEPTUAL_FRAMEWORK +- **C4.I2** → Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models → CORE_METHOD + - additional: DEFAME: Dynamic Evidence-based FAct-checking with Multimodal Experts +- **C4.I3** → NONE → CORE_METHOD + +--- + +# 8. Final Rationales And Evidence + +## Claim C1: Dataset + +### C1.I1 +- **ingredient:** Source fact-checking articles containing real-world multimodal claims +- **canonical study:** NONE +- **additional studies:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web; AMMeBa: A Large-Scale Survey and Dataset of Media-Based Misinformation In-The-Wild +- **role:** DATA_SOURCE +- **contribution:** Supplies the raw fact-checking articles and multimodal claims that are normalized and annotated to form the dataset. +- **rationale:** The dataset's source material is inherently composite, drawing from multiple prior fact-checking repositories and APIs. No single prior study represents the entire source mixture, making `NONE` the correct canonical grounding, with the primary contributing datasets listed as additional studies. +- **evidence_span:** “we incorporated partially annotated articles from three sources: 1) filtered articles from AVeriTec... 2) articles verifying image-related claims from AMMEBA... and 3) all true claims from ClaimReview” + +### C1.I2 +- **ingredient:** Question-answering representation for fact-checking rationales +- **canonical study:** Complex Claim Verification with Evidence Retrieved in the Wild +- **additional studies:** The Art of SOCRATIC QUESTIONING: Recursive Thinking with Large Language Models; Fact-Checking Complex Claims with Program-Guided Reasoning +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the conceptual framework of using question-answering pairs to represent the decomposed reasoning steps required for claim verification. +- **rationale:** The paper explicitly adopts QA as the reasoning representation for fact-checking, citing prior works that established this paradigm. Chen et al. (CLAIMDECOMP) is a clean representative study for this QA-based verification framing. +- **evidence_span:** “To capture the rationale in claim verification, a complex reasoning task, various reasoning representations have been explored, including... QA [Chen et al., 2024, Qi et al., 2023, Pan et al., 2023a].” + +### C1.I3 +- **ingredient:** Multi-stage human annotation protocol for evidence retrieval and sufficiency checking +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the manual dataset-construction process ensuring that all claims are normalized, annotated with evidence, and verified for sufficiency. +- **rationale:** Constructing this dataset requires a complex, multi-stage human annotation workflow. Because this specific five-phase pipeline is realized in the target paper (though extending prior work), `NONE` is the correct grounding. +- **evidence_span:** “The five-phase annotation pipeline is illustrated in Figure 2, extending the annotation process proposed in [Schlichtkrull et al., 2023] to the domain of image-text claims.” + +## Claim C2: Metric + +### C2.I1 +- **ingredient:** Reference-based evaluation framework for textual evidence using LLMs +- **canonical study:** Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the base reference-based evaluation methodology that the paper extends to handle multimodal evidence. +- **rationale:** The paper explicitly builds its evaluation method upon the Ev2R framework, which demonstrated that LLM-based reference evaluation aligns well with human assessments for fact-checking evidence. +- **evidence_span:** “Recent research [Akhtar et al., 2024] showed that reference-based evaluation of evidence with large language models (LLMs) aligns best with human assessments. We extend this framework to a multimodal setting” + +### C2.I2 +- **ingredient:** Large multimodal model for scoring textual and visual alignments +- **canonical study:** NONE +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Acts as the automated evaluator that scores the alignment between retrieved multimodal evidence and human-annotated references. +- **rationale:** The metric operationally depends on an MLLM to perform the scoring. The paper uses Gemini 2.0 Flash, which is a commercial API/model rather than a canonical prior academic study defining the ingredient, making `NONE` the most appropriate grounding. +- **evidence_span:** “We exploit Gemini-2.0-Flash [DeepMind, 2024] as the scoring model for both steps in the reference-based evaluation” + +## Claim C3: Method (Retrieval Methodology) + +### C3.I1 +- **ingredient:** Conceptual framing of temporal leakage in fact-checking datasets +- **canonical study:** Missing Counter-Evidence Renders NLP Fact-Checking Unrealistic for Misinformation +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Identifies the problem of temporal leakage in fact-checking, motivating the need for date-restricted evidence retrieval. +- **rationale:** The paper explicitly cites Glockner et al. (2022) as the motivation for restricting search results to prevent temporal leakage, making it the clean canonical grounding for this concept. +- **evidence_span:** “All retrieved pages from the interface are restricted to dates prior to the claim date to prevent temporal leakage [Glockner et al., 2022].” + +### C3.I2 +- **ingredient:** Date-restricted web search and scraping tooling +- **canonical study:** NONE +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the operational mechanism to enforce the temporal constraints during evidence retrieval. +- **rationale:** The methodology relies on standard APIs and Python packages (like htmldate) to enforce date restrictions. These are general software tools rather than canonical prior research studies, so `NONE` is the correct grounding. +- **evidence_span:** “we use a post-hoc method by leveraging the Python package htmldate.fine_date to filter out pages published before claim dates.” + +## Claim C4: Method (Baseline Framework) + +### C4.I1 +- **ingredient:** Question-guided multi-hop reasoning framework for fact-checking +- **canonical study:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the foundational architecture of using generated questions to guide the fact-checking process. +- **rationale:** The paper explicitly builds its question generation approach upon the framework established by AVeriTeC, making it the cleanest canonical grounding. +- **evidence_span:** “A straightforward approach is to generate all verification questions at once, given an image-text claim, as in prior work [Schlichtkrull et al., 2023].” + +### C4.I2 +- **ingredient:** Tool-augmented LLM framework for answering questions +- **canonical study:** Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models +- **additional studies:** DEFAME: Dynamic Evidence-based FAct-checking with Multimodal Experts +- **role:** CORE_METHOD +- **contribution:** Provides the architecture for integrating specialized external tools into the LLM's answer generation process. +- **rationale:** The paper explicitly adopts the tool-augmented LLM paradigm to handle multimodal evidence retrieval, citing Visual ChatGPT and DEFAME as inspirations. Visual ChatGPT is a clean canonical representative for this broader paradigm. +- **evidence_span:** “Inspired by recent research on tool usage [Wu et al., 2023, Cao et al., 2024, Braun et al., 2024], we integrate a set of specialized tools into the answer generation module” + +### C4.I3 +- **ingredient:** Multimodal large language models for question generation and visual question answering +- **canonical study:** NONE +- **role:** CORE_METHOD +- **contribution:** Powers the internal reasoning, question generation, and visual analysis capabilities of the baseline framework. +- **rationale:** The framework is designed to be model-agnostic and tests several different MLLMs (Gemini, Qwen-VL, Gemma, LLaVA). Because the structural dependency is on the class of MLLMs rather than one specific canonical model, `NONE` is the correct grounding. +- **evidence_span:** “All three strategies employ an MLLM for question generation, leveraging the model's internal decomposition capability.” \ No newline at end of file diff --git a/replay_traces/2505.17978/two_pass_outputs/20260726T204323Z/pass_1_reasoning.prompt.txt b/replay_traces/2505.17978/two_pass_outputs/20260726T204323Z/pass_1_reasoning.prompt.txt new file mode 100644 index 0000000000000000000000000000000000000000..ad0ed0d830bbaeb3037b50a7baab12c9bee8ae8f --- /dev/null +++ b/replay_traces/2505.17978/two_pass_outputs/20260726T204323Z/pass_1_reasoning.prompt.txt @@ -0,0 +1,3748 @@ + +Task: Produce the reasoning memo for cluster-first claim derivation and enabling-ingredient annotation. + +Important: +- Think cluster-first. +- Do not start from a paper-level discovery summary. +- Return the smallest structurally sufficient ingredient set. + +Required procedure: +1. Identify downstream-used claims from the cluster evidence. +2. Split claims only when downstream-used contributions are distinct. +3. Rewrite each claim at the abstraction level matching the intended ingredients. +4. For each claim, first list broad candidate ingredients and possible grounding studies. +5. Collapse those candidates into the minimal structurally sufficient ingredient set. +6. Explicitly exclude tempting non-ingredients. E.g. + - one-task tools + - local preprocessing + - balancing tricks + - token/language-tag tricks + - helper models used for one small step +7. Only then decide canonical vs additional vs NONE. + + +Benchmark/dataset claims: include only what is needed to build the artifact itself. Favor the benchmark framing, core conceputal idea, source tasks/data, and benchmark-wide annotation or verification protocol. Exclude anything the benchmark could still exist without. +Method/resource claims: include only what defines the artifact: core architecture, core objective, core conceputal idea, essential training data, or essential initialization. Exclude non-essential implementation details. +Finding claims: include only what is needed to observe and measure the empirical result: the evaluation protocol, core conceputal idea, studied model/artifact type, task substrate, and measurement/extraction procedure. Exclude recommendations, fixes, and general background unless they are operationally necessary. + +For each final ingredient, provide: +- ingredient +- why it is structurally necessary +- why tempting alternatives are excluded +- canonical grounding decision +- additional groundings if any +- role +- contribution +- rationale +- evidence:one verbatim quote + +Return a markdown reasoning memo only. +-------------------------------------------------- +REFERENCE EXAMPLE 1 +-------------------------------------------------- + +# EXAMPLE: Pick-a-Pic / PickScore + +## 1. Cluster Evidence + +**The PickScore metric, a CLIP-based scoring function that predicts human preferences to evaluate and rank text-to-image generation models.** + +The paper introduces two downstream-used contributions: PickScore and the Pick-a-Pic dataset. Downstream evidence shows separate reuse: + +- PickScore is reused as an evaluation metric / reward model. +- Pick-a-Pic is reused as a dataset for training and preference tuning. + +So the bundled contribution should be split into two atomic claims. + +--- + +## 2. Claim Split Decision + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Tool +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C1 + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Dataset +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C2 + +### Why this split is correct + +The two artifacts have distinct downstream roles and distinct ingredient decompositions: + +- **PickScore** is reused as a scoring / evaluation / reward model. +- **Pick-a-Pic** is reused as a preference dataset for training and benchmarking. + +They should therefore be annotated separately. + +--- + +## 3. Rewritten Claims + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Tool +- **rewritten_claim:** **"Tool: A human-preference prompt-conditioned image scorer, enabling automatic ranking of text-to-image model outputs** +- **why_this_is_atomic:** This claim isolates the scorer itself as the reused artifact. It does not bundle the dataset resource used to train it. +- **optional cluster_id:** C1 +- **decision:** YES_SUFFICIENT + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Dataset +- **rewritten_claim:** **Dataset: An open prompt-conditioned human-preference dataset for generated images, enabling training and evaluation of preference-aligned text-to-image systems.** +- **why_this_is_atomic:** This claim isolates the dataset as the released reusable resource, separate from the scoring model trained on it. +- **optional cluster_id:** C2 +- **decision:** YES_SUFFICIENT + +--- + +## 4. Minimal Ingredient Set Per Claim + +## Claim C1: PickScore + +### Ingredient C1.I1 +- **ingredient_id:** C1.I1 +- **ingredient:** Pre-trained vision language model for prompt-conditioned image scoring +- **why_structurally_necessary:** PickScore is explicitly a CLIP-based scoring function. Without a pretrained joint text-image model, the scorer would not exist in its claimed form. +- **why_not_lower_level_substeps:** This should stay at the level of pretrained vision-language initialization, not be decomposed into separate text encoder, image encoder, or projection details. +- **why_not_adjacent_implementation_details:** Specific OpenCLIP variants are narrower implementation choices rather than the core structural dependency. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Learning Transferable Visual Models From Natural Language Supervision +- **additional groundings if any:** none +- **role:** MODEL_INITIALIZATION +- **contribution:** Provides the pre-trained text and image representations that are fine-tuned into a human preference score. +- **rationale:** To score a generated image relative to a prompt, the function needs a pretrained joint text–image representation. CLIP is the cleanest representative study for such capability and is the direct architectural basis of PickScore. +- **evidence_span:** “PickScore follows the architecture of CLIP [12]” + +### Ingredient C1.I2 +- **ingredient_id:** C1.I2 +- **ingredient:** Large-scale dataset of human-preference judgements over text-to-image outputs +- **why_structurally_necessary:** The scorer must learn which of two generated images humans prefer for a given prompt. Without this training signal, it would not become a human-preference metric. +- **why_not_lower_level_substeps:** This should remain a single training-data ingredient rather than being split into prompts, pairs, ties, logging, or collection mechanics. +- **why_not_adjacent_implementation_details:** Dataset construction details belong to the dataset claim, not the scorer’s minimal training-data dependency. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** TRAINING_DATA +- **contribution:** Provides the pairwise user choices needed to fit the scorer. +- **rationale:** The model needs direct supervision about which of two generated images users prefer for a given prompt. This operative dataset is released in the target paper, so the right grounding is `NONE`. +- **evidence_span:** “We finetune CLIP-H [7] using our framework8 on the Pick-a-Pic training set.” + +### Ingredient C1.I3 +- **ingredient_id:** C1.I3 +- **ingredient:** Preference-learning objective to model the reward +- **why_structurally_necessary:** PickScore is trained to assign higher scores to preferred images. Without a preference-learning objective, it would not become a ranking metric that evaluates outputs by predicted human choice. +- **why_not_lower_level_substeps:** This should remain one core-method ingredient rather than being decomposed into loss cases, tie handling, or optimization details. +- **why_not_adjacent_implementation_details:** Training schedule and optimizer are implementation details; the structural dependency is the reward-model-style objective itself. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Training language models to follow instructions with human feedback +- **additional groundings if any:** none +- **role:** CORE_METHOD +- **contribution:** Learns scores whose differences reflect human preference between candidate images for the same prompt. +- **rationale:** Without that preference-learning step, the system would not become a ranking metric that can evaluate outputs. The paper explicitly grounds this in a variant of InstructGPT’s reward model objective. +- **evidence_span:** “We train the PickScore scoring function over Pick-a-Pic by combining a CLIP-style model with a variant of In-structGPT’s reward model objective” + +--- + +## Claim C2: Pick-a-Pic + +### Ingredient C2.I1 +- **ingredient_id:** C2.I1 +- **ingredient:** Framework for pairwise human preference collection using prompt-conditioned image comparisons +- **why_structurally_necessary:** The defining property of the dataset is that it records human preferences over generated image pairs for prompts. Without a collection framework that elicits those comparisons, the dataset would not exist in its claimed form. +- **why_not_lower_level_substeps:** This should remain one collection-framework ingredient rather than being split into web app interface details, session flow, or tie-button mechanics. +- **why_not_adjacent_implementation_details:** The structural point is the pairwise human-preference collection framework, not the exact UI implementation. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the framework to collect human preferences on a large scale. +- **rationale:** A framework is necessary to collect human preference choices at scale. The authors implement a web app that lets users generate images and specify preferences, and that workflow is a defining part of the dataset rather than a prior reusable study. +- **evidence_span:** “To address this issue, we create a web app that enables text-to-image users to generate images and specify their preferences.” + +### Ingredient C2.I2 +- **ingredient_id:** C2.I2 +- **ingredient:** Diffusion models for image generation from text +- **why_structurally_necessary:** The dataset consists of preferences over generated images. Without text-to-image generators, there would be no candidate images to compare and annotate. +- **why_not_lower_level_substeps:** This should remain at the level of text-to-image diffusion generation, not be split into individual backbone variants or guidance settings. +- **why_not_adjacent_implementation_details:** The exact backbone mix is less important than the general generation capability used to produce the images. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** High-Resolution Image Synthesis with Latent Diffusion Models +- **additional groundings if any:** none +- **role:** CORE_METHOD +- **contribution:** Provides the diffusion models needed to generate the images in the dataset. +- **rationale:** The dataset requires text-to-image generation backbones to convert user prompts into candidate images for preference annotation. Latent Diffusion Models is the cleanest representative prior study for that capability. +- **evidence_span:** “The images in the dataset were generated by employing multiple backbone models, namely, Stable Diffusion 2.1, Dreamlike Photoreal 2.0, and Stable Diffusion XL variants” + +### Ingredient C2.I3 +- **ingredient_id:** C2.I3 +- **ingredient:** Quality-control and preprocessing protocol for reliable open preference-data collection +- **why_structurally_necessary:** The dataset is released as an open, reusable human-preference resource. Without moderation, filtering, and preprocessing, it would lose one of its defining properties: being a reliable preference dataset rather than a raw interaction log. +- **why_not_lower_level_substeps:** This should remain one protocol-level ingredient rather than being split into NSFW phrase lists, account filtering, banning, or split construction separately. +- **why_not_adjacent_implementation_details:** These local measures are all parts of the same broader quality-control workflow and should not be promoted to standalone ingredients. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the moderation, filtering, and preprocessing workflow that makes the open preference dataset reliable and usable. +- **rationale:** Constructing an open preference dataset requires more than collecting raw comparisons. The paper includes moderation, harmful-content filtering, and preprocessing decisions that are structurally important to the quality of the released resource. This is paper-specific and therefore maps cleanly to `NONE`. +- **evidence_span:** “we closely monitor user activity logs and take action to ban users” + +--- + +## 5. Excluded Tempting Non-Ingredients + +### Claim C1 exclusions +- **Pick-a-Pic web app** + - Excluded because for the scorer claim, the structural dependency is the availability of human preference training data, not the full data-collection workflow. + +- **Preference-based evaluation setup for ranking candidate images by predicted human choice** + - Excluded because this is already captured by the rewritten claim and by the combination of human preference training data plus the reward-model objective. Adding it would double count the same functional role. + +### Claim C2 exclusions +- **List of NSFW phrases** + - Excluded because it is only one local component of the broader quality-control and preprocessing protocol, not a standalone structural ingredient. + +- **Specific diffusion backbone variants** + - Excluded because the structural dependency is text-to-image diffusion generation as a class, not each particular model instance. + +- **PickScore** + - Excluded because the dataset is a prerequisite for the tool, not the reverse. + +--- + +## 6. Candidate Grounding Decisions + +## Claim C1: PickScore + +### Ingredient C1.I1 +- **Candidate:** Learning Transferable Visual Models From Natural Language Supervision +- **decision:** accepted_canonical +- **why:** Provides the pretrained joint text-image architecture used as the basis for prompt-conditioned scoring. + +### Ingredient C1.I2 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The operative human preference dataset is released in the target paper itself. + +### Ingredient C1.I3 +- **Candidate:** Training language models to follow instructions with human feedback +- **decision:** accepted_canonical +- **why:** Provides the reward-model-style objective explicitly adapted by the paper. + +## Claim C2: Pick-a-Pic + +### Ingredient C2.I1 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The pairwise collection framework is implemented in the paper and not cleanly inherited from one prior study. + +### Ingredient C2.I2 +- **Candidate:** High-Resolution Image Synthesis with Latent Diffusion Models +- **decision:** accepted_canonical +- **why:** Cleanest representative of the text-to-image diffusion generation capability used to create the candidate images. + +### Ingredient C2.I3 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The quality-control and preprocessing workflow is a paper-specific protocol rather than one canonical prior study. + +--- + +## 7. Final Grounding Decisions + +## Claim C1 +- **C1.I1** → Learning Transferable Visual Models From Natural Language Supervision → MODEL_INITIALIZATION +- **C1.I2** → NONE → TRAINING_DATA +- **C1.I3** → Training language models to follow instructions with human feedback → CORE_METHOD + +## Claim C2 +- **C2.I1** → NONE → EVALUATION_PROTOCOL +- **C2.I2** → High-Resolution Image Synthesis with Latent Diffusion Models → CORE_METHOD +- **C2.I3** → NONE → EVALUATION_PROTOCOL + +--- + +## 8. Final Rationales And Evidence + +## Claim C1: PickScore + +### C1.I1 +- **ingredient:** Pre-trained vision language model for prompt-conditioned image scoring +- **canonical study:** Learning Transferable Visual Models From Natural Language Supervision +- **role:** MODEL_INITIALIZATION +- **contribution:** Provides the pre-trained text and image representations that are fine-tuned into a human preference score. +- **rationale:** To score a generated image relative to a prompt, the function needs a pretrained joint text–image representation. CLIP is the cleanest representative study for such capability and is the direct architectural basis of PickScore. +- **evidence_span:** “PickScore follows the architecture of CLIP [12]; given a prompt x and an image y, our scoring function s computes a real number by representing x using a transformer text encoder and y using a transformer image encoder as d-dimensional vectors, and returning their inner product” + +### C1.I2 +- **ingredient:** Large-scale dataset of human-preference judgements over text-to-image outputs +- **canonical study:** NONE +- **role:** TRAINING_DATA +- **contribution:** Provides the pairwise user choices needed to fit the scorer. +- **rationale:** The model needs direct supervision about which of two generated images users prefer for a given prompt. This operative dataset is released in the target paper, so the correct grounding is `NONE`. +- **evidence_span:** “We finetune CLIP-H [7] using our framework8 on the Pick-a-Pic training set.” + +### C1.I3 +- **ingredient:** Preference-learning objective to model the reward +- **canonical study:** Training language models to follow instructions with human feedback +- **role:** CORE_METHOD +- **contribution:** Learns scores whose differences reflect human preference between candidate images for the same prompt. +- **rationale:** Without that preference-learning step, the system would not become a ranking metric that can evaluate outputs. The paper explicitly grounds this in a variant of InstructGPT’s reward model objective. +- **evidence_span:** “We train the PickScore scoring function over Pick-a-Pic by combining a CLIP-style model with a variant of In-structGPT’s reward model objective [10].” + +## Claim C2: Pick-a-Pic + +### C2.I1 +- **ingredient:** Framework for pairwise human preference collection using prompt-conditioned image comparisons +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the framework to collect human preferences on a large scale. +- **rationale:** A framework is necessary to collect human preference choices at a large scale. The authors implement a web app to collect this, and that workflow is a defining part of the dataset. +- **evidence_span:** “To address this issue, we create a web app that enables text-to-image users to generate images and specify their preferences.” + +### C2.I2 +- **ingredient:** Diffusion models for image generation from text +- **canonical study:** High-Resolution Image Synthesis with Latent Diffusion Models +- **role:** CORE_METHOD +- **contribution:** Provides the diffusion models needed to generate the images in the dataset. +- **rationale:** The dataset requires text-to-image generation backbones to convert prompts into candidate images for annotation. Latent Diffusion Models is the cleanest representative prior study for that capability. +- **evidence_span:** “The images in the dataset were generated by employing multiple backbone models, namely, Stable Diffusion 2.1, Dreamlike Photoreal 2.0 5 , and Stable Diffusion XL variants [13] while sampling different classifier-free guidance scale values [6].” + +### C2.I3 +- **ingredient:** Quality-control and preprocessing protocol for reliable open preference-data collection +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the moderation, filtering, and preprocessing workflow that makes the open preference dataset reliable and usable. +- **rationale:** Constructing an open preference dataset requires more than collecting raw interactions. The released resource depends on moderation, filtering, and preprocessing steps that preserve data quality and usability. Because this workflow is paper-specific, `NONE` is the right grounding. +- **evidence_span:** “we closely monitor user activity logs and take action to ban users” + +-------------------------------------------------- +REFERENCE EXAMPLE 2 +-------------------------------------------------- + +# EXAMPLE: Offline RL bottleneck finding + +## 1. Cluster Evidence + +**Finding that policy extraction and test-time generalization (rather than just value learning) are the main bottlenecks in offline RL, enabling improved algorithm design via better policy extraction objectives.** + +This original claim identifies a real downstream-used empirical contribution. It should remain a **single finding claim** rather than being split into separate findings about value learning, policy extraction, and test-time generalization, because the paper’s contribution is one integrated diagnosis of what limits offline RL performance in practice. + +--- + +## 2. Claim Split Decision + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Finding +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C1 + +### Why this split is correct + +The downstream reuse is of a single empirical conclusion: that offline RL performance is often limited more by **policy extraction and deployment-time generalization** than by value learning alone. The later algorithm-design implication is a consequence of that diagnosis, not a separate discovery claim. + +--- + +## 3. Rewritten Claims + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Finding +- **rewritten_claim:** **Finding: In offline reinforcement learning, policy extraction and test-time policy generalization often limit performance more than value learning alone, making policy-extraction objective choice a primary determinant of final returns.** +- **why_this_is_atomic:** This claim isolates one empirical result about the source of performance bottlenecks in offline RL. It does not split off the downstream design recommendation as a separate claim. +- **optional cluster_id:** C1 +- **decision:** YES_SUFFICIENT + +--- + +## 4. Minimal Ingredient Set Per Claim + +## Claim C1: Offline RL bottleneck finding + +### Ingredient C1.I1 +- **ingredient_id:** C1.I1 +- **ingredient:** Offline RL methods with decoupled value learning and policy extraction phases +- **why_structurally_necessary:** The finding depends on separating the quality of the learned critic from the quality of the extracted policy. Without a decoupled setup, the paper could not attribute performance differences to policy extraction rather than entangled actor-critic training. +- **why_not_lower_level_substeps:** This should remain one high-level ingredient about the decoupled offline RL setup, not be split into each individual algorithm separately. +- **why_not_adjacent_implementation_details:** Specific architectural or optimization choices inside one method are not the point; the structural dependency is the existence of a method family where value learning and policy extraction can be independently analyzed. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** Offline RL Without Off-Policy Evaluation; Offline Reinforcement Learning with Implicit Q-Learning; Contrastive Learning as Goal-Conditioned Reinforcement Learning +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the offline RL setup in which value learning is trained independently of policy extraction, enabling the paper’s bottleneck analysis. +- **rationale:** To show that policy extraction rather than value learning is often the limiting factor, the paper must analyze settings where the critic can be held fixed while different extraction procedures are compared. That dependence is structurally on a family of decoupled offline RL methods rather than on one single canonical paper, so `NONE` is the faithful canonical choice and the relevant prior methods are better represented as additional studies. +- **evidence_span:** “we focus on offline RL methods with decoupled value and policy training phases” + +### Ingredient C1.I2 +- **ingredient_id:** C1.I2 +- **ingredient:** Behavior-constrained policy-gradient extraction objectives for policies learned from fixed critics +- **why_structurally_necessary:** The finding is not only that some extraction methods underperform, but that behavior-constrained policy-gradient extraction often performs better and scales better than common alternatives under the same learned critic. +- **why_not_lower_level_substeps:** This should remain one extraction-family ingredient rather than being split into every equation or policy-update detail. +- **why_not_adjacent_implementation_details:** Hyperparameters and optimizer details are irrelevant here. The structural ingredient is the extraction objective family itself. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** A Minimalist Approach to Offline Reinforcement Learning +- **additional groundings if any:** Advantage-Weighted Regression: Simple and Scalable Off-Policy Reinforcement Learning; Offline Reinforcement Learning via High-Fidelity Generative Behavior Modeling +- **role:** CORE_METHOD +- **contribution:** Provides the representative behavior-regularized policy-gradient extraction objective that the paper identifies as stronger and more scalable than common alternatives. +- **rationale:** The paper’s empirical conclusion depends on contrasting extraction families and finding that DDPG+BC-style extraction often outperforms widely used value-weighted or sampling-based alternatives. DDPG+BC is the clearest representative study for that stronger extraction family, while AWR and SfBC-style methods serve as additional comparison groundings rather than the canonical representative. +- **evidence_span:** “switching to behavior-constrained policy gradient objectives (e.g., DDPG+BC) often leads to substantial improvements” + +### Ingredient C1.I3 +- **ingredient_id:** C1.I3 +- **ingredient:** Comparative protocol that evaluates extraction objectives across data regimes and deployment generalization settings +- **why_structurally_necessary:** The claim is a comparative empirical finding about bottlenecks and scaling behavior. It requires an analysis protocol that systematically varies data properties and observes how extraction choices affect final performance and generalization. +- **why_not_lower_level_substeps:** This should remain one protocol-level ingredient rather than being split into separate dataset-size sweeps, quality sweeps, and metric definitions. +- **why_not_adjacent_implementation_details:** Individual plots, metrics, or visualization choices are not separate ingredients; they are local parts of the broader comparative protocol. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the systematic empirical protocol that isolates policy extraction and generalization bottlenecks across environments and data regimes. +- **rationale:** A finding about what the main bottleneck is cannot be established from one aggregate return table alone. The paper needs a structured comparative protocol that varies data size, quality, and coverage while comparing extraction methods under shared critics. That protocol is built in the target paper itself, so `NONE` is the correct grounding. +- **evidence_span:** “We use data size, quality, and coverage as levers for systematically controlling their impacts” + +--- + +## 5. Excluded Tempting Non-Ingredients + +### Claim C1 exclusions +- **Offline RL setting with fixed logged data and no online interaction** + - Excluded because it is too broad and field-level. The more precise structural ingredient is the **decoupled offline RL setup** that lets the paper isolate value learning from extraction. + +- **Advantage-weighted regression as its own standalone ingredient** + - Excluded as a separate final ingredient because it is better represented as an **additional grounding** under the broader comparative extraction-objective ingredient, rather than being promoted to its own parallel ingredient. + +- **Standard offline RL benchmark datasets as a separate ingredient** + - Excluded because the finding depends more centrally on the paper’s broader comparative protocol than on one benchmark substrate alone. The datasets are part of that empirical protocol rather than a standalone higher-level ingredient. + +- **Test-time generalization as a separate conceptual ingredient** + - Excluded because it is already captured within the rewritten claim and within the comparative evaluation protocol ingredient. Keeping it separate would over-decompose the finding. + +--- + +## 6. Candidate Grounding Decisions + +## Claim C1: Offline RL bottleneck finding + +### Ingredient C1.I1 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The finding relies on a family of decoupled offline RL methods rather than one single representative study. +- **Candidate:** Offline RL Without Off-Policy Evaluation +- **decision:** accepted_additional +- **why:** Representative prior decoupled offline RL method explicitly listed by the paper. +- **Candidate:** Offline Reinforcement Learning with Implicit Q-Learning +- **decision:** accepted_additional +- **why:** Another representative decoupled method used to motivate the analysis setup. +- **Candidate:** Contrastive Learning as Goal-Conditioned Reinforcement Learning +- **decision:** accepted_additional +- **why:** Included as a representative decoupled method family member rather than the single canonical grounding. + +### Ingredient C1.I2 +- **Candidate:** A Minimalist Approach to Offline Reinforcement Learning +- **decision:** accepted_canonical +- **why:** Cleanest representative study for behavior-constrained policy-gradient extraction, which the paper identifies as the stronger extraction family. +- **Candidate:** Advantage-Weighted Regression: Simple and Scalable Off-Policy Reinforcement Learning +- **decision:** accepted_additional +- **why:** Necessary comparison grounding for the weaker value-weighted extraction family contrasted in the finding. +- **Candidate:** Offline Reinforcement Learning via High-Fidelity Generative Behavior Modeling +- **decision:** accepted_additional +- **why:** Relevant additional comparison study for alternative policy extraction objectives. + +### Ingredient C1.I3 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The comparative data-scaling and deployment-generalization protocol is constructed in the target paper rather than inherited from one prior study. + +--- + +## 7. Final Grounding Decisions + +## Claim C1 +- **C1.I1** → NONE → CONCEPTUAL_FRAMEWORK + - additional: Offline RL Without Off-Policy Evaluation; Offline Reinforcement Learning with Implicit Q-Learning; Contrastive Learning as Goal-Conditioned Reinforcement Learning +- **C1.I2** → A Minimalist Approach to Offline Reinforcement Learning → CORE_METHOD + - additional: Advantage-Weighted Regression: Simple and Scalable Off-Policy Reinforcement Learning; Offline Reinforcement Learning via High-Fidelity Generative Behavior Modeling +- **C1.I3** → NONE → EVALUATION_PROTOCOL + +--- + +## 8. Final Rationales And Evidence + +## Claim C1: Offline RL bottleneck finding + +### C1.I1 +- **ingredient:** Offline RL methods with decoupled value learning and policy extraction phases +- **canonical study:** NONE +- **additional studies:** Offline RL Without Off-Policy Evaluation; Offline Reinforcement Learning with Implicit Q-Learning; Contrastive Learning as Goal-Conditioned Reinforcement Learning +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the offline RL setup in which value learning is trained independently of policy extraction, enabling the paper’s bottleneck analysis. +- **rationale:** To show that policy extraction rather than value learning is often the limiting factor, the paper must analyze settings where the critic can be held fixed while different extraction procedures are compared. That dependence is structurally on a family of decoupled offline RL methods rather than on one single canonical paper, so `NONE` is the faithful canonical choice and the relevant prior methods are better represented as additional studies. +- **evidence_span:** “we focus on offline RL methods with decoupled value and policy training phases” + +### C1.I2 +- **ingredient:** Behavior-constrained policy-gradient extraction objectives for policies learned from fixed critics +- **canonical study:** A Minimalist Approach to Offline Reinforcement Learning +- **additional studies:** Advantage-Weighted Regression: Simple and Scalable Off-Policy Reinforcement Learning; Offline Reinforcement Learning via High-Fidelity Generative Behavior Modeling +- **role:** CORE_METHOD +- **contribution:** Provides the representative behavior-regularized policy-gradient extraction objective that the paper identifies as stronger and more scalable than common alternatives. +- **rationale:** The paper’s empirical conclusion depends on contrasting extraction families and finding that DDPG+BC-style extraction often outperforms widely used value-weighted or sampling-based alternatives. DDPG+BC is the clearest representative study for that stronger extraction family, while AWR and SfBC-style methods serve as additional comparison groundings rather than the canonical representative. +- **evidence_span:** “switching to behavior-constrained policy gradient objectives (e.g., DDPG+BC) often leads to substantial improvements” + +### C1.I3 +- **ingredient:** Comparative protocol that evaluates extraction objectives across data regimes and deployment generalization settings +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the systematic empirical protocol that isolates policy extraction and generalization bottlenecks across environments and data regimes. +- **rationale:** A finding about what the main bottleneck is cannot be established from one aggregate return table alone. The paper needs a structured comparative protocol that varies data size, quality, and coverage while comparing extraction methods under shared critics. That protocol is built in the target paper itself, so `NONE` is the correct grounding. +- **evidence_span:** “We use data size, quality, and coverage as levers for systematically controlling their impacts” + + +-------------------------------------------------- +REFERENCE EXAMPLE 3 +-------------------------------------------------- + +# EXAMPLE: IndicXTREME / IndicBERT + +## 1. Cluster Evidence + +Based on downstream cluster contributions, the paper introduces two distinct downstream-used contributions: + +- a **benchmark** for multilingual zero-shot evaluation on Indic languages +- a **pretrained multilingual encoder** for Indic NLU + +Downstream evidence shows these are reused differently and should therefore be split into two atomic claims. + +--- + +## 2. Claim Split Decision + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Benchmark +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C5 + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Resource +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C2 + +### Why this split is correct + +The paper contributes two separate artifacts with distinct downstream roles: + +- **IndicXTREME** is reused as a multilingual benchmark for zero-shot evaluation. +- **IndicBERT v2** is reused as a pretrained model for transfer and downstream fine-tuning. + +These have different structural ingredients, so they should not be annotated as one merged claim. + +--- + +## 3. Rewritten Claims + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Benchmark +- **rewritten_claim:** **Benchmark: A multilingual Indic-language evaluation suite, enabling standardized zero-shot assessment of pretrained multilingual models across diverse NLU tasks.** +- **why_this_is_atomic:** This claim isolates the benchmark artifact and its evaluation role, without bundling the pretrained model. +- **optional cluster_id:** C5 +- **decision:** YES_SUFFICIENT + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Resource +- **rewritten_claim:** **Resource: A pretrained multilingual encoder for Indic languages, enabling zero-shot transfer and downstream adaptation on Indic NLU tasks.** +- **why_this_is_atomic:** This claim isolates the pretrained encoder as the reused artifact, separate from the benchmark used to evaluate such models. +- **optional cluster_id:** C2 +- **decision:** YES_SUFFICIENT + +--- + +## 4. Minimal Ingredient Set Per Claim + +## Claim C1: Indic benchmark + +### Ingredient C1.I1 +- **ingredient_id:** C1.I1 +- **ingredient:** Multilingual, multi-task, zero-shot evaluation framework for cross-lingual NLU benchmarking +- **why_structurally_necessary:** The benchmark is explicitly framed as a multilingual zero-shot evaluation suite across tasks and languages. Without this benchmark framing, it would not exist in its claimed form as a standardized cross-lingual evaluation resource. +- **why_not_lower_level_substeps:** This should remain one benchmark-paradigm ingredient rather than being split into separate benchmark papers or separate task families. +- **why_not_adjacent_implementation_details:** The structural dependency is the zero-shot multilingual benchmark framing itself, not local benchmark formatting or reporting choices. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** XTREME: A Massively Multilingual Multi-task Benchmark for Evaluating Cross-lingual Generalization +- **additional groundings if any:** XTREME-R: Towards More Challenging and Nuanced Multilingual Evaluation; XGLUE: A New Benchmark Dataset for Cross-lingual Pre-training, Understanding and Generation +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the multilingual zero-shot benchmark framing that the paper adapts to Indic languages. +- **rationale:** To construct a benchmark whose point is standardized zero-shot evaluation across many languages and tasks, the paper depends on an existing multilingual benchmark paradigm. XTREME is the cleanest canonical grounding because it most directly established that framing; XTREME-R and XGLUE are relevant supporting studies but are less direct representatives. +- **evidence_span:** “aims to test the multilingual zero-shot capabilities of pretrained language models” + +### Ingredient C1.I2 +- **ingredient_id:** C1.I2 +- **ingredient:** NLU task datasets and source resources used to assemble the benchmark +- **why_structurally_necessary:** A multi-task benchmark requires the actual upstream task substrates from which its evaluation sets are translated, adapted, corrected, or incorporated. +- **why_not_lower_level_substeps:** This should remain one composite data-source ingredient rather than being split into separate ingredients for each task dataset. +- **why_not_adjacent_implementation_details:** The key dependency is the heterogeneous task substrate itself, not each local adaptation step used for one task. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** Choice of Plausible Alternatives: An Evaluation of Commonsense Causal Reasoning; XNLI: Evaluating Cross-lingual Sentence Representations; IndicXNLI: Evaluating Multilingual Inference for Indian Languages; Naamapadam: A Large-Scale Named Entity Annotated Data for Indic Languages; The Flores-101 Evaluation Benchmark for Low-Resource and Multilingual Machine Translation; MASSIVE: A 1M-Example Multilingual Natural Language Understanding Dataset with 51 Typologically-Diverse Languages +- **role:** DATA_SOURCE +- **contribution:** Supplies the task definitions and raw/source evaluation material that the benchmark translates, subsets, verifies, or incorporates. +- **rationale:** Constructing a multi-task benchmark requires the underlying task substrates that define its evaluation sets. Because these substrates are inherently composite across multiple tasks and datasets, they do not map cleanly to a single prior study, making `NONE` the appropriate canonical choice, with the task-specific sources listed as additional studies. +- **evidence_span:** “We manually translate the COPA test set into 18 Indic languages to create IndicCOPA.” + +### Ingredient C1.I3 +- **ingredient_id:** C1.I3 +- **ingredient:** Human-supervised multilingual curation, translation, and verification protocol +- **why_structurally_necessary:** Human supervision is one of the benchmark’s defining claimed properties. Without it, the benchmark would not exist in the same form as a human-validated evaluation resource. +- **why_not_lower_level_substeps:** This should remain one protocol-level ingredient rather than being split into manual translation, verification, correction, and annotation as separate ingredients. +- **why_not_adjacent_implementation_details:** The structural dependency is the benchmark-wide human-supervision workflow, not each local annotation step. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the manual benchmark-construction process ensuring that all included evaluation sets are created, translated, edited, or verified by humans. +- **rationale:** This ingredient is necessary because the benchmark’s defining property is that all evaluation sets are created or verified under human supervision. Removing it would fundamentally change the nature of the benchmark, reducing it to a weaker, non-human-validated collection of datasets. Because this workflow is benchmark-specific and realized in the target paper, `NONE` is the correct grounding. +- **evidence_span:** “ALL the evaluation sets included in IndicXTREME were created with human supervision” + +--- + +## Claim C2: Indic multilingual encoder + +### Ingredient C2.I1 +- **ingredient_id:** C2.I1 +- **ingredient:** Transformer-based masked language modeling architecture +- **why_structurally_necessary:** The resource is explicitly a BERT-style pretrained encoder. Without the transformer MLM architecture, it would not exist in its claimed form. +- **why_not_lower_level_substeps:** This should remain one architecture-level ingredient rather than being split into encoder layers, attention heads, or hyperparameter choices. +- **why_not_adjacent_implementation_details:** Model size, vocabulary size, and optimizer settings are implementation details; the structural dependency is the BERT-style MLM architecture itself. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding +- **additional groundings if any:** none +- **role:** CORE_METHOD +- **contribution:** Provides the transformer encoder architecture and masked language modeling objective used to pretrain the model. +- **rationale:** To build a multilingual BERT-style encoder, the paper requires a transformer architecture trained with masked language modeling. BERT is the clearest and most faithful canonical grounding for that dependency. +- **evidence_span:** “We use the default hyperparameters of BERT-Base” + +### Ingredient C2.I2 +- **ingredient_id:** C2.I2 +- **ingredient:** Translation Language Modeling objective for cross-lingual alignment +- **why_structurally_necessary:** The paper explicitly uses TLM as a second core objective to improve cross-lingual transfer by aligning languages through parallel data. +- **why_not_lower_level_substeps:** This should remain one objective-level ingredient rather than being split into masking details, parallel-pair formatting, or loss implementation details. +- **why_not_adjacent_implementation_details:** The structural dependency is the TLM objective itself, not specific data-preparation or training-engineering choices around it. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Cross-lingual Language Model Pretraining +- **additional groundings if any:** none +- **role:** CORE_METHOD +- **contribution:** Provides the cross-lingual pretraining objective used to align representations across languages. +- **rationale:** The model is not only trained with MLM; it explicitly incorporates TLM as a core objective for cross-lingual alignment. That objective maps cleanly to XLM as the canonical prior study. +- **evidence_span:** “Translation Language Modeling (Conneau and Lample, 2019, TLM)” + +### Ingredient C2.I3 +- **ingredient_id:** C2.I3 +- **ingredient:** Large-scale Indic-language pretraining text +- **why_structurally_necessary:** A multilingual encoder for Indic languages cannot be pretrained in its claimed form without a large-scale Indic-language text substrate. +- **why_not_lower_level_substeps:** This should remain one training-data ingredient rather than being split into IndicCorp, Wikipedia, OSCAR, Samanantar-derived text, or synthetic translations separately. +- **why_not_adjacent_implementation_details:** Specific corpus components and augmentation choices are subordinate pieces of the broader pretraining-text ingredient and should not be promoted to standalone structural ingredients. +- **necessary:** true +- **from_prior_work:** partially +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** IndicNLPSuite: Monolingual Corpora, Evaluation Benchmarks and Pre-trained Multilingual Language Models for Indian Languages; Samanantar: The Largest Publicly Available Parallel Corpora Collection for 11 Indic Languages +- **role:** TRAINING_DATA +- **contribution:** Provides the large-scale Indic-language text substrate used to pretrain the multilingual encoder. +- **rationale:** Recreating the model requires the actual large-scale Indic-language training text. But the operative pretraining text is structurally composite: it is dominated by corpus material created or assembled in the target paper, while also incorporating prior resources such as IndicCorp lineage and Samanantar-derived text. Because no single prior study cleanly represents that full training substrate, `NONE` is the correct canonical choice, with prior contributing resources listed as additional studies. +- **evidence_span:** “we merge data from IndicCorp v2 with Indic language data from Wikipedia and OSCAR” + +--- + +## 5. Excluded Tempting Non-Ingredients + +### Claim C1 exclusions +- **Machine translation model for auto-translating some source datasets** + - Excluded because it is not structural to the benchmark as a whole and is only used to help construct a small part of one task family rather than defining the benchmark itself. + +- **English training datasets used for zero-shot evaluation** + - Excluded because these are used to apply the benchmark in evaluation setups, not to construct the benchmark artifact itself. + +- **Individual task datasets as separate top-level ingredients** + - Excluded because the benchmark’s task substrate is better represented as one composite DATA_SOURCE ingredient rather than over-decomposed into one ingredient per task. + +### Claim C2 exclusions +- **IndicTrans as a standalone ingredient** + - Excluded because it is a helper tool for generating some synthetic parallel data, not a core structural ingredient of the model in its claimed form. + +- **Temperature-based upsampling / balancing tricks** + - Excluded because these are recipe-level training choices rather than artifact-defining ingredients. + +- **Language ID tokens / tokenizer / WordPiece design** + - Excluded because these are implementation-level configuration details, not part of the minimal structurally sufficient ingredient set. + +- **Samanantar as the canonical grounding for the full training-text ingredient** + - Excluded because it represents only one smaller component of the broader pretraining substrate; assigning it as canonical would distort the true structure of the ingredient. + +--- + +## 6. Candidate Grounding Decisions + +## Claim C1: Indic benchmark + +### Ingredient C1.I1 +- **Candidate:** XTREME: A Massively Multilingual Multi-task Benchmark for Evaluating Cross-lingual Generalization +- **decision:** accepted_canonical +- **why:** Cleanest representative study for the multilingual multi-task zero-shot benchmark framing adapted here. +- **Candidate:** XTREME-R: Towards More Challenging and Nuanced Multilingual Evaluation +- **decision:** accepted_additional +- **why:** Important related multilingual benchmark extension, but not the clearest canonical origin. +- **Candidate:** XGLUE: A New Benchmark Dataset for Cross-lingual Pre-training, Understanding and Generation +- **decision:** accepted_additional +- **why:** Relevant supporting multilingual benchmark study, but less direct as the primary framing. + +### Ingredient C1.I2 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The task substrate is structurally composite across multiple datasets and source resources. +- **Candidate:** Choice of Plausible Alternatives: An Evaluation of Commonsense Causal Reasoning +- **decision:** accepted_additional +- **why:** Source task for IndicCOPA. +- **Candidate:** XNLI: Evaluating Cross-lingual Sentence Representations +- **decision:** accepted_additional +- **why:** Important upstream NLI source task. +- **Candidate:** IndicXNLI: Evaluating Multilingual Inference for Indian Languages +- **decision:** accepted_additional +- **why:** Important Indic-specific benchmark source. +- **Candidate:** Naamapadam: A Large-Scale Named Entity Annotated Data for Indic Languages +- **decision:** accepted_additional +- **why:** Upstream NER task resource. +- **Candidate:** The Flores-101 Evaluation Benchmark for Low-Resource and Multilingual Machine Translation +- **decision:** accepted_additional +- **why:** Upstream retrieval/evaluation source resource. +- **Candidate:** MASSIVE: A 1M-Example Multilingual Natural Language Understanding Dataset with 51 Typologically-Diverse Languages +- **decision:** accepted_additional +- **why:** Upstream intent/slot resource. + +### Ingredient C1.I3 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The human-supervised curation and verification workflow is benchmark-specific and realized in the target paper. + +## Claim C2: Indic multilingual encoder + +### Ingredient C2.I1 +- **Candidate:** BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding +- **decision:** accepted_canonical +- **why:** Cleanest representative study for the transformer MLM architecture used by the model. + +### Ingredient C2.I2 +- **Candidate:** Cross-lingual Language Model Pretraining +- **decision:** accepted_canonical +- **why:** Cleanest representative study for the TLM objective explicitly used by the paper. + +### Ingredient C2.I3 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The pretraining text is dominated by same-paper corpus construction and is structurally composite across several sources. +- **Candidate:** IndicNLPSuite: Monolingual Corpora, Evaluation Benchmarks and Pre-trained Multilingual Language Models for Indian Languages +- **decision:** accepted_additional +- **why:** Important prior Indic corpus lineage that the paper extends. +- **Candidate:** Samanantar: The Largest Publicly Available Parallel Corpora Collection for 11 Indic Languages +- **decision:** accepted_additional +- **why:** Important contributing prior text source, but not representative enough to be canonical for the full pretraining-text ingredient. + +--- + +## 7. Final Grounding Decisions + +## Claim C1 +- **C1.I1** → XTREME: A Massively Multilingual Multi-task Benchmark for Evaluating Cross-lingual Generalization → CONCEPTUAL_FRAMEWORK + - additional: XTREME-R: Towards More Challenging and Nuanced Multilingual Evaluation; XGLUE: A New Benchmark Dataset for Cross-lingual Pre-training, Understanding and Generation +- **C1.I2** → NONE → DATA_SOURCE + - additional: Choice of Plausible Alternatives: An Evaluation of Commonsense Causal Reasoning; XNLI: Evaluating Cross-lingual Sentence Representations; IndicXNLI: Evaluating Multilingual Inference for Indian Languages; Naamapadam: A Large-Scale Named Entity Annotated Data for Indic Languages; The Flores-101 Evaluation Benchmark for Low-Resource and Multilingual Machine Translation; MASSIVE: A 1M-Example Multilingual Natural Language Understanding Dataset with 51 Typologically-Diverse Languages +- **C1.I3** → NONE → EVALUATION_PROTOCOL + +## Claim C2 +- **C2.I1** → BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding → CORE_METHOD +- **C2.I2** → Cross-lingual Language Model Pretraining → CORE_METHOD +- **C2.I3** → NONE → TRAINING_DATA + - additional: IndicNLPSuite: Monolingual Corpora, Evaluation Benchmarks and Pre-trained Multilingual Language Models for Indian Languages; Samanantar: The Largest Publicly Available Parallel Corpora Collection for 11 Indic Languages + +--- + +## 8. Final Rationales And Evidence + +## Claim C1: Indic benchmark + +### C1.I1 +- **ingredient:** Multilingual, multi-task, zero-shot evaluation framework for cross-lingual NLU benchmarking +- **canonical study:** XTREME: A Massively Multilingual Multi-task Benchmark for Evaluating Cross-lingual Generalization +- **additional studies:** XTREME-R: Towards More Challenging and Nuanced Multilingual Evaluation; XGLUE: A New Benchmark Dataset for Cross-lingual Pre-training, Understanding and Generation +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the multilingual zero-shot benchmark framing that the paper adapts to Indic languages. +- **rationale:** To construct a benchmark whose point is standardized zero-shot evaluation across many languages and tasks, the paper depends on an existing multilingual benchmark paradigm. XTREME is the cleanest canonical grounding because it most directly established that framing; XTREME-R and XGLUE are relevant supporting studies but are less direct representatives. +- **evidence_span:** “aims to test the multilingual zero-shot capabilities of pretrained language models” + +### C1.I2 +- **ingredient:** NLU task datasets and source resources used to assemble the benchmark +- **canonical study:** NONE +- **additional studies:** Choice of Plausible Alternatives: An Evaluation of Commonsense Causal Reasoning; XNLI: Evaluating Cross-lingual Sentence Representations; IndicXNLI: Evaluating Multilingual Inference for Indian Languages; Naamapadam: A Large-Scale Named Entity Annotated Data for Indic Languages; The Flores-101 Evaluation Benchmark for Low-Resource and Multilingual Machine Translation; MASSIVE: A 1M-Example Multilingual Natural Language Understanding Dataset with 51 Typologically-Diverse Languages +- **role:** DATA_SOURCE +- **contribution:** Supplies the task definitions and raw/source evaluation material that the benchmark translates, subsets, verifies, or incorporates. +- **rationale:** Constructing a multi-task benchmark requires the underlying task substrates that define its evaluation sets. Because these substrates are inherently composite across multiple tasks and datasets, they do not map cleanly to a single prior study, making `NONE` the appropriate canonical choice, with the task-specific sources listed as additional studies. +- **evidence_span:** “We manually translate the COPA test set into 18 Indic languages to create IndicCOPA.” + +### C1.I3 +- **ingredient:** Human-supervised multilingual curation, translation, and verification protocol +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the manual benchmark-construction process ensuring that all included evaluation sets are created, translated, edited, or verified by humans. +- **rationale:** This ingredient is necessary because the benchmark’s defining property is that all evaluation sets are created or verified under human supervision. Removing it would fundamentally change the nature of the benchmark, reducing it to a weaker, non-human-validated collection of datasets. Because this workflow is benchmark-specific and realized in the target paper, `NONE` is the correct grounding. +- **evidence_span:** “ALL the evaluation sets included in IndicXTREME were created with human supervision” + +## Claim C2: Indic multilingual encoder + +### C2.I1 +- **ingredient:** Transformer-based masked language modeling architecture +- **canonical study:** BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding +- **role:** CORE_METHOD +- **contribution:** Provides the transformer encoder architecture and masked language modeling objective used to pretrain the model. +- **rationale:** To build a multilingual BERT-style encoder, the paper requires a transformer architecture trained with masked language modeling. BERT is the clearest and most faithful canonical grounding for that dependency. +- **evidence_span:** “We use the default hyperparameters of BERT-Base” + +### C2.I2 +- **ingredient:** Translation Language Modeling objective for cross-lingual alignment +- **canonical study:** Cross-lingual Language Model Pretraining +- **role:** CORE_METHOD +- **contribution:** Provides the cross-lingual pretraining objective used to align representations across languages. +- **rationale:** The model is not only trained with MLM; it explicitly incorporates TLM as a core objective for cross-lingual alignment. That objective maps cleanly to XLM as the canonical prior study. +- **evidence_span:** “Translation Language Modeling (Conneau and Lample, 2019, TLM)” + +### C2.I3 +- **ingredient:** Large-scale Indic-language pretraining text +- **canonical study:** NONE +- **additional studies:** IndicNLPSuite: Monolingual Corpora, Evaluation Benchmarks and Pre-trained Multilingual Language Models for Indian Languages; Samanantar: The Largest Publicly Available Parallel Corpora Collection for 11 Indic Languages +- **role:** TRAINING_DATA +- **contribution:** Provides the large-scale Indic-language text substrate used to pretrain the multilingual encoder. +- **rationale:** Recreating the model requires the actual large-scale Indic-language training text. But the operative pretraining text is structurally composite: it is dominated by corpus material created or assembled in the target paper, while also incorporating prior resources such as IndicCorp lineage and Samanantar-derived text. Because no single prior study cleanly represents that full training substrate, `NONE` is the correct canonical choice, with prior contributing resources listed as additional studies. +- **evidence_span:** “we merge data from IndicCorp v2 with Indic language data from Wikipedia and OSCAR” + + +-------------------------------------------------- +REFERENCE EXAMPLE 4 +-------------------------------------------------- + +# EXAMPLE: MassiveDS datastore scaling resource + +## 1. Cluster Evidence + +**The MassiveDS datastore resource** is used in the literature + +The paper also presents a pipeline reordering trick that makes datastore scaling computationally feasible. However, the downstream evidence does **not** show that this pipeline is reused as a separate artifact on its own. Therefore, the correct annotation is **one resource claim**, with the pipeline-reordering idea represented as an important ingredient rather than a separate claim. + +--- + +## 2. Claim Split Decision + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Resource +- **rewritten_claim:** **Dataset: A large-scale multi-domain datastore and experimental pipeline, enabling systematic study and evaluation of retrieval-based language model scaling.** +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C3 + +### Why this split is correct + +The downstream reuse is centered on **one released resource**: the MassiveDS datastore. + +- Later work uses the **datastore itself** as the reusable artifact. +- The paper also introduces a pipeline-reordering idea, but the evidence does **not** show that this pipeline is being reused downstream as a separate artifact. +- Because claim splitting is driven by **distinct downstream-used contributions**, the correct annotation is to keep **one claim**. + +The pipeline reordering is still important, but its role is different: + +- it is a **structural ingredient** that makes the datastore resource computationally feasible in its claimed form +- it is **not** a separate discovery claim unless downstream evidence shows reuse of the pipeline itself + +--- + +## 3. Rewritten Claims + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Resource +- **rewritten_claim:** **Resource: A massive open multi-domain datastore that combines general web and domain-specific text, enabling large-scale retrieval-in-context language modeling and datastore-scaling studies.** +- **why_this_is_atomic:** This claim isolates the released datastore resource that downstream work reuses. It does not split off the pipeline-reordering idea as a separate claim because that idea is not independently substantiated as a downstream-used artifact. +- **optional cluster_id:** C3 +- **decision:** YES_SUFFICIENT + +--- + +## 4. Minimal Ingredient Set Per Claim + +## Claim C1: MassiveDS datastore resource + +### Ingredient C1.I1 +- **ingredient_id:** C1.I1 +- **ingredient:** Multi-source open text corpora spanning general web and domain-specific sources +- **why_structurally_necessary:** The datastore is defined by both its scale and its domain diversity. Without large open corpora spanning general web and specialized domains, the released resource would not exist in its claimed form. +- **why_not_lower_level_substeps:** This should remain one composite data-source ingredient rather than being split into separate final ingredients for web data, books, scientific papers, math, biomedical text, code, and other source families. +- **why_not_adjacent_implementation_details:** The structural dependency is the multi-source raw text substrate itself, not every individual source-selection or curation decision. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** RedPajama: An Open Source Recipe to Reproduce LLaMA Training Dataset; PubMed Baseline Repository; S2ORC: The Semantic Scholar Open Research Corpus; peS2o (Pretraining Efficiently on S2ORC) Dataset; NaturalProofs: Mathematical Theorem Proving in Natural Language; Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer +- **role:** DATA_SOURCE +- **contribution:** Supplies the large-scale open text sources used to assemble the datastore across both general web and specialized domains. +- **rationale:** Recreating MassiveDS requires the raw text substrate that gives it both scale and domain coverage. That substrate is inherently composite: the datastore mixes general web data with multiple distinct domain-specific resources. No single prior study faithfully represents that entire source mixture, so `NONE` is the correct canonical grounding, with the main contributing source resources listed as additional studies. +- **evidence_span:** “comprising 1.4 trillion tokens of both general web data and domain specific data” + +### Ingredient C1.I2 +- **ingredient_id:** C1.I2 +- **ingredient:** Pretrained dense retriever for document embedding +- **why_structurally_necessary:** The released datastore is not just raw text; it is an operational retrieval datastore with dense-vector retrieval over passages. Without a pretrained retriever to embed documents, it would not function in its claimed form. +- **why_not_lower_level_substeps:** This should remain one retriever ingredient rather than being split into encoder architecture, embedding dimensionality, or indexing internals. +- **why_not_adjacent_implementation_details:** Retriever hyperparameters and ablation choices are implementation details. The structural dependency is the pretrained dense retriever used to embed the datastore. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Unsupervised Dense Information Retrieval with Contrastive Learning +- **additional groundings if any:** none +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the pretrained retriever used to represent datastore documents as dense vectors for retrieval. +- **rationale:** To reproduce the datastore as a usable retrieval resource, the corpus must be embedded into a searchable dense space. The paper explicitly uses Contriever-MSMARCO for this purpose, making Contriever the cleanest canonical grounding. +- **evidence_span:** “we use CONTRIEVER-MSMARCO (Izacard et al., 2022), which represents every document in the datastore as a dense vector” + +### Ingredient C1.I3 +- **ingredient_id:** C1.I3 +- **ingredient:** Retrieval-in-context language modeling framework +- **why_structurally_necessary:** MassiveDS is introduced specifically to support retrieve-in-context language models rather than as a generic corpus release. Without this framing, the datastore would lose the main capability it is meant to enable. +- **why_not_lower_level_substeps:** This should remain one framework-level ingredient rather than being split into retrieval prompting, concatenation order, or model-specific usage details. +- **why_not_adjacent_implementation_details:** The structural dependency is the inference-time retrieval framework itself, not one particular LM or evaluation setup. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** In-Context Retrieval-Augmented Language Models +- **additional groundings if any:** none +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Defines the retrieval-in-context framing in which documents are retrieved from an external datastore and prepended to LM inputs at inference time. +- **rationale:** The datastore is meaningful here because it supports retrieval-based language models that use retrieved documents in context. Ram et al. is the cleanest canonical grounding because it directly represents the retrieve-in-context setup that the paper focuses on. +- **evidence_span:** “We focus on retrieve-in-context language models (RIC-LMs)” + +### Ingredient C1.I4 +- **ingredient_id:** C1.I4 +- **ingredient:** Data filtering workflow for deduplication, decontamination, and quality control +- **why_structurally_necessary:** The released resource is not merely a raw trillion-token dump; it is a constructed datastore whose quality depends on filtering, deduplication, and decontamination steps. Without this workflow, the datastore would be substantially weaker as a reusable retrieval resource. +- **why_not_lower_level_substeps:** This should remain one filtering-workflow ingredient rather than being split into deduplication, decontamination, and quality filtering as separate ingredients. +- **why_not_adjacent_implementation_details:** Individual thresholds, heuristics, or local filtering choices are not the structural dependency; the key dependency is the broader filtering workflow that makes the datastore usable. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Dolma: an Open Corpus of Three Trillion Tokens for Language Model Pretraining Research +- **additional groundings if any:** none +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the filtering workflow used to clean the datastore through deduplication, decontamination, and quality controls. +- **rationale:** Building a reusable datastore at this scale requires systematic filtering so the resulting resource is not just large, but also usable and reasonably clean. The paper explicitly ties this step to Dolma, making it the best canonical grounding. +- **evidence_span:** “data filtering, including deduplication, decontamination, and quality filters” + +### Ingredient C1.I5 +- **ingredient_id:** C1.I5 +- **ingredient:** Pipeline reordering to amortize expensive indexing and retrieval across datastore variants +- **why_structurally_necessary:** The datastore is released together with an experimental pipeline that makes large-scale datastore scaling feasible on modest compute. Without this reordering idea, the resource would lose an important defining property: practical support for scalable datastore studies. +- **why_not_lower_level_substeps:** This should remain one pipeline-level ingredient rather than being split into retrieval-overfetching, post-hoc subsampling, and late-stage filtering as separate ingredients. +- **why_not_adjacent_implementation_details:** The structural dependency is the reordering principle itself—running expensive steps once and sharing them—not the local code or systems details used to implement it. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the core pipeline idea that makes datastore scaling computationally feasible by sharing expensive indexing and retrieval across many variants. +- **rationale:** A key part of the released contribution is not only the datastore’s scale, but the fact that the paper makes large-scale scaling studies feasible by reordering operations so indexing and retrieval are amortized. This idea is introduced in the target paper rather than inherited from a single prior study, so `NONE` is the correct grounding. +- **evidence_span:** “the most expensive ones—indexing and retrieval—are run only once at the start” + +--- + +## 5. Excluded Tempting Non-Ingredients + +### Claim C1 exclusions +- **Pipeline reordering as a separate second claim** + - Excluded because the downstream reuse evidence centers on the datastore resource, not on the pipeline as an independently reused artifact. The pipeline is important, but it is best represented as a structural ingredient of the single datastore claim. + +- **Passage chunking and indexing into fixed-length retrieval units** + - Excluded because this is a lower-level construction step rather than one of the smallest higher-level structural ingredients. It is subordinate to the broader retriever-and-datastore construction pipeline. + +- **FAISS / exact top-K retrieval implementation** + - Excluded because this is too implementation-specific. It supports the pipeline but is not needed at the abstraction level of the minimal structurally sufficient ingredient set. + +- **Separate final ingredients for general web data and domain-specific data** + - Excluded because the higher-level composite ingredient is the multi-source open text substrate spanning both. Splitting them would over-decompose the datastore’s source basis. + +- **Contriever as MODEL_INITIALIZATION instead of IMPLEMENTATION_TOOLING** + - Excluded because its role here is not initializing the target discovery as a model checkpoint, but operationally embedding the datastore documents for retrieval. + +--- + +## 6. Candidate Grounding Decisions + +## Claim C1: MassiveDS datastore resource + +### Ingredient C1.I1 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The datastore’s source substrate is structurally composite across general web and many domain-specific resources. +- **Candidate:** RedPajama: An Open Source Recipe to Reproduce LLaMA Training Dataset +- **decision:** accepted_additional +- **why:** Representative open source for books and other large-scale text components, but not sufficient alone to represent the whole source mixture. +- **Candidate:** PubMed Baseline Repository +- **decision:** accepted_additional +- **why:** Important biomedical source contributing to the domain-specific side of the datastore. +- **Candidate:** S2ORC: The Semantic Scholar Open Research Corpus +- **decision:** accepted_additional +- **why:** Important scientific-paper source for the domain-specific mixture. +- **Candidate:** peS2o (Pretraining Efficiently on S2ORC) Dataset +- **decision:** accepted_additional +- **why:** Additional scientific corpus contribution. +- **Candidate:** NaturalProofs: Mathematical Theorem Proving in Natural Language +- **decision:** accepted_additional +- **why:** Representative mathematical-language source for the domain-specific mixture. +- **Candidate:** Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer +- **decision:** accepted_additional +- **why:** Relevant additional grounding because C4 is explicitly cited as one of the general web data sources. + +### Ingredient C1.I2 +- **Candidate:** Unsupervised Dense Information Retrieval with Contrastive Learning +- **decision:** accepted_canonical +- **why:** Cleanest representative prior study for Contriever, the pretrained dense retriever actually used to embed datastore passages. + +### Ingredient C1.I3 +- **Candidate:** In-Context Retrieval-Augmented Language Models +- **decision:** accepted_canonical +- **why:** Cleanest representative framing for the retrieval-in-context LM setup the datastore is designed to support. + +### Ingredient C1.I4 +- **Candidate:** Dolma: an Open Corpus of Three Trillion Tokens for Language Model Pretraining Research +- **decision:** accepted_canonical +- **why:** Explicitly cited by the paper for the filtering stage and is the cleanest representative for the deduplication / decontamination / quality-control workflow. + +### Ingredient C1.I5 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The pipeline-reordering idea is introduced in the target paper itself and is not cleanly attributable to one prior study. + +--- + +## 7. Final Grounding Decisions + +## Claim C1 +- **C1.I1** → NONE → DATA_SOURCE + - additional: RedPajama: An Open Source Recipe to Reproduce LLaMA Training Dataset; PubMed Baseline Repository; S2ORC: The Semantic Scholar Open Research Corpus; peS2o (Pretraining Efficiently on S2ORC) Dataset; NaturalProofs: Mathematical Theorem Proving in Natural Language; Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer +- **C1.I2** → Unsupervised Dense Information Retrieval with Contrastive Learning → IMPLEMENTATION_TOOLING +- **C1.I3** → In-Context Retrieval-Augmented Language Models → CONCEPTUAL_FRAMEWORK +- **C1.I4** → Dolma: an Open Corpus of Three Trillion Tokens for Language Model Pretraining Research → IMPLEMENTATION_TOOLING +- **C1.I5** → NONE → CONCEPTUAL_FRAMEWORK + +--- + +## 8. Final Rationales And Evidence + +## Claim C1: MassiveDS datastore resource + +### C1.I1 +- **ingredient:** Multi-source open text corpora spanning general web and domain-specific sources +- **canonical study:** NONE +- **additional studies:** RedPajama: An Open Source Recipe to Reproduce LLaMA Training Dataset; PubMed Baseline Repository; S2ORC: The Semantic Scholar Open Research Corpus; peS2o (Pretraining Efficiently on S2ORC) Dataset; NaturalProofs: Mathematical Theorem Proving in Natural Language; Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer +- **role:** DATA_SOURCE +- **contribution:** Supplies the large-scale open text sources used to assemble the datastore across both general web and specialized domains. +- **rationale:** Recreating MassiveDS requires the raw text substrate that gives it both scale and domain coverage. That substrate is inherently composite: the datastore mixes general web data with multiple distinct domain-specific resources. No single prior study faithfully represents that entire source mixture, so `NONE` is the correct canonical grounding, with the main contributing source resources listed as additional studies. +- **evidence_span:** “comprising 1.4 trillion tokens of both general web data and domain specific data” + +### C1.I2 +- **ingredient:** Pretrained dense retriever for document embedding +- **canonical study:** Unsupervised Dense Information Retrieval with Contrastive Learning +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the pretrained retriever used to represent datastore documents as dense vectors for retrieval. +- **rationale:** To reproduce the datastore as a usable retrieval resource, the corpus must be embedded into a searchable dense space. The paper explicitly uses Contriever-MSMARCO for this purpose, making Contriever the cleanest canonical grounding. +- **evidence_span:** “we use CONTRIEVER-MSMARCO (Izacard et al., 2022), which represents every document in the datastore as a dense vector” + +### C1.I3 +- **ingredient:** Retrieval-in-context language modeling framework +- **canonical study:** In-Context Retrieval-Augmented Language Models +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Defines the retrieval-in-context framing in which documents are retrieved from an external datastore and prepended to LM inputs at inference time. +- **rationale:** The datastore is meaningful here because it supports retrieval-based language models that use retrieved documents in context. Ram et al. is the cleanest canonical grounding because it directly represents the retrieve-in-context setup that the paper focuses on. +- **evidence_span:** “We focus on retrieve-in-context language models (RIC-LMs)” + +### C1.I4 +- **ingredient:** Data filtering workflow for deduplication, decontamination, and quality control +- **canonical study:** Dolma: an Open Corpus of Three Trillion Tokens for Language Model Pretraining Research +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the filtering workflow used to clean the datastore through deduplication, decontamination, and quality controls. +- **rationale:** Building a reusable datastore at this scale requires systematic filtering so the resulting resource is not just large, but also usable and reasonably clean. The paper explicitly ties this step to Dolma, making it the best canonical grounding. +- **evidence_span:** “data filtering, including deduplication, decontamination, and quality filters” + +### C1.I5 +- **ingredient:** Pipeline reordering to amortize expensive indexing and retrieval across datastore variants +- **canonical study:** NONE +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the core pipeline idea that makes datastore scaling computationally feasible by sharing expensive indexing and retrieval across many variants. +- **rationale:** A key part of the released contribution is not only the datastore’s scale, but the fact that the paper makes large-scale scaling studies feasible by reordering operations so indexing and retrieval are amortized. This idea is introduced in the target paper rather than inherited from a single prior study, so `NONE` is the correct grounding. +- **evidence_span:** “the most expensive ones—indexing and retrieval—are run only once at the start” + + +-------------------------------------------------- +TARGET PAPER CONTEXT +-------------------------------------------------- +Target paper context: + +TITLE: AVerImaTeC: A Dataset for Automatic Verification of Image-Text Claims with Evidence from the Web +YEAR: 2025 +VENUE: arXiv.org +PAPER_ID: + +REFINED DOWNSTREAM CLUSTER EVIDENCE: +[ + { + "cluster_id": "C1", + "representative_claim": "Extends target dataset for new dataset creation", + "cluster_title": "Extends target dataset for new dataset creation", + "count": "1", + "merge_rationale": "Directly builds upon the target dataset by extracting its pairs to construct a new dataset." + }, + { + "cluster_id": "C2", + "representative_claim": "Uses target dataset and knowledge stores for evaluation", + "cluster_title": "Uses target dataset and knowledge stores for evaluation", + "count": "2", + "merge_rationale": "Directly uses the target dataset and its provided knowledge stores to evaluate proposed claim verification methods." + }, + { + "cluster_id": "C3", + "representative_claim": "Extends reference-based evaluation method for a novel setting", + "cluster_title": "Extends reference-based evaluation method for a novel setting", + "count": "1", + "merge_rationale": "Adapts and extends the evaluation method introduced in the target paper to a new multimodal meme understanding setting." + }, + { + "cluster_id": "C4", + "representative_claim": "Adopts temporally constrained retrieval methodology for dataset creation", + "cluster_title": "Adopts temporally constrained retrieval methodology for dataset creation", + "count": "1", + "merge_rationale": "Adopts the target paper's specific methodology of restricting evidence retrieval by date to create a new dataset." + }, + { + "cluster_id": "C5", + "representative_claim": "Extends target baseline framework for multimodal claim verification", + "cluster_title": "Extends target baseline framework for multimodal claim verification", + "count": "1", + "merge_rationale": "Builds upon the baseline verification framework introduced by the target paper to develop a new system." + } +] + +TARGET PAPER CONTEXT: +[ + { + "cluster_id": "C1", + "representative_claim": "Extends target dataset for new dataset creation", + "cluster_title": "Extends target dataset for new dataset creation", + "count": "1", + "merge_rationale": "Directly builds upon the target dataset by extracting its pairs to construct a new dataset." + }, + { + "cluster_id": "C2", + "representative_claim": "Uses target dataset and knowledge stores for evaluation", + "cluster_title": "Uses target dataset and knowledge stores for evaluation", + "count": "2", + "merge_rationale": "Directly uses the target dataset and its provided knowledge stores to evaluate proposed claim verification methods." + }, + { + "cluster_id": "C3", + "representative_claim": "Extends reference-based evaluation method for a novel setting", + "cluster_title": "Extends reference-based evaluation method for a novel setting", + "count": "1", + "merge_rationale": "Adapts and extends the evaluation method introduced in the target paper to a new multimodal meme understanding setting." + }, + { + "cluster_id": "C4", + "representative_claim": "Adopts temporally constrained retrieval methodology for dataset creation", + "cluster_title": "Adopts temporally constrained retrieval methodology for dataset creation", + "count": "1", + "merge_rationale": "Adopts the target paper's specific methodology of restricting evidence retrieval by date to create a new dataset." + }, + { + "cluster_id": "C5", + "representative_claim": "Extends target baseline framework for multimodal claim verification", + "cluster_title": "Extends target baseline framework for multimodal claim verification", + "count": "1", + "merge_rationale": "Builds upon the baseline verification framework introduced by the target paper to develop a new system." + } +] + +FULL PAPER CONTENT: +\documentclass{article} + +% if you need to pass options to natbib, use, e.g.: +% \PassOptionsToPackage{numbers, compress}{natbib} +% before loading neurips_2024 + + +% ready for submission +%\usepackage{neurips_2024} +\usepackage[final]{neurips_2024} + +% to compile a preprint version, e.g., for submission to arXiv, add add the +% [preprint] option: +% \usepackage[preprint]{neurips_2024} + + +% to compile a camera-ready version, add the [final] option, e.g.: +% \usepackage[final]{neurips_2024} + + +% to avoid loading the natbib package, add option nonatbib: +% \usepackage[nonatbib]{neurips_2024} + + +\usepackage[utf8]{inputenc} % allow utf-8 input +\usepackage[T1]{fontenc} % use 8-bit T1 fonts +\usepackage{hyperref} % hyperlinks +\usepackage{url} % simple URL typesetting +\usepackage{booktabs} % professional-quality tables +\usepackage{amsfonts} % blackboard math symbols +%\usepackage{unicode-math} +%\DeclareUnicodeCharacter{1E31}{$\kappa$} +\usepackage{nicefrac} % compact symbols for 1/2, etc. +\usepackage{microtype} % microtypography +%\usepackage{xcolor} % colors +\usepackage[table]{xcolor} +\usepackage{multirow,multicol} +\usepackage{graphicx} +\usepackage{pifont} +\usepackage{bbding} +\newcommand{\crcomment}[1]{\textcolor{cyan}{[~Rui: #1~]}} +\newcommand{\crtext}[1]{\textcolor{magenta}{#1}} +\newcommand{\redtext}[1]{\textcolor{red}{#1}} +\newcommand{\greentext}[1]{\textcolor{teal}{#1}} + +\usepackage{xspace} +\newcommand{\our}{\textsc{AVerImaTeC}\xspace} + +\title{\textsc{AVerImaTeC}: A Dataset for Automatic Verification of Image-Text Claims with Evidence from the Web} + + +% The \author macro works with any number of authors. There are two commands +% used to separate the names and addresses of multiple authors: \And and \AND. +% +% Using \And between authors leaves it to LaTeX to determine where to break the +% lines. Using \AND forces a line break at that point. So, if LaTeX puts 3 of 4 +% authors names on the first line, and the last on the second line, try using +% \AND instead of \And before the third author name. + + +\author{% + Rui Cao$^\heartsuit$, Zifeng Ding$^\heartsuit$, Zhijiang Guo$^\heartsuit$, Michael Schlichtkrull$^\diamondsuit$, Andreas Vlachos$^\heartsuit$\\ + University of Cambridge$^\heartsuit$, Queen Mary University of London$^\diamondsuit$\\ + \texttt{\{rc990,zd320,zg283,av308\}@cam.ac.uk}, \texttt{m.schlichtkrull@qmul.ac.uk} \\ + % examples of more authors + % \And + % Coauthor \\ + % Affiliation \\ + % Address \\ + % \texttt{email} \\ + % \AND + % Coauthor \\ + % Affiliation \\ + % Address \\ + % \texttt{email} \\ + % \And + % Coauthor \\ + % Affiliation \\ + % Address \\ + % \texttt{email} \\ + % \And + % Coauthor \\ + % Affiliation \\ + % Address \\ + % \texttt{email} \\ +} + + +\begin{document} + + +\maketitle + + +\begin{abstract} +Textual claims are often accompanied by images to enhance their credibility and spread on social media, but this also raises concerns about the spread of misinformation. +Existing datasets for automated verification of image-text claims remain limited, as +they often consist of synthetic claims and lack evidence annotations to capture the reasoning behind the verdict. +In this work, we introduce \textbf{\our}, a dataset consisting of 1,297 real-world image-text claims. Each claim is annotated with question-answer (QA) pairs containing evidence from the web, reflecting a decomposed reasoning regarding the verdict. +We mitigate common challenges in fact-checking datasets such as contextual dependence, temporal leakage, and evidence insufficiency, via claim normalization, temporally constrained evidence annotation, and a two-stage sufficiency check. We assess the consistency of the annotation in \our via inter-annotator studies, achieving a $\kappa=0.742$ on verdicts and $74.7\%$ consistency on QA pairs. +We also propose a novel evaluation method for evidence retrieval and conduct extensive experiments to establish baselines for verifying image-text claims using open-web evidence. +\end{abstract} + +\section{Introduction} +\label{sec:intro} +Misinformation has become a public concern due to its potential impact on elections, public health and safety~\citep{99b421ca-b79f-3991-b65a-e1460d47174f,doi:10.1126/science.aaw8243,DBLP:journals/cacm/BarPF23,article-election-pub-health-covid}. +To curb its spread, professional fact-checkers are employed to identify misleading content. +However, they are unable to keep up with the vast volume of information online~\citep{article-fighting-misinfo-scale,10.1093/pnasnexus/pgae217}. +The severity of the problem, along with the limitations of manual verification, has motivated the development of automated fact-checking (AFC)~\citep{DBLP:journals/tacl/GuoSV22,ijcai2021p619}. + +To support research in AFC, the research community has created various benchmark datasets~\citep{DBLP:conf/naacl/ThorneVCM18,DBLP:conf/nips/AlyGST00CM21,DBLP:conf/nips/SchlichtkrullG023,DBLP:conf/emnlp/AlhindiPM18,DBLP:conf/sigir/YaoS0CH23,DBLP:conf/naacl/ChenKSDC24}, aiming to enhance the effectiveness and interpretability of fact-checking systems. +However, most existing benchmarks focus exclusively on textual claims, overlooking the important role of media in the dissemination of misinformation. +Recent studies estimate that approximately $80\%$ of online claims are multimodal involving both text and media~\citep{DBLP:journals/corr/abs-2405-11697}, as media can enhance perceived credibility~\citep{Newman2012NonprobativeP} and increase exposure~\citep{doi:10.1177/0022243719881113}. +Among these, images are the most prevalent media type~\citep{DBLP:journals/corr/abs-2405-11697}. + +\begin{figure*}[t] + \centering + \includegraphics[width=0.95\linewidth]{claim-exp.pdf} + \caption{ +\textbf{An annotated claim from \our.} The rationale for verifying an image-text claim has been decomposed into a sequence of QA pairs, which could be potentially multimodal. + } + \label{fig:intro-exp} +\end{figure*} +While several datasets have been developed for image-text AFC, many are synthetic, generated by manually manipulating either the textual or visual modality of image-text pairs~\citep{DBLP:conf/emnlp/LuoDR21,DBLP:journals/ijmir/PapadopoulosKPP24,DBLP:conf/cvpr/JiaHZJCL23}. +Due to discrepancies between synthetic data and real-world data~\citep{DBLP:conf/emnlp/ZengLGP24}, models that perform well on synthetic benchmarks may fail to generalize to real-world claims. +Moreover, recent work~\citep{DBLP:journals/corr/abs-2407-13488} showed that models can achieve high performance on such datasets by exploiting superficial correlations, such as image-text similarity, without examining factuality and logical consistency. +Some benchmarks~\citep{DBLP:conf/emnlp/ZlatkovaNK19,DBLP:conf/lrec/NakamuraLW20,DBLP:journals/bigdata/ShuMWLL20} attempt to include real-world image-text claims extracted from fact-checking articles. As noted in prior work~\citep{DBLP:conf/emnlp/OusidhoumY022,DBLP:conf/nips/SchlichtkrullG023}, this may result in omitting critical contextual information +for verification, such as context to resolve coreferences. +Additionally, both synthetic and real-world image-text datasets typically lack annotated evidence, making it impossible to evaluate models’ reasoning process. + + +To address the limitations above, we propose the \textbf{A}utomated \textbf{Ver}ification of \textbf{Ima}ge-\textbf{Te}xt \textbf{C}laim (\textbf{\textsc{AVerImaTeC}}) dataset, where the verification of real-world image-text claims has been decomposed into a sequence of question-answering with evidence from the web. +In addition, each claim is annotated with metadata information, a veracity label based on retrieved evidence and a textual justification, explaining how the verdict is reached, as shown in the example in Figure~\ref{fig:intro-exp}. + +To construct \our, +initially annotators are asked to identify and normalize image-text claims from fact-checking articles, incorporating necessary contextual information while providing associated metadata. +Next, annotators convert the verification rationale from the articles into QA pairs, while being restricted to using only online evidence published before the claim's date. +Given the multimodal nature of the task, both questions and answers may involve images. Finally, we conduct two rounds of quality control to ensure that each annotated claim is supported by sufficient evidence for the annotated verdict. +The resulting dataset, contains 1,297 image-text claims. +To assess the consistency of the verdict labels we conducted an inter-annotator agreement study in which we obtained a Randolph's~\citep{randolph} free-marginal $\kappa$ of $0.742$ over 100 re-annotated claims. +The re-annotation +recovered $74.7\%$ of the original QA pairs, confirming that the annotations capture reasoning paths for verifying image-text claims consistently. + +We further introduce a baseline for image-text claim verification, which operates by generating evidence-seeking questions aimed at fact-checking and answering these questions with a set of expert tools. +Since \our is the first image-text claim verification dataset to incorporate QA annotations that explicitly reflect reasoning paths and evidence, we develop a reference-based evaluation method to assess models' generated questions and retrieved evidence. Using this evidence evaluation, we report conditional verdict accuracy which measures the correctness of predicted verdicts only when the associated evidence score exceeds a predefined threshold\footnote{The dataset is available here: https://huggingface.co/datasets/Rui4416/AVerImaTeC. The code can be accessed here: https://github.com/abril4416/AVerImaTeC}. + +\section{Related Works} +\label{sec:related} + +\begin{table*}[t] +\centering +\small +\caption{\textbf{Comparison of fact-checking datasets.} +\textit{Indep.} (independence) denotes whether extracted claims are context independent (e.g., understandable without fact-checking articles). \textit{Img.}, \textit{Suff.}, \textit{Retr.}, and \textit{Unleak.} are abbreviations for image, sufficiency, retrieval and unleaked evidence, respectively. +Sufficiency indicates whether there is sufficient supporting evidence to reach annotated verdicts; Retrieval refers to whether open-world evidence retrieval is performed; Unleaked represents whether annotated evidence contains temporal leakage such as including evidence published after claims. +} + \label{tab:rela-dataset} + \begin{tabular}{lccccccc} + \toprule + \multirow{2.5}{*}{\textbf{Dataset}}& + \multicolumn{2}{c}{\textbf{Claim}} & \multicolumn{4}{c}{\textbf{Evidence}} & \multirow{2.5}{*}{\textbf{\# Claims}} \\ + \cmidrule(lr){2-3} \cmidrule(lr){4-7} + & \textit{Real} & \textit{Indep.} & + \textit{Img.} & + \textit{Suff.} & \textit{Retr.} & \textit{Unleak.} &\\ + \midrule + FEVER~\citep{DBLP:conf/naacl/ThorneVCM18} & \redtext{\ding{56}} &\greentext{\checkmark} &\redtext{\ding{56}} & \greentext{\checkmark}&\greentext{\checkmark} &- &185,445\\ + FEVEROUS~\citep{DBLP:conf/nips/AlyGST00CM21} & \redtext{\ding{56}}&\greentext{\checkmark} &\redtext{\ding{56}}&\greentext{\checkmark} &\greentext{\checkmark} & -&87,026\\ + Liar-Plus~\citep{DBLP:conf/emnlp/AlhindiPM18} &\greentext{\checkmark} &\redtext{\ding{56}} &\redtext{\ding{56}} &\greentext{\checkmark} &\redtext{\ding{56}} & \redtext{\ding{56}} &12,836\\ + Snopes~\citep{DBLP:conf/conll/HanselowskiSSLG19} &\greentext{\checkmark} &\redtext{\ding{56}} & \redtext{\ding{56}}& \redtext{\ding{56}}&\redtext{\ding{56}} & \greentext{\checkmark} &6,422\\ + MultiFC~\citep{DBLP:conf/emnlp/AugensteinLWLHH19} &\greentext{\checkmark} &\redtext{\ding{56}} & \redtext{\ding{56}}&\redtext{\ding{56}} &\greentext{\checkmark} & \redtext{\ding{56}} &36,534\\ + AVeriTec~\citep{DBLP:conf/nips/SchlichtkrullG023} &\greentext{\checkmark} &\greentext{\checkmark} & \redtext{\ding{56}}&\greentext{\checkmark} &\greentext{\checkmark} & \greentext{\checkmark} &4,568\\ + CLAIMDECOMP~\citep{DBLP:conf/naacl/ChenKSDC24} & \greentext{\checkmark} & \redtext{\ding{56}} & \redtext{\ding{56}}&\redtext{\ding{56}} &\greentext{\checkmark} & \greentext{\checkmark} &1,200\\ + MOCHEG~\citep{DBLP:conf/sigir/YaoS0CH23} &\greentext{\checkmark} & \redtext{\ding{56}}&\greentext{\checkmark} &\redtext{\ding{56}} & \redtext{\ding{56}}& \redtext{\ding{56}} &15,601\\ + \midrule + NewsCLIPpings~\citep{DBLP:conf/emnlp/LuoDR21} &\redtext{\ding{56}} &\greentext{\checkmark} & -&- & -& - &988,283\\ + InfoSurgeon~\citep{DBLP:conf/acl/FungTRPJCMBS20} &\redtext{\ding{56}} &\greentext{\checkmark} & -&- & -& - &30,000\\ + Autosplice~\citep{DBLP:conf/cvpr/JiaHZJCL23} &\redtext{\ding{56}} &\greentext{\checkmark} & -&- & -& - &5,894\\ + DGM~\citep{DBLP:conf/cvpr/Shao0L23} & \redtext{\ding{56}}& \greentext{\checkmark}& -&- & -& - &230,000\\ + MMFake~\citep{DBLP:journals/corr/abs-2406-08772} &\redtext{\ding{56}} &\greentext{\checkmark} & -&- & -& - &11,000\\ + Verite~\citep{DBLP:journals/ijmir/PapadopoulosKPP24} &\redtext{\ding{56}} &\greentext{\checkmark} & -&- & -& - &1,000\\ + COSMOS~\citep{DBLP:journals/corr/abs-2101-06278} &Mix &\redtext{\ding{56}} & -&- & -& - &201,700\\ + FACTIFY~\citep{DBLP:conf/aaai/MishraSBCRPD0SE22} & Mix & \redtext{\ding{56}}& -&- &- & - &50,000\\ + FACTIFY 2~\citep{DBLP:conf/defactify/SuryavardanMPCR23} & Mix & \redtext{\ding{56}}& -&- &- & - &50,000\\ + MMOOC~\citep{DBLP:conf/acisp/XuDCLY24} & Mix & \greentext{\checkmark}& -&- &- & - &364,000\\ + Fauxtography~\citep{DBLP:conf/emnlp/ZlatkovaNK19} & \greentext{\checkmark}&\redtext{\ding{56}} &- &- &- &- &1,233\\ + Fakeddit~\citep{DBLP:conf/lrec/NakamuraLW20} &\greentext{\checkmark} & \redtext{\ding{56}}& -&- & -& - &1,063,106\\ + Qprop~\citep{DBLP:journals/ipm/Barron-CedenoJM19} & \greentext{\checkmark}& \redtext{\ding{56}} & -&- & -& - &51,294\\ + FakeNewsNet~\citep{DBLP:journals/bigdata/ShuMWLL20} & \greentext{\checkmark}&\redtext{\ding{56}} & -&- & -& - &23,196\\ + MuMiN~\citep{DBLP:conf/sigir/NielsenM22} & \greentext{\checkmark}&\redtext{\ding{56}} & -&- & -& - &12,914\\ + \midrule + \our& \greentext{\checkmark}&\greentext{\checkmark}&\greentext{\checkmark}&\greentext{\checkmark}&\greentext{\checkmark}&\greentext{\checkmark} &1,297\\ + \bottomrule +\end{tabular} +\end{table*} + +Automated fact-checking (AFC) has become increasingly important due to the pressing need to curb the spread of misinformation~\citep{DBLP:journals/tacl/GuoSV22,ijcai2021p619}. To support AFC research, several datasets focusing primarily on text-based claims have been proposed (see the top block of Table~\ref{tab:rela-dataset}). +Motivated by the prevalence of images in claims, fact-checking datasets for image-text claims were introduced (listed in the bottom block of Table~\ref{tab:rela-dataset}). Some studies~\citep{DBLP:conf/emnlp/LuoDR21,DBLP:journals/ijmir/PapadopoulosKPP24,DBLP:conf/cvpr/JiaHZJCL23} have generated synthetic claims by applying manipulation techniques to the visual and textual modalities of image-text pairs. However, there are discrepancies between synthetic data and real-world image-text claims~\citep{DBLP:conf/emnlp/ZengLGP24,DBLP:journals/corr/abs-2407-13488}, raising concerns about the generalization of models to real-world image-text claim verification. +Although some benchmarks~\citep{DBLP:conf/emnlp/ZlatkovaNK19,DBLP:journals/ipm/Barron-CedenoJM19,DBLP:journals/bigdata/ShuMWLL20} focus on real-world claims (e.g., those derived from fact-checking articles), they all suffer from context dependence. Moreover, all existing datasets with image-text claims lack evidence annotations, limiting transparency, and the ability to understand the rationale behind fact-checking verdicts. +%\crcomment{Table~\ref{tab:rela-dataset} has been updated by adding the number of claims.} + +To capture the rationale in claim verification, a complex reasoning task, various reasoning representations have been explored, including natural logic~\citep{DBLP:conf/emnlp/StrongA024}, functional programs~\citep{DBLP:conf/acl/PanWLLWKN23}, and QA~\citep{DBLP:conf/naacl/ChenKSDC24,DBLP:conf/emnlp/QiXSLJWH23,DBLP:conf/emnlp/PanLKN23}. +In AVerImaTeC, we adopt QA as the reasoning representation, as the QA format is more intuitive and accessible for annotators, enabling efficient and consistent annotation by non-experts. Furthermore, other reasoning forms can often be mapped into QA-style representations, ensuring compatibility and flexibility for future extensions. + +\section{Annotation Schema} +\label{sec:anno_struct} +Each claim in \our is normalised to be understandable alone without additional context, such as the original social media post or the associated fact-checking article. +For each claim, we provide key metadata such as the \textit{speaker}, \textit{publisher}, \textit{publication date}, and the relevant \textit{location}. +These metadata elements can serve as valuable evidence for claim verification. +We further annotate each claim with +\textit{claim type} (e.g., \textit{quote verification}, which determines whether a quote was actually attributed to the correct speaker), and \textit{fact-checking strategy} (e.g., \textit{reverse image search} to find background information about images). +While not all of these annotations are directly used during verification, they offer insights for developing fact-checking models. + +The sequence of QA pairs reflects the reasoning process involved in evidence retrieval and verification. +Given the multimodal nature of the claims, both questions and answers may include images (i.e., image-related questions and image answers). +A single question may have multiple answers, +as there may be conflicts or disagreements in the evidence. +Questions may refer to previous questions and their answers as long as they are understandable on their own, thus capturing multi-hop reasoning in fact-checking. +Answers (other than \textit{``No answer could be found.''} or derived via image analysis without external information) must be supported by a \textit{source url} linking to a web page. +To ensure long-term accessibility, all source pages are archived on the internet archive.\footnote{https://archive.org/} +We also provide metadata for each annotated QA pair, including the \textit{question type} (e.g., \textit{image-related} or \textit{metadata-related}), \textit{answering method}, \textit{answer type} (e.g., \textit{extractive}, \textit{abstractive}, \textit{boolean} or \textit{image-based}), and \textit{source medium type} (type of web content used as evidence). + +We follow the four-way veracity labeling schema from~\cite{DBLP:conf/nips/SchlichtkrullG023}: \textit{supported}, \textit{refuted}, \textit{not enough evidence}, and \textit{conflicting/cherry-picking}. +An image-text claim can be refuted due to its textual part (the textual part of is factually wrong) +and/or due to \textit{misuse of images} (e.g., mis-interpreting the context of an image). +For instance, the claim in Figure~\ref{fig:intro-exp} is refuted due to both textual refutation and image misuse. +\textit{Not enough evidence} refers to cases where evidence is insufficient to either support or refute a claim. \textit{Conflicting/Cherry-picking} covers cherry-picking claims, +true-but-misleading claims, as well as claims with conflicting evidence. Conflicts among evidence has been extensively studied in the context of QA~\citep{DBLP:journals/corr/abs-2410-12311,DBLP:journals/corr/abs-2404-12447}. + +A textual justification is added to the claims to explain how verdicts can be reached on the basis of the evidence found. Considering that the evidence can contain multiple images, we assign unique and special tokens to the images in the evidence (e.g., \texttt{[IMG\_1]}) for annotators to refer to in justifications. +They may also include commonsense reasoning or inductive reasoning beyond the retrieved evidence. For instance, the evidence of the claim in Figure~\ref{fig:intro-exp} proves the image was taken in 2016 while Kamala Harris' mother died in 2009. The justification infers it is impossible that Harris \textit{was with her parents in +the image}, taken in 2016 as her mother died in 2009 already. + +\section{Annotation Process} +\label{sec:anno_pipe} +\begin{figure*}[t] + \centering + \includegraphics[width=\linewidth]{annotation-pipeline.pdf} + \caption{ + \textbf{Annotation pipeline.} We first + normalize the claim, then perform QA annotation to structure evidence retrieval. + Two rounds of evidence sufficiency checks ensure annotation quality.} + \label{fig:anno-pipeline} +\end{figure*} + +The five-phase annotation pipeline is illustrated in Figure~\ref{fig:anno-pipeline}, extending the annotation process proposed in~\cite{DBLP:conf/nips/SchlichtkrullG023} to the domain of image-text claims. +% +In phase one, an annotator extracts and normalize valid image-text claims, given a fact-checking article. For each extracted claim, a different annotator in the second phase generates questions and answers to reflect the rationale of fact-checking based on the article, using evidence from the web. A provisional veracity label is annotated as well. Thirdly, a first round of evidence sufficiency checking will be conducted by a third annotator who provides a justification and a verdict solely based on the QA pairs, without considering the fact-checking article. +Different verdicts in the second and third phases suggest possible insufficient evidence in the second phase. In this case, the claim is forwarded to a fourth phase to add or modify existing QA pairs and +a fifth phase for an additional round of evidence sufficiency check. +Any claims for with unresolved conflicts in the verdicts are discarded. +We ensure that each annotation phase of a claim is done independently +by using different annotators (i.e., the same claim will not be annotated by the same annotator twice). +Details for annotation guidelines and annotators' demographics are provided in Appendix~\ref{sec:app-annotation-guidelines} and~\ref{sec:app-anno-detail-annotator-demo}, respectively. + +\noindent\textbf{Claim Extraction \& Normalization. }Given a fact-checking article, an annotator extracts all claims from it, as multiple claims may pertain to a single event. +The modality types of claims are annotated, and only \textit{image-text} claims (i.e., textual claims paired with images) are forwarded to phase two. +For identified image-text claims, sufficient context must be provided to ensure that the claim, specifically its textual component, can be understood independently of the surrounding article. +In some cases, metadata (e.g., date, location, speaker) is sufficient to disambiguate a claim. However, for ambiguities beyond metadata, such as unresolved coreference or missing referents, annotators are instructed to enrich the extracted claims so that they can be interpreted independently. +The associated images are uploaded and normalized (e.g., separating collages or locating original, unaltered versions). +Relevant metadata is also annotated. +We exclude unverifiable claims (e.g., speculation or personal opinions) +and claims where images are not used in verification or involve manipulated content. + +\noindent\textbf{Question Generation \& Answering. }Annotators in this phase are instructed to transform the rationales derived from fact-checking articles into a sequence of QA pairs. +Each question is annotated with its \textit{question type}, \textit{answering method} and \textit{answers}. +For \textit{image-related} questions, annotators must select relevant images from either the claim images or images from previous answers. If a question is not marked as \textit{unanswerable} or pertains to \textit{image analysis}, annotators are required to provide supporting urls of the evidence source. +We advise annotators to prioritize evidence sources linked within the fact-checking articles but to exclude anything published after the claim date, including the article itself. +When linked sources are unavailable (e.g., dead links) or insufficient, annotators are provided with a custom Google search interface that supports both text and image queries. +All retrieved pages from the interface are restricted to dates \textit{prior} to the claim date to prevent temporal leakage~\citep{DBLP:conf/emnlp/Glockner0G22}. Based on the generated QA pairs, annotators assign a verdict. + +\noindent\textbf{Evidence Sufficiency Check. } +In this phase, +a third annotator, who does not have access to the fact-checking article, is presented with the extracted claim and its associated annotated QA pairs. +The annotator is tasked with assigning a verdict and providing a textual justification. +This verdict is then compared to the one generated during the QA annotation phase. A discrepancy between the two verdicts indicates insufficient evidence, and the QA annotation process is repeated to refine the QA pairs, followed by a second round of evidence sufficiency assessment with new annotators. +\section{Dataset Statistics} +\label{sec:data_stat} +\begin{table*}[t] +\small +\centering +\caption{\textbf{Data statistics for dataset splits.} End date refers to the latest publication date of claims included in each split. The start date of each \textit{dev} and \textit{test} split corresponds to the end date of the preceding split. The final row of the table reports the distribution of claim labels across four categories: \textit{supported} (S), \textit{refuted} (R), \textit{conflicting/cherry-picking} (C), and \textit{not enough evidence} (N).} + \label{tab:data-statistics} + \begin{tabular}{l|ccc} + \toprule + \textbf{Split} &\textbf{Train} &\textbf{Dev} &\textbf{Test} \\ + \midrule + \# Claims & 793&152 &352 \\ + \# Images / Claim & 1.49&1.38 & 1.38\\ + \# QA Pairs / Claim & 2.86& 2.84& 3.11\\ + Reannotated (\%) &15.0 &15.8 &9.4 \\ + End Date &31-05-2023 &31-07-2023 &21-03-2025 \\ + Labels (S / R / C / N) (\%) & 1.6 / 95.3 / 0.8 / 2.3& 2.6 / 92.8 / 0.7 / 3.9& 13.9 / 78.1 / 2.0 / 6.0\\ + \bottomrule +\end{tabular} +\end{table*} + +\noindent\textbf{Data Distribution. } We began with 2,353 fact-checking articles +After discarding those that were inaccessible, did not focus on image-text claims, or contained unresolved annotation conflicts, we obtained 1,297 annotated image-text claims using the annotation pipeline described in the previous section. +Further details on article sources and the filtering process are provided in Appendix~\ref{sec:app-data-p1-data-filter} and~\ref{sec:app-discard-claims}, respectively. +%, and after discarding +%\crtext{inaccessible ones, ones not focusing on image-text claims and those with unresolved annotation conflicts}, we obtained 1,297 annotated image-text claims using the annotation pipeline described in the previous section. +%\crtext{Details for article sources and discarding claims are provided in Appendix~\ref{sec:app-data-p1-data-filter}} +The splits of our dataset are temporally organized, and detailed statistics are presented in Table~\ref{tab:data-statistics}. +We find that $23.7\%$ image-text claims include more than one claim image, highlighting the need to understand multiple visual inputs. +On average, each claim is annotated with $2.92$ questions. Of these, $3.5\%$ questions have more than one answer, and +$62.5\%$ are image-related, emphasizing the importance of visual context in verifying image-text claims. +To address these questions, annotators frequently selected +\textit{Image-search} ($53.9\%$) as the answering method, indicating the necessity for tools that support image-centric information retrieval. +Regarding answer types, +$58.8\%$ are \textit{extractive}, consistent with our annotation guidelines. Additionally, $1.6\%$ of answers are images themselves, underscoring the importance of supporting image retrieval as direct answers. +A small proportion ($2.6\%$) of questions are marked as \textit{unanswerable}, reflecting cases where no supporting evidence could be found online. +Further metadata statistics, such as \textit{claim types} and \textit{answer types}, are provided in Appendix~\ref{sec:app-data-statistics-metadatao}. +% +The dataset shows a label imbalance, with most claims being \textit{refuted}, which is expected given that misleading content is more likely to be scrutinized. + +\noindent\textbf{Inter-Annotator Agreement. } +We re-annotated $100$ claims with a different group of annotators, following the same annotation pipeline as described in Section~\ref{sec:anno_pipe}. As in prior work~\citep{DBLP:conf/nips/SchlichtkrullG023}, we assume that +the first phase of annotation +has already been completed, and thus the re-annotation process begins from the second phase. We evaluate inter-annotator agreement for both verdict labels and QA annotations. +For verdict agreement, we use +Randolph's~\citep{randolph} free-marginal multi-rater $\kappa$, designed for unbalanced datasets~\citep{10.1007/s11634-010-0073-4}. We obtained an agreement score of $\kappa = \mathbf{0.742}$. For comparison, +AVeriTec~\citep{DBLP:conf/nips/SchlichtkrullG023} reported agreement of $0.619$. +Using Fleiss’ $\kappa$, a more traditional metric, our annotation process achieves an agreement score of $0.450$. +To evaluate QA annotations, the three best performing annotators were provided with an extracted claim and two independently annotated sets of QA pairs, and asked to determine how many QA pairs in one set were covered by the other. +We compute recall and precision by comparing the original annotated QA pairs against those from re-annotation. The recall rate is $\mathbf{74.7\%}$ and the precision is $\mathbf{67.2\%}$. +The substantial overlap between the two sets suggests strong agreement between annotators. +\section{Evaluation} +\label{sec:eval} +The evaluation of model accuracy on our dataset considers both the \textit{retrieved evidence} and the \textit{veracity}. +Following~\citep{DBLP:conf/naacl/ThorneVCM18,DBLP:conf/nips/SchlichtkrullG023}, we first assess the quality of the retrieved evidence by comparing it against human-annotated references, and report veracity prediction accuracy conditioned on the evidence scores. I.e., the accuracy of a verdict prediction is considered only if the associated evidence score exceeds a predefined threshold $\lambda$, otherwise the claim is considered to be labeled incorrectly. +This reflects the requirement that an effective fact-checking system should not only predict return verdicts on claims but also provide appropriate evidence. + +Recent research~\citep{DBLP:journals/corr/abs-2411-05375} showed that reference-based evaluation of evidence with large language models (LLMs) aligns best with human assessments. +We extend this framework to a multimodal setting where evidence from the web may comprise text and images. We transform each QA pair into an evidence statement following~\citep{DBLP:journals/corr/abs-2411-05375}, where images are represented with special image tokens (e.g., \texttt{[IMG\_1]}). For instance, the first QA pair in Figure~\ref{fig:intro-exp} is transformed to \textit{``[IMG\_1] was published in 2016.''}. +We then conduct separate reference-based evaluations for the textual and visual components. For the textual part, we use a validation method similar to~\citep{DBLP:journals/corr/abs-2411-05375}. +If a matched evidence item is found within the ground-truth set, we proceed to a second step to compare the associated images. If image similarity falls below a threshold +the match is deemed invalid due to image mismatch. +We exploit Gemini-2.0-Flash~\citep{gemini-deepmind} as the scoring model for both steps in the reference-based evaluation inspired by its power~\citep{DBLP:journals/corr/abs-2411-05375}. +We report evidence \textit{recall}, defined as the percentage of ground-truth evidence instances successfully retrieved. + + +We performed alignment checks and robustness checks (against adversarial attacks) of different reference-based evaluation schemes (text-only, interleaved and the separated evaluation) in our setting (details in Appendix~\ref{sec:app-eval-evid-eval} and~\ref{sec:app-eval-robust}). Our separated reference-based evaluation achieved the highest alignment with human assessments, with a Spearman correlation coefficient ($\rho$)~\citep{spearman04} of $0.332$ and a Pearson correlation coefficient ($r$)~\citep{pearson1896mathematical} of $0.381$. Furthermore, the separated evaluation method is the most robust towards adversarial attacks. + +Our evaluation primarily focuses on evidence retrieval and verdict prediction. +Additionally, given the availability of justification annotations, we assess the quality of model-generated justifications by comparing them to the human-annotated ground truth with the traditional evaluation metric, ROUGE-1~\citep{lin-2004-rouge}, a standard metric that is relatively tolerant of longer outputs. +While ROUGE-1 provides a baseline assessment, we recognize that more sophisticated and targeted evaluation methods may be necessary. +Similarly, we adopt a conditional justification generation score that any claim for which the +evidence score is below $\lambda$ receives a score for justification of 0. +The prompts used for the QA pair conversion and evaluation are provided in Appendix~\ref{sec:app-prompt-qa-to-evid} and~\ref{sec:app-prompt-evaluation}, respectively. + +\section{Experiments} +\label{sec:experiment} + +\subsection{Baselines} +\label{sec:exp-baseline} +Our baseline system guides the fact-checking process by sequentially posing and answering essential questions to verify image-text claims. It consists of four components: +a \textit{question generator}, an \textit{answer generator}, a \textit{verifier} and a \textit{justification generator}. +Throughout the question-answering process, the system maintains an evolving \textit{evidence history context}, which is continuously updated with summarized evidence derived from each QA pair. +After the QA stage, the verifier receives the claim and the accumulated evidence context to predict a veracity label. +Finally, the justification generator produces a rationale that explains how the predicted verdict is supported by the evidence. + +\noindent\textbf{Question Generator.} A straightforward approach is to generate all verification questions at once, given an image-text claim, as in prior work~\citep{DBLP:conf/nips/SchlichtkrullG023}. We refer to this strategy as \textit{paralleled} question generation (\textbf{PQG}). +However, since later questions often depend on earlier ones (e.g., as shown in Figure~\ref{fig:intro-exp}), we also propose a \textit{dynamic} question generation (\textbf{DQG}) method, where each subsequent question is generated based on both the claim and the evolving evidence history. +To combine the strengths of both approaches, we introduce a \textit{hybrid} question generation (\textbf{HQA}) method, which generates the first few questions in parallel and then switches to dynamic generation for the remaining ones. +All three strategies employ an MLLM for question generation, leveraging the model's internal decomposition capability. +The prompts used for each strategy are detailed in Appendix~\ref{sec:app-prompt-qg}. +\begin{table*}[t] +\centering +\small +\caption{\textbf{Experimental results of baselines on \our.} \textbf{Q-Eval} and \textbf{Evid-Eval} denote for \textit{recall} scores of generated questions and retrieved evidence, with reference of ground-truth questions and evidence. We report verdict prediction and justification generation scores conditioned on evidence retrieval performance, specifically only considering verdict accuracy and justification generation performance when the evidence score is above 0.2, 0.3 and 0.4.} + \label{tab:exp-results-main} + \begin{tabular}{ll|cc|ccc|ccc} + \toprule + \textbf{LLM} &\textbf{MLLM} &\textbf{Q-Eval} &\textbf{Evid-Eval} & \multicolumn{3}{c}{\textbf{Veracity} \@ (.2/.3/.4)} & \multicolumn{3}{c}{\textbf{Justifications} \@ (.2/.3/.4)}\\ + \midrule + \rowcolor{lightgray} + \multicolumn{10}{c}{\textit{Paralleled Question Generation}}\\ + \midrule + Gemini & Gemini &0.42 & 0.15&0.15 &0.13 & 0.08& 0.15&0.11 &0.07 \\ + Qwen & Qwen-VL &0.43 &0.18 &0.09& 0.08&0.05 &0.13 &0.11 &0.07 \\ + Gemma & Gemma & 0.39& 0.21&0.14 & 0.12&0.09 &0.17 & 0.14& 0.10\\ + Qwen & LLaVA & 0.37& 0.16& 0.09&0.08 &0.05 &0.12 & 0.10&0.06 \\ + \midrule + \rowcolor{lightgray} + \multicolumn{10}{c}{\textit{Dynamic Question Generation}}\\ + \midrule + Gemini & Gemini &0.33& 0.22&0.17 &0.16 &0.12 &0.17 &0.16&0.11\\ + Qwen & Qwen-VL &0.27 & 0.12&0.10 & 0.09&0.05 &0.09 &0.08 &0.05\\ + Gemma & Gemma & 0.27& 0.19& 0.15&0.13 &0.10 & 0.15& 0.13&0.09 \\ + Qwen & LLaVA & 0.32&0.16 & 0.13& 0.11& 0.08&0.11 & 0.10&0.08 \\ + \midrule + \rowcolor{lightgray} + \multicolumn{10}{c}{\textit{Hybrid Question Generation}}\\ + \midrule + Gemini & Gemini & 0.36&0.19 &0.18 &0.17 & 0.10&0.17 &0.15& 0.09\\ + Qwen & Qwen-VL &0.37 &0.16 &0.11 &0.09 & 0.06& 0.12& 0.10&0.06 \\ + Gemma & Gemma &0.26 &0.25 &0.16 &0.15 &0.11 &0.19 &0.17& 0.12\\ + Qwen & LLaVA & 0.30&0.17 &0.09 &0.09 & 0.07&0.12 &0.11&0.07 \\ + \bottomrule +\end{tabular} +\end{table*} + +\noindent\textbf{Answer Generator. } +Given a generated question, the answer generator is responsible for generating an answer to the question. +Inspired by recent research on tool usage~\citep{DBLP:journals/corr/abs-2303-04671,DBLP:journals/corr/abs-2411-13697,DBLP:journals/corr/abs-2412-10510}, we integrate a set of specialized tools into the answer generation module, along with a tool selector that automatically selects the appropriate tool for a given question. +The system includes tools for: (1) \textit{reverse image search} (RIS) to retrieve text-based information associated with an image; (2) \textit{web search for texts} (WST) to retrieve relevant web texts given a textual query; (3) \textit{web search for images} (WSI) to retrieve relevant images using a textual query; and (4) \textit{visual question answering} (VQA) to answer questions directly based on input images, including comparison and detail analysis. +Among these, VQA, implemented with an MLLM, directly outputs an answer, while RIS and WST are followed by an LLM to leverage the retrieved text to generate an answer. +WSI is followed by an MLLM, which incorporates the retrieved image into the answer generation process. +Prompts for tool selection and answer generation are provided in Appendix~\ref{sec:app-prompt-tool-sel} and~\ref{sec:app-prompt-ans-gen}, respectively. + +\noindent\textbf{Verifier. }The verifier takes the claim to be verified, the multimodal evidence context and the verdict definitions as introduced in Section~\ref{sec:anno_struct} to predict a veracity label for the claim +(prompts in use shown in Appendix~\ref{sec:app-prompt-verdict-gen}). An MLLM will serve as the verifier. + +\noindent\textbf{Justification Generator. }Given the predicted verdict, in this step, the justification generator is asked to provide an explanation for the prediction. It also has access to the claim and evidence history (detailed prompts in Appendix~\ref{sec:app-prompt-justi-gen}). +We rely on an MLLM for justification generation. + +\noindent\textbf{Model Implementation. } +Our baseline system includes both an LLM and an MLLM, which could take different roles in components. +We experimented with four combinations of LLMs and MLLMs: 1) Gemini-2.0-flash-001~\citep{gemini-deepmind} (\textbf{Gemini}) as both the LLM and the MLLM; 2) Qwen/Qwen2.5-7B-Instruct~\citep{DBLP:journals/corr/abs-2412-15115} (\textbf{Qwen}) acts as the LLM and Qwen2.5-VL-7B-Instruct~\citep{DBLP:journals/corr/abs-2502-13923} (\textbf{Qwen-VL}) serves as the MLLM; 3) Gemma-3-12B~\citep{DBLP:journals/corr/abs-2503-19786} (\textbf{Gemma}), capable of both unimodal and multimodal understanding, is used as both the LLM and the MLLM; and 4) Qwen and LLaVA-Next-7B~\citep{DBLP:journals/corr/abs-2407-07895} (\textbf{LLaVA}) work as the LLM and MLLM respectively. +More details on model implementation and experiment settings are in Appendix~\ref{sec:app-detail-of-exp}. + + +\begin{table*}[t] +\centering +\small +\caption{\textbf{Baseline performance on verdict prediction and justification generation with ground-truth evidence (first block) and without accessing any external evidence (second block).} +NEE is for \textit{Not Enough Evidence} and Conflict.\ is for \textit{Conflicting/Cherry-picking}. +Justi.\ is for the performance of justification generation.} + \label{tab:exp-ablation-results} + \begin{tabular}{l|ll|cccc|c|c} + \toprule + \textbf{Evid. Source} & \textbf{LLM} &\textbf{MLLM} &\textbf{Refuted} &\textbf{Supported} & \textbf{NEE} &\textbf{Conflict.} &\textbf{Overall}& \textbf{Justi.}\\ + \midrule + \multirow{4}{*}{Ground-Truth}& Gemini & Gemini & 0.84&0.92 &0.52& 0.00&0.82&0.50\\ + & Qwen & Qwen-VL &0.87 & 0.47&0.62 & 0.00&0.78&0.44\\ + & Gemma & Gemma &0.63 &0.84 & 0.62& 0.00&0.64& 0.49\\ + & Qwen & LLaVA &0.55 & 0.47&0.90 &0.00 &0.55 & 0.43\\ + \midrule + No Search & Qwen &Qwen-VL &0.01 &0.02 &0.14 & 0.00&0.02 & 0.04 \\ + \bottomrule +\end{tabular} +\end{table*} + +\subsection{Main Results} +\label{sec:exp-results} +We investigated the performance of baseline models both under zero-shot and few-shot settings. +In the latter, three training instances were used as demonstrations to guide question generation (details in Appendix~\ref{sec:app-detail-exp-setting}). +Overall, few-shot baselines slightly outperform their zero-shot counterparts. +Few-shot performance results for various combinations of LLMs and MLLMs are presented in Table~\ref{tab:exp-results-main} (zero-shot model performance in Appendix~\ref{sec:app-add-exp-zero-shot}, and the findings hold for both settings). +We report conditional veracity accuracy and ROUGE-1 scores for justification generation under varying evidence evaluation thresholds, $\lambda = \{0.2, 0.3, 0.4\}$. + +\noindent\textbf{Comparison of Question Generation Strategies. } +Among the three question generation strategies, the parallel approach consistently outperformed others in producing critical questions for fact-checking. +The dynamic strategy, where questions are generated based on evolving evidence, yielded a weaker performance. +This suggests that the increased complexity of reasoning over evolving interleaved image-text evidence poses significant challenges for current MLLMs. +Open-source MLLMs often generated repetitive questions when using the dynamic strategy, further highlighting their limitations in handling complex multimodal inputs. + +\textbf{Performance of Evidence Retrieval. } +Evidence scores across all models are significantly lower than their question evaluation scores, underscoring the difficulty of retrieving appropriate evidence for image-text verification. +One reason for the failure of evidence retrieval is that models exhibited a bias toward using VQA as the answering tool (e.g., Qwen + Qwen-VL in PQG selected VQA as the answering tool for $30\%$ questions), diverging from human fact-checkers' preferences, despite being provided with tool selection demonstration examples +(more elaborations in Appendix~\ref{sec:app-detail-of-exp}). +MLLMs tend to rely more on internal image details rather than external contextual information to address image-related questions. +Additionally, approximately $13\%$ images failed to retrieve any contextual information via RIS (i.e., no web pages published before claim dates could be found), consistent with the findings of~\citep{DBLP:conf/emnlp/TongletMG24}. +A notable portion of questions, e.g., $30\%$ questions of the Qwen + Qwen-VL baseline with PQG, elicited responses such as \textit{``No answer could be found''}, based on retrieved evidence. +This can be attributed to two main factors besides the failure of RIS: (1) many web pages retrieved by RIS were non-scrapable (e.g., Instagram posts), and (2) the baseline evidence employed a naive ranking method, BM25~\citep{DBLP:journals/ftir/RobertsonZ09}, which neither considered the visual content of images nor incorporated fine-grained re-ranking. +Interestingly, higher scores for question generation did not always translate into better evidence retrieval. +For instance, under the hybrid strategy, the Gemini-based baseline model achieved a 0.1-point higher question generation score than the Gemma-based model, but had worse evidence retrieval performance. +Further analysis showed that Gemini generated a higher proportion of RIS-dependent questions ($40.1\%$ vs. $32.4\%$), and the majority ($62.6\%$ vs. $31.6\%$) of these were unanswerable using the retrieved evidence. + +\subsection{Analysis and Discussion} +\label{sec:exp-ablations} +%\crcomment{Table~\ref{tab:exp-ablation-results} has been updated for clarification. } +We conducted analysis to assess baseline model performance under two conditions: (1) using ground-truth evidence (first block of Table~\ref{tab:exp-ablation-results}), and (2) disabling web-based evidence retrieval (second block of Table~\ref{tab:exp-ablation-results}). For both settings, we report conditional accuracy and justification generation scores with the evidence threshold set to $\lambda=0.3$. + +\noindent\textbf{Baselines' Performance with Ground-truth Evidence. } +The results obtained using golden evidence represent upper-bound performance, highlighting the models’ full potential. +Most baselines perform well in verdict prediction under this setting, underscoring both the importance and difficulty of effective evidence retrieval. +Notably, the Gemini-based baseline achieves the highest scores for both prediction and justification generation. +In contrast, baselines using LLaVA as the MLLM demonstrate the weakest performance. This is expected, as LLaVA~\citep{DBLP:journals/corr/abs-2407-07895} was not pre-trained on interleaved image-text documents. +Across all baselines, we observe consistent failure in identifying conflicting claims. This result aligns with prior findings on the challenge of detecting conflicting textual claims~\citep{DBLP:conf/nips/SchlichtkrullG023}. Moreover, our dataset contains a relatively small number of conflicting claims, which can cause instability in model performance. +It is also worth noting that our models do not explicitly model conflict within evidence, in contrast to~\citep{DBLP:conf/nips/SchlichtkrullG023}, which predicted verdicts per evidence piece and then examined whether these verdicts conflicted. +In our setup, we observe stronger dependencies between multiple QA pairs that individual QA pairs are often insufficient to support or refute a claim. For example, only by combining the first and second QA pairs in Figure~\ref{fig:intro-exp} can the model conclusively refute the claim. + +Although baselines used ground-truth evidence, they still achieved low justification generation scores. We attribute this to the limitations of ROUGE-1 for evaluating justification generation. To address this, we experimented with a reference-based evaluation method, Ev2R~\citep{DBLP:journals/corr/abs-2411-05375}, which has been shown to align well with human assessments in open-ended generation tasks. Using this approach, the justification scores were much higher and more encouraging (details in Appendix~\ref{sec:app-add-exp-justi-scores}). Moving forward, we plan to explore and incorporate more appropriate evaluation methods for justification generation. + +\noindent\textbf{Baselines' Performance without Searching. } +For the baseline without external evidence retrieval, we selected the combination of Qwen + Qwen-VL under PQG, as it demonstrated the strongest performance in question generation for claim verification. In this setting, all questions that require external search, those invoking the RIS, WST, or WSI tools, are marked with the response \textit{``No answer could be found.''}. +As shown in the second block of Table~\ref{tab:exp-ablation-results}, the model achieves an evidence score of $0.06$, reflecting a significant drop in evidence retrieval. +This result underscores the critical role of web-based information retrieval in real-world claim verification tasks. Nonetheless, in a few isolated cases, the model was still able to generate accurate evidence. These instances typically involved questions about locations or events depicted in the image, which the VQA tool could address correctly, likely because the model encountered these images during pretraining. +However, such behavior also raises concerns about potential data leakage from MLLMs. +Since the model's responses are not grounded in externally retrieved evidence, there is a risk that its predictions stem from memorized content or prior exposure to fact-checking articles during training. + +\section{Limitations} +\label{sec:limitations} +AVerImaTeC has a relatively limited scale, as it is sourced from real-world claims and constructed through detailed human annotation. This is consistent with other human-annotated datasets of real-world claims, such as AVeriTeC~\citep{DBLP:conf/nips/SchlichtkrullG023}, which also contain only a few thousand claims. +Since the claims in AVerImaTeC originate from fact-checking articles, the dataset may inherit biases inherent in these sources—for example, selection bias~\citep{10.1111/jcom.12284,Barnoy10122019}, leading to imbalanced label distributions. + +We made considerable efforts to prevent temporal leakage. Specifically, we provided annotators with custom search bars to retrieve online evidence published prior to the claim date. However, the exact dates of a small portion of claims ($5\%$) were unavailable. +In such cases, we used the dates of the corresponding fact-checking articles, which may have been published a few days after the original claims. +Additionally, we relied on Google Search and the Python package \textit{htmldate.find\_date} to estimate publication dates of web pages, which are coarse approximations. + +Furthermore, we exploited a reference-based evaluation strategy for both the generated questions and the retrieved evidence. Though it aligns well with human assessments, it has limitations in cases where model predictions are reasonable but not reflected in the reference annotations. +In such cases, evaluation scores may be undeservedly low due to poor alignment with the references. + + +\section{Ethical Statement} +\label{sec:ethics} +The datasets and models described in this paper are not intended for use in truth-telling tasks, such as automated content moderation systems. The labels and justifications included in the dataset reflect only the evidence recovered by annotators and are therefore subject to the biases of both annotators and journalists. In addition, the QA annotations may introduce framing bias. + +Annotators were instructed to prioritize evidence sources referenced in the original fact-checking articles, as we consider these sources, curated by professional fact-checkers, to be more trustworthy. Nonetheless, annotators were also permitted to draw on evidence retrieved by our customized search engine. +Using the list of common misinformative sources from~\citep{DBLP:conf/nips/SchlichtkrullG023}, we observed that 8 answers relied on a flagged source. However, this list was not applied during annotation, as it may be incomplete or contain false positives. Moreover, source credibility is often context-dependent, varying across topics and over time, which makes a static list insufficient for reliable filtering. + +We acknowledge that some fact-checking articles included in our dataset may have been exposed to large pre-trained models during their pre-training phase. This is an open and ongoing challenge in constructing human-annotated datasets. To help mitigate this issue, the splits of our dataset are temporally organized. In particular, if the training data of a language model is cut off prior to the temporal start of our test set, then data leakage from pre-training into evaluation cannot occur. + +We did not anonymize the data in AVerImaTeC, as all claims are derived from publicly available journalistic sources and primarily concern public figures and events. Preserving these references is essential to ensure the accuracy and integrity of fact-checking. +Nevertheless, we recognize that some individuals may not wish to appear in the dataset. Accordingly, we have established an opt-out policy: if any individual featured in the dataset, as a claim speaker, person depicted in an image, subject of a claim, or author of a fact-checking article underlying a claim, wishes to be removed, they may submit a request, and the relevant content will be deleted from the dataset. + +\section{Conclusion} +\label{sec:conclusion} + +We present a real-world image-text claim verification dataset, annotated with QA pairs that capture the reasoning and evidence retrieval processes involved in claim verification. +To ensure high annotation quality, we employed a multi-stage evidence sufficiency validation process, resulting in substantial inter-annotator agreement on both verdicts and QA annotations. +In addition, we introduce a reference-based evaluation framework for open-web multimodal evidence retrieval, along with a set of baseline models for image-text claim verification that leverage web-sourced evidence. +These contributions provide a foundation for advancing research in image-text claim verification. + +\clearpage + +\section*{Acknowledgement} +This research was supported by the Alan Turing Institute and DSO National Laboratories in Singapore Partnership (ref DCfP2\textbackslash100063). Zifeng Ding and Andreas Vlachos were further supported by the +ERC grant AVeriTeC (GA 865958). Andreas Vlachos is also supported by the DARPA program SciFy. +Michael Schlichtkrull is supported by the Engineering and Physical Sciences Research Council (grant number EP/Y009800/1), through funding from Responsible AI UK (KP0016). + +{ +\small +\bibliography{ref} +\bibliographystyle{plainnat} +} + + +\section*{NeurIPS Paper Checklist} + +%%% BEGIN INSTRUCTIONS %%% +%The checklist is designed to encourage best practices for responsible machine learning research, addressing issues of reproducibility, transparency, research ethics, and societal impact. Do not remove the checklist: {\bf The papers not including the checklist will be desk rejected.} The checklist should follow the references and follow the (optional) supplemental material. The checklist does NOT count towards the page +%limit. + +%Please read the checklist guidelines carefully for information on how to answer these questions. For each question in the checklist: +%\begin{itemize} +% \item You should answer \answerYes{}, \answerNo{}, or \answerNA{}. +% \item \answerNA{} means either that the question is Not Applicable for that particular paper or the relevant information is Not Available. +% \item Please provide a short (1–2 sentence) justification right after your answer (even for NA). + % \item {\bf The papers not including the checklist will be desk rejected.} +%\end{itemize} + +%{\bf The checklist answers are an integral part of your paper submission.} They are visible to the reviewers, area chairs, senior area chairs, and ethics reviewers. You will be asked to also include it (after eventual revisions) with the final version of your paper, and its final version will be published with the paper. + +%The reviewers of your paper will be asked to use the checklist as one of the factors in their evaluation. While "\answerYes{}" is generally preferable to "\answerNo{}", it is perfectly acceptable to answer "\answerNo{}" provided a proper justification is given (e.g., "error bars are not reported because it would be too computationally expensive" or "we were unable to find the license for the dataset we used"). In general, answering "\answerNo{}" or "\answerNA{}" is not grounds for rejection. While the questions are phrased in a binary way, we acknowledge that the true answer is often more nuanced, so please just use your best judgment and write a justification to elaborate. All supporting evidence can appear either in the main paper or the supplemental material, provided in appendix. If you answer \answerYes{} to a question, in the justification please point to the section(s) where related material for the question can be found. + +%IMPORTANT, please: +%\begin{itemize} +% \item {\bf Delete this instruction block, but keep the section heading ``NeurIPS Paper Checklist"}, +% \item {\bf Keep the checklist subsection headings, questions/answers and guidelines below.} +% \item {\bf Do not modify the questions and only use the provided macros for your answers}. +%\end{itemize} + + +%%% END INSTRUCTIONS %%% + + +\begin{enumerate} + +\item {\bf Claims} + \item[] Question: Do the main claims made in the abstract and introduction accurately reflect the paper's contributions and scope? + \item[] Answer: \answerYes{} % Replace by \answerYes{}, \answerNo{}, or \answerNA{}. + \item[] Justification: Yes, main claims made in the abstract and introduction accurately reflect the +paper’s contributions and scope. + \item[] Guidelines: + \begin{itemize} + \item The answer NA means that the abstract and introduction do not include the claims made in the paper. + \item The abstract and/or introduction should clearly state the claims made, including the contributions made in the paper and important assumptions and limitations. A No or NA answer to this question will not be perceived well by the reviewers. + \item The claims made should match theoretical and experimental results, and reflect how much the results can be expected to generalize to other settings. + \item It is fine to include aspirational goals as motivation as long as it is clear that these goals are not attained by the paper. + \end{itemize} + +\item {\bf Limitations} + \item[] Question: Does the paper discuss the limitations of the work performed by the authors? + \item[] Answer: \answerYes{} % Replace by \answerYes{}, \answerNo{}, or \answerNA{}. + \item[] Justification: Limitations of the work are discussed in Section~\ref{sec:limitations}. + \item[] Guidelines: + \begin{itemize} + \item The answer NA means that the paper has no limitation while the answer No means that the paper has limitations, but those are not discussed in the paper. + \item The authors are encouraged to create a separate "Limitations" section in their paper. + \item The paper should point out any strong assumptions and how robust the results are to violations of these assumptions (e.g., independence assumptions, noiseless settings, model well-specification, asymptotic approximations only holding locally). The authors should reflect on how these assumptions might be violated in practice and what the implications would be. + \item The authors should reflect on the scope of the claims made, e.g., if the approach was only tested on a few datasets or with a few runs. In general, empirical results often depend on implicit assumptions, which should be articulated. + \item The authors should reflect on the factors that influence the performance of the approach. For example, a facial recognition algorithm may perform poorly when image resolution is low or images are taken in low lighting. Or a speech-to-text system might not be used reliably to provide closed captions for online lectures because it fails to handle technical jargon. + \item The authors should discuss the computational efficiency of the proposed algorithms and how they scale with dataset size. + \item If applicable, the authors should discuss possible limitations of their approach to address problems of privacy and fairness. + \item While the authors might fear that complete honesty about limitations might be used by reviewers as grounds for rejection, a worse outcome might be that reviewers discover limitations that aren't acknowledged in the paper. The authors should use their best judgment and recognize that individual actions in favor of transparency play an important role in developing norms that preserve the integrity of the community. Reviewers will be specifically instructed to not penalize honesty concerning limitations. + \end{itemize} + +\item {\bf Theory assumptions and proofs} + \item[] Question: For each theoretical result, does the paper provide the full set of assumptions and a complete (and correct) proof? + \item[] Answer: \answerNA{} + %Replace by \answerYes{}, \answerNo{}, or \answerNA{}. + \item[] Justification: There are no theoretical results in the paper. + \item[] Guidelines: + \begin{itemize} + \item The answer NA means that the paper does not include theoretical results. + \item All the theorems, formulas, and proofs in the paper should be numbered and cross-referenced. + \item All assumptions should be clearly stated or referenced in the statement of any theorems. + \item The proofs can either appear in the main paper or the supplemental material, but if they appear in the supplemental material, the authors are encouraged to provide a short proof sketch to provide intuition. + \item Inversely, any informal proof provided in the core of the paper should be complemented by formal proofs provided in appendix or supplemental material. + \item Theorems and Lemmas that the proof relies upon should be properly referenced. + \end{itemize} + + \item {\bf Experimental result reproducibility} + \item[] Question: Does the paper fully disclose all the information needed to reproduce the main experimental results of the paper to the extent that it affects the main claims and/or conclusions of the paper (regardless of whether the code and data are provided or not)? + \item[] Answer: \answerYes{} % Replace by \answerYes{}, \answerNo{}, or \answerNA{}. + \item[] Justification: The experiment settings and implementation details can be found in Section~\ref{sec:exp-baseline} (Model Implementation) and Appendix~\ref{sec:app-detail-model-imp}. + \item[] Guidelines: + \begin{itemize} + \item The answer NA means that the paper does not include experiments. + \item If the paper includes experiments, a No answer to this question will not be perceived well by the reviewers: Making the paper reproducible is important, regardless of whether the code and data are provided or not. + \item If the contribution is a dataset and/or model, the authors should describe the steps taken to make their results reproducible or verifiable. + \item Depending on the contribution, reproducibility can be accomplished in various ways. For example, if the contribution is a novel architecture, describing the architecture fully might suffice, or if the contribution is a specific model and empirical evaluation, it may be necessary to either make it possible for others to replicate the model with the same dataset, or provide access to the model. In general. releasing code and data is often one good way to accomplish this, but reproducibility can also be provided via detailed instructions for how to replicate the results, access to a hosted model (e.g., in the case of a large language model), releasing of a model checkpoint, or other means that are appropriate to the research performed. + \item While NeurIPS does not require releasing code, the conference does require all submissions to provide some reasonable avenue for reproducibility, which may depend on the nature of the contribution. For example + \begin{enumerate} + \item If the contribution is primarily a new algorithm, the paper should make it clear how to reproduce that algorithm. + \item If the contribution is primarily a new model architecture, the paper should describe the architecture clearly and fully. + \item If the contribution is a new model (e.g., a large language model), then there should either be a way to access this model for reproducing the results or a way to reproduce the model (e.g., with an open-source dataset or instructions for how to construct the dataset). + \item We recognize that reproducibility may be tricky in some cases, in which case authors are welcome to describe the particular way they provide for reproducibility. In the case of closed-source models, it may be that access to the model is limited in some way (e.g., to registered users), but it should be possible for other researchers to have some path to reproducing or verifying the results. + \end{enumerate} + \end{itemize} + + +\item {\bf Open access to data and code} + \item[] Question: Does the paper provide open access to the data and code, with sufficient instructions to faithfully reproduce the main experimental results, as described in supplemental material? + \item[] Answer: \answerYes{} % Replace by \answerYes{}, \answerNo{}, or \answerNA{}. + \item[] Justification: Both the data and code are accessible to reviewers (see Appendix~\ref{sec:data_stat}). As we anticipate using the dataset in a future shared task, we are as of submission +time only releasing the training and development splits. We will make the test split available privately +to reviewers upon request. We will make our data and code publicly available and maintain them on GitHub upon acceptance. + \item[] Guidelines: + \begin{itemize} + \item The answer NA means that paper does not include experiments requiring code. + \item Please see the NeurIPS code and data submission guidelines (\url{https://nips.cc/public/guides/CodeSubmissionPolicy}) for more details. + \item While we encourage the release of code and data, we understand that this might not be possible, so “No” is an acceptable answer. Papers cannot be rejected simply for not including code, unless this is central to the contribution (e.g., for a new open-source benchmark). + \item The instructions should contain the exact command and environment needed to run to reproduce the results. See the NeurIPS code and data submission guidelines (\url{https://nips.cc/public/guides/CodeSubmissionPolicy}) for more details. + \item The authors should provide instructions on data access and preparation, including how to access the raw data, preprocessed data, intermediate data, and generated data, etc. + \item The authors should provide scripts to reproduce all experimental results for the new proposed method and baselines. If only a subset of experiments are reproducible, they should state which ones are omitted from the script and why. + \item At submission time, to preserve anonymity, the authors should release anonymized versions (if applicable). + \item Providing as much information as possible in supplemental material (appended to the paper) is recommended, but including URLs to data and code is permitted. + \end{itemize} + + +\item {\bf Experimental setting/details} + \item[] Question: Does the paper specify all the training and test details (e.g., data splits, hyperparameters, how they were chosen, type of optimizer, etc.) necessary to understand the results? + \item[] Answer: \answerYes{} % Replace by \answerYes{}, \answerNo{}, or \answerNA{}. + \item[] Justification: + \item[] Guidelines: The information for data splits can be found in Section~\ref{sec:data_stat}. Testing details are provided in Appendix~\ref{sec:app-detail-model-imp}. + \begin{itemize} + \item The answer NA means that the paper does not include experiments. + \item The experimental setting should be presented in the core of the paper to a level of detail that is necessary to appreciate the results and make sense of them. + \item The full details can be provided either with the code, in appendix, or as supplemental material. + \end{itemize} + +\item {\bf Experiment statistical significance} + \item[] Question: Does the paper report error bars suitably and correctly defined or other appropriate information about the statistical significance of the experiments? + \item[] Answer: \answerNA{} % Replace by \answerYes{}, \answerNo{}, or \answerNA{}. + \item[] Justification: Our goal is not to advance in model design but to establish baselines for image-text claim verification. Besides, following the practice of~\citep{DBLP:conf/nips/SchlichtkrullG023} and prior FEVER workshops (https://fever.ai/workshop.html), we do not require multiple runs of each model. + \item[] Guidelines: + \begin{itemize} + \item The answer NA means that the paper does not include experiments. + \item The authors should answer "Yes" if the results are accompanied by error bars, confidence intervals, or statistical significance tests, at least for the experiments that support the main claims of the paper. + \item The factors of variability that the error bars are capturing should be clearly stated (for example, train/test split, initialization, random drawing of some parameter, or overall run with given experimental conditions). + \item The method for calculating the error bars should be explained (closed form formula, call to a library function, bootstrap, etc.) + \item The assumptions made should be given (e.g., Normally distributed errors). + \item It should be clear whether the error bar is the standard deviation or the standard error of the mean. + \item It is OK to report 1-sigma error bars, but one should state it. The authors should preferably report a 2-sigma error bar than state that they have a 96\% CI, if the hypothesis of Normality of errors is not verified. + \item For asymmetric distributions, the authors should be careful not to show in tables or figures symmetric error bars that would yield results that are out of range (e.g. negative error rates). + \item If error bars are reported in tables or plots, The authors should explain in the text how they were calculated and reference the corresponding figures or tables in the text. + \end{itemize} + +\item {\bf Experiments compute resources} + \item[] Question: For each experiment, does the paper provide sufficient information on the computer resources (type of compute workers, memory, time of execution) needed to reproduce the experiments? + \item[] Answer: \answerYes{} % Replace by \answerYes{}, \answerNo{}, or \answerNA{}. + \item[] Justification: The information for computation resources can be found in Appendix~\ref{sec:app-detail-model-imp}. + \item[] Guidelines: + \begin{itemize} + \item The answer NA means that the paper does not include experiments. + \item The paper should indicate the type of compute workers CPU or GPU, internal cluster, or cloud provider, including relevant memory and storage. + \item The paper should provide the amount of compute required for each of the individual experimental runs as well as estimate the total compute. + \item The paper should disclose whether the full research project required more compute than the experiments reported in the paper (e.g., preliminary or failed experiments that didn't make it into the paper). + \end{itemize} + +\item {\bf Code of ethics} + \item[] Question: Does the research conducted in the paper conform, in every respect, with the NeurIPS Code of Ethics \url{https://neurips.cc/public/EthicsGuidelines}? + \item[] Answer: \answerYes{} % Replace by \answerYes{}, \answerNo{}, or \answerNA{}. + \item[] Justification: The research in the paper conforms, in every respect, with the NeurIPS Code of Ethics. + \item[] Guidelines: + \begin{itemize} + \item The answer NA means that the authors have not reviewed the NeurIPS Code of Ethics. + \item If the authors answer No, they should explain the special circumstances that require a deviation from the Code of Ethics. + \item The authors should make sure to preserve anonymity (e.g., if there is a special consideration due to laws or regulations in their jurisdiction). + \end{itemize} + + +\item {\bf Broader impacts} + \item[] Question: Does the paper discuss both potential positive societal impacts and negative societal impacts of the work performed? + \item[] Answer: \answerYes{} % Replace by \answerYes{}, \answerNo{}, or \answerNA{}. + \item[] Justification: The potential societal impact of the work has been discussed in Section~\ref{sec:intro} and Section~\ref{sec:related}. + \item[] Guidelines: + \begin{itemize} + \item The answer NA means that there is no societal impact of the work performed. + \item If the authors answer NA or No, they should explain why their work has no societal impact or why the paper does not address societal impact. + \item Examples of negative societal impacts include potential malicious or unintended uses (e.g., disinformation, generating fake profiles, surveillance), fairness considerations (e.g., deployment of technologies that could make decisions that unfairly impact specific groups), privacy considerations, and security considerations. + \item The conference expects that many papers will be foundational research and not tied to particular applications, let alone deployments. However, if there is a direct path to any negative applications, the authors should point it out. For example, it is legitimate to point out that an improvement in the quality of generative models could be used to generate deepfakes for disinformation. On the other hand, it is not needed to point out that a generic algorithm for optimizing neural networks could enable people to train models that generate Deepfakes faster. + \item The authors should consider possible harms that could arise when the technology is being used as intended and functioning correctly, harms that could arise when the technology is being used as intended but gives incorrect results, and harms following from (intentional or unintentional) misuse of the technology. + \item If there are negative societal impacts, the authors could also discuss possible mitigation strategies (e.g., gated release of models, providing defenses in addition to attacks, mechanisms for monitoring misuse, mechanisms to monitor how a system learns from feedback over time, improving the efficiency and accessibility of ML). + \end{itemize} + +\item {\bf Safeguards} + \item[] Question: Does the paper describe safeguards that have been put in place for responsible release of data or models that have a high risk for misuse (e.g., pretrained language models, image generators, or scraped datasets)? + \item[] Answer: \answerNA{} % Replace by \answerYes{}, \answerNo{}, or \answerNA{}. + \item[] Justification: The paper poses no such risks. + \item[] Guidelines: + \begin{itemize} + \item The answer NA means that the paper poses no such risks. + \item Released models that have a high risk for misuse or dual-use should be released with necessary safeguards to allow for controlled use of the model, for example by requiring that users adhere to usage guidelines or restrictions to access the model or implementing safety filters. + \item Datasets that have been scraped from the Internet could pose safety risks. The authors should describe how they avoided releasing unsafe images. + \item We recognize that providing effective safeguards is challenging, and many papers do not require this, but we encourage authors to take this into account and make a best faith effort. + \end{itemize} + +\item {\bf Licenses for existing assets} + \item[] Question: Are the creators or original owners of assets (e.g., code, data, models), used in the paper, properly credited and are the license and terms of use explicitly mentioned and properly respected? + \item[] Answer: \answerYes{} % Replace by \answerYes{}, \answerNo{}, or \answerNA{}. + \item[] Justification: We used partially annotated data from AVeriTeC, AMMeBA and ClaimReview. They are explictly mentioned and discussed in Appendix~\ref{sec:app-state-dataset} + \item[] Guidelines: + \begin{itemize} + \item The answer NA means that the paper does not use existing assets. + \item The authors should cite the original paper that produced the code package or dataset. + \item The authors should state which version of the asset is used and, if possible, include a URL. + \item The name of the license (e.g., CC-BY 4.0) should be included for each asset. + \item For scraped data from a particular source (e.g., website), the copyright and terms of service of that source should be provided. + \item If assets are released, the license, copyright information, and terms of use in the package should be provided. For popular datasets, \url{paperswithcode.com/datasets} has curated licenses for some datasets. Their licensing guide can help determine the license of a dataset. + \item For existing datasets that are re-packaged, both the original license and the license of the derived asset (if it has changed) should be provided. + \item If this information is not available online, the authors are encouraged to reach out to the asset's creators. + \end{itemize} + +\item {\bf New assets} + \item[] Question: Are new assets introduced in the paper well documented and is the documentation provided alongside the assets? + \item[] Answer: \answerYes{} % Replace by \answerYes{}, \answerNo{}, or \answerNA{}. + \item[] Justification: + The dataset and code have been made available to reviewers. + Our dataset and baseline will be made publicly +available under a CC-BY-NC-4.0 license if accepted. Detailed dataset information can be found in Section~\ref{sec:anno_struct},~\ref{sec:anno_pipe} and~\ref{sec:data_stat}. + \item[] Guidelines: + \begin{itemize} + \item The answer NA means that the paper does not release new assets. + \item Researchers should communicate the details of the dataset/code/model as part of their submissions via structured templates. This includes details about training, license, limitations, etc. + \item The paper should discuss whether and how consent was obtained from people whose asset is used. + \item At submission time, remember to anonymize your assets (if applicable). You can either create an anonymized URL or include an anonymized zip file. + \end{itemize} + +\item {\bf Crowdsourcing and research with human subjects} + \item[] Question: For crowdsourcing experiments and research with human subjects, does the paper include the full text of instructions given to participants and screenshots, if applicable, as well as details about compensation (if any)? + \item[] Answer: \answerYes{} % Replace by \answerYes{}, \answerNo{}, or \answerNA{}. + \item[] Justification: Data collection was carried out with the help of Appen (https://appen.com/). The payment details and annotator demographics can be found in Appendix~\ref{sec:app-anno-detail-annotator-demo}. Guidelines for annotators are provided Appendix~\ref{sec:app-annotation-guidelines}. + \item[] Guidelines: + \begin{itemize} + \item The answer NA means that the paper does not involve crowdsourcing nor research with human subjects. + \item Including this information in the supplemental material is fine, but if the main contribution of the paper involves human subjects, then as much detail as possible should be included in the main paper. + \item According to the NeurIPS Code of Ethics, workers involved in data collection, curation, or other labor should be paid at least the minimum wage in the country of the data collector. + \end{itemize} + +\item {\bf Institutional review board (IRB) approvals or equivalent for research with human subjects} + \item[] Question: Does the paper describe potential risks incurred by study participants, whether such risks were disclosed to the subjects, and whether Institutional Review Board (IRB) approvals (or an equivalent approval/review based on the requirements of your country or institution) were obtained? + \item[] Answer: \answerNA{} % Replace by \answerYes{}, \answerNo{}, or \answerNA{}. + \item[] Justification: We do not involve human subjects. All annotators are recruited by a company, Appen (See Appendix~\ref{sec:app-anno-detail-annotator-demo}). + \item[] Guidelines: + \begin{itemize} + \item The answer NA means that the paper does not involve crowdsourcing nor research with human subjects. + \item Depending on the country in which research is conducted, IRB approval (or equivalent) may be required for any human subjects research. If you obtained IRB approval, you should clearly state this in the paper. + \item We recognize that the procedures for this may vary significantly between institutions and locations, and we expect authors to adhere to the NeurIPS Code of Ethics and the guidelines for their institution. + \item For initial submissions, do not include any information that would break anonymity (if applicable), such as the institution conducting the review. + \end{itemize} + +\item {\bf Declaration of LLM usage} + \item[] Question: Does the paper describe the usage of LLMs if it is an important, original, or non-standard component of the core methods in this research? Note that if the LLM is used only for writing, editing, or formatting purposes and does not impact the core methodology, scientific rigorousness, or originality of the research, declaration is not required. + %this research? + \item[] Answer: \answerNA{} % Replace by \answerYes{}, \answerNo{}, or \answerNA{}. + \item[] Justification: The core method development in this research does not involve LLMs as any important, original, or non-standard components. LLMs are only used for grammar error correction. + \item[] Guidelines: + \begin{itemize} + \item The answer NA means that the core method development in this research does not involve LLMs as any important, original, or non-standard components. + \item Please refer to our LLM policy (\url{https://neurips.cc/Conferences/2025/LLM}) for what should or should not be described. + \end{itemize} + +\end{enumerate} + + +\clearpage +\appendix +%% WARNING: missing file /Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/job_1785098515_4b6a9bbd/processed_papers/2505.17978/tex_2505.17978/Sections/Appendix.tex + +\section{Statement of Dataset} +\label{sec:app-state-dataset} +We provide the training and development splits of our dataset for review.\footnote{https://kaggle.com/datasets/a1ebcc27f233ca16e80fa7007d1a8cea6c491656d7b4ca6d87531df61c5089fd} +However, due to naming inconsistencies with Kaggle, we are unable to upload the images there. Instead, we have made the image files available on Harvard Dataverse.\footnote{https://dataverse.harvard.edu/previewurl.xhtml?token=5fbd9e3c-bc9e-49d7-a109-eace6f1bdacd} +We cannot upload the entire dataset (including the .json file and images) to Harvard Dataverse, as it requires publishing the dataset when generating the Croissant file. +To avoid unexpected inaccessibility of images on Harvard Dataverse, we also upload the whole development set (images and json file) to \textbf{supplementary}, due to file size limits. + +Since we plan to use the test split for future shared tasks, we are withholding it at the time of submission. However, the test split will be made privately available to reviewers upon request. + +If accepted, we will publicly release the training and development splits of the dataset and maintain both the data and baseline code on Harvard Dataverse and GitHub. The dataset and baselines will be licensed under the CC BY-NC 4.0 license. + + +\section{Annotation Platform and Guidelines} +\label{sec:app-annotation-guidelines} +The annotation was performed using a custom-developed web platform specifically designed for this task from our team. We will make the platform’s source code available upon request. + +Annotators were provided with detailed guidelines outlining the annotation process. +Due to the length of the guideline, we attached it to the \textbf{supplementary}. +After reviewing the instructions, they underwent training for each phase. +Specifically, we provide $14$, $12$ and $12$ training instances for the first three phases, respectively. Phases four and five replicate phases two and three; therefore, no additional training instances were provided. +We offered continuous feedback to annotators throughout both the training and annotation phases. + +\section{Annotation Fees and Annotator Demographics} +\label{sec:app-anno-detail-annotator-demo} +The annotation was conducted with the assistance of Appen,\footnote{https://www.appen.com/} a private company that provides machine learning services. Annotators were recruited through the company, which ensures fair compensation practices in accordance with their fair pay guidelines.\footnote{https://success.appen.com/hc/en-us/articles/9557008940941-Guide-to-Fair-Pay} +A total of 20 annotators participated in the project, 12 women ($60\%$) and 8 men ($40\%$). +Fourteen annotators were based in the United States and six in the United Kingdom. +Regarding age distribution, $20\%$ of annotators were between 18 and 30 years old, $35\%$ between 31 and 45, $25\%$ were older than 45, and $20\%$ chose not to disclose their age. + + +\section{Additional Dataset Information} +\label{sec:app-data-info} +In this section, we provide additional dataset statistics. +\subsection{Statistics for Data Filtering in Phase One} +\label{sec:app-data-p1-data-filter} +We began with $2,353$ fact-checking articles in Phase One (P1) of the annotation process. To increase the proportion of image-text claims, we incorporated partially annotated articles from three sources: +1) filtered articles from AVeriTec~\citep{DBLP:conf/nips/SchlichtkrullG023} containing multimodal claims, 2) articles verifying image-related claims from AMMEBA~\citep{DBLP:journals/corr/abs-2405-11697} and 3) all \textit{true} claims from ClaimReview\footnote{https://developers.google.com/fact-check/tools/api} over the past two years that included keywords such as \textit{phtos} or \textit{pictures}. +Among the 2,353 articles, we identified 91 duplicate claims (e.g., those verified by multiple fact-checking organizations). +Duplicates were detected automatically by computing cosine similarity and Jaccard distance between N-gram representations of textual part of claims. +We manually reviewed claim pairs with a cosine similarity or Jaccard distance greater than 0.3. +Despite these efforts, a small number of duplicates (fewer than 5\%) remained, as fact-checkers may paraphrase similar claims differently. +However, due to the temporal split of our dataset, such duplication does not cause data leakage across the splits. Paywalled or inaccessible articles were reported by annotators and subsequently discarded. + +Although we leveraged coarsely annotated articles that verify multimodal claims, not all of them involved image-text relationships. +In the first step, we filtered out claims of other modalities and retained only the image-text claims (around $85\%$) for further P1 annotation, which included image uploading, metadata collection, and verdict annotation. +We further filter out image-text claims, where images are not used in claim verification (14.7\%) or which involve image manipulations(4.0\%). Both types were excluded from subsequent phases. +Following~\citep{DBLP:conf/nips/SchlichtkrullG023}, we discard \textit{speculative} claims, personal opinions and claims relying solely on \textit{fact-checking reference} as the fact-checking strategy. +After the P1 annotation, we retained a total of $\mathbf{1,457}$ valid image-text claims, for further processing in subsequent phases. + +\subsection{Discarded Claims} +\label{sec:app-discard-claims} +In Phase Two, annotators can report and skip invalid claims, particularly when images associated with claims are inaccessible due to regional restrictions. A total of $48$ such cases were discarded. + +As described in Section~\ref{sec:anno_pipe}, +when there are conflicting verdicts between annotators in Phases Two and Three, the affected claims proceed to Phases Four and Five for quality assurance updates and a second round of evidence sufficiency checks. In our dataset, $14.6\%$ of claims required this second round of evaluation. + +Claims with unresolved conflicts after the five-phase annotation process are also discarded. Overall, approximately $7\%$ of claims were excluded due to irreconcilable disagreements in annotated verdicts, even after the second round of evidence sufficiency checks. + +\subsection{Metadata Distributions} +\label{sec:app-data-statistics-metadatao} +Following~\citep{DBLP:conf/nips/SchlichtkrullG023}, we provide details of distributions of metadata of data splits in this part. + +\begin{table*}[ht!] +\centering +\caption{\textbf{Distributions of claim types.} } + \label{tab:app-meta-claim-type} + \begin{tabular}{lccc} + \toprule + \textbf{Claim Type} &\textbf{Train} &\textbf{Dev} &\textbf{Test} \\ + \midrule + Event/Property Claim &85.4 &91.4 &93.5\\ + Causal Claim & 5.7&0.7 & 3.4\\ + Numerical Claim & 3.2&3.9 &1.4 \\ + Media Analysis Claim &21.1&14.5 &30.1 \\ + Media Publishing Claim & 2.1& 2.0&2.0 \\ + Position Statement & 1.2&0.7 & 0.3\\ + Quote Verification & 1.4& 0.7& 0.3\\ + \bottomrule +\end{tabular} +\end{table*} + +\begin{table*}[ht!] +\centering +\caption{\textbf{Distributions of fact-checking strategies.} } + \label{tab:app-meta-factcheck} + \begin{tabular}{lccc} + \toprule + \textbf{Claim Type} &\textbf{Train} &\textbf{Dev} &\textbf{Test} \\ + \midrule + Media Source Discovery &26.4 &20.4 &26.4 \\ + Keyword Search &19.5 & 19.7 &14.8 \\ + Written Evidence & 87.5& 89.5&84.7 \\ + Reverse Image Search &50.2 & 57.9 & 67.3\\ + Image Analysis &20.7 & 24.3 & 21.3\\ + Consultation &30.0 &24.3 & 16.8\\ + Video Analysis & 6.8&4.6 & 8.2\\ + Numerical Comparison & 2.9& 3.3&1.4 \\ + Satirical Source Identification &1.9 & 3.3&1.4 \\ + Fact-checker Reference & 11.0&11.8 &3.7 \\ + Geolocation & 4.0&5.3&5.4\\ + \bottomrule +\end{tabular} +\end{table*} + +\noindent\textbf{Metadata about Claims. }We present the distribution of claim types and fact-checking strategies in Table~\ref{tab:app-meta-claim-type} and~\ref{tab:app-meta-factcheck}, respectively. It is important to note that a single claim may belong to multiple claim types and can be verified using several fact-checking strategies. +\begin{table*}[ht!] +\centering +\caption{\textbf{Counts of locations associated with claims.} Countries are represented with their ISO country code. We do not show countries with fewer than ten occurrences in the table, whereas the complete location information is available in our dataset.} + \label{tab:app-meta-location} + \begin{tabular}{lccc} + \toprule + \textbf{Country code} &\textbf{Counts} \\ + \midrule + IN & 417 \\ + US & 309 \\ + GB & 74 \\ + UA &37 \\ + PK & 30 \\ + IL &27\\ + NG &20\\ + LK &19 \\ + TR &18 \\ + PS &17 \\ + KR &17\\ + KE &15\\ + AU &14 \\ + RU &13 \\ + BD &12 \\ + JP &11 \\ + TH &10 \\ + MY & 10 \\ + CN &10 \\ + CA &10\\ + CR &10 \\ + \bottomrule +\end{tabular} +\end{table*} +The most relevant locations associated with claims are listed in Table~\ref{tab:app-meta-location}. +We observed a bias in the geographic distribution of claims. As our dataset includes claims verified by fact-checkers, it may inherit any biases present in the original fact-checking sources~\citep{10.1111/jcom.12284,Barnoy10122019}. + +\begin{table*}[ht!] +\centering +\caption{\textbf{Distributions of question types.} } + \label{tab:app-meta-qt} + \begin{tabular}{lccc} + \toprule + \textbf{Question Type} &\textbf{Train} &\textbf{Dev} &\textbf{Test} \\ + \midrule + Text-related & 35.2 & 38.9& 31.8 \\ + Image-related &63.4 & 58.8&62.2 \\ + Metadata-related & 3.2 &4.4 & 8.3 \\ + Commonsense-related & 0.8 &0.9 & 1.1 \\ + \bottomrule +\end{tabular} +\end{table*} + +\begin{table*}[ht!] +\centering +\caption{\textbf{Distributions of metadata related to answers.} The first block is the distribution for answering methods, the second for answer types and the third for source medium.} + \label{tab:app-meta-ans} + \begin{tabular}{lccc} + \toprule + \textbf{Question Type} &\textbf{Train} &\textbf{Dev} &\textbf{Test} \\ + \midrule + Image-search &54.6 & 54.6&51.9\\ + Text-search &40.1&41.0 &37.4\\ + Metadata & 1.8 &3.0 &7.7\\ + Image Analysis &3.2 & 1.4&2.5\\ + \midrule + Abstractive & 16.4 &15.0 & 19.1 \\ + Extractive & 57.9&61.3 & 58.9\\ + Unanswerable &2.2 &3.2 & 3.2 \\ + Boolean & 21.9& 18.8& 17.3 \\ + Image & 1.6 &1.6 &1.5\\ + \midrule + Web text & 85.2&85.2 &78.9\\ + PDF & 0.8 &1.2 & 0.8 \\ + Metadata & 1.2 & 2.1& 7.1 \\ + Video & 2.5&1.2 &2.9 \\ + Image/graphic & 4.4 & 5.3& 4.6 \\ + Web table & 0.3 & 0.5& 0.1 \\ + Other & 0.2 &0.0 & 0.1 \\ + \bottomrule +\end{tabular} +\end{table*} +\noindent\textbf{Metadata about QA Annotations. } +In addition to metadata for claims, we also provide metadata associated with the annotated QA pairs, including question types, answering methods, answer types and source mediums. +Metadata statistics for questions and answers are presented in Table~\ref{tab:app-meta-qt} and Table~\ref{tab:app-meta-ans}, respectively. +Notably, a single question may belong to multiple question types. + +\section{Inter-Annotator Agreement} +\label{sec:app-inter-annotator-agreement} +\begin{figure*}[ht!] + \centering + \includegraphics[width=\linewidth]{app-human-eval-inter.pdf} + \caption{ + \textbf{Platform and instructions for validating annotators' agreement on QA annotations.}} + \label{fig:app-human-eval-inter} +\end{figure*} +To assess the quality of our annotated data, we recruited a different set of annotators to re-annotate $100$ randomly sampled claims from our dataset and performed an inter-annotator agreement check. +During re-annotation, we assumed that claim extraction and normalization had already been completed, and the annotators proceeded with the remaining phases. We ensured that the sample included at least five claims for each veracity label. +The inter-annotator agreement check was done regarding both the agreement on verdicts and agreements on annotated QA pairs. + +\noindent\textbf{Verdict Agreement. } +To evaluate agreement on verdicts, we used Randolph's~\citep{10.1007/s11634-010-0073-4} free-marginal multi-rater $\kappa$, which is well-suited for unbalanced datasets, following previous practices~\citep{DBLP:conf/nips/SchlichtkrullG023,DBLP:conf/emnlp/OusidhoumY022}. We achieved an agreement score of 0.742 on the double-annotated claims. + +\noindent\textbf{QA Pair Agreement. } +For assessing agreement on annotated QA pairs, we recruited three best performing annotators to compare the similarities between the original and re-annotated QA annotations. Specifically, we instructed them to evaluate 1) whether the annotated verdict from our dataset could be supported by the original QA pairs; 2) how many original QA pairs for a claim are covered by the re-annotated QA pairs, and 3) how many QA pairs from the re-annotation are covered by the original QA pairs. The platform used for this agreement check is shown in Figure~\ref{fig:app-human-eval-inter}. + +\noindent\textbf{Justification Evaluation. } +Although we primarily focused on verdict and QA annotation agreement, we also conducted a small-scale human evaluation to assess the quality of justifications. We randomly sampled 20 justifications from our dataset and from~\citep{DBLP:conf/nips/SchlichtkrullG023}, and asked a human evaluator to rate them on a scale from 0 to 5. Our justifications achieved an average score of 4.2, compared to an average score of 1.95 from~\citep{DBLP:conf/nips/SchlichtkrullG023}. + +\section{Human Alignment of Evaluation Metrics} +\label{sec:app-human-align-eval-metrics} +Following~\citep{DBLP:conf/nips/SchlichtkrullG023}, we evaluate the quality of generated questions and retrieved evidence. Motivated by a recent study~\citep{DBLP:journals/corr/abs-2411-05375}, +we adopt a reference-based evaluation method that compares model responses to human-annotated ground-truth data. This method is applied to both question and evidence evaluations. +To assess the reliability of the reference-based evaluation, we compare the resulting scores with human judgments obtained from independent raters (see Appendix~\ref{sec:app-eval-ques-eval} and \ref{sec:app-eval-evid-eval}). +Additionally, we conduct \textit{checklist} tests to evaluate the sensitivity of the reference-based evidence evaluation method, following the approach outlined in~\citep{DBLP:journals/corr/abs-2411-05375} (Appendix~\ref{sec:app-eval-robust}). + +\subsection{Alignment Check on Question Evaluation. } +\label{sec:app-eval-ques-eval} +Regarding question evaluation, we focus on assessing the semantics (i.e., textual content) of the questions. This evaluation approach aligns with that used in~\citep{DBLP:journals/corr/abs-2411-05375} and has also been applied at the FEVER workshop,\footnote{https://huggingface.co/spaces/fever/AVeriTeCFEVER8} demonstrating its reliability and strong correlation with human ratings. + +To further ensure alignment with human judgment in our setting, we conducted a small-scale human evaluation to validate the reference-based evaluation of questions. Specifically, we invited two NLP researchers (both authors of this paper) to participate in the assessment. They were presented with model-generated questions from the first 20 testing claims in our dataset and were asked to compare these questions against ground-truth annotated questions. The comparison was made based on two criteria: +\textbf{Relevance} to claim verification (how relevant the generated questions are for verifying the claim) and \textbf{Coverage} of the ground-truth questions (how many ground-truth questions are accurately covered by the predicted questions.). +The agreement between human and the evaluation metric was measured using +Spearman ($\rho$)~\citep{spearman04}, achieving a value of \textbf{0.705}, and Pearson correlation coefficients ($r$)~\citep{pearson1896mathematical} achieving a value of \textbf{0.791}. These results indicate a strong alignment between human judgments and the reference-based evaluation scores on question evaluation. + +\begin{table*}[ht!] +\centering +\caption{\textbf{Correlation between human evaluation and automatic evaluation metrics.} We presents both the Spearman ($\rho$) and Pearson ($r$) correlation coefficients.} + \label{tab:app-align-evid-eval} + \begin{tabular}{lccc} + \toprule + \textbf{Scorer} &\textbf{Text-only} &\textbf{Interleaved} &\textbf{Separate} \\ + \midrule + $\rho$ &0.263 &0.08 &0.332 \\ + $r$ &0.286 & 0.14&0.381\\ + \bottomrule +\end{tabular} +\end{table*} + +\subsection{Alignment Check on Evidence Evaluation. } +\label{sec:app-eval-evid-eval} +We leveraged re-annotation and re-used the human judgment on QA annotations as described in Appendix~\ref{sec:app-inter-annotator-agreement} to validate the alignment between our automatic evidence evaluation method and human judgment. +To assess alignment, we treated the original annotations from our dataset as predictions and the re-annotations as ground truth. We then applied the automatic evaluation methods to compare these \textit{``predictions''} against the ground truth. Subsequently, we computed alignment scores by comparing the recall obtained from the automatic evaluation methods with human annotations. +The correlation between the automatic evaluation scores and human assessments is presented using Spearman ($\rho$) and Pearson ($r$) correlation coefficients, as shown in Table~\ref{tab:app-align-evid-eval}. +Our analysis shows that the separated evaluation method aligns most closely with human assessments, demonstrating the effectiveness of our approach. + + +\begin{table*}[ht!] +\centering + +\caption{\textbf{Adversarial attack results for the assessment of the robustness of our evaluation method.} Results in the table are obtained by computing the evidence evaluation score difference (in \%) between initial evidence and +manipulated evidence with checklists.} + \label{tab:app-checklist-evid-eval} + \begin{tabular}{lccc} + \toprule + \textbf{Check.} &\textbf{Text-only}&\textbf{Interleaved} &\textbf{Separate} \\ + \midrule + Completeness & -29.2&-19.8& -28.5 \\ + Shuffle &-10.2 &4.1& -17.9 \\ + \textbf{Irrelevant Image} & 0.0&9.4 &-32.7\\ + \midrule + Inv. contract &-1.6 & -7.4&1.5 \\ + Inv. num2text & 1.8&21.3 &-11.3 \\ + Inv. text2num & -0.3&24.5 &0.6 \\ + Inv. synonyms &-11.8 &88.8 &-0.6\\ + Redundant words & -3.1& 85.4&-4.2 \\ + Fluency &-4.9& 31.4& -0.7\\ + Argument structure & -1.6& 30.8& 4.8\\ + \textbf{Image Invariance} & 0.0&20.0 & -1.4 \\ + \bottomrule +\end{tabular} +\end{table*} + +\subsection{Robustness Check on Evidence Evaluation. } +\label{sec:app-eval-robust} +Motivate by~\citep{DBLP:journals/corr/abs-2411-05375,ribeiro-etal-2020-beyond}, we not only assess alignment but also validate the robustness of our evidence evaluation method using \textit{checklist} tests. Notably, the sensitivity of reference-based evaluation for textual data has already been examined in~\citep{DBLP:journals/corr/abs-2411-05375}, and we adopt their adversarial attack design on texts. Details for these attacks (\textit{completeness}, \textit{shuffle}, \textit{Inv. contract}, \textit{Inv. num2text}, \textit{Inv. text2num}, \textit{Inv. synonyms}, \textit{Redundant words}, \textit{Fluency}, \textit{Argument structure}) can be found in the original study. + +Given the multimodal nature of our evaluation, we extend the robustness tests to include visual adversarial checks. Specifically, we evaluate the robustness of the evidence evaluation method against two types of image-based perturbations: 1) \textbf{Irrelevant images}, replacing images in predicted evidence with unrelated images and 2) \textbf{Image invariance}, applying invariant manipulations, such as resizing and rotation, to images within the predicted evidence + +We conducted robustness testing on $20$ claims and compared the difference between the original evidence evaluation scores and the scores obtained after introducing adversarial evidence. The results are summarized in +Table~\ref{tab:app-checklist-evid-eval}. +A robust evaluation method is expected to show a significant performance drop when facing adversarial attacks in the first block of the table while maintaining consistent scores (i.e., minimal deviation from previous scores) when subjected to attacks in the second block. + +The results indicate that both the Text-only and Separated reference-based evaluations exhibit robustness against textual adversarial attacks. In contrast, the interleaved evaluation is sensitive and unstable when faced with such attacks. This finding suggests that even advanced MLLMs, such as Gemini, may be prone to instability when handling interleaved image-text comprehension. + +Regarding visual adversarial attacks (i.e., irrelevant image replacement and image invariance), the Text-only evaluation method fails to maintain stability, while our Separated evaluation method demonstrates robust performance. + +\section{Additional Experimental Results} +\label{sec:app-add-exp-results} + +\subsection{Zero-shot Performance of Baselines} +\label{sec:app-add-exp-zero-shot} +\begin{table*}[ht!] +\centering +\caption{\textbf{Zero-shot performance of baselines.} \textbf{Q-Eval} and \textbf{Evid-Eval} denote for \textit{recall} scores of generated questions and retrieved evidence, with reference of ground-truth questions and evidence. We report verdict prediction and justification generation scores conditioned on evidence retrieval performance, specifically only considering verdict accuracy and justification generation performance when the evidence score is above 0.2, 0.3 and 0.4.} + \label{tab:app-add-exp-result-main} + \begin{tabular}{ll|cc|ccc|ccc} + \toprule + \textbf{LLM} &\textbf{MLLM} &\textbf{Q-Eval} &\textbf{Evid-Eval} & \multicolumn{3}{c}{\textbf{Veracity} \@ (.2/.3/.4)} & \multicolumn{3}{c}{\textbf{Justifications} \@ (.2/.3/.4)}\\ + \midrule + \rowcolor{lightgray} + \multicolumn{10}{c}{\textit{Paralleled Question Generation}}\\ + \midrule + Gemini & Gemini &0.40 &0.18 &0.13&0.11 &0.08 &0.14 &0.13 & 0.08\\ + Qwen & Qwen-VL &0.41 &0.16 &0.08 &0.07 & 0.04&0.11&0.09 & 0.07\\ + Gemma & Gemma &0.37 &0.22 &0.12 & 0.12& 0.08& 0.17&0.15 & 0.10\\ + Qwen & LLaVA & 0.37&0.18 &0.10 &0.09 & 0.07&0.12 &0.11 &0.07 \\ + \midrule + \rowcolor{lightgray} + \multicolumn{10}{c}{\textit{Dynamic Question Generation}}\\ + \midrule + Gemini & Gemini &0.35 &0.19 & 0.15& 0.14&0.09 &0.15 &0.13 & 0.10\\ + Qwen & Qwen-VL &0.30 &0.14 & 0.11&0.10&0.06 &0.11 &0.09 & 0.06\\ + Gemma & Gemma & 0.27& 0.19&0.12 & 0.11&0.07 &0.14& 0.13& 0.08\\ + Qwen & LLaVA & 0.24&0.16 & 0.11&0.11 &0.06 & 0.12&0.11 & 0.06\\ + \midrule + \rowcolor{lightgray} + \multicolumn{10}{c}{\textit{Hybrid Question Generation}}\\ + \midrule + Gemini & Gemini &0.35 & 0.16&0.13 &0.12 & 0.09&0.12 &0.11 & 0.08\\ + Qwen & Qwen-VL &0.34 &0.14 &0.09 &0.08 &0.05 &0.10 & 0.09&0.05 \\ + Gemma & Gemma & 0.26& 0.25& 0.15& 0.14& 0.09& 0.19&0.18 &0.12 \\ + Qwen & LLaVA &0.26 &0.17&0.07 & 0.07& 0.05&0.12 &0.11 &0.08 \\ + \bottomrule +\end{tabular} +\end{table*} + +Due to the limitation of space, we provide the zero-shot performance of baselines in Table~\ref{tab:app-add-exp-result-main}. The findings discussed in Section~\ref{sec:exp-results} hold for baselines under the zero-shot setting as well. + +\begin{table*}[ht!] +\centering +\caption{\textbf{Justification evaluation scores with ROGUE-1 and Ev2R } when baselines are using ground-truth evidence.} + \label{tab:ev2r-justi} + \begin{tabular}{ll|c|c} + \toprule + \textbf{LLM} &\textbf{MLLM} &\textbf{ROGUE-1} &\textbf{Ev2R} \\ + \midrule + Gemini & Gemini &0.50 &0.78 \\ + Qwen & Qwen-VL &0.44 &0.68 \\ + Gemma & Gemma &0.43 &0.75 \\ + Qwen & LLaVA & 0.49& 0.55\\ + \bottomrule +\end{tabular} +\end{table*} +\subsection{Justification Generation Scores with Ev2R} +\label{sec:app-add-exp-justi-scores} +The justification evaluation scores of the baselines, even when provided with ground-truth evidence, are relatively low, as shown in Table~\ref{tab:exp-ablation-results}. We attribute this to the limitations of ROUGE-1 in assessing open-ended generation. To address this, we adopted Ev2R~\citep{DBLP:journals/corr/abs-2411-05375}, a reference-based evaluation method shown to perform well for open-ended generation tasks. The corresponding results are reported in Table~\ref{tab:ev2r-justi}. + +A comparison between ROUGE-1 and Ev2R reveals that Ev2R produces significantly higher and more reasonable scores, which we find encouraging. As part of future work, we plan to conduct human alignment studies and incorporate such more sophisticated evaluation approaches for justification generation. + +%\subsection{Verdict Accuracy with Varying Evidence Score Thresholds} +%\label{sec:app-add-exp-verdict-curve} + +\section{Details of Experiments} +\label{sec:app-detail-of-exp} + +\subsection{Model Implementation} +\label{sec:app-detail-model-imp} +\noindent\textbf{Hyper-parameters and Implementation. } +In the main experiments, we set the total number of generated questions to be $5$ for all QG strategies. For HQG, the first two questions are generated in parallel while the rest three are generated exploited DQG. +In cases of textual evidence retrieval (i.e., leveraging the tools of RIS and WST), we truncated retrieved texts into chunks with the maximum length of $128$ and applied BM25~\citep{DBLP:journals/ftir/RobertsonZ09} to select the most relevant chunks to the given query. +With the increasing capability of long-context understanding of existing LLMs, we keep the top $30$ most related chunks, without a second stage of fine-grained re-ranking as what previous works have done~\citep{DBLP:conf/naacl/ChenKSDC24,DBLP:conf/nips/SchlichtkrullG023}. +For the retrieved images returned by WSI, we compute their similarity scores with the given textual query with CLIP~\citep{DBLP:conf/icml/RadfordKHRGASAM21} and select the most related one as the image evidence source. + +For the choice of LLMs in baselines, we have tried using LLaMA-3.1-8B-Instruct~\citep{llama-meta} as the LLM, whereas the model got stuck in loops, the same as reported by other users.\footnote{https://www.reddit.com/r/LocalLLaMA/comments/1c858ac/llama3\_seems\_to\_get\_stuck\_in\_loops\_sometimes/} +For LaVA-Next~\citep{DBLP:journals/corr/abs-2407-07895}, which is not designed for interleaved image-text, we only consider the textual part of evidence in verdict prediction and justification as we observed some issues with model generation with complex interleaved image-text information. + +For the searching related tools, specifically WST, WSI and RIS, we used the API provided by Google. For web search with textual queries (WST and WSI), we first tokenize and post-tag words in queries and only keep verbs, nouns and adjectives as the search term~\citep{DBLP:conf/ranlp/KaradzhovNMBK17}. +We set temporal constraints with input arguments, limiting all returned web pages published before claim dates. We keep the first 30 search results. +For RIS, we employed the google cloud vision service for detecting web pages containing matched images with the querying image. However, the service does not embed arguments to set temporal constraints. Alternatively, we use a post-hoc method by leveraging the Python package \textit{htmldate.fine\_date} to filter out pages published before claim dates. We noticed a lot of web pages returned by RIS are social media posts, which are non-scrapable. For these pages, we use their page titles as the scraping content. + +\noindent\textbf{Few-shot Setting. }To encourage models generate more critical questions for fact-checking, we exploit a computationally efficient method, few-shot learning. Specifically, we use a few training examples to guide models in question generation. Selecting similar claims to the inference one is important as similar claims may have similar reasoning path for claim verification. We rank the similarity between training claims and the inference claim with BM25~\citep{DBLP:journals/ftir/RobertsonZ09} by comparing their textual part. We set the number of shots to be $3$ to balance between the input length and information from demonstrations. + +For the PQG strategy, we directly provide models the ground-truth questions from selected demonstrations. For the DQG setting, models are provided the textual part of image-text claims and their first questions to generate the initial questions. For generating subsequent questions, each demonstration contains the textual part of a claim, its evidence history from previous QA pairs and the next question to be asked. + +\noindent\textbf{Guidance for Tool Selection. }Besides the guidance for tool selection in prompts as provided in Appendix~\ref{sec:app-prompt-tool-sel}, we also used few demonstrations for inspire models to select proper tools. +We provide few examples to guide tool selection as in the preliminary experiment we observed a heavy rely on VQA as the answering tool. We leveraged the metadata annotation of QA pairs, the answering method, for tool selection. \textit{Image-search} will be mapped to RIS, \textit{Text-search} with an image answer will be converted to WSI while with a textual answer will be mapped into WST. \textit{Image analysis} will be converted to selecting VQA as the answering tool. + +%ICL for tool selection +\noindent\textbf{Computation Resources. } +All experiments are conducted with two GPUs each with 40G dedicated memory. +Specifically, we exploited either A100 or L40 for our experiments. +The Qwen + Qwen-VL baselines and the Qwen + LLaVA baselines take about three hours with A100 and Gemma-based baselines take about seven hours for inference on the test split. +Models have a faster inference speed on L40, saving one third of inference time. +The inference time of Gemini-based models varies, probably depending on the volume of API calls. +Also, we observed instability of Gemini API (e.g., the API call returned \textit{503}, saying the service is not available), maybe because of too many requests at the same time. + +\subsection{Experiment Environment and Packages} +\label{sec:app-detail-exp-setting} +In this section, we introduce the experiment environment and packages in use. +We implement all models under the +PyTorch Library (version 2.4.0+cu121), with CUDA version 12.1. For the implementation of open-source LLMs and MLLMs, we leverage the HuggingFace Library, with the \textit{Qwen/Qwen2.5-7B-Instruct} (Qwen), \textit{google/gemma-3-12b-it} (Gemma), \textit{Qwen/Qwen2.5-VL-7B-Instruct} (Qwen-VL) and \textit{llava-hf/llava-v1.6-mistral-7b-hf} (LLaVA), respectively. The version of Huggingface is 4.50.2. For Gemini, we exploited the API, \textit{gemini-2.0-flash-001}. For the CLIP model employed in image-text similarity computation for evidence rankin, we adopt the checkpoint, \textit{openai/clip-vit-base-patch32}, from Huggingface as well. + +\section{Prompts in Use} +\label{sec:app-prompts-in-use} +In this section, we provide the exact prompts in use for baselines. + +\subsection{Prompts for QA Conversion to Evidence Statement} +\label{sec:app-prompt-qa-to-evid} + +\begin{figure*}[ht!] + \centering + \includegraphics[width=\linewidth]{app-qa-to-evid.pdf} + \caption{ + \textbf{The prompt in use for converting QA pairs to evidence statement.}} + \label{fig:appprompt-qa-to-evid} +\end{figure*} +Following~\citep{DBLP:journals/corr/abs-2411-05375}, we convert QA pairs to evidence statement, for both evidence evaluation and maintaining the evidence history. We consider a text-only conversion for simplicity and use special tokens as placeholders for image. These placeholders could be placed with the exact images in the future. +The prompt is demonstrated in Figure~\ref{fig:appprompt-qa-to-evid}. + +\subsection{Prompts for Evaluation} +\label{sec:app-prompt-evaluation} +\begin{figure*}[ht!] + \centering + \includegraphics[width=\linewidth]{app-eval-ques.pdf} + \caption{ + \textbf{The evaluation prompt for generated questions.}} + \label{fig:appprompt-eval-ques} +\end{figure*} +We adopt a reference based evaluation strategy, which compare predictions against references, for both question evaluation and evidence retrieval evaluation. + +\noindent\textbf{Question Evaluation. }Though questions could be multimodal, the semantics are the most informative. Therefore, we leverage a vanilla reference based evaluation scheme to compare the textual part of predicted questions and annotated questions. The exact prompt in use is shown in Figure~\ref{fig:appprompt-eval-ques}. + +\begin{figure*}[ht!] + \centering + \includegraphics[width=\linewidth]{app-eval-evid.pdf} + \caption{ + \textbf{The evaluation prompt for retrieved evidence.}} + \label{fig:appprompt-eval-evid} +\end{figure*} + +\noindent\textbf{Evidence Evaluation. }We conducted a two-stage reference-based evaluation of evidence, as described in Section~\ref{sec:eval}. +In the first stage, we consider compare the textual of retrieved evidence and ground-truth evidence. Hence, the prompt used for evaluation is similar to that in question evaluation. The difference is that there are special image tokens in evidence and we need the evaluator to output the index of aligned predictions and ground-truth annotations. The prompt is illustrated in Figure~\ref{fig:appprompt-eval-evid}. + +\subsection{Prompts for Question Generation} +\label{sec:app-prompt-qg} + +\begin{figure*}[ht!] + \centering + \includegraphics[width=\linewidth]{app-ques-gen-dynamic.pdf} + \caption{ + \textbf{The prompt for dynamic question generation.} In the few-shot setting, the second paragraph is replaced with the few-shot demonstrations.} + \label{fig:appprompt-qg-dynamic} +\end{figure*} + +\begin{figure*}[ht!] + \centering + \includegraphics[width=\linewidth]{app-ques-gen-para.pdf} + \caption{ + \textbf{The prompt for paralleled question generation.} In the few-shot setting, the third paragraph is replaced with the few-shot demonstrations.} + \label{fig:appprompt-qg-para} +\end{figure*} + +We considered three strategies for question generation as introduced in Section~\ref{sec:exp-baseline}. The \textit{hybrid} generation is the combination of the \textit{paralleled} and the \textit{dynamic} question generation strategy. Below, we provide prompts for the DQG and PQG strategies. +For DQG, we use the prompt shown in Figure~\ref{fig:appprompt-qg-dynamic} and the PQG prompt is shown in Figure~\ref{fig:appprompt-qg-para}. + +For the few-shot question generation setting, we utilize the same prompts while adding a few demonstrations before the information of inference instances. + +\subsection{Prompts for Tool Selection} +\label{sec:app-prompt-tool-sel} +\begin{figure*}[ht!] + \centering + \includegraphics[width=\linewidth]{app-tool-selection.pdf} + \caption{ + \textbf{The prompt in use for selecting tools to answer questions.} [DEMONSTRATIONS] are placeholders of examples provided to guide the tool selection.} + \label{fig:appprompt-tool-sel} +\end{figure*} +As mentioned in Section~\ref{sec:exp-results}, we observed bias of models for heavily relying on VQA as the answering tool, diverging from fact-checkers' choice. This leads to failures for retrieving essential evidence. + +Considering the issue, besides the tool definitions, we provide a few demonstrations, each consisting of a question, a question type and the tool should be selected. Specifically, we leverage the annotated metadata information of questions. For questions annotated with the \textit{answering method} of image-search, we consider the tool of RIS for such cases. For questions with the answering method as text-search while the answers are not images, we regard WST as the tool to be selected; if there are image answers, then the WSI should be the tool. For questions answered by image analysis, we would consider VQA as the answering method. + +The prompt for tool selection is shown in Figure~\ref{fig:appprompt-tool-sel}. +\subsection{Prompts for Answer Generation} +\label{sec:app-prompt-ans-gen} +As introduced in Section~\ref{sec:exp-baseline}, when leveraging the tools of RIS, WST and WSI, there follows an answering model (either an LLM or an MLLM) to leverage retrieved evidence to address the question. + +For using an LLM to leverage textual evidence, we use the prompt: \textit{You need to answer a question according to a set of retrieved documents.} +\textit{Question:} \texttt{[QUES]}; +\textit{Document:} \texttt{[RETRIEVED\_DOC]}. +\textit{If the question is not answerable according to the provided document, please answer as: No answer can be found. Start you answer as: **ANSWER:** } + +For VQA with an MLLM, we prompt models with the template below for an answer: +\textit{Question:} \texttt{[QUES]}; +\textit{Related images to the question:} \texttt{[IMAGES]}. + +\subsection{Prompts for Verdict Prediction} +\label{sec:app-prompt-verdict-gen} +The verifier receives the claim and the retrieved evidence for predicting a veracity label of the claim. Below is the prompt exploited for the verifier: +\textit{You need to select a verdict for a given Image-Text claim when provided a set of evidence. [IMG] is a placeholder for images. We provide four verdict labels and the definitions of them below: Supported: The claim is supported by the evidence presented. Refuted: The claim (either the text or the image part) is contradicted by the evidence presented. Not Enough Evidence: There is not enough evidence (NEE) to support or refute the claim. Conflicting: The claim is misleading due to conflicting evidence/cherry-picking, but not explicitly refuted. You need to response with the verdict for the claim (i.e., Supported, Refuted, Not Enough Evidence or Conflicting) and please do not respond with any other words.} +\textit{The metadata of the claim:} \texttt{[DATE\_AND\_LOCATION]}. +\textit{Claim:} \texttt{[CLAIM\_TEXT]}; +\textit{Claim images:} \texttt{[CLAIM\_IMAGES]}. +\textit{Here is the evidence:} \texttt{[EVID]}. +\textit{Verdict:} + +\subsection{Prompts for Justification Generation} +\label{sec:app-prompt-justi-gen} +The prompt for justification generation receives the information about the claim (textual part and claim images), retrieved evidence and predicted verdict to explain how the verdict could be reached. +Below is the exact prompt in use: + +\textit{Given an image-text claim and a set of evidence for verifying the claim, a fact-checker predict a veracity label for the claim. You need to explain how the verdict is reached for the image-text claim. Below is information for the image-text claim:} +\textit{The metadata of the claim:} \texttt{[DATE\_AND\_LOCATION]}. +\textit{Claim:} \texttt{[CLAIM\_TEXT]}; +\textit{Claim images:} \texttt{[CLAIM\_IMAGES]}. +\textit{The predicted verdict is:} \texttt{[PRED\_VERDICT]}. +\textit{Here is the evidence:} \texttt{[EVID]}. \textit{Please generate your justification (i.e., explanation) for the verdict:} + +Outputs from MLLMs are verbose, whereas human annotated justifications are concise. Therefore, we conduct one step further to prompt the corresponding LLMs to summarize the generated jsutifications in one or two sentences. + +\end{document} + +BIBLIOGRAPHY: +[ + { + "ref_id": "adac25c0aebd013f1225ec02636edfcc0bf4cf7c", + "title": "Gemma 3 Technical Report", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "journals/corr/abs-2503-19786", + "ArXiv": "2503.19786", + "DOI": "10.48550/arXiv.2503.19786", + "CorpusId": 277313563 + } + }, + { + "ref_id": "f61cc9b5583c6295d5cd756ec0f34e4c003aab29", + "title": "Qwen2.5-VL Technical Report", + "authors": null, + "year": null, + "external_ids": { + "ArXiv": "2502.13923", + "DBLP": "journals/corr/abs-2502-13923", + "DOI": "10.48550/arXiv.2502.13923", + "CorpusId": 276449796 + } + }, + { + "ref_id": "88aa6b1f37d1fd8e0a40499ce9bb87873f03aaa8", + "title": "Qwen2.5 Technical Report", + "authors": null, + "year": null, + "external_ids": { + "ArXiv": "2412.15115", + "DBLP": "journals/corr/abs-2412-15115", + "DOI": "10.48550/arXiv.2412.15115", + "CorpusId": 274859421 + } + }, + { + "ref_id": "5fcd69259de58e354e9927372c37053337e14225", + "title": "DEFAME: Dynamic Evidence-based FAct-checking with Multimodal Experts", + "authors": null, + "year": null, + "external_ids": { + "ArXiv": "2412.10510", + "DBLP": "journals/corr/abs-2412-10510", + "DOI": "10.48550/arXiv.2412.10510", + "CorpusId": 274777339 + } + }, + { + "ref_id": "e6a5394593ef0466032946ca323931ba2e76ac91", + "title": "Decompose and Leverage Preferences from Expert Models for Improving Trustworthiness of MLLMs", + "authors": null, + "year": null, + "external_ids": { + "ArXiv": "2411.13697", + "DBLP": "journals/corr/abs-2411-13697", + "DOI": "10.48550/arXiv.2411.13697", + "CorpusId": 274165881 + } + }, + { + "ref_id": "8821d5567174b7d56e556a841fb34a892b2cefe6", + "title": "Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "journals/corr/abs-2411-05375", + "ArXiv": "2411.05375", + "DOI": "10.48550/arXiv.2411.05375", + "CorpusId": 273950610 + } + }, + { + "ref_id": "74c76f0e60e4866641a95f37fb413acaef5badff", + "title": "Open Domain Question Answering with Conflicting Contexts", + "authors": null, + "year": null, + "external_ids": { + "ArXiv": "2410.12311", + "DBLP": "journals/corr/abs-2410-12311", + "DOI": "10.48550/arXiv.2410.12311", + "CorpusId": 273375445 + } + }, + { + "ref_id": "4917d60c82ce041afad1b63887d49d0062031f66", + "title": "Zero-Shot Fact Verification via Natural Logic and Large Language Models", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "journals/corr/abs-2410-03341", + "ArXiv": "2410.03341", + "DOI": "10.48550/arXiv.2410.03341", + "CorpusId": 273162463 + } + }, + { + "ref_id": "570aefa173b255a083c6c5e625a8f8393e446ecd", + "title": "Multimodal Misinformation Detection by Learning from Synthetic Data with Multimodal LLMs", + "authors": null, + "year": null, + "external_ids": { + "ArXiv": "2409.19656", + "DBLP": "conf/emnlp/ZengLGP24", + "DOI": "10.48550/arXiv.2409.19656", + "CorpusId": 272986611 + } + }, + { + "ref_id": "b1efbe7540b7a15a234ae4316198f34fa6574202", + "title": "Similarity Over Factuality: Are we Making Progress on Multimodal Out-of-Context Misinformation Detection?", + "authors": null, + "year": null, + "external_ids": { + "ArXiv": "2407.13488", + "DBLP": "conf/wacv/PapadopoulosKPP25", + "DOI": "10.1109/WACV61041.2025.00544", + "CorpusId": 271270044 + } + }, + { + "ref_id": "2e230519ef242e592b3c19765639008c7046424f", + "title": "LLaVA-NeXT-Interleave: Tackling Multi-image, Video, and 3D in Large Multimodal Models", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "journals/corr/abs-2407-07895", + "ArXiv": "2407.07895", + "DOI": "10.48550/arXiv.2407.07895", + "CorpusId": 271088459 + } + }, + { + "ref_id": "5f9fd071b4cee742893f62f8e7f3ad6f95bae97c", + "title": "MMFakeBench: A Mixed-Source Multimodal Misinformation Detection Benchmark for LVLMs", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "conf/iclr/Liu0LHXCHDH25", + "ArXiv": "2406.08772", + "DOI": "10.48550/arXiv.2406.08772", + "CorpusId": 270440419 + } + }, + { + "ref_id": "38795c0f380cf5f57d5c4de6259c70ea52ddcf17", + "title": "Community notes increase trust in fact-checking on social media", + "authors": null, + "year": null, + "external_ids": { + "PubMedCentral": "11212665", + "DOI": "10.1093/pnasnexus/pgae217", + "CorpusId": 270188060, + "PubMed": "38948016" + } + }, + { + "ref_id": "bcd6cb75279b19279d7b499a4ee92c30781bc177", + "title": "AMMeBa: A Large-Scale Survey and Dataset of Media-Based Misinformation In-The-Wild", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "journals/corr/abs-2405-11697", + "ArXiv": "2405.11697", + "DOI": "10.48550/arXiv.2405.11697", + "CorpusId": 269921753 + } + }, + { + "ref_id": "8dc4876573fe0b3fa11dc8fbd08ad44bd5e4d97e", + "title": "AmbigDocs: Reasoning across Documents on Different Entities under the Same Name", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "journals/corr/abs-2404-12447", + "ArXiv": "2404.12447", + "DOI": "10.48550/arXiv.2404.12447", + "CorpusId": 269282807 + } + }, + { + "ref_id": "eda4655ad91229a109db9075b81ad22a4774499b", + "title": "QACHECK: A Demonstration System for Question-Guided Multi-Hop Fact-Checking", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "journals/corr/abs-2310-07609", + "ArXiv": "2310.07609", + "DOI": "10.48550/arXiv.2310.07609", + "CorpusId": 263835046 + } + }, + { + "ref_id": "69335077fcacbff7a7cf25697da1949e6bdfa968", + "title": "The Art of SOCRATIC QUESTIONING: Recursive Thinking with Large Language Models", + "authors": null, + "year": null, + "external_ids": { + "ArXiv": "2305.14999", + "DBLP": "conf/emnlp/QiXSLJWH23", + "DOI": "10.18653/v1/2023.emnlp-main.255", + "CorpusId": 264935025 + } + }, + { + "ref_id": "767f2f4f22c4f87d6f3a948596f36e4c4c5f1ad4", + "title": "AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "journals/corr/abs-2305-13117", + "ArXiv": "2305.13117", + "DOI": "10.48550/arXiv.2305.13117", + "CorpusId": 258832336 + } + }, + { + "ref_id": "8793066d170b6a742c4fcdb478d4f100c1e4bf17", + "title": "Fact-Checking Complex Claims with Program-Guided Reasoning", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "journals/corr/abs-2305-12744", + "ACL": "2023.acl-long.386", + "ArXiv": "2305.12744", + "DOI": "10.48550/arXiv.2305.12744", + "CorpusId": 258833449 + } + }, + { + "ref_id": "4fb0e1283194c0d20ac0396bc375fdf91c14da1e", + "title": "Complex Claim Verification with Evidence Retrieved in the Wild", + "authors": null, + "year": null, + "external_ids": { + "ACL": "2024.naacl-long.196", + "ArXiv": "2305.11859", + "DBLP": "journals/corr/abs-2305-11859", + "DOI": "10.48550/arXiv.2305.11859", + "CorpusId": 258822852 + } + }, + { + "ref_id": "c8fdcffdd866f10dd39d9fb16bacda0ec4085eb3", + "title": "VERITE: a Robust benchmark for multimodal misinformation detection accounting for unimodal bias", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "journals/ijmir/PapadopoulosKPP24", + "ArXiv": "2304.14133", + "DOI": "10.1007/s13735-023-00312-6", + "CorpusId": 260091840 + } + }, + { + "ref_id": "aa6a8a321de1edd2545a6c8bc1cf4fea6bfe5fb0", + "title": "AutoSplice: A Text-prompt Manipulated Image Dataset for Media Forensics", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "conf/cvpr/JiaHZJCL23", + "ArXiv": "2304.06870", + "DOI": "10.1109/CVPRW59228.2023.00096", + "CorpusId": 258170444 + } + }, + { + "ref_id": "1a23ed979aee638ae05c030dae6d7d1f13ee3f78", + "title": "Factify 2: A multimodal fake news and satire news dataset", + "authors": null, + "year": null, + "external_ids": { + "ArXiv": "2304.03897", + "DBLP": "conf/defactify/SuryavardanMPCR23", + "DOI": "10.48550/arXiv.2304.03897", + "CorpusId": 258049045 + } + }, + { + "ref_id": "96704efeb471a64178a3356662c5fb42e451f16e", + "title": "Detecting and Grounding Multi-Modal Media Manipulation", + "authors": null, + "year": null, + "external_ids": { + "ArXiv": "2304.02556", + "DBLP": "journals/corr/abs-2304-02556", + "DOI": "10.1109/CVPR52729.2023.00667", + "CorpusId": 257952257 + } + }, + { + "ref_id": "af997821231898a5f8d0fd78dad4eec526acabe5", + "title": "Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "journals/corr/abs-2303-04671", + "ArXiv": "2303.04671", + "DOI": "10.48550/arXiv.2303.04671", + "CorpusId": 257404891 + } + }, + { + "ref_id": "47a1930329ce5a9b148ea4d97b1f29f77c9841a4", + "title": "New Threats to Society from Free-Speech Social Media Platforms", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "journals/corr/abs-2302-01229", + "ArXiv": "2302.01229", + "DOI": "10.1145/3587094", + "CorpusId": 256503989 + } + }, + { + "ref_id": "bb6f8161be27877e0299af65a6a7caf3f0a1f220", + "title": "Missing Counter-Evidence Renders NLP Fact-Checking Unrealistic for Misinformation", + "authors": null, + "year": null, + "external_ids": { + "ArXiv": "2210.13865", + "ACL": "2022.emnlp-main.397", + "DBLP": "conf/emnlp/Glockner0G22", + "DOI": "10.48550/arXiv.2210.13865", + "CorpusId": 253107194 + } + }, + { + "ref_id": "be050e69aadd6461ff3bc35dcfa7a551742ef840", + "title": "Varifocal Question Generation for Fact-checking", + "authors": null, + "year": null, + "external_ids": { + "ArXiv": "2210.12400", + "ACL": "2022.emnlp-main.163", + "DBLP": "conf/emnlp/OusidhoumY022", + "DOI": "10.48550/arXiv.2210.12400", + "CorpusId": 253098619 + } + }, + { + "ref_id": "4da2b6b1677c0178416f3a613fcafdedf9ac25c4", + "title": "End-to-End Multimodal Fact-Checking and Explanation Generation: A Challenging Dataset and Models", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "conf/sigir/YaoS0CH23", + "ArXiv": "2205.12487", + "DOI": "10.1145/3539618.3591879", + "CorpusId": 249062580 + } + }, + { + "ref_id": "e78a5096031e97f43906b907fe2e0a6a827afcb1", + "title": "MuMiN: A Large-Scale Multilingual Multimodal Fact-Checked Misinformation Social Network Dataset", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "conf/sigir/NielsenM22", + "ArXiv": "2202.11684", + "DOI": "10.1145/3477495.3531744", + "CorpusId": 247058367 + } + }, + { + "ref_id": "3d233f11e779eea67884ad871df457d5c9ff55ea", + "title": "The impact of fake news on social media and its influence on health during the COVID-19 pandemic: a systematic review", + "authors": null, + "year": null, + "external_ids": { + "PubMedCentral": "8502082", + "DOI": "10.1007/s10389-021-01658-z", + "CorpusId": 238530721, + "PubMed": "34660175" + } + }, + { + "ref_id": "02e46711fc86877bdd279c736abe5415a2415e48", + "title": "A Survey on Automated Fact-Checking", + "authors": null, + "year": null, + "external_ids": { + "ArXiv": "2108.11896", + "DBLP": "journals/tacl/GuoSV22", + "ACL": "2022.tacl-1.11", + "DOI": "10.1162/tacl_a_00454", + "CorpusId": 237304047 + } + }, + { + "ref_id": "98bb75dcb7dfe8e675781fe2008170e8f00a5dee", + "title": "FEVEROUS: Fact Extraction and VERification Over Unstructured and Structured information", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "conf/nips/AlyGST00CM21", + "ArXiv": "2106.05707", + "CorpusId": 235391052 + } + }, + { + "ref_id": "89e7626d9b94e05c4154e7f6fb6df146fa4eb8fa", + "title": "NewsCLIPpings: Automatic Generation of Out-of-Context Multimodal Media", + "authors": null, + "year": null, + "external_ids": { + "ArXiv": "2104.05893", + "ACL": "2021.emnlp-main.545", + "DBLP": "conf/emnlp/LuoDR21", + "DOI": "10.18653/v1/2021.emnlp-main.545", + "CorpusId": 233219387 + } + }, + { + "ref_id": "38d243b9f6e2c786699dbc83513fb190372cde07", + "title": "Automated Fact-Checking for Assisting Human Fact-Checkers", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "conf/ijcai/NakovCHAEBPSM21", + "ArXiv": "2103.07769", + "DOI": "10.24963/ijcai.2021/619", + "CorpusId": 232233764 + } + }, + { + "ref_id": "6f870f7f02a8c59c3e23f407f3ef00dd1dcf8fc4", + "title": "Learning Transferable Visual Models From Natural Language Supervision", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "conf/icml/RadfordKHRGASAM21", + "ArXiv": "2103.00020", + "CorpusId": 231591445 + } + }, + { + "ref_id": "33ec7eb2168e37e3007d1059aa96b9a63254b4da", + "title": "Beyond Accuracy: Behavioral Testing of NLP Models with CheckList", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "conf/ijcai/RibeiroWG021", + "ACL": "2020.acl-main.442", + "ArXiv": "2005.04118", + "MAG": "3023533951", + "DOI": "10.18653/v1/2020.acl-main.442", + "CorpusId": 218551201 + } + }, + { + "ref_id": "8ff6feeb256fa324b440744f9df3fbc766826003", + "title": "Fakeddit: A New Multimodal Benchmark Dataset for Fine-grained Fake News Detection", + "authors": null, + "year": null, + "external_ids": { + "ACL": "2020.lrec-1.755", + "DBLP": "conf/lrec/NakamuraLW20", + "MAG": "3030648110", + "CorpusId": 207852737 + } + }, + { + "ref_id": "ae456fa5af4e277c2125c094059451efc4dae346", + "title": "Is a Picture Worth a Thousand Words? An Empirical Study of Image Content and Social Media Engagement", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2984184072", + "DOI": "10.1177/0022243719881113", + "CorpusId": 211736212 + } + }, + { + "ref_id": "d8cb11d4be955f9869387a18967dee366eb851d9", + "title": "MultiFC: A Real-World Multi-Domain Dataset for Evidence-Based Fact Checking of Claims", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "journals/corr/abs-1909-03242", + "ACL": "D19-1475", + "MAG": "2971911738", + "ArXiv": "1909.03242", + "DOI": "10.18653/v1/D19-1475", + "CorpusId": 202541363 + } + }, + { + "ref_id": "2cd27f7a7498e812c4859c15e0371c8b7ec019e2", + "title": "Proppy: Organizing the news based on their propagandistic content", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2946595845", + "DBLP": "journals/ipm/Barron-CedenoJM19", + "DOI": "10.1016/J.IPM.2019.03.005", + "CorpusId": 181473906 + } + }, + { + "ref_id": "6897deee6838921dabcc4ef2c958b995e1404893", + "title": "A Richly Annotated Corpus for Different Tasks in Automated Fact-Checking", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "conf/conll/HanselowskiSSLG19", + "ACL": "K19-1046", + "MAG": "2988245244", + "ArXiv": "1911.01214", + "DOI": "10.18653/v1/K19-1046", + "CorpusId": 207779874 + } + }, + { + "ref_id": "06896fc0a0cb2a39ed9e9018e314977f3920ca25", + "title": "Protecting elections from social media manipulation", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2970610561", + "DOI": "10.1126/science.aaw8243", + "CorpusId": 201672410, + "PubMed": "31467206" + } + }, + { + "ref_id": "6fa367e31ca69baead660940f66ee2151134aec6", + "title": "Fact-Checking Meets Fauxtography: Verifying Claims About Images", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "journals/corr/abs-1908-11722", + "MAG": "2971291998", + "ACL": "D19-1216", + "ArXiv": "1908.11722", + "DOI": "10.18653/v1/D19-1216", + "CorpusId": 201698324 + } + }, + { + "ref_id": "14303b1724c2bb623d5fca314d180fddc59e5f5c", + "title": "The When, Why, How and So-What of Verifications", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2924319510", + "DOI": "10.1080/1461670X.2019.1593881", + "CorpusId": 150434814 + } + }, + { + "ref_id": "9861c0c52c8c5c933983d9fb2dbb9381eb393f04", + "title": "Fighting misinformation on social media using crowdsourced judgments of news source quality", + "authors": null, + "year": null, + "external_ids": { + "PubMedCentral": "6377495", + "MAG": "2913081045", + "DOI": "10.1073/pnas.1806781116", + "CorpusId": 59340859, + "PubMed": "30692252" + } + }, + { + "ref_id": "eed1a4b3ec3b6de0fd1f0b8b2ec969b540fe41a0", + "title": "FakeNewsNet: A Data Repository with News Content, Social Context, and Spatiotemporal Information for Studying Fake News on Social Media", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "journals/bigdata/ShuMWLL20", + "MAG": "3031781733", + "ArXiv": "1809.01286", + "DOI": "10.1089/big.2020.0062", + "CorpusId": 85528899, + "PubMed": "32491943" + } + }, + { + "ref_id": "b1d24e8e08435b7c52335485a0d635abf9bc604c", + "title": "FEVER: a Large-scale Dataset for Fact Extraction and VERification", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2789566302", + "ACL": "N18-1074", + "ArXiv": "1803.05355", + "DBLP": "conf/naacl/ThorneVCM18", + "DOI": "10.18653/v1/N18-1074", + "CorpusId": 4711425 + } + }, + { + "ref_id": "e1fb1c8f76dfa1044b8f969940780635d92e7599", + "title": "Fully Automated Fact Checking Using External Sources", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2963857245", + "ArXiv": "1710.00341", + "DBLP": "conf/ranlp/KaradzhovNMBK17", + "ACL": "R17-1046", + "DOI": "10.26615/978-954-452-049-6_046", + "CorpusId": 7473712 + } + }, + { + "ref_id": "c99c3ec3008e2124b831a20786189f58ece2532c", + "title": "Partisan Selective Sharing: The Biased Diffusion of Fact-Checking Messages on Social Media", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2592103299", + "DOI": "10.1111/JCOM.12284", + "CorpusId": 151470507 + } + }, + { + "ref_id": "751b04d346fc9a70781bbfea23953f424ff7deec", + "title": "The proof and measurement of association between two things.", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2151967815", + "DOI": "10.1093/ije/dyq191", + "CorpusId": 14780428, + "PubMed": "21051364" + } + }, + { + "ref_id": "8694fc7b887c9a2ff553851c165ae377ed343059", + "title": "Nonprobative photographs (or words) inflate truthiness", + "authors": null, + "year": null, + "external_ids": { + "MAG": "1972102087", + "DOI": "10.3758/S13423-012-0292-0", + "CorpusId": 8677269, + "PubMed": "22869334" + } + }, + { + "ref_id": "62c46828aef5c68bfa610ce7b1a067020287830d", + "title": "Inequalities between multi-rater kappas", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "journals/adac/Warrens10", + "MAG": "2029553347", + "DOI": "10.1007/s11634-010-0073-4", + "CorpusId": 312052 + } + }, + { + "ref_id": "47ced790a563344efae66588b5fb7fe6cca29ed3", + "title": "The Probabilistic Relevance Framework: BM25 and Beyond", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2155482025", + "DBLP": "journals/ftir/RobertsonZ09", + "DOI": "10.1561/1500000019", + "CorpusId": 207178704 + } + }, + { + "ref_id": "799b424118419cf7fc78ed1bc407e3af03aba01f", + "title": "Free-Marginal Multirater Kappa (multirater K[free]): An Alternative to Fleiss' Fixed-Marginal Multirater Kappa.", + "authors": null, + "year": null, + "external_ids": { + "MAG": "112197792", + "CorpusId": 59676845 + } + }, + { + "ref_id": "60b05f32c32519a809f21642ef1eb3eaf3848008", + "title": "ROUGE: A Package for Automatic Evaluation of Summaries", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2154652894", + "ACL": "W04-1013", + "CorpusId": 964287 + } + }, + { + "ref_id": "c710d25e5af106a90e7cca30b97d24cc734a2a1a", + "title": "MMOOC: A Multimodal Misinformation Dataset for Out-of-Context News Analysis", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "conf/acisp/XuDCLY24", + "DOI": "10.1007/978-981-97-5101-3_24", + "CorpusId": 271571949 + } + }, + { + "ref_id": "c0532d8d69af3bc0836b88c5aae2ce6166ac5136", + "title": "FACTIFY: A Multi-Modal Fact Verification Dataset", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "conf/aaai/MishraSBCRPD0SE22", + "CorpusId": 252016186 + } + }, + { + "ref_id": "bbfed74eed1796b4534bcce6811b2c7c0b74024a", + "title": "InfoSurgeon: Cross-Media Fine-grained Information Consistency Checking for Fake News Detection", + "authors": null, + "year": null, + "external_ids": { + "ACL": "2021.acl-long.133", + "DBLP": "conf/acl/FungTRPJCMBS20", + "DOI": "10.18653/v1/2021.acl-long.133", + "CorpusId": 236460326 + } + }, + { + "ref_id": null, + "title": "Catching out-of-context misinformation with self-supervised learning", + "authors": null, + "year": null, + "external_ids": {} + }, + { + "ref_id": "bd0f43f8962ef1abf10e4ed4048d69156d5e8713", + "title": "Where is Your Evidence: Improving Fact-checking by Justification Modeling", + "authors": null, + "year": null, + "external_ids": { + "DBLP": "conf/emnlp/AlhindiPM18", + "MAG": "2954646118", + "ACL": "W18-5513", + "DOI": "10.18653/v1/W18-5513", + "CorpusId": 53640239 + } + }, + { + "ref_id": "6f78b5608fed43f106da192f12e09d9edbd2fce0", + "title": "Social Media and Fake News in the 2016 Election", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2582561810", + "DOI": "10.1257/JEP.31.2.211", + "CorpusId": 32730475 + } + }, + { + "ref_id": "89e8c86c6144c752885baa07bf3dbe11142e5fab", + "title": "Mathematical Contributions to the Theory of Evolution. III. Regression, Heredity, and Panmixia", + "authors": null, + "year": null, + "external_ids": { + "MAG": "2005825070", + "DOI": "10.1098/RSTA.1896.0007", + "CorpusId": 119875807 + } + }, + { + "ref_id": null, + "title": "Introducing gemini 2.0: our new ai model for the agentic era", + "authors": null, + "year": null, + "external_ids": {} + }, + { + "ref_id": null, + "title": "image, tell me your story!", + "authors": null, + "year": null, + "external_ids": {} + }, + { + "ref_id": null, + "title": "Claim images", + "authors": null, + "year": null, + "external_ids": {} + }, + { + "ref_id": null, + "title": "Please refer to our LLM policy", + "authors": null, + "year": null, + "external_ids": {} + } +] + +CITATION CONTEXTS: +[ + { + "ref_id": "adac25c0aebd013f1225ec02636edfcc0bf4cf7c", + "citation_marker": "Gemma 3 Technical Report", + "text": "5-VL-7B-Instruct [Bai et al., 2025] ( Qwen-VL ) serves as the MLLM; 3) Gemma-3-12B [Kamath et al., 2025] ( Gemma ), capable of both unimodal and multimodal understanding, is used as both the LLM and the MLLM; and 4) Qwen and LLaVA-Next-7B [Li et al., 2024] ( LLaVA ) work as the LLM and MLLM\u2026", + "section": null, + "intents": [] + }, + { + "ref_id": "f61cc9b5583c6295d5cd756ec0f34e4c003aab29", + "citation_marker": "Qwen2.5-VL Technical Report", + "text": "5-VL-7B-Instruct [Bai et al., 2025] ( Qwen-VL ) serves as the MLLM; 3) Gemma-3-12B [Kamath et al., 2025] ( Gemma ), capable of both unimodal and multimodal understanding, is used as both the LLM and the MLLM; and 4) Qwen and LLaVA-Next-7B [Li et al., 2024] ( LLaVA ) work as the LLM and MLLM\u2026", + "section": null, + "intents": [] + }, + { + "ref_id": "88aa6b1f37d1fd8e0a40499ce9bb87873f03aaa8", + "citation_marker": "Qwen2.5 Technical Report", + "text": "5-7B-Instruct [Yang et al., 2024] ( Qwen ) acts as the LLM and Qwen2.", + "section": null, + "intents": [] + }, + { + "ref_id": "5fcd69259de58e354e9927372c37053337e14225", + "citation_marker": "DEFAME: Dynamic Evidence-based FAct-checking with Multimodal Experts", + "text": "Inspired by recent research on tool usage [Wu et al., 2023, Cao et al., 2024, Braun et al., 2024], we integrate a set of specialized tools into the answer generation module, along with a tool selector that automatically selects the appropriate tool for a given question.", + "section": null, + "intents": [] + }, + { + "ref_id": "e6a5394593ef0466032946ca323931ba2e76ac91", + "citation_marker": "Decompose and Leverage Preferences from Expert Models for Improving Trustworthiness of MLLMs", + "text": "Inspired by recent research on tool usage [Wu et al., 2023, Cao et al., 2024, Braun et al., 2024], we integrate a set of specialized tools into the answer generation module, along with a tool selector that automatically selects the appropriate tool for a given question.", + "section": null, + "intents": [] + }, + { + "ref_id": "8821d5567174b7d56e556a841fb34a892b2cefe6", + "citation_marker": "Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking", + "text": "We transform each QA pair into an evidence statement following [Akhtar et al., 2024], where images are represented with special image tokens (e.g., [IMG_1] ).", + "section": null, + "intents": [] + }, + { + "ref_id": "8821d5567174b7d56e556a841fb34a892b2cefe6", + "citation_marker": "Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking", + "text": "Motivate by [Akhtar et al., 2024, Ribeiro et al., 2020], we not only assess alignment but also validate the robustness of our evidence evaluation method using checklist tests.", + "section": null, + "intents": [] + }, + { + "ref_id": "8821d5567174b7d56e556a841fb34a892b2cefe6", + "citation_marker": "Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking", + "text": "Motivated by a recent study [Akhtar et al., 2024], we adopt a reference-based evaluation method that compares model responses to human-annotated ground-truth data.", + "section": null, + "intents": [] + }, + { + "ref_id": "8821d5567174b7d56e556a841fb34a892b2cefe6", + "citation_marker": "Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking", + "text": "Recent research [Akhtar et al., 2024] showed that reference-based evaluation of evidence with large language models (LLMs) aligns best with human assessments.", + "section": null, + "intents": [] + }, + { + "ref_id": "8821d5567174b7d56e556a841fb34a892b2cefe6", + "citation_marker": "Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking", + "text": "This evaluation approach aligns with that used in [Akhtar et al., 2024] and has also been applied at the FEVER workshop, 8 demonstrating its reliability and strong correlation with human ratings.", + "section": null, + "intents": [] + }, + { + "ref_id": "8821d5567174b7d56e556a841fb34a892b2cefe6", + "citation_marker": "Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking", + "text": "0-Flash [DeepMind, 2024] as the scoring model for both steps in the reference-based evaluation inspired by its power [Akhtar et al., 2024].", + "section": null, + "intents": [] + }, + { + "ref_id": "8821d5567174b7d56e556a841fb34a892b2cefe6", + "citation_marker": "Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking", + "text": "Additionally, we conduct checklist tests to evaluate the sensitivity of the reference-based evidence evaluation method, following the approach outlined in [Akhtar et al., 2024] (Appendix F.3).", + "section": null, + "intents": [] + }, + { + "ref_id": "8821d5567174b7d56e556a841fb34a892b2cefe6", + "citation_marker": "Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking", + "text": "To address this, we experimented with a reference-based evaluation method, Ev2R [Akhtar et al., 2024], which has been shown to align well with human assessments in open-ended generation tasks.", + "section": null, + "intents": [] + }, + { + "ref_id": "8821d5567174b7d56e556a841fb34a892b2cefe6", + "citation_marker": "Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking", + "text": "Notably, the sensitivity of reference-based evaluation for textual data has already been examined in [Akhtar et al., 2024], and we adopt their adversarial attack design on texts.", + "section": null, + "intents": [] + }, + { + "ref_id": "8821d5567174b7d56e556a841fb34a892b2cefe6", + "citation_marker": "Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking", + "text": "For the textual part, we use a validation method similar to [Akhtar et al., 2024].", + "section": null, + "intents": [] + }, + { + "ref_id": "4917d60c82ce041afad1b63887d49d0062031f66", + "citation_marker": "Zero-Shot Fact Verification via Natural Logic and Large Language Models", + "text": "To capture the rationale in claim verification, a complex reasoning task, various reasoning representations have been explored, including natural logic [Strong et al., 2024], functional programs [Pan et al., 2023b], and QA [Chen et al., 2024, Qi et al., 2023, Pan et al., 2023a].", + "section": null, + "intents": [] + }, + { + "ref_id": "570aefa173b255a083c6c5e625a8f8393e446ecd", + "citation_marker": "Multimodal Misinformation Detection by Learning from Synthetic Data with Multimodal LLMs", + "text": "However, there are discrepancies between synthetic data and real-world image-text claims [Zeng et al., 2024, Papadopoulos et al., 2025], raising concerns about the generalization of models to real-world image-text claim verification.", + "section": null, + "intents": [] + }, + { + "ref_id": "570aefa173b255a083c6c5e625a8f8393e446ecd", + "citation_marker": "Multimodal Misinformation Detection by Learning from Synthetic Data with Multimodal LLMs", + "text": "Due to discrepancies between synthetic data and real-world data [Zeng et al., 2024], models that perform well on synthetic benchmarks may fail to generalize to real-world claims.", + "section": null, + "intents": [] + }, + { + "ref_id": "b1efbe7540b7a15a234ae4316198f34fa6574202", + "citation_marker": "Similarity Over Factuality: Are we Making Progress on Multimodal Out-of-Context Misinformation Detection?", + "text": "However, there are discrepancies between synthetic data and real-world image-text claims [Zeng et al., 2024, Papadopoulos et al., 2025], raising concerns about the generalization of models to real-world image-text claim verification.", + "section": null, + "intents": [] + }, + { + "ref_id": "b1efbe7540b7a15a234ae4316198f34fa6574202", + "citation_marker": "Similarity Over Factuality: Are we Making Progress on Multimodal Out-of-Context Misinformation Detection?", + "text": "Moreover, recent work [Papadopoulos et al., 2025] showed that models can achieve high performance on such datasets by exploiting superficial correlations, such as image-text similarity, without examining factuality and logical consistency.", + "section": null, + "intents": [] + }, + { + "ref_id": "2e230519ef242e592b3c19765639008c7046424f", + "citation_marker": "LLaVA-NeXT-Interleave: Tackling Multi-image, Video, and 3D in Large Multimodal Models", + "text": "\u2026[Bai et al., 2025] ( Qwen-VL ) serves as the MLLM; 3) Gemma-3-12B [Kamath et al., 2025] ( Gemma ), capable of both unimodal and multimodal understanding, is used as both the LLM and the MLLM; and 4) Qwen and LLaVA-Next-7B [Li et al., 2024] ( LLaVA ) work as the LLM and MLLM respectively.", + "section": null, + "intents": [] + }, + { + "ref_id": "2e230519ef242e592b3c19765639008c7046424f", + "citation_marker": "LLaVA-NeXT-Interleave: Tackling Multi-image, Video, and 3D in Large Multimodal Models", + "text": "This is expected, as LLaVA [Li et al., 2024] was not pre-trained on interleaved image-text documents.", + "section": null, + "intents": [] + }, + { + "ref_id": "2e230519ef242e592b3c19765639008c7046424f", + "citation_marker": "LLaVA-NeXT-Interleave: Tackling Multi-image, Video, and 3D in Large Multimodal Models", + "text": "For LaVA-Next [Li et al., 2024], which is not designed for interleaved image-text, we only consider the textual part of evidence in verdict prediction and justification as we observed some issues with model generation with complex interleaved image-text information.", + "section": null, + "intents": [] + }, + { + "ref_id": "5f9fd071b4cee742893f62f8e7f3ad6f95bae97c", + "citation_marker": "MMFakeBench: A Mixed-Source Multimodal Misinformation Detection Benchmark for LVLMs", + "text": "\u2026al., 2019 [Fung et al., 2021] \u2718 \u2713 - - - - 30,000 Autosplice [Jia et al., 2023] \u2718 \u2713 - - - - 5,894 DGM [Shao et al., 2023] \u2718 \u2713 - - - - 230,000 MMFake [Liu et al., 2024b] \u2718 \u2713 - - - - 11,000 Verite [Papadopoulos et al., 2024] \u2718 \u2713 - - - - 1,000 COSMOS [Aneja et al., 2021] Mix \u2718 - - - - 201,700 FACTIFY\u2026", + "section": null, + "intents": [] + }, + { + "ref_id": "38795c0f380cf5f57d5c4de6259c70ea52ddcf17", + "citation_marker": "Community notes increase trust in fact-checking on social media", + "text": "However, they are unable to keep up with the vast volume of information online [Pennycook, 2019, Drolsbach et al., 2024].", + "section": null, + "intents": [] + }, + { + "ref_id": "bcd6cb75279b19279d7b499a4ee92c30781bc177", + "citation_marker": "AMMeBa: A Large-Scale Survey and Dataset of Media-Based Misinformation In-The-Wild", + "text": "\u2026sources: 1) filtered articles from AVeriTec [Schlichtkrull et al., 2023] containing multimodal claims, 2) articles verifying image-related claims from AMMEBA [Dufour et al., 2024] and 3) all true claims from ClaimReview 7 over the past two years that included keywords such as phtos or pictures .", + "section": null, + "intents": [] + }, + { + "ref_id": "bcd6cb75279b19279d7b499a4ee92c30781bc177", + "citation_marker": "AMMeBa: A Large-Scale Survey and Dataset of Media-Based Misinformation In-The-Wild", + "text": "Recent studies estimate that approximately 80% of online claims are multimodal involving both text and media [Dufour et al., 2024], as media can enhance perceived credibility [Newman et al., 2012] and increase exposure [Li and Xie, 2020].", + "section": null, + "intents": [] + }, + { + "ref_id": "bcd6cb75279b19279d7b499a4ee92c30781bc177", + "citation_marker": "AMMeBa: A Large-Scale Survey and Dataset of Media-Based Misinformation In-The-Wild", + "text": "Among these, images are the most prevalent media type [Dufour et al., 2024].", + "section": null, + "intents": [] + }, + { + "ref_id": "8dc4876573fe0b3fa11dc8fbd08ad44bd5e4d97e", + "citation_marker": "AmbigDocs: Reasoning across Documents on Different Entities under the Same Name", + "text": "Conflicts among evidence has been extensively studied in the context of QA [Liu et al., 2024a, Lee et al., 2024].", + "section": null, + "intents": [] + }, + { + "ref_id": "eda4655ad91229a109db9075b81ad22a4774499b", + "citation_marker": "QACHECK: A Demonstration System for Question-Guided Multi-Hop Fact-Checking", + "text": "To capture the rationale in claim verification, a complex reasoning task, various reasoning representations have been explored, including natural logic [Strong et al., 2024], functional programs [Pan et al., 2023b], and QA [Chen et al., 2024, Qi et al., 2023, Pan et al., 2023a].", + "section": null, + "intents": [] + }, + { + "ref_id": "69335077fcacbff7a7cf25697da1949e6bdfa968", + "citation_marker": "The Art of SOCRATIC QUESTIONING: Recursive Thinking with Large Language Models", + "text": "To capture the rationale in claim verification, a complex reasoning task, various reasoning representations have been explored, including natural logic [Strong et al., 2024], functional programs [Pan et al., 2023b], and QA [Chen et al., 2024, Qi et al., 2023, Pan et al., 2023a].", + "section": null, + "intents": [] + }, + { + "ref_id": "767f2f4f22c4f87d6f3a948596f36e4c4c5f1ad4", + "citation_marker": "AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web", + "text": "It is also worth noting that our models do not explicitly model conflict within evidence, in contrast to [Schlichtkrull et al., 2023], which predicted verdicts per evidence piece and then examined whether these verdicts conflicted.", + "section": null, + "intents": [] + }, + { + "ref_id": "767f2f4f22c4f87d6f3a948596f36e4c4c5f1ad4", + "citation_marker": "AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web", + "text": "This is consistent with other human-annotated datasets of real-world claims, such as AVeriTeC [Schlichtkrull et al., 2023], which also contain only a few thousand claims.", + "section": null, + "intents": [] + }, + { + "ref_id": "767f2f4f22c4f87d6f3a948596f36e4c4c5f1ad4", + "citation_marker": "AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web", + "text": "Following [Thorne et al., 2018, Schlichtkrull et al., 2023], we first assess the quality of the retrieved evidence by comparing it against human-annotated references, and report veracity prediction accuracy conditioned on the evidence scores.", + "section": null, + "intents": [] + }, + { + "ref_id": "767f2f4f22c4f87d6f3a948596f36e4c4c5f1ad4", + "citation_marker": "AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web", + "text": "For comparison, AVeriTec [Schlichtkrull et al., 2023] reported agreement of 0 .", + "section": null, + "intents": [] + }, + { + "ref_id": "767f2f4f22c4f87d6f3a948596f36e4c4c5f1ad4", + "citation_marker": "AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web", + "text": "This result aligns with prior findings on the challenge of detecting conflicting textual claims [Schlichtkrull et al., 2023].", + "section": null, + "intents": [] + }, + { + "ref_id": "767f2f4f22c4f87d6f3a948596f36e4c4c5f1ad4", + "citation_marker": "AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web", + "text": "As in prior work [Schlichtkrull et al., 2023], we assume that the first phase of annotation has already been completed, and thus the re-annotation process begins from the second phase.", + "section": null, + "intents": [] + }, + { + "ref_id": "767f2f4f22c4f87d6f3a948596f36e4c4c5f1ad4", + "citation_marker": "AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web", + "text": "A straightforward approach is to generate all verification questions at once, given an image-text claim, as in prior work [Schlichtkrull et al., 2023].", + "section": null, + "intents": [] + }, + { + "ref_id": "767f2f4f22c4f87d6f3a948596f36e4c4c5f1ad4", + "citation_marker": "AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web", + "text": "Using the list of common misinformative sources from [Schlichtkrull et al., 2023], we observed that 8 answers relied on a flagged source.", + "section": null, + "intents": [] + }, + { + "ref_id": "767f2f4f22c4f87d6f3a948596f36e4c4c5f1ad4", + "citation_marker": "AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web", + "text": "As noted in prior work [Ousidhoum et al., 2022, Schlichtkrull et al., 2023], this may result in omitting critical contextual information for verification, such as context to resolve coreferences.", + "section": null, + "intents": [] + }, + { + "ref_id": "767f2f4f22c4f87d6f3a948596f36e4c4c5f1ad4", + "citation_marker": "AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web", + "text": "To support research in AFC, the research community has created various benchmark datasets [Thorne et al., 2018, Aly et al., 2021, Schlichtkrull et al., 2023, Alhindi et al., 2018, Yao et al., 2023, Chen et al., 2024], aiming to enhance the effectiveness and interpretability of fact-checking systems.", + "section": null, + "intents": [] + }, + { + "ref_id": "8793066d170b6a742c4fcdb478d4f100c1e4bf17", + "citation_marker": "Fact-Checking Complex Claims with Program-Guided Reasoning", + "text": "To capture the rationale in claim verification, a complex reasoning task, various reasoning representations have been explored, including natural logic [Strong et al., 2024], functional programs [Pan et al., 2023b], and QA [Chen et al., 2024, Qi et al., 2023, Pan et al., 2023a].", + "section": null, + "intents": [] + }, + { + "ref_id": "4fb0e1283194c0d20ac0396bc375fdf91c14da1e", + "citation_marker": "Complex Claim Verification with Evidence Retrieved in the Wild", + "text": "To support research in AFC, the research community has created various benchmark datasets [Thorne et al., 2018, Aly et al., 2021, Schlichtkrull et al., 2023, Alhindi et al., 2018, Yao et al., 2023, Chen et al., 2024], aiming to enhance the effectiveness and interpretability of fact-checking systems.", + "section": null, + "intents": [] + }, + { + "ref_id": "4fb0e1283194c0d20ac0396bc375fdf91c14da1e", + "citation_marker": "Complex Claim Verification with Evidence Retrieved in the Wild", + "text": "To capture the rationale in claim verification, a complex reasoning task, various reasoning representations have been explored, including natural logic [Strong et al., 2024], functional programs [Pan et al., 2023b], and QA [Chen et al., 2024, Qi et al., 2023, Pan et al., 2023a].", + "section": null, + "intents": [] + }, + { + "ref_id": "4fb0e1283194c0d20ac0396bc375fdf91c14da1e", + "citation_marker": "Complex Claim Verification with Evidence Retrieved in the Wild", + "text": "With the increasing capability of long-context understanding of existing LLMs, we keep the top 30 most related chunks, without a second stage of fine-grained re-ranking as what previous works have done [Chen et al., 2024, Schlichtkrull et al., 2023].", + "section": null, + "intents": [] + }, + { + "ref_id": "c8fdcffdd866f10dd39d9fb16bacda0ec4085eb3", + "citation_marker": "VERITE: a Robust benchmark for multimodal misinformation detection accounting for unimodal bias", + "text": "While several datasets have been developed for image-text AFC, many are synthetic, generated by manually manipulating either the textual or visual modality of image-text pairs [Luo et al., 2021, Papadopoulos et al., 2024, Jia et al., 2023].", + "section": null, + "intents": [] + }, + { + "ref_id": "c8fdcffdd866f10dd39d9fb16bacda0ec4085eb3", + "citation_marker": "VERITE: a Robust benchmark for multimodal misinformation detection accounting for unimodal bias", + "text": "Some studies [Luo et al., 2021, Papadopoulos et al., 2024, Jia et al., 2023] have generated synthetic claims by applying manipulation techniques to the visual and textual modalities of image-text pairs.", + "section": null, + "intents": [] + }, + { + "ref_id": "c8fdcffdd866f10dd39d9fb16bacda0ec4085eb3", + "citation_marker": "VERITE: a Robust benchmark for multimodal misinformation detection accounting for unimodal bias", + "text": "\u2026Autosplice [Jia et al., 2023] \u2718 \u2713 - - - - 5,894 DGM [Shao et al., 2023] \u2718 \u2713 - - - - 230,000 MMFake [Liu et al., 2024b] \u2718 \u2713 - - - - 11,000 Verite [Papadopoulos et al., 2024] \u2718 \u2713 - - - - 1,000 COSMOS [Aneja et al., 2021] Mix \u2718 - - - - 201,700 FACTIFY [Mishra et al., 2022] Mix \u2718 - - - - 50,000\u2026", + "section": null, + "intents": [] + }, + { + "ref_id": "aa6a8a321de1edd2545a6c8bc1cf4fea6bfe5fb0", + "citation_marker": "AutoSplice: A Text-prompt Manipulated Image Dataset for Media Forensics", + "text": "While several datasets have been developed for image-text AFC, many are synthetic, generated by manually manipulating either the textual or visual modality of image-text pairs [Luo et al., 2021, Papadopoulos et al., 2024, Jia et al., 2023].", + "section": null, + "intents": [] + }, + { + "ref_id": "aa6a8a321de1edd2545a6c8bc1cf4fea6bfe5fb0", + "citation_marker": "AutoSplice: A Text-prompt Manipulated Image Dataset for Media Forensics", + "text": "Some studies [Luo et al., 2021, Papadopoulos et al., 2024, Jia et al., 2023] have generated synthetic claims by applying manipulation techniques to the visual and textual modalities of image-text pairs.", + "section": null, + "intents": [] + }, + { + "ref_id": "aa6a8a321de1edd2545a6c8bc1cf4fea6bfe5fb0", + "citation_marker": "AutoSplice: A Text-prompt Manipulated Image Dataset for Media Forensics", + "text": "\u2026\u2713 \u2718 \u2718 12,836 Snopes [Hanselowski et al., 2019] \u2713 \u2718 \u2718 \u2718 \u2718 \u2713 6,422 MultiFC [Augenstein et al., 2019 [Fung et al., 2021] \u2718 \u2713 - - - - 30,000 Autosplice [Jia et al., 2023] \u2718 \u2713 - - - - 5,894 DGM [Shao et al., 2023] \u2718 \u2713 - - - - 230,000 MMFake [Liu et al., 2024b] \u2718 \u2713 - - - - 11,000 Verite [Papadopoulos et\u2026", + "section": null, + "intents": [] + }, + { + "ref_id": "1a23ed979aee638ae05c030dae6d7d1f13ee3f78", + "citation_marker": "Factify 2: A multimodal fake news and satire news dataset", + "text": "\u2026et al., 2024] \u2718 \u2713 - - - - 1,000 COSMOS [Aneja et al., 2021] Mix \u2718 - - - - 201,700 FACTIFY [Mishra et al., 2022] Mix \u2718 - - - - 50,000 FACTIFY 2 [Suryavardan et al., 2023] Mix \u2718 - - - - 50,000 MMOOC [Xu et al., 2024] Mix \u2713 - - - - 364,000 Fauxtography [Zlatkova et al., 2019] \u2713 \u2718 - - - - 1,233\u2026", + "section": null, + "intents": [] + }, + { + "ref_id": "96704efeb471a64178a3356662c5fb42e451f16e", + "citation_marker": "Detecting and Grounding Multi-Modal Media Manipulation", + "text": "\u20262019] \u2713 \u2718 \u2718 \u2718 \u2718 \u2713 6,422 MultiFC [Augenstein et al., 2019 [Fung et al., 2021] \u2718 \u2713 - - - - 30,000 Autosplice [Jia et al., 2023] \u2718 \u2713 - - - - 5,894 DGM [Shao et al., 2023] \u2718 \u2713 - - - - 230,000 MMFake [Liu et al., 2024b] \u2718 \u2713 - - - - 11,000 Verite [Papadopoulos et al., 2024] \u2718 \u2713 - - - - 1,000 COSMOS\u2026", + "section": null, + "intents": [] + }, + { + "ref_id": "96704efeb471a64178a3356662c5fb42e451f16e", + "citation_marker": "Detecting and Grounding Multi-Modal Media Manipulation", + "text": "Inspired by recent research on tool usage [Wu et al., 2023, Cao et al., 2024, Braun et al., 2024], we integrate a set of specialized tools into the answer generation module, along with a tool selector that automatically selects the appropriate tool for a given question.", + "section": null, + "intents": [] + }, + { + "ref_id": "47a1930329ce5a9b148ea4d97b1f29f77c9841a4", + "citation_marker": "New Threats to Society from Free-Speech Social Media Platforms", + "text": "Misinformation has become a public concern due to its potential impact on elections, public health and safety [Allcott and Gentzkow, 2017, Aral and Eckles, 2019, B\u00e4r et al., 2023, Rocha et al., 2021].", + "section": null, + "intents": [] + }, + { + "ref_id": "bb6f8161be27877e0299af65a6a7caf3f0a1f220", + "citation_marker": "Missing Counter-Evidence Renders NLP Fact-Checking Unrealistic for Misinformation", + "text": "All retrieved pages from the interface are restricted to dates prior to the claim date to prevent temporal leakage [Glockner et al., 2022].", + "section": null, + "intents": [] + }, + { + "ref_id": "be050e69aadd6461ff3bc35dcfa7a551742ef840", + "citation_marker": "Varifocal Question Generation for Fact-checking", + "text": "To evaluate agreement on verdicts, we used Randolph\u2019s [Warrens, 2010] free-marginal multi-rater \u03ba , which is well-suited for unbalanced datasets, following previous practices [Schlichtkrull et al., 2023, Ousidhoum et al., 2022].", + "section": null, + "intents": [] + }, + { + "ref_id": "be050e69aadd6461ff3bc35dcfa7a551742ef840", + "citation_marker": "Varifocal Question Generation for Fact-checking", + "text": "As noted in prior work [Ousidhoum et al., 2022, Schlichtkrull et al., 2023], this may result in omitting critical contextual information for verification, such as context to resolve coreferences.", + "section": null, + "intents": [] + }, + { + "ref_id": "4da2b6b1677c0178416f3a613fcafdedf9ac25c4", + "citation_marker": "End-to-End Multimodal Fact-Checking and Explanation Generation: A Challenging Dataset and Models", + "text": "To support research in AFC, the research community has created various benchmark datasets [Thorne et al., 2018, Aly et al., 2021, Schlichtkrull et al., 2023, Alhindi et al., 2018, Yao et al., 2023, Chen et al., 2024], aiming to enhance the effectiveness and interpretability of fact-checking systems.", + "section": null, + "intents": [] + }, + { + "ref_id": "3d233f11e779eea67884ad871df457d5c9ff55ea", + "citation_marker": "The impact of fake news on social media and its influence on health during the COVID-19 pandemic: a systematic review", + "text": "Misinformation has become a public concern due to its potential impact on elections, public health and safety [Allcott and Gentzkow, 2017, Aral and Eckles, 2019, B\u00e4r et al., 2023, Rocha et al., 2021].", + "section": null, + "intents": [] + } +] diff --git a/replay_traces/2505.17978/two_pass_outputs/20260726T204323Z/pass_1_reasoning.selected.md b/replay_traces/2505.17978/two_pass_outputs/20260726T204323Z/pass_1_reasoning.selected.md new file mode 100644 index 0000000000000000000000000000000000000000..0a2a92983754b27de715053b6032f1ef8b7ba0cd --- /dev/null +++ b/replay_traces/2505.17978/two_pass_outputs/20260726T204323Z/pass_1_reasoning.selected.md @@ -0,0 +1,399 @@ +# 1. Cluster Evidence + +Based on the downstream cluster evidence, the target paper introduces four distinct downstream-used contributions: +- **AVerImaTeC Dataset:** Reused for new dataset creation (C1) and as an evaluation benchmark (C2). +- **Reference-based Evaluation Method:** Reused and extended for evaluating multimodal evidence in novel settings (C3). +- **Temporally Constrained Retrieval Methodology:** Reused as a protocol for dataset creation to prevent temporal leakage (C4). +- **Baseline Verification Framework:** Reused and extended for multimodal claim verification systems (C5). + +These contributions have distinct downstream roles and require different structural ingredients, so they should be split into four atomic claims. + +--- + +# 2. Claim Split Decision + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C1, C2 + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Method +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C3 + +### Claim C3 +- **claim_id:** C3 +- **artifact_type:** Protocol +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C4 + +### Claim C4 +- **claim_id:** C4 +- **artifact_type:** Method +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C5 + +### Why this split is correct +The paper bundles a dataset, an evaluation method, a data-collection protocol, and a baseline modeling framework. Downstream evidence shows that later works reuse these artifacts independently—some use the dataset for benchmarking, some adapt the evaluation method for new multimodal tasks, some adopt the temporal retrieval protocol for their own data collection, and others build on the baseline verification framework. They must be annotated separately to reflect their distinct downstream impact and structural dependencies. + +--- + +# 3. Rewritten Claims + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **rewritten_claim:** **Dataset: A real-world image-text claim verification dataset annotated with question-answer reasoning paths and web evidence, enabling the training and evaluation of multimodal fact-checking systems.** +- **why_this_is_atomic:** Isolates the dataset artifact from the evaluation methods and baseline models built around it. +- **decision:** YES_SUFFICIENT + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Method +- **rewritten_claim:** **Method: A reference-based evaluation method using large language models to separately assess the textual and visual components of retrieved evidence, enabling automated alignment with human judgments in multimodal fact-checking.** +- **why_this_is_atomic:** Isolates the specific evaluation metric/methodology from the dataset it was used to evaluate. +- **decision:** YES_SUFFICIENT + +### Claim C3 +- **claim_id:** C3 +- **artifact_type:** Protocol +- **rewritten_claim:** **Protocol: A temporally constrained evidence retrieval methodology that restricts search results to dates prior to the claim, enabling the construction of fact-checking datasets without temporal leakage.** +- **why_this_is_atomic:** Isolates the data-collection protocol from the final dataset, as downstream work reuses this specific methodology for their own dataset creation. +- **decision:** YES_SUFFICIENT + +### Claim C4 +- **claim_id:** C4 +- **artifact_type:** Method +- **rewritten_claim:** **Method: A multimodal claim verification framework combining dynamic question generation and tool-augmented answer generation, enabling automated fact-checking of image-text claims using open-web evidence.** +- **why_this_is_atomic:** Isolates the baseline modeling framework from the dataset and evaluation metrics. +- **decision:** YES_SUFFICIENT + +--- + +# 4. Minimal Ingredient Set Per Claim + +## Claim C1: AVerImaTeC Dataset + +### Ingredient C1.I1 +- **ingredient_id:** C1.I1 +- **ingredient:** Source fact-checking articles containing multimodal claims +- **why_structurally_necessary:** The dataset is built by extracting and normalizing real-world claims from existing fact-checking articles. Without this raw substrate, the dataset would not exist. +- **why_not_lower_level_substeps:** This should remain a composite data-source ingredient rather than being split into individual source datasets. +- **why_not_adjacent_implementation_details:** The exact filtering heuristics are implementation details; the structural dependency is the source text itself. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web; AMMeBa: A Large-Scale Survey and Dataset of Media-Based Misinformation In-The-Wild +- **role:** DATA_SOURCE +- **contribution:** Provides the raw real-world image-text claims that are normalized and annotated to form the dataset. +- **rationale:** The dataset is a composite of claims extracted from multiple prior fact-checking resources. No single prior study fully represents this substrate, making `NONE` the correct canonical grounding, with the primary sources listed as additional groundings. +- **evidence_span:** “we incorporated partially annotated articles from three sources: 1) filtered articles from AVeriTec... 2) articles verifying image-related claims from AMMEBA” + +### Ingredient C1.I2 +- **ingredient_id:** C1.I2 +- **ingredient:** Question-answering representation for verification rationale +- **why_structurally_necessary:** The defining feature of the dataset is that the reasoning process for verification is decomposed into QA pairs. Without this conceptual framing, it would just be a standard claim-verdict dataset. +- **why_not_lower_level_substeps:** This is a high-level conceptual framing and should not be decomposed into specific question types or answer formats. +- **why_not_adjacent_implementation_details:** The choice of QA over natural logic is the structural dependency, not the specific UI used to collect the QA pairs. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Complex Claim Verification with Evidence Retrieved in the Wild +- **additional groundings if any:** none +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the QA-based reasoning representation used to structure the evidence retrieval and verification process. +- **rationale:** The paper explicitly adopts QA as the reasoning representation for fact-checking, citing prior work that established this paradigm. Chen et al. (CLAIMDECOMP) is a clean representative for QA-based claim decomposition. +- **evidence_span:** “To capture the rationale in claim verification, a complex reasoning task, various reasoning representations have been explored... and QA” + +### Ingredient C1.I3 +- **ingredient_id:** C1.I3 +- **ingredient:** Four-way veracity labeling schema +- **why_structurally_necessary:** The dataset categorizes claims into specific veracity labels. This schema defines the classification task the dataset supports. +- **why_not_lower_level_substeps:** This is a single conceptual ingredient defining the label space. +- **why_not_adjacent_implementation_details:** The specific definitions of each label are part of this broader schema ingredient. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web +- **additional groundings if any:** none +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the specific four-way label space (supported, refuted, not enough evidence, conflicting) used to annotate the claims. +- **rationale:** The paper explicitly adopts the exact four-way veracity labeling schema introduced in AVeriTeC. +- **evidence_span:** “We follow the four-way veracity labeling schema from [Schlichtkrull et al., 2023]: supported, refuted, not enough evidence, and conflicting/cherry-picking.” + +### Ingredient C1.I4 +- **ingredient_id:** C1.I4 +- **ingredient:** Multi-stage annotation pipeline with evidence sufficiency checks +- **why_structurally_necessary:** The dataset's quality and reliability depend on a structured annotation workflow that includes independent sufficiency checks to resolve conflicts. +- **why_not_lower_level_substeps:** This should remain one protocol-level ingredient rather than being split into claim extraction, QA generation, and sufficiency checking separately. +- **why_not_adjacent_implementation_details:** The structural dependency is the pipeline itself, not the specific number of annotators or their demographics. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the foundational multi-stage annotation and verification workflow that the paper extends for image-text claims. +- **rationale:** The paper explicitly states that its annotation pipeline is an extension of the process proposed in AVeriTeC, making it the canonical grounding for this workflow. +- **evidence_span:** “extending the annotation process proposed in [Schlichtkrull et al., 2023] to the domain of image-text claims.” + +--- + +## Claim C2: Reference-based Evaluation Method + +### Ingredient C2.I1 +- **ingredient_id:** C2.I1 +- **ingredient:** LLM-based reference evaluation framework for evidence retrieval +- **why_structurally_necessary:** The method relies on using an LLM to compare retrieved evidence against human-annotated references, rather than relying on exact string matching. +- **why_not_lower_level_substeps:** This is the core conceptual basis of the evaluation method and should not be split into specific prompt designs. +- **why_not_adjacent_implementation_details:** The choice of Gemini as the specific LLM is an implementation detail; the structural dependency is the LLM-as-a-judge framework. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the foundation for using LLMs to evaluate evidence retrieval by comparing it to references. +- **rationale:** The paper explicitly adopts the Ev2R framework for its reference-based evaluation of textual evidence. +- **evidence_span:** “Recent research [Akhtar et al., 2024] showed that reference-based evaluation of evidence with large language models (LLMs) aligns best with human assessments.” + +### Ingredient C2.I2 +- **ingredient_id:** C2.I2 +- **ingredient:** Separated reference-based evaluation for textual and visual components +- **why_structurally_necessary:** The core novelty of the method is extending reference-based evaluation to multimodal evidence by evaluating text and images separately, which the paper shows is more robust than interleaved evaluation. +- **why_not_lower_level_substeps:** This should remain a single protocol ingredient rather than being split into the text-check step and the image-check step. +- **why_not_adjacent_implementation_details:** The specific similarity thresholds used are implementation details. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the novel mechanism for robustly evaluating multimodal evidence by decoupling the modalities during the LLM judgment phase. +- **rationale:** This separated evaluation strategy is the target paper's own novel extension to prior text-only reference evaluation methods, so it maps to `NONE`. +- **evidence_span:** “We then conduct separate reference-based evaluations for the textual and visual components.” + +--- + +## Claim C3: Temporally Constrained Retrieval Methodology + +### Ingredient C3.I1 +- **ingredient_id:** C3.I1 +- **ingredient:** Identification of temporal leakage as a critical flaw in fact-checking datasets +- **why_structurally_necessary:** The entire protocol is designed to solve the problem of temporal leakage (where models use future information to verify past claims). Without this conceptual framing, the protocol has no purpose. +- **why_not_lower_level_substeps:** This is a high-level conceptual ingredient. +- **why_not_adjacent_implementation_details:** N/A +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Missing Counter-Evidence Renders NLP Fact-Checking Unrealistic for Misinformation +- **additional groundings if any:** none +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the theoretical motivation for restricting evidence retrieval to prevent models from accessing future knowledge. +- **rationale:** The paper explicitly cites Glockner et al. as the motivation for preventing temporal leakage during evidence collection. +- **evidence_span:** “restricted to dates prior to the claim date to prevent temporal leakage [Glockner et al., 2022].” + +### Ingredient C3.I2 +- **ingredient_id:** C3.I2 +- **ingredient:** Date-restricted search interface and post-hoc filtering mechanism +- **why_structurally_necessary:** This is the operational mechanism of the protocol. Without restricting the search API and filtering results by date, the temporal constraint cannot be enforced. +- **why_not_lower_level_substeps:** This should remain one mechanism ingredient rather than being split into the Google Search API arguments and the Python htmldate package. +- **why_not_adjacent_implementation_details:** The specific Python packages used are implementation details; the structural dependency is the date-restriction mechanism itself. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the practical tooling to enforce the temporal constraints during dataset construction. +- **rationale:** The specific implementation of the custom search interface and post-hoc date filtering is developed in the target paper, mapping to `NONE`. +- **evidence_span:** “annotators are provided with a custom Google search interface... restricted to dates prior to the claim date” + +--- + +## Claim C4: Baseline Verification Framework + +### Ingredient C4.I1 +- **ingredient_id:** C4.I1 +- **ingredient:** Tool-augmented answer generation module with specialized search and VQA tools +- **why_structurally_necessary:** The baseline framework relies on an LLM selecting and using external tools (RIS, WST, WSI, VQA) to answer generated questions. Without this tool-use architecture, the framework could not retrieve open-web multimodal evidence. +- **why_not_lower_level_substeps:** This should remain one architectural ingredient rather than being split into each individual tool. +- **why_not_adjacent_implementation_details:** The specific prompts used for tool selection are implementation details. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models +- **additional groundings if any:** none +- **role:** CORE_METHOD +- **contribution:** Provides the paradigm of augmenting language models with specialized visual and search tools to solve complex multimodal tasks. +- **rationale:** The paper explicitly grounds its tool-augmented answer generation module in recent research on tool usage, with Visual ChatGPT (Wu et al.) being a canonical representative for integrating visual tools with LLMs. +- **evidence_span:** “we integrate a set of specialized tools into the answer generation module, along with a tool selector” + +### Ingredient C4.I2 +- **ingredient_id:** C4.I2 +- **ingredient:** Dynamic question generation strategy based on evolving evidence history +- **why_structurally_necessary:** The framework introduces dynamic and hybrid question generation to handle multi-hop reasoning where subsequent questions depend on previous answers. +- **why_not_lower_level_substeps:** This is a core methodological strategy and should not be decomposed into specific prompt templates. +- **why_not_adjacent_implementation_details:** The choice of specific MLLMs (like Qwen-VL) is an implementation detail; the structural dependency is the dynamic generation strategy. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** CORE_METHOD +- **contribution:** Provides the mechanism for generating sequential verification questions conditioned on the accumulated evidence context. +- **rationale:** The dynamic and hybrid question generation strategies are proposed by the target paper as novel components of their baseline framework, mapping to `NONE`. +- **evidence_span:** “we also propose a dynamic question generation (DQG) method, where each subsequent question is generated based on both the claim and the evolving evidence history.” + +--- + +# 5. Excluded Tempting Non-Ingredients + +- **Specific LLMs (Gemini, Qwen, Gemma, LLaVA)** + - Excluded because they are interchangeable implementation choices for the baseline framework and evaluation method, rather than structural prerequisites defining the artifacts themselves. +- **BM25 for text chunk retrieval** + - Excluded because it is a low-level implementation detail for the text search tool, not a high-level structural ingredient of the baseline framework. +- **Python htmldate package** + - Excluded because it is merely a local tool used to implement the date-restriction mechanism, which is already captured at a higher structural level. + +--- + +# 6. Candidate Grounding Decisions + +### Claim C1: AVerImaTeC Dataset +- **C1.I1:** NONE (accepted_none). The source data is composite. AVeriTeC and AMMeBa are accepted as additional groundings. +- **C1.I2:** Complex Claim Verification with Evidence Retrieved in the Wild (accepted_canonical). Clean representative for QA-based claim decomposition. +- **C1.I3:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web (accepted_canonical). Directly provides the four-way veracity schema. +- **C1.I4:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web (accepted_canonical). Directly provides the foundational annotation pipeline. + +### Claim C2: Reference-based Evaluation Method +- **C2.I1:** Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking (accepted_canonical). Directly provides the LLM-based reference evaluation framework. +- **C2.I2:** NONE (accepted_none). The separated multimodal evaluation strategy is introduced in the target paper. + +### Claim C3: Temporally Constrained Retrieval Methodology +- **C3.I1:** Missing Counter-Evidence Renders NLP Fact-Checking Unrealistic for Misinformation (accepted_canonical). Provides the conceptual motivation regarding temporal leakage. +- **C3.I2:** NONE (accepted_none). The specific date-restricted search implementation is from the target paper. + +### Claim C4: Baseline Verification Framework +- **C4.I1:** Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models (accepted_canonical). Representative for tool-augmented LLM architectures. +- **C4.I2:** NONE (accepted_none). The dynamic question generation strategy is proposed in the target paper. + +--- + +# 7. Final Grounding Decisions + +## Claim C1 +- **C1.I1** → NONE → DATA_SOURCE + - additional: AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web; AMMeBa: A Large-Scale Survey and Dataset of Media-Based Misinformation In-The-Wild +- **C1.I2** → Complex Claim Verification with Evidence Retrieved in the Wild → CONCEPTUAL_FRAMEWORK +- **C1.I3** → AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web → CONCEPTUAL_FRAMEWORK +- **C1.I4** → AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web → EVALUATION_PROTOCOL + +## Claim C2 +- **C2.I1** → Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking → EVALUATION_PROTOCOL +- **C2.I2** → NONE → EVALUATION_PROTOCOL + +## Claim C3 +- **C3.I1** → Missing Counter-Evidence Renders NLP Fact-Checking Unrealistic for Misinformation → CONCEPTUAL_FRAMEWORK +- **C3.I2** → NONE → IMPLEMENTATION_TOOLING + +## Claim C4 +- **C4.I1** → Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models → CORE_METHOD +- **C4.I2** → NONE → CORE_METHOD + +--- + +# 8. Final Rationales And Evidence + +## Claim C1: AVerImaTeC Dataset + +### C1.I1 +- **ingredient:** Source fact-checking articles containing multimodal claims +- **canonical study:** NONE +- **additional studies:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web; AMMeBa: A Large-Scale Survey and Dataset of Media-Based Misinformation In-The-Wild +- **role:** DATA_SOURCE +- **contribution:** Provides the raw real-world image-text claims that are normalized and annotated to form the dataset. +- **rationale:** The dataset is a composite of claims extracted from multiple prior fact-checking resources. No single prior study fully represents this substrate, making `NONE` the correct canonical grounding, with the primary sources listed as additional groundings. +- **evidence_span:** “we incorporated partially annotated articles from three sources: 1) filtered articles from AVeriTec... 2) articles verifying image-related claims from AMMEBA” + +### C1.I2 +- **ingredient:** Question-answering representation for verification rationale +- **canonical study:** Complex Claim Verification with Evidence Retrieved in the Wild +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the QA-based reasoning representation used to structure the evidence retrieval and verification process. +- **rationale:** The paper explicitly adopts QA as the reasoning representation for fact-checking, citing prior work that established this paradigm. Chen et al. (CLAIMDECOMP) is a clean representative for QA-based claim decomposition. +- **evidence_span:** “To capture the rationale in claim verification, a complex reasoning task, various reasoning representations have been explored... and QA” + +### C1.I3 +- **ingredient:** Four-way veracity labeling schema +- **canonical study:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the specific four-way label space (supported, refuted, not enough evidence, conflicting) used to annotate the claims. +- **rationale:** The paper explicitly adopts the exact four-way veracity labeling schema introduced in AVeriTeC. +- **evidence_span:** “We follow the four-way veracity labeling schema from [Schlichtkrull et al., 2023]: supported, refuted, not enough evidence, and conflicting/cherry-picking.” + +### C1.I4 +- **ingredient:** Multi-stage annotation pipeline with evidence sufficiency checks +- **canonical study:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the foundational multi-stage annotation and verification workflow that the paper extends for image-text claims. +- **rationale:** The paper explicitly states that its annotation pipeline is an extension of the process proposed in AVeriTeC, making it the canonical grounding for this workflow. +- **evidence_span:** “extending the annotation process proposed in [Schlichtkrull et al., 2023] to the domain of image-text claims.” + +## Claim C2: Reference-based Evaluation Method + +### C2.I1 +- **ingredient:** LLM-based reference evaluation framework for evidence retrieval +- **canonical study:** Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the foundation for using LLMs to evaluate evidence retrieval by comparing it to references. +- **rationale:** The paper explicitly adopts the Ev2R framework for its reference-based evaluation of textual evidence. +- **evidence_span:** “Recent research [Akhtar et al., 2024] showed that reference-based evaluation of evidence with large language models (LLMs) aligns best with human assessments.” + +### C2.I2 +- **ingredient:** Separated reference-based evaluation for textual and visual components +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the novel mechanism for robustly evaluating multimodal evidence by decoupling the modalities during the LLM judgment phase. +- **rationale:** This separated evaluation strategy is the target paper's own novel extension to prior text-only reference evaluation methods, so it maps to `NONE`. +- **evidence_span:** “We then conduct separate reference-based evaluations for the textual and visual components.” + +## Claim C3: Temporally Constrained Retrieval Methodology + +### C3.I1 +- **ingredient:** Identification of temporal leakage as a critical flaw in fact-checking datasets +- **canonical study:** Missing Counter-Evidence Renders NLP Fact-Checking Unrealistic for Misinformation +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the theoretical motivation for restricting evidence retrieval to prevent models from accessing future knowledge. +- **rationale:** The paper explicitly cites Glockner et al. as the motivation for preventing temporal leakage during evidence collection. +- **evidence_span:** “restricted to dates prior to the claim date to prevent temporal leakage [Glockner et al., 2022].” + +### C3.I2 +- **ingredient:** Date-restricted search interface and post-hoc filtering mechanism +- **canonical study:** NONE +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the practical tooling to enforce the temporal constraints during dataset construction. +- **rationale:** The specific implementation of the custom search interface and post-hoc date filtering is developed in the target paper, mapping to `NONE`. +- **evidence_span:** “annotators are provided with a custom Google search interface... restricted to dates prior to the claim date” + +## Claim C4: Baseline Verification Framework + +### C4.I1 +- **ingredient:** Tool-augmented answer generation module with specialized search and VQA tools +- **canonical study:** Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models +- **role:** CORE_METHOD +- **contribution:** Provides the paradigm of augmenting language models with specialized visual and search tools to solve complex multimodal tasks. +- **rationale:** The paper explicitly grounds its tool-augmented answer generation module in recent research on tool usage, with Visual ChatGPT (Wu et al.) being a canonical representative for integrating visual tools with LLMs. +- **evidence_span:** “we integrate a set of specialized tools into the answer generation module, along with a tool selector” + +### C4.I2 +- **ingredient:** Dynamic question generation strategy based on evolving evidence history +- **canonical study:** NONE +- **role:** CORE_METHOD +- **contribution:** Provides the mechanism for generating sequential verification questions conditioned on the accumulated evidence context. +- **rationale:** The dynamic and hybrid question generation strategies are proposed by the target paper as novel components of their baseline framework, mapping to `NONE`. +- **evidence_span:** “we also propose a dynamic question generation (DQG) method, where each subsequent question is generated based on both the claim and the evolving evidence history.” \ No newline at end of file diff --git a/replay_traces/2505.17978/two_pass_outputs/20260726T204323Z/pass_2_formatter.prompt.txt b/replay_traces/2505.17978/two_pass_outputs/20260726T204323Z/pass_2_formatter.prompt.txt new file mode 100644 index 0000000000000000000000000000000000000000..e76ce94fadc58e7c1e4545a81d4c7f38ee1a99d3 --- /dev/null +++ b/replay_traces/2505.17978/two_pass_outputs/20260726T204323Z/pass_2_formatter.prompt.txt @@ -0,0 +1,549 @@ +Task: Convert the reasoning memo into the final annotation UI JSON payload. + +This is a strict formatting pass. +Do not do new reasoning unless needed to resolve minor schema ambiguities. +Follow the memo’s final decisions faithfully. + +Critical formatting priorities: +1. Preserve the final claim split. +2. Preserve the final minimal ingredient set. +3. Respect exclusions of tempting non-ingredients. +4. Preserve canonical vs additional vs NONE decisions exactly. +5. Do not inflate the ingredient set. + +Target schema: +{ + "target_paper_id": "...", + "target_title": "...", + "target_year": 2024, + "annotator_id": "streamlit_hf_space", + "active_claim_id": "...", + "claims": [ + { + "claim_id": "...", + "text": "...", + "rewritten_claim": "...", + "cluster_id": "...", + "decision": "YES_SUFFICIENT | NO_NOT_DISCOVERY | UNCERTAIN", + "notes": "", + "ingredients": [ + { + "ingredient_id": "...", + "ingredient": "...", + "canonical_ref_id": "__NONE__ or ref_id", + "canonical_grounding": null or { + "ref_id": "...", + "bib_key": "...", + "paper_id": "...", + "external_ids": {}, + "ref_title": "...", + "ref_year": "...", + "ref_authors": "..." + }, + "additional_ref_ids": ["..."], + "additional_groundings": [ + { + "ref_id": "...", + "bib_key": "...", + "paper_id": "...", + "external_ids": {}, + "ref_title": "...", + "ref_year": "...", + "ref_authors": "..." + } + ], + "canonical_annotation": { + "role": null or "CONCEPTUAL_FRAMEWORK" or "CORE_METHOD" or "DATA_SOURCE" or "MODEL_INITIALIZATION" or "EVALUATION_PROTOCOL" or "IMPLEMENTATION_TOOLING" or "TRAINING_DATA", + "roles": ["..."], + "contribution": "...", + "rationale": "...", + "evidence_span": "..." + } + } + ], + "enabling_discoveries": [ + { + "ref_id": "...", + "bib_key": "...", + "paper_id": "...", + "external_ids": {}, + "ref_title": "...", + "ref_year": "...", + "ref_authors": "...", + "ingredient_id": "...", + "ingredient": "...", + "role": null or "CONCEPTUAL_FRAMEWORK" or "CORE_METHOD" or "DATA_SOURCE" or "MODEL_INITIALIZATION" or "EVALUATION_PROTOCOL" or "IMPLEMENTATION_TOOLING" or "TRAINING_DATA", + "roles": ["..."], + "contribution": "...", + "rationale": "...", + "evidence_span": "..." + } + ] + } + ] +} + +Formatting rules: +- `text` must equal `rewritten_claim` +- `active_claim_id` should be the first claim_id unless the reasoning memo explicitly recommends another active claim +- if canonical grounding is NONE: + - `canonical_ref_id` must be "__NONE__" + - `canonical_grounding` must be null + - do not include that ingredient in `enabling_discoveries` +- if roles has length 1, `role` should equal that role +- if roles has length != 1, `role` should be null +- `additional_ref_ids` must match `additional_groundings[].ref_id` +- `enabling_discoveries` should contain only canonical non-NONE ingredients +- use only the allowed role set + +REASONING MEMO: +# 1. Cluster Evidence + +Based on the downstream cluster evidence, the target paper introduces four distinct downstream-used contributions: +- **AVerImaTeC Dataset:** Reused for new dataset creation (C1) and as an evaluation benchmark (C2). +- **Reference-based Evaluation Method:** Reused and extended for evaluating multimodal evidence in novel settings (C3). +- **Temporally Constrained Retrieval Methodology:** Reused as a protocol for dataset creation to prevent temporal leakage (C4). +- **Baseline Verification Framework:** Reused and extended for multimodal claim verification systems (C5). + +These contributions have distinct downstream roles and require different structural ingredients, so they should be split into four atomic claims. + +--- + +# 2. Claim Split Decision + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C1, C2 + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Method +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C3 + +### Claim C3 +- **claim_id:** C3 +- **artifact_type:** Protocol +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C4 + +### Claim C4 +- **claim_id:** C4 +- **artifact_type:** Method +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C5 + +### Why this split is correct +The paper bundles a dataset, an evaluation method, a data-collection protocol, and a baseline modeling framework. Downstream evidence shows that later works reuse these artifacts independently—some use the dataset for benchmarking, some adapt the evaluation method for new multimodal tasks, some adopt the temporal retrieval protocol for their own data collection, and others build on the baseline verification framework. They must be annotated separately to reflect their distinct downstream impact and structural dependencies. + +--- + +# 3. Rewritten Claims + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Dataset +- **rewritten_claim:** **Dataset: A real-world image-text claim verification dataset annotated with question-answer reasoning paths and web evidence, enabling the training and evaluation of multimodal fact-checking systems.** +- **why_this_is_atomic:** Isolates the dataset artifact from the evaluation methods and baseline models built around it. +- **decision:** YES_SUFFICIENT + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Method +- **rewritten_claim:** **Method: A reference-based evaluation method using large language models to separately assess the textual and visual components of retrieved evidence, enabling automated alignment with human judgments in multimodal fact-checking.** +- **why_this_is_atomic:** Isolates the specific evaluation metric/methodology from the dataset it was used to evaluate. +- **decision:** YES_SUFFICIENT + +### Claim C3 +- **claim_id:** C3 +- **artifact_type:** Protocol +- **rewritten_claim:** **Protocol: A temporally constrained evidence retrieval methodology that restricts search results to dates prior to the claim, enabling the construction of fact-checking datasets without temporal leakage.** +- **why_this_is_atomic:** Isolates the data-collection protocol from the final dataset, as downstream work reuses this specific methodology for their own dataset creation. +- **decision:** YES_SUFFICIENT + +### Claim C4 +- **claim_id:** C4 +- **artifact_type:** Method +- **rewritten_claim:** **Method: A multimodal claim verification framework combining dynamic question generation and tool-augmented answer generation, enabling automated fact-checking of image-text claims using open-web evidence.** +- **why_this_is_atomic:** Isolates the baseline modeling framework from the dataset and evaluation metrics. +- **decision:** YES_SUFFICIENT + +--- + +# 4. Minimal Ingredient Set Per Claim + +## Claim C1: AVerImaTeC Dataset + +### Ingredient C1.I1 +- **ingredient_id:** C1.I1 +- **ingredient:** Source fact-checking articles containing multimodal claims +- **why_structurally_necessary:** The dataset is built by extracting and normalizing real-world claims from existing fact-checking articles. Without this raw substrate, the dataset would not exist. +- **why_not_lower_level_substeps:** This should remain a composite data-source ingredient rather than being split into individual source datasets. +- **why_not_adjacent_implementation_details:** The exact filtering heuristics are implementation details; the structural dependency is the source text itself. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web; AMMeBa: A Large-Scale Survey and Dataset of Media-Based Misinformation In-The-Wild +- **role:** DATA_SOURCE +- **contribution:** Provides the raw real-world image-text claims that are normalized and annotated to form the dataset. +- **rationale:** The dataset is a composite of claims extracted from multiple prior fact-checking resources. No single prior study fully represents this substrate, making `NONE` the correct canonical grounding, with the primary sources listed as additional groundings. +- **evidence_span:** “we incorporated partially annotated articles from three sources: 1) filtered articles from AVeriTec... 2) articles verifying image-related claims from AMMEBA” + +### Ingredient C1.I2 +- **ingredient_id:** C1.I2 +- **ingredient:** Question-answering representation for verification rationale +- **why_structurally_necessary:** The defining feature of the dataset is that the reasoning process for verification is decomposed into QA pairs. Without this conceptual framing, it would just be a standard claim-verdict dataset. +- **why_not_lower_level_substeps:** This is a high-level conceptual framing and should not be decomposed into specific question types or answer formats. +- **why_not_adjacent_implementation_details:** The choice of QA over natural logic is the structural dependency, not the specific UI used to collect the QA pairs. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Complex Claim Verification with Evidence Retrieved in the Wild +- **additional groundings if any:** none +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the QA-based reasoning representation used to structure the evidence retrieval and verification process. +- **rationale:** The paper explicitly adopts QA as the reasoning representation for fact-checking, citing prior work that established this paradigm. Chen et al. (CLAIMDECOMP) is a clean representative for QA-based claim decomposition. +- **evidence_span:** “To capture the rationale in claim verification, a complex reasoning task, various reasoning representations have been explored... and QA” + +### Ingredient C1.I3 +- **ingredient_id:** C1.I3 +- **ingredient:** Four-way veracity labeling schema +- **why_structurally_necessary:** The dataset categorizes claims into specific veracity labels. This schema defines the classification task the dataset supports. +- **why_not_lower_level_substeps:** This is a single conceptual ingredient defining the label space. +- **why_not_adjacent_implementation_details:** The specific definitions of each label are part of this broader schema ingredient. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web +- **additional groundings if any:** none +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the specific four-way label space (supported, refuted, not enough evidence, conflicting) used to annotate the claims. +- **rationale:** The paper explicitly adopts the exact four-way veracity labeling schema introduced in AVeriTeC. +- **evidence_span:** “We follow the four-way veracity labeling schema from [Schlichtkrull et al., 2023]: supported, refuted, not enough evidence, and conflicting/cherry-picking.” + +### Ingredient C1.I4 +- **ingredient_id:** C1.I4 +- **ingredient:** Multi-stage annotation pipeline with evidence sufficiency checks +- **why_structurally_necessary:** The dataset's quality and reliability depend on a structured annotation workflow that includes independent sufficiency checks to resolve conflicts. +- **why_not_lower_level_substeps:** This should remain one protocol-level ingredient rather than being split into claim extraction, QA generation, and sufficiency checking separately. +- **why_not_adjacent_implementation_details:** The structural dependency is the pipeline itself, not the specific number of annotators or their demographics. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the foundational multi-stage annotation and verification workflow that the paper extends for image-text claims. +- **rationale:** The paper explicitly states that its annotation pipeline is an extension of the process proposed in AVeriTeC, making it the canonical grounding for this workflow. +- **evidence_span:** “extending the annotation process proposed in [Schlichtkrull et al., 2023] to the domain of image-text claims.” + +--- + +## Claim C2: Reference-based Evaluation Method + +### Ingredient C2.I1 +- **ingredient_id:** C2.I1 +- **ingredient:** LLM-based reference evaluation framework for evidence retrieval +- **why_structurally_necessary:** The method relies on using an LLM to compare retrieved evidence against human-annotated references, rather than relying on exact string matching. +- **why_not_lower_level_substeps:** This is the core conceptual basis of the evaluation method and should not be split into specific prompt designs. +- **why_not_adjacent_implementation_details:** The choice of Gemini as the specific LLM is an implementation detail; the structural dependency is the LLM-as-a-judge framework. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the foundation for using LLMs to evaluate evidence retrieval by comparing it to references. +- **rationale:** The paper explicitly adopts the Ev2R framework for its reference-based evaluation of textual evidence. +- **evidence_span:** “Recent research [Akhtar et al., 2024] showed that reference-based evaluation of evidence with large language models (LLMs) aligns best with human assessments.” + +### Ingredient C2.I2 +- **ingredient_id:** C2.I2 +- **ingredient:** Separated reference-based evaluation for textual and visual components +- **why_structurally_necessary:** The core novelty of the method is extending reference-based evaluation to multimodal evidence by evaluating text and images separately, which the paper shows is more robust than interleaved evaluation. +- **why_not_lower_level_substeps:** This should remain a single protocol ingredient rather than being split into the text-check step and the image-check step. +- **why_not_adjacent_implementation_details:** The specific similarity thresholds used are implementation details. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the novel mechanism for robustly evaluating multimodal evidence by decoupling the modalities during the LLM judgment phase. +- **rationale:** This separated evaluation strategy is the target paper's own novel extension to prior text-only reference evaluation methods, so it maps to `NONE`. +- **evidence_span:** “We then conduct separate reference-based evaluations for the textual and visual components.” + +--- + +## Claim C3: Temporally Constrained Retrieval Methodology + +### Ingredient C3.I1 +- **ingredient_id:** C3.I1 +- **ingredient:** Identification of temporal leakage as a critical flaw in fact-checking datasets +- **why_structurally_necessary:** The entire protocol is designed to solve the problem of temporal leakage (where models use future information to verify past claims). Without this conceptual framing, the protocol has no purpose. +- **why_not_lower_level_substeps:** This is a high-level conceptual ingredient. +- **why_not_adjacent_implementation_details:** N/A +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Missing Counter-Evidence Renders NLP Fact-Checking Unrealistic for Misinformation +- **additional groundings if any:** none +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the theoretical motivation for restricting evidence retrieval to prevent models from accessing future knowledge. +- **rationale:** The paper explicitly cites Glockner et al. as the motivation for preventing temporal leakage during evidence collection. +- **evidence_span:** “restricted to dates prior to the claim date to prevent temporal leakage [Glockner et al., 2022].” + +### Ingredient C3.I2 +- **ingredient_id:** C3.I2 +- **ingredient:** Date-restricted search interface and post-hoc filtering mechanism +- **why_structurally_necessary:** This is the operational mechanism of the protocol. Without restricting the search API and filtering results by date, the temporal constraint cannot be enforced. +- **why_not_lower_level_substeps:** This should remain one mechanism ingredient rather than being split into the Google Search API arguments and the Python htmldate package. +- **why_not_adjacent_implementation_details:** The specific Python packages used are implementation details; the structural dependency is the date-restriction mechanism itself. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the practical tooling to enforce the temporal constraints during dataset construction. +- **rationale:** The specific implementation of the custom search interface and post-hoc date filtering is developed in the target paper, mapping to `NONE`. +- **evidence_span:** “annotators are provided with a custom Google search interface... restricted to dates prior to the claim date” + +--- + +## Claim C4: Baseline Verification Framework + +### Ingredient C4.I1 +- **ingredient_id:** C4.I1 +- **ingredient:** Tool-augmented answer generation module with specialized search and VQA tools +- **why_structurally_necessary:** The baseline framework relies on an LLM selecting and using external tools (RIS, WST, WSI, VQA) to answer generated questions. Without this tool-use architecture, the framework could not retrieve open-web multimodal evidence. +- **why_not_lower_level_substeps:** This should remain one architectural ingredient rather than being split into each individual tool. +- **why_not_adjacent_implementation_details:** The specific prompts used for tool selection are implementation details. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models +- **additional groundings if any:** none +- **role:** CORE_METHOD +- **contribution:** Provides the paradigm of augmenting language models with specialized visual and search tools to solve complex multimodal tasks. +- **rationale:** The paper explicitly grounds its tool-augmented answer generation module in recent research on tool usage, with Visual ChatGPT (Wu et al.) being a canonical representative for integrating visual tools with LLMs. +- **evidence_span:** “we integrate a set of specialized tools into the answer generation module, along with a tool selector” + +### Ingredient C4.I2 +- **ingredient_id:** C4.I2 +- **ingredient:** Dynamic question generation strategy based on evolving evidence history +- **why_structurally_necessary:** The framework introduces dynamic and hybrid question generation to handle multi-hop reasoning where subsequent questions depend on previous answers. +- **why_not_lower_level_substeps:** This is a core methodological strategy and should not be decomposed into specific prompt templates. +- **why_not_adjacent_implementation_details:** The choice of specific MLLMs (like Qwen-VL) is an implementation detail; the structural dependency is the dynamic generation strategy. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** CORE_METHOD +- **contribution:** Provides the mechanism for generating sequential verification questions conditioned on the accumulated evidence context. +- **rationale:** The dynamic and hybrid question generation strategies are proposed by the target paper as novel components of their baseline framework, mapping to `NONE`. +- **evidence_span:** “we also propose a dynamic question generation (DQG) method, where each subsequent question is generated based on both the claim and the evolving evidence history.” + +--- + +# 5. Excluded Tempting Non-Ingredients + +- **Specific LLMs (Gemini, Qwen, Gemma, LLaVA)** + - Excluded because they are interchangeable implementation choices for the baseline framework and evaluation method, rather than structural prerequisites defining the artifacts themselves. +- **BM25 for text chunk retrieval** + - Excluded because it is a low-level implementation detail for the text search tool, not a high-level structural ingredient of the baseline framework. +- **Python htmldate package** + - Excluded because it is merely a local tool used to implement the date-restriction mechanism, which is already captured at a higher structural level. + +--- + +# 6. Candidate Grounding Decisions + +### Claim C1: AVerImaTeC Dataset +- **C1.I1:** NONE (accepted_none). The source data is composite. AVeriTeC and AMMeBa are accepted as additional groundings. +- **C1.I2:** Complex Claim Verification with Evidence Retrieved in the Wild (accepted_canonical). Clean representative for QA-based claim decomposition. +- **C1.I3:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web (accepted_canonical). Directly provides the four-way veracity schema. +- **C1.I4:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web (accepted_canonical). Directly provides the foundational annotation pipeline. + +### Claim C2: Reference-based Evaluation Method +- **C2.I1:** Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking (accepted_canonical). Directly provides the LLM-based reference evaluation framework. +- **C2.I2:** NONE (accepted_none). The separated multimodal evaluation strategy is introduced in the target paper. + +### Claim C3: Temporally Constrained Retrieval Methodology +- **C3.I1:** Missing Counter-Evidence Renders NLP Fact-Checking Unrealistic for Misinformation (accepted_canonical). Provides the conceptual motivation regarding temporal leakage. +- **C3.I2:** NONE (accepted_none). The specific date-restricted search implementation is from the target paper. + +### Claim C4: Baseline Verification Framework +- **C4.I1:** Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models (accepted_canonical). Representative for tool-augmented LLM architectures. +- **C4.I2:** NONE (accepted_none). The dynamic question generation strategy is proposed in the target paper. + +--- + +# 7. Final Grounding Decisions + +## Claim C1 +- **C1.I1** → NONE → DATA_SOURCE + - additional: AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web; AMMeBa: A Large-Scale Survey and Dataset of Media-Based Misinformation In-The-Wild +- **C1.I2** → Complex Claim Verification with Evidence Retrieved in the Wild → CONCEPTUAL_FRAMEWORK +- **C1.I3** → AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web → CONCEPTUAL_FRAMEWORK +- **C1.I4** → AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web → EVALUATION_PROTOCOL + +## Claim C2 +- **C2.I1** → Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking → EVALUATION_PROTOCOL +- **C2.I2** → NONE → EVALUATION_PROTOCOL + +## Claim C3 +- **C3.I1** → Missing Counter-Evidence Renders NLP Fact-Checking Unrealistic for Misinformation → CONCEPTUAL_FRAMEWORK +- **C3.I2** → NONE → IMPLEMENTATION_TOOLING + +## Claim C4 +- **C4.I1** → Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models → CORE_METHOD +- **C4.I2** → NONE → CORE_METHOD + +--- + +# 8. Final Rationales And Evidence + +## Claim C1: AVerImaTeC Dataset + +### C1.I1 +- **ingredient:** Source fact-checking articles containing multimodal claims +- **canonical study:** NONE +- **additional studies:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web; AMMeBa: A Large-Scale Survey and Dataset of Media-Based Misinformation In-The-Wild +- **role:** DATA_SOURCE +- **contribution:** Provides the raw real-world image-text claims that are normalized and annotated to form the dataset. +- **rationale:** The dataset is a composite of claims extracted from multiple prior fact-checking resources. No single prior study fully represents this substrate, making `NONE` the correct canonical grounding, with the primary sources listed as additional groundings. +- **evidence_span:** “we incorporated partially annotated articles from three sources: 1) filtered articles from AVeriTec... 2) articles verifying image-related claims from AMMEBA” + +### C1.I2 +- **ingredient:** Question-answering representation for verification rationale +- **canonical study:** Complex Claim Verification with Evidence Retrieved in the Wild +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the QA-based reasoning representation used to structure the evidence retrieval and verification process. +- **rationale:** The paper explicitly adopts QA as the reasoning representation for fact-checking, citing prior work that established this paradigm. Chen et al. (CLAIMDECOMP) is a clean representative for QA-based claim decomposition. +- **evidence_span:** “To capture the rationale in claim verification, a complex reasoning task, various reasoning representations have been explored... and QA” + +### C1.I3 +- **ingredient:** Four-way veracity labeling schema +- **canonical study:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the specific four-way label space (supported, refuted, not enough evidence, conflicting) used to annotate the claims. +- **rationale:** The paper explicitly adopts the exact four-way veracity labeling schema introduced in AVeriTeC. +- **evidence_span:** “We follow the four-way veracity labeling schema from [Schlichtkrull et al., 2023]: supported, refuted, not enough evidence, and conflicting/cherry-picking.” + +### C1.I4 +- **ingredient:** Multi-stage annotation pipeline with evidence sufficiency checks +- **canonical study:** AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the foundational multi-stage annotation and verification workflow that the paper extends for image-text claims. +- **rationale:** The paper explicitly states that its annotation pipeline is an extension of the process proposed in AVeriTeC, making it the canonical grounding for this workflow. +- **evidence_span:** “extending the annotation process proposed in [Schlichtkrull et al., 2023] to the domain of image-text claims.” + +## Claim C2: Reference-based Evaluation Method + +### C2.I1 +- **ingredient:** LLM-based reference evaluation framework for evidence retrieval +- **canonical study:** Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the foundation for using LLMs to evaluate evidence retrieval by comparing it to references. +- **rationale:** The paper explicitly adopts the Ev2R framework for its reference-based evaluation of textual evidence. +- **evidence_span:** “Recent research [Akhtar et al., 2024] showed that reference-based evaluation of evidence with large language models (LLMs) aligns best with human assessments.” + +### C2.I2 +- **ingredient:** Separated reference-based evaluation for textual and visual components +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the novel mechanism for robustly evaluating multimodal evidence by decoupling the modalities during the LLM judgment phase. +- **rationale:** This separated evaluation strategy is the target paper's own novel extension to prior text-only reference evaluation methods, so it maps to `NONE`. +- **evidence_span:** “We then conduct separate reference-based evaluations for the textual and visual components.” + +## Claim C3: Temporally Constrained Retrieval Methodology + +### C3.I1 +- **ingredient:** Identification of temporal leakage as a critical flaw in fact-checking datasets +- **canonical study:** Missing Counter-Evidence Renders NLP Fact-Checking Unrealistic for Misinformation +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the theoretical motivation for restricting evidence retrieval to prevent models from accessing future knowledge. +- **rationale:** The paper explicitly cites Glockner et al. as the motivation for preventing temporal leakage during evidence collection. +- **evidence_span:** “restricted to dates prior to the claim date to prevent temporal leakage [Glockner et al., 2022].” + +### C3.I2 +- **ingredient:** Date-restricted search interface and post-hoc filtering mechanism +- **canonical study:** NONE +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the practical tooling to enforce the temporal constraints during dataset construction. +- **rationale:** The specific implementation of the custom search interface and post-hoc date filtering is developed in the target paper, mapping to `NONE`. +- **evidence_span:** “annotators are provided with a custom Google search interface... restricted to dates prior to the claim date” + +## Claim C4: Baseline Verification Framework + +### C4.I1 +- **ingredient:** Tool-augmented answer generation module with specialized search and VQA tools +- **canonical study:** Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models +- **role:** CORE_METHOD +- **contribution:** Provides the paradigm of augmenting language models with specialized visual and search tools to solve complex multimodal tasks. +- **rationale:** The paper explicitly grounds its tool-augmented answer generation module in recent research on tool usage, with Visual ChatGPT (Wu et al.) being a canonical representative for integrating visual tools with LLMs. +- **evidence_span:** “we integrate a set of specialized tools into the answer generation module, along with a tool selector” + +### C4.I2 +- **ingredient:** Dynamic question generation strategy based on evolving evidence history +- **canonical study:** NONE +- **role:** CORE_METHOD +- **contribution:** Provides the mechanism for generating sequential verification questions conditioned on the accumulated evidence context. +- **rationale:** The dynamic and hybrid question generation strategies are proposed by the target paper as novel components of their baseline framework, mapping to `NONE`. +- **evidence_span:** “we also propose a dynamic question generation (DQG) method, where each subsequent question is generated based on both the claim and the evolving evidence history.” + +TARGET PAPER METADATA: +{ + "paperId": "3446200921606209cd3db3954ce913bf90070030", + "externalIds": { + "ArXiv": "2505.17978", + "DBLP": "journals/corr/abs-2505-17978", + "DOI": "10.48550/arXiv.2505.17978", + "CorpusId": 278886414 + }, + "url": "https://www.semanticscholar.org/paper/3446200921606209cd3db3954ce913bf90070030", + "title": "AVerImaTeC: A Dataset for Automatic Verification of Image-Text Claims with Evidence from the Web", + "venue": "arXiv.org", + "year": 2025, + "referenceCount": 67, + "citationCount": 16, + "openAccessPdf": { + "url": "", + "status": null, + "license": null, + "disclaimer": "Notice: Paper or abstract available at https://arxiv.org/abs/2505.17978, which is subject to the license by the author or copyright owner provided with this content. Please go to the source to verify the license and copyright information for your use." + }, + "tldr": { + "model": "tldr@v2.0.0", + "text": "This work introduces AVerImaTeC, a dataset consisting of 1,297 real-world image-text claims, and mitigate common challenges in fact-checking datasets such as contextual dependence, temporal leakage, and evidence insufficiency, via claim normalization, temporally constrained evidence annotation, and a two-stage sufficiency check." + }, + "publicationDate": "2025-05-23", + "authors": [ + { + "authorId": "2331613625", + "name": "Rui Cao" + }, + { + "authorId": "2348499942", + "name": "Zifeng Ding" + }, + { + "authorId": "2681038", + "name": "Zhijiang Guo" + }, + { + "authorId": "8804828", + "name": "M. Schlichtkrull" + }, + { + "authorId": "2273888020", + "name": "Andreas Vlachos" + } + ], + "abstract": "Textual claims are often accompanied by images to enhance their credibility and spread on social media, but this also raises concerns about the spread of misinformation. Existing datasets for automated verification of image-text claims remain limited, as they often consist of synthetic claims and lack evidence annotations to capture the reasoning behind the verdict. In this work, we introduce AVerImaTeC, a dataset consisting of 1,297 real-world image-text claims. Each claim is annotated with question-answer (QA) pairs containing evidence from the web, reflecting a decomposed reasoning regarding the verdict. We mitigate common challenges in fact-checking datasets such as contextual dependence, temporal leakage, and evidence insufficiency, via claim normalization, temporally constrained evidence annotation, and a two-stage sufficiency check. We assess the consistency of the annotation in AVerImaTeC via inter-annotator studies, achieving a $\\kappa=0.742$ on verdicts and $74.7\\%$ consistency on QA pairs. We also propose a novel evaluation method for evidence retrieval and conduct extensive experiments to establish baselines for verifying image-text claims using open-web evidence." +} diff --git a/replay_traces/2505.17978/two_pass_outputs/20260726T204323Z/pass_2_ui_payload.json b/replay_traces/2505.17978/two_pass_outputs/20260726T204323Z/pass_2_ui_payload.json new file mode 100644 index 0000000000000000000000000000000000000000..eeee5999c2c41596a3852f6aefe6964e4e29645a --- /dev/null +++ b/replay_traces/2505.17978/two_pass_outputs/20260726T204323Z/pass_2_ui_payload.json @@ -0,0 +1,405 @@ +{ + "target_paper_id": "3446200921606209cd3db3954ce913bf90070030", + "target_title": "AVerImaTeC: A Dataset for Automatic Verification of Image-Text Claims with Evidence from the Web", + "target_year": 2025, + "annotator_id": "streamlit_hf_space", + "active_claim_id": "C1", + "claims": [ + { + "claim_id": "C1", + "text": "Dataset: A real-world image-text claim verification dataset annotated with question-answer reasoning paths and web evidence, enabling the training and evaluation of multimodal fact-checking systems.", + "rewritten_claim": "Dataset: A real-world image-text claim verification dataset annotated with question-answer reasoning paths and web evidence, enabling the training and evaluation of multimodal fact-checking systems.", + "cluster_id": "C1, C2", + "decision": "YES_SUFFICIENT", + "notes": "", + "ingredients": [ + { + "ingredient_id": "C1.I1", + "ingredient": "Source fact-checking articles containing multimodal claims", + "canonical_ref_id": "__NONE__", + "canonical_grounding": null, + "additional_ref_ids": [ + "ref_averitec", + "ref_ammeba" + ], + "additional_groundings": [ + { + "ref_id": "ref_averitec", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web", + "ref_year": "", + "ref_authors": "" + }, + { + "ref_id": "ref_ammeba", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "AMMeBa: A Large-Scale Survey and Dataset of Media-Based Misinformation In-The-Wild", + "ref_year": "", + "ref_authors": "" + } + ], + "canonical_annotation": { + "role": "DATA_SOURCE", + "roles": [ + "DATA_SOURCE" + ], + "contribution": "Provides the raw real-world image-text claims that are normalized and annotated to form the dataset.", + "rationale": "The dataset is a composite of claims extracted from multiple prior fact-checking resources. No single prior study fully represents this substrate, making NONE the correct canonical grounding, with the primary sources listed as additional groundings.", + "evidence_span": "we incorporated partially annotated articles from three sources: 1) filtered articles from AVeriTec... 2) articles verifying image-related claims from AMMEBA" + } + }, + { + "ingredient_id": "C1.I2", + "ingredient": "Question-answering representation for verification rationale", + "canonical_ref_id": "ref_complex", + "canonical_grounding": { + "ref_id": "ref_complex", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "Complex Claim Verification with Evidence Retrieved in the Wild", + "ref_year": "", + "ref_authors": "" + }, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "CONCEPTUAL_FRAMEWORK", + "roles": [ + "CONCEPTUAL_FRAMEWORK" + ], + "contribution": "Provides the QA-based reasoning representation used to structure the evidence retrieval and verification process.", + "rationale": "The paper explicitly adopts QA as the reasoning representation for fact-checking, citing prior work that established this paradigm. Chen et al. (CLAIMDECOMP) is a clean representative for QA-based claim decomposition.", + "evidence_span": "To capture the rationale in claim verification, a complex reasoning task, various reasoning representations have been explored... and QA" + } + }, + { + "ingredient_id": "C1.I3", + "ingredient": "Four-way veracity labeling schema", + "canonical_ref_id": "ref_averitec", + "canonical_grounding": { + "ref_id": "ref_averitec", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web", + "ref_year": "", + "ref_authors": "" + }, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "CONCEPTUAL_FRAMEWORK", + "roles": [ + "CONCEPTUAL_FRAMEWORK" + ], + "contribution": "Provides the specific four-way label space (supported, refuted, not enough evidence, conflicting) used to annotate the claims.", + "rationale": "The paper explicitly adopts the exact four-way veracity labeling schema introduced in AVeriTeC.", + "evidence_span": "We follow the four-way veracity labeling schema from [Schlichtkrull et al., 2023]: supported, refuted, not enough evidence, and conflicting/cherry-picking." + } + }, + { + "ingredient_id": "C1.I4", + "ingredient": "Multi-stage annotation pipeline with evidence sufficiency checks", + "canonical_ref_id": "ref_averitec", + "canonical_grounding": { + "ref_id": "ref_averitec", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web", + "ref_year": "", + "ref_authors": "" + }, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "EVALUATION_PROTOCOL", + "roles": [ + "EVALUATION_PROTOCOL" + ], + "contribution": "Provides the foundational multi-stage annotation and verification workflow that the paper extends for image-text claims.", + "rationale": "The paper explicitly states that its annotation pipeline is an extension of the process proposed in AVeriTeC, making it the canonical grounding for this workflow.", + "evidence_span": "extending the annotation process proposed in [Schlichtkrull et al., 2023] to the domain of image-text claims." + } + } + ], + "enabling_discoveries": [ + { + "ref_id": "ref_complex", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "Complex Claim Verification with Evidence Retrieved in the Wild", + "ref_year": "", + "ref_authors": "", + "ingredient_id": "C1.I2", + "ingredient": "Question-answering representation for verification rationale", + "role": "CONCEPTUAL_FRAMEWORK", + "roles": [ + "CONCEPTUAL_FRAMEWORK" + ], + "contribution": "Provides the QA-based reasoning representation used to structure the evidence retrieval and verification process.", + "rationale": "The paper explicitly adopts QA as the reasoning representation for fact-checking, citing prior work that established this paradigm. Chen et al. (CLAIMDECOMP) is a clean representative for QA-based claim decomposition.", + "evidence_span": "To capture the rationale in claim verification, a complex reasoning task, various reasoning representations have been explored... and QA" + }, + { + "ref_id": "ref_averitec", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web", + "ref_year": "", + "ref_authors": "", + "ingredient_id": "C1.I3", + "ingredient": "Four-way veracity labeling schema", + "role": "CONCEPTUAL_FRAMEWORK", + "roles": [ + "CONCEPTUAL_FRAMEWORK" + ], + "contribution": "Provides the specific four-way label space (supported, refuted, not enough evidence, conflicting) used to annotate the claims.", + "rationale": "The paper explicitly adopts the exact four-way veracity labeling schema introduced in AVeriTeC.", + "evidence_span": "We follow the four-way veracity labeling schema from [Schlichtkrull et al., 2023]: supported, refuted, not enough evidence, and conflicting/cherry-picking." + }, + { + "ref_id": "ref_averitec", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web", + "ref_year": "", + "ref_authors": "", + "ingredient_id": "C1.I4", + "ingredient": "Multi-stage annotation pipeline with evidence sufficiency checks", + "role": "EVALUATION_PROTOCOL", + "roles": [ + "EVALUATION_PROTOCOL" + ], + "contribution": "Provides the foundational multi-stage annotation and verification workflow that the paper extends for image-text claims.", + "rationale": "The paper explicitly states that its annotation pipeline is an extension of the process proposed in AVeriTeC, making it the canonical grounding for this workflow.", + "evidence_span": "extending the annotation process proposed in [Schlichtkrull et al., 2023] to the domain of image-text claims." + } + ] + }, + { + "claim_id": "C2", + "text": "Method: A reference-based evaluation method using large language models to separately assess the textual and visual components of retrieved evidence, enabling automated alignment with human judgments in multimodal fact-checking.", + "rewritten_claim": "Method: A reference-based evaluation method using large language models to separately assess the textual and visual components of retrieved evidence, enabling automated alignment with human judgments in multimodal fact-checking.", + "cluster_id": "C3", + "decision": "YES_SUFFICIENT", + "notes": "", + "ingredients": [ + { + "ingredient_id": "C2.I1", + "ingredient": "LLM-based reference evaluation framework for evidence retrieval", + "canonical_ref_id": "ref_ev2r", + "canonical_grounding": { + "ref_id": "ref_ev2r", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking", + "ref_year": "", + "ref_authors": "" + }, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "EVALUATION_PROTOCOL", + "roles": [ + "EVALUATION_PROTOCOL" + ], + "contribution": "Provides the foundation for using LLMs to evaluate evidence retrieval by comparing it to references.", + "rationale": "The paper explicitly adopts the Ev2R framework for its reference-based evaluation of textual evidence.", + "evidence_span": "Recent research [Akhtar et al., 2024] showed that reference-based evaluation of evidence with large language models (LLMs) aligns best with human assessments." + } + }, + { + "ingredient_id": "C2.I2", + "ingredient": "Separated reference-based evaluation for textual and visual components", + "canonical_ref_id": "__NONE__", + "canonical_grounding": null, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "EVALUATION_PROTOCOL", + "roles": [ + "EVALUATION_PROTOCOL" + ], + "contribution": "Provides the novel mechanism for robustly evaluating multimodal evidence by decoupling the modalities during the LLM judgment phase.", + "rationale": "This separated evaluation strategy is the target paper's own novel extension to prior text-only reference evaluation methods, so it maps to NONE.", + "evidence_span": "We then conduct separate reference-based evaluations for the textual and visual components." + } + } + ], + "enabling_discoveries": [ + { + "ref_id": "ref_ev2r", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking", + "ref_year": "", + "ref_authors": "", + "ingredient_id": "C2.I1", + "ingredient": "LLM-based reference evaluation framework for evidence retrieval", + "role": "EVALUATION_PROTOCOL", + "roles": [ + "EVALUATION_PROTOCOL" + ], + "contribution": "Provides the foundation for using LLMs to evaluate evidence retrieval by comparing it to references.", + "rationale": "The paper explicitly adopts the Ev2R framework for its reference-based evaluation of textual evidence.", + "evidence_span": "Recent research [Akhtar et al., 2024] showed that reference-based evaluation of evidence with large language models (LLMs) aligns best with human assessments." + } + ] + }, + { + "claim_id": "C3", + "text": "Protocol: A temporally constrained evidence retrieval methodology that restricts search results to dates prior to the claim, enabling the construction of fact-checking datasets without temporal leakage.", + "rewritten_claim": "Protocol: A temporally constrained evidence retrieval methodology that restricts search results to dates prior to the claim, enabling the construction of fact-checking datasets without temporal leakage.", + "cluster_id": "C4", + "decision": "YES_SUFFICIENT", + "notes": "", + "ingredients": [ + { + "ingredient_id": "C3.I1", + "ingredient": "Identification of temporal leakage as a critical flaw in fact-checking datasets", + "canonical_ref_id": "ref_missing", + "canonical_grounding": { + "ref_id": "ref_missing", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "Missing Counter-Evidence Renders NLP Fact-Checking Unrealistic for Misinformation", + "ref_year": "", + "ref_authors": "" + }, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "CONCEPTUAL_FRAMEWORK", + "roles": [ + "CONCEPTUAL_FRAMEWORK" + ], + "contribution": "Provides the theoretical motivation for restricting evidence retrieval to prevent models from accessing future knowledge.", + "rationale": "The paper explicitly cites Glockner et al. as the motivation for preventing temporal leakage during evidence collection.", + "evidence_span": "restricted to dates prior to the claim date to prevent temporal leakage [Glockner et al., 2022]." + } + }, + { + "ingredient_id": "C3.I2", + "ingredient": "Date-restricted search interface and post-hoc filtering mechanism", + "canonical_ref_id": "__NONE__", + "canonical_grounding": null, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "IMPLEMENTATION_TOOLING", + "roles": [ + "IMPLEMENTATION_TOOLING" + ], + "contribution": "Provides the practical tooling to enforce the temporal constraints during dataset construction.", + "rationale": "The specific implementation of the custom search interface and post-hoc date filtering is developed in the target paper, mapping to NONE.", + "evidence_span": "annotators are provided with a custom Google search interface... restricted to dates prior to the claim date" + } + } + ], + "enabling_discoveries": [ + { + "ref_id": "ref_missing", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "Missing Counter-Evidence Renders NLP Fact-Checking Unrealistic for Misinformation", + "ref_year": "", + "ref_authors": "", + "ingredient_id": "C3.I1", + "ingredient": "Identification of temporal leakage as a critical flaw in fact-checking datasets", + "role": "CONCEPTUAL_FRAMEWORK", + "roles": [ + "CONCEPTUAL_FRAMEWORK" + ], + "contribution": "Provides the theoretical motivation for restricting evidence retrieval to prevent models from accessing future knowledge.", + "rationale": "The paper explicitly cites Glockner et al. as the motivation for preventing temporal leakage during evidence collection.", + "evidence_span": "restricted to dates prior to the claim date to prevent temporal leakage [Glockner et al., 2022]." + } + ] + }, + { + "claim_id": "C4", + "text": "Method: A multimodal claim verification framework combining dynamic question generation and tool-augmented answer generation, enabling automated fact-checking of image-text claims using open-web evidence.", + "rewritten_claim": "Method: A multimodal claim verification framework combining dynamic question generation and tool-augmented answer generation, enabling automated fact-checking of image-text claims using open-web evidence.", + "cluster_id": "C5", + "decision": "YES_SUFFICIENT", + "notes": "", + "ingredients": [ + { + "ingredient_id": "C4.I1", + "ingredient": "Tool-augmented answer generation module with specialized search and VQA tools", + "canonical_ref_id": "ref_visual", + "canonical_grounding": { + "ref_id": "ref_visual", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models", + "ref_year": "", + "ref_authors": "" + }, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "CORE_METHOD", + "roles": [ + "CORE_METHOD" + ], + "contribution": "Provides the paradigm of augmenting language models with specialized visual and search tools to solve complex multimodal tasks.", + "rationale": "The paper explicitly grounds its tool-augmented answer generation module in recent research on tool usage, with Visual ChatGPT (Wu et al.) being a canonical representative for integrating visual tools with LLMs.", + "evidence_span": "we integrate a set of specialized tools into the answer generation module, along with a tool selector" + } + }, + { + "ingredient_id": "C4.I2", + "ingredient": "Dynamic question generation strategy based on evolving evidence history", + "canonical_ref_id": "__NONE__", + "canonical_grounding": null, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "CORE_METHOD", + "roles": [ + "CORE_METHOD" + ], + "contribution": "Provides the mechanism for generating sequential verification questions conditioned on the accumulated evidence context.", + "rationale": "The dynamic and hybrid question generation strategies are proposed by the target paper as novel components of their baseline framework, mapping to NONE.", + "evidence_span": "we also propose a dynamic question generation (DQG) method, where each subsequent question is generated based on both the claim and the evolving evidence history." + } + } + ], + "enabling_discoveries": [ + { + "ref_id": "ref_visual", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models", + "ref_year": "", + "ref_authors": "", + "ingredient_id": "C4.I1", + "ingredient": "Tool-augmented answer generation module with specialized search and VQA tools", + "role": "CORE_METHOD", + "roles": [ + "CORE_METHOD" + ], + "contribution": "Provides the paradigm of augmenting language models with specialized visual and search tools to solve complex multimodal tasks.", + "rationale": "The paper explicitly grounds its tool-augmented answer generation module in recent research on tool usage, with Visual ChatGPT (Wu et al.) being a canonical representative for integrating visual tools with LLMs.", + "evidence_span": "we integrate a set of specialized tools into the answer generation module, along with a tool selector" + } + ] + } + ] +} diff --git a/replay_traces/2505.17978/two_pass_outputs/20260726T204323Z/run_output.json b/replay_traces/2505.17978/two_pass_outputs/20260726T204323Z/run_output.json new file mode 100644 index 0000000000000000000000000000000000000000..0395cc7daf8425daabee0575426d4cf7cba70380 --- /dev/null +++ b/replay_traces/2505.17978/two_pass_outputs/20260726T204323Z/run_output.json @@ -0,0 +1,432 @@ +{ + "paper_id": "2505.17978", + "paper_dir": "/Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/job_1785098515_4b6a9bbd/processed_papers/2505.17978", + "generated_at": "2026-07-26T20:47:28.337476+00:00", + "reasoner_model": "gemini/gemini-3.1-pro-preview", + "formatter_model": "gemini/gemini-3.1-pro-preview", + "judge_model": "gemini/gemini-3.1-pro-preview", + "candidate_count": 3, + "include_reference_examples": true, + "prompt_profile": "full", + "reasoning_candidate_paths": [ + "/Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/job_1785098515_4b6a9bbd/two_pass_outputs/2505.17978/20260726T204323Z/pass_1_reasoning.output.candidate_1.md", + "/Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/job_1785098515_4b6a9bbd/two_pass_outputs/2505.17978/20260726T204323Z/pass_1_reasoning.output.candidate_2.md", + "/Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/job_1785098515_4b6a9bbd/two_pass_outputs/2505.17978/20260726T204323Z/pass_1_reasoning.output.candidate_3.md" + ], + "selected_reasoning_candidate": "candidate_1", + "selected_candidate_index": 0, + "selected_reasoning_path": "/Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/job_1785098515_4b6a9bbd/two_pass_outputs/2505.17978/20260726T204323Z/pass_1_reasoning.selected.md", + "judge_output_path": "/Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/job_1785098515_4b6a9bbd/two_pass_outputs/2505.17978/20260726T204323Z/pass_1_reasoning.judge.output.json", + "formatter_attempts": [ + { + "attempt": 1, + "status": "success" + } + ], + "ui_payload_path": "/Users/dingzifeng/SciFy-UMBC/demo_2026_07/hf_space/runs/job_1785098515_4b6a9bbd/two_pass_outputs/2505.17978/20260726T204323Z/pass_2_ui_payload.json", + "ui_payload": { + "target_paper_id": "3446200921606209cd3db3954ce913bf90070030", + "target_title": "AVerImaTeC: A Dataset for Automatic Verification of Image-Text Claims with Evidence from the Web", + "target_year": 2025, + "annotator_id": "streamlit_hf_space", + "active_claim_id": "C1", + "claims": [ + { + "claim_id": "C1", + "text": "Dataset: A real-world image-text claim verification dataset annotated with question-answer reasoning paths and web evidence, enabling the training and evaluation of multimodal fact-checking systems.", + "rewritten_claim": "Dataset: A real-world image-text claim verification dataset annotated with question-answer reasoning paths and web evidence, enabling the training and evaluation of multimodal fact-checking systems.", + "cluster_id": "C1, C2", + "decision": "YES_SUFFICIENT", + "notes": "", + "ingredients": [ + { + "ingredient_id": "C1.I1", + "ingredient": "Source fact-checking articles containing multimodal claims", + "canonical_ref_id": "__NONE__", + "canonical_grounding": null, + "additional_ref_ids": [ + "ref_averitec", + "ref_ammeba" + ], + "additional_groundings": [ + { + "ref_id": "ref_averitec", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web", + "ref_year": "", + "ref_authors": "" + }, + { + "ref_id": "ref_ammeba", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "AMMeBa: A Large-Scale Survey and Dataset of Media-Based Misinformation In-The-Wild", + "ref_year": "", + "ref_authors": "" + } + ], + "canonical_annotation": { + "role": "DATA_SOURCE", + "roles": [ + "DATA_SOURCE" + ], + "contribution": "Provides the raw real-world image-text claims that are normalized and annotated to form the dataset.", + "rationale": "The dataset is a composite of claims extracted from multiple prior fact-checking resources. No single prior study fully represents this substrate, making NONE the correct canonical grounding, with the primary sources listed as additional groundings.", + "evidence_span": "we incorporated partially annotated articles from three sources: 1) filtered articles from AVeriTec... 2) articles verifying image-related claims from AMMEBA" + } + }, + { + "ingredient_id": "C1.I2", + "ingredient": "Question-answering representation for verification rationale", + "canonical_ref_id": "ref_complex", + "canonical_grounding": { + "ref_id": "ref_complex", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "Complex Claim Verification with Evidence Retrieved in the Wild", + "ref_year": "", + "ref_authors": "" + }, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "CONCEPTUAL_FRAMEWORK", + "roles": [ + "CONCEPTUAL_FRAMEWORK" + ], + "contribution": "Provides the QA-based reasoning representation used to structure the evidence retrieval and verification process.", + "rationale": "The paper explicitly adopts QA as the reasoning representation for fact-checking, citing prior work that established this paradigm. Chen et al. (CLAIMDECOMP) is a clean representative for QA-based claim decomposition.", + "evidence_span": "To capture the rationale in claim verification, a complex reasoning task, various reasoning representations have been explored... and QA" + } + }, + { + "ingredient_id": "C1.I3", + "ingredient": "Four-way veracity labeling schema", + "canonical_ref_id": "ref_averitec", + "canonical_grounding": { + "ref_id": "ref_averitec", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web", + "ref_year": "", + "ref_authors": "" + }, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "CONCEPTUAL_FRAMEWORK", + "roles": [ + "CONCEPTUAL_FRAMEWORK" + ], + "contribution": "Provides the specific four-way label space (supported, refuted, not enough evidence, conflicting) used to annotate the claims.", + "rationale": "The paper explicitly adopts the exact four-way veracity labeling schema introduced in AVeriTeC.", + "evidence_span": "We follow the four-way veracity labeling schema from [Schlichtkrull et al., 2023]: supported, refuted, not enough evidence, and conflicting/cherry-picking." + } + }, + { + "ingredient_id": "C1.I4", + "ingredient": "Multi-stage annotation pipeline with evidence sufficiency checks", + "canonical_ref_id": "ref_averitec", + "canonical_grounding": { + "ref_id": "ref_averitec", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web", + "ref_year": "", + "ref_authors": "" + }, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "EVALUATION_PROTOCOL", + "roles": [ + "EVALUATION_PROTOCOL" + ], + "contribution": "Provides the foundational multi-stage annotation and verification workflow that the paper extends for image-text claims.", + "rationale": "The paper explicitly states that its annotation pipeline is an extension of the process proposed in AVeriTeC, making it the canonical grounding for this workflow.", + "evidence_span": "extending the annotation process proposed in [Schlichtkrull et al., 2023] to the domain of image-text claims." + } + } + ], + "enabling_discoveries": [ + { + "ref_id": "ref_complex", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "Complex Claim Verification with Evidence Retrieved in the Wild", + "ref_year": "", + "ref_authors": "", + "ingredient_id": "C1.I2", + "ingredient": "Question-answering representation for verification rationale", + "role": "CONCEPTUAL_FRAMEWORK", + "roles": [ + "CONCEPTUAL_FRAMEWORK" + ], + "contribution": "Provides the QA-based reasoning representation used to structure the evidence retrieval and verification process.", + "rationale": "The paper explicitly adopts QA as the reasoning representation for fact-checking, citing prior work that established this paradigm. Chen et al. (CLAIMDECOMP) is a clean representative for QA-based claim decomposition.", + "evidence_span": "To capture the rationale in claim verification, a complex reasoning task, various reasoning representations have been explored... and QA" + }, + { + "ref_id": "ref_averitec", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web", + "ref_year": "", + "ref_authors": "", + "ingredient_id": "C1.I3", + "ingredient": "Four-way veracity labeling schema", + "role": "CONCEPTUAL_FRAMEWORK", + "roles": [ + "CONCEPTUAL_FRAMEWORK" + ], + "contribution": "Provides the specific four-way label space (supported, refuted, not enough evidence, conflicting) used to annotate the claims.", + "rationale": "The paper explicitly adopts the exact four-way veracity labeling schema introduced in AVeriTeC.", + "evidence_span": "We follow the four-way veracity labeling schema from [Schlichtkrull et al., 2023]: supported, refuted, not enough evidence, and conflicting/cherry-picking." + }, + { + "ref_id": "ref_averitec", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web", + "ref_year": "", + "ref_authors": "", + "ingredient_id": "C1.I4", + "ingredient": "Multi-stage annotation pipeline with evidence sufficiency checks", + "role": "EVALUATION_PROTOCOL", + "roles": [ + "EVALUATION_PROTOCOL" + ], + "contribution": "Provides the foundational multi-stage annotation and verification workflow that the paper extends for image-text claims.", + "rationale": "The paper explicitly states that its annotation pipeline is an extension of the process proposed in AVeriTeC, making it the canonical grounding for this workflow.", + "evidence_span": "extending the annotation process proposed in [Schlichtkrull et al., 2023] to the domain of image-text claims." + } + ] + }, + { + "claim_id": "C2", + "text": "Method: A reference-based evaluation method using large language models to separately assess the textual and visual components of retrieved evidence, enabling automated alignment with human judgments in multimodal fact-checking.", + "rewritten_claim": "Method: A reference-based evaluation method using large language models to separately assess the textual and visual components of retrieved evidence, enabling automated alignment with human judgments in multimodal fact-checking.", + "cluster_id": "C3", + "decision": "YES_SUFFICIENT", + "notes": "", + "ingredients": [ + { + "ingredient_id": "C2.I1", + "ingredient": "LLM-based reference evaluation framework for evidence retrieval", + "canonical_ref_id": "ref_ev2r", + "canonical_grounding": { + "ref_id": "ref_ev2r", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking", + "ref_year": "", + "ref_authors": "" + }, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "EVALUATION_PROTOCOL", + "roles": [ + "EVALUATION_PROTOCOL" + ], + "contribution": "Provides the foundation for using LLMs to evaluate evidence retrieval by comparing it to references.", + "rationale": "The paper explicitly adopts the Ev2R framework for its reference-based evaluation of textual evidence.", + "evidence_span": "Recent research [Akhtar et al., 2024] showed that reference-based evaluation of evidence with large language models (LLMs) aligns best with human assessments." + } + }, + { + "ingredient_id": "C2.I2", + "ingredient": "Separated reference-based evaluation for textual and visual components", + "canonical_ref_id": "__NONE__", + "canonical_grounding": null, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "EVALUATION_PROTOCOL", + "roles": [ + "EVALUATION_PROTOCOL" + ], + "contribution": "Provides the novel mechanism for robustly evaluating multimodal evidence by decoupling the modalities during the LLM judgment phase.", + "rationale": "This separated evaluation strategy is the target paper's own novel extension to prior text-only reference evaluation methods, so it maps to NONE.", + "evidence_span": "We then conduct separate reference-based evaluations for the textual and visual components." + } + } + ], + "enabling_discoveries": [ + { + "ref_id": "ref_ev2r", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "Ev2R: Evaluating Evidence Retrieval in Automated Fact-Checking", + "ref_year": "", + "ref_authors": "", + "ingredient_id": "C2.I1", + "ingredient": "LLM-based reference evaluation framework for evidence retrieval", + "role": "EVALUATION_PROTOCOL", + "roles": [ + "EVALUATION_PROTOCOL" + ], + "contribution": "Provides the foundation for using LLMs to evaluate evidence retrieval by comparing it to references.", + "rationale": "The paper explicitly adopts the Ev2R framework for its reference-based evaluation of textual evidence.", + "evidence_span": "Recent research [Akhtar et al., 2024] showed that reference-based evaluation of evidence with large language models (LLMs) aligns best with human assessments." + } + ] + }, + { + "claim_id": "C3", + "text": "Protocol: A temporally constrained evidence retrieval methodology that restricts search results to dates prior to the claim, enabling the construction of fact-checking datasets without temporal leakage.", + "rewritten_claim": "Protocol: A temporally constrained evidence retrieval methodology that restricts search results to dates prior to the claim, enabling the construction of fact-checking datasets without temporal leakage.", + "cluster_id": "C4", + "decision": "YES_SUFFICIENT", + "notes": "", + "ingredients": [ + { + "ingredient_id": "C3.I1", + "ingredient": "Identification of temporal leakage as a critical flaw in fact-checking datasets", + "canonical_ref_id": "ref_missing", + "canonical_grounding": { + "ref_id": "ref_missing", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "Missing Counter-Evidence Renders NLP Fact-Checking Unrealistic for Misinformation", + "ref_year": "", + "ref_authors": "" + }, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "CONCEPTUAL_FRAMEWORK", + "roles": [ + "CONCEPTUAL_FRAMEWORK" + ], + "contribution": "Provides the theoretical motivation for restricting evidence retrieval to prevent models from accessing future knowledge.", + "rationale": "The paper explicitly cites Glockner et al. as the motivation for preventing temporal leakage during evidence collection.", + "evidence_span": "restricted to dates prior to the claim date to prevent temporal leakage [Glockner et al., 2022]." + } + }, + { + "ingredient_id": "C3.I2", + "ingredient": "Date-restricted search interface and post-hoc filtering mechanism", + "canonical_ref_id": "__NONE__", + "canonical_grounding": null, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "IMPLEMENTATION_TOOLING", + "roles": [ + "IMPLEMENTATION_TOOLING" + ], + "contribution": "Provides the practical tooling to enforce the temporal constraints during dataset construction.", + "rationale": "The specific implementation of the custom search interface and post-hoc date filtering is developed in the target paper, mapping to NONE.", + "evidence_span": "annotators are provided with a custom Google search interface... restricted to dates prior to the claim date" + } + } + ], + "enabling_discoveries": [ + { + "ref_id": "ref_missing", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "Missing Counter-Evidence Renders NLP Fact-Checking Unrealistic for Misinformation", + "ref_year": "", + "ref_authors": "", + "ingredient_id": "C3.I1", + "ingredient": "Identification of temporal leakage as a critical flaw in fact-checking datasets", + "role": "CONCEPTUAL_FRAMEWORK", + "roles": [ + "CONCEPTUAL_FRAMEWORK" + ], + "contribution": "Provides the theoretical motivation for restricting evidence retrieval to prevent models from accessing future knowledge.", + "rationale": "The paper explicitly cites Glockner et al. as the motivation for preventing temporal leakage during evidence collection.", + "evidence_span": "restricted to dates prior to the claim date to prevent temporal leakage [Glockner et al., 2022]." + } + ] + }, + { + "claim_id": "C4", + "text": "Method: A multimodal claim verification framework combining dynamic question generation and tool-augmented answer generation, enabling automated fact-checking of image-text claims using open-web evidence.", + "rewritten_claim": "Method: A multimodal claim verification framework combining dynamic question generation and tool-augmented answer generation, enabling automated fact-checking of image-text claims using open-web evidence.", + "cluster_id": "C5", + "decision": "YES_SUFFICIENT", + "notes": "", + "ingredients": [ + { + "ingredient_id": "C4.I1", + "ingredient": "Tool-augmented answer generation module with specialized search and VQA tools", + "canonical_ref_id": "ref_visual", + "canonical_grounding": { + "ref_id": "ref_visual", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models", + "ref_year": "", + "ref_authors": "" + }, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "CORE_METHOD", + "roles": [ + "CORE_METHOD" + ], + "contribution": "Provides the paradigm of augmenting language models with specialized visual and search tools to solve complex multimodal tasks.", + "rationale": "The paper explicitly grounds its tool-augmented answer generation module in recent research on tool usage, with Visual ChatGPT (Wu et al.) being a canonical representative for integrating visual tools with LLMs.", + "evidence_span": "we integrate a set of specialized tools into the answer generation module, along with a tool selector" + } + }, + { + "ingredient_id": "C4.I2", + "ingredient": "Dynamic question generation strategy based on evolving evidence history", + "canonical_ref_id": "__NONE__", + "canonical_grounding": null, + "additional_ref_ids": [], + "additional_groundings": [], + "canonical_annotation": { + "role": "CORE_METHOD", + "roles": [ + "CORE_METHOD" + ], + "contribution": "Provides the mechanism for generating sequential verification questions conditioned on the accumulated evidence context.", + "rationale": "The dynamic and hybrid question generation strategies are proposed by the target paper as novel components of their baseline framework, mapping to NONE.", + "evidence_span": "we also propose a dynamic question generation (DQG) method, where each subsequent question is generated based on both the claim and the evolving evidence history." + } + } + ], + "enabling_discoveries": [ + { + "ref_id": "ref_visual", + "bib_key": "", + "paper_id": "", + "external_ids": {}, + "ref_title": "Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models", + "ref_year": "", + "ref_authors": "", + "ingredient_id": "C4.I1", + "ingredient": "Tool-augmented answer generation module with specialized search and VQA tools", + "role": "CORE_METHOD", + "roles": [ + "CORE_METHOD" + ], + "contribution": "Provides the paradigm of augmenting language models with specialized visual and search tools to solve complex multimodal tasks.", + "rationale": "The paper explicitly grounds its tool-augmented answer generation module in recent research on tool usage, with Visual ChatGPT (Wu et al.) being a canonical representative for integrating visual tools with LLMs.", + "evidence_span": "we integrate a set of specialized tools into the answer generation module, along with a tool selector" + } + ] + } + ] + } +} diff --git a/replay_traces/index.json b/replay_traces/index.json new file mode 100644 index 0000000000000000000000000000000000000000..e814bbdf4d9b5e571e07ed1b304b7772516327aa --- /dev/null +++ b/replay_traces/index.json @@ -0,0 +1,38 @@ +{ + "built_at": "2026-07-26T21:22:24.857616+00:00", + "examples": [ + { + "label": "PASTA: Participant States in Narratives", + "arxiv_id": "2208.00329", + "status": "Completed", + "trace_dir": "/Users/dingzifeng/SciFy-UMBC/demo_2026_07/replay_traces/2208.00329", + "cluster_count": 2, + "claim_count": 2, + "annotation_skipped_reason": null, + "pipeline_failed_reason": null, + "pipeline_stopped_reason": null + }, + { + "label": "CSCD-NS (2022)", + "arxiv_id": "2211.08788", + "status": "Completed", + "trace_dir": "/Users/dingzifeng/SciFy-UMBC/demo_2026_07/replay_traces/2211.08788", + "cluster_count": 2, + "claim_count": 1, + "annotation_skipped_reason": null, + "pipeline_failed_reason": null, + "pipeline_stopped_reason": null + }, + { + "label": "AVerImaTeC", + "arxiv_id": "2505.17978", + "status": "Completed", + "trace_dir": "/Users/dingzifeng/SciFy-UMBC/demo_2026_07/replay_traces/2505.17978", + "cluster_count": 5, + "claim_count": 4, + "annotation_skipped_reason": null, + "pipeline_failed_reason": null, + "pipeline_stopped_reason": null + } + ] +} \ No newline at end of file diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 0000000000000000000000000000000000000000..55a10702105ba6bbbf4a9853529e072938d1629a --- /dev/null +++ b/requirements.txt @@ -0,0 +1 @@ +-r hf_space/requirements.txt diff --git a/scripts/deploy_hf_space.sh b/scripts/deploy_hf_space.sh new file mode 100644 index 0000000000000000000000000000000000000000..3b89e84914f2c8bbcca1679d68c726438e4c869e --- /dev/null +++ b/scripts/deploy_hf_space.sh @@ -0,0 +1,73 @@ +#!/usr/bin/env bash +# Deploy demo_2026_07 to Hugging Face Space beanbagdzf/SciPaths-Demo +set -euo pipefail +ROOT="$(cd "$(dirname "$0")/.." && pwd)" +SPACE_ID="${SPACE_ID:-Beanbagdzf/SciPaths-Demo}" +cd "$ROOT" + +if [[ -z "${HF_TOKEN:-}" ]]; then + echo "Set HF_TOKEN to a Hugging Face write token for Beanbagdzf, then re-run." + echo "Create one at: https://huggingface.co/settings/tokens" + exit 1 +fi + +export HUGGINGFACE_HUB_TOKEN="$HF_TOKEN" +export SPACE_ID +PY="${ROOT}/.venv/bin/python" +if [[ ! -x "$PY" ]]; then PY="$(command -v python3)"; fi + +"$PY" - <<'PY' +import os +from pathlib import Path +from huggingface_hub import HfApi, login + +token = os.environ["HF_TOKEN"] +login(token=token, add_to_git_credential=False) +api = HfApi(token=token) +who = api.whoami() +print("logged_in_as", who.get("name")) +space_id = os.environ.get("SPACE_ID", "Beanbagdzf/SciPaths-Demo") +url = api.create_repo(repo_id=space_id, repo_type="space", space_sdk="docker", exist_ok=True, private=False) +print("space", url) + +# Upload folder (respects .gitignore / skips secrets) +api.upload_folder( + folder_path=".", + repo_id=space_id, + repo_type="space", + ignore_patterns=[ + ".git*", + ".venv/**", + ".env", + ".playwright/**", + "hf_space/runs/**", + "runs/**", + ".streamlit/secrets.toml", + "**/__pycache__/**", + "*.pyc", + ".DS_Store", + ], +) +print("uploaded") + +# Space secret for Gemini (from local .env) +env = Path(".env") +gemini = "" +if env.exists(): + for line in env.read_text().splitlines(): + line=line.strip() + if not line or line.startswith("#") or "=" not in line: + continue + k,v = line.split("=",1) + if k.strip() in {"GEMINI_API_KEY","GOOGLE_GENAI_API_KEY","GOOGLE_API_KEY"} and v.strip(): + gemini = v.strip().strip('"').strip("'") + break +if gemini: + api.add_space_secret(space_id, "GEMINI_API_KEY", gemini) + api.add_space_secret(space_id, "GOOGLE_GENAI_API_KEY", gemini) + api.add_space_secret(space_id, "GOOGLE_API_KEY", gemini) + print("secrets_set GEMINI_API_KEY") +else: + print("WARN: no GEMINI key found in .env — set Space secret manually") +print(f"URL: https://huggingface.co/spaces/{space_id}") +PY diff --git a/scripts/setup_playwright.sh b/scripts/setup_playwright.sh new file mode 100644 index 0000000000000000000000000000000000000000..7e7bb26efd4ad66a773cb7ee74044a14802f5cce --- /dev/null +++ b/scripts/setup_playwright.sh @@ -0,0 +1,53 @@ +#!/usr/bin/env bash +# Install Playwright Chromium into the demo-local cache used by Live Run. +set -euo pipefail + +ROOT="$(cd "$(dirname "$0")/.." && pwd)" +CACHE="${PLAYWRIGHT_BROWSERS_PATH:-$ROOT/.playwright}" +PY="${ROOT}/.venv/bin/python" +if [[ ! -x "$PY" ]]; then + PY="$(command -v python3)" +fi + +mkdir -p "$CACHE" +export PLAYWRIGHT_BROWSERS_PATH="$CACHE" + +echo "Installing Playwright Chromium into: $CACHE" +"$PY" -m playwright install chromium + +# Ensure .env points at the local cache (override Cursor sandbox temp paths). +ENV_FILE="$ROOT/.env" +touch "$ENV_FILE" +if grep -q '^PLAYWRIGHT_BROWSERS_PATH=' "$ENV_FILE"; then + # portable in-place replace + tmp="$(mktemp)" + awk -v p="$CACHE" ' + BEGIN { done=0 } + /^PLAYWRIGHT_BROWSERS_PATH=/ { print "PLAYWRIGHT_BROWSERS_PATH=" p; done=1; next } + { print } + END { if (!done) print "PLAYWRIGHT_BROWSERS_PATH=" p } + ' "$ENV_FILE" > "$tmp" + mv "$tmp" "$ENV_FILE" +else + printf '\nPLAYWRIGHT_BROWSERS_PATH=%s\n' "$CACHE" >> "$ENV_FILE" +fi + +echo "Done. Restart Streamlit so the new PLAYWRIGHT_BROWSERS_PATH is picked up." +echo "Quick check:" +PLAYWRIGHT_BROWSERS_PATH="$CACHE" "$PY" - <<'PY' +import os +from playwright.sync_api import sync_playwright +os.environ.setdefault( + "PLAYWRIGHT_BROWSERS_PATH", + os.environ.get("PLAYWRIGHT_BROWSERS_PATH", ""), +) +with sync_playwright() as p: + try: + browser = p.chromium.launch(channel="chrome", headless=True) + except Exception: + browser = p.chromium.launch(headless=True) + page = browser.new_page() + page.goto("https://example.com", timeout=20000) + print("ok:", page.title()) + browser.close() +PY diff --git a/src/common/__init__.py b/src/common/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..e69de29bb2d1d6434b8b29ae775ad8c2e48c5391 diff --git a/src/common/llm_client.py b/src/common/llm_client.py new file mode 100644 index 0000000000000000000000000000000000000000..66c0cb8b77d7478bcecdc4fd89b88a2d1de9f282 --- /dev/null +++ b/src/common/llm_client.py @@ -0,0 +1,49 @@ +import os +from typing import Optional + +import google.generativeai as genai +from google.generativeai.types import GenerationConfig + + +class LLMClient: + def __init__(self): + self.provider = os.getenv("LLM_PROVIDER", "gemini").lower() + self.model_name = os.getenv("LLM_MODEL", "gemini-3.1-pro-preview") + + if self.provider == "gemini": + if genai is None: + raise ImportError("google-generativeai not installed.") + key = os.getenv("GEMINI_API_KEY") + if not key: + raise ValueError("GEMINI_API_KEY not set.") + genai.configure(api_key=key) + self.model = genai.GenerativeModel(self.model_name) + else: + raise NotImplementedError("Only Gemini provider is wired for now.") + + def call(self, prompt: str, schema: Optional[dict] = None) -> str: + """ + Call the underlying LLM. + + If `schema` is provided (as a plain JSON schema dict), and provider is Gemini, + use it as response_schema with JSON mime type. + """ + if self.provider == "gemini": + if schema and GenerationConfig is not None: + config = GenerationConfig( + response_schema=schema, + response_mime_type="application/json", + ) + response = self.model.generate_content( + prompt, + generation_config=config, + ) + else: + response = self.model.generate_content(prompt) + + text = getattr(response, "text", "") + if not text: + raise RuntimeError("LLM response did not contain text.") + return text + + raise NotImplementedError("Schema-based calls only wired for Gemini right now.") diff --git a/src/common/model_client.py b/src/common/model_client.py new file mode 100644 index 0000000000000000000000000000000000000000..18d9af7cc430d57588563c55c5500d53c9bafc55 --- /dev/null +++ b/src/common/model_client.py @@ -0,0 +1,143 @@ +from __future__ import annotations + +import json +import os +import re +from dataclasses import dataclass +from typing import Any, Type + +import litellm +from litellm import completion +from pydantic import BaseModel, ValidationError + + +@dataclass +class ModelConfig: + provider: str + model: str + temperature: float = 0.2 + max_tokens: int = 12000 + + @property + def model_name(self) -> str: + if "/" in self.model: + return self.model + if self.provider.lower() == "openai": + return f"openai/{self.model}" + if self.provider.lower() == "gemini": + return f"gemini/{self.model}" + return self.model + + +class MultiProviderLLMClient: + def __init__(self, default_config: ModelConfig, stage_models: dict[str, str] | None = None): + self.default_config = default_config + self.stage_models = stage_models or {} + litellm.drop_params = True + self._validate_env(default_config.provider) + + def _validate_env(self, provider: str) -> None: + provider = provider.lower() + if provider == "openai" and not os.getenv("OPENAI_API_KEY"): + raise ValueError("OPENAI_API_KEY is required for provider=openai") + if provider == "gemini" and not os.getenv("GEMINI_API_KEY"): + raise ValueError("GEMINI_API_KEY is required for provider=gemini") + + def config_for_stage(self, stage_name: str) -> ModelConfig: + model_override = self.stage_models.get(stage_name) + if not model_override: + return self.default_config + provider = self.default_config.provider + model = model_override + if "/" in model_override: + provider, model = model_override.split("/", 1) + self._validate_env(provider) + return ModelConfig( + provider=provider, + model=model, + temperature=self.default_config.temperature, + max_tokens=self.default_config.max_tokens, + ) + + def generate_structured( + self, + *, + stage_name: str, + system_prompt: str, + user_prompt: str, + response_model: Type[BaseModel], + ) -> BaseModel: + config = self.config_for_stage(stage_name) + completion_kwargs = { + "model": config.model_name, + "messages": [ + {"role": "system", "content": system_prompt}, + {"role": "user", "content": user_prompt}, + ], + "max_tokens": config.max_tokens, + "response_format": {"type": "json_object"}, + } + temperature = self._temperature_for_model(config) + if temperature is not None: + completion_kwargs["temperature"] = temperature + + response = completion( + **completion_kwargs, + ) + content = response.choices[0].message.content or "" + payload = self._parse_json(content) + try: + return response_model.model_validate(payload) + except ValidationError as exc: + if isinstance(payload, list) and len(payload) == 1 and isinstance(payload[0], dict): + try: + return response_model.model_validate(payload[0]) + except ValidationError: + pass + raise ValueError( + f"Stage {stage_name} returned invalid JSON for {response_model.__name__}: {exc}\nRaw content:\n{content}" + ) from exc + + def generate_text( + self, + *, + stage_name: str, + system_prompt: str, + user_prompt: str, + ) -> str: + config = self.config_for_stage(stage_name) + completion_kwargs = { + "model": config.model_name, + "messages": [ + {"role": "system", "content": system_prompt}, + {"role": "user", "content": user_prompt}, + ], + "max_tokens": config.max_tokens, + } + temperature = self._temperature_for_model(config) + if temperature is not None: + completion_kwargs["temperature"] = temperature + response = completion(**completion_kwargs) + return (response.choices[0].message.content or "").strip() + + @staticmethod + def _parse_json(text: str) -> Any: + text = text.strip() + if text.startswith("```"): + match = re.search(r"```(?:json)?\s*(.*?)```", text, flags=re.S) + if match: + text = match.group(1).strip() + try: + return json.loads(text) + except json.JSONDecodeError: + match = re.search(r"(\{.*\}|\[.*\])", text, flags=re.S) + if match: + return json.loads(match.group(1)) + raise + + @staticmethod + def _temperature_for_model(config: ModelConfig) -> float | None: + model_name = config.model_name.lower() + if "gpt-5" in model_name: + return None + return config.temperature diff --git a/src/common/paper_package.py b/src/common/paper_package.py new file mode 100644 index 0000000000000000000000000000000000000000..7adc82d402bc54f2e7419f03ad2bb13310923c4a --- /dev/null +++ b/src/common/paper_package.py @@ -0,0 +1,288 @@ +from __future__ import annotations + +import json +import re +from pathlib import Path +from typing import Any, Dict, List + +from pydantic import BaseModel + + +SECTION_FILES = [ + "abstract.txt", + "introduction.tex", + "related_work.tex", + "tldr.txt", +] + + +class PaperPackage(BaseModel): + paper_dir: Path + paper_metadata: Dict[str, Any] + extracted_discovery_claim: str + downstream_cluster_evidence: List[Dict[str, Any]] + paper_text: Dict[str, str] + full_processed_text: str + bibliography: List[Dict[str, Any]] + citation_contexts: List[Dict[str, Any]] + + def to_prompt_payload(self) -> Dict[str, Any]: + return { + "paper_metadata": self.paper_metadata, + "extracted_discovery_claim": self.extracted_discovery_claim, + "downstream_cluster_evidence": self.downstream_cluster_evidence, + "paper_text": self.paper_text, + "full_processed_text": self.full_processed_text, + "bibliography": self.bibliography, + "citation_contexts": self.citation_contexts, + } + + +def _load_json(path: Path, default: Any) -> Any: + try: + return json.loads(path.read_text()) + except Exception: + return default + + +def _read_text(path: Path) -> str: + try: + return path.read_text() + except Exception: + return "" + + +def _normalize_dict_payload(value: Any) -> Dict[str, Any]: + if isinstance(value, dict): + return value + if isinstance(value, list): + for item in value: + if isinstance(item, dict): + return item + return {} + + +def _collect_sections(paper_dir: Path) -> Dict[str, str]: + sections_dir = paper_dir / "sections" + out: Dict[str, str] = {} + for name in SECTION_FILES: + text = _read_text(sections_dir / name).strip() + if text: + out[name] = text[:12000] + if not out: + processed = _read_text(paper_dir / "processed_main.tex").strip() + if processed: + out["processed_main.tex"] = processed[:24000] + return out + + +def _collect_full_processed_text(paper_dir: Path) -> str: + processed = _read_text(paper_dir / "processed_main.tex").strip() + if processed: + return processed + + sections_dir = paper_dir / "sections" + parts: List[str] = [] + if sections_dir.exists(): + for path in sorted(sections_dir.iterdir()): + if not path.is_file(): + continue + text = _read_text(path).strip() + if text: + parts.append(f"[{path.name}]\n{text}") + return "\n\n".join(parts) + + +def _extract_year(value: Any) -> Any: + if value: + return value + return None + + +def _normalise_reference_record(ref: Dict[str, Any]) -> Dict[str, Any]: + cited = ref.get("citedPaper") + source = cited if isinstance(cited, dict) else ref + external_ids = source.get("external_ids") or source.get("externalIds") or {} + return { + "ref_id": ( + ref.get("ref_id") + or ref.get("bib_key") + or source.get("ref_id") + or source.get("bib_key") + or source.get("paperId") + or source.get("paper_id") + or external_ids.get("ACL") + or external_ids.get("ArXiv") + or external_ids.get("DOI") + ), + "title": source.get("title") or source.get("ref_title"), + "authors": source.get("authors") or source.get("ref_authors"), + "year": _extract_year(source.get("year") or source.get("ref_year")), + "external_ids": external_ids, + } + + +def _parse_bibtex_entries(text: str, limit: int) -> List[Dict[str, Any]]: + entries: List[Dict[str, Any]] = [] + for match in re.finditer(r"@\w+\s*\{\s*([^,]+),(.*?)(?=\n@\w+\s*\{|\Z)", text, re.S): + key = match.group(1).strip() + body = match.group(2) + fields: Dict[str, str] = {} + for field in ("title", "author", "year", "doi", "url", "eprint"): + field_match = re.search( + rf"\b{field}\s*=\s*(\{{(?:[^{{}}]|\{{[^{{}}]*\}})*\}}|\"[^\"]*\"|[^,\n]+)", + body, + re.I | re.S, + ) + if field_match: + value = field_match.group(1).strip().strip(",") + if (value.startswith("{") and value.endswith("}")) or ( + value.startswith('"') and value.endswith('"') + ): + value = value[1:-1] + fields[field] = re.sub(r"\s+", " ", value).strip() + if fields: + external_ids: Dict[str, Any] = {} + if fields.get("doi"): + external_ids["DOI"] = fields["doi"] + if fields.get("eprint"): + external_ids["ArXiv"] = fields["eprint"] + entries.append( + { + "ref_id": key, + "title": fields.get("title"), + "authors": fields.get("author"), + "year": fields.get("year"), + "external_ids": external_ids, + } + ) + if len(entries) >= limit: + break + return entries + + +def _collect_bibtex_citation_contexts(paper_dir: Path, limit: int = 60) -> List[Dict[str, Any]]: + bibtex = _read_text(paper_dir / "references.bib") + processed = _read_text(paper_dir / "processed_main.tex") + if not bibtex or not processed: + return [] + + refs = _parse_bibtex_entries(bibtex, limit=500) + out: List[Dict[str, Any]] = [] + seen: set[tuple[str, int]] = set() + for ref in refs: + ref_id = ref.get("ref_id") + if not ref_id: + continue + for match in re.finditer(rf"\\cite\w*\s*(?:\[[^\]]*\]\s*)*\{{[^}}]*\b{re.escape(str(ref_id))}\b[^}}]*\}}", processed): + key = (str(ref_id), match.start()) + if key in seen: + continue + seen.add(key) + start = max(0, match.start() - 350) + end = min(len(processed), match.end() + 350) + snippet = re.sub(r"\s+", " ", processed[start:end]).strip() + out.append( + { + "ref_id": ref_id, + "citation_marker": ref.get("title") or ref_id, + "text": snippet, + "section": None, + "intents": [], + } + ) + if len(out) >= limit: + return out + return out + + +def _collect_bibliography(paper_dir: Path, limit: int = 80) -> List[Dict[str, Any]]: + refs = _load_json(paper_dir / "references_metadata.json", []) + if isinstance(refs, list) and refs: + return [_normalise_reference_record(ref) for ref in refs[:limit] if isinstance(ref, dict)] + + bibtex = _read_text(paper_dir / "references.bib") + if bibtex: + return _parse_bibtex_entries(bibtex, limit) + return [] + + +def _collect_citation_contexts(paper_dir: Path, limit: int = 60) -> List[Dict[str, Any]]: + refs = _load_json(paper_dir / "references_metadata.json", []) + out = [] + if isinstance(refs, list): + for ref in refs: + if not isinstance(ref, dict): + continue + ref_record = _normalise_reference_record(ref) + for context in ref.get("contextsWithIntent") or []: + if not isinstance(context, dict): + continue + text = context.get("context") or context.get("text") or "" + if not text: + continue + out.append( + { + "ref_id": ref_record.get("ref_id"), + "citation_marker": ref_record.get("title"), + "text": text, + "section": context.get("section"), + "intents": context.get("intents", []), + } + ) + if len(out) >= limit: + return out + contexts = _load_json(paper_dir / "usage_contexts.json", []) + if isinstance(contexts, list): + for item in contexts: + entry = { + "ref_id": item.get("ref_id") or item.get("bib_key"), + "citation_marker": item.get("citation_marker"), + "text": item.get("text") or item.get("text_raw") or "", + "section": item.get("section"), + } + if entry["text"]: + out.append(entry) + if len(out) >= limit: + break + if not out: + out = _collect_bibtex_citation_contexts(paper_dir, limit=limit) + return out + + +def _collect_downstream_cluster_evidence(paper_dir: Path) -> List[Dict[str, Any]]: + discovery = _normalize_dict_payload(_load_json(paper_dir / "usage_discovery_from_contributions.json", {})) + clusters = discovery.get("clusters", []) + out = [] + for cluster in clusters: + out.append( + { + "cluster_id": cluster.get("cluster_id"), + "representative_claim": cluster.get("representative_claim") or cluster.get("cluster_title"), + "cluster_title": cluster.get("cluster_title"), + "count": cluster.get("count"), + "merge_rationale": cluster.get("merge_rationale"), + } + ) + return out + + +def load_paper_package(paper_dir: str | Path, extracted_claim_override: str | None = None) -> PaperPackage: + paper_dir = Path(paper_dir) + discovery = _normalize_dict_payload(_load_json(paper_dir / "usage_discovery_from_contributions.json", {})) + paper_metadata = _normalize_dict_payload(_load_json(paper_dir / "paper_metadata.json", {})) + claim = extracted_claim_override or ( + discovery.get("most_impactful_contribution_self_contained") + or discovery.get("most_impactful_contribution") + or "" + ) + return PaperPackage( + paper_dir=paper_dir, + paper_metadata=paper_metadata, + extracted_discovery_claim=claim, + downstream_cluster_evidence=_collect_downstream_cluster_evidence(paper_dir), + paper_text=_collect_sections(paper_dir), + full_processed_text=_collect_full_processed_text(paper_dir), + bibliography=_collect_bibliography(paper_dir), + citation_contexts=_collect_citation_contexts(paper_dir), + ) diff --git a/src/step_01_fetch/config.py b/src/step_01_fetch/config.py new file mode 100644 index 0000000000000000000000000000000000000000..acf30a41f10cc3b84816b2167908c7f8926ebf40 --- /dev/null +++ b/src/step_01_fetch/config.py @@ -0,0 +1,6 @@ +from pathlib import Path +import os + +ACL_IDS_PATH = Path("input_ids.json") +PAPERS_DIR = Path("papers") +SEMANTIC_SCHOLAR_API_KEY = os.getenv("SEMANTIC_SCHOLAR_API_KEY", "") diff --git a/src/step_01_fetch/fetch_metadata.py b/src/step_01_fetch/fetch_metadata.py new file mode 100644 index 0000000000000000000000000000000000000000..736970e73ac46f446cfb6d6b7a1ab323e4f5df82 --- /dev/null +++ b/src/step_01_fetch/fetch_metadata.py @@ -0,0 +1,440 @@ +from pathlib import Path +import argparse +import json +import os +import random +import re +import tarfile +import time + +import arxiv +import requests + +from config import ACL_IDS_PATH +from process_tex_source import preprocess_tex, extract_introduction_and_related +from semanticscholar_client import get_paper, get_paper_links, search_by_title + + +def load_ids(path: Path): + return json.loads(path.read_text(encoding="utf-8")) + + +def ensure_dir(path: Path): + path.mkdir(parents=True, exist_ok=True) + + +_ARXIV_LAST_TS = 0.0 + + +def _cleanup_partial_source_dir(source_dir: Path) -> None: + for pattern in ("*.tar.gz", "*.tgz", "*.tar"): + for path in source_dir.glob(pattern): + try: + path.unlink() + except Exception: + pass + + +def _download_arxiv_source_with_retries(paper, source_dir: Path, arxiv_id: str) -> Path | None: + max_retries = int(os.getenv("ARXIV_SOURCE_MAX_RETRIES", "4")) + base_sleep = float(os.getenv("ARXIV_SOURCE_BASE_SLEEP", "2.0")) + max_sleep = float(os.getenv("ARXIV_MAX_BACKOFF", "60")) + last_exc = None + + for attempt in range(max_retries): + _cleanup_partial_source_dir(source_dir) + try: + _arxiv_min_interval_sleep() + tar_path = Path(paper.download_source(dirpath=str(source_dir))) + if not tar_path.exists(): + raise FileNotFoundError(f"download_source returned {tar_path}, but the file does not exist") + if tar_path.stat().st_size < 1024: + raise IOError(f"downloaded source archive is unexpectedly small ({tar_path.stat().st_size} bytes)") + return tar_path + except Exception as exc: + last_exc = exc + sleep = min(base_sleep * (2**attempt), max_sleep) + random.uniform(0.0, 0.5) + print(f"[WARN] Failed to download source for {arxiv_id} on attempt {attempt + 1}/{max_retries}: {exc}") + if attempt + 1 < max_retries: + print(f"[INFO] Retrying source download in {sleep:.2f}s") + time.sleep(sleep) + + print(f"[WARN] Source download failed for {arxiv_id} after {max_retries} attempts: {last_exc}") + return None + + +def _arxiv_min_interval_sleep() -> None: + """Global throttle to avoid arXiv API rate limits.""" + global _ARXIV_LAST_TS + min_interval = float(os.getenv("ARXIV_MIN_INTERVAL", "1.0")) + now = time.monotonic() + elapsed = now - _ARXIV_LAST_TS + if elapsed < min_interval: + time.sleep(min_interval - elapsed) + _ARXIV_LAST_TS = time.monotonic() + + +def download_arxiv_tex(arxiv_id: str, base_dir: Path) -> Path | None: + """ + Download LaTeX source from arXiv and return the path to a merged TeX file. + + - arxiv_id: e.g. "2410.22815" + - base_dir: paper directory where source should be unpacked + """ + source_dir = base_dir / f"tex_{arxiv_id}" + source_dir.mkdir(parents=True, exist_ok=True) + search = arxiv.Search(id_list=[arxiv_id]) + max_retries = int(os.getenv("ARXIV_MAX_RETRIES", "6")) + base_sleep = float(os.getenv("ARXIV_BASE_SLEEP", "2.0")) + max_sleep = float(os.getenv("ARXIV_MAX_BACKOFF", "60")) + paper = None + + for attempt in range(max_retries): + try: + _arxiv_min_interval_sleep() + paper = next(search.results()) + break + except StopIteration: + print(f"[WARN] No arXiv paper found for ID {arxiv_id}") + return None + except arxiv.HTTPError as exc: + if getattr(exc, "status", None) == 429 or "429" in str(exc): + sleep = min(base_sleep * (2**attempt), max_sleep) + random.uniform(0.0, 0.5) + print(f"[WARN] arXiv 429 → retrying in {sleep:.2f}s") + time.sleep(sleep) + continue + print(f"[WARN] arXiv HTTP error for {arxiv_id}: {exc}") + return None + except Exception as exc: + sleep = min(base_sleep * (2**attempt), max_sleep) + random.uniform(0.0, 0.5) + print(f"[WARN] arXiv error {exc} → retrying in {sleep:.2f}s") + time.sleep(sleep) + continue + + if paper is None: + print(f"[ERROR] Giving up after {max_retries} attempts for arXiv ID {arxiv_id}") + return None + + tar_path = _download_arxiv_source_with_retries(paper, source_dir, arxiv_id) + if tar_path is None: + return None + + try: + with tarfile.open(tar_path) as tar: + tar.extractall(path=source_dir) + os.remove(tar_path) + except Exception as exc: + print(f"[WARN] Failed to extract source for {arxiv_id}: {exc}") + return None + + processed_tex = preprocess_tex(source_dir) + if processed_tex: + extract_introduction_and_related(processed_tex) + + if not processed_tex or not processed_tex.exists(): + print(f"[WARN] Could not produce merged TeX for {arxiv_id}") + return None + + print(f"[INFO] Processed LaTeX for {arxiv_id} at {processed_tex}") + return processed_tex + + +def _extract_arxiv_id_from_text(text: str) -> str | None: + if not text: + return None + match = re.search(r"\b(\d{4}\.\d{4,5}(?:v\d+)?)\b", text) + if match: + return match.group(1) + match = re.search(r"arxiv[:\s/]*(\d{4}\.\d{4,5}(?:v\d+)?)", text, re.IGNORECASE) + if match: + return match.group(1) + return None + + +def _safe_write_json(path: Path, payload) -> None: + path.write_text(json.dumps(payload, indent=2), encoding="utf-8") + + +def _safe_write_text(path: Path, text: str) -> None: + path.write_text(text, encoding="utf-8") + + +def _query_openreview_for_paper(openreview_id: str) -> dict | None: + """Query OpenReview using a real OpenReview note/forum id.""" + if not openreview_id: + return None + + try_urls = [ + f"https://api.openreview.net/notes?forum={openreview_id}", + f"https://api2.openreview.net/notes?forum={openreview_id}", + f"https://api.openreview.net/notes?id={openreview_id}", + f"https://api2.openreview.net/notes?id={openreview_id}", + ] + + for url in try_urls: + try: + response = requests.get(url, timeout=20) + if response.status_code != 200: + continue + payload = response.json() + except Exception: + continue + + notes = None + if isinstance(payload, dict) and isinstance(payload.get("notes"), list): + notes = payload["notes"] + elif isinstance(payload, dict) and payload.get("content"): + notes = [payload] + elif isinstance(payload, list): + notes = payload + + if not notes: + continue + + note = notes[0] + content = note.get("content") if isinstance(note, dict) else None + title = None + arxiv_id = None + pdf_url = None + + if isinstance(content, dict): + raw_title = content.get("title") or content.get("paperTitle") + title = raw_title.get("value") if isinstance(raw_title, dict) else raw_title + + raw_pdf = content.get("pdf") + pdf_url = raw_pdf.get("value") if isinstance(raw_pdf, dict) else raw_pdf + + for value in content.values(): + if isinstance(value, dict): + value = value.get("value") + if isinstance(value, list): + value = " ".join(str(item) for item in value) + if isinstance(value, str): + arxiv_id = _extract_arxiv_id_from_text(value) + if arxiv_id: + break + + if not title and isinstance(note, dict): + title = note.get("title") or note.get("forumTitle") + + if not arxiv_id and isinstance(note, dict): + for value in note.values(): + if isinstance(value, str): + arxiv_id = _extract_arxiv_id_from_text(value) + if arxiv_id: + break + + return { + "title": title, + "arxiv_id": arxiv_id, + "pdf_url": pdf_url, + "openreview_id": openreview_id, + "source_url": url, + } + + return None + + +def _treat_as_openreview(paper: dict) -> bool: + acl_id = str(paper.get("id", "")).lower() + id_type = str(paper.get("id_type", "")).lower() + return ( + id_type == "openreview" + or bool(paper.get("openreview_id")) + or acl_id.startswith("neurips-") + or acl_id.startswith("icml-") + ) + + +def _fetch_s2_by_title(title: str, acl_id: str) -> tuple[int, dict | None]: + if not title: + print(f"[WARN] no title available for {acl_id} → skipping.") + return 0, None + hit = search_by_title(title) + if not hit: + print(f"[WARN] no S2 match for {acl_id} ({title}) → skipping.") + return 0, None + s2_id = hit["paperId"] + print(f"[DEBUG] title search matched semantic scholar paperId={s2_id}") + return get_paper(s2_id, id_type="SemanticScholar") + + +def _best_arxiv_id(*values: str) -> str | None: + for value in values: + arxiv_id = _extract_arxiv_id_from_text(value or "") + if arxiv_id: + return arxiv_id + return None + + +def _write_openreview_snapshot(paper_dir: Path, payload: dict) -> None: + if payload: + _safe_write_json(paper_dir / "openreview_metadata.json", payload) + + +def _write_metadata_outputs(paper_dir: Path, acl_id: str, data: dict) -> None: + meta_path = paper_dir / "paper_metadata.json" + _safe_write_json(meta_path, [data]) + print(f"[DEBUG] wrote metadata to {meta_path}") + + external_ids = data.get("externalIds", {}) or {} + arxiv_id = external_ids.get("ArXiv") + if arxiv_id: + download_arxiv_tex(arxiv_id=arxiv_id, base_dir=paper_dir) + + sections_dir = paper_dir / "sections" + sections_dir.mkdir(exist_ok=True) + + abstract = data.get("abstract") + if abstract: + _safe_write_text(sections_dir / "abstract.txt", abstract) + + tldr_obj = data.get("tldr") + if isinstance(tldr_obj, dict) and tldr_obj.get("text"): + _safe_write_text(sections_dir / "tldr.txt", tldr_obj["text"]) + + semantic_id = data.get("paperId") + if not semantic_id: + print(f"[WARN] no semantic_id for {acl_id} → skip refs/cites.") + return + + citation_count = data.get("citationCount", 0) + reference_count = data.get("referenceCount", 0) + + ref_status, refs = get_paper_links(semantic_id, "references", reference_count) + if ref_status == 200: + _safe_write_json(paper_dir / "references_metadata.json", refs) + + cit_status, cits = get_paper_links(semantic_id, "citations", citation_count) + if cit_status == 200: + _safe_write_json(paper_dir / "citations_metadata.json", cits) + + if "ArXiv" not in external_ids: + _safe_write_text(paper_dir / "no_arxiv.txt", "no arxiv for this paper") + + +def fetch_one_acl_id(paper: dict, base_dir: Path): + acl_id = paper["id"] + title = (paper.get("title") or "").strip() + id_type = paper.get("id_type", "ACL") + openreview_id = paper.get("openreview_id", "") + input_pdf_url = paper.get("pdf_url", "") + s2_key = os.getenv("SEMANTIC_SCHOLAR_API_KEY", "") + print( + f"[DEBUG] fetch_one_acl_id: id={acl_id} id_type={id_type} " + f"title_len={len(title)} s2_key_present={'yes' if bool(s2_key) else 'no'} " + f"s2_key_len={len(s2_key)}" + ) + + paper_dir = base_dir / acl_id + ensure_dir(paper_dir) + meta_path = paper_dir / "paper_metadata.json" + + if meta_path.exists(): + return + + status, data = 0, None + fetch_label = f"{id_type}:{acl_id}" + is_openreview = _treat_as_openreview(paper) + openreview_meta = None + attempted_title_search = False + + if is_openreview: + try: + openreview_meta = _query_openreview_for_paper(openreview_id or acl_id) + except Exception as exc: + print(f"[WARN] OpenReview lookup failed for {acl_id}: {exc}") + openreview_meta = None + + if openreview_meta: + _write_openreview_snapshot(paper_dir, openreview_meta) + or_title = (openreview_meta.get("title") or title or "").strip() + arxiv_id = ( + _best_arxiv_id( + openreview_meta.get("arxiv_id", ""), + openreview_meta.get("pdf_url", ""), + input_pdf_url, + ) + or "" + ) + if arxiv_id: + print(f"[DEBUG] OpenReview -> found ArXiv {arxiv_id} for {acl_id}") + status, data = get_paper(arxiv_id, id_type="ArXiv") + fetch_label = f"ArXiv:{arxiv_id}" + title = or_title or title + elif or_title: + print(f"[DEBUG] OpenReview -> no arXiv for {acl_id}, title-searching") + status, data = _fetch_s2_by_title(or_title, acl_id) + fetch_label = f"title:{or_title[:80]}" + title = or_title + attempted_title_search = True + else: + print(f"[WARN] OpenReview metadata for {acl_id} had neither title nor arXiv") + else: + print(f"[WARN] no OpenReview metadata for {acl_id} (openreview_id={openreview_id or acl_id})") + + if data is None and title and not attempted_title_search: + print(f"[DEBUG] OpenReview fallback -> title-searching extracted title for {acl_id}") + status, data = _fetch_s2_by_title(title, acl_id) + fetch_label = f"title:{title[:80]}" + attempted_title_search = True + + if data is None and not is_openreview: + status, data = get_paper(acl_id, id_type=id_type) + fetch_label = f"{id_type}:{acl_id}" + + if data is None and not attempted_title_search: + print( + f"[WARN] direct fetch failed for {fetch_label} " + f"(status={status}) → trying title search with title_len={len(title)}" + ) + status, data = _fetch_s2_by_title(title, acl_id) + + if status != 200 or data is None: + print(f"[WARN] still no data for {acl_id} → skipping.") + return + + _write_metadata_outputs(paper_dir, acl_id, data) + print("[SUCCESS]") + + +def fetch_all_metadata(acl_ids_path: Path, out_dir: Path, start_from: str | None = None, resume: bool = False): + raw = json.loads(acl_ids_path.read_text(encoding="utf-8")) + papers = raw if isinstance(raw[0], dict) else [{"id": x, "title": ""} for x in raw] + + start_seen = start_from is None + for paper in papers: + pid = str(paper.get("id", "")) + if not start_seen: + if pid == start_from: + start_seen = True + else: + continue + if resume: + paper_dir = out_dir / pid + if (paper_dir / "paper_metadata.json").exists(): + continue + fetch_one_acl_id(paper, out_dir) + return "Meta Data Completed" + + +if __name__ == "__main__": + parser = argparse.ArgumentParser() + parser.add_argument("--ids", type=str, required=True, help="Path to JSON file with paper IDs.") + parser.add_argument("--outdir", type=str, default="papers", help="Output directory for metadata.") + parser.add_argument("--start-from", type=str, default=None, help="Start from this paper ID.") + parser.add_argument("--resume", action="store_true", help="Skip papers that already have paper_metadata.json.") + args = parser.parse_args() + + ACL_IDS_PATH = Path(args.ids).expanduser().resolve() + OUTDIR = Path(args.outdir).expanduser().resolve() + + if not ACL_IDS_PATH.exists(): + raise FileNotFoundError(f"Could not find {ACL_IDS_PATH}") + + print(f"[INFO] Using ID list from {ACL_IDS_PATH}") + print(f"[INFO] Output will be saved to {OUTDIR}") + + start = time.time() + fetch_all_metadata(acl_ids_path=ACL_IDS_PATH, out_dir=OUTDIR, start_from=args.start_from, resume=args.resume) + print("done in", time.time() - start, "s") diff --git a/src/step_01_fetch/process_tex_source.py b/src/step_01_fetch/process_tex_source.py new file mode 100644 index 0000000000000000000000000000000000000000..769a4bee60eea87064c5d7a7a72527576592432d --- /dev/null +++ b/src/step_01_fetch/process_tex_source.py @@ -0,0 +1,203 @@ +import os +import re +from pathlib import Path +import shutil + +def read_tex(path: Path) -> str: + try: + return path.read_text(encoding="utf-8", errors="ignore") + except Exception: + return "" + + +def resolve_inputs(tex: str, base_dir: Path, seen=None) -> str: + """ + Recursively replace \\input{...} and \\include{...} with file contents. + """ + if seen is None: + seen = set() + + pattern = r'\\(?:input|include)\{([^}]+)\}' + + def repl(match): + name = match.group(1) + if not name.endswith(".tex"): + name += ".tex" + + full = base_dir / name + + if full in seen: + return f"% WARNING: skipped circular input {full}\n" + + if not full.exists(): + return f"% WARNING: missing file {full}\n" + + seen.add(full) + content = read_tex(full) + return resolve_inputs(content, full.parent, seen) + + return re.sub(pattern, repl, tex) + + +def find_main_tex(source_dir: Path) -> Path | None: + """ + Heuristic to find the main .tex file: + 1. match .bbl → .tex + 2. else top-level .tex that contains \\begin{document} + 3. else first .tex in directory + """ + bbls = list(source_dir.glob("*.bbl")) + if bbls: + main_candidate = source_dir / (bbls[0].stem + ".tex") + if main_candidate.exists(): + return main_candidate + + for tex in source_dir.glob("*.tex"): + if "\\begin{document}" in read_tex(tex): + return tex + + tex_files = list(source_dir.glob("*.tex")) + return tex_files[0] if tex_files else None + + +def preprocess_tex(source_dir: Path) -> Path | None: + """ + Given an extracted arXiv source directory, produce: + - a merged TeX file named 'processed_main.tex' + - a concatenated BibTeX file named 'references.bib' + Both are written in the parent directory of source_dir (the paper dir). + + Then delete the extracted source_dir. + """ + main_tex = find_main_tex(source_dir) + if not main_tex: + print(f"[WARN] No main .tex found in {source_dir}") + shutil.rmtree(source_dir, ignore_errors=True) + return None + + raw = read_tex(main_tex) + merged = resolve_inputs(raw, main_tex.parent) + + paper_dir = source_dir.parent + out_tex_path = paper_dir / "processed_main.tex" + out_tex_path.write_text(merged, encoding="utf-8") + + bib_files = list(source_dir.rglob("*.bib")) + if bib_files: + bib_texts = [] + for bib in bib_files: + try: + bib_texts.append(bib.read_text(encoding="utf-8", errors="ignore")) + except Exception: + print(f"[WARN] Could not read bib file {bib}") + if bib_texts: + bib_out = paper_dir / "references.bib" + bib_out.write_text("\n\n".join(bib_texts), encoding="utf-8") + print(f"[INFO] Wrote combined BibTeX to {bib_out}") + + shutil.rmtree(source_dir, ignore_errors=True) + + return out_tex_path + +def _load_tex(path: Path) -> str: + return path.read_text(encoding="utf-8", errors="ignore") + + +SECTION_PATTERN = re.compile( + r'\\section\*?\{([^}]*)\}', + flags=re.IGNORECASE +) + + +def _split_into_sections(tex: str): + """ + Returns a list of (section_title, content) in order. + Title is the raw LaTeX title text (without braces). + Content is the text from this \\section line up to (but not including) + the next \\section or end of document. + """ + sections = [] + matches = list(SECTION_PATTERN.finditer(tex)) + + if not matches: + return sections + + for i, m in enumerate(matches): + title = m.group(1).strip() + start = m.start() + end = matches[i + 1].start() if i + 1 < len(matches) else len(tex) + content = tex[start:end] + sections.append((title, content)) + + return sections + + +def _normalize_title(title: str) -> str: + """Lowercase and strip punctuation-ish stuff for robust matching.""" + t = title.lower() + t = re.sub(r'[^a-z0-9\s]', ' ', t) + t = re.sub(r'\s+', ' ', t).strip() + return t + + +def _find_best_section(sections, candidates): + """ + sections: list of (raw_title, content) + candidates: list of strings to match against normalized title + Returns the content of the best-matching section or None. + """ + norm_candidates = [c.lower() for c in candidates] + + for raw_title, content in sections: + nt = _normalize_title(raw_title) + for cand in norm_candidates: + if nt == cand or cand in nt: + return content + return None + + +def extract_introduction_and_related( + processed_tex_path: Path, + out_dir: Path | None = None, +) -> dict: + """ + Given path to processed_main.tex, extract Introduction and Related Work sections + into separate .tex files. + + Returns a dict with keys: + { + "introduction": Path | None, + "related_work": Path | None + } + """ + if out_dir is None: + out_dir = processed_tex_path.parent / "sections" + + out_dir.mkdir(parents=True, exist_ok=True) + + tex = _load_tex(processed_tex_path) + sections = _split_into_sections(tex) + + intro_candidates = ["introduction"] + related_candidates = ["related work"] + + intro_content = _find_best_section(sections, intro_candidates) + related_content = _find_best_section(sections, related_candidates) + + results = {"introduction": None, "related_work": None} + + if intro_content: + intro_path = out_dir / "introduction.tex" + intro_path.write_text(intro_content, encoding="utf-8") + results["introduction"] = intro_path + else: + print(f"[WARN] No Introduction section found in {processed_tex_path}") + + if related_content: + rw_path = out_dir / "related_work.tex" + rw_path.write_text(related_content, encoding="utf-8") + results["related_work"] = rw_path + else: + print(f"[WARN] No Related Work section found in {processed_tex_path}") + + return results diff --git a/src/step_01_fetch/semanticscholar_client.py b/src/step_01_fetch/semanticscholar_client.py new file mode 100644 index 0000000000000000000000000000000000000000..86ca3a76b6f117ebd7fbea5417091189b45ec07c --- /dev/null +++ b/src/step_01_fetch/semanticscholar_client.py @@ -0,0 +1,158 @@ +import time +import random +import requests +from typing import Optional, Tuple, Any +from config import SEMANTIC_SCHOLAR_API_KEY +import os + +BASE_URL = "https://api.semanticscholar.org/graph/v1/paper" + +_LAST_REQUEST_TS = 0.0 + + +def _min_interval_sleep() -> None: + """Global throttle to avoid hammering Semantic Scholar.""" + global _LAST_REQUEST_TS + min_interval = float(os.getenv("S2_MIN_INTERVAL", "1.0")) + now = time.monotonic() + elapsed = now - _LAST_REQUEST_TS + if elapsed < min_interval: + time.sleep(min_interval - elapsed) + _LAST_REQUEST_TS = time.monotonic() + + +def robust_request(url, params=None, headers=None, max_retries=8, base_sleep=2.0): + """ + Make a GET request with exponential backoff. + Retries on: + - connection errors + - 429 (Too Many Requests) + - 500–599 server errors + - invalid JSON + Returns (status_code, json_or_None). + """ + + for attempt in range(max_retries): + try: + _min_interval_sleep() + resp = requests.get(url, params=params, headers=headers, timeout=30) + status = resp.status_code + + if status == 200: + try: + return 200, resp.json() + except Exception: + print(f"[WARN] JSON decode failed on attempt {attempt+1}/{max_retries}") + + if status == 429: + retry_after = resp.headers.get("Retry-After") + if retry_after: + try: + sleep = float(retry_after) + except Exception: + sleep = base_sleep * (2 ** attempt) + else: + sleep = base_sleep * (2 ** attempt) + max_sleep = float(os.getenv("S2_MAX_BACKOFF", "60")) + sleep = min(sleep, max_sleep) + sleep += random.uniform(0.0, 0.5) + print(f"[WARN] 429 Too Many Requests → retrying in {sleep:.2f}s") + time.sleep(sleep) + continue + + if 500 <= status < 600: + sleep = base_sleep * (2 ** attempt) + max_sleep = float(os.getenv("S2_MAX_BACKOFF", "60")) + sleep = min(sleep, max_sleep) + sleep += random.uniform(0.0, 0.5) + print(f"[WARN] Server error {status} → retrying in {sleep:.2f}s") + time.sleep(sleep) + continue + + return status, None + + except requests.exceptions.RequestException as e: + sleep = base_sleep * (2 ** attempt) + max_sleep = float(os.getenv("S2_MAX_BACKOFF", "60")) + sleep = min(sleep, max_sleep) + sleep += random.uniform(0.0, 0.5) + print(f"[WARN] Network error {e} → retrying in {sleep:.2f}s") + time.sleep(sleep) + continue + + print(f"[ERROR] Giving up after {max_retries} attempts for URL: {url}") + return None, None +def get_paper(paper_id: str, id_type: str = "ACL") -> Tuple[int, Optional[dict]]: + """ + id_type can be "ACL" or "SemanticScholar" or "ArXiv" etc. + """ + if id_type == "SemanticScholar": + full_id = paper_id + else: + full_id = f"{id_type}:{paper_id}" + + url = f"{BASE_URL}/{full_id}" + params = { + "fields": ( + "title,year,publicationDate,authors,url,venue,externalIds," + "tldr,abstract,citationCount,referenceCount,openAccessPdf" + ) + } + + headers = {"x-api-key": SEMANTIC_SCHOLAR_API_KEY} if SEMANTIC_SCHOLAR_API_KEY else {} + + status, data = robust_request(url, params=params, headers=headers, max_retries=5, base_sleep=1.0) + if status == 200 and data is not None: + return status, data + else: + print(f"[WARN] {status} on {full_id}") + return status or 0, None + + +def get_paper_links(semantic_id: str, target_type: str, total: int, limit: int = 1000): + headers = {"x-api-key": SEMANTIC_SCHOLAR_API_KEY} if SEMANTIC_SCHOLAR_API_KEY else {} + loops = total // limit + 1 if total else 0 + collected = [] + + for i in range(loops): + offset = i * limit + url = f"{BASE_URL}/{semantic_id}/{target_type}" + params = { + "offset": offset, + "limit": limit, + "fields": "paperId,title,isInfluential,externalIds,contextsWithIntent,openAccessPdf", + } + + status, data = robust_request(url, params=params, headers=headers, max_retries=5, base_sleep=1.0) + + if status != 200 or data is None: + print(f"[WARN] {target_type} fetch failed for {semantic_id} (status {status})") + return status or 0, [] + + items = data.get("data") + if not isinstance(items, list): + print(f"[WARN] malformed {target_type} response for {semantic_id}") + return status, [] + + collected.extend(items) + + return 200, collected + + +def search_by_title(title: str, limit: int = 1): + """Search Semantic Scholar by paper title.""" + url = "https://api.semanticscholar.org/graph/v1/paper/search" + params = { + "query": title, + "limit": limit, + "fields": "paperId,title,year,venue,externalIds", + } + headers = {"x-api-key": SEMANTIC_SCHOLAR_API_KEY} if SEMANTIC_SCHOLAR_API_KEY else {} + + status, data = robust_request(url, params=params, headers=headers, max_retries=5, base_sleep=1.0) + if status == 200 and data is not None: + items = data.get("data", []) + return items[0] if items else None + else: + print(f"[WARN] title search failed for '{title[:60]}...' (status {status})") + return None diff --git a/src/step_02_mark_citations/replace_citation_markers.py b/src/step_02_mark_citations/replace_citation_markers.py new file mode 100644 index 0000000000000000000000000000000000000000..e58676a68c6eacf455c8e723ebd387f9eade7fc6 --- /dev/null +++ b/src/step_02_mark_citations/replace_citation_markers.py @@ -0,0 +1,440 @@ +import argparse +import json +import re +from pathlib import Path +from typing import Any, Dict, List, Tuple + + +PAPER_META_FILE = "paper_metadata.json" +USAGE_CLAIMS_FILE = "usage_claims.json" +USAGE_CONTEXTS_FILE = "usage_contexts.json" +CITATIONS_FILE = "citations_metadata.json" +PROCESSED_MAIN_FILE = "processed_main.tex" +REFERENCES_META_FILE = "references_metadata.json" + + +def load_json(path: Path) -> Any | None: + if not path.exists(): + return None + try: + return json.loads(path.read_text(encoding="utf-8")) + except Exception: + return None + + +def save_json(path: Path, data: Any) -> None: + path.write_text(json.dumps(data, indent=2), encoding="utf-8") + + +def iter_paper_dirs(root: Path) -> List[Path]: + out: List[Path] = [] + for child in root.iterdir(): + if child.is_dir() and (child / PAPER_META_FILE).exists(): + out.append(child) + return out + + +def load_paper_metadata(paper_dir: Path) -> Dict[str, Any]: + meta = load_json(paper_dir / PAPER_META_FILE) + if isinstance(meta, list) and meta: + return meta[0] + if isinstance(meta, dict): + return meta + return {} + + +def _is_structurally_complete(paper_dir: Path) -> bool: + return ( + (paper_dir / PAPER_META_FILE).exists() + and (paper_dir / PROCESSED_MAIN_FILE).exists() + and (paper_dir / REFERENCES_META_FILE).exists() + ) + + +def _author_last_names(authors: List[Any]) -> List[str]: + last_names: List[str] = [] + for author in authors: + if isinstance(author, dict): + name = author.get("name") + else: + name = author + if not isinstance(name, str): + continue + parts = [p for p in re.split(r"\s+", name.strip()) if p] + if not parts: + continue + last_names.append(parts[-1]) + return list(dict.fromkeys(last_names)) + + +def _title_aliases(title: str) -> List[str]: + aliases = [title] + if ":" in title: + aliases.append(title.split(":", 1)[0]) + acronym = "".join([c for c in title if c.isupper()]) + if 3 <= len(acronym) <= 10: + aliases.append(acronym) + return list(dict.fromkeys([a for a in aliases if a])) + + +def _artifact_aliases(paper_dir: Path) -> List[str]: + aliases: List[str] = [] + usage_claims = load_json(paper_dir / USAGE_CLAIMS_FILE) + if isinstance(usage_claims, dict): + caps = usage_claims.get("capabilities") or [] + if isinstance(caps, list): + for cap in caps: + if not isinstance(cap, dict): + continue + name = cap.get("artifact_name") + if isinstance(name, str) and name.strip(): + aliases.append(name.strip()) + return list(dict.fromkeys(aliases)) + + +def _loose_alias_pattern(alias: str) -> str: + parts = re.split(r"[^A-Za-z0-9]+", alias) + parts = [p for p in parts if p] + if not parts: + return "" + return r"\b" + r"[-\s]*".join(map(re.escape, parts)) + r"\b" + + +def build_patterns( + meta: Dict[str, Any], + paper_dir: Path, +) -> Tuple[List[re.Pattern], List[re.Pattern], str | None]: + year = meta.get("year") + year_str = str(year) if isinstance(year, int) else None + authors = meta.get("authors") if isinstance(meta.get("authors"), list) else [] + last_names = _author_last_names(authors) + title = meta.get("title") if isinstance(meta.get("title"), str) else "" + aliases = _title_aliases(title) + _artifact_aliases(paper_dir) + aliases = [a for a in aliases if a] + + author_patterns: List[re.Pattern] = [] + alias_patterns: List[re.Pattern] = [] + + if year_str and last_names: + year_pat = rf"{re.escape(year_str)}[a-z]?" + first_last = re.escape(last_names[0]) + author_patterns.append( + re.compile( + rf"\b{first_last}\s+et\s+al\.?\s*(?:,\s*|\s*){year_pat}", + re.IGNORECASE, + ) + ) + + for alias in aliases: + pat = _loose_alias_pattern(alias) + if pat: + alias_patterns.append(re.compile(pat, re.IGNORECASE)) + + first_last = last_names[0] if last_names else None + return author_patterns, alias_patterns, first_last + + +def _replace_author_span(text: str, first_last: str) -> Tuple[str, bool]: + occurrences = list(re.finditer(rf"\b{re.escape(first_last)}\b", text, re.IGNORECASE)) + if len(occurrences) != 1: + return text, False + author_pat = re.compile( + rf"\(?\b{re.escape(first_last)}\b" + rf"\s+(?:et\s+al\.?|and|&)\s*" + rf"(?:,?\s*\(?\d{{4}}[a-z]?\)?)?" + rf"\)?", + re.IGNORECASE, + ) + new_text, count = author_pat.subn("", text, count=1) + return new_text, count > 0 + + +_BRACKET_NUM_RE = re.compile(r"\[[0-9,;\s]+\]") +_BRACKET_GROUP_RE = re.compile(r"\[([0-9,;\s]+)\]") + + +def _extract_bracket_numbers(text: str) -> List[str]: + numbers: List[str] = [] + for match in _BRACKET_GROUP_RE.finditer(text): + parts = re.split(r"[,\s;]+", match.group(1).strip()) + for part in parts: + if part.isdigit(): + numbers.append(part) + return numbers + + +def _dominant_bracket(contexts: List[Dict[str, Any]]) -> str | None: + counts: Dict[str, int] = {} + for ctx in contexts: + if not isinstance(ctx, dict): + continue + text = ctx.get("context") or ctx.get("text") + if not isinstance(text, str): + continue + for num in _extract_bracket_numbers(text): + counts[num] = counts.get(num, 0) + 1 + if not counts: + return None + best = max(counts.values()) + winners = [num for num, count in counts.items() if count == best] + if len(winners) == 1: + return winners[0] + return None + + +def _single_bracket_candidate(contexts: List[Dict[str, Any]]) -> str | None: + counts: Dict[str, int] = {} + for ctx in contexts: + if not isinstance(ctx, dict): + continue + text = ctx.get("context") or ctx.get("text") + if not isinstance(text, str): + continue + matches = list(_BRACKET_GROUP_RE.finditer(text)) + if len(matches) == 1: + nums = _extract_bracket_numbers(text) + if len(nums) != 1: + continue + num = nums[0] + counts[num] = counts.get(num, 0) + 1 + if not counts: + return None + best = max(counts.values()) + winners = [num for num, count in counts.items() if count == best] + if len(winners) == 1: + return winners[0] + return None + + +def _replace_single_bracket(text: str, dominant: str | None) -> Tuple[str, bool]: + matches = list(_BRACKET_GROUP_RE.finditer(text)) + if len(matches) != 1: + return text, False + nums = _extract_bracket_numbers(text) + if len(nums) != 1: + return text, False + num = nums[0] + if dominant is not None and num != dominant: + return text, False + start, end = matches[0].span() + return text[:start] + "" + text[end:], True + + +def replace_with_marker( + text: str, + author_patterns: List[re.Pattern], + alias_patterns: List[re.Pattern], + dominant_bracket: str | None = None, + first_author_last: str | None = None, +) -> Tuple[str, bool]: + def _collapse_markers(value: str) -> str: + value = re.sub(r"(?:[\s()\[\],;:]*){2,}", " ", value) + value = re.sub(r"(?:\s+)+", "", value) + return value.strip() + + updated = text + changed = False + + author_changed = False + if first_author_last: + new, author_changed = _replace_author_span(updated, first_author_last) + if author_changed: + changed = True + updated = _collapse_markers(new) + + if dominant_bracket: + def _replace_if_contains(match: re.Match) -> str: + nums = re.split(r"[,\s;]+", match.group(1).strip()) + if any(n == dominant_bracket for n in nums if n.isdigit()): + return "" + return match.group(0) + + new = _BRACKET_GROUP_RE.sub(_replace_if_contains, updated) + if new != updated: + changed = True + updated = _collapse_markers(new) + + for pat in author_patterns: + new = pat.sub("", updated) + if new != updated: + changed = True + updated = _collapse_markers(new) + + if not author_changed: + for pat in alias_patterns: + new = pat.sub("", updated) + if new != updated: + changed = True + updated = _collapse_markers(new) + + new, bracket_changed = _replace_single_bracket(updated, dominant_bracket) + if bracket_changed: + changed = True + updated = _collapse_markers(new) + + updated = _collapse_markers(updated) + return updated, changed + + +def _process_contexts( + contexts: List[Dict[str, Any]], + author_patterns: List[re.Pattern], + alias_patterns: List[re.Pattern], + dominant_bracket: str | None, + first_author_last: str | None, +) -> Tuple[int, int]: + updated_count = 0 + total = 0 + for ctx in contexts: + if not isinstance(ctx, dict): + continue + text = ctx.get("context") or ctx.get("text") + if not isinstance(text, str): + continue + total += 1 + new_text, changed = replace_with_marker( + text, + author_patterns=author_patterns, + alias_patterns=alias_patterns, + dominant_bracket=dominant_bracket, + first_author_last=first_author_last, + ) + if changed: + updated_count += 1 + ctx["context_with_marker"] = new_text + return updated_count, total + + +def update_citations_file( + paper_dir: Path, + author_patterns: List[re.Pattern], + alias_patterns: List[re.Pattern], + first_author_last: str | None, +) -> Tuple[int, int]: + path = paper_dir / CITATIONS_FILE + data = load_json(path) + if not isinstance(data, list): + return 0, 0 + updated = 0 + total = 0 + for entry in data: + if not isinstance(entry, dict): + continue + ctxs = entry.get("contextsWithIntent") or [] + if isinstance(ctxs, list): + dominant = _dominant_bracket(ctxs) + if dominant is None: + dominant = _single_bracket_candidate(ctxs) + upd, tot = _process_contexts( + ctxs, + author_patterns, + alias_patterns, + dominant, + first_author_last, + ) + updated += upd + total += tot + save_json(path, data) + return updated, total + + +def update_usage_contexts_file( + paper_dir: Path, + author_patterns: List[re.Pattern], + alias_patterns: List[re.Pattern], + first_author_last: str | None, +) -> Tuple[int, int]: + path = paper_dir / USAGE_CONTEXTS_FILE + data = load_json(path) + if not isinstance(data, dict): + return 0, 0 + updated = 0 + total = 0 + for entry in data.get("citing_papers", []) or []: + if not isinstance(entry, dict): + continue + ctxs = entry.get("contexts") or [] + if isinstance(ctxs, list): + dominant = _dominant_bracket(ctxs) + if dominant is None: + dominant = _single_bracket_candidate(ctxs) + upd, tot = _process_contexts( + ctxs, + author_patterns, + alias_patterns, + dominant, + first_author_last, + ) + updated += upd + total += tot + save_json(path, data) + return updated, total + + +def main() -> None: + parser = argparse.ArgumentParser( + description="Replace citation mentions with in context fields." + ) + parser.add_argument( + "--root", + type=str, + default="runs/processed_papers", + help="Root directory containing processed paper directories.", + ) + parser.add_argument( + "--usage-contexts", + action="store_true", + help="Also update usage_contexts.json.", + ) + args = parser.parse_args() + + root = Path(args.root).expanduser().resolve() + if not root.exists(): + raise SystemExit(f"Root directory does not exist: {root}") + + paper_dirs = sorted(iter_paper_dirs(root), key=lambda p: p.name) + print(f"[INFO] Found {len(paper_dirs)} paper dirs under {root}") + total_updated = 0 + total_contexts = 0 + skipped_incomplete = 0 + + for paper_dir in paper_dirs: + if not _is_structurally_complete(paper_dir): + skipped_incomplete += 1 + continue + meta = load_paper_metadata(paper_dir) + if not meta: + continue + author_patterns, alias_patterns, first_author_last = build_patterns(meta, paper_dir) + if not (author_patterns or alias_patterns): + continue + updated, total = update_citations_file( + paper_dir, + author_patterns, + alias_patterns, + first_author_last, + ) + total_updated += updated + total_contexts += total + if args.usage_contexts: + upd_usage, tot_usage = update_usage_contexts_file( + paper_dir, + author_patterns, + alias_patterns, + first_author_last, + ) + updated += upd_usage + total += tot_usage + total_updated += upd_usage + total_contexts += tot_usage + if total: + print(f"[OK] {paper_dir.name}: updated {updated} contexts over {total}") + + print( + f"[SUMMARY] total_updated={total_updated} over {total_contexts}; " + f"skipped_incomplete={skipped_incomplete}" + ) + + +if __name__ == "__main__": + main() diff --git a/src/step_03_usage_contexts/build_usage_contexts.py b/src/step_03_usage_contexts/build_usage_contexts.py new file mode 100644 index 0000000000000000000000000000000000000000..a79c8686ffa8ea7c96833b3545dc34c89f775692 --- /dev/null +++ b/src/step_03_usage_contexts/build_usage_contexts.py @@ -0,0 +1,184 @@ +import argparse +import json +from pathlib import Path +from typing import Any, Dict, List, Optional + + +PAPER_META_FILE = "paper_metadata.json" +CITATIONS_FILE = "citations_metadata.json" +DEFAULT_OUT_NAME = "usage_contexts.json" + + +def load_json(path: Path) -> Any | None: + if not path.exists(): + return None + try: + return json.loads(path.read_text(encoding="utf-8")) + except Exception as e: + print(f"[WARN] could not parse JSON at {path}: {e}") + return None + + +def iter_paper_dirs(root: Path) -> List[Path]: + out: List[Path] = [] + for child in root.iterdir(): + if child.is_dir() and (child / PAPER_META_FILE).exists(): + out.append(child) + return out + + +def _extract_contexts(item: Dict[str, Any]) -> List[Dict[str, Any]]: + contexts: List[Dict[str, Any]] = [] + + raw = item.get("contextsWithIntent") or [] + if isinstance(raw, list) and raw: + for entry in raw: + if not isinstance(entry, dict): + continue + text_raw = (entry.get("context") or "").strip() + text = (entry.get("context_with_marker") or text_raw).strip() + intents = entry.get("intents") or [] + contexts.append( + { + "text": text, + "text_raw": text_raw, + "intents": intents, + } + ) + + # Fallback for older schema that only stores raw context strings. + if not contexts: + raw_alt = item.get("contexts") or [] + if isinstance(raw_alt, list): + for text in raw_alt: + if not isinstance(text, str): + continue + text = text.strip() + if text: + contexts.append( + { + "text": text, + "intents": [], + } + ) + + return contexts + + +def build_usage_contexts_for_paper(paper_dir: Path) -> Optional[Dict[str, Any]]: + citations_path = paper_dir / CITATIONS_FILE + data = load_json(citations_path) + if data is None: + return None + + if not isinstance(data, list): + print(f"[WARN] {paper_dir.name}: {CITATIONS_FILE} is not a list") + return None + + citing_entries: List[Dict[str, Any]] = [] + total_contexts = 0 + citing_with_context = 0 + influential_citations = 0 + influential_with_context = 0 + influential_contexts: List[Dict[str, Any]] = [] + + for item in data: + if not isinstance(item, dict): + continue + citing = item.get("citingPaper") or {} + + contexts = _extract_contexts(item) + is_influential = bool(item.get("isInfluential", False)) + if is_influential: + influential_citations += 1 + if contexts: + citing_with_context += 1 + total_contexts += len(contexts) + if is_influential: + influential_with_context += 1 + + citing_entries.append( + { + "citing_paper_id": citing.get("paperId"), + "title": citing.get("title"), + "external_ids": citing.get("externalIds") or {}, + "is_influential": is_influential, + "contexts": contexts, + } + ) + if is_influential and contexts: + influential_contexts.append( + { + "citing_paper_id": citing.get("paperId"), + "title": citing.get("title"), + "external_ids": citing.get("externalIds") or {}, + "contexts": contexts, + } + ) + + payload = { + "paper_id": paper_dir.name, + "total_citations": len(data), + "num_contexts": total_contexts, + "num_citing_with_context": citing_with_context, + "num_citing_without_context": len(data) - citing_with_context, + "num_influential_citations": influential_citations, + "num_influential_with_context": influential_with_context, + "influential_contexts": influential_contexts, + "citing_papers": citing_entries, + } + return payload + + +def run(root: Path, out_name: str, overwrite: bool) -> None: + root = root.resolve() + if not root.exists(): + raise SystemExit(f"Root directory does not exist: {root}") + + paper_dirs = sorted(iter_paper_dirs(root), key=lambda p: p.name) + print(f"[INFO] Found {len(paper_dirs)} paper dirs under {root}") + + for paper_dir in paper_dirs: + out_path = paper_dir / out_name + if out_path.exists() and not overwrite: + print(f"[SKIP] {paper_dir.name}: {out_name} already exists") + continue + payload = build_usage_contexts_for_paper(paper_dir) + if payload is None: + continue + + out_path.write_text(json.dumps(payload, indent=2), encoding="utf-8") + print( + f"[OK] {paper_dir.name}: wrote {out_name} " + f"({payload['num_contexts']} contexts from {payload['total_citations']} citations)" + ) + + +def main() -> None: + parser = argparse.ArgumentParser( + description="Build usage_contexts.json from citations_metadata.json files." + ) + parser.add_argument( + "--root", + type=str, + default="processed_papers/acl_2024", + help="Root directory containing processed_papers/acl_2024/ dirs.", + ) + parser.add_argument( + "--out-name", + type=str, + default=DEFAULT_OUT_NAME, + help="Output filename to write inside each paper dir.", + ) + parser.add_argument( + "--overwrite", + action="store_true", + help="Overwrite existing usage_contexts.json files.", + ) + args = parser.parse_args() + + run(Path(args.root), out_name=args.out_name, overwrite=args.overwrite) + + +if __name__ == "__main__": + main() diff --git a/src/step_04_label_citations/label_citation_functions.py b/src/step_04_label_citations/label_citation_functions.py new file mode 100644 index 0000000000000000000000000000000000000000..9f85da4a53948c006b2bed172dab1a7ab0610910 --- /dev/null +++ b/src/step_04_label_citations/label_citation_functions.py @@ -0,0 +1,373 @@ +import argparse +import json +import sys +from pathlib import Path +from typing import Any, Dict, List + +DEEP_CITATION_ROOT = Path(__file__).resolve().parents[2] / "Deep-Citation" +if not DEEP_CITATION_ROOT.exists(): + raise SystemExit(f"Deep-Citation repo not found at {DEEP_CITATION_ROOT}") + +sys.path.insert(0, str(DEEP_CITATION_ROOT)) + +from data import CollateFn, create_data_channels +from Model import MultiHeadLanguageModel +import torch +from torch.utils.data import DataLoader + + +PAPER_META_FILE = "paper_metadata.json" +USAGE_CONTEXTS_FILE = "usage_contexts.json" +OUT_FILE = "usage_context_labels.json" + +LABEL_SET = [ + "Background", + "Uses", + "Extends", + "CompareOrContrast", + "Motivation", + "Future", +] + + +def load_json(path: Path) -> Any | None: + if not path.exists(): + return None + try: + return json.loads(path.read_text(encoding="utf-8")) + except Exception: + return None + + +def iter_paper_dirs(root: Path) -> List[Path]: + out: List[Path] = [] + for child in root.iterdir(): + if child.is_dir() and (child / PAPER_META_FILE).exists(): + out.append(child) + return out + + +def flatten_contexts(usage: Dict[str, Any]) -> List[Dict[str, Any]]: + contexts: List[Dict[str, Any]] = [] + idx = 1 + for entry in usage.get("citing_papers", []) or []: + if not isinstance(entry, dict): + continue + citing_title = entry.get("title") or "Unknown citing paper" + citing_paper_id = entry.get("citing_paper_id") or "" + for c in entry.get("contexts", []) or []: + if not isinstance(c, dict): + continue + text = (c.get("text") or "").strip() + if not text: + continue + contexts.append( + { + "id": idx, + "text": text, + "citing_title": citing_title, + "citing_paper_id": citing_paper_id, + } + ) + idx += 1 + return contexts + + +def _resolve_model_name(lm: str) -> str: + if lm == "scibert": + return "allenai/scibert_scivocab_uncased" + if lm == "bert": + return "bert-base-uncased" + if lm == "deberta": + return "microsoft/deberta-v3-base" + if lm == "deberta-large": + return "microsoft/deberta-v3-large" + return lm + + +def _infer_head_sizes(state_dict: Dict[str, Any]) -> List[int]: + head_weights = [ + (k, v) for k, v in state_dict.items() if k.startswith("lns.") and k.endswith(".weight") + ] + head_weights.sort(key=lambda x: int(x[0].split(".")[1])) + return [int(weight.shape[0]) for _, weight in head_weights] + + +class _ContextDataset: + def __init__(self, texts: List[str]): + self.texts = texts + + def __len__(self) -> int: + return len(self.texts) + + def __getitem__(self, idx: int): + return (self.texts[idx], torch.tensor(0), torch.tensor(0)) + + +def label_with_model( + contexts: List[Dict[str, Any]], + model_path: Path, + data_dir: Path, + class_definition: Path, + lm: str, + device: str, + batch_size: int, +) -> Dict[int, Dict[str, Any]]: + data_file = data_dir / "acl.tsv" + train_data, _, _, label_names = create_data_channels( + str(data_file), + str(class_definition), + lmbd=1.0, + ) + modelname = _resolve_model_name(lm) + state_dict = torch.load(model_path, map_location=device) + head_sizes = _infer_head_sizes(state_dict) + model = MultiHeadLanguageModel( + modelname=modelname, + device=device, + readout="ch", + num_classes=head_sizes, + ).to(device) + model.load_state_dict(state_dict) + model.eval() + + collate_fn = CollateFn( + modelname=modelname, + class_definitions=train_data.class_definitions, + instance_weights=False, + ) + dataset = _ContextDataset([ctx["text"] for ctx in contexts]) + loader = DataLoader(dataset, batch_size=batch_size, shuffle=False, collate_fn=collate_fn) + + outputs: Dict[int, Dict[str, Any]] = {} + idx_offset = 0 + with torch.no_grad(): + for batched_text, labels, ds_indices, class_tokens, class_ds_indices in loader: + ds_indices = ds_indices.to(device) + class_ds_indices = class_ds_indices.to(device) + logits = model(batched_text, ds_indices, class_tokens, class_ds_indices)[0] + probs = torch.softmax(logits, dim=1) + preds = logits.argmax(dim=1).cpu().tolist() + pred_confidences = probs.max(dim=1).values.cpu().tolist() + top2 = torch.topk(probs, k=2, dim=1).values.cpu() + margins = (top2[:, 0] - top2[:, 1]).tolist() + for i, pred in enumerate(preds): + raw_label = label_names[pred] + outputs[idx_offset + i + 1] = { + "id": idx_offset + i + 1, + "label": raw_label, + "confidence": float(pred_confidences[i]), + "confidence_margin": float(margins[i]), + "cue_span": "", + "rationale": "scibert_model", + } + idx_offset += len(preds) + return outputs + + +def aggregate_citing_labels(labels: List[Dict[str, Any]]) -> List[Dict[str, Any]]: + by_citing: Dict[str, List[Dict[str, Any]]] = {} + for item in labels: + citing_id = item.get("citing_paper_id") or "" + by_citing.setdefault(citing_id, []).append(item) + + aggregated: List[Dict[str, Any]] = [] + for citing_id, items in by_citing.items(): + title = items[0].get("citing_title", "") + labels_set = {it.get("label") for it in items} + + if "Extends" in labels_set: + label = "Extends" + evidence_ids = [it["id"] for it in items if it.get("label") == "Extends"] + elif "Uses" in labels_set: + label = "Uses" + evidence_ids = [it["id"] for it in items if it.get("label") == "Uses"] + elif "CompareOrContrast" in labels_set: + label = "CompareOrContrast" + evidence_ids = [ + it["id"] for it in items if it.get("label") == "CompareOrContrast" + ] + else: + label = "Background" + evidence_ids = [] + + aggregated.append( + { + "citing_paper_id": citing_id, + "citing_title": title, + "label": label, + "evidence_context_ids": evidence_ids, + } + ) + + return aggregated + + +def aggregate_final_label(citing_labels: List[Dict[str, Any]]) -> str: + labels_set = {item.get("label") for item in citing_labels} + if "Extends" in labels_set: + return "Extends" + if "Uses" in labels_set: + return "Uses" + if "CompareOrContrast" in labels_set: + return "CompareOrContrast" + return "Background" + + +def score_for_paper( + paper_dir: Path, + batch_size: int, + overwrite: bool, + model_path: Path, + model_data_dir: Path, + model_class_def: Path, + model_lm: str, + device: str, +) -> str: + usage_path = paper_dir / USAGE_CONTEXTS_FILE + usage = load_json(usage_path) + if not isinstance(usage, dict): + return "missing_usage" + + contexts = flatten_contexts(usage) + if not contexts: + return "empty_contexts" + + out_path = paper_dir / OUT_FILE + if out_path.exists() and not overwrite: + return "skipped" + + labeled = label_with_model( + contexts=contexts, + model_path=model_path, + data_dir=model_data_dir, + class_definition=model_class_def, + lm=model_lm, + device=device, + batch_size=batch_size, + ) + + labels_sorted = [] + for context in contexts: + context_id = context["id"] + item = labeled.get(context_id) + if not item: + item = { + "id": context_id, + "label": "Background", + "confidence": 0.0, + "cue_span": "", + "rationale": "missing label", + } + item = dict(item) + item["citing_paper_id"] = context.get("citing_paper_id", "") + item["citing_title"] = context.get("citing_title", "") + item["text"] = context.get("text", "") + labels_sorted.append(item) + + citing_labels = aggregate_citing_labels(labels_sorted) + payload = { + "paper_id": usage.get("paper_id"), + "num_contexts": len(contexts), + "label_set": LABEL_SET, + "labels": labels_sorted, + "citing_paper_labels": citing_labels, + "final_label": aggregate_final_label(citing_labels), + } + out_path.write_text(json.dumps(payload, indent=2), encoding="utf-8") + return "labeled" + + +def main() -> None: + parser = argparse.ArgumentParser( + description="Label citation functions using a Deep-Citation checkpoint." + ) + parser.add_argument( + "--root", + type=str, + default="runs/processed_papers", + help="Root directory containing processed paper directories.", + ) + parser.add_argument( + "--batch-size", + type=int, + default=32, + help="Batch size for model inference.", + ) + parser.add_argument( + "--overwrite", + action="store_true", + help="Overwrite existing usage_context_labels.json files.", + ) + parser.add_argument( + "--model-path", + type=str, + required=True, + help="Path to Deep-Citation best_model.pt checkpoint.", + ) + parser.add_argument( + "--model-data-dir", + type=str, + default="Deep-Citation/Data", + help="Deep-Citation data directory (for label order).", + ) + parser.add_argument( + "--model-class-def", + type=str, + default="Deep-Citation/Data/class_def.json", + help="Deep-Citation class_def.json path.", + ) + parser.add_argument( + "--model-lm", + type=str, + default="scibert", + help="Model name used for the Deep-Citation checkpoint.", + ) + parser.add_argument( + "--device", + type=str, + default="cuda", + help="Device for model inference (cuda/cpu).", + ) + args = parser.parse_args() + + model_path = Path(args.model_path).expanduser().resolve() + if not model_path.exists(): + raise SystemExit(f"Model path does not exist: {model_path}") + + root = Path(args.root).expanduser().resolve() + if not root.exists(): + raise SystemExit(f"Root directory does not exist: {root}") + + paper_dirs = sorted(iter_paper_dirs(root), key=lambda p: p.name) + print(f"[INFO] Found {len(paper_dirs)} paper dirs under {root}") + + counts = { + "labeled": 0, + "skipped": 0, + "missing_usage": 0, + "empty_contexts": 0, + } + + for paper_dir in paper_dirs: + status = score_for_paper( + paper_dir, + args.batch_size, + args.overwrite, + model_path=model_path, + model_data_dir=Path(args.model_data_dir).expanduser().resolve(), + model_class_def=Path(args.model_class_def).expanduser().resolve(), + model_lm=args.model_lm, + device=args.device, + ) + counts[status] = counts.get(status, 0) + 1 + print(f"[{status.upper()}] {paper_dir.name}") + + print( + "[SUMMARY] labeled={labeled}, skipped={skipped}, missing_usage={missing_usage}, " + "empty_contexts={empty_contexts}".format(**counts) + ) + + +if __name__ == "__main__": + main() diff --git a/src/step_05_verify_uses_extends/prompts.py b/src/step_05_verify_uses_extends/prompts.py new file mode 100644 index 0000000000000000000000000000000000000000..19fa403a7eda8acc0b5b84d1df3a955f97a7dee4 --- /dev/null +++ b/src/step_05_verify_uses_extends/prompts.py @@ -0,0 +1,115 @@ +from typing import Dict, List + + +USES_DEFINITION = ( + "USES: The CITING_PAPER explicitly uses/adopts/evaluates on/includes/relies on " + "a dataset, benchmark, method, tool, or reported results from TARGET_PAPER " + "as part of the CITING_PAPER's own methodology or evaluation." +) + +EXTENDS_DEFINITION = ( + "EXTENDS: The CITING_PAPER explicitly extends/modifies/adapts/builds upon " + "TARGET_PAPER's method/dataset/benchmark/tool." +) + +NOTES_DEFINITION = ( + "NOT USES/EXTENDS: Merely describing what TARGET_PAPER introduces/offers/proposes " + "or listing it among related work or benchmarks (without stating adoption). " + "If no explicit adoption/extension cue is present, label NOT_CONFIRMED." +) + + +FEW_SHOT_USES = [ + "We use the same splits as .", + "The Praat tool was used ( ) .", + "CCGBank ( ) is used to train the model .", + "This design idea was adopted from TANKA ( b ) .", + "Our strategy is based on the approach presented by .", +] + +FEW_SHOT_EXTENDS = [ + "The features can be easily obtained by modifying the TAT extraction algorithm described in ( ) .", + "Our own work ( ) extends the first idea to paraphrase fragment extraction on monolingual parallel and comparable corpora .", + "This article represents an extension of our previous work on unsupervised event coreference resolution ( Bejan et al. 2009 ; ) .", + "This evaluation set-up is an improvement versus the one we previously reported ( ) , in which fixed partitions were used for training , development , and testing .", + "The computational treatment of lexical rules proposed can be seen as an extension to the principled method discussed by Gotz and , 1996 , 1997b ) for encoding the main building block of HPSG grammars -- the implicative constraints -- as a logic program .", +] + +FEW_SHOT_NOT_CONFIRMED = [ + " introduced factored SMT .", + "See ( ) for a discussion .", + "See , among others , ( ) .", + " reported a correlation of r = .69 .", + "See for further discussion .", +] + + +def build_uses_extends_verification_prompt( + target_info: Dict[str, str], + candidates: List[Dict[str, str]], +) -> str: + header = [ + "You are verifying citation function for a TARGET paper inside a citing sentence.", + "Be strict: lists of related work or benchmarks are NOT USES/EXTENDS unless there is an explicit action", + "like \"use\", \"build on\", \"adopt\", \"extend\", \"based on\", \"trained on\", \"evaluate on\", \"implement\".", + "", + "Actor test (CRITICAL for USES/EXTENSION):", + "- Only label USES or EXTENSION if the ACTION is performed by the CITING_PAPER.", + "- The cue_span for USES/EXTENSION must include an explicit citing-paper actor phrase such as:", + " \"we\", \"our\", \"in this work\", \"in this paper\", \"we use\", \"we evaluate\",", + " \"our evaluation includes\", \"we extend\", \"we build on\", \"we adapt\".", + "- If the context says the TARGET_PAPER (or some other paper/system) uses/extends something", + " (e.g., \"TARGET_PAPER uses...\", \"TARGET_PAPER extends...\"),", + " then it is NOT USES/EXTENSION. Label NOT_CONFIRMED.", + "", + "Task: Label each sentence as USES, EXTENDS, or NOT_CONFIRMED.", + "Return JSON only with one entry per input sentence.", + "", + "Definitions:", + f"- {USES_DEFINITION}", + f"- {EXTENDS_DEFINITION}", + f"- {NOTES_DEFINITION}", + "", + "Output rules:", + "- label must be one of: USES, EXTENDS, NOT_CONFIRMED", + "- cue_span: exact substring from the sentence that justifies USES/EXTENDS, else empty", + "- rationale: one short sentence", + "- If cue_span is empty => label must be NOT_CONFIRMED", + "", + "Few-shot examples:", + "USES:", + ] + for ex in FEW_SHOT_USES: + header.append(f"- {ex}") + header.append("EXTENDS:") + for ex in FEW_SHOT_EXTENDS: + header.append(f"- {ex}") + header.append("NOT_CONFIRMED:") + for ex in FEW_SHOT_NOT_CONFIRMED: + header.append(f"- {ex}") + + header.extend( + [ + "", + "TARGET_PAPER:", + f"- title: {target_info.get('title', '')}", + f"- first_author_last: {target_info.get('first_author_last', '')}", + f"- year: {target_info.get('year', '')}", + "", + "CANDIDATES:", + ] + ) + + for item in candidates: + header.extend( + [ + f"ID: {item['id']}", + f"Citing paper: {item.get('citing_title', '')}", + f"Sentence: {item.get('text', '')}", + "", + ] + ) + + header.append("JSON OUTPUT:") + header.append("{\"labels\": [{\"id\": 1, \"label\": \"USES\", \"cue_span\": \"...\", \"rationale\": \"...\"}]}") + return "\n".join(header) diff --git a/src/step_05_verify_uses_extends/schemas.py b/src/step_05_verify_uses_extends/schemas.py new file mode 100644 index 0000000000000000000000000000000000000000..0c05d5122fdc24d960ae76cc8b48be9e67b99e07 --- /dev/null +++ b/src/step_05_verify_uses_extends/schemas.py @@ -0,0 +1,22 @@ +USES_EXTENDS_VERIFICATION_JSON_SCHEMA = { + "type": "object", + "properties": { + "labels": { + "type": "array", + "items": { + "type": "object", + "properties": { + "id": {"type": "integer"}, + "label": { + "type": "string", + "enum": ["USES", "EXTENDS", "NOT_CONFIRMED"], + }, + "cue_span": {"type": "string"}, + "rationale": {"type": "string"}, + }, + "required": ["id", "label", "cue_span", "rationale"], + }, + }, + }, + "required": ["labels"], +} diff --git a/src/step_05_verify_uses_extends/verify_uses_extends.py b/src/step_05_verify_uses_extends/verify_uses_extends.py new file mode 100644 index 0000000000000000000000000000000000000000..41fb9227a959fdc2e81e5ae73112409706f6b39d --- /dev/null +++ b/src/step_05_verify_uses_extends/verify_uses_extends.py @@ -0,0 +1,296 @@ +import argparse +import json +import sys +from pathlib import Path +from typing import Any, Dict, List + +SRC_ROOT = Path(__file__).resolve().parents[1] +if str(SRC_ROOT) not in sys.path: + sys.path.insert(0, str(SRC_ROOT)) + +from common.llm_client import LLMClient + +from prompts import build_uses_extends_verification_prompt +from schemas import USES_EXTENDS_VERIFICATION_JSON_SCHEMA + + +PAPER_META_FILE = "paper_metadata.json" +USAGE_LABELS_FILE = "usage_context_labels.json" +OUT_FILE = "usage_uses_extends_verified.json" + +USE_LABELS = {"Uses", "Extends"} + + +def load_json(path: Path) -> Any | None: + if not path.exists(): + return None + try: + return json.loads(path.read_text(encoding="utf-8")) + except Exception: + return None + + +def iter_paper_dirs(root: Path) -> List[Path]: + out: List[Path] = [] + for child in root.iterdir(): + if child.is_dir() and (child / PAPER_META_FILE).exists(): + out.append(child) + return out + + +def _normalize_author_last(name: str) -> str: + parts = [p for p in (name or "").split() if p.strip()] + return parts[-1] if parts else "" + + +def extract_target_info(meta: Any) -> Dict[str, str]: + if isinstance(meta, list) and meta: + meta = meta[0] + if not isinstance(meta, dict): + return {"title": "", "first_author_last": "", "year": ""} + authors = meta.get("authors") or [] + first_author = authors[0]["name"] if authors else "" + return { + "title": meta.get("title", ""), + "first_author_last": _normalize_author_last(first_author), + "year": str(meta.get("year", "")), + } + + +def verify_candidates( + client: LLMClient, + target_info: Dict[str, str], + candidates: List[Dict[str, Any]], +) -> List[Dict[str, Any]]: + prompt = build_uses_extends_verification_prompt(target_info, candidates) + try: + raw = client.call(prompt, schema=USES_EXTENDS_VERIFICATION_JSON_SCHEMA) + except Exception as exc: + print(f"[WARN] LLM call failed: {exc}. Marking all candidates NOT_CONFIRMED.") + return [ + { + "id": item.get("id"), + "label": "NOT_CONFIRMED", + "cue_span": "", + "rationale": "", + "text": item.get("text", ""), + "citing_paper_id": item.get("citing_paper_id", ""), + "citing_title": item.get("citing_title", ""), + "original_label": item.get("original_label", ""), + } + for item in candidates + ] + data = _parse_llm_json(raw) + if not isinstance(data, dict): + print("[WARN] Failed to parse LLM JSON response; marking all candidates NOT_CONFIRMED.") + return [ + { + "id": item.get("id"), + "label": "NOT_CONFIRMED", + "cue_span": "", + "rationale": "", + "text": item.get("text", ""), + "citing_paper_id": item.get("citing_paper_id", ""), + "citing_title": item.get("citing_title", ""), + "original_label": item.get("original_label", ""), + } + for item in candidates + ] + labels = data.get("labels", []) + by_id = {item.get("id"): item for item in labels if isinstance(item, dict)} + + verified: List[Dict[str, Any]] = [] + for candidate in candidates: + item_id = candidate["id"] + model = by_id.get(item_id, {}) + label = model.get("label", "NOT_CONFIRMED") + cue_span = model.get("cue_span", "") + if not cue_span: + label = "NOT_CONFIRMED" + verified.append( + { + "id": item_id, + "label": label, + "cue_span": cue_span, + "rationale": model.get("rationale", ""), + "text": candidate.get("text", ""), + "citing_paper_id": candidate.get("citing_paper_id", ""), + "citing_title": candidate.get("citing_title", ""), + "original_label": candidate.get("original_label", ""), + } + ) + return verified + + +def _parse_llm_json(raw: str) -> Any | None: + try: + return json.loads(raw) + except json.JSONDecodeError: + pass + + cleaned = raw.strip() + if cleaned.startswith("```"): + cleaned = cleaned.strip("`") + cleaned = cleaned.replace("json", "", 1).strip() + + start = cleaned.find("{") + end = cleaned.rfind("}") + if start == -1 or end == -1 or end <= start: + return None + + snippet = cleaned[start : end + 1] + try: + return json.loads(snippet) + except json.JSONDecodeError: + return None + + +def process_paper( + paper_dir: Path, + client: LLMClient, + k: int, + batch_size: int, + overwrite: bool, + resume: bool, +) -> str: + labels_path = paper_dir / USAGE_LABELS_FILE + payload = load_json(labels_path) + if not isinstance(payload, dict): + return "missing_labels" + + out_path = paper_dir / OUT_FILE + if out_path.exists() and (resume or not overwrite): + return "skipped" + + labels = payload.get("labels", []) + candidates_all = [] + for item in labels: + if item.get("label") in USE_LABELS: + candidates_all.append( + { + "id": item.get("id"), + "text": item.get("text", ""), + "citing_paper_id": item.get("citing_paper_id", ""), + "citing_title": item.get("citing_title", ""), + "original_label": item.get("label"), + "confidence": float(item.get("confidence", 0.0) or 0.0), + } + ) + + if not candidates_all: + result = { + "paper_id": payload.get("paper_id"), + "target": {}, + "candidates_total": 0, + "candidates_considered": 0, + "verified": [], + "confirmed": [], + } + out_path.write_text(json.dumps(result, indent=2), encoding="utf-8") + return "no_candidates" + + # Keep top-k highest-confidence USES/EXTENDS contexts for LLM verification. + # If k <= 0, verify all candidates. + candidates_all = sorted( + candidates_all, + key=lambda x: x.get("confidence", 0.0), + reverse=True, + ) + candidates = candidates_all if k <= 0 else candidates_all[:k] + + target_info = extract_target_info(load_json(paper_dir / PAPER_META_FILE)) + verified: List[Dict[str, Any]] = [] + if batch_size <= 0: + batch_size = 25 + for i in range(0, len(candidates), batch_size): + batch = candidates[i : i + batch_size] + verified.extend(verify_candidates(client, target_info, batch)) + confirmed = [v for v in verified if v["label"] in {"USES", "EXTENDS"}] + if any(item["label"] == "EXTENDS" for item in confirmed): + final_label = "EXTENDS" + elif confirmed: + final_label = "USES" + else: + final_label = "NOT_CONFIRMED" + + result = { + "paper_id": payload.get("paper_id"), + "target": target_info, + "candidates_total": len(candidates_all), + "candidates_considered": len(candidates), + "verification_batch_size": int(batch_size), + "verification_num_batches": (len(candidates) + batch_size - 1) // batch_size if candidates else 0, + "candidates_selected": len(confirmed), + "verified": verified, + "confirmed": confirmed, + "confirmed_extends": sum(1 for x in confirmed if x.get("label") == "EXTENDS"), + "confirmed_uses": sum(1 for x in confirmed if x.get("label") == "USES"), + "final_label": final_label, + } + out_path.write_text(json.dumps(result, indent=2), encoding="utf-8") + return "verified" + + +def main() -> None: + parser = argparse.ArgumentParser( + description="Verify USES/EXTENDS candidates via LLM and select top-K." + ) + parser.add_argument( + "--root", + type=str, + default="runs/processed_papers", + help="Root directory containing processed paper directories.", + ) + parser.add_argument( + "--k", + type=int, + default=0, + help="Verify top-k USES/EXTENDS candidates ranked by classifier confidence (<=0 means all).", + ) + parser.add_argument( + "--batch-size", + type=int, + default=25, + help="Number of candidates per LLM verification batch.", + ) + parser.add_argument( + "--overwrite", + action="store_true", + help="Overwrite existing usage_uses_extends_verified.json files.", + ) + parser.add_argument( + "--resume", + action="store_true", + help="Skip papers with existing output files (even if --overwrite is set).", + ) + args = parser.parse_args() + + root = Path(args.root).expanduser().resolve() + if not root.exists(): + raise SystemExit(f"Root directory does not exist: {root}") + + client = LLMClient() + paper_dirs = sorted(iter_paper_dirs(root), key=lambda p: p.name) + print(f"[INFO] Found {len(paper_dirs)} paper dirs under {root}") + + counts = {"verified": 0, "skipped": 0, "missing_labels": 0, "no_candidates": 0} + for paper_dir in paper_dirs: + status = process_paper( + paper_dir, + client, + args.k, + args.batch_size, + args.overwrite, + args.resume, + ) + counts[status] = counts.get(status, 0) + 1 + print(f"[{status.upper()}] {paper_dir.name}") + + print( + "[SUMMARY] verified={verified}, skipped={skipped}, missing_labels={missing_labels}, " + "no_candidates={no_candidates}".format(**counts) + ) + + +if __name__ == "__main__": + main() diff --git a/src/step_06_extract_paragraphs/extract_arxiv_paragraphs.py b/src/step_06_extract_paragraphs/extract_arxiv_paragraphs.py new file mode 100644 index 0000000000000000000000000000000000000000..e808734cbf457afdff96b2acb41fdf2640635956 --- /dev/null +++ b/src/step_06_extract_paragraphs/extract_arxiv_paragraphs.py @@ -0,0 +1,488 @@ +import argparse +import json +import random +import re +import sys +import tarfile +import tempfile +import time +import urllib.request +from pathlib import Path +from typing import Any, Dict, List, Optional, Tuple +import os + +SRC_ROOT = Path(__file__).resolve().parents[1] +if str(SRC_ROOT) not in sys.path: + sys.path.insert(0, str(SRC_ROOT)) + + +PAPER_META_FILE = "paper_metadata.json" +USAGE_CONTEXTS_FILE = "usage_contexts.json" +VERIFIED_FILE = "usage_uses_extends_verified.json" +OUT_FILE = "usage_citing_paragraphs.json" + + +def load_json(path: Path) -> Any | None: + if not path.exists(): + return None + try: + return json.loads(path.read_text(encoding="utf-8")) + except Exception: + return None + + +def iter_paper_dirs(root: Path) -> List[Path]: + out: List[Path] = [] + for child in root.iterdir(): + if child.is_dir() and (child / PAPER_META_FILE).exists(): + out.append(child) + return out + + +def safe_extract(tar: tarfile.TarFile, path: Path) -> None: + for member in tar.getmembers(): + member_path = path / member.name + if not str(member_path.resolve()).startswith(str(path.resolve())): + raise RuntimeError(f"Blocked path traversal in tar: {member.name}") + tar.extractall(path) + + +_ARXIV_LAST_TS = 0.0 + + +def _arxiv_min_interval_sleep() -> None: + """Global throttle to avoid arXiv API rate limits.""" + global _ARXIV_LAST_TS + min_interval = float(os.getenv("ARXIV_MIN_INTERVAL", "1.0")) + now = time.monotonic() + elapsed = now - _ARXIV_LAST_TS + if elapsed < min_interval: + time.sleep(min_interval - elapsed) + _ARXIV_LAST_TS = time.monotonic() + + +def download_arxiv_source(arxiv_id: str, tmpdir: Path) -> Optional[Path]: + url = f"https://arxiv.org/e-print/{arxiv_id}" + archive_path = tmpdir / f"{arxiv_id.replace('/', '_')}.tar" + max_retries = int(os.getenv("ARXIV_MAX_RETRIES", "6")) + base_sleep = float(os.getenv("ARXIV_BASE_SLEEP", "2.0")) + max_sleep = float(os.getenv("ARXIV_MAX_BACKOFF", "60")) + + for attempt in range(max_retries): + try: + _arxiv_min_interval_sleep() + urllib.request.urlretrieve(url, archive_path) # noqa: S310 + try: + with tarfile.open(archive_path) as tar: + safe_extract(tar, tmpdir) + return tmpdir + except tarfile.ReadError as exc: + print(f"[WARN] Invalid arXiv archive for {arxiv_id}: {exc}") + return None + except Exception as exc: + # arXiv sometimes returns 429; treat any network error as retryable. + sleep = min(base_sleep * (2 ** attempt), max_sleep) + random.uniform(0.0, 0.5) + print(f"[WARN] Failed to download arXiv source for {arxiv_id}: {exc}") + print(f"[WARN] arXiv download retrying in {sleep:.2f}s") + time.sleep(sleep) + continue + + print(f"[ERROR] Giving up after {max_retries} attempts for arXiv {arxiv_id}") + return None + + +def find_main_tex(root: Path) -> Optional[Path]: + tex_files = list(root.rglob("*.tex")) + if not tex_files: + return None + + candidates: List[Tuple[int, Path]] = [] + for path in tex_files: + try: + text = path.read_text(encoding="utf-8", errors="ignore") + except Exception: + continue + score = 0 + if "\\begin{document}" in text: + score += 3 + if "\\documentclass" in text: + score += 2 + score += len(text) // 1000 + candidates.append((score, path)) + + candidates.sort(key=lambda x: x[0], reverse=True) + return candidates[0][1] if candidates else None + + +def read_bib_files(root: Path) -> Dict[str, str]: + bibs: Dict[str, str] = {} + for path in root.rglob("*.bib"): + try: + bibs[str(path.relative_to(root))] = path.read_text(encoding="utf-8", errors="ignore") + except Exception: + continue + return bibs + + +def normalize_text(text: str) -> str: + text = re.sub(r"[^a-z0-9\s]", " ", text.lower()) + return re.sub(r"\s+", " ", text).strip() + + +def tokenize(text: str) -> List[str]: + return [t for t in normalize_text(text).split() if t] + + +def paragraphize(text: str) -> List[str]: + text = text.replace("\r\n", "\n") + text = re.sub(r"\n\s*\n", "\n\n", text) + paragraphs = [p.strip() for p in re.split(r"\n\s*\n", text) if p.strip()] + return paragraphs + + +def strip_latex_comments(text: str) -> str: + # Remove explicit comment environments first. + text = re.sub(r"\\begin\{comment\}.*?\\end\{comment\}", "", text, flags=re.DOTALL) + + cleaned_lines: List[str] = [] + for line in text.splitlines(): + out_chars: List[str] = [] + i = 0 + while i < len(line): + ch = line[i] + if ch == "%": + # Keep escaped percent (\%) and continue parsing. + if i > 0 and line[i - 1] == "\\": + out_chars.append(ch) + i += 1 + continue + # Unescaped percent starts a LaTeX comment; ignore rest of the line. + break + out_chars.append(ch) + i += 1 + cleaned_lines.append("".join(out_chars)) + return "\n".join(cleaned_lines) + + +def parse_bib_entries(bib_text: str) -> List[Dict[str, str]]: + entries: List[Dict[str, str]] = [] + matches = list(re.finditer(r"@[\w]+\s*\{\s*([^,]+),", bib_text)) + for i, match in enumerate(matches): + key = match.group(1).strip() + start = match.end() + end = matches[i + 1].start() if i + 1 < len(matches) else len(bib_text) + body = bib_text[start:end] + fields = {} + for f_match in re.finditer(r"(\w+)\s*=\s*[{|\"](.+?)[}|\"]\s*,", body, re.DOTALL): + fields[f_match.group(1).lower()] = f_match.group(2).strip() + entries.append({"key": key, **fields}) + return entries + + +def find_target_bib_keys( + bib_texts: Dict[str, str], + target_info: Dict[str, str], +) -> List[str]: + target_title = normalize_text(target_info.get("title", "")) + target_author = normalize_text(target_info.get("first_author_last", "")) + target_year = target_info.get("year", "") + if not target_title and not target_author: + return [] + + keys: List[str] = [] + for bib_text in bib_texts.values(): + for entry in parse_bib_entries(bib_text): + title = normalize_text(entry.get("title", "")) + author = normalize_text(entry.get("author", "")) + year = str(entry.get("year", "")) + has_title = bool(title) + title_match = target_title and (target_title in title or title in target_title) + author_match = target_author and target_author in author + year_match = target_year and target_year in year + + if title_match and author_match: + keys.append(entry["key"]) + elif not has_title and author_match and year_match: + keys.append(entry["key"]) + elif author_match and year_match: + keys.append(entry["key"]) + return keys + + +def replace_target_citations(text: str, target_keys: List[str], target_info: Dict[str, str]) -> str: + key_set = set(target_keys or []) + author = target_info.get("first_author_last", "").lower() + year = target_info.get("year", "") + alt_years = {year} + if year.isdigit(): + alt_years.add(str(int(year) - 1)) + alt_years.add(str(int(year) + 1)) + + def repl(match: re.Match) -> str: + keys = [k.strip() for k in match.group(1).split(",")] + for key in keys: + if key in key_set: + return "" + key_lc = key.lower() + if author and author in key_lc and any(y in key_lc for y in alt_years if y): + return "" + return match.group(0) + + return re.sub(r"\\cite[a-zA-Z]*\s*\{([^}]+)\}", repl, text) + + +def match_paragraphs( + paragraphs: List[str], + contexts: List[Dict[str, str]], +) -> List[Dict[str, Any]]: + results: List[Dict[str, Any]] = [] + para_tokens = [set(tokenize(p)) for p in paragraphs] + + for idx, ctx in enumerate(contexts, start=1): + ctx_text = ctx.get("text", "") + ctx_tokens = set(tokenize(ctx_text)) + if not ctx_tokens: + continue + best = None + best_score = 0.0 + for p_idx, tokens in enumerate(para_tokens): + if not tokens: + continue + overlap = len(ctx_tokens & tokens) / max(1, len(ctx_tokens)) + if overlap > best_score: + best = p_idx + best_score = overlap + if best is not None and best_score >= 0.5: + paragraph = paragraphs[best] + results.append( + { + "context_id": idx, + "context": ctx_text, + "context_with_marker": ctx.get("text_with_marker", ctx_text), + "paragraph": paragraph, + "overlap": round(best_score, 3), + } + ) + return results + + +def _normalize_text(text: str) -> str: + return " ".join(text.split()).strip().lower() + + +def _normalize_for_match(text: str) -> str: + text = text.replace("", "") + text = re.sub(r"\[[^\]]+\]", "", text) + return _normalize_text(text) + + +def _normalize_author_last(name: str) -> str: + parts = [p for p in (name or "").split() if p.strip()] + return parts[-1] if parts else "" + + +def extract_target_info(meta: Any) -> Dict[str, str]: + if isinstance(meta, list) and meta: + meta = meta[0] + if not isinstance(meta, dict): + return {"title": "", "first_author_last": "", "year": ""} + authors = meta.get("authors") or [] + first_author = authors[0]["name"] if authors else "" + return { + "title": meta.get("title", ""), + "first_author_last": _normalize_author_last(first_author), + "year": str(meta.get("year", "")), + } + + +def build_citing_contexts_map( + usage: Dict[str, Any], + confirmed_texts_by_citing: Dict[str, set] | None, +) -> Dict[str, Dict[str, Any]]: + citing_map: Dict[str, Dict[str, Any]] = {} + for entry in usage.get("citing_papers", []) or []: + if not isinstance(entry, dict): + continue + citing_id = entry.get("citing_paper_id") or "" + allowed_texts = confirmed_texts_by_citing.get(citing_id) if confirmed_texts_by_citing else None + allowed_norms = ( + {_normalize_for_match(text) for text in allowed_texts} if allowed_texts else None + ) + contexts = [] + seen = set() + for c in entry.get("contexts", []) or []: + if not isinstance(c, dict): + continue + text_raw = (c.get("text") or "").strip() + text_with_marker = (c.get("context_with_marker") or text_raw).strip() + if not text_raw: + continue + norm = _normalize_for_match(text_raw) + if allowed_norms is not None and norm not in allowed_norms: + continue + if norm in seen: + continue + seen.add(norm) + contexts.append({"text": text_raw, "text_with_marker": text_with_marker}) + if allowed_texts is not None and not contexts: + for text in allowed_texts: + norm = _normalize_for_match(text) + if norm in seen: + continue + seen.add(norm) + contexts.append({"text": text, "text_with_marker": text}) + citing_map[citing_id] = { + "title": entry.get("title", ""), + "paper_id": citing_id, + "arxiv_id": (entry.get("external_ids") or {}).get("ArXiv", ""), + "contexts": contexts, + } + return citing_map + + +def process_citing_paper(citing: Dict[str, Any]) -> Dict[str, Any]: + target_info = citing.get("target_info", {}) + arxiv_id = citing.get("arxiv_id", "") + if not arxiv_id: + return {"error": "missing_arxiv_id", **citing} + + with tempfile.TemporaryDirectory() as tmp: + tmpdir = Path(tmp) + if not download_arxiv_source(arxiv_id, tmpdir): + return {"error": "bad_arxiv_archive", **citing} + main_tex = find_main_tex(tmpdir) + if not main_tex: + return {"error": "missing_main_tex", **citing} + + tex_text = main_tex.read_text(encoding="utf-8", errors="ignore") + tex_text = strip_latex_comments(tex_text) + bibs = read_bib_files(tmpdir) + target_keys = find_target_bib_keys(bibs, target_info) + tex_text = replace_target_citations(tex_text, target_keys, target_info) + paragraphs = paragraphize(tex_text) + target_citing_paragraphs = [p for p in paragraphs if "" in p] + matched = match_paragraphs(paragraphs, citing.get("contexts", [])) + + return { + "citing_paper_id": citing.get("paper_id", ""), + "citing_title": citing.get("title", ""), + "arxiv_id": arxiv_id, + "main_tex_file": str(main_tex.relative_to(tmpdir)), + "bib_files": list(bibs.keys()), + "bib_texts": bibs, + "target_bib_keys": target_keys, + "contexts": citing.get("contexts", []), + "target_citing_paragraphs": target_citing_paragraphs, + "matched_paragraphs": matched, + } + + +def process_paper(root: Path, overwrite: bool, include_all: bool, resume: bool) -> str: + usage = load_json(root / USAGE_CONTEXTS_FILE) + if not isinstance(usage, dict): + return "missing_usage" + + out_path = root / OUT_FILE + if out_path.exists() and (resume or not overwrite): + return "skipped" + + verified = None + confirmed_texts_by_citing: Dict[str, set] = {} + if not include_all: + verified = load_json(root / VERIFIED_FILE) + if not isinstance(verified, dict): + return "missing_verified" + for item in verified.get("confirmed", []) or []: + citing_id = item.get("citing_paper_id") or "" + text = item.get("text") or "" + if not citing_id or not text: + continue + confirmed_texts_by_citing.setdefault(citing_id, set()).add(text) + + target_info = extract_target_info(load_json(root / PAPER_META_FILE)) + citing_map = build_citing_contexts_map( + usage, + confirmed_texts_by_citing if confirmed_texts_by_citing else None, + ) + if not citing_map: + out_path.write_text( + json.dumps({"paper_id": usage.get("paper_id"), "citing_papers": []}, indent=2), + encoding="utf-8", + ) + return "empty_citing" + + confirmed_ids: Optional[set] = None + if not include_all and isinstance(verified, dict): + confirmed = verified.get("confirmed", []) + confirmed_ids = { + item.get("citing_paper_id") + for item in confirmed + if item.get("citing_paper_id") + } + + citing_papers = [] + for citing_id, citing in citing_map.items(): + if confirmed_ids is not None and citing_id not in confirmed_ids: + continue + citing["target_info"] = target_info + citing_papers.append(process_citing_paper(citing)) + + payload = {"paper_id": usage.get("paper_id"), "citing_papers": citing_papers} + out_path.write_text(json.dumps(payload, indent=2), encoding="utf-8") + return "processed" + + +def main() -> None: + parser = argparse.ArgumentParser( + description="Download arXiv sources and extract citation-local paragraphs." + ) + parser.add_argument( + "--root", + type=str, + default="runs/processed_papers", + help="Root directory containing processed paper directories.", + ) + parser.add_argument( + "--overwrite", + action="store_true", + help="Overwrite existing usage_citing_paragraphs.json files.", + ) + parser.add_argument( + "--all", + action="store_true", + help="Process all citing papers (not just confirmed USES/EXTENDS).", + ) + parser.add_argument( + "--resume", + action="store_true", + help="Skip papers with existing output files (even if --overwrite is set).", + ) + args = parser.parse_args() + + root = Path(args.root).expanduser().resolve() + if not root.exists(): + raise SystemExit(f"Root directory does not exist: {root}") + + paper_dirs = sorted(iter_paper_dirs(root), key=lambda p: p.name) + print(f"[INFO] Found {len(paper_dirs)} paper dirs under {root}") + + counts = { + "processed": 0, + "skipped": 0, + "missing_usage": 0, + "missing_verified": 0, + "empty_citing": 0, + } + for paper_dir in paper_dirs: + status = process_paper(paper_dir, args.overwrite, args.all, args.resume) + counts[status] = counts.get(status, 0) + 1 + print(f"[{status.upper()}] {paper_dir.name}") + + print( + "[SUMMARY] processed={processed}, skipped={skipped}, missing_usage={missing_usage}, " + "missing_verified={missing_verified}, empty_citing={empty_citing}".format(**counts) + ) + + +if __name__ == "__main__": + main() diff --git a/src/step_07_extract_and_refine/extract_contributions_from_citations.py b/src/step_07_extract_and_refine/extract_contributions_from_citations.py new file mode 100644 index 0000000000000000000000000000000000000000..d53fd39ab7d0fdc21a3bf570472deb9e1e1a8e30 --- /dev/null +++ b/src/step_07_extract_and_refine/extract_contributions_from_citations.py @@ -0,0 +1,329 @@ +import argparse +import json +import sys +from pathlib import Path +from typing import Any, Dict, List + +SRC_ROOT = Path(__file__).resolve().parents[1] +if str(SRC_ROOT) not in sys.path: + sys.path.insert(0, str(SRC_ROOT)) + +from common.llm_client import LLMClient + +from prompts import build_contribution_prompt +from schemas import CONTRIBUTION_JSON_SCHEMA + + +PAPER_META_FILE = "paper_metadata.json" +USAGE_CONTEXTS_FILE = "usage_contexts.json" +ARXIV_PARAGRAPHS_FILE = "usage_citing_paragraphs.json" +VERIFIED_FILE = "usage_uses_extends_verified.json" +OUT_FILE = "usage_contributions.json" + + +def load_json(path: Path) -> Any | None: + if not path.exists(): + return None + try: + return json.loads(path.read_text(encoding="utf-8")) + except Exception: + return None + + +def iter_paper_dirs(root: Path) -> List[Path]: + out: List[Path] = [] + for child in root.iterdir(): + if child.is_dir() and (child / PAPER_META_FILE).exists(): + out.append(child) + return out + + +def _normalize_author_last(name: str) -> str: + parts = [p for p in (name or "").split() if p.strip()] + return parts[-1] if parts else "" + + +def extract_target_info(meta: Any) -> Dict[str, str]: + if isinstance(meta, list) and meta: + meta = meta[0] + if not isinstance(meta, dict): + return { + "title": "", + "first_author_last": "", + "year": "", + "tldr": "", + "abstract": "", + } + authors = meta.get("authors") or [] + first_author = authors[0]["name"] if authors else "" + tldr = "" + tldr_obj = meta.get("tldr") + if isinstance(tldr_obj, dict): + tldr = tldr_obj.get("text", "") + return { + "title": meta.get("title", ""), + "first_author_last": _normalize_author_last(first_author), + "year": str(meta.get("year", "")), + "tldr": tldr, + "abstract": meta.get("abstract", ""), + } + + +def build_citing_contexts_map_from_paragraphs( + arxiv_data: Dict[str, Any], +) -> Dict[str, Dict[str, Any]]: + citing_map: Dict[str, Dict[str, Any]] = {} + for entry in arxiv_data.get("citing_papers", []) or []: + if not isinstance(entry, dict): + continue + citing_id = entry.get("citing_paper_id") or "" + contexts = [] + seen = set() + for paragraph in entry.get("target_citing_paragraphs", []) or []: + paragraph = (paragraph or "").strip() + if not paragraph: + continue + combined = f"Target-citing paragraph: {paragraph}" + norm = " ".join(combined.split()).lower() + if norm in seen: + continue + seen.add(norm) + contexts.append(combined) + citing_map[citing_id] = { + "title": entry.get("citing_title", ""), + "paper_id": citing_id, + "contexts": contexts, + "source": "arxiv_paragraphs", + } + return citing_map + + +def build_citing_contexts_map_from_usage( + usage: Dict[str, Any], + confirmed_texts_by_citing: Dict[str, set] | None, +) -> Dict[str, Dict[str, Any]]: + citing_map: Dict[str, Dict[str, Any]] = {} + for entry in usage.get("citing_papers", []) or []: + if not isinstance(entry, dict): + continue + citing_id = entry.get("citing_paper_id") or "" + allowed_texts = confirmed_texts_by_citing.get(citing_id) if confirmed_texts_by_citing else None + contexts = [] + seen = set() + for c in entry.get("contexts", []) or []: + if not isinstance(c, dict): + continue + text = (c.get("context_with_marker") or c.get("text") or "").strip() + if not text: + continue + if allowed_texts is not None and text not in allowed_texts: + continue + norm = " ".join(text.split()).lower() + if norm in seen: + continue + seen.add(norm) + contexts.append(f"Target sentence: {text}") + citing_map[citing_id] = { + "title": entry.get("title", ""), + "paper_id": citing_id, + "contexts": contexts, + "source": "usage_contexts_fallback", + } + return citing_map + + +def extract_contribution( + client: LLMClient, + target_info: Dict[str, str], + citing_info: Dict[str, Any], +) -> Dict[str, Any]: + contexts = citing_info.get("contexts", []) + prompt = build_contribution_prompt(target_info, citing_info, contexts) + raw = client.call(prompt, schema=CONTRIBUTION_JSON_SCHEMA) + data = _parse_llm_json(raw) + if not isinstance(data, dict): + return { + "citing_paper_id": citing_info.get("paper_id", ""), + "citing_title": citing_info.get("title", ""), + "label": "NOT_CONFIRMED", + "paper_claim": "", + "claim": "", + "cluster_title": "", + "cluster_key": "", + "evidence_span": "", + "rationale": "", + "contexts": contexts, + "source": citing_info.get("source", "unknown"), + } + label = data.get("label", "NOT_CONFIRMED") + paper_claim = data.get("paper_claim", "") or data.get("claim", "") + cluster_title = data.get("cluster_title", "") or data.get("cluster_claim", "") + cluster_key = data.get("cluster_key", "") + evidence_span = data.get("evidence_span", "") + if not evidence_span: + label = "NOT_CONFIRMED" + paper_claim = "" + cluster_title = "" + cluster_key = "" + if label in {"USES", "EXTENDS"} and not cluster_title: + cluster_title = paper_claim + if label in {"USES", "EXTENDS"} and not cluster_key: + cluster_key = f"{label}|contribution|unspecified" + return { + "citing_paper_id": citing_info.get("paper_id", ""), + "citing_title": citing_info.get("title", ""), + "label": label, + "paper_claim": paper_claim, + "claim": paper_claim, + "cluster_title": cluster_title, + "cluster_key": cluster_key, + "evidence_span": evidence_span, + "rationale": data.get("rationale", ""), + "contexts": contexts, + "source": citing_info.get("source", "unknown"), + } + + +def _parse_llm_json(raw: str) -> Any | None: + try: + return json.loads(raw) + except json.JSONDecodeError: + pass + + cleaned = raw.strip() + if cleaned.startswith("```"): + cleaned = cleaned.strip("`") + cleaned = cleaned.replace("json", "", 1).strip() + + start = cleaned.find("{") + end = cleaned.rfind("}") + if start == -1 or end == -1 or end <= start: + return None + + snippet = cleaned[start : end + 1] + try: + return json.loads(snippet) + except json.JSONDecodeError: + return None + + +def process_paper( + paper_dir: Path, + client: LLMClient, + overwrite: bool, + resume: bool, +) -> str: + verified = load_json(paper_dir / VERIFIED_FILE) + if not isinstance(verified, dict): + return "missing_verified" + out_path = paper_dir / OUT_FILE + if out_path.exists() and (resume or not overwrite): + return "skipped" + + if verified.get("final_label") == "NOT_CONFIRMED": + payload = { + "paper_id": verified.get("paper_id"), + "final_label": "NOT_CONFIRMED", + "contributions": [], + } + out_path.write_text(json.dumps(payload, indent=2), encoding="utf-8") + return "no_confirmed" + + arxiv_data = load_json(paper_dir / ARXIV_PARAGRAPHS_FILE) + if not isinstance(arxiv_data, dict): + return "missing_arxiv_paragraphs" + + target_info = extract_target_info(load_json(paper_dir / PAPER_META_FILE)) + citing_map = build_citing_contexts_map_from_paragraphs(arxiv_data) + usage = load_json(paper_dir / USAGE_CONTEXTS_FILE) + confirmed_texts_by_citing: Dict[str, set] = {} + for item in verified.get("confirmed", []) or []: + citing_id = item.get("citing_paper_id") or "" + text = item.get("text") or "" + if not citing_id or not text: + continue + confirmed_texts_by_citing.setdefault(citing_id, set()).add(text) + usage_map = ( + build_citing_contexts_map_from_usage(usage, confirmed_texts_by_citing) + if isinstance(usage, dict) + else {} + ) + + confirmed = verified.get("confirmed", []) + confirmed_ids = {item.get("citing_paper_id") for item in confirmed if item.get("citing_paper_id")} + contributions: List[Dict[str, Any]] = [] + fallback_citing_ids: List[str] = [] + for citing_id in confirmed_ids: + citing_info = citing_map.get(citing_id) + if citing_info and not citing_info.get("contexts"): + citing_info = None + if not citing_info: + fallback = usage_map.get(citing_id) + if fallback and fallback.get("contexts"): + citing_info = fallback + fallback_citing_ids.append(citing_id) + else: + continue + contributions.append(extract_contribution(client, target_info, citing_info)) + + payload = { + "paper_id": verified.get("paper_id"), + "final_label": verified.get("final_label"), + "contributions": contributions, + "source": "arxiv_paragraphs", + "fallback_citing_ids": fallback_citing_ids, + } + out_path.write_text(json.dumps(payload, indent=2), encoding="utf-8") + return "labeled" + + +def main() -> None: + parser = argparse.ArgumentParser( + description="Extract per-citing-paper contribution claims from verified USES/EXTENDS." + ) + parser.add_argument( + "--root", + type=str, + default="runs/processed_papers", + help="Root directory containing processed paper directories.", + ) + parser.add_argument( + "--overwrite", + action="store_true", + help="Overwrite existing usage_contributions.json files.", + ) + parser.add_argument( + "--resume", + action="store_true", + help="Skip papers with existing output files (even if --overwrite is set).", + ) + args = parser.parse_args() + + root = Path(args.root).expanduser().resolve() + if not root.exists(): + raise SystemExit(f"Root directory does not exist: {root}") + + client = LLMClient() + paper_dirs = sorted(iter_paper_dirs(root), key=lambda p: p.name) + print(f"[INFO] Found {len(paper_dirs)} paper dirs under {root}") + + counts = { + "labeled": 0, + "skipped": 0, + "missing_verified": 0, + "missing_arxiv_paragraphs": 0, + "no_confirmed": 0, + } + for paper_dir in paper_dirs: + status = process_paper(paper_dir, client, args.overwrite, args.resume) + counts[status] = counts.get(status, 0) + 1 + print(f"[{status.upper()}] {paper_dir.name}") + + print( + "[SUMMARY] labeled={labeled}, skipped={skipped}, missing_verified={missing_verified}, " + "missing_arxiv_paragraphs={missing_arxiv_paragraphs}, no_confirmed={no_confirmed}".format(**counts) + ) + + +if __name__ == "__main__": + main() diff --git a/src/step_07_extract_and_refine/prompts.py b/src/step_07_extract_and_refine/prompts.py new file mode 100644 index 0000000000000000000000000000000000000000..604d6f3f8df65f51da856a7a7c88a2626d930495 --- /dev/null +++ b/src/step_07_extract_and_refine/prompts.py @@ -0,0 +1,65 @@ +from typing import Dict, List + + +def build_contribution_prompt( + target_info: Dict[str, str], + citing_info: Dict[str, str], + contexts: List[str], +) -> str: + header = [ + "You are extracting how a citing paper uses or extends a target paper.", + "Read the paragraph(s) below and write ONE concise contribution claim.", + "Focus only on what the citing paper actually does with the target paper.", + "", + "Rules:", + "- If the citing paper explicitly uses/adopts/evaluates on the target paper's method/data/benchmark, label USES.", + "- If it explicitly extends/modifies/adapts/builds upon the target paper, label EXTENDS.", + "- If the paragraph is only descriptive/background or only compares/mentions the target paper, return label NOT_CONFIRMED and empty fields.", + "- Do not output comparison-only claims (e.g., 'compares to '); those are NOT_CONFIRMED.", + "- Output paper_claim: one concise, paper-specific contribution claim.", + "- Output cluster_title: concise natural-language cluster summary (6-14 words), generic across papers.", + "- Also output cluster_key in this exact format: RELATION|artifact|purpose", + "- cluster_key must be generic and reusable across papers.", + "- artifact and purpose must be short snake_case phrases (e.g., dataset, evaluation_protocol, evaluation).", + "- cluster_key RELATION must exactly match label.", + "- Avoid overly specific keys (no paper names, no model/version numbers, no citation keys).", + "- Prefer stable generic keys such as: USES|dataset|evaluation, EXTENDS|dataset|dataset_creation, USES|evaluation_protocol|evaluation.", + "- If label is NOT_CONFIRMED, paper_claim, cluster_title, cluster_key, and evidence_span must be empty.", + "- The evidence_span must be a verbatim substring from the provided contexts.", + "- The TARGET_PAPER abstract/TLDR is for background only; do not use it as evidence.", + "", + "Negative example (NOT_CONFIRMED):", + "Paragraph: \"We compare our method to and other baselines.\"", + "Output: {\"label\":\"NOT_CONFIRMED\",\"paper_claim\":\"\",\"cluster_title\":\"\",\"cluster_key\":\"\",\"evidence_span\":\"\",\"rationale\":\"Comparison only.\"}", + "", + "Return JSON only.", + "", + "TARGET_PAPER:", + f"- title: {target_info.get('title', '')}", + f"- first_author_last: {target_info.get('first_author_last', '')}", + f"- year: {target_info.get('year', '')}", + f"- tldr: {target_info.get('tldr', '')}", + f"- abstract: {target_info.get('abstract', '')}", + "", + "CITING_PAPER:", + f"- title: {citing_info.get('title', '')}", + f"- paper_id: {citing_info.get('paper_id', '')}", + "", + "CONTEXTS (verbatim, same order as extracted):", + ] + for i, text in enumerate(contexts, start=1): + header.append(f"({i}) {text}") + + header.append("") + header.append("JSON OUTPUT:") + header.append( + "{" + "\"label\":\"USES\"," + "\"paper_claim\":\"...\"," + "\"cluster_title\":\"Uses target dataset for evaluation\"," + "\"cluster_key\":\"USES|dataset|evaluation\"," + "\"evidence_span\":\"...\"," + "\"rationale\":\"...\"" + "}" + ) + return "\n".join(header) diff --git a/src/step_07_extract_and_refine/refine_and_filter_clusters_llm.py b/src/step_07_extract_and_refine/refine_and_filter_clusters_llm.py new file mode 100644 index 0000000000000000000000000000000000000000..bcacac3c8620c8b9ef00aa2bf6e0edf7ded9dc92 --- /dev/null +++ b/src/step_07_extract_and_refine/refine_and_filter_clusters_llm.py @@ -0,0 +1,402 @@ +import argparse +import json +import sys +from pathlib import Path +from typing import Any, Dict, List, Tuple + +SRC_ROOT = Path(__file__).resolve().parents[1] +if str(SRC_ROOT) not in sys.path: + sys.path.insert(0, str(SRC_ROOT)) + +from common.llm_client import LLMClient + +PAPER_META_FILE = "paper_metadata.json" +CONTRIB_FILE = "usage_contributions.json" +DISCOVERY_FILE = "usage_discovery_from_contributions.json" +OUT_FILE = "usage_discovery_from_contributions_refined.json" + +REFINE_SCHEMA = { + "type": "object", + "properties": { + "kept_groups": { + "type": "array", + "items": { + "type": "object", + "properties": { + "cluster_ids": {"type": "array", "items": {"type": "string"}}, + "merged_title": {"type": "string"}, + "merged_key": {"type": "string"}, + "rationale": {"type": "string"}, + }, + "required": ["cluster_ids", "merged_title", "merged_key", "rationale"], + }, + }, + "dropped_clusters": { + "type": "array", + "items": { + "type": "object", + "properties": { + "cluster_id": {"type": "string"}, + "reason": {"type": "string"}, + }, + "required": ["cluster_id", "reason"], + }, + }, + }, + "required": ["kept_groups", "dropped_clusters"], +} + + +def load_json(path: Path) -> Any | None: + if not path.exists(): + return None + try: + return json.loads(path.read_text(encoding="utf-8")) + except Exception: + return None + + +def iter_paper_dirs(root: Path) -> List[Path]: + return sorted([p for p in root.iterdir() if p.is_dir() and (p / PAPER_META_FILE).exists()], key=lambda p: p.name) + + +def _to_int_indices(raw_indices: List[Any]) -> List[int]: + out: List[int] = [] + for i in raw_indices or []: + try: + out.append(int(i)) + except Exception: + continue + return out + + +def _parse_key(key: str) -> Tuple[str, str, str]: + parts = [p.strip() for p in str(key or "").split("|")] + if len(parts) >= 3: + return parts[0].upper(), parts[1], parts[2] + return "", "", "" + + +def _dominant_key(member_clusters: List[Dict[str, Any]]) -> str: + rel_count: Dict[str, int] = {} + art_count: Dict[str, int] = {} + pur_count: Dict[str, int] = {} + for c in member_clusters: + rel, art, pur = _parse_key(c.get("cluster_key", "")) + if rel: + rel_count[rel] = rel_count.get(rel, 0) + 1 + if art: + art_count[art] = art_count.get(art, 0) + 1 + if pur: + pur_count[pur] = pur_count.get(pur, 0) + 1 + rel = max(rel_count, key=rel_count.get) if rel_count else "USES" + art = max(art_count, key=art_count.get) if art_count else "contribution" + pur = max(pur_count, key=pur_count.get) if pur_count else "unspecified" + return f"{rel}|{art}|{pur}" + + +def _extract_title(meta: Any) -> str: + if isinstance(meta, list) and meta: + meta = meta[0] + if not isinstance(meta, dict): + return "" + return str(meta.get("title", "")) + + +def _title_from_cluster_key(cluster_key: str) -> str: + parts = [p.strip() for p in str(cluster_key or "").split("|")] + if len(parts) >= 3: + relation, artifact, purpose = parts[0], parts[1], parts[2] + relation_txt = "Uses" if relation.upper() == "USES" else "Extends" + artifact_txt = artifact.replace("_", " ") + purpose_txt = purpose.replace("_", " ") + return f"{relation_txt} {artifact_txt} for {purpose_txt}".strip() + return cluster_key or "" + + +def _cluster_by_exact_keys(keys: List[str]) -> List[List[int]]: + groups: Dict[str, List[int]] = {} + order: List[str] = [] + for i, key in enumerate(keys): + k = (key or "").strip() + if not k: + k = f"__EMPTY__::{i}" + if k not in groups: + groups[k] = [] + order.append(k) + groups[k].append(i) + return [groups[k] for k in order] + + +def _build_initial_clusters_from_contributions(contrib: Dict[str, Any]) -> List[Dict[str, Any]]: + contributions = [ + c for c in contrib.get("contributions", []) or [] + if c.get("label") in {"USES", "EXTENDS"} and (c.get("paper_claim") or c.get("claim")) + ] + if not contributions: + return [] + cluster_keys_all: List[str] = [] + for c in contributions: + key = (c.get("cluster_key") or "").strip() + if not key: + label = str(c.get("label", "USES")).upper() + if label not in {"USES", "EXTENDS"}: + label = "USES" + key = f"{label}|contribution|unspecified" + cluster_keys_all.append(key) + clusters = _cluster_by_exact_keys(cluster_keys_all) + out: List[Dict[str, Any]] = [] + for idx, cluster in enumerate(clusters, start=1): + first = contributions[cluster[0]] + key = cluster_keys_all[cluster[0]] + title = (first.get("cluster_title") or "").strip() or _title_from_cluster_key(key) + out.append({ + "cluster_id": f"C{idx}", + "count": str(len(cluster)), + "representative_claim": title, + "cluster_key": key, + "cluster_title": title, + "claim_indices": [str(i) for i in cluster], + }) + return out + + +def _cluster_support_summary(cluster: Dict[str, Any], contributions: List[Dict[str, Any]]) -> Dict[str, Any]: + indices = _to_int_indices(cluster.get("claim_indices") or []) + items: List[Dict[str, Any]] = [] + for i in indices: + if 0 <= i < len(contributions): + items.append(contributions[i]) + labels = [str(item.get("label", "")).upper() for item in items if item.get("label")] + examples: List[str] = [] + for item in items: + text = str(item.get("paper_claim") or item.get("claim") or "").strip() + if text: + examples.append(text) + if len(examples) >= 3: + break + rationales = [str(item.get("rationale", "")).strip() for item in items if item.get("rationale")][:2] + use_count = sum(1 for x in labels if x == "USES") + ext_count = sum(1 for x in labels if x == "EXTENDS") + return { + "examples": examples, + "rationales": rationales, + "uses_count": use_count, + "extends_count": ext_count, + "member_count": len(items), + } + + +def build_prompt(paper_title: str, centroids: List[Dict[str, Any]]) -> str: + lines: List[str] = [ + "You are refining downstream citation contribution clusters for one target paper.", + "Input clusters are already built. Your job is to (a) conservatively merge near-duplicate downstream-usage clusters and (b) drop clusters that do not actually show substantive downstream usage of the target contribution.", + "", + f"Target paper: {paper_title}", + "", + "Rules:", + "- Operate only at cluster level. Do not invent new instances.", + "- Prefer conservative merges. If unsure, keep clusters separate.", + "- You may drop clusters only when they fail to show real downstream use or extension of the target contribution.", + "- Drop clusters that are clearly mere mention, loose comparison, background citation, noisy extraction, or off-target usage.", + "- Never merge USES and EXTENDS clusters together.", + "- Every input cluster_id must either appear in exactly one kept group or in dropped_clusters.", + "- kept merged_key must be in format RELATION|artifact|purpose.", + "- merged_title must be a short natural-language summary (5-12 words).", + "", + "Input clusters:", + ] + for c in centroids: + lines.append( + f"- {c['cluster_id']}: key={c.get('cluster_key','')}; title={c.get('cluster_title','')}; count={c.get('count', 0)}; uses={c.get('uses_count',0)}; extends={c.get('extends_count',0)}; examples={' | '.join(c.get('examples',[])[:2])}; rationales={' | '.join(c.get('rationales',[])[:1])}" + ) + lines += [ + "", + "Return JSON only with this shape:", + "{", + ' "kept_groups": [', + " {", + ' "cluster_ids": ["C1","C3"],', + ' "merged_title": "Uses target dataset for evaluation",', + ' "merged_key": "USES|dataset|evaluation",', + ' "rationale": "Both clusters describe the same downstream dataset use."', + " }", + " ],", + ' "dropped_clusters": [', + ' {"cluster_id": "C7", "reason": "Only background mention; no substantive downstream use."}', + " ]", + "}", + ] + return "\n".join(lines) + + +def _normalize_decision(data: Dict[str, Any], original_clusters: List[Dict[str, Any]]) -> Tuple[List[Dict[str, Any]], List[Dict[str, Any]]]: + valid_ids = [c.get("cluster_id", "") for c in original_clusters if c.get("cluster_id")] + valid_set = set(valid_ids) + assigned = set() + kept: List[Dict[str, Any]] = [] + dropped: List[Dict[str, Any]] = [] + + for item in data.get("dropped_clusters") or []: + cid = item.get("cluster_id") + if cid in valid_set and cid not in assigned: + assigned.add(cid) + dropped.append({"cluster_id": cid, "reason": str(item.get("reason", "")).strip() or "Dropped by LLM filter."}) + + for g in data.get("kept_groups") or []: + ids = [cid for cid in (g.get("cluster_ids") or []) if cid in valid_set and cid not in assigned] + if not ids: + continue + for cid in ids: + assigned.add(cid) + kept.append({ + "cluster_ids": ids, + "merged_title": str(g.get("merged_title", "")).strip(), + "merged_key": str(g.get("merged_key", "")).strip(), + "rationale": str(g.get("rationale", "")).strip(), + }) + + for cid in valid_ids: + if cid not in assigned: + kept.append({ + "cluster_ids": [cid], + "merged_title": "", + "merged_key": "", + "rationale": "Auto-singleton fallback.", + }) + + order = {cid: i for i, cid in enumerate(valid_ids)} + kept.sort(key=lambda g: min(order[cid] for cid in g["cluster_ids"])) + dropped.sort(key=lambda x: order.get(x["cluster_id"], 10**9)) + return kept, dropped + + +def refine_paper(paper_dir: Path, overwrite: bool, inplace: bool) -> str: + disc_path = paper_dir / DISCOVERY_FILE + contrib_path = paper_dir / CONTRIB_FILE + meta_path = paper_dir / PAPER_META_FILE + + disc = load_json(disc_path) + contrib = load_json(contrib_path) + meta = load_json(meta_path) + if not isinstance(contrib, dict): + return "missing_inputs" + + if not isinstance(disc, dict): + disc = {"paper_id": contrib.get("paper_id"), "decision": "", "justification": "", "clusters": []} + + clusters = disc.get("clusters") or [] + if not clusters: + clusters = _build_initial_clusters_from_contributions(contrib) + if not clusters: + payload = dict(disc) + payload["clusters"] = [] + payload["dropped_clusters"] = [] + payload["cluster_refine_method"] = "llm_centroid_merge_filter" + payload["cluster_refine_source"] = CONTRIB_FILE + out_path = disc_path if inplace else (paper_dir / OUT_FILE) + out_path.write_text(json.dumps(payload, indent=2), encoding="utf-8") + return "empty_clusters" + + out_path = disc_path if inplace else (paper_dir / OUT_FILE) + if out_path.exists() and not overwrite: + return "skipped" + + contributions = contrib.get("contributions") or [] + centroids: List[Dict[str, Any]] = [] + auto_dropped: List[Dict[str, Any]] = [] + active_clusters: List[Dict[str, Any]] = [] + + for c in clusters: + cid = c.get("cluster_id", "") + summary = _cluster_support_summary(c, contributions) + rel, _, _ = _parse_key(c.get("cluster_key", "")) + if summary["uses_count"] + summary["extends_count"] == 0 or rel not in {"USES", "EXTENDS"}: + auto_dropped.append({"cluster_id": cid, "reason": "No verified USES/EXTENDS support in member contributions."}) + continue + row = { + "cluster_id": cid, + "cluster_key": c.get("cluster_key", ""), + "cluster_title": c.get("cluster_title") or c.get("representative_claim") or "", + "count": int(c.get("count", summary["member_count"]) or summary["member_count"]), + **summary, + } + centroids.append(row) + active_clusters.append(c) + + if not active_clusters: + payload = dict(disc) + payload["clusters"] = [] + payload["dropped_clusters"] = auto_dropped + payload["cluster_refine_method"] = "llm_centroid_merge_filter" + payload["cluster_refine_source"] = CONTRIB_FILE if not load_json(disc_path) else DISCOVERY_FILE + out_path.write_text(json.dumps(payload, indent=2), encoding="utf-8") + return "refined" + + prompt = build_prompt(_extract_title(meta), centroids) + client = LLMClient() + raw = client.call(prompt, schema=REFINE_SCHEMA) + data = json.loads(raw) + kept_groups, llm_dropped = _normalize_decision(data, active_clusters) + + id_to_cluster = {c.get("cluster_id"): c for c in active_clusters if c.get("cluster_id")} + merged_clusters: List[Dict[str, Any]] = [] + for idx, g in enumerate(kept_groups, start=1): + member_ids = g["cluster_ids"] + members = [id_to_cluster[mid] for mid in member_ids if mid in id_to_cluster] + merged_indices: List[int] = [] + for m in members: + for i in _to_int_indices(m.get("claim_indices") or []): + if i not in merged_indices: + merged_indices.append(i) + merged_indices.sort() + merged_key = g.get("merged_key") or _dominant_key(members) + rel, _, _ = _parse_key(merged_key) + if rel not in {"USES", "EXTENDS"}: + merged_key = _dominant_key(members) + merged_title = g.get("merged_title") or (members[0].get("cluster_title") if members else "") + if not merged_title: + merged_title = members[0].get("representative_claim", "") if members else "" + merged_clusters.append({ + "cluster_id": f"C{idx}", + "count": str(len(merged_indices)), + "representative_claim": merged_title, + "cluster_key": merged_key, + "cluster_title": merged_title, + "claim_indices": [str(i) for i in merged_indices], + "source_cluster_ids": member_ids, + "merge_rationale": g.get("rationale", ""), + }) + + payload = dict(disc) + payload["clusters"] = merged_clusters + payload["dropped_clusters"] = auto_dropped + llm_dropped + payload["cluster_refine_method"] = "llm_centroid_merge_filter" + payload["cluster_refine_source"] = CONTRIB_FILE if not load_json(disc_path) else DISCOVERY_FILE + out_path.write_text(json.dumps(payload, indent=2), encoding="utf-8") + return "refined" + + +def main() -> None: + parser = argparse.ArgumentParser(description="LLM centroid-level merge/filter pass for downstream contribution clusters.") + parser.add_argument("--root", type=str, default="runs/processed_papers", help="Root directory containing processed paper directories.") + parser.add_argument("--overwrite", action="store_true", help="Overwrite output file if it exists.") + parser.add_argument("--inplace", action="store_true", help="Write back to usage_discovery_from_contributions.json.") + args = parser.parse_args() + + root = Path(args.root).expanduser().resolve() + if not root.exists(): + raise SystemExit(f"Root directory does not exist: {root}") + + paper_dirs = iter_paper_dirs(root) + print(f"[INFO] Found {len(paper_dirs)} paper dirs under {root}") + counts = {"refined": 0, "skipped": 0, "missing_inputs": 0, "empty_clusters": 0} + for paper_dir in paper_dirs: + status = refine_paper(paper_dir, overwrite=args.overwrite, inplace=args.inplace) + counts[status] = counts.get(status, 0) + 1 + print(f"[{status.upper()}] {paper_dir.name}") + print("[SUMMARY] refined={refined}, skipped={skipped}, missing_inputs={missing_inputs}, empty_clusters={empty_clusters}".format(**counts)) + + +if __name__ == "__main__": + main() diff --git a/src/step_07_extract_and_refine/schemas.py b/src/step_07_extract_and_refine/schemas.py new file mode 100644 index 0000000000000000000000000000000000000000..fe30dcdb59f11369dfae1d2b107e8e07aa2c800d --- /dev/null +++ b/src/step_07_extract_and_refine/schemas.py @@ -0,0 +1,12 @@ +CONTRIBUTION_JSON_SCHEMA = { + "type": "object", + "properties": { + "label": {"type": "string", "enum": ["USES", "EXTENDS", "NOT_CONFIRMED"]}, + "paper_claim": {"type": "string"}, + "cluster_title": {"type": "string"}, + "cluster_key": {"type": "string"}, + "evidence_span": {"type": "string"}, + "rationale": {"type": "string"}, + }, + "required": ["label", "paper_claim", "cluster_title", "cluster_key", "evidence_span", "rationale"], +} diff --git a/src/step_08_annotation/__init__.py b/src/step_08_annotation/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..48c7c57a5599e6ce2faf6aedb5789992321fa439 --- /dev/null +++ b/src/step_08_annotation/__init__.py @@ -0,0 +1,3 @@ +from .pipeline import TwoPassAnnotationPipeline, TwoPassPipelineResult + +__all__ = ["TwoPassAnnotationPipeline", "TwoPassPipelineResult"] diff --git a/src/step_08_annotation/cli.py b/src/step_08_annotation/cli.py new file mode 100644 index 0000000000000000000000000000000000000000..72545193a73ecef504be89ae4b76f0a86a2b6192 --- /dev/null +++ b/src/step_08_annotation/cli.py @@ -0,0 +1,99 @@ +from __future__ import annotations + +import json +from pathlib import Path + +import typer + +from .paper_package import load_paper_package + +from .pipeline import TwoPassAnnotationPipeline + + +app = typer.Typer(help="Run step 8: derive target contributions, enabling contributions, and groundings.") + + +def _default_output_root() -> Path: + return Path("runs/two_pass_outputs") + + +@app.command() +def run( + paper_dir: Path = typer.Option(..., exists=True, file_okay=False, dir_okay=True), + provider: str = typer.Option("openai", help="Provider family: openai or gemini."), + model: str = typer.Option("openai/gpt-5", help="Reasoning model used for target-contribution derivation and annotation."), + formatter_model: str | None = typer.Option( + None, + help="Optional model override for pass 2 formatting, e.g. openai/gpt-5-mini or openai/gpt-5.4-pro.", + ), + judge_model: str | None = typer.Option( + None, + help="Optional model override for pass 1 candidate ranking. Ignored when --candidate-count=1.", + ), + candidate_count: int = typer.Option( + 1, + help="Number of reasoning candidates to generate. If set to 1, no judge call is made.", + ), + formatter_max_attempts: int = typer.Option( + 3, + help="Formatter-only retry attempts after pass 1 has succeeded.", + ), + include_reference_examples: bool = typer.Option( + True, + "--include-reference-examples/--no-include-reference-examples", + help="Include the built-in reference examples in the pass-1 reasoning prompt.", + ), + prompt_profile: str = typer.Option( + "full", + help="Reasoning prompt profile: full or generic.", + ), + output_root: Path = typer.Option( + _default_output_root(), + help="Directory to store run outputs.", + ), + run_label: str | None = typer.Option(None, help="Optional label to include in the saved run directory name."), + annotator_id: str = typer.Option("llm", help="Annotator id to embed in the final UI payload."), + extracted_claim: str | None = typer.Option(None, help="Optional override for the extracted target contribution."), +) -> None: + paper = load_paper_package(paper_dir, extracted_claim_override=extracted_claim) + pipeline = TwoPassAnnotationPipeline( + provider=provider, + model=model, + formatter_model=formatter_model, + judge_model=judge_model, + output_root=output_root, + run_label=run_label, + annotator_id=annotator_id, + candidate_count=candidate_count, + formatter_max_attempts=formatter_max_attempts, + include_reference_examples=include_reference_examples, + prompt_profile=prompt_profile, + progress_callback=typer.echo, + ) + result = pipeline.run(paper) + typer.echo(str(result.run_dir / "run_output.json")) + + +@app.command() +def summarize(run_output: Path = typer.Option(..., exists=True, dir_okay=False, file_okay=True)) -> None: + data = json.loads(run_output.read_text()) + payload = data.get("ui_payload") or {} + claims = payload.get("claims") or [] + summary = { + "paper_id": data.get("paper_id"), + "target_contribution_count": len(claims), + "target_contributions": [ + { + "claim_id": claim.get("claim_id"), + "rewritten_claim": claim.get("rewritten_claim"), + "decision": claim.get("decision"), + "enabling_contribution_count": len(claim.get("ingredients") or []), + } + for claim in claims + ], + } + typer.echo(json.dumps(summary, indent=2)) + + +if __name__ == "__main__": + app() diff --git a/src/step_08_annotation/final_prompts.py b/src/step_08_annotation/final_prompts.py new file mode 100644 index 0000000000000000000000000000000000000000..cf51c47bc1bba98b8f292dd345efd059f17e4e28 --- /dev/null +++ b/src/step_08_annotation/final_prompts.py @@ -0,0 +1,2026 @@ +from __future__ import annotations + +import json +from typing import Any, Dict + +SYSTEM_TWO_PASS_REASONING = """ +You are an expert annotator for scientific enabling-discovery pathways. + +This is the reasoning pass. +Do NOT output JSON. +Return a structured reasoning memo in markdown. + +Your goal is to recover one or more downstream-used claims directly from the full refined cluster set, then find the SMALLEST STRUCTURALLY SUFFICIENT set of enabling ingredients for each derived claim. + +Core objective: +- Identify one or more downstream-used claims with real downstream impact, directly from the refined citation clusters. +- Split bundled paper contributions into separate atomic claims only when downstream usage differs. +- For each claim, recover the minimal recall-oriented ingredient set needed to recreate the discovery in its claimed form. +- Focus on artifact-defining ingredients rather than narrow operational ones. +- Distinguish structural prerequisites from implementation details, helper tools, and local recipe choices. + +Primary principle: +A valid ingredient is something whose absence would prevent the discovery from existing in its claimed form. +If the discovery would still exist with the same essential identity without it, exclude it. + +================================================== +I. WHAT COUNTS AS A DISCOVERY CLAIM +================================================== + +A claim here is a downstream-used contribution recovered from the refined citation clusters, such that later work depends on it to build, evaluate, extend, or operationalize its own work, and removing it would make those downstream capabilities unavailable or substantially weaker. + +Allowed discovery types: +- Resource +- Dataset +- Benchmark +- Method +- Tool +- Finding + +Do NOT annotate a paper contribution as a claim if: +- there is no downstream evidence of meaningful use +- it is only mentioned or compared against +- it is only an internal supporting artifact for another downstream-used contribution +- it is a same-paper asset with no independent downstream role + +Claim splitting rule: +Split one paper into multiple claims only when: +- the paper contributes distinct artifacts/findings +- they have different downstream usage clusters +- they would require different ingredient decompositions + +================================================== +II. HOW CLAIMS SHOULD BE REWRITTEN +================================================== + +Each rewritten claim must be a compact functional abstraction: + +[artifact type]: [object] + [key property] + [what it enables] + +Requirements: +- atomic: exactly one contribution +- abstracted: not just the paper name or title +- functional: says what the contribution is/does +- causal: states what it enables +- decomposable: should support ingredient decomposition +- aligned to downstream evidence, not just the abstract + +Rewriting rules: +- rewrite the contribution, do not paraphrase the paper abstract +- avoid names unless necessary +- avoid vague phrases like “improves performance” +- do not include recommendations or speculative implications +- for findings: state the empirical result and setting, not advice + +================================================== +III. RECALL-ORIENTED INGREDIENT SELECTION +================================================== + +You are doing recall-oriented ingredient annotation. + +This means: +- return the smallest structurally sufficient set +- prefer strong ingredients over a long implementation checklist +- include only what is structural to the discovery itself + +Default bias: +- prefer higher-level structural ingredients, but still be as specific as possible such that the ingredients allow you to reconstruct the discovery in its claimed form. +- prefer artifact-defining ingredients +- prefer benchmark-construction ingredients over benchmark-use ingredients +- prefer model-defining ingredients over training conveniences + +Exclude by default unless the ingredient is clearly structural to the discovery as claimed: +- replaceable ingredients used only for one small step, one ablation, or one small subcomponent +- local preprocessing, filtering, balancing, prompting, or augmentation steps +- low-level recipe choices such as language tags, tokenization/vocabulary choices, optimizer settings, or similar configuration details +- procedures used only to apply, evaluate, or use the artifact rather than construct it +- anything that is merely one possible implementation of a local substep rather than a defining prerequisite of the artifact + +If unsure whether something is structural or just operational: +exclude it unless removing it would fundamentally change the discovery’s claimed form. + +================================================== +IV. BENCHMARK / DATASET RULES +================================================== + +For benchmark or dataset claims, include only ingredients needed to construct the artifact itself. + +Prefer benchmark/dataset ingredients such as: +- benchmark framing / evaluation paradigm +- source-task or source-data substrate +- benchmark-wide annotation / curation / verification workflow +- core conceputal trick or idea +- benchmark-defining task formulation or protocol + +Benchmark test: +Ask: +- If removed, would the benchmark still exist as the same benchmark? +If the benchmark still exists, exclude it. + +================================================== +V. METHOD / RESOURCE RULES +================================================== + +Prefer method/resource ingredients such as: +- core architecture +- core training objective(s) +- core conceputal trick or idea +- structurally necessary training data +- essential initialization/checkpoint + +Method test: +Ask: +- Is this part of what makes the method the method? +- Or is it just part of one particular implementation detail? +If it is implementation-level rather than artifact-defining, exclude it. + +================================================== +VI. FINDING RULES +================================================== + +For finding claims, include ingredients needed to make the empirical finding possible, such as: +- the evaluation protocol under examination +- the model family or artifact type whose behavior is studied +- core conceputal trick or idea +- the benchmark/task substrate needed to observe the finding +- the extraction or measurement protocol required to quantify the finding + +Do NOT include: +- recommendations derived from the finding +- proposed fixes unless they are part of the finding itself +- broad background theories unless operationally required + +================================================== +VII. CANONICAL VS ADDITIONAL VS NONE +================================================== + +For every ingredient, decide: +- one canonical grounding +- additional groundings +- or NONE + +Canonical grounding: +The single best representative prior study or resource that cleanly provides the ingredient. +It should be: +- necessary for the discovery in its claimed form +- directly used, clearly instantiated, or operationally inherited by the target paper +- minimal, not overly broad +- the most faithful representative, not merely the earliest or most famous + +Additional groundings: +Use when: +- the ingredient is composite +- multiple prior studies materially contribute +- one canonical study is representative but incomplete + +NONE grounding: +Use "__NONE__" when no single prior study or resource cleanly represents the ingredient. + +Use NONE especially when: +- the ingredient is composite across multiple independent sources +- the main grounding is released in the target paper +- the ingredient is a benchmark-wide or corpus-wide workflow not attributable to one prior study +- a prior cited study is only a minor component of the true ingredient +- forcing one canonical study would distort the real structure of the ingredient +- the ingredient is field-level or distributed across many sources + +Important NONE rule: +If the dominant realization of an ingredient comes from the target paper itself, but smaller contributing prior sources exist, use NONE and list those smaller prior sources as additional groundings. + +Never choose a canonical study just because: +- it is cited +- it is famous +- it is somewhat related +- it helps implement a local substep + +================================================== +VIII. ROLE DEFINITIONS +================================================== + +Use only these roles: +- CONCEPTUAL_FRAMEWORK +- CORE_METHOD +- DATA_SOURCE +- TRAINING_DATA +- MODEL_INITIALIZATION +- EVALUATION_PROTOCOL +- IMPLEMENTATION_TOOLING + +Role guide: +- CONCEPTUAL_FRAMEWORK: task definition, problem framing, representational framing, theoretical framing, empirical framing +- CORE_METHOD: architecture, algorithm, training objective, inference mechanism, optimization mechanism, computational method +- DATA_SOURCE: upstream resource used to build another dataset/resource +- TRAINING_DATA: data used directly to train or fine-tune the discovery +- MODEL_INITIALIZATION: pretrained checkpoint or initialization necessary to the discovery +- EVALUATION_PROTOCOL: benchmark, metric, annotation scheme, or evaluation setup reused as part of the discovery +- IMPLEMENTATION_TOOLING: software/tooling/infrastructure explicitly required to implement the contribution in principle + +Use DATA_SOURCE when a resource is used to build another artifact. +Use TRAINING_DATA when it is directly used to train the artifact. +If a resource is mostly same-paper and only partly prior-work, prefer NONE. + +================================================== +IX. REQUIRED REASONING PROCEDURE +================================================== + +You must reason in this order: + +Step 1: Cluster-first claim identification +- inspect the refined downstream citation clusters first +- derive one or more atomic claims from those clusters +- do not annotate unused paper outputs as claims + +Step 2: Claim split +- split only when downstream-used contributions are distinct +- choose one representative cluster per claim if possible + +Step 3: Rewrite claim +- rewrite at the abstraction level matching the intended ingredients +- do not write a claim so specific that it forces recipe-level decomposition + +Step 4: Broad candidate ingredient generation +- for each claim, first list the broad candidate ingredients that could be necessary +- then list the prior studies/resources that might ground each candidate ingredient +- at this stage, be inclusive rather than minimal +- do not yet decide canonical vs additional vs NONE +- do not treat every cited paper as an ingredient; only include studies that could plausibly ground a necessary ingredient + +Step 5: Minimal ingredient refinement +- collapse the broad candidate set into the smallest structurally sufficient ingredient set +For every tempting ingredient or study, explicitly ask: +- is it structural or merely helpful? +- would removing it change the identity of the discovery as claimed? + +Step 7: Grounding pass +Only after finalizing the minimal ingredient set: +- decide canonical vs additional vs NONE +- prefer NONE when the ingredient is composite, same-paper dominant, field-level, or not cleanly attributable to one prior study +- use additional groundings when several studies jointly support the ingredient but none alone fully represents it + +================================================== +X. EVIDENCE REQUIREMENT +================================================== + +For each final ingredient, provide exactly one short verbatim quote from the target paper. +Requirements: +- 5–25 words +- one fully written quote only +- no section references without quotation +- quote must directly support the ingredient +- prefer the most concrete sentence available + +================================================== +XI. OUTPUT FORMAT +================================================== + +Return a structured markdown memo with these exact section headings: + +1. Cluster Evidence +2. Claim Split Decision +3. Rewritten Claims +4. Minimal Ingredient Set Per Claim +5. Excluded Tempting Non-Ingredients +6. Candidate Grounding Decisions +7. Final Grounding Decisions +8. Final Rationales And Evidence + +For each rewritten claim, include: +- claim_id +- artifact_type +- rewritten_claim +- why_this_is_atomic +- optional cluster_id if inferable +- decision: YES_SUFFICIENT / NO_NOT_DISCOVERY / UNCERTAIN + +For each ingredient, include: +- ingredient_id +- ingredient +- why_structurally_necessary +- why_not_lower_level_substeps +- why_not_adjacent_implementation_details +- necessary +- from_prior_work +- maps_cleanly_to_one_study +- canonical grounding decision +- additional groundings if any +- role +- contribution +- rationale +- evidence_span + +For excluded tempting non-ingredients, include only the strongest ones and say briefly why they are excluded. + +Memo style: +- concise but explicit +- compact bullets preferred +- no JSON +- do not restate the full paper +- do not list every citation +- keep reasoning focused on structural sufficiency +""" + +def compact_json(data: Any) -> str: + return json.dumps(data, indent=2, ensure_ascii=True) + +def _target_paper_context(paper) -> str: + return f"""-------------------------------------------------- +TARGET PAPER CONTEXT +-------------------------------------------------- +Target paper context: + +TITLE: {paper["paper_metadata"].get("title", "")} +YEAR: {paper["paper_metadata"].get("year", "")} +VENUE: {paper["paper_metadata"].get("venue", "")} +PAPER_ID: {paper["paper_metadata"].get("paper_id", "")} + +REFINED DOWNSTREAM CLUSTER EVIDENCE: +{compact_json(paper.get("downstream_cluster_evidence") or [])} + +TARGET PAPER CONTEXT: +{compact_json(paper.get("downstream_cluster_evidence") or [])} + +FULL PAPER CONTENT: +{paper.get("full_processed_text") or compact_json(paper.get("paper_text") or {})} + +BIBLIOGRAPHY: +{compact_json(paper.get("bibliography") or [])} + +CITATION CONTEXTS: +{compact_json(paper.get("citation_contexts") or [])} +""" + + +def reasoning_prompt(paper, include_reference_examples: bool = True, prompt_profile: str = "full"): + if prompt_profile not in {"full", "generic"}: + raise ValueError(f"Unknown prompt_profile: {prompt_profile}") + + if prompt_profile == "generic": + return f""" +Task: Produce the reasoning memo for cluster-first claim derivation and enabling-ingredient annotation. + +Identify one or more downstream-used claims directly from the full refined citation-cluster evidence. For each claim, rewrite the claim clearly and list the enabling contributions needed to realize the discovery. + +For each enabling contribution, provide: +- ingredient +- role +- contribution +- rationale +- canonical grounding decision +- additional groundings if any +- evidence: one verbatim quote + +Guidelines: +- Use the paper context and refined cluster evidence to decide what claims exist. +- Include contributions that are necessary for the claimed discovery. +- If a contribution is introduced in the target paper and does not map cleanly to one prior study, use NONE. +- Return a markdown reasoning memo only. + +{_target_paper_context(paper)}""" + + reference_examples = "" + if include_reference_examples: + reference_examples = f"""-------------------------------------------------- +REFERENCE EXAMPLE 1 +-------------------------------------------------- +{EXAMPLE_1} + +-------------------------------------------------- +REFERENCE EXAMPLE 2 +-------------------------------------------------- +{EXAMPLE_2} + +-------------------------------------------------- +REFERENCE EXAMPLE 3 +-------------------------------------------------- +{EXAMPLE_3} + +-------------------------------------------------- +REFERENCE EXAMPLE 4 +-------------------------------------------------- +{EXAMPLE_4} + +""" + + return f""" +Task: Produce the reasoning memo for cluster-first claim derivation and enabling-ingredient annotation. + +Important: +- Think cluster-first. +- Do not start from a paper-level discovery summary. +- Return the smallest structurally sufficient ingredient set. + +Required procedure: +1. Identify downstream-used claims from the cluster evidence. +2. Split claims only when downstream-used contributions are distinct. +3. Rewrite each claim at the abstraction level matching the intended ingredients. +4. For each claim, first list broad candidate ingredients and possible grounding studies. +5. Collapse those candidates into the minimal structurally sufficient ingredient set. +6. Explicitly exclude tempting non-ingredients. E.g. + - one-task tools + - local preprocessing + - balancing tricks + - token/language-tag tricks + - helper models used for one small step +7. Only then decide canonical vs additional vs NONE. + + +Benchmark/dataset claims: include only what is needed to build the artifact itself. Favor the benchmark framing, core conceputal idea, source tasks/data, and benchmark-wide annotation or verification protocol. Exclude anything the benchmark could still exist without. +Method/resource claims: include only what defines the artifact: core architecture, core objective, core conceputal idea, essential training data, or essential initialization. Exclude non-essential implementation details. +Finding claims: include only what is needed to observe and measure the empirical result: the evaluation protocol, core conceputal idea, studied model/artifact type, task substrate, and measurement/extraction procedure. Exclude recommendations, fixes, and general background unless they are operationally necessary. + +For each final ingredient, provide: +- ingredient +- why it is structurally necessary +- why tempting alternatives are excluded +- canonical grounding decision +- additional groundings if any +- role +- contribution +- rationale +- evidence:one verbatim quote + +Return a markdown reasoning memo only. +{reference_examples}{_target_paper_context(paper)}""" + +SYSTEM_TWO_PASS_FORMATTER = """ +You are a strict annotation JSON formatter. + +Your job is to convert a reasoning memo into the exact annotation UI JSON payload. + +You do not do substantive annotation. +You do not invent new ingredients, claims, studies, or roles. +You only translate the memo into schema-valid JSON as faithfully and conservatively as possible. + +Primary objective: +- preserve the final decisions in the reasoning memo +- keep the ingredient set minimal +- do not re-expand collapsed ingredients +- do not promote excluded non-ingredients back into the JSON +- prefer conservative encoding when the memo is ambiguous + +Hard rules: +- Return valid JSON only. +- Do not add prose. +- Do not add markdown. +- Do not invent field names. +- Do not invent enum values. +- Do not invent missing studies or metadata. +- Do not change the number of claims or ingredients unless the memo explicitly requires it. +- Do not convert local implementation details into ingredients if the memo excluded them. +- Do not override a NONE decision. +- Do not create canonical groundings from related studies when the memo chose NONE. +- Do not add enabling_discoveries entries for NONE ingredients. +- Do not infer additional studies unless they are explicitly present in the memo. +- Do not split one ingredient into multiple JSON ingredients. + +Allowed roles only: +- CONCEPTUAL_FRAMEWORK +- CORE_METHOD +- DATA_SOURCE +- TRAINING_DATA +- MODEL_INITIALIZATION +- EVALUATION_PROTOCOL +- IMPLEMENTATION_TOOLING + +Normalization rules: +1. text must equal rewritten_claim +2. active_claim_id should be the first claim_id unless the memo explicitly selects another +3. if canonical grounding is NONE: + - canonical_ref_id = "__NONE__" + - canonical_grounding = null + - ingredient must not appear in enabling_discoveries +4. if roles has length 1, role must equal that role +5. if roles has length != 1, role must be null +6. additional_ref_ids must exactly match additional_groundings[].ref_id in order +7. enabling_discoveries should contain only canonical non-NONE ingredients +8. if a study is mentioned only as rejected, do not include it anywhere in final JSON +9. if an ingredient has contribution/rationale/evidence in the memo, preserve them closely +10. if metadata for a study is partially missing in the memo, use empty strings or empty objects rather than inventing values + +Conservative ambiguity policy: +- If the memo contains both broader and narrower formulations, choose the broader one if it is the stated final ingredient. +- If the memo distinguishes structural ingredients from excluded implementation details, encode only the structural ingredients. +- If a canonical study is uncertain but the memo leans NONE, choose NONE. +- If the memo gives additional studies for a composite NONE ingredient, include them only as additional_groundings. +- If the memo includes an “Excluded Tempting Non-Ingredients” section, treat those exclusions as binding. + +Your output must exactly follow the target schema provided in the user prompt. +""" + + +def formatter_prompt(paper: Dict[str, Any], reasoning_text: str, annotator_id: str) -> str: + return f"""Task: Convert the reasoning memo into the final annotation UI JSON payload. + +This is a strict formatting pass. +Do not do new reasoning unless needed to resolve minor schema ambiguities. +Follow the memo’s final decisions faithfully. + +Critical formatting priorities: +1. Preserve the final claim split. +2. Preserve the final minimal ingredient set. +3. Respect exclusions of tempting non-ingredients. +4. Preserve canonical vs additional vs NONE decisions exactly. +5. Do not inflate the ingredient set. + +Target schema: +{{ + "target_paper_id": "...", + "target_title": "...", + "target_year": 2024, + "annotator_id": "{annotator_id}", + "active_claim_id": "...", + "claims": [ + {{ + "claim_id": "...", + "text": "...", + "rewritten_claim": "...", + "cluster_id": "...", + "decision": "YES_SUFFICIENT | NO_NOT_DISCOVERY | UNCERTAIN", + "notes": "", + "ingredients": [ + {{ + "ingredient_id": "...", + "ingredient": "...", + "canonical_ref_id": "__NONE__ or ref_id", + "canonical_grounding": null or {{ + "ref_id": "...", + "bib_key": "...", + "paper_id": "...", + "external_ids": {{}}, + "ref_title": "...", + "ref_year": "...", + "ref_authors": "..." + }}, + "additional_ref_ids": ["..."], + "additional_groundings": [ + {{ + "ref_id": "...", + "bib_key": "...", + "paper_id": "...", + "external_ids": {{}}, + "ref_title": "...", + "ref_year": "...", + "ref_authors": "..." + }} + ], + "canonical_annotation": {{ + "role": null or "CONCEPTUAL_FRAMEWORK" or "CORE_METHOD" or "DATA_SOURCE" or "MODEL_INITIALIZATION" or "EVALUATION_PROTOCOL" or "IMPLEMENTATION_TOOLING" or "TRAINING_DATA", + "roles": ["..."], + "contribution": "...", + "rationale": "...", + "evidence_span": "..." + }} + }} + ], + "enabling_discoveries": [ + {{ + "ref_id": "...", + "bib_key": "...", + "paper_id": "...", + "external_ids": {{}}, + "ref_title": "...", + "ref_year": "...", + "ref_authors": "...", + "ingredient_id": "...", + "ingredient": "...", + "role": null or "CONCEPTUAL_FRAMEWORK" or "CORE_METHOD" or "DATA_SOURCE" or "MODEL_INITIALIZATION" or "EVALUATION_PROTOCOL" or "IMPLEMENTATION_TOOLING" or "TRAINING_DATA", + "roles": ["..."], + "contribution": "...", + "rationale": "...", + "evidence_span": "..." + }} + ] + }} + ] +}} + +Formatting rules: +- `text` must equal `rewritten_claim` +- `active_claim_id` should be the first claim_id unless the reasoning memo explicitly recommends another active claim +- if canonical grounding is NONE: + - `canonical_ref_id` must be "__NONE__" + - `canonical_grounding` must be null + - do not include that ingredient in `enabling_discoveries` +- if roles has length 1, `role` should equal that role +- if roles has length != 1, `role` should be null +- `additional_ref_ids` must match `additional_groundings[].ref_id` +- `enabling_discoveries` should contain only canonical non-NONE ingredients +- use only the allowed role set + +REASONING MEMO: +{reasoning_text} + +TARGET PAPER METADATA: +{compact_json(paper.get("paper_metadata") or {})} +""" + +SYSTEM_TWO_PASS_JUDGE = """ +You are an expert judge for scientific enabling-discovery annotation. + +This is the critic/judge pass. +You are evaluating candidate reasoning memos produced by the reasoner. +You are NOT formatting JSON yet. + +Your goal is to select the candidate that best recovers the STRUCTURALLY SUFFICIENT set of enabling ingredients for each downstream-used discovery claim. + +================================================== +I. JUDGING OBJECTIVE +================================================== + +Choose the candidate that is best aligned with the annotation objective: + +- downstream-first claim identification +- correct claim splitting +- compact functional claim rewriting +- minimal recall-oriented ingredient decomposition +- exclusion of implementation details and local helpers +- correct canonical vs additional vs NONE decisions +- clear, direct evidence for each final ingredient + +The best candidate is NOT the one with: +- the most details +- the most cited papers +- the most plausible implementation story +- the richest operational decomposition + +The best candidate IS the one with: +- the right claims +- the fewest correct structural ingredients +- the cleanest exclusions +- the least distortion from overly specific groundings + +================================================== +II. PRIMARY EVALUATION PRINCIPLE +================================================== + +Judge candidates by this question: + +Does this memo recover the smallest structurally sufficient ingredient set for the downstream-used artifact or finding? + +If a candidate includes extra ingredients that are merely helpful, local, one-task, or implementation-level, penalize it. + +If two candidates are similarly correct, prefer: +- fewer ingredients +- higher-level but still reconstructive ingredients +- better use of NONE for composite or same-paper-dominant ingredients +- clearer separation between artifact-defining ingredients and recipe details + +================================================== +III. CLAIM-LEVEL JUDGING RULES +================================================== + +A good candidate: +- starts from downstream evidence, not from the paper abstract or citations +- annotates only contributions with meaningful downstream impact +- splits claims only when downstream-used contributions are distinct +- assigns one representative cluster per claim when possible +- rewrites claims as compact functional abstractions: + [artifact type]: [object] + [key property] + [what it enables] + +Penalize candidates that: +- annotate unused paper outputs as claims +- bundle distinct downstream-used artifacts into one claim +- oversplit one artifact into multiple claims without downstream justification +- rewrite claims too specifically, forcing recipe-level decomposition +- include names or paper phrasing where abstraction is expected +- include recommendations or speculative implications inside finding claims + +================================================== +IV. INGREDIENT-LEVEL JUDGING RULES +================================================== + +The judge must strongly prefer recall-oriented minimality. + +A good candidate: +- returns the smallest structurally sufficient ingredient set +- includes only ingredients whose absence would prevent the discovery from existing in its claimed form +- prefers artifact-defining ingredients over narrow operational substeps + +Penalize candidates that include: +- one-task helpers +- local preprocessing/filtering/balancing tricks +- token/language-tag/vocabulary/configuration details +- helper models used only for one small construction step +- benchmark-use ingredients rather than benchmark-construction ingredients +- model-training conveniences rather than model-defining ingredients +- evaluation-time training data that is only needed to use the benchmark, not construct it + +If a candidate includes extra plausible ingredients that are not structural, penalize it even if they are real and cited. + +================================================== +V. DISCOVERY-TYPE RULES +================================================== + +For benchmark/dataset claims, prefer candidates that include only: +- benchmark framing / evaluation paradigm +- source-task or source-data substrate +- benchmark-wide annotation / curation / verification protocol +- core conceputal trick or idea +- benchmark-defining task formulation or protocol + +Penalize benchmark candidates that include: +- tools used for only one benchmark subtask +- benchmark-use training data +- local translation or preprocessing tools unless structurally benchmark-wide + +For method/resource claims, prefer candidates that include only: +- core architecture +- core conceputal trick or idea +- core training objective(s) +- structurally necessary training data +- essential initialization/checkpoint + +Penalize method/resource candidates that include: +- augmentation tools +- balancing tricks +- language tags +- tokenizer/vocabulary subchoices +- narrow training recipe details +unless clearly artifact-defining + +For finding claims, prefer candidates that include only: +- the evaluation protocol under examination +- the model family or artifact type being studied +- core conceputal trick or idea +- the benchmark/task substrate needed to observe the finding +- the measurement/extraction protocol needed to quantify it + +Penalize candidates that include: +- recommendations derived from the finding +- proposed fixes unless part of the finding itself +- broad background theories not operationally required + +================================================== +VI. GROUNDING RULES +================================================== + +Judge canonical/additional/NONE decisions by structural faithfulness. + +Canonical grounding should be chosen only when one prior study or resource cleanly represents the ingredient. + +Prefer NONE when: +- the ingredient is composite across multiple independent sources +- the dominant realization is released in the target paper +- the ingredient is benchmark-wide or corpus-wide and not attributable to one study +- a prior cited study is only a minor part of the true ingredient +- forcing one canonical study would distort the actual structure + +Prefer candidates that: +- use NONE conservatively but correctly +- use additional groundings for partial contributors +- avoid choosing famous papers just because they are famous +- avoid promoting local helper tools into canonical ingredients + +Penalize candidates that: +- force a canonical grounding where NONE is more faithful +- omit additional groundings when the ingredient is clearly composite +- choose groundings based on citation salience rather than operational necessity + +================================================== +VII. ROLE FIDELITY +================================================== + +Allowed roles: +- CONCEPTUAL_FRAMEWORK +- CORE_METHOD +- DATA_SOURCE +- TRAINING_DATA +- MODEL_INITIALIZATION +- EVALUATION_PROTOCOL +- IMPLEMENTATION_TOOLING + +Prefer candidates that assign roles conservatively and correctly. + +Penalize: +- TRAINING_DATA when the resource is really a source for constructing another artifact +- DATA_SOURCE when the resource is directly used to train the artifact +- IMPLEMENTATION_TOOLING for things that are merely optional recipe choices +- inflated CONCEPTUAL_FRAMEWORK assignments for concrete tooling or datasets + +================================================== +VIII. EVIDENCE QUALITY +================================================== + +A good candidate provides, for each final ingredient: +- one direct short quote from the target paper +- a quote that directly supports the ingredient +- evidence that matches the contribution/rationale + +Penalize: +- vague evidence +- evidence that supports only a local implementation detail +- quotes that do not justify the claimed level of abstraction +- multiple stitched spans when one direct quote should suffice + +================================================== +IX. HOW TO DECIDE BETWEEN CANDIDATES +================================================== + +Use this ranking order: + +1. Correct downstream-used claim selection +2. Correct claim splitting and abstraction level +3. Smallest structurally sufficient ingredient set +4. Correct exclusion of tempting non-ingredients +5. Correct canonical vs additional vs NONE decisions +6. Correct role assignments +7. Evidence quality and clarity +8. Overall readiness for formatter conversion + +If one candidate is more detailed but another is more minimal and structurally faithful, prefer the more minimal and structurally faithful one. + +================================================== +X. OUTPUT FORMAT +================================================== + +Return valid JSON only with this schema: + +{ + "selected_candidate_index": 0, + "selected_candidate_id": "candidate_1", + "selected_reason": "...", + "candidate_scores": [ + { + "candidate_id": "candidate_1", + "candidate_index": 0, + "score": 1, + "assessment": "..." + } + ] +} + +Scoring: +- Use integer scores from 1 to 10. +- 10 = best candidate by the above criteria. +- The selected candidate must have the highest score. +- In `selected_reason`, explain why it best matches the minimal structurally sufficient annotation objective. +- In each `assessment`, mention both strengths and weaknesses, especially over-decomposition, bad exclusions, or bad NONE choices. +""" + +def judge_prompt(paper: Dict[str, Any], candidate_texts: list[str]) -> str: + candidate_blocks = [] + for index, text in enumerate(candidate_texts): + candidate_blocks.append( + f"CANDIDATE {index + 1}: candidate_{index + 1}\n" + + "-" * 60 + + "\n" + + text.strip() + + "\n" + ) + candidates_section = "\n\n".join(candidate_blocks) + + return f"""Task: Compare candidate reasoning memos for a single target paper. + +You will be given: +1. target paper metadata +2. the original extracted discovery claim +3. optional downstream usage evidence +4. multiple candidate reasoning outputs from pass 1 + +Choose the candidate that best matches the minimal structurally sufficient enabling-discovery annotation objective. + +Important: +- Judge candidates using the full critic policy from the system instructions. +- Do not reward verbosity by itself. +- Do not reward confidence by itself. +- Do not reward richer implementation stories, more citations, or more operational detail unless they are structurally necessary. +- Prefer candidates that recover the smallest structurally sufficient ingredient set for each downstream-used claim. +- Prefer candidates that minimize formatter guesswork. + +Apply these priorities in order: +1. Correct downstream-used claim selection +2. Correct claim splitting and abstraction level +3. Smallest structurally sufficient ingredient set +4. Correct exclusion of tempting non-ingredients +5. Correct canonical vs additional vs "__NONE__" grounding decisions +6. Correct role assignments +7. Evidence quality and clarity +8. Overall readiness for formatter conversion + +Specific judging reminders: +- Prefer downstream-first claim identification over abstract-first or citation-first reasoning. +- Penalize candidates that annotate unused paper outputs as claims. +- Penalize candidates that oversplit one artifact into multiple claims without downstream justification. +- Penalize candidates that include one-task helpers, local preprocessing, balancing tricks, token/language-tag details, helper models for one small step, benchmark-use ingredients, model-training conveniences, or other non-structural details. +- For benchmark/dataset claims, prefer benchmark framing, source-task/source-data substrate, benchmark-wide annotation/curation/verification protocol, benchmark-defining task formulation/protocol, and any core conceptual trick or idea. +- For method/resource claims, prefer core architecture, core conceptual trick or idea, core objective(s), structurally necessary training data, and essential initialization/checkpoints. +- For finding claims, prefer the model/artifact type under study, task/benchmark substrate, evaluation/measurement protocol, and any core conceptual trick or idea needed to make the finding observable. +- Prefer "__NONE__" when an ingredient is composite, same-paper dominant, benchmark-wide/corpus-wide, or would be distorted by forcing one canonical prior study. +- Penalize candidates that force canonical studies where "__NONE__" is more faithful. +- Penalize candidates that choose famous or cited papers rather than the most structurally faithful grounding. +- Prefer role assignments that are conservative and faithful to the ingredient's function. + +TARGET PAPER METADATA: +{compact_json(paper.get('paper_metadata') or {})} + +ORIGINAL DISCOVERY CLAIM: +{paper.get("extracted_discovery_claim") or ""} + +OPTIONAL DOWNSTREAM USAGE EVIDENCE: +{compact_json(paper.get("downstream_cluster_evidence") or [])} + +CANDIDATE REASONING OUTPUTS: +{candidates_section} + +Return only valid JSON with this exact schema: + +{{ + "selected_candidate_index": 0, + "selected_candidate_id": "candidate_1", + "selected_reason": "...", + "candidate_scores": [ + {{ + "candidate_id": "candidate_1", + "candidate_index": 0, + "score": 1, + "assessment": "..." + }} + ] +}} + +Scoring rules: +- Use integer scores from 1 to 10. +- 10 = best candidate by the minimal structurally sufficient annotation objective. +- The selected candidate must have the highest score. +- In "selected_reason", explain why it best matches the downstream-first, minimal-ingredient, structurally faithful objective. +- In each "assessment", mention both strengths and weaknesses, especially over-decomposition, poor exclusions, weak abstraction level, bad role choices, and bad canonical vs "__NONE__" decisions. + +Return only valid JSON. Do not add any prose outside the JSON object. +""" + +EXAMPLE_1 = """ +# EXAMPLE: Pick-a-Pic / PickScore + +## 1. Cluster Evidence + +**The PickScore metric, a CLIP-based scoring function that predicts human preferences to evaluate and rank text-to-image generation models.** + +The paper introduces two downstream-used contributions: PickScore and the Pick-a-Pic dataset. Downstream evidence shows separate reuse: + +- PickScore is reused as an evaluation metric / reward model. +- Pick-a-Pic is reused as a dataset for training and preference tuning. + +So the bundled contribution should be split into two atomic claims. + +--- + +## 2. Claim Split Decision + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Tool +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C1 + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Dataset +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C2 + +### Why this split is correct + +The two artifacts have distinct downstream roles and distinct ingredient decompositions: + +- **PickScore** is reused as a scoring / evaluation / reward model. +- **Pick-a-Pic** is reused as a preference dataset for training and benchmarking. + +They should therefore be annotated separately. + +--- + +## 3. Rewritten Claims + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Tool +- **rewritten_claim:** **"Tool: A human-preference prompt-conditioned image scorer, enabling automatic ranking of text-to-image model outputs** +- **why_this_is_atomic:** This claim isolates the scorer itself as the reused artifact. It does not bundle the dataset resource used to train it. +- **optional cluster_id:** C1 +- **decision:** YES_SUFFICIENT + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Dataset +- **rewritten_claim:** **Dataset: An open prompt-conditioned human-preference dataset for generated images, enabling training and evaluation of preference-aligned text-to-image systems.** +- **why_this_is_atomic:** This claim isolates the dataset as the released reusable resource, separate from the scoring model trained on it. +- **optional cluster_id:** C2 +- **decision:** YES_SUFFICIENT + +--- + +## 4. Minimal Ingredient Set Per Claim + +## Claim C1: PickScore + +### Ingredient C1.I1 +- **ingredient_id:** C1.I1 +- **ingredient:** Pre-trained vision language model for prompt-conditioned image scoring +- **why_structurally_necessary:** PickScore is explicitly a CLIP-based scoring function. Without a pretrained joint text-image model, the scorer would not exist in its claimed form. +- **why_not_lower_level_substeps:** This should stay at the level of pretrained vision-language initialization, not be decomposed into separate text encoder, image encoder, or projection details. +- **why_not_adjacent_implementation_details:** Specific OpenCLIP variants are narrower implementation choices rather than the core structural dependency. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Learning Transferable Visual Models From Natural Language Supervision +- **additional groundings if any:** none +- **role:** MODEL_INITIALIZATION +- **contribution:** Provides the pre-trained text and image representations that are fine-tuned into a human preference score. +- **rationale:** To score a generated image relative to a prompt, the function needs a pretrained joint text–image representation. CLIP is the cleanest representative study for such capability and is the direct architectural basis of PickScore. +- **evidence_span:** “PickScore follows the architecture of CLIP [12]” + +### Ingredient C1.I2 +- **ingredient_id:** C1.I2 +- **ingredient:** Large-scale dataset of human-preference judgements over text-to-image outputs +- **why_structurally_necessary:** The scorer must learn which of two generated images humans prefer for a given prompt. Without this training signal, it would not become a human-preference metric. +- **why_not_lower_level_substeps:** This should remain a single training-data ingredient rather than being split into prompts, pairs, ties, logging, or collection mechanics. +- **why_not_adjacent_implementation_details:** Dataset construction details belong to the dataset claim, not the scorer’s minimal training-data dependency. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** TRAINING_DATA +- **contribution:** Provides the pairwise user choices needed to fit the scorer. +- **rationale:** The model needs direct supervision about which of two generated images users prefer for a given prompt. This operative dataset is released in the target paper, so the right grounding is `NONE`. +- **evidence_span:** “We finetune CLIP-H [7] using our framework8 on the Pick-a-Pic training set.” + +### Ingredient C1.I3 +- **ingredient_id:** C1.I3 +- **ingredient:** Preference-learning objective to model the reward +- **why_structurally_necessary:** PickScore is trained to assign higher scores to preferred images. Without a preference-learning objective, it would not become a ranking metric that evaluates outputs by predicted human choice. +- **why_not_lower_level_substeps:** This should remain one core-method ingredient rather than being decomposed into loss cases, tie handling, or optimization details. +- **why_not_adjacent_implementation_details:** Training schedule and optimizer are implementation details; the structural dependency is the reward-model-style objective itself. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Training language models to follow instructions with human feedback +- **additional groundings if any:** none +- **role:** CORE_METHOD +- **contribution:** Learns scores whose differences reflect human preference between candidate images for the same prompt. +- **rationale:** Without that preference-learning step, the system would not become a ranking metric that can evaluate outputs. The paper explicitly grounds this in a variant of InstructGPT’s reward model objective. +- **evidence_span:** “We train the PickScore scoring function over Pick-a-Pic by combining a CLIP-style model with a variant of In-structGPT’s reward model objective” + +--- + +## Claim C2: Pick-a-Pic + +### Ingredient C2.I1 +- **ingredient_id:** C2.I1 +- **ingredient:** Framework for pairwise human preference collection using prompt-conditioned image comparisons +- **why_structurally_necessary:** The defining property of the dataset is that it records human preferences over generated image pairs for prompts. Without a collection framework that elicits those comparisons, the dataset would not exist in its claimed form. +- **why_not_lower_level_substeps:** This should remain one collection-framework ingredient rather than being split into web app interface details, session flow, or tie-button mechanics. +- **why_not_adjacent_implementation_details:** The structural point is the pairwise human-preference collection framework, not the exact UI implementation. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the framework to collect human preferences on a large scale. +- **rationale:** A framework is necessary to collect human preference choices at scale. The authors implement a web app that lets users generate images and specify preferences, and that workflow is a defining part of the dataset rather than a prior reusable study. +- **evidence_span:** “To address this issue, we create a web app that enables text-to-image users to generate images and specify their preferences.” + +### Ingredient C2.I2 +- **ingredient_id:** C2.I2 +- **ingredient:** Diffusion models for image generation from text +- **why_structurally_necessary:** The dataset consists of preferences over generated images. Without text-to-image generators, there would be no candidate images to compare and annotate. +- **why_not_lower_level_substeps:** This should remain at the level of text-to-image diffusion generation, not be split into individual backbone variants or guidance settings. +- **why_not_adjacent_implementation_details:** The exact backbone mix is less important than the general generation capability used to produce the images. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** High-Resolution Image Synthesis with Latent Diffusion Models +- **additional groundings if any:** none +- **role:** CORE_METHOD +- **contribution:** Provides the diffusion models needed to generate the images in the dataset. +- **rationale:** The dataset requires text-to-image generation backbones to convert user prompts into candidate images for preference annotation. Latent Diffusion Models is the cleanest representative prior study for that capability. +- **evidence_span:** “The images in the dataset were generated by employing multiple backbone models, namely, Stable Diffusion 2.1, Dreamlike Photoreal 2.0, and Stable Diffusion XL variants” + +### Ingredient C2.I3 +- **ingredient_id:** C2.I3 +- **ingredient:** Quality-control and preprocessing protocol for reliable open preference-data collection +- **why_structurally_necessary:** The dataset is released as an open, reusable human-preference resource. Without moderation, filtering, and preprocessing, it would lose one of its defining properties: being a reliable preference dataset rather than a raw interaction log. +- **why_not_lower_level_substeps:** This should remain one protocol-level ingredient rather than being split into NSFW phrase lists, account filtering, banning, or split construction separately. +- **why_not_adjacent_implementation_details:** These local measures are all parts of the same broader quality-control workflow and should not be promoted to standalone ingredients. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the moderation, filtering, and preprocessing workflow that makes the open preference dataset reliable and usable. +- **rationale:** Constructing an open preference dataset requires more than collecting raw comparisons. The paper includes moderation, harmful-content filtering, and preprocessing decisions that are structurally important to the quality of the released resource. This is paper-specific and therefore maps cleanly to `NONE`. +- **evidence_span:** “we closely monitor user activity logs and take action to ban users” + +--- + +## 5. Excluded Tempting Non-Ingredients + +### Claim C1 exclusions +- **Pick-a-Pic web app** + - Excluded because for the scorer claim, the structural dependency is the availability of human preference training data, not the full data-collection workflow. + +- **Preference-based evaluation setup for ranking candidate images by predicted human choice** + - Excluded because this is already captured by the rewritten claim and by the combination of human preference training data plus the reward-model objective. Adding it would double count the same functional role. + +### Claim C2 exclusions +- **List of NSFW phrases** + - Excluded because it is only one local component of the broader quality-control and preprocessing protocol, not a standalone structural ingredient. + +- **Specific diffusion backbone variants** + - Excluded because the structural dependency is text-to-image diffusion generation as a class, not each particular model instance. + +- **PickScore** + - Excluded because the dataset is a prerequisite for the tool, not the reverse. + +--- + +## 6. Candidate Grounding Decisions + +## Claim C1: PickScore + +### Ingredient C1.I1 +- **Candidate:** Learning Transferable Visual Models From Natural Language Supervision +- **decision:** accepted_canonical +- **why:** Provides the pretrained joint text-image architecture used as the basis for prompt-conditioned scoring. + +### Ingredient C1.I2 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The operative human preference dataset is released in the target paper itself. + +### Ingredient C1.I3 +- **Candidate:** Training language models to follow instructions with human feedback +- **decision:** accepted_canonical +- **why:** Provides the reward-model-style objective explicitly adapted by the paper. + +## Claim C2: Pick-a-Pic + +### Ingredient C2.I1 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The pairwise collection framework is implemented in the paper and not cleanly inherited from one prior study. + +### Ingredient C2.I2 +- **Candidate:** High-Resolution Image Synthesis with Latent Diffusion Models +- **decision:** accepted_canonical +- **why:** Cleanest representative of the text-to-image diffusion generation capability used to create the candidate images. + +### Ingredient C2.I3 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The quality-control and preprocessing workflow is a paper-specific protocol rather than one canonical prior study. + +--- + +## 7. Final Grounding Decisions + +## Claim C1 +- **C1.I1** → Learning Transferable Visual Models From Natural Language Supervision → MODEL_INITIALIZATION +- **C1.I2** → NONE → TRAINING_DATA +- **C1.I3** → Training language models to follow instructions with human feedback → CORE_METHOD + +## Claim C2 +- **C2.I1** → NONE → EVALUATION_PROTOCOL +- **C2.I2** → High-Resolution Image Synthesis with Latent Diffusion Models → CORE_METHOD +- **C2.I3** → NONE → EVALUATION_PROTOCOL + +--- + +## 8. Final Rationales And Evidence + +## Claim C1: PickScore + +### C1.I1 +- **ingredient:** Pre-trained vision language model for prompt-conditioned image scoring +- **canonical study:** Learning Transferable Visual Models From Natural Language Supervision +- **role:** MODEL_INITIALIZATION +- **contribution:** Provides the pre-trained text and image representations that are fine-tuned into a human preference score. +- **rationale:** To score a generated image relative to a prompt, the function needs a pretrained joint text–image representation. CLIP is the cleanest representative study for such capability and is the direct architectural basis of PickScore. +- **evidence_span:** “PickScore follows the architecture of CLIP [12]; given a prompt x and an image y, our scoring function s computes a real number by representing x using a transformer text encoder and y using a transformer image encoder as d-dimensional vectors, and returning their inner product” + +### C1.I2 +- **ingredient:** Large-scale dataset of human-preference judgements over text-to-image outputs +- **canonical study:** NONE +- **role:** TRAINING_DATA +- **contribution:** Provides the pairwise user choices needed to fit the scorer. +- **rationale:** The model needs direct supervision about which of two generated images users prefer for a given prompt. This operative dataset is released in the target paper, so the correct grounding is `NONE`. +- **evidence_span:** “We finetune CLIP-H [7] using our framework8 on the Pick-a-Pic training set.” + +### C1.I3 +- **ingredient:** Preference-learning objective to model the reward +- **canonical study:** Training language models to follow instructions with human feedback +- **role:** CORE_METHOD +- **contribution:** Learns scores whose differences reflect human preference between candidate images for the same prompt. +- **rationale:** Without that preference-learning step, the system would not become a ranking metric that can evaluate outputs. The paper explicitly grounds this in a variant of InstructGPT’s reward model objective. +- **evidence_span:** “We train the PickScore scoring function over Pick-a-Pic by combining a CLIP-style model with a variant of In-structGPT’s reward model objective [10].” + +## Claim C2: Pick-a-Pic + +### C2.I1 +- **ingredient:** Framework for pairwise human preference collection using prompt-conditioned image comparisons +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the framework to collect human preferences on a large scale. +- **rationale:** A framework is necessary to collect human preference choices at a large scale. The authors implement a web app to collect this, and that workflow is a defining part of the dataset. +- **evidence_span:** “To address this issue, we create a web app that enables text-to-image users to generate images and specify their preferences.” + +### C2.I2 +- **ingredient:** Diffusion models for image generation from text +- **canonical study:** High-Resolution Image Synthesis with Latent Diffusion Models +- **role:** CORE_METHOD +- **contribution:** Provides the diffusion models needed to generate the images in the dataset. +- **rationale:** The dataset requires text-to-image generation backbones to convert prompts into candidate images for annotation. Latent Diffusion Models is the cleanest representative prior study for that capability. +- **evidence_span:** “The images in the dataset were generated by employing multiple backbone models, namely, Stable Diffusion 2.1, Dreamlike Photoreal 2.0 5 , and Stable Diffusion XL variants [13] while sampling different classifier-free guidance scale values [6].” + +### C2.I3 +- **ingredient:** Quality-control and preprocessing protocol for reliable open preference-data collection +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the moderation, filtering, and preprocessing workflow that makes the open preference dataset reliable and usable. +- **rationale:** Constructing an open preference dataset requires more than collecting raw interactions. The released resource depends on moderation, filtering, and preprocessing steps that preserve data quality and usability. Because this workflow is paper-specific, `NONE` is the right grounding. +- **evidence_span:** “we closely monitor user activity logs and take action to ban users”""" + +EXAMPLE_2 = """ +# EXAMPLE: Offline RL bottleneck finding + +## 1. Cluster Evidence + +**Finding that policy extraction and test-time generalization (rather than just value learning) are the main bottlenecks in offline RL, enabling improved algorithm design via better policy extraction objectives.** + +This original claim identifies a real downstream-used empirical contribution. It should remain a **single finding claim** rather than being split into separate findings about value learning, policy extraction, and test-time generalization, because the paper’s contribution is one integrated diagnosis of what limits offline RL performance in practice. + +--- + +## 2. Claim Split Decision + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Finding +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C1 + +### Why this split is correct + +The downstream reuse is of a single empirical conclusion: that offline RL performance is often limited more by **policy extraction and deployment-time generalization** than by value learning alone. The later algorithm-design implication is a consequence of that diagnosis, not a separate discovery claim. + +--- + +## 3. Rewritten Claims + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Finding +- **rewritten_claim:** **Finding: In offline reinforcement learning, policy extraction and test-time policy generalization often limit performance more than value learning alone, making policy-extraction objective choice a primary determinant of final returns.** +- **why_this_is_atomic:** This claim isolates one empirical result about the source of performance bottlenecks in offline RL. It does not split off the downstream design recommendation as a separate claim. +- **optional cluster_id:** C1 +- **decision:** YES_SUFFICIENT + +--- + +## 4. Minimal Ingredient Set Per Claim + +## Claim C1: Offline RL bottleneck finding + +### Ingredient C1.I1 +- **ingredient_id:** C1.I1 +- **ingredient:** Offline RL methods with decoupled value learning and policy extraction phases +- **why_structurally_necessary:** The finding depends on separating the quality of the learned critic from the quality of the extracted policy. Without a decoupled setup, the paper could not attribute performance differences to policy extraction rather than entangled actor-critic training. +- **why_not_lower_level_substeps:** This should remain one high-level ingredient about the decoupled offline RL setup, not be split into each individual algorithm separately. +- **why_not_adjacent_implementation_details:** Specific architectural or optimization choices inside one method are not the point; the structural dependency is the existence of a method family where value learning and policy extraction can be independently analyzed. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** Offline RL Without Off-Policy Evaluation; Offline Reinforcement Learning with Implicit Q-Learning; Contrastive Learning as Goal-Conditioned Reinforcement Learning +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the offline RL setup in which value learning is trained independently of policy extraction, enabling the paper’s bottleneck analysis. +- **rationale:** To show that policy extraction rather than value learning is often the limiting factor, the paper must analyze settings where the critic can be held fixed while different extraction procedures are compared. That dependence is structurally on a family of decoupled offline RL methods rather than on one single canonical paper, so `NONE` is the faithful canonical choice and the relevant prior methods are better represented as additional studies. +- **evidence_span:** “we focus on offline RL methods with decoupled value and policy training phases” + +### Ingredient C1.I2 +- **ingredient_id:** C1.I2 +- **ingredient:** Behavior-constrained policy-gradient extraction objectives for policies learned from fixed critics +- **why_structurally_necessary:** The finding is not only that some extraction methods underperform, but that behavior-constrained policy-gradient extraction often performs better and scales better than common alternatives under the same learned critic. +- **why_not_lower_level_substeps:** This should remain one extraction-family ingredient rather than being split into every equation or policy-update detail. +- **why_not_adjacent_implementation_details:** Hyperparameters and optimizer details are irrelevant here. The structural ingredient is the extraction objective family itself. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** A Minimalist Approach to Offline Reinforcement Learning +- **additional groundings if any:** Advantage-Weighted Regression: Simple and Scalable Off-Policy Reinforcement Learning; Offline Reinforcement Learning via High-Fidelity Generative Behavior Modeling +- **role:** CORE_METHOD +- **contribution:** Provides the representative behavior-regularized policy-gradient extraction objective that the paper identifies as stronger and more scalable than common alternatives. +- **rationale:** The paper’s empirical conclusion depends on contrasting extraction families and finding that DDPG+BC-style extraction often outperforms widely used value-weighted or sampling-based alternatives. DDPG+BC is the clearest representative study for that stronger extraction family, while AWR and SfBC-style methods serve as additional comparison groundings rather than the canonical representative. +- **evidence_span:** “switching to behavior-constrained policy gradient objectives (e.g., DDPG+BC) often leads to substantial improvements” + +### Ingredient C1.I3 +- **ingredient_id:** C1.I3 +- **ingredient:** Comparative protocol that evaluates extraction objectives across data regimes and deployment generalization settings +- **why_structurally_necessary:** The claim is a comparative empirical finding about bottlenecks and scaling behavior. It requires an analysis protocol that systematically varies data properties and observes how extraction choices affect final performance and generalization. +- **why_not_lower_level_substeps:** This should remain one protocol-level ingredient rather than being split into separate dataset-size sweeps, quality sweeps, and metric definitions. +- **why_not_adjacent_implementation_details:** Individual plots, metrics, or visualization choices are not separate ingredients; they are local parts of the broader comparative protocol. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the systematic empirical protocol that isolates policy extraction and generalization bottlenecks across environments and data regimes. +- **rationale:** A finding about what the main bottleneck is cannot be established from one aggregate return table alone. The paper needs a structured comparative protocol that varies data size, quality, and coverage while comparing extraction methods under shared critics. That protocol is built in the target paper itself, so `NONE` is the correct grounding. +- **evidence_span:** “We use data size, quality, and coverage as levers for systematically controlling their impacts” + +--- + +## 5. Excluded Tempting Non-Ingredients + +### Claim C1 exclusions +- **Offline RL setting with fixed logged data and no online interaction** + - Excluded because it is too broad and field-level. The more precise structural ingredient is the **decoupled offline RL setup** that lets the paper isolate value learning from extraction. + +- **Advantage-weighted regression as its own standalone ingredient** + - Excluded as a separate final ingredient because it is better represented as an **additional grounding** under the broader comparative extraction-objective ingredient, rather than being promoted to its own parallel ingredient. + +- **Standard offline RL benchmark datasets as a separate ingredient** + - Excluded because the finding depends more centrally on the paper’s broader comparative protocol than on one benchmark substrate alone. The datasets are part of that empirical protocol rather than a standalone higher-level ingredient. + +- **Test-time generalization as a separate conceptual ingredient** + - Excluded because it is already captured within the rewritten claim and within the comparative evaluation protocol ingredient. Keeping it separate would over-decompose the finding. + +--- + +## 6. Candidate Grounding Decisions + +## Claim C1: Offline RL bottleneck finding + +### Ingredient C1.I1 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The finding relies on a family of decoupled offline RL methods rather than one single representative study. +- **Candidate:** Offline RL Without Off-Policy Evaluation +- **decision:** accepted_additional +- **why:** Representative prior decoupled offline RL method explicitly listed by the paper. +- **Candidate:** Offline Reinforcement Learning with Implicit Q-Learning +- **decision:** accepted_additional +- **why:** Another representative decoupled method used to motivate the analysis setup. +- **Candidate:** Contrastive Learning as Goal-Conditioned Reinforcement Learning +- **decision:** accepted_additional +- **why:** Included as a representative decoupled method family member rather than the single canonical grounding. + +### Ingredient C1.I2 +- **Candidate:** A Minimalist Approach to Offline Reinforcement Learning +- **decision:** accepted_canonical +- **why:** Cleanest representative study for behavior-constrained policy-gradient extraction, which the paper identifies as the stronger extraction family. +- **Candidate:** Advantage-Weighted Regression: Simple and Scalable Off-Policy Reinforcement Learning +- **decision:** accepted_additional +- **why:** Necessary comparison grounding for the weaker value-weighted extraction family contrasted in the finding. +- **Candidate:** Offline Reinforcement Learning via High-Fidelity Generative Behavior Modeling +- **decision:** accepted_additional +- **why:** Relevant additional comparison study for alternative policy extraction objectives. + +### Ingredient C1.I3 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The comparative data-scaling and deployment-generalization protocol is constructed in the target paper rather than inherited from one prior study. + +--- + +## 7. Final Grounding Decisions + +## Claim C1 +- **C1.I1** → NONE → CONCEPTUAL_FRAMEWORK + - additional: Offline RL Without Off-Policy Evaluation; Offline Reinforcement Learning with Implicit Q-Learning; Contrastive Learning as Goal-Conditioned Reinforcement Learning +- **C1.I2** → A Minimalist Approach to Offline Reinforcement Learning → CORE_METHOD + - additional: Advantage-Weighted Regression: Simple and Scalable Off-Policy Reinforcement Learning; Offline Reinforcement Learning via High-Fidelity Generative Behavior Modeling +- **C1.I3** → NONE → EVALUATION_PROTOCOL + +--- + +## 8. Final Rationales And Evidence + +## Claim C1: Offline RL bottleneck finding + +### C1.I1 +- **ingredient:** Offline RL methods with decoupled value learning and policy extraction phases +- **canonical study:** NONE +- **additional studies:** Offline RL Without Off-Policy Evaluation; Offline Reinforcement Learning with Implicit Q-Learning; Contrastive Learning as Goal-Conditioned Reinforcement Learning +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the offline RL setup in which value learning is trained independently of policy extraction, enabling the paper’s bottleneck analysis. +- **rationale:** To show that policy extraction rather than value learning is often the limiting factor, the paper must analyze settings where the critic can be held fixed while different extraction procedures are compared. That dependence is structurally on a family of decoupled offline RL methods rather than on one single canonical paper, so `NONE` is the faithful canonical choice and the relevant prior methods are better represented as additional studies. +- **evidence_span:** “we focus on offline RL methods with decoupled value and policy training phases” + +### C1.I2 +- **ingredient:** Behavior-constrained policy-gradient extraction objectives for policies learned from fixed critics +- **canonical study:** A Minimalist Approach to Offline Reinforcement Learning +- **additional studies:** Advantage-Weighted Regression: Simple and Scalable Off-Policy Reinforcement Learning; Offline Reinforcement Learning via High-Fidelity Generative Behavior Modeling +- **role:** CORE_METHOD +- **contribution:** Provides the representative behavior-regularized policy-gradient extraction objective that the paper identifies as stronger and more scalable than common alternatives. +- **rationale:** The paper’s empirical conclusion depends on contrasting extraction families and finding that DDPG+BC-style extraction often outperforms widely used value-weighted or sampling-based alternatives. DDPG+BC is the clearest representative study for that stronger extraction family, while AWR and SfBC-style methods serve as additional comparison groundings rather than the canonical representative. +- **evidence_span:** “switching to behavior-constrained policy gradient objectives (e.g., DDPG+BC) often leads to substantial improvements” + +### C1.I3 +- **ingredient:** Comparative protocol that evaluates extraction objectives across data regimes and deployment generalization settings +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the systematic empirical protocol that isolates policy extraction and generalization bottlenecks across environments and data regimes. +- **rationale:** A finding about what the main bottleneck is cannot be established from one aggregate return table alone. The paper needs a structured comparative protocol that varies data size, quality, and coverage while comparing extraction methods under shared critics. That protocol is built in the target paper itself, so `NONE` is the correct grounding. +- **evidence_span:** “We use data size, quality, and coverage as levers for systematically controlling their impacts” +""" + + +EXAMPLE_3 = """ +# EXAMPLE: IndicXTREME / IndicBERT + +## 1. Cluster Evidence + +Based on downstream cluster contributions, the paper introduces two distinct downstream-used contributions: + +- a **benchmark** for multilingual zero-shot evaluation on Indic languages +- a **pretrained multilingual encoder** for Indic NLU + +Downstream evidence shows these are reused differently and should therefore be split into two atomic claims. + +--- + +## 2. Claim Split Decision + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Benchmark +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C5 + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Resource +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C2 + +### Why this split is correct + +The paper contributes two separate artifacts with distinct downstream roles: + +- **IndicXTREME** is reused as a multilingual benchmark for zero-shot evaluation. +- **IndicBERT v2** is reused as a pretrained model for transfer and downstream fine-tuning. + +These have different structural ingredients, so they should not be annotated as one merged claim. + +--- + +## 3. Rewritten Claims + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Benchmark +- **rewritten_claim:** **Benchmark: A multilingual Indic-language evaluation suite, enabling standardized zero-shot assessment of pretrained multilingual models across diverse NLU tasks.** +- **why_this_is_atomic:** This claim isolates the benchmark artifact and its evaluation role, without bundling the pretrained model. +- **optional cluster_id:** C5 +- **decision:** YES_SUFFICIENT + +### Claim C2 +- **claim_id:** C2 +- **artifact_type:** Resource +- **rewritten_claim:** **Resource: A pretrained multilingual encoder for Indic languages, enabling zero-shot transfer and downstream adaptation on Indic NLU tasks.** +- **why_this_is_atomic:** This claim isolates the pretrained encoder as the reused artifact, separate from the benchmark used to evaluate such models. +- **optional cluster_id:** C2 +- **decision:** YES_SUFFICIENT + +--- + +## 4. Minimal Ingredient Set Per Claim + +## Claim C1: Indic benchmark + +### Ingredient C1.I1 +- **ingredient_id:** C1.I1 +- **ingredient:** Multilingual, multi-task, zero-shot evaluation framework for cross-lingual NLU benchmarking +- **why_structurally_necessary:** The benchmark is explicitly framed as a multilingual zero-shot evaluation suite across tasks and languages. Without this benchmark framing, it would not exist in its claimed form as a standardized cross-lingual evaluation resource. +- **why_not_lower_level_substeps:** This should remain one benchmark-paradigm ingredient rather than being split into separate benchmark papers or separate task families. +- **why_not_adjacent_implementation_details:** The structural dependency is the zero-shot multilingual benchmark framing itself, not local benchmark formatting or reporting choices. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** XTREME: A Massively Multilingual Multi-task Benchmark for Evaluating Cross-lingual Generalization +- **additional groundings if any:** XTREME-R: Towards More Challenging and Nuanced Multilingual Evaluation; XGLUE: A New Benchmark Dataset for Cross-lingual Pre-training, Understanding and Generation +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the multilingual zero-shot benchmark framing that the paper adapts to Indic languages. +- **rationale:** To construct a benchmark whose point is standardized zero-shot evaluation across many languages and tasks, the paper depends on an existing multilingual benchmark paradigm. XTREME is the cleanest canonical grounding because it most directly established that framing; XTREME-R and XGLUE are relevant supporting studies but are less direct representatives. +- **evidence_span:** “aims to test the multilingual zero-shot capabilities of pretrained language models” + +### Ingredient C1.I2 +- **ingredient_id:** C1.I2 +- **ingredient:** NLU task datasets and source resources used to assemble the benchmark +- **why_structurally_necessary:** A multi-task benchmark requires the actual upstream task substrates from which its evaluation sets are translated, adapted, corrected, or incorporated. +- **why_not_lower_level_substeps:** This should remain one composite data-source ingredient rather than being split into separate ingredients for each task dataset. +- **why_not_adjacent_implementation_details:** The key dependency is the heterogeneous task substrate itself, not each local adaptation step used for one task. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** Choice of Plausible Alternatives: An Evaluation of Commonsense Causal Reasoning; XNLI: Evaluating Cross-lingual Sentence Representations; IndicXNLI: Evaluating Multilingual Inference for Indian Languages; Naamapadam: A Large-Scale Named Entity Annotated Data for Indic Languages; The Flores-101 Evaluation Benchmark for Low-Resource and Multilingual Machine Translation; MASSIVE: A 1M-Example Multilingual Natural Language Understanding Dataset with 51 Typologically-Diverse Languages +- **role:** DATA_SOURCE +- **contribution:** Supplies the task definitions and raw/source evaluation material that the benchmark translates, subsets, verifies, or incorporates. +- **rationale:** Constructing a multi-task benchmark requires the underlying task substrates that define its evaluation sets. Because these substrates are inherently composite across multiple tasks and datasets, they do not map cleanly to a single prior study, making `NONE` the appropriate canonical choice, with the task-specific sources listed as additional studies. +- **evidence_span:** “We manually translate the COPA test set into 18 Indic languages to create IndicCOPA.” + +### Ingredient C1.I3 +- **ingredient_id:** C1.I3 +- **ingredient:** Human-supervised multilingual curation, translation, and verification protocol +- **why_structurally_necessary:** Human supervision is one of the benchmark’s defining claimed properties. Without it, the benchmark would not exist in the same form as a human-validated evaluation resource. +- **why_not_lower_level_substeps:** This should remain one protocol-level ingredient rather than being split into manual translation, verification, correction, and annotation as separate ingredients. +- **why_not_adjacent_implementation_details:** The structural dependency is the benchmark-wide human-supervision workflow, not each local annotation step. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the manual benchmark-construction process ensuring that all included evaluation sets are created, translated, edited, or verified by humans. +- **rationale:** This ingredient is necessary because the benchmark’s defining property is that all evaluation sets are created or verified under human supervision. Removing it would fundamentally change the nature of the benchmark, reducing it to a weaker, non-human-validated collection of datasets. Because this workflow is benchmark-specific and realized in the target paper, `NONE` is the correct grounding. +- **evidence_span:** “ALL the evaluation sets included in IndicXTREME were created with human supervision” + +--- + +## Claim C2: Indic multilingual encoder + +### Ingredient C2.I1 +- **ingredient_id:** C2.I1 +- **ingredient:** Transformer-based masked language modeling architecture +- **why_structurally_necessary:** The resource is explicitly a BERT-style pretrained encoder. Without the transformer MLM architecture, it would not exist in its claimed form. +- **why_not_lower_level_substeps:** This should remain one architecture-level ingredient rather than being split into encoder layers, attention heads, or hyperparameter choices. +- **why_not_adjacent_implementation_details:** Model size, vocabulary size, and optimizer settings are implementation details; the structural dependency is the BERT-style MLM architecture itself. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding +- **additional groundings if any:** none +- **role:** CORE_METHOD +- **contribution:** Provides the transformer encoder architecture and masked language modeling objective used to pretrain the model. +- **rationale:** To build a multilingual BERT-style encoder, the paper requires a transformer architecture trained with masked language modeling. BERT is the clearest and most faithful canonical grounding for that dependency. +- **evidence_span:** “We use the default hyperparameters of BERT-Base” + +### Ingredient C2.I2 +- **ingredient_id:** C2.I2 +- **ingredient:** Translation Language Modeling objective for cross-lingual alignment +- **why_structurally_necessary:** The paper explicitly uses TLM as a second core objective to improve cross-lingual transfer by aligning languages through parallel data. +- **why_not_lower_level_substeps:** This should remain one objective-level ingredient rather than being split into masking details, parallel-pair formatting, or loss implementation details. +- **why_not_adjacent_implementation_details:** The structural dependency is the TLM objective itself, not specific data-preparation or training-engineering choices around it. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Cross-lingual Language Model Pretraining +- **additional groundings if any:** none +- **role:** CORE_METHOD +- **contribution:** Provides the cross-lingual pretraining objective used to align representations across languages. +- **rationale:** The model is not only trained with MLM; it explicitly incorporates TLM as a core objective for cross-lingual alignment. That objective maps cleanly to XLM as the canonical prior study. +- **evidence_span:** “Translation Language Modeling (Conneau and Lample, 2019, TLM)” + +### Ingredient C2.I3 +- **ingredient_id:** C2.I3 +- **ingredient:** Large-scale Indic-language pretraining text +- **why_structurally_necessary:** A multilingual encoder for Indic languages cannot be pretrained in its claimed form without a large-scale Indic-language text substrate. +- **why_not_lower_level_substeps:** This should remain one training-data ingredient rather than being split into IndicCorp, Wikipedia, OSCAR, Samanantar-derived text, or synthetic translations separately. +- **why_not_adjacent_implementation_details:** Specific corpus components and augmentation choices are subordinate pieces of the broader pretraining-text ingredient and should not be promoted to standalone structural ingredients. +- **necessary:** true +- **from_prior_work:** partially +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** IndicNLPSuite: Monolingual Corpora, Evaluation Benchmarks and Pre-trained Multilingual Language Models for Indian Languages; Samanantar: The Largest Publicly Available Parallel Corpora Collection for 11 Indic Languages +- **role:** TRAINING_DATA +- **contribution:** Provides the large-scale Indic-language text substrate used to pretrain the multilingual encoder. +- **rationale:** Recreating the model requires the actual large-scale Indic-language training text. But the operative pretraining text is structurally composite: it is dominated by corpus material created or assembled in the target paper, while also incorporating prior resources such as IndicCorp lineage and Samanantar-derived text. Because no single prior study cleanly represents that full training substrate, `NONE` is the correct canonical choice, with prior contributing resources listed as additional studies. +- **evidence_span:** “we merge data from IndicCorp v2 with Indic language data from Wikipedia and OSCAR” + +--- + +## 5. Excluded Tempting Non-Ingredients + +### Claim C1 exclusions +- **Machine translation model for auto-translating some source datasets** + - Excluded because it is not structural to the benchmark as a whole and is only used to help construct a small part of one task family rather than defining the benchmark itself. + +- **English training datasets used for zero-shot evaluation** + - Excluded because these are used to apply the benchmark in evaluation setups, not to construct the benchmark artifact itself. + +- **Individual task datasets as separate top-level ingredients** + - Excluded because the benchmark’s task substrate is better represented as one composite DATA_SOURCE ingredient rather than over-decomposed into one ingredient per task. + +### Claim C2 exclusions +- **IndicTrans as a standalone ingredient** + - Excluded because it is a helper tool for generating some synthetic parallel data, not a core structural ingredient of the model in its claimed form. + +- **Temperature-based upsampling / balancing tricks** + - Excluded because these are recipe-level training choices rather than artifact-defining ingredients. + +- **Language ID tokens / tokenizer / WordPiece design** + - Excluded because these are implementation-level configuration details, not part of the minimal structurally sufficient ingredient set. + +- **Samanantar as the canonical grounding for the full training-text ingredient** + - Excluded because it represents only one smaller component of the broader pretraining substrate; assigning it as canonical would distort the true structure of the ingredient. + +--- + +## 6. Candidate Grounding Decisions + +## Claim C1: Indic benchmark + +### Ingredient C1.I1 +- **Candidate:** XTREME: A Massively Multilingual Multi-task Benchmark for Evaluating Cross-lingual Generalization +- **decision:** accepted_canonical +- **why:** Cleanest representative study for the multilingual multi-task zero-shot benchmark framing adapted here. +- **Candidate:** XTREME-R: Towards More Challenging and Nuanced Multilingual Evaluation +- **decision:** accepted_additional +- **why:** Important related multilingual benchmark extension, but not the clearest canonical origin. +- **Candidate:** XGLUE: A New Benchmark Dataset for Cross-lingual Pre-training, Understanding and Generation +- **decision:** accepted_additional +- **why:** Relevant supporting multilingual benchmark study, but less direct as the primary framing. + +### Ingredient C1.I2 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The task substrate is structurally composite across multiple datasets and source resources. +- **Candidate:** Choice of Plausible Alternatives: An Evaluation of Commonsense Causal Reasoning +- **decision:** accepted_additional +- **why:** Source task for IndicCOPA. +- **Candidate:** XNLI: Evaluating Cross-lingual Sentence Representations +- **decision:** accepted_additional +- **why:** Important upstream NLI source task. +- **Candidate:** IndicXNLI: Evaluating Multilingual Inference for Indian Languages +- **decision:** accepted_additional +- **why:** Important Indic-specific benchmark source. +- **Candidate:** Naamapadam: A Large-Scale Named Entity Annotated Data for Indic Languages +- **decision:** accepted_additional +- **why:** Upstream NER task resource. +- **Candidate:** The Flores-101 Evaluation Benchmark for Low-Resource and Multilingual Machine Translation +- **decision:** accepted_additional +- **why:** Upstream retrieval/evaluation source resource. +- **Candidate:** MASSIVE: A 1M-Example Multilingual Natural Language Understanding Dataset with 51 Typologically-Diverse Languages +- **decision:** accepted_additional +- **why:** Upstream intent/slot resource. + +### Ingredient C1.I3 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The human-supervised curation and verification workflow is benchmark-specific and realized in the target paper. + +## Claim C2: Indic multilingual encoder + +### Ingredient C2.I1 +- **Candidate:** BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding +- **decision:** accepted_canonical +- **why:** Cleanest representative study for the transformer MLM architecture used by the model. + +### Ingredient C2.I2 +- **Candidate:** Cross-lingual Language Model Pretraining +- **decision:** accepted_canonical +- **why:** Cleanest representative study for the TLM objective explicitly used by the paper. + +### Ingredient C2.I3 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The pretraining text is dominated by same-paper corpus construction and is structurally composite across several sources. +- **Candidate:** IndicNLPSuite: Monolingual Corpora, Evaluation Benchmarks and Pre-trained Multilingual Language Models for Indian Languages +- **decision:** accepted_additional +- **why:** Important prior Indic corpus lineage that the paper extends. +- **Candidate:** Samanantar: The Largest Publicly Available Parallel Corpora Collection for 11 Indic Languages +- **decision:** accepted_additional +- **why:** Important contributing prior text source, but not representative enough to be canonical for the full pretraining-text ingredient. + +--- + +## 7. Final Grounding Decisions + +## Claim C1 +- **C1.I1** → XTREME: A Massively Multilingual Multi-task Benchmark for Evaluating Cross-lingual Generalization → CONCEPTUAL_FRAMEWORK + - additional: XTREME-R: Towards More Challenging and Nuanced Multilingual Evaluation; XGLUE: A New Benchmark Dataset for Cross-lingual Pre-training, Understanding and Generation +- **C1.I2** → NONE → DATA_SOURCE + - additional: Choice of Plausible Alternatives: An Evaluation of Commonsense Causal Reasoning; XNLI: Evaluating Cross-lingual Sentence Representations; IndicXNLI: Evaluating Multilingual Inference for Indian Languages; Naamapadam: A Large-Scale Named Entity Annotated Data for Indic Languages; The Flores-101 Evaluation Benchmark for Low-Resource and Multilingual Machine Translation; MASSIVE: A 1M-Example Multilingual Natural Language Understanding Dataset with 51 Typologically-Diverse Languages +- **C1.I3** → NONE → EVALUATION_PROTOCOL + +## Claim C2 +- **C2.I1** → BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding → CORE_METHOD +- **C2.I2** → Cross-lingual Language Model Pretraining → CORE_METHOD +- **C2.I3** → NONE → TRAINING_DATA + - additional: IndicNLPSuite: Monolingual Corpora, Evaluation Benchmarks and Pre-trained Multilingual Language Models for Indian Languages; Samanantar: The Largest Publicly Available Parallel Corpora Collection for 11 Indic Languages + +--- + +## 8. Final Rationales And Evidence + +## Claim C1: Indic benchmark + +### C1.I1 +- **ingredient:** Multilingual, multi-task, zero-shot evaluation framework for cross-lingual NLU benchmarking +- **canonical study:** XTREME: A Massively Multilingual Multi-task Benchmark for Evaluating Cross-lingual Generalization +- **additional studies:** XTREME-R: Towards More Challenging and Nuanced Multilingual Evaluation; XGLUE: A New Benchmark Dataset for Cross-lingual Pre-training, Understanding and Generation +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the multilingual zero-shot benchmark framing that the paper adapts to Indic languages. +- **rationale:** To construct a benchmark whose point is standardized zero-shot evaluation across many languages and tasks, the paper depends on an existing multilingual benchmark paradigm. XTREME is the cleanest canonical grounding because it most directly established that framing; XTREME-R and XGLUE are relevant supporting studies but are less direct representatives. +- **evidence_span:** “aims to test the multilingual zero-shot capabilities of pretrained language models” + +### C1.I2 +- **ingredient:** NLU task datasets and source resources used to assemble the benchmark +- **canonical study:** NONE +- **additional studies:** Choice of Plausible Alternatives: An Evaluation of Commonsense Causal Reasoning; XNLI: Evaluating Cross-lingual Sentence Representations; IndicXNLI: Evaluating Multilingual Inference for Indian Languages; Naamapadam: A Large-Scale Named Entity Annotated Data for Indic Languages; The Flores-101 Evaluation Benchmark for Low-Resource and Multilingual Machine Translation; MASSIVE: A 1M-Example Multilingual Natural Language Understanding Dataset with 51 Typologically-Diverse Languages +- **role:** DATA_SOURCE +- **contribution:** Supplies the task definitions and raw/source evaluation material that the benchmark translates, subsets, verifies, or incorporates. +- **rationale:** Constructing a multi-task benchmark requires the underlying task substrates that define its evaluation sets. Because these substrates are inherently composite across multiple tasks and datasets, they do not map cleanly to a single prior study, making `NONE` the appropriate canonical choice, with the task-specific sources listed as additional studies. +- **evidence_span:** “We manually translate the COPA test set into 18 Indic languages to create IndicCOPA.” + +### C1.I3 +- **ingredient:** Human-supervised multilingual curation, translation, and verification protocol +- **canonical study:** NONE +- **role:** EVALUATION_PROTOCOL +- **contribution:** Provides the manual benchmark-construction process ensuring that all included evaluation sets are created, translated, edited, or verified by humans. +- **rationale:** This ingredient is necessary because the benchmark’s defining property is that all evaluation sets are created or verified under human supervision. Removing it would fundamentally change the nature of the benchmark, reducing it to a weaker, non-human-validated collection of datasets. Because this workflow is benchmark-specific and realized in the target paper, `NONE` is the correct grounding. +- **evidence_span:** “ALL the evaluation sets included in IndicXTREME were created with human supervision” + +## Claim C2: Indic multilingual encoder + +### C2.I1 +- **ingredient:** Transformer-based masked language modeling architecture +- **canonical study:** BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding +- **role:** CORE_METHOD +- **contribution:** Provides the transformer encoder architecture and masked language modeling objective used to pretrain the model. +- **rationale:** To build a multilingual BERT-style encoder, the paper requires a transformer architecture trained with masked language modeling. BERT is the clearest and most faithful canonical grounding for that dependency. +- **evidence_span:** “We use the default hyperparameters of BERT-Base” + +### C2.I2 +- **ingredient:** Translation Language Modeling objective for cross-lingual alignment +- **canonical study:** Cross-lingual Language Model Pretraining +- **role:** CORE_METHOD +- **contribution:** Provides the cross-lingual pretraining objective used to align representations across languages. +- **rationale:** The model is not only trained with MLM; it explicitly incorporates TLM as a core objective for cross-lingual alignment. That objective maps cleanly to XLM as the canonical prior study. +- **evidence_span:** “Translation Language Modeling (Conneau and Lample, 2019, TLM)” + +### C2.I3 +- **ingredient:** Large-scale Indic-language pretraining text +- **canonical study:** NONE +- **additional studies:** IndicNLPSuite: Monolingual Corpora, Evaluation Benchmarks and Pre-trained Multilingual Language Models for Indian Languages; Samanantar: The Largest Publicly Available Parallel Corpora Collection for 11 Indic Languages +- **role:** TRAINING_DATA +- **contribution:** Provides the large-scale Indic-language text substrate used to pretrain the multilingual encoder. +- **rationale:** Recreating the model requires the actual large-scale Indic-language training text. But the operative pretraining text is structurally composite: it is dominated by corpus material created or assembled in the target paper, while also incorporating prior resources such as IndicCorp lineage and Samanantar-derived text. Because no single prior study cleanly represents that full training substrate, `NONE` is the correct canonical choice, with prior contributing resources listed as additional studies. +- **evidence_span:** “we merge data from IndicCorp v2 with Indic language data from Wikipedia and OSCAR” +""" + + +EXAMPLE_4 = """ +# EXAMPLE: MassiveDS datastore scaling resource + +## 1. Cluster Evidence + +**The MassiveDS datastore resource** is used in the literature + +The paper also presents a pipeline reordering trick that makes datastore scaling computationally feasible. However, the downstream evidence does **not** show that this pipeline is reused as a separate artifact on its own. Therefore, the correct annotation is **one resource claim**, with the pipeline-reordering idea represented as an important ingredient rather than a separate claim. + +--- + +## 2. Claim Split Decision + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Resource +- **rewritten_claim:** **Dataset: A large-scale multi-domain datastore and experimental pipeline, enabling systematic study and evaluation of retrieval-based language model scaling.** +- **decision:** YES_SUFFICIENT +- **evidence_cluster_id:** C3 + +### Why this split is correct + +The downstream reuse is centered on **one released resource**: the MassiveDS datastore. + +- Later work uses the **datastore itself** as the reusable artifact. +- The paper also introduces a pipeline-reordering idea, but the evidence does **not** show that this pipeline is being reused downstream as a separate artifact. +- Because claim splitting is driven by **distinct downstream-used contributions**, the correct annotation is to keep **one claim**. + +The pipeline reordering is still important, but its role is different: + +- it is a **structural ingredient** that makes the datastore resource computationally feasible in its claimed form +- it is **not** a separate discovery claim unless downstream evidence shows reuse of the pipeline itself + +--- + +## 3. Rewritten Claims + +### Claim C1 +- **claim_id:** C1 +- **artifact_type:** Resource +- **rewritten_claim:** **Resource: A massive open multi-domain datastore that combines general web and domain-specific text, enabling large-scale retrieval-in-context language modeling and datastore-scaling studies.** +- **why_this_is_atomic:** This claim isolates the released datastore resource that downstream work reuses. It does not split off the pipeline-reordering idea as a separate claim because that idea is not independently substantiated as a downstream-used artifact. +- **optional cluster_id:** C3 +- **decision:** YES_SUFFICIENT + +--- + +## 4. Minimal Ingredient Set Per Claim + +## Claim C1: MassiveDS datastore resource + +### Ingredient C1.I1 +- **ingredient_id:** C1.I1 +- **ingredient:** Multi-source open text corpora spanning general web and domain-specific sources +- **why_structurally_necessary:** The datastore is defined by both its scale and its domain diversity. Without large open corpora spanning general web and specialized domains, the released resource would not exist in its claimed form. +- **why_not_lower_level_substeps:** This should remain one composite data-source ingredient rather than being split into separate final ingredients for web data, books, scientific papers, math, biomedical text, code, and other source families. +- **why_not_adjacent_implementation_details:** The structural dependency is the multi-source raw text substrate itself, not every individual source-selection or curation decision. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** RedPajama: An Open Source Recipe to Reproduce LLaMA Training Dataset; PubMed Baseline Repository; S2ORC: The Semantic Scholar Open Research Corpus; peS2o (Pretraining Efficiently on S2ORC) Dataset; NaturalProofs: Mathematical Theorem Proving in Natural Language; Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer +- **role:** DATA_SOURCE +- **contribution:** Supplies the large-scale open text sources used to assemble the datastore across both general web and specialized domains. +- **rationale:** Recreating MassiveDS requires the raw text substrate that gives it both scale and domain coverage. That substrate is inherently composite: the datastore mixes general web data with multiple distinct domain-specific resources. No single prior study faithfully represents that entire source mixture, so `NONE` is the correct canonical grounding, with the main contributing source resources listed as additional studies. +- **evidence_span:** “comprising 1.4 trillion tokens of both general web data and domain specific data” + +### Ingredient C1.I2 +- **ingredient_id:** C1.I2 +- **ingredient:** Pretrained dense retriever for document embedding +- **why_structurally_necessary:** The released datastore is not just raw text; it is an operational retrieval datastore with dense-vector retrieval over passages. Without a pretrained retriever to embed documents, it would not function in its claimed form. +- **why_not_lower_level_substeps:** This should remain one retriever ingredient rather than being split into encoder architecture, embedding dimensionality, or indexing internals. +- **why_not_adjacent_implementation_details:** Retriever hyperparameters and ablation choices are implementation details. The structural dependency is the pretrained dense retriever used to embed the datastore. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Unsupervised Dense Information Retrieval with Contrastive Learning +- **additional groundings if any:** none +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the pretrained retriever used to represent datastore documents as dense vectors for retrieval. +- **rationale:** To reproduce the datastore as a usable retrieval resource, the corpus must be embedded into a searchable dense space. The paper explicitly uses Contriever-MSMARCO for this purpose, making Contriever the cleanest canonical grounding. +- **evidence_span:** “we use CONTRIEVER-MSMARCO (Izacard et al., 2022), which represents every document in the datastore as a dense vector” + +### Ingredient C1.I3 +- **ingredient_id:** C1.I3 +- **ingredient:** Retrieval-in-context language modeling framework +- **why_structurally_necessary:** MassiveDS is introduced specifically to support retrieve-in-context language models rather than as a generic corpus release. Without this framing, the datastore would lose the main capability it is meant to enable. +- **why_not_lower_level_substeps:** This should remain one framework-level ingredient rather than being split into retrieval prompting, concatenation order, or model-specific usage details. +- **why_not_adjacent_implementation_details:** The structural dependency is the inference-time retrieval framework itself, not one particular LM or evaluation setup. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** In-Context Retrieval-Augmented Language Models +- **additional groundings if any:** none +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Defines the retrieval-in-context framing in which documents are retrieved from an external datastore and prepended to LM inputs at inference time. +- **rationale:** The datastore is meaningful here because it supports retrieval-based language models that use retrieved documents in context. Ram et al. is the cleanest canonical grounding because it directly represents the retrieve-in-context setup that the paper focuses on. +- **evidence_span:** “We focus on retrieve-in-context language models (RIC-LMs)” + +### Ingredient C1.I4 +- **ingredient_id:** C1.I4 +- **ingredient:** Data filtering workflow for deduplication, decontamination, and quality control +- **why_structurally_necessary:** The released resource is not merely a raw trillion-token dump; it is a constructed datastore whose quality depends on filtering, deduplication, and decontamination steps. Without this workflow, the datastore would be substantially weaker as a reusable retrieval resource. +- **why_not_lower_level_substeps:** This should remain one filtering-workflow ingredient rather than being split into deduplication, decontamination, and quality filtering as separate ingredients. +- **why_not_adjacent_implementation_details:** Individual thresholds, heuristics, or local filtering choices are not the structural dependency; the key dependency is the broader filtering workflow that makes the datastore usable. +- **necessary:** true +- **from_prior_work:** true +- **maps_cleanly_to_one_study:** true +- **canonical grounding decision:** Dolma: an Open Corpus of Three Trillion Tokens for Language Model Pretraining Research +- **additional groundings if any:** none +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the filtering workflow used to clean the datastore through deduplication, decontamination, and quality controls. +- **rationale:** Building a reusable datastore at this scale requires systematic filtering so the resulting resource is not just large, but also usable and reasonably clean. The paper explicitly ties this step to Dolma, making it the best canonical grounding. +- **evidence_span:** “data filtering, including deduplication, decontamination, and quality filters” + +### Ingredient C1.I5 +- **ingredient_id:** C1.I5 +- **ingredient:** Pipeline reordering to amortize expensive indexing and retrieval across datastore variants +- **why_structurally_necessary:** The datastore is released together with an experimental pipeline that makes large-scale datastore scaling feasible on modest compute. Without this reordering idea, the resource would lose an important defining property: practical support for scalable datastore studies. +- **why_not_lower_level_substeps:** This should remain one pipeline-level ingredient rather than being split into retrieval-overfetching, post-hoc subsampling, and late-stage filtering as separate ingredients. +- **why_not_adjacent_implementation_details:** The structural dependency is the reordering principle itself—running expensive steps once and sharing them—not the local code or systems details used to implement it. +- **necessary:** true +- **from_prior_work:** false +- **maps_cleanly_to_one_study:** false +- **canonical grounding decision:** NONE +- **additional groundings if any:** none +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the core pipeline idea that makes datastore scaling computationally feasible by sharing expensive indexing and retrieval across many variants. +- **rationale:** A key part of the released contribution is not only the datastore’s scale, but the fact that the paper makes large-scale scaling studies feasible by reordering operations so indexing and retrieval are amortized. This idea is introduced in the target paper rather than inherited from a single prior study, so `NONE` is the correct grounding. +- **evidence_span:** “the most expensive ones—indexing and retrieval—are run only once at the start” + +--- + +## 5. Excluded Tempting Non-Ingredients + +### Claim C1 exclusions +- **Pipeline reordering as a separate second claim** + - Excluded because the downstream reuse evidence centers on the datastore resource, not on the pipeline as an independently reused artifact. The pipeline is important, but it is best represented as a structural ingredient of the single datastore claim. + +- **Passage chunking and indexing into fixed-length retrieval units** + - Excluded because this is a lower-level construction step rather than one of the smallest higher-level structural ingredients. It is subordinate to the broader retriever-and-datastore construction pipeline. + +- **FAISS / exact top-K retrieval implementation** + - Excluded because this is too implementation-specific. It supports the pipeline but is not needed at the abstraction level of the minimal structurally sufficient ingredient set. + +- **Separate final ingredients for general web data and domain-specific data** + - Excluded because the higher-level composite ingredient is the multi-source open text substrate spanning both. Splitting them would over-decompose the datastore’s source basis. + +- **Contriever as MODEL_INITIALIZATION instead of IMPLEMENTATION_TOOLING** + - Excluded because its role here is not initializing the target discovery as a model checkpoint, but operationally embedding the datastore documents for retrieval. + +--- + +## 6. Candidate Grounding Decisions + +## Claim C1: MassiveDS datastore resource + +### Ingredient C1.I1 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The datastore’s source substrate is structurally composite across general web and many domain-specific resources. +- **Candidate:** RedPajama: An Open Source Recipe to Reproduce LLaMA Training Dataset +- **decision:** accepted_additional +- **why:** Representative open source for books and other large-scale text components, but not sufficient alone to represent the whole source mixture. +- **Candidate:** PubMed Baseline Repository +- **decision:** accepted_additional +- **why:** Important biomedical source contributing to the domain-specific side of the datastore. +- **Candidate:** S2ORC: The Semantic Scholar Open Research Corpus +- **decision:** accepted_additional +- **why:** Important scientific-paper source for the domain-specific mixture. +- **Candidate:** peS2o (Pretraining Efficiently on S2ORC) Dataset +- **decision:** accepted_additional +- **why:** Additional scientific corpus contribution. +- **Candidate:** NaturalProofs: Mathematical Theorem Proving in Natural Language +- **decision:** accepted_additional +- **why:** Representative mathematical-language source for the domain-specific mixture. +- **Candidate:** Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer +- **decision:** accepted_additional +- **why:** Relevant additional grounding because C4 is explicitly cited as one of the general web data sources. + +### Ingredient C1.I2 +- **Candidate:** Unsupervised Dense Information Retrieval with Contrastive Learning +- **decision:** accepted_canonical +- **why:** Cleanest representative prior study for Contriever, the pretrained dense retriever actually used to embed datastore passages. + +### Ingredient C1.I3 +- **Candidate:** In-Context Retrieval-Augmented Language Models +- **decision:** accepted_canonical +- **why:** Cleanest representative framing for the retrieval-in-context LM setup the datastore is designed to support. + +### Ingredient C1.I4 +- **Candidate:** Dolma: an Open Corpus of Three Trillion Tokens for Language Model Pretraining Research +- **decision:** accepted_canonical +- **why:** Explicitly cited by the paper for the filtering stage and is the cleanest representative for the deduplication / decontamination / quality-control workflow. + +### Ingredient C1.I5 +- **Candidate:** NONE +- **decision:** accepted_none +- **why:** The pipeline-reordering idea is introduced in the target paper itself and is not cleanly attributable to one prior study. + +--- + +## 7. Final Grounding Decisions + +## Claim C1 +- **C1.I1** → NONE → DATA_SOURCE + - additional: RedPajama: An Open Source Recipe to Reproduce LLaMA Training Dataset; PubMed Baseline Repository; S2ORC: The Semantic Scholar Open Research Corpus; peS2o (Pretraining Efficiently on S2ORC) Dataset; NaturalProofs: Mathematical Theorem Proving in Natural Language; Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer +- **C1.I2** → Unsupervised Dense Information Retrieval with Contrastive Learning → IMPLEMENTATION_TOOLING +- **C1.I3** → In-Context Retrieval-Augmented Language Models → CONCEPTUAL_FRAMEWORK +- **C1.I4** → Dolma: an Open Corpus of Three Trillion Tokens for Language Model Pretraining Research → IMPLEMENTATION_TOOLING +- **C1.I5** → NONE → CONCEPTUAL_FRAMEWORK + +--- + +## 8. Final Rationales And Evidence + +## Claim C1: MassiveDS datastore resource + +### C1.I1 +- **ingredient:** Multi-source open text corpora spanning general web and domain-specific sources +- **canonical study:** NONE +- **additional studies:** RedPajama: An Open Source Recipe to Reproduce LLaMA Training Dataset; PubMed Baseline Repository; S2ORC: The Semantic Scholar Open Research Corpus; peS2o (Pretraining Efficiently on S2ORC) Dataset; NaturalProofs: Mathematical Theorem Proving in Natural Language; Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer +- **role:** DATA_SOURCE +- **contribution:** Supplies the large-scale open text sources used to assemble the datastore across both general web and specialized domains. +- **rationale:** Recreating MassiveDS requires the raw text substrate that gives it both scale and domain coverage. That substrate is inherently composite: the datastore mixes general web data with multiple distinct domain-specific resources. No single prior study faithfully represents that entire source mixture, so `NONE` is the correct canonical grounding, with the main contributing source resources listed as additional studies. +- **evidence_span:** “comprising 1.4 trillion tokens of both general web data and domain specific data” + +### C1.I2 +- **ingredient:** Pretrained dense retriever for document embedding +- **canonical study:** Unsupervised Dense Information Retrieval with Contrastive Learning +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the pretrained retriever used to represent datastore documents as dense vectors for retrieval. +- **rationale:** To reproduce the datastore as a usable retrieval resource, the corpus must be embedded into a searchable dense space. The paper explicitly uses Contriever-MSMARCO for this purpose, making Contriever the cleanest canonical grounding. +- **evidence_span:** “we use CONTRIEVER-MSMARCO (Izacard et al., 2022), which represents every document in the datastore as a dense vector” + +### C1.I3 +- **ingredient:** Retrieval-in-context language modeling framework +- **canonical study:** In-Context Retrieval-Augmented Language Models +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Defines the retrieval-in-context framing in which documents are retrieved from an external datastore and prepended to LM inputs at inference time. +- **rationale:** The datastore is meaningful here because it supports retrieval-based language models that use retrieved documents in context. Ram et al. is the cleanest canonical grounding because it directly represents the retrieve-in-context setup that the paper focuses on. +- **evidence_span:** “We focus on retrieve-in-context language models (RIC-LMs)” + +### C1.I4 +- **ingredient:** Data filtering workflow for deduplication, decontamination, and quality control +- **canonical study:** Dolma: an Open Corpus of Three Trillion Tokens for Language Model Pretraining Research +- **role:** IMPLEMENTATION_TOOLING +- **contribution:** Provides the filtering workflow used to clean the datastore through deduplication, decontamination, and quality controls. +- **rationale:** Building a reusable datastore at this scale requires systematic filtering so the resulting resource is not just large, but also usable and reasonably clean. The paper explicitly ties this step to Dolma, making it the best canonical grounding. +- **evidence_span:** “data filtering, including deduplication, decontamination, and quality filters” + +### C1.I5 +- **ingredient:** Pipeline reordering to amortize expensive indexing and retrieval across datastore variants +- **canonical study:** NONE +- **role:** CONCEPTUAL_FRAMEWORK +- **contribution:** Provides the core pipeline idea that makes datastore scaling computationally feasible by sharing expensive indexing and retrieval across many variants. +- **rationale:** A key part of the released contribution is not only the datastore’s scale, but the fact that the paper makes large-scale scaling studies feasible by reordering operations so indexing and retrieval are amortized. This idea is introduced in the target paper rather than inherited from a single prior study, so `NONE` is the correct grounding. +- **evidence_span:** “the most expensive ones—indexing and retrieval—are run only once at the start” +""" diff --git a/src/step_08_annotation/paper_package.py b/src/step_08_annotation/paper_package.py new file mode 100644 index 0000000000000000000000000000000000000000..e2a7a646d646bc58b04e7ea9aa932dd624b05a4d --- /dev/null +++ b/src/step_08_annotation/paper_package.py @@ -0,0 +1,52 @@ +from __future__ import annotations + +import json +from pathlib import Path +from typing import Any, Dict, List + +from common.paper_package import ( + PaperPackage, + _collect_bibliography, + _collect_citation_contexts, + _collect_full_processed_text, + _collect_sections, + _load_json, + _normalize_dict_payload, +) + + +def _collect_all_cluster_evidence(paper_dir: Path) -> List[Dict[str, Any]]: + discovery = _normalize_dict_payload(_load_json(paper_dir / "usage_discovery_from_contributions.json", {})) + clusters = discovery.get("clusters", []) + out = [] + for cluster in clusters: + out.append( + { + "cluster_id": cluster.get("cluster_id"), + "representative_claim": cluster.get("representative_claim") or cluster.get("cluster_title"), + "cluster_title": cluster.get("cluster_title"), + "count": cluster.get("count"), + "cluster_key": cluster.get("cluster_key"), + "claim_indices": cluster.get("claim_indices", []), + "source_cluster_ids": cluster.get("source_cluster_ids", []), + "merge_rationale": cluster.get("merge_rationale"), + } + ) + return out + + +def load_paper_package(paper_dir: str | Path, extracted_claim_override: str | None = None) -> PaperPackage: + paper_dir = Path(paper_dir) + paper_metadata = _normalize_dict_payload(_load_json(paper_dir / "paper_metadata.json", {})) + cluster_evidence = _collect_all_cluster_evidence(paper_dir) + seed = extracted_claim_override or "" + return PaperPackage( + paper_dir=paper_dir, + paper_metadata=paper_metadata, + extracted_discovery_claim=seed, + downstream_cluster_evidence=cluster_evidence, + paper_text=_collect_sections(paper_dir), + full_processed_text=_collect_full_processed_text(paper_dir), + bibliography=_collect_bibliography(paper_dir), + citation_contexts=_collect_citation_contexts(paper_dir), + ) diff --git a/src/step_08_annotation/pipeline.py b/src/step_08_annotation/pipeline.py new file mode 100644 index 0000000000000000000000000000000000000000..b8a9a109795d6738157b22b48684b7d4273af067 --- /dev/null +++ b/src/step_08_annotation/pipeline.py @@ -0,0 +1,256 @@ +from __future__ import annotations + +import json +import traceback +from dataclasses import dataclass +from datetime import datetime, timezone +from pathlib import Path +from typing import Any, Callable, Dict + +from common.model_client import ModelConfig, MultiProviderLLMClient +from common.paper_package import PaperPackage + +from .final_prompts import ( + SYSTEM_TWO_PASS_FORMATTER, + SYSTEM_TWO_PASS_JUDGE, + SYSTEM_TWO_PASS_REASONING, + formatter_prompt, + judge_prompt, + reasoning_prompt, +) +from .schemas import JudgeResult, UIPayload + + +@dataclass +class TwoPassPipelineResult: + run_dir: Path + result: Dict[str, Any] + + +class FormatterStageError(RuntimeError): + def __init__(self, message: str, run_dir: Path): + super().__init__(message) + self.run_dir = run_dir + + +class TwoPassAnnotationPipeline: + def __init__( + self, + *, + provider: str, + model: str, + formatter_model: str | None, + judge_model: str | None, + output_root: Path, + run_label: str | None = None, + annotator_id: str = "llm", + temperature: float = 0.2, + max_tokens: int = 16000, + candidate_count: int = 1, + formatter_max_attempts: int = 3, + include_reference_examples: bool = True, + prompt_profile: str = "full", + progress_callback: Callable[[str], None] | None = None, + ): + self.output_root = output_root + self.annotator_id = annotator_id + self.progress_callback = progress_callback + self.run_label = run_label + self.candidate_count = max(1, candidate_count) + self.formatter_max_attempts = max(1, formatter_max_attempts) + self.include_reference_examples = include_reference_examples + self.prompt_profile = prompt_profile + self.use_judge = self.candidate_count > 1 + stage_models = {} + if formatter_model: + stage_models["two_pass_formatter"] = formatter_model + if judge_model and self.use_judge: + stage_models["two_pass_judge"] = judge_model + self.client = MultiProviderLLMClient( + default_config=ModelConfig( + provider=provider, + model=model, + temperature=temperature, + max_tokens=max_tokens, + ), + stage_models=stage_models, + ) + + def run(self, paper: PaperPackage) -> TwoPassPipelineResult: + run_dir = self._make_run_dir(paper) + payload = { + **paper.to_prompt_payload(), + "paper_dir": paper.paper_dir, + "full_processed_text": self._load_full_processed_text(paper), + } + formatter_config = self.client.config_for_stage("two_pass_formatter") + self._log( + f"[run] paper={paper.paper_dir.name} provider={self.client.default_config.provider} model={self.client.default_config.model_name}" + ) + self._log(f"[run] formatter_model={formatter_config.model_name}") + self._log(f"[run] include_reference_examples={self.include_reference_examples}") + self._log(f"[run] prompt_profile={self.prompt_profile}") + if self.use_judge: + judge_config = self.client.config_for_stage("two_pass_judge") + self._log(f"[run] judge_model={judge_config.model_name}") + else: + self._log("[run] judge_model=disabled (candidate_count=1)") + self._log(f"[run] output={run_dir}") + + reasoning_user_prompt = reasoning_prompt(payload, include_reference_examples=self.include_reference_examples, prompt_profile=self.prompt_profile) + self._write_text(run_dir / "pass_1_reasoning.prompt.txt", reasoning_user_prompt) + self._log( + f"[pass 1] free-form reasoning ({self.candidate_count} candidate{'s' if self.candidate_count != 1 else ''})" + ) + + candidate_texts: list[str] = [] + candidate_paths: list[str] = [] + for index in range(self.candidate_count): + reasoning_text = self.client.generate_text( + stage_name="two_pass_reasoning", + system_prompt=SYSTEM_TWO_PASS_REASONING, + user_prompt=reasoning_user_prompt, + ) + candidate_id = f"candidate_{index + 1}" + candidate_path = run_dir / f"pass_1_reasoning.output.{candidate_id}.md" + self._write_text(candidate_path, reasoning_text) + candidate_texts.append(reasoning_text) + candidate_paths.append(str(candidate_path)) + + selected_candidate_index = 0 + selected_candidate_id = "candidate_1" + selected_reasoning_text = candidate_texts[0] + judge_output_path: Path | None = None + + if self.use_judge: + judge_user_prompt = judge_prompt(payload, candidate_texts) + self._write_text(run_dir / "pass_1_reasoning.judge.prompt.txt", judge_user_prompt) + self._log("[pass 1] candidate judging") + judge_result = self.client.generate_structured( + stage_name="two_pass_judge", + system_prompt=SYSTEM_TWO_PASS_JUDGE, + user_prompt=judge_user_prompt, + response_model=JudgeResult, + ) + judge_output_path = run_dir / "pass_1_reasoning.judge.output.json" + self._write_json(judge_output_path, judge_result.model_dump()) + selected_candidate_index = judge_result.selected_candidate_index + selected_candidate_id = judge_result.selected_candidate_id + selected_reasoning_text = candidate_texts[selected_candidate_index] + + selected_reasoning_path = run_dir / "pass_1_reasoning.selected.md" + self._write_text(selected_reasoning_path, selected_reasoning_text) + + formatter_user_prompt = formatter_prompt(payload, selected_reasoning_text, self.annotator_id) + self._write_text(run_dir / "pass_2_formatter.prompt.txt", formatter_user_prompt) + final_payload: UIPayload | None = None + formatter_attempts: list[dict[str, Any]] = [] + for attempt in range(1, self.formatter_max_attempts + 1): + self._log( + f"[pass 2] strict ui json formatting (attempt {attempt}/{self.formatter_max_attempts})" + ) + try: + final_payload = self.client.generate_structured( + stage_name="two_pass_formatter", + system_prompt=SYSTEM_TWO_PASS_FORMATTER, + user_prompt=formatter_user_prompt, + response_model=UIPayload, + ) + formatter_attempts.append({"attempt": attempt, "status": "success"}) + break + except Exception as exc: + error_text = "".join(traceback.format_exception(exc)).strip() + error_path = run_dir / f"pass_2_formatter.attempt_{attempt}.error.txt" + self._write_text(error_path, error_text) + formatter_attempts.append( + { + "attempt": attempt, + "status": "failed", + "error": str(exc), + "error_path": str(error_path), + } + ) + if attempt < self.formatter_max_attempts: + self._log("[pass 2] formatter failed; retrying formatter only") + + if final_payload is None: + self._write_json(run_dir / "formatter_attempts.json", {"attempts": formatter_attempts}) + raise FormatterStageError( + f"Formatter failed after {self.formatter_max_attempts} attempts; pass 1 outputs kept in {run_dir}", + run_dir, + ) + + self._write_json(run_dir / "formatter_attempts.json", {"attempts": formatter_attempts}) + self._write_json(run_dir / "pass_2_ui_payload.json", final_payload.model_dump()) + + result = { + "paper_id": paper.paper_dir.name, + "paper_dir": str(paper.paper_dir), + "generated_at": datetime.now(timezone.utc).isoformat(), + "reasoner_model": self.client.default_config.model_name, + "formatter_model": formatter_config.model_name, + "judge_model": judge_config.model_name if self.use_judge else None, + "candidate_count": self.candidate_count, + "include_reference_examples": self.include_reference_examples, + "prompt_profile": self.prompt_profile, + "reasoning_candidate_paths": [str(path) for path in candidate_paths], + "selected_reasoning_candidate": selected_candidate_id, + "selected_candidate_index": selected_candidate_index, + "selected_reasoning_path": str(selected_reasoning_path), + "judge_output_path": str(judge_output_path) if judge_output_path is not None else None, + "formatter_attempts": formatter_attempts, + "ui_payload_path": str(run_dir / "pass_2_ui_payload.json"), + "ui_payload": final_payload.model_dump(), + } + self._write_json(run_dir / "run_output.json", result) + self._log("[run] complete") + return TwoPassPipelineResult(run_dir=run_dir, result=result) + + def _make_run_dir(self, paper: PaperPackage) -> Path: + stamp = datetime.now(timezone.utc).strftime("%Y%m%dT%H%M%SZ") + run_name = stamp + if self.run_label: + run_name = f"{self._slugify(self.run_label)}__{stamp}" + run_dir = self.output_root / paper.paper_dir.name / run_name + run_dir.mkdir(parents=True, exist_ok=True) + return run_dir + + def _write_json(self, path: Path, payload: Dict[str, Any]) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps(payload, indent=2, ensure_ascii=True) + "\n") + + def _write_text(self, path: Path, text: str) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(text) + + def _log(self, message: str) -> None: + if self.progress_callback: + self.progress_callback(message) + + @staticmethod + def _slugify(value: str) -> str: + slug = "".join(ch if ch.isalnum() or ch in {"-", "_", "."} else "-" for ch in value.strip()) + slug = "-".join(part for part in slug.split("-") if part) + return slug[:160] or "run" + + def _load_full_processed_text(self, paper: PaperPackage) -> str: + processed_path = paper.paper_dir / "processed_main.tex" + if processed_path.exists(): + try: + return processed_path.read_text() + except Exception: + pass + + sections_dir = paper.paper_dir / "sections" + parts: list[str] = [] + if sections_dir.exists(): + for path in sorted(sections_dir.iterdir()): + if not path.is_file(): + continue + try: + text = path.read_text().strip() + except Exception: + continue + if text: + parts.append(f"[{path.name}]\n{text}") + return "\n\n".join(parts) diff --git a/src/step_08_annotation/schemas.py b/src/step_08_annotation/schemas.py new file mode 100644 index 0000000000000000000000000000000000000000..2351ab56ebb974d767f2f967ef90892925c434a5 --- /dev/null +++ b/src/step_08_annotation/schemas.py @@ -0,0 +1,127 @@ +from __future__ import annotations + +from typing import Any, Dict, List, Literal, Optional + +from pydantic import BaseModel + + +ALLOWED_ARTIFACT_TYPES = ["Resource", "Finding", "Method", "Benchmark", "Dataset", "Tool", "Other"] +ALLOWED_ROLES = [ + "CONCEPTUAL_FRAMEWORK", + "CORE_METHOD", + "DATA_SOURCE", + "MODEL_INITIALIZATION", + "EVALUATION_PROTOCOL", +] + + +class ReasoningCandidate(BaseModel): + study: str + decision: Literal["accepted_canonical", "accepted_additional", "accepted_none", "rejected_candidate"] + why: str + + +class ReasoningIngredient(BaseModel): + ingredient_id: str + ingredient: str + necessary: bool + from_prior_work: bool + maps_cleanly_to_one_study: bool + notes: str + canonical_grounding_decision: Dict[str, Any] + additional_groundings: List[Dict[str, Any]] + candidate_studies_considered: List[ReasoningCandidate] + role: Optional[Literal["CONCEPTUAL_FRAMEWORK", "CORE_METHOD", "DATA_SOURCE", "MODEL_INITIALIZATION", "EVALUATION_PROTOCOL", "IMPLEMENTATION_TOOLING", "TRAINING_DATA"]] = None + contribution: str + rationale: str + evidence_span: str + + +class ReasoningClaim(BaseModel): + claim_id: str + artifact_type: Literal["Resource", "Finding", "Method", "Benchmark", "Dataset", "Tool", "Other"] + rewritten_claim: str + cluster_id: str = "" + decision: Literal["YES_SUFFICIENT", "NO_NOT_DISCOVERY"] = "YES_SUFFICIENT" + notes: str = "" + why_this_is_atomic: str + ingredients: List[ReasoningIngredient] + + +class ReasoningOutput(BaseModel): + original_discovery_claim: str + claim_split_decision: Dict[str, Any] + rewritten_claims: List[ReasoningClaim] + paper_level_notes: str = "" + + +class GroundingRecord(BaseModel): + ref_id: Optional[str] = None + bib_key: Optional[str] = None + paper_id: Optional[str] = None + external_ids: Optional[Dict[str, Any]] = None + ref_title: Optional[str] = None + ref_year: Optional[str] = None + ref_authors: Optional[str] = None + + +class CanonicalAnnotation(BaseModel): + role: Optional[Literal["CONCEPTUAL_FRAMEWORK", "CORE_METHOD", "DATA_SOURCE", "MODEL_INITIALIZATION", "EVALUATION_PROTOCOL", "IMPLEMENTATION_TOOLING", "TRAINING_DATA"]] = None + roles: List[Literal["CONCEPTUAL_FRAMEWORK", "CORE_METHOD", "DATA_SOURCE", "MODEL_INITIALIZATION", "EVALUATION_PROTOCOL", "IMPLEMENTATION_TOOLING", "TRAINING_DATA"]] + contribution: str + rationale: str + evidence_span: str + + +class IngredientPayload(BaseModel): + ingredient_id: str + ingredient: str + canonical_ref_id: str + canonical_grounding: Optional[GroundingRecord] = None + additional_ref_ids: List[str] + additional_groundings: List[GroundingRecord] + canonical_annotation: CanonicalAnnotation + + +class EnablingDiscoveryPayload(GroundingRecord): + ingredient_id: str + ingredient: str + role: Optional[Literal["CONCEPTUAL_FRAMEWORK", "CORE_METHOD", "DATA_SOURCE", "MODEL_INITIALIZATION", "EVALUATION_PROTOCOL", "IMPLEMENTATION_TOOLING", "TRAINING_DATA"]] = None + roles: List[Literal["CONCEPTUAL_FRAMEWORK", "CORE_METHOD", "DATA_SOURCE", "MODEL_INITIALIZATION", "EVALUATION_PROTOCOL", "IMPLEMENTATION_TOOLING", "TRAINING_DATA"]] + contribution: str + rationale: str + evidence_span: str + + +class ClaimPayload(BaseModel): + claim_id: str + text: str + rewritten_claim: str + cluster_id: str = "" + decision: Literal["YES_SUFFICIENT", "NO_NOT_DISCOVERY", "UNCERTAIN"] = "YES_SUFFICIENT" + notes: str = "" + ingredients: List[IngredientPayload] + enabling_discoveries: List[EnablingDiscoveryPayload] + + +class JudgeCandidateScore(BaseModel): + candidate_id: str + candidate_index: int + score: int + assessment: str + + +class JudgeResult(BaseModel): + selected_candidate_index: int + selected_candidate_id: str + selected_reason: str + candidate_scores: List[JudgeCandidateScore] + + +class UIPayload(BaseModel): + target_paper_id: str + target_title: Optional[str] = None + target_year: Optional[int] = None + annotator_id: str + active_claim_id: Optional[str] = None + claims: List[ClaimPayload]