{"claims": [{"text": "About half of language utterances consist of multi-word expressions", "quote_or_locator": "Abstract and Introduction: 'half of language utterances consist of multi-word expressions' (Biber et al., 2004; Conklin & Schmitt, 2008; Muraki et al., 2023)"}, {"text": "GPT-4o showed a correlation of r = .80 with human concreteness ratings for 62,889 multi-word expressions", "quote_or_locator": "Study 1, Table 1: 'The observed correlation of .8 comes close to the maximum value that can be expected' (N = 62,889)"}, {"text": "The reliability of Muraki et al. concreteness ratings is estimated at r = .84", "quote_or_locator": "Study 1: 'Given that the reliability of the Muraki et al. (2023) ratings is estimated at r = .84, the observed correlation of .8 comes close to the maximum value that can be expected.'"}, {"text": "For opaque idioms (525 most frequent idioms from Hsu 2020), GPT-4o achieved a correlation of r = .56", "quote_or_locator": "Study 1: 'There were 486 idioms with exactly the same wordings in our list of GPT-4o estimates. These gave a correlation of r = .56, considerably lower than the .81 observed in Table 1'"}, {"text": "GPT-4o's valence estimates for single words correlated r = .90 with Warriner et al. human ratings", "quote_or_locator": "Study 2, Table 2: GPT4 column shows correlation of .90 with Warriner ratings (upper half, Pearson)"}, {"text": "GPT-4o's arousal estimates for single words correlated r = .74 with Warriner et al. human ratings", "quote_or_locator": "Study 2, Table 3: GPT4 column shows correlation of .74 with Warriner ratings (upper half, Pearson)"}, {"text": "In validation Study 4, GPT-4o valence estimates correlated r = .95 with newly collected human ratings for 96 multi-word expressions", "quote_or_locator": "Study 4: 'The correlation between GPT estimates and mean participant ratings was r(94) = .95' (Fig. 6)"}, {"text": "In validation Study 5, GPT-4o arousal estimates correlated r = .92 with newly collected human ratings for 96 multi-word expressions", "quote_or_locator": "Study 5: 'The correlation between GPT estimates and mean participant ratings was r(94) = .92' (Fig. 7)"}, {"text": "The authors provide datasets with LLM-generated norms for 126,397 English single words and 63,680 multi-word expressions", "quote_or_locator": "Abstract and Discussion: 'we provide datasets with LLM-generated norms of concreteness, valence, and arousal for 126,397 English single words and 63,680 multi-word expressions'"}, {"text": "Reliability of newly collected valence ratings in Study 4 was omega total = .97", "quote_or_locator": "Study 4: 'Reliability of the valence ratings was omega total = .97'"}, {"text": "Reliability of newly collected arousal ratings in Study 5 was omega total = .96", "quote_or_locator": "Study 5: 'Reliability of the arousal ratings was omega total = .96'"}], "prompt_version": "p1.0", "verdicts": [{"claim": "About half of language utterances consist of multi-word expressions", "verdict": "supported", "evidence": "half of language utterances consist of multi-word expressions (Biber et al., 2004; Conklin & Schmitt, 2008; Muraki et al., 2023)", "note": null}, {"claim": "GPT-4o showed a correlation of r = .80 with human concreteness ratings for 62,889 multi-word expressions", "verdict": "supported", "evidence": "In Study 1, GPT-4o showed strong correlations with human concreteness ratings (r = .8) for multi-word expressions. Only the 62,889 expressions known by the participants of Muraki et al. (2023) were included.", "note": null}, {"claim": "The reliability of Muraki et al. concreteness ratings is estimated at r = .84", "verdict": "supported", "evidence": "Given that the reliability of the Muraki et al. (2023) ratings is estimated at r = .84, the observed correlation of .8 comes close to the maximum value that can be expected.", "note": null}, {"claim": "For opaque idioms (525 most frequent idioms from Hsu 2020), GPT-4o achieved a correlation of r = .56", "verdict": "distorted", "evidence": "There were 486 idioms with exactly the same wordings in our list of GPT-4o estimates. These gave a correlation of r = .56, considerably lower than the .81 observed in Table 1", "note": "The claim states 525 idioms from Hsu 2020, but the source clarifies that only 486 of those 525 idioms had exactly matching wordings in the GPT-4o list. The correlation r = .56 is correct, but the sample size is 486, not 525."}, {"claim": "GPT-4o's valence estimates for single words correlated r = .90 with Warriner et al. human ratings", "verdict": "supported", "evidence": "Table 2 shows GPT4 column with correlation of .90 with Warriner ratings (upper half, Pearson correlation)", "note": null}, {"claim": "GPT-4o's arousal estimates for single words correlated r = .74 with Warriner et al. human ratings", "verdict": "supported", "evidence": "Table 3 shows GPT4 column with correlation of .74 with Warriner ratings (upper half, Pearson correlation)", "note": null}, {"claim": "In validation Study 4, GPT-4o valence estimates correlated r = .95 with newly collected human ratings for 96 multi-word expressions", "verdict": "supported", "evidence": "The correlation between GPT estimates and mean participant ratings was r(94) = .95 (see Fig. 6)", "note": null}, {"claim": "In validation Study 5, GPT-4o arousal estimates correlated r = .92 with newly collected human ratings for 96 multi-word expressions", "verdict": "supported", "evidence": "The correlation between GPT estimates and mean participant ratings was r(94) = .92 (see Fig. 7)", "note": null}, {"claim": "The authors provide datasets with LLM-generated norms for 126,397 English single words and 63,680 multi-word expressions", "verdict": "supported", "evidence": "we provide datasets with LLM-generated norms of concreteness, valence, and arousal for 126,397 English single words and 63,680 multi-word expressions", "note": null}, {"claim": "Reliability of newly collected valence ratings in Study 4 was omega total = .97", "verdict": "supported", "evidence": "Reliability of the valence ratings was omega total = .97", "note": null}, {"claim": "Reliability of newly collected arousal ratings in Study 5 was omega total = .96", "verdict": "supported", "evidence": "Reliability of the arousal ratings was omega total = .96", "note": null}]}