diff --git a/data/gutenburg_sample.txt b/data/gutenburg_sample.txt new file mode 100644 index 0000000..12cde16 --- /dev/null +++ b/data/gutenburg_sample.txt @@ -0,0 +1,387 @@ +The Project Gutenberg eBook, Addison, by William John Courthope + + +This eBook is for the use of anyone anywhere at no cost and with +almost no restrictions whatsoever. You may copy it, give it away or +re-use it under the terms of the Project Gutenberg License included +with this eBook or online at www.gutenberg.org + + + + + +Title: Addison + + +Author: William John Courthope + + + +Release Date: November 27, 2012 [eBook #41496] + +Language: English + +Character set encoding: ISO-8859-1 + + +***START OF THE PROJECT GUTENBERG EBOOK ADDISON*** + + +E-text prepared by the Online Distributed Proofreading Team +(http://www.pgdp.net) from page images generously made available by +Internet Archive (http://archive.org) + + + +Note: Images of the original pages are available through + Internet Archive. See + http://archive.org/details/addison_00cour + + +Transcriber's note: + + Text enclosed by underscores is in italics (_italics_). + + Text enclosed by curly brackets is superscripted + (example: y{e}). + + + + + +English Men of Letters + +Edited by John Morley + +ADDISON + +by + +W. J. COURTHOPE + + + + + + + +Harper & Brothers Publishers +New York and London +1902 + + * * * * * + +ENGLISH MEN OF LETTERS. + +EDITED BY JOHN MORLEY. + + JOHNSON Leslie Stephen. + GIBBON J. C. Morison. + SCOTT R. H. Hutton. + SHELLEY J. A. Symonds. + HUME T. H. Huxley. + GOLDSMITH William Black. + DEFOE William Minto. + BURNS J. C. Shairp. + SPENSER R. W. Church. + THACKERAY Anthony Trollope. + BURKE John Morley. + MILTON Mark Pattison. + HAWTHORNE Henry James, Jr. + SOUTHEY E. Dowden. + CHAUCER A. W. Ward. + BUNYAN J. A. Froude. + COWPER Goldwin Smith. + POPE Leslie Stephen. + BYRON John Nichol. + LOCKE Thomas Fowler. + WORDSWORTH F. Myers. + DRYDEN G. Saintsbury. + LANDOR Sidney Colvin. + DE QUINCEY David Masson. + LAMB Alfred Ainger. + BENTLEY R. C. Jebb. + DICKENS A. W. Ward. + GRAY E. W. Gosse. + SWIFT Leslie Stephen. + STERNE H. D. Traill. + MACAULAY J. Cotter Morison. + FIELDING Austin Dobson. + SHERIDAN Mrs. Oliphant. + ADDISON W. J. Courthope. + BACON R. W. Church. + COLERIDGE H. D. Traill. + SIR PHILIP SIDNEY J. A. Symonds. + KEATS Sidney Colvin. + CARLYLE John Nichol. + +12mo, Cloth, 75 cents per volume. + +_Other volumes in preparation._ + +PUBLISHED BY HARPER & BROTHERS, NEW YORK. + +_Any of the above works will be sent by mail, postage prepaid, to any part +of the United States, Canada, or Mexico, on receipt of the price._ + + * * * * * + + + +CONTENTS. + + + PAGE + + CHAPTER I. + THE STATE OF ENGLISH SOCIETY AND LETTERS + AFTER THE RESTORATION 1 + + CHAPTER II. + ADDISON'S FAMILY AND EDUCATION 21 + + CHAPTER III. + ADDISON ON HIS TRAVELS 38 + + CHAPTER IV. + HIS EMPLOYMENT IN AFFAIRS OF STATE 53 + + CHAPTER V. + THE "TATLER" AND "SPECTATOR" 78 + + CHAPTER VI. + "CATO" 110 + + CHAPTER VII. + ADDISON'S QUARREL WITH POPE 125 + + CHAPTER VIII. + THE LAST YEARS OF HIS LIFE 139 + + CHAPTER IX. + THE GENIUS OF ADDISON 153 + + + + +ADDISON. + + + + +CHAPTER I. + +THE STATE OF ENGLISH SOCIETY AND LETTERS AFTER THE RESTORATION. + + +Of the four English men of letters whose writings most fully embody the +spirit of the eighteenth century, the one who provides the biographer with +the scantiest materials is Addison. In his _Journal to Stella_, his social +verses, and his letters to his friends, we have a vivid picture of those +relations with women and that protracted suffering which invest with such +tragic interest the history of Swift. Pope, by the publication of his own +correspondence, has enabled us, in a way that he never intended, to +understand the strange moral twist which distorted a nature by no means +devoid of noble instincts. Johnson was fortunate in the companionship of +perhaps the best biographer who ever lived. But of the real life and +character of Addison scarcely any contemporary record remains. The formal +narrative prefixed to his works by Tickell is, by that writer's own +admission, little more than a bibliography. Steele, who might have told us +more than any man about his boyhood and his manner of life in London, had +become estranged from his old friend before his death. No writer has +taken the trouble to preserve any account of the wit and wisdom that +enlivened the "little senate" at Button's. His own letters are, as a rule, +compositions as finished as his papers in the _Spectator_. Those features +in his character which excite the greatest interest have been delineated +by the hand of an enemy--an enemy who possessed an unrivalled power of +satirical portrait-painting, and was restrained by no regard for truth +from creating in the public mind such impressions about others as might +serve to heighten the favourable opinion of himself. + +This absence of dramatic incident in Addison's life would lead us +naturally to conclude that he was deficient in the energy and passion +which cause a powerful nature to leave a mark upon its age. Yet such a +judgment would certainly be erroneous. Shy and reserved as he was, the +unanimous verdict of his most illustrious contemporaries is decisive as to +the respect and admiration which he excited among them. The man who could +exert so potent an influence over the mercurial Steele, who could +fascinate the haughty and cynical intellect of Swift, whose conversation, +by the admission of his satirist Pope, had in it something more charming +than that of any other man; of whom it was said that he might have been +chosen king if he wished it; such a man, though to the coarse perception +of Mandeville he might have seemed no more than "a parson in a tye-wig," +can hardly have been deficient in force of character. + +Nor would it have been possible for a writer distinguished by mere +elegance and refinement to leave a lasting impress on the literature and +society of his country. In one generation after another, men representing +opposing elements of rank, class, interest, and taste, have agreed in +acknowledging Addison's extraordinary merits. "Whoever wishes," says +Johnson--at the end of a biography strongly coloured with the +prepossessions of a semi-Jacobite Tory--"whoever wishes to attain an +English style, familiar but not coarse, and elegant but not ostentatious, +must give his days and nights to the volumes of Addison." "Such a mark of +national respect," says Macaulay, the best representative of middle-class +opinion in the present century, speaking of the statue erected to Addison +in Westminster Abbey, "was due to the unsullied statesman, to the +accomplished scholar, to the master of pure English eloquence, to the +consummate painter of life and manners. It was due, above all, to the +great satirist who alone knew how to use ridicule without abusing it; who, +without inflicting a wound, effected a great social reform, and who +reconciled wit and virtue after a long and disastrous separation, during +which wit had been led astray by profligacy, and virtue by fanaticism." + +This verdict of a great critic is accepted by an age to which the grounds +of it are, perhaps, not very apparent. The author of any ideal creation--a +poem, a drama, or a novel--has an imprescriptible property in the fame of +his work. But to harmonise conflicting social elements, to bring order out +of chaos in the sphere of criticism, to form right ways of thinking about +questions of morals, taste, and breeding, are operations of which the +credit, though it is certainly to be ascribed to particular individuals, +is generally absorbed by society itself. Macaulay's eulogy is as just as +it is eloquent, but the pages of the _Spectator_ alone will hardly show +the reader why Addison should be so highly praised for having reconciled +wit with virtue. Nor, looking at him as a critic, will it appear a great +achievement to have pointed out to English society the beauties of +_Paradise Lost_, unless it be remembered that the taste of the preceding +generation still influenced Addison's contemporaries, and that in that +generation Cowley was accounted a greater poet than Milton. + +To estimate Addison at his real value we must regard him as the chief +architect of Public Opinion in the eighteenth century. But here again we +are met by an initial difficulty, because it has become almost a +commonplace of contemporary criticism to represent the eighteenth century +as a period of sheer destruction. It is tacitly assumed by a school of +distinguished philosophical writers that we have arrived at a stage in the +world's history in which it is possible to take a positive and scientific +view of human affairs. As it is of course necessary that from such a +system all belief in the supernatural shall be jealously excluded, it has +not seemed impossible to write the history of Thought itself in the +eighteenth century. And in tracing the course of this supposed continuous +stream it is natural that all the great English writers of the period +should be described as in one way or another helping to pull down, or +vainly to strengthen, the theological barriers erected by centuries of +bigotry against the irresistible tide of enlightened progress. + +It would be of course entirely out of place to discuss here the merits of +this new school of history. Those who consider that, whatever glimpses we +may obtain of the law and order of the universe, man is, as he always has +been and always will be, a mystery to himself, will hardly allow that the +operations of the human spirit can be traced in the dissecting-room. But +it is, in any case, obvious that to treat the great _imaginative_ writers +of any age as if they were only mechanical agents in an evolution of +thought is to do them grave injustice. Such writers are, above all things, +creative. Their first aim is to "show the very age and body of the time +his form and pressure." No work of the eighteenth century, composed in a +consciously destructive spirit, has taken its place among the acknowledged +classics of the language. Even the _Tale of a Tub_ is to be regarded as a +satire upon the aberrations of theologians from right reason, not upon the +principles of Christianity itself. The _Essay on Man_ has, no doubt, +logically a tendency towards Deism, but nobody ever read the poem for the +sake of its philosophy; and it is well known that Pope was much alarmed +when it was pointed out to him that his conclusions might be represented +as incompatible with the doctrines of revealed religion. + +The truth indeed seems to be the exact converse of what is alleged by the +scientific historians. So far from the eighteenth century in England being +an age of destructive analysis, its energies were chiefly devoted to +political, social, and literary reconstruction. Whatever revolution in +faith and manners the English nation had undergone had been the work of +the two preceding centuries, and though the historic foundations of +society remained untouched, the whole form of the superstructure had been +profoundly modified. + + "So tenacious are we," said Burke, towards the close of the last + century, "of our old ecclesiastical modes and fashions of institution + that very little change has been made in them since the fourteenth or + fifteenth centuries, adhering in this particular as in all else to our + old settled maxim never entirely nor at once to depart from antiquity. + We found these institutions on the whole favourable to morality and + discipline, and we thought they were susceptible of amendment without + altering the ground. We thought they were capable of receiving and + meliorating, and, above all, of preserving the accessories of science + and literature as the order of Providence should successively produce + them. And after all, with this Gothic and monkish education (for such + it is the groundwork), we may put in our claim to as ample and early + a share in all the improvements in science, in arts, and in literature + which have illuminated the modern world as any other nation in Europe. + We think one main cause of this improvement was our not despising the + patrimony of knowledge which was left us by our forefathers." + +All this is, in substance, true of our political as well as our +ecclesiastical institutions. And yet, when Burke wrote, the great feudal +and mediæval structure of England had been so transformed by the Wars of +the Roses, the Reformation, the Rebellion, and the Revolution, that its +ancient outlines were barely visible. In so far, therefore, as his words +seem to imply that the social evolution he describes was produced by an +imperceptible and almost mechanical process of national instinct, the +impression they tend to create is entirely erroneous. + +If we have been hitherto saved from such corruption as undermined the +republics of Italy, from the religious wars that so long enfeebled and +divided Germany, and from the Revolution that has severed modern France +from her ancient history, thanks for this are due partly, no doubt, to +favouring conditions of nature and society, but quite as much to the +genius of great individuals who prepared the mind of the nation for the +gradual assimilation of new ideas. Thus Langland and Wycliffe and their +numerous followers, long before the Reformation, had so familiarised the +minds of the people with their ideas of the Christian religion that the +Sovereign was able to assume the Headship of the Church without the shock +of a social convulsion. Fresh feelings and instincts grew up in the hearts +of whole classes of the nation without at first producing any change in +outward habits of life, and even without arousing a sense of their logical +incongruity. These mixed ideas were constantly brought before the +imagination in the works of the poets. Shakespeare abounds with passages +in which, side by side with the old feudal, monarchical, catholic, and +patriotic instincts of Englishmen, we find the sentiments of the Italian +Renaissance. Spenser conveys Puritan doctrines sometimes by the mouth of +shepherds, whose originals he had found in Theocritus and Virgil; +sometimes under allegorical forms derived from books of chivalry and the +ceremonial of the Catholic Church. Milton, the most rigidly Calvinistic of +all the English poets in his opinions, is also the most severely classical +in his style. + +It was the task of Addison to carry on the reconciling traditions of our +literature. It is his praise to have accomplished his task under +conditions far more difficult than any that his predecessors had +experienced. What they had done was to give instinctive and characteristic +expression to the floating ideas of the society about them; what Addison +and his contemporaries did was to found a public opinion by a conscious +effort of reason and persuasion. Before the Civil Wars there had been at +least no visible breach in the principle of Authority in Church and State. +At the beginning of the eighteenth century constituted authority had been +recently overthrown; one king had been beheaded, another had been +expelled; the Episcopalian form of Church Government had been violently +displaced in favour of the Presbyterian, and had been with almost equal +violence restored. Whole classes of the population had been drawn into +opposing camps during the Civil War, and still stood confronting each +other with all the harsh antagonism of sentiment inherited from that +conflict. Such a bare summary alone is sufficient to indicate the nature +of the difficulties Addison had to encounter in his efforts to harmonise +public opinion; but a more detailed examination of the state of society +after the Restoration is required to place in its full light the +extraordinary merits of the success that he achieved. + +There was, to begin with, a vehement opposition between town and country. +In the country the old ideas of Feudalism, modified by circumstances, but +vigorous and deep-rooted, still prevailed. True, the military system of +land-tenure had disappeared with the Restoration, but it was not so with +the relations of life, and the habits of thought and feeling which the +system had created. The features of surviving Feudalism have been +inimitably preserved for us in the character of Sir Roger de Coverley. +Living in the patriarchal fashion, in the midst of tenants and retainers, +who looked up to him as their chief, and for whose welfare and protection +he considered himself responsible, the country gentleman valued above all +things the principle of Loyalty. To the moneyed classes in the towns he +was instinctively opposed; he regarded their interests, both social and +commercial, as contrary to his own; he looked with dislike and suspicion +on the economical principles of government and conduct on which these +classes naturally rely. Even the younger sons of county families had in +Addison's day abandoned the custom, common enough in the feudal times, of +seeking their fortune in trade. Many a Will Wimble now spent his whole +life in the country, training dogs for his neighbours, fishing their +streams, making whips for their young heirs, and even garters for their +wives and daughters.[1] + + + diff --git a/run_eval.py b/run_eval.py index 5e3f390..a6dc5bf 100644 --- a/run_eval.py +++ b/run_eval.py @@ -149,6 +149,11 @@ if __name__ == "__main__": type=float, default=1.0, ) + parser.add_argument( + "--flip_ctx_inp", + action="store_true", + help="Flip the order of context and input", + ) cli_args = vars(parser.parse_args()) # setup_logging(output_dir, debug=os.getenv("DEBUG", False)) diff --git a/src/ctx_to_lora/data/definitions.py b/src/ctx_to_lora/data/definitions.py index 92f8f99..fa20dd5 100644 --- a/src/ctx_to_lora/data/definitions.py +++ b/src/ctx_to_lora/data/definitions.py @@ -202,6 +202,18 @@ DS_KWARGS = { split="train[180:]", ), ), + "squad_negative": dict( + test=dict(path="data/raw_datasets/squad", split="validation"), + ), + "squad_assistant_ctx": dict( + test=dict(path="data/raw_datasets/squad", split="validation"), + ), + "squad_negative_no_passage": dict( + test=dict(path="data/raw_datasets/squad", split="validation"), + ), + "squad_assistant_ctx_no_passage": dict( + test=dict(path="data/raw_datasets/squad", split="validation"), + ), "fw_qa_xl": dict( train=dict( path="parquet", @@ -569,6 +581,30 @@ DS_KWARGS = { # split="train", # ), # ), + "gsm8k_assistant_ctx": dict( + train=dict( + path="openai/gsm8k", + name="main", + split="train[100:]", + ), + validation=dict( + path="openai/gsm8k", + name="main", + split="train[:100]", + ), + test=dict( + path="openai/gsm8k", + name="main", + split="test", + ), + ), + "gsm8k_negative": dict( + test=dict( + path="openai/gsm8k", + name="main", + split="test", + ), + ), "gsm8k": dict( train=dict( path="openai/gsm8k", @@ -603,6 +639,16 @@ DS_KWARGS = { split="test", ), ), + "metaicl": dict( + validation=dict( + path="SakanaAI/metaicl", + split="train[:100]", + ), + test=dict( + path="SakanaAI/metaicl", + split="train", + ), + ), "openmathintx-2": dict( train=dict( path="nvidia/OpenMathInstruct-2", @@ -766,6 +812,10 @@ CLOSED_QA_DATASETS = { "longbench/musique", "hotpot_qa", "squad", + "squad_negative", + "squad_assistant_ctx", + "squad_negative_no_passage", + "squad_assistant_ctx_no_passage", "triviaqa_retrieved", "negative_nq", "ropes", @@ -780,6 +830,10 @@ MULTI_ANSWER_DATASETS = { "longbench/2wikimqa", "longbench/musique", "squad", + "squad_negative", + "squad_assistant_ctx", + "squad_negative_no_passage", + "squad_assistant_ctx_no_passage", "drop", } @@ -793,7 +847,8 @@ for ds_name in list(MULTI_ANSWER_DATASETS): if ds_name.startswith("longbench/"): MULTI_ANSWER_DATASETS.add(f"{ds_name}_e") -GSM8K_DATASETS = {"gsm8k", "gsm8k_fewshot"} +GSM8K_DATASETS = {"gsm8k", "gsm8k_fewshot", "gsm8k_assistant_ctx", "gsm8k_negative"} +MULTI_CHOICE_DATASETS = {"metaicl"} # for training closed qa datasets, e.g., hotpot_qa, squad, etc. CLOSED_QA_INTX_TEMPLATES = [ @@ -829,6 +884,10 @@ EVAL_INTX_TEMPLATES = { "drop": "Answer the following question. Output only the answer and do not output any other words.\n\nQuestion: {input}", # short-ctx extractive qa "squad": "Answer the following question. Output only the answer and do not output any other words.\n\nQuestion: {input}", + "squad_negative": "Answer the following question. Output only the answer and do not output any other words.\n\nQuestion: {input}", + "squad_assistant_ctx": "Answer the following question. Output only the answer and do not output any other words.\n\nQuestion: {input}", + "squad_negative_no_passage": "Answer the following question. Output only the answer and do not output any other words.\n\nQuestion: {input}", + "squad_assistant_ctx_no_passage": "Answer the following question. Output only the answer and do not output any other words.\n\nQuestion: {input}", # retrieved noisy ctx? "triviaqa_retrieved": "Answer the following question. Output only the answer and do not output any other words.\n\nQuestion: {input}", # doc w/ distractors qa diff --git a/src/ctx_to_lora/data/preprocessing_fn.py b/src/ctx_to_lora/data/preprocessing_fn.py index f8973fe..7f816c5 100644 --- a/src/ctx_to_lora/data/preprocessing_fn.py +++ b/src/ctx_to_lora/data/preprocessing_fn.py @@ -148,6 +148,46 @@ def get_preprocessing_fn( "response": sample["answers"]["text"][0], } + elif ds_name == "squad_assistant_ctx": + + def f(sample): + return { + "context": "You are a useful AI assistant.", + "prompt": sample["context"] + "\n\n" + sample["question"], + "response": sample["answers"]["text"][0], + } + + elif ds_name == "squad_negative": + with open("data/gutenburg_sample.txt") as f: + gutenburg_sample = f.read() + + def f(sample): + return { + "context": gutenburg_sample, + "prompt": sample["context"] + "\n\n" + sample["question"], + "response": sample["answers"]["text"][0], + } + + elif ds_name == "squad_negative_no_passage": + with open("data/gutenburg_sample.txt") as f: + gutenburg_sample = f.read() + + def f(sample): + return { + "context": gutenburg_sample, + "prompt": sample["question"], + "response": sample["answers"]["text"][0], + } + + elif ds_name == "squad_assistant_ctx_no_passage": + + def f(sample): + return { + "context": "You are a useful AI assistant.", + "prompt": sample["question"], + "response": sample["answers"]["text"][0], + } + elif ds_name == "drop": def f(sample): @@ -264,6 +304,30 @@ def get_preprocessing_fn( # "prompt": sample["question"] + "\n" + instruction_prompt, # "response": sample["answer"], # } + elif ds_name == "gsm8k_negative": + with open("data/gutenburg_sample.txt") as f: + gutenburg_sample = f.read() + + def f(sample): + return { + "prompt": sample["question"], + "context": gutenburg_sample, + "response": sample["answer"], + } + + elif ds_name == "gsm8k_assistant_ctx": + + def f(sample): + return { + "prompt": sample["question"], + "context": "You are a useful AI assistant.", + "response": sample["answer"], + } + # return { + # "context": sample["question"], + # "prompt": sample["question"] + "\n" + instruction_prompt, + # "response": sample["answer"], + # } elif ds_name == "gsm8k_fewshot": N_FEW_SHOT_EXAMPLES = 5 @@ -306,6 +370,19 @@ def get_preprocessing_fn( "context": few_shot_examples_txt, "response": sample["answer"], } + elif ds_name == "metaicl": + + def f(sample): + prompt = f"You are a classifier. Reply must be of the following format: 'Answer: $LETTER' (without quotes) where LETTER is one of {sample['options']['options']}.\nInput:\n{sample['input']}" + context = "Examples\n" + for example in sample["examples"]: + context += f"Input: {example['input']}\nAnswer: {example['output']}\n\n" + + return { + "context": context, + "prompt": prompt, + "response": sample["output"], + } elif "opencoder-edu" in ds_name: diff --git a/src/ctx_to_lora/data/processing.py b/src/ctx_to_lora/data/processing.py index aa8e404..d94ac9b 100644 --- a/src/ctx_to_lora/data/processing.py +++ b/src/ctx_to_lora/data/processing.py @@ -77,7 +77,7 @@ def load_answers(ds_name, split): def extract_ans(sample): return {"answers": sample["answers"]} - elif ds_name == "squad": + elif "squad" in ds_name: def extract_ans(sample): return {"answers": sample["answers"]["text"]} @@ -102,35 +102,36 @@ def get_ds_kwargs(ds_name: str, split: str) -> dict[str, Any]: skip = slice.split(":")[0] take = slice.split(":")[1] - if ds_name.startswith("self_gen/"): - if ds_name.endswith(".parquet"): - # ds_name is a glob pattern - files = glob(f"{RAW_DATA_DIR}/{ds_name}") - if not files: - raise FileNotFoundError( - f"The provided pattern does not match any files: {RAW_DATA_DIR}/{ds_name}" - ) - else: - # e.g., "self_gen/google/gemma-2-2b-it/pwc" - base_model_name = "/".join(ds_name.split("/")[1:3]) - base_ds = "/".join(ds_name.split("/")[3:]) - if ("[" in split) and split.endswith("]"): - kwargs["split"], slice = split.split("[") - slice = slice.strip("]") - skip = slice.split(":")[0] - if skip: - kwargs["skip"] = int(skip) - take = slice.split(":")[1] - if take: - kwargs["take"] = int(take) - files = glob( - f"{SELF_GEN_DATA_DIR}/{base_model_name}/{base_ds}/{split}/*.parquet" + if ds_name.endswith(".parquet"): + # ds_name is a glob pattern + files = glob(f"{RAW_DATA_DIR}/{ds_name}") + if not files: + raise FileNotFoundError( + f"The provided pattern does not match any files: {RAW_DATA_DIR}/{ds_name}" + ) + kwargs = dict(path="parquet", data_files=files, split="train") + + elif ds_name.startswith("self_gen/"): + # e.g., "self_gen/google/gemma-2-2b-it/pwc" + base_model_name = "/".join(ds_name.split("/")[1:3]) + base_ds = "/".join(ds_name.split("/")[3:]) + if ("[" in split) and split.endswith("]"): + kwargs["split"], slice = split.split("[") + slice = slice.strip("]") + skip = slice.split(":")[0] + if skip: + kwargs["skip"] = int(skip) + take = slice.split(":")[1] + if take: + kwargs["take"] = int(take) + files = glob( + f"{SELF_GEN_DATA_DIR}/{base_model_name}/{base_ds}/{split}/*.parquet" + ) + if not files: + raise FileNotFoundError( + f"No self-gen files found for base model {base_model_name} " + f"in {SELF_GEN_DATA_DIR}/{base_model_name}/{base_ds}/" ) - if not files: - raise FileNotFoundError( - f"No self-gen files found for base model {base_model_name} " - f"in {SELF_GEN_DATA_DIR}/{base_model_name}/{base_ds}/" - ) kwargs = dict(path="parquet", data_files=files, split="train") elif (ds_name not in DS_KWARGS) or (split not in DS_KWARGS[ds_name]): kwargs = dict(path=ds_name, split=split) @@ -220,6 +221,7 @@ def get_tokenized_dataset( set_format: str | None = None, truncate_if_too_long_inp: bool = False, truncate_if_too_long_ctx: bool = False, + flip_ctx_inp: bool = False, ) -> dict[str, Any]: if max_qas_len > 0: assert max_qas_len <= base_model_max_len, ( @@ -282,6 +284,23 @@ def get_tokenized_dataset( "is not present in the dataset." ) logger.info(f"Constructing and tokenizing {ds_name} with {split} split...") + if flip_ctx_inp: + + def squeeze(sample, column): + first_id = sample[column][0] + if check_is_iterable(first_id): + sample[column] = first_id + return sample + + def unsqueeze(sample, column): + sample[column] = [sample[column]] + return sample + + ds = ds.rename_column("context", "context_temp") + ds = ds.rename_column("prompts", "context") + ds = ds.rename_column("context_temp", "prompts") + ds = ds.map(squeeze, fn_kwargs={"column": "context"}) + ds = ds.map(unsqueeze, fn_kwargs={"column": "prompts"}) tokenized_ds = construct_and_tokenize_ctx_qa( ds=ds, @@ -304,6 +323,7 @@ def get_tokenized_dataset( tokenized_ds = tokenized_ds.remove_columns( ["logprobs_vals", "logprobs_indices"] ) + return tokenized_ds diff --git a/src/ctx_to_lora/eval_utils.py b/src/ctx_to_lora/eval_utils.py index 00e64b6..307ab8b 100644 --- a/src/ctx_to_lora/eval_utils.py +++ b/src/ctx_to_lora/eval_utils.py @@ -14,7 +14,7 @@ import numpy as np import pandas as pd import torch import yaml -from datasets import disable_caching +from datasets import disable_caching, load_dataset from peft import get_peft_model from transformers import ( PreTrainedModel, @@ -33,13 +33,19 @@ from ctx_to_lora.data.definitions import ( LONGBENCH_E_TASKS, LONGBENCH_TASKS, MULTI_ANSWER_DATASETS, + MULTI_CHOICE_DATASETS, +) +from ctx_to_lora.data.processing import ( + get_ds_kwargs, + get_tokenized_dataset, + load_answers, ) -from ctx_to_lora.data.processing import get_tokenized_dataset, load_answers from ctx_to_lora.data.self_gen_template import SELF_QA_INTX from ctx_to_lora.metrics import ( LENGTH_BINS, Evaluator, compute_gsm8k_acc, + compute_macro_f1_score, compute_metrics, compute_per_token_acc, compute_perplexity, @@ -195,16 +201,20 @@ def split_string(s: str) -> list[str]: return [x for x in out if x] # remove empty spaces -def f1_score(prediction: str, ground_truth: str) -> float: - """Compute F1 score between prediction and ground truth strings.""" +def f1_score(prediction: str, ground_truth: str) -> tuple[float, float, float]: + """Compute F1 score, precision, and recall between prediction and ground truth strings.""" common = Counter(prediction) & Counter(ground_truth) num_same = sum(common.values()) if num_same == 0: - return 0 - precision = 1.0 * num_same / len(prediction) - recall = 1.0 * num_same / len(ground_truth) - f1 = (2 * precision * recall) / (precision + recall) - return f1 + return 0, 0, 0 + precision = 1.0 * num_same / len(prediction) if len(prediction) > 0 else 0 + recall = 1.0 * num_same / len(ground_truth) if len(ground_truth) > 0 else 0 + f1 = ( + (2 * precision * recall) / (precision + recall) + if (precision + recall) > 0 + else 0 + ) + return f1, precision, recall def compute_qa_f1_score( @@ -214,17 +224,35 @@ def compute_qa_f1_score( Word-level F1 score for evaluating question answering systems. Order of the words does not matter. """ - res = [] + f1_scores = [] + precisions = [] + recalls = [] + for prediction, answers in zip(pred_texts, answers_list): normalized_prediction = normalize_answer(prediction) prediction_words = split_string(normalized_prediction) - score = 0 + best_f1 = 0 + best_precision = 0 + best_recall = 0 + for answer in answers: normalized_label = normalize_answer(answer) label_words = split_string(normalized_label) - score = max(score, f1_score(prediction_words, label_words)) - res.append(score) - return dict(qa_f1_score=np.mean(res)), dict(qa_f1_score=res) + f1, precision, recall = f1_score(prediction_words, label_words) + if f1 > best_f1: + best_f1 = f1 + best_precision = precision + best_recall = recall + + f1_scores.append(best_f1) + precisions.append(best_precision) + recalls.append(best_recall) + + return dict( + qa_f1_score=np.mean(f1_scores), + qa_precision=np.mean(precisions), + qa_recall=np.mean(recalls), + ), dict(qa_f1_score=f1_scores, qa_precision=precisions, qa_recall=recalls) def add_longbench_tasks(ds_names: list[str]) -> None: @@ -251,11 +279,12 @@ def save_generated_text( os.makedirs(split_dir, exist_ok=True) metric_keys = list(per_sample_metric.keys()) - assert len(metric_keys) == 1 + # assert len(metric_keys) == 1 metric_name = metric_keys[0] with open(f"{output_dir}/{split}_generated_text.jsonl", "w") as f: for sample, metric_val in zip(samples, per_sample_metric[metric_name]): - sample[f"{metric_name}"] = metric_val + for metric_name in metric_keys: + sample[f"{metric_name}"] = metric_val f.write(json.dumps(sample) + "\n") @@ -576,6 +605,7 @@ def eval_generation( tokenizer, ctx_tokenizer, datasets, + original_datasets, answers, split, remove_context, @@ -623,6 +653,18 @@ def eval_generation( ) for k, v in gsm8k_acc_metric.items(): eval_result.metrics[f"{split_name}_{k}"] = v + + elif ds_name in MULTI_CHOICE_DATASETS: + print("Computing Multi-Choice Accuracy") + original_ds = original_datasets[ds_name] + multi_choice_acc_metric, per_sample_metric = compute_macro_f1_score( + pred_texts, + label_texts, + [x["options"] for x in original_ds["options"]], + original_ds["task"], + ) + for k, v in multi_choice_acc_metric.items(): + eval_result.metrics[f"{split_name}_{k}"] = v else: rouge_metrics, per_sample_metric = compute_rouge(pred_texts, label_texts) for k, v in rouge_metrics.items(): @@ -912,9 +954,11 @@ def evaluate( add_self_distill_template=use_cd, # only for eval truncate_if_too_long_inp=args.truncate_if_too_long_inp, # only for eval truncate_if_too_long_ctx=args.truncate_if_too_long_ctx, # only for eval + flip_ctx_inp=args.flip_ctx_inp, # only for eval ) datasets = dict() + original_datasets = dict() answers = dict() ds_names = args.val_ds_names if split == "validation" else args.test_ds_names add_longbench_tasks(ds_names) @@ -923,6 +967,11 @@ def evaluate( # handling cases where there are multiple answers if ds_name in MULTI_ANSWER_DATASETS: answers[ds_name] = load_answers(ds_name, split) + if ds_name in MULTI_CHOICE_DATASETS: + ds_kwargs = get_ds_kwargs(ds_name, split) + original_datasets[ds_name] = load_dataset( + **ds_kwargs, trust_remote_code=True + ) print(f"Datasets: {datasets}") print(f"Answers: {answers}") @@ -935,6 +984,10 @@ def evaluate( datasets[ds_name] = ds.select(val_indices) if ds_name in answers: answers[ds_name] = answers[ds_name].select(val_indices) + if ds_name in original_datasets: + original_datasets[ds_name] = original_datasets[ds_name].select( + val_indices + ) max_test_samples_per_ds = getattr(args, "max_test_samples_per_ds", 0) if split == "test" and max_test_samples_per_ds > 0: @@ -944,6 +997,10 @@ def evaluate( datasets[ds_name] = ds.select(test_indices) if ds_name in answers: answers[ds_name] = answers[ds_name].select(test_indices) + if ds_name in original_datasets: + original_datasets[ds_name] = original_datasets[ds_name].select( + test_indices + ) print(f"Datasets: {datasets}") print(f"Answers: {answers}") @@ -1032,6 +1089,7 @@ def evaluate( tokenizer, ctx_tokenizer, {ds_name: ds}, + original_datasets, answers, split, args.remove_context, @@ -1074,6 +1132,7 @@ def run_eval( add_ctx_to_input: bool = False, truncate_if_too_long_inp: bool = False, truncate_if_too_long_ctx: bool = False, + flip_ctx_inp: bool = False, gen_lora_scaling: float = 1, ) -> None: """Run evaluation with the specified parameters.""" @@ -1155,6 +1214,7 @@ def run_eval( args.gen_lora_scaling = gen_lora_scaling args.truncate_if_too_long_inp = truncate_if_too_long_inp args.truncate_if_too_long_ctx = truncate_if_too_long_ctx + args.flip_ctx_inp = flip_ctx_inp setup_logging(args.logging_dir) logger.debug(f"CMD: {' '.join(os.sys.argv)}") diff --git a/src/ctx_to_lora/metrics.py b/src/ctx_to_lora/metrics.py index 00c4d3d..6961ea6 100644 --- a/src/ctx_to_lora/metrics.py +++ b/src/ctx_to_lora/metrics.py @@ -35,10 +35,6 @@ def get_length_bin(length: int): def compute_gsm8k_acc(pred_texts, label_texts): out = defaultdict(list) for pred_text, label_text in zip(pred_texts, label_texts): - if "Answer" not in pred_text: - out["gsm8k_acc"].append(0) - continue - label_ans = float(label_text.split("####")[-1].strip().replace(",", "")) # find all the numbers (including decimals) in the string numbers = re.findall(r"\d+\.?\d*", pred_text) @@ -60,6 +56,225 @@ def compute_gsm8k_acc(pred_texts, label_texts): return out_mean, out +def compute_macro_f1_score( + pred_texts: list[str], + label_texts: list[str], + choices: list[list[str]], # per-sample choices + tasks: list[str], # per-sample task name +): + """ + Compute macro-averaged precision/recall/F1 for multi-class classification. + - Parses labels/preds using: + * Prefer token immediately following case-insensitive "Answer:" / "Answer:" + * Otherwise exact token match of any provided choice (case-insensitive, token-boundary) + - Invalid predictions count as errors for accuracy and as FN for the true class. + - Macro metrics exclude zero-support classes *within each task*. + - Final macro scores are the simple average across tasks (each task has equal weight). + Returns: + out_mean: dict with task-averaged macro metrics + per-task metrics like macro_f1_ + out_details: dict with: + - "accuracy": per-sample correctness (0/1) + - "per_task": raw per-task metric dicts + """ + import re + from collections import defaultdict + + import numpy as np + + assert len(pred_texts) == len(label_texts) == len(choices) == len(tasks), ( + "All input lists must have the same length." + ) + + # --- helpers --- + def _norm_label(text: str) -> str: + m = re.search(r"answer\s*[::]\s*([^\n\r]+)", text, flags=re.I) + cand = (m.group(1) if m else text).strip().rstrip(".") + return cand + + def _extract_choice(text: str, per_sample_choices: list[str]): + # 1) token after "Answer:" + m = re.search(r"answer\s*[::]\s*([A-Za-z0-9\-\._]+)", text, flags=re.I) + if m: + raw = m.group(1).strip().rstrip(".") + for ch in per_sample_choices: + if raw.lower() == ch.lower(): + return ch + # 2) exact token match (case-insensitive) with boundaries + for ch in per_sample_choices: + pattern = rf"(? int: + if val is None: + return -1 + for i, ch in enumerate(choice_list): + if val.lower() == ch.lower(): + return i + return -1 + + def _slugify(name: str) -> str: + # lowercase, replace non-alnum with underscores, collapse repeats, trim edges + s = re.sub(r"[^A-Za-z0-9]+", "_", name.strip().lower()) + s = re.sub(r"_+", "_", s).strip("_") + return s or "task" + + # Normalize ground-truth labels + norm_labels = [_norm_label(lbl) for lbl in label_texts] + + # Group indices by task + by_task = defaultdict(list) + for i, tname in enumerate(tasks): + by_task[tname].append(i) + + per_sample_correct = [0] * len(pred_texts) + task_metrics = {} # name -> dict + + # Compute per-task metrics + for tname, idxs in by_task.items(): + # Union of classes across samples in this task + class_set = [] + seen = set() + for i in idxs: + for ch in choices[i]: + key = ch.lower() + if key not in seen: + seen.add(key) + class_set.append(ch) + + y_true_idx, y_pred_idx = [], [] + for i in idxs: + t_idx = _choice_to_idx(norm_labels[i], class_set) + pred_choice = _extract_choice(pred_texts[i], choices[i]) + p_idx = _choice_to_idx(pred_choice, class_set) + if p_idx == -1: + # debug logging if desired + print( + f"Extracted {pred_choice} from {pred_texts[i]} which is not in {class_set}" + ) + + if t_idx == -1: + print(f"True label {norm_labels[i]} not in class set {class_set}") + + y_true_idx.append(t_idx) + y_pred_idx.append(p_idx) + + per_sample_correct[i] = int(t_idx != -1 and p_idx == t_idx) + + # Per-class stats + n_classes = len(class_set) + f1s, precs, recs, supports = [], [], [], [] + for ci in range(n_classes): + tp = sum(1 for t, p in zip(y_true_idx, y_pred_idx) if t == ci and p == ci) + fp = sum( + 1 + for t, p in zip(y_true_idx, y_pred_idx) + if p == ci and t != ci and p != -1 + ) + fn = sum(1 for t, p in zip(y_true_idx, y_pred_idx) if t == ci and p != ci) + support = sum(1 for t in y_true_idx if t == ci) + supports.append(support) + + precision = tp / (tp + fp) if (tp + fp) > 0 else 0.0 + recall = tp / (tp + fn) if (tp + fn) > 0 else 0.0 + f1 = ( + (2 * precision * recall / (precision + recall)) + if (precision + recall) > 0 + else 0.0 + ) + + precs.append(precision) + recs.append(recall) + f1s.append(f1) + + present = [i for i, s in enumerate(supports) if s > 0] + if present: + task_macro_f1 = float(np.mean([f1s[i] for i in present])) + task_macro_prec = float(np.mean([precs[i] for i in present])) + task_macro_rec = float(np.mean([recs[i] for i in present])) + else: + task_macro_f1 = task_macro_prec = task_macro_rec = 0.0 + + task_acc = ( + float( + np.mean( + [ + 1 if (t != -1 and p == t) else 0 + for t, p in zip(y_true_idx, y_pred_idx) + ] + ) + ) + if idxs + else 0.0 + ) + + # --- valid sample ratios (both label and prediction valid) --- + valid_both = [ + 1 if (t != -1 and p != -1) else 0 for t, p in zip(y_true_idx, y_pred_idx) + ] + task_valid_sample_ratio = float(np.mean(valid_both)) if idxs else 0.0 + + task_metrics[tname] = { + "macro_f1": task_macro_f1, + "macro_precision": task_macro_prec, + "macro_recall": task_macro_rec, + "accuracy": task_acc, + "valid_sample_ratio": task_valid_sample_ratio, + "n_samples": len(idxs), + "n_classes_present": len(present), + } + + # Aggregate across tasks (equal weight per task) + task_names = list(task_metrics.keys()) + if task_names: + macro_f1_over_tasks = float( + np.mean([task_metrics[t]["macro_f1"] for t in task_names]) + ) + macro_prec_over_tasks = float( + np.mean([task_metrics[t]["macro_precision"] for t in task_names]) + ) + macro_rec_over_tasks = float( + np.mean([task_metrics[t]["macro_recall"] for t in task_names]) + ) + acc_over_tasks = float( + np.mean([task_metrics[t]["accuracy"] for t in task_names]) + ) + valid_ratio_task_avg = float( + np.mean([task_metrics[t]["valid_sample_ratio"] for t in task_names]) + ) + else: + macro_f1_over_tasks = macro_prec_over_tasks = macro_rec_over_tasks = ( + acc_over_tasks + ) = valid_ratio_task_avg = 0.0 + + out_mean = { + # Task-averaged (equal task weights) + "macro_f1": macro_f1_over_tasks, + "macro_precision": macro_prec_over_tasks, + "macro_recall": macro_rec_over_tasks, + "accuracy_task_avg": acc_over_tasks, + "valid_sample_ratio_task_avg": valid_ratio_task_avg, + } + + # --- add per-task metrics to out_mean with slugified keys --- + for tname, metrics in task_metrics.items(): + slug = _slugify(tname) + out_mean[f"macro_f1_{slug}"] = metrics["macro_f1"] + # out_mean[f"macro_precision_{slug}"] = metrics["macro_precision"] + # out_mean[f"macro_recall_{slug}"] = metrics["macro_recall"] + # out_mean[f"accuracy_{slug}"] = metrics["accuracy"] + out_mean[f"valid_sample_ratio_{slug}"] = metrics["valid_sample_ratio"] + # out_mean[f"n_samples_{slug}"] = metrics["n_samples"] + # out_mean[f"n_classes_present_{slug}"] = metrics["n_classes_present"] + + out_details = { + "accuracy": per_sample_correct, # per-sample 0/1 + } + + return out_mean, out_details + + def compute_rouge(pred_texts, label_texts): out = defaultdict(list) scorer = rouge_scorer.RougeScorer(["rougeL"], use_stemmer=True) diff --git a/token_length_distributions.pdf b/token_length_distributions.pdf new file mode 100644 index 0000000..a591e67 Binary files /dev/null and b/token_length_distributions.pdf differ diff --git a/webui/app.py b/webui/app.py index 57cf117..e674771 100644 --- a/webui/app.py +++ b/webui/app.py @@ -592,13 +592,38 @@ def chat(): ctx_tokenizer = None with torch.inference_mode(), torch.amp.autocast(str(device)): # Get the contexts and tokenize them - contexts = request.form.getlist("contexts[]") - if not contexts: - contexts = [""] # Use empty context if none provided + raw_contexts = request.form.getlist("contexts[]") + # Build scalers aligned with contexts (default to 1.0) + raw_scalers = request.form.getlist("scalers[]") + # Parse bias scaler (singular), default to 1.0 if missing/invalid + bias_scaler_str = request.form.get("bias_scaler", "1.0") + try: + bias_scaler = float(bias_scaler_str) + except Exception: + bias_scaler = 1.0 - # Ensure we have at least one context + # Clean contexts (drop empty), mirror selection for scalers + pairs = list(zip(raw_contexts, raw_scalers)) + contexts = [] + kept_scalers = [] + for ctx, sc in pairs: + if ctx.strip(): + contexts.append(ctx) + try: + kept_scalers.append(float(sc)) + except Exception: + kept_scalers.append(1.0) + + # If all contexts are empty, use a single empty context and scaler 1.0 + if not contexts: + contexts = [""] + kept_scalers = [1.0] + + # Ensure we have at least one context (safety) if len(contexts) == 1 and not contexts[0].strip(): contexts = [""] + if not kept_scalers: + kept_scalers = [1.0] print( f"Processing {len(contexts)} contexts for response generation" @@ -621,17 +646,18 @@ def chat(): ctx_inputs = process_multiple_contexts( contexts, ctx_tokenizer, - # max_length=( - # modulated_model.ctx_encoder_args.max_ctx_len - # if hasattr(modulated_model.ctx_encoder_args, "max_ctx_len") - # else 512 - # ), ) ctx_ids = ctx_inputs["ctx_ids"].to(device) ctx_attn_mask = ctx_inputs["ctx_attn_mask"].to(device) + scalers_tensor = torch.tensor( + kept_scalers, dtype=torch.float32, device=device + ) + print(f"chat_history: {chat_history}") + print(f"scalers: {scalers_tensor}") + print(f"bias_scaler: {bias_scaler}") # Tokenize the chat history model_inputs = base_tokenizer.apply_chat_template( @@ -655,6 +681,8 @@ def chat(): n_ctx_chunks=torch.tensor( [len(ctx_ids)], device=ctx_ids.device ), + scalers=scalers_tensor, # pass per-context scalers + bias_scaler=bias_scaler, # pass singular bias scaler input_ids=model_inputs, max_new_tokens=512, do_sample=False, diff --git a/webui/templates/visualize.html b/webui/templates/visualize.html index a81623e..d25045e 100644 --- a/webui/templates/visualize.html +++ b/webui/templates/visualize.html @@ -489,6 +489,21 @@ .highlight-diff { background-color: #fff8e1; } + + .context-scaler-controls { + display: flex; + gap: 10px; + align-items: center; + margin-top: 8px; + } + + .context-scaler-controls input[type="range"] { + flex: 1; + } + + .context-scaler-controls input[type="number"] { + width: 90px; + } {% endblock %} @@ -651,10 +666,23 @@ + +
+ + +
+ +
+ + +

A single scalar applied to bias; independent of the number of contexts.

+
+
@@ -806,14 +834,24 @@ let formData = new FormData(); formData.append('message', message); - // Check if we're using hypernetwork and include contexts if (document.getElementById('chat-model-name').textContent.includes('Hypernetwork')) { const contextInputs = document.querySelectorAll('.context-input'); - - // Add all contexts to the form data Array.from(contextInputs).forEach(input => { formData.append('contexts[]', input.value); }); + + const scalerInputs = document.querySelectorAll('.context-scaler'); + Array.from(scalerInputs).forEach(input => { + const v = (input.value || '').trim(); + formData.append('scalers[]', v.length ? v : '1.0'); + }); + + // Include singular bias scaler + const biasScalerInput = document.getElementById('bias-scaler'); + if (biasScalerInput) { + const bv = (biasScalerInput.value || '').trim(); + formData.append('bias_scaler', bv.length ? bv : '1.0'); + } } fetch('/chat', { @@ -845,6 +883,43 @@ }); } + // Utility to wire up slider-number sync inside a context-field + function attachScalerSync(fieldEl) { + const slider = fieldEl.querySelector('.context-scaler-slider'); + const number = fieldEl.querySelector('.context-scaler'); + + if (!slider || !number) return; + + // Keep slider within [-2,2], but allow number to be arbitrary + const clampToSlider = (x) => { + const min = parseFloat(slider.min || '-2'); + const max = parseFloat(slider.max || '2'); + if (isNaN(x)) return 1.0; + return Math.min(Math.max(x, min), max); + }; + + slider.addEventListener('input', () => { + number.value = slider.value; + }); + + number.addEventListener('input', () => { + const parsed = parseFloat(number.value); + if (isNaN(parsed)) { + number.value = '1.0'; + slider.value = '1.0'; + } else { + // do not clamp number; only clamp slider representation + slider.value = clampToSlider(parsed); + } + }); + } + + // Bind initial scaler sync + document.addEventListener('DOMContentLoaded', function () { + const firstField = document.querySelector('.context-fields .context-field'); + if (firstField) attachScalerSync(firstField); + }); + // Handle hypernetwork checkpoint loading document.addEventListener('DOMContentLoaded', function () { const applyHypernetworkButton = document.getElementById('apply-hypernetwork-button'); @@ -1113,10 +1188,18 @@ + +
+ + +
`; contextFieldsContainer.appendChild(newField); + // Wire up slider-number sync for this field + attachScalerSync(newField); + // Enable the remove button if we have more than one context if (contextCount > 1) { removeContextButton.disabled = false; @@ -1128,8 +1211,6 @@ if (contextCount > 1) { contextFieldsContainer.removeChild(contextFieldsContainer.lastChild); contextCount--; - - // Disable the remove button if we're back to just one context if (contextCount === 1) { removeContextButton.disabled = true; }