diff --git a/README.md b/README.md index 800fbb4b..54fdeaed 100644 --- a/README.md +++ b/README.md @@ -47,6 +47,7 @@ wikibaseintegrator~=0.11.3 - [Set lemma on lexeme](#set-lemma-on-lexeme) - [Add gloss to a sense on lexeme](#add-gloss-to-a-sense-on-lexeme) - [Add form to a lexeme](#add-form-to-a-lexeme) + - [Add a form or a sense to an existing lexeme](#add-a-form-or-a-sense-to-an-existing-lexeme) - [Other projects](#other-projects) - [Installation](#installation) - [Installation of the development environment](#installation-of-the-development-environment) @@ -313,6 +314,30 @@ form.claims.add(claim) lexeme.forms.add(form) ``` +#### Add a form or a sense to an existing lexeme + +Contrary to `write()`, `write_form()` and `write_sense()` only send the new Form or Sense to the Wikibase instance (with +the `wbladdform` and `wbladdsense` actions), the rest of the lexeme is left untouched. They return the id assigned by the +instance, which is also set on the object. + +`write_forms()` and `write_senses()` add every Form or Sense of the lexeme without an id, one request per Form or Sense. + +From [lexeme_write.ipynb](notebooks/lexeme_write.ipynb) + +```python +lexeme = wbi.lexeme.get('L5') + +form = Form() +form.representations.set(language='en', value='English form representation') +form.grammatical_features = ['Q146786'] +lexeme.write_form(form) # 'L5-F3' + +sense = Sense() +sense.glosses.set(language='en', value='English gloss') +lexeme.senses.add(sense) +lexeme.write_senses() # ['L5-S2'] +``` + ## Other projects ## Here is a list of different projects that use the library: diff --git a/notebooks/lexeme_write.ipynb b/notebooks/lexeme_write.ipynb index c4a4147a..a67f596a 100644 --- a/notebooks/lexeme_write.ipynb +++ b/notebooks/lexeme_write.ipynb @@ -396,6 +396,85 @@ "name": "#%%\n" } } + }, + { + "cell_type": "markdown", + "source": [ + "# Add a form and a sense to an existing lexeme\n", + "\n", + "`write_form()` and `write_sense()` only send the new Form or Sense to the Wikibase instance (with the `wbladdform` and `wbladdsense` actions), the rest of the lexeme is left untouched. The id assigned by the instance is returned and set on the object." + ], + "metadata": { + "collapsed": false, + "pycharm": { + "name": "#%% md\n" + } + } + }, + { + "cell_type": "code", + "execution_count": null, + "outputs": [], + "source": [ + "new_form = Form()\n", + "new_form.representations.set(language='en', value='Another English form representation')\n", + "new_form.grammatical_features = ['Q146786']\n", + "\n", + "lexeme.write_form(new_form)" + ], + "metadata": { + "collapsed": false, + "pycharm": { + "name": "#%%\n" + } + } + }, + { + "cell_type": "code", + "execution_count": null, + "outputs": [], + "source": [ + "new_sense = Sense()\n", + "new_sense.glosses.set(language='en', value='Another English gloss')\n", + "\n", + "lexeme.write_sense(new_sense)" + ], + "metadata": { + "collapsed": false, + "pycharm": { + "name": "#%%\n" + } + } + }, + { + "cell_type": "markdown", + "source": [ + "`write_forms()` and `write_senses()` add every Form or Sense of the lexeme without an id, one request per Form or Sense. Those already on the Wikibase instance are skipped." + ], + "metadata": { + "collapsed": false, + "pycharm": { + "name": "#%% md\n" + } + } + }, + { + "cell_type": "code", + "execution_count": null, + "outputs": [], + "source": [ + "another_sense = Sense()\n", + "another_sense.glosses.set(language='en', value='A third English gloss')\n", + "lexeme.senses.add(another_sense)\n", + "\n", + "lexeme.write_senses()" + ], + "metadata": { + "collapsed": false, + "pycharm": { + "name": "#%%\n" + } + } } ], "metadata": { diff --git a/test/conftest.py b/test/conftest.py index da813660..032c183d 100644 --- a/test/conftest.py +++ b/test/conftest.py @@ -321,6 +321,35 @@ def _apply_claims(self, current: dict, submitted: dict | list, entity_id: str) - return result + def _action_wbladdform(self, params: dict[str, str]) -> dict: + return self._add_lexeme_sub_entity(params, section='forms', response_key='form', id_prefix='F', + defaults={'representations': {}, 'grammaticalFeatures': [], 'claims': {}}) + + def _action_wbladdsense(self, params: dict[str, str]) -> dict: + return self._add_lexeme_sub_entity(params, section='senses', response_key='sense', id_prefix='S', + defaults={'glosses': {}, 'claims': {}}) + + def _add_lexeme_sub_entity(self, params: dict[str, str], section: str, response_key: str, id_prefix: str, defaults: dict) -> dict: + """Shared implementation of the wbladdform and wbladdsense actions.""" + data = json.loads(params['data']) + self.edits.append({'params': params, 'data': data}) + + lexeme_id = params['lexemeId'] + if lexeme_id not in self.entities: + return {'error': {'code': 'not-found', 'info': f'Could not find an entity with the ID "{lexeme_id}".'}, 'servedby': 'mock'} + + lexeme = self.entities[lexeme_id] + existing = lexeme.setdefault(section, []) + + sub_entity = {**defaults, **deepcopy(data)} + # A real instance assigns the id, incrementing a counter never reused after a removal. + sub_entity['id'] = f'{lexeme_id}-{id_prefix}{len(existing) + 1}' + existing.append(sub_entity) + + lexeme['lastrevid'] = lexeme.get('lastrevid', 0) + 1 + + return {response_key: deepcopy(sub_entity), 'lastrevid': lexeme['lastrevid'], 'success': 1} + def _action_query(self, params: dict[str, str]) -> dict: if params.get('meta') == 'tokens': if params.get('type') == 'login': diff --git a/test/integration/test_wikibase_roundtrip.py b/test/integration/test_wikibase_roundtrip.py index 1a7ca72a..796b791d 100644 --- a/test/integration/test_wikibase_roundtrip.py +++ b/test/integration/test_wikibase_roundtrip.py @@ -12,9 +12,10 @@ import pytest from wikibaseintegrator.datatypes import Item, String +from wikibaseintegrator.models import Form, Sense from wikibaseintegrator.wbi_enums import ActionIfExists from wikibaseintegrator.wbi_exceptions import MissingEntityException -from wikibaseintegrator.wbi_helpers import search_entities +from wikibaseintegrator.wbi_helpers import mediawiki_api_call_helper, search_entities pytestmark = pytest.mark.integration @@ -95,6 +96,66 @@ def test_qualifier_and_reference_roundtrip(self, wbi, string_property): fetched.delete(reason='WikibaseIntegrator integration test cleanup') +@pytest.fixture(scope='module') +def lexeme_prerequisites(login): + """The items used as language and lexical category of the lexemes created for this test run.""" + from wikibaseintegrator import WikibaseIntegrator + wbi = WikibaseIntegrator(login=login) + + # The WikibaseLexeme extension is optional on a Wikibase instance + extensions = mediawiki_api_call_helper(data={'action': 'query', 'meta': 'siteinfo', 'siprop': 'extensions'}, allow_anonymous=True)['query']['extensions'] + if not any(extension.get('name') == 'WikibaseLexeme' for extension in extensions): + pytest.skip('The WikibaseLexeme extension is not installed on the instance') + + items = {} + for role in ('language', 'lexical category'): + item = wbi.item.new() + item.labels.set(language='en', value=f'WBI integration test {role} {RUN_ID}') + items[role] = item.write(summary='WikibaseIntegrator integration test setup') + + yield items + + for item in items.values(): + item.delete(reason='WikibaseIntegrator integration test cleanup') + + +class TestLexemeFormsAndSenses: + def test_write_form_and_sense(self, wbi, lexeme_prerequisites): + lexeme = wbi.lexeme.new(language=lexeme_prerequisites['language'].id, lexical_category=lexeme_prerequisites['lexical category'].id) + lexeme.lemmas.set(language='en', value=f'wbi-lemma-{RUN_ID}') + lexeme.write(summary='WikibaseIntegrator integration test: create lexeme') + assert lexeme.id + + # A single Form, with wbladdform + form = Form(grammatical_features=lexeme_prerequisites['lexical category'].id) + form.representations.set(language='en', value=f'wbi-form-{RUN_ID}') + form_id = lexeme.write_form(form) + assert form_id.startswith(f'{lexeme.id}-F') + assert form.id == form_id + + # Several Senses at once with wbladdsense, the one marked as removed is skipped + for gloss in ('first', 'second'): + sense = Sense() + sense.glosses.set(language='en', value=f'{gloss} gloss {RUN_ID}') + lexeme.senses.add(sense) + removed_sense = Sense() + removed_sense.glosses.set(language='en', value='removed gloss') + lexeme.senses.add(removed_sense.remove()) + + sense_ids = lexeme.write_senses() + assert len(sense_ids) == 2 + assert all(sense_id.startswith(f'{lexeme.id}-S') for sense_id in sense_ids) + + # Read back from the instance + fetched = wbi.lexeme.get(lexeme.id) + assert fetched.forms.get(form_id).representations.get('en').value == f'wbi-form-{RUN_ID}' + assert fetched.forms.get(form_id).grammatical_features == [lexeme_prerequisites['lexical category'].id] + assert [fetched.senses.get(sense_id).glosses.get('en').value for sense_id in sense_ids] == [f'first gloss {RUN_ID}', f'second gloss {RUN_ID}'] + assert len(fetched.senses) == 2 + + fetched.delete(reason='WikibaseIntegrator integration test cleanup') + + class TestSearch: def test_search_finds_created_entity(self, wbi, string_property): # The property created for this run must be findable by its label. diff --git a/test/test_entity_lexeme.py b/test/test_entity_lexeme.py index b93d154e..e77e9e3b 100644 --- a/test/test_entity_lexeme.py +++ b/test/test_entity_lexeme.py @@ -3,11 +3,24 @@ """ import pytest -from wikibaseintegrator import WikibaseIntegrator +from wikibaseintegrator import WikibaseIntegrator, datatypes, wbi_login +from wikibaseintegrator.models import Form, Sense wbi = WikibaseIntegrator() +def new_form(representation='pinos', grammatical_features=None): + form = Form(grammatical_features=grammatical_features or ['Q146786']) + form.representations.set(language='es', value=representation) + return form + + +def new_sense(gloss='pine tree'): + sense = Sense() + sense.glosses.set(language='en', value=gloss) + return sense + + @pytest.fixture def lexeme_l5(wikibase): return wikibase.add_fixture('lexeme_L5') @@ -72,3 +85,145 @@ def test_write_roundtrip(self, wikibase, lexeme_l5): assert edit['data']['lexicalCategory'] == 'Q1084' assert written.id == 'L5' assert written.lemmas.get('es') == 'pino' + + +class TestWriteForm: + def test_write_form(self, wikibase, lexeme_l5): + lexeme = wbi.lexeme.get('L5') + form = new_form() + form.claims.add(datatypes.String(prop_nr='P828', value='a claim on a form')) + + assert lexeme.write_form(form, allow_anonymous=True) == 'L5-F3' + + edit = wikibase.last_edit + assert edit['params']['action'] == 'wbladdform' + assert edit['params']['lexemeId'] == 'L5' + assert edit['data']['representations']['es'] == {'language': 'es', 'value': 'pinos'} + assert edit['data']['grammaticalFeatures'] == ['Q146786'] + assert 'P828' in edit['data']['claims'] + # 'add' is a wbeditentity marker, the wbladdform action doesn't expect it + assert 'add' not in edit['data'] + assert 'id' not in edit['data'] + + # The id assigned by the instance is reported back on the local object + assert form.id == 'L5-F3' + assert wikibase.entities['L5']['forms'][-1]['id'] == 'L5-F3' + + def test_write_form_requires_a_lexeme_id(self, wikibase): + with pytest.raises(ValueError, match='Lexeme id'): + wbi.lexeme.new().write_form(new_form(), allow_anonymous=True) + + assert wikibase.requests == [] + + def test_write_form_refuses_an_existing_form(self, wikibase, lexeme_l5): + lexeme = wbi.lexeme.get('L5') + requests_before = len(wikibase.requests) + + # Sending an existing Form to wbladdform would silently duplicate it + with pytest.raises(ValueError, match='L5-F1'): + lexeme.write_form(lexeme.forms.get('L5-F1'), allow_anonymous=True) + + assert len(wikibase.requests) == requests_before + + def test_write_forms_only_writes_the_new_ones(self, wikibase, lexeme_l5): + lexeme = wbi.lexeme.get('L5') + lexeme.forms.add(new_form(representation='pinillo')) + lexeme.forms.add(new_form(representation='pinito')) + + # L5-F1 and L5-F2 already exist on the instance and must be skipped + assert lexeme.write_forms(allow_anonymous=True) == ['L5-F3', 'L5-F4'] + + added = [request for request in wikibase.requests if request.get('action') == 'wbladdform'] + assert len(added) == 2 + assert [form.id for form in lexeme.forms] == ['L5-F1', 'L5-F2', 'L5-F3', 'L5-F4'] + + # Calling it again is a no-op: every Form now has an id + assert lexeme.write_forms(allow_anonymous=True) == [] + + +class TestWriteSense: + def test_write_sense(self, wikibase, lexeme_l5): + lexeme = wbi.lexeme.get('L5') + + assert lexeme.write_sense(new_sense(gloss='a pine'), allow_anonymous=True) == 'L5-S2' + + edit = wikibase.last_edit + assert edit['params']['action'] == 'wbladdsense' + assert edit['params']['lexemeId'] == 'L5' + assert edit['data']['glosses']['en'] == {'language': 'en', 'value': 'a pine'} + assert 'add' not in edit['data'] + assert 'id' not in edit['data'] + + def test_write_sense_requires_a_lexeme_id(self, wikibase): + with pytest.raises(ValueError, match='Lexeme id'): + wbi.lexeme.new().write_sense(new_sense(), allow_anonymous=True) + + assert wikibase.requests == [] + + def test_write_sense_refuses_an_existing_sense(self, wikibase, lexeme_l5): + lexeme = wbi.lexeme.get('L5') + requests_before = len(wikibase.requests) + + with pytest.raises(ValueError, match='L5-S1'): + lexeme.write_sense(lexeme.senses.get('L5-S1'), allow_anonymous=True) + + assert len(wikibase.requests) == requests_before + + def test_write_senses_only_writes_the_new_ones(self, wikibase, lexeme_l5): + lexeme = wbi.lexeme.get('L5') + lexeme.senses.add(new_sense(gloss='a pine')) + + assert lexeme.write_senses(allow_anonymous=True) == ['L5-S2'] + assert [sense.id for sense in lexeme.senses] == ['L5-S1', 'L5-S2'] + assert lexeme.write_senses(allow_anonymous=True) == [] + + def test_write_sense_refuses_a_removed_sense(self, wikibase, lexeme_l5): + lexeme = wbi.lexeme.get('L5') + requests_before = len(wikibase.requests) + + # The 'remove' marker would be sent to wbladdsense, which doesn't expect it + with pytest.raises(ValueError, match='removed'): + lexeme.write_sense(new_sense().remove(), allow_anonymous=True) + + assert len(wikibase.requests) == requests_before + + def test_write_senses_skips_the_removed_ones(self, wikibase, lexeme_l5): + lexeme = wbi.lexeme.get('L5') + lexeme.senses.add(new_sense(gloss='a removed pine').remove()) + lexeme.senses.add(new_sense(gloss='a pine')) + + assert lexeme.write_senses(allow_anonymous=True) == ['L5-S2'] + assert wikibase.last_edit['data']['glosses']['en']['value'] == 'a pine' + assert len(wikibase.entities['L5']['senses']) == 2 + + +class TestWriteSubEntityAuthentication: + def test_anonymous_write_is_refused_without_a_login(self, wikibase, lexeme_l5): + lexeme = wbi.lexeme.get('L5') + + # Default allow_anonymous is False, an explicit login is required + with pytest.raises(ValueError, match='allow_anonymous'): + lexeme.write_form(new_form()) + + with pytest.raises(ValueError, match='allow_anonymous'): + lexeme.write_sense(new_sense()) + + def test_login_and_is_bot_are_taken_from_the_api_instance(self, wikibase, lexeme_l5): + wikibase.valid_credentials['TestUser@bot'] = 'botpassword' + login = wbi_login.Login(user='TestUser@bot', password='botpassword') + authenticated_wbi = WikibaseIntegrator(login=login, is_bot=True) + + lexeme = authenticated_wbi.lexeme.get('L5') + assert lexeme.write_form(new_form(), allow_anonymous=False) == 'L5-F3' + + params = wikibase.last_edit['params'] + assert params['token'] == wikibase.csrf_token + assert 'bot' in params + + def test_login_can_be_passed_explicitly(self, wikibase, lexeme_l5): + wikibase.valid_credentials['TestUser@bot'] = 'botpassword' + login = wbi_login.Login(user='TestUser@bot', password='botpassword') + + lexeme = wbi.lexeme.get('L5') + assert lexeme.write_sense(new_sense(), login=login) == 'L5-S2' + assert wikibase.last_edit['params']['token'] == wikibase.csrf_token diff --git a/test/test_models.py b/test/test_models.py index 366af3ca..27448a48 100644 --- a/test/test_models.py +++ b/test/test_models.py @@ -11,7 +11,7 @@ from wikibaseintegrator import WikibaseIntegrator, datatypes from wikibaseintegrator.datatypes import Item, MonolingualText, String from wikibaseintegrator.entities import ItemEntity -from wikibaseintegrator.models import Claims, Descriptions, Form, Qualifiers +from wikibaseintegrator.models import Claims, Descriptions, Form, Forms, Qualifiers, Sense, Senses from wikibaseintegrator.wbi_enums import ActionIfExists, WikibaseSnakType from .conftest import load_fixture @@ -369,6 +369,85 @@ def test_grammatical_features_setter(self): with pytest.raises(TypeError): Form(grammatical_features=True) + def test_forms_are_iterable(self): + forms = Forms() + forms.add(Form(form_id='L5-F1')) + forms.add(Form(form_id='L5-F2')) + + assert [form.id for form in forms] == ['L5-F1', 'L5-F2'] + # An iterator must be exhausted by the first loop, the container itself must not + assert [form.id for form in forms] == ['L5-F1', 'L5-F2'] + assert list(iter(forms)) == forms.forms + + def test_equality_compares_content(self): + def build_form(form_id=None, representation='pinos', feature='Q146786', value='a claim'): + form = Form(form_id=form_id, grammatical_features=feature) + form.representations.set(language='es', value=representation) + form.claims.add(datatypes.String(prop_nr='P828', value=value)) + return form + + # The id is assigned by the instance and isn't part of the comparison + assert build_form() == build_form(form_id='L5-F1') + + assert build_form() != build_form(representation='pino') + assert build_form() != build_form(feature='Q110786') + assert build_form() != build_form(value='another claim') + assert Form() == Form() + + def test_equality_with_unrelated_types(self): + assert Form() != 'L5-F1' + assert Form() != Sense() + assert Form().__eq__(None) is NotImplemented # pylint: disable=unnecessary-dunder-call + + def test_forms_are_hashable(self): + form = Form(grammatical_features='Q146786') + form.representations.set(language='es', value='pinos') + same_form = Form(form_id='L5-F1', grammatical_features='Q146786') + same_form.representations.set(language='es', value='pinos') + + assert hash(form) == hash(same_form) + assert len({form, same_form, Form()}) == 2 + + +class TestSenses: + def test_senses_are_iterable(self): + senses = Senses() + senses.add(Sense(sense_id='L5-S1')) + senses.add(Sense(sense_id='L5-S2')) + + assert [sense.id for sense in senses] == ['L5-S1', 'L5-S2'] + assert [sense.id for sense in senses] == ['L5-S1', 'L5-S2'] + assert list(iter(senses)) == senses.senses + + def test_equality_compares_content(self): + def build_sense(sense_id=None, gloss='a gloss', value='a claim'): + sense = Sense(sense_id=sense_id) + sense.glosses.set(language='en', value=gloss) + sense.claims.add(datatypes.String(prop_nr='P828', value=value)) + return sense + + # The id is assigned by the instance and isn't part of the comparison + assert build_sense() == build_sense(sense_id='L5-S1') + + assert build_sense() != build_sense(gloss='another gloss') + assert build_sense() != build_sense(value='another claim') + assert Sense() == Sense() + + def test_equality_with_unrelated_types(self): + # Comparing with anything else must return False instead of raising + assert Sense() != 'L5-S1' + assert Sense() != Form() + assert Sense().__eq__(None) is NotImplemented # pylint: disable=unnecessary-dunder-call + + def test_senses_are_hashable(self): + sense = Sense() + sense.glosses.set(language='en', value='pine tree') + same_sense = Sense(sense_id='L5-S1') + same_sense.glosses.set(language='en', value='pine tree') + + assert hash(sense) == hash(same_sense) + assert len({sense, same_sense, Sense()}) == 2 + class TestTermsEntity: def test_term_setters_validate_type_across_entities(self): diff --git a/wikibaseintegrator/entities/lexeme.py b/wikibaseintegrator/entities/lexeme.py index 054ead91..00325d9e 100644 --- a/wikibaseintegrator/entities/lexeme.py +++ b/wikibaseintegrator/entities/lexeme.py @@ -4,10 +4,12 @@ from typing import Any from wikibaseintegrator.entities.baseentity import BaseEntity -from wikibaseintegrator.models.forms import Forms +from wikibaseintegrator.models.forms import Form, Forms from wikibaseintegrator.models.lemmas import Lemmas -from wikibaseintegrator.models.senses import Senses +from wikibaseintegrator.models.senses import Sense, Senses from wikibaseintegrator.wbi_config import config +from wikibaseintegrator.wbi_helpers import lexeme_add_form, lexeme_add_sense +from wikibaseintegrator.wbi_login import _Login class LexemeEntity(BaseEntity): @@ -168,3 +170,86 @@ def write(self, **kwargs: Any) -> LexemeEntity: """ json_data = super()._write(data=self.get_json(), **kwargs) return self.from_json(json_data=json_data) + + def write_form(self, form: Form, login: _Login | None = None, allow_anonymous: bool = False, is_bot: bool | None = None, **kwargs: Any) -> str: + """ + Add a single Form to the Lexeme with the wbladdform action. + + Contrary to write(), only the Form is sent to the Wikibase instance, the rest of the Lexeme is left untouched. + + :param form: The Form to add. It must be a new Form, without an id. + :param login: A login instance + :param allow_anonymous: Force a check if the query can be anonymous or not + :param is_bot: Add the bot flag to the query + :param kwargs: More arguments for lexeme_add_form and Python requests + :return: The id of the newly created Form, e.g. L10-F2 + """ + if not self.id: + raise ValueError('You must set a Lexeme id before writing a Form.') + + if form.id: + raise ValueError(f"The Form {form.id} already exists, adding it again would create a duplicate.") + + data = form.get_json() + # 'add' is a marker used by wbeditentity, the wbladdform action doesn't expect it. + data.pop('add', None) + + login = login or self.api.login + is_bot = is_bot if is_bot is not None else self.api.is_bot + + form.id = lexeme_add_form(lexeme_id=self.id, data=data, login=login, allow_anonymous=allow_anonymous, is_bot=is_bot, **kwargs)['form']['id'] + + return form.id + + def write_forms(self, **kwargs: Any) -> list[str]: + """ + Add all the new Forms of the Lexeme, one wbladdform action per Form. The Forms already existing on the + Wikibase instance are skipped. + + :param kwargs: Arguments passed to write_form() + :return: The ids of the newly created Forms + """ + return [self.write_form(form, **kwargs) for form in self.forms if not form.id] + + def write_sense(self, sense: Sense, login: _Login | None = None, allow_anonymous: bool = False, is_bot: bool | None = None, **kwargs: Any) -> str: + """ + Add a single Sense to the Lexeme with the wbladdsense action. + + Contrary to write(), only the Sense is sent to the Wikibase instance, the rest of the Lexeme is left untouched. + + :param sense: The Sense to add. It must be a new Sense, without an id. + :param login: A login instance + :param allow_anonymous: Force a check if the query can be anonymous or not + :param is_bot: Add the bot flag to the query + :param kwargs: More arguments for lexeme_add_sense and Python requests + :return: The id of the newly created Sense, e.g. L10-S2 + """ + if not self.id: + raise ValueError('You must set a Lexeme id before writing a Sense.') + + if sense.id: + raise ValueError(f"The Sense {sense.id} already exists, adding it again would create a duplicate.") + + if sense.removed: + raise ValueError('The Sense is marked as removed, it cannot be added.') + + data = sense.get_json() + # 'add' is a marker used by wbeditentity, the wbladdsense action doesn't expect it. + data.pop('add', None) + + login = login or self.api.login + is_bot = is_bot if is_bot is not None else self.api.is_bot + + sense.id = lexeme_add_sense(lexeme_id=self.id, data=data, login=login, allow_anonymous=allow_anonymous, is_bot=is_bot, **kwargs)['sense']['id'] + + return sense.id + + def write_senses(self, **kwargs: Any) -> list[str]: + """ + Add all the new Senses of the Lexeme, one wbladdsense action per Sense. The Senses already existing on the + Wikibase instance and the Senses marked as removed are skipped. + + :param kwargs: Arguments passed to write_sense() + :return: The ids of the newly created Senses + """ + return [self.write_sense(sense, **kwargs) for sense in self.senses if not sense.id and not sense.removed] diff --git a/wikibaseintegrator/models/forms.py b/wikibaseintegrator/models/forms.py index 4b68a4da..ebfaa33d 100644 --- a/wikibaseintegrator/models/forms.py +++ b/wikibaseintegrator/models/forms.py @@ -1,5 +1,6 @@ from __future__ import annotations +import json from typing import Any from wikibaseintegrator.models.basemodel import BaseModel @@ -46,6 +47,9 @@ def get_json(self) -> list[dict]: return json_data + def __iter__(self): + return iter(self.forms) + def __len__(self): return len(self.forms) @@ -123,6 +127,20 @@ def get_json(self) -> dict[str, str | dict | list]: return json_data + def _content(self) -> str: + # The id is assigned by the Wikibase instance, so two forms are considered equal when they hold the same content + return json.dumps([self.representations.get_json(), self.grammatical_features, self.claims.get_json()], sort_keys=True) + + def __eq__(self, other): + if not isinstance(other, Form): + return NotImplemented + + return self._content() == other._content() + + def __hash__(self): + # Based on the content, like __eq__: a Form modified after being added to a set or used as a dict key won't be found anymore + return hash(self._content()) + class Representations(LanguageValues): pass diff --git a/wikibaseintegrator/models/senses.py b/wikibaseintegrator/models/senses.py index 4778fa78..d62fbd6b 100644 --- a/wikibaseintegrator/models/senses.py +++ b/wikibaseintegrator/models/senses.py @@ -1,5 +1,6 @@ from __future__ import annotations +import json from typing import Any from wikibaseintegrator.models.basemodel import BaseModel @@ -37,6 +38,9 @@ def get_json(self) -> list[dict]: return json_data + def __iter__(self): + return iter(self.senses) + def __len__(self): return len(self.senses) @@ -75,6 +79,20 @@ def remove(self) -> Sense: self.removed = True return self + def _content(self) -> str: + # The id is assigned by the Wikibase instance, so two senses are considered equal when they hold the same content + return json.dumps([self.glosses.get_json(), self.claims.get_json()], sort_keys=True) + + def __eq__(self, other): + if not isinstance(other, Sense): + return NotImplemented + + return self._content() == other._content() + + def __hash__(self): + # Based on the content, like __eq__: a Sense modified after being added to a set or used as a dict key won't be found anymore + return hash(self._content()) + class Glosses(LanguageValues): pass