Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
25 changes: 25 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -47,6 +47,7 @@ wikibaseintegrator~=0.11.3
- [Set lemma on lexeme](#set-lemma-on-lexeme)
- [Add gloss to a sense on lexeme](#add-gloss-to-a-sense-on-lexeme)
- [Add form to a lexeme](#add-form-to-a-lexeme)
- [Add a form or a sense to an existing lexeme](#add-a-form-or-a-sense-to-an-existing-lexeme)
- [Other projects](#other-projects)
- [Installation](#installation)
- [Installation of the development environment](#installation-of-the-development-environment)
Expand Down Expand Up @@ -313,6 +314,30 @@ form.claims.add(claim)
lexeme.forms.add(form)
```

#### Add a form or a sense to an existing lexeme

Contrary to `write()`, `write_form()` and `write_sense()` only send the new Form or Sense to the Wikibase instance (with
the `wbladdform` and `wbladdsense` actions), the rest of the lexeme is left untouched. They return the id assigned by the
instance, which is also set on the object.

`write_forms()` and `write_senses()` add every Form or Sense of the lexeme without an id, one request per Form or Sense.

From [lexeme_write.ipynb](notebooks/lexeme_write.ipynb)

```python
lexeme = wbi.lexeme.get('L5')

form = Form()
form.representations.set(language='en', value='English form representation')
form.grammatical_features = ['Q146786']
lexeme.write_form(form) # 'L5-F3'

sense = Sense()
sense.glosses.set(language='en', value='English gloss')
lexeme.senses.add(sense)
lexeme.write_senses() # ['L5-S2']
```

## Other projects ##

Here is a list of different projects that use the library:
Expand Down
79 changes: 79 additions & 0 deletions notebooks/lexeme_write.ipynb
Original file line number Diff line number Diff line change
Expand Up @@ -396,6 +396,85 @@
"name": "#%%\n"
}
}
},
{
"cell_type": "markdown",
"source": [
"# Add a form and a sense to an existing lexeme\n",
"\n",
"`write_form()` and `write_sense()` only send the new Form or Sense to the Wikibase instance (with the `wbladdform` and `wbladdsense` actions), the rest of the lexeme is left untouched. The id assigned by the instance is returned and set on the object."
],
"metadata": {
"collapsed": false,
"pycharm": {
"name": "#%% md\n"
}
}
},
{
"cell_type": "code",
"execution_count": null,
"outputs": [],
"source": [
"new_form = Form()\n",
"new_form.representations.set(language='en', value='Another English form representation')\n",
"new_form.grammatical_features = ['Q146786']\n",
"\n",
"lexeme.write_form(new_form)"
],
"metadata": {
"collapsed": false,
"pycharm": {
"name": "#%%\n"
}
}
},
{
"cell_type": "code",
"execution_count": null,
"outputs": [],
"source": [
"new_sense = Sense()\n",
"new_sense.glosses.set(language='en', value='Another English gloss')\n",
"\n",
"lexeme.write_sense(new_sense)"
],
"metadata": {
"collapsed": false,
"pycharm": {
"name": "#%%\n"
}
}
},
{
"cell_type": "markdown",
"source": [
"`write_forms()` and `write_senses()` add every Form or Sense of the lexeme without an id, one request per Form or Sense. Those already on the Wikibase instance are skipped."
],
"metadata": {
"collapsed": false,
"pycharm": {
"name": "#%% md\n"
}
}
},
{
"cell_type": "code",
"execution_count": null,
"outputs": [],
"source": [
"another_sense = Sense()\n",
"another_sense.glosses.set(language='en', value='A third English gloss')\n",
"lexeme.senses.add(another_sense)\n",
"\n",
"lexeme.write_senses()"
],
"metadata": {
"collapsed": false,
"pycharm": {
"name": "#%%\n"
}
}
}
],
"metadata": {
Expand Down
29 changes: 29 additions & 0 deletions test/conftest.py
Original file line number Diff line number Diff line change
Expand Up @@ -321,6 +321,35 @@ def _apply_claims(self, current: dict, submitted: dict | list, entity_id: str) -

return result

def _action_wbladdform(self, params: dict[str, str]) -> dict:
return self._add_lexeme_sub_entity(params, section='forms', response_key='form', id_prefix='F',
defaults={'representations': {}, 'grammaticalFeatures': [], 'claims': {}})

def _action_wbladdsense(self, params: dict[str, str]) -> dict:
return self._add_lexeme_sub_entity(params, section='senses', response_key='sense', id_prefix='S',
defaults={'glosses': {}, 'claims': {}})

def _add_lexeme_sub_entity(self, params: dict[str, str], section: str, response_key: str, id_prefix: str, defaults: dict) -> dict:
"""Shared implementation of the wbladdform and wbladdsense actions."""
data = json.loads(params['data'])
self.edits.append({'params': params, 'data': data})

lexeme_id = params['lexemeId']
if lexeme_id not in self.entities:
return {'error': {'code': 'not-found', 'info': f'Could not find an entity with the ID "{lexeme_id}".'}, 'servedby': 'mock'}

lexeme = self.entities[lexeme_id]
existing = lexeme.setdefault(section, [])

sub_entity = {**defaults, **deepcopy(data)}
# A real instance assigns the id, incrementing a counter never reused after a removal.
sub_entity['id'] = f'{lexeme_id}-{id_prefix}{len(existing) + 1}'
existing.append(sub_entity)

lexeme['lastrevid'] = lexeme.get('lastrevid', 0) + 1

return {response_key: deepcopy(sub_entity), 'lastrevid': lexeme['lastrevid'], 'success': 1}

def _action_query(self, params: dict[str, str]) -> dict:
if params.get('meta') == 'tokens':
if params.get('type') == 'login':
Expand Down
63 changes: 62 additions & 1 deletion test/integration/test_wikibase_roundtrip.py
Original file line number Diff line number Diff line change
Expand Up @@ -12,9 +12,10 @@
import pytest

from wikibaseintegrator.datatypes import Item, String
from wikibaseintegrator.models import Form, Sense
from wikibaseintegrator.wbi_enums import ActionIfExists
from wikibaseintegrator.wbi_exceptions import MissingEntityException
from wikibaseintegrator.wbi_helpers import search_entities
from wikibaseintegrator.wbi_helpers import mediawiki_api_call_helper, search_entities

pytestmark = pytest.mark.integration

Expand Down Expand Up @@ -95,6 +96,66 @@ def test_qualifier_and_reference_roundtrip(self, wbi, string_property):
fetched.delete(reason='WikibaseIntegrator integration test cleanup')


@pytest.fixture(scope='module')
def lexeme_prerequisites(login):
"""The items used as language and lexical category of the lexemes created for this test run."""
from wikibaseintegrator import WikibaseIntegrator
wbi = WikibaseIntegrator(login=login)

# The WikibaseLexeme extension is optional on a Wikibase instance
extensions = mediawiki_api_call_helper(data={'action': 'query', 'meta': 'siteinfo', 'siprop': 'extensions'}, allow_anonymous=True)['query']['extensions']
if not any(extension.get('name') == 'WikibaseLexeme' for extension in extensions):
pytest.skip('The WikibaseLexeme extension is not installed on the instance')

items = {}
for role in ('language', 'lexical category'):
item = wbi.item.new()
item.labels.set(language='en', value=f'WBI integration test {role} {RUN_ID}')
items[role] = item.write(summary='WikibaseIntegrator integration test setup')

yield items

for item in items.values():
item.delete(reason='WikibaseIntegrator integration test cleanup')


class TestLexemeFormsAndSenses:
def test_write_form_and_sense(self, wbi, lexeme_prerequisites):
lexeme = wbi.lexeme.new(language=lexeme_prerequisites['language'].id, lexical_category=lexeme_prerequisites['lexical category'].id)
lexeme.lemmas.set(language='en', value=f'wbi-lemma-{RUN_ID}')
lexeme.write(summary='WikibaseIntegrator integration test: create lexeme')
assert lexeme.id

# A single Form, with wbladdform
form = Form(grammatical_features=lexeme_prerequisites['lexical category'].id)
form.representations.set(language='en', value=f'wbi-form-{RUN_ID}')
form_id = lexeme.write_form(form)
assert form_id.startswith(f'{lexeme.id}-F')
assert form.id == form_id

# Several Senses at once with wbladdsense, the one marked as removed is skipped
for gloss in ('first', 'second'):
sense = Sense()
sense.glosses.set(language='en', value=f'{gloss} gloss {RUN_ID}')
lexeme.senses.add(sense)
removed_sense = Sense()
removed_sense.glosses.set(language='en', value='removed gloss')
lexeme.senses.add(removed_sense.remove())

sense_ids = lexeme.write_senses()
assert len(sense_ids) == 2
assert all(sense_id.startswith(f'{lexeme.id}-S') for sense_id in sense_ids)

# Read back from the instance
fetched = wbi.lexeme.get(lexeme.id)
assert fetched.forms.get(form_id).representations.get('en').value == f'wbi-form-{RUN_ID}'
assert fetched.forms.get(form_id).grammatical_features == [lexeme_prerequisites['lexical category'].id]
assert [fetched.senses.get(sense_id).glosses.get('en').value for sense_id in sense_ids] == [f'first gloss {RUN_ID}', f'second gloss {RUN_ID}']
assert len(fetched.senses) == 2

fetched.delete(reason='WikibaseIntegrator integration test cleanup')


class TestSearch:
def test_search_finds_created_entity(self, wbi, string_property):
# The property created for this run must be findable by its label.
Expand Down
Loading
Loading