#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""A set of utility functions that identify irrelevant images
and unsuitable article or section candidates.
"""
import string
from pyspark.sql import Column, DataFrame, SparkSession
from pyspark.sql import functions as F
from image_suggestions import shared
#: ASCII punctuation characters to be stripped from section titles.
#: Include the ASCII white space, don't strip round brackets.
STRIP_CHARS = string.punctuation.replace('()', ' ')
#: All kinds of white space to be substituted for the ASCII one;
#: underscores turn into spaces as well.
SUBSTITUTE_PATTERN = r'[\s_]'
#: If an article's Wikidata `item <https://www.wikidata.org/wiki/WD:GLOSS#Item>`_ is an
#: `instance of <https://www.wikidata.org/wiki/Property:P31>`_ one of the items in this list,
#: then it's not suitable for getting suggestions.
UNILLUSTRATABLE_P31 = (
'Q577', # Year
'Q29964144', # Year BC
'Q3186692', # Calendar year
'Q3311614', # Century leap year
'Q14795564', # Recurrent timeframe
'Q101352', # Family name
'Q82799', # Name
'Q21199', # Natural number
'Q28920044', # Positive integer
'Q28920052', # Non-negative integer
'Q13406463', # List
'Q4167410', # Disambiguation page
'Q22808320', # Wikimedia human name disambiguation page
'Q98645843', # Wikimedia music-related list
'Q17099416', # Wikimedia list of songs
'Q100775261', # Wikimedia list of musical works
)
# We have a stored DataFrame containing images that are in placeholder categories.
# See `maintenance/generate_image_placeholders.py`.
PLACEHOLDER_IMAGE_PARQUET = 'image_placeholders'
#: Image file names containing these substrings are probably icons or placeholders.
PLACEHOLDER_IMAGE_SUBSTRINGS = (
'flag',
'noantimage',
'no_free_image',
'image_manquante',
'replace_this_image',
'disambig',
'regions',
'map',
'default',
'defaut',
'falta_imagem_',
'imageNA',
'noimage',
'noenzyimage',
)
PLACEHOLDER_IMAGE_ALLOWED_FILE_EXTENSIONS = (
'.bmp',
'.jpeg',
'.jpg',
'.png',
'.tif',
'.tiff',
)
[docs]
def get_disallowed_substrings_regex(
substrings: tuple[str, ...] = PLACEHOLDER_IMAGE_SUBSTRINGS,
) -> str:
"""Build a regular expression to detect image file names that may be icons or placeholders.
:param substrings: a tuple of substrings that indicate icons or placeholders in image file names
:return: the regular expression that matches the given substrings
"""
return '(?i)({})'.format('|'.join(substrings))
[docs]
def get_allowed_suffixes_regex(
suffixes: tuple[str, ...] = PLACEHOLDER_IMAGE_ALLOWED_FILE_EXTENSIONS,
) -> str:
"""Build a regular expression to detect image file extensions that typically hold valid images.
:param suffixes: a tuple of suffixes that indicate valid image file extensions
:return: the regular expression that matches the given suffixes
"""
return '(?i)({})$'.format('|'.join(suffixes))
[docs]
def read_section_images_parquet(
spark: SparkSession, section_images_parquet: str
) -> DataFrame: # pragma: no cover
"""Load images available in all sections of all articles of all Wikipedias,
as output by :mod:`imagerec.article_images`.
:param spark: an active Spark session
:param section_images_parquet: a HDFS path to a parquet generated by :mod:`imagerec.article_images`
:return: the dataframe of:
- item_id (string) - page Wikidata `QID <https://www.wikidata.org/wiki/Wikidata:Glossary#QID>`_
- page_id (string) - page ID
- page_title (string) - page title, in original case and underscored
- article_images (array<struct<heading:string,images:array<string>>>) - images per section per page
- wiki_db (string) - wiki project
"""
return spark.read.parquet(section_images_parquet)
[docs]
def get_section_images(
spark: SparkSession, section_images_parquet: str
) -> DataFrame: # pragma: no cover
"""Explode a dataframe as loaded by :func:`read_section_images_parquet` for easier processing.
:param spark: an active Spark session
:param section_images_parquet: a HDFS path to a parquet generated by :mod:`imagerec.article_images`
:return: the dataframe of:
- wiki_db (string) - wiki project
- page_id (string) - page ID
- page_title (string) - page title, in original case and underscored
- section_heading (string) - page section, in URL anchor format.
More details in :func:`section_topics.pipeline.wikitext_headings_to_anchors`
- image (string) - Commons image file name
"""
article_images = read_section_images_parquet(spark, section_images_parquet)
return (
article_images.select('*', F.explode('article_images').alias('ais_array'))
.select('*', F.explode('ais_array.images').alias('image'))
.where(F.col('image').isNotNull())
.select(
'wiki_db',
'page_id',
'page_title',
F.col('ais_array.heading').alias('section_heading'),
'image',
)
.orderBy('page_title', 'section_heading', 'image')
)
[docs]
def get_non_illustratable_item_ids(
spark: SparkSession, hive_db: str, weekly_snapshot: str
) -> DataFrame:
"""Gather Wikidata QIDs that aren't suitable for getting suggestions.
See :const:`UNILLUSTRATABLE_P31`.
:param spark: an active Spark session
:param hive_db: a Data Lake's `Hive <https://hive.apache.org/>`_ database name
:param weekly_snapshot: a ``YYYY-MM-DD`` date
:return: the dataframe of Wikidata QIDs
"""
wikidata_items_with_p31 = shared.load_wikidata_items_with_p31(
spark, hive_db, weekly_snapshot
)
return wikidata_items_with_p31.where(
F.col('value').isin(list(UNILLUSTRATABLE_P31))
).select('item_id')
[docs]
def read_denylist_parquet(
spark: SparkSession, denylist_parquet: str
) -> DataFrame: # pragma: no cover
"""Load denylisted section titles.
:param spark: an active Spark session
:param denylist_parquet: a HDFS path to a parquet generated by
:mod:`section_topics.scripts.gather_section_titles_denylist`
:return: the dataframe of:
- wiki_db (string) - wiki project
- section_heading (string) - page section, in URL anchor format.
"""
return spark.read.parquet(denylist_parquet)
[docs]
def get_non_illustratable_sections(
spark: SparkSession,
denylist_parquet: str,
dataframe: DataFrame,
wiki_column: Column,
heading_column: Column,
) -> DataFrame:
"""Gather all Wikipedia article section headings that aren't suitable for getting suggestions.
:param spark: an active Spark session
:param denylist_parquet: a HDFS path to a parquet generated by
:mod:`section_topics.scripts.gather_section_titles_denylist`
:param dataframe: a dataframe of irrelevant section headings
:param wiki_column: a ``dataframe``'s column of wikis
:param heading_column: a ``dataframe``'s column of section headings
:return: the dataframe of:
- wiki_db (string) - wiki project
- section_heading (string) - page section, in URL anchor format.
More details in :func:`section_topics.pipeline.wikitext_headings_to_anchors`
"""
denylist_dataframe = read_denylist_parquet(spark, denylist_parquet)
return denylist_dataframe.join(
dataframe,
on=[
denylist_dataframe.wiki_db == wiki_column,
normalize_heading_column(denylist_dataframe.section_heading)
== normalize_heading_column(heading_column),
],
how='inner',
).select(
wiki_column,
heading_column.alias('section_heading'),
)
[docs]
def get_images_in_placeholder_categories(
spark: SparkSession,
) -> DataFrame: # pragma: no cover
"""Load images that belong to the placeholder Commons category.
:param spark: an active Spark session
:return: the dataframe of:
- cl_from (bigint) - Commons page ID
- cl_to (string) - Commons category page title, in original case and underscored
- cl_type (string) - ``'file'``
- page_title (string) - Commons page title, in original case and underscored
"""
return spark.read.parquet(PLACEHOLDER_IMAGE_PARQUET)
# Copied from https://gitlab.wikimedia.org/repos/structured-data/section-topics/-/blob/main/section_topics/pipeline.py
[docs]
def normalize_heading_column(
column: Column,
substitute_pattern: str = SUBSTITUTE_PATTERN,
strip_chars: str = STRIP_CHARS,
) -> Column:
"""Same as :func:`section_topics.pipeline.normalize_heading_column`."""
# These chars will be surrounded by `[]` to form a regexp group, so escape literal `[` and `]`.
# Also double the backslash to prevent it from being an escape char itself.
escaped_strip_chars = (
strip_chars.replace('\\', '\\\\').replace('[', r'\[').replace(']', r'\]')
)
return F.lower(
F.regexp_replace(
F.regexp_replace(
F.regexp_replace(
column,
'_[0-9]+$',
'',
),
substitute_pattern,
' ',
),
f'^[{escaped_strip_chars}]*(.*?)[{escaped_strip_chars}]*$',
'$1',
)
)