music-kraken-core/src/music_kraken/objects/source.py

143 lines
4.6 KiB
Python
Raw Normal View History

2023-03-09 21:14:39 +00:00
from collections import defaultdict
2023-01-12 15:25:50 +00:00
from enum import Enum
from typing import List, Dict, Set, Tuple, Optional
from urllib.parse import urlparse
2023-01-12 15:25:50 +00:00
2023-04-18 09:18:17 +00:00
from ..utils.enums.source import SourcePages, SourceTypes
2023-06-19 13:10:57 +00:00
from ..utils.shared import ALL_YOUTUBE_URLS
2023-03-10 08:09:35 +00:00
from .metadata import Mapping, Metadata
2023-03-09 21:14:39 +00:00
from .parents import DatabaseObject
from .collection import Collection
2023-01-12 15:25:50 +00:00
2023-01-20 22:05:15 +00:00
2023-03-10 09:13:35 +00:00
class Source(DatabaseObject):
2023-01-12 15:25:50 +00:00
"""
create somehow like that
```python
# url won't be a valid one due to it being just an example
Source(src="youtube", url="https://youtu.be/dfnsdajlhkjhsd")
```
"""
2023-03-13 13:33:17 +00:00
COLLECTION_ATTRIBUTES = tuple()
SIMPLE_ATTRIBUTES = {
"page_enum": None,
2023-04-18 09:18:17 +00:00
"url": None,
"referer_page": None,
"audio_url": None
}
2023-01-12 15:25:50 +00:00
2023-04-18 09:18:17 +00:00
def __init__(
self,
page_enum: SourcePages,
url: str = None,
2023-04-18 09:18:17 +00:00
id_: str = None,
referer_page: SourcePages = None,
2023-04-18 09:18:17 +00:00
adio_url: str = None
) -> None:
2023-01-12 15:25:50 +00:00
DatabaseObject.__init__(self, id_=id_)
2023-01-20 22:05:15 +00:00
self.page_enum = page_enum
2023-04-18 09:18:17 +00:00
self.referer_page = page_enum if referer_page is None else referer_page
2023-01-20 22:05:15 +00:00
2023-01-12 15:25:50 +00:00
self.url = url
2023-04-18 09:18:17 +00:00
self.audio_url = adio_url
2023-01-12 15:25:50 +00:00
@classmethod
def match_url(cls, url: str, referer_page: SourcePages) -> Optional["Source"]:
"""
this shouldn't be used, unlesse you are not certain what the source is for
the reason is that it is more inefficient
"""
parsed = urlparse(url)
url = parsed.geturl()
2023-03-30 10:31:37 +00:00
if "musify" in parsed.netloc:
return cls(SourcePages.MUSIFY, url, referer_page=referer_page)
2023-06-19 13:10:57 +00:00
if parsed.netloc in [url.netloc for url in ALL_YOUTUBE_URLS]:
return cls(SourcePages.YOUTUBE, url, referer_page=referer_page)
if url.startswith("https://www.deezer"):
return cls(SourcePages.DEEZER, url, referer_page=referer_page)
if url.startswith("https://open.spotify.com"):
return cls(SourcePages.SPOTIFY, url, referer_page=referer_page)
if "bandcamp" in url:
return cls(SourcePages.BANDCAMP, url, referer_page=referer_page)
2023-03-18 16:06:12 +00:00
if "wikipedia" in parsed.netloc:
return cls(SourcePages.WIKIPEDIA, url, referer_page=referer_page)
2023-03-18 16:06:12 +00:00
if url.startswith("https://www.metal-archives.com/"):
return cls(SourcePages.ENCYCLOPAEDIA_METALLUM, url, referer_page=referer_page)
# the less important once
if url.startswith("https://www.facebook"):
return cls(SourcePages.FACEBOOK, url, referer_page=referer_page)
if url.startswith("https://www.instagram"):
return cls(SourcePages.INSTAGRAM, url, referer_page=referer_page)
if url.startswith("https://twitter"):
return cls(SourcePages.TWITTER, url, referer_page=referer_page)
if url.startswith("https://myspace.com"):
return cls(SourcePages.MYSPACE, url, referer_page=referer_page)
2023-03-10 09:13:35 +00:00
def get_song_metadata(self) -> Metadata:
return Metadata({
2023-01-30 13:41:02 +00:00
Mapping.FILE_WEBPAGE_URL: [self.url],
Mapping.SOURCE_WEBPAGE_URL: [self.homepage]
})
2023-03-10 09:13:35 +00:00
def get_artist_metadata(self) -> Metadata:
return Metadata({
2023-01-30 13:41:02 +00:00
Mapping.ARTIST_WEBPAGE_URL: [self.url]
})
2023-03-10 08:09:35 +00:00
@property
def metadata(self) -> Metadata:
return self.get_song_metadata()
2023-01-12 16:14:21 +00:00
@property
def indexing_values(self) -> List[Tuple[str, object]]:
return [
('id', self.id),
('url', self.url),
('audio_url', self.audio_url),
]
2023-01-12 15:25:50 +00:00
def __str__(self):
return self.__repr__()
2023-01-12 15:25:50 +00:00
2023-01-30 13:41:02 +00:00
def __repr__(self) -> str:
return f"Src({self.page_enum.value}: {self.url}, {self.audio_url})"
2023-01-30 13:41:02 +00:00
2023-01-20 22:05:15 +00:00
page_str = property(fget=lambda self: self.page_enum.value)
2023-01-20 09:56:40 +00:00
type_str = property(fget=lambda self: self.type_enum.value)
2023-01-20 22:05:15 +00:00
homepage = property(fget=lambda self: SourcePages.get_homepage(self.page_enum))
2023-01-25 13:14:15 +00:00
2023-03-09 21:14:39 +00:00
class SourceCollection(Collection):
def __init__(self, source_list: List[Source]):
self._page_to_source_list: Dict[SourcePages, List[Source]] = defaultdict(list)
2023-03-10 09:13:35 +00:00
super().__init__(data=source_list, element_type=Source)
2023-03-09 21:14:39 +00:00
def map_element(self, source: Source):
super().map_element(source)
self._page_to_source_list[source.page_enum].append(source)
@property
def source_pages(self) -> Set[SourcePages]:
return set(source.page_enum for source in self._data)
2023-03-09 21:14:39 +00:00
def get_sources_from_page(self, source_page: SourcePages) -> List[Source]:
"""
getting the sources for a specific page like
YouTube or musify
"""
return self._page_to_source_list[source_page].copy()