diff --git a/Pipfile b/Pipfile index f8e2543..9a95eed 100644 --- a/Pipfile +++ b/Pipfile @@ -35,6 +35,7 @@ demagnetize = "*" redis = {extras = ["hiredis"], version = "*"} dramatiq = {extras = ["redis", "watch"], version = "*"} gunicorn = "*" +scrapy = "*" [dev-packages] pysocks = "*" diff --git a/Pipfile.lock b/Pipfile.lock index ca44e6f..a48506c 100644 --- a/Pipfile.lock +++ b/Pipfile.lock @@ -1,7 +1,7 @@ { "_meta": { "hash": { - "sha256": "58f27737820bf86f5b3c3f2c68770725c909770690f04d240f3ee3c0e7b564c4" + "sha256": "30880a6cb492bb705c7671889e64ef1113f718c771ed9c6545b3431e37e1f396" }, "pipfile-spec": 6, "requires": { @@ -49,6 +49,13 @@ "markers": "python_version >= '3.7'", "version": "==23.2.0" }, + "automat": { + "hashes": [ + "sha256:c3164f8742b9dc440f3682482d32aaff7bb53f71740dd018533f9de286b64180", + "sha256:e56beb84edad19dcc11d30e8d9b895f75deeb5ef5e96b84a467066b3b84bb04e" + ], + "version": "==22.10.0" + }, "beanie": { "hashes": [ "sha256:4436ac740718ccd62b21576778679ac972359fce2938557890c576adbbf5e244", @@ -82,6 +89,64 @@ "markers": "python_version >= '3.6'", "version": "==2024.2.2" }, + "cffi": { + "hashes": [ + "sha256:0c9ef6ff37e974b73c25eecc13952c55bceed9112be2d9d938ded8e856138bcc", + "sha256:131fd094d1065b19540c3d72594260f118b231090295d8c34e19a7bbcf2e860a", + "sha256:1b8ebc27c014c59692bb2664c7d13ce7a6e9a629be20e54e7271fa696ff2b417", + "sha256:2c56b361916f390cd758a57f2e16233eb4f64bcbeee88a4881ea90fca14dc6ab", + "sha256:2d92b25dbf6cae33f65005baf472d2c245c050b1ce709cc4588cdcdd5495b520", + "sha256:31d13b0f99e0836b7ff893d37af07366ebc90b678b6664c955b54561fc36ef36", + "sha256:32c68ef735dbe5857c810328cb2481e24722a59a2003018885514d4c09af9743", + "sha256:3686dffb02459559c74dd3d81748269ffb0eb027c39a6fc99502de37d501faa8", + "sha256:582215a0e9adbe0e379761260553ba11c58943e4bbe9c36430c4ca6ac74b15ed", + "sha256:5b50bf3f55561dac5438f8e70bfcdfd74543fd60df5fa5f62d94e5867deca684", + "sha256:5bf44d66cdf9e893637896c7faa22298baebcd18d1ddb6d2626a6e39793a1d56", + "sha256:6602bc8dc6f3a9e02b6c22c4fc1e47aa50f8f8e6d3f78a5e16ac33ef5fefa324", + "sha256:673739cb539f8cdaa07d92d02efa93c9ccf87e345b9a0b556e3ecc666718468d", + "sha256:68678abf380b42ce21a5f2abde8efee05c114c2fdb2e9eef2efdb0257fba1235", + "sha256:68e7c44931cc171c54ccb702482e9fc723192e88d25a0e133edd7aff8fcd1f6e", + "sha256:6b3d6606d369fc1da4fd8c357d026317fbb9c9b75d36dc16e90e84c26854b088", + "sha256:748dcd1e3d3d7cd5443ef03ce8685043294ad6bd7c02a38d1bd367cfd968e000", + "sha256:7651c50c8c5ef7bdb41108b7b8c5a83013bfaa8a935590c5d74627c047a583c7", + "sha256:7b78010e7b97fef4bee1e896df8a4bbb6712b7f05b7ef630f9d1da00f6444d2e", + "sha256:7e61e3e4fa664a8588aa25c883eab612a188c725755afff6289454d6362b9673", + "sha256:80876338e19c951fdfed6198e70bc88f1c9758b94578d5a7c4c91a87af3cf31c", + "sha256:8895613bcc094d4a1b2dbe179d88d7fb4a15cee43c052e8885783fac397d91fe", + "sha256:88e2b3c14bdb32e440be531ade29d3c50a1a59cd4e51b1dd8b0865c54ea5d2e2", + "sha256:8f8e709127c6c77446a8c0a8c8bf3c8ee706a06cd44b1e827c3e6a2ee6b8c098", + "sha256:9cb4a35b3642fc5c005a6755a5d17c6c8b6bcb6981baf81cea8bfbc8903e8ba8", + "sha256:9f90389693731ff1f659e55c7d1640e2ec43ff725cc61b04b2f9c6d8d017df6a", + "sha256:a09582f178759ee8128d9270cd1344154fd473bb77d94ce0aeb2a93ebf0feaf0", + "sha256:a6a14b17d7e17fa0d207ac08642c8820f84f25ce17a442fd15e27ea18d67c59b", + "sha256:a72e8961a86d19bdb45851d8f1f08b041ea37d2bd8d4fd19903bc3083d80c896", + "sha256:abd808f9c129ba2beda4cfc53bde801e5bcf9d6e0f22f095e45327c038bfe68e", + "sha256:ac0f5edd2360eea2f1daa9e26a41db02dd4b0451b48f7c318e217ee092a213e9", + "sha256:b29ebffcf550f9da55bec9e02ad430c992a87e5f512cd63388abb76f1036d8d2", + "sha256:b2ca4e77f9f47c55c194982e10f058db063937845bb2b7a86c84a6cfe0aefa8b", + "sha256:b7be2d771cdba2942e13215c4e340bfd76398e9227ad10402a8767ab1865d2e6", + "sha256:b84834d0cf97e7d27dd5b7f3aca7b6e9263c56308ab9dc8aae9784abb774d404", + "sha256:b86851a328eedc692acf81fb05444bdf1891747c25af7529e39ddafaf68a4f3f", + "sha256:bcb3ef43e58665bbda2fb198698fcae6776483e0c4a631aa5647806c25e02cc0", + "sha256:c0f31130ebc2d37cdd8e44605fb5fa7ad59049298b3f745c74fa74c62fbfcfc4", + "sha256:c6a164aa47843fb1b01e941d385aab7215563bb8816d80ff3a363a9f8448a8dc", + "sha256:d8a9d3ebe49f084ad71f9269834ceccbf398253c9fac910c4fd7053ff1386936", + "sha256:db8e577c19c0fda0beb7e0d4e09e0ba74b1e4c092e0e40bfa12fe05b6f6d75ba", + "sha256:dc9b18bf40cc75f66f40a7379f6a9513244fe33c0e8aa72e2d56b0196a7ef872", + "sha256:e09f3ff613345df5e8c3667da1d918f9149bd623cd9070c983c013792a9a62eb", + "sha256:e4108df7fe9b707191e55f33efbcb2d81928e10cea45527879a4749cbe472614", + "sha256:e6024675e67af929088fda399b2094574609396b1decb609c55fa58b028a32a1", + "sha256:e70f54f1796669ef691ca07d046cd81a29cb4deb1e5f942003f401c0c4a2695d", + "sha256:e715596e683d2ce000574bae5d07bd522c781a822866c20495e52520564f0969", + "sha256:e760191dd42581e023a68b758769e2da259b5d52e3103c6060ddc02c9edb8d7b", + "sha256:ed86a35631f7bfbb28e108dd96773b9d5a6ce4811cf6ea468bb6a359b256b1e4", + "sha256:ee07e47c12890ef248766a6e55bd38ebfb2bb8edd4142d56db91b21ea68b7627", + "sha256:fa3a0128b152627161ce47201262d3140edb5a5c3da88d73a1b790a959126956", + "sha256:fcc8eb6d5902bb1cf6dc4f187ee3ea80a1eba0a89aba40a5cb20a5087d961357" + ], + "markers": "platform_python_implementation != 'PyPy'", + "version": "==1.16.0" + }, "charset-normalizer": { "hashes": [ "sha256:06435b539f889b1f6f4ac1758871aae42dc3a8c0e24ac9e60c2384973ad73027", @@ -215,6 +280,60 @@ "markers": "python_version >= '3.6'", "version": "==6.8.2" }, + "constantly": { + "hashes": [ + "sha256:3fd9b4d1c3dc1ec9757f3c52aef7e53ad9323dbe39f51dfd4c43853b68dfa3f9", + "sha256:aa92b70a33e2ac0bb33cd745eb61776594dc48764b06c35e0efd050b7f1c7cbd" + ], + "markers": "python_version >= '3.8'", + "version": "==23.10.4" + }, + "cryptography": { + "hashes": [ + "sha256:0270572b8bd2c833c3981724b8ee9747b3ec96f699a9665470018594301439ee", + "sha256:111a0d8553afcf8eb02a4fea6ca4f59d48ddb34497aa8706a6cf536f1a5ec576", + "sha256:16a48c23a62a2f4a285699dba2e4ff2d1cff3115b9df052cdd976a18856d8e3d", + "sha256:1b95b98b0d2af784078fa69f637135e3c317091b615cd0905f8b8a087e86fa30", + "sha256:1f71c10d1e88467126f0efd484bd44bca5e14c664ec2ede64c32f20875c0d413", + "sha256:2424ff4c4ac7f6b8177b53c17ed5d8fa74ae5955656867f5a8affaca36a27abb", + "sha256:2bce03af1ce5a5567ab89bd90d11e7bbdff56b8af3acbbec1faded8f44cb06da", + "sha256:329906dcc7b20ff3cad13c069a78124ed8247adcac44b10bea1130e36caae0b4", + "sha256:37dd623507659e08be98eec89323469e8c7b4c1407c85112634ae3dbdb926fdd", + "sha256:3eaafe47ec0d0ffcc9349e1708be2aaea4c6dd4978d76bf6eb0cb2c13636c6fc", + "sha256:5e6275c09d2badf57aea3afa80d975444f4be8d3bc58f7f80d2a484c6f9485c8", + "sha256:6fe07eec95dfd477eb9530aef5bead34fec819b3aaf6c5bd6d20565da607bfe1", + "sha256:7367d7b2eca6513681127ebad53b2582911d1736dc2ffc19f2c3ae49997496bc", + "sha256:7cde5f38e614f55e28d831754e8a3bacf9ace5d1566235e39d91b35502d6936e", + "sha256:9481ffe3cf013b71b2428b905c4f7a9a4f76ec03065b05ff499bb5682a8d9ad8", + "sha256:98d8dc6d012b82287f2c3d26ce1d2dd130ec200c8679b6213b3c73c08b2b7940", + "sha256:a011a644f6d7d03736214d38832e030d8268bcff4a41f728e6030325fea3e400", + "sha256:a2913c5375154b6ef2e91c10b5720ea6e21007412f6437504ffea2109b5a33d7", + "sha256:a30596bae9403a342c978fb47d9b0ee277699fa53bbafad14706af51fe543d16", + "sha256:b03c2ae5d2f0fc05f9a2c0c997e1bc18c8229f392234e8a0194f202169ccd278", + "sha256:b6cd2203306b63e41acdf39aa93b86fb566049aeb6dc489b70e34bcd07adca74", + "sha256:b7ffe927ee6531c78f81aa17e684e2ff617daeba7f189f911065b2ea2d526dec", + "sha256:b8cac287fafc4ad485b8a9b67d0ee80c66bf3574f655d3b97ef2e1082360faf1", + "sha256:ba334e6e4b1d92442b75ddacc615c5476d4ad55cc29b15d590cc6b86efa487e2", + "sha256:ba3e4a42397c25b7ff88cdec6e2a16c2be18720f317506ee25210f6d31925f9c", + "sha256:c41fb5e6a5fe9ebcd58ca3abfeb51dffb5d83d6775405305bfa8715b76521922", + "sha256:cd2030f6650c089aeb304cf093f3244d34745ce0cfcc39f20c6fbfe030102e2a", + "sha256:cd65d75953847815962c84a4654a84850b2bb4aed3f26fadcc1c13892e1e29f6", + "sha256:e4985a790f921508f36f81831817cbc03b102d643b5fcb81cd33df3fa291a1a1", + "sha256:e807b3188f9eb0eaa7bbb579b462c5ace579f1cedb28107ce8b48a9f7ad3679e", + "sha256:f12764b8fffc7a123f641d7d049d382b73f96a34117e0b637b80643169cec8ac", + "sha256:f8837fe1d6ac4a8052a9a8ddab256bc006242696f03368a4009be7ee3075cdb7" + ], + "markers": "python_version >= '3.7'", + "version": "==42.0.5" + }, + "cssselect": { + "hashes": [ + "sha256:666b19839cfaddb9ce9d36bfe4c969132c647b92fc9088c4e23f786b30f1b3dc", + "sha256:da1885f0c10b60c03ed5eccbb6b68d6eff248d91976fcde348f395d54c9fd35e" + ], + "markers": "python_version >= '3.7'", + "version": "==1.2.0" + }, "demagnetize": { "hashes": [ "sha256:a3ac825bb8d949d3d1ea17fb67b48d41c22a22bf0eae6c70817042aa7db9f01c", @@ -263,6 +382,14 @@ "markers": "python_version >= '3.8'", "version": "==0.110.0" }, + "filelock": { + "hashes": [ + "sha256:521f5f56c50f8426f5e03ad3b281b490a87ef15bc6c526f168290f0c7148d44e", + "sha256:57dbda9b35157b05fb3e58ee91448612eb674172fab98ee235ccb0b5bee19a1c" + ], + "markers": "python_version >= '3.8'", + "version": "==3.13.1" + }, "flatbencode": { "hashes": [ "sha256:77397d5d0108835404f6f7cad640d403c7025a11812afc7e1df4b0f92589ba77" @@ -569,6 +696,13 @@ "markers": "python_version >= '3.8'", "version": "==0.27.0" }, + "hyperlink": { + "hashes": [ + "sha256:427af957daa58bc909471c6c40f74c5450fa123dd093fc53efd2e91d2705a56b", + "sha256:e6b14c37ecb73e89c77d78cdb4c2cc8f3fb59a885c5b3f819ff4ed80f25af1b4" + ], + "version": "==21.0.0" + }, "idna": { "hashes": [ "sha256:9ecdbbd083b06798ae1e86adcbfe8ab1479cf864e4ee30fe4e46a003d12491ca", @@ -577,6 +711,29 @@ "markers": "python_version >= '3.5'", "version": "==3.6" }, + "incremental": { + "hashes": [ + "sha256:912feeb5e0f7e0188e6f42241d2f450002e11bbc0937c65865045854c24c0bd0", + "sha256:b864a1f30885ee72c5ac2835a761b8fe8aa9c28b9395cacf27286602688d3e51" + ], + "version": "==22.10.0" + }, + "itemadapter": { + "hashes": [ + "sha256:2ac1fbcc363b789a18639935ca322e50a65a0a7dfdd8d973c34e2c468e6c0f94", + "sha256:77758485fb0ac10730d4b131363e37d65cb8db2450bfec7a57c3f3271f4a48a9" + ], + "markers": "python_version >= '3.7'", + "version": "==0.8.0" + }, + "itemloaders": { + "hashes": [ + "sha256:21d81c61da6a08b48e5996288cdf3031c0f92e5d0075920a0242527523e14a48", + "sha256:c8c82fe0c11fc4cdd08ec04df0b3c43f3cb7190002edb517e02d55de8efc2aeb" + ], + "markers": "python_version >= '3.7'", + "version": "==1.1.0" + }, "jinja2": { "hashes": [ "sha256:7d6d50dd97d52cbc355597bd845fabfbac3f551e1f99619e39a35ce8c370b5fa", @@ -586,6 +743,14 @@ "markers": "python_version >= '3.7'", "version": "==3.1.3" }, + "jmespath": { + "hashes": [ + "sha256:02e2e4cc71b5bcab88332eebf907519190dd9e6e82107fa7f83b1003a6252980", + "sha256:90261b206d6defd58fdd5e85f478bf633a2901798906be2ad389150c5c60edbe" + ], + "markers": "python_version >= '3.7'", + "version": "==1.0.1" + }, "lazy-model": { "hashes": [ "sha256:57c0e91e171530c4fca7aebc3ac05a163a85cddd941bf7527cc46c0ddafca47c", @@ -861,6 +1026,14 @@ "git": "git+https://github.com/mhdzumair/parse-torrent-title", "ref": "5f4c12b3ac6ed108a68258f1c67e69d4c12a3c71" }, + "parsel": { + "hashes": [ + "sha256:2708fc74daeeb4ce471e2c2e9089b650ec940c7a218053e57421e69b5b00f82c", + "sha256:aff28e68c9b3f1a901db2a4e3f158d8480a38724d7328ee751c1a4e1c1801e39" + ], + "markers": "python_version >= '3.7'", + "version": "==1.8.1" + }, "pikpakapi": { "git": "git+https://github.com/mhdzumair/PikPakAPI.git", "markers": "python_full_version >= '3.8.3' and python_full_version < '4.0.0'", @@ -972,6 +1145,37 @@ "markers": "python_version >= '3.8'", "version": "==0.20.0" }, + "protego": { + "hashes": [ + "sha256:04228bffde4c6bcba31cf6529ba2cfd6e1b70808fdc1d2cb4301be6b28d6c568", + "sha256:db38f6a945839d8162a4034031a21490469566a2726afb51d668497c457fb0aa" + ], + "markers": "python_version >= '3.7'", + "version": "==0.3.0" + }, + "pyasn1": { + "hashes": [ + "sha256:4439847c58d40b1d0a573d07e3856e95333f1976294494c325775aeca506eb58", + "sha256:6d391a96e59b23130a5cfa74d6fd7f388dbbe26cc8f1edf39fdddf08d9d6676c" + ], + "markers": "python_version >= '2.7' and python_version not in '3.0, 3.1, 3.2, 3.3, 3.4, 3.5'", + "version": "==0.5.1" + }, + "pyasn1-modules": { + "hashes": [ + "sha256:5bd01446b736eb9d31512a30d46c1ac3395d676c6f3cafa4c03eb54b9925631c", + "sha256:d3ccd6ed470d9ffbc716be08bd90efbd44d0734bc9303818f7336070984a162d" + ], + "markers": "python_version >= '2.7' and python_version not in '3.0, 3.1, 3.2, 3.3, 3.4, 3.5'", + "version": "==0.3.0" + }, + "pycparser": { + "hashes": [ + "sha256:8ee45429555515e1f6b185e78100aea234072576aa43ab53aefcae078162fca9", + "sha256:e644fdec12f7872f86c58ff790da456218b10f863970249516d60a5eaca77206" + ], + "version": "==2.21" + }, "pycryptodome": { "hashes": [ "sha256:06d6de87c19f967f03b4cf9b34e538ef46e99a337e9a61a77dbe44b2cbcf0690", @@ -1114,6 +1318,14 @@ "markers": "python_version >= '3.8'", "version": "==2.2.1" }, + "pydispatcher": { + "hashes": [ + "sha256:96543bea04115ffde08f851e1d45cacbfd1ee866ac42127d9b476dc5aefa7de0", + "sha256:b777c6ad080dc1bad74a4c29d6a46914fa6701ac70f94b0d66fbcfde62f5be31" + ], + "markers": "platform_python_implementation == 'CPython'", + "version": "==2.0.7" + }, "pyee": { "hashes": [ "sha256:2770c4928abc721f46b705e6a72b0c59480c4a69c9a83ca0b00bb994f1ea4b32", @@ -1212,6 +1424,14 @@ "markers": "python_version >= '3.7'", "version": "==4.6.2" }, + "pyopenssl": { + "hashes": [ + "sha256:6aa33039a93fffa4563e655b61d11364d01264be8ccb49906101e02a334530bf", + "sha256:ba07553fb6fd6a7a2259adb9b84e12302a9a8a75c44046e8bb5d3e5ee887e3c3" + ], + "markers": "python_version >= '3.7'", + "version": "==24.0.0" + }, "pyparsing": { "hashes": [ "sha256:32c7c0b711493c72ff18a981d24f28aaf9c1fb7ed5e9667c9e84e3db623bdbfb", @@ -1300,6 +1520,14 @@ ], "version": "==6.0.1" }, + "queuelib": { + "hashes": [ + "sha256:4b207267f2642a8699a1f806045c56eb7ad1a85a10c0e249884580d139c2fcd2", + "sha256:4b96d48f650a814c6fb2fd11b968f9c46178b683aad96d68f930fe13a8574d19" + ], + "markers": "python_version >= '3.5'", + "version": "==1.6.2" + }, "rapidfuzz": { "hashes": [ "sha256:01835d02acd5d95c1071e1da1bb27fe213c84a013b899aba96380ca9962364bc", @@ -1416,6 +1644,13 @@ "markers": "python_version >= '3.7'", "version": "==2.31.0" }, + "requests-file": { + "hashes": [ + "sha256:20c5931629c558fda566cacc10cfe2cd502433e628f568c34c80d96a0cc95972", + "sha256:3e493d390adb44aa102ebea827a48717336d5268968c370eaf19abaf5cae13bf" + ], + "version": "==2.0.0" + }, "requests-toolbelt": { "hashes": [ "sha256:7681a0a3d047012b5bdc0ee37d7f8f07ebe76ab08caeccfc3921ce23c88d5bc6", @@ -1424,6 +1659,15 @@ "markers": "python_version >= '2.7' and python_version not in '3.0, 3.1, 3.2, 3.3'", "version": "==1.0.0" }, + "scrapy": { + "hashes": [ + "sha256:733a039c7423e52b69bf2810b5332093d4e42a848460359c07b02ecff8f73ebe", + "sha256:f1edee0cd214512054c01a8d031a8d213dddb53492b02c9e66256e3efe90d175" + ], + "index": "pypi", + "markers": "python_version >= '3.8'", + "version": "==2.11.1" + }, "seedrcc": { "hashes": [ "sha256:6c41d35b49118b5e3abb5ea643e848afad68125ccd328ebcc5e8c9a11420dd5d", @@ -1433,6 +1677,14 @@ "markers": "python_version >= '3.0'", "version": "==1.0.1" }, + "service-identity": { + "hashes": [ + "sha256:6829c9d62fb832c2e1c435629b0a8c476e1929881f28bee4d20bc24161009221", + "sha256:a28caf8130c8a5c1c7a6f5293faaf239bbfb7751e4862436920ee6f2616f568a" + ], + "markers": "python_version >= '3.8'", + "version": "==24.1.0" + }, "setuptools": { "hashes": [ "sha256:02fa291a0471b3a18b2b2481ed902af520c69e8ae0919c13da936542754b4c56", @@ -1537,6 +1789,14 @@ "markers": "python_version >= '3.8'", "version": "==0.22.1" }, + "tldextract": { + "hashes": [ + "sha256:9b6dbf803cb5636397f0203d48541c0da8ba53babaf0e8a6feda2d88746813d4", + "sha256:b9c4510a8766d377033b6bace7e9f1f17a891383ced3c5d50c150f181e9e1cc2" + ], + "markers": "python_version >= '3.8'", + "version": "==5.1.1" + }, "toml": { "hashes": [ "sha256:806143ae5bfb6a3c6e736a764057db0e6a0e05e338b5630894a5f779cabb4f9b", @@ -1553,6 +1813,14 @@ "markers": "python_version >= '3.7'", "version": "==4.2.4" }, + "twisted": { + "hashes": [ + "sha256:4ae8bce12999a35f7fe6443e7f1893e6fe09588c8d2bed9c35cdce8ff2d5b444", + "sha256:987847a0790a2c597197613686e2784fd54167df3a55d0fb17c8412305d76ce5" + ], + "markers": "python_full_version >= '3.8.0'", + "version": "==23.10.0" + }, "typing-extensions": { "hashes": [ "sha256:23478f88c37f27d76ac8aee6c905017a143b0b1b886c3c9f66bc2fd94f9f5783", @@ -1633,6 +1901,14 @@ "markers": "python_version >= '3.8'", "version": "==0.22.0" }, + "w3lib": { + "hashes": [ + "sha256:c4432926e739caa8e3f49f5de783f336df563d9490416aebd5d39fb896d264e7", + "sha256:ed5b74e997eea2abe3c1321f916e344144ee8e9072a6f33463ee8e57f858a4b1" + ], + "markers": "python_version >= '3.7'", + "version": "==2.1.2" + }, "watchdog": { "hashes": [ "sha256:11e12fafb13372e18ca1bbf12d50f593e7280646687463dd47730fd4f4d5d257", diff --git a/db/crud.py b/db/crud.py index c5496dd..521e54a 100644 --- a/db/crud.py +++ b/db/crud.py @@ -302,7 +302,9 @@ async def get_series_meta(meta_id: str): metadata["meta"]["videos"].append( { "id": stream_id, - "title": f"S{stream.season.season_number} EP{episode.episode_number}", + "title": f"S{stream.season.season_number} EP{episode.episode_number}" + if not episode.title + else episode.title, "season": stream.season.season_number, "episode": episode.episode_number, "released": stream.created_at.strftime( diff --git a/db/models.py b/db/models.py index 6864e66..34fa663 100644 --- a/db/models.py +++ b/db/models.py @@ -3,7 +3,7 @@ from typing import Optional, Any import pymongo from beanie import Document, Link -from pydantic import BaseModel, Field, field_validator +from pydantic import BaseModel, Field from pymongo import IndexModel, ASCENDING @@ -12,6 +12,7 @@ class Episode(BaseModel): filename: str | None = None size: int | None = None file_index: int | None = None + title: str | None = None class Season(BaseModel): diff --git a/mediafusion_scrapy/__init__.py b/mediafusion_scrapy/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/mediafusion_scrapy/items.py b/mediafusion_scrapy/items.py new file mode 100644 index 0000000..4aa9e42 --- /dev/null +++ b/mediafusion_scrapy/items.py @@ -0,0 +1,12 @@ +# Define here the models for your scraped items +# +# See documentation in: +# https://docs.scrapy.org/en/latest/topics/items.html + +import scrapy + + +class MediafusionScrapyItem(scrapy.Item): + # define the fields for your item here like: + # name = scrapy.Field() + pass diff --git a/mediafusion_scrapy/middlewares.py b/mediafusion_scrapy/middlewares.py new file mode 100644 index 0000000..8d0c2e2 --- /dev/null +++ b/mediafusion_scrapy/middlewares.py @@ -0,0 +1,119 @@ +# Define here the models for your spider middleware +# +# See documentation in: +# https://docs.scrapy.org/en/latest/topics/spider-middleware.html + +from scrapy import signals + +# useful for handling different item types with a single interface +from itemadapter import is_item, ItemAdapter + +from db import database + + +class MediafusionScrapySpiderMiddleware: + # Not all methods need to be defined. If a method is not defined, + # scrapy acts as if the spider middleware does not modify the + # passed objects. + + @classmethod + def from_crawler(cls, crawler): + # This method is used by Scrapy to create your spiders. + s = cls() + crawler.signals.connect(s.spider_opened, signal=signals.spider_opened) + return s + + def process_spider_input(self, response, spider): + # Called for each response that goes through the spider + # middleware and into the spider. + + # Should return None or raise an exception. + return None + + def process_spider_output(self, response, result, spider): + # Called with the results returned from the Spider, after + # it has processed the response. + + # Must return an iterable of Request, or item objects. + for i in result: + yield i + + def process_spider_exception(self, response, exception, spider): + # Called when a spider or process_spider_input() method + # (from other spider middleware) raises an exception. + + # Should return either None or an iterable of Request or item objects. + pass + + def process_start_requests(self, start_requests, spider): + # Called with the start requests of the spider, and works + # similarly to the process_spider_output() method, except + # that it doesn’t have a response associated. + + # Must return only requests (not items). + for r in start_requests: + yield r + + def spider_opened(self, spider): + spider.logger.info("Spider opened: %s" % spider.name) + + +class MediafusionScrapyDownloaderMiddleware: + # Not all methods need to be defined. If a method is not defined, + # scrapy acts as if the downloader middleware does not modify the + # passed objects. + + @classmethod + def from_crawler(cls, crawler): + # This method is used by Scrapy to create your spiders. + s = cls() + crawler.signals.connect(s.spider_opened, signal=signals.spider_opened) + return s + + def process_request(self, request, spider): + # Called for each request that goes through the downloader + # middleware. + + # Must either: + # - return None: continue processing this request + # - or return a Response object + # - or return a Request object + # - or raise IgnoreRequest: process_exception() methods of + # installed downloader middleware will be called + return None + + def process_response(self, request, response, spider): + # Called with the response returned from the downloader. + + # Must either; + # - return a Response object + # - return a Request object + # - or raise IgnoreRequest + return response + + def process_exception(self, request, exception, spider): + # Called when a download handler or a process_request() + # (from other downloader middleware) raises an exception. + + # Must either: + # - return None: continue processing this exception + # - return a Response object: stops process_exception() chain + # - return a Request object: stops process_exception() chain + pass + + def spider_opened(self, spider): + spider.logger.info("Spider opened: %s" % spider.name) + + +class DatabaseInitializationMiddleware: + @classmethod + def from_crawler(cls, crawler): + # This method is used by Scrapy to create your middleware instance + middleware = cls() + crawler.signals.connect(middleware.spider_opened, signal=signals.spider_opened) + return middleware + + async def spider_opened(self, spider): + # Initialize your database here + await database.init() + spider.logger.info("Database initialized successfully.") diff --git a/mediafusion_scrapy/pipelines.py b/mediafusion_scrapy/pipelines.py new file mode 100644 index 0000000..606f18b --- /dev/null +++ b/mediafusion_scrapy/pipelines.py @@ -0,0 +1,69 @@ +import logging + +from beanie import WriteRules +from itemadapter import ItemAdapter +from scrapy.exceptions import DropItem + +from db import database +from db.models import TorrentStreams, Season, MediaFusionSeriesMetaData + + +class TorrentDuplicatesPipeline: + def __init__(self): + self.info_hashes_seen = set() + + def process_item(self, item, spider): + adapter = ItemAdapter(item) + if adapter["info_hash"] in self.info_hashes_seen: + raise DropItem(f"Duplicate item found: {item!r}") + else: + self.info_hashes_seen.add(adapter["info_hash"]) + return item + + +class FormulaStorePipeline: + async def process_item(self, item, spider): + if "unique_id" not in item: + logging.warning(f"unique_id not found in item: {item}") + return item + + # Construct the meta_id + meta_id = f"mf{item['unique_id']}" + + # Create a season object + season = Season(season_number=1, episodes=item["episodes"]) + + # Create the stream + stream = TorrentStreams( + id=item["info_hash"], + torrent_name=item["torrent_name"], + announce_list=item["announce_list"], + size=item["total_size"], + languages=item["languages"], + resolution=item.get("resolution"), + codec=item.get("codec"), + quality=item.get("quality"), + audio=item.get("audio"), + encoder=item.get("encoder"), + source=item["source"], + catalog=item["catalog"], + created_at=item["created_at"], + season=season, + meta_id=meta_id, + seeders=item["seeders"], + ) + + series = MediaFusionSeriesMetaData( + id=meta_id, + title=item["title"], + year=item.get("year"), + poster=item.get("poster"), + background=item.get("background"), + streams=[stream], + type="series", + ) + + await series.insert(link_rule=WriteRules.WRITE) + logging.info(f"Inserted new formula: {item['torrent_name']}") + + return item diff --git a/mediafusion_scrapy/settings.py b/mediafusion_scrapy/settings.py new file mode 100644 index 0000000..c699ae5 --- /dev/null +++ b/mediafusion_scrapy/settings.py @@ -0,0 +1,93 @@ +# Scrapy settings for mediafusion_scrapy project +# +# For simplicity, this file contains only settings considered important or +# commonly used. You can find more settings consulting the documentation: +# +# https://docs.scrapy.org/en/latest/topics/settings.html +# https://docs.scrapy.org/en/latest/topics/downloader-middleware.html +# https://docs.scrapy.org/en/latest/topics/spider-middleware.html + +BOT_NAME = "mediafusion_scrapy" + +SPIDER_MODULES = ["mediafusion_scrapy.spiders"] +NEWSPIDER_MODULE = "mediafusion_scrapy.spiders" + + +# Crawl responsibly by identifying yourself (and your website) on the user-agent +# USER_AGENT = "mediafusion_scrapy (+http://www.yourdomain.com)" + +# Obey robots.txt rules +ROBOTSTXT_OBEY = False + +# Configure maximum concurrent requests performed by Scrapy (default: 16) +# CONCURRENT_REQUESTS = 32 + +# Configure a delay for requests for the same website (default: 0) +# See https://docs.scrapy.org/en/latest/topics/settings.html#download-delay +# See also autothrottle settings and docs +# DOWNLOAD_DELAY = 3 +# The download delay setting will honor only one of: +# CONCURRENT_REQUESTS_PER_DOMAIN = 16 +# CONCURRENT_REQUESTS_PER_IP = 16 + +# Disable cookies (enabled by default) +# COOKIES_ENABLED = False + +# Disable Telnet Console (enabled by default) +# TELNETCONSOLE_ENABLED = False + +# Override the default request headers: +# DEFAULT_REQUEST_HEADERS = { +# "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8", +# "Accept-Language": "en", +# } + +# Enable or disable spider middlewares +# See https://docs.scrapy.org/en/latest/topics/spider-middleware.html +SPIDER_MIDDLEWARES = { + "mediafusion_scrapy.middlewares.DatabaseInitializationMiddleware": 100, +} + +# Enable or disable downloader middlewares +# See https://docs.scrapy.org/en/latest/topics/downloader-middleware.html +# DOWNLOADER_MIDDLEWARES = { +# "mediafusion_scrapy.middlewares.MediafusionScrapyDownloaderMiddleware": 543, +# } + +# Enable or disable extensions +# See https://docs.scrapy.org/en/latest/topics/extensions.html +# EXTENSIONS = { +# "scrapy.extensions.telnet.TelnetConsole": None, +# } + +# Configure item pipelines +# See https://docs.scrapy.org/en/latest/topics/item-pipeline.html +# ITEM_PIPELINES = { +# "mediafusion_scrapy.pipelines.MediafusionScrapyPipeline": 300, +# } + +# Enable and configure the AutoThrottle extension (disabled by default) +# See https://docs.scrapy.org/en/latest/topics/autothrottle.html +AUTOTHROTTLE_ENABLED = True +# The initial download delay +# AUTOTHROTTLE_START_DELAY = 5 +# The maximum download delay to be set in case of high latencies +# AUTOTHROTTLE_MAX_DELAY = 60 +# The average number of requests Scrapy should be sending in parallel to +# each remote server +# AUTOTHROTTLE_TARGET_CONCURRENCY = 1.0 +# Enable showing throttling stats for every response received: +# AUTOTHROTTLE_DEBUG = False + +# Enable and configure HTTP caching (disabled by default) +# See https://docs.scrapy.org/en/latest/topics/downloader-middleware.html#httpcache-middleware-settings +# HTTPCACHE_ENABLED = True +# HTTPCACHE_EXPIRATION_SECS = 0 +# HTTPCACHE_DIR = "httpcache" +# HTTPCACHE_IGNORE_HTTP_CODES = [] +# HTTPCACHE_STORAGE = "scrapy.extensions.httpcache.FilesystemCacheStorage" + +# Set settings whose default value is deprecated to a future-proof value +REQUEST_FINGERPRINTER_IMPLEMENTATION = "2.7" +TWISTED_REACTOR = "twisted.internet.asyncioreactor.AsyncioSelectorReactor" +FEED_EXPORT_ENCODING = "utf-8" diff --git a/mediafusion_scrapy/spiders/__init__.py b/mediafusion_scrapy/spiders/__init__.py new file mode 100644 index 0000000..ebd689a --- /dev/null +++ b/mediafusion_scrapy/spiders/__init__.py @@ -0,0 +1,4 @@ +# This package will contain the spiders of your Scrapy project +# +# Please refer to the documentation for information on how to create and manage +# your spiders. diff --git a/mediafusion_scrapy/spiders/formula_tgx.py b/mediafusion_scrapy/spiders/formula_tgx.py new file mode 100644 index 0000000..562da16 --- /dev/null +++ b/mediafusion_scrapy/spiders/formula_tgx.py @@ -0,0 +1,213 @@ +import re +from datetime import datetime + +import scrapy + +from db.models import TorrentStreams, Episode +from utils.parser import convert_size_to_bytes +from utils.torrent import parse_magnet + + +class FormulaTgxSpider(scrapy.Spider): + name = "formula_tgx" + allowed_domains = ["torrentgalaxy.to", "tgx.rs"] + start_urls = [ + "https://torrentgalaxy.to/profile/egortech/torrents/0", + "https://tgx.rs/profile/egortech/torrents/0", + ] + formula1_keyword_patterns = re.compile(r"formula[ .+]*[1234e]+", re.IGNORECASE) + uploader_parsing_functions = { + "egortech": "parse_torrent_details_egortech", + } + + custom_settings = { + "ITEM_PIPELINES": { + "mediafusion_scrapy.pipelines.TorrentDuplicatesPipeline": 100, + "mediafusion_scrapy.pipelines.FormulaStorePipeline": 200, + } + } + + async def parse(self, response, **kwargs): + uploader_profile_name = response.url.split("/")[4] + self.logger.info(f"Scraping torrents from {uploader_profile_name}") + parsing_function_name = self.uploader_parsing_functions[uploader_profile_name] + parsing_function = getattr(self, parsing_function_name, None) + + # Extract the last page number only once at the beginning + if response.url in self.start_urls: + last_page_number = response.css( + "ul.pagination li.page-item:not(.disabled) a::attr(href)" + ).re(r"/profile/.*/torrents/(\d+)")[-2] + last_page_number = ( + int(last_page_number) if last_page_number.isdigit() else 0 + ) + + # Generate requests for all pages + for page_number in range(1, last_page_number + 1): + next_page_url = ( + f"{response.url.split('/torrents/')[0]}/torrents/{page_number}" + ) + yield response.follow(next_page_url, self.parse) + + # Extract torrents from the page + for torrent in response.css("div.tgxtablerow.txlight"): + urls = torrent.css("div.tgxtablecell a::attr(href)").getall() + + torrent_name = torrent.css( + "div.tgxtablecell.clickable-row.click.textshadow.rounded.txlight a b::text" + ).get() + + if not self.formula1_keyword_patterns.search(torrent_name): + continue + + tgx_unique_id = urls[0].split("/")[-2] + torrent_page_link = response.urljoin(urls[0]) + torrent_link = urls[1] + magnet_link = urls[2] + info_hash, announce_list = parse_magnet(magnet_link) + if not info_hash: + self.logger.warning( + f"Failed to parse magnet link: {response.url}, {torrent_name}" + ) + continue + + seeders = torrent.css( + "div.tgxtablecell span[title='Seeders/Leechers'] font[color='green'] b::text" + ).get() + + seeders = int(seeders) if seeders and seeders.isdigit() else None + + torrent_data = { + "info_hash": info_hash, + "torrent_name": torrent_name, + "torrent_link": torrent_link, + "magnet_link": magnet_link, + "seeders": seeders, + "torrent_page_link": torrent_page_link, + "unique_id": tgx_unique_id, + "source": f"TorrentGalaxy ({uploader_profile_name})", + "announce_list": announce_list, + "catalog": ["formula_racing"], + } + + torrent_stream = await TorrentStreams.get(info_hash) + if torrent_stream: + self.logger.info(f"Torrent stream already exists: {torrent_name}") + torrent_stream.seeders = seeders + await torrent_stream.save() + else: + yield response.follow( + torrent_page_link, + parsing_function, + meta={"torrent_data": torrent_data}, + ) + + def parse_torrent_details_egortech(self, response): + torrent_data = response.meta["torrent_data"] + + # Extracting file details and sizes + file_details = [] + for row in response.xpath('//table[contains(@class, "table-striped")]/tr'): + file_name = row.xpath('td[@class="table_col1"]/text()').get() + file_size = row.xpath('td[@class="table_col2"]/text()').get() + if file_name and file_size: + file_details.append({"file_name": file_name, "file_size": file_size}) + + cover_image_url = response.xpath( + "//center/img[contains(@class, 'img-responsive') and contains(@data-src, '.png')]/@data-src" + ).get() + torrent_data["poster"] = cover_image_url + torrent_data["background"] = cover_image_url + + # Processing the description for video, audio, and other details + torrent_description = "".join( + response.xpath( + "//font/following-sibling::*[1]/following-sibling::text() | //font/following-sibling::*[1]/following-sibling::*//text()" + ).extract() + ) + + quality_match = re.search(r"Quality:\s*(\S+)", torrent_description) + video_match = re.search( + r"Video:\s*([^,]+),\s*([0-9]+x[0-9]+[A-Za-z]+),\s*([0-9]+\s*fps),\s*([0-9]+\s*kb/s)", + torrent_description, + ) + audio_match = re.search( + r"Audio:\s*([^,]+),\s*([0-9]+\s*KHz),\s*([0-9]+\s*kb/s)\s*(English)?", + torrent_description, + ) + + if quality_match: + torrent_data["quality"] = quality_match.group(1) + if video_match: + torrent_data[ + "codec" + ] = f"{video_match.group(1)} {video_match.group(3)} {video_match.group(4)}" + torrent_data["resolution"] = video_match.group(2).lower().split("x")[1] + if audio_match: + torrent_data[ + "audio" + ] = f"{audio_match.group(1)} {audio_match.group(2)} {audio_match.group(3)}" + + contains_index = torrent_description.find("Contains:") + episodes = [] + + if contains_index != -1: + contents_section = torrent_description[ + contains_index + len("Contains:") : + ].strip() + + items = [ + item.strip() + for item in re.split(r"\r?\n", contents_section) + if item.strip() + ] + + for index, item in enumerate(items): + file_detail = file_details[index] + episodes.append( + Episode( + episode_number=index + 1, + filename=file_detail.get("file_name"), + size=convert_size_to_bytes(file_detail.get("file_size")), + file_index=index, + title=item, + ) + ) + + torrent_data["episodes"] = episodes + + total_size = response.xpath( + "//div[b='Total Size:']/following-sibling::div/text()" + ).get() + if total_size: + torrent_data["total_size"] = convert_size_to_bytes(total_size) + + # Extracting date created + date_created = response.xpath( + "//div[b[contains(., 'Added:')]]/following-sibling::div/text()" + ).get() + if date_created: + # Processing to extract the date and time + torrent_data["created_at"] = datetime.strptime( + date_created.strip(), "%d-%m-%Y %H:%M" + ) + + # Extracting language + language = ( + response.xpath("//div[b='Language:']/following-sibling::div/text()") + .get(default="Unknown") + .strip() + ) + torrent_data["languages"] = [language] + + # cleanup "." from torrent name for and add unique_id for title to be unique for indexing + torrent_data[ + "title" + ] = f"{torrent_data['torrent_name'].replace('.', '')} {torrent_data['unique_id']}" + + # Extract year from the torrent name + year_match = re.search(r"\b(19|20)\d{2}\b", torrent_data["torrent_name"]) + if year_match: + torrent_data["year"] = int(year_match.group()) + + yield torrent_data diff --git a/resources/manifest.json b/resources/manifest.json index 794347b..4470ad7 100644 --- a/resources/manifest.json +++ b/resources/manifest.json @@ -438,6 +438,17 @@ "isRequired": true } ] + }, + { + "id": "formula_racing", + "type": "series", + "name": "Formula Racing", + "extra": [ + { + "name": "skip", + "isRequired": false + } + ] } ] } \ No newline at end of file diff --git a/scrapy.cfg b/scrapy.cfg new file mode 100644 index 0000000..40b8a3b --- /dev/null +++ b/scrapy.cfg @@ -0,0 +1,11 @@ +# Automatically created by: scrapy startproject +# +# For more information about the [deploy] section see: +# https://scrapyd.readthedocs.io/en/latest/deploy.html + +[settings] +default = mediafusion_scrapy.settings + +[deploy] +#url = http://localhost:6800/ +project = mediafusion_scrapy diff --git a/utils/const.py b/utils/const.py index 76ee473..2d6c773 100644 --- a/utils/const.py +++ b/utils/const.py @@ -33,6 +33,7 @@ CATALOG_ID_DATA = [ "mediafusion_search_tv", "torrentio_streams", "prowlarr_streams", + "formula_racing", ] CATALOG_NAME_DATA = [ @@ -70,6 +71,7 @@ CATALOG_NAME_DATA = [ "MediaFusion Search TV", "Torrentio Streams", "Prowlarr Streams", + "Formula Racing", ] RESOLUTIONS = [ diff --git a/utils/parser.py b/utils/parser.py index 1997fcc..5b7c789 100644 --- a/utils/parser.py +++ b/utils/parser.py @@ -194,11 +194,9 @@ async def parse_stream_data( description_parts = [ torrent_name or quality_detail, - convert_bytes_to_readable( - episode_data.size or stream_data.size - if episode_data - else stream_data.size - ), + f"{convert_bytes_to_readable(episode_data.size)} - {convert_bytes_to_readable(stream_data.size)}" + if episode_data and episode_data.size + else convert_bytes_to_readable(stream_data.size), seeders, " + ".join(stream_data.languages), stream_data.source, diff --git a/utils/torrent.py b/utils/torrent.py index 15f99a5..c97d34c 100644 --- a/utils/torrent.py +++ b/utils/torrent.py @@ -15,7 +15,7 @@ from anyio import ( ) from anyio.streams.memory import MemoryObjectSendStream from demagnetize.core import Demagnetizer -from torf import Magnet +from torf import Magnet, MagnetError from utils.parser import is_contain_18_plus_keywords @@ -183,3 +183,14 @@ async def init_best_trackers(): logging.info(f"Loaded {len(trackers)} trackers. Total: {len(TRACKERS)}") else: logging.error(f"Failed to load trackers: {response.status_code}") + + +def parse_magnet(magnet_link: str) -> tuple[str, list[str]]: + """ + Parse magnet link and return info hash and trackers + """ + try: + magnet = Magnet.from_string(magnet_link) + except MagnetError: + return "", [] + return magnet.infohash, magnet.tr